From 0135e1d634c4f9068fcd345e305767e76dbf5a4e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 00:01:44 -0700 Subject: [PATCH 001/157] fix(oci): pin one response id per streamed completion, skip the [DONE] sentinel OCIStreamWrapper.chunk_creator built every chunk straight from the apiFormat handlers, so it never reached model_response_creator and OCI streams came back with a fresh chatcmpl id, a drifting created value and no model on every chunk. Both exits now go through the shared creator. The GENERIC apiFormat also closes its stream with a literal `data: [DONE]` line, which chunk_creator json-parsed and turned into a 500 on every OCI streaming completion. It is skipped now. --- litellm/llms/oci/chat/transformation.py | 16 +- .../oci/chat/test_oci_chat_transformation.py | 145 ++++++++++++++++++ 2 files changed, 157 insertions(+), 4 deletions(-) diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index 98e23a59eea..17aed74b9d3 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -745,13 +745,21 @@ class OCIStreamWrapper(CustomStreamWrapper): # single-event case (terminal chunk carries the only copy of the text). self._cohere_text_emitted = False - def chunk_creator(self, chunk: Any) -> ModelResponseStream: + def _with_stream_identity(self, parsed: ModelResponseStream) -> ModelResponseStream: + model_response: Final = self.model_response_creator() + model_response.choices = parsed.choices + return model_response + + def chunk_creator(self, chunk: Any) -> ModelResponseStream | None: if not isinstance(chunk, str): raise ValueError(f"Chunk is not a string: {chunk}") if not chunk.startswith("data:"): raise ValueError(f"Chunk does not start with 'data:': {chunk}") + payload: Final = chunk[5:].strip() + if payload == "[DONE]": + return None try: - dict_chunk: Final = json.loads(chunk[5:]) + dict_chunk: Final = json.loads(payload) except json.JSONDecodeError as e: raise OCIError( status_code=500, @@ -774,8 +782,8 @@ class OCIStreamWrapper(CustomStreamWrapper): if getattr(choice.delta, "content", None): self._cohere_text_emitted = True break - return result - return handle_generic_stream_chunk(dict_chunk) + return self._with_stream_identity(result) + return self._with_stream_identity(handle_generic_stream_chunk(dict_chunk)) __all__ = [ diff --git a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py index 86c534c73c2..6601de2853d 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py @@ -1900,3 +1900,148 @@ class TestOCIImageUrlTransformation: adapt_messages_to_generic_oci_standard(messages) assert "image_url" in str(exc_info.value) + + +# --------------------------------------------------------------------------- +# OCIStreamWrapper: per-stream identity and the GENERIC `[DONE]` sentinel +# --------------------------------------------------------------------------- + +import itertools +from unittest.mock import patch + +from litellm.llms.oci.chat.transformation import OCIStreamWrapper, _iter_sse_events + +_STREAM_GENERIC_MODEL = "xai.grok-4" +_STREAM_COHERE_MODEL = "cohere.command-latest" + +_GENERIC_TEXT_EVENT = ( + 'data: {{"index":0,"message":{{"role":"ASSISTANT","content":[{{"type":"TEXT","text":"{text}"}}]}},"pad":"aaa"}}' +) +_GENERIC_TERMINAL_EVENT = ( + 'data: {"message":{"role":"ASSISTANT","content":[{"type":"TEXT","text":""}]},"finishReason":"stop","pad":"a"}' +) +_COHERE_TEXT_EVENT = 'data: {{"apiFormat":"COHERE","text":"{text}","pad":"aaaaaa"}}' +_COHERE_TERMINAL_EVENT = ( + 'data: {"apiFormat":"COHERE","text":"123","finishReason":"COMPLETE",' + '"chatHistory":[{"role":"USER","message":"count"},{"role":"CHATBOT","message":"123"}]}' +) + + +def _make_stream_wrapper(model: str) -> OCIStreamWrapper: + logging_obj = MagicMock() + logging_obj.model_call_details = {"custom_llm_provider": "oci", "litellm_params": {}} + return OCIStreamWrapper( + completion_stream=iter([]), + model=model, + custom_llm_provider="oci", + logging_obj=logging_obj, + ) + + +def _ticking_clock(): + """A ``time.time`` stand-in that advances a full second on every call. + + Without it the whole test runs inside one wall-clock second, so a per-chunk + ``created`` would coincidentally match and the drift would go unnoticed. + """ + return itertools.count(1_700_000_000.0) + + +class TestOCIStreamWrapperIdentityPinning: + """One OCI streaming completion must present one id, one created and the + wrapper's model on every chunk, the way every other provider does.""" + + def test_generic_stream_shares_one_id_created_and_model(self): + wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) + events = [ + _GENERIC_TEXT_EVENT.format(text="1"), + _GENERIC_TEXT_EVENT.format(text="2"), + _GENERIC_TEXT_EVENT.format(text="3"), + _GENERIC_TERMINAL_EVENT, + ] + + with patch("time.time", side_effect=_ticking_clock()): + chunks = [wrapper.chunk_creator(event) for event in events] + + assert len(chunks) == 4 + assert len({chunk.id for chunk in chunks}) == 1 + assert chunks[0].id.startswith("chatcmpl-") + assert len({chunk.created for chunk in chunks}) == 1 + assert {chunk.model for chunk in chunks} == {_STREAM_GENERIC_MODEL} + assert [chunk.choices[0].delta.content for chunk in chunks[:3]] == ["1", "2", "3"] + assert chunks[-1].choices[0].finish_reason == "stop" + assert all(chunk._hidden_params["custom_llm_provider"] == "oci" for chunk in chunks) + + def test_cohere_stream_shares_one_id_created_and_model(self): + """Rebuilding each chunk through the shared creator must not disturb the + Cohere bookkeeping that suppresses the terminal event's repeated text.""" + wrapper = _make_stream_wrapper(_STREAM_COHERE_MODEL) + events = [ + _COHERE_TEXT_EVENT.format(text="1"), + _COHERE_TEXT_EVENT.format(text="2"), + _COHERE_TEXT_EVENT.format(text="3"), + _COHERE_TERMINAL_EVENT, + ] + + with patch("time.time", side_effect=_ticking_clock()): + chunks = [wrapper.chunk_creator(event) for event in events] + + assert len(chunks) == 4 + assert len({chunk.id for chunk in chunks}) == 1 + assert len({chunk.created for chunk in chunks}) == 1 + assert {chunk.model for chunk in chunks} == {_STREAM_COHERE_MODEL} + assert [chunk.choices[0].delta.content for chunk in chunks[:3]] == ["1", "2", "3"] + assert chunks[-1].choices[0].finish_reason == "stop" + assert chunks[-1].choices[0].delta.content is None + assert wrapper._cohere_text_emitted is True + + def test_id_is_pinned_to_the_wrapper_response_id(self): + wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) + + first = wrapper.chunk_creator(_GENERIC_TEXT_EVENT.format(text="1")) + + assert wrapper.response_id == first.id + assert wrapper.created == first.created + + +class TestOCIStreamWrapperDoneSentinel: + """OCI's GENERIC apiFormat closes the stream with a literal `[DONE]` line; + parsing it as JSON turned every streaming completion into a 500.""" + + @pytest.mark.parametrize("done_event", ["data: [DONE]", "data:[DONE]", "data: [DONE] "]) + def test_done_sentinel_returns_none(self, done_event): + wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) + assert wrapper.chunk_creator(done_event) is None + + def test_done_sentinel_off_the_sse_splitter_is_skipped(self): + wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) + wire = ( + f"{_GENERIC_TEXT_EVENT.format(text='1')}\n\n" + f"{_GENERIC_TEXT_EVENT.format(text='2')}\n\n" + f"{_GENERIC_TERMINAL_EVENT}\n\n" + "data: [DONE]\n\n" + ) + + events = list(_iter_sse_events(iter([wire]))) + assert events[-1] == "data: [DONE]" + + chunks = [wrapper.chunk_creator(event) for event in events] + assert chunks[-1] is None + + emitted = [chunk for chunk in chunks if chunk is not None] + assert len(emitted) == 3 + assert len({chunk.id for chunk in emitted}) == 1 + + def test_unparseable_payload_still_raises_oci_error(self): + from litellm.llms.oci.common_utils import OCIError + + wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) + with pytest.raises(OCIError, match="Chunk cannot be parsed as JSON"): + wrapper.chunk_creator("data: not-json-at-all") + + def test_done_lookalike_payload_still_raises_oci_error(self): + from litellm.llms.oci.common_utils import OCIError + + wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) + with pytest.raises(OCIError, match="Chunk cannot be parsed as JSON"): + wrapper.chunk_creator("data: [DONE] trailing garbage") From fd2104a61efe91eb1428d40c3261527c363ea5d6 Mon Sep 17 00:00:00 2001 From: mateo-berri Date: Thu, 3 Sep 2026 00:27:54 -0700 Subject: [PATCH 002/157] refactor(oci): build the identity-pinned stream chunk in one shot Address the two Greptile P2 notes: construct the chunk through the shared creator instead of mutating its choices afterwards, and drop the decorative divider comment from the new tests. --- litellm/llms/oci/chat/transformation.py | 4 +--- .../llms/oci/chat/test_oci_chat_transformation.py | 4 ---- 2 files changed, 1 insertion(+), 7 deletions(-) diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index 17aed74b9d3..b1167d447fc 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -746,9 +746,7 @@ class OCIStreamWrapper(CustomStreamWrapper): self._cohere_text_emitted = False def _with_stream_identity(self, parsed: ModelResponseStream) -> ModelResponseStream: - model_response: Final = self.model_response_creator() - model_response.choices = parsed.choices - return model_response + return self.model_response_creator(chunk={"choices": parsed.choices}) def chunk_creator(self, chunk: Any) -> ModelResponseStream | None: if not isinstance(chunk, str): diff --git a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py index 6601de2853d..e77b2c24d01 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py @@ -1902,10 +1902,6 @@ class TestOCIImageUrlTransformation: assert "image_url" in str(exc_info.value) -# --------------------------------------------------------------------------- -# OCIStreamWrapper: per-stream identity and the GENERIC `[DONE]` sentinel -# --------------------------------------------------------------------------- - import itertools from unittest.mock import patch From 445ccc14b29bb4a93e3b21bf907a7b2a26da507a Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 00:53:55 -0700 Subject: [PATCH 003/157] fix(vector-stores): surface retrieval failures to the API caller A vector store search that fails is swallowed by the pre-call hook, so the request goes to the model with an un-augmented prompt and the caller gets a 200 answering from the model's own knowledge with no way to tell the knowledge base was skipped. Failed searches now ride the same channel their successes already use: a vector_store_search_failures entry on provider_specific_fields naming the store id, provider, and error. That is additive and always on. For callers who would rather fail than answer ungrounded, litellm_settings vector_store_search_failure_mode: error raises VectorStoreSearchError (400) instead; the default stays annotate, today's permissive behavior. The hook's outer catch-all also now names the requested vector store ids in its log line, and only wraps the augmentation itself, so the fail-closed raise is not swallowed by it. --- litellm/__init__.py | 3 + litellm/exceptions.py | 25 ++ .../vector_store_pre_call_hook.py | 378 ++++++++++-------- litellm/types/vector_stores.py | 13 +- .../test_vector_store_pre_call_hook.py | 199 +++++++++ 5 files changed, 460 insertions(+), 158 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 41a3789ab0d..76daf771c4b 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1360,6 +1360,7 @@ from .exceptions import ( InvalidRequestError, BadRequestError, ImageFetchError, + VectorStoreSearchError, NotFoundError, PermissionDeniedError, RateLimitError, @@ -1461,9 +1462,11 @@ from .vector_stores.vector_store_registry import ( VectorStoreRegistry, VectorStoreIndexRegistry, ) +from .types.vector_stores import VectorStoreSearchFailureMode vector_store_registry: Optional[VectorStoreRegistry] = None vector_store_index_registry: Optional[VectorStoreIndexRegistry] = None +vector_store_search_failure_mode: VectorStoreSearchFailureMode = "annotate" ### RAG ### from . import rag diff --git a/litellm/exceptions.py b/litellm/exceptions.py index 286f7528896..4387370e58c 100644 --- a/litellm/exceptions.py +++ b/litellm/exceptions.py @@ -10,12 +10,14 @@ ## LiteLLM versions of the OpenAI Exception Types import enum +from collections.abc import Sequence from typing import Any, Final import httpx import openai from litellm.types.utils import LiteLLMCommonStrings +from litellm.types.vector_stores import VectorStoreSearchFailure class RateLimitErrorCategory(str, enum.Enum): @@ -288,6 +290,29 @@ class ImageFetchError(BadRequestError): ) +VECTOR_STORE_SEARCH_FAILED_CODE: Final = "vector_store_search_failed" + + +class VectorStoreSearchError(BadRequestError): + def __init__( + self, + failures: Sequence[VectorStoreSearchFailure], + model: str | None = None, + llm_provider: str | None = None, + ) -> None: + self.failures: Final[tuple[VectorStoreSearchFailure, ...]] = tuple(failures) + detail: Final = "; ".join(f"{failure['vector_store_id']}: {failure['error']}" for failure in self.failures) + super().__init__( + message=( + "The request could not be grounded in every configured vector store. " + f"{len(self.failures)} vector store search(es) failed: {detail}" + ), + model=model, + llm_provider=llm_provider, + body={"type": "invalid_request_error", "code": VECTOR_STORE_SEARCH_FAILED_CODE}, + ) + + class UnprocessableEntityError(openai.UnprocessableEntityError): def __init__( self, diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index 12ff38ce4ba..d5c893c4b20 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -5,20 +5,21 @@ This hook is called before making an LLM request when a vector store is configur It searches the vector store for relevant context and appends it to the messages. """ -from collections.abc import Awaitable, Callable +from collections.abc import Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, Protocol, cast +from typing import TYPE_CHECKING, Any, Final, Protocol, assert_never, cast import litellm import litellm.vector_stores from litellm._logging import verbose_logger +from litellm.exceptions import VectorStoreSearchError from litellm.integrations.custom_logger import CustomLogger from litellm.types.llms.openai import AllMessageValues, ChatCompletionUserMessage from litellm.types.prompts.init_prompts import PromptSpec from litellm.types.utils import CallTypes, StandardCallbackDynamicParams from litellm.types.vector_stores import ( LiteLLM_ManagedVectorStore, - VectorStoreResultContent, + VectorStoreSearchFailure, VectorStoreSearchResponse, VectorStoreSearchResult, ) @@ -30,6 +31,8 @@ if TYPE_CHECKING: else: LiteLLMLoggingObj = Any +SEARCH_FAILURES_FIELD: Final = "vector_store_search_failures" + class ProxyRuntime(Protocol): def llm_router(self) -> "Router | None": ... @@ -54,11 +57,31 @@ class ProxyServerRuntime: return prisma_client +@dataclass(frozen=True, slots=True) +class SearchSucceeded: + response: VectorStoreSearchResponse + + +@dataclass(frozen=True, slots=True) +class SearchFailed: + failure: VectorStoreSearchFailure + + +SearchOutcome = SearchSucceeded | SearchFailed + + +@dataclass(frozen=True, slots=True) +class VectorStoreAugmentation: + messages: tuple[AllMessageValues, ...] + search_results: tuple[VectorStoreSearchResponse, ...] + failures: tuple[VectorStoreSearchFailure, ...] + + class VectorStorePreCallHook(CustomLogger): CONTENT_PREFIX_STRING = "Context:\n\n" """ Custom logger that handles vector store searches before LLM calls. - + When a vector store is configured, this hook: 1. Extracts the query from the last user message 2. Calls litellm.vector_stores.search() to get relevant context @@ -101,100 +124,152 @@ class VectorStorePreCallHook(CustomLogger): Returns: Tuple of (model, modified_messages, non_default_params) """ + requested_vector_store_ids: Final = _requested_vector_store_ids(non_default_params) try: - # Check if vector store is configured - if litellm.vector_store_registry is None: - return model, messages, non_default_params - - prisma_client: Final = self.proxy_runtime.prisma_client() - llm_router: Final = self.proxy_runtime.llm_router() - - # Use database fallback to ensure synchronization across instances - vector_stores_to_run: list[ - LiteLLM_ManagedVectorStore - ] = await litellm.vector_store_registry.pop_vector_stores_to_run_with_db_fallback( + augmentation: VectorStoreAugmentation | None = await self._augment_messages( + messages=messages, non_default_params=non_default_params, tools=tools, - prisma_client=prisma_client, + litellm_logging_obj=litellm_logging_obj, ) - - if not vector_stores_to_run: - return model, messages, non_default_params - - # Extract the query from the last user message - query: Final = self._extract_query_from_messages(messages) - - if not query: - verbose_logger.debug("No query found in messages for vector store search") - return model, messages, non_default_params - - modified_messages: list[AllMessageValues] = messages.copy() - all_search_results: Final[list[VectorStoreSearchResponse]] = [] - - for vector_store_to_run in vector_stores_to_run: - # Get vector store id from the vector store config - vector_store_id = vector_store_to_run.get("vector_store_id", "") - custom_llm_provider = vector_store_to_run.get("custom_llm_provider") - litellm_params_for_vector_store = vector_store_to_run.get("litellm_params", {}) or {} - request_litellm_params = litellm_logging_obj.model_call_details.get("litellm_params", {}) - request_metadata = ( - request_litellm_params.get("metadata", {}) if isinstance(request_litellm_params, dict) else {} - ) - if llm_router is not None: - search_function = cast( # cast-ok: normalize router search callable - Callable[..., Awaitable[VectorStoreSearchResponse]], - llm_router.avector_store_search, - ) - else: - search_function = cast( # cast-ok: normalize SDK search callable - Callable[..., Awaitable[VectorStoreSearchResponse]], - litellm.vector_stores.asearch, - ) - try: - search_response = await search_function( - **{ - "vector_store_id": vector_store_id, - "query": query, - "custom_llm_provider": custom_llm_provider, - "metadata": request_metadata, - **litellm_params_for_vector_store, - }, - ) - except Exception as search_error: - verbose_logger.warning( - "Vector store search failed for vector_store_id=%s, continuing without its context: %s", - vector_store_id, - search_error, - ) - continue - - verbose_logger.debug("search_response: %s", search_response) - - # Store search results for later use in citations - all_search_results.append(search_response) - - # Process search results and append as context - modified_messages = self._append_search_results_to_messages( - messages=modified_messages, search_response=search_response - ) - - # Get the number of results for logging - num_results = 0 - num_results = len(search_response.get("data", []) or []) - verbose_logger.debug("Vector store search completed. Added context from %s results", num_results) - - # Store search results as-is (already in OpenAI-compatible format) - if litellm_logging_obj and all_search_results: - litellm_logging_obj.model_call_details["search_results"] = all_search_results - - return model, modified_messages, non_default_params - except Exception as e: - verbose_logger.exception("Error in VectorStorePreCallHook: %s", e) - # Return original parameters on error + verbose_logger.exception( + "Error in VectorStorePreCallHook for vector_store_ids=%s: %s", + requested_vector_store_ids, + e, + ) return model, messages, non_default_params - def _extract_query_from_messages(self, messages: list[AllMessageValues]) -> str | None: + if augmentation is None: + return model, messages, non_default_params + + for detail, value in ( + ("search_results", list(augmentation.search_results)), + (SEARCH_FAILURES_FIELD, augmentation.failures), + ): + if value: + litellm_logging_obj.model_call_details[detail] = value + + if augmentation.failures: + match litellm.vector_store_search_failure_mode: + case "error": + raise VectorStoreSearchError(failures=augmentation.failures, model=model) + case "annotate": + pass + case unreachable: + assert_never(unreachable) + + return model, list(augmentation.messages), non_default_params + + async def _augment_messages( + self, + messages: Sequence[AllMessageValues], + non_default_params: dict, + tools: list[dict] | None, + litellm_logging_obj: LiteLLMLoggingObj, + ) -> VectorStoreAugmentation | None: + if litellm.vector_store_registry is None: + return None + + prisma_client: Final = self.proxy_runtime.prisma_client() + llm_router: Final = self.proxy_runtime.llm_router() + + # Use database fallback to ensure synchronization across instances + vector_stores_to_run: Final[ + Sequence[LiteLLM_ManagedVectorStore] + ] = await litellm.vector_store_registry.pop_vector_stores_to_run_with_db_fallback( + non_default_params=non_default_params, + tools=tools, + prisma_client=prisma_client, + ) + + if not vector_stores_to_run: + return None + + query: Final = self._extract_query_from_messages(messages) + + if not query: + verbose_logger.debug("No query found in messages for vector store search") + return None + + request_litellm_params: Final = litellm_logging_obj.model_call_details.get("litellm_params", {}) + request_metadata: Final = ( + request_litellm_params.get("metadata", {}) if isinstance(request_litellm_params, dict) else {} + ) + search_function: Final = ( + cast( # cast-ok: normalize router search callable + Callable[..., Awaitable[VectorStoreSearchResponse]], + llm_router.avector_store_search, + ) + if llm_router is not None + else cast( # cast-ok: normalize SDK search callable + Callable[..., Awaitable[VectorStoreSearchResponse]], + litellm.vector_stores.asearch, + ) + ) + + outcomes: Final = tuple( + [ + await self._search_one( + vector_store=vector_store_to_run, + query=query, + request_metadata=request_metadata, + search_function=search_function, + ) + for vector_store_to_run in vector_stores_to_run + ] + ) + search_results: Final = tuple(outcome.response for outcome in outcomes if isinstance(outcome, SearchSucceeded)) + failures: Final = tuple(outcome.failure for outcome in outcomes if isinstance(outcome, SearchFailed)) + + return VectorStoreAugmentation( + messages=self._messages_with_context(messages=messages, search_results=search_results), + search_results=search_results, + failures=failures, + ) + + async def _search_one( + self, + vector_store: LiteLLM_ManagedVectorStore, + query: str, + request_metadata: Mapping[str, object], + search_function: Callable[..., Awaitable[VectorStoreSearchResponse]], + ) -> SearchOutcome: + vector_store_id: Final = vector_store.get("vector_store_id", "") + custom_llm_provider: Final = vector_store.get("custom_llm_provider") + litellm_params_for_vector_store: Final = vector_store.get("litellm_params", {}) or {} + try: + search_response: Final = await search_function( + **{ + "vector_store_id": vector_store_id, + "query": query, + "custom_llm_provider": custom_llm_provider, + "metadata": request_metadata, + **litellm_params_for_vector_store, + }, + ) + except Exception as search_error: + verbose_logger.warning( + "Vector store search failed for vector_store_id=%s, continuing without its context: %s", + vector_store_id, + search_error, + ) + return SearchFailed( + failure=VectorStoreSearchFailure( + vector_store_id=vector_store_id, + custom_llm_provider=custom_llm_provider, + error=str(search_error), + ) + ) + + verbose_logger.debug( + "Vector store search completed for vector_store_id=%s. Added context from %s results", + vector_store_id, + len(search_response.get("data", []) or []), + ) + return SearchSucceeded(response=search_response) + + def _extract_query_from_messages(self, messages: Sequence[AllMessageValues]) -> str | None: """ Extract the query from the last user message. @@ -223,48 +298,40 @@ class VectorStorePreCallHook(CustomLogger): return None - def _append_search_results_to_messages( + def _messages_with_context( self, - messages: list[AllMessageValues], - search_response: VectorStoreSearchResponse, - ) -> list[AllMessageValues]: - """ - Append search results as context to the messages. + messages: Sequence[AllMessageValues], + search_results: Sequence[VectorStoreSearchResponse], + ) -> tuple[AllMessageValues, ...]: + context_messages: Final = tuple( + context_message + for search_response in search_results + if (context_message := self._context_message(search_response)) is not None + ) + if not context_messages: + return tuple(messages) + return (*messages[:-1], *context_messages, *messages[-1:]) - Args: - messages: Original list of messages - search_response: Response from vector store search - - Returns: - Modified list of messages with context appended - """ - search_response_data: Final[list[VectorStoreSearchResult] | None] = search_response.get("data") + def _context_message(self, search_response: VectorStoreSearchResponse) -> AllMessageValues | None: + """Build the context message for one vector store's results, or None when it returned nothing usable.""" + search_response_data: Final[Sequence[VectorStoreSearchResult] | None] = search_response.get("data") if not search_response_data: - return messages + return None - context_content = self.CONTENT_PREFIX_STRING + context_texts: Final = tuple( + content_text + for result in search_response_data + for content_item in (result.get("content") or ()) + if (content_text := content_item.get("text")) + ) + if not context_texts: + return None - for result in search_response_data: - result_content: list[VectorStoreResultContent] | None = result.get("content") - if result_content: - for content_item in result_content: - content_text: str | None = content_item.get("text") - if content_text: - context_content += content_text + "\n\n" - - # Only add context if we found any content - if context_content != "Context:\n\n": - # Create a copy of messages to avoid modifying the original - modified_messages: Final = messages.copy() - # Add context as a new message before the last user message - context_message: Final[ChatCompletionUserMessage] = { - "role": "user", - "content": context_content, - } - modified_messages.insert(-1, cast(AllMessageValues, context_message)) - return modified_messages - - return messages + context_message: Final[ChatCompletionUserMessage] = { + "role": "user", + "content": self.CONTENT_PREFIX_STRING + "".join(f"{text}\n\n" for text in context_texts), + } + return cast(AllMessageValues, context_message) async def async_post_call_success_deployment_hook( self, @@ -287,34 +354,29 @@ class VectorStorePreCallHook(CustomLogger): verbose_logger.debug("No litellm_logging_obj in request_data") return None - verbose_logger.debug("model_call_details keys: %s", list(litellm_logging_obj.model_call_details.keys())) - # Get search results from model_call_details (already in OpenAI format) - search_results: Final[list[VectorStoreSearchResponse] | None] = litellm_logging_obj.model_call_details.get( - "search_results" + search_results: Final[Sequence[VectorStoreSearchResponse] | None] = ( + litellm_logging_obj.model_call_details.get("search_results") + ) + search_failures: Final[Sequence[VectorStoreSearchFailure] | None] = ( + litellm_logging_obj.model_call_details.get(SEARCH_FAILURES_FIELD) ) - verbose_logger.debug("Search results found: %s", search_results is not None) - - if not search_results: - verbose_logger.debug("No search results found") + if not search_results and not search_failures: + verbose_logger.debug("No search results or search failures found") return None # Add search results to response object if hasattr(response, "choices") and response.choices: for choice in response.choices: if hasattr(choice, "message") and choice.message: - # Get existing provider_specific_fields or create new dict provider_fields = getattr(choice.message, "provider_specific_fields", None) or {} - - # Add search results (already in OpenAI-compatible format) - provider_fields["search_results"] = search_results - - # Set the provider_specific_fields + if search_results: + provider_fields["search_results"] = search_results + if search_failures: + provider_fields[SEARCH_FAILURES_FIELD] = search_failures setattr(choice.message, "provider_specific_fields", provider_fields) - verbose_logger.debug("Added %s search results to response", len(search_results)) - # Return modified response return response @@ -339,29 +401,24 @@ class VectorStorePreCallHook(CustomLogger): verbose_logger.debug("VectorStorePreCallHook.async_post_call_streaming_deployment_hook called") # Get search results from model_call_details (already in OpenAI format) - search_results: Final[list[VectorStoreSearchResponse] | None] = request_data.get("search_results") + search_results: Final[Sequence[VectorStoreSearchResponse] | None] = request_data.get("search_results") + search_failures: Final[Sequence[VectorStoreSearchFailure] | None] = request_data.get(SEARCH_FAILURES_FIELD) - verbose_logger.debug("Search results found for streaming chunk: %s", search_results is not None) - - if not search_results: - verbose_logger.debug("No search results found for streaming chunk") + if not search_results and not search_failures: + verbose_logger.debug("No search results or search failures found for streaming chunk") return response_chunk # Add search results to streaming chunk if hasattr(response_chunk, "choices") and response_chunk.choices: for choice in response_chunk.choices: if hasattr(choice, "delta") and choice.delta: - # Get existing provider_specific_fields or create new dict provider_fields = getattr(choice.delta, "provider_specific_fields", None) or {} - - # Add search results (already in OpenAI-compatible format) - provider_fields["search_results"] = search_results - - # Set the provider_specific_fields + if search_results: + provider_fields["search_results"] = search_results + if search_failures: + provider_fields[SEARCH_FAILURES_FIELD] = search_failures choice.delta.provider_specific_fields = provider_fields - verbose_logger.debug("Added %s search results to streaming chunk", len(search_results)) - # Return modified chunk return response_chunk @@ -369,3 +426,10 @@ class VectorStorePreCallHook(CustomLogger): verbose_logger.exception("Error adding search results to streaming chunk: %s", e) # Don't fail the request if search results fail to be added return response_chunk + + +def _requested_vector_store_ids(non_default_params: Mapping[str, object]) -> tuple[str, ...]: + requested: Final = non_default_params.get("vector_store_ids") + if not isinstance(requested, (list, tuple)): + return () + return tuple(str(vector_store_id) for vector_store_id in requested) diff --git a/litellm/types/vector_stores.py b/litellm/types/vector_stores.py index 474c652ff3a..9051287b30a 100644 --- a/litellm/types/vector_stores.py +++ b/litellm/types/vector_stores.py @@ -4,7 +4,7 @@ from enum import Enum from typing import Any, Literal from pydantic import BaseModel -from typing_extensions import TypedDict +from typing_extensions import ReadOnly, TypedDict class SupportedVectorStoreIntegrations(str, Enum): @@ -95,6 +95,17 @@ class VectorStoreSearchResponse(TypedDict, total=False): data: list[VectorStoreSearchResult] | None +VectorStoreSearchFailureMode = Literal["annotate", "error"] + + +class VectorStoreSearchFailure(TypedDict): + """A configured vector store whose search failed, as reported back to the API caller""" + + vector_store_id: ReadOnly[str] + custom_llm_provider: ReadOnly[str | None] + error: ReadOnly[str] + + class VectorStoreSearchOptionalRequestParams(TypedDict, total=False): """TypedDict for Optional parameters supported by the vector store search API.""" diff --git a/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py b/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py index ae5cffd8ab0..64ee9474e44 100644 --- a/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py +++ b/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py @@ -12,6 +12,15 @@ from litellm.integrations.vector_store_integrations.vector_store_pre_call_hook i VectorStorePreCallHook, ) from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import ( + CallTypes, + Choices, + Delta, + Message, + ModelResponse, + ModelResponseStream, + StreamingChoices, +) from litellm.types.vector_stores import ( VectorStoreResultContent, VectorStoreSearchResponse, @@ -36,6 +45,18 @@ def _search_response(text: str) -> VectorStoreSearchResponse: ) +def _first_message(response: ModelResponse) -> Message: + choice = response.choices[0] + assert isinstance(choice, Choices) + return choice.message + + +@dataclass(frozen=True) +class ExplodingRegistry: + async def pop_vector_stores_to_run_with_db_fallback(self, **kwargs: object) -> list[LiteLLM_ManagedVectorStore]: + raise RuntimeError("the registry blew up") + + @dataclass class RecordingRouter: failing_vector_store_ids: frozenset[str] = frozenset() @@ -285,3 +306,181 @@ def test_the_default_runtime_follows_the_proxy_globals(monkeypatch: pytest.Monke assert runtime.llm_router() is router assert runtime.prisma_client() is prisma + + +@pytest.mark.asyncio +async def test_a_failing_vector_store_is_reported_back_to_the_caller( + registry_with: RegisterStores, +) -> None: + """Regression (LIT-6809): a silently dropped store left the caller with an un-augmented answer and no signal.""" + registry_with("vs-broken", "vs-healthy") + logging_obj = FakeLoggingObj({}) + + await _run_hook( + VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime(router=RecordingRouter(failing_vector_store_ids=frozenset({"vs-broken"}))) + ), + ["vs-broken", "vs-healthy"], + logging_obj, + ) + + response = ModelResponse(choices=[Choices(message=Message(content="an answer"))]) + await VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=None)).async_post_call_success_deployment_hook( + request_data={"litellm_logging_obj": logging_obj}, + response=response, + call_type=CallTypes.acompletion, + ) + + provider_specific_fields = _first_message(response).provider_specific_fields or {} + assert provider_specific_fields["vector_store_search_failures"] == ( + { + "vector_store_id": "vs-broken", + "custom_llm_provider": "bedrock", + "error": "litellm.BadRequestError: no healthy deployments for vs-broken", + }, + ) + assert len(provider_specific_fields["search_results"]) == 1 + + +@pytest.mark.asyncio +async def test_a_healthy_vector_store_alone_reports_no_failures(registry_with: RegisterStores) -> None: + registry_with("vs-healthy") + logging_obj = FakeLoggingObj({}) + + await _run_hook( + VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=RecordingRouter())), + ["vs-healthy"], + logging_obj, + ) + + response = ModelResponse(choices=[Choices(message=Message(content="an answer"))]) + await VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=None)).async_post_call_success_deployment_hook( + request_data={"litellm_logging_obj": logging_obj}, + response=response, + call_type=CallTypes.acompletion, + ) + + assert "vector_store_search_failures" not in (_first_message(response).provider_specific_fields or {}) + + +@pytest.mark.asyncio +async def test_a_failing_vector_store_is_reported_on_the_streaming_chunk(registry_with: RegisterStores) -> None: + registry_with("vs-broken") + logging_obj = FakeLoggingObj({}) + + await _run_hook( + VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime(router=RecordingRouter(failing_vector_store_ids=frozenset({"vs-broken"}))) + ), + ["vs-broken"], + logging_obj, + ) + + chunk = ModelResponseStream(choices=[StreamingChoices(delta=Delta(content="an answer"))]) + await VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime(router=None) + ).async_post_call_streaming_deployment_hook( + request_data=logging_obj.model_call_details, + response_chunk=chunk, + call_type=CallTypes.acompletion, + ) + + assert (chunk.choices[0].delta.provider_specific_fields or {})["vector_store_search_failures"] == ( + { + "vector_store_id": "vs-broken", + "custom_llm_provider": "bedrock", + "error": "litellm.BadRequestError: no healthy deployments for vs-broken", + }, + ) + + +@pytest.mark.asyncio +async def test_error_mode_fails_the_request_instead_of_answering_without_the_knowledge_base( + registry_with: RegisterStores, + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Regression (LIT-6809): opting in must turn an ungrounded answer into a 400 the caller can act on.""" + registry_with("vs-broken", "vs-healthy") + monkeypatch.setattr(litellm, "vector_store_search_failure_mode", "error") + + with pytest.raises(litellm.VectorStoreSearchError) as raised: + await _run_hook( + VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime( + router=RecordingRouter(failing_vector_store_ids=frozenset({"vs-broken"})) + ) + ), + ["vs-broken", "vs-healthy"], + FakeLoggingObj({}), + ) + + assert raised.value.status_code == 400 + assert raised.value.failures == ( + { + "vector_store_id": "vs-broken", + "custom_llm_provider": "bedrock", + "error": "litellm.BadRequestError: no healthy deployments for vs-broken", + }, + ) + assert "vs-broken: litellm.BadRequestError: no healthy deployments for vs-broken" in raised.value.message + + +@pytest.mark.asyncio +async def test_error_mode_leaves_a_fully_healthy_request_alone( + registry_with: RegisterStores, + monkeypatch: pytest.MonkeyPatch, +) -> None: + registry_with("vs-healthy") + monkeypatch.setattr(litellm, "vector_store_search_failure_mode", "error") + + _, messages, _ = await _run_hook( + VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=RecordingRouter())), + ["vs-healthy"], + FakeLoggingObj({}), + ) + + assert messages[0]["content"] == "Context:\n\ncontext from vs-healthy\n\n" + + +@pytest.mark.asyncio +async def test_error_mode_does_not_swallow_the_raise_in_the_hooks_own_catch_all( + registry_with: RegisterStores, + monkeypatch: pytest.MonkeyPatch, + warnings: list[logging.LogRecord], +) -> None: + """Regression (LIT-6809): the catch-all around the hook must not turn the opted-in failure back into a 200.""" + registry_with("vs-broken") + monkeypatch.setattr(litellm, "vector_store_search_failure_mode", "error") + + with pytest.raises(litellm.VectorStoreSearchError): + await _run_hook( + VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime( + router=RecordingRouter(failing_vector_store_ids=frozenset({"vs-broken"})) + ) + ), + ["vs-broken"], + FakeLoggingObj({}), + ) + + assert [record.levelname for record in warnings] == ["WARNING"] + + +@pytest.mark.asyncio +async def test_a_crash_outside_the_search_names_the_requested_vector_stores( + monkeypatch: pytest.MonkeyPatch, + warnings: list[logging.LogRecord], +) -> None: + """Regression (LIT-6809): the catch-all logged no store id, so an operator could not tell which store broke.""" + monkeypatch.setattr(litellm, "vector_store_registry", ExplodingRegistry()) + + _, messages, _ = await _run_hook( + VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=None)), + ["vs-one", "vs-two"], + FakeLoggingObj({}), + ) + + assert messages == [{"role": "user", "content": "what is litellm?"}] + assert [record.getMessage() for record in warnings] == [ + "Error in VectorStorePreCallHook for vector_store_ids=('vs-one', 'vs-two'): the registry blew up" + ] From 6499ca349f1cfe5bc4e92d6128a11c13aea3420c Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:11:22 -0700 Subject: [PATCH 004/157] fix(vector-stores): import assert_never from typing_extensions for Python 3.10 --- .../vector_store_integrations/vector_store_pre_call_hook.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index d5c893c4b20..ed125b915b4 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -7,7 +7,9 @@ It searches the vector store for relevant context and appends it to the messages from collections.abc import Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, Protocol, assert_never, cast +from typing import TYPE_CHECKING, Any, Final, Protocol, cast + +from typing_extensions import assert_never import litellm import litellm.vector_stores From e464e749e00ea680dbc59702218210ea772a59e6 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:23:28 -0700 Subject: [PATCH 005/157] fix(vector-stores): fall back to annotate when the failure mode is unrecognized litellm_settings keys are set on the litellm module with no allowlist, so a typo in vector_store_search_failure_mode reached assert_never and turned every vector-store request into a 500. Validate the configured value and fall back to the permissive default with a warning naming the supported modes. --- .../vector_store_pre_call_hook.py | 21 +++++++++++-- .../test_vector_store_pre_call_hook.py | 30 +++++++++++++++++++ 2 files changed, 49 insertions(+), 2 deletions(-) diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index ed125b915b4..274cb496b8f 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -7,8 +7,9 @@ It searches the vector store for relevant context and appends it to the messages from collections.abc import Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Any, Final, Protocol, cast +from typing import TYPE_CHECKING, Any, Final, Protocol, cast, get_args +from pydantic import TypeAdapter, ValidationError from typing_extensions import assert_never import litellm @@ -22,6 +23,7 @@ from litellm.types.utils import CallTypes, StandardCallbackDynamicParams from litellm.types.vector_stores import ( LiteLLM_ManagedVectorStore, VectorStoreSearchFailure, + VectorStoreSearchFailureMode, VectorStoreSearchResponse, VectorStoreSearchResult, ) @@ -34,6 +36,8 @@ else: LiteLLMLoggingObj = Any SEARCH_FAILURES_FIELD: Final = "vector_store_search_failures" +_DEFAULT_FAILURE_MODE: Final[VectorStoreSearchFailureMode] = "annotate" +_FAILURE_MODE_ADAPTER: Final = TypeAdapter(VectorStoreSearchFailureMode) class ProxyRuntime(Protocol): @@ -153,7 +157,7 @@ class VectorStorePreCallHook(CustomLogger): litellm_logging_obj.model_call_details[detail] = value if augmentation.failures: - match litellm.vector_store_search_failure_mode: + match _configured_failure_mode(): case "error": raise VectorStoreSearchError(failures=augmentation.failures, model=model) case "annotate": @@ -435,3 +439,16 @@ def _requested_vector_store_ids(non_default_params: Mapping[str, object]) -> tup if not isinstance(requested, (list, tuple)): return () return tuple(str(vector_store_id) for vector_store_id in requested) + + +def _configured_failure_mode() -> VectorStoreSearchFailureMode: + try: + return _FAILURE_MODE_ADAPTER.validate_python(litellm.vector_store_search_failure_mode) + except ValidationError: + verbose_logger.warning( + "Unsupported vector_store_search_failure_mode=%r, falling back to %r. Supported modes: %s", + litellm.vector_store_search_failure_mode, + _DEFAULT_FAILURE_MODE, + ", ".join(get_args(VectorStoreSearchFailureMode)), + ) + return _DEFAULT_FAILURE_MODE diff --git a/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py b/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py index 64ee9474e44..eae53becc14 100644 --- a/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py +++ b/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py @@ -425,6 +425,36 @@ async def test_error_mode_fails_the_request_instead_of_answering_without_the_kno assert "vs-broken: litellm.BadRequestError: no healthy deployments for vs-broken" in raised.value.message +@pytest.mark.asyncio +async def test_a_misspelled_failure_mode_annotates_instead_of_erroring_the_request( + registry_with: RegisterStores, + monkeypatch: pytest.MonkeyPatch, + warnings: list[logging.LogRecord], +) -> None: + """Regression (LIT-6809): litellm_settings takes any value, so a typo must not become a 500.""" + registry_with("vs-broken") + monkeypatch.setattr(litellm, "vector_store_search_failure_mode", "erorr") + + logging_obj = FakeLoggingObj({}) + _, messages, _ = await _run_hook( + VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime(router=RecordingRouter(failing_vector_store_ids=frozenset({"vs-broken"}))) + ), + ["vs-broken"], + logging_obj, + ) + + assert messages[0]["content"] == "what is litellm?" + assert logging_obj.model_call_details["vector_store_search_failures"] == ( + { + "vector_store_id": "vs-broken", + "custom_llm_provider": "bedrock", + "error": "litellm.BadRequestError: no healthy deployments for vs-broken", + }, + ) + assert any("erorr" in record.getMessage() for record in warnings) + + @pytest.mark.asyncio async def test_error_mode_leaves_a_fully_healthy_request_alone( registry_with: RegisterStores, From 2647c2890fc6b8a32ab5601ecd8ed04b1cc0ebff Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:25:32 -0700 Subject: [PATCH 006/157] fix(oci): keep the stream's own finish reason instead of a synthetic stop chunk The OCI wrapper overrides chunk_creator wholesale, so it never recorded the finish reason or marked the terminal chunk as sent. The shared end-of-stream finalizer then appended a synthetic chunk whose finish_reason was always stop, which downgraded a tool_calls completion for any client that reads the finish reason off the last chunk. --- litellm/llms/oci/chat/transformation.py | 12 +++-- .../oci/chat/test_oci_chat_transformation.py | 54 +++++++++++++++++++ 2 files changed, 63 insertions(+), 3 deletions(-) diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index b1167d447fc..8e4e41b4ac1 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -745,7 +745,13 @@ class OCIStreamWrapper(CustomStreamWrapper): # single-event case (terminal chunk carries the only copy of the text). self._cohere_text_emitted = False - def _with_stream_identity(self, parsed: ModelResponseStream) -> ModelResponseStream: + def _emit_chunk(self, parsed: ModelResponseStream) -> ModelResponseStream: + for choice in parsed.choices: + if getattr(choice.delta, "tool_calls", None): + self.tool_call = True + if choice.finish_reason is not None: + self.received_finish_reason = choice.finish_reason + self.sent_last_chunk = True return self.model_response_creator(chunk={"choices": parsed.choices}) def chunk_creator(self, chunk: Any) -> ModelResponseStream | None: @@ -780,8 +786,8 @@ class OCIStreamWrapper(CustomStreamWrapper): if getattr(choice.delta, "content", None): self._cohere_text_emitted = True break - return self._with_stream_identity(result) - return self._with_stream_identity(handle_generic_stream_chunk(dict_chunk)) + return self._emit_chunk(result) + return self._emit_chunk(handle_generic_stream_chunk(dict_chunk)) __all__ = [ diff --git a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py index e77b2c24d01..4c9bd29b337 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py @@ -2041,3 +2041,57 @@ class TestOCIStreamWrapperDoneSentinel: wrapper = _make_stream_wrapper(_STREAM_GENERIC_MODEL) with pytest.raises(OCIError, match="Chunk cannot be parsed as JSON"): wrapper.chunk_creator("data: [DONE] trailing garbage") + + +_GENERIC_TOOL_CALL_EVENT = ( + 'data: {"index":0,"message":{"role":"ASSISTANT","content":[],' + '"toolCalls":[{"type":"FUNCTION","id":"call_1","name":"get_weather","arguments":"{}"}]}}' +) +_GENERIC_TOOL_TERMINAL_EVENT = ( + 'data: {"index":0,"message":{"role":"ASSISTANT","content":[]},"finishReason":"TOOL_CALLS"}' +) + + +def _drain_stream(model: str, events: list[str]) -> list: + logging_obj = MagicMock() + logging_obj.model_call_details = {"custom_llm_provider": "oci", "litellm_params": {}} + wrapper = OCIStreamWrapper( + completion_stream=iter(events), + model=model, + custom_llm_provider="oci", + logging_obj=logging_obj, + ) + return list(wrapper) + + +class TestOCIStreamWrapperTerminalChunk: + """OCI's ``chunk_creator`` override bypasses the shared handler's + finish-reason bookkeeping, so the shared end-of-stream finalizer used to + append a synthetic ``stop`` chunk after OCI's own terminal chunk, silently + downgrading a ``tool_calls`` completion for any client that reads the + finish reason off the last chunk.""" + + def test_generic_tool_call_stream_ends_on_tool_calls(self): + chunks = _drain_stream( + _STREAM_GENERIC_MODEL, + [_GENERIC_TOOL_CALL_EVENT, _GENERIC_TOOL_TERMINAL_EVENT, "data: [DONE]"], + ) + + assert [chunk.choices[0].finish_reason for chunk in chunks] == [None, "tool_calls"] + assert len({chunk.id for chunk in chunks}) == 1 + + def test_generic_text_stream_emits_exactly_one_finish_reason(self): + chunks = _drain_stream( + _STREAM_GENERIC_MODEL, + [_GENERIC_TEXT_EVENT.format(text="1"), _GENERIC_TERMINAL_EVENT, "data: [DONE]"], + ) + + assert [chunk.choices[0].finish_reason for chunk in chunks] == [None, "stop"] + + def test_cohere_stream_emits_exactly_one_finish_reason(self): + chunks = _drain_stream( + _STREAM_COHERE_MODEL, + [_COHERE_TEXT_EVENT.format(text="123"), _COHERE_TERMINAL_EVENT], + ) + + assert [chunk.choices[0].finish_reason for chunk in chunks] == [None, "stop"] From b55833b1f9269753fa35f5f6e68fc9782b83936f Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:40:00 -0700 Subject: [PATCH 007/157] fix(ai-gateway): build the release image again and cover it in CI The image had two independent breaks. The Dockerfile pinned rust 1.90 while the repo pins 1.98 in rust-toolchain.toml and never copied it in, so the first cargo call died on crates needing a newer rustc. The runtime stage then ran pip install on the root pyproject, which builds with maturin against the python-bridge crate, so metadata generation failed with no Cargo manifest and no Rust toolchain in that stage. Copy rust-toolchain.toml into the builder so every cargo call uses the pinned channel, build the wheel in the builder stage where cargo and python3-dev already live, and have the runtime stage install that artifact instead of compiling anything. Add the ai-gateway image job to the rust workflow so a broken build fails a PR instead of surfacing on a release. --- .github/workflows/test-rust.yml | 39 +++++++++++++++++++ litellm-rust/crates/ai-gateway/Dockerfile | 31 ++++++++++----- .../crates/ai-gateway/Dockerfile.dockerignore | 6 ++- 3 files changed, 66 insertions(+), 10 deletions(-) diff --git a/.github/workflows/test-rust.yml b/.github/workflows/test-rust.yml index 9b8b132df62..99de9a25fbe 100644 --- a/.github/workflows/test-rust.yml +++ b/.github/workflows/test-rust.yml @@ -133,3 +133,42 @@ jobs: - name: Test native route wheel run: python tests/test_litellm/rust_bridge/native_route_wheel_test.py dist/*.whl + + ai-gateway-image: + name: ai-gateway release image + runs-on: ubuntu-latest + timeout-minutes: 60 + permissions: + contents: read + + steps: + - name: Checkout repository + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Build the release image + run: docker build -f litellm-rust/crates/ai-gateway/Dockerfile -t litellm-ai-gateway:${{ github.sha }} . + + - name: Start the gateway and wait for readiness + env: + IMAGE: litellm-ai-gateway:${{ github.sha }} + run: | + docker run -d --name ai-gateway -p 4001:4001 \ + -e LITELLM_MASTER_KEY=sk-ci-not-a-real-key \ + -e OPENAI_API_KEY=sk-ci-not-a-real-key \ + "$IMAGE" + for _ in $(seq 1 60); do + if curl -fsS http://127.0.0.1:4001/health/readiness; then + echo "gateway is serving readiness" + exit 0 + fi + sleep 2 + done + echo "gateway never became ready" >&2 + docker logs ai-gateway >&2 + exit 1 + + - name: Stop the gateway + if: always() + run: docker rm -f ai-gateway || true diff --git a/litellm-rust/crates/ai-gateway/Dockerfile b/litellm-rust/crates/ai-gateway/Dockerfile index 2bc3c05ad7e..370d6b3f08f 100644 --- a/litellm-rust/crates/ai-gateway/Dockerfile +++ b/litellm-rust/crates/ai-gateway/Dockerfile @@ -14,15 +14,20 @@ # ---- Chef ------------------------------------------------------------------- # cargo-chef caches the dependency build so only the gateway crate recompiles on # a source-only change. python3-dev is present in every rust stage because the -# `python-config` feature links libpython via pyo3 (even in the cook step). -FROM rust:1.90-slim-bookworm AS chef +# `python-config` feature links libpython via pyo3 (even in the cook step), and +# python3-pip builds the litellm wheel in the builder stage. +FROM rust:1.98-slim-bookworm AS chef ENV PYO3_PYTHON=python3.11 +# rustup reads rust-toolchain.toml from any parent of the working directory, so +# copying it in is what keeps every cargo call below on the repo's pinned +# channel rather than on whatever the base image happens to ship. +COPY rust-toolchain.toml /build/rust-toolchain.toml +WORKDIR /build/litellm-rust RUN apt-get update \ && apt-get install -y --no-install-recommends \ - python3 python3-dev pkg-config libssl-dev clang \ + python3 python3-dev python3-pip pkg-config libssl-dev clang \ && rm -rf /var/lib/apt/lists/* \ && cargo install cargo-chef --locked --version 0.1.77 -WORKDIR /build/litellm-rust # ---- Planner ---------------------------------------------------------------- # Produce the dependency recipe from the rust workspace manifests + Cargo.lock. @@ -43,6 +48,14 @@ RUN cargo chef cook --locked --release \ COPY litellm-rust/ . RUN cargo build --locked --release -p litellm-ai-gateway --bin litellm-ai-gateway --features server,python-config +# The root pyproject builds with maturin against litellm-rust/crates/python-bridge, +# so the wheel is built here, next to the crate sources and the cargo toolchain, +# and the runtime stage installs the artifact instead of compiling anything. +COPY pyproject.toml README.md LICENSE /build/ +COPY litellm/ /build/litellm/ +COPY enterprise/ /build/enterprise/ +RUN pip3 wheel --no-cache-dir --no-deps --wheel-dir /build/dist /build + # ---- Runtime ---------------------------------------------------------------- # python:3.11-slim-bookworm ships libpython3.11, matching the builder's PyO3 # 3.11 ABI so the embedded interpreter links and imports cleanly. @@ -56,11 +69,11 @@ RUN apt-get update \ WORKDIR /app # Install litellm (with proxy extras) FROM THIS REPO'S SOURCE so -# `import litellm.proxy.read_model_list` works — it is not on PyPI yet. Copy the -# package + packaging metadata, then pip install the proxy extra. -COPY pyproject.toml README.md LICENSE ./ -COPY litellm/ ./litellm/ -RUN pip install --no-cache-dir ".[proxy]" +# `import litellm.proxy.read_model_list` works — it is not on PyPI yet. +COPY --from=builder /build/dist/*.whl /tmp/wheels/ +RUN wheel="$(ls /tmp/wheels/litellm-*.whl)" \ + && pip install --no-cache-dir "${wheel}[proxy]" \ + && rm -rf /tmp/wheels # The compiled gateway binary (pure-Rust realtime hot path; Python is load-time # only). diff --git a/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore b/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore index 030ee6a37c5..4fec13d18db 100644 --- a/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore +++ b/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore @@ -9,13 +9,17 @@ # Strategy: ignore everything, then re-include only what the build needs: # - litellm/ (pip install . needs the full package + proxy reader) # - litellm-rust/ (the rust workspace; Cargo.lock + crate sources) -# - pyproject.toml / README.md / LICENSE (packaging metadata for pip install) +# - enterprise/ (litellm/proxy/enterprise symlinks into it; maturin walks it) +# - pyproject.toml / README.md / LICENSE (packaging metadata for the wheel build) +# - rust-toolchain.toml (the pinned channel every cargo call in the build uses) * # --- re-include the build inputs --- !litellm/ !litellm-rust/ +!enterprise/ !pyproject.toml +!rust-toolchain.toml !README.md !LICENSE From fbb799e240eec1e4d8c41d9c905d3699b7bfaad7 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:44:20 -0700 Subject: [PATCH 008/157] ci: drop the ai-gateway image from the coverage allowlist now that a job builds it --- .github/ci-coverage-allowlist.yml | 5 ----- 1 file changed, 5 deletions(-) diff --git a/.github/ci-coverage-allowlist.yml b/.github/ci-coverage-allowlist.yml index f7a785b3b80..810265f4b40 100644 --- a/.github/ci-coverage-allowlist.yml +++ b/.github/ci-coverage-allowlist.yml @@ -96,11 +96,6 @@ dockerfiles: and lint workflows already exercise that output, so building the image adds no signal about it paths: - ui/Dockerfile - - reason: >- - The Rust gateway ships as its own chart and package with a separate release pipeline, so its - image is not part of this repo's Python image set - paths: - - litellm-rust/crates/ai-gateway/Dockerfile - reason: >- An example image under cookbook/ that is documentation rather than a shipped artifact paths: From 1d1eb4264f2ea33beeda9f4a4b84c45ce35ebb0e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:49:40 -0700 Subject: [PATCH 009/157] build(ai-gateway): keep the committed enterprise wheels out of the build context --- litellm-rust/crates/ai-gateway/Dockerfile.dockerignore | 2 ++ 1 file changed, 2 insertions(+) diff --git a/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore b/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore index 4fec13d18db..c8e789b5e8c 100644 --- a/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore +++ b/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore @@ -26,6 +26,8 @@ # --- prune heavy / irrelevant subpaths back out of the re-included trees --- # Rust build artifacts (huge; regenerated in the builder). **/target/ +# Committed python distribution artifacts; the wheel build does not read them. +enterprise/dist/ # Python caches and compiled bytecode. **/__pycache__/ **/*.pyc From dc28f0eb4b9aeee5933368055daeb3a09d5fbab1 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:54:09 -0700 Subject: [PATCH 010/157] fix(vector-stores): report search failures on the responses API surface too --- .../vector_store_pre_call_hook.py | 11 +++- .../test_vector_store_pre_call_hook.py | 55 ++++++++++++++++++- 2 files changed, 64 insertions(+), 2 deletions(-) diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index 274cb496b8f..849e87fae40 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -17,7 +17,11 @@ import litellm.vector_stores from litellm._logging import verbose_logger from litellm.exceptions import VectorStoreSearchError from litellm.integrations.custom_logger import CustomLogger -from litellm.types.llms.openai import AllMessageValues, ChatCompletionUserMessage +from litellm.types.llms.openai import ( + AllMessageValues, + ChatCompletionUserMessage, + ResponsesAPIResponse, +) from litellm.types.prompts.init_prompts import PromptSpec from litellm.types.utils import CallTypes, StandardCallbackDynamicParams from litellm.types.vector_stores import ( @@ -372,6 +376,11 @@ class VectorStorePreCallHook(CustomLogger): verbose_logger.debug("No search results or search failures found") return None + if isinstance(response, ResponsesAPIResponse): + if search_failures: + setattr(response, SEARCH_FAILURES_FIELD, list(search_failures)) + return response + # Add search results to response object if hasattr(response, "choices") and response.choices: for choice in response.choices: diff --git a/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py b/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py index eae53becc14..f1f9f7c3f3f 100644 --- a/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py +++ b/tests/test_litellm/integrations/vector_store_integrations/test_vector_store_pre_call_hook.py @@ -11,7 +11,7 @@ from litellm.integrations.vector_store_integrations.vector_store_pre_call_hook i ProxyServerRuntime, VectorStorePreCallHook, ) -from litellm.types.llms.openai import AllMessageValues +from litellm.types.llms.openai import AllMessageValues, ResponsesAPIResponse from litellm.types.utils import ( CallTypes, Choices, @@ -363,6 +363,59 @@ async def test_a_healthy_vector_store_alone_reports_no_failures(registry_with: R assert "vector_store_search_failures" not in (_first_message(response).provider_specific_fields or {}) +@pytest.mark.asyncio +async def test_a_failing_vector_store_is_reported_on_the_responses_api_response( + registry_with: RegisterStores, +) -> None: + """Regression (LIT-6809): /v1/responses answered 200 with no sign the knowledge base was missing.""" + registry_with("vs-broken") + logging_obj = FakeLoggingObj({}) + + await _run_hook( + VectorStorePreCallHook( + proxy_runtime=FakeProxyRuntime(router=RecordingRouter(failing_vector_store_ids=frozenset({"vs-broken"}))) + ), + ["vs-broken"], + logging_obj, + ) + + response = ResponsesAPIResponse(id="resp-lit6809", created_at=0, output=[]) + await VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=None)).async_post_call_success_deployment_hook( + request_data={"litellm_logging_obj": logging_obj}, + response=response, + call_type=CallTypes.aresponses, + ) + + assert response.model_dump()["vector_store_search_failures"] == [ + { + "vector_store_id": "vs-broken", + "custom_llm_provider": "bedrock", + "error": "litellm.BadRequestError: no healthy deployments for vs-broken", + } + ] + + +@pytest.mark.asyncio +async def test_a_healthy_vector_store_leaves_the_responses_api_response_alone(registry_with: RegisterStores) -> None: + registry_with("vs-healthy") + logging_obj = FakeLoggingObj({}) + + await _run_hook( + VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=RecordingRouter())), + ["vs-healthy"], + logging_obj, + ) + + response = ResponsesAPIResponse(id="resp-lit6809", created_at=0, output=[]) + await VectorStorePreCallHook(proxy_runtime=FakeProxyRuntime(router=None)).async_post_call_success_deployment_hook( + request_data={"litellm_logging_obj": logging_obj}, + response=response, + call_type=CallTypes.aresponses, + ) + + assert "vector_store_search_failures" not in response.model_dump() + + @pytest.mark.asyncio async def test_a_failing_vector_store_is_reported_on_the_streaming_chunk(registry_with: RegisterStores) -> None: registry_with("vs-broken") From 376f74c058cdaa00cf5abcda1ba83fb23615c801 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 4 Sep 2026 17:26:45 -0700 Subject: [PATCH 011/157] refactor(vector_stores): bind the failure mode before matching on it basedpyright counts a named capture after an exhaustive match under reportUnnecessaryComparison, which pushed the merged tree one over the budget; the wildcard case with assert_never on the bound subject is the shape the rest of the codebase uses --- .../vector_store_pre_call_hook.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index 849e87fae40..636e076a229 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -161,13 +161,14 @@ class VectorStorePreCallHook(CustomLogger): litellm_logging_obj.model_call_details[detail] = value if augmentation.failures: - match _configured_failure_mode(): + failure_mode: Final = _configured_failure_mode() + match failure_mode: case "error": raise VectorStoreSearchError(failures=augmentation.failures, model=model) case "annotate": pass - case unreachable: - assert_never(unreachable) + case _: + assert_never(failure_mode) return model, list(augmentation.messages), non_default_params From c4d09a31e0631f03c87fe27c1fd6dcfce16e2c6c Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 4 Sep 2026 20:20:07 -0700 Subject: [PATCH 012/157] fix(vector_stores): default the search-count debug log to a tuple so LIT002 stays within budget --- .../vector_store_integrations/vector_store_pre_call_hook.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index 636e076a229..b2243060c6c 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -276,7 +276,7 @@ class VectorStorePreCallHook(CustomLogger): verbose_logger.debug( "Vector store search completed for vector_store_id=%s. Added context from %s results", vector_store_id, - len(search_response.get("data", []) or []), + len(search_response.get("data") or ()), ) return SearchSucceeded(response=search_response) From 025b9cb75168233ef06f17078e7159d6b3ae6f08 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:52:10 -0700 Subject: [PATCH 013/157] build(ai-gateway): build the sibling wheels from the repo so the image never waits on PyPI --- litellm-rust/crates/ai-gateway/Dockerfile | 16 +++++++++++++--- .../crates/ai-gateway/Dockerfile.dockerignore | 3 +++ 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/litellm-rust/crates/ai-gateway/Dockerfile b/litellm-rust/crates/ai-gateway/Dockerfile index 370d6b3f08f..72ac25ce1d6 100644 --- a/litellm-rust/crates/ai-gateway/Dockerfile +++ b/litellm-rust/crates/ai-gateway/Dockerfile @@ -51,10 +51,15 @@ RUN cargo build --locked --release -p litellm-ai-gateway --bin litellm-ai-gatewa # The root pyproject builds with maturin against litellm-rust/crates/python-bridge, # so the wheel is built here, next to the crate sources and the cargo toolchain, # and the runtime stage installs the artifact instead of compiling anything. +# litellm[proxy] pins litellm-enterprise and litellm-proxy-extras to the versions +# in this repo, and those hit PyPI hours after every version bump merges, so both +# wheels are built from the repo too instead of being resolved from PyPI. COPY pyproject.toml README.md LICENSE /build/ COPY litellm/ /build/litellm/ COPY enterprise/ /build/enterprise/ -RUN pip3 wheel --no-cache-dir --no-deps --wheel-dir /build/dist /build +COPY litellm-proxy-extras/ /build/litellm-proxy-extras/ +RUN pip3 wheel --no-cache-dir --no-deps --wheel-dir /build/dist \ + /build /build/enterprise /build/litellm-proxy-extras # ---- Runtime ---------------------------------------------------------------- # python:3.11-slim-bookworm ships libpython3.11, matching the builder's PyO3 @@ -69,10 +74,15 @@ RUN apt-get update \ WORKDIR /app # Install litellm (with proxy extras) FROM THIS REPO'S SOURCE so -# `import litellm.proxy.read_model_list` works — it is not on PyPI yet. +# `import litellm.proxy.read_model_list` works — it is not on PyPI yet. The two +# sibling wheels come from the builder as well, so the pins in litellm[proxy] +# resolve against them and never wait on a PyPI publish. COPY --from=builder /build/dist/*.whl /tmp/wheels/ RUN wheel="$(ls /tmp/wheels/litellm-*.whl)" \ - && pip install --no-cache-dir "${wheel}[proxy]" \ + && pip install --no-cache-dir \ + /tmp/wheels/litellm_enterprise-*.whl \ + /tmp/wheels/litellm_proxy_extras-*.whl \ + "${wheel}[proxy]" \ && rm -rf /tmp/wheels # The compiled gateway binary (pure-Rust realtime hot path; Python is load-time diff --git a/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore b/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore index c8e789b5e8c..d1386ff684d 100644 --- a/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore +++ b/litellm-rust/crates/ai-gateway/Dockerfile.dockerignore @@ -10,6 +10,7 @@ # - litellm/ (pip install . needs the full package + proxy reader) # - litellm-rust/ (the rust workspace; Cargo.lock + crate sources) # - enterprise/ (litellm/proxy/enterprise symlinks into it; maturin walks it) +# - litellm-proxy-extras/ (built into a wheel alongside enterprise/ for litellm[proxy]) # - pyproject.toml / README.md / LICENSE (packaging metadata for the wheel build) # - rust-toolchain.toml (the pinned channel every cargo call in the build uses) * @@ -18,6 +19,7 @@ !litellm/ !litellm-rust/ !enterprise/ +!litellm-proxy-extras/ !pyproject.toml !rust-toolchain.toml !README.md @@ -28,6 +30,7 @@ **/target/ # Committed python distribution artifacts; the wheel build does not read them. enterprise/dist/ +litellm-proxy-extras/dist/ # Python caches and compiled bytecode. **/__pycache__/ **/*.pyc From d1f76e81d25b47887e85204ad6a24c103e8c79c0 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 5 Sep 2026 17:14:32 -0700 Subject: [PATCH 014/157] ci(ai-gateway): build the release image in its own workflow and assert the config loads --- .github/workflows/ai-gateway-image.yml | 73 ++++++++++++++++++++++++++ .github/workflows/test-rust.yml | 39 -------------- 2 files changed, 73 insertions(+), 39 deletions(-) create mode 100644 .github/workflows/ai-gateway-image.yml diff --git a/.github/workflows/ai-gateway-image.yml b/.github/workflows/ai-gateway-image.yml new file mode 100644 index 00000000000..3f690f566b0 --- /dev/null +++ b/.github/workflows/ai-gateway-image.yml @@ -0,0 +1,73 @@ +name: ai-gateway image + +on: + push: + paths: + - "litellm-rust/**" + - "litellm/**" + - "enterprise/**" + - "litellm-proxy-extras/**" + - "pyproject.toml" + - "rust-toolchain.toml" + - ".github/workflows/ai-gateway-image.yml" + pull_request: + branches: + - main + - litellm_internal_staging + - litellm_oss_staging + - "litellm_**" + paths: + - "litellm-rust/**" + - "litellm/**" + - "enterprise/**" + - "litellm-proxy-extras/**" + - "pyproject.toml" + - "rust-toolchain.toml" + - ".github/workflows/ai-gateway-image.yml" + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + ai-gateway-image: + name: ai-gateway release image + runs-on: ubuntu-latest + timeout-minutes: 60 + permissions: + contents: read + steps: + - name: Checkout repository + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + - name: Build the release image + run: docker build -f litellm-rust/crates/ai-gateway/Dockerfile -t litellm-ai-gateway:${{ github.sha }} . + - name: Start the gateway and wait for readiness + env: + IMAGE: litellm-ai-gateway:${{ github.sha }} + run: | + docker run -d --name ai-gateway -p 4001:4001 \ + -e LITELLM_MASTER_KEY=sk-ci-not-a-real-key \ + -e OPENAI_API_KEY=sk-ci-not-a-real-key \ + "$IMAGE" + for _ in $(seq 1 60); do + if curl -fsS http://127.0.0.1:4001/health/readiness; then + echo "gateway is serving readiness" + exit 0 + fi + sleep 2 + done + echo "gateway never became ready" >&2 + docker logs ai-gateway >&2 + exit 1 + - name: Assert the gateway loaded the baked config + run: | + docker logs ai-gateway 2>&1 | tee gateway.log + grep 'via python config reader' gateway.log + - name: Stop the gateway + if: always() + run: docker rm -f ai-gateway || true diff --git a/.github/workflows/test-rust.yml b/.github/workflows/test-rust.yml index 99de9a25fbe..9b8b132df62 100644 --- a/.github/workflows/test-rust.yml +++ b/.github/workflows/test-rust.yml @@ -133,42 +133,3 @@ jobs: - name: Test native route wheel run: python tests/test_litellm/rust_bridge/native_route_wheel_test.py dist/*.whl - - ai-gateway-image: - name: ai-gateway release image - runs-on: ubuntu-latest - timeout-minutes: 60 - permissions: - contents: read - - steps: - - name: Checkout repository - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Build the release image - run: docker build -f litellm-rust/crates/ai-gateway/Dockerfile -t litellm-ai-gateway:${{ github.sha }} . - - - name: Start the gateway and wait for readiness - env: - IMAGE: litellm-ai-gateway:${{ github.sha }} - run: | - docker run -d --name ai-gateway -p 4001:4001 \ - -e LITELLM_MASTER_KEY=sk-ci-not-a-real-key \ - -e OPENAI_API_KEY=sk-ci-not-a-real-key \ - "$IMAGE" - for _ in $(seq 1 60); do - if curl -fsS http://127.0.0.1:4001/health/readiness; then - echo "gateway is serving readiness" - exit 0 - fi - sleep 2 - done - echo "gateway never became ready" >&2 - docker logs ai-gateway >&2 - exit 1 - - - name: Stop the gateway - if: always() - run: docker rm -f ai-gateway || true From 8c5b29519e6609d743913bc8d3c85070fc5cdf16 Mon Sep 17 00:00:00 2001 From: Jon Walton Date: Wed, 9 Sep 2026 18:15:53 +0800 Subject: [PATCH 015/157] fix(proxy): emit internal user budget alerts --- .../SlackAlerting/slack_alerting.py | 2 +- litellm/proxy/auth/auth_checks.py | 13 ++ .../SlackAlerting/test_slack_alerting.py | 15 ++ .../proxy/auth/test_auth_checks.py | 131 +++++++++++++++++- 4 files changed, 158 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/SlackAlerting/slack_alerting.py b/litellm/integrations/SlackAlerting/slack_alerting.py index dc41c7dadc8..caac8e888fd 100644 --- a/litellm/integrations/SlackAlerting/slack_alerting.py +++ b/litellm/integrations/SlackAlerting/slack_alerting.py @@ -673,7 +673,7 @@ class SlackAlerting(CustomBatchLogger): Create a standard message for a budget alert """ _all_fields_as_dict: Final[dict[str, object]] = user_info.model_dump(exclude_none=True) - _all_fields_as_dict.pop("token") + _all_fields_as_dict.pop("token", None) msg = "" for k, v in _all_fields_as_dict.items(): if isinstance(v, Litellm_EntityType): diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index dc693317de0..d26474778d9 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -1020,6 +1020,19 @@ async def common_checks( fallback_spend=user_object.spend or 0.0, max_budget=user_budget, ) + call_info: Final = CallInfo( + spend=user_spend, + max_budget=user_budget, + user_id=user_object.user_id, + user_email=user_object.user_email, + event_group=Litellm_EntityType.USER, + ) + asyncio.create_task( + proxy_logging_obj.budget_alerts( + type="user_budget", + user_info=call_info, + ) + ) if math.isfinite(user_budget) and user_spend >= user_budget: raise litellm.BudgetExceededError( current_cost=user_spend, diff --git a/tests/test_litellm/integrations/SlackAlerting/test_slack_alerting.py b/tests/test_litellm/integrations/SlackAlerting/test_slack_alerting.py index 55e2dcdc270..44dda57dd27 100644 --- a/tests/test_litellm/integrations/SlackAlerting/test_slack_alerting.py +++ b/tests/test_litellm/integrations/SlackAlerting/test_slack_alerting.py @@ -45,6 +45,21 @@ class TestSlackAlerting(unittest.TestCase): result = self.slack_alerting._get_percent_of_max_budget_left(user_info) self.assertEqual(result, -0.2) + def test_get_user_info_str_omits_absent_token_for_user_alert(self): + user_info = CallInfo( + spend=85.0, + max_budget=100.0, + user_id="user-1", + user_email="person@example.com", + event_group=Litellm_EntityType.USER, + ) + + result = self.slack_alerting._get_user_info_str(user_info) + + self.assertIn("*user_id:* `user-1`", result) + self.assertIn("*user_email:* `person@example.com`", result) + self.assertNotIn("*token:*", result) + def test_get_event_and_event_message_max_budget(self): # Initial setup with no event event = None diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index 2284a05b2e9..1fc2e0b3dc1 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -1,7 +1,7 @@ import asyncio import json from types import SimpleNamespace -from typing import TYPE_CHECKING, Optional +from typing import TYPE_CHECKING, Final, Literal, Optional from unittest.mock import AsyncMock, MagicMock, patch if TYPE_CHECKING: @@ -29,6 +29,7 @@ from litellm.proxy._types import ( ProxyException, SSOUserDefinedValues, UserAPIKeyAuth, + WebhookEvent, ) from litellm.proxy.auth.auth_checks import ( ExperimentalUIJWTToken, @@ -52,6 +53,7 @@ from litellm.proxy.auth.auth_checks import ( ) from litellm.caching.in_memory_cache import InMemoryCache from litellm.caching.redis_cache import RedisCache +from litellm.integrations.SlackAlerting.slack_alerting import SlackAlerting from litellm.constants import ( DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL, END_USER_RESTRICTED_REGISTRY_MAX_SIZE, @@ -5375,6 +5377,127 @@ async def test_common_checks_personal_user_budget_blocks_in_gather(): assert "User=u1" in str(over.value) +async def _run_internal_user_budget_alert( + *, + spend: float, +) -> tuple[AsyncMock, litellm.BudgetExceededError | None]: + from fastapi import Request + + from litellm.proxy.auth.auth_checks import common_checks + + user: Final = LiteLLM_UserTable( + user_id="user-1", + user_email="person@example.com", + spend=0.0, + max_budget=100.0, + ) + token: Final = UserAPIKeyAuth(token="hashed-key-1", user_id="user-1") + slack_alerting: Final = SlackAlerting(alerting=["webhook"]) + send_alert: Final = AsyncMock() + alert_finished: Final = asyncio.Event() + + async def _get_spend( + counter_key: str, + fallback_spend: float, + max_budget: float | None = None, + **kwargs: object, + ) -> float: + assert counter_key == "spend:user:user-1" + assert fallback_spend == 0.0 + assert max_budget == 100.0 + return spend + + async def _budget_alerts( + *, + type: Literal["user_budget"], + user_info: CallInfo, + ) -> None: + try: + await slack_alerting.budget_alerts(type=type, user_info=user_info) + finally: + alert_finished.set() + + proxy_logging_obj: Final = MagicMock(budget_alerts=_budget_alerts) + + async def _check() -> bool: + return await common_checks( + request_body={"messages": [{"role": "user", "content": "hi"}]}, + team_object=None, + user_object=user, + end_user_object=None, + global_proxy_spend=None, + general_settings={}, + route="/chat/completions", + llm_router=None, + proxy_logging_obj=proxy_logging_obj, + valid_token=token, + request=MagicMock(spec=Request), + ) + + async def _check_for_error() -> litellm.BudgetExceededError | None: + if spend < 100.0: + assert await _check() is True + return None + + with pytest.raises(litellm.BudgetExceededError) as raised: + await _check() + return raised.value + + with ( + patch("litellm.proxy.proxy_server.prisma_client", None), # test-quality-ok: common_checks has no database seam + patch("litellm.proxy.proxy_server.get_current_spend", _get_spend), # test-quality-ok: common_checks imports it locally + patch.object(slack_alerting, "send_alert", send_alert), + ): + error: Final = await _check_for_error() + await asyncio.wait_for(alert_finished.wait(), timeout=1.0) + + return send_alert, error + + +@pytest.mark.asyncio +async def test_common_checks_internal_user_budget_below_threshold_does_not_emit_alert(): + send_alert, error = await _run_internal_user_budget_alert(spend=84.0) + + assert error is None + send_alert.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_common_checks_internal_user_budget_emits_user_threshold_event(): + send_alert, error = await _run_internal_user_budget_alert(spend=85.0) + + assert error is None + send_alert.assert_awaited_once() + event: Final = send_alert.await_args.kwargs["user_info"] + assert isinstance(event, WebhookEvent) + assert event.event == "threshold_crossed" + assert event.event_group == Litellm_EntityType.USER + assert event.user_id == "user-1" + assert event.user_email == "person@example.com" + assert event.spend == 85.0 + assert event.max_budget == 100.0 + assert event.token is None + assert event.key_alias is None + assert event.team_id is None + assert event.organization_id is None + + +@pytest.mark.asyncio +async def test_common_checks_internal_user_budget_emits_crossed_event_and_rejects(): + send_alert, error = await _run_internal_user_budget_alert(spend=100.0) + + assert error is not None + assert error.current_cost == 100.0 + assert error.max_budget == 100.0 + send_alert.assert_awaited_once() + event: Final = send_alert.await_args.kwargs["user_info"] + assert isinstance(event, WebhookEvent) + assert event.event == "budget_crossed" + assert event.event_group == Litellm_EntityType.USER + assert event.user_id == "user-1" + assert event.user_email == "person@example.com" + + @pytest.mark.asyncio async def test_common_checks_personal_user_budget_skipped_for_team_key(): """A user's personal max_budget does not apply to a team-scoped key. @@ -5398,6 +5521,9 @@ async def test_common_checks_personal_user_budget_skipped_for_team_key(): async def _no_membership(*args, **kwargs): return None + proxy_logging_obj: Final = MagicMock() + proxy_logging_obj.budget_alerts = AsyncMock() + with ( patch("litellm.proxy.proxy_server.prisma_client", None), patch("litellm.proxy.proxy_server.get_current_spend", _spend_by_counter), @@ -5412,11 +5538,12 @@ async def test_common_checks_personal_user_budget_skipped_for_team_key(): general_settings={}, route="/chat/completions", llm_router=None, - proxy_logging_obj=MagicMock(), + proxy_logging_obj=proxy_logging_obj, valid_token=token, request=MagicMock(spec=Request), ) assert result is True + proxy_logging_obj.budget_alerts.assert_not_awaited() @pytest.mark.asyncio From 6d6659a81bd1b6d92c5e8c43383a37b95711aa68 Mon Sep 17 00:00:00 2001 From: Jon Walton Date: Wed, 9 Sep 2026 18:39:36 +0800 Subject: [PATCH 016/157] test(proxy): harden user budget alert coverage --- .../proxy/auth/test_auth_checks.py | 19 ++++++++++++++++--- 1 file changed, 16 insertions(+), 3 deletions(-) diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index 1fc2e0b3dc1..f9e25e9beaf 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -5356,6 +5356,9 @@ async def test_common_checks_personal_user_budget_blocks_in_gather(): async def _spend_by_counter(counter_key, fallback_spend, max_budget=None, **kwargs): return 999.0 if counter_key == "spend:user:u1" else 0.0 + proxy_logging_obj: Final = MagicMock() + proxy_logging_obj.budget_alerts = AsyncMock() + with ( patch("litellm.proxy.proxy_server.prisma_client", None), patch("litellm.proxy.proxy_server.get_current_spend", _spend_by_counter), @@ -5370,10 +5373,11 @@ async def test_common_checks_personal_user_budget_blocks_in_gather(): general_settings={}, route="/chat/completions", llm_router=None, - proxy_logging_obj=MagicMock(), + proxy_logging_obj=proxy_logging_obj, valid_token=token, request=MagicMock(spec=Request), ) + await asyncio.sleep(0) assert "User=u1" in str(over.value) @@ -5412,6 +5416,7 @@ async def _run_internal_user_budget_alert( type: Literal["user_budget"], user_info: CallInfo, ) -> None: + assert type == "user_budget" try: await slack_alerting.budget_alerts(type=type, user_info=user_info) finally: @@ -5568,6 +5573,9 @@ async def test_common_checks_personal_user_budget_enforced_on_team_key_when_flag async def _no_membership(*args, **kwargs): return None + proxy_logging_obj: Final = MagicMock() + proxy_logging_obj.budget_alerts = AsyncMock() + with ( patch("litellm.proxy.proxy_server.prisma_client", None), patch("litellm.proxy.proxy_server.get_current_spend", _spend_by_counter), @@ -5583,10 +5591,11 @@ async def test_common_checks_personal_user_budget_enforced_on_team_key_when_flag general_settings={"apply_user_budget_to_team_keys": True}, route="/chat/completions", llm_router=None, - proxy_logging_obj=MagicMock(), + proxy_logging_obj=proxy_logging_obj, valid_token=token, request=MagicMock(spec=Request), ) + await asyncio.sleep(0) assert "ExceededBudget: User=u1" in str(exc_info.value) @@ -5603,6 +5612,9 @@ async def test_common_checks_personal_user_budget_still_enforced_on_personal_key async def _spend_by_counter(counter_key, fallback_spend, max_budget=None, **kwargs): return 999.0 if counter_key == "spend:user:u1" else 0.0 + proxy_logging_obj: Final = MagicMock() + proxy_logging_obj.budget_alerts = AsyncMock() + with ( patch("litellm.proxy.proxy_server.prisma_client", None), patch("litellm.proxy.proxy_server.get_current_spend", _spend_by_counter), @@ -5617,10 +5629,11 @@ async def test_common_checks_personal_user_budget_still_enforced_on_personal_key general_settings={"apply_user_budget_to_team_keys": True}, route="/chat/completions", llm_router=None, - proxy_logging_obj=MagicMock(), + proxy_logging_obj=proxy_logging_obj, valid_token=token, request=MagicMock(spec=Request), ) + await asyncio.sleep(0) @pytest.mark.parametrize( From 1425c71c10634bc86a2966ac4cfc977b7b9b39de Mon Sep 17 00:00:00 2001 From: jesus Date: Wed, 9 Sep 2026 19:00:24 +0000 Subject: [PATCH 017/157] fix(mcp): log upstream request method, body and response on MCP tool-list and OAuth2 token failures Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../_experimental/mcp_server/mcp_debug.py | 65 +++++++++++++++++++ .../mcp_server/mcp_server_manager.py | 12 +++- .../client_credentials.py | 6 ++ .../mcp_server/test_mcp_debug.py | 58 +++++++++++++++++ .../test_mcp_oauth_passthrough_tools.py | 27 +++++++- 5 files changed, 165 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_debug.py b/litellm/proxy/_experimental/mcp_server/mcp_debug.py index 6aaa00ae415..7a04901c2ca 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_debug.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_debug.py @@ -85,11 +85,14 @@ Usage with curl:: http://localhost:4000/mcp/atlassian_mcp """ +import re from typing import TYPE_CHECKING, Final +import httpx from starlette.types import Message, Send from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker +from litellm.proxy._experimental.mcp_server.faults.traversal import iter_exception_tree if TYPE_CHECKING: from litellm.types.mcp_server.mcp_server_manager import MCPServer @@ -127,6 +130,10 @@ class MCPDebug: @staticmethod def _mask(value: str | None) -> str: """Mask a single value for safe display in headers.""" + return MCPDebug.mask_secret(value) + + @staticmethod + def mask_secret(value: str | None) -> str: if not value: return "(none)" return MCPDebug._masker._mask_value(value) @@ -311,3 +318,61 @@ class MCPDebug: server_url=server_url, server_auth_type=server_auth_type, ) + + +_BODY_PREVIEW_CHARS: Final = 512 +_SENSITIVE_HEADER_NAMES: Final = frozenset({"authorization", "proxy-authorization", "cookie", "x-api-key"}) +_SENSITIVE_BODY_FIELD: Final = re.compile( + r'(?P"?(?:client_secret|client_assertion|refresh_token|access_token|id_token|password|code)"?\s*[=:]\s*"?)' + r'(?P[^&"\s,}]+)' +) + + +def _mask_body_match(match: re.Match[str]) -> str: + return f"{match.group('key')}{MCPDebug.mask_secret(match.group('value'))}" + + +def _preview(raw: bytes) -> str: + text: Final = _SENSITIVE_BODY_FIELD.sub(_mask_body_match, raw.decode("utf-8", errors="replace")) + return ( + text + if len(text) <= _BODY_PREVIEW_CHARS + else f"{text[:_BODY_PREVIEW_CHARS]}...(+{len(text) - _BODY_PREVIEW_CHARS} chars)" + ) + + +def _masked_headers(headers: httpx.Headers) -> str: + return ", ".join( + f"{name}={MCPDebug.mask_secret(value) if name.lower() in _SENSITIVE_HEADER_NAMES else value}" + for name, value in headers.items() + ) + + +def _request_body_preview(request: httpx.Request) -> str: + try: + return _preview(request.content) or "(empty)" + except httpx.RequestNotRead: + return "(streamed, not captured)" + + +def _response_body_preview(response: httpx.Response) -> str: + try: + return _preview(response.content) or "(empty)" + except httpx.ResponseNotRead: + return "(not read)" + + +def describe_upstream_http_failure(exc: BaseException) -> str | None: + """One line per upstream ``httpx.Response`` in the exception tree: the request method, URL, + masked request headers and JSON-RPC body that were sent, plus the status and body that came back. + ``None`` when the failure never reached an HTTP response (DNS, refused connection, timeout).""" + lines: Final = tuple( + f"{response.request.method} {response.request.url} -> HTTP {response.status_code} {response.reason_phrase}" + f" | request headers: {_masked_headers(response.request.headers)}" + f" | request body: {_request_body_preview(response.request)}" + f" | response body: {_response_body_preview(response)}" + for current in iter_exception_tree(exc) + for response in (getattr(current, "response", None),) + if isinstance(response, httpx.Response) + ) + return "\n".join(lines) or None diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index d7f238f142c..1247623f491 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -80,6 +80,7 @@ from litellm.proxy._experimental.mcp_server.faults.list_outcomes import ( raise_classified_list_failure, upstream_auth_challenge, ) +from litellm.proxy._experimental.mcp_server.mcp_debug import describe_upstream_http_failure from litellm.proxy._experimental.mcp_server.oauth2_token_cache import ( MCPPerUserTokenCache, mcp_per_user_token_cache, @@ -1353,6 +1354,11 @@ def _extract_upstream_auth_failure( return upstream_auth_challenge(exc) +def _upstream_failure_suffix(exc: BaseException) -> str: + detail: Final = describe_upstream_http_failure(exc) + return f"\n upstream exchange: {detail}" if detail else "" + + def _obo_retry_applies(server: MCPServer, subject_token: str | None) -> bool: """Whether an upstream 401/403 should invalidate the minted credential and retry once. @@ -4259,7 +4265,9 @@ class MCPServerManager: except MCPServerListError: raise except Exception as e: - verbose_logger.warning("Failed to get tools from server %s: %s", server.name, e) + verbose_logger.warning( + "Failed to get tools from server %s: %s%s", server.name, e, _upstream_failure_suffix(e) + ) raise_classified_list_failure(e, server.name, suppress_challenge=server.is_dcr_bridge) async def get_prompts_from_server( @@ -5004,7 +5012,7 @@ class MCPServerManager: verbose_logger.warning("Connection error while listing tools from %s: %s", server_name, e) raise MCPServerListError(ServerListFault(tag="unreachable"), server_name) from e except Exception as e: - verbose_logger.warning("Error listing tools from %s: %s", server_name, e) + verbose_logger.warning("Error listing tools from %s: %s%s", server_name, e, _upstream_failure_suffix(e)) raise_classified_list_failure(e, server_name) _SHORT_PREFIX_MAX_REHASH_ATTEMPTS = 1024 diff --git a/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py b/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py index ad18d1bb10f..f962148c7ff 100644 --- a/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py +++ b/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py @@ -37,6 +37,8 @@ import httpx from pydantic import BaseModel, ConfigDict, Field, SecretStr, TypeAdapter, ValidationError from typing_extensions import assert_never +from litellm._logging import verbose_logger +from litellm.proxy._experimental.mcp_server.mcp_debug import describe_upstream_http_failure from litellm.proxy._experimental.mcp_server.outbound_credentials.oauth_token_store import ( InMemoryTokenCacheBackend, OAuthToken, @@ -110,6 +112,10 @@ async def post_client_credentials_grant( ) except httpx.HTTPStatusError as status_err: status_code: Final = status_err.response.status_code + verbose_logger.warning( + "OAuth2 client_credentials token request denied:\n upstream exchange: %s", + describe_upstream_http_failure(status_err), + ) return TokenEndpointDenied(status_code=status_code, detail=f"token endpoint returned HTTP {status_code}") except Exception as exc: # noqa: BLE001 # any transport failure is the same outcome: unreachable return TokenEndpointUnreachable(detail=str(exc)) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index d299239f68e..b37e738579c 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -5,9 +5,12 @@ Tests for MCPDebug — MCP OAuth2 debug response headers. import asyncio from unittest.mock import MagicMock +import httpx + from litellm.proxy._experimental.mcp_server.mcp_debug import ( MCP_DEBUG_REQUEST_HEADER, MCPDebug, + describe_upstream_http_failure, ) @@ -269,3 +272,58 @@ class TestWrapSendWithDebugHeaders: asyncio.run(wrapped(body_msg)) assert captured[0] == body_msg + + +class TestDescribeUpstreamHttpFailure: + @staticmethod + def _status_error(*, body: bytes, response_body: bytes | None = None) -> httpx.HTTPStatusError: + request = httpx.Request( + "POST", + "https://upstream.example/apis/mcp", + headers={"Authorization": "Bearer secret-token-abcdef0123456789", "Content-Type": "application/json"}, + content=body, + ) + response = ( + httpx.Response(500, request=request, content=response_body) + if response_body is not None + else httpx.Response(500, request=request, stream=httpx.ByteStream(b'{"error":"boom"}')) + ) + return httpx.HTTPStatusError("500", request=request, response=response) + + def test_includes_method_url_status_and_request_body(self): + exc = self._status_error( + body=b'{"method":"initialize","jsonrpc":"2.0","id":0}', + response_body=b'{"error":"boom"}', + ) + described = describe_upstream_http_failure(exc) + assert described is not None + assert "POST https://upstream.example/apis/mcp -> HTTP 500" in described + assert '{"method":"initialize"' in described + assert 'response body: {"error":"boom"}' in described + + def test_masks_authorization_header_and_secret_body_fields(self): + exc = self._status_error( + body=b"grant_type=client_credentials&client_id=abc&client_secret=super-secret-value-1234", + response_body=b"{}", + ) + described = describe_upstream_http_failure(exc) + assert described is not None + assert "secret-token-abcdef0123456789" not in described + assert "super-secret-value-1234" not in described + assert "client_id=abc" in described + assert "client_secret=" in described + + def test_reports_unread_streamed_response_body(self): + described = describe_upstream_http_failure(self._status_error(body=b"{}")) + assert described is not None + assert "response body: (not read)" in described + + def test_finds_response_behind_cause_chain(self): + wrapper = RuntimeError("token minting failed") + wrapper.__cause__ = self._status_error(body=b"{}", response_body=b'{"error":"invalid_client"}') + described = describe_upstream_http_failure(wrapper) + assert described is not None + assert "invalid_client" in described + + def test_returns_none_without_http_response(self): + assert describe_upstream_http_failure(ConnectionError("refused")) is None diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py index 9667224de98..c7f3ea47fb3 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py @@ -1,5 +1,6 @@ """Unit tests for MCP OAuth passthrough tool-fetch behavior.""" +import logging import sys from unittest.mock import AsyncMock, MagicMock @@ -11,7 +12,7 @@ if sys.version_info < (3, 11): from exceptiongroup import ExceptionGroup -from litellm.proxy._experimental.mcp_server.exceptions import MCPUpstreamAuthError +from litellm.proxy._experimental.mcp_server.exceptions import MCPServerListError, MCPUpstreamAuthError from litellm.proxy._experimental.mcp_server.mcp_server_manager import ( MCPServerManager, _extract_upstream_auth_failure, @@ -434,3 +435,27 @@ async def test_aggregate_with_single_accessible_server_still_absorbs(): assert listing.tools == [] assert listing.outcomes["delegate_docs"].tag == "auth_required" + + +@pytest.mark.asyncio +async def test_fetch_tools_logs_upstream_request_details_on_500(caplog): + manager = MCPServerManager() + request = httpx.Request( + "POST", + "https://upstream/apis/mcp", + headers={"Authorization": "Bearer upstream-token-0123456789"}, + content=b'{"method":"initialize","jsonrpc":"2.0","id":0}', + ) + response = httpx.Response(500, request=request) + mock_client = MagicMock() + mock_client.list_tools = AsyncMock( + side_effect=httpx.HTTPStatusError("500", request=request, response=response) + ) + + with caplog.at_level(logging.WARNING, logger="LiteLLM"): + with pytest.raises(MCPServerListError): + await manager._fetch_tools_with_timeout(mock_client, "sample_docs") + + assert "POST https://upstream/apis/mcp -> HTTP 500" in caplog.text + assert '"method":"initialize"' in caplog.text + assert "upstream-token-0123456789" not in caplog.text From 058ff8c63c222d52392814b0fda84d5b030a84ec Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 15:33:44 -0700 Subject: [PATCH 018/157] test(e2e): keep answering while every Redis command times out Add tests/e2e/router/test_redis_timeout_e2e.py against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml: a real Redis with socket_timeout 0.001, so every command times out and the circuit breaker opens, plus a primary deployment that always fails and falls back to a healthy one, so every request carries retry breadcrumbs into cost tracking. The test drives twenty chat requests through the proxy and asserts each answers within ten seconds, the last third is no slower than the first, /health/liveliness stays fast, and every request still reaches the spend log. Gate it behind the redis_timeout marker and E2E_REDIS_TIMEOUT, exclude it from the per-PR e2e-changed selector, register the reliability.circuit_breaker.redis_timeout.stays_responsive cell, and run it as its own job in the weekly load anomaly workflow with a Postgres and Valkey service. Against a v1.100.0 proxy the run wedges the worker: requests time out and liveliness stops answering (LIT-6780). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- .github/e2e-stack/select_tests.py | 1 + .github/workflows/weekly_load_anomaly.yml | 84 +++++++++++++++++ tests/e2e/CLAUDE.md | 2 +- tests/e2e/CONTRIBUTING.md | 2 +- tests/e2e/conftest.py | 9 +- tests/e2e/coverage_registry/reliability.yaml | 1 + tests/e2e/e2e_config.py | 5 +- tests/e2e/gateway/redis_timeout_ci_config.yml | 29 ++++++ tests/e2e/pytest.ini | 1 + tests/e2e/router/conftest.py | 21 +++-- tests/e2e/router/test_redis_timeout_e2e.py | 93 +++++++++++++++++++ 11 files changed, 234 insertions(+), 14 deletions(-) create mode 100644 tests/e2e/gateway/redis_timeout_ci_config.yml create mode 100644 tests/e2e/router/test_redis_timeout_e2e.py diff --git a/.github/e2e-stack/select_tests.py b/.github/e2e-stack/select_tests.py index a62358f81ff..10f26bd3b8a 100644 --- a/.github/e2e-stack/select_tests.py +++ b/.github/e2e-stack/select_tests.py @@ -8,6 +8,7 @@ UNSUPPORTED: Final = re.compile( r"|^tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e\.py$" r"|^tests/e2e/batches/test_managed_files_enforcement_e2e\.py$" r"|^tests/e2e/guardrails/test_presidio_masking_e2e\.py$" + r"|^tests/e2e/router/test_redis_timeout_e2e\.py$" ) HARNESS: Final = re.compile( r"^tests/e2e/[A-Za-z0-9_.-]+\.(py|ini)$" diff --git a/.github/workflows/weekly_load_anomaly.yml b/.github/workflows/weekly_load_anomaly.yml index 3e1fca89645..6e564cf7e78 100644 --- a/.github/workflows/weekly_load_anomaly.yml +++ b/.github/workflows/weekly_load_anomaly.yml @@ -83,3 +83,87 @@ jobs: - name: Show proxy log on failure if: failure() run: tail -n 300 proxy.log + + redis-timeout-e2e: + if: github.event_name != 'schedule' || github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + timeout-minutes: 30 + services: + postgres: + image: postgres:16.6 + env: + POSTGRES_USER: llmproxy + POSTGRES_PASSWORD: dbpassword9090 + POSTGRES_DB: litellm + ports: + - 5432:5432 + options: >- + --health-cmd "pg_isready -U llmproxy" + --health-interval 5s + --health-timeout 5s + --health-retries 10 + valkey: + image: valkey/valkey:8.1.4@sha256:81db6d39e1bba3b3ff32bd3a1b19a6d69690f94a3954ec131277b9a26b95b3aa + ports: + - 6379:6379 + options: >- + --health-cmd "valkey-cli ping" + --health-interval 5s + --health-timeout 5s + --health-retries 10 + env: + DATABASE_URL: postgresql://llmproxy:dbpassword9090@localhost:5432/litellm + LITELLM_MASTER_KEY: sk-redis-timeout-e2e + LITELLM_LOG: WARNING + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache the Rust build + uses: ./.github/actions/cache-cargo-build + + - name: Install dependencies + run: | + .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra proxy + + - name: Cache Prisma binaries + uses: ./.github/actions/cache-prisma-binaries + + - name: Generate Prisma client + run: | + uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma + + - name: Start the proxy with a Redis that times out on every command + run: | + nohup uv run --no-sync litellm --config tests/e2e/gateway/redis_timeout_ci_config.yml --port 4000 > proxy.log 2>&1 & + for _ in $(seq 1 90); do + if curl -fs http://localhost:4000/health/liveliness > /dev/null; then + exit 0 + fi + sleep 2 + done + echo "proxy never became live" + tail -n 100 proxy.log + exit 1 + + - name: Run the Redis timeout e2e test + env: + E2E_REDIS_TIMEOUT: "1" + LITELLM_PROXY_URL: http://localhost:4000 + run: | + uv run --no-sync pytest tests/e2e/router/test_redis_timeout_e2e.py -v --tb=short -rA + + - name: Show proxy log on failure + if: failure() + run: tail -n 300 proxy.log diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 89c04208d65..53342f66b67 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -17,7 +17,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `mcp/` - the MCP server surface over api_key auth against the real Datadog remote MCP server (see "MCP suite: real Datadog only" below); plus the gateway-managed OAuth (authorization_code) path exercised through `/chat/completions`, the one behavior Datadog's static-header auth cannot reach, seeding the per-user upstream token via the interactive authorize dance driven with the mcp SDK's own OAuth client (headless-browser consent from a saved session) and asserting the completion lists and executes the server's tools with the stored per-user token - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection -- `router/` - routing and reliability behavior (fallbacks, cooldowns) +- `router/` - routing and reliability behavior (fallbacks, cooldowns). Also holds `test_redis_timeout_e2e.py`, which needs a proxy booted from `gateway/redis_timeout_ci_config.yml` (real Redis, `socket_timeout: 0.001`, so every command times out); it is marked `redis_timeout`, deselected unless `E2E_REDIS_TIMEOUT` is set, excluded from the per-PR check, and driven by `.github/workflows/weekly_load_anomaly.yml` - `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What remains here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`) and markerless harness unit tests for the Locust/session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index 1183096b81e..ccd27df6f46 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -65,7 +65,7 @@ A couple of logging destinations are configured on the proxy rather than by the ### The pull request check -Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start +Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, `guardrails/test_presidio_masking_e2e.py`, and `router/test_redis_timeout_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start, and the Redis timeout test needs a proxy whose Redis times out on every command (`gateway/redis_timeout_ci_config.yml`), which the weekly load workflow boots Every selected file must execute at least one passing test in each pass, and any test failure, collection error, or entirely skipped or deselected file fails the check. A file whose tests are all marked skip therefore cannot pass this check, so unskip at least one of them, or add the file to `UNSUPPORTED` in `select_tests.py` with the reason, before changing one. A failed pass stops the run. The public log prints pytest's one-line summary for each pass, including the rerun count, and names each failed or errored test as `classname::name`, so a retried network error or a failing test is visible without the raw output. The final `e2e-changed-tests` job succeeds only when no supported test files changed or the approved run completed all three passes. Fork PRs with selected tests fail this gate until a maintainer brings the reviewed change onto a same-repository branch diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index e1b987cbfd9..ff3d2dcc6bd 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -20,16 +20,14 @@ from datetime import datetime, timezone import pytest import requests - from e2e_config import CONTROL_PLANE_BASE_URL, FIXTURE_DIR, FIXTURE_MODE_RAW, PROXY_BASE_URL from e2e_db import RESET_OPT_IN_ENV, reset_spend_logs, run_spend_log_cleanup from fixture_mode import fixture_mode_collection_error, fixture_report_lines -from provider_edge import replay_leftover_error from junit_properties import attach_result_properties from lifecycle import ProxyClientProvider, ResourceManager +from provider_edge import replay_leftover_error from proxy_client import ProxyClient, build_proxy_client - _E2E_TEST_RAN = pytest.StashKey[bool]() _CALL_PASSED = pytest.StashKey[bool]() @@ -60,6 +58,11 @@ def pytest_configure(config: pytest.Config) -> None: "markers", "managed_files: needs a proxy running with require_managed_files enabled; deselected unless E2E_MANAGED_FILES_STACK is set", ) + config.addinivalue_line( + "markers", + "redis_timeout: needs a proxy booted from gateway/redis_timeout_ci_config.yml whose Redis times out on every " + "command; deselected unless E2E_REDIS_TIMEOUT is set", + ) def pytest_sessionstart(session: pytest.Session) -> None: diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index 6b69677d490..d62f0105931 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,6 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/e2e_config.py b/tests/e2e/e2e_config.py index 691335ffdd5..6cb59ea4e90 100644 --- a/tests/e2e/e2e_config.py +++ b/tests/e2e/e2e_config.py @@ -13,7 +13,6 @@ from pathlib import Path from typing import Final from dotenv import load_dotenv - from fixture_mode import deterministic_marker, parse_fixture_mode from provider_edge import provider_edge_api_base @@ -144,6 +143,7 @@ LOAD_MIN_CONCURRENCY_EFFICIENCY = float(os.environ.get("E2E_LOAD_MIN_CONCURRENCY WEEKLY_ANOMALY_OPT_IN_ENV = "E2E_WEEKLY_ANOMALY" MANAGED_FILES_OPT_IN_ENV = "E2E_MANAGED_FILES_STACK" +REDIS_TIMEOUT_OPT_IN_ENV = "E2E_REDIS_TIMEOUT" ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6")) ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6")) ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3")) @@ -181,8 +181,7 @@ def datadog_mcp_url(*, toolsets: str = "core") -> str: site = ( os.environ.get("DD_SITE", DD_SITE) or "datadoghq.com" ).strip().removeprefix("https://").removeprefix("http://").rstrip("/") - if site.startswith("app."): - site = site[len("app.") :] + site = site.removeprefix("app.") host = "mcp.datadoghq.com" if site in ("", "datadoghq.com") else f"mcp.{site}" base = f"https://{host}/v1/mcp" return f"{base}?toolsets={toolsets}" if toolsets else base diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_timeout_ci_config.yml new file mode 100644 index 00000000000..69ab2ee14dc --- /dev/null +++ b/tests/e2e/gateway/redis_timeout_ci_config.yml @@ -0,0 +1,29 @@ +general_settings: + master_key: os.environ/LITELLM_MASTER_KEY + store_model_in_db: true + +litellm_settings: + cache: true + cache_params: + type: redis + host: 127.0.0.1 + port: 6379 + socket_timeout: 0.001 + +router_settings: + num_retries: 1 + fallbacks: + - redis-timeout-primary: + - redis-timeout-backup + +model_list: + - model_name: redis-timeout-primary + litellm_params: + model: openai/gpt-5-mini + api_key: sk-redis-timeout-primary-not-used + mock_response: "litellm.InternalServerError" + - model_name: redis-timeout-backup + litellm_params: + model: openai/gpt-5-mini + api_key: sk-redis-timeout-backup-not-used + mock_response: "ok" diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index c3f8865f218..f69d5d058f3 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -9,3 +9,4 @@ markers = load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set managed_files: needs a proxy running with require_managed_files enabled; deselected unless E2E_MANAGED_FILES_STACK is set + redis_timeout: needs a proxy booted from gateway/redis_timeout_ci_config.yml whose Redis times out on every command; deselected unless E2E_REDIS_TIMEOUT is set diff --git a/tests/e2e/router/conftest.py b/tests/e2e/router/conftest.py index 98501f9bd7c..57fde729f41 100644 --- a/tests/e2e/router/conftest.py +++ b/tests/e2e/router/conftest.py @@ -10,13 +10,12 @@ proxy does not already list it (compose has it in static config; stage does not) from __future__ import annotations +import os from collections.abc import Iterator import pytest -from requests import RequestException - from complexity_router_client import ComplexityRouterClient, build_client -from proxy_client import ProxyClient +from e2e_config import REDIS_TIMEOUT_OPT_IN_ENV from e2e_http import NoBody, Success from lifecycle import ResourceManager from models import ( @@ -26,6 +25,8 @@ from models import ( LiteLLMParamsBody, ModelsListResponse, ) +from proxy_client import ProxyClient +from requests import RequestException ROUTER_MODEL = "complexity-smart-router" ROUTER_PARAMS = LiteLLMParamsBody( @@ -45,6 +46,16 @@ ROUTER_PARAMS = LiteLLMParamsBody( ROUTER_KEY_MODELS = [ROUTER_MODEL, "gpt-5.5", "claude-haiku-4-5"] +def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None: + if os.environ.get(REDIS_TIMEOUT_OPT_IN_ENV): + return + deselected = [item for item in items if item.get_closest_marker("redis_timeout") is not None] + if not deselected: + return + config.hook.pytest_deselected(items=deselected) + items[:] = [item for item in items if item.get_closest_marker("redis_timeout") is None] + + @pytest.fixture(scope="session") def client(proxy: ProxyClient) -> ComplexityRouterClient: return build_client(proxy) @@ -120,8 +131,6 @@ def _ensure_complexity_smart_router( # pyright: ignore[reportUnusedFunction] # @pytest.fixture def complexity_key(resources: ResourceManager, client: ComplexityRouterClient) -> str: """Per-test key allowed to call the complexity router and its tier backends.""" - key = client.proxy.generate_key( - KeyGenerateBody(models=ROUTER_KEY_MODELS, user_id="e2e-complexity-router") - ) + key = client.proxy.generate_key(KeyGenerateBody(models=ROUTER_KEY_MODELS, user_id="e2e-complexity-router")) resources.defer(lambda: client.proxy.delete_key(key)) return key diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py new file mode 100644 index 00000000000..eeb5224e051 --- /dev/null +++ b/tests/e2e/router/test_redis_timeout_e2e.py @@ -0,0 +1,93 @@ +"""Live e2e: the proxy keeps answering while every Redis command times out. + +Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which +points cache_params at a real Redis with socket_timeout 0.001 so every command times out and +the circuit breaker opens. Each request fails its primary deployment, retries, falls back to the +backup and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend +counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0 +that string doubled per request until the worker hung (LIT-6780). Deselected unless +E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. +""" + +from __future__ import annotations + +import time +from typing import Final + +import pytest +from complexity_router_client import ComplexityRouterClient +from e2e_config import unique_marker +from e2e_http import NoBody, Success +from lifecycle import ResourceManager +from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody + +pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout] + +PRIMARY_MODEL: Final = "redis-timeout-primary" +BACKUP_MODEL: Final = "redis-timeout-backup" +REQUESTS: Final = 20 +MAX_SECONDS_PER_REQUEST: Final = 10.0 +MAX_LATENCY_GROWTH_RATIO: Final = 3.0 +MAX_LIVELINESS_SECONDS: Final = 2.0 + + +class TestRedisTimeout: + @pytest.mark.covers( + "reliability.circuit_breaker.redis_timeout.stays_responsive", + exercised_on=["chat_completions"], + ) + def test_retries_under_redis_timeouts_keep_answering( + self, client: ComplexityRouterClient, resources: ResourceManager + ) -> None: + proxy = client.proxy + key = proxy.generate_key( + KeyGenerateBody(models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{unique_marker()}") + ) + resources.defer(lambda: proxy.delete_key(key)) + + latencies: list[float] = [] + for request_number in range(1, REQUESTS + 1): + started = time.monotonic() + result = proxy.transport.post( + "/chat/completions", + headers=proxy.transport.bearer(key), + json=ChatBody( + model=PRIMARY_MODEL, + messages=[ChatMessage(role="user", content=f"redis timeout {unique_marker()} {request_number}")], + max_tokens=5, + ), + response_type=ChatResponse, + timeout=MAX_SECONDS_PER_REQUEST, + ) + elapsed = time.monotonic() - started + assert isinstance(result, Success), ( + f"request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " + f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" + ) + assert result.data.choices, f"request {request_number}: fallback to {BACKUP_MODEL} returned no choices" + assert elapsed < MAX_SECONDS_PER_REQUEST, ( + f"request {request_number} took {elapsed:.1f}s with Redis timing out; " + f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" + ) + latencies.append(elapsed) + + third = REQUESTS // 3 + early = sum(latencies[:third]) / third + late = sum(latencies[-third:]) / third + assert late <= max(early * MAX_LATENCY_GROWTH_RATIO, 0.5), ( + f"per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests " + "while Redis timed out; the proxy is paying more for each failed request" + ) + + started = time.monotonic() + probe = proxy.transport.probe("/health/liveliness", params=NoBody()) + liveliness_seconds = time.monotonic() - started + assert probe.healthy, f"/health/liveliness returned {probe.status_code} after the Redis timeout loop" + assert liveliness_seconds < MAX_LIVELINESS_SECONDS, ( + f"/health/liveliness took {liveliness_seconds:.1f}s after the loop; the worker is stalled" + ) + + rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS) + assert len(rows) >= REQUESTS, ( + f"only {len(rows)} of {REQUESTS} requests reached the spend log; a Redis outage must not lose spend rows" + ) From d9e822d2562ddd716d8f6a5e86d3d8fedcb30df1 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 15:51:11 -0700 Subject: [PATCH 019/157] ci: run the Redis timeout e2e test from its own weekly workflow It is a functional e2e test, not a load test, so give it its own workflow instead of a job inside the load anomaly run. It keeps the Saturday 12:00 UTC cadence and manual dispatch, and boots the timeout-config proxy with Postgres and Valkey services exactly as before. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- .github/workflows/test-e2e-redis-timeout.yml | 94 ++++++++++++++++++++ .github/workflows/weekly_load_anomaly.yml | 84 ----------------- tests/e2e/CLAUDE.md | 2 +- tests/e2e/CONTRIBUTING.md | 2 +- 4 files changed, 96 insertions(+), 86 deletions(-) create mode 100644 .github/workflows/test-e2e-redis-timeout.yml diff --git a/.github/workflows/test-e2e-redis-timeout.yml b/.github/workflows/test-e2e-redis-timeout.yml new file mode 100644 index 00000000000..a956ed79590 --- /dev/null +++ b/.github/workflows/test-e2e-redis-timeout.yml @@ -0,0 +1,94 @@ +name: "Weekly Redis Timeout E2E" + +on: + schedule: + - cron: "0 12 * * 6" + workflow_dispatch: + +permissions: + contents: read + +jobs: + redis-timeout-e2e: + if: github.event_name != 'schedule' || github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + timeout-minutes: 30 + services: + postgres: + image: postgres:16.6 + env: + POSTGRES_USER: llmproxy + POSTGRES_PASSWORD: dbpassword9090 + POSTGRES_DB: litellm + ports: + - 5432:5432 + options: >- + --health-cmd "pg_isready -U llmproxy" + --health-interval 5s + --health-timeout 5s + --health-retries 10 + valkey: + image: valkey/valkey:8.1.4@sha256:81db6d39e1bba3b3ff32bd3a1b19a6d69690f94a3954ec131277b9a26b95b3aa + ports: + - 6379:6379 + options: >- + --health-cmd "valkey-cli ping" + --health-interval 5s + --health-timeout 5s + --health-retries 10 + env: + DATABASE_URL: postgresql://llmproxy:dbpassword9090@localhost:5432/litellm + LITELLM_MASTER_KEY: sk-redis-timeout-e2e + LITELLM_LOG: WARNING + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache the Rust build + uses: ./.github/actions/cache-cargo-build + + - name: Install dependencies + run: | + .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra proxy + + - name: Cache Prisma binaries + uses: ./.github/actions/cache-prisma-binaries + + - name: Generate Prisma client + run: | + uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma + + - name: Start the proxy with a Redis that times out on every command + run: | + nohup uv run --no-sync litellm --config tests/e2e/gateway/redis_timeout_ci_config.yml --port 4000 > proxy.log 2>&1 & + for _ in $(seq 1 90); do + if curl -fs http://localhost:4000/health/liveliness > /dev/null; then + exit 0 + fi + sleep 2 + done + echo "proxy never became live" + tail -n 100 proxy.log + exit 1 + + - name: Run the Redis timeout e2e test + env: + E2E_REDIS_TIMEOUT: "1" + LITELLM_PROXY_URL: http://localhost:4000 + run: | + uv run --no-sync pytest tests/e2e/router/test_redis_timeout_e2e.py -v --tb=short -rA + + - name: Show proxy log on failure + if: failure() + run: tail -n 300 proxy.log diff --git a/.github/workflows/weekly_load_anomaly.yml b/.github/workflows/weekly_load_anomaly.yml index 6e564cf7e78..3e1fca89645 100644 --- a/.github/workflows/weekly_load_anomaly.yml +++ b/.github/workflows/weekly_load_anomaly.yml @@ -83,87 +83,3 @@ jobs: - name: Show proxy log on failure if: failure() run: tail -n 300 proxy.log - - redis-timeout-e2e: - if: github.event_name != 'schedule' || github.repository == 'BerriAI/litellm' - runs-on: ubuntu-latest - timeout-minutes: 30 - services: - postgres: - image: postgres:16.6 - env: - POSTGRES_USER: llmproxy - POSTGRES_PASSWORD: dbpassword9090 - POSTGRES_DB: litellm - ports: - - 5432:5432 - options: >- - --health-cmd "pg_isready -U llmproxy" - --health-interval 5s - --health-timeout 5s - --health-retries 10 - valkey: - image: valkey/valkey:8.1.4@sha256:81db6d39e1bba3b3ff32bd3a1b19a6d69690f94a3954ec131277b9a26b95b3aa - ports: - - 6379:6379 - options: >- - --health-cmd "valkey-cli ping" - --health-interval 5s - --health-timeout 5s - --health-retries 10 - env: - DATABASE_URL: postgresql://llmproxy:dbpassword9090@localhost:5432/litellm - LITELLM_MASTER_KEY: sk-redis-timeout-e2e - LITELLM_LOG: WARNING - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Set up uv - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - - name: Cache the Rust build - uses: ./.github/actions/cache-cargo-build - - - name: Install dependencies - run: | - .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra proxy - - - name: Cache Prisma binaries - uses: ./.github/actions/cache-prisma-binaries - - - name: Generate Prisma client - run: | - uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma - - - name: Start the proxy with a Redis that times out on every command - run: | - nohup uv run --no-sync litellm --config tests/e2e/gateway/redis_timeout_ci_config.yml --port 4000 > proxy.log 2>&1 & - for _ in $(seq 1 90); do - if curl -fs http://localhost:4000/health/liveliness > /dev/null; then - exit 0 - fi - sleep 2 - done - echo "proxy never became live" - tail -n 100 proxy.log - exit 1 - - - name: Run the Redis timeout e2e test - env: - E2E_REDIS_TIMEOUT: "1" - LITELLM_PROXY_URL: http://localhost:4000 - run: | - uv run --no-sync pytest tests/e2e/router/test_redis_timeout_e2e.py -v --tb=short -rA - - - name: Show proxy log on failure - if: failure() - run: tail -n 300 proxy.log diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 53342f66b67..76e8550cb9a 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -17,7 +17,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `mcp/` - the MCP server surface over api_key auth against the real Datadog remote MCP server (see "MCP suite: real Datadog only" below); plus the gateway-managed OAuth (authorization_code) path exercised through `/chat/completions`, the one behavior Datadog's static-header auth cannot reach, seeding the per-user upstream token via the interactive authorize dance driven with the mcp SDK's own OAuth client (headless-browser consent from a saved session) and asserting the completion lists and executes the server's tools with the stored per-user token - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection -- `router/` - routing and reliability behavior (fallbacks, cooldowns). Also holds `test_redis_timeout_e2e.py`, which needs a proxy booted from `gateway/redis_timeout_ci_config.yml` (real Redis, `socket_timeout: 0.001`, so every command times out); it is marked `redis_timeout`, deselected unless `E2E_REDIS_TIMEOUT` is set, excluded from the per-PR check, and driven by `.github/workflows/weekly_load_anomaly.yml` +- `router/` - routing and reliability behavior (fallbacks, cooldowns). Also holds `test_redis_timeout_e2e.py`, which needs a proxy booted from `gateway/redis_timeout_ci_config.yml` (real Redis, `socket_timeout: 0.001`, so every command times out); it is marked `redis_timeout`, deselected unless `E2E_REDIS_TIMEOUT` is set, excluded from the per-PR check, and driven weekly by `.github/workflows/test-e2e-redis-timeout.yml` - `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What remains here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`) and markerless harness unit tests for the Locust/session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index ccd27df6f46..7a859965c80 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -65,7 +65,7 @@ A couple of logging destinations are configured on the proxy rather than by the ### The pull request check -Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, `guardrails/test_presidio_masking_e2e.py`, and `router/test_redis_timeout_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start, and the Redis timeout test needs a proxy whose Redis times out on every command (`gateway/redis_timeout_ci_config.yml`), which the weekly load workflow boots +Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, `guardrails/test_presidio_masking_e2e.py`, and `router/test_redis_timeout_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start, and the Redis timeout test needs a proxy whose Redis times out on every command (`gateway/redis_timeout_ci_config.yml`), which `.github/workflows/test-e2e-redis-timeout.yml` boots on a weekly schedule Every selected file must execute at least one passing test in each pass, and any test failure, collection error, or entirely skipped or deselected file fails the check. A file whose tests are all marked skip therefore cannot pass this check, so unskip at least one of them, or add the file to `UNSUPPORTED` in `select_tests.py` with the reason, before changing one. A failed pass stops the run. The public log prints pytest's one-line summary for each pass, including the rerun count, and names each failed or errored test as `classname::name`, so a retried network error or a failing test is visible without the raw output. The final `e2e-changed-tests` job succeeds only when no supported test files changed or the approved run completed all three passes. Fork PRs with selected tests fail this gate until a maintainer brings the reviewed change onto a same-repository branch From 02351da51f14d644660c4a5583968389752af46b Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 15:51:35 -0700 Subject: [PATCH 020/157] test(e2e): register the Redis timeout cell at tier P1 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- tests/e2e/coverage_registry/reliability.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index d62f0105931..38427b647b9 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,7 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} From 939ac04a932baae63f2f8b9440ee43ed70570763 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 16:03:36 -0700 Subject: [PATCH 021/157] test(e2e): cover /v1/responses in the Redis timeout test and fail the primary for real The Responses path returns a mock for any mock_response string, so the InternalServerError sentinel never failed there. Point the primary deployment's api_base at a closed port instead, which fails every endpoint the same way, then parametrize the test over /chat/completions and /v1/responses and register both on the coverage cell. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- tests/e2e/coverage_registry/reliability.yaml | 2 +- tests/e2e/gateway/redis_timeout_ci_config.yml | 2 +- tests/e2e/router/test_redis_timeout_e2e.py | 105 +++++++++++++----- 3 files changed, 81 insertions(+), 28 deletions(-) diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index 38427b647b9..3b378b56870 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,7 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_timeout_ci_config.yml index 69ab2ee14dc..735285adb36 100644 --- a/tests/e2e/gateway/redis_timeout_ci_config.yml +++ b/tests/e2e/gateway/redis_timeout_ci_config.yml @@ -21,7 +21,7 @@ model_list: litellm_params: model: openai/gpt-5-mini api_key: sk-redis-timeout-primary-not-used - mock_response: "litellm.InternalServerError" + api_base: http://127.0.0.1:1 - model_name: redis-timeout-backup litellm_params: model: openai/gpt-5-mini diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py index eeb5224e051..d10f32b6482 100644 --- a/tests/e2e/router/test_redis_timeout_e2e.py +++ b/tests/e2e/router/test_redis_timeout_e2e.py @@ -1,25 +1,29 @@ """Live e2e: the proxy keeps answering while every Redis command times out. Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which -points cache_params at a real Redis with socket_timeout 0.001 so every command times out and -the circuit breaker opens. Each request fails its primary deployment, retries, falls back to the -backup and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend -counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0 -that string doubled per request until the worker hung (LIT-6780). Deselected unless -E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. +points cache_params at a real Redis with socket_timeout 0.001 so commands time out and the +circuit breaker opens. Each request fails its primary deployment, whose api_base is a closed +port, retries, falls back to the backup and succeeds, so it carries retry breadcrumbs; its cost +tracking then fails on the spend counter increment and stringifies the request metadata into a +failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung +(LIT-6780). Deselected unless E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. """ from __future__ import annotations import time +from collections.abc import Callable +from dataclasses import dataclass from typing import Final import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker -from e2e_http import NoBody, Success +from e2e_http import NoBody, Result, Success from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody +from proxy_client import ProxyClient +from pydantic import BaseModel pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout] @@ -31,42 +35,90 @@ MAX_LATENCY_GROWTH_RATIO: Final = 3.0 MAX_LIVELINESS_SECONDS: Final = 2.0 +class ResponsesBody(BaseModel): + model: str + input: str + max_output_tokens: int = 5 + + +class ResponsesObject(BaseModel): + id: str | None = None + status: str | None = None + output: list[object] = [] + + +@dataclass(frozen=True, slots=True) +class Endpoint: + name: str + send: Callable[[ProxyClient, str, str], Result[BaseModel]] + served: Callable[[BaseModel], bool] + + +def _send_chat(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]: + return proxy.transport.post( + "/chat/completions", + headers=proxy.transport.bearer(key), + json=ChatBody(model=PRIMARY_MODEL, messages=[ChatMessage(role="user", content=marker)], max_tokens=5), + response_type=ChatResponse, + timeout=MAX_SECONDS_PER_REQUEST, + ) + + +def _send_responses(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]: + return proxy.transport.post( + "/v1/responses", + headers=proxy.transport.bearer(key), + json=ResponsesBody(model=PRIMARY_MODEL, input=marker), + response_type=ResponsesObject, + timeout=MAX_SECONDS_PER_REQUEST, + ) + + +ENDPOINTS: Final = ( + Endpoint( + name="chat_completions", + send=_send_chat, + served=lambda data: isinstance(data, ChatResponse) and bool(data.choices), + ), + Endpoint( + name="responses", + send=_send_responses, + served=lambda data: isinstance(data, ResponsesObject) and bool(data.output), + ), +) + + class TestRedisTimeout: + @pytest.mark.parametrize("endpoint", ENDPOINTS, ids=[endpoint.name for endpoint in ENDPOINTS]) @pytest.mark.covers( "reliability.circuit_breaker.redis_timeout.stays_responsive", - exercised_on=["chat_completions"], + exercised_on=["chat_completions", "responses"], ) def test_retries_under_redis_timeouts_keep_answering( - self, client: ComplexityRouterClient, resources: ResourceManager + self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint ) -> None: proxy = client.proxy key = proxy.generate_key( - KeyGenerateBody(models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{unique_marker()}") + KeyGenerateBody( + models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}" + ) ) resources.defer(lambda: proxy.delete_key(key)) latencies: list[float] = [] for request_number in range(1, REQUESTS + 1): started = time.monotonic() - result = proxy.transport.post( - "/chat/completions", - headers=proxy.transport.bearer(key), - json=ChatBody( - model=PRIMARY_MODEL, - messages=[ChatMessage(role="user", content=f"redis timeout {unique_marker()} {request_number}")], - max_tokens=5, - ), - response_type=ChatResponse, - timeout=MAX_SECONDS_PER_REQUEST, - ) + result = endpoint.send(proxy, key, f"redis timeout {unique_marker()} {request_number}") elapsed = time.monotonic() - started assert isinstance(result, Success), ( - f"request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " + f"{endpoint.name} request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" ) - assert result.data.choices, f"request {request_number}: fallback to {BACKUP_MODEL} returned no choices" + assert endpoint.served(result.data), ( + f"{endpoint.name} request {request_number}: fallback to {BACKUP_MODEL} returned no output" + ) assert elapsed < MAX_SECONDS_PER_REQUEST, ( - f"request {request_number} took {elapsed:.1f}s with Redis timing out; " + f"{endpoint.name} request {request_number} took {elapsed:.1f}s with Redis timing out; " f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" ) latencies.append(elapsed) @@ -75,7 +127,7 @@ class TestRedisTimeout: early = sum(latencies[:third]) / third late = sum(latencies[-third:]) / third assert late <= max(early * MAX_LATENCY_GROWTH_RATIO, 0.5), ( - f"per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests " + f"{endpoint.name} per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests " "while Redis timed out; the proxy is paying more for each failed request" ) @@ -89,5 +141,6 @@ class TestRedisTimeout: rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS) assert len(rows) >= REQUESTS, ( - f"only {len(rows)} of {REQUESTS} requests reached the spend log; a Redis outage must not lose spend rows" + f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; " + "a Redis outage must not lose spend rows" ) From 2d7e5999b78f726c97a7142231de566d788d4fc0 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 16:26:03 -0700 Subject: [PATCH 022/157] test(e2e): pause Redis writes for the run and prove the timeouts from /metrics A loopback Redis answers many commands inside the 1 ms socket timeout, so nothing guaranteed the failure path ran. The test now holds the proxy's Redis in CLIENT PAUSE WRITE for its duration, so every write the proxy sends, the spend counter increment included, outlives the timeout, and lifts the pause in teardown. Reads stay live so the control connection can do that. Enable the prometheus callback in the gateway config and assert from /metrics that the proxy counted at least the breaker's five timeouts and that, during each case, it saw fresh timeouts, a breaker transition, or an open breaker rejecting every call. The open breaker is the state a customer's worker sits in, and cost tracking fails on every request either way. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- .github/workflows/test-e2e-redis-timeout.yml | 2 + tests/e2e/gateway/redis_timeout_ci_config.yml | 2 + tests/e2e/router/test_redis_timeout_e2e.py | 60 +++++++++++++++++-- 3 files changed, 60 insertions(+), 4 deletions(-) diff --git a/.github/workflows/test-e2e-redis-timeout.yml b/.github/workflows/test-e2e-redis-timeout.yml index a956ed79590..9c318a1346c 100644 --- a/.github/workflows/test-e2e-redis-timeout.yml +++ b/.github/workflows/test-e2e-redis-timeout.yml @@ -86,6 +86,8 @@ jobs: env: E2E_REDIS_TIMEOUT: "1" LITELLM_PROXY_URL: http://localhost:4000 + REDIS_HOST: 127.0.0.1 + REDIS_PORT: "6379" run: | uv run --no-sync pytest tests/e2e/router/test_redis_timeout_e2e.py -v --tb=short -rA diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_timeout_ci_config.yml index 735285adb36..375a51f6127 100644 --- a/tests/e2e/gateway/redis_timeout_ci_config.yml +++ b/tests/e2e/gateway/redis_timeout_ci_config.yml @@ -3,6 +3,8 @@ general_settings: store_model_in_db: true litellm_settings: + callbacks: ["prometheus"] + require_auth_for_metrics_endpoint: false cache: true cache_params: type: redis diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py index d10f32b6482..e99bd2fbe80 100644 --- a/tests/e2e/router/test_redis_timeout_e2e.py +++ b/tests/e2e/router/test_redis_timeout_e2e.py @@ -1,8 +1,10 @@ """Live e2e: the proxy keeps answering while every Redis command times out. Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which -points cache_params at a real Redis with socket_timeout 0.001 so commands time out and the -circuit breaker opens. Each request fails its primary deployment, whose api_base is a closed +points cache_params at a real Redis with socket_timeout 0.001. The test holds that Redis in +CLIENT PAUSE WRITE for its duration, so every write the proxy sends, the spend counter increment +included, hangs past the timeout, and it proves the degradation was real from the breaker metrics on /metrics: fresh timeouts, a breaker transition, +or an already-open breaker rejecting every call, which is the state a customer's worker sits in. Each request fails its primary deployment, whose api_base is a closed port, retries, falls back to the backup and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung @@ -11,12 +13,15 @@ failed-tracking alert. On v1.100.0 that string doubled per request until the wor from __future__ import annotations +import os +import re import time -from collections.abc import Callable +from collections.abc import Callable, Iterator from dataclasses import dataclass from typing import Final import pytest +import redis from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import NoBody, Result, Success @@ -33,6 +38,11 @@ REQUESTS: Final = 20 MAX_SECONDS_PER_REQUEST: Final = 10.0 MAX_LATENCY_GROWTH_RATIO: Final = 3.0 MAX_LIVELINESS_SECONDS: Final = 2.0 +REDIS_PAUSE_MS: Final = 600_000 +BREAKER_FAILURE_THRESHOLD: Final = 5 +TIMEOUT_FAILURES_RE: Final = re.compile( + r'^litellm_redis_circuit_breaker_failures_total\{failure_class="timeout"\} ([0-9.e+]+)$', re.M +) class ResponsesBody(BaseModel): @@ -86,6 +96,32 @@ ENDPOINTS: Final = ( served=lambda data: isinstance(data, ResponsesObject) and bool(data.output), ), ) +BREAKER_OPEN_RE: Final = re.compile(r'^litellm_redis_circuit_breaker_state\{state="open"\} ([0-9.e+]+)$', re.M) +BREAKER_TRANSITIONS_RE: Final = re.compile( + r'^litellm_redis_circuit_breaker_transitions_total\{state="[a-z_]+"\} ([0-9.e+]+)$', re.M +) + + +@pytest.fixture +def paused_redis() -> Iterator[None]: + """Hold the proxy's Redis in CLIENT PAUSE WRITE so every write it sends outlives the 1 ms socket + timeout. A loopback Redis otherwise answers many commands inside that budget. Reads stay live so + this control connection can lift the pause in teardown.""" + host = os.environ.get("REDIS_HOST") + port = os.environ.get("REDIS_PORT") + assert host and port, "REDIS_HOST and REDIS_PORT must name the Redis the proxy under test uses" + control = redis.Redis(host=host, port=int(port), socket_timeout=5) + control.client_pause(REDIS_PAUSE_MS, all=False) # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any + try: + yield + finally: + control.client_unpause() # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any + control.close() + + +def _metric(proxy: ProxyClient, pattern: re.Pattern[str]) -> float: + body = proxy.probe("/metrics", params=NoBody()).body + return sum(float(match.group(1)) for match in pattern.finditer(body)) class TestRedisTimeout: @@ -95,9 +131,11 @@ class TestRedisTimeout: exercised_on=["chat_completions", "responses"], ) def test_retries_under_redis_timeouts_keep_answering( - self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint + self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint, paused_redis: None ) -> None: proxy = client.proxy + timeouts_before = _metric(proxy, TIMEOUT_FAILURES_RE) + transitions_before = _metric(proxy, BREAKER_TRANSITIONS_RE) key = proxy.generate_key( KeyGenerateBody( models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}" @@ -139,6 +177,20 @@ class TestRedisTimeout: f"/health/liveliness took {liveliness_seconds:.1f}s after the loop; the worker is stalled" ) + timeouts_total = _metric(proxy, TIMEOUT_FAILURES_RE) + timeouts = timeouts_total - timeouts_before + transitions = _metric(proxy, BREAKER_TRANSITIONS_RE) - transitions_before + breaker_open = _metric(proxy, BREAKER_OPEN_RE) >= 1 + assert timeouts_total >= BREAKER_FAILURE_THRESHOLD, ( + f"the proxy counted only {timeouts_total:.0f} Redis timeouts in its lifetime; the write-paused Redis " + "never made its spend counter writes time out, so this run proved nothing" + ) + assert timeouts >= REQUESTS or transitions >= 1 or breaker_open, ( + f"during {REQUESTS} {endpoint.name} requests the breaker counted {timeouts:.0f} new timeouts, " + f"{transitions:.0f} state transitions, and ended {'open' if breaker_open else 'closed'}; " + "Redis was healthy for this case, so it proved nothing" + ) + rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS) assert len(rows) >= REQUESTS, ( f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; " From 2e4461ec458756c940f3a8863ea805bc44c33907 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 16:33:18 -0700 Subject: [PATCH 023/157] test(e2e): register the Redis timeout deployments through /model/new The e2e directive has every test create its deployments through the management API and delete them on teardown. Drop the static model_list from the gateway config; the test now registers the closed-port primary and the mock backup itself, and the fallback map stays in router_settings where proxy-level config belongs. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- tests/e2e/gateway/redis_timeout_ci_config.yml | 12 ----------- tests/e2e/router/test_redis_timeout_e2e.py | 20 ++++++++++++++++--- 2 files changed, 17 insertions(+), 15 deletions(-) diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_timeout_ci_config.yml index 375a51f6127..9bcf65162d6 100644 --- a/tests/e2e/gateway/redis_timeout_ci_config.yml +++ b/tests/e2e/gateway/redis_timeout_ci_config.yml @@ -17,15 +17,3 @@ router_settings: fallbacks: - redis-timeout-primary: - redis-timeout-backup - -model_list: - - model_name: redis-timeout-primary - litellm_params: - model: openai/gpt-5-mini - api_key: sk-redis-timeout-primary-not-used - api_base: http://127.0.0.1:1 - - model_name: redis-timeout-backup - litellm_params: - model: openai/gpt-5-mini - api_key: sk-redis-timeout-backup-not-used - mock_response: "ok" diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py index e99bd2fbe80..58d10f1d8ea 100644 --- a/tests/e2e/router/test_redis_timeout_e2e.py +++ b/tests/e2e/router/test_redis_timeout_e2e.py @@ -4,8 +4,8 @@ Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config. points cache_params at a real Redis with socket_timeout 0.001. The test holds that Redis in CLIENT PAUSE WRITE for its duration, so every write the proxy sends, the spend counter increment included, hangs past the timeout, and it proves the degradation was real from the breaker metrics on /metrics: fresh timeouts, a breaker transition, -or an already-open breaker rejecting every call, which is the state a customer's worker sits in. Each request fails its primary deployment, whose api_base is a closed -port, retries, falls back to the backup and succeeds, so it carries retry breadcrumbs; its cost +or an already-open breaker rejecting every call, which is the state a customer's worker sits in. The test registers two deployments through /model/new: a primary whose api_base is a closed port +and a backup that answers with a mock. Each request fails the primary, retries, falls back and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung (LIT-6780). Deselected unless E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. @@ -26,7 +26,7 @@ from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import NoBody, Result, Success from lifecycle import ResourceManager -from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody +from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody from proxy_client import ProxyClient from pydantic import BaseModel @@ -34,6 +34,8 @@ pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout] PRIMARY_MODEL: Final = "redis-timeout-primary" BACKUP_MODEL: Final = "redis-timeout-backup" +BACKING_MODEL: Final = "openai/gpt-5-mini" +CLOSED_PORT_API_BASE: Final = "http://127.0.0.1:1" REQUESTS: Final = 20 MAX_SECONDS_PER_REQUEST: Final = 10.0 MAX_LATENCY_GROWTH_RATIO: Final = 3.0 @@ -134,6 +136,18 @@ class TestRedisTimeout: self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint, paused_redis: None ) -> None: proxy = client.proxy + primary_id = proxy.create_model( + PRIMARY_MODEL, + LiteLLMParamsBody( + model=BACKING_MODEL, api_key="sk-redis-timeout-primary-not-used", api_base=CLOSED_PORT_API_BASE + ), + ) + resources.defer(lambda: proxy.delete_model(primary_id)) + backup_id = proxy.create_model( + BACKUP_MODEL, + LiteLLMParamsBody(model=BACKING_MODEL, api_key="sk-redis-timeout-backup-not-used", mock_response="ok"), + ) + resources.defer(lambda: proxy.delete_model(backup_id)) timeouts_before = _metric(proxy, TIMEOUT_FAILURES_RE) transitions_before = _metric(proxy, BREAKER_TRANSITIONS_RE) key = proxy.generate_key( From 1213d7d39929500883fee4c3823f8dfbc073fdb8 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Wed, 9 Sep 2026 17:17:33 -0700 Subject: [PATCH 024/157] test(e2e): cover /embeddings and assert memory and fallbacks in the Redis timeout test Add an embeddings case with its own closed-port primary and mock backup (the fallback map in the gateway config gains the pair; LiteLLMParamsBody.mock_response accepts the list an embedding mock needs). Assert from /metrics that the proxy's resident memory grows by no more than 200 MB across each case where the process collector reports it (Linux), that the router counted a successful fallback for every request, and that every spend row is a success. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T --- tests/e2e/coverage_registry/reliability.yaml | 2 +- tests/e2e/gateway/redis_timeout_ci_config.yml | 2 + tests/e2e/models.py | 2 +- tests/e2e/router/test_redis_timeout_e2e.py | 142 +++++++++++++----- 4 files changed, 110 insertions(+), 38 deletions(-) diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index 3b378b56870..a42e4b2779a 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,7 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses, embeddings], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_timeout_ci_config.yml index 9bcf65162d6..fd283dbc7df 100644 --- a/tests/e2e/gateway/redis_timeout_ci_config.yml +++ b/tests/e2e/gateway/redis_timeout_ci_config.yml @@ -17,3 +17,5 @@ router_settings: fallbacks: - redis-timeout-primary: - redis-timeout-backup + - redis-timeout-embed-primary: + - redis-timeout-embed-backup diff --git a/tests/e2e/models.py b/tests/e2e/models.py index 62810e6cfd9..f3c111bdf06 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -920,7 +920,7 @@ class LiteLLMParamsBody(BaseModel): auto_router_default_model: str | None = None auto_router_embedding_model: str | None = None tags: list[str] | None = None - mock_response: str | None = None + mock_response: str | list[float] | None = None timeout: float | None = None tpm: int | None = None weight: int | None = None diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py index 58d10f1d8ea..3f1984c0696 100644 --- a/tests/e2e/router/test_redis_timeout_e2e.py +++ b/tests/e2e/router/test_redis_timeout_e2e.py @@ -3,12 +3,12 @@ Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which points cache_params at a real Redis with socket_timeout 0.001. The test holds that Redis in CLIENT PAUSE WRITE for its duration, so every write the proxy sends, the spend counter increment -included, hangs past the timeout, and it proves the degradation was real from the breaker metrics on /metrics: fresh timeouts, a breaker transition, -or an already-open breaker rejecting every call, which is the state a customer's worker sits in. The test registers two deployments through /model/new: a primary whose api_base is a closed port -and a backup that answers with a mock. Each request fails the primary, retries, falls back and succeeds, so it carries retry breadcrumbs; its cost -tracking then fails on the spend counter increment and stringifies the request metadata into a -failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung -(LIT-6780). Deselected unless E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. +included, hangs past the timeout. For each endpoint it registers two deployments through +/model/new: a primary whose api_base is a closed port and a backup that answers with a mock. Each request fails the primary, +retries, falls back and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on +the spend counter increment and stringifies the request metadata into a failed-tracking alert. On +v1.100.0 that string doubled per request until the worker hung (LIT-6780). Deselected unless +E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. """ from __future__ import annotations @@ -26,25 +26,29 @@ from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import NoBody, Result, Success from lifecycle import ResourceManager -from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody +from models import ChatBody, ChatMessage, ChatResponse, EmbedBody, KeyGenerateBody, LiteLLMParamsBody from proxy_client import ProxyClient from pydantic import BaseModel pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout] -PRIMARY_MODEL: Final = "redis-timeout-primary" -BACKUP_MODEL: Final = "redis-timeout-backup" -BACKING_MODEL: Final = "openai/gpt-5-mini" -CLOSED_PORT_API_BASE: Final = "http://127.0.0.1:1" REQUESTS: Final = 20 MAX_SECONDS_PER_REQUEST: Final = 10.0 MAX_LATENCY_GROWTH_RATIO: Final = 3.0 MAX_LIVELINESS_SECONDS: Final = 2.0 +MAX_RSS_GROWTH_BYTES: Final = 200 * 1024 * 1024 REDIS_PAUSE_MS: Final = 600_000 BREAKER_FAILURE_THRESHOLD: Final = 5 +CLOSED_PORT_API_BASE: Final = "http://127.0.0.1:1" TIMEOUT_FAILURES_RE: Final = re.compile( r'^litellm_redis_circuit_breaker_failures_total\{failure_class="timeout"\} ([0-9.e+]+)$', re.M ) +BREAKER_OPEN_RE: Final = re.compile(r'^litellm_redis_circuit_breaker_state\{state="open"\} ([0-9.e+]+)$', re.M) +BREAKER_TRANSITIONS_RE: Final = re.compile( + r'^litellm_redis_circuit_breaker_transitions_total\{state="[a-z_]+"\} ([0-9.e+]+)$', re.M +) +FALLBACKS_RE: Final = re.compile(r"^litellm_deployment_successful_fallbacks_total\{[^}]*\} ([0-9.e+]+)$", re.M) +RSS_RE: Final = re.compile(r"^process_resident_memory_bytes ([0-9.e+]+)$", re.M) class ResponsesBody(BaseModel): @@ -59,48 +63,95 @@ class ResponsesObject(BaseModel): output: list[object] = [] +class EmbeddingsObject(BaseModel): + model: str | None = None + data: list[object] = [] + + @dataclass(frozen=True, slots=True) class Endpoint: + """One endpoint's deployments and request shape.""" + name: str - send: Callable[[ProxyClient, str, str], Result[BaseModel]] + primary: str + backup: str + primary_params: LiteLLMParamsBody + backup_params: LiteLLMParamsBody + send: Callable[[ProxyClient, str, str, str], Result[BaseModel]] served: Callable[[BaseModel], bool] -def _send_chat(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]: +def _send_chat(proxy: ProxyClient, key: str, model: str, marker: str) -> Result[BaseModel]: return proxy.transport.post( "/chat/completions", headers=proxy.transport.bearer(key), - json=ChatBody(model=PRIMARY_MODEL, messages=[ChatMessage(role="user", content=marker)], max_tokens=5), + json=ChatBody(model=model, messages=[ChatMessage(role="user", content=marker)], max_tokens=5), response_type=ChatResponse, timeout=MAX_SECONDS_PER_REQUEST, ) -def _send_responses(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]: +def _send_responses(proxy: ProxyClient, key: str, model: str, marker: str) -> Result[BaseModel]: return proxy.transport.post( "/v1/responses", headers=proxy.transport.bearer(key), - json=ResponsesBody(model=PRIMARY_MODEL, input=marker), + json=ResponsesBody(model=model, input=marker), response_type=ResponsesObject, timeout=MAX_SECONDS_PER_REQUEST, ) +def _send_embeddings(proxy: ProxyClient, key: str, model: str, marker: str) -> Result[BaseModel]: + return proxy.transport.post( + "/embeddings", + headers=proxy.transport.bearer(key), + json=EmbedBody(model=model, input=marker), + response_type=EmbeddingsObject, + timeout=MAX_SECONDS_PER_REQUEST, + ) + + +_CHAT_PRIMARY: Final = LiteLLMParamsBody( + model="openai/gpt-5-mini", api_key="sk-redis-timeout-primary-not-used", api_base=CLOSED_PORT_API_BASE +) +_CHAT_BACKUP: Final = LiteLLMParamsBody( + model="openai/gpt-5-nano", api_key="sk-redis-timeout-backup-not-used", mock_response="ok" +) +_EMBED_PRIMARY: Final = LiteLLMParamsBody( + model="openai/text-embedding-3-small", api_key="sk-redis-timeout-primary-not-used", api_base=CLOSED_PORT_API_BASE +) +_EMBED_BACKUP: Final = LiteLLMParamsBody( + model="openai/text-embedding-3-large", api_key="sk-redis-timeout-backup-not-used", mock_response=[0.1, 0.2, 0.3] +) + ENDPOINTS: Final = ( Endpoint( name="chat_completions", + primary="redis-timeout-primary", + backup="redis-timeout-backup", + primary_params=_CHAT_PRIMARY, + backup_params=_CHAT_BACKUP, send=_send_chat, served=lambda data: isinstance(data, ChatResponse) and bool(data.choices), ), Endpoint( name="responses", + primary="redis-timeout-primary", + backup="redis-timeout-backup", + primary_params=_CHAT_PRIMARY, + backup_params=_CHAT_BACKUP, send=_send_responses, served=lambda data: isinstance(data, ResponsesObject) and bool(data.output), ), -) -BREAKER_OPEN_RE: Final = re.compile(r'^litellm_redis_circuit_breaker_state\{state="open"\} ([0-9.e+]+)$', re.M) -BREAKER_TRANSITIONS_RE: Final = re.compile( - r'^litellm_redis_circuit_breaker_transitions_total\{state="[a-z_]+"\} ([0-9.e+]+)$', re.M + Endpoint( + name="embeddings", + primary="redis-timeout-embed-primary", + backup="redis-timeout-embed-backup", + primary_params=_EMBED_PRIMARY, + backup_params=_EMBED_BACKUP, + send=_send_embeddings, + served=lambda data: isinstance(data, EmbeddingsObject) and bool(data.data), + ), ) @@ -126,48 +177,51 @@ def _metric(proxy: ProxyClient, pattern: re.Pattern[str]) -> float: return sum(float(match.group(1)) for match in pattern.finditer(body)) +def _rss_bytes(proxy: ProxyClient) -> float | None: + """The proxy's resident memory from the Prometheus process collector, which reads /proc and + so reports on Linux only; None where the metric is absent.""" + body = proxy.probe("/metrics", params=NoBody()).body + match = RSS_RE.search(body) + return float(match.group(1)) if match else None + + class TestRedisTimeout: @pytest.mark.parametrize("endpoint", ENDPOINTS, ids=[endpoint.name for endpoint in ENDPOINTS]) @pytest.mark.covers( "reliability.circuit_breaker.redis_timeout.stays_responsive", - exercised_on=["chat_completions", "responses"], + exercised_on=["chat_completions", "responses", "embeddings"], ) def test_retries_under_redis_timeouts_keep_answering( self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint, paused_redis: None ) -> None: proxy = client.proxy - primary_id = proxy.create_model( - PRIMARY_MODEL, - LiteLLMParamsBody( - model=BACKING_MODEL, api_key="sk-redis-timeout-primary-not-used", api_base=CLOSED_PORT_API_BASE - ), - ) + primary_id = proxy.create_model(endpoint.primary, endpoint.primary_params) resources.defer(lambda: proxy.delete_model(primary_id)) - backup_id = proxy.create_model( - BACKUP_MODEL, - LiteLLMParamsBody(model=BACKING_MODEL, api_key="sk-redis-timeout-backup-not-used", mock_response="ok"), - ) + backup_id = proxy.create_model(endpoint.backup, endpoint.backup_params) resources.defer(lambda: proxy.delete_model(backup_id)) - timeouts_before = _metric(proxy, TIMEOUT_FAILURES_RE) - transitions_before = _metric(proxy, BREAKER_TRANSITIONS_RE) key = proxy.generate_key( KeyGenerateBody( - models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}" + models=[endpoint.primary, endpoint.backup], + key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}", ) ) resources.defer(lambda: proxy.delete_key(key)) + timeouts_before = _metric(proxy, TIMEOUT_FAILURES_RE) + transitions_before = _metric(proxy, BREAKER_TRANSITIONS_RE) + fallbacks_before = _metric(proxy, FALLBACKS_RE) + rss_before = _rss_bytes(proxy) latencies: list[float] = [] for request_number in range(1, REQUESTS + 1): started = time.monotonic() - result = endpoint.send(proxy, key, f"redis timeout {unique_marker()} {request_number}") + result = endpoint.send(proxy, key, endpoint.primary, f"redis timeout {unique_marker()} {request_number}") elapsed = time.monotonic() - started assert isinstance(result, Success), ( f"{endpoint.name} request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" ) assert endpoint.served(result.data), ( - f"{endpoint.name} request {request_number}: fallback to {BACKUP_MODEL} returned no output" + f"{endpoint.name} request {request_number}: fallback to {endpoint.backup} returned no output" ) assert elapsed < MAX_SECONDS_PER_REQUEST, ( f"{endpoint.name} request {request_number} took {elapsed:.1f}s with Redis timing out; " @@ -191,6 +245,20 @@ class TestRedisTimeout: f"/health/liveliness took {liveliness_seconds:.1f}s after the loop; the worker is stalled" ) + if rss_before is not None: + rss_after = _rss_bytes(proxy) + assert rss_after is not None + assert rss_after - rss_before <= MAX_RSS_GROWTH_BYTES, ( + f"proxy RSS grew {(rss_after - rss_before) / 2**20:.0f} MB across {REQUESTS} {endpoint.name} requests " + "with Redis timing out; on v1.100.0 this path grew by gigabytes" + ) + + fallbacks = _metric(proxy, FALLBACKS_RE) - fallbacks_before + assert fallbacks >= REQUESTS, ( + f"only {fallbacks:.0f} successful fallbacks were counted across {REQUESTS} {endpoint.name} requests; " + "the closed-port primary did not fail every request, so the retry path was not exercised" + ) + timeouts_total = _metric(proxy, TIMEOUT_FAILURES_RE) timeouts = timeouts_total - timeouts_before transitions = _metric(proxy, BREAKER_TRANSITIONS_RE) - transitions_before @@ -210,3 +278,5 @@ class TestRedisTimeout: f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; " "a Redis outage must not lose spend rows" ) + failed_rows = [row.status for row in rows if row.status not in (None, "success")] + assert not failed_rows, f"{len(failed_rows)} {endpoint.name} spend rows are not successes: {failed_rows[:3]}" From 8b965880efd8e28321c1db74850fbb2b6f59242c Mon Sep 17 00:00:00 2001 From: dclark Date: Thu, 10 Sep 2026 11:29:37 +0100 Subject: [PATCH 025/157] fix(proxy): prevent spend counter double counting --- litellm/proxy/db/spend_counter_reseed.py | 24 ++++++++-- .../proxy/db/test_spend_counter_reseed.py | 48 +++++++++++++++++++ 2 files changed, 69 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/db/spend_counter_reseed.py b/litellm/proxy/db/spend_counter_reseed.py index e35b1c8c82b..c6060a47f84 100644 --- a/litellm/proxy/db/spend_counter_reseed.py +++ b/litellm/proxy/db/spend_counter_reseed.py @@ -245,7 +245,16 @@ class SpendCounterReseed: value=current_value, ) else: - await spend_counter_cache.async_increment_cache(key=counter_key, value=db_spend, refresh_ttl=True) + # Repair/reservations can populate the counter during the DB read. + # Seed a floor without adding the database balance again. + # No await between read/compare/write: atomic within this worker. + cached = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) + current_value = float(db_spend) + if cached is not None: + current_value = max(current_value, float(cached)) + spend_counter_cache.in_memory_cache.set_cache( + key=counter_key, value=current_value + ) except Exception: verbose_proxy_logger.exception( "SpendCounterReseed.coalesced: failed to warm counter %s", @@ -438,11 +447,20 @@ class SpendCounterReseed: value=current_value, ) else: - await spend_counter_cache.async_increment_cache(key=counter_key, value=window_spend) + # Repair/reservations can populate the counter during the DB read. + # Seed a floor without adding the database balance again. + # No await between read/compare/write: atomic within this worker. + cached = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) + current_value = float(window_spend) + if cached is not None: + current_value = max(current_value, float(cached)) + spend_counter_cache.in_memory_cache.set_cache( + key=counter_key, value=current_value + ) except Exception: verbose_proxy_logger.exception( "SpendCounterReseed.coalesced_window: failed to warm counter %s", counter_key, ) raise - return window_spend + return current_value diff --git a/tests/test_litellm/proxy/db/test_spend_counter_reseed.py b/tests/test_litellm/proxy/db/test_spend_counter_reseed.py index 0361d5cfe8f..a0cf9128cc9 100644 --- a/tests/test_litellm/proxy/db/test_spend_counter_reseed.py +++ b/tests/test_litellm/proxy/db/test_spend_counter_reseed.py @@ -15,6 +15,7 @@ from typing import Final import pytest from litellm.caching.dual_cache import DualCache +from litellm.caching.in_memory_cache import InMemoryCache from litellm.constants import PROXY_DB_LOOKUP_MAX_CONCURRENCY from litellm.proxy.db.spend_counter_reseed import SpendCounterReseed @@ -270,6 +271,53 @@ async def test_coalesced_window_seeds_a_cold_counter_from_the_row(): assert prisma.db.litellm_spendlogs.call_count == 0 +@pytest.mark.asyncio +@pytest.mark.parametrize("window", [False, True], ids=["primary", "window"]) +@pytest.mark.parametrize("concurrent_spend", [989.01459411, 995.0, 900.0]) +async def test_cold_reseed_does_not_add_database_spend_to_concurrent_cache( + monkeypatch: pytest.MonkeyPatch, + window: bool, + concurrent_spend: float, +): + """A repair/reservation write may populate the counter while DB read runs. + + Reseeding must establish the larger value, not increment the concurrent + value by the same authoritative spend a second time. + """ + cache = DualCache(in_memory_cache=InMemoryCache()) + counter_key = ( + "spend:team:team-1:window:1d" if window else "spend:user:user-1" + ) + db_spend = 989.01459411 + + async def read_db(*args, **kwargs): + cache.in_memory_cache.set_cache(key=counter_key, value=concurrent_spend) + return db_spend + + if window: + monkeypatch.setattr(SpendCounterReseed, "window_from_db", staticmethod(read_db)) + result = await SpendCounterReseed.coalesced_window( + prisma_client=None, + spend_counter_cache=cache, + counter_key=counter_key, + entity_type="Team", + entity_id="team-1", + window_duration="1d", + window_start=WINDOW_START, + ) + else: + monkeypatch.setattr(SpendCounterReseed, "from_db", staticmethod(read_db)) + result = await SpendCounterReseed.coalesced( + prisma_client=None, + spend_counter_cache=cache, + counter_key=counter_key, + ) + + expected = max(db_spend, concurrent_spend) + assert cache.in_memory_cache.get_cache(key=counter_key) == expected + assert result == expected + + @pytest.mark.asyncio async def test_end_user_from_db_reads_the_end_user_row_by_user_id(): prisma: Final = _FakePrismaClient(end_user_row=SimpleNamespace(user_id="customer-42", spend=0.0)) From a39c35878569f74c50de3295320457dd7bd6adb2 Mon Sep 17 00:00:00 2001 From: dclark Date: Thu, 10 Sep 2026 13:44:08 +0100 Subject: [PATCH 026/157] fix(proxy): preserve local spend adjustments during reseeding --- litellm/proxy/db/spend_counter_reseed.py | 37 +++--- litellm/proxy/proxy_server.py | 12 +- .../proxy/db/test_spend_counter_reseed.py | 117 ++++++++++++------ 3 files changed, 102 insertions(+), 64 deletions(-) diff --git a/litellm/proxy/db/spend_counter_reseed.py b/litellm/proxy/db/spend_counter_reseed.py index c6060a47f84..0131a67db8b 100644 --- a/litellm/proxy/db/spend_counter_reseed.py +++ b/litellm/proxy/db/spend_counter_reseed.py @@ -104,6 +104,15 @@ class SpendCounterReseed: SpendCounterReseed._locks.popitem(last=False) return lock + @staticmethod + async def increment_in_memory(spend_counter_cache: "DualCache", counter_key: str, increment: float) -> float | None: + """Apply local deltas after an in-flight reseed establishes the spend balance.""" + lock: Final = await SpendCounterReseed._get_lock(counter_key) + async with lock: + return await spend_counter_cache.async_increment_cache( + key=counter_key, value=increment, local_only=True, refresh_ttl=True + ) + @staticmethod async def from_db(prisma_client: Optional["PrismaClient"], counter_key: str) -> float | None: """ @@ -245,16 +254,10 @@ class SpendCounterReseed: value=current_value, ) else: - # Repair/reservations can populate the counter during the DB read. - # Seed a floor without adding the database balance again. - # No await between read/compare/write: atomic within this worker. - cached = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) - current_value = float(db_spend) - if cached is not None: - current_value = max(current_value, float(cached)) - spend_counter_cache.in_memory_cache.set_cache( - key=counter_key, value=current_value - ) + cached_spend: Final = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) + seeded_spend: Final = max(db_spend, float(cached_spend)) if cached_spend is not None else db_spend + spend_counter_cache.in_memory_cache.set_cache(key=counter_key, value=seeded_spend) + return seeded_spend except Exception: verbose_proxy_logger.exception( "SpendCounterReseed.coalesced: failed to warm counter %s", @@ -447,16 +450,12 @@ class SpendCounterReseed: value=current_value, ) else: - # Repair/reservations can populate the counter during the DB read. - # Seed a floor without adding the database balance again. - # No await between read/compare/write: atomic within this worker. - cached = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) - current_value = float(window_spend) - if cached is not None: - current_value = max(current_value, float(cached)) - spend_counter_cache.in_memory_cache.set_cache( - key=counter_key, value=current_value + cached_spend: Final = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) + seeded_spend: Final = ( + max(window_spend, float(cached_spend)) if cached_spend is not None else window_spend ) + spend_counter_cache.in_memory_cache.set_cache(key=counter_key, value=seeded_spend) + return seeded_spend except Exception: verbose_proxy_logger.exception( "SpendCounterReseed.coalesced_window: failed to warm counter %s", diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9269fd48e6c..e8608661c48 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -3330,10 +3330,8 @@ async def _increment_spend_counter_cache(counter_key: str, increment: float): ) return current_value - return await spend_counter_cache.async_increment_cache( - key=counter_key, - value=increment, - refresh_ttl=True, + return await SpendCounterReseed.increment_in_memory( + spend_counter_cache=spend_counter_cache, counter_key=counter_key, increment=increment ) @@ -3356,10 +3354,8 @@ async def _apply_spend_counter_increments(pending: Sequence[_PendingSpendIncreme redis_cache: Final = spend_counter_cache.redis_cache if redis_cache is None: for item in pending: - await spend_counter_cache.async_increment_cache( - key=item.counter_key, - value=item.increment, - refresh_ttl=True, + await SpendCounterReseed.increment_in_memory( + spend_counter_cache=spend_counter_cache, counter_key=item.counter_key, increment=item.increment ) return ttl: Final = redis_cache.get_ttl() diff --git a/tests/test_litellm/proxy/db/test_spend_counter_reseed.py b/tests/test_litellm/proxy/db/test_spend_counter_reseed.py index a0cf9128cc9..69ec174903b 100644 --- a/tests/test_litellm/proxy/db/test_spend_counter_reseed.py +++ b/tests/test_litellm/proxy/db/test_spend_counter_reseed.py @@ -8,6 +8,7 @@ allowed to run: only when the row is missing or belongs to an older window. from __future__ import annotations import asyncio +from collections.abc import Mapping from datetime import datetime, timedelta, timezone from types import SimpleNamespace from typing import Final @@ -79,6 +80,39 @@ def _row(window_start: datetime, spend: float) -> SimpleNamespace: return SimpleNamespace(window_start=window_start, spend=spend) +class _PausedSpendTable: + def __init__(self, spend: float) -> None: + self.spend: Final = spend + self.read_started: Final = asyncio.Event() + self.resume_read: Final = asyncio.Event() + + async def find_unique(self, where: Mapping[str, object]) -> SimpleNamespace: + self.read_started.set() + await self.resume_read.wait() + return _row(WINDOW_START, self.spend) + + +async def _reseed_with_paused_table( + table: _PausedSpendTable, cache: DualCache, counter_key: str, window: bool +) -> float | None: + prisma: Final = SimpleNamespace(db=SimpleNamespace(litellm_usertable=table, litellm_budgetwindowspend=table)) + if window: + return await SpendCounterReseed.coalesced_window( + prisma_client=prisma, + spend_counter_cache=cache, + counter_key=counter_key, + entity_type="Team", + entity_id="team-1", + window_duration="1d", + window_start=WINDOW_START, + ) + return await SpendCounterReseed.coalesced( + prisma_client=prisma, + spend_counter_cache=cache, + counter_key=counter_key, + ) + + @pytest.mark.asyncio async def test_window_from_table_reads_row_by_primary_key(): """The lookup must use the table's own entity_type values ("key"), not the @@ -275,49 +309,59 @@ async def test_coalesced_window_seeds_a_cold_counter_from_the_row(): @pytest.mark.parametrize("window", [False, True], ids=["primary", "window"]) @pytest.mark.parametrize("concurrent_spend", [989.01459411, 995.0, 900.0]) async def test_cold_reseed_does_not_add_database_spend_to_concurrent_cache( - monkeypatch: pytest.MonkeyPatch, window: bool, concurrent_spend: float, -): - """A repair/reservation write may populate the counter while DB read runs. +) -> None: + cache: Final = DualCache(in_memory_cache=InMemoryCache()) + counter_key: Final = "spend:team:team-1:window:1d" if window else "spend:user:user-1" + db_spend: Final = 989.01459411 + table: Final = _PausedSpendTable(db_spend) + reseed_task: Final = asyncio.create_task(_reseed_with_paused_table(table, cache, counter_key, window)) - Reseeding must establish the larger value, not increment the concurrent - value by the same authoritative spend a second time. - """ - cache = DualCache(in_memory_cache=InMemoryCache()) - counter_key = ( - "spend:team:team-1:window:1d" if window else "spend:user:user-1" - ) - db_spend = 989.01459411 + await asyncio.wait_for(table.read_started.wait(), timeout=5) + cache.in_memory_cache.set_cache(key=counter_key, value=concurrent_spend) + table.resume_read.set() + result: Final = await asyncio.wait_for(reseed_task, timeout=5) - async def read_db(*args, **kwargs): - cache.in_memory_cache.set_cache(key=counter_key, value=concurrent_spend) - return db_spend - - if window: - monkeypatch.setattr(SpendCounterReseed, "window_from_db", staticmethod(read_db)) - result = await SpendCounterReseed.coalesced_window( - prisma_client=None, - spend_counter_cache=cache, - counter_key=counter_key, - entity_type="Team", - entity_id="team-1", - window_duration="1d", - window_start=WINDOW_START, - ) - else: - monkeypatch.setattr(SpendCounterReseed, "from_db", staticmethod(read_db)) - result = await SpendCounterReseed.coalesced( - prisma_client=None, - spend_counter_cache=cache, - counter_key=counter_key, - ) - - expected = max(db_spend, concurrent_spend) + expected: Final = max(db_spend, concurrent_spend) assert cache.in_memory_cache.get_cache(key=counter_key) == expected assert result == expected +@pytest.mark.asyncio +@pytest.mark.parametrize("window", [False, True], ids=["primary", "window"]) +@pytest.mark.parametrize("batch", [False, True], ids=["single_increment", "batch_increment"]) +@pytest.mark.parametrize("increment", [5.0, -5.0], ids=["charge", "refund"]) +async def test_cold_reseed_preserves_concurrent_local_increment( + monkeypatch: pytest.MonkeyPatch, window: bool, batch: bool, increment: float +) -> None: + from litellm.proxy import proxy_server + + cache: Final = DualCache(in_memory_cache=InMemoryCache()) + counter_key: Final = ( + f"spend:team:concurrent-{batch}-{increment}:window:1d" + if window + else f"spend:user:concurrent-{batch}-{increment}" + ) + table: Final = _PausedSpendTable(100.0) + monkeypatch.setattr(proxy_server, "spend_counter_cache", cache) + reseed_task: Final = asyncio.create_task(_reseed_with_paused_table(table, cache, counter_key, window)) + await asyncio.wait_for(table.read_started.wait(), timeout=5) + + increment_task: Final = asyncio.create_task( + proxy_server._apply_spend_counter_increments( + pending=(proxy_server._PendingSpendIncrement(counter_key=counter_key, increment=increment),) + ) + if batch + else proxy_server._increment_spend_counter_cache(counter_key=counter_key, increment=increment) + ) + await asyncio.sleep(0) + table.resume_read.set() + await asyncio.wait_for(asyncio.gather(reseed_task, increment_task), timeout=5) + + assert cache.in_memory_cache.get_cache(key=counter_key) == 100.0 + increment + + @pytest.mark.asyncio async def test_end_user_from_db_reads_the_end_user_row_by_user_id(): prisma: Final = _FakePrismaClient(end_user_row=SimpleNamespace(user_id="customer-42", spend=0.0)) @@ -352,8 +396,7 @@ async def test_end_user_from_db_ignores_other_counter_kinds_without_touching_the @pytest.mark.asyncio async def test_end_user_from_db_returns_none_without_a_row_a_client_or_on_db_error(): assert ( - await SpendCounterReseed.end_user_from_db(prisma_client=None, counter_key="spend:end_user:customer-42") - is None + await SpendCounterReseed.end_user_from_db(prisma_client=None, counter_key="spend:end_user:customer-42") is None ) assert ( await SpendCounterReseed.end_user_from_db( From 60c7dd8348cb1634784efd908afa046233858585 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 19:13:49 +0000 Subject: [PATCH 027/157] fix(model_prices): add deepseek-flash and gpt-live-1, bill DeepSeek legacy flash aliases at Flash rates Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 77 +++++++++++++++---- model_prices_and_context_window.json | 77 +++++++++++++++---- tests/test_litellm/test_utils.py | 31 ++++++-- 3 files changed, 146 insertions(+), 39 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d8160a3d842..7f3e0565f88 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -56459,17 +56459,43 @@ "supports_reasoning": true, "source": "https://serverless.tensormesh.ai/v1/models/openrouter" }, - "deepseek-v4-flash": { + "deepseek-flash": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "deepseek-v4-flash": { + "cache_creation_input_token_cost": 0.0, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, + "litellm_provider": "deepseek", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56487,15 +56513,15 @@ }, "deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56539,15 +56565,15 @@ }, "deepseek/deepseek-v4-flash": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56565,15 +56591,15 @@ }, "deepseek/deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56929,6 +56955,23 @@ ], "supports_audio_input": true }, + "gpt-live-1": { + "input_cost_per_second": 0.0008333333333333334, + "litellm_provider": "openai", + "mode": "realtime", + "source": "https://developers.openai.com/api/docs/models/gpt-live-1", + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true + }, "gpt-realtime-translate": { "input_cost_per_second": 0.0005666666666666667, "litellm_provider": "openai", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d8160a3d842..7f3e0565f88 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -56459,17 +56459,43 @@ "supports_reasoning": true, "source": "https://serverless.tensormesh.ai/v1/models/openrouter" }, - "deepseek-v4-flash": { + "deepseek-flash": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "deepseek-v4-flash": { + "cache_creation_input_token_cost": 0.0, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, + "litellm_provider": "deepseek", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56487,15 +56513,15 @@ }, "deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56539,15 +56565,15 @@ }, "deepseek/deepseek-v4-flash": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56565,15 +56591,15 @@ }, "deepseek/deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, - "cache_read_input_token_cost": 1.4e-08, - "input_cost_per_token": 4.4e-07, - "input_cost_per_token_cache_hit": 1.4e-08, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, "litellm_provider": "deepseek", "max_input_tokens": 1000000, "max_output_tokens": 393216, "max_tokens": 393216, "mode": "chat", - "output_cost_per_token": 1.32e-06, + "output_cost_per_token": 1.2e-06, "source": "https://api-docs.deepseek.com/quick_start/pricing", "supported_endpoints": [ "/v1/chat/completions" @@ -56929,6 +56955,23 @@ ], "supports_audio_input": true }, + "gpt-live-1": { + "input_cost_per_second": 0.0008333333333333334, + "litellm_provider": "openai", + "mode": "realtime", + "source": "https://developers.openai.com/api/docs/models/gpt-live-1", + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true + }, "gpt-realtime-translate": { "input_cost_per_second": 0.0005666666666666667, "litellm_provider": "openai", diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 8b186be43e5..34b7f263dcd 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -4071,7 +4071,7 @@ def test_deepseek_v4_models_in_cost_map(): configured in model_prices_and_context_window.json. Prices sourced from https://api-docs.deepseek.com/quick_start/pricing: - - deepseek-v4-flash: $0.44/M input, $1.32/M output + - deepseek-v4-flash: $0.30/M input, $1.20/M output - deepseek-v4-pro: $1.32/M input, $3.96/M output Closes https://github.com/BerriAI/litellm/issues/26709 @@ -4085,7 +4085,7 @@ def test_deepseek_v4_models_in_cost_map(): # --- bare model names --- for key, expected_input, expected_output, expected_cache in [ - ("deepseek-v4-flash", 4.4e-07, 1.32e-06, 1.4e-08), + ("deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), ("deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), ]: info = model_cost.get(key) @@ -4101,7 +4101,7 @@ def test_deepseek_v4_models_in_cost_map(): # --- provider-prefixed names --- for key, expected_input, expected_output, expected_cache in [ - ("deepseek/deepseek-v4-flash", 4.4e-07, 1.32e-06, 1.4e-08), + ("deepseek/deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), ("deepseek/deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), ]: info = model_cost.get(key) @@ -4129,7 +4129,7 @@ def test_deepseek_v4_models_in_backup_cost_map(): # --- bare model names --- for key, expected_input, expected_output, expected_cache in [ - ("deepseek-v4-flash", 4.4e-07, 1.32e-06, 1.4e-08), + ("deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), ("deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), ]: info = model_cost.get(key) @@ -4143,7 +4143,7 @@ def test_deepseek_v4_models_in_backup_cost_map(): # --- provider-prefixed names --- for key, expected_input, expected_output, expected_cache in [ - ("deepseek/deepseek-v4-flash", 4.4e-07, 1.32e-06, 1.4e-08), + ("deepseek/deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), ("deepseek/deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), ]: info = model_cost.get(key) @@ -4155,6 +4155,27 @@ def test_deepseek_v4_models_in_backup_cost_map(): assert info["cache_read_input_token_cost"] == expected_cache +def test_deepseek_flash_completion_cost(): + from litellm.types.utils import ModelResponse + + response = ModelResponse( + model="deepseek-flash", + usage=Usage( + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + total_tokens=2_000_000, + ), + ) + + cost = litellm.completion_cost( + completion_response=response, + model="deepseek-flash", + custom_llm_provider="deepseek", + ) + + assert cost == pytest.approx(1.50, abs=1e-9) + + _FIREWORKS_MODELS = [ ( "accounts/fireworks/models/glm-5p2", From 896f0dff2e0f59d8d1d2ac8cc0c75c1ee611f881 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 19:14:28 +0000 Subject: [PATCH 028/157] fix(model_prices): add provider-prefixed deepseek/deepseek-flash Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 26 +++++++++++++++++++ model_prices_and_context_window.json | 26 +++++++++++++++++++ 2 files changed, 52 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7f3e0565f88..91a24bfae2e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -56563,6 +56563,32 @@ "supports_tool_choice": true, "supports_vision": false }, + "deepseek/deepseek-flash": { + "cache_creation_input_token_cost": 0.0, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, + "litellm_provider": "deepseek", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true + }, "deepseek/deepseek-v4-flash": { "cache_creation_input_token_cost": 0.0, "cache_read_input_token_cost": 6e-09, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7f3e0565f88..91a24bfae2e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -56563,6 +56563,32 @@ "supports_tool_choice": true, "supports_vision": false }, + "deepseek/deepseek-flash": { + "cache_creation_input_token_cost": 0.0, + "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 3e-07, + "input_cost_per_token_cache_hit": 6e-09, + "litellm_provider": "deepseek", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true + }, "deepseek/deepseek-v4-flash": { "cache_creation_input_token_cost": 0.0, "cache_read_input_token_cost": 6e-09, From ace6de88a4cb3dc052908f394f64c17a73e5b816 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 19:31:11 +0000 Subject: [PATCH 029/157] fix(tests): load local model costs for DeepSeek flash regression Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/test_litellm/test_utils.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 34b7f263dcd..a2d430e8807 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -4155,6 +4155,7 @@ def test_deepseek_v4_models_in_backup_cost_map(): assert info["cache_read_input_token_cost"] == expected_cache +@pytest.mark.usefixtures("local_model_cost_map") def test_deepseek_flash_completion_cost(): from litellm.types.utils import ModelResponse From 8deb465346317baf550acb948ff2c72bdf8b9315 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 13:18:52 -0700 Subject: [PATCH 030/157] fix(realtime): dial Azure's GA realtime upstream for GA clients Azure realtime defaulted to the beta upstream whenever realtime_protocol was not configured, so a GA client's session.update (session.type, output_modalities, nested audio) was forwarded unchanged to /openai/realtime and Azure rejected it with "Unknown parameter: 'session.type'" on gpt-realtime and gpt-realtime-1.5. The unset default now follows the client the way the OpenAI handler already does: a client that sends OpenAI-Beta: realtime=v1 keeps the beta upstream, any other client gets /openai/v1/realtime. An explicit realtime_protocol in litellm_params or LITELLM_AZURE_REALTIME_PROTOCOL still wins. --- litellm/realtime_api/main.py | 22 ++++++-- tests/test_litellm/realtime_api/test_main.py | 59 +++++++++++++++++++- 2 files changed, 74 insertions(+), 7 deletions(-) diff --git a/litellm/realtime_api/main.py b/litellm/realtime_api/main.py index b824a5928c6..5310efba1c4 100644 --- a/litellm/realtime_api/main.py +++ b/litellm/realtime_api/main.py @@ -14,6 +14,7 @@ from litellm.constants import ( request_timeout, ) from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider +from litellm.litellm_core_utils.realtime_streaming import client_sent_openai_beta_realtime_header from litellm.llms.base_llm.realtime.transformation import BaseRealtimeConfig from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from litellm.llms.xai.common_utils import XAIModelInfo @@ -413,14 +414,14 @@ async def _arealtime( api_version = api_version or litellm_params.api_version or "2024-10-01-preview" - realtime_protocol = ( + configured_realtime_protocol: Final = ( kwargs.get("realtime_protocol") or litellm_params.get("realtime_protocol") or os.environ.get("LITELLM_AZURE_REALTIME_PROTOCOL") ) - if realtime_protocol is None and (query_params or {}).get("intent") == "transcription": - realtime_protocol = "GA" - realtime_protocol = realtime_protocol or "beta" + realtime_protocol: Final = _azure_realtime_protocol_for_client( + configured_realtime_protocol, query_params=query_params, websocket=websocket + ) resolved_azure_ad_token: Final = ( None if api_key else get_azure_ad_token(GenericLiteLLMParams(**kwargs, azure_ad_token=azure_ad_token)) ) @@ -576,6 +577,19 @@ def _is_transcription_only_realtime_model(model: str, custom_llm_provider: str) _TRANSCRIPTION_QUERY_PARAMS: Final[RealtimeQueryParams] = {"intent": "transcription"} +def _azure_realtime_protocol_for_client( + configured_protocol: object, + *, + query_params: RealtimeQueryParams | None, + websocket: "WebSocket", +) -> str: + if isinstance(configured_protocol, str) and configured_protocol: + return configured_protocol + if (query_params or {}).get("intent") == "transcription": + return "GA" + return "beta" if client_sent_openai_beta_realtime_header(websocket) else "GA" + + def _azure_realtime_health_protocol( model: str, realtime_protocol: str | None, model_params: Mapping[str, object] ) -> tuple[str, RealtimeQueryParams | None]: diff --git a/tests/test_litellm/realtime_api/test_main.py b/tests/test_litellm/realtime_api/test_main.py index 761e87ac764..8a9abe819e5 100644 --- a/tests/test_litellm/realtime_api/test_main.py +++ b/tests/test_litellm/realtime_api/test_main.py @@ -327,7 +327,60 @@ async def test_arealtime_azure_ai_on_a_foundry_host_connects_to_the_azure_openai api_key="fake-key", litellm_logging_obj=FakeLogging(), ) - assert connect.url == ( - "wss://my-project.services.ai.azure.com/openai/realtime" - "?api-version=2024-10-01-preview&deployment=gpt-realtime-mini" + assert connect.url == "wss://my-project.services.ai.azure.com/openai/v1/realtime?model=gpt-realtime-mini" + + +class _ClientWebSocketWithHeaders: + def __init__(self, headers: tuple[tuple[bytes, bytes], ...]) -> None: + self.scope: Final = {"headers": headers} + + +_GA_CLIENT: Final = _ClientWebSocketWithHeaders(headers=()) +_BETA_CLIENT: Final = _ClientWebSocketWithHeaders(headers=((b"openai-beta", b"realtime=v1"),)) + + +async def _azure_backend_url_dialed_for(websocket: _ClientWebSocketWithHeaders, **kwargs: object) -> str | None: + connect: Final = _ConnectThatStopsAfterCapturingTheUrl() + with patch("websockets.connect", connect): + await realtime_main._arealtime.__wrapped__( + model="azure/gpt-realtime", + websocket=websocket, + api_base="https://my-endpoint.openai.azure.com", + api_key="fake-key", + litellm_logging_obj=FakeLogging(), + **kwargs, + ) + return connect.url + + +@pytest.mark.asyncio +async def test_arealtime_azure_ga_client_without_beta_header_dials_the_ga_upstream(monkeypatch): + monkeypatch.delenv("LITELLM_AZURE_REALTIME_PROTOCOL", raising=False) + assert ( + await _azure_backend_url_dialed_for(_GA_CLIENT) + == "wss://my-endpoint.openai.azure.com/openai/v1/realtime?model=gpt-realtime" + ) + + +@pytest.mark.asyncio +async def test_arealtime_azure_beta_header_client_keeps_the_beta_upstream(monkeypatch): + monkeypatch.delenv("LITELLM_AZURE_REALTIME_PROTOCOL", raising=False) + assert await _azure_backend_url_dialed_for(_BETA_CLIENT) == ( + "wss://my-endpoint.openai.azure.com/openai/realtime?api-version=2024-10-01-preview&deployment=gpt-realtime" + ) + + +@pytest.mark.asyncio +async def test_arealtime_azure_explicit_beta_protocol_wins_over_a_ga_client(monkeypatch): + monkeypatch.delenv("LITELLM_AZURE_REALTIME_PROTOCOL", raising=False) + assert await _azure_backend_url_dialed_for(_GA_CLIENT, realtime_protocol="beta") == ( + "wss://my-endpoint.openai.azure.com/openai/realtime?api-version=2024-10-01-preview&deployment=gpt-realtime" + ) + + +@pytest.mark.asyncio +async def test_arealtime_azure_env_beta_protocol_wins_over_a_ga_client(monkeypatch): + monkeypatch.setenv("LITELLM_AZURE_REALTIME_PROTOCOL", "beta") + assert await _azure_backend_url_dialed_for(_GA_CLIENT) == ( + "wss://my-endpoint.openai.azure.com/openai/realtime?api-version=2024-10-01-preview&deployment=gpt-realtime" ) From 8fe2094a5579e048c5d79f374f4a6ad7d7261326 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 13:46:13 -0700 Subject: [PATCH 031/157] refactor(realtime): move the Azure protocol picker into the Azure realtime handler --- litellm/llms/azure/realtime/handler.py | 23 +++++++++++++++++++++-- litellm/realtime_api/main.py | 18 ++---------------- 2 files changed, 23 insertions(+), 18 deletions(-) diff --git a/litellm/llms/azure/realtime/handler.py b/litellm/llms/azure/realtime/handler.py index e9913f0108d..88813c21cd9 100644 --- a/litellm/llms/azure/realtime/handler.py +++ b/litellm/llms/azure/realtime/handler.py @@ -6,17 +6,23 @@ This requires websockets, and is currently only supported on LiteLLM Proxy. from collections.abc import Mapping from types import MappingProxyType -from typing import Any, Final, Protocol, cast +from typing import TYPE_CHECKING, Any, Final, Protocol, cast from litellm._logging import _redact_string, verbose_proxy_logger from litellm.constants import REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES from litellm.types.realtime import RealtimeQueryParams from ....litellm_core_utils.litellm_logging import Logging as LiteLLMLogging -from ....litellm_core_utils.realtime_streaming import RealTimeStreaming +from ....litellm_core_utils.realtime_streaming import ( + RealTimeStreaming, + client_sent_openai_beta_realtime_header, +) from ....llms.custom_httpx.http_handler import get_shared_realtime_ssl_context from ..azure import AzureChatCompletion +if TYPE_CHECKING: + from fastapi import WebSocket + # BACKEND_WS_URL = "ws://localhost:8080/v1/realtime?model=gpt-4o-realtime-preview-2024-10-01" @@ -31,6 +37,19 @@ async def forward_messages(client_ws: Any, backend_ws: Any): pass +def azure_realtime_protocol_for_client( + configured_protocol: object, + *, + query_params: RealtimeQueryParams | None, + websocket: "WebSocket", +) -> str: + if isinstance(configured_protocol, str) and configured_protocol: + return configured_protocol + if (query_params or {}).get("intent") == "transcription": + return "GA" + return "beta" if client_sent_openai_beta_realtime_header(websocket) else "GA" + + class _ProxyClientWebSocket(Protocol): """Client-facing websocket handle: this path only closes it after a failed handshake.""" diff --git a/litellm/realtime_api/main.py b/litellm/realtime_api/main.py index 5310efba1c4..d42e7e18b75 100644 --- a/litellm/realtime_api/main.py +++ b/litellm/realtime_api/main.py @@ -14,7 +14,6 @@ from litellm.constants import ( request_timeout, ) from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider -from litellm.litellm_core_utils.realtime_streaming import client_sent_openai_beta_realtime_header from litellm.llms.base_llm.realtime.transformation import BaseRealtimeConfig from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from litellm.llms.xai.common_utils import XAIModelInfo @@ -34,7 +33,7 @@ from litellm.utils import ProviderConfigManager from ..litellm_core_utils.get_litellm_params import get_litellm_params from ..litellm_core_utils.litellm_logging import Logging as LiteLLMLogging from ..llms.azure.common_utils import get_azure_ad_token -from ..llms.azure.realtime.handler import AzureOpenAIRealtime +from ..llms.azure.realtime.handler import AzureOpenAIRealtime, azure_realtime_protocol_for_client from ..llms.bedrock.realtime.handler import BedrockRealtime from ..llms.custom_httpx.http_handler import get_shared_realtime_ssl_context from ..llms.openai.realtime.handler import OpenAIRealtime @@ -419,7 +418,7 @@ async def _arealtime( or litellm_params.get("realtime_protocol") or os.environ.get("LITELLM_AZURE_REALTIME_PROTOCOL") ) - realtime_protocol: Final = _azure_realtime_protocol_for_client( + realtime_protocol: Final = azure_realtime_protocol_for_client( configured_realtime_protocol, query_params=query_params, websocket=websocket ) resolved_azure_ad_token: Final = ( @@ -577,19 +576,6 @@ def _is_transcription_only_realtime_model(model: str, custom_llm_provider: str) _TRANSCRIPTION_QUERY_PARAMS: Final[RealtimeQueryParams] = {"intent": "transcription"} -def _azure_realtime_protocol_for_client( - configured_protocol: object, - *, - query_params: RealtimeQueryParams | None, - websocket: "WebSocket", -) -> str: - if isinstance(configured_protocol, str) and configured_protocol: - return configured_protocol - if (query_params or {}).get("intent") == "transcription": - return "GA" - return "beta" if client_sent_openai_beta_realtime_header(websocket) else "GA" - - def _azure_realtime_health_protocol( model: str, realtime_protocol: str | None, model_params: Mapping[str, object] ) -> tuple[str, RealtimeQueryParams | None]: From 4a3950cf678fa1b65acfc468c61ae3dcbb67b472 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 13:57:48 -0700 Subject: [PATCH 032/157] refactor(realtime): type the Azure protocol picker with the streaming module's websocket protocol --- litellm/litellm_core_utils/realtime_streaming.py | 8 ++++---- litellm/llms/azure/realtime/handler.py | 8 +++----- 2 files changed, 7 insertions(+), 9 deletions(-) diff --git a/litellm/litellm_core_utils/realtime_streaming.py b/litellm/litellm_core_utils/realtime_streaming.py index 75046f2cf87..4923bdda305 100644 --- a/litellm/litellm_core_utils/realtime_streaming.py +++ b/litellm/litellm_core_utils/realtime_streaming.py @@ -90,12 +90,12 @@ class _ResponseDoneBody(TypedDict, total=False): output: ReadOnly[Sequence[Mapping[str, object]]] -class _ScopedWebSocket(Protocol): +class ScopedWebSocket(Protocol): @property def scope(self) -> _ASGIScope: ... -class _ClientWebSocket(_ScopedWebSocket, Protocol): +class _ClientWebSocket(ScopedWebSocket, Protocol): async def send_text(self, data: str) -> None: ... async def receive_text(self) -> str: ... async def close(self, code: int = 1000, reason: str | None = None) -> None: ... @@ -1149,7 +1149,7 @@ class RealTimeStreaming: ) @staticmethod - def _detect_beta_header(websocket: _ScopedWebSocket) -> bool: + def _detect_beta_header(websocket: ScopedWebSocket) -> bool: """Return True if the client sent 'OpenAI-Beta: realtime=v1'. Checks the raw ASGI scope headers so it works for both FastAPI WebSocket @@ -1584,6 +1584,6 @@ class RealTimeStreaming: verbose_logger.debug("Could not relay the upstream close to the client: %s", e) -def client_sent_openai_beta_realtime_header(websocket: _ScopedWebSocket) -> bool: +def client_sent_openai_beta_realtime_header(websocket: ScopedWebSocket) -> bool: """True when the client WebSocket includes ``OpenAI-Beta: realtime=v1``.""" return RealTimeStreaming._detect_beta_header(websocket) diff --git a/litellm/llms/azure/realtime/handler.py b/litellm/llms/azure/realtime/handler.py index 88813c21cd9..146915dd6fd 100644 --- a/litellm/llms/azure/realtime/handler.py +++ b/litellm/llms/azure/realtime/handler.py @@ -6,7 +6,7 @@ This requires websockets, and is currently only supported on LiteLLM Proxy. from collections.abc import Mapping from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Protocol, cast +from typing import Any, Final, Protocol, cast from litellm._logging import _redact_string, verbose_proxy_logger from litellm.constants import REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES @@ -15,14 +15,12 @@ from litellm.types.realtime import RealtimeQueryParams from ....litellm_core_utils.litellm_logging import Logging as LiteLLMLogging from ....litellm_core_utils.realtime_streaming import ( RealTimeStreaming, + ScopedWebSocket, client_sent_openai_beta_realtime_header, ) from ....llms.custom_httpx.http_handler import get_shared_realtime_ssl_context from ..azure import AzureChatCompletion -if TYPE_CHECKING: - from fastapi import WebSocket - # BACKEND_WS_URL = "ws://localhost:8080/v1/realtime?model=gpt-4o-realtime-preview-2024-10-01" @@ -41,7 +39,7 @@ def azure_realtime_protocol_for_client( configured_protocol: object, *, query_params: RealtimeQueryParams | None, - websocket: "WebSocket", + websocket: ScopedWebSocket, ) -> str: if isinstance(configured_protocol, str) and configured_protocol: return configured_protocol From 32b7daf69116907678c03f4d78f713080db52317 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 20:58:41 +0000 Subject: [PATCH 033/157] fix(caching): make an open Redis circuit breaker a quiet cache miss An open breaker raised a generic Exception on every skipped call, and DualCache caught it and logged a full ERROR traceback each time. Under load that became hundreds of traceback formats per second on every replica and pinned the proxies at 100% CPU. Raise a typed RedisCircuitBreakerOpenError instead and have DualCache return its in-memory result without logging for it. Classify redis-py pool exhaustion (ConnectionError chained from TimeoutError) as a timeout so a latency blip goes through the duration gate. Track a breaker generation so a call admitted before the breaker opened cannot close it, leaving that to the HALF_OPEN probe. Rate limit the LoggingWorker callback-error traceback to one per interval so a stalled logging backend cannot start a second traceback storm. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/dual_cache.py | 15 +- litellm/caching/redis_cache.py | 76 ++++++++--- litellm/constants.py | 1 + litellm/litellm_core_utils/logging_worker.py | 23 +++- tests/test_litellm/caching/test_dual_cache.py | 62 +++++++++ .../test_litellm/caching/test_redis_cache.py | 128 ++++++++++++++++++ .../litellm_core_utils/test_logging_worker.py | 45 ++++++ 7 files changed, 326 insertions(+), 24 deletions(-) diff --git a/litellm/caching/dual_cache.py b/litellm/caching/dual_cache.py index ec17cc1d809..98c6da0f02f 100644 --- a/litellm/caching/dual_cache.py +++ b/litellm/caching/dual_cache.py @@ -23,7 +23,7 @@ from litellm.constants import DEFAULT_MAX_REDIS_BATCH_CACHE_SIZE from .base_cache import BaseCache from .in_memory_cache import InMemoryCache -from .redis_cache import RedisCache +from .redis_cache import RedisCache, RedisCircuitBreakerOpenError if TYPE_CHECKING: from opentelemetry.trace import Span as _Span @@ -250,6 +250,8 @@ class DualCache(BaseCache): print_verbose(f"get cache: cache result: {result}") return result + except RedisCircuitBreakerOpenError: + return None except Exception: verbose_logger.error(traceback.format_exc()) @@ -319,6 +321,9 @@ class DualCache(BaseCache): redis_result: Final = await self.redis_cache.async_batch_get_cache( sublist_keys, parent_otel_span=parent_otel_span ) + except RedisCircuitBreakerOpenError: + self._rollback_redis_batch_key_reservations(previous_access_times) + return result except Exception: # Do not throttle subsequent callers if the Redis read fails. self._rollback_redis_batch_key_reservations(previous_access_times) @@ -352,6 +357,8 @@ class DualCache(BaseCache): if self.redis_cache is not None and local_only is False: await self.redis_cache.async_set_cache(key, value, **kwargs) + except RedisCircuitBreakerOpenError: + return except Exception as e: verbose_logger.exception("LiteLLM Cache: Excepton async add_cache: %s", e) @@ -371,6 +378,8 @@ class DualCache(BaseCache): await self.redis_cache.async_set_cache_pipeline( cache_list=cache_list, ttl=kwargs.pop("ttl", None), **kwargs ) + except RedisCircuitBreakerOpenError: + return except Exception as e: verbose_logger.exception("LiteLLM Cache: Excepton async add_cache: %s", e) @@ -408,6 +417,8 @@ class DualCache(BaseCache): refresh_ttl=refresh_ttl, ) + return result + except RedisCircuitBreakerOpenError: return result except Exception as e: verbose_logger.warning( @@ -437,6 +448,8 @@ class DualCache(BaseCache): parent_otel_span=parent_otel_span, ) + return result + except RedisCircuitBreakerOpenError: return result except Exception as e: verbose_logger.warning( diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 106c1580110..de4dcb5aa27 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -17,6 +17,7 @@ import json import time from collections.abc import Awaitable, Callable, Sequence from contextvars import ContextVar +from dataclasses import dataclass from datetime import timedelta from typing import TYPE_CHECKING, Any, Final, Protocol, TypeVar, cast @@ -148,6 +149,10 @@ def _get_call_stack_info(num_frames: int = 2) -> str: return "unknown" +class RedisCircuitBreakerOpenError(Exception): + """Expected fast-fail while the breaker is open; optional-cache callers treat it as a miss.""" + + class RedisCircuitBreaker: """ Tracks Redis health for a RedisCache instance. @@ -163,8 +168,12 @@ class RedisCircuitBreaker: (no success or hard failure in between) that reaches failure_threshold and spans timeout_min_duration seconds OPEN -> HALF_OPEN after recovery_timeout seconds - HALF_OPEN -> CLOSED on success - HALF_OPEN -> OPEN on failure (resets timer) + HALF_OPEN -> CLOSED on the recovery probe's success + HALF_OPEN -> OPEN on the recovery probe's failure (resets timer) + + Every OPEN transition starts a new generation. A call reports its outcome only for the + generation it was admitted under, so a success from a call that was already in flight + when the breaker opened cannot close it and the HALF_OPEN probe is the only call that can. Timeouts are accounted separately from hard connectivity failures because the async Redis timeout includes time waiting for the worker event loop to resume: one loop @@ -194,9 +203,14 @@ class RedisCircuitBreaker: self._timeout_count = 0 self._timeout_streak_started_at: float | None = None self._opened_at: float | None = None + self._generation = 0 self._state = self.CLOSED _breaker_metrics().record_state_change(None, self._state) + @property + def generation(self) -> int: + return self._generation + def is_open(self) -> bool: """Returns True if Redis calls should be skipped.""" if not self.enabled: @@ -241,18 +255,19 @@ class RedisCircuitBreaker: if self._state != self.OPEN: verbose_logger.warning( "Redis circuit breaker OPENED after %d consecutive failures" - " (%d hard connectivity) — fast-failing Redis calls for %ds", + " (%d hard connectivity), fast-failing Redis calls for %ds", self._failure_count, self._hard_failure_count, self.recovery_timeout, ) + self._generation += 1 self._set_state(self.OPEN) def record_success(self) -> None: if not self.enabled: return if self._state == self.HALF_OPEN: - verbose_logger.info("Redis circuit breaker CLOSED — Redis recovered") + verbose_logger.info("Redis circuit breaker CLOSED, Redis recovered") self._failure_count = 0 self._hard_failure_count = 0 self._timeout_count = 0 @@ -320,7 +335,10 @@ def _redis_timeout_error_types() -> tuple[type, ...]: def _is_redis_timeout_failure(exc: BaseException) -> bool: - return isinstance(exc, _redis_timeout_error_types()) + """Follows __cause__: a blocking pool wait raises ConnectionError from asyncio.TimeoutError.""" + if isinstance(exc, _redis_timeout_error_types()): + return True + return exc.__cause__ is not None and _is_redis_timeout_failure(exc.__cause__) class _BreakerMetrics: @@ -391,23 +409,37 @@ def _record_swallowed_redis_failure(breaker: RedisCircuitBreaker, exc: BaseExcep _swallowed_redis_failures.set(_swallowed_redis_failures.get() + 1) -def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> int: - """Reject the call if the breaker is open, else return the swallowed-failure count to compare against.""" +@dataclass(frozen=True, slots=True) +class _BreakerAdmission: + swallowed_before: int + generation: int + + +def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> _BreakerAdmission: + """Reject the call if the breaker is open, else snapshot what its outcome will be judged against.""" if breaker.is_open(): - raise Exception(f"Redis circuit breaker is open — skipping {name}") - return _swallowed_redis_failures.get() + raise RedisCircuitBreakerOpenError(f"Redis circuit breaker is open, skipping {name}") + return _BreakerAdmission(swallowed_before=_swallowed_redis_failures.get(), generation=breaker.generation) -def _exit_circuit_breaker(breaker: RedisCircuitBreaker, swallowed_before: int) -> None: - """Record success only when nothing failed while the call ran. +def _exit_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission) -> None: + """Record success only when nothing failed while the call ran and the breaker has not opened since. Several Redis methods catch their own connection errors and return a default, so a method that returned is not on its own proof of a healthy Redis. """ - if _swallowed_redis_failures.get() == swallowed_before: + if admission.generation != breaker.generation: + return + if _swallowed_redis_failures.get() == admission.swallowed_before: breaker.record_success() +def _fail_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission, exc: BaseException) -> None: + if admission.generation != breaker.generation or not _is_redis_health_failure(exc): + return + breaker.record_failure(is_timeout=_is_redis_timeout_failure(exc)) + + async def _run_under_circuit_breaker( breaker: RedisCircuitBreaker, name: str, @@ -418,14 +450,13 @@ async def _run_under_circuit_breaker( Shared by the method decorator and the Lua script executor so both feed the same health signal. """ - swallowed_before: Final = _enter_circuit_breaker(breaker, name) + admission: Final = _enter_circuit_breaker(breaker, name) try: result: Final = await call() except Exception as e: - if _is_redis_health_failure(e): - breaker.record_failure(is_timeout=_is_redis_timeout_failure(e)) + _fail_circuit_breaker(breaker, admission, e) raise - _exit_circuit_breaker(breaker, swallowed_before) + _exit_circuit_breaker(breaker, admission) return result @@ -435,14 +466,13 @@ def _run_under_circuit_breaker_sync( call: Callable[[], _RedisCallResult], ) -> _RedisCallResult: """Run one blocking Redis call under a circuit breaker, feeding the same health signal as the async path.""" - swallowed_before: Final = _enter_circuit_breaker(breaker, name) + admission: Final = _enter_circuit_breaker(breaker, name) try: result: Final = call() except Exception as e: - if _is_redis_health_failure(e): - breaker.record_failure() + _fail_circuit_breaker(breaker, admission, e) raise - _exit_circuit_breaker(breaker, swallowed_before) + _exit_circuit_breaker(breaker, admission) return result @@ -1382,10 +1412,10 @@ class RedisCache(BaseCache): start_time: Final = time.time() try: - swallowed_before: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") + admission: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") _keys: Final = [self.check_and_fix_namespace(key=cache_key or "") for cache_key in _key_list] results: Final = self._run_redis_mget_operation(keys=_keys) - _exit_circuit_breaker(self._circuit_breaker, swallowed_before) + _exit_circuit_breaker(self._circuit_breaker, admission) end_time: Final = time.time() _duration: Final = end_time - start_time self.service_logger_obj.service_success_hook( @@ -1409,6 +1439,8 @@ class RedisCache(BaseCache): decoded_results[k] = v return decoded_results + except RedisCircuitBreakerOpenError: + return key_value_dict except Exception as e: failed_at: Final = time.time() self.service_logger_obj.service_failure_hook( diff --git a/litellm/constants.py b/litellm/constants.py index 0a7ef363e92..421419fecc0 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -571,6 +571,7 @@ ANTHROPIC_MESSAGES_MAX_DETACHED_STREAM_DRAINS: Final = int( LOGGING_WORKER_CONCURRENCY: Final = int(os.getenv("LOGGING_WORKER_CONCURRENCY", 100)) # Must be above 0 LOGGING_WORKER_MAX_QUEUE_SIZE: Final = int(os.getenv("LOGGING_WORKER_MAX_QUEUE_SIZE", 50_000)) LOGGING_WORKER_MAX_TIME_PER_COROUTINE: Final = float(os.getenv("LOGGING_WORKER_MAX_TIME_PER_COROUTINE", 20.0)) +LOGGING_WORKER_ERROR_TRACEBACK_INTERVAL_SECONDS: Final = 60.0 LOGGING_WORKER_CLEAR_PERCENTAGE: Final = int( os.getenv("LOGGING_WORKER_CLEAR_PERCENTAGE", 50) ) # Percentage of queue to clear (default: 50%) diff --git a/litellm/litellm_core_utils/logging_worker.py b/litellm/litellm_core_utils/logging_worker.py index 1d74595781a..dfaace1bb10 100644 --- a/litellm/litellm_core_utils/logging_worker.py +++ b/litellm/litellm_core_utils/logging_worker.py @@ -6,6 +6,7 @@ import atexit import contextvars import inspect import logging +import time from collections.abc import Coroutine, Iterator from typing import Final @@ -16,6 +17,7 @@ from litellm.constants import ( LOGGING_WORKER_AGGRESSIVE_CLEAR_COOLDOWN_SECONDS, LOGGING_WORKER_CLEAR_PERCENTAGE, LOGGING_WORKER_CONCURRENCY, + LOGGING_WORKER_ERROR_TRACEBACK_INTERVAL_SECONDS, LOGGING_WORKER_MAX_QUEUE_SIZE, LOGGING_WORKER_MAX_TIME_PER_COROUTINE, MAX_ITERATIONS_TO_CLEAR_QUEUE, @@ -47,10 +49,14 @@ class LoggingWorker: timeout: float = LOGGING_WORKER_MAX_TIME_PER_COROUTINE, max_queue_size: int = LOGGING_WORKER_MAX_QUEUE_SIZE, concurrency: int = LOGGING_WORKER_CONCURRENCY, + error_traceback_interval: float = LOGGING_WORKER_ERROR_TRACEBACK_INTERVAL_SECONDS, ): self.timeout = timeout self.max_queue_size = max_queue_size self.concurrency = concurrency + self.error_traceback_interval = error_traceback_interval + self._last_error_traceback_at: float | None = None + self._errors_since_traceback: int = 0 self._queue: asyncio.Queue[LoggingTask] | None = None self._worker_task: asyncio.Task | None = None self._running_tasks: set[asyncio.Task] = set() @@ -163,7 +169,7 @@ class LoggingWorker: timeout=self.timeout, ) except Exception as e: - verbose_logger.exception("LoggingWorker error: %s", e) + self._log_task_error(e) finally: self._untrack_dequeued(task) self._queue.task_done() @@ -171,6 +177,21 @@ class LoggingWorker: # Always release semaphore, even if queue is None sem.release() + def _log_task_error(self, error: Exception) -> None: + """One traceback per interval: a stalled backend fails every in-flight task at once.""" + now: Final = time.monotonic() + last_traceback_at: Final = self._last_error_traceback_at + if last_traceback_at is not None and now - last_traceback_at < self.error_traceback_interval: + self._errors_since_traceback += 1 + return + verbose_logger.exception( + "LoggingWorker error (%d more suppressed since the last traceback): %r", + self._errors_since_traceback, + error, + ) + self._last_error_traceback_at = now + self._errors_since_traceback = 0 + async def _worker_loop(self) -> None: """Main worker loop that gets tasks and schedules them to run concurrently.""" try: diff --git a/tests/test_litellm/caching/test_dual_cache.py b/tests/test_litellm/caching/test_dual_cache.py index ded3be26630..46dd2687e48 100644 --- a/tests/test_litellm/caching/test_dual_cache.py +++ b/tests/test_litellm/caching/test_dual_cache.py @@ -1,4 +1,5 @@ import asyncio +import logging import time import uuid from unittest.mock import AsyncMock, MagicMock, patch @@ -576,3 +577,64 @@ async def test_dual_cache_late_attach_redis_wires_writes_and_ttl_async(): assert mock_redis.async_set_cache.call_args[0][:2] == (key_after, val_after) assert in_memory.get_cache(key_after) == val_after + + +@pytest.fixture +def dual_cache_with_open_breaker(): + """A DualCache whose Redis tier is behind an already-open circuit breaker. + + The Redis client is a mock that fails the test if anything reaches it, so every + guarded call has to be short-circuited by the breaker. + """ + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import _is_redis_timeout_failure + from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + + with ( + patch("asyncio.get_running_loop", side_effect=RuntimeError("No running event loop")), + patch( # test-quality-ok: RedisCache.__init__ builds its client eagerly, with no injection point + "litellm._redis.get_redis_client", return_value=MagicMock() + ), + ): + redis_cache = RedisCache(host="127.0.0.1", port=6379) + unreachable = AsyncMock() + unreachable.get.side_effect = AssertionError("an open breaker must not touch Redis") + unreachable.mget.side_effect = AssertionError("an open breaker must not touch Redis") + unreachable.set.side_effect = AssertionError("an open breaker must not touch Redis") + unreachable.pipeline.side_effect = AssertionError("an open breaker must not touch Redis") + for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): + redis_cache._circuit_breaker.record_failure(is_timeout=_is_redis_timeout_failure(RedisConnectionError("refused"))) + assert redis_cache._circuit_breaker.is_open() is True + with patch.object(redis_cache, "init_async_client", return_value=unreachable): + yield DualCache(in_memory_cache=InMemoryCache(), redis_cache=redis_cache) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "call, expected", + [ + pytest.param(lambda c: c.async_get_cache("lit7468"), lambda n: None, id="async_get_cache"), + pytest.param( + lambda c: c.async_batch_get_cache(["lit7468", "lit7460"]), lambda n: [None, None], id="async_batch_get_cache" + ), + pytest.param(lambda c: c.async_set_cache("lit7468", "v"), lambda n: None, id="async_set_cache"), + pytest.param(lambda c: c.async_set_cache_pipeline([("lit7468", "v")]), lambda n: None, id="async_set_cache_pipeline"), + pytest.param(lambda c: c.async_increment_cache("lit7468", 1.0, ttl=60), float, id="async_increment_cache"), + ], +) +async def test_open_breaker_is_a_quiet_cache_miss(dual_cache_with_open_breaker, call, expected, caplog): + """While the breaker is open, every cache operation must degrade to the in-memory result + without logging anything above DEBUG. + + Before this, each skipped call raised a generic exception that DualCache caught and logged + as a full ERROR traceback. Under production request rates that was hundreds of stack + formats per second per replica, enough to pin every proxy at 100% CPU on a Redis blip. + """ + caplog.set_level(logging.DEBUG, logger="LiteLLM") + + for n in range(1, 51): + assert await call(dual_cache_with_open_breaker) == expected(n) + + noisy = [r for r in caplog.records if r.levelno > logging.DEBUG] + assert noisy == [], f"an open breaker must be silent per call, got {[r.getMessage() for r in noisy]}" diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index 6b2df118611..856f9b49c56 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -1013,3 +1013,131 @@ async def test_breaker_metrics_track_state_and_failure_class(): breaker.record_success() assert sample("litellm_redis_circuit_breaker_state", {"state": "open"}) == open_gauge_before assert sample("litellm_redis_circuit_breaker_state", {"state": "closed"}) == closed_gauge_before + 1 + + +@pytest.mark.asyncio +async def test_pool_exhaustion_counts_as_a_timeout_not_a_hard_failure(): + """redis-py reports a blocking pool that waited out its timeout as a ConnectionError + chained from the underlying TimeoutError. That is Redis being slow, the same signal as + a read timeout, so a burst of them must go through the duration gate instead of opening + the breaker on the fifth one as though Redis had refused the connection. + """ + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import ( + RedisCircuitBreaker, + _is_redis_timeout_failure, + _run_under_circuit_breaker, + ) + + def pool_exhausted() -> RedisConnectionError: + """Built the way redis-py's BlockingConnectionPool.get_connection raises it.""" + try: + try: + raise asyncio.TimeoutError() + except asyncio.TimeoutError as err: + raise RedisConnectionError("No connection available.") from err + except RedisConnectionError as chained: + return chained + + assert _is_redis_timeout_failure(pool_exhausted()) is True + assert _is_redis_timeout_failure(RedisConnectionError("Connection refused")) is False + + breaker = RedisCircuitBreaker(failure_threshold=3, recovery_timeout=60, timeout_min_duration=5.0) + + async def pool_exhausted_call(): + raise pool_exhausted() + + for _ in range(breaker.failure_threshold + 1): + with pytest.raises(RedisConnectionError, match="No connection available"): + await _run_under_circuit_breaker(breaker, "op", pool_exhausted_call) + + assert breaker.is_open() is False, "an instantaneous burst of pool waits must not open the breaker" + + +@pytest.mark.asyncio +async def test_success_admitted_before_the_breaker_opened_cannot_close_it(): + """Only the HALF_OPEN recovery probe may close the breaker. + + A call that was already in flight when the breaker opened knows nothing about whether + Redis has recovered. Letting its late success close the breaker made the state flap + OPEN -> CLOSED -> OPEN under load, and every OPEN transition re-logged the warning while + the next five failures each paid the full socket timeout again. + """ + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import ( + RedisCircuitBreaker, + RedisCircuitBreakerOpenError, + _run_under_circuit_breaker, + ) + + breaker = RedisCircuitBreaker(failure_threshold=2, recovery_timeout=60) + release_slow_success = asyncio.Event() + + async def slow_success(): + await release_slow_success.wait() + return "ok" + + async def refused(): + raise RedisConnectionError("refused") + + in_flight = asyncio.create_task(_run_under_circuit_breaker(breaker, "slow", slow_success)) + await asyncio.sleep(0) + for _ in range(breaker.failure_threshold): + with pytest.raises(RedisConnectionError): + await _run_under_circuit_breaker(breaker, "op", refused) + assert breaker.is_open() is True + + release_slow_success.set() + assert await in_flight == "ok" + + assert breaker.is_open() is True, "a pre-open success is not a recovery probe" + with pytest.raises(RedisCircuitBreakerOpenError): + await _run_under_circuit_breaker(breaker, "op", slow_success) + + +@pytest.mark.asyncio +async def test_open_breaker_raises_its_own_exception_type(): + """Callers with an optional cache need to tell the expected fast-fail apart from a real error.""" + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import ( + RedisCircuitBreaker, + RedisCircuitBreakerOpenError, + _run_under_circuit_breaker, + ) + + breaker = RedisCircuitBreaker(failure_threshold=1, recovery_timeout=60) + + async def refused(): + raise RedisConnectionError("refused") + + with pytest.raises(RedisConnectionError): + await _run_under_circuit_breaker(breaker, "op", refused) + + with pytest.raises(RedisCircuitBreakerOpenError, match="circuit breaker is open"): + await _run_under_circuit_breaker(breaker, "op", refused) + + +@pytest.mark.asyncio +async def test_recovery_probe_still_closes_the_breaker(): + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import RedisCircuitBreaker, _run_under_circuit_breaker + + breaker = RedisCircuitBreaker(failure_threshold=1, recovery_timeout=0.05) + + async def refused(): + raise RedisConnectionError("refused") + + async def recovered(): + return "ok" + + with pytest.raises(RedisConnectionError): + await _run_under_circuit_breaker(breaker, "op", refused) + assert breaker.is_open() is True + + await asyncio.sleep(0.06) + assert await _run_under_circuit_breaker(breaker, "probe", recovered) == "ok" + assert breaker.is_open() is False diff --git a/tests/test_litellm/litellm_core_utils/test_logging_worker.py b/tests/test_litellm/litellm_core_utils/test_logging_worker.py index eb4e893adb8..f93a2e0b08a 100644 --- a/tests/test_litellm/litellm_core_utils/test_logging_worker.py +++ b/tests/test_litellm/litellm_core_utils/test_logging_worker.py @@ -525,3 +525,48 @@ class TestLoggingWorker: asyncio.run(rebind_on_second_loop()) assert sorted(executed) == [0, 1, 2, 3, 4] + + @pytest.mark.asyncio + async def test_callback_timeout_burst_logs_one_traceback_per_interval(self, caplog): + """A slow logging backend times out every in-flight callback at once. Logging a full + traceback for each of them turns that stall into a CPU-bound log storm on every replica, + so the worker must emit one traceback per interval and count the rest. + """ + caplog.set_level(logging.DEBUG, logger="LiteLLM") + worker = LoggingWorker(timeout=0.05, max_queue_size=200, concurrency=100, error_traceback_interval=60.0) + worker.start() + + async def stalled_callback(): + await asyncio.sleep(10) + + for _ in range(40): + worker.enqueue(stalled_callback()) + + await asyncio.sleep(0.5) + await worker.stop() + + errors = [r for r in caplog.records if r.levelno >= logging.ERROR and "LoggingWorker error" in r.getMessage()] + assert len(errors) == 1, f"expected one traceback for the burst, got {len(errors)}" + assert errors[0].exc_info is not None + + @pytest.mark.asyncio + async def test_error_traceback_resumes_after_interval_with_suppressed_count(self, caplog): + caplog.set_level(logging.DEBUG, logger="LiteLLM") + worker = LoggingWorker(timeout=0.05, max_queue_size=200, concurrency=100, error_traceback_interval=0.2) + worker.start() + + async def stalled_callback(): + await asyncio.sleep(10) + + for _ in range(5): + worker.enqueue(stalled_callback()) + await asyncio.sleep(0.3) + for _ in range(3): + worker.enqueue(stalled_callback()) + await asyncio.sleep(0.3) + await worker.stop() + + messages = [r.getMessage() for r in caplog.records if "LoggingWorker error" in r.getMessage()] + assert len(messages) == 2, messages + assert "(0 more suppressed" in messages[0] + assert "(4 more suppressed" in messages[1] From 1957bd388ed1c8771f767c01ccf25aabaf19ac37 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 21:15:02 +0000 Subject: [PATCH 034/157] fix(proxy): treat an open Redis breaker as a dropped spend counter update, not a cost tracking failure Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/proxy_server.py | 7 +++- .../proxy/proxy_server/test_spend_counters.py | 42 +++++++++++++++++++ 2 files changed, 47 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9269fd48e6c..ff8f8ac8647 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -240,6 +240,7 @@ import litellm._redis from litellm import Router from litellm._logging import _redact_string, verbose_proxy_logger, verbose_router_logger from litellm.caching.caching import DualCache, RedisCache +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.caching.redis_cluster_cache import RedisClusterCache from litellm.constants import ( _REALTIME_BODY_CACHE_SIZE, @@ -3369,9 +3370,11 @@ async def _apply_spend_counter_increments(pending: Sequence[_PendingSpendIncreme ] try: results: Final = await redis_cache.async_increment_pipeline(increment_list=increment_list) - except Exception: + except Exception as e: await asyncio.gather(*(_invalidate_spend_counter(counter_key=item.counter_key) for item in pending)) - raise + if not isinstance(e, RedisCircuitBreakerOpenError): + raise + return for item, current_value in zip(pending, results or ()): spend_counter_cache.in_memory_cache.set_cache(key=item.counter_key, value=current_value) diff --git a/tests/test_litellm/proxy/proxy_server/test_spend_counters.py b/tests/test_litellm/proxy/proxy_server/test_spend_counters.py index c343652efd9..f5965838f05 100644 --- a/tests/test_litellm/proxy/proxy_server/test_spend_counters.py +++ b/tests/test_litellm/proxy/proxy_server/test_spend_counters.py @@ -1348,6 +1348,48 @@ async def test_is_spend_counter_cache_warm_redis_error_falls_back_to_in_memory( assert result is False +# --------------------------------------------------------------------------- +# _apply_spend_counter_increments +# --------------------------------------------------------------------------- + + +@pytest.mark.asyncio +async def test_apply_spend_counter_increments_open_breaker_invalidates_and_returns(monkeypatch): + """An open Redis breaker fast-fails the pipeline on every request, so the callback + must not turn each one into a tracking-cost failure: drop the stale local copies + and return like a miss, without raising.""" + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + + fake_cache = _make_spend_counter_cache() + fake_cache.redis_cache.async_increment_pipeline = AsyncMock( + side_effect=RedisCircuitBreakerOpenError("Redis circuit breaker is open, skipping async_increment_pipeline") + ) + monkeypatch.setattr(ps, "spend_counter_cache", fake_cache) + pending: Final = ( + ps._PendingSpendIncrement(counter_key="spend:key:k", increment=1.0), + ps._PendingSpendIncrement(counter_key="spend:team:t", increment=1.0), + ) + + await ps._apply_spend_counter_increments(pending=pending) + + deleted: Final = sorted(call.kwargs["key"] for call in fake_cache.in_memory_cache.delete_cache.call_args_list) + assert deleted == ["spend:key:k", "spend:team:t"] + assert fake_cache.in_memory_cache.set_cache.called is False + + +@pytest.mark.asyncio +async def test_apply_spend_counter_increments_other_redis_error_invalidates_and_raises(monkeypatch): + fake_cache = _make_spend_counter_cache() + fake_cache.redis_cache.async_increment_pipeline = AsyncMock(side_effect=RuntimeError("incr fail")) + monkeypatch.setattr(ps, "spend_counter_cache", fake_cache) + pending: Final = (ps._PendingSpendIncrement(counter_key="spend:key:k", increment=1.0),) + + with pytest.raises(RuntimeError): + await ps._apply_spend_counter_increments(pending=pending) + + assert fake_cache.in_memory_cache.delete_cache.called is True + + # --------------------------------------------------------------------------- # _increment_spend_counter_cache # --------------------------------------------------------------------------- From 14b5915d4c74dcfc1490fb4e58caf3af42110253 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 21:24:29 +0000 Subject: [PATCH 035/157] fix(proxy): log the open-breaker TTL preservation fallback at debug instead of per request warning Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../hooks/parallel_request_limiter_v3.py | 9 +++++-- .../hooks/test_parallel_request_limiter_v3.py | 24 +++++++++++++++++++ 2 files changed, 31 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/hooks/parallel_request_limiter_v3.py b/litellm/proxy/hooks/parallel_request_limiter_v3.py index c6c3dde4b6e..7fdc919de2c 100644 --- a/litellm/proxy/hooks/parallel_request_limiter_v3.py +++ b/litellm/proxy/hooks/parallel_request_limiter_v3.py @@ -26,6 +26,7 @@ from typing_extensions import NotRequired, ReadOnly from litellm import DualCache from litellm._logging import verbose_proxy_logger +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.constants import DYNAMIC_RATE_LIMIT_ERROR_THRESHOLD_PER_MINUTE, INTERNAL_CALL_ORIGIN_METADATA_KEY from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.prompt_templates.common_utils import ( @@ -3856,8 +3857,12 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger): ) except Exception as e: - verbose_proxy_logger.warning("TTL preservation failed, falling back to regular pipeline: %s", e) - # Fallback to regular pipeline on error + log: Final = ( + verbose_proxy_logger.debug + if isinstance(e, RedisCircuitBreakerOpenError) + else verbose_proxy_logger.warning + ) + log("TTL preservation failed, falling back to regular pipeline: %s", e) await self.internal_usage_cache.dual_cache.async_increment_cache_pipeline( increment_list=pipeline_operations, litellm_parent_otel_span=parent_otel_span, diff --git a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py index 6d382370f5f..2d49338753d 100644 --- a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py +++ b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py @@ -1795,6 +1795,30 @@ async def test_async_increment_tokens_fallback_behavior(): ), "Fallback method should be called when Lua script is not available" +@pytest.mark.asyncio +async def test_async_increment_tokens_open_breaker_falls_back_without_warning(caplog): + """An open Redis breaker fast-fails the Lua script on every request, so it must fall + back to the regular pipeline quietly instead of emitting a WARNING per request.""" + from unittest.mock import AsyncMock + + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + + handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(DualCache())) + handler.token_increment_script = object() + handler._execute_token_increment_script = AsyncMock( + side_effect=RedisCircuitBreakerOpenError("Redis circuit breaker is open, skipping run_script") + ) + fallback = AsyncMock() + handler.internal_usage_cache.dual_cache.async_increment_cache_pipeline = fallback + pipeline_operations = [RedisPipelineIncrementOperation(key="test_breaker_key", increment_value=10.0, ttl=60)] + + with caplog.at_level(logging.DEBUG, logger="LiteLLM Proxy"): + await handler.async_increment_tokens_with_ttl_preservation(pipeline_operations=pipeline_operations) + + assert fallback.await_count == 1 + assert [r.levelno for r in caplog.records if "TTL preservation failed" in r.getMessage()] == [logging.DEBUG] + + # Redis Cluster Compatibility Tests def test_group_keys_by_hash_tag_regular_redis(): """ From ad78a8f8d0a4e6200a98d5ef0ddae252b1e5c154 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 21:35:49 +0000 Subject: [PATCH 036/157] refactor(caching): walk exception causes iteratively for the breaker timeout check Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/redis_cache.py | 21 ++++++++++++++++----- 1 file changed, 16 insertions(+), 5 deletions(-) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index de4dcb5aa27..742c7144784 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -15,7 +15,7 @@ import hashlib import inspect import json import time -from collections.abc import Awaitable, Callable, Sequence +from collections.abc import Awaitable, Callable, Iterator, Sequence from contextvars import ContextVar from dataclasses import dataclass from datetime import timedelta @@ -334,11 +334,22 @@ def _redis_timeout_error_types() -> tuple[type, ...]: return (RedisTimeoutError, TimeoutError) +_MAX_EXCEPTION_CAUSE_DEPTH: Final = 20 + + +def _exception_cause_chain(exc: BaseException) -> Iterator[BaseException]: + current = exc # rebind-ok: advances one link per iteration of the bounded walk + for _ in range(_MAX_EXCEPTION_CAUSE_DEPTH): + yield current + if current.__cause__ is None: + return + current = current.__cause__ + + def _is_redis_timeout_failure(exc: BaseException) -> bool: - """Follows __cause__: a blocking pool wait raises ConnectionError from asyncio.TimeoutError.""" - if isinstance(exc, _redis_timeout_error_types()): - return True - return exc.__cause__ is not None and _is_redis_timeout_failure(exc.__cause__) + """Walks __cause__: a blocking pool wait raises ConnectionError from asyncio.TimeoutError.""" + timeout_types: Final = _redis_timeout_error_types() + return any(isinstance(cause, timeout_types) for cause in _exception_cause_chain(exc)) class _BreakerMetrics: From df8a9c72ab44c9fa5552b1757f36d2fe99e39a7f Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 21:45:46 +0000 Subject: [PATCH 037/157] fix(caching): keep the sync Redis read and the top-level cache wrapper quiet when the breaker is open The sync RedisCache.get_cache ran outside the breaker and logged with a stray positional argument, so every failure printed a logging-module stack dump. Cache.add_cache and its async twins logged a full traceback for the expected open-breaker fast fail Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/caching.py | 8 +++++- litellm/caching/dual_cache.py | 2 ++ litellm/caching/redis_cache.py | 4 ++- tests/test_litellm/caching/test_caching.py | 26 +++++++++++++++++++ tests/test_litellm/caching/test_dual_cache.py | 15 +++++++++++ .../test_litellm/caching/test_redis_cache.py | 23 +++++++++++++++- 6 files changed, 75 insertions(+), 3 deletions(-) diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 884d095793c..b5423998401 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -32,7 +32,7 @@ from .dual_cache import DualCache # noqa: F401 from .gcs_cache import GCSCache from .in_memory_cache import InMemoryCache from .qdrant_semantic_cache import QdrantSemanticCache -from .redis_cache import RedisCache +from .redis_cache import RedisCache, RedisCircuitBreakerOpenError from .redis_cluster_cache import RedisClusterCache from .redis_semantic_cache import RedisSemanticCache from .s3_cache import S3Cache @@ -677,6 +677,8 @@ class Cache: return cache_key, cached_data, kwargs = self._add_cache_logic(result=result, **kwargs) self.cache.set_cache(cache_key, cached_data, **kwargs) + except RedisCircuitBreakerOpenError as e: + verbose_logger.debug("LiteLLM Cache: skipped add_cache: %s", e) except Exception as e: verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e) @@ -696,6 +698,8 @@ class Cache: await dynamic_cache_object.async_set_cache(cache_key, cached_data, **kwargs) else: await self.cache.async_set_cache(cache_key, cached_data, **kwargs) + except RedisCircuitBreakerOpenError as e: + verbose_logger.debug("LiteLLM Cache: skipped add_cache: %s", e) except Exception as e: verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e) @@ -875,6 +879,8 @@ class Cache: await dynamic_cache_object.async_set_cache_pipeline(cache_list=cache_list, **kwargs) else: await self.cache.async_set_cache_pipeline(cache_list=cache_list, **kwargs) + except RedisCircuitBreakerOpenError as e: + verbose_logger.debug("LiteLLM Cache: skipped add_cache: %s", e) except Exception as e: verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e) diff --git a/litellm/caching/dual_cache.py b/litellm/caching/dual_cache.py index 98c6da0f02f..6fb918ce748 100644 --- a/litellm/caching/dual_cache.py +++ b/litellm/caching/dual_cache.py @@ -177,6 +177,8 @@ class DualCache(BaseCache): print_verbose(f"get cache: cache result: {result}") return result + except RedisCircuitBreakerOpenError: + return None except Exception: verbose_logger.error(traceback.format_exc()) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 742c7144784..6c19b018c1d 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -1364,6 +1364,7 @@ class RedisCache(BaseCache): except Exception: return ast.literal_eval(decoded) + @_redis_circuit_breaker_guard_sync def get_cache(self, key, parent_otel_span: Span | None = None, **kwargs): try: key = self.check_and_fix_namespace(key=key) @@ -1384,7 +1385,8 @@ class RedisCache(BaseCache): return self._get_cache_logic(cached_response=cached_response) except Exception as e: # NON blocking - notify users Redis is throwing an exception - verbose_logger.error("litellm.caching.caching: get() - Got exception from REDIS: ", e) + verbose_logger.error("litellm.caching.caching: get() - Got exception from REDIS: %s", e) + _record_swallowed_redis_failure(self._circuit_breaker, e) def _run_redis_mget_operation(self, keys: list[str]) -> Sequence[bytes | str | None]: """ diff --git a/tests/test_litellm/caching/test_caching.py b/tests/test_litellm/caching/test_caching.py index 4d0ec0fb677..5d326a9ece0 100644 --- a/tests/test_litellm/caching/test_caching.py +++ b/tests/test_litellm/caching/test_caching.py @@ -252,3 +252,29 @@ def test_exact_cache_key_includes_anthropic_messages_params(anthropic_param): assert baseline != cache.get_cache_key( model="claude-sonnet-4-5", messages=messages, **anthropic_param ) + + +@pytest.mark.asyncio +async def test_async_add_cache_treats_an_open_breaker_as_a_quiet_skip(caplog): + """The top-level write wrapper logs a full ERROR traceback for any failure. An open Redis + breaker fails every write instantly, so under load that wrapper alone was hundreds of + stack formats per second per replica. Unexpected failures must still get the traceback. + """ + from unittest.mock import AsyncMock + + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + + caplog.set_level(logging.DEBUG, logger="LiteLLM") + cache = Cache(type=LiteLLMCacheType.LOCAL) + cache.cache.async_set_cache = AsyncMock(side_effect=RedisCircuitBreakerOpenError("open")) + + await cache.async_add_cache("result", model="gpt-5.4-nano", messages=[{"role": "user", "content": "hi"}]) + + assert [r.levelno for r in caplog.records if "add_cache" in r.getMessage()] == [logging.DEBUG] + cache.cache.async_set_cache.assert_awaited_once() + + cache.cache.async_set_cache = AsyncMock(side_effect=OSError("disk full")) + await cache.async_add_cache("result", model="gpt-5.4-nano", messages=[{"role": "user", "content": "hi"}]) + + errors = [r for r in caplog.records if r.levelno == logging.ERROR] + assert len(errors) == 1 and errors[0].exc_info is not None diff --git a/tests/test_litellm/caching/test_dual_cache.py b/tests/test_litellm/caching/test_dual_cache.py index 46dd2687e48..3e21362b3f0 100644 --- a/tests/test_litellm/caching/test_dual_cache.py +++ b/tests/test_litellm/caching/test_dual_cache.py @@ -638,3 +638,18 @@ async def test_open_breaker_is_a_quiet_cache_miss(dual_cache_with_open_breaker, noisy = [r for r in caplog.records if r.levelno > logging.DEBUG] assert noisy == [], f"an open breaker must be silent per call, got {[r.getMessage() for r in noisy]}" + + +def test_open_breaker_is_a_quiet_cache_miss_on_the_sync_read_path(dual_cache_with_open_breaker, caplog): + """The sync read runs in the request thread pool for /v1/messages and /v1/responses, so it + must short-circuit on an open breaker like the async path instead of dialing Redis per call. + """ + caplog.set_level(logging.DEBUG, logger="LiteLLM") + redis_client = dual_cache_with_open_breaker.redis_cache.redis_client + redis_client.get.side_effect = AssertionError("an open breaker must not touch Redis") + + for _ in range(50): + assert dual_cache_with_open_breaker.get_cache("lit7468") is None + + noisy = [r for r in caplog.records if r.levelno > logging.DEBUG] + assert noisy == [], f"an open breaker must be silent per call, got {[r.getMessage() for r in noisy]}" diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index 856f9b49c56..1a884d44db1 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -1,4 +1,5 @@ import asyncio +import logging from collections.abc import Iterator from unittest.mock import AsyncMock, MagicMock, patch @@ -646,7 +647,6 @@ def test_sync_batch_get_cache_survives_a_service_callback_that_raises( from concurrent.futures import ThreadPoolExecutor import litellm - from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD cache, service_logger = sync_batch_cache_with_service_logger @@ -1141,3 +1141,24 @@ async def test_recovery_probe_still_closes_the_breaker(): await asyncio.sleep(0.06) assert await _run_under_circuit_breaker(breaker, "probe", recovered) == "ok" assert breaker.is_open() is False + + +def test_sync_get_cache_failure_feeds_the_breaker_and_logs_a_well_formed_record(sync_batch_redis_cache, caplog): + """The sync read used to log with a stray positional arg, so every Redis failure produced a + `--- Logging error ---` stack dump on stderr, and it sat outside the breaker so it kept dialing + Redis on every request even after the async paths had opened it. + """ + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + + caplog.set_level(logging.ERROR, logger="LiteLLM") + sync_batch_redis_cache.redis_client.get.side_effect = RedisConnectionError("refused") + + for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): + assert sync_batch_redis_cache.get_cache("lit7468") is None + + assert sync_batch_redis_cache._circuit_breaker.is_open() is True + assert sync_batch_redis_cache.redis_client.get.call_count == REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + assert all("refused" in record.getMessage() for record in caplog.records) + assert len(caplog.records) == REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD From be7dce30a3ab832bdd104b2d2f15bace5d622b4d Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Thu, 10 Sep 2026 15:16:24 -0700 Subject: [PATCH 038/157] ci(e2e): pin the Redis timeout workflow's Postgres image by digest Co-Authored-By: Claude Code --- .github/workflows/test-e2e-redis-timeout.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test-e2e-redis-timeout.yml b/.github/workflows/test-e2e-redis-timeout.yml index 9c318a1346c..501af0f240f 100644 --- a/.github/workflows/test-e2e-redis-timeout.yml +++ b/.github/workflows/test-e2e-redis-timeout.yml @@ -15,7 +15,7 @@ jobs: timeout-minutes: 30 services: postgres: - image: postgres:16.6 + image: postgres:16.6@sha256:557fea37a744d5f4c8faab304b0a90858b53ab119735a88c131fd19dab802f36 env: POSTGRES_USER: llmproxy POSTGRES_PASSWORD: dbpassword9090 From 7658cd53aaf5f8cc217b9a9a25891851eaa64cbc Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 22:25:03 +0000 Subject: [PATCH 039/157] fix(caching): judge swallowed Redis failures per admission and throttle worker tracebacks per error type Swallowed failures are now reported to the breaker when the admitted call exits, so a call admitted before the breaker opened cannot refresh the open timer or knock out the recovery probe. The sync batch read raises the typed open-breaker error like its async twin so DualCache releases its batch reservations, and LoggingWorker throttles tracebacks per exception class instead of worker-wide Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/dual_cache.py | 3 + litellm/caching/redis_cache.py | 60 ++++++++++------- litellm/litellm_core_utils/logging_worker.py | 20 +++--- tests/test_litellm/caching/test_dual_cache.py | 20 ++++++ .../test_litellm/caching/test_redis_cache.py | 65 +++++++++++++++++-- .../litellm_core_utils/test_logging_worker.py | 32 ++++++++- 6 files changed, 162 insertions(+), 38 deletions(-) diff --git a/litellm/caching/dual_cache.py b/litellm/caching/dual_cache.py index 6fb918ce748..5dd0493cd65 100644 --- a/litellm/caching/dual_cache.py +++ b/litellm/caching/dual_cache.py @@ -206,6 +206,9 @@ class DualCache(BaseCache): redis_result: Final = self.redis_cache.batch_get_cache( key_list=sublist_keys, parent_otel_span=parent_otel_span ) + except RedisCircuitBreakerOpenError: + self._rollback_redis_batch_key_reservations(previous_access_times) + return result except Exception: # Do not throttle subsequent callers if the Redis read fails. self._rollback_redis_batch_key_reservations(previous_access_times) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 6c19b018c1d..3ecf805ed0e 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -285,7 +285,9 @@ class RedisCircuitBreaker: _RedisCallResult = TypeVar("_RedisCallResult") -_swallowed_redis_failures: Final[ContextVar[int]] = ContextVar("litellm_swallowed_redis_failures", default=0) +_swallowed_redis_failures: Final[ContextVar[tuple[bool, ...]]] = ContextVar( + "litellm_swallowed_redis_failures", default=() +) def _opaque_kwarg_key(value: object) -> str: @@ -405,8 +407,8 @@ def _breaker_metrics() -> _BreakerMetrics: return _BreakerMetrics() -def _record_swallowed_redis_failure(breaker: RedisCircuitBreaker, exc: BaseException) -> None: - """Record a Redis failure that the calling method is about to swallow. +def _record_swallowed_redis_failure(exc: BaseException) -> None: + """Note a Redis failure that the calling method is about to swallow, for the breaker exit to judge. The marker is a ContextVar rather than a counter on the breaker because breakers are shared by every concurrent caller. A plain shared counter cannot tell "my call failed" @@ -416,8 +418,7 @@ def _record_swallowed_redis_failure(breaker: RedisCircuitBreaker, exc: BaseExcep """ if not _is_redis_health_failure(exc): return - breaker.record_failure(is_timeout=_is_redis_timeout_failure(exc)) - _swallowed_redis_failures.set(_swallowed_redis_failures.get() + 1) + _swallowed_redis_failures.set((*_swallowed_redis_failures.get(), _is_redis_timeout_failure(exc))) @dataclass(frozen=True, slots=True) @@ -430,25 +431,42 @@ def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> _BreakerA """Reject the call if the breaker is open, else snapshot what its outcome will be judged against.""" if breaker.is_open(): raise RedisCircuitBreakerOpenError(f"Redis circuit breaker is open, skipping {name}") - return _BreakerAdmission(swallowed_before=_swallowed_redis_failures.get(), generation=breaker.generation) + return _BreakerAdmission(swallowed_before=len(_swallowed_redis_failures.get()), generation=breaker.generation) + + +def _take_swallowed_failures(admission: _BreakerAdmission) -> tuple[bool, ...]: + """Return the is_timeout flag of every failure this call swallowed, and drop them from the context.""" + all_swallowed: Final = _swallowed_redis_failures.get() + _swallowed_redis_failures.set(all_swallowed[: admission.swallowed_before]) + return all_swallowed[admission.swallowed_before :] def _exit_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission) -> None: - """Record success only when nothing failed while the call ran and the breaker has not opened since. + """Report the call's outcome to the breaker generation it was admitted under. Several Redis methods catch their own connection errors and return a default, so a - method that returned is not on its own proof of a healthy Redis. + method that returned is not on its own proof of a healthy Redis. A call admitted + before the breaker opened reports nothing: its failures would refresh the open + timer or knock out the recovery probe, and its success would close it early. """ + swallowed: Final = _take_swallowed_failures(admission) if admission.generation != breaker.generation: return - if _swallowed_redis_failures.get() == admission.swallowed_before: + if not swallowed: breaker.record_success() + return + for is_timeout in swallowed: + breaker.record_failure(is_timeout=is_timeout) def _fail_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission, exc: BaseException) -> None: - if admission.generation != breaker.generation or not _is_redis_health_failure(exc): + swallowed: Final = _take_swallowed_failures(admission) + if admission.generation != breaker.generation: return - breaker.record_failure(is_timeout=_is_redis_timeout_failure(exc)) + for is_timeout in swallowed: + breaker.record_failure(is_timeout=is_timeout) + if _is_redis_health_failure(exc): + breaker.record_failure(is_timeout=_is_redis_timeout_failure(exc)) async def _run_under_circuit_breaker( @@ -1056,7 +1074,7 @@ class RedisCache(BaseCache): str(e), value, ) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) async def _pipeline_helper( self, @@ -1143,7 +1161,7 @@ class RedisCache(BaseCache): str(e), cache_value, ) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) async def _set_cache_sadd_helper( self, @@ -1228,7 +1246,7 @@ class RedisCache(BaseCache): str(e), value, ) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) @_redis_circuit_breaker_guard async def batch_cache_write(self, key, value, **kwargs): @@ -1386,7 +1404,7 @@ class RedisCache(BaseCache): except Exception as e: # NON blocking - notify users Redis is throwing an exception verbose_logger.error("litellm.caching.caching: get() - Got exception from REDIS: %s", e) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) def _run_redis_mget_operation(self, keys: list[str]) -> Sequence[bytes | str | None]: """ @@ -1423,9 +1441,9 @@ class RedisCache(BaseCache): key_value_dict = {} _key_list: Final = [key for key in key_list if key is not None] start_time: Final = time.time() + admission: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") try: - admission: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") _keys: Final = [self.check_and_fix_namespace(key=cache_key or "") for cache_key in _key_list] results: Final = self._run_redis_mget_operation(keys=_keys) _exit_circuit_breaker(self._circuit_breaker, admission) @@ -1452,8 +1470,6 @@ class RedisCache(BaseCache): decoded_results[k] = v return decoded_results - except RedisCircuitBreakerOpenError: - return key_value_dict except Exception as e: failed_at: Final = time.time() self.service_logger_obj.service_failure_hook( @@ -1466,7 +1482,7 @@ class RedisCache(BaseCache): parent_otel_span=parent_otel_span, ) verbose_logger.error("Error occurred in batch get cache - %s", e) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _fail_circuit_breaker(self._circuit_breaker, admission, e) return key_value_dict @_redis_circuit_breaker_guard @@ -1513,7 +1529,7 @@ class RedisCache(BaseCache): ) ) print_verbose(f"litellm.caching.caching: async get() - Got exception from REDIS: {e}") - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) @_redis_circuit_breaker_guard async def async_batch_get_cache( @@ -1585,7 +1601,7 @@ class RedisCache(BaseCache): ) ) verbose_logger.error("Error occurred in async batch get cache - %s", e) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) return key_value_dict def sync_ping(self) -> bool: @@ -1837,7 +1853,7 @@ class RedisCache(BaseCache): return ttl except Exception as e: verbose_logger.debug("Redis TTL Error: %s", e) - _record_swallowed_redis_failure(self._circuit_breaker, e) + _record_swallowed_redis_failure(e) return None @_redis_circuit_breaker_guard diff --git a/litellm/litellm_core_utils/logging_worker.py b/litellm/litellm_core_utils/logging_worker.py index dfaace1bb10..de3999d63ea 100644 --- a/litellm/litellm_core_utils/logging_worker.py +++ b/litellm/litellm_core_utils/logging_worker.py @@ -55,8 +55,8 @@ class LoggingWorker: self.max_queue_size = max_queue_size self.concurrency = concurrency self.error_traceback_interval = error_traceback_interval - self._last_error_traceback_at: float | None = None - self._errors_since_traceback: int = 0 + self._last_error_traceback_at: dict[type[BaseException], float] = {} + self._errors_since_traceback: dict[type[BaseException], int] = {} self._queue: asyncio.Queue[LoggingTask] | None = None self._worker_task: asyncio.Task | None = None self._running_tasks: set[asyncio.Task] = set() @@ -178,19 +178,21 @@ class LoggingWorker: sem.release() def _log_task_error(self, error: Exception) -> None: - """One traceback per interval: a stalled backend fails every in-flight task at once.""" + """One traceback per error type per interval: a stalled backend fails every in-flight task at once.""" now: Final = time.monotonic() - last_traceback_at: Final = self._last_error_traceback_at + error_type: Final = type(error) + last_traceback_at: Final = self._last_error_traceback_at.get(error_type) if last_traceback_at is not None and now - last_traceback_at < self.error_traceback_interval: - self._errors_since_traceback += 1 + self._errors_since_traceback[error_type] = self._errors_since_traceback.get(error_type, 0) + 1 return verbose_logger.exception( - "LoggingWorker error (%d more suppressed since the last traceback): %r", - self._errors_since_traceback, + "LoggingWorker error (%d more %s suppressed since the last traceback): %r", + self._errors_since_traceback.get(error_type, 0), + error_type.__name__, error, ) - self._last_error_traceback_at = now - self._errors_since_traceback = 0 + self._last_error_traceback_at[error_type] = now + self._errors_since_traceback[error_type] = 0 async def _worker_loop(self) -> None: """Main worker loop that gets tasks and schedules them to run concurrently.""" diff --git a/tests/test_litellm/caching/test_dual_cache.py b/tests/test_litellm/caching/test_dual_cache.py index 3e21362b3f0..b2462b297fc 100644 --- a/tests/test_litellm/caching/test_dual_cache.py +++ b/tests/test_litellm/caching/test_dual_cache.py @@ -653,3 +653,23 @@ def test_open_breaker_is_a_quiet_cache_miss_on_the_sync_read_path(dual_cache_wit noisy = [r for r in caplog.records if r.levelno > logging.DEBUG] assert noisy == [], f"an open breaker must be silent per call, got {[r.getMessage() for r in noisy]}" + + +def test_open_breaker_does_not_leave_sync_batch_reservations_behind(dual_cache_with_open_breaker, caplog): + """A batch read skipped by the breaker must not hold its keys for the batch expiry window. + + The sync read reserves keys before dialing Redis so concurrent callers do not all hit it. + When the breaker rejects the read, those reservations have to be released, otherwise the + first read after Redis recovers is still throttled for up to default_redis_batch_cache_expiry. + """ + caplog.set_level(logging.DEBUG, logger="LiteLLM") + dual_cache_with_open_breaker.redis_cache.redis_client.mget.side_effect = AssertionError( + "an open breaker must not touch Redis" + ) + + assert dual_cache_with_open_breaker.batch_get_cache(keys=["lit7468", "lit7460"]) == [None, None] + + assert "lit7468" not in dual_cache_with_open_breaker.last_redis_batch_access_time + assert "lit7460" not in dual_cache_with_open_breaker.last_redis_batch_access_time + noisy = [r for r in caplog.records if r.levelno > logging.DEBUG] + assert noisy == [], f"an open breaker must be silent per call, got {[r.getMessage() for r in noisy]}" diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index 1a884d44db1..3bc3aa1e047 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -516,14 +516,20 @@ async def test_circuit_breaker_opens_when_method_swallows_redis_failure(call_met await call_method(cache) -def test_circuit_breaker_open_keeps_sync_batch_get_cache_as_a_miss(sync_batch_redis_cache): - """An open breaker must preserve the sync batch read's dictionary fallback.""" +def test_sync_batch_get_cache_swallowed_failures_open_the_breaker_and_then_fast_fail(sync_batch_redis_cache): + """The sync batch read hides its Redis error behind an empty dict, but the breaker must still + count it, and once open the read has to raise the typed error like its async twin so DualCache + can release the batch reservations it took before the call. + """ + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): assert sync_batch_redis_cache.batch_get_cache(key_list=["lit6729"]) == {} - assert sync_batch_redis_cache.batch_get_cache(key_list=["lit6729"]) == {} + with pytest.raises(RedisCircuitBreakerOpenError): + sync_batch_redis_cache.batch_get_cache(key_list=["lit6729"]) + assert sync_batch_redis_cache.redis_client.mget.call_count == REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD def test_batch_get_counts_raises_where_batch_get_cache_reports_a_miss(sync_batch_redis_cache): @@ -647,6 +653,7 @@ def test_sync_batch_get_cache_survives_a_service_callback_that_raises( from concurrent.futures import ThreadPoolExecutor import litellm + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD cache, service_logger = sync_batch_cache_with_service_logger @@ -661,7 +668,8 @@ def test_sync_batch_get_cache_survives_a_service_callback_that_raises( with ThreadPoolExecutor(max_workers=1) as pool: assert pool.submit(cache.batch_get_cache, key_list=["lit6729"]).result() == {} - assert cache.batch_get_cache(key_list=["lit6729"]) == {} + with pytest.raises(RedisCircuitBreakerOpenError): + cache.batch_get_cache(key_list=["lit6729"]) def test_call_stack_info_skips_breaker_guard_frames(): @@ -801,7 +809,7 @@ async def test_concurrent_success_is_not_cancelled_by_another_calls_failure(): # starts would leave its snapshot correct and prove nothing. async def swallows_a_failure(): await asyncio.sleep(0.02) - _record_swallowed_redis_failure(breaker, RedisConnectionError("redis unreachable")) + _record_swallowed_redis_failure(RedisConnectionError("redis unreachable")) async def succeeds_while_the_other_fails(): await asyncio.sleep(0.05) @@ -1097,6 +1105,53 @@ async def test_success_admitted_before_the_breaker_opened_cannot_close_it(): await _run_under_circuit_breaker(breaker, "op", slow_success) +@pytest.mark.asyncio +async def test_swallowed_failure_admitted_before_the_breaker_opened_cannot_delay_recovery(): + """A stale in-flight call that swallows its Redis error must not restart the open timer. + + Guarded methods that return a default instead of raising still report their failure to + the breaker on exit. If that report landed against a newer breaker generation, every + slow call that was already dialing Redis when the breaker opened would push recovery + back by its own socket timeout, and knock out the HALF_OPEN probe if it landed then. + """ + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import ( + RedisCircuitBreaker, + _record_swallowed_redis_failure, + _run_under_circuit_breaker, + ) + + breaker = RedisCircuitBreaker(failure_threshold=2, recovery_timeout=0.05) + release_stale_call = asyncio.Event() + + async def stale_call_that_swallows_its_failure(): + await release_stale_call.wait() + _record_swallowed_redis_failure(RedisConnectionError("refused")) + return {} + + async def refused(): + raise RedisConnectionError("refused") + + async def recovered(): + return "ok" + + in_flight = asyncio.create_task(_run_under_circuit_breaker(breaker, "stale", stale_call_that_swallows_its_failure)) + await asyncio.sleep(0) + for _ in range(breaker.failure_threshold): + with pytest.raises(RedisConnectionError): + await _run_under_circuit_breaker(breaker, "op", refused) + assert breaker.is_open() is True + + await asyncio.sleep(0.04) + release_stale_call.set() + assert await in_flight == {} + await asyncio.sleep(0.03) + + assert await _run_under_circuit_breaker(breaker, "probe", recovered) == "ok" + assert breaker.is_open() is False + + @pytest.mark.asyncio async def test_open_breaker_raises_its_own_exception_type(): """Callers with an optional cache need to tell the expected fast-fail apart from a real error.""" diff --git a/tests/test_litellm/litellm_core_utils/test_logging_worker.py b/tests/test_litellm/litellm_core_utils/test_logging_worker.py index f93a2e0b08a..66d49b22c6b 100644 --- a/tests/test_litellm/litellm_core_utils/test_logging_worker.py +++ b/tests/test_litellm/litellm_core_utils/test_logging_worker.py @@ -568,5 +568,33 @@ class TestLoggingWorker: messages = [r.getMessage() for r in caplog.records if "LoggingWorker error" in r.getMessage()] assert len(messages) == 2, messages - assert "(0 more suppressed" in messages[0] - assert "(4 more suppressed" in messages[1] + assert "(0 more TimeoutError suppressed" in messages[0] + assert "(4 more TimeoutError suppressed" in messages[1] + + @pytest.mark.asyncio + async def test_traceback_throttle_is_per_error_type(self, caplog): + """A timeout burst from one stalled backend must not hide the first traceback of a different + failure, otherwise a misconfigured callback stays invisible for the whole interval. + """ + caplog.set_level(logging.DEBUG, logger="LiteLLM") + worker = LoggingWorker(timeout=0.05, max_queue_size=200, concurrency=100, error_traceback_interval=60.0) + worker.start() + + async def stalled_callback(): + await asyncio.sleep(10) + + async def misconfigured_callback(): + raise KeyError("missing api key") + + for _ in range(20): + worker.enqueue(stalled_callback()) + await asyncio.sleep(0.2) + for _ in range(3): + worker.enqueue(misconfigured_callback()) + await asyncio.sleep(0.2) + await worker.stop() + + messages = [r.getMessage() for r in caplog.records if "LoggingWorker error" in r.getMessage()] + assert len(messages) == 2, messages + assert "TimeoutError" in messages[0] + assert "KeyError" in messages[1] and "missing api key" in messages[1] From fbd923190e04583e56184c6ff38e1b2e41ad6c36 Mon Sep 17 00:00:00 2001 From: mateo Date: Thu, 10 Sep 2026 22:53:46 +0000 Subject: [PATCH 040/157] fix(caching): let the outer breaker guard judge failures a nested guarded call swallowed batch_cache_write flushes through async_set_cache_pipeline, both guarded. The inner guard consumed the swallowed pipeline failure and the outer then recorded a success, so a dead Redis never tripped the breaker on the batch write path. Only the outermost admission now reports, and the sync batch read runs under the same guard. Also covers the sync add_cache and embedding pipeline wrappers, the increment pipeline, sadd, and the stale raise path in tests. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/redis_cache.py | 36 ++++++-- tests/test_litellm/caching/test_caching.py | 33 +++++-- tests/test_litellm/caching/test_dual_cache.py | 5 ++ .../test_litellm/caching/test_redis_cache.py | 90 +++++++++++++++++-- 4 files changed, 141 insertions(+), 23 deletions(-) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 3ecf805ed0e..2c1ca75212c 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -288,6 +288,7 @@ _RedisCallResult = TypeVar("_RedisCallResult") _swallowed_redis_failures: Final[ContextVar[tuple[bool, ...]]] = ContextVar( "litellm_swallowed_redis_failures", default=() ) +_breaker_depth: Final[ContextVar[int]] = ContextVar("litellm_redis_breaker_depth", default=0) def _opaque_kwarg_key(value: object) -> str: @@ -425,13 +426,20 @@ def _record_swallowed_redis_failure(exc: BaseException) -> None: class _BreakerAdmission: swallowed_before: int generation: int + nested: bool def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> _BreakerAdmission: """Reject the call if the breaker is open, else snapshot what its outcome will be judged against.""" if breaker.is_open(): raise RedisCircuitBreakerOpenError(f"Redis circuit breaker is open, skipping {name}") - return _BreakerAdmission(swallowed_before=len(_swallowed_redis_failures.get()), generation=breaker.generation) + depth: Final = _breaker_depth.get() + _breaker_depth.set(depth + 1) + return _BreakerAdmission( + swallowed_before=len(_swallowed_redis_failures.get()), + generation=breaker.generation, + nested=depth > 0, + ) def _take_swallowed_failures(admission: _BreakerAdmission) -> tuple[bool, ...]: @@ -448,7 +456,14 @@ def _exit_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmis method that returned is not on its own proof of a healthy Redis. A call admitted before the breaker opened reports nothing: its failures would refresh the open timer or knock out the recovery probe, and its success would close it early. + + A guarded method calling another guarded method is one Redis interaction, so only + the outermost admission reports. The inner one leaves its swallowed failures in the + context for the outer to judge, otherwise the outer would read a clean context and + reset the streak the inner just fed. """ + if admission.nested: + return swallowed: Final = _take_swallowed_failures(admission) if admission.generation != breaker.generation: return @@ -459,7 +474,13 @@ def _exit_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmis breaker.record_failure(is_timeout=is_timeout) +def _leave_circuit_breaker() -> None: + _breaker_depth.set(_breaker_depth.get() - 1) + + def _fail_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission, exc: BaseException) -> None: + if admission.nested: + return swallowed: Final = _take_swallowed_failures(admission) if admission.generation != breaker.generation: return @@ -485,6 +506,8 @@ async def _run_under_circuit_breaker( except Exception as e: _fail_circuit_breaker(breaker, admission, e) raise + finally: + _leave_circuit_breaker() _exit_circuit_breaker(breaker, admission) return result @@ -501,6 +524,8 @@ def _run_under_circuit_breaker_sync( except Exception as e: _fail_circuit_breaker(breaker, admission, e) raise + finally: + _leave_circuit_breaker() _exit_circuit_breaker(breaker, admission) return result @@ -1441,12 +1466,12 @@ class RedisCache(BaseCache): key_value_dict = {} _key_list: Final = [key for key in key_list if key is not None] start_time: Final = time.time() - admission: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") try: _keys: Final = [self.check_and_fix_namespace(key=cache_key or "") for cache_key in _key_list] - results: Final = self._run_redis_mget_operation(keys=_keys) - _exit_circuit_breaker(self._circuit_breaker, admission) + results: Final = _run_under_circuit_breaker_sync( + self._circuit_breaker, "batch_get_cache", lambda: self._run_redis_mget_operation(keys=_keys) + ) end_time: Final = time.time() _duration: Final = end_time - start_time self.service_logger_obj.service_success_hook( @@ -1470,6 +1495,8 @@ class RedisCache(BaseCache): decoded_results[k] = v return decoded_results + except RedisCircuitBreakerOpenError: + raise except Exception as e: failed_at: Final = time.time() self.service_logger_obj.service_failure_hook( @@ -1482,7 +1509,6 @@ class RedisCache(BaseCache): parent_otel_span=parent_otel_span, ) verbose_logger.error("Error occurred in batch get cache - %s", e) - _fail_circuit_breaker(self._circuit_breaker, admission, e) return key_value_dict @_redis_circuit_breaker_guard diff --git a/tests/test_litellm/caching/test_caching.py b/tests/test_litellm/caching/test_caching.py index 5d326a9ece0..d135d2892c5 100644 --- a/tests/test_litellm/caching/test_caching.py +++ b/tests/test_litellm/caching/test_caching.py @@ -255,26 +255,41 @@ def test_exact_cache_key_includes_anthropic_messages_params(anthropic_param): @pytest.mark.asyncio -async def test_async_add_cache_treats_an_open_breaker_as_a_quiet_skip(caplog): - """The top-level write wrapper logs a full ERROR traceback for any failure. An open Redis - breaker fails every write instantly, so under load that wrapper alone was hundreds of +@pytest.mark.parametrize("write", ["add_cache", "async_add_cache", "async_add_cache_pipeline"]) +async def test_add_cache_wrappers_treat_an_open_breaker_as_a_quiet_skip(caplog, write: str): + """The top-level write wrappers log a full ERROR traceback for any failure. An open Redis + breaker fails every write instantly, so under load those wrappers alone were hundreds of stack formats per second per replica. Unexpected failures must still get the traceback. """ - from unittest.mock import AsyncMock + from unittest.mock import AsyncMock, MagicMock from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + from litellm.types.utils import Embedding, EmbeddingResponse caplog.set_level(logging.DEBUG, logger="LiteLLM") cache = Cache(type=LiteLLMCacheType.LOCAL) - cache.cache.async_set_cache = AsyncMock(side_effect=RedisCircuitBreakerOpenError("open")) + embedding = EmbeddingResponse(model="text-embedding-3-small", data=[Embedding(embedding=[0.1], index=0, object="embedding")]) - await cache.async_add_cache("result", model="gpt-5.4-nano", messages=[{"role": "user", "content": "hi"}]) + async def run_write() -> None: + if write == "add_cache": + cache.add_cache("result", model="gpt-5.4-nano", messages=[{"role": "user", "content": "hi"}]) + elif write == "async_add_cache": + await cache.async_add_cache("result", model="gpt-5.4-nano", messages=[{"role": "user", "content": "hi"}]) + else: + await cache.async_add_cache_pipeline(embedding, model="text-embedding-3-small", input="hi") + + def install_backend(exc: Exception) -> None: + cache.cache.set_cache = MagicMock(side_effect=exc) + cache.cache.async_set_cache = AsyncMock(side_effect=exc) + cache.cache.async_set_cache_pipeline = AsyncMock(side_effect=exc) + + install_backend(RedisCircuitBreakerOpenError("open")) + await run_write() assert [r.levelno for r in caplog.records if "add_cache" in r.getMessage()] == [logging.DEBUG] - cache.cache.async_set_cache.assert_awaited_once() - cache.cache.async_set_cache = AsyncMock(side_effect=OSError("disk full")) - await cache.async_add_cache("result", model="gpt-5.4-nano", messages=[{"role": "user", "content": "hi"}]) + install_backend(OSError("disk full")) + await run_write() errors = [r for r in caplog.records if r.levelno == logging.ERROR] assert len(errors) == 1 and errors[0].exc_info is not None diff --git a/tests/test_litellm/caching/test_dual_cache.py b/tests/test_litellm/caching/test_dual_cache.py index b2462b297fc..4177c215d0b 100644 --- a/tests/test_litellm/caching/test_dual_cache.py +++ b/tests/test_litellm/caching/test_dual_cache.py @@ -621,6 +621,11 @@ def dual_cache_with_open_breaker(): pytest.param(lambda c: c.async_set_cache("lit7468", "v"), lambda n: None, id="async_set_cache"), pytest.param(lambda c: c.async_set_cache_pipeline([("lit7468", "v")]), lambda n: None, id="async_set_cache_pipeline"), pytest.param(lambda c: c.async_increment_cache("lit7468", 1.0, ttl=60), float, id="async_increment_cache"), + pytest.param( + lambda c: c.async_increment_cache_pipeline([{"key": "lit7468", "increment_value": 1.0, "ttl": 60}]), + lambda n: [float(n)], + id="async_increment_cache_pipeline", + ), ], ) async def test_open_breaker_is_a_quiet_cache_miss(dual_cache_with_open_breaker, call, expected, caplog): diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index 3bc3aa1e047..8c994a4742e 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -494,6 +494,7 @@ def _closed_port() -> int: pytest.param(lambda c: c.async_batch_get_cache(["lit4930"]), id="async_batch_get_cache"), pytest.param(lambda c: c.async_set_cache("lit4930", "v"), id="async_set_cache"), pytest.param(lambda c: c.async_get_ttl("lit4930"), id="async_get_ttl"), + pytest.param(lambda c: c.async_set_cache_sadd("lit4930", ["v"], ttl=None), id="async_set_cache_sadd"), ], ) async def test_circuit_breaker_opens_when_method_swallows_redis_failure(call_method): @@ -505,6 +506,7 @@ async def test_circuit_breaker_opens_when_method_swallows_redis_failure(call_met breaker could never open. An unreachable Redis then stayed in the pool and every request kept paying the full socket timeout on it. """ + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD cache = await asyncio.to_thread(RedisCache, host="127.0.0.1", port=_closed_port(), socket_timeout=0.5) @@ -512,10 +514,73 @@ async def test_circuit_breaker_opens_when_method_swallows_redis_failure(call_met for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): await call_method(cache) - with pytest.raises(Exception, match="circuit breaker is open"): + with pytest.raises(RedisCircuitBreakerOpenError): await call_method(cache) +@pytest.mark.asyncio +async def test_nested_guarded_flush_failures_still_open_the_breaker(): + """A guarded method that delegates to another guarded method is one Redis call, not two. + + batch_cache_write is guarded and flushes through the guarded async_set_cache_pipeline, + which swallows the pipeline error. If the inner guard consumes that failure, the outer + guard sees a clean run and records a success, so the streak resets on every write and + a dead Redis keeps receiving flushes instead of tripping the breaker. + """ + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + + cache = await asyncio.to_thread( + RedisCache, host="127.0.0.1", port=_closed_port(), socket_timeout=0.5, redis_flush_size=1 + ) + + for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): + await cache.batch_cache_write("lit7468", "v") + + with pytest.raises(RedisCircuitBreakerOpenError): + await cache.batch_cache_write("lit7468", "v") + assert cache._circuit_breaker._failure_count == REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + + +@pytest.mark.asyncio +async def test_nested_guard_that_raises_counts_one_failure_for_the_outer_call(): + """An inner guarded call that raises through the outer one is a single Redis failure, and a + swallowed inner failure followed by an outer raise is two, so the count follows what Redis + actually refused rather than how many guard frames the error crossed. + """ + from redis.exceptions import ConnectionError as RedisConnectionError + + from litellm.caching.redis_cache import ( + RedisCircuitBreaker, + _record_swallowed_redis_failure, + _run_under_circuit_breaker, + ) + + breaker = RedisCircuitBreaker(failure_threshold=10, recovery_timeout=60) + + async def refused(): + raise RedisConnectionError("refused") + + async def outer_delegating_to_inner(): + return await _run_under_circuit_breaker(breaker, "inner", refused) + + async def outer_swallowing_then_raising(): + async def inner_swallowing(): + _record_swallowed_redis_failure(RedisConnectionError("refused")) + return {} + + await _run_under_circuit_breaker(breaker, "inner", inner_swallowing) + raise RedisConnectionError("refused") + + with pytest.raises(RedisConnectionError): + await _run_under_circuit_breaker(breaker, "outer", outer_delegating_to_inner) + assert breaker._failure_count == 1 + + with pytest.raises(RedisConnectionError): + await _run_under_circuit_breaker(breaker, "outer", outer_swallowing_then_raising) + assert breaker._failure_count == 3 + + def test_sync_batch_get_cache_swallowed_failures_open_the_breaker_and_then_fast_fail(sync_batch_redis_cache): """The sync batch read hides its Redis error behind an empty dict, but the breaker must still count it, and once open the read has to raise the typed error like its async twin so DualCache @@ -1106,12 +1171,13 @@ async def test_success_admitted_before_the_breaker_opened_cannot_close_it(): @pytest.mark.asyncio -async def test_swallowed_failure_admitted_before_the_breaker_opened_cannot_delay_recovery(): - """A stale in-flight call that swallows its Redis error must not restart the open timer. +@pytest.mark.parametrize("stale_call_raises", [False, True], ids=["swallows", "raises"]) +async def test_failure_admitted_before_the_breaker_opened_cannot_delay_recovery(stale_call_raises: bool): + """A stale in-flight call that fails after the breaker opened must not restart the open timer. - Guarded methods that return a default instead of raising still report their failure to - the breaker on exit. If that report landed against a newer breaker generation, every - slow call that was already dialing Redis when the breaker opened would push recovery + Whether the method swallows its Redis error and returns a default or lets it propagate, + the failure is reported on exit. If that report landed against a newer breaker generation, + every slow call that was already dialing Redis when the breaker opened would push recovery back by its own socket timeout, and knock out the HALF_OPEN probe if it landed then. """ from redis.exceptions import ConnectionError as RedisConnectionError @@ -1125,8 +1191,10 @@ async def test_swallowed_failure_admitted_before_the_breaker_opened_cannot_delay breaker = RedisCircuitBreaker(failure_threshold=2, recovery_timeout=0.05) release_stale_call = asyncio.Event() - async def stale_call_that_swallows_its_failure(): + async def stale_call(): await release_stale_call.wait() + if stale_call_raises: + raise RedisConnectionError("refused") _record_swallowed_redis_failure(RedisConnectionError("refused")) return {} @@ -1136,7 +1204,7 @@ async def test_swallowed_failure_admitted_before_the_breaker_opened_cannot_delay async def recovered(): return "ok" - in_flight = asyncio.create_task(_run_under_circuit_breaker(breaker, "stale", stale_call_that_swallows_its_failure)) + in_flight = asyncio.create_task(_run_under_circuit_breaker(breaker, "stale", stale_call)) await asyncio.sleep(0) for _ in range(breaker.failure_threshold): with pytest.raises(RedisConnectionError): @@ -1145,7 +1213,11 @@ async def test_swallowed_failure_admitted_before_the_breaker_opened_cannot_delay await asyncio.sleep(0.04) release_stale_call.set() - assert await in_flight == {} + if stale_call_raises: + with pytest.raises(RedisConnectionError): + await in_flight + else: + assert await in_flight == {} await asyncio.sleep(0.03) assert await _run_under_circuit_breaker(breaker, "probe", recovered) == "ok" From e9c388d79917cd4018a568181f31969ab223cf4e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 17:49:20 -0700 Subject: [PATCH 041/157] fix(caching): keep an open Redis circuit breaker quiet on the sync read and spend counter paths The sync get path was unguarded, logged with a stray format argument, and never fed the breaker. The sync batch read swallowed the breaker's refusal as an ERROR plus a service failure event per call, so DualCache dropped its in-memory hits and left batch reservations behind. record_success closed an OPEN breaker on stale in-flight successes, skipping the recovery timeout and the half-open probe. The spend counter pipeline re-raised the refusal into the cost callback, which logged an ERROR and fired the failed-tracking alert per request. --- litellm/caching/dual_cache.py | 12 ++- litellm/caching/redis_cache.py | 9 +- litellm/proxy/management_endpoints/ui_sso.py | 18 ++-- litellm/proxy/proxy_server.py | 5 +- tests/test_litellm/caching/test_dual_cache.py | 27 ++++++ .../test_litellm/caching/test_redis_cache.py | 83 +++++++++++++++++-- .../proxy/management_endpoints/test_ui_sso.py | 22 +++++ .../proxy/proxy_server/test_spend_counters.py | 48 +++++++++++ 8 files changed, 205 insertions(+), 19 deletions(-) diff --git a/litellm/caching/dual_cache.py b/litellm/caching/dual_cache.py index baabfad6852..be761e1258b 100644 --- a/litellm/caching/dual_cache.py +++ b/litellm/caching/dual_cache.py @@ -23,7 +23,7 @@ from litellm.constants import DEFAULT_MAX_REDIS_BATCH_CACHE_SIZE from .base_cache import BaseCache from .in_memory_cache import InMemoryCache -from .redis_cache import RedisCache, log_redis_failure +from .redis_cache import RedisCache, RedisCircuitBreakerOpenError, log_redis_failure if TYPE_CHECKING: from opentelemetry.trace import Span as _Span @@ -206,9 +206,12 @@ class DualCache(BaseCache): redis_result: Final = self.redis_cache.batch_get_cache( key_list=sublist_keys, parent_otel_span=parent_otel_span ) - except Exception: + except Exception as e: # Do not throttle subsequent callers if the Redis read fails. self._rollback_redis_batch_key_reservations(previous_access_times) + if isinstance(e, RedisCircuitBreakerOpenError): + verbose_logger.debug("LiteLLM Cache: batch_get_cache served from memory only: %s", e) + return result raise if self.in_memory_cache is not None: @@ -325,9 +328,12 @@ class DualCache(BaseCache): redis_result: Final = await self.redis_cache.async_batch_get_cache( sublist_keys, parent_otel_span=parent_otel_span ) - except Exception: + except Exception as e: # Do not throttle subsequent callers if the Redis read fails. self._rollback_redis_batch_key_reservations(previous_access_times) + if isinstance(e, RedisCircuitBreakerOpenError): + verbose_logger.debug("LiteLLM Cache: async_batch_get_cache served from memory only: %s", e) + return result raise # Short-circuit if redis_result is None or contains only None values diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 6b93529e456..5b1a9739eb8 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -250,7 +250,7 @@ class RedisCircuitBreaker: self._set_state(self.OPEN) def record_success(self) -> None: - if not self.enabled: + if not self.enabled or self._state == self.OPEN: return if self._state == self.HALF_OPEN: verbose_logger.info("Redis circuit breaker CLOSED — Redis recovered") @@ -1337,6 +1337,7 @@ class RedisCache(BaseCache): except Exception: return ast.literal_eval(decoded) + @_redis_circuit_breaker_guard_sync def get_cache(self, key, parent_otel_span: Span | None = None, **kwargs): try: key = self.check_and_fix_namespace(key=key) @@ -1356,8 +1357,8 @@ class RedisCache(BaseCache): print_verbose(f"Got Redis Cache: key: {key}, cached_response {cached_response}") return self._get_cache_logic(cached_response=cached_response) except Exception as e: - # NON blocking - notify users Redis is throwing an exception - verbose_logger.error("litellm.caching.caching: get() - Got exception from REDIS: ", e) + verbose_logger.error("litellm.caching.caching: get() - Got exception from REDIS: %s", e) + _record_swallowed_redis_failure(self._circuit_breaker, e) def _run_redis_mget_operation(self, keys: list[str]) -> Sequence[bytes | str | None]: """ @@ -1394,9 +1395,9 @@ class RedisCache(BaseCache): key_value_dict = {} _key_list: Final = [key for key in key_list if key is not None] start_time: Final = time.time() + swallowed_before: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") try: - swallowed_before: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") _keys: Final = [self.check_and_fix_namespace(key=cache_key or "") for cache_key in _key_list] results: Final = self._run_redis_mget_operation(keys=_keys) _exit_circuit_breaker(self._circuit_breaker, swallowed_before) diff --git a/litellm/proxy/management_endpoints/ui_sso.py b/litellm/proxy/management_endpoints/ui_sso.py index c60888e298f..1ba90725eff 100644 --- a/litellm/proxy/management_endpoints/ui_sso.py +++ b/litellm/proxy/management_endpoints/ui_sso.py @@ -47,6 +47,7 @@ import litellm from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid from litellm.caching.dual_cache import DualCache +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.constants import ( CLI_SSO_CLAIM_MAP, CLI_SSO_CLAIM_MAX_SCALAR_LENGTH, @@ -336,6 +337,16 @@ def _check_cli_sso_start_rate_limit( ) +def _read_cli_sso_flow(cache: DualCache, cache_key: str) -> object: + redis_cache: Final = cache.redis_cache + if redis_cache is None: + return cache.get_cache(key=cache_key) + try: + return redis_cache.get_cache(key=cache_key) + except RedisCircuitBreakerOpenError: + return None + + def _get_cli_sso_flow_or_raise(login_id: str | None, cache: DualCache) -> dict: if isinstance(login_id, str) and login_id.startswith("sk-"): raise HTTPException( @@ -348,12 +359,7 @@ def _get_cli_sso_flow_or_raise(login_id: str | None, cache: DualCache) -> dict: if not _is_valid_cli_sso_login_id(login_id): raise HTTPException(status_code=400, detail="Invalid CLI login session id") - cache_key: Final = _get_cli_sso_flow_cache_key(cast(str, login_id)) - redis_cache: Final = cache.redis_cache - if redis_cache is not None: - flow = redis_cache.get_cache(key=cache_key) - else: - flow = cache.get_cache(key=cache_key) + flow = _read_cli_sso_flow(cache, _get_cli_sso_flow_cache_key(cast(str, login_id))) if isinstance(flow, str): try: flow = _as_object(json.loads(flow)) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 0f24acb8bb4..94e74b20297 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -251,6 +251,7 @@ import litellm._redis from litellm import Router from litellm._logging import _redact_string, verbose_proxy_logger, verbose_router_logger from litellm.caching.caching import DualCache, RedisCache +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.caching.redis_cluster_cache import RedisClusterCache from litellm.constants import ( _REALTIME_BODY_CACHE_SIZE, @@ -3412,8 +3413,10 @@ async def _apply_spend_counter_increments(pending: Sequence[_PendingSpendIncreme ] try: results: Final = await redis_cache.async_increment_pipeline(increment_list=increment_list) - except Exception: + except Exception as e: await asyncio.gather(*(_invalidate_spend_counter(counter_key=item.counter_key) for item in pending)) + if isinstance(e, RedisCircuitBreakerOpenError): + return raise for item, current_value in zip(pending, results or ()): spend_counter_cache.in_memory_cache.set_cache(key=item.counter_key, value=current_value) diff --git a/tests/test_litellm/caching/test_dual_cache.py b/tests/test_litellm/caching/test_dual_cache.py index eae8bbfdaff..4c9068722b8 100644 --- a/tests/test_litellm/caching/test_dual_cache.py +++ b/tests/test_litellm/caching/test_dual_cache.py @@ -677,3 +677,30 @@ async def test_a_real_redis_failure_still_logs_an_error(caplog): errors = [record for record in caplog.records if record.levelno == logging.ERROR] assert [record.getMessage() for record in errors] == ["LiteLLM Cache: exception in async_get_cache: redis is down"] assert errors[0].exc_info is not None + + +def _dual_cache_with_open_breaker_and_a_memory_hit() -> DualCache: + in_memory = InMemoryCache() + in_memory.set_cache("k1", "v1") + return DualCache(in_memory_cache=in_memory, redis_cache=_OpenBreakerRedis(), default_redis_batch_cache_expiry=10) # pyright: ignore[reportArgumentType] # duck-typed Redis double + + +def test_open_breaker_keeps_sync_batch_read_memory_hits_and_releases_reservations(): + """A refused Redis batch read must still answer with the in-memory hits and hold no reservation. + + The refusal was logged and turned into a bare None, so a caller lost its in-memory hits + for as long as the breaker stayed open, and the reserved keys stayed throttled until + the batch expiry passed even though nothing was ever read for them. + """ + cache = _dual_cache_with_open_breaker_and_a_memory_hit() + + assert list(cache.batch_get_cache(["k1", "k2"])) == ["v1", None] + assert "k2" not in cache.last_redis_batch_access_time + + +@pytest.mark.asyncio +async def test_open_breaker_keeps_async_batch_read_memory_hits_and_releases_reservations(): + cache = _dual_cache_with_open_breaker_and_a_memory_hit() + + assert list(await cache.async_batch_get_cache(["k1", "k2"])) == ["v1", None] + assert "k2" not in cache.last_redis_batch_access_time diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index bcaa58c9c40..c6f0caa1dca 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -1,11 +1,12 @@ import asyncio +import time from collections.abc import Iterator from unittest.mock import AsyncMock, MagicMock, patch import pytest from litellm._service_logger import ServiceLogging -from litellm.caching.redis_cache import RedisCache +from litellm.caching.redis_cache import RedisCache, RedisCircuitBreakerOpenError @pytest.fixture @@ -515,14 +516,46 @@ async def test_circuit_breaker_opens_when_method_swallows_redis_failure(call_met await call_method(cache) -def test_circuit_breaker_open_keeps_sync_batch_get_cache_as_a_miss(sync_batch_redis_cache): - """An open breaker must preserve the sync batch read's dictionary fallback.""" +def test_circuit_breaker_open_makes_sync_batch_get_cache_fast_fail(sync_batch_redis_cache, caplog): + """Once the breaker is open the sync batch read refuses with the typed error instead of a miss. + + Swallowing the refusal into `{}` made every sync batch read on an open breaker emit an ERROR + log and a service failure event per call, and the DualCache caller could not tell the + refusal from a dead Redis, so it dropped its in-memory hits too. + """ from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): assert sync_batch_redis_cache.batch_get_cache(key_list=["lit6729"]) == {} - assert sync_batch_redis_cache.batch_get_cache(key_list=["lit6729"]) == {} + caplog.clear() + with caplog.at_level("INFO"): + with pytest.raises(RedisCircuitBreakerOpenError): + sync_batch_redis_cache.batch_get_cache(key_list=["lit6729"]) + sync_batch_redis_cache.redis_client.mget.assert_called() + assert caplog.records == [] + + +def test_sync_get_cache_failure_feeds_the_breaker_and_logs_a_well_formed_record(sync_batch_redis_cache, caplog): + """The sync get path swallowed its Redis error without recording it, and its log call was malformed. + + `verbose_logger.error("...: ", e)` passes the exception as a format argument to a message + with no placeholder, so the record carried no error text. Nothing fed the breaker either, + so a dead Redis read through this path never opened it. + """ + from litellm.constants import REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + + sync_batch_redis_cache.redis_client.get.side_effect = OSError("redis unavailable") + + with caplog.at_level("ERROR"): + for _ in range(REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD): + assert sync_batch_redis_cache.get_cache("lit7468") is None + + assert all("redis unavailable" in record.getMessage() for record in caplog.records) + assert len(caplog.records) == REDIS_CIRCUIT_BREAKER_FAILURE_THRESHOLD + assert sync_batch_redis_cache._circuit_breaker.is_open() is True + with pytest.raises(RedisCircuitBreakerOpenError): + sync_batch_redis_cache.get_cache("lit7468") def test_batch_get_counts_raises_where_batch_get_cache_reports_a_miss(sync_batch_redis_cache): @@ -661,7 +694,8 @@ def test_sync_batch_get_cache_survives_a_service_callback_that_raises( with ThreadPoolExecutor(max_workers=1) as pool: assert pool.submit(cache.batch_get_cache, key_list=["lit6729"]).result() == {} - assert cache.batch_get_cache(key_list=["lit6729"]) == {} + with pytest.raises(RedisCircuitBreakerOpenError): + cache.batch_get_cache(key_list=["lit6729"]) def test_call_stack_info_skips_breaker_guard_frames(): @@ -1010,6 +1044,8 @@ async def test_breaker_metrics_track_state_and_failure_class(): assert sample("litellm_redis_circuit_breaker_state", {"state": "open"}) == open_gauge_before + 1 assert sample("litellm_redis_circuit_breaker_state", {"state": "closed"}) == closed_gauge_before + breaker._opened_at = time.time() - 9999 + assert breaker.is_open() is False breaker.record_success() assert sample("litellm_redis_circuit_breaker_state", {"state": "open"}) == open_gauge_before assert sample("litellm_redis_circuit_breaker_state", {"state": "closed"}) == closed_gauge_before + 1 @@ -1030,3 +1066,40 @@ def test_sync_guard_counts_a_timeout_as_a_timeout(): _run_under_circuit_breaker_sync(breaker, "op", timing_out_call) assert breaker.is_open() is False + + +def test_success_admitted_before_the_breaker_opened_cannot_close_it(): + """A stale in-flight success must not close a breaker that opened while it ran. + + Calls admitted while the breaker was still closed finish after later failures opened it. + Recording their success unconditionally closed the breaker again, skipping the recovery + timeout and the single half-open probe, so the breaker flapped between open and closed + on every straggler while Redis was still down. + """ + from litellm.caching.redis_cache import RedisCircuitBreaker + + breaker = RedisCircuitBreaker(failure_threshold=3, recovery_timeout=60) + for _ in range(3): + breaker.record_failure() + assert breaker._state == breaker.OPEN + + breaker.record_success() + + assert breaker._state == breaker.OPEN + assert breaker.is_open() is True + + +def test_recovery_probe_still_closes_the_breaker(): + from litellm.caching.redis_cache import RedisCircuitBreaker + + breaker = RedisCircuitBreaker(failure_threshold=3, recovery_timeout=60) + for _ in range(3): + breaker.record_failure() + breaker._opened_at = time.time() - 9999 + assert breaker.is_open() is False + assert breaker._state == breaker.HALF_OPEN + + breaker.record_success() + + assert breaker._state == breaker.CLOSED + assert breaker.is_open() is False diff --git a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py index 8d8bc15f9be..2050e65d2a1 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py +++ b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py @@ -2616,6 +2616,28 @@ class TestCLIKeyRegenerationFlow: ) cache.set_cache.assert_not_called() + def test_cli_sso_flow_lookup_treats_an_open_redis_breaker_as_a_miss(self): + """A Redis read refused by the open circuit breaker is a missing session, not a server error. + + The direct Redis read is what keeps the flow authoritative across workers, so the + refusal must not fall back to a possibly stale in-memory copy either. + """ + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + from litellm.proxy.management_endpoints.ui_sso import _get_cli_sso_flow_or_raise + + redis_cache = MagicMock() + redis_cache.get_cache.side_effect = RedisCircuitBreakerOpenError("Redis circuit breaker is open") + cache = MagicMock() + cache.redis_cache = redis_cache + cache.get_cache.return_value = {"poll_secret_hash": "stale", "sso_complete": False} + + with pytest.raises(HTTPException) as exc_info: + _get_cli_sso_flow_or_raise(login_id="cli-breaker_open_1234567890", cache=cache) + + assert exc_info.value.status_code == 400 + assert "not found or expired" in exc_info.value.detail + cache.get_cache.assert_not_called() + def test_cli_sso_flow_with_enum_survives_redis_round_trip(self): """ RedisCache stores values via str(value) and reads them back through diff --git a/tests/test_litellm/proxy/proxy_server/test_spend_counters.py b/tests/test_litellm/proxy/proxy_server/test_spend_counters.py index c343652efd9..2f47736a398 100644 --- a/tests/test_litellm/proxy/proxy_server/test_spend_counters.py +++ b/tests/test_litellm/proxy/proxy_server/test_spend_counters.py @@ -1146,6 +1146,54 @@ async def test_prepare_window_spend_counter_increment_missing_window_start_inval assert fake_cache.redis_cache.async_increment.called is False +# --------------------------------------------------------------------------- +# _apply_spend_counter_increments +# --------------------------------------------------------------------------- + + +def _two_pending_increments() -> tuple[ps._PendingSpendIncrement, ...]: + return ( + ps._PendingSpendIncrement(counter_key="spend:key:k", increment=1.5), + ps._PendingSpendIncrement(counter_key="spend:team:t", increment=1.5), + ) + + +@pytest.mark.asyncio +async def test_apply_spend_counter_increments_open_breaker_invalidates_and_returns(monkeypatch): + """An open Redis circuit breaker is a known, already-logged state, not a per-request tracking failure. + + Re-raising the refusal sent every request through the cost callback's error path, which + logged an ERROR and fired the failed-tracking alert once per request for the whole outage. + """ + from litellm.caching.redis_cache import RedisCircuitBreakerOpenError + + fake_cache = _make_spend_counter_cache() + fake_cache.redis_cache.async_increment_pipeline = AsyncMock( + side_effect=RedisCircuitBreakerOpenError("Redis circuit breaker is open") + ) + monkeypatch.setattr(ps, "spend_counter_cache", fake_cache) + + await ps._apply_spend_counter_increments(_two_pending_increments()) + + deleted_keys = sorted(call.kwargs["key"] for call in fake_cache.in_memory_cache.delete_cache.call_args_list) + assert deleted_keys == ["spend:key:k", "spend:team:t"] + fake_cache.in_memory_cache.set_cache.assert_not_called() + + +@pytest.mark.asyncio +async def test_apply_spend_counter_increments_other_redis_error_invalidates_and_raises(monkeypatch): + fake_cache = _make_spend_counter_cache() + fake_cache.redis_cache.async_increment_pipeline = AsyncMock(side_effect=ConnectionError("redis down")) + monkeypatch.setattr(ps, "spend_counter_cache", fake_cache) + + with pytest.raises(ConnectionError, match="redis down"): + await ps._apply_spend_counter_increments(_two_pending_increments()) + + deleted_keys = sorted(call.kwargs["key"] for call in fake_cache.in_memory_cache.delete_cache.call_args_list) + assert deleted_keys == ["spend:key:k", "spend:team:t"] + fake_cache.in_memory_cache.set_cache.assert_not_called() + + # --------------------------------------------------------------------------- # _ensure_spend_counter_initialized # --------------------------------------------------------------------------- From 857b9ad7d344b2cb6d76f0f0b4072b464ad3b297 Mon Sep 17 00:00:00 2001 From: ryan Date: Fri, 11 Sep 2026 01:08:20 +0000 Subject: [PATCH 042/157] fix(ui): jump straight to the last Request Logs page instead of advancing one page Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend_management_endpoints.py | 11 +++-- .../test_spend_management_endpoints.py | 42 +++++++++++++++++++ .../view_logs/RequestLogsPanel.test.tsx | 38 +++++++++++++++++ .../components/view_logs/RequestLogsPanel.tsx | 7 ++-- 4 files changed, 91 insertions(+), 7 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index da79328fa59..4a5995167f7 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -2944,7 +2944,10 @@ async def _ui_session_grouped_spend_logs( next ``page_size`` sessions ordered by ``(MAX(startTime), session_key, api_key)``, resumed from the ``session_cursor`` keyset ``'||'`` instead of an OFFSET, so - page depth does not degrade the query plan. Each session is represented + page depth does not degrade the query plan. A request for ``page > 1`` + without a cursor (the UI jumping straight to the last page, or back to a + page it never walked through) falls back to ``OFFSET (page - 1) * + page_size``, bounded by the capped total. Each session is represented by its newest non-MCP row, enriched by ``_build_ui_spend_logs_response`` exactly like the flat listing, and the response carries ``next_session_cursor`` / ``has_more`` while ``total`` counts sessions @@ -2963,6 +2966,8 @@ async def _ui_session_grouped_spend_logs( ) cursor_params: Final[tuple[object, ...]] = cursor if cursor else () limit_index: Final = next_param_index + len(cursor_params) + offset_params: Final[tuple[int, ...]] = ((page - 1) * page_size,) if cursor is None and page > 1 else () + offset_clause: Final = f"OFFSET ${limit_index + 1}" if offset_params else "" page_query: Final = f""" SELECT {_SESSION_KEY_EXPR} AS session_key, @@ -2973,10 +2978,10 @@ async def _ui_session_grouped_spend_logs( GROUP BY {_SESSION_GROUP_KEY_SQL} {having_clause} ORDER BY MAX("startTime") {direction}, {_SESSION_KEY_EXPR} {direction}, api_key {direction} - LIMIT ${limit_index} + LIMIT ${limit_index} {offset_clause} """ page_rows: Final[Sequence[_SessionPageRow]] = await _query_raw( - prisma_client, page_query, *sql_params, *cursor_params, page_size + 1 + prisma_client, page_query, *sql_params, *cursor_params, page_size + 1, *offset_params ) has_more: Final = len(page_rows) > page_size diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 671a8ae63fc..5052cf3c085 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -6742,6 +6742,48 @@ async def test_ui_view_spend_logs_group_by_session_cursor_page(client, monkeypat app.dependency_overrides.pop(ps.user_api_key_auth, None) +@pytest.mark.asyncio +async def test_ui_view_spend_logs_group_by_session_jumps_to_page_without_cursor(client, monkeypatch): + """page > 1 with no session_cursor (the UI's last-page jump) skips (page - 1) * page_size sessions by OFFSET.""" + page_rows = [_session_page_row("sess-3", "2026-08-29 06:00:00")] + reps = [_session_representative_row("req-3", "sess-3")] + mock_prisma = _session_grouped_mock_prisma(page_rows, 60, reps) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma) + monkeypatch.setattr( + "litellm.proxy.spend_tracking.spend_management_endpoints._is_admin_view_safe", + lambda user_api_key_dict: True, + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_user" + ) + try: + start_date, end_date = _default_date_range() + response = client.get( + "/spend/logs/ui", + params={ + "start_date": start_date, + "end_date": end_date, + "group_by_session": "true", + "page": 3, + "page_size": 25, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + assert response.status_code == 200, response.text + data = response.json() + assert data["page"] == 3 + assert data["has_more"] is False + assert [row["request_id"] for row in data["data"]] == ["req-3"] + + page_query_call = mock_prisma.db.query_raw.await_args_list[0] + page_query_sql = page_query_call.args[0] + assert "HAVING" not in page_query_sql + assert "OFFSET" in page_query_sql + assert page_query_call.args[-2:] == (26, 50), "LIMIT page_size + 1 then OFFSET (page - 1) * page_size" + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + @pytest.mark.asyncio async def test_ui_view_spend_logs_group_by_session_offset_for_non_starttime_sort( client, monkeypatch diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.test.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.test.tsx index 295446186b1..e760d0b8dc3 100644 --- a/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.test.tsx @@ -258,6 +258,44 @@ describe("RequestLogsPanel", () => { }); }); + it("jumps straight to the last page without a cursor when the last-page button is clicked", async () => { + const firstPage = Array.from({ length: 25 }, (_, index) => logEntry({ request_id: `req-${index}` })); + const lastPage = Array.from({ length: 10 }, (_, index) => logEntry({ request_id: `req-last-${index}` })); + vi.mocked(uiSpendLogsCall).mockImplementation(async ({ page }) => + page === 3 + ? { + data: lastPage, + total: 60, + page: 3, + page_size: 25, + total_pages: 3, + next_session_cursor: null, + has_more: false, + } + : { + data: firstPage, + total: 60, + page: 1, + page_size: 25, + total_pages: 3, + next_session_cursor: "2026-07-07 09:50:13|key-1|sess-1", + has_more: true, + }, + ); + renderPanel(); + + await waitFor(() => expect(row("req-0")).not.toBeNull()); + expect(screen.getByTestId("pagination-page")).toHaveTextContent("Page 1 of 3"); + fireEvent.click(screen.getByTestId("pagination-last")); + + await waitFor(() => expect(row("req-last-0")).not.toBeNull()); + expect(lastCall()?.page).toBe(3); + expect(lastCall()?.params?.session_cursor).toBeUndefined(); + expect(screen.getByTestId("pagination-page")).toHaveTextContent("Page 3 of 3"); + expect(screen.getByTestId("pagination-range")).toHaveTextContent("Showing 51-60 of 60"); + expect(vi.mocked(uiSpendLogsCall).mock.calls.filter(([options]) => options.page === 2)).toHaveLength(0); + }); + it("drops the cursor and returns to the first page when a filter changes", async () => { const firstPage = Array.from({ length: 50 }, (_, index) => logEntry({ request_id: `req-${index}` })); vi.mocked(uiSpendLogsCall).mockResolvedValue({ diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.tsx index 6e984297bf2..4f39bb3b79b 100644 --- a/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/RequestLogsPanel.tsx @@ -209,15 +209,14 @@ export default function RequestLogsPanel({ accessToken, token, userRole, userID, setPagination({ ...requested, pageIndex: 0 }); return; } - if (requested.pageIndex <= pagination.pageIndex) { + if (requested.pageIndex !== pagination.pageIndex + 1) { setPagination(requested); return; } const nextCursor = filteredLogs.next_session_cursor; if (!nextCursor || logsQuery.isPlaceholderData) return; - const nextPageIndex = pagination.pageIndex + 1; - setSessionCursors((previous) => ({ ...previous, [nextPageIndex]: nextCursor })); - setPagination({ ...requested, pageIndex: nextPageIndex }); + setSessionCursors((previous) => ({ ...previous, [requested.pageIndex]: nextCursor })); + setPagination(requested); }, [usesSessionCursor, pagination, filteredLogs.next_session_cursor, logsQuery.isPlaceholderData], ); From 7169ddaef6ad4847f630f2c2bcefa2799db19c92 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 18:16:59 -0700 Subject: [PATCH 043/157] fix(router): treat a breaker-refused Redis read as a miss in the health state cache The sync Redis read now raises while the circuit breaker is open, and the health state merge caught that as a generic error, skipping the local write and logging an error on every background health check cycle. Read the shared snapshot through a helper that treats the refused read as a miss so the merge falls back to the pod-local copy the way a swallowed connection error already did --- litellm/router_utils/health_state_cache.py | 20 ++++++++++---- .../router_utils/test_health_state_cache.py | 27 +++++++++++++++++++ 2 files changed, 42 insertions(+), 5 deletions(-) diff --git a/litellm/router_utils/health_state_cache.py b/litellm/router_utils/health_state_cache.py index 22d816e13e9..c8ca7105392 100644 --- a/litellm/router_utils/health_state_cache.py +++ b/litellm/router_utils/health_state_cache.py @@ -12,6 +12,7 @@ from typing_extensions import TypedDict from litellm import verbose_logger from litellm.caching.caching import DualCache +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError if TYPE_CHECKING: from opentelemetry.trace import Span as _Span @@ -27,6 +28,16 @@ class DeploymentHealthStateValue(TypedDict): reason: str +def _read_shared_health_snapshot(cache: DualCache, key: str) -> object: + redis_cache: Final = cache.redis_cache + if redis_cache is None: + return None + try: + return redis_cache.get_cache(key) + except RedisCircuitBreakerOpenError: + return None + + class DeploymentHealthCache: """ Cache for deployment health states produced by background health checks. @@ -50,13 +61,12 @@ class DeploymentHealthCache: coexist on the one shared entry without erasing each other's results. The snapshot is read from Redis when available, since a pod-local read would only ever see this writer's own previous merge. When the Redis - read comes back empty (a miss, or a swallowed connection error), the - pod-local copy of the last merge is used so peers are not erased. + read comes back empty (a miss, a swallowed connection error, or a read + refused by the open circuit breaker), the pod-local copy of the last + merge is used so peers are not erased. """ try: - redis_raw: Final = ( - self.cache.redis_cache.get_cache(self.CACHE_KEY) if self.cache.redis_cache is not None else None - ) + redis_raw: Final = _read_shared_health_snapshot(self.cache, self.CACHE_KEY) raw: Final = redis_raw if isinstance(redis_raw, dict) else self.cache.get_cache(key=self.CACHE_KEY) existing: Final = raw if isinstance(raw, dict) else {} expiry_seconds: Final = self.staleness_threshold * 1.5 diff --git a/tests/test_litellm/router_utils/test_health_state_cache.py b/tests/test_litellm/router_utils/test_health_state_cache.py index ffd031f9b7d..aa976bb1002 100644 --- a/tests/test_litellm/router_utils/test_health_state_cache.py +++ b/tests/test_litellm/router_utils/test_health_state_cache.py @@ -7,6 +7,7 @@ import time import pytest from litellm.caching.caching import DualCache +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.router_utils.health_state_cache import DeploymentHealthCache @@ -145,8 +146,11 @@ class _SharedRedisFake: def __init__(self): self.store = {} self.fail_get = False + self.breaker_open = False def get_cache(self, key, parent_otel_span=None, **kwargs): + if self.breaker_open: + raise RedisCircuitBreakerOpenError("Redis circuit breaker is open - skipping get_cache") if self.fail_get: return None # RedisCache.get_cache swallows connection errors and returns None return self.store.get(key) @@ -192,3 +196,26 @@ def test_failed_redis_read_falls_back_to_local_copy(): {"prod-bad": {"is_healthy": False, "timestamp": time.time(), "reason": "check_failed"}} ) assert set(redis_fake.store[DeploymentHealthCache.CACHE_KEY]) == {"prod-bad", "internal-bad"} + + +def test_open_circuit_breaker_read_still_merges_into_local_copy(caplog): + """A read refused by the open breaker is a miss, so the merge and local write still happen quietly.""" + redis_fake = _SharedRedisFake() + pod_a = DeploymentHealthCache(cache=DualCache(redis_cache=redis_fake), staleness_threshold=60.0) + pod_b = DeploymentHealthCache(cache=DualCache(redis_cache=redis_fake), staleness_threshold=60.0) + pod_a.set_deployment_health_states( + {"prod-bad": {"is_healthy": False, "timestamp": time.time(), "reason": "check_failed"}} + ) + pod_b.set_deployment_health_states( + {"internal-bad": {"is_healthy": False, "timestamp": time.time(), "reason": "timeout"}} + ) + pod_a.set_deployment_health_states( + {"prod-bad": {"is_healthy": False, "timestamp": time.time(), "reason": "check_failed"}} + ) + redis_fake.breaker_open = True + with caplog.at_level("ERROR"): + pod_a.set_deployment_health_states( + {"prod-new-bad": {"is_healthy": False, "timestamp": time.time(), "reason": "check_failed"}} + ) + assert caplog.records == [] + assert pod_a.get_unhealthy_deployment_ids() == {"prod-bad", "internal-bad", "prod-new-bad"} From dcdd884352b1f39f51dd0719bba4eca0046859f8 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 18:36:19 -0700 Subject: [PATCH 044/157] fix(caching): let only the recovery probe close a half-open Redis breaker A call admitted before the breaker opened could finish while the breaker was HALF_OPEN and close it before the designated probe reported, so Redis traffic resumed on a stale answer. The admission now records whether the call is the probe and only the probe's success closes a half-open breaker. The cron job lock manager also logged an error every cycle the open breaker refused its Redis call, one line per job per pod. That refusal is now a debug line like every other guarded call, while real Redis errors still log at error --- litellm/caching/redis_cache.py | 43 ++++++++++------ .../db_transaction_queue/pod_lock_manager.py | 7 +-- .../test_litellm/caching/test_redis_cache.py | 49 +++++++++++++++++++ .../test_pod_lock_manager.py | 18 +++++++ 4 files changed, 100 insertions(+), 17 deletions(-) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 5b1a9739eb8..812b435ae5f 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -18,6 +18,7 @@ import logging import time from collections.abc import Awaitable, Callable, Sequence from contextvars import ContextVar +from dataclasses import dataclass from datetime import timedelta from typing import TYPE_CHECKING, Any, Final, Protocol, TypeVar, cast @@ -198,6 +199,9 @@ class RedisCircuitBreaker: self._state = self.CLOSED _breaker_metrics().record_state_change(None, self._state) + def is_half_open(self) -> bool: + return self._state == self.HALF_OPEN + def is_open(self) -> bool: """Returns True if Redis calls should be skipped.""" if not self.enabled: @@ -405,21 +409,32 @@ def log_redis_failure( logger.log(level, "%s: %s", message, exc, exc_info=exc if with_traceback else None) -def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> int: - """Reject the call if the breaker is open, else return the swallowed-failure count to compare against.""" +@dataclass(frozen=True, slots=True) +class _BreakerAdmission: + swallowed_before: int + is_probe: bool + + +def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> _BreakerAdmission: + """Reject the call if the breaker is open, else record what its success may later prove.""" if breaker.is_open(): raise RedisCircuitBreakerOpenError(f"Redis circuit breaker is open — skipping {name}") - return _swallowed_redis_failures.get() + return _BreakerAdmission(swallowed_before=_swallowed_redis_failures.get(), is_probe=breaker.is_half_open()) -def _exit_circuit_breaker(breaker: RedisCircuitBreaker, swallowed_before: int) -> None: - """Record success only when nothing failed while the call ran. +def _exit_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission) -> None: + """Record success only when nothing failed while the call ran and the call may vouch for Redis. Several Redis methods catch their own connection errors and return a default, so a - method that returned is not on its own proof of a healthy Redis. + method that returned is not on its own proof of a healthy Redis. While the breaker is + half open only the designated recovery probe may close it: a call admitted before the + breaker opened that finishes late says nothing about whether Redis recovered. """ - if _swallowed_redis_failures.get() == swallowed_before: - breaker.record_success() + if _swallowed_redis_failures.get() != admission.swallowed_before: + return + if breaker.is_half_open() and not admission.is_probe: + return + breaker.record_success() async def _run_under_circuit_breaker( @@ -432,14 +447,14 @@ async def _run_under_circuit_breaker( Shared by the method decorator and the Lua script executor so both feed the same health signal. """ - swallowed_before: Final = _enter_circuit_breaker(breaker, name) + admission: Final = _enter_circuit_breaker(breaker, name) try: result: Final = await call() except Exception as e: if _is_redis_health_failure(e): breaker.record_failure(is_timeout=_is_redis_timeout_failure(e)) raise - _exit_circuit_breaker(breaker, swallowed_before) + _exit_circuit_breaker(breaker, admission) return result @@ -449,14 +464,14 @@ def _run_under_circuit_breaker_sync( call: Callable[[], _RedisCallResult], ) -> _RedisCallResult: """Run one blocking Redis call under a circuit breaker, feeding the same health signal as the async path.""" - swallowed_before: Final = _enter_circuit_breaker(breaker, name) + admission: Final = _enter_circuit_breaker(breaker, name) try: result: Final = call() except Exception as e: if _is_redis_health_failure(e): breaker.record_failure(is_timeout=_is_redis_timeout_failure(e)) raise - _exit_circuit_breaker(breaker, swallowed_before) + _exit_circuit_breaker(breaker, admission) return result @@ -1395,12 +1410,12 @@ class RedisCache(BaseCache): key_value_dict = {} _key_list: Final = [key for key in key_list if key is not None] start_time: Final = time.time() - swallowed_before: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") + admission: Final = _enter_circuit_breaker(self._circuit_breaker, "batch_get_cache") try: _keys: Final = [self.check_and_fix_namespace(key=cache_key or "") for cache_key in _key_list] results: Final = self._run_redis_mget_operation(keys=_keys) - _exit_circuit_breaker(self._circuit_breaker, swallowed_before) + _exit_circuit_breaker(self._circuit_breaker, admission) end_time: Final = time.time() _duration: Final = end_time - start_time self.service_logger_obj.service_success_hook( diff --git a/litellm/proxy/db/db_transaction_queue/pod_lock_manager.py b/litellm/proxy/db/db_transaction_queue/pod_lock_manager.py index 4be1331e955..bc67617e444 100644 --- a/litellm/proxy/db/db_transaction_queue/pod_lock_manager.py +++ b/litellm/proxy/db/db_transaction_queue/pod_lock_manager.py @@ -1,10 +1,11 @@ import asyncio import json +import logging from typing import TYPE_CHECKING, Any, Final from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid -from litellm.caching.redis_cache import RedisCache +from litellm.caching.redis_cache import RedisCache, log_redis_failure from litellm.constants import DEFAULT_CRON_JOB_LOCK_TTL_SECONDS from litellm.proxy.db.db_transaction_queue.base_update_queue import service_logger_obj from litellm.types.services import ServiceTypes @@ -109,7 +110,7 @@ end ) return False except Exception as e: - verbose_proxy_logger.error("Error acquiring Redis lock for %s: %s", cronjob_id, e) + log_redis_failure(verbose_proxy_logger, logging.ERROR, f"Error acquiring Redis lock for {cronjob_id}", e) return False async def release_lock( @@ -151,7 +152,7 @@ end cronjob_id, ) except Exception as e: - verbose_proxy_logger.error("Error releasing Redis lock for %s: %s", cronjob_id, e) + log_redis_failure(verbose_proxy_logger, logging.ERROR, f"Error releasing Redis lock for {cronjob_id}", e) async def _compare_and_delete_lock(self, lock_key: str) -> int: """ diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index c6f0caa1dca..9df9c44a30f 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -1103,3 +1103,52 @@ def test_recovery_probe_still_closes_the_breaker(): assert breaker._state == breaker.CLOSED assert breaker.is_open() is False + + +@pytest.mark.asyncio +async def test_stale_success_during_the_recovery_probe_leaves_the_breaker_to_the_probe(): + """A call admitted before the trip that finishes while HALF_OPEN must not close the breaker. + + Only the one call designated as the recovery probe has actually reached Redis after the + outage, so closing on the straggler's success resumed full Redis traffic before the probe + had proven anything. + """ + from litellm.caching.redis_cache import RedisCircuitBreaker, _run_under_circuit_breaker + + breaker = RedisCircuitBreaker(failure_threshold=3, recovery_timeout=60) + stale_admitted = asyncio.Event() + stale_release = asyncio.Event() + probe_admitted = asyncio.Event() + probe_release = asyncio.Event() + + async def stale_call() -> str: + stale_admitted.set() + await stale_release.wait() + return "stale" + + async def probe_call() -> str: + probe_admitted.set() + await probe_release.wait() + return "probe" + + stale = asyncio.ensure_future(_run_under_circuit_breaker(breaker, "op", stale_call)) + await stale_admitted.wait() + for _ in range(3): + breaker.record_failure() + assert breaker._state == breaker.OPEN + breaker._opened_at = time.time() - 9999 + probe = asyncio.ensure_future(_run_under_circuit_breaker(breaker, "op", probe_call)) + await probe_admitted.wait() + assert breaker._state == breaker.HALF_OPEN + + stale_release.set() + assert await stale == "stale" + + assert breaker._state == breaker.HALF_OPEN, "the straggler must not close the breaker for the probe" + assert breaker.is_open() is True + + probe_release.set() + assert await probe == "probe" + + assert breaker._state == breaker.CLOSED + assert breaker.is_open() is False diff --git a/tests/test_litellm/proxy/db/db_transaction_queue/test_pod_lock_manager.py b/tests/test_litellm/proxy/db/db_transaction_queue/test_pod_lock_manager.py index ecd5c5f50c0..4684c3213d6 100644 --- a/tests/test_litellm/proxy/db/db_transaction_queue/test_pod_lock_manager.py +++ b/tests/test_litellm/proxy/db/db_transaction_queue/test_pod_lock_manager.py @@ -1,4 +1,5 @@ import json +import logging from datetime import datetime, timedelta, timezone from unittest.mock import AsyncMock, MagicMock, patch @@ -6,6 +7,7 @@ import pytest from fastapi.testclient import TestClient +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError from litellm.constants import DEFAULT_CRON_JOB_LOCK_TTL_SECONDS from litellm.proxy.db.db_transaction_queue.pod_lock_manager import PodLockManager @@ -215,6 +217,22 @@ async def test_redis_error_handling(pod_lock_manager, mock_redis): ) +@pytest.mark.asyncio +async def test_lock_refused_by_the_open_circuit_breaker_is_not_logged_as_an_error(pod_lock_manager, mock_redis, caplog): + """Every cron job retries its lock on a timer, so an open breaker must not add an error line per cycle.""" + refused = RedisCircuitBreakerOpenError("Redis circuit breaker is open - skipping async_set_cache") + mock_redis.async_set_cache.side_effect = refused + mock_redis.async_get_cache.return_value = pod_lock_manager.pod_id + mock_redis.async_delete_cache.side_effect = refused + + with caplog.at_level(logging.ERROR): + acquired = await pod_lock_manager.acquire_lock(cronjob_id="test_job") + await pod_lock_manager.release_lock(cronjob_id="test_job") + + assert acquired is False + assert caplog.records == [] + + @pytest.mark.asyncio async def test_bytes_handling(pod_lock_manager, mock_redis): """ From 01c6b50564c16375023dd240fd376ccca11fa42b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 18:57:24 -0700 Subject: [PATCH 045/157] fix(caching): let a Redis breaker success count only for the state that admitted the call --- litellm/caching/redis_cache.py | 23 +++++---- .../test_litellm/caching/test_redis_cache.py | 50 +++++++++++++++++++ 2 files changed, 64 insertions(+), 9 deletions(-) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 812b435ae5f..2c36995c4f8 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -197,10 +197,13 @@ class RedisCircuitBreaker: self._timeout_streak_started_at: float | None = None self._opened_at: float | None = None self._state = self.CLOSED + self._generation = 0 _breaker_metrics().record_state_change(None, self._state) - def is_half_open(self) -> bool: - return self._state == self.HALF_OPEN + @property + def generation(self) -> int: + """Counts state transitions, so a call can tell whether the breaker moved while it ran.""" + return self._generation def is_open(self) -> bool: """Returns True if Redis calls should be skipped.""" @@ -270,6 +273,7 @@ class RedisCircuitBreaker: _breaker_metrics().record_transition(state) _breaker_metrics().record_state_change(self._state, state) self._state = state + self._generation += 1 _RedisCallResult = TypeVar("_RedisCallResult") @@ -412,27 +416,28 @@ def log_redis_failure( @dataclass(frozen=True, slots=True) class _BreakerAdmission: swallowed_before: int - is_probe: bool + generation: int def _enter_circuit_breaker(breaker: RedisCircuitBreaker, name: str) -> _BreakerAdmission: """Reject the call if the breaker is open, else record what its success may later prove.""" if breaker.is_open(): raise RedisCircuitBreakerOpenError(f"Redis circuit breaker is open — skipping {name}") - return _BreakerAdmission(swallowed_before=_swallowed_redis_failures.get(), is_probe=breaker.is_half_open()) + return _BreakerAdmission(swallowed_before=_swallowed_redis_failures.get(), generation=breaker.generation) def _exit_circuit_breaker(breaker: RedisCircuitBreaker, admission: _BreakerAdmission) -> None: - """Record success only when nothing failed while the call ran and the call may vouch for Redis. + """Record success only when nothing failed while the call ran and the breaker has not moved since. Several Redis methods catch their own connection errors and return a default, so a - method that returned is not on its own proof of a healthy Redis. While the breaker is - half open only the designated recovery probe may close it: a call admitted before the - breaker opened that finishes late says nothing about whether Redis recovered. + method that returned is not on its own proof of a healthy Redis. A success also vouches + only for the breaker state that admitted the call: a call admitted before the breaker + opened, or a probe admitted before a later failure reopened it, finishes knowing nothing + about whether Redis has recovered since, so only the current probe may close the breaker. """ if _swallowed_redis_failures.get() != admission.swallowed_before: return - if breaker.is_half_open() and not admission.is_probe: + if breaker.generation != admission.generation: return breaker.record_success() diff --git a/tests/test_litellm/caching/test_redis_cache.py b/tests/test_litellm/caching/test_redis_cache.py index 9df9c44a30f..bcae33b976e 100644 --- a/tests/test_litellm/caching/test_redis_cache.py +++ b/tests/test_litellm/caching/test_redis_cache.py @@ -1152,3 +1152,53 @@ async def test_stale_success_during_the_recovery_probe_leaves_the_breaker_to_the assert breaker._state == breaker.CLOSED assert breaker.is_open() is False + + +@pytest.mark.asyncio +async def test_a_probe_overtaken_by_a_later_outage_leaves_the_breaker_to_the_new_probe(): + """A probe still in flight when a late failure reopens the breaker must not close it for the next probe. + + Once the breaker has reopened, only the probe admitted after that outage has reached + Redis, so the older probe's success no longer says anything about whether Redis recovered. + """ + from litellm.caching.redis_cache import RedisCircuitBreaker, _run_under_circuit_breaker + + breaker = RedisCircuitBreaker(failure_threshold=3, recovery_timeout=60) + old_probe_admitted = asyncio.Event() + old_probe_release = asyncio.Event() + new_probe_admitted = asyncio.Event() + new_probe_release = asyncio.Event() + + async def old_probe_call() -> str: + old_probe_admitted.set() + await old_probe_release.wait() + return "old probe" + + async def new_probe_call() -> str: + new_probe_admitted.set() + await new_probe_release.wait() + return "new probe" + + for _ in range(3): + breaker.record_failure() + breaker._opened_at = time.time() - 9999 + old_probe = asyncio.ensure_future(_run_under_circuit_breaker(breaker, "op", old_probe_call)) + await old_probe_admitted.wait() + assert breaker._state == breaker.HALF_OPEN + + breaker.record_failure() + assert breaker._state == breaker.OPEN + breaker._opened_at = time.time() - 9999 + new_probe = asyncio.ensure_future(_run_under_circuit_breaker(breaker, "op", new_probe_call)) + await new_probe_admitted.wait() + assert breaker._state == breaker.HALF_OPEN + + old_probe_release.set() + assert await old_probe == "old probe" + + assert breaker._state == breaker.HALF_OPEN, "the overtaken probe must not close the breaker for the new probe" + assert breaker.is_open() is True + + new_probe_release.set() + assert await new_probe == "new probe" + assert breaker._state == breaker.CLOSED From 0ffe6512de83f4907ff8a387d673b04fb14eb7db Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:15:40 -0700 Subject: [PATCH 046/157] fix(router): keep the routing and budget sync loops quiet while the Redis breaker is open --- .../router_strategy/base_routing_strategy.py | 5 ++- litellm/router_strategy/budget_limiter.py | 17 ++++----- .../test_base_routing_strategy.py | 18 ++++++++- .../test_budget_limiter_hotpath.py | 37 +++++++++++++++++++ 4 files changed, 64 insertions(+), 13 deletions(-) diff --git a/litellm/router_strategy/base_routing_strategy.py b/litellm/router_strategy/base_routing_strategy.py index 8235761ca98..686d57e2b77 100644 --- a/litellm/router_strategy/base_routing_strategy.py +++ b/litellm/router_strategy/base_routing_strategy.py @@ -3,12 +3,13 @@ Base class across routing strategies to abstract commmon functions like batch in """ import asyncio +import logging from abc import ABC from typing import Final from litellm._logging import verbose_router_logger from litellm.caching.caching import DualCache -from litellm.caching.redis_cache import RedisPipelineIncrementOperation +from litellm.caching.redis_cache import RedisPipelineIncrementOperation, log_redis_failure from litellm.constants import DEFAULT_REDIS_SYNC_INTERVAL @@ -147,7 +148,7 @@ class BaseRoutingStrategy(ABC): return return_result except Exception as e: - verbose_router_logger.error("Error syncing in-memory cache with Redis: %s", e) + log_redis_failure(verbose_router_logger, logging.ERROR, "Error syncing in-memory cache with Redis", e) self.redis_increment_operation_queue = [] def add_to_in_memory_keys_to_update(self, key: str): diff --git a/litellm/router_strategy/budget_limiter.py b/litellm/router_strategy/budget_limiter.py index a8d51f95e45..acdc57d706d 100644 --- a/litellm/router_strategy/budget_limiter.py +++ b/litellm/router_strategy/budget_limiter.py @@ -20,6 +20,7 @@ anthropic: import asyncio import builtins +import logging from collections.abc import Mapping from datetime import datetime, timedelta, timezone from typing import Any, Final @@ -27,7 +28,7 @@ from typing import Any, Final import litellm from litellm._logging import verbose_router_logger from litellm.caching.caching import DualCache -from litellm.caching.redis_cache import RedisPipelineIncrementOperation +from litellm.caching.redis_cache import RedisPipelineIncrementOperation, log_redis_failure from litellm.integrations.custom_logger import CustomLogger, Span from litellm.litellm_core_utils.core_helpers import ( get_metadata_variable_name_from_kwargs, @@ -536,17 +537,13 @@ class RouterBudgetLimiting(CustomLogger): "Pushing Redis Increment Pipeline for queue: %s", self.redis_increment_operation_queue, ) - if len(self.redis_increment_operation_queue) > 0: - asyncio.create_task( - self.dual_cache.redis_cache.async_increment_pipeline( - increment_list=self.redis_increment_operation_queue, - ) - ) - + queued: Final = self.redis_increment_operation_queue self.redis_increment_operation_queue = [] + if queued: + await self.dual_cache.redis_cache.async_increment_pipeline(increment_list=queued) except Exception as e: - verbose_router_logger.error("Error syncing in-memory cache with Redis: %s", e) + log_redis_failure(verbose_router_logger, logging.ERROR, "Error syncing in-memory cache with Redis", e) async def _sync_in_memory_spend_with_redis(self): """ @@ -601,7 +598,7 @@ class RouterBudgetLimiting(CustomLogger): verbose_router_logger.debug("Updated in-memory cache for %s: %s", key, value) except Exception as e: - verbose_router_logger.error("Error syncing in-memory cache with Redis: %s", e) + log_redis_failure(verbose_router_logger, logging.ERROR, "Error syncing in-memory cache with Redis", e) def _get_budget_config_for_deployment( self, diff --git a/tests/test_litellm/router_strategy/test_base_routing_strategy.py b/tests/test_litellm/router_strategy/test_base_routing_strategy.py index 154042692d0..dc75c3d2916 100644 --- a/tests/test_litellm/router_strategy/test_base_routing_strategy.py +++ b/tests/test_litellm/router_strategy/test_base_routing_strategy.py @@ -1,4 +1,5 @@ import json +import logging from typing import Any, Dict, List, Optional, Set, Union import pytest @@ -9,7 +10,7 @@ from unittest.mock import MagicMock, patch from litellm.caching.caching import DualCache -from litellm.caching.redis_cache import RedisPipelineIncrementOperation +from litellm.caching.redis_cache import RedisCircuitBreakerOpenError, RedisPipelineIncrementOperation from litellm.router_strategy.base_routing_strategy import BaseRoutingStrategy @@ -146,3 +147,18 @@ async def test_cache_keys_management(base_strategy): # Test resetting cache keys base_strategy.reset_in_memory_keys_to_update() assert len(base_strategy.get_in_memory_keys_to_update()) == 0 + + +@pytest.mark.asyncio +async def test_push_refused_by_the_open_circuit_breaker_is_not_logged_as_an_error(base_strategy, mock_dual_cache, caplog): + """The sync loop pushes every 100 ms under usage-based routing, so an open breaker must not add an error line per cycle.""" + mock_dual_cache.redis_cache.async_increment_pipeline.side_effect = RedisCircuitBreakerOpenError( + "Redis circuit breaker is open - skipping async_increment_pipeline" + ) + base_strategy.redis_increment_operation_queue = [{"key": "k", "increment_value": 1.0, "ttl": 60}] + + with caplog.at_level(logging.ERROR): + await base_strategy._push_in_memory_increments_to_redis() + + assert caplog.records == [] + assert base_strategy.redis_increment_operation_queue == [] diff --git a/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py b/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py index 36fa38bacb5..128b2dbd842 100644 --- a/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py +++ b/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py @@ -1,7 +1,13 @@ +import asyncio +import gc +import logging +from unittest.mock import AsyncMock, MagicMock + import pytest import litellm from litellm.caching.caching import DualCache +from litellm.caching.redis_cache import RedisCache, RedisCircuitBreakerOpenError from litellm.router_strategy.budget_limiter import RouterBudgetLimiting from litellm.types.router import LiteLLM_Params from litellm.types.utils import BudgetConfig @@ -303,3 +309,34 @@ def test_router_add_deployment_registers_deployment_budget( ) assert config is not None assert config.max_budget == 0.000000000001 + + +@pytest.mark.asyncio +async def test_sync_refused_by_the_open_circuit_breaker_is_quiet_and_leaks_no_task(disable_budget_sync, caplog): + """The budget sync runs every second, so an open breaker must not add an error line or an unretrieved task exception per cycle.""" + refused = RedisCircuitBreakerOpenError("Redis circuit breaker is open - skipping async_increment_pipeline") + redis_cache = MagicMock(spec=RedisCache) + redis_cache.async_increment_pipeline = AsyncMock(side_effect=refused) + redis_cache.async_batch_get_cache = AsyncMock(side_effect=refused) + limiter = RouterBudgetLimiting( + dual_cache=DualCache(redis_cache=redis_cache), + provider_budget_config={"openai": BudgetConfig(max_budget=1.0, budget_duration="1d")}, + ) + await asyncio.gather(*(task for task in asyncio.all_tasks() if task is not asyncio.current_task())) + limiter.redis_increment_operation_queue = [{"key": "provider_spend:openai:1d", "increment_value": 0.5, "ttl": 60}] + loop = asyncio.get_running_loop() + unretrieved = MagicMock() + loop.set_exception_handler(unretrieved) + + try: + with caplog.at_level(logging.ERROR): + await limiter._sync_in_memory_spend_with_redis() + await asyncio.sleep(0) + gc.collect() + finally: + loop.set_exception_handler(None) + + assert caplog.records == [] + unretrieved.assert_not_called() + assert limiter.redis_increment_operation_queue == [] + assert redis_cache.async_increment_pipeline.await_count == 1 From 8c82c325aca247c09da4ef48e7db0058af3a590d Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:14:15 -0700 Subject: [PATCH 047/157] fix(mcp): check OpenAPI specifications without native MCP handshakes --- .../mcp_server/mcp_server_manager.py | 35 +++++- .../mcp_server/test_mcp_server_manager.py | 110 ++++++++++++++++++ 2 files changed, 144 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index af25b0e919a..5a9000d2f23 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -883,6 +883,27 @@ def _sanitized_error_text(exc: Exception) -> str: return re.sub(r"https?://\S+", "", str(exc))[:200] +async def _openapi_spec_health( + spec_path: str, *, timeout: float +) -> tuple[Literal["healthy", "unhealthy", "unknown"], str | None]: + """Check specification availability, not upstream operations or user credentials.""" + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import load_openapi_spec_async + + if not spec_path.startswith(("http://", "https://")): + return "unknown", "OpenAPI servers have no protocol-level health probe" + try: + await asyncio.wait_for(load_openapi_spec_async(spec_path), timeout=timeout) + except asyncio.TimeoutError: + return "unhealthy", f"OpenAPI specification check timed out after {timeout} seconds" + except asyncio.CancelledError: + return "unknown", "OpenAPI specification check was cancelled" + except HTTPStatusError as exc: + return "unhealthy", f"OpenAPI specification request failed (HTTP {exc.response.status_code})" + except (httpx.RequestError, ValueError, OSError) as exc: + return "unhealthy", f"OpenAPI specification could not be loaded ({type(exc).__name__})" + return "healthy", None + + def _discovery_failure_leaves_needs_unresolved( *, needs_authorization_url: bool, @@ -6665,7 +6686,7 @@ class MCPServerManager: Returns: Dict containing health check results """ - from datetime import datetime + from datetime import datetime, timezone server: Final = self.get_mcp_server_by_id(server_id) if not server: @@ -6679,6 +6700,18 @@ class MCPServerManager: last_health_check=datetime.now(), ) + if server.spec_path: + spec_status, spec_error = await _openapi_spec_health(server.spec_path, timeout=MCP_HEALTH_CHECK_TIMEOUT) + return self._build_mcp_server_table(server).model_copy( + update=MappingProxyType( + { + "status": spec_status, + "health_check_error": spec_error, + "last_health_check": datetime.now(timezone.utc), + } + ) + ) + status: Literal["healthy", "unhealthy", "unknown"] = "unknown" health_check_error = None diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py index adf985e9a21..e3172d4939d 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py @@ -4473,6 +4473,116 @@ class TestMCPServerManager: assert len(result) == 1 assert result[0].name == "github_tool_1" + @pytest.mark.asyncio + @pytest.mark.parametrize("auth_type", [MCPAuth.none, MCPAuth.bearer_token, MCPAuth.api_key, MCPAuth.oauth2]) + @pytest.mark.parametrize("is_byok", [False, True]) + @pytest.mark.parametrize("scheme", ["http", "https"]) + async def test_openapi_health_loads_spec_without_mcp_handshake(self, respx_mock, monkeypatch, auth_type, is_byok, scheme): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + manager = MCPServerManager() + server = MCPServer( + server_id="openapi-health", + name="openapi-health", + transport=MCPTransport.http, + url="https://rest.example.com", + spec_path=f"{scheme}://93.184.216.34/openapi.json", + auth_type=auth_type, + is_byok=is_byok, + authentication_token=None if is_byok else "shared-secret", + static_headers={"Authorization": "Bearer static-secret"}, + ) + manager.registry = {server.server_id: server} + route = respx_mock.get(server.spec_path).respond(200, json={"openapi": "3.0.0", "paths": {}}) + result = await manager.health_check_server(server.server_id, mcp_auth_header="caller-secret") + assert result.status == "healthy" + assert result.health_check_error is None + assert result.last_health_check is not None + assert result.spec_path == server.spec_path + assert route.call_count == 1 + assert "authorization" not in route.calls[0].request.headers + assert "x-api-key" not in route.calls[0].request.headers + + @pytest.mark.asyncio + @pytest.mark.parametrize("auth_type", [MCPAuth.none, MCPAuth.bearer_token]) + @pytest.mark.parametrize("spec_path", ["/config/openapi.json", "relative/openapi.json"]) + async def test_openapi_local_spec_health_is_unknown(self, respx_mock, auth_type, spec_path): + manager = MCPServerManager() + server = MCPServer( + server_id="local-openapi-health", + name="local-openapi-health", + transport=MCPTransport.http, + url="https://rest.example.com", + spec_path=spec_path, + auth_type=auth_type, + is_byok=True, + ) + manager.registry = {server.server_id: server} + result = await manager.health_check_server(server.server_id) + assert result.status == "unknown" + assert result.health_check_error == "OpenAPI servers have no protocol-level health probe" + assert result.last_health_check is not None + assert not respx_mock.calls + + @pytest.mark.asyncio + @pytest.mark.parametrize( + ("failure", "expected_status", "expected_error"), + [ + (httpx.Response(401, text="secret response content"), "unhealthy", "OpenAPI specification request failed (HTTP 401)"), + (httpx.Response(404), "unhealthy", "OpenAPI specification request failed (HTTP 404)"), + (httpx.Response(500), "unhealthy", "OpenAPI specification request failed (HTTP 500)"), + (httpx.ConnectError("secret network details"), "unhealthy", "OpenAPI specification could not be loaded (ConnectError)"), + (httpx.Response(200, text="secret invalid JSON body"), "unhealthy", "OpenAPI specification could not be loaded (JSONDecodeError)"), + ], + ) + async def test_openapi_health_reports_safe_failures(self, respx_mock, monkeypatch, failure, expected_status, expected_error): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + manager = MCPServerManager() + server = MCPServer( + server_id="failed-openapi-health", + name="failed-openapi-health", + transport=MCPTransport.http, + url="https://rest.example.com", + spec_path="https://93.184.216.34/key-secret?token=query-secret", + auth_type=MCPAuth.bearer_token, + is_byok=True, + ) + manager.registry = {server.server_id: server} + route = respx_mock.get(server.spec_path).mock(side_effect=[failure]) + result = await manager.health_check_server(server.server_id) + assert result.status == expected_status + assert result.health_check_error == expected_error + assert result.last_health_check is not None + assert route.call_count == 1 + + @pytest.mark.asyncio + @pytest.mark.parametrize("cancel", [False, True]) + async def test_openapi_health_timeout_and_cancellation_cleanup(self, respx_mock, monkeypatch, cancel): + from litellm.proxy._experimental.mcp_server.mcp_server_manager import _openapi_spec_health + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + started = asyncio.Event() + cancelled = asyncio.Event() + + async def slow_load(request): + started.set() + try: + await asyncio.Event().wait() + finally: + cancelled.set() + + respx_mock.get("https://93.184.216.34/slow.json").mock(side_effect=slow_load) + task = asyncio.create_task(_openapi_spec_health("https://93.184.216.34/slow.json", timeout=0.1)) + await asyncio.wait_for(started.wait(), timeout=1) + if cancel: + task.cancel() + status, error = await task + assert status == ("unknown" if cancel else "unhealthy") + assert error == ( + "OpenAPI specification check was cancelled" + if cancel else "OpenAPI specification check timed out after 0.1 seconds" + ) + assert cancelled.is_set() + @pytest.mark.asyncio async def test_health_check_server_healthy(self): """Test health check for a healthy server""" From 49809a814bf6794cf9190104de3cb1b20be166c8 Mon Sep 17 00:00:00 2001 From: Hayden Moulds Date: Thu, 10 Sep 2026 16:03:53 +1000 Subject: [PATCH 048/157] fix(proxy): preserve metadata for public team aliases --- litellm/proxy/proxy_server.py | 7 +- .../test_team_alias_listing_metadata.py | 94 +++++++++++++++++++ 2 files changed, 98 insertions(+), 3 deletions(-) create mode 100644 tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 0f24acb8bb4..ccabc4ed572 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -10827,14 +10827,15 @@ async def model_info( # Use the actual litellm model from the deployment to get provider info _, provider, _, _ = litellm.get_llm_provider(model=deployment.litellm_params.model) - response_id: Final = internal_to_public.get(resolved_model_id, model_id) - return create_model_info_response( - model_id=response_id, + response = create_model_info_response( + model_id=resolved_model_id, provider=provider, include_metadata=False, fallback_type=None, llm_router=llm_router, ) + response["id"] = internal_to_public.get(resolved_model_id, model_id) + return response def _blocked_response_usage(original_response: object | None) -> "litellm.Usage": diff --git a/tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py b/tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py new file mode 100644 index 00000000000..ce637d2e19f --- /dev/null +++ b/tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py @@ -0,0 +1,94 @@ +"""Regression coverage for metadata on public team model aliases.""" + +from unittest.mock import MagicMock + +import pytest + +import litellm.proxy.proxy_server as ps +from litellm.proxy._types import UserAPIKeyAuth +from litellm.types.router import DeploymentModelListingInfo + + +def _team_router(*, public_name: str, internal_name: str, underlying_model: str, listing_info): + deployment = { + "model_name": internal_name, + "litellm_params": {"model": underlying_model}, + "model_info": { + "id": "deployment-id", + "team_id": "teamx", + "team_public_model_name": public_name, + "access_groups": ["team-access"], + }, + } + router = MagicMock() + router.get_model_names.return_value = [internal_name] + router.get_model_access_groups.return_value = {"team-access": [internal_name]} + router.get_fully_blocked_model_names.return_value = set() + router.get_model_listing_info.return_value = listing_info + router.get_model_group_info.return_value = None + router.model_list = [deployment] + router.get_model_list.return_value = [deployment] + return router + + +@pytest.mark.asyncio +async def test_team_alias_inherits_deployment_token_limits_and_chat_mode(monkeypatch): + router = _team_router( + public_name="GPT Terra", + internal_name="model_name_teamx_terra_uuid", + underlying_model="azure/gpt-4.1", + listing_info=DeploymentModelListingInfo( + cost_map_keys=("azure/gpt-4.1",), + max_input_tokens=876000, + max_output_tokens=128000, + ), + ) + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {"use_team_public_model_name": True}) + + key = UserAPIKeyAuth(user_id="user", api_key="***", models=["team-access"], team_models=[]) + response = await ps.model_list(user_api_key_dict=key, include_metadata=True) + + assert response["data"] == [ + { + "id": "GPT Terra", + "object": "model", + "created": 1677610602, + "owned_by": "openai", + "mode": "chat", + "max_input_tokens": 876000, + "max_output_tokens": 128000, + "metadata": {"fallbacks": []}, + } + ] + + +@pytest.mark.asyncio +async def test_team_image_alias_inherits_image_generation_mode(monkeypatch): + router = _team_router( + public_name="image", + internal_name="model_name_teamx_image_uuid", + underlying_model="openai/gpt-image-1", + listing_info=DeploymentModelListingInfo( + cost_map_keys=("openai/gpt-image-1",), + max_input_tokens=None, + max_output_tokens=None, + ), + ) + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {"use_team_public_model_name": True}) + + key = UserAPIKeyAuth(user_id="user", api_key="***", models=["team-access"], team_models=[]) + response = await ps.model_list(user_api_key_dict=key) + + assert response["data"] == [ + { + "id": "image", + "object": "model", + "created": 1677610602, + "owned_by": "openai", + "mode": "image_generation", + } + ] From e664500003201dff80ceeab4008be7895109487f Mon Sep 17 00:00:00 2001 From: Hayden Moulds Date: Thu, 10 Sep 2026 16:28:48 +1000 Subject: [PATCH 049/157] test(proxy): cover team alias retrieve metadata --- .../proxy_server/test_team_model_name_translation.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py index 300cac5cdb2..acb352bcc8c 100644 --- a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py +++ b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py @@ -27,6 +27,7 @@ from litellm.proxy.proxy_server import ( _get_proxy_model_info, _translate_model_name_for_response, ) +from litellm.types.router import DeploymentModelListingInfo def _team_row() -> dict: @@ -1427,8 +1428,13 @@ async def test_retrieve_model_by_public_name_returns_200(monkeypatch): team_row = _team_row() router = _public_named_router(team_row) deployment = MagicMock() - deployment.litellm_params.model = "azure/gpt-5.2-low-rpm-testing" + deployment.litellm_params.model = "azure/gpt-4.1" router.get_deployment_by_model_group_name.return_value = deployment + router.get_model_listing_info.return_value = DeploymentModelListingInfo( + cost_map_keys=("azure/gpt-4.1",), + max_input_tokens=16384, + max_output_tokens=4096, + ) monkeypatch.setattr(ps, "llm_router", router) monkeypatch.setattr(ps, "general_settings", {}) @@ -1445,6 +1451,9 @@ async def test_retrieve_model_by_public_name_returns_200(monkeypatch): resp = await ps.model_info(model_id="team-claude-sonnet", user_api_key_dict=key) assert resp["id"] == "team-claude-sonnet" + assert resp.get("mode") == "chat" + assert resp.get("max_input_tokens") == 16384 + assert resp.get("max_output_tokens") == 4096 # lookup happened by the internal routing key, not the public name router.get_deployment_by_model_group_name.assert_called_once_with( "model_name_team-abc-123_4a6b8" From e18d766f53c7d4e5a808f2056fdca6da96f5d75a Mon Sep 17 00:00:00 2001 From: Hayden Moulds Date: Thu, 10 Sep 2026 17:37:32 +1000 Subject: [PATCH 050/157] test(proxy): consolidate team alias metadata coverage --- .../test_team_alias_listing_metadata.py | 94 ------------------- .../test_team_model_name_translation.py | 87 +++++++++++++++++ 2 files changed, 87 insertions(+), 94 deletions(-) delete mode 100644 tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py diff --git a/tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py b/tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py deleted file mode 100644 index ce637d2e19f..00000000000 --- a/tests/test_litellm/proxy/proxy_server/test_team_alias_listing_metadata.py +++ /dev/null @@ -1,94 +0,0 @@ -"""Regression coverage for metadata on public team model aliases.""" - -from unittest.mock import MagicMock - -import pytest - -import litellm.proxy.proxy_server as ps -from litellm.proxy._types import UserAPIKeyAuth -from litellm.types.router import DeploymentModelListingInfo - - -def _team_router(*, public_name: str, internal_name: str, underlying_model: str, listing_info): - deployment = { - "model_name": internal_name, - "litellm_params": {"model": underlying_model}, - "model_info": { - "id": "deployment-id", - "team_id": "teamx", - "team_public_model_name": public_name, - "access_groups": ["team-access"], - }, - } - router = MagicMock() - router.get_model_names.return_value = [internal_name] - router.get_model_access_groups.return_value = {"team-access": [internal_name]} - router.get_fully_blocked_model_names.return_value = set() - router.get_model_listing_info.return_value = listing_info - router.get_model_group_info.return_value = None - router.model_list = [deployment] - router.get_model_list.return_value = [deployment] - return router - - -@pytest.mark.asyncio -async def test_team_alias_inherits_deployment_token_limits_and_chat_mode(monkeypatch): - router = _team_router( - public_name="GPT Terra", - internal_name="model_name_teamx_terra_uuid", - underlying_model="azure/gpt-4.1", - listing_info=DeploymentModelListingInfo( - cost_map_keys=("azure/gpt-4.1",), - max_input_tokens=876000, - max_output_tokens=128000, - ), - ) - monkeypatch.setattr(ps, "llm_router", router) - monkeypatch.setattr(ps, "user_model", None) - monkeypatch.setattr(ps, "general_settings", {"use_team_public_model_name": True}) - - key = UserAPIKeyAuth(user_id="user", api_key="***", models=["team-access"], team_models=[]) - response = await ps.model_list(user_api_key_dict=key, include_metadata=True) - - assert response["data"] == [ - { - "id": "GPT Terra", - "object": "model", - "created": 1677610602, - "owned_by": "openai", - "mode": "chat", - "max_input_tokens": 876000, - "max_output_tokens": 128000, - "metadata": {"fallbacks": []}, - } - ] - - -@pytest.mark.asyncio -async def test_team_image_alias_inherits_image_generation_mode(monkeypatch): - router = _team_router( - public_name="image", - internal_name="model_name_teamx_image_uuid", - underlying_model="openai/gpt-image-1", - listing_info=DeploymentModelListingInfo( - cost_map_keys=("openai/gpt-image-1",), - max_input_tokens=None, - max_output_tokens=None, - ), - ) - monkeypatch.setattr(ps, "llm_router", router) - monkeypatch.setattr(ps, "user_model", None) - monkeypatch.setattr(ps, "general_settings", {"use_team_public_model_name": True}) - - key = UserAPIKeyAuth(user_id="user", api_key="***", models=["team-access"], team_models=[]) - response = await ps.model_list(user_api_key_dict=key) - - assert response["data"] == [ - { - "id": "image", - "object": "model", - "created": 1677610602, - "owned_by": "openai", - "mode": "image_generation", - } - ] diff --git a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py index acb352bcc8c..e9c2ff492a0 100644 --- a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py +++ b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py @@ -1090,6 +1090,93 @@ async def test_v1_models_metadata_does_not_leak_other_team_fallbacks(monkeypatch ] +@pytest.mark.asyncio +async def test_v1_models_team_alias_inherits_token_limits_and_chat_mode(monkeypatch): + team_dep = { + "model_name": "model_name_teamX_terra_uuid", + "litellm_params": {"model": "azure/gpt-4.1"}, + "model_info": { + "id": "id-terra", + "team_id": "teamX", + "team_public_model_name": "GPT Terra", + "access_groups": ["grp-a"], + "mode": "chat", + "max_input_tokens": 876000, + "max_output_tokens": 128000, + }, + } + router = MagicMock() + router.get_model_names.return_value = ["model_name_teamX_terra_uuid"] + router.get_model_access_groups.return_value = {"grp-a": ["model_name_teamX_terra_uuid"]} + router.get_fully_blocked_model_names.return_value = set() + router.get_configured_token_limits.return_value = (876000, 128000) + router.get_configured_mode.return_value = "chat" + router.model_list = [team_dep] + router.get_model_list.return_value = [team_dep] + router.get_model_group_info.return_value = None + + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {"use_team_public_model_name": True}) + + key = UserAPIKeyAuth(user_id="user", api_key="***", models=["grp-a"], team_models=[]) + response = await ps.model_list(user_api_key_dict=key, include_metadata=True) + + assert response["data"] == [ + { + "id": "GPT Terra", + "object": "model", + "created": 1677610602, + "owned_by": "openai", + "mode": "chat", + "max_input_tokens": 876000, + "max_output_tokens": 128000, + "metadata": {"fallbacks": []}, + } + ] + + +@pytest.mark.asyncio +async def test_v1_models_team_image_alias_inherits_image_generation_mode(monkeypatch): + team_dep = { + "model_name": "model_name_teamX_image_uuid", + "litellm_params": {"model": "openai/gpt-image-1"}, + "model_info": { + "id": "id-image", + "team_id": "teamX", + "team_public_model_name": "image", + "access_groups": ["grp-a"], + "mode": "image_generation", + }, + } + router = MagicMock() + router.get_model_names.return_value = ["model_name_teamX_image_uuid"] + router.get_model_access_groups.return_value = {"grp-a": ["model_name_teamX_image_uuid"]} + router.get_fully_blocked_model_names.return_value = set() + router.get_configured_token_limits.return_value = (None, None) + router.get_configured_mode.return_value = "image_generation" + router.model_list = [team_dep] + router.get_model_list.return_value = [team_dep] + router.get_model_group_info.return_value = None + + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {"use_team_public_model_name": True}) + + key = UserAPIKeyAuth(user_id="user", api_key="***", models=["grp-a"], team_models=[]) + response = await ps.model_list(user_api_key_dict=key) + + assert response["data"] == [ + { + "id": "image", + "object": "model", + "created": 1677610602, + "owned_by": "openai", + "mode": "image_generation", + } + ] + + def test_translate_team_model_names_for_listing_swaps_and_dedupes(): """Internal team routing keys -> public name; sibling deployments sharing a public name collapse to one entry (order preserved); globals untouched.""" From 2f8f3d3d21e071fd4a85b5bc02de4445a23078e0 Mon Sep 17 00:00:00 2001 From: Hayden Moulds Date: Thu, 10 Sep 2026 18:54:06 +1000 Subject: [PATCH 051/157] fix(proxy): build public team alias response immutably --- litellm/proxy/proxy_server.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index ccabc4ed572..c3c45855c5f 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -10827,15 +10827,14 @@ async def model_info( # Use the actual litellm model from the deployment to get provider info _, provider, _, _ = litellm.get_llm_provider(model=deployment.litellm_params.model) - response = create_model_info_response( + response: Final = create_model_info_response( model_id=resolved_model_id, provider=provider, include_metadata=False, fallback_type=None, llm_router=llm_router, ) - response["id"] = internal_to_public.get(resolved_model_id, model_id) - return response + return {**response, "id": internal_to_public.get(resolved_model_id, model_id)} def _blocked_response_usage(original_response: object | None) -> "litellm.Usage": From 5b9244105c3989f1edf12fca5a64e9fe4fd3deb0 Mon Sep 17 00:00:00 2001 From: Hayden Moulds Date: Fri, 11 Sep 2026 12:23:48 +1000 Subject: [PATCH 052/157] test(proxy): use deployment listing metadata in alias coverage --- .../proxy_server/test_team_model_name_translation.py | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py index e9c2ff492a0..bc346874e0d 100644 --- a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py +++ b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py @@ -1109,7 +1109,11 @@ async def test_v1_models_team_alias_inherits_token_limits_and_chat_mode(monkeypa router.get_model_names.return_value = ["model_name_teamX_terra_uuid"] router.get_model_access_groups.return_value = {"grp-a": ["model_name_teamX_terra_uuid"]} router.get_fully_blocked_model_names.return_value = set() - router.get_configured_token_limits.return_value = (876000, 128000) + router.get_model_listing_info.return_value = DeploymentModelListingInfo( + cost_map_keys=("azure/gpt-4.1",), + max_input_tokens=876000, + max_output_tokens=128000, + ) router.get_configured_mode.return_value = "chat" router.model_list = [team_dep] router.get_model_list.return_value = [team_dep] @@ -1153,7 +1157,11 @@ async def test_v1_models_team_image_alias_inherits_image_generation_mode(monkeyp router.get_model_names.return_value = ["model_name_teamX_image_uuid"] router.get_model_access_groups.return_value = {"grp-a": ["model_name_teamX_image_uuid"]} router.get_fully_blocked_model_names.return_value = set() - router.get_configured_token_limits.return_value = (None, None) + router.get_model_listing_info.return_value = DeploymentModelListingInfo( + cost_map_keys=("openai/gpt-image-1",), + max_input_tokens=None, + max_output_tokens=None, + ) router.get_configured_mode.return_value = "image_generation" router.model_list = [team_dep] router.get_model_list.return_value = [team_dep] From 2fc520329ffcbf6ec98b86d8f1c21e0da314e337 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:26:28 -0700 Subject: [PATCH 053/157] fix(router): keep the budget push off the request callback path The provider budget push runs inside the request success callback, so awaiting the Redis pipeline there made every request wait for the round trip. Hand it back to a task whose failure is logged through the breaker aware logger, so an open breaker stays a debug line and a real Redis error is one error line instead of an unretrieved task traceback --- litellm/router_strategy/budget_limiter.py | 11 +++- .../test_budget_limiter_hotpath.py | 56 +++++++++++++++++++ 2 files changed, 65 insertions(+), 2 deletions(-) diff --git a/litellm/router_strategy/budget_limiter.py b/litellm/router_strategy/budget_limiter.py index acdc57d706d..3e094df7ac8 100644 --- a/litellm/router_strategy/budget_limiter.py +++ b/litellm/router_strategy/budget_limiter.py @@ -28,7 +28,7 @@ from typing import Any, Final import litellm from litellm._logging import verbose_router_logger from litellm.caching.caching import DualCache -from litellm.caching.redis_cache import RedisPipelineIncrementOperation, log_redis_failure +from litellm.caching.redis_cache import RedisCache, RedisPipelineIncrementOperation, log_redis_failure from litellm.integrations.custom_logger import CustomLogger, Span from litellm.litellm_core_utils.core_helpers import ( get_metadata_variable_name_from_kwargs, @@ -93,6 +93,13 @@ class _LiteLLMParamsDictView: return dict(self._params) +async def _push_increments_to_redis(redis_cache: RedisCache, queued: list[RedisPipelineIncrementOperation]) -> None: + try: + await redis_cache.async_increment_pipeline(increment_list=queued) + except Exception as e: + log_redis_failure(verbose_router_logger, logging.ERROR, "Error syncing in-memory cache with Redis", e) + + class RouterBudgetLimiting(CustomLogger): def __init__( self, @@ -540,7 +547,7 @@ class RouterBudgetLimiting(CustomLogger): queued: Final = self.redis_increment_operation_queue self.redis_increment_operation_queue = [] if queued: - await self.dual_cache.redis_cache.async_increment_pipeline(increment_list=queued) + asyncio.create_task(_push_increments_to_redis(self.dual_cache.redis_cache, queued)) except Exception as e: log_redis_failure(verbose_router_logger, logging.ERROR, "Error syncing in-memory cache with Redis", e) diff --git a/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py b/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py index 128b2dbd842..4cc8fe78811 100644 --- a/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py +++ b/tests/test_litellm/router_strategy/test_budget_limiter_hotpath.py @@ -340,3 +340,59 @@ async def test_sync_refused_by_the_open_circuit_breaker_is_quiet_and_leaks_no_ta unretrieved.assert_not_called() assert limiter.redis_increment_operation_queue == [] assert redis_cache.async_increment_pipeline.await_count == 1 + + +async def _limiter_with_redis(redis_cache: MagicMock) -> RouterBudgetLimiting: + limiter = RouterBudgetLimiting( + dual_cache=DualCache(redis_cache=redis_cache), + provider_budget_config={"openai": BudgetConfig(max_budget=1.0, budget_duration="1d")}, + ) + await asyncio.gather(*(task for task in asyncio.all_tasks() if task is not asyncio.current_task())) + limiter.redis_increment_operation_queue = [{"key": "provider_spend:openai:1d", "increment_value": 0.5, "ttl": 60}] + return limiter + + +@pytest.mark.asyncio +async def test_push_returns_before_redis_answers(disable_budget_sync): + """The push runs inside the request success callback, so it must hand the Redis round trip to a task instead of waiting on it.""" + redis_answered = asyncio.Event() + + async def wait_for_redis(**_: object) -> None: + await redis_answered.wait() + + redis_cache = MagicMock(spec=RedisCache) + redis_cache.async_increment_pipeline = AsyncMock(side_effect=wait_for_redis) + limiter = await _limiter_with_redis(redis_cache) + + await asyncio.wait_for(limiter._push_in_memory_increments_to_redis(), timeout=1) + await asyncio.sleep(0) + + assert not redis_answered.is_set() + assert redis_cache.async_increment_pipeline.await_count == 1 + assert limiter.redis_increment_operation_queue == [] + redis_answered.set() + await asyncio.gather(*(task for task in asyncio.all_tasks() if task is not asyncio.current_task())) + + +@pytest.mark.asyncio +async def test_push_task_failure_is_logged_once_and_not_leaked(disable_budget_sync, caplog): + """A real Redis failure on the background push must surface as one error line, never as an unretrieved task exception.""" + redis_cache = MagicMock(spec=RedisCache) + redis_cache.async_increment_pipeline = AsyncMock(side_effect=ConnectionError("Error 61 connecting to 127.0.0.1:6379")) + limiter = await _limiter_with_redis(redis_cache) + loop = asyncio.get_running_loop() + unretrieved = MagicMock() + loop.set_exception_handler(unretrieved) + + try: + with caplog.at_level(logging.ERROR): + await limiter._push_in_memory_increments_to_redis() + await asyncio.sleep(0) + gc.collect() + finally: + loop.set_exception_handler(None) + + assert [record.getMessage() for record in caplog.records] == [ + "Error syncing in-memory cache with Redis: Error 61 connecting to 127.0.0.1:6379" + ] + unretrieved.assert_not_called() From fc95d223670541d1cbd43f4ae76fd3d8e87ec51b Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:18:38 -0700 Subject: [PATCH 054/157] fix(mcp): accept VS Code OAuth registration callbacks --- .../mcp_server/gateway_dcr_flow.py | 8 +- .../mcp_server/test_gateway_dcr_flow.py | 74 +++++++++++++++++-- 2 files changed, 72 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/gateway_dcr_flow.py b/litellm/proxy/_experimental/mcp_server/gateway_dcr_flow.py index f4889008e94..f3fdd54b39d 100644 --- a/litellm/proxy/_experimental/mcp_server/gateway_dcr_flow.py +++ b/litellm/proxy/_experimental/mcp_server/gateway_dcr_flow.py @@ -122,13 +122,13 @@ _USED_CODE_CACHE_PREFIX: Final = "mcp_gateway_dcr_code_used:" _USED_FLOW_CACHE_PREFIX: Final = "mcp_gateway_dcr_flow_used:" _USED_REFRESH_CACHE_PREFIX: Final = "mcp_gateway_dcr_refresh_used:" -MAX_REDIRECT_URIS: Final = 3 +MAX_REDIRECT_URIS: Final = 4 MAX_REDIRECT_URI_LENGTH: Final = 256 MAX_CLIENT_ID_LENGTH: Final = 2048 """Registration bounds. They exist to bound the sealed client_id, which rides inside -every session-token claim set: 3 URIs of 256 bytes seal to roughly 1.2KB, comfortably -under this cap and under the session token's own 4KB ceiling. Claude Desktop and MCP -Inspector register one or two redirect URIs.""" +every session-token claim set. Four 256-character ASCII URIs seal to roughly 1.5KB; +the encoded client_id is checked against its own cap before registration succeeds. +VS Code registers four callbacks for its web and desktop environments.""" MAX_STATE_LENGTH: Final = 1024 """Bound on the client ``state`` sealed into the flow cookie and echoed on the auth-code diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_gateway_dcr_flow.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_gateway_dcr_flow.py index 8aa4cfc5619..7c80ee77cd7 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_gateway_dcr_flow.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_gateway_dcr_flow.py @@ -6,6 +6,7 @@ import re from base64 import urlsafe_b64encode from datetime import datetime, timedelta, timezone from http.cookies import SimpleCookie +from typing import Final from urllib.parse import parse_qs, urlparse import pytest @@ -18,6 +19,7 @@ from litellm.proxy._experimental.mcp_server.gateway_dcr_flow import ( GATEWAY_AUTH_CODE_PREFIX, GATEWAY_AUTH_CODE_TTL_SECONDS, MANUAL_DELIVERY_AUTH_CODE_TTL_SECONDS, + MAX_CLIENT_ID_LENGTH, ConsentTeam, MintedProxyCredential, _GatewayAuthCode, @@ -53,6 +55,13 @@ from litellm.proxy._experimental.mcp_server.outbound_credentials.session_token i MASTER_KEY = "sk-gateway-dcr-flow-tests" REDIRECT_URI = "https://claude.ai/api/mcp/auth_callback" +VSCODE_REDIRECT_URIS: Final = ( + "https://insiders.vscode.dev/redirect", + "https://vscode.dev/redirect", + "http://127.0.0.1/", + "http://127.0.0.1:33418/", +) +MAX_LENGTH_REDIRECT_URIS: Final = tuple(f"https://client.example/{index}/".ljust(256, "a") for index in range(4)) CODE_VERIFIER = "verifier-" + "v" * 43 CODE_CHALLENGE = urlsafe_b64encode(hashlib.sha256(CODE_VERIFIER.encode("ascii")).digest()).rstrip(b"=").decode("ascii") @@ -104,6 +113,58 @@ async def test_register_mints_stateless_public_client(): assert record.redirect_uris == (REDIRECT_URI,) +@pytest.mark.asyncio +@pytest.mark.parametrize("redirect_uris", [VSCODE_REDIRECT_URIS, MAX_LENGTH_REDIRECT_URIS]) +async def test_register_four_callbacks_preserves_metadata(redirect_uris: tuple[str, ...]) -> None: + response: Final = await register_aggregate_client( + request=_request(path="/register", method="POST"), + request_body={ + "client_name": "Visual Studio Code", + "client_uri": "https://code.visualstudio.com", + "grant_types": ["authorization_code", "refresh_token"], + "response_types": ["code"], + "redirect_uris": list(redirect_uris), + "token_endpoint_auth_method": "none", + "application_type": "native", + }, + ) + assert response.status_code == 201 + body: Final = json.loads(response.body) + assert body["redirect_uris"] == list(redirect_uris) + assert body["token_endpoint_auth_method"] == "none" + assert "client_secret" not in body + assert len(body["client_id"]) <= MAX_CLIENT_ID_LENGTH + record: Final = open_gateway_dcr_client(body["client_id"]) + assert record is not None + assert record.redirect_uris == redirect_uris + + +@pytest.mark.asyncio +async def test_register_rejects_five_valid_callbacks() -> None: + response: Final = await register_aggregate_client( + request=_request(path="/register", method="POST"), + request_body={"redirect_uris": [*VSCODE_REDIRECT_URIS, "http://127.0.0.1:33419/"]}, + ) + assert response.status_code == 400 + assert json.loads(response.body) == { + "error": "invalid_redirect_uri", + "error_description": "redirect_uris must be a list of 1 to 4 URIs", + } + + +@pytest.mark.asyncio +async def test_register_four_callbacks_preserves_encoded_size_guard() -> None: + response: Final = await register_aggregate_client( + request=_request(path="/register", method="POST"), + request_body={"redirect_uris": [f"https://client.example/{index}/".ljust(256, "é") for index in range(4)]}, + ) + assert response.status_code == 400 + assert json.loads(response.body) == { + "error": "invalid_client_metadata", + "error_description": "registered metadata is too large", + } + + @pytest.mark.asyncio async def test_register_allows_loopback_http_for_dev_clients(): body = await _register(["http://localhost:6274/oauth/callback"]) @@ -162,7 +223,6 @@ async def test_register_rejects_userinfo_spoofed_origin(): ["https://claude.ai/cb#fragment"], ["ftp://claude.ai/cb"], ["https://a.example.com/" + "p" * 300], - ["https://a.example.com/1", "https://a.example.com/2", "https://a.example.com/3", "https://a.example.com/4"], [12345], ], ) @@ -248,12 +308,14 @@ def _flow_cookie_from(response) -> tuple: @pytest.mark.asyncio -async def test_full_walk_register_authorize_complete_token_and_replay(): +@pytest.mark.parametrize("redirect_uris", [(REDIRECT_URI,), VSCODE_REDIRECT_URIS, MAX_LENGTH_REDIRECT_URIS]) +async def test_full_walk_register_authorize_complete_token_and_replay(redirect_uris: tuple[str, ...]): """The whole front door on one deterministic walk: register -> authorize -> complete -> token, then the security edges on the same artifacts (user mismatch, PKCE mismatch, single-use replay, refresh rotation, cross-client refresh).""" - client_id = (await _register([REDIRECT_URI]))["client_id"] - authorize_response = _authorize(client_id, session_user_id="u1") + redirect_uri: Final = redirect_uris[-1] + client_id = (await _register(list(redirect_uris)))["client_id"] + authorize_response = _authorize(client_id, session_user_id="u1", redirect_uri=redirect_uri) handle, cookies = _flow_cookie_from(authorize_response) denied = await complete_connect_flow( @@ -280,7 +342,7 @@ async def test_full_walk_register_authorize_complete_token_and_replay(): ) assert completed.status_code == 303 redirect = urlparse(completed.headers["location"]) - assert f"{redirect.scheme}://{redirect.netloc}{redirect.path}" == REDIRECT_URI + assert f"{redirect.scheme}://{redirect.netloc}{redirect.path}" == redirect_uri params = parse_qs(redirect.query) assert params["state"] == ["client-state-123"] code = params["code"][0] @@ -293,7 +355,7 @@ async def test_full_walk_register_authorize_complete_token_and_replay(): "request": _request("/token", method="POST"), "grant_type": "authorization_code", "code": code, - "redirect_uri": REDIRECT_URI, + "redirect_uri": redirect_uri, "client_id": client_id, "code_verifier": CODE_VERIFIER, "refresh_token": None, From 86c5cd736b7ac60c661a7a55e0656bfa0aea180b Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:29:19 -0700 Subject: [PATCH 055/157] chore: sync API descriptions from the updated base branch --- ui/litellm-dashboard/src/lib/http/schema.d.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 6826cded6f5..29435c31aee 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -35064,7 +35064,7 @@ export interface components { classification_prompt?: string | null; /** * Classifier Context Budget Chars - * @description Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the caller's system prompt sit outside this budget and are always sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. + * @description Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. * @default 8000 */ classifier_context_budget_chars: number; @@ -35081,7 +35081,7 @@ export interface components { classifier_context_per_turn_chars?: number | null; /** * Classifier Context Window Size - * @description Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call already carries the current user ask and the caller's system prompt in full. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. + * @description Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. * @default 3 */ classifier_context_window_size: number; From 22b4656cb50586042ecf13ae4e7d8973a713cc19 Mon Sep 17 00:00:00 2001 From: Hayden Moulds Date: Fri, 11 Sep 2026 12:40:02 +1000 Subject: [PATCH 056/157] chore: synchronize generated proxy API types --- litellm/proxy/proxy_server.py | 2 +- ui/litellm-dashboard/src/lib/http/schema.d.ts | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index c3c45855c5f..8b12b75d7fd 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -10834,7 +10834,7 @@ async def model_info( fallback_type=None, llm_router=llm_router, ) - return {**response, "id": internal_to_public.get(resolved_model_id, model_id)} + return {**response, "id": internal_to_public.get(resolved_model_id, model_id)} # mutable-ok: response id differs def _blocked_response_usage(original_response: object | None) -> "litellm.Usage": diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 6826cded6f5..29435c31aee 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -35064,7 +35064,7 @@ export interface components { classification_prompt?: string | null; /** * Classifier Context Budget Chars - * @description Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the caller's system prompt sit outside this budget and are always sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. + * @description Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. * @default 8000 */ classifier_context_budget_chars: number; @@ -35081,7 +35081,7 @@ export interface components { classifier_context_per_turn_chars?: number | null; /** * Classifier Context Window Size - * @description Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call already carries the current user ask and the caller's system prompt in full. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. + * @description Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. * @default 3 */ classifier_context_window_size: number; From 25ed0abfc955fe886fe214af580153b6976dde8b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:42:27 -0700 Subject: [PATCH 057/157] chore(ui): regenerate schema.d.ts after merging the base --- ui/litellm-dashboard/src/lib/http/schema.d.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 6826cded6f5..29435c31aee 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -35064,7 +35064,7 @@ export interface components { classification_prompt?: string | null; /** * Classifier Context Budget Chars - * @description Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the caller's system prompt sit outside this budget and are always sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. + * @description Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. * @default 8000 */ classifier_context_budget_chars: number; @@ -35081,7 +35081,7 @@ export interface components { classifier_context_per_turn_chars?: number | null; /** * Classifier Context Window Size - * @description Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call already carries the current user ask and the caller's system prompt in full. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. + * @description Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. * @default 3 */ classifier_context_window_size: number; From 576c1bc5d6f5421956f638a6e54a9e534031bbf6 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:45:44 -0700 Subject: [PATCH 058/157] fix(mcp): bound and coalesce OpenAPI health probes --- .../mcp_server/mcp_server_manager.py | 35 +++++- .../mcp_server/openapi_to_mcp_generator.py | 58 +++++++++- .../mcp_server/test_mcp_server_manager.py | 76 +++++++++++++ .../test_openapi_to_mcp_generator.py | 102 ++++++++++++++++++ 4 files changed, 264 insertions(+), 7 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index 5a9000d2f23..4177e1b16dc 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -25,6 +25,7 @@ from collections.abc import ( ) from contextlib import asynccontextmanager from dataclasses import dataclass, replace +from functools import lru_cache from types import MappingProxyType from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypedDict, cast from urllib.parse import ParseResult, urlparse @@ -887,23 +888,46 @@ async def _openapi_spec_health( spec_path: str, *, timeout: float ) -> tuple[Literal["healthy", "unhealthy", "unknown"], str | None]: """Check specification availability, not upstream operations or user credentials.""" - from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import load_openapi_spec_async + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import ( + OpenAPISpecProbeLimitError, + load_openapi_spec_async, + ) if not spec_path.startswith(("http://", "https://")): return "unknown", "OpenAPI servers have no protocol-level health probe" try: - await asyncio.wait_for(load_openapi_spec_async(spec_path), timeout=timeout) + await asyncio.wait_for(load_openapi_spec_async(spec_path, max_bytes=10 * 1024 * 1024), timeout=timeout) except asyncio.TimeoutError: return "unhealthy", f"OpenAPI specification check timed out after {timeout} seconds" except asyncio.CancelledError: return "unknown", "OpenAPI specification check was cancelled" except HTTPStatusError as exc: return "unhealthy", f"OpenAPI specification request failed (HTTP {exc.response.status_code})" + except OpenAPISpecProbeLimitError as exc: + return "unknown", str(exc) except (httpx.RequestError, ValueError, OSError) as exc: return "unhealthy", f"OpenAPI specification could not be loaded ({type(exc).__name__})" return "healthy", None +class _OpenAPIHealthProbe: + def __init__(self, spec_path: str, clock: Callable[[], float] = time.monotonic) -> None: + self.spec_path = spec_path + self.clock = clock + self.lock = asyncio.Lock() + self.checked_at = float("-inf") + self.result: tuple[Literal["healthy", "unhealthy", "unknown"], str | None, datetime.datetime] | None = None + + async def check(self) -> tuple[Literal["healthy", "unhealthy", "unknown"], str | None, datetime.datetime]: + async with self.lock: + if self.result is not None and self.clock() - self.checked_at < 30.0: + return self.result + status, error = await _openapi_spec_health(self.spec_path, timeout=MCP_HEALTH_CHECK_TIMEOUT) + self.result = (status, error, datetime.datetime.now(datetime.timezone.utc)) + self.checked_at = self.clock() + return self.result + + def _discovery_failure_leaves_needs_unresolved( *, needs_authorization_url: bool, @@ -1770,6 +1794,7 @@ class MCPServerManager: token_exchanger=build_token_exchanger(), ) self.registry: dict[str, MCPServer] = {} + self._openapi_health_probes: Callable[[str], _OpenAPIHealthProbe] = lru_cache(maxsize=128)(_OpenAPIHealthProbe) self.config_mcp_servers: dict[str, MCPServer] = {} """ eg. @@ -6686,7 +6711,7 @@ class MCPServerManager: Returns: Dict containing health check results """ - from datetime import datetime, timezone + from datetime import datetime server: Final = self.get_mcp_server_by_id(server_id) if not server: @@ -6701,13 +6726,13 @@ class MCPServerManager: ) if server.spec_path: - spec_status, spec_error = await _openapi_spec_health(server.spec_path, timeout=MCP_HEALTH_CHECK_TIMEOUT) + spec_status, spec_error, spec_checked_at = await self._openapi_health_probes(server.spec_path).check() return self._build_mcp_server_table(server).model_copy( update=MappingProxyType( { "status": spec_status, "health_check_error": spec_error, - "last_health_check": datetime.now(timezone.utc), + "last_health_check": spec_checked_at, } ) ) diff --git a/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py b/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py index 16f58ef5b76..ba72f7bea4d 100644 --- a/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py +++ b/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py @@ -8,7 +8,10 @@ import json import os import re from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from io import BytesIO from pathlib import PurePosixPath +from types import MappingProxyType from typing import Any, Final, TypedDict from urllib.parse import quote @@ -163,10 +166,61 @@ def load_openapi_spec(filepath: str) -> dict[str, Any]: return asyncio.run(load_openapi_spec_async(filepath)) -async def load_openapi_spec_async(filepath: str) -> dict[str, Any]: +class OpenAPISpecProbeLimitError(ValueError): + pass + + +@dataclass(frozen=True) +class _BoundedOpenAPIFetcher: + client: httpx.AsyncClient + max_bytes: int + + async def get( + self, + url: str, + *, + headers: dict[str, str] | None = None, + follow_redirects: bool = False, + redirects_remaining: int = 10, + ) -> httpx.Response: + async with self.client.stream( + "GET", + url, + headers=MappingProxyType({**(headers or MappingProxyType({})), "Accept-Encoding": "identity"}), + follow_redirects=False, + ) as response: + if response.is_redirect and follow_redirects: + if redirects_remaining == 0: + raise ValueError("Too many specification redirects") + await response.aclose() + return await self.get( + str(response.url.join(response.headers["location"])), + headers=headers, + follow_redirects=True, + redirects_remaining=redirects_remaining - 1, + ) + if response.is_redirect or response.is_error: + return httpx.Response(response.status_code, headers=response.headers, request=response.request) + if response.headers.get("content-encoding", "identity").lower() != "identity": + raise OpenAPISpecProbeLimitError("OpenAPI specification probe requires an uncompressed response") + if int(response.headers.get("content-length", "0")) > self.max_bytes: + raise OpenAPISpecProbeLimitError("OpenAPI specification exceeds the health-check size limit") + with BytesIO() as body: + async for chunk in response.aiter_bytes(chunk_size=65536): + if body.tell() + len(chunk) > self.max_bytes: + raise OpenAPISpecProbeLimitError("OpenAPI specification exceeds the health-check size limit") + body.write(chunk) + return httpx.Response( + response.status_code, headers=response.headers, content=body.getvalue(), request=response.request + ) + + +async def load_openapi_spec_async(filepath: str, *, max_bytes: int | None = None) -> dict[str, Any]: if filepath.startswith("http://") or filepath.startswith("https://"): client: Final = get_async_httpx_client(llm_provider=httpxSpecialProvider.MCP) - r: Final[httpx.Response] = await async_safe_get(client, filepath) + r: Final[httpx.Response] = await async_safe_get( + client if max_bytes is None else _BoundedOpenAPIFetcher(client.client, max_bytes), filepath + ) r.raise_for_status() return r.json() diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py index e3172d4939d..7eff19f3ed9 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py @@ -12786,3 +12786,79 @@ async def test_debug_reports_legacy_signing_and_non_http_transport(transport: Li assert "Credential=AKIDEXAMPLE/" in request.headers["Authorization"] finally: request_ctx.reset(token) + +@pytest.mark.asyncio +async def test_openapi_health_coalesces_concurrent_checks_and_reuses_results(respx_mock, monkeypatch): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + manager = MCPServerManager() + server = MCPServer( + server_id="coalesced", + name="coalesced", + transport=MCPTransport.http, + spec_path="https://93.184.216.34/coalesced.json", + auth_type=MCPAuth.none, + ) + manager.registry = {server.server_id: server} + started = asyncio.Event() + release = asyncio.Event() + + async def serve(request): + started.set() + await release.wait() + return httpx.Response(200, json={"paths": {}}) + + route = respx_mock.get(server.spec_path).mock(side_effect=serve) + tasks = [asyncio.create_task(manager.health_check_server(server.server_id)) for _ in range(4)] + await asyncio.wait_for(started.wait(), timeout=1) + release.set() + results = await asyncio.gather(*tasks) + cached = await manager.health_check_server(server.server_id) + assert [result.status for result in results] == ["healthy"] * 4 + assert cached.status == "healthy" + assert {result.last_health_check for result in [*results, cached]} == {results[0].last_health_check} + assert route.call_count == 1 + + +@pytest.mark.asyncio +async def test_openapi_health_cache_expires_at_thirty_seconds(respx_mock, monkeypatch): + from litellm.proxy._experimental.mcp_server.mcp_server_manager import _OpenAPIHealthProbe + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + clock = iter([0.0, 29.0, 30.0, 30.0]) + probe = _OpenAPIHealthProbe("https://93.184.216.34/expiry.json", clock=clock.__next__) + route = respx_mock.get(probe.spec_path).mock( + side_effect=[ + httpx.Response(200, json={"paths": {}}), + httpx.Response(503), + ] + ) + first = await probe.check() + assert first[0] == "healthy" + assert await probe.check() == first + refreshed = await probe.check() + assert refreshed[0] == "unhealthy" + assert refreshed[1] == "OpenAPI specification request failed (HTTP 503)" + assert refreshed[2] >= first[2] + assert route.call_count == 2 + + +@pytest.mark.asyncio +async def test_openapi_health_reports_size_limit_as_unknown_and_caches_failure(respx_mock, monkeypatch): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + manager = MCPServerManager() + server = MCPServer( + server_id="oversized", + name="oversized", + transport=MCPTransport.http, + spec_path="https://93.184.216.34/large.json", + auth_type=MCPAuth.none, + ) + manager.registry = {server.server_id: server} + route = respx_mock.get(server.spec_path).respond(200, headers={"content-length": str(12 * 1024 * 1024)}) + result = await manager.health_check_server(server.server_id) + cached = await manager.health_check_server(server.server_id) + assert result.status == "unknown" + assert result.health_check_error == "OpenAPI specification exceeds the health-check size limit" + assert cached.health_check_error == result.health_check_error + assert cached.last_health_check == result.last_health_check + assert route.call_count == 1 diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py index e59616e53c1..af0b188d22d 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py @@ -1378,3 +1378,105 @@ class TestUpstreamStatusIsClassified: assert exc.value.status_code == status_code assert secret_body not in str(exc.value) assert str(exc.value) == f"upstream returned HTTP {status_code}" + +class TestBoundedOpenAPISpecLoading: + @pytest.mark.asyncio + @pytest.mark.parametrize("max_bytes", [12, 13]) + async def test_exact_size_and_smaller_specs_load(self, respx_mock, monkeypatch, max_bytes): + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import load_openapi_spec_async + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + route = respx_mock.get("https://93.184.216.34/spec.json").respond(200, content=b'{"paths":{}}') + assert await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=max_bytes) == {"paths": {}} + assert route.calls[0].request.headers["accept-encoding"] == "identity" + + @pytest.mark.asyncio + @pytest.mark.parametrize("headers", [{"content-length": "1000000"}, {"content-encoding": "gzip"}]) + async def test_unsafe_response_headers_reject_before_reading(self, respx_mock, monkeypatch, headers): + import httpx + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import ( + OpenAPISpecProbeLimitError, + load_openapi_spec_async, + ) + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + closed = [] + + class UnreadableStream(httpx.AsyncByteStream): + async def __aiter__(self): + pytest.fail("Oversized or compressed response must not be consumed") + yield b"" + + async def aclose(self): + closed.append(True) + + respx_mock.get("https://93.184.216.34/spec.json").respond(200, headers=headers, stream=UnreadableStream()) + with pytest.raises(OpenAPISpecProbeLimitError): + await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=12) + assert closed == [True] + + @pytest.mark.asyncio + async def test_chunked_response_is_bounded_and_closed(self, respx_mock, monkeypatch): + import httpx + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import ( + OpenAPISpecProbeLimitError, + load_openapi_spec_async, + ) + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + consumed = [] + closed = [] + + class ChunkedStream(httpx.AsyncByteStream): + async def __aiter__(self): + for index in range(10): + consumed.append(index) + yield b"x" * 65536 + + async def aclose(self): + closed.append(True) + + respx_mock.get("https://93.184.216.34/spec.json").respond(200, stream=ChunkedStream()) + with pytest.raises(OpenAPISpecProbeLimitError, match="size limit"): + await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=65536) + assert consumed == [0, 1] + assert closed == [True] + + @pytest.mark.asyncio + @pytest.mark.parametrize("target", ["https://93.184.216.35/final.json", "http://127.0.0.1/private.json"]) + async def test_bounded_spec_redirects_preserve_ssrf_protection(self, respx_mock, monkeypatch, target): + from litellm.litellm_core_utils.url_utils import SSRFError + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import load_openapi_spec_async + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + respx_mock.get("https://93.184.216.34/spec.json").respond(302, headers={"location": target}) + destination = respx_mock.get(target).respond(200, json={"paths": {}}) + if "127.0.0.1" in target: + with pytest.raises(SSRFError): + await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=100) + assert not destination.called + else: + assert await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=100) == {"paths": {}} + assert destination.call_count == 1 + + @pytest.mark.asyncio + @pytest.mark.parametrize("loop", [False, True]) + async def test_redirects_are_bounded_when_validation_is_disabled(self, respx_mock, loop): + import httpx + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import _BoundedOpenAPIFetcher + + source = respx_mock.get("https://example.com/spec.json").respond( + 302, headers={"location": "/spec.json" if loop else "/final.json"} + ) + destination = respx_mock.get("https://example.com/final.json").respond(200, json={"paths": {}}) + async with httpx.AsyncClient() as client: + fetcher = _BoundedOpenAPIFetcher(client, 100) + if loop: + with pytest.raises(ValueError, match="Too many specification redirects"): + await fetcher.get("https://example.com/spec.json", follow_redirects=True) + assert source.call_count == 11 + assert not destination.called + else: + response = await fetcher.get("https://example.com/spec.json", follow_redirects=True) + assert response.json() == {"paths": {}} + assert destination.call_count == 1 From da1dfcdb2404d4a24573dd1ffdbf560e683ceba1 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:59:07 -0700 Subject: [PATCH 059/157] refactor(mcp): reuse the shared HTTP handler for bounded probes --- litellm/llms/custom_httpx/http_handler.py | 67 +++++++++++++++++++ .../mcp_server/mcp_server_manager.py | 10 ++- .../mcp_server/openapi_to_mcp_generator.py | 58 ++-------------- .../llms/custom_httpx/test_http_handler.py | 63 +++++++++++++++++ .../mcp_server/test_mcp_server_manager.py | 2 +- .../test_openapi_to_mcp_generator.py | 30 ++------- 6 files changed, 143 insertions(+), 87 deletions(-) diff --git a/litellm/llms/custom_httpx/http_handler.py b/litellm/llms/custom_httpx/http_handler.py index 12e515e19a8..f4883b57fbc 100644 --- a/litellm/llms/custom_httpx/http_handler.py +++ b/litellm/llms/custom_httpx/http_handler.py @@ -9,6 +9,7 @@ import threading import time from collections.abc import AsyncIterable, Callable, Iterable, Mapping from http.cookiejar import CookieJar, DefaultCookiePolicy +from io import BytesIO from types import MappingProxyType from typing import TYPE_CHECKING, Any, ClassVar, Final, NoReturn, Optional, TypeAlias, TypedDict, TypeVar @@ -505,6 +506,10 @@ async def _raise_masked_async_error(e: httpx.HTTPStatusError, stream: bool) -> N raise MaskedHTTPStatusError(e, message=_text, text=_text) from None +class HTTPResponseLimitError(ValueError): + pass + + class MaskedHTTPStatusError(httpx.HTTPStatusError): def __init__(self, original_error, message: str | None = None, text: str | None = None): # Create a new error with the masked URL @@ -654,6 +659,7 @@ class AsyncHTTPHandler: headers: dict | None = None, follow_redirects: bool | None = None, timeout: float | httpx.Timeout | None = None, + max_response_bytes: int | None = None, ): # Set follow_redirects to UseClientDefault if None _follow_redirects: Final = follow_redirects if follow_redirects is not None else USE_CLIENT_DEFAULT @@ -661,6 +667,16 @@ class AsyncHTTPHandler: params = params or {} params.update(HTTPHandler.extract_query_params(url)) + if max_response_bytes is not None: + return await self._get_with_response_limit( + url, + params=httpx.QueryParams(params), + headers=httpx.Headers(headers), + max_bytes=max_response_bytes, + follow_redirects=self.client.follow_redirects if follow_redirects is None else follow_redirects, + timeout=self.client.timeout if timeout is None else httpx.Timeout(timeout), + ) + response: Final = await self.client.get( url, params=params, @@ -670,6 +686,57 @@ class AsyncHTTPHandler: ) return response + async def _get_with_response_limit( + self, + url: str, + *, + params: httpx.QueryParams, + headers: httpx.Headers, + timeout: httpx.Timeout, + max_bytes: int, + follow_redirects: bool, + ) -> httpx.Response: + request: Final = self.client.build_request( + "GET", + url, + headers=MappingProxyType({**headers, "accept-encoding": "identity"}), + params=params, + timeout=timeout, + ) + response: Final = await self.client.send(request, stream=True, follow_redirects=False) + return await self._read_with_response_limit(response, max_bytes=max_bytes, follow_redirects=follow_redirects) + + async def _read_with_response_limit( + self, response: httpx.Response, *, max_bytes: int, follow_redirects: bool, redirects_remaining: int = 10 + ) -> httpx.Response: + try: + if response.next_request is not None and follow_redirects: + if redirects_remaining == 0: + raise ValueError("Too many redirects") + await response.aclose() + following: Final = await self.client.send( + response.next_request, auth=None, stream=True, follow_redirects=False + ) + return await self._read_with_response_limit( + following, max_bytes=max_bytes, follow_redirects=True, redirects_remaining=redirects_remaining - 1 + ) + if response.is_redirect or response.is_error: + return httpx.Response(response.status_code, headers=response.headers, request=response.request) + if response.headers.get("content-encoding", "identity").lower() != "identity": + raise HTTPResponseLimitError("Response size limits require an uncompressed response") + if int(response.headers.get("content-length", "0")) > max_bytes: + raise HTTPResponseLimitError("Response exceeds the configured size limit") + with BytesIO() as body: + async for chunk in response.aiter_bytes(chunk_size=65536): + if body.tell() + len(chunk) > max_bytes: + raise HTTPResponseLimitError("Response exceeds the configured size limit") + body.write(chunk) + return httpx.Response( + response.status_code, headers=response.headers, content=body.getvalue(), request=response.request + ) + finally: + await response.aclose() + @track_llm_api_timing() async def post( self, diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index 4177e1b16dc..d0b0c83a23c 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -888,10 +888,8 @@ async def _openapi_spec_health( spec_path: str, *, timeout: float ) -> tuple[Literal["healthy", "unhealthy", "unknown"], str | None]: """Check specification availability, not upstream operations or user credentials.""" - from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import ( - OpenAPISpecProbeLimitError, - load_openapi_spec_async, - ) + from litellm.llms.custom_httpx.http_handler import HTTPResponseLimitError + from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import load_openapi_spec_async if not spec_path.startswith(("http://", "https://")): return "unknown", "OpenAPI servers have no protocol-level health probe" @@ -903,8 +901,8 @@ async def _openapi_spec_health( return "unknown", "OpenAPI specification check was cancelled" except HTTPStatusError as exc: return "unhealthy", f"OpenAPI specification request failed (HTTP {exc.response.status_code})" - except OpenAPISpecProbeLimitError as exc: - return "unknown", str(exc) + except HTTPResponseLimitError as exc: + return "unknown", f"OpenAPI specification probe refused: {exc}" except (httpx.RequestError, ValueError, OSError) as exc: return "unhealthy", f"OpenAPI specification could not be loaded ({type(exc).__name__})" return "healthy", None diff --git a/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py b/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py index ba72f7bea4d..d115eb8b3c1 100644 --- a/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py +++ b/litellm/proxy/_experimental/mcp_server/openapi_to_mcp_generator.py @@ -8,10 +8,7 @@ import json import os import re from collections.abc import Mapping, Sequence -from dataclasses import dataclass -from io import BytesIO from pathlib import PurePosixPath -from types import MappingProxyType from typing import Any, Final, TypedDict from urllib.parse import quote @@ -166,60 +163,13 @@ def load_openapi_spec(filepath: str) -> dict[str, Any]: return asyncio.run(load_openapi_spec_async(filepath)) -class OpenAPISpecProbeLimitError(ValueError): - pass - - -@dataclass(frozen=True) -class _BoundedOpenAPIFetcher: - client: httpx.AsyncClient - max_bytes: int - - async def get( - self, - url: str, - *, - headers: dict[str, str] | None = None, - follow_redirects: bool = False, - redirects_remaining: int = 10, - ) -> httpx.Response: - async with self.client.stream( - "GET", - url, - headers=MappingProxyType({**(headers or MappingProxyType({})), "Accept-Encoding": "identity"}), - follow_redirects=False, - ) as response: - if response.is_redirect and follow_redirects: - if redirects_remaining == 0: - raise ValueError("Too many specification redirects") - await response.aclose() - return await self.get( - str(response.url.join(response.headers["location"])), - headers=headers, - follow_redirects=True, - redirects_remaining=redirects_remaining - 1, - ) - if response.is_redirect or response.is_error: - return httpx.Response(response.status_code, headers=response.headers, request=response.request) - if response.headers.get("content-encoding", "identity").lower() != "identity": - raise OpenAPISpecProbeLimitError("OpenAPI specification probe requires an uncompressed response") - if int(response.headers.get("content-length", "0")) > self.max_bytes: - raise OpenAPISpecProbeLimitError("OpenAPI specification exceeds the health-check size limit") - with BytesIO() as body: - async for chunk in response.aiter_bytes(chunk_size=65536): - if body.tell() + len(chunk) > self.max_bytes: - raise OpenAPISpecProbeLimitError("OpenAPI specification exceeds the health-check size limit") - body.write(chunk) - return httpx.Response( - response.status_code, headers=response.headers, content=body.getvalue(), request=response.request - ) - - async def load_openapi_spec_async(filepath: str, *, max_bytes: int | None = None) -> dict[str, Any]: if filepath.startswith("http://") or filepath.startswith("https://"): client: Final = get_async_httpx_client(llm_provider=httpxSpecialProvider.MCP) - r: Final[httpx.Response] = await async_safe_get( - client if max_bytes is None else _BoundedOpenAPIFetcher(client.client, max_bytes), filepath + r: Final[httpx.Response] = ( + await async_safe_get(client, filepath) + if max_bytes is None + else await async_safe_get(client, filepath, max_response_bytes=max_bytes) ) r.raise_for_status() return r.json() diff --git a/tests/test_litellm/llms/custom_httpx/test_http_handler.py b/tests/test_litellm/llms/custom_httpx/test_http_handler.py index 3d4ba264c1c..f8868cfaf83 100644 --- a/tests/test_litellm/llms/custom_httpx/test_http_handler.py +++ b/tests/test_litellm/llms/custom_httpx/test_http_handler.py @@ -1612,3 +1612,66 @@ async def test_a_retried_put_stays_a_put_and_still_refuses_redirects(): assert attempts == [("PUT", "/first"), ("PUT", "/first")] finally: await handler.client.aclose() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("target", ["https://example.com/final.json?next=1", "https://other.example/final.json?next=1"]) +async def test_bounded_get_preserves_sdk_redirect_auth_and_query_handling(respx_mock, monkeypatch, target): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + respx_mock.get("https://example.com/spec.json?original=1").respond(302, headers={"location": target}) + destination = respx_mock.get(target).respond(200, json={"paths": {}}) + handler = AsyncHTTPHandler() + try: + response = await handler.get( + "https://example.com/spec.json?original=1", max_response_bytes=100, follow_redirects=True, + headers={"Authorization": "Bearer sentinel", "Accept-Encoding": "gzip"}, timeout=2.0, + ) + finally: + await handler.close() + assert response.json() == {"paths": {}} + request = destination.calls[0].request + assert request.headers.get("authorization") == (None if "other.example" in target else "Bearer sentinel") + assert request.headers["accept-encoding"] == "identity" + assert str(request.url) == target + assert request.extensions["timeout"]["read"] == 2.0 + + +@pytest.mark.asyncio +async def test_bounded_get_stops_redirect_loops(respx_mock, monkeypatch): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + route = respx_mock.get("https://example.com/spec.json").respond(302, headers={"location": "/spec.json"}) + handler = AsyncHTTPHandler() + try: + with pytest.raises(ValueError, match="Too many redirects"): + await handler.get("https://example.com/spec.json", max_response_bytes=100, follow_redirects=True) + finally: + await handler.close() + assert route.call_count == 11 + + +@pytest.mark.asyncio +async def test_bounded_get_closes_stream_on_cancellation(respx_mock, monkeypatch): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + started = asyncio.Event() + closed = asyncio.Event() + + class SlowStream(httpx.AsyncByteStream): + async def __aiter__(self): + yield b"x" + started.set() + await asyncio.Event().wait() + + async def aclose(self): + closed.set() + + respx_mock.get("https://example.com/slow.json").respond(200, stream=SlowStream()) + handler = AsyncHTTPHandler() + try: + task = asyncio.create_task(handler.get("https://example.com/slow.json", max_response_bytes=100)) + await asyncio.wait_for(started.wait(), timeout=1) + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + finally: + await handler.close() + assert closed.is_set() diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py index 7eff19f3ed9..e3dd92485fc 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py @@ -12858,7 +12858,7 @@ async def test_openapi_health_reports_size_limit_as_unknown_and_caches_failure(r result = await manager.health_check_server(server.server_id) cached = await manager.health_check_server(server.server_id) assert result.status == "unknown" - assert result.health_check_error == "OpenAPI specification exceeds the health-check size limit" + assert result.health_check_error == "OpenAPI specification probe refused: Response exceeds the configured size limit" assert cached.health_check_error == result.health_check_error assert cached.last_health_check == result.last_health_check assert route.call_count == 1 diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py index af0b188d22d..5fa202224e3 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_to_mcp_generator.py @@ -1394,8 +1394,8 @@ class TestBoundedOpenAPISpecLoading: @pytest.mark.parametrize("headers", [{"content-length": "1000000"}, {"content-encoding": "gzip"}]) async def test_unsafe_response_headers_reject_before_reading(self, respx_mock, monkeypatch, headers): import httpx + from litellm.llms.custom_httpx.http_handler import HTTPResponseLimitError from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import ( - OpenAPISpecProbeLimitError, load_openapi_spec_async, ) @@ -1411,15 +1411,15 @@ class TestBoundedOpenAPISpecLoading: closed.append(True) respx_mock.get("https://93.184.216.34/spec.json").respond(200, headers=headers, stream=UnreadableStream()) - with pytest.raises(OpenAPISpecProbeLimitError): + with pytest.raises(HTTPResponseLimitError): await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=12) assert closed == [True] @pytest.mark.asyncio async def test_chunked_response_is_bounded_and_closed(self, respx_mock, monkeypatch): import httpx + from litellm.llms.custom_httpx.http_handler import HTTPResponseLimitError from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import ( - OpenAPISpecProbeLimitError, load_openapi_spec_async, ) @@ -1437,7 +1437,7 @@ class TestBoundedOpenAPISpecLoading: closed.append(True) respx_mock.get("https://93.184.216.34/spec.json").respond(200, stream=ChunkedStream()) - with pytest.raises(OpenAPISpecProbeLimitError, match="size limit"): + with pytest.raises(HTTPResponseLimitError, match="size limit"): await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=65536) assert consumed == [0, 1] assert closed == [True] @@ -1458,25 +1458,3 @@ class TestBoundedOpenAPISpecLoading: else: assert await load_openapi_spec_async("https://93.184.216.34/spec.json", max_bytes=100) == {"paths": {}} assert destination.call_count == 1 - - @pytest.mark.asyncio - @pytest.mark.parametrize("loop", [False, True]) - async def test_redirects_are_bounded_when_validation_is_disabled(self, respx_mock, loop): - import httpx - from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import _BoundedOpenAPIFetcher - - source = respx_mock.get("https://example.com/spec.json").respond( - 302, headers={"location": "/spec.json" if loop else "/final.json"} - ) - destination = respx_mock.get("https://example.com/final.json").respond(200, json={"paths": {}}) - async with httpx.AsyncClient() as client: - fetcher = _BoundedOpenAPIFetcher(client, 100) - if loop: - with pytest.raises(ValueError, match="Too many specification redirects"): - await fetcher.get("https://example.com/spec.json", follow_redirects=True) - assert source.call_count == 11 - assert not destination.called - else: - response = await fetcher.get("https://example.com/spec.json", follow_redirects=True) - assert response.json() == {"paths": {}} - assert destination.call_count == 1 From ca03c889c9b72ec47a5d4f604626ac0b7cfdfadd Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 20:26:53 -0700 Subject: [PATCH 060/157] fix(mcp): avoid caching cancelled OpenAPI health probes --- .../mcp_server/mcp_server_manager.py | 11 ++-- .../mcp_server/test_mcp_server_manager.py | 51 ++++++++++++++++--- 2 files changed, 53 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index d0b0c83a23c..17d3098bf4c 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -897,8 +897,6 @@ async def _openapi_spec_health( await asyncio.wait_for(load_openapi_spec_async(spec_path, max_bytes=10 * 1024 * 1024), timeout=timeout) except asyncio.TimeoutError: return "unhealthy", f"OpenAPI specification check timed out after {timeout} seconds" - except asyncio.CancelledError: - return "unknown", "OpenAPI specification check was cancelled" except HTTPStatusError as exc: return "unhealthy", f"OpenAPI specification request failed (HTTP {exc.response.status_code})" except HTTPResponseLimitError as exc: @@ -920,7 +918,14 @@ class _OpenAPIHealthProbe: async with self.lock: if self.result is not None and self.clock() - self.checked_at < 30.0: return self.result - status, error = await _openapi_spec_health(self.spec_path, timeout=MCP_HEALTH_CHECK_TIMEOUT) + try: + status, error = await _openapi_spec_health(self.spec_path, timeout=MCP_HEALTH_CHECK_TIMEOUT) + except asyncio.CancelledError: + return ( + "unknown", + "OpenAPI specification check was cancelled", + datetime.datetime.now(datetime.timezone.utc), + ) self.result = (status, error, datetime.datetime.now(datetime.timezone.utc)) self.checked_at = self.clock() return self.result diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py index e3dd92485fc..a6e43686e2c 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py @@ -4575,12 +4575,12 @@ class TestMCPServerManager: await asyncio.wait_for(started.wait(), timeout=1) if cancel: task.cancel() - status, error = await task - assert status == ("unknown" if cancel else "unhealthy") - assert error == ( - "OpenAPI specification check was cancelled" - if cancel else "OpenAPI specification check timed out after 0.1 seconds" - ) + with pytest.raises(asyncio.CancelledError): + await task + else: + status, error = await task + assert status == "unhealthy" + assert error == "OpenAPI specification check timed out after 0.1 seconds" assert cancelled.is_set() @pytest.mark.asyncio @@ -12862,3 +12862,42 @@ async def test_openapi_health_reports_size_limit_as_unknown_and_caches_failure(r assert cached.health_check_error == result.health_check_error assert cached.last_health_check == result.last_health_check assert route.call_count == 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("already_waiting", [False, True]) +async def test_openapi_health_cancellation_does_not_poison_cache(respx_mock, monkeypatch, already_waiting): + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + manager = MCPServerManager() + server = MCPServer( + server_id="cancelled-cache", name="cancelled-cache", transport=MCPTransport.http, + spec_path="https://93.184.216.34/cancelled-cache.json", auth_type=MCPAuth.none, + ) + manager.registry = {server.server_id: server} + started = asyncio.Event() + attempts = [] + + async def serve(request): + attempts.append(request.url) + if not started.is_set(): + started.set() + await asyncio.Event().wait() + return httpx.Response(200, json={"paths": {}}) + + route = respx_mock.get(server.spec_path).mock(side_effect=serve) + leader = asyncio.create_task(manager.health_check_server(server.server_id)) + await asyncio.wait_for(started.wait(), timeout=1) + follower = asyncio.create_task(manager.health_check_server(server.server_id)) if already_waiting else None + await asyncio.sleep(0) + leader.cancel() + cancelled = await leader + assert cancelled.status == "unknown" + assert cancelled.health_check_error == "OpenAPI specification check was cancelled" + recovered = await follower if follower is not None else await manager.health_check_server(server.server_id) + assert recovered.status == "healthy" + assert recovered.health_check_error is None + cached = await manager.health_check_server(server.server_id) + assert cached.last_health_check == recovered.last_health_check + assert cached.status == "healthy" + assert len(attempts) == 2 + assert route.call_count == 1 From f04fb748c532edea6606e1ee9f7d110a9d82b4b2 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 21:02:46 -0700 Subject: [PATCH 061/157] fix(mcp): explain refused OAuth registration and bound discovery retries --- README.md | 2 + .../mcp_server/faults/__init__.py | 2 + .../mcp_server/faults/classify.py | 5 +- .../mcp_server/faults/render_oauth.py | 21 ++++- .../_experimental/mcp_server/faults/types.py | 10 ++- .../mcp_server/mcp_server_manager.py | 22 ++++- .../mcp_server/faults/test_classify.py | 29 +++++++ .../mcp_server/faults/test_render_oauth.py | 18 ++++ .../mcp_server/test_discoverable_endpoints.py | 45 ++++++++++ .../mcp_server/test_mcp_server_manager.py | 82 +++++++++++++++++++ 10 files changed, 228 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 92757fcbbc1..901cc5b0cea 100644 --- a/README.md +++ b/README.md @@ -262,6 +262,8 @@ curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ } ``` +For MCP OAuth, an upstream may advertise dynamic client registration but refuse requests with HTTP 401 or 403. If the provider requires a pre-registered OAuth app, configure its `credentials.client_id` and, when required, `credentials.client_secret` on the MCP server. This skips dynamic registration in the gateway sign-in flow. The provider must approve the app for MCP access; reaching its authorization page does not establish that login or tool calls will succeed + [**Docs: MCP Gateway**](https://docs.litellm.ai/docs/mcp) diff --git a/litellm/proxy/_experimental/mcp_server/faults/__init__.py b/litellm/proxy/_experimental/mcp_server/faults/__init__.py index 1b9ee77d795..de7ee5c866a 100644 --- a/litellm/proxy/_experimental/mcp_server/faults/__init__.py +++ b/litellm/proxy/_experimental/mcp_server/faults/__init__.py @@ -22,6 +22,7 @@ from litellm.proxy._experimental.mcp_server.faults.types import ( GatewayRejected, UpstreamOAuthFault, UpstreamProtocolFault, + UpstreamRegistrationRefused, UpstreamReportedFault, ) @@ -31,6 +32,7 @@ __all__ = [ "GatewayRejected", "UpstreamOAuthFault", "UpstreamProtocolFault", + "UpstreamRegistrationRefused", "UpstreamReportedFault", "classify_upstream_dcr_rejection", "classify_upstream_token_rejection", diff --git a/litellm/proxy/_experimental/mcp_server/faults/classify.py b/litellm/proxy/_experimental/mcp_server/faults/classify.py index 2162d078c09..bb41436b495 100644 --- a/litellm/proxy/_experimental/mcp_server/faults/classify.py +++ b/litellm/proxy/_experimental/mcp_server/faults/classify.py @@ -21,6 +21,7 @@ from litellm.proxy._experimental.mcp_server.faults.types import ( GatewayRejected, UpstreamOAuthFault, UpstreamProtocolFault, + UpstreamRegistrationRefused, UpstreamReportedFault, ) @@ -122,11 +123,13 @@ def classify_upstream_dcr_rejection(response: httpx.Response, log_context: str) """Classify a dynamic-client-registration rejection. RFC 7591 §3.2.2 errors carry ``error`` / ``error_description`` and go through the same blame assignment as token errors (registration sends no client credentials, so credential codes stay caller-actionable); anything - without a usable ``error`` field is an upstream protocol fault.""" + without a usable ``error`` field is a registration refusal for 401/403 and a protocol fault otherwise.""" parsed: Final = _safe_json(response) fields: Final = parsed if isinstance(parsed, dict) else {} code: Final = _bounded_field(fields.get("error")) if code is None: + if response.status_code == 401 or response.status_code == 403: + return UpstreamRegistrationRefused(status_code=response.status_code) _log_out_of_contract("registration", response, log_context) return UpstreamProtocolFault(note=f"upstream registration failed with HTTP {response.status_code}") return _classify_oauth_error_code( diff --git a/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py b/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py index 64d14140a5b..b3464382142 100644 --- a/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py +++ b/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py @@ -11,7 +11,7 @@ from typing import Final from fastapi.responses import JSONResponse from typing_extensions import assert_never -from litellm.proxy._experimental.mcp_server.faults.types import UpstreamOAuthFault +from litellm.proxy._experimental.mcp_server.faults.types import CallerRejected, UpstreamOAuthFault from litellm.proxy._experimental.mcp_server.oauth_utils import TOKEN_NO_CACHE_HEADERS @@ -35,6 +35,14 @@ def _upstream_reported_status_and_description(code: str) -> tuple[int, str]: return 502, "the upstream authorization server reported an internal error" +def _registration_refused_description(status_code: int) -> str: + return ( + f"the upstream authorization server refused dynamic client registration (HTTP {status_code}). " + "This provider may require a pre-registered OAuth client. Configure client_id and, if required " + "by the provider, client_secret for this MCP server to skip dynamic registration" + ) + + def render_token_fault(fault: UpstreamOAuthFault) -> JSONResponse: """RFC 6749 §5.2 response for a token-endpoint fault. Caller-actionable rejections relay the upstream's code on the status that code implies (401 for invalid_client per §5.2, else 400); @@ -65,6 +73,13 @@ def render_token_fault(fault: UpstreamOAuthFault) -> JSONResponse: content={"error": fault.code, "error_description": description}, headers=TOKEN_NO_CACHE_HEADERS, ) + case "upstream_registration_refused": + return render_token_fault( + CallerRejected( + code="unauthorized_client", + description=_registration_refused_description(fault.status_code), + ) + ) case "upstream_protocol_fault": return JSONResponse( status_code=502, @@ -78,7 +93,7 @@ def render_token_fault(fault: UpstreamOAuthFault) -> JSONResponse: def dcr_fault_detail(fault: UpstreamOAuthFault) -> tuple[int, str]: """Status and detail string for a registration fault, raised as HTTPException by the caller. RFC 7591 §3.2.2 defines registration errors as 400, so a contract-conformant rejection is 400 - regardless of the status the upstream chose; everything else is a 502 upstream fault.""" + regardless of the upstream status; a bare 401/403 is a registration refusal rendered as 403.""" match fault.tag: case "caller_rejected": detail: Final = f"{fault.code}: {fault.description}" if fault.description else fault.code @@ -87,6 +102,8 @@ def dcr_fault_detail(fault: UpstreamOAuthFault) -> tuple[int, str]: return 502, _gateway_rejected_description(fault.code) case "upstream_reported_fault": return _upstream_reported_status_and_description(fault.code) + case "upstream_registration_refused": + return 403, _registration_refused_description(fault.status_code) case "upstream_protocol_fault": return 502, fault.note case _: diff --git a/litellm/proxy/_experimental/mcp_server/faults/types.py b/litellm/proxy/_experimental/mcp_server/faults/types.py index 4b9505ad801..d081d9d735e 100644 --- a/litellm/proxy/_experimental/mcp_server/faults/types.py +++ b/litellm/proxy/_experimental/mcp_server/faults/types.py @@ -77,4 +77,12 @@ class UpstreamProtocolFault(BaseModel): note: str -UpstreamOAuthFault: TypeAlias = CallerRejected | GatewayRejected | UpstreamReportedFault | UpstreamProtocolFault +class UpstreamRegistrationRefused(BaseModel): + model_config = ConfigDict(frozen=True) + tag: Literal["upstream_registration_refused"] = "upstream_registration_refused" + status_code: Literal[401, 403] + + +UpstreamOAuthFault: TypeAlias = ( + CallerRejected | GatewayRejected | UpstreamReportedFault | UpstreamProtocolFault | UpstreamRegistrationRefused +) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index af25b0e919a..d5f4fa2b256 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -1910,7 +1910,7 @@ class MCPServerManager: elif server.server_id in self.config_mcp_servers: self.config_mcp_servers[server.server_id] = server else: - return None + return server self._remove_oauth_discovery_slot(server.server_id) return server @@ -2002,6 +2002,12 @@ class MCPServerManager: if slot.task is not None: if not slot.task.done() or _oauth_discovery_now() < slot.retry_not_before: return slot.task, slot.generation + if ( + not slot.task.cancelled() + and slot.task.exception() is None + and isinstance(slot.task.result(), _OAuthDiscoveryResolved) + ): + return slot.task, slot.generation task: Final = asyncio.create_task( self._run_oauth_metadata_resolution(self._registered_server(server), slot.generation) ) @@ -2031,7 +2037,7 @@ class MCPServerManager: if should_defer != has_slot: self._set_oauth_discovery_deferred(server.server_id, should_defer) - async def ensure_oauth_metadata_discovered(self, server: MCPServer) -> MCPServer: + async def ensure_oauth_metadata_discovered(self, server: MCPServer, *, _retry_stale: bool = True) -> MCPServer: """Join the bounded discovery task and return the resolved server. Concurrent callers share one task per server. A failed attempt remains @@ -2058,13 +2064,13 @@ class MCPServerManager: outcome: Final = await asyncio.shield(task) except asyncio.CancelledError: if task.cancelled() and not self._oauth_discovery_slot_is_current(server.server_id, generation): - return await self.ensure_oauth_metadata_discovered(server) + return await self._rejoin_oauth_metadata_discovery(server, retry_stale=_retry_stale) raise match outcome: case _OAuthDiscoveryResolved(resolved_server): return resolved_server case _OAuthDiscoveryStale(): - return await self.ensure_oauth_metadata_discovered(server) + return await self._rejoin_oauth_metadata_discovery(server, retry_stale=_retry_stale) case _OAuthDiscoveryFailed(timed_out=timed_out): current: Final = self._registered_server(server) if current.is_client_forwarded_token: @@ -2076,6 +2082,14 @@ class MCPServerManager: detail=f"OAuth metadata discovery {reason} for MCP server {server_ref!r}", ) + async def _rejoin_oauth_metadata_discovery(self, server: MCPServer, *, retry_stale: bool) -> MCPServer: + if retry_stale: + return await self.ensure_oauth_metadata_discovered(server, _retry_stale=False) + current: Final = self._registered_server(server) + if not _oauth_endpoints_unresolved(current) or current.is_client_forwarded_token: + return current + raise HTTPException(status_code=503, detail="OAuth metadata discovery changed repeatedly; retry shortly") + def _remember_upstream_initialize_instructions(self, server: MCPServer, client: MCPClient) -> None: raw: Final[str | None] = getattr(client, "_last_initialize_instructions", None) if raw and str(raw).strip(): diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_classify.py b/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_classify.py index dc20d664a53..30107db4055 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_classify.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_classify.py @@ -1,6 +1,9 @@ """Classification matrix for upstream OAuth/DCR rejections: who is blamed depends only on the §5.2 code and whose credentials the gateway presented, never on the upstream's HTTP status.""" +from typing import Final + +import pytest import httpx from litellm.proxy._experimental.mcp_server.faults.classify import ( @@ -12,6 +15,7 @@ from litellm.proxy._experimental.mcp_server.faults.types import ( GatewayRejected, UpstreamProtocolFault, UpstreamReportedFault, + UpstreamRegistrationRefused, ) @@ -145,3 +149,28 @@ def test_dcr_server_error_code_is_not_blamed_on_caller(): log_context="srv", ) assert isinstance(fault, UpstreamReportedFault) + + +@pytest.mark.parametrize("status_code", [401, 403]) +@pytest.mark.parametrize("body", ["Forbidden", 'private upstream details', '{"error": ""}', '{"error": 12}']) +def test_dcr_access_refusal_without_oauth_error(status_code: int, body: str) -> None: + fault: Final = classify_upstream_dcr_rejection(_response(status_code, text_body=body), log_context="srv") + assert isinstance(fault, UpstreamRegistrationRefused) + assert fault.status_code == status_code + + +@pytest.mark.parametrize("status_code", [401, 403]) +def test_dcr_access_refusal_preserves_oauth_error(status_code: int) -> None: + fault: Final = classify_upstream_dcr_rejection( + _response(status_code, json_body={"error": "invalid_redirect_uri", "error_description": "not allowed"}), + log_context="srv", + ) + assert fault == CallerRejected(code="invalid_redirect_uri", description="not allowed") + + +@pytest.mark.parametrize("status_code", [401, 403]) +def test_token_access_refusal_remains_protocol_fault(status_code: int) -> None: + fault: Final = classify_upstream_token_rejection( + _response(status_code, text_body="Forbidden"), credential_source="gateway_stored", log_context="srv" + ) + assert isinstance(fault, UpstreamProtocolFault) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_render_oauth.py b/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_render_oauth.py index 78513e315a7..a6807ae1454 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_render_oauth.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/faults/test_render_oauth.py @@ -2,6 +2,9 @@ code can never ship on a server-fault status and gateway-side faults never carry provider prose.""" import json +from typing import Final, Literal + +import pytest from litellm.proxy._experimental.mcp_server.faults.render_oauth import ( dcr_fault_detail, @@ -12,6 +15,7 @@ from litellm.proxy._experimental.mcp_server.faults.types import ( GatewayRejected, UpstreamProtocolFault, UpstreamReportedFault, + UpstreamRegistrationRefused, ) @@ -94,3 +98,17 @@ def test_dcr_upstream_reported_fault_maps_to_5xx(): status_code, detail = dcr_fault_detail(UpstreamReportedFault(code="server_error")) assert status_code == 502 assert "internal error" in detail + + +@pytest.mark.parametrize("upstream_status", [401, 403]) +def test_registration_refusal_gives_configuration_guidance(upstream_status: Literal[401, 403]) -> None: + fault: Final = UpstreamRegistrationRefused(status_code=upstream_status) + status, detail = dcr_fault_detail(fault) + assert status == 403 + assert f"HTTP {upstream_status}" in detail + assert "may require a pre-registered OAuth client" in detail + assert "client_id" in detail and "client_secret" in detail + response: Final = render_token_fault(fault) + assert response.status_code == 400 + assert json.loads(response.body) == {"error": "unauthorized_client", "error_description": detail} + assert response.headers["cache-control"] == "no-store" diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py index 5a99139a67f..18ff50a881a 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py @@ -11141,3 +11141,48 @@ async def test_enforced_login_warms_verified_token_readable_without_database_loo assert token.identity_binding_proof == proof assert token.refresh_token is None read.assert_not_awaited() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("upstream_status", [401, 403]) +@pytest.mark.parametrize("auth_type", [MCPAuth.true_passthrough, MCPAuth.oauth_delegate]) +@pytest.mark.parametrize("dcr_bridge", [False, True]) +@pytest.mark.parametrize("flow", ["register", "mint"]) +async def test_dcr_refusal_is_actionable_without_upstream_body( + upstream_status: int, auth_type: MCPAuth, dcr_bridge: bool, flow: str, monkeypatch: pytest.MonkeyPatch +) -> None: + import httpx + from typing import Final + + from litellm.proxy._experimental.mcp_server.discoverable_endpoints import ( + mint_ephemeral_dcr_client, + register_client_with_server, + ) + + server: Final = _bridge_server( + auth_type=auth_type, dcr_bridge=dcr_bridge, server_id=f"refused-{auth_type}-{dcr_bridge}-{flow}-{upstream_status}", + client_id=None, + ) + import respx + + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + with respx.mock as upstream: + registration: Final = upstream.post(server.registration_url).mock( + return_value=httpx.Response(upstream_status, text="Forbidden private upstream details") + ) + operation: Final = ( + mint_ephemeral_dcr_client(_bridge_mock_request(), server) + if flow == "mint" + else register_client_with_server( + request=_bridge_mock_request(), mcp_server=server, client_name="Test client", + grant_types=None, response_types=None, token_endpoint_auth_method=None, + client_redirect_uris=["http://localhost:9999/callback"], + ) + ) + with pytest.raises(HTTPException) as exc: + await operation + assert registration.call_count == 1 + assert exc.value.status_code == 403 + assert f"HTTP {upstream_status}" in str(exc.value.detail) + assert "pre-registered OAuth client" in str(exc.value.detail) + assert "private upstream details" not in str(exc.value.detail) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py index adf985e9a21..ecf78359191 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py @@ -12676,3 +12676,85 @@ async def test_debug_reports_legacy_signing_and_non_http_transport(transport: Li assert "Credential=AKIDEXAMPLE/" in request.headers["Authorization"] finally: request_ctx.reset(token) + + +@pytest.mark.asyncio +async def test_temporary_server_discovery_reuses_resolved_metadata_without_publishing() -> None: + manager: Final = MCPServerManager() + server: Final = MCPServer( + server_id="temporary-oauth-discovery", name="temporary", url="https://idp.example.com/mcp", + transport=MCPTransport.http, auth_type=MCPAuth.true_passthrough, + ) + manager._set_oauth_discovery_deferred(server.server_id, True) + metadata: Final = MCPOAuthMetadata( + authorization_url="https://idp.example.com/authorize", token_url="https://idp.example.com/token", + registration_url="https://idp.example.com/register", + ) + with patch.object(manager, "_discover_oauth_metadata_for_server", AsyncMock(return_value=metadata)) as discovery: + resolved: Final = await manager.ensure_oauth_metadata_discovered(server) + repeated: Final = await manager.ensure_oauth_metadata_discovered(server) + assert resolved.authorization_url == metadata.authorization_url + assert resolved.token_url == metadata.token_url + assert resolved.registration_url == metadata.registration_url + assert repeated is resolved + assert server.server_id not in manager.registry + assert server.server_id not in manager.config_mcp_servers + discovery.assert_awaited_once() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("auth_type", [MCPAuth.oauth2, MCPAuth.true_passthrough]) +async def test_repeated_stale_oauth_discovery_is_bounded(auth_type: MCPAuth) -> None: + manager: Final = MCPServerManager() + server: Final = MCPServer( + server_id="repeated-stale", name="stale", url="https://idp.example.com/mcp", + transport=MCPTransport.http, auth_type=auth_type, oauth2_flow="authorization_code", + ) + manager.registry[server.server_id] = server + manager._set_oauth_discovery_deferred(server.server_id, True) + metadata: Final = MCPOAuthMetadata( + authorization_url="https://idp.example.com/authorize", token_url="https://idp.example.com/token", + ) + with ( + patch.object(manager, "_discover_oauth_metadata_for_server", AsyncMock(return_value=metadata)) as discovery, + patch.object(manager, "_publish_resolved_oauth_server", return_value=None), + ): + if auth_type == MCPAuth.true_passthrough: + assert await manager.ensure_oauth_metadata_discovered(server) is server + else: + with pytest.raises(HTTPException) as exc: + await manager.ensure_oauth_metadata_discovered(server) + assert exc.value.status_code == 503 + assert "changed repeatedly" in str(exc.value.detail) + assert discovery.await_count == 2 + + +@pytest.mark.asyncio +async def test_stale_discovery_falls_back_to_resolved_registered_server() -> None: + manager: Final = MCPServerManager() + original: Final = MCPServer( + server_id="resolved-replacement", name="replacement", url="https://old.example.com/mcp", + transport=MCPTransport.http, auth_type=MCPAuth.oauth2, oauth2_flow="authorization_code", + ) + replacement: Final = original.model_copy(update={ + "url": "https://new.example.com/mcp", "authorization_url": "https://new.example.com/authorize", + "token_url": "https://new.example.com/token", + }) + manager.registry[original.server_id] = replacement + assert await manager._rejoin_oauth_metadata_discovery(original, retry_stale=False) is replacement + + +def test_stale_discovery_cannot_overwrite_new_registered_server() -> None: + manager: Final = MCPServerManager() + original: Final = MCPServer( + server_id="stale-publication", name="publication", url="https://old.example.com/mcp", + transport=MCPTransport.http, auth_type=MCPAuth.oauth2, + ) + manager._set_oauth_discovery_deferred(original.server_id, True) + original_slot: Final = manager._oauth_discovery_slot(original.server_id) + assert original_slot is not None + replacement: Final = original.model_copy(update={"url": "https://new.example.com/mcp"}) + manager.registry[original.server_id] = replacement + manager._set_oauth_discovery_deferred(original.server_id, True) + assert manager._publish_resolved_oauth_server(original, original_slot.generation) is None + assert manager.registry[original.server_id] is replacement From bb9b4c4aef48a244bbe4ea40cc56a60f9ffbfbce Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Thu, 10 Sep 2026 21:54:06 -0700 Subject: [PATCH 062/157] fix(mcp): render registration refusals without recursion --- .../mcp_server/faults/render_oauth.py | 20 +++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py b/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py index b3464382142..3ecf5310482 100644 --- a/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py +++ b/litellm/proxy/_experimental/mcp_server/faults/render_oauth.py @@ -43,6 +43,16 @@ def _registration_refused_description(status_code: int) -> str: ) +def _render_caller_rejected(fault: CallerRejected) -> JSONResponse: + content: Final = { + "error": fault.code, + **({"error_description": fault.description} if fault.description else {}), + **({"error_uri": fault.error_uri} if fault.error_uri else {}), + } + status_code: Final = 401 if fault.code == "invalid_client" else 400 + return JSONResponse(status_code=status_code, content=content, headers=TOKEN_NO_CACHE_HEADERS) + + def render_token_fault(fault: UpstreamOAuthFault) -> JSONResponse: """RFC 6749 §5.2 response for a token-endpoint fault. Caller-actionable rejections relay the upstream's code on the status that code implies (401 for invalid_client per §5.2, else 400); @@ -50,13 +60,7 @@ def render_token_fault(fault: UpstreamOAuthFault) -> JSONResponse: blamed for, or shown the internals of, a failure only the operator can fix.""" match fault.tag: case "caller_rejected": - content: Final = { - "error": fault.code, - **({"error_description": fault.description} if fault.description else {}), - **({"error_uri": fault.error_uri} if fault.error_uri else {}), - } - status_code = 401 if fault.code == "invalid_client" else 400 - return JSONResponse(status_code=status_code, content=content, headers=TOKEN_NO_CACHE_HEADERS) + return _render_caller_rejected(fault) case "gateway_rejected": return JSONResponse( status_code=502, @@ -74,7 +78,7 @@ def render_token_fault(fault: UpstreamOAuthFault) -> JSONResponse: headers=TOKEN_NO_CACHE_HEADERS, ) case "upstream_registration_refused": - return render_token_fault( + return _render_caller_rejected( CallerRejected( code="unauthorized_client", description=_registration_refused_description(fault.status_code), From 33ec56ed759a3d2876e821d970198cc30563f8d6 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Thu, 10 Sep 2026 21:55:23 -0700 Subject: [PATCH 063/157] test(e2e): rewrite the Redis timeout test as a locust chaos load test The sequential version sent one request at a time, so a Redis outage never reached the concurrency where the failed-tracking alert body actually grows. This drives the proxy with locust against one model group of three mock deployments, two failing at order 1 and one serving at order 2, so every request spends its retries on the failing pair and lands on the serving deployment through the order-based fallback. Two phases, a healthy baseline and a CLIENT PAUSE WRITE window, and every request must succeed in both. Latency, RSS and CPU are reported as p50/p90/p99 per phase rather than asserted on: RSS and CPU come from psutil on the proxy's process tree, since a multi-worker proxy serves /metrics from the prometheus multiprocess collector and that drops the process collector's series. Thresholds stay open until weekly runs give real baselines. Co-Authored-By: Claude Code --- .github/e2e-stack/select_tests.py | 1 - ...s-timeout.yml => test-e2e-redis-chaos.yml} | 21 +- pyproject.toml | 1 + tests/e2e/CLAUDE.md | 4 +- tests/e2e/CONTRIBUTING.md | 2 +- tests/e2e/conftest.py | 4 +- tests/e2e/coverage_registry/reliability.yaml | 2 +- tests/e2e/e2e_config.py | 2 +- ...i_config.yml => redis_chaos_ci_config.yml} | 11 +- tests/e2e/load/conftest.py | 22 +- tests/e2e/load/locust_load.py | 103 ++++++- tests/e2e/load/locustfile.py | 36 +++ tests/e2e/load/proxy_usage.py | 156 ++++++++++ tests/e2e/load/test_locust_load.py | 44 ++- tests/e2e/load/test_proxy_usage.py | 59 ++++ tests/e2e/load/test_redis_chaos_e2e.py | 258 ++++++++++++++++ tests/e2e/models.py | 1 + tests/e2e/pytest.ini | 2 +- tests/e2e/router/conftest.py | 12 - tests/e2e/router/test_redis_timeout_e2e.py | 282 ------------------ uv.lock | 4 +- 21 files changed, 673 insertions(+), 354 deletions(-) rename .github/workflows/{test-e2e-redis-timeout.yml => test-e2e-redis-chaos.yml} (82%) rename tests/e2e/gateway/{redis_timeout_ci_config.yml => redis_chaos_ci_config.yml} (59%) create mode 100644 tests/e2e/load/locustfile.py create mode 100644 tests/e2e/load/proxy_usage.py create mode 100644 tests/e2e/load/test_proxy_usage.py create mode 100644 tests/e2e/load/test_redis_chaos_e2e.py delete mode 100644 tests/e2e/router/test_redis_timeout_e2e.py diff --git a/.github/e2e-stack/select_tests.py b/.github/e2e-stack/select_tests.py index 10f26bd3b8a..a62358f81ff 100644 --- a/.github/e2e-stack/select_tests.py +++ b/.github/e2e-stack/select_tests.py @@ -8,7 +8,6 @@ UNSUPPORTED: Final = re.compile( r"|^tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e\.py$" r"|^tests/e2e/batches/test_managed_files_enforcement_e2e\.py$" r"|^tests/e2e/guardrails/test_presidio_masking_e2e\.py$" - r"|^tests/e2e/router/test_redis_timeout_e2e\.py$" ) HARNESS: Final = re.compile( r"^tests/e2e/[A-Za-z0-9_.-]+\.(py|ini)$" diff --git a/.github/workflows/test-e2e-redis-timeout.yml b/.github/workflows/test-e2e-redis-chaos.yml similarity index 82% rename from .github/workflows/test-e2e-redis-timeout.yml rename to .github/workflows/test-e2e-redis-chaos.yml index 501af0f240f..2ce20836db7 100644 --- a/.github/workflows/test-e2e-redis-timeout.yml +++ b/.github/workflows/test-e2e-redis-chaos.yml @@ -1,4 +1,4 @@ -name: "Weekly Redis Timeout E2E" +name: "Weekly Redis Chaos E2E" on: schedule: @@ -9,9 +9,9 @@ permissions: contents: read jobs: - redis-timeout-e2e: + redis-chaos-e2e: if: github.event_name != 'schedule' || github.repository == 'BerriAI/litellm' - runs-on: ubuntu-latest + runs-on: ubuntu-latest-16-cores timeout-minutes: 30 services: postgres: @@ -38,7 +38,7 @@ jobs: --health-retries 10 env: DATABASE_URL: postgresql://llmproxy:dbpassword9090@localhost:5432/litellm - LITELLM_MASTER_KEY: sk-redis-timeout-e2e + LITELLM_MASTER_KEY: sk-redis-chaos-e2e LITELLM_LOG: WARNING steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 @@ -60,7 +60,7 @@ jobs: - name: Install dependencies run: | - .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra proxy + .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --group e2e-dev --extra proxy - name: Cache Prisma binaries uses: ./.github/actions/cache-prisma-binaries @@ -69,9 +69,10 @@ jobs: run: | uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma - - name: Start the proxy with a Redis that times out on every command + - name: Start a multi-worker proxy on the chaos config run: | - nohup uv run --no-sync litellm --config tests/e2e/gateway/redis_timeout_ci_config.yml --port 4000 > proxy.log 2>&1 & + nohup uv run --no-sync litellm --config tests/e2e/gateway/redis_chaos_ci_config.yml --port 4000 --num_workers 4 > proxy.log 2>&1 & + echo "E2E_PROXY_PID=$!" >> "$GITHUB_ENV" for _ in $(seq 1 90); do if curl -fs http://localhost:4000/health/liveliness > /dev/null; then exit 0 @@ -82,14 +83,14 @@ jobs: tail -n 100 proxy.log exit 1 - - name: Run the Redis timeout e2e test + - name: Run the Redis chaos load test env: - E2E_REDIS_TIMEOUT: "1" + E2E_REDIS_CHAOS: "1" LITELLM_PROXY_URL: http://localhost:4000 REDIS_HOST: 127.0.0.1 REDIS_PORT: "6379" run: | - uv run --no-sync pytest tests/e2e/router/test_redis_timeout_e2e.py -v --tb=short -rA + uv run --no-sync pytest tests/e2e/load/test_redis_chaos_e2e.py -v --tb=short -rA -s - name: Show proxy log on failure if: failure() diff --git a/pyproject.toml b/pyproject.toml index 04f2f3fd1dd..ae1698feb5f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -220,6 +220,7 @@ e2e-dev = [ "playwright==1.61.0", "websockets>=15.0.1,<16.0", "locust==2.45.0", + "psutil==7.2.2", "mcp>=1.28.1,<2.0", ] proxy-dev = [ diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 76e8550cb9a..dc197996ac2 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -17,8 +17,8 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `mcp/` - the MCP server surface over api_key auth against the real Datadog remote MCP server (see "MCP suite: real Datadog only" below); plus the gateway-managed OAuth (authorization_code) path exercised through `/chat/completions`, the one behavior Datadog's static-header auth cannot reach, seeding the per-user upstream token via the interactive authorize dance driven with the mcp SDK's own OAuth client (headless-browser consent from a saved session) and asserting the completion lists and executes the server's tools with the stored per-user token - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection -- `router/` - routing and reliability behavior (fallbacks, cooldowns). Also holds `test_redis_timeout_e2e.py`, which needs a proxy booted from `gateway/redis_timeout_ci_config.yml` (real Redis, `socket_timeout: 0.001`, so every command times out); it is marked `redis_timeout`, deselected unless `E2E_REDIS_TIMEOUT` is set, excluded from the per-PR check, and driven weekly by `.github/workflows/test-e2e-redis-timeout.yml` -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What remains here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`) and markerless harness unit tests for the Locust/session-anomaly aggregation logic +- `router/` - routing and reliability behavior (fallbacks, cooldowns) +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments with `CLIENT PAUSE WRITE` on the proxy's Redis mid-run, asserting zero failed requests and reporting latency, RSS, and CPU as p50/p90/p99 per phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index 7a859965c80..b1331de190a 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -65,7 +65,7 @@ A couple of logging destinations are configured on the proxy rather than by the ### The pull request check -Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, `guardrails/test_presidio_masking_e2e.py`, and `router/test_redis_timeout_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start, and the Redis timeout test needs a proxy whose Redis times out on every command (`gateway/redis_timeout_ci_config.yml`), which `.github/workflows/test-e2e-redis-timeout.yml` boots on a weekly schedule +Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots on a weekly schedule Every selected file must execute at least one passing test in each pass, and any test failure, collection error, or entirely skipped or deselected file fails the check. A file whose tests are all marked skip therefore cannot pass this check, so unskip at least one of them, or add the file to `UNSUPPORTED` in `select_tests.py` with the reason, before changing one. A failed pass stops the run. The public log prints pytest's one-line summary for each pass, including the rerun count, and names each failed or errored test as `classname::name`, so a retried network error or a failing test is visible without the raw output. The final `e2e-changed-tests` job succeeds only when no supported test files changed or the approved run completed all three passes. Fork PRs with selected tests fail this gate until a maintainer brings the reviewed change onto a same-repository branch diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index ff3d2dcc6bd..99a3f4967c3 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -60,8 +60,8 @@ def pytest_configure(config: pytest.Config) -> None: ) config.addinivalue_line( "markers", - "redis_timeout: needs a proxy booted from gateway/redis_timeout_ci_config.yml whose Redis times out on every " - "command; deselected unless E2E_REDIS_TIMEOUT is set", + "redis_chaos: load test that pauses the proxy's Redis writes mid-run; needs a proxy booted from " + "gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set", ) diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index a42e4b2779a..23163ab666c 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,7 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses, embeddings], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "Under locust load with every request retrying through failing mock deployments, pausing Redis writes mid-run trips the breaker and every request still succeeds, with latency, RSS, and CPU reported as p50/p90/p99 against the pre-pause baseline; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/e2e_config.py b/tests/e2e/e2e_config.py index 6cb59ea4e90..7344a2b6cec 100644 --- a/tests/e2e/e2e_config.py +++ b/tests/e2e/e2e_config.py @@ -143,7 +143,7 @@ LOAD_MIN_CONCURRENCY_EFFICIENCY = float(os.environ.get("E2E_LOAD_MIN_CONCURRENCY WEEKLY_ANOMALY_OPT_IN_ENV = "E2E_WEEKLY_ANOMALY" MANAGED_FILES_OPT_IN_ENV = "E2E_MANAGED_FILES_STACK" -REDIS_TIMEOUT_OPT_IN_ENV = "E2E_REDIS_TIMEOUT" +REDIS_CHAOS_OPT_IN_ENV = "E2E_REDIS_CHAOS" ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6")) ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6")) ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3")) diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_chaos_ci_config.yml similarity index 59% rename from tests/e2e/gateway/redis_timeout_ci_config.yml rename to tests/e2e/gateway/redis_chaos_ci_config.yml index fd283dbc7df..37b024502f7 100644 --- a/tests/e2e/gateway/redis_timeout_ci_config.yml +++ b/tests/e2e/gateway/redis_chaos_ci_config.yml @@ -5,17 +5,14 @@ general_settings: litellm_settings: callbacks: ["prometheus"] require_auth_for_metrics_endpoint: false + enable_redis_auth_cache: true cache: true cache_params: type: redis host: 127.0.0.1 port: 6379 - socket_timeout: 0.001 + socket_timeout: 0.1 router_settings: - num_retries: 1 - fallbacks: - - redis-timeout-primary: - - redis-timeout-backup - - redis-timeout-embed-primary: - - redis-timeout-embed-backup + num_retries: 2 + disable_cooldowns: true diff --git a/tests/e2e/load/conftest.py b/tests/e2e/load/conftest.py index 3a926ef2a61..e6f3c7aa538 100644 --- a/tests/e2e/load/conftest.py +++ b/tests/e2e/load/conftest.py @@ -4,25 +4,23 @@ import os import pytest -from e2e_config import WEEKLY_ANOMALY_OPT_IN_ENV +from e2e_config import REDIS_CHAOS_OPT_IN_ENV, WEEKLY_ANOMALY_OPT_IN_ENV from load_client import LoadClient, build_client from proxy_client import ProxyClient +_OPT_IN_MARKERS = ( + ("weekly", WEEKLY_ANOMALY_OPT_IN_ENV), + ("redis_chaos", REDIS_CHAOS_OPT_IN_ENV), +) -def pytest_collection_modifyitems( - config: pytest.Config, items: list[pytest.Item] -) -> None: - if os.environ.get(WEEKLY_ANOMALY_OPT_IN_ENV): - return - deselected = [ - item for item in items if item.get_closest_marker("weekly") is not None - ] + +def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None: + opted_out = {marker for marker, opt_in_env in _OPT_IN_MARKERS if not os.environ.get(opt_in_env)} + deselected = [item for item in items if any(item.get_closest_marker(marker) is not None for marker in opted_out)] if not deselected: return config.hook.pytest_deselected(items=deselected) - items[:] = [ - item for item in items if item.get_closest_marker("weekly") is None - ] + items[:] = [item for item in items if item not in deselected] @pytest.fixture(scope="session") diff --git a/tests/e2e/load/locust_load.py b/tests/e2e/load/locust_load.py index e0da8ba70c2..a1668c5bfc5 100644 --- a/tests/e2e/load/locust_load.py +++ b/tests/e2e/load/locust_load.py @@ -1,12 +1,19 @@ from __future__ import annotations import csv +import os +import subprocess +import sys +import tempfile from dataclasses import dataclass from itertools import accumulate from pathlib import Path +from typing import Final from pydantic import BaseModel, TypeAdapter +_LOCUSTFILE = Path(__file__).with_name("locustfile.py") +_CSV_PREFIX = "locust" _GENERATOR_SATURATION_MARKER = "CPU usage above" _MAX_REPORTED_ERRORS = 5 @@ -34,7 +41,9 @@ class LoadResult: requests: int failures: int requests_per_second: float - median_response_seconds: float + p50_seconds: float + p90_seconds: float + p99_seconds: float errors: tuple[LoadError, ...] generator_warnings: tuple[str, ...] @@ -53,18 +62,24 @@ class LoadResult: lines.append("locust recorded no error breakdown") return "; ".join((*lines, *self.generator_warnings)) + def latency_summary(self) -> str: + return f"p50 {self.p50_seconds:.3f}s, p90 {self.p90_seconds:.3f}s, p99 {self.p99_seconds:.3f}s" -def median_seconds(entries: list[LocustStatEntry]) -> float: - samples = sorted( - (milliseconds, count) for entry in entries for milliseconds, count in entry.response_times.items() - ) + +def percentile_seconds(entries: list[LocustStatEntry], fraction: float) -> float: + """The response time at `fraction` of the merged histograms, in seconds. + + Locust buckets response times by millisecond, so this reads the first bucket whose + running count reaches the rank, the same lower-sample convention locust's own + percentiles use. + """ + samples = sorted((milliseconds, count) for entry in entries for milliseconds, count in entry.response_times.items()) total = sum(count for _, count in samples) if total == 0: return 0.0 running = accumulate(count for _, count in samples) - return next( - milliseconds for (milliseconds, _), seen in zip(samples, running) if seen >= total / 2 - ) / 1000.0 + rank: Final = total * fraction + return next(milliseconds for (milliseconds, _), seen in zip(samples, running) if seen >= rank) / 1000.0 def aggregate_stats( @@ -79,7 +94,9 @@ def aggregate_stats( requests=requests, failures=failures, requests_per_second=0.0, - median_response_seconds=0.0, + p50_seconds=0.0, + p90_seconds=0.0, + p99_seconds=0.0, errors=errors, generator_warnings=generator_warnings, ) @@ -88,7 +105,9 @@ def aggregate_stats( requests=requests, failures=failures, requests_per_second=requests / elapsed if elapsed > 0 else 0.0, - median_response_seconds=median_seconds(entries), + p50_seconds=percentile_seconds(entries, 0.5), + p90_seconds=percentile_seconds(entries, 0.9), + p99_seconds=percentile_seconds(entries, 0.99), errors=errors, generator_warnings=generator_warnings, ) @@ -121,3 +140,67 @@ def read_generator_warnings(stderr: str) -> tuple[str, ...]: if _GENERATOR_SATURATION_MARKER in line ) return tuple(dict.fromkeys(saturated)) + + +def run_chat_load( + *, + base_url: str, + api_keys: tuple[str, ...], + model: str, + users: int, + spawn_rate: float, + duration_seconds: float, +) -> LoadResult: + """Drive /chat/completions from headless locust and aggregate what it reported. + + Each simulated user picks one of `api_keys`, so auth and budget lookups spread over a + pool of virtual keys instead of keeping one key's cache entry permanently warm. + """ + with tempfile.TemporaryDirectory(prefix="e2e-load-") as report_dir: + csv_prefix = Path(report_dir) / _CSV_PREFIX + completed = subprocess.run( + [ + sys.executable, + "-m", + "locust", + "--headless", + "--json", + "--csv", + str(csv_prefix), + "--locustfile", + str(_LOCUSTFILE), + "--host", + base_url, + "--users", + str(users), + "--spawn-rate", + str(spawn_rate), + "--run-time", + f"{int(duration_seconds)}s", + "--exit-code-on-error", + "0", + ], + env={**os.environ, "LOAD_API_KEYS": ",".join(api_keys), "LOAD_MODEL": model}, + capture_output=True, + text=True, + timeout=duration_seconds + 120, + check=False, + ) + if completed.returncode != 0: + raise RuntimeError( + f"locust exited {completed.returncode} before it could report throughput " + f"(a startup failure, not request failures, which are folded into the JSON summary via " + f"--exit-code-on-error 0):\n{completed.stderr}" + ) + try: + entries = _STATS_ADAPTER.validate_json(completed.stdout) + except ValueError as exc: + raise RuntimeError( + f"locust exited 0 but did not print a parseable --json throughput summary on stdout; " + f"got stdout={completed.stdout!r}, stderr={completed.stderr!r}" + ) from exc + return aggregate_stats( + entries, + read_errors(csv_prefix.with_name(f"{_CSV_PREFIX}_failures.csv")), + read_generator_warnings(completed.stderr), + ) diff --git a/tests/e2e/load/locustfile.py b/tests/e2e/load/locustfile.py new file mode 100644 index 00000000000..3d9ef7c83ea --- /dev/null +++ b/tests/e2e/load/locustfile.py @@ -0,0 +1,36 @@ +from __future__ import annotations + +import os +import random +import uuid +from typing import Final + +from locust import FastHttpUser, constant, task + +_MODEL: Final = os.environ["LOAD_MODEL"] +_API_KEYS: Final = tuple(os.environ["LOAD_API_KEYS"].split(",")) + + +def _payload() -> dict[str, object]: + """A prompt no other request sent, so the response cache never answers for the deployment.""" + return { + "model": _MODEL, + "messages": [{"role": "user", "content": f"load test ping {uuid.uuid4().hex}"}], + "max_tokens": 16, + } + + +class ChatUser(FastHttpUser): + wait_time = constant(0) + + def on_start(self) -> None: + self.headers = {"Authorization": f"Bearer {random.choice(_API_KEYS)}"} + + @task + def chat(self) -> None: + self.client.post( # pyright: ignore[reportUnknownMemberType] # locust FastHttpSession.post types json/**kwargs as Any + "/chat/completions", + json=_payload(), + headers=self.headers, + name="/chat/completions", + ) diff --git a/tests/e2e/load/proxy_usage.py b/tests/e2e/load/proxy_usage.py new file mode 100644 index 00000000000..b48129f0099 --- /dev/null +++ b/tests/e2e/load/proxy_usage.py @@ -0,0 +1,156 @@ +"""Resident memory and CPU of the proxy process tree, sampled on a background thread. + +The proxy under load runs several worker processes, and `/metrics` cannot report their +memory: litellm sets PROMETHEUS_MULTIPROC_DIR when num_workers > 1, and the multiprocess +collector drops the process collector's `process_resident_memory_bytes` / +`process_cpu_seconds_total` entirely. So the test measures the tree itself through psutil, +which needs the proxy to run on the same host as the test. +""" + +from __future__ import annotations + +import math +import threading +import time +from dataclasses import dataclass +from typing import Final + +import psutil +from pydantic import BaseModel, ConfigDict + + +class _MemoryInfo(BaseModel): + model_config = ConfigDict(from_attributes=True) + + rss: int + + +@dataclass(frozen=True, slots=True) +class UsageSample: + elapsed_seconds: float + rss_bytes: int + cpu_seconds: float + + +@dataclass(frozen=True, slots=True) +class UsageWindow: + """The samples taken across one phase, plus what they say about that phase.""" + + samples: tuple[UsageSample, ...] + + def rss_percentile(self, fraction: float) -> int: + if not self.samples: + return 0 + ordered: Final = sorted(sample.rss_bytes for sample in self.samples) + return ordered[_rank(len(ordered), fraction)] + + def cpu_seconds_consumed(self) -> float: + """CPU seconds the tree burned across the window, from its monotonic counter.""" + if len(self.samples) < 2: + return 0.0 + return self.samples[-1].cpu_seconds - self.samples[0].cpu_seconds + + def cpu_utilization_percentiles(self) -> tuple[float, float, float]: + """Per-interval CPU utilization (cores busy) at p50, p90 and p99. + + Derived from consecutive samples of the cumulative counter rather than + psutil's own cpu_percent, so it covers every process in the tree including + workers that came and went between samples. + """ + rates: Final = sorted( + (later.cpu_seconds - earlier.cpu_seconds) / (later.elapsed_seconds - earlier.elapsed_seconds) + for earlier, later in zip(self.samples, self.samples[1:]) + if later.elapsed_seconds > earlier.elapsed_seconds + ) + if not rates: + return 0.0, 0.0, 0.0 + return ( + rates[_rank(len(rates), 0.5)], + rates[_rank(len(rates), 0.9)], + rates[_rank(len(rates), 0.99)], + ) + + def summary(self) -> str: + p50_cpu, p90_cpu, p99_cpu = self.cpu_utilization_percentiles() + return ( + f"RSS p50 {self.rss_percentile(0.5) / 2**20:.0f} MB, " + f"p90 {self.rss_percentile(0.9) / 2**20:.0f} MB, " + f"p99 {self.rss_percentile(0.99) / 2**20:.0f} MB; " + f"CPU cores busy p50 {p50_cpu:.2f}, p90 {p90_cpu:.2f}, p99 {p99_cpu:.2f}; " + f"{self.cpu_seconds_consumed():.1f} CPU seconds consumed" + ) + + +def _rank(count: int, fraction: float) -> int: + """Index of the sample at `fraction`, the same lower-sample convention as locust's percentiles.""" + return min(count - 1, max(0, math.ceil(count * fraction) - 1)) + + +def _read_process(process: psutil.Process) -> tuple[int, float] | None: + try: + with process.oneshot(): + memory: Final = _MemoryInfo.model_validate(process.memory_info()) + times: Final = process.cpu_times() + return memory.rss, times.user + times.system + except (psutil.NoSuchProcess, psutil.AccessDenied): + return None + + +class ProxyUsageSampler: + """Samples the proxy process tree every `interval_seconds` until stopped. + + `split()` returns the samples taken so far and starts a new window, so one sampler + covers a baseline phase and a chaos phase without a gap between them. + """ + + def __init__(self, pid: int, interval_seconds: float = 1.0) -> None: + self._process: Final = psutil.Process(pid) + self._interval: Final = interval_seconds + self._stop: Final = threading.Event() + self._lock: Final = threading.Lock() + self._samples: list[UsageSample] = [] # mutable-ok: a sampling buffer the reader drains under a lock + self._started: Final = time.monotonic() + self._thread: Final = threading.Thread(target=self._run, name="proxy-usage-sampler", daemon=True) + + def __enter__(self) -> ProxyUsageSampler: + self._thread.start() + return self + + def __exit__(self, *_: object) -> None: + self._stop.set() + self._thread.join(timeout=self._interval * 5) + + def _tree(self) -> tuple[psutil.Process, ...]: + try: + return (self._process, *self._process.children(recursive=True)) + except psutil.NoSuchProcess: + return () + + def _sample(self) -> UsageSample | None: + readings: Final = tuple(reading for process in self._tree() if (reading := _read_process(process)) is not None) + if not readings: + return None + return UsageSample( + elapsed_seconds=time.monotonic() - self._started, + rss_bytes=sum(rss for rss, _ in readings), + cpu_seconds=sum(cpu for _, cpu in readings), + ) + + def _run(self) -> None: + while not self._stop.is_set(): + sample = self._sample() + if sample is not None: + with self._lock: + self._samples.append(sample) + self._stop.wait(self._interval) + + def split(self) -> UsageWindow: + """The window that ends now; the next one starts from this window's last sample. + + The boundary sample is carried into the next window so its CPU counter has a + starting point, which is what makes the two windows' utilization comparable. + """ + with self._lock: + taken = tuple(self._samples) + self._samples = [taken[-1]] if taken else [] # rebind-ok: drains the buffer under the lock + return UsageWindow(samples=taken) diff --git a/tests/e2e/load/test_locust_load.py b/tests/e2e/load/test_locust_load.py index e3cc6f3efd5..40a238ef624 100644 --- a/tests/e2e/load/test_locust_load.py +++ b/tests/e2e/load/test_locust_load.py @@ -7,7 +7,7 @@ from locust_load import ( LoadResult, LocustStatEntry, aggregate_stats, - median_seconds, + percentile_seconds, read_errors, read_generator_warnings, ) @@ -41,35 +41,47 @@ def _result( requests=10, failures=10, requests_per_second=1.0, - median_response_seconds=0.05, + p50_seconds=0.05, + p90_seconds=0.08, + p99_seconds=0.1, errors=errors, generator_warnings=generator_warnings, ) -class TestSerialLatency: +class TestPercentiles: def test_median_is_the_middle_sample_not_the_mean_a_slow_tail_would_drag(self) -> None: # Nine fast requests and one very slow one: the mean is 1.99s, the median is 20ms. entry = _entry(num_requests=10, response_times={20: 9, 20000: 1}) - assert median_seconds([entry]) == 0.02 + assert percentile_seconds([entry], 0.5) == 0.02 - def test_median_merges_the_histograms_of_every_stats_entry(self) -> None: + def test_the_tail_percentiles_reach_the_slow_samples_the_median_hides(self) -> None: + # 100 samples: 89 fast, 10 slow, 1 very slow. p50 sits in the fast bucket, p90 in the + # slow one, and p99 lands on the single very slow sample. + entry = _entry(num_requests=100, response_times={20: 89, 500: 10, 20000: 1}) + + assert percentile_seconds([entry], 0.5) == 0.02 + assert percentile_seconds([entry], 0.9) == 0.5 + assert percentile_seconds([entry], 0.99) == 0.5 + assert percentile_seconds([entry], 1.0) == 20.0 + + def test_percentiles_merge_the_histograms_of_every_stats_entry(self) -> None: # Per entry the median would be 10ms and 90ms; merged, the middle of the five samples is 90ms. entries = [ _entry(num_requests=2, response_times={10: 2}), _entry(num_requests=3, response_times={90: 3}), ] - assert median_seconds(entries) == 0.09 + assert percentile_seconds(entries, 0.5) == 0.09 def test_an_even_split_takes_the_lower_middle_sample_as_locust_itself_does(self) -> None: entry = _entry(num_requests=4, response_times={10: 2, 90: 2}) - assert median_seconds([entry]) == 0.01 + assert percentile_seconds([entry], 0.5) == 0.01 def test_no_samples_reports_zero_rather_than_dividing_by_an_empty_histogram(self) -> None: - assert median_seconds([]) == 0.0 + assert percentile_seconds([], 0.5) == 0.0 class TestAggregate: @@ -84,9 +96,20 @@ class TestAggregate: result = aggregate_stats([entry], (), ()) assert result.requests_per_second == 3.0 - assert result.median_response_seconds == 0.057 + assert result.p50_seconds == 0.057 + assert result.p99_seconds == 0.057 assert result.failure_ratio == 0.0 + def test_tail_percentiles_come_from_the_slow_end_of_the_histogram(self) -> None: + entry = _entry(num_requests=100, response_times={20: 89, 500: 10, 3000: 1}) + + result = aggregate_stats([entry], (), ()) + + assert result.p50_seconds == 0.02 + assert result.p90_seconds == 0.5 + assert result.p99_seconds == 0.5 + assert result.latency_summary() == "p50 0.020s, p90 0.500s, p99 0.500s" + def test_throughput_spans_from_the_earliest_start_when_locust_reports_several_entries(self) -> None: entries = [ _entry(num_requests=60, start_time=1000.0, last_request_timestamp=1030.0), @@ -133,8 +156,7 @@ class TestErrorBreakdown: def test_diagnosis_caps_the_list_and_says_how_many_it_left_out(self) -> None: result = _result( errors=tuple( - LoadError(name="/chat/completions", error=f"error-{index}", occurrences=index) - for index in range(1, 9) + LoadError(name="/chat/completions", error=f"error-{index}", occurrences=index) for index in range(1, 9) ) ) diff --git a/tests/e2e/load/test_proxy_usage.py b/tests/e2e/load/test_proxy_usage.py new file mode 100644 index 00000000000..8d9f848825f --- /dev/null +++ b/tests/e2e/load/test_proxy_usage.py @@ -0,0 +1,59 @@ +from __future__ import annotations + +from typing import Final + +from proxy_usage import UsageSample, UsageWindow + +_MB: Final = 2**20 + + +def _window(*points: tuple[float, int, float]) -> UsageWindow: + return UsageWindow( + samples=tuple( + UsageSample(elapsed_seconds=elapsed, rss_bytes=rss, cpu_seconds=cpu) for elapsed, rss, cpu in points + ) + ) + + +class TestRssPercentiles: + def test_the_tail_percentiles_reach_the_peak_the_median_hides(self) -> None: + # 100 one-second samples: 89 flat, 10 elevated, 1 spike. The median stays flat, p90 sees the + # elevated plateau, and only the max reaches the spike. + window: Final = _window( + *((float(i), 100 * _MB, float(i)) for i in range(89)), + *((float(89 + i), 300 * _MB, float(89 + i)) for i in range(10)), + (99.0, 900 * _MB, 99.0), + ) + + assert window.rss_percentile(0.5) == 100 * _MB + assert window.rss_percentile(0.9) == 300 * _MB + assert window.rss_percentile(0.99) == 300 * _MB + assert window.rss_percentile(1.0) == 900 * _MB + + def test_an_empty_window_reports_zero_rather_than_indexing_nothing(self) -> None: + assert _window().rss_percentile(0.5) == 0 + + +class TestCpuUtilization: + def test_utilization_is_the_counter_delta_over_the_interval_not_the_counter_itself(self) -> None: + # The counter climbs 0.5 CPU seconds per second, then 4.0 per second: half a core, then four. + window: Final = _window((0.0, _MB, 0.0), (1.0, _MB, 0.5), (2.0, _MB, 1.0), (3.0, _MB, 5.0)) + + p50, p90, p99 = window.cpu_utilization_percentiles() + + assert (p50, p90, p99) == (0.5, 4.0, 4.0) + assert window.cpu_seconds_consumed() == 5.0 + + def test_a_single_sample_has_no_interval_and_reports_zero(self) -> None: + window: Final = _window((0.0, _MB, 3.0)) + + assert window.cpu_utilization_percentiles() == (0.0, 0.0, 0.0) + assert window.cpu_seconds_consumed() == 0.0 + + def test_summary_reports_every_percentile_in_human_units(self) -> None: + window: Final = _window((0.0, 200 * _MB, 0.0), (1.0, 200 * _MB, 1.5), (2.0, 200 * _MB, 3.0)) + + assert window.summary() == ( + "RSS p50 200 MB, p90 200 MB, p99 200 MB; " + "CPU cores busy p50 1.50, p90 1.50, p99 1.50; 3.0 CPU seconds consumed" + ) diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py new file mode 100644 index 00000000000..f91b9c2fa09 --- /dev/null +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -0,0 +1,258 @@ +"""Live e2e: the proxy under load keeps serving every request while Redis writes time out. + +Runs against a proxy booted from tests/e2e/gateway/redis_chaos_ci_config.yml, which points +cache_params at a real Redis with litellm's default socket_timeout. That one client backs all +three Redis touchpoints on the request path: the virtual-key auth cache, the response cache, +and the cross-pod spend counter the cost-tracking callback awaits. + +The load runs in two phases against one model group of three mock deployments. The two at +order 1 raise InternalServerError and the one at order 2 serves, so every request burns its +retries on the failing pair (a 500 is retryable, so retries keep re-picking inside the lowest +order) and the router's order-based fallback then re-targets order 2. Every request is expected +to succeed, and each one carries retry breadcrumbs into cost tracking. + +Phase A is a baseline with Redis healthy; phase B holds Redis in CLIENT PAUSE WRITE, so the +spend counter increment times out and the callback stringifies the request metadata, +breadcrumbs included, into a failed-tracking alert. On v1.100.0 that string doubled per request +until the worker hung (LIT-6780), which is what the per-phase RSS and CPU percentiles are here +to catch. + +Needs the proxy on the same host, since RSS and CPU come from psutil on its process tree: +a multi-worker proxy serves /metrics from the prometheus multiprocess collector, which drops +the process collector's memory and CPU series. Deselected unless E2E_REDIS_CHAOS is set. +""" + +from __future__ import annotations + +import os +import re +from collections.abc import Iterator +from dataclasses import dataclass +from typing import Final + +import pytest +import redis +from e2e_config import PROXY_BASE_URL, unique_marker +from e2e_http import NoBody +from lifecycle import ResourceManager +from load_client import LoadClient +from locust_load import LoadResult, run_chat_load +from models import KeyGenerateBody, LiteLLMParamsBody +from proxy_client import ProxyClient +from proxy_usage import ProxyUsageSampler, UsageWindow + +pytestmark = [pytest.mark.e2e, pytest.mark.redis_chaos] + +MODEL_GROUP: Final = "redis-chaos-fable" +MOCK_MODEL: Final = "anthropic/claude-fable-5-1" +FAILING_DEPLOYMENTS: Final = 2 +SERVING_DEPLOYMENTS: Final = 1 +FAILING_ORDER: Final = 1 +SERVING_ORDER: Final = 2 +KEY_POOL_SIZE: Final = 8 +LOCUST_USERS: Final = 50 +LOCUST_SPAWN_RATE: Final = 50.0 +BASELINE_SECONDS: Final = 60.0 +CHAOS_SECONDS: Final = 90.0 +REDIS_PAUSE_MS: Final = 600_000 +BASELINE_TIMEOUT_RATE_CEILING: Final = 0.05 +CHAOS_TIMEOUT_RATE_FLOOR: Final = 0.20 + +TIMEOUT_FAILURES_RE: Final = re.compile( + r'^litellm_redis_circuit_breaker_failures_total\{failure_class="timeout"\} ([0-9.e+]+)$', re.M +) +BREAKER_OPEN_RE: Final = re.compile(r'^litellm_redis_circuit_breaker_state\{state="open"\} ([0-9.e+]+)$', re.M) +BREAKER_TRANSITIONS_RE: Final = re.compile( + r'^litellm_redis_circuit_breaker_transitions_total\{state="[a-z_]+"\} ([0-9.e+]+)$', re.M +) +RETRIES_RE: Final = re.compile(r"^litellm_deployment_failure_responses_total\{[^}]*\} ([0-9.e+]+)$", re.M) +COOLDOWN_RE: Final = re.compile(r"^litellm_deployment_cooled_down_total\{[^}]*\} ([0-9.e+]+)$", re.M) + + +@dataclass(frozen=True, slots=True) +class Phase: + """One load phase's traffic and what the proxy's process tree did during it.""" + + name: str + load: LoadResult + usage: UsageWindow + + def report(self) -> str: + return ( + f"{self.name}: {self.load.requests} requests, {self.load.failures} failures, " + f"{self.load.requests_per_second:.0f} rps, {self.load.latency_summary()}; {self.usage.summary()}" + ) + + +def _failing_params() -> LiteLLMParamsBody: + return LiteLLMParamsBody( + model=MOCK_MODEL, + api_key="sk-redis-chaos-not-used", + mock_response="litellm.InternalServerError", + order=FAILING_ORDER, + ) + + +def _serving_params() -> LiteLLMParamsBody: + return LiteLLMParamsBody( + model=MOCK_MODEL, + api_key="sk-redis-chaos-not-used", + mock_response="redis chaos ok", + order=SERVING_ORDER, + ) + + +@pytest.fixture +def proxy_pid() -> int: + """The proxy's PID, which the workflow exports after starting it. + + Required rather than discovered: picking a process out of the table by name would be + ambiguous on a developer machine running more than one proxy. + """ + pid: Final = os.environ.get("E2E_PROXY_PID") + assert pid and pid.isdigit(), ( + "E2E_PROXY_PID must hold the PID of the proxy under test; RSS and CPU are read from " + "its process tree because a multi-worker proxy does not report them on /metrics" + ) + return int(pid) + + +@pytest.fixture +def redis_control() -> Iterator[redis.Redis[bytes]]: + """A control connection to the proxy's Redis, which unpauses writes in teardown. + + Only writes are paused: CLIENT PAUSE ALL would freeze this connection too, leaving + nothing able to lift the pause. + """ + host: Final = os.environ.get("REDIS_HOST") + port: Final = os.environ.get("REDIS_PORT") + assert host and port, "REDIS_HOST and REDIS_PORT must name the Redis the proxy under test uses" + control: Final = redis.Redis(host=host, port=int(port), socket_timeout=5) + try: + yield control + finally: + control.client_unpause() # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any + control.close() + + +def _metric(proxy: ProxyClient, pattern: re.Pattern[str]) -> float: + body: Final = proxy.probe("/metrics", params=NoBody()).body + return sum(float(match.group(1)) for match in pattern.finditer(body)) + + +def _register_deployments(proxy: ProxyClient, resources: ResourceManager) -> None: + for _ in range(FAILING_DEPLOYMENTS): + failing_id = proxy.create_model(MODEL_GROUP, _failing_params()) + resources.defer(lambda model_id=failing_id: proxy.delete_model(model_id)) + for _ in range(SERVING_DEPLOYMENTS): + serving_id = proxy.create_model(MODEL_GROUP, _serving_params()) + resources.defer(lambda model_id=serving_id: proxy.delete_model(model_id)) + + +def _generate_key_pool(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, ...]: + """A pool of virtual keys so auth and budget lookups are not one permanently warm + cache entry; each locust user picks one, so Redis auth reads actually happen.""" + keys: Final = tuple( + proxy.generate_key( + KeyGenerateBody(models=[MODEL_GROUP], key_alias=f"e2e-redis-chaos-{unique_marker()}-{index}") + ) + for index in range(KEY_POOL_SIZE) + ) + for key in keys: + resources.defer(lambda doomed=key: proxy.delete_key(doomed)) + return keys + + +def _drive(keys: tuple[str, ...], seconds: float) -> LoadResult: + return run_chat_load( + base_url=PROXY_BASE_URL, + api_keys=keys, + model=MODEL_GROUP, + users=LOCUST_USERS, + spawn_rate=LOCUST_SPAWN_RATE, + duration_seconds=seconds, + ) + + +class TestRedisChaos: + @pytest.mark.covers( + "reliability.circuit_breaker.redis_timeout.stays_responsive", + exercised_on=["chat_completions"], + ) + def test_load_survives_redis_write_timeouts( + self, + client: LoadClient, + resources: ResourceManager, + proxy_pid: int, + redis_control: redis.Redis[bytes], + ) -> None: + proxy: Final = client.proxy + _register_deployments(proxy, resources) + keys: Final = _generate_key_pool(proxy, resources) + + timeouts_at_start: Final = _metric(proxy, TIMEOUT_FAILURES_RE) + retries_before: Final = _metric(proxy, RETRIES_RE) + cooldowns_before: Final = _metric(proxy, COOLDOWN_RE) + + with ProxyUsageSampler(proxy_pid) as sampler: + baseline: Final = Phase(name="baseline", load=_drive(keys, BASELINE_SECONDS), usage=sampler.split()) + timeouts_after_baseline: Final = _metric(proxy, TIMEOUT_FAILURES_RE) + + redis_control.client_pause(REDIS_PAUSE_MS, all=False) # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any + chaos: Final = Phase(name="chaos", load=_drive(keys, CHAOS_SECONDS), usage=sampler.split()) + + report: Final = f"{baseline.report()} | {chaos.report()}" + + for phase in (baseline, chaos): + assert phase.load.requests > 0, ( + f"{phase.name} drove no traffic at all, so it proved nothing: {phase.load.diagnosis()}. {report}" + ) + assert phase.load.failures == 0, ( + f"{phase.name} had {phase.load.failures} of {phase.load.requests} requests fail. Every request " + f"must succeed: the failing deployments sit at order {FAILING_ORDER} and the serving one at order " + f"{SERVING_ORDER}, so once the retries on order {FAILING_ORDER} are spent the order-based fallback " + f"lands on the serving deployment. Failures mean it was cooled down, the fallback did not run, or " + f"a Redis failure reached the response path. {phase.load.diagnosis()}. {report}" + ) + + cooldowns: Final = _metric(proxy, COOLDOWN_RE) - cooldowns_before + assert cooldowns == 0, ( + f"{cooldowns:.0f} deployments were cooled down during the run; the failing deployments are supposed " + f"to stay in rotation so every request keeps exercising the retry path. {report}" + ) + + retries: Final = _metric(proxy, RETRIES_RE) - retries_before + assert retries >= baseline.load.requests + chaos.load.requests, ( + f"only {retries:.0f} deployment failures were counted across " + f"{baseline.load.requests + chaos.load.requests} requests; the mock deployments did not fail, so no " + f"request carried retry breadcrumbs into cost tracking and the regression path was never entered. " + f"{report}" + ) + + baseline_timeout_rate: Final = (timeouts_after_baseline - timeouts_at_start) / baseline.load.requests + assert baseline_timeout_rate <= BASELINE_TIMEOUT_RATE_CEILING, ( + f"a healthy Redis timed out on {baseline_timeout_rate:.1%} of baseline requests, over the " + f"{BASELINE_TIMEOUT_RATE_CEILING:.0%} this test tolerates; at litellm's default socket_timeout a " + f"loaded Redis does time out occasionally, but this much means the baseline is already degraded and " + f"the two phases are not comparable. {report}" + ) + + chaos_timeouts: Final = _metric(proxy, TIMEOUT_FAILURES_RE) - timeouts_after_baseline + chaos_timeout_rate: Final = chaos_timeouts / chaos.load.requests + transitions: Final = _metric(proxy, BREAKER_TRANSITIONS_RE) + breaker_open: Final = _metric(proxy, BREAKER_OPEN_RE) >= 1 + assert chaos_timeout_rate >= CHAOS_TIMEOUT_RATE_FLOOR or transitions >= 1 or breaker_open, ( + f"with writes paused the breaker saw Redis time out on only {chaos_timeout_rate:.1%} of requests " + f"against {baseline_timeout_rate:.1%} at baseline, under the {CHAOS_TIMEOUT_RATE_FLOOR:.0%} a real " + f"outage produces, and it counted {transitions:.0f} state transitions and ended " + f"{'open' if breaker_open else 'closed'}. The spend counter increment never failed, so this run " + f"proved nothing. {report}" + ) + + rows: Final = proxy.poll_logs_for_key(keys[0], min_rows=1) + assert rows, ( + f"no spend rows landed for the first key in the pool; a Redis outage must not cost the proxy its " + f"spend logs, which are written to Postgres through a queue rather than through Redis. {report}" + ) + + print(f"\nredis chaos load: {report}") # noqa: T201 # the numbers this test exists to report, read off the CI log diff --git a/tests/e2e/models.py b/tests/e2e/models.py index f3c111bdf06..31eb6e31401 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -924,6 +924,7 @@ class LiteLLMParamsBody(BaseModel): timeout: float | None = None tpm: int | None = None weight: int | None = None + order: int | None = None ModelMode = Literal["batch", "realtime", "image_generation"] diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index f69d5d058f3..1dcb50a6fab 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -9,4 +9,4 @@ markers = load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set managed_files: needs a proxy running with require_managed_files enabled; deselected unless E2E_MANAGED_FILES_STACK is set - redis_timeout: needs a proxy booted from gateway/redis_timeout_ci_config.yml whose Redis times out on every command; deselected unless E2E_REDIS_TIMEOUT is set + redis_chaos: load test that pauses the proxy's Redis writes mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set diff --git a/tests/e2e/router/conftest.py b/tests/e2e/router/conftest.py index 57fde729f41..2d5e546dc6c 100644 --- a/tests/e2e/router/conftest.py +++ b/tests/e2e/router/conftest.py @@ -10,12 +10,10 @@ proxy does not already list it (compose has it in static config; stage does not) from __future__ import annotations -import os from collections.abc import Iterator import pytest from complexity_router_client import ComplexityRouterClient, build_client -from e2e_config import REDIS_TIMEOUT_OPT_IN_ENV from e2e_http import NoBody, Success from lifecycle import ResourceManager from models import ( @@ -46,16 +44,6 @@ ROUTER_PARAMS = LiteLLMParamsBody( ROUTER_KEY_MODELS = [ROUTER_MODEL, "gpt-5.5", "claude-haiku-4-5"] -def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None: - if os.environ.get(REDIS_TIMEOUT_OPT_IN_ENV): - return - deselected = [item for item in items if item.get_closest_marker("redis_timeout") is not None] - if not deselected: - return - config.hook.pytest_deselected(items=deselected) - items[:] = [item for item in items if item.get_closest_marker("redis_timeout") is None] - - @pytest.fixture(scope="session") def client(proxy: ProxyClient) -> ComplexityRouterClient: return build_client(proxy) diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py deleted file mode 100644 index 3f1984c0696..00000000000 --- a/tests/e2e/router/test_redis_timeout_e2e.py +++ /dev/null @@ -1,282 +0,0 @@ -"""Live e2e: the proxy keeps answering while every Redis command times out. - -Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which -points cache_params at a real Redis with socket_timeout 0.001. The test holds that Redis in -CLIENT PAUSE WRITE for its duration, so every write the proxy sends, the spend counter increment -included, hangs past the timeout. For each endpoint it registers two deployments through -/model/new: a primary whose api_base is a closed port and a backup that answers with a mock. Each request fails the primary, -retries, falls back and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on -the spend counter increment and stringifies the request metadata into a failed-tracking alert. On -v1.100.0 that string doubled per request until the worker hung (LIT-6780). Deselected unless -E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. -""" - -from __future__ import annotations - -import os -import re -import time -from collections.abc import Callable, Iterator -from dataclasses import dataclass -from typing import Final - -import pytest -import redis -from complexity_router_client import ComplexityRouterClient -from e2e_config import unique_marker -from e2e_http import NoBody, Result, Success -from lifecycle import ResourceManager -from models import ChatBody, ChatMessage, ChatResponse, EmbedBody, KeyGenerateBody, LiteLLMParamsBody -from proxy_client import ProxyClient -from pydantic import BaseModel - -pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout] - -REQUESTS: Final = 20 -MAX_SECONDS_PER_REQUEST: Final = 10.0 -MAX_LATENCY_GROWTH_RATIO: Final = 3.0 -MAX_LIVELINESS_SECONDS: Final = 2.0 -MAX_RSS_GROWTH_BYTES: Final = 200 * 1024 * 1024 -REDIS_PAUSE_MS: Final = 600_000 -BREAKER_FAILURE_THRESHOLD: Final = 5 -CLOSED_PORT_API_BASE: Final = "http://127.0.0.1:1" -TIMEOUT_FAILURES_RE: Final = re.compile( - r'^litellm_redis_circuit_breaker_failures_total\{failure_class="timeout"\} ([0-9.e+]+)$', re.M -) -BREAKER_OPEN_RE: Final = re.compile(r'^litellm_redis_circuit_breaker_state\{state="open"\} ([0-9.e+]+)$', re.M) -BREAKER_TRANSITIONS_RE: Final = re.compile( - r'^litellm_redis_circuit_breaker_transitions_total\{state="[a-z_]+"\} ([0-9.e+]+)$', re.M -) -FALLBACKS_RE: Final = re.compile(r"^litellm_deployment_successful_fallbacks_total\{[^}]*\} ([0-9.e+]+)$", re.M) -RSS_RE: Final = re.compile(r"^process_resident_memory_bytes ([0-9.e+]+)$", re.M) - - -class ResponsesBody(BaseModel): - model: str - input: str - max_output_tokens: int = 5 - - -class ResponsesObject(BaseModel): - id: str | None = None - status: str | None = None - output: list[object] = [] - - -class EmbeddingsObject(BaseModel): - model: str | None = None - data: list[object] = [] - - -@dataclass(frozen=True, slots=True) -class Endpoint: - """One endpoint's deployments and request shape.""" - - name: str - primary: str - backup: str - primary_params: LiteLLMParamsBody - backup_params: LiteLLMParamsBody - send: Callable[[ProxyClient, str, str, str], Result[BaseModel]] - served: Callable[[BaseModel], bool] - - -def _send_chat(proxy: ProxyClient, key: str, model: str, marker: str) -> Result[BaseModel]: - return proxy.transport.post( - "/chat/completions", - headers=proxy.transport.bearer(key), - json=ChatBody(model=model, messages=[ChatMessage(role="user", content=marker)], max_tokens=5), - response_type=ChatResponse, - timeout=MAX_SECONDS_PER_REQUEST, - ) - - -def _send_responses(proxy: ProxyClient, key: str, model: str, marker: str) -> Result[BaseModel]: - return proxy.transport.post( - "/v1/responses", - headers=proxy.transport.bearer(key), - json=ResponsesBody(model=model, input=marker), - response_type=ResponsesObject, - timeout=MAX_SECONDS_PER_REQUEST, - ) - - -def _send_embeddings(proxy: ProxyClient, key: str, model: str, marker: str) -> Result[BaseModel]: - return proxy.transport.post( - "/embeddings", - headers=proxy.transport.bearer(key), - json=EmbedBody(model=model, input=marker), - response_type=EmbeddingsObject, - timeout=MAX_SECONDS_PER_REQUEST, - ) - - -_CHAT_PRIMARY: Final = LiteLLMParamsBody( - model="openai/gpt-5-mini", api_key="sk-redis-timeout-primary-not-used", api_base=CLOSED_PORT_API_BASE -) -_CHAT_BACKUP: Final = LiteLLMParamsBody( - model="openai/gpt-5-nano", api_key="sk-redis-timeout-backup-not-used", mock_response="ok" -) -_EMBED_PRIMARY: Final = LiteLLMParamsBody( - model="openai/text-embedding-3-small", api_key="sk-redis-timeout-primary-not-used", api_base=CLOSED_PORT_API_BASE -) -_EMBED_BACKUP: Final = LiteLLMParamsBody( - model="openai/text-embedding-3-large", api_key="sk-redis-timeout-backup-not-used", mock_response=[0.1, 0.2, 0.3] -) - -ENDPOINTS: Final = ( - Endpoint( - name="chat_completions", - primary="redis-timeout-primary", - backup="redis-timeout-backup", - primary_params=_CHAT_PRIMARY, - backup_params=_CHAT_BACKUP, - send=_send_chat, - served=lambda data: isinstance(data, ChatResponse) and bool(data.choices), - ), - Endpoint( - name="responses", - primary="redis-timeout-primary", - backup="redis-timeout-backup", - primary_params=_CHAT_PRIMARY, - backup_params=_CHAT_BACKUP, - send=_send_responses, - served=lambda data: isinstance(data, ResponsesObject) and bool(data.output), - ), - Endpoint( - name="embeddings", - primary="redis-timeout-embed-primary", - backup="redis-timeout-embed-backup", - primary_params=_EMBED_PRIMARY, - backup_params=_EMBED_BACKUP, - send=_send_embeddings, - served=lambda data: isinstance(data, EmbeddingsObject) and bool(data.data), - ), -) - - -@pytest.fixture -def paused_redis() -> Iterator[None]: - """Hold the proxy's Redis in CLIENT PAUSE WRITE so every write it sends outlives the 1 ms socket - timeout. A loopback Redis otherwise answers many commands inside that budget. Reads stay live so - this control connection can lift the pause in teardown.""" - host = os.environ.get("REDIS_HOST") - port = os.environ.get("REDIS_PORT") - assert host and port, "REDIS_HOST and REDIS_PORT must name the Redis the proxy under test uses" - control = redis.Redis(host=host, port=int(port), socket_timeout=5) - control.client_pause(REDIS_PAUSE_MS, all=False) # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any - try: - yield - finally: - control.client_unpause() # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any - control.close() - - -def _metric(proxy: ProxyClient, pattern: re.Pattern[str]) -> float: - body = proxy.probe("/metrics", params=NoBody()).body - return sum(float(match.group(1)) for match in pattern.finditer(body)) - - -def _rss_bytes(proxy: ProxyClient) -> float | None: - """The proxy's resident memory from the Prometheus process collector, which reads /proc and - so reports on Linux only; None where the metric is absent.""" - body = proxy.probe("/metrics", params=NoBody()).body - match = RSS_RE.search(body) - return float(match.group(1)) if match else None - - -class TestRedisTimeout: - @pytest.mark.parametrize("endpoint", ENDPOINTS, ids=[endpoint.name for endpoint in ENDPOINTS]) - @pytest.mark.covers( - "reliability.circuit_breaker.redis_timeout.stays_responsive", - exercised_on=["chat_completions", "responses", "embeddings"], - ) - def test_retries_under_redis_timeouts_keep_answering( - self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint, paused_redis: None - ) -> None: - proxy = client.proxy - primary_id = proxy.create_model(endpoint.primary, endpoint.primary_params) - resources.defer(lambda: proxy.delete_model(primary_id)) - backup_id = proxy.create_model(endpoint.backup, endpoint.backup_params) - resources.defer(lambda: proxy.delete_model(backup_id)) - key = proxy.generate_key( - KeyGenerateBody( - models=[endpoint.primary, endpoint.backup], - key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}", - ) - ) - resources.defer(lambda: proxy.delete_key(key)) - timeouts_before = _metric(proxy, TIMEOUT_FAILURES_RE) - transitions_before = _metric(proxy, BREAKER_TRANSITIONS_RE) - fallbacks_before = _metric(proxy, FALLBACKS_RE) - rss_before = _rss_bytes(proxy) - - latencies: list[float] = [] - for request_number in range(1, REQUESTS + 1): - started = time.monotonic() - result = endpoint.send(proxy, key, endpoint.primary, f"redis timeout {unique_marker()} {request_number}") - elapsed = time.monotonic() - started - assert isinstance(result, Success), ( - f"{endpoint.name} request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " - f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" - ) - assert endpoint.served(result.data), ( - f"{endpoint.name} request {request_number}: fallback to {endpoint.backup} returned no output" - ) - assert elapsed < MAX_SECONDS_PER_REQUEST, ( - f"{endpoint.name} request {request_number} took {elapsed:.1f}s with Redis timing out; " - f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" - ) - latencies.append(elapsed) - - third = REQUESTS // 3 - early = sum(latencies[:third]) / third - late = sum(latencies[-third:]) / third - assert late <= max(early * MAX_LATENCY_GROWTH_RATIO, 0.5), ( - f"{endpoint.name} per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests " - "while Redis timed out; the proxy is paying more for each failed request" - ) - - started = time.monotonic() - probe = proxy.transport.probe("/health/liveliness", params=NoBody()) - liveliness_seconds = time.monotonic() - started - assert probe.healthy, f"/health/liveliness returned {probe.status_code} after the Redis timeout loop" - assert liveliness_seconds < MAX_LIVELINESS_SECONDS, ( - f"/health/liveliness took {liveliness_seconds:.1f}s after the loop; the worker is stalled" - ) - - if rss_before is not None: - rss_after = _rss_bytes(proxy) - assert rss_after is not None - assert rss_after - rss_before <= MAX_RSS_GROWTH_BYTES, ( - f"proxy RSS grew {(rss_after - rss_before) / 2**20:.0f} MB across {REQUESTS} {endpoint.name} requests " - "with Redis timing out; on v1.100.0 this path grew by gigabytes" - ) - - fallbacks = _metric(proxy, FALLBACKS_RE) - fallbacks_before - assert fallbacks >= REQUESTS, ( - f"only {fallbacks:.0f} successful fallbacks were counted across {REQUESTS} {endpoint.name} requests; " - "the closed-port primary did not fail every request, so the retry path was not exercised" - ) - - timeouts_total = _metric(proxy, TIMEOUT_FAILURES_RE) - timeouts = timeouts_total - timeouts_before - transitions = _metric(proxy, BREAKER_TRANSITIONS_RE) - transitions_before - breaker_open = _metric(proxy, BREAKER_OPEN_RE) >= 1 - assert timeouts_total >= BREAKER_FAILURE_THRESHOLD, ( - f"the proxy counted only {timeouts_total:.0f} Redis timeouts in its lifetime; the write-paused Redis " - "never made its spend counter writes time out, so this run proved nothing" - ) - assert timeouts >= REQUESTS or transitions >= 1 or breaker_open, ( - f"during {REQUESTS} {endpoint.name} requests the breaker counted {timeouts:.0f} new timeouts, " - f"{transitions:.0f} state transitions, and ended {'open' if breaker_open else 'closed'}; " - "Redis was healthy for this case, so it proved nothing" - ) - - rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS) - assert len(rows) >= REQUESTS, ( - f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; " - "a Redis outage must not lose spend rows" - ) - failed_rows = [row.status for row in rows if row.status not in (None, "success")] - assert not failed_rows, f"{len(failed_rows)} {endpoint.name} spend rows are not successes: {failed_rows[:3]}" diff --git a/uv.lock b/uv.lock index 0fe787645a2..be124e2ead7 100644 --- a/uv.lock +++ b/uv.lock @@ -10,7 +10,7 @@ resolution-markers = [ ] [options] -exclude-newer = "2026-09-06T00:40:30.433549Z" +exclude-newer = "2026-09-08T03:56:24.358378Z" exclude-newer-span = "P3D" [manifest] @@ -4559,6 +4559,7 @@ e2e-dev = [ { name = "locust" }, { name = "mcp" }, { name = "playwright" }, + { name = "psutil" }, { name = "websockets" }, ] healthcheck = [ @@ -4747,6 +4748,7 @@ e2e-dev = [ { name = "locust", specifier = "==2.45.0" }, { name = "mcp", specifier = ">=1.28.1,<2.0" }, { name = "playwright", specifier = "==1.61.0" }, + { name = "psutil", specifier = "==7.2.2" }, { name = "websockets", specifier = ">=15.0.1,<16.0" }, ] healthcheck = [ From 5fa693896fea7ac6ab2455bbfbfbb5f11171eef7 Mon Sep 17 00:00:00 2001 From: Chetan Soni Date: Fri, 11 Sep 2026 01:21:15 -0700 Subject: [PATCH 064/157] fix(guardrails): stop 500s on POST /responses when a guardrail rewrites input --- .../guardrail_translation/handler.py | 13 +- .../crowdstrike_aidr/crowdstrike_aidr.py | 8 +- .../proxy/policy_engine/pipeline_executor.py | 9 ++ .../guardrail_hooks/test_crowdstrike_aidr.py | 117 ++++++++++++++++++ 4 files changed, 143 insertions(+), 4 deletions(-) diff --git a/litellm/llms/openai/responses/guardrail_translation/handler.py b/litellm/llms/openai/responses/guardrail_translation/handler.py index 33e0a0a923f..2fe11d9f7bd 100644 --- a/litellm/llms/openai/responses/guardrail_translation/handler.py +++ b/litellm/llms/openai/responses/guardrail_translation/handler.py @@ -497,9 +497,14 @@ class OpenAIResponsesHandler(BaseTranslation): guardrailed_texts: Final = guardrailed_inputs.get("texts") or () data["input"] = guardrailed_texts[0] if guardrailed_texts else input_data # rebind-ok: data is an out-param else: + rewritten_texts: Final = guardrailed_inputs.get("texts") or () + if len(rewritten_texts) != len(extracted.task_mappings): + from litellm.proxy.policy_engine.pipeline_executor import UnappliableRequestRewrite + + raise UnappliableRequestRewrite(guardrail_to_apply.guardrail_name or "unknown") await self._apply_guardrail_responses_to_input( messages=input_data, - responses=guardrailed_inputs.get("texts") or (), + responses=rewritten_texts, task_mappings=extracted.task_mappings, ) verbose_proxy_logger.debug("OpenAI Responses API: Processed input messages: %s", data.get("input")) @@ -635,10 +640,12 @@ class OpenAIResponsesHandler(BaseTranslation): """ Apply guardrail responses back to input messages. + ``responses`` pairs positionally with ``task_mappings``; the caller rejects + the request when the two disagree, so this never has to guess an alignment. + Override this method to customize how responses are applied. """ - for task_idx, guardrail_response in enumerate(responses): - mapping = task_mappings[task_idx] + for guardrail_response, mapping in zip(responses, task_mappings): msg_idx = cast(int, mapping[0]) content_idx_optional = cast(int | None, mapping[1]) diff --git a/litellm/proxy/guardrails/guardrail_hooks/crowdstrike_aidr/crowdstrike_aidr.py b/litellm/proxy/guardrails/guardrail_hooks/crowdstrike_aidr/crowdstrike_aidr.py index f7f500b1adc..8fed1f906e5 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/crowdstrike_aidr/crowdstrike_aidr.py +++ b/litellm/proxy/guardrails/guardrail_hooks/crowdstrike_aidr/crowdstrike_aidr.py @@ -18,6 +18,7 @@ from litellm.llms.base_llm.guardrail_translation.utils import ( effective_skip_tool_message_for_guardrail, ) from litellm.llms.custom_httpx.http_handler import ( + AsyncHTTPHandler, get_async_httpx_client, httpxSpecialProvider, ) @@ -261,6 +262,7 @@ class CrowdStrikeAIDRHandler(CustomGuardrail): fail_on_error: bool | None = True, streaming_end_of_stream_only: bool | None = None, streaming_sampling_rate: int | None = None, + async_handler: AsyncHTTPHandler | None = None, **kwargs, ) -> None: """ @@ -273,9 +275,13 @@ class CrowdStrikeAIDRHandler(CustomGuardrail): streaming_end_of_stream_only (bool | None): Scan streamed output once at end of stream instead of every streaming_sampling_rate chunks. Defaults to False. streaming_sampling_rate (int | None): Scan the accumulated streamed output every Nth chunk. Defaults to 5. + async_handler (AsyncHTTPHandler | None): HTTP client to call AI Guard with. Defaults to the shared + guardrail-callback client. **kwargs: Additional arguments passed to the CustomGuardrail base class. """ - self.async_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.GuardrailCallback) + self.async_handler = async_handler or get_async_httpx_client( + llm_provider=httpxSpecialProvider.GuardrailCallback + ) self.fail_on_error = True if fail_on_error is None else fail_on_error self._set_streaming_params( CrowdStrikeAIDRGuardrailConfigModelOptionalParams( diff --git a/litellm/proxy/policy_engine/pipeline_executor.py b/litellm/proxy/policy_engine/pipeline_executor.py index ed193c7f434..ad45781d5d2 100644 --- a/litellm/proxy/policy_engine/pipeline_executor.py +++ b/litellm/proxy/policy_engine/pipeline_executor.py @@ -58,6 +58,15 @@ class UndeliverableStreamRewrite(Exception): self.guardrail_name: Final = guardrail_name +class UnappliableRequestRewrite(Exception): + def __init__(self, guardrail_name: str) -> None: + super().__init__( + f"Guardrail '{guardrail_name}' rewrote the request in a way this endpoint cannot apply, " + "so the request was rejected rather than sent unrewritten" + ) + self.guardrail_name: Final = guardrail_name + + def _tool_call_shape(tool_call: object) -> tuple[object, object]: plain: Final = tool_call.model_dump() if isinstance(tool_call, BaseModel) else tool_call function: Final = plain.get("function") if isinstance(plain, Mapping) else None diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_crowdstrike_aidr.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_crowdstrike_aidr.py index a07157396df..a80dae27d01 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_crowdstrike_aidr.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_crowdstrike_aidr.py @@ -1,3 +1,6 @@ +from collections.abc import AsyncIterator +from contextlib import asynccontextmanager +from typing import Final, cast from unittest.mock import patch import httpx @@ -7,6 +10,9 @@ from pydantic import ValidationError import litellm from litellm.exceptions import Timeout +from litellm.integrations.custom_guardrail import CustomGuardrail +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler +from litellm.llms.openai.responses.guardrail_translation.handler import OpenAIResponsesHandler from litellm.litellm_core_utils.core_helpers import get_or_create_metadata_bucket from litellm.proxy.guardrails.guardrail_hooks.crowdstrike_aidr import initialize_guardrail from litellm.proxy.guardrails.guardrail_hooks.crowdstrike_aidr.crowdstrike_aidr import ( @@ -1719,3 +1725,114 @@ async def test_streaming_params_from_config_control_output_scan_cadence( handler = _initialize_from_config(mode="post_call", **configured) assert await _guard_calls_for_stream(handler, list("ABCDEFGHIJ")) == expected_calls + + +@asynccontextmanager +async def _guardrail_transforming_into(messages: list[dict[str, object]]) -> AsyncIterator[CrowdStrikeAIDRHandler]: + """A guardrail whose AI Guard returns ``messages`` as its rewrite.""" + + def respond(request: httpx.Request) -> httpx.Response: + return httpx.Response( + status_code=200, + json={ + "result": { + "blocked": False, + "transformed": True, + "guard_output": {"messages": messages}, + }, + }, + request=request, + ) + + async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as client: + handler: Final = AsyncHTTPHandler() + handler.client = client + yield CrowdStrikeAIDRHandler( + mode="pre_call", + guardrail_name="crowdstrike-aidr-guard", + api_key="pts_crowdstrike_tokenid", + api_base="https://api.crowdstrike.com/aidr/aiguard", + async_handler=handler, + ) + + +class _MessageShapedGuardrail(CustomGuardrail): + """Returns one text per chat message and no ``structured_messages`` rewrite. + + Prompt Security and friends scan messages rather than Responses text parts, + which is the shape that outnumbers the endpoint's own bookkeeping. + """ + + def __init__(self, redacted: str) -> None: + super().__init__(guardrail_name="message-shaped") + self.redacted: Final = redacted + + async def apply_guardrail( + self, + inputs: GenericGuardrailAPIInputs, + request_data: dict, + input_type: str, + logging_obj: object = None, + ) -> GenericGuardrailAPIInputs: + messages: Final = inputs.get("structured_messages") or () + return {"texts": [self.redacted for _ in messages]} + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("case", "instructions", "responses_input"), + [ + ( + "instructions add a system message", + "be terse", + [{"role": "user", "content": [{"type": "input_text", "text": "my ssn is 078-05-1120"}]}], + ), + ( + "tool items add messages that carry no text", + None, + [ + {"role": "user", "content": [{"type": "input_text", "text": "my ssn is 078-05-1120"}]}, + {"type": "function_call", "call_id": "c1", "name": "get_x", "arguments": "{}"}, + {"type": "function_call_output", "call_id": "c1", "output": "42"}, + ], + ), + ], +) +async def test_unalignable_rewrite_is_rejected_never_sent_unredacted( + case: str, + instructions: str | None, + responses_input: list[dict[str, object]], +) -> None: + """An unalignable rewrite must fail the request, not forward the raw prompt. + + Skipping the write-back would hand the model the unredacted text, so a + guardrail could be bypassed by adding ``instructions`` or a tool call. + """ + from litellm.proxy.policy_engine.pipeline_executor import UnappliableRequestRewrite + + data: dict[str, object] = {"model": "gpt-4o", "input": responses_input} + if instructions is not None: + data["instructions"] = instructions + + with pytest.raises(UnappliableRequestRewrite): + await OpenAIResponsesHandler().process_input_messages( + data=data, + guardrail_to_apply=_MessageShapedGuardrail("my ssn is "), + ) + + assert "078-05-1120" in str(responses_input), case + + +@pytest.mark.asyncio +async def test_aligned_rewrite_is_written_back() -> None: + """Matching counts must still redact the input in place.""" + responses_input: list[dict[str, object]] = [ + {"role": "user", "content": [{"type": "input_text", "text": "my ssn is 078-05-1120"}]} + ] + + await OpenAIResponsesHandler().process_input_messages( + data={"model": "gpt-4o", "input": responses_input}, + guardrail_to_apply=_MessageShapedGuardrail("my ssn is "), + ) + + assert cast(list, responses_input[0]["content"])[0]["text"] == "my ssn is " From 4382b86b0f3a18bf5a5c5bc4abc29cc012fb27fa Mon Sep 17 00:00:00 2001 From: mateo Date: Fri, 11 Sep 2026 13:14:37 +0000 Subject: [PATCH 065/157] registry audit 2026-09-11: xai/groq deprecation dates, deepseek-v4-flash vision, perplexity nemotron reasoning Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 10 ++++++++-- model_prices_and_context_window.json | 10 ++++++++-- 2 files changed, 16 insertions(+), 4 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 729a5623fbe..ad992ff92eb 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -33076,6 +33076,7 @@ "supports_tool_choice": true }, "groq/gemma-7b-it": { + "deprecation_date": "2024-12-18", "input_cost_per_token": 5e-08, "litellm_provider": "groq", "max_input_tokens": 8192, @@ -56515,7 +56516,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, - "supports_vision": false + "supports_vision": true }, "deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, @@ -56619,7 +56620,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, - "supports_vision": false + "supports_vision": true }, "deepseek/deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, @@ -60108,6 +60109,7 @@ ] }, "xai/grok-imagine-image-quality": { + "deprecation_date": "2026-11-02", "input_cost_per_image": 0.05, "litellm_provider": "xai", "mode": "image_generation", @@ -60124,6 +60126,7 @@ ] }, "xai/grok-imagine-image-quality-20260403": { + "deprecation_date": "2026-11-02", "input_cost_per_image": 0.05, "litellm_provider": "xai", "mode": "image_generation", @@ -60140,6 +60143,7 @@ ] }, "xai/grok-imagine-image-quality-latest": { + "deprecation_date": "2026-11-02", "input_cost_per_image": 0.05, "litellm_provider": "xai", "mode": "image_generation", @@ -61929,6 +61933,7 @@ "mode": "responses", "supports_web_search": true, "supports_function_calling": true, + "supports_reasoning": true, "input_cost_per_token": 1.15e-08, "output_cost_per_token": 1.7e-07, "cache_read_input_token_cost": 1.15e-09, @@ -61939,6 +61944,7 @@ "mode": "responses", "supports_web_search": true, "supports_function_calling": true, + "supports_reasoning": true, "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2.5e-06, "cache_read_input_token_cost": 2.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 729a5623fbe..ad992ff92eb 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -33076,6 +33076,7 @@ "supports_tool_choice": true }, "groq/gemma-7b-it": { + "deprecation_date": "2024-12-18", "input_cost_per_token": 5e-08, "litellm_provider": "groq", "max_input_tokens": 8192, @@ -56515,7 +56516,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, - "supports_vision": false + "supports_vision": true }, "deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, @@ -56619,7 +56620,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, - "supports_vision": false + "supports_vision": true }, "deepseek/deepseek-v4-flash-vision-exp": { "cache_creation_input_token_cost": 0.0, @@ -60108,6 +60109,7 @@ ] }, "xai/grok-imagine-image-quality": { + "deprecation_date": "2026-11-02", "input_cost_per_image": 0.05, "litellm_provider": "xai", "mode": "image_generation", @@ -60124,6 +60126,7 @@ ] }, "xai/grok-imagine-image-quality-20260403": { + "deprecation_date": "2026-11-02", "input_cost_per_image": 0.05, "litellm_provider": "xai", "mode": "image_generation", @@ -60140,6 +60143,7 @@ ] }, "xai/grok-imagine-image-quality-latest": { + "deprecation_date": "2026-11-02", "input_cost_per_image": 0.05, "litellm_provider": "xai", "mode": "image_generation", @@ -61929,6 +61933,7 @@ "mode": "responses", "supports_web_search": true, "supports_function_calling": true, + "supports_reasoning": true, "input_cost_per_token": 1.15e-08, "output_cost_per_token": 1.7e-07, "cache_read_input_token_cost": 1.15e-09, @@ -61939,6 +61944,7 @@ "mode": "responses", "supports_web_search": true, "supports_function_calling": true, + "supports_reasoning": true, "input_cost_per_token": 2.5e-07, "output_cost_per_token": 2.5e-06, "cache_read_input_token_cost": 2.5e-07, From 8762c664b791fa34c85265c42a18e3ccdedc6b3a Mon Sep 17 00:00:00 2001 From: mateo Date: Fri, 11 Sep 2026 13:28:35 +0000 Subject: [PATCH 066/157] test(registry): cover nemotron reasoning, v4-flash vision and xai/groq deprecation dates Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/test_litellm/test_utils.py | 56 +++++++++++++++++++++++++------- 1 file changed, 44 insertions(+), 12 deletions(-) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index e09f2fb71c7..c7e46829aba 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -198,6 +198,8 @@ def test_get_model_info_resolves_provider_prefixed_model_ids(local_model_cost_ma ("perplexity/perplexity/kimi-k3", True), ("perplexity/perplexity/deepseek-v4-flash-0731", True), ("perplexity/perplexity/kimi-k2.7-code", False), + ("perplexity/perplexity/nemotron-3.5-lightning-30b-a3b", True), + ("perplexity/perplexity/nemotron-3-ultra-550b-a55b", True), ): assert litellm.supports_reasoning(model=model) is reasoning, model @@ -209,6 +211,18 @@ def test_get_model_info_resolves_provider_prefixed_model_ids(local_model_cost_ma assert via_provider["output_cost_per_token"] == 4.4e-06 assert via_provider["mode"] == "responses" + lightning = litellm.get_model_info( + model="perplexity/nemotron-3.5-lightning-30b-a3b", custom_llm_provider="perplexity" + ) + assert lightning["key"] == "perplexity/perplexity/nemotron-3.5-lightning-30b-a3b" + assert lightning["input_cost_per_token"] == 1.15e-08 + assert lightning["output_cost_per_token"] == 1.7e-07 + assert lightning["cache_read_input_token_cost"] == 1.15e-09 + assert lightning["mode"] == "responses" + + ultra = litellm.get_model_info(model="perplexity/perplexity/nemotron-3-ultra-550b-a55b") + assert ultra["key"] == "perplexity/perplexity/nemotron-3-ultra-550b-a55b" + def test_get_model_info_strips_openai_finetune_ids_without_a_custom_suffix(local_model_cost_map): info = litellm.get_model_info(model="ft:gpt-4o-2024-08-06:my-org::abc123", custom_llm_provider="openai") @@ -4088,9 +4102,9 @@ def test_deepseek_v4_models_in_cost_map(): model_cost = json.load(f) # --- bare model names --- - for key, expected_input, expected_output, expected_cache in [ - ("deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), - ("deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), + for key, expected_input, expected_output, expected_cache, expected_vision in [ + ("deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09, True), + ("deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08, False), ]: info = model_cost.get(key) assert info is not None, f"{key} missing from model_prices_and_context_window.json" @@ -4102,11 +4116,12 @@ def test_deepseek_v4_models_in_cost_map(): assert info["max_input_tokens"] == 1_000_000 assert info["supports_function_calling"] is True assert info["supports_tool_choice"] is True + assert info.get("supports_vision", False) is expected_vision # --- provider-prefixed names --- - for key, expected_input, expected_output, expected_cache in [ - ("deepseek/deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), - ("deepseek/deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), + for key, expected_input, expected_output, expected_cache, expected_vision in [ + ("deepseek/deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09, True), + ("deepseek/deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08, False), ]: info = model_cost.get(key) assert info is not None, f"{key} missing from model_prices_and_context_window.json" @@ -4117,6 +4132,7 @@ def test_deepseek_v4_models_in_cost_map(): assert info["cache_read_input_token_cost"] == expected_cache assert info["supports_function_calling"] is True assert info["supports_tool_choice"] is True + assert info.get("supports_vision", False) is expected_vision def test_deepseek_v4_models_in_backup_cost_map(): @@ -4132,9 +4148,9 @@ def test_deepseek_v4_models_in_backup_cost_map(): model_cost = json.load(f) # --- bare model names --- - for key, expected_input, expected_output, expected_cache in [ - ("deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), - ("deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), + for key, expected_input, expected_output, expected_cache, expected_vision in [ + ("deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09, True), + ("deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08, False), ]: info = model_cost.get(key) assert info is not None, f"{key} missing from backup JSON" @@ -4144,11 +4160,12 @@ def test_deepseek_v4_models_in_backup_cost_map(): assert info["output_cost_per_token"] == expected_output assert info["cache_read_input_token_cost"] == expected_cache assert info["max_input_tokens"] == 1_000_000 + assert info.get("supports_vision", False) is expected_vision # --- provider-prefixed names --- - for key, expected_input, expected_output, expected_cache in [ - ("deepseek/deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09), - ("deepseek/deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08), + for key, expected_input, expected_output, expected_cache, expected_vision in [ + ("deepseek/deepseek-v4-flash", 3e-07, 1.2e-06, 6e-09, True), + ("deepseek/deepseek-v4-pro", 1.32e-06, 3.96e-06, 4.4e-08, False), ]: info = model_cost.get(key) assert info is not None, f"{key} missing from backup JSON" @@ -4157,6 +4174,21 @@ def test_deepseek_v4_models_in_backup_cost_map(): assert info["input_cost_per_token"] == expected_input assert info["output_cost_per_token"] == expected_output assert info["cache_read_input_token_cost"] == expected_cache + assert info.get("supports_vision", False) is expected_vision + + +def test_deprecation_dates_for_retired_xai_and_groq_models(): + import json + from pathlib import Path + + json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json" + with open(json_path) as f: + model_cost = json.load(f) + + assert model_cost["xai/grok-imagine-image-quality"]["deprecation_date"] == "2026-11-02" + assert model_cost["xai/grok-imagine-image-quality-latest"]["deprecation_date"] == "2026-11-02" + assert model_cost["xai/grok-imagine-image-quality-20260403"]["deprecation_date"] == "2026-11-02" + assert model_cost["groq/gemma-7b-it"]["deprecation_date"] == "2024-12-18" @pytest.mark.usefixtures("local_model_cost_map") From 3fc483d42436a3a5b4e05a9de327283c2c2b38a5 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 06:58:40 -0700 Subject: [PATCH 067/157] fix(mcp): capture bounded error diagnostics without exposing credentials --- litellm/experimental_mcp_client/client.py | 6 +- .../_experimental/mcp_server/mcp_debug.py | 147 ++++++++++++++---- .../mcp_server/mcp_server_manager.py | 6 +- .../client_credentials.py | 17 +- .../mcp_server/rest_endpoints.py | 4 +- .../test_client_credentials.py | 44 ++++++ .../mcp_server/test_mcp_debug.py | 139 ++++++++++++++++- .../test_mcp_oauth_passthrough_tools.py | 16 ++ 8 files changed, 341 insertions(+), 38 deletions(-) diff --git a/litellm/experimental_mcp_client/client.py b/litellm/experimental_mcp_client/client.py index 3503468c735..23bcc3b5242 100644 --- a/litellm/experimental_mcp_client/client.py +++ b/litellm/experimental_mcp_client/client.py @@ -10,6 +10,7 @@ from contextlib import AbstractAsyncContextManager from datetime import timedelta from functools import partial from importlib import metadata +from types import MappingProxyType from typing import Any, Final, Protocol, TypeAlias, TypeVar import httpx @@ -69,6 +70,7 @@ from litellm._logging import verbose_logger from litellm.constants import MCP_CLIENT_TIMEOUT, MCP_NPM_CACHE_DIR, MCP_TOOL_LISTING_TIMEOUT from litellm.experimental_mcp_client.tools import list_tools_with_pagination from litellm.llms.custom_httpx.http_handler import get_ssl_configuration +from litellm.proxy._experimental.mcp_server.mcp_debug import capture_upstream_error_response from litellm.types.llms.custom_http import VerifyTypes from litellm.types.mcp import ( MCPAuth, @@ -607,7 +609,9 @@ class MCPClient: auth=effective_auth, verify=ssl_config, follow_redirects=True, - event_hooks={"request": [guard]} if guard else {}, + event_hooks=MappingProxyType( + {"response": [capture_upstream_error_response], "request": [guard] if guard else []} + ), # mutable-ok: httpx types require lists of hooks ) return factory diff --git a/litellm/proxy/_experimental/mcp_server/mcp_debug.py b/litellm/proxy/_experimental/mcp_server/mcp_debug.py index 7a04901c2ca..c608e0a73f0 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_debug.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_debug.py @@ -85,12 +85,18 @@ Usage with curl:: http://localhost:4000/mcp/atlassian_mcp """ -import re +import asyncio +import json +from collections.abc import AsyncIterator +from itertools import islice from typing import TYPE_CHECKING, Final +from urllib.parse import parse_qsl, urlencode import httpx +from pydantic import JsonValue, TypeAdapter, ValidationError from starlette.types import Message, Send +from litellm.litellm_core_utils.secret_redaction import REDACTED, redact_string from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker from litellm.proxy._experimental.mcp_server.faults.traversal import iter_exception_tree @@ -321,58 +327,137 @@ class MCPDebug: _BODY_PREVIEW_CHARS: Final = 512 -_SENSITIVE_HEADER_NAMES: Final = frozenset({"authorization", "proxy-authorization", "cookie", "x-api-key"}) -_SENSITIVE_BODY_FIELD: Final = re.compile( - r'(?P"?(?:client_secret|client_assertion|refresh_token|access_token|id_token|password|code)"?\s*[=:]\s*"?)' - r'(?P[^&"\s,}]+)' -) +_BODY_CAPTURE_BYTES: Final = 16384 +_CAPTURE_TIMEOUT_SECONDS: Final = 1.0 +_CAPTURE_EXTENSION: Final = "litellm_mcp_error_preview" +_SAFE_HEADER_NAMES: Final = frozenset({"content-type", "content-length", "accept"}) +_JSON_BODY: Final = TypeAdapter(JsonValue) +_LOG_MASKER: Final = SensitiveDataMasker(visible_prefix=0, visible_suffix=0) -def _mask_body_match(match: re.Match[str]) -> str: - return f"{match.group('key')}{MCPDebug.mask_secret(match.group('value'))}" +def _safe_text(value: str, limit: int = _BODY_PREVIEW_CHARS) -> str: + escaped: Final = "".join(json.dumps(char)[1:-1] if ord(char) < 32 or ord(char) == 127 else char for char in value) + return escaped if len(escaped) <= limit else f"{escaped[:limit]}...(truncated)" -def _preview(raw: bytes) -> str: - text: Final = _SENSITIVE_BODY_FIELD.sub(_mask_body_match, raw.decode("utf-8", errors="replace")) - return ( - text - if len(text) <= _BODY_PREVIEW_CHARS - else f"{text[:_BODY_PREVIEW_CHARS]}...(+{len(text) - _BODY_PREVIEW_CHARS} chars)" - ) +def safe_upstream_url(url: httpx.URL) -> str: + return _safe_text(str(url.copy_with(username="", password="", query=None, fragment=None))) + + +def _sensitive_field(key: str) -> bool: + return key.lower() in ("code", "cookie", "client_assertion") or _LOG_MASKER.is_sensitive_key(key) + + +def _redact_json(value: JsonValue, depth: int = 0) -> str: + if depth >= 16: + return json.dumps("(depth limit)") + if isinstance(value, dict): + return ( + "{" + + ",".join( + json.dumps(key) + + ":" + + (json.dumps(REDACTED) if _sensitive_field(key) else _redact_json(item, depth + 1)) + for key, item in value.items() + ) + + "}" + ) + if isinstance(value, list): + return "[" + ",".join(_redact_json(item, depth + 1) for item in value) + "]" + return json.dumps(redact_string(value) if isinstance(value, str) else value) + + +def _preview(raw: bytes, content_type: str = "") -> str: + if not raw: + return "(empty)" + if len(raw) > _BODY_CAPTURE_BYTES: + return "(omitted: body exceeds capture limit)" + try: + parsed: Final = _JSON_BODY.validate_json(raw) + except ValidationError: + text: Final = raw.decode("utf-8", errors="replace") + if ( + content_type.split(";", 1)[0].strip().lower() != "application/x-www-form-urlencoded" + or "=" not in text + or any(char in text for char in "<>\n\r") + ): + return "(omitted: unstructured body)" + fields: Final = parse_qsl(text, keep_blank_values=True) + return _safe_text( + urlencode( + tuple((key, REDACTED if _sensitive_field(key) else redact_string(value)) for key, value in fields) + ) + ) + if not isinstance(parsed, (dict, list)): + return "(omitted: unstructured body)" + return _safe_text(_redact_json(parsed)) def _masked_headers(headers: httpx.Headers) -> str: - return ", ".join( - f"{name}={MCPDebug.mask_secret(value) if name.lower() in _SENSITIVE_HEADER_NAMES else value}" - for name, value in headers.items() - ) + return _safe_text(", ".join(f"{name}={value}" for name, value in headers.items() if name in _SAFE_HEADER_NAMES)) def _request_body_preview(request: httpx.Request) -> str: try: - return _preview(request.content) or "(empty)" + return _preview(request.content, request.headers.get("content-type", "")) except httpx.RequestNotRead: return "(streamed, not captured)" def _response_body_preview(response: httpx.Response) -> str: + captured: Final = response.extensions.get(_CAPTURE_EXTENSION) + if isinstance(captured, str): + return captured try: - return _preview(response.content) or "(empty)" + return _preview(response.content, response.headers.get("content-type", "")) except httpx.ResponseNotRead: return "(not read)" -def describe_upstream_http_failure(exc: BaseException) -> str | None: - """One line per upstream ``httpx.Response`` in the exception tree: the request method, URL, - masked request headers and JSON-RPC body that were sent, plus the status and body that came back. - ``None`` when the failure never reached an HTTP response (DNS, refused connection, timeout).""" - lines: Final = tuple( - f"{response.request.method} {response.request.url} -> HTTP {response.status_code} {response.reason_phrase}" - f" | request headers: {_masked_headers(response.request.headers)}" - f" | request body: {_request_body_preview(response.request)}" +async def _read_error_prefix(chunks: AsyncIterator[bytes], remaining: int) -> bytes: + chunk: Final = await anext(chunks, b"") + if not chunk or len(chunk) >= remaining: + return chunk[:remaining] + return chunk + await _read_error_prefix(chunks, remaining - len(chunk)) + + +async def capture_upstream_error_response(response: httpx.Response) -> None: + if not response.is_error: + return + try: + prefix: Final = await asyncio.wait_for( + _read_error_prefix(response.aiter_bytes(chunk_size=4096), _BODY_CAPTURE_BYTES + 1), + timeout=_CAPTURE_TIMEOUT_SECONDS, + ) + response._content = prefix # pyright: ignore[reportPrivateUsage] # rebind-ok: httpx has no public setter to retain consumed bytes for auth retries + preview: Final = _preview(prefix, response.headers.get("content-type", "")) + except (TimeoutError, httpx.HTTPError): + response._content = b"" # pyright: ignore[reportPrivateUsage] # rebind-ok: httpx auth retries must survive diagnostic read failures + response.extensions[_CAPTURE_EXTENSION] = ( + "(unavailable: error body read failed)" # rebind-ok: httpx response hooks communicate through extensions + ) + return + response.extensions[_CAPTURE_EXTENSION] = preview # rebind-ok: httpx response hooks communicate through extensions + + +def describe_upstream_response(response: httpx.Response) -> str: + try: + request: Final = response.request + except RuntimeError: + return f"HTTP {response.status_code} | request unavailable" + return ( + f"{_safe_text(request.method)} {safe_upstream_url(request.url)} -> HTTP {response.status_code}" + f" | request headers: {_masked_headers(request.headers)}" + f" | request body: {_request_body_preview(request)}" f" | response body: {_response_body_preview(response)}" - for current in iter_exception_tree(exc) + ) + + +def describe_upstream_http_failure(exc: BaseException) -> str | None: + lines: Final = tuple( + describe_upstream_response(response) + for current in islice(iter_exception_tree(exc), 16) for response in (getattr(current, "response", None),) if isinstance(response, httpx.Response) ) - return "\n".join(lines) or None + return " | ".join(lines) or None diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index 1247623f491..b6712d68f72 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -4266,7 +4266,7 @@ class MCPServerManager: raise except Exception as e: verbose_logger.warning( - "Failed to get tools from server %s: %s%s", server.name, e, _upstream_failure_suffix(e) + "Failed to get tools from server %s: %s%s", server.name, type(e).__name__, _upstream_failure_suffix(e) ) raise_classified_list_failure(e, server.name, suppress_challenge=server.is_dcr_bridge) @@ -5012,7 +5012,9 @@ class MCPServerManager: verbose_logger.warning("Connection error while listing tools from %s: %s", server_name, e) raise MCPServerListError(ServerListFault(tag="unreachable"), server_name) from e except Exception as e: - verbose_logger.warning("Error listing tools from %s: %s%s", server_name, e, _upstream_failure_suffix(e)) + verbose_logger.warning( + "Error listing tools from %s: %s%s", server_name, type(e).__name__, _upstream_failure_suffix(e) + ) raise_classified_list_failure(e, server_name) _SHORT_PREFIX_MAX_REHASH_ATTEMPTS = 1024 diff --git a/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py b/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py index f962148c7ff..513fc2f2036 100644 --- a/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py +++ b/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py @@ -38,7 +38,11 @@ from pydantic import BaseModel, ConfigDict, Field, SecretStr, TypeAdapter, Valid from typing_extensions import assert_never from litellm._logging import verbose_logger -from litellm.proxy._experimental.mcp_server.mcp_debug import describe_upstream_http_failure +from litellm.proxy._experimental.mcp_server.mcp_debug import ( + describe_upstream_http_failure, + describe_upstream_response, + safe_upstream_url, +) from litellm.proxy._experimental.mcp_server.outbound_credentials.oauth_token_store import ( InMemoryTokenCacheBackend, OAuthToken, @@ -118,13 +122,22 @@ async def post_client_credentials_grant( ) return TokenEndpointDenied(status_code=status_code, detail=f"token endpoint returned HTTP {status_code}") except Exception as exc: # noqa: BLE001 # any transport failure is the same outcome: unreachable - return TokenEndpointUnreachable(detail=str(exc)) + verbose_logger.warning( + "OAuth2 client_credentials POST %s failed: %s", safe_upstream_url(httpx.URL(url)), type(exc).__name__ + ) + return TokenEndpointUnreachable(detail=type(exc).__name__) try: body: Final = _TOKEN_BODY_ADAPTER.validate_json(response.content) except ValidationError: + verbose_logger.warning("OAuth2 client_credentials invalid response: %s", describe_upstream_response(response)) return TokenEndpointDenied( status_code=response.status_code, detail="token endpoint returned a non-JSON-object body" ) + access_token: Final = body.get("access_token") + if not isinstance(access_token, str) or not access_token: + verbose_logger.warning( + "OAuth2 client_credentials response has no access token | %s", describe_upstream_response(response) + ) return TokenEndpointSuccess(body=body) diff --git a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py index 329dddbdf05..384a1858545 100644 --- a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py +++ b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py @@ -900,7 +900,9 @@ if MCP_AVAILABLE: apply_tool_filters=apply_tool_filters, ) except Exception as e: - verbose_logger.exception("Error getting tools from %s: %s", server.name, e) + verbose_logger.warning( + "Error getting tools from %s: %s", server.name, classify_list_exception(e).tag + ) return (), classify_list_exception(e) return tools_result, ServerListOk(tool_count=len(tools_result)) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py b/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py index 010e7e14d39..1550b812a87 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py @@ -478,3 +478,47 @@ async def test_bearer_auth_advertises_the_header_it_will_occupy(): assert ClientCredentialsBearerAuth("t", refetch, ClientCredentialsConfig()).header_name == "Authorization" default_carrier = ClientCredentialsConfig(header_name="esb-oauth") assert ClientCredentialsBearerAuth("t", refetch, default_carrier).header_name == "esb-oauth" + + +@pytest.mark.asyncio +@pytest.mark.parametrize("mode", ["denied", "invalid", "missing", "success", "timeout", "connect", "cancel"]) +async def test_token_exchange_failure_diagnostics(mode, monkeypatch, caplog): + import asyncio + import logging + from litellm.llms.custom_httpx import http_handler + from litellm.proxy._experimental.mcp_server.outbound_credentials.client_credentials import post_client_credentials_grant + + class Poster: + async def post(self, url, headers, data): + request = httpx.Request("POST", url, headers=headers, data=data) + if mode == "timeout": + raise httpx.ReadTimeout("private-transport-message", request=request) + if mode == "connect": + raise httpx.ConnectError("private-transport-message", request=request) + if mode == "cancel": + raise asyncio.CancelledError + response = httpx.Response(401 if mode == "denied" else 200, request=request, + content=b"not-json-private" if mode == "invalid" else None, + json=None if mode == "invalid" else {"error": "invalid_client", "client_secret":"first second", **({"access_token":"private-token"} if mode == "success" else {})}) + response.raise_for_status() + return response + + monkeypatch.setattr(http_handler, "get_async_httpx_client", lambda **kwargs: Poster()) + with caplog.at_level(logging.WARNING, logger="LiteLLM"): + if mode == "cancel": + with pytest.raises(asyncio.CancelledError): + await post_client_credentials_grant("https://idp/token", {}, {}) + assert not caplog.text + return + result = await post_client_credentials_grant("https://idp/token?key=query-secret", {"client_secret":"first second"}, {"X-Custom":"header-secret"}) + for secret in ("first", "second", "query-secret", "header-secret", "private-token", "not-json-private", "private-transport-message"): + assert secret not in caplog.text + if mode == "success": + assert isinstance(result, TokenEndpointSuccess) and result.body["access_token"] == "private-token" + assert not caplog.text + elif mode in {"timeout", "connect"}: + assert isinstance(result, TokenEndpointUnreachable) + assert "POST https://idp/token failed" in caplog.text + else: + assert "POST https://idp/token -> HTTP" in caplog.text + assert {"denied":"denied", "invalid":"invalid response", "missing":"no access token"}[mode] in caplog.text diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index b37e738579c..8ca9edc1ff9 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -6,6 +6,7 @@ import asyncio from unittest.mock import MagicMock import httpx +import pytest from litellm.proxy._experimental.mcp_server.mcp_debug import ( MCP_DEBUG_REQUEST_HEADER, @@ -280,7 +281,7 @@ class TestDescribeUpstreamHttpFailure: request = httpx.Request( "POST", "https://upstream.example/apis/mcp", - headers={"Authorization": "Bearer secret-token-abcdef0123456789", "Content-Type": "application/json"}, + headers={"Authorization": "Bearer secret-token-abcdef0123456789", "Content-Type": "application/json" if body.startswith(b"{") else "application/x-www-form-urlencoded"}, content=body, ) response = ( @@ -327,3 +328,139 @@ class TestDescribeUpstreamHttpFailure: def test_returns_none_without_http_response(self): assert describe_upstream_http_failure(ConnectionError("refused")) is None + + +@pytest.mark.parametrize("body", [ + b'{"password":"first second","token":"demo-secret"}', + b'{"nested":[{"access_token":"first,second"}]}', + b'client%5Fsecret=first+second&token=demo-secret', +]) +def test_failure_log_fully_redacts_structured_secrets(body): + request = httpx.Request("POST", "https://upstream/mcp?credential=query-secret", + headers={"X-Custom-Credential": "custom-secret"}, content=body) + response = httpx.Response(500, request=request, content=body) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) + assert detail is not None + for secret in ("first", "second", "demo-secret", "custom-secret", "query-secret"): + assert secret not in detail + + +def test_failure_log_omits_unstructured_body(): + request = httpx.Request("POST", "https://upstream/mcp", content=b"arbitrary-secret") + response = httpx.Response(500, request=request, content=b"arbitrary-secret") + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) + assert detail is not None + assert "arbitrary-secret" not in detail + assert "omitted" in detail + + +@pytest.mark.asyncio +@pytest.mark.parametrize("mode", ["error", "large", "timeout", "read_failure", "success", "cancel"]) +async def test_error_capture_is_bounded_and_preserves_success_and_cancellation(mode): + from litellm.proxy._experimental.mcp_server.mcp_debug import capture_upstream_error_response + + class Stream(httpx.AsyncByteStream): + def __init__(self): + self.reads = 0 + + async def __aiter__(self): + self.reads += 1 + if mode == "timeout": + await asyncio.sleep(10) + if mode == "read_failure": + raise httpx.ReadError("private-read-error") + if mode == "cancel": + raise asyncio.CancelledError + yield b'{"error":"missing_scope","password":"first second"}' if mode != "large" else b"x" * 20000 + + stream = Stream() + request = httpx.Request("POST", "https://upstream/mcp") + response = httpx.Response(200 if mode == "success" else 500, request=request, stream=stream) + if mode == "cancel": + with pytest.raises(asyncio.CancelledError): + await capture_upstream_error_response(response) + return + await capture_upstream_error_response(response) + if mode == "success": + assert stream.reads == 0 + assert await response.aread() == b'{"error":"missing_scope","password":"first second"}' + return + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) + assert detail is not None + assert "first" not in detail and "second" not in detail and "private-read-error" not in detail + expected = {"error": "missing_scope", "large": "capture limit", "timeout": "read failed", "read_failure": "read failed"} + assert expected[mode] in detail + if mode == "error": + assert await response.aread() == b'{"error":"missing_scope","password":"first second"}' + + +@pytest.mark.parametrize("body", [b"", b'"scalar"', b'{"hint":"line1\\nline2"}', b'{"hint":"' + b'x' * 600 + b'"}']) +def test_failure_preview_handles_empty_scalar_control_and_long_bodies(body): + request = httpx.Request("POST", "https://user:secret@upstream/mcp?key=private#private", content=body) + response = httpx.Response(500, request=request, content=body) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) + assert detail is not None + assert "private" not in detail and "user:secret" not in detail and "\n" not in detail + if not body: + assert "(empty)" in detail + elif body.startswith(b'"'): + assert "omitted" in detail + elif len(body) > 512: + assert "truncated" in detail and len(detail) < 1300 + else: + assert "line1\\nline2" in detail + + +@pytest.mark.asyncio +async def test_error_capture_preserves_httpx_auth_retry(): + from litellm.proxy._experimental.mcp_server.mcp_debug import capture_upstream_error_response + + class RetryAuth(httpx.Auth): + def auth_flow(self, request): + response = yield request + if response.status_code == 401: + request.headers["Authorization"] = "Bearer refreshed" + yield request + + def upstream(request): + if request.headers.get("Authorization"): + return httpx.Response(200, json={"ok": True}) + return httpx.Response(401, json={"error":"expired_token"}) + + async with httpx.AsyncClient(transport=httpx.MockTransport(upstream), auth=RetryAuth(), + event_hooks={"response":[capture_upstream_error_response]}) as client: + response = await client.get("https://upstream/mcp") + assert response.status_code == 200 and response.json() == {"ok":True} + assert response.history[0].json() == {"error":"expired_token"} + + +def test_failure_diagnostics_without_request_and_with_streamed_request(): + response = httpx.Response(503) + exc = httpx.HTTPStatusError("failed", request=httpx.Request("GET", "https://upstream"), response=response) + assert describe_upstream_http_failure(exc) == "HTTP 503 | request unavailable" + request = httpx.Request("POST", "https://upstream", content=iter((b"private-body",))) + response = httpx.Response(503, request=request) + described = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert described is not None and "streamed, not captured" in described and "private-body" not in described + + + +def test_deep_error_body_is_bounded_without_exposing_nested_values(): + import json + body = b'{"nested":' * 18 + b'{"password":"hidden-value"}' + b'}' * 18 + request = httpx.Request("POST", "https://upstream/mcp", content=body) + response = httpx.Response(500, request=request, content=body) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) + assert detail is not None and "depth limit" in detail and "hidden-value" not in detail + assert json.loads(detail.split("response body: ")[1])["nested"] + + +@pytest.mark.parametrize("body", [b'client%5Fsecret=first+second&client_id=visible', b'client_secret=first%26second&client_id=visible']) +def test_encoded_form_credentials_are_decoded_before_redaction(body): + request = httpx.Request("POST", "https://upstream/token", content=body, + headers={"Content-Type":"application/x-www-form-urlencoded"}) + response = httpx.Response(400, request=request, content=body, + headers={"Content-Type":"application/x-www-form-urlencoded"}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) + assert detail is not None and "client_id=visible" in detail + assert "first" not in detail and "second" not in detail diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py index c7f3ea47fb3..f8e4cd0c6ae 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py @@ -459,3 +459,19 @@ async def test_fetch_tools_logs_upstream_request_details_on_500(caplog): assert "POST https://upstream/apis/mcp -> HTTP 500" in caplog.text assert '"method":"initialize"' in caplog.text assert "upstream-token-0123456789" not in caplog.text + + + +@pytest.mark.asyncio +async def test_client_creation_failure_logs_sanitized_exchange(monkeypatch, caplog): + manager = MCPServerManager() + server = MCPServer(server_id="sample", name="sample", url="https://upstream/mcp", transport=MCPTransport.http, auth_type=MCPAuth.none) + request = httpx.Request("POST", "https://upstream/mcp?credential=query-secret") + response = httpx.Response(500, request=request, json={"error":"missing_scope"}) + error = httpx.HTTPStatusError("query-secret", request=request, response=response) + monkeypatch.setattr(manager, "_create_mcp_client", AsyncMock(side_effect=error)) + with caplog.at_level(logging.WARNING, logger="LiteLLM"): + with pytest.raises(MCPServerListError): + await manager._get_tools_from_server(server) + assert "POST https://upstream/mcp -> HTTP 500" in caplog.text + assert "missing_scope" in caplog.text and "query-secret" not in caplog.text From 8d5a675878f22fb387c7e5a8bd642913408f67c8 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 07:01:09 -0700 Subject: [PATCH 068/157] fix(mcp): expire temporary OAuth discovery results --- .../mcp_server/mcp_server_manager.py | 11 ++++++ .../mcp_server/test_mcp_server_manager.py | 36 +++++++++++++++++++ 2 files changed, 47 insertions(+) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index d5f4fa2b256..5f9384c7eea 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -254,6 +254,7 @@ _TRUE_ENV_VALUES: Final = frozenset(("1", "true", "yes", "on")) _OAUTH_DISCOVERY_RETRY_DELAYS_SECONDS: Final = (0.05, 0.15) _OAUTH_DISCOVERY_RETRY_BASE_SECONDS: Final = 30.0 _OAUTH_DISCOVERY_RETRY_MAX_SECONDS: Final = 900.0 +_OAUTH_TEMPORARY_DISCOVERY_TTL_SECONDS: Final = 300.0 def _oauth_discovery_now() -> float: @@ -1898,6 +1899,10 @@ class MCPServerManager: slot: Final = self._oauth_discovery_slot(server_id) return slot is not None and slot.generation == generation + def _expire_temporary_oauth_discovery(self, server_id: str, generation: int) -> None: + if self._oauth_discovery_slot_is_current(server_id, generation): + self._remove_oauth_discovery_slot(server_id) + def _publish_resolved_oauth_server( self, server: MCPServer, @@ -1910,6 +1915,12 @@ class MCPServerManager: elif server.server_id in self.config_mcp_servers: self.config_mcp_servers[server.server_id] = server else: + asyncio.get_running_loop().call_later( + _OAUTH_TEMPORARY_DISCOVERY_TTL_SECONDS, + self._expire_temporary_oauth_discovery, + server.server_id, + generation, + ) return server self._remove_oauth_discovery_slot(server.server_id) return server diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py index ecf78359191..f0998cff583 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py @@ -12758,3 +12758,39 @@ def test_stale_discovery_cannot_overwrite_new_registered_server() -> None: manager._set_oauth_discovery_deferred(original.server_id, True) assert manager._publish_resolved_oauth_server(original, original_slot.generation) is None assert manager.registry[original.server_id] is replacement + + +@pytest.mark.asyncio +async def test_temporary_oauth_discovery_expires_without_more_requests() -> None: + manager: Final = MCPServerManager() + server: Final = MCPServer( + server_id="expiring-session", name="temporary", url="https://idp.example.com/mcp", + transport=MCPTransport.http, auth_type=MCPAuth.true_passthrough, + authorization_url="https://idp.example.com/authorize", token_url="https://idp.example.com/token", + ) + manager._set_oauth_discovery_deferred(server.server_id, True) + resolved: Final = await manager.ensure_oauth_metadata_discovered(server) + assert manager._oauth_discovery_slot(server.server_id) is not None + loop: Final = asyncio.get_running_loop() + expired: Final = loop.create_future() + with patch.object(loop, "time", return_value=loop.time() + 301): + loop.call_later(0, expired.set_result, None) + await expired + assert resolved.authorization_url == server.authorization_url + assert manager._oauth_discovery_slot(server.server_id) is None + + +def test_old_temporary_discovery_expiry_preserves_replacement() -> None: + manager: Final = MCPServerManager() + manager._set_oauth_discovery_deferred("reused-session", True) + old_slot: Final = manager._oauth_discovery_slot("reused-session") + assert old_slot is not None + manager._set_oauth_discovery_deferred("reused-session", True) + replacement: Final = manager._oauth_discovery_slot("reused-session") + manager._expire_temporary_oauth_discovery("reused-session", old_slot.generation) + assert manager._oauth_discovery_slot("reused-session") is replacement + assert replacement is not None + manager._expire_temporary_oauth_discovery("reused-session", replacement.generation) + assert manager._oauth_discovery_slot("reused-session") is None + manager._expire_temporary_oauth_discovery("reused-session", replacement.generation) + assert manager._oauth_discovery_slot("reused-session") is None From 0ee9e1e448d36e219148a35d2369a1f0dcf3eab5 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 07:20:48 -0700 Subject: [PATCH 069/157] fix(mcp): redact reflected credentials and avoid import cycles --- .../_experimental/mcp_server/mcp_debug.py | 166 ++++++++++++++---- .../client_credentials.py | 10 +- .../test_mcp_client.py | 12 ++ .../mcp_server/test_mcp_debug.py | 72 +++++++- 4 files changed, 211 insertions(+), 49 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_debug.py b/litellm/proxy/_experimental/mcp_server/mcp_debug.py index 07e77856894..ccce0aaf615 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_debug.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_debug.py @@ -101,21 +101,24 @@ Usage with curl:: """ import asyncio +import base64 +import io import json +import re from collections.abc import AsyncIterator, Callable, Mapping +from http.cookies import CookieError, SimpleCookie from itertools import islice from types import MappingProxyType from typing import Final -from urllib.parse import parse_qsl, urlencode +from urllib.parse import parse_qsl, quote, quote_plus, unquote_plus, urlencode import httpx -from pydantic import JsonValue, TypeAdapter, ValidationError +from pydantic import JsonValue, TypeAdapter from starlette.requests import HTTPConnection from starlette.types import Message, Send from litellm.litellm_core_utils.secret_redaction import REDACTED, redact_string from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker -from litellm.proxy._experimental.mcp_server.faults.traversal import iter_exception_tree from litellm.proxy._experimental.mcp_server.outbound_credentials.types import AuthResolution # Header the client sends to opt into debug mode @@ -396,6 +399,7 @@ _BODY_CAPTURE_BYTES: Final = 16384 _CAPTURE_TIMEOUT_SECONDS: Final = 1.0 _CAPTURE_EXTENSION: Final = "litellm_mcp_error_preview" _SAFE_HEADER_NAMES: Final = frozenset({"content-type", "content-length", "accept"}) +_PUBLIC_HEADER_NAMES: Final = _SAFE_HEADER_NAMES | frozenset(("host", "user-agent", "accept-encoding", "connection")) _JSON_BODY: Final = TypeAdapter(JsonValue) _LOG_MASKER: Final = SensitiveDataMasker(visible_prefix=0, visible_suffix=0) @@ -413,33 +417,105 @@ def _sensitive_field(key: str) -> bool: return key.lower() in ("code", "cookie", "client_assertion") or _LOG_MASKER.is_sensitive_key(key) -def _redact_json(value: JsonValue, depth: int = 0) -> str: - if depth >= 16: - return json.dumps("(depth limit)") - if isinstance(value, dict): - return ( - "{" - + ",".join( - json.dumps(key) - + ":" - + (json.dumps(REDACTED) if _sensitive_field(key) else _redact_json(item, depth + 1)) - for key, item in value.items() - ) - + "}" +def _redact_object( + fields: Mapping[str, JsonValue], +) -> dict[str, JsonValue]: # mutable-ok: the standard JSON encoder requires dict objects + return { # mutable-ok: construct the JSON object once for the standard parser and encoder + key: REDACTED if _sensitive_field(key) else value for key, value in fields.items() + } + + +def _header_secret_values(name: str, value: str) -> tuple[str, ...]: + if name == "cookie": + cookie: Final = SimpleCookie[str]() + try: + cookie.load(value) + except CookieError: + return (value,) + return (value, *(item.value for item in cookie.values())) + if name not in ("authorization", "proxy-authorization"): + return (value,) + scheme, _, credential = value.partition(" ") + if scheme.lower() != "basic": + return (value, credential) + try: + decoded: Final = base64.b64decode(credential, validate=True).decode("utf-8") + except ValueError: + return (value, credential) + password: Final = decoded.partition(":")[2] + return (value, credential, decoded, password, unquote_plus(password)) + + +def _body_secret_values(request: httpx.Request) -> tuple[str, ...] | None: + try: + raw: Final = request.content + except httpx.RequestNotRead: + return None + if not raw: + return () + if len(raw) > _BODY_CAPTURE_BYTES: + return None + if request.headers.get("content-type", "").split(";", 1)[0].strip().lower() == "application/x-www-form-urlencoded": + return tuple(value for key, value in parse_qsl(raw.decode("utf-8", errors="replace")) if _sensitive_field(key)) + try: + body: Final = _JSON_BODY.validate_json(raw) + except ValueError: + return None + from litellm.proxy._experimental.mcp_server.utils import ( # noqa: PLC0415 # MCP utils imports clients; inspect bodies only after initialization + json_string_leaves, + ) + + leaves: Final = json_string_leaves(body) + if leaves is None: + return None + return tuple( + value + for path, value in leaves + if not path or any(isinstance(part, str) and _sensitive_field(part) for part in path) + ) + + +def _request_secret_values(request: httpx.Request) -> tuple[str, ...] | None: + body_values: Final = _body_secret_values(request) + if body_values is None: + return None + values: Final = ( + *body_values, + request.url.password, + *(value for _, value in request.url.params.multi_items()), + *( + secret + for name, value in request.headers.items() + if name not in _PUBLIC_HEADER_NAMES + for secret in _header_secret_values(name, value) + ), + ) + return tuple(sorted(frozenset(value for value in values if value), key=len, reverse=True)) + + +def _mask_known_values(value: str, secrets: tuple[str, ...]) -> str: + variants: Final = tuple( + sorted( + frozenset( + variant + for secret in secrets + for variant in (secret, json.dumps(secret)[1:-1], quote(secret, safe=""), quote_plus(secret)) + ), + key=len, + reverse=True, ) - if isinstance(value, list): - return "[" + ",".join(_redact_json(item, depth + 1) for item in value) + "]" - return json.dumps(redact_string(value) if isinstance(value, str) else value) + ) + return re.sub("|".join(re.escape(secret) for secret in variants), REDACTED, value) if variants else value -def _preview(raw: bytes, content_type: str = "") -> str: +def _preview(raw: bytes, content_type: str = "", secrets: tuple[str, ...] = ()) -> str: if not raw: return "(empty)" if len(raw) > _BODY_CAPTURE_BYTES: return "(omitted: body exceeds capture limit)" try: - parsed: Final = _JSON_BODY.validate_json(raw) - except ValidationError: + parsed: Final = _JSON_BODY.validate_python(json.loads(raw, object_hook=_redact_object)) + except (ValueError, RecursionError): text: Final = raw.decode("utf-8", errors="replace") if ( content_type.split(";", 1)[0].strip().lower() != "application/x-www-form-urlencoded" @@ -449,41 +525,45 @@ def _preview(raw: bytes, content_type: str = "") -> str: return "(omitted: unstructured body)" fields: Final = parse_qsl(text, keep_blank_values=True) return _safe_text( - urlencode( - tuple((key, REDACTED if _sensitive_field(key) else redact_string(value)) for key, value in fields) + _mask_known_values( + urlencode(tuple((key, REDACTED if _sensitive_field(key) else value) for key, value in fields)), secrets ) ) if not isinstance(parsed, (dict, list)): return "(omitted: unstructured body)" - return _safe_text(_redact_json(parsed)) + return _safe_text(redact_string(_mask_known_values(json.dumps(parsed, separators=(",", ":")), secrets))) def _masked_headers(headers: httpx.Headers) -> str: return _safe_text(", ".join(f"{name}={value}" for name, value in headers.items() if name in _SAFE_HEADER_NAMES)) -def _request_body_preview(request: httpx.Request) -> str: +def _request_body_preview(request: httpx.Request, secrets: tuple[str, ...] | None) -> str: try: - return _preview(request.content, request.headers.get("content-type", "")) + return _preview(request.content, request.headers.get("content-type", ""), secrets or ()) except httpx.RequestNotRead: return "(streamed, not captured)" -def _response_body_preview(response: httpx.Response) -> str: +def _response_body_preview(response: httpx.Response, secrets: tuple[str, ...] | None) -> str: + if secrets is None: + return "(omitted: request credentials unavailable)" captured: Final = response.extensions.get(_CAPTURE_EXTENSION) if isinstance(captured, str): return captured try: - return _preview(response.content, response.headers.get("content-type", "")) + return _preview(response.content, response.headers.get("content-type", ""), secrets) except httpx.ResponseNotRead: return "(not read)" -async def _read_error_prefix(chunks: AsyncIterator[bytes], remaining: int) -> bytes: - chunk: Final = await anext(chunks, b"") - if not chunk or len(chunk) >= remaining: - return chunk[:remaining] - return chunk + await _read_error_prefix(chunks, remaining - len(chunk)) +async def _read_error_prefix(chunks: AsyncIterator[bytes], limit: int) -> bytes: + buffer: Final = io.BytesIO() + async for chunk in chunks: + buffer.write(chunk[: limit - buffer.tell()]) + if buffer.tell() >= limit: + break + return buffer.getvalue() async def capture_upstream_error_response(response: httpx.Response) -> None: @@ -495,8 +575,13 @@ async def capture_upstream_error_response(response: httpx.Response) -> None: timeout=_CAPTURE_TIMEOUT_SECONDS, ) response._content = prefix # pyright: ignore[reportPrivateUsage] # rebind-ok: httpx has no public setter to retain consumed bytes for auth retries - preview: Final = _preview(prefix, response.headers.get("content-type", "")) - except (TimeoutError, httpx.HTTPError): + secrets: Final = _request_secret_values(response.request) + preview: Final = ( + _preview(prefix, response.headers.get("content-type", ""), secrets) + if secrets is not None + else "(omitted: request credentials unavailable)" + ) + except (TimeoutError, httpx.HTTPError, httpx.StreamError): response._content = b"" # pyright: ignore[reportPrivateUsage] # rebind-ok: httpx auth retries must survive diagnostic read failures response.extensions[_CAPTURE_EXTENSION] = ( "(unavailable: error body read failed)" # rebind-ok: httpx response hooks communicate through extensions @@ -510,15 +595,20 @@ def describe_upstream_response(response: httpx.Response) -> str: request: Final = response.request except RuntimeError: return f"HTTP {response.status_code} | request unavailable" + secrets: Final = _request_secret_values(request) return ( f"{_safe_text(request.method)} {safe_upstream_url(request.url)} -> HTTP {response.status_code}" f" | request headers: {_masked_headers(request.headers)}" - f" | request body: {_request_body_preview(request)}" - f" | response body: {_response_body_preview(response)}" + f" | request body: {_request_body_preview(request, secrets)}" + f" | response body: {_response_body_preview(response, secrets)}" ) def describe_upstream_http_failure(exc: BaseException) -> str | None: + from litellm.proxy._experimental.mcp_server.faults.traversal import ( # noqa: PLC0415 # fault package initialization imports the credential resolver + iter_exception_tree, + ) + lines: Final = tuple( describe_upstream_response(response) for current in islice(iter_exception_tree(exc), 16) diff --git a/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py b/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py index 513fc2f2036..43d97abe4db 100644 --- a/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py +++ b/litellm/proxy/_experimental/mcp_server/outbound_credentials/client_credentials.py @@ -38,11 +38,6 @@ from pydantic import BaseModel, ConfigDict, Field, SecretStr, TypeAdapter, Valid from typing_extensions import assert_never from litellm._logging import verbose_logger -from litellm.proxy._experimental.mcp_server.mcp_debug import ( - describe_upstream_http_failure, - describe_upstream_response, - safe_upstream_url, -) from litellm.proxy._experimental.mcp_server.outbound_credentials.oauth_token_store import ( InMemoryTokenCacheBackend, OAuthToken, @@ -107,6 +102,11 @@ async def post_client_credentials_grant( from litellm.llms.custom_httpx.http_handler import ( # noqa: PLC0415 # defer heavy handler import to call time get_async_httpx_client, # pyright: ignore[reportUnknownVariableType] # handler factory params are coarsely typed ) + from litellm.proxy._experimental.mcp_server.mcp_debug import ( # noqa: PLC0415 # diagnostics import credential enums through this package + describe_upstream_http_failure, + describe_upstream_response, + safe_upstream_url, + ) from litellm.types.llms.custom_http import httpxSpecialProvider # noqa: PLC0415 # deferred with the handler import try: diff --git a/tests/test_litellm/experimental_mcp_client/test_mcp_client.py b/tests/test_litellm/experimental_mcp_client/test_mcp_client.py index 713330a8280..3283af26cf7 100644 --- a/tests/test_litellm/experimental_mcp_client/test_mcp_client.py +++ b/tests/test_litellm/experimental_mcp_client/test_mcp_client.py @@ -1895,3 +1895,15 @@ async def test_optional_discovery_preserves_cancellation(method: str) -> None: task.cancel() with pytest.raises(asyncio.CancelledError): await asyncio.wait_for(task, timeout=3) + + + +def test_client_import_before_proxy_credentials_succeeds_in_fresh_process(): + import subprocess + + result = subprocess.run( + [sys.executable, "-c", "import litellm.experimental_mcp_client.client; from litellm.proxy._experimental.mcp_server.mcp_server_manager import MCPServerManager; print(MCPServerManager.__name__)"], + capture_output=True, text=True, timeout=60, check=False, + ) + assert result.returncode == 0, result.stderr + assert result.stdout.strip() == "MCPServerManager" diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index 0ed78fcfc9d..dced871e146 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -290,7 +290,7 @@ def test_failure_log_omits_unstructured_body(): @pytest.mark.asyncio -@pytest.mark.parametrize("mode", ["error", "large", "timeout", "read_failure", "success", "cancel"]) +@pytest.mark.parametrize("mode", ["error", "empty", "large", "timeout", "read_failure", "closed", "success", "cancel"]) async def test_error_capture_is_bounded_and_preserves_success_and_cancellation(mode): from litellm.proxy._experimental.mcp_server.mcp_debug import capture_upstream_error_response @@ -302,11 +302,13 @@ async def test_error_capture_is_bounded_and_preserves_success_and_cancellation(m self.reads += 1 if mode == "timeout": await asyncio.sleep(10) + if mode == "closed": + raise httpx.StreamClosed() if mode == "read_failure": raise httpx.ReadError("private-read-error") if mode == "cancel": raise asyncio.CancelledError - yield b'{"error":"missing_scope","password":"first second"}' if mode != "large" else b"x" * 20000 + yield b"" if mode == "empty" else b'{"error":"missing_scope","password":"first second"}' if mode != "large" else b"x" * 20000 stream = Stream() request = httpx.Request("POST", "https://upstream/mcp") @@ -323,7 +325,7 @@ async def test_error_capture_is_bounded_and_preserves_success_and_cancellation(m detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) assert detail is not None assert "first" not in detail and "second" not in detail and "private-read-error" not in detail - expected = {"error": "missing_scope", "large": "capture limit", "timeout": "read failed", "read_failure": "read failed"} + expected = {"empty": "(empty)", "error": "missing_scope", "large": "capture limit", "timeout": "read failed", "read_failure": "read failed", "closed":"read failed"} assert expected[mode] in detail if mode == "error": assert await response.aread() == b'{"error":"missing_scope","password":"first second"}' @@ -381,13 +383,12 @@ def test_failure_diagnostics_without_request_and_with_streamed_request(): def test_deep_error_body_is_bounded_without_exposing_nested_values(): - import json body = b'{"nested":' * 18 + b'{"password":"hidden-value"}' + b'}' * 18 request = httpx.Request("POST", "https://upstream/mcp", content=body) response = httpx.Response(500, request=request, content=body) detail = describe_upstream_http_failure(httpx.HTTPStatusError("failure", request=request, response=response)) - assert detail is not None and "depth limit" in detail and "hidden-value" not in detail - assert json.loads(detail.split("response body: ")[1])["nested"] + assert detail is not None and "hidden-value" not in detail + assert "nested" in detail and "REDACTED" in detail @pytest.mark.parametrize("body", [b'client%5Fsecret=first+second&client_id=visible', b'client_secret=first%26second&client_id=visible']) @@ -485,3 +486,62 @@ async def test_concurrent_mcp_messages_record_on_their_own_http_scope() -> None: await asyncio.gather(record(first, AuthResolution.stored_user_token), record(second, AuthResolution.per_request_header)) assert first.resolution() == "stored-user-token" assert second.resolution() == "per-request-header" + + +@pytest.mark.parametrize("source", ["header", "bearer", "basic", "cookie", "query", "form", "json"]) +def test_reflected_credentials_are_removed_from_normal_response_fields(source): + import base64 + + secret = "generic-credential-123" + headers = {"X-Custom":secret} if source == "header" else {"Authorization":"Bearer " + secret} if source == "bearer" else {"Authorization":"Basic " + base64.b64encode(("client:" + secret).encode()).decode()} if source == "basic" else {"Cookie":"session=" + secret} if source == "cookie" else {} + request = httpx.Request("POST", "https://upstream/token" + ("?credential=" + secret if source == "query" else ""), + headers=headers, data={"client_secret":secret} if source == "form" else None, + json={"nested":{"client_secret":secret}} if source == "json" else None) + response = httpx.Response(401, request=request, json={"error":"invalid_client", "error_description":"Rejected " + secret}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "invalid_client" in detail + assert secret not in detail and "REDACTED" in detail + + + +@pytest.mark.parametrize("secret", ['value"with\ncharacters€', "R"]) +def test_reflected_values_are_redacted_before_truncation_without_expanding_replacements(secret): + request = httpx.Request("POST", "https://upstream/token", json={"client_secret":secret}) + response = httpx.Response(401, request=request, json={"error":"invalid_client", "detail":"x" * 460 + secret}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "invalid_client" in detail + assert "value" not in detail and "characters" not in detail and len(detail) < 1400 + + +@pytest.mark.parametrize("headers", [{"Authorization":"Basic !!!"}, {"Cookie":"bad@key=opaque"}]) +def test_malformed_auth_headers_do_not_break_failure_diagnostics(headers): + request = httpx.Request("POST", "https://upstream/token", headers=headers) + response = httpx.Response(401, request=request, json={"error":"invalid_client"}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "invalid_client" in detail + assert "!!!" not in detail and "opaque" not in detail + + +def test_oversized_request_omits_potentially_reflected_response_credentials(): + request = httpx.Request("POST", "https://upstream/token", content=b"x" * 17000) + response = httpx.Response(401, request=request, json={"error_description":"unknown-secret"}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "capture limit" in detail and "credentials unavailable" in detail + assert "unknown-secret" not in detail + + + +@pytest.mark.asyncio +async def test_streamed_error_redacts_reflected_credentials_before_capture(): + import json + from litellm.proxy._experimental.mcp_server.mcp_debug import capture_upstream_error_response + + secret = "generic-credential-123" + request = httpx.Request("POST", "https://upstream/token", data={"client_secret":secret}) + raw = json.dumps({"error":"invalid_client", "error_description":"Rejected " + secret}).encode() + response = httpx.Response(401, request=request, stream=httpx.ByteStream(raw)) + await capture_upstream_error_response(response) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "invalid_client" in detail and "Rejected" in detail + assert secret not in detail and "REDACTED" in detail + assert await response.aread() == raw From 5b13dfcc59b51d637d98bdf6363da0f2e13e6e53 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 07:27:10 -0700 Subject: [PATCH 070/157] fix(mcp): omit credential-bearing paths from failure logs --- .../proxy/_experimental/mcp_server/mcp_debug.py | 2 +- .../test_client_credentials.py | 4 ++-- .../_experimental/mcp_server/test_mcp_debug.py | 14 +++++++++++++- .../mcp_server/test_mcp_oauth_passthrough_tools.py | 4 ++-- 4 files changed, 18 insertions(+), 6 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_debug.py b/litellm/proxy/_experimental/mcp_server/mcp_debug.py index ccce0aaf615..06b17bc59c9 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_debug.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_debug.py @@ -410,7 +410,7 @@ def _safe_text(value: str, limit: int = _BODY_PREVIEW_CHARS) -> str: def safe_upstream_url(url: httpx.URL) -> str: - return _safe_text(str(url.copy_with(username="", password="", query=None, fragment=None))) + return _safe_text(str(url.copy_with(username="", password="", path="/", query=None, fragment=None))) def _sensitive_field(key: str) -> bool: diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py b/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py index 1550b812a87..774cd022703 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/outbound_credentials/test_client_credentials.py @@ -518,7 +518,7 @@ async def test_token_exchange_failure_diagnostics(mode, monkeypatch, caplog): assert not caplog.text elif mode in {"timeout", "connect"}: assert isinstance(result, TokenEndpointUnreachable) - assert "POST https://idp/token failed" in caplog.text + assert "POST https://idp/ failed" in caplog.text else: - assert "POST https://idp/token -> HTTP" in caplog.text + assert "POST https://idp/ -> HTTP" in caplog.text assert {"denied":"denied", "invalid":"invalid response", "missing":"no access token"}[mode] in caplog.text diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index dced871e146..14e14882f3f 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -233,7 +233,7 @@ class TestDescribeUpstreamHttpFailure: ) described = describe_upstream_http_failure(exc) assert described is not None - assert "POST https://upstream.example/apis/mcp -> HTTP 500" in described + assert "POST https://upstream.example/ -> HTTP 500" in described assert '{"method":"initialize"' in described assert 'response body: {"error":"boom"}' in described @@ -545,3 +545,15 @@ async def test_streamed_error_redacts_reflected_credentials_before_capture(): assert detail is not None and "invalid_client" in detail and "Rejected" in detail assert secret not in detail and "REDACTED" in detail assert await response.aread() == raw + + +@pytest.mark.parametrize("path", ["/credential-path-value/mcp", "/oauth/credential-path-value/token"]) +def test_failure_diagnostics_omit_credential_bearing_url_paths(path): + request = httpx.Request("POST", "https://upstream.example" + path) + response = httpx.Response(401, request=request, json={"error": "access_denied"}) + error = httpx.HTTPStatusError("denied", request=request, response=response) + diagnostic = describe_upstream_http_failure(error) + assert diagnostic is not None + assert "credential-path-value" not in diagnostic + assert "POST https://upstream.example/ -> HTTP 401" in diagnostic + assert "access_denied" in diagnostic diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py index f8e4cd0c6ae..3f5d4ad83ea 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py @@ -456,7 +456,7 @@ async def test_fetch_tools_logs_upstream_request_details_on_500(caplog): with pytest.raises(MCPServerListError): await manager._fetch_tools_with_timeout(mock_client, "sample_docs") - assert "POST https://upstream/apis/mcp -> HTTP 500" in caplog.text + assert "POST https://upstream/ -> HTTP 500" in caplog.text assert '"method":"initialize"' in caplog.text assert "upstream-token-0123456789" not in caplog.text @@ -473,5 +473,5 @@ async def test_client_creation_failure_logs_sanitized_exchange(monkeypatch, capl with caplog.at_level(logging.WARNING, logger="LiteLLM"): with pytest.raises(MCPServerListError): await manager._get_tools_from_server(server) - assert "POST https://upstream/mcp -> HTTP 500" in caplog.text + assert "POST https://upstream/ -> HTTP 500" in caplog.text assert "missing_scope" in caplog.text and "query-secret" not in caplog.text From e8c411fb43d0662be9d6a3795a22e383a2bb4fef Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 07:32:38 -0700 Subject: [PATCH 071/157] test(mcp): cover deeply nested credential inspection limits --- .../proxy/_experimental/mcp_server/test_mcp_debug.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index 14e14882f3f..5cd5b0aa81e 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -557,3 +557,15 @@ def test_failure_diagnostics_omit_credential_bearing_url_paths(path): assert "credential-path-value" not in diagnostic assert "POST https://upstream.example/ -> HTTP 401" in diagnostic assert "access_denied" in diagnostic + + +def test_deep_request_omits_response_when_credentials_cannot_be_inspected(): + from litellm.proxy._experimental.mcp_server.utils import MAX_STRUCTURED_CONTENT_SCAN_DEPTH + + raw = "[" * (MAX_STRUCTURED_CONTENT_SCAN_DEPTH + 1) + '{"client_secret":"nested-credential"}' + "]" * (MAX_STRUCTURED_CONTENT_SCAN_DEPTH + 1) + request = httpx.Request("POST", "https://upstream/token", content=raw, headers={"Content-Type": "application/json"}) + response = httpx.Response(401, request=request, json={"error_description": "Rejected nested-credential"}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "HTTP 401" in detail + assert "response body: (omitted: request credentials unavailable)" in detail + assert "nested-credential" not in detail From 5c190e69bf058afb991cd79fca5625fda88e715c Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 08:00:39 -0700 Subject: [PATCH 072/157] fix(mcp): preserve timeout fallback on Python 3.10 --- .github/workflows/test-code-quality.yml | 6 ++++++ .../proxy/_experimental/mcp_server/mcp_debug.py | 2 +- .../_experimental/mcp_server/test_mcp_debug.py | 15 ++++++++++++--- 3 files changed, 19 insertions(+), 4 deletions(-) diff --git a/.github/workflows/test-code-quality.yml b/.github/workflows/test-code-quality.yml index e6d2264fbf0..4abcc47df5e 100644 --- a/.github/workflows/test-code-quality.yml +++ b/.github/workflows/test-code-quality.yml @@ -187,3 +187,9 @@ jobs: - name: Check litellm CLI run: uv run --no-sync litellm --version + + - name: Verify MCP timeout fallback and auth retries on Python 3.10 + run: >- + uv run --no-sync pytest --noconftest + tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py + -k error_capture -q diff --git a/litellm/proxy/_experimental/mcp_server/mcp_debug.py b/litellm/proxy/_experimental/mcp_server/mcp_debug.py index 06b17bc59c9..0ca17d1c336 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_debug.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_debug.py @@ -581,7 +581,7 @@ async def capture_upstream_error_response(response: httpx.Response) -> None: if secrets is not None else "(omitted: request credentials unavailable)" ) - except (TimeoutError, httpx.HTTPError, httpx.StreamError): + except (asyncio.TimeoutError, httpx.HTTPError, httpx.StreamError): response._content = b"" # pyright: ignore[reportPrivateUsage] # rebind-ok: httpx auth retries must survive diagnostic read failures response.extensions[_CAPTURE_EXTENSION] = ( "(unavailable: error body read failed)" # rebind-ok: httpx response hooks communicate through extensions diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index 5cd5b0aa81e..88de9df5d39 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -349,7 +349,8 @@ def test_failure_preview_handles_empty_scalar_control_and_long_bodies(body): @pytest.mark.asyncio -async def test_error_capture_preserves_httpx_auth_retry(): +@pytest.mark.parametrize("slow_error", [False, True]) +async def test_error_capture_preserves_httpx_auth_retry(slow_error): from litellm.proxy._experimental.mcp_server.mcp_debug import capture_upstream_error_response class RetryAuth(httpx.Auth): @@ -359,16 +360,24 @@ async def test_error_capture_preserves_httpx_auth_retry(): request.headers["Authorization"] = "Bearer refreshed" yield request + class SlowStream(httpx.AsyncByteStream): + async def __aiter__(self): + await asyncio.sleep(10) + yield b'{"error":"expired_token"}' + def upstream(request): if request.headers.get("Authorization"): return httpx.Response(200, json={"ok": True}) - return httpx.Response(401, json={"error":"expired_token"}) + return httpx.Response(401, stream=SlowStream()) if slow_error else httpx.Response(401, json={"error":"expired_token"}) async with httpx.AsyncClient(transport=httpx.MockTransport(upstream), auth=RetryAuth(), event_hooks={"response":[capture_upstream_error_response]}) as client: response = await client.get("https://upstream/mcp") assert response.status_code == 200 and response.json() == {"ok":True} - assert response.history[0].json() == {"error":"expired_token"} + if slow_error: + assert response.history[0].content == b"" + else: + assert response.history[0].json() == {"error":"expired_token"} def test_failure_diagnostics_without_request_and_with_streamed_request(): From 3883a891f0809a4358ca4b35ec974a2347c29b74 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 08:10:19 -0700 Subject: [PATCH 073/157] fix(mcp): redact compact credential field names --- litellm/proxy/_experimental/mcp_server/mcp_debug.py | 5 ++++- .../_experimental/mcp_server/test_mcp_debug.py | 13 +++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/_experimental/mcp_server/mcp_debug.py b/litellm/proxy/_experimental/mcp_server/mcp_debug.py index 0ca17d1c336..1f157aefdc3 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_debug.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_debug.py @@ -414,7 +414,10 @@ def safe_upstream_url(url: httpx.URL) -> str: def _sensitive_field(key: str) -> bool: - return key.lower() in ("code", "cookie", "client_assertion") or _LOG_MASKER.is_sensitive_key(key) + normalized: Final = re.sub(r"[^a-z0-9]", "", key.casefold()) + return normalized in ("code", "clientassertion") or any( + pattern in normalized for pattern in _LOG_MASKER.sensitive_patterns + ) def _redact_object( diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py index 88de9df5d39..b6535e6326a 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py @@ -578,3 +578,16 @@ def test_deep_request_omits_response_when_credentials_cannot_be_inspected(): assert detail is not None and "HTTP 401" in detail assert "response body: (omitted: request credentials unavailable)" in detail assert "nested-credential" not in detail + + +@pytest.mark.parametrize("field", ["accessToken", "refreshToken", "clientSecret", "apikey", "CLIENTASSERTION", "cost_token"]) +@pytest.mark.parametrize("encoding", ["json", "form"]) +def test_compact_credential_fields_and_reflected_values_are_redacted(field, encoding): + secret = "generic-private-value" + fields = {field: secret} + request = httpx.Request("POST", "https://upstream/token", json=fields if encoding == "json" else None, + data=fields if encoding == "form" else None) + response = httpx.Response(401, request=request, json={field: secret, "error": "invalid_client", "detail": "Rejected " + secret}) + detail = describe_upstream_http_failure(httpx.HTTPStatusError("failed", request=request, response=response)) + assert detail is not None and "invalid_client" in detail + assert "REDACTED" in detail and secret not in detail From 43b56e8707355b135b94eb7a1561d4d2786d26d4 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 15:36:39 +0000 Subject: [PATCH 074/157] fix(cost-map): advertise xhigh reasoning effort on bedrock-hosted openai gpt rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 20 +++++++++++++++++++ model_prices_and_context_window.json | 20 +++++++++++++++++++ .../test_litellm/test_model_prices_schema.py | 19 ++++++++++++++++++ 3 files changed, 59 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ad992ff92eb..0b4a7c08884 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -55278,6 +55278,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/openai.gpt-5.6-terra": { @@ -55317,6 +55318,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/openai.gpt-5.6-cyber": { @@ -55345,6 +55347,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "bedrock_mantle/openai.gpt-daybreak-blue-5.6-sol": { @@ -55377,6 +55380,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-daybreak-blue-56-sol.html" }, @@ -55417,6 +55421,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "us.openai.gpt-5.6-sol": { @@ -55443,6 +55448,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "global.openai.gpt-5.6-sol": { @@ -55469,6 +55475,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "us.openai.gpt-5.6-terra": { @@ -55495,6 +55502,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "global.openai.gpt-5.6-terra": { @@ -55521,6 +55529,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "us.openai.gpt-5.6-luna": { @@ -55547,6 +55556,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "global.openai.gpt-5.6-luna": { @@ -55573,6 +55583,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "bedrock_mantle/openai.gpt-6-astra": { @@ -55606,6 +55617,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html" }, @@ -55634,6 +55646,7 @@ "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html" }, @@ -55662,6 +55675,7 @@ "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html" }, @@ -55699,6 +55713,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/openai.gpt-5.4": { @@ -55735,6 +55750,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/google.gemma-4-31b": { @@ -61170,6 +61186,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 2.64e-06, "input_cost_per_token_above_272k_tokens": 5.28e-06, @@ -61203,6 +61220,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 2.64e-07, "input_cost_per_token_above_272k_tokens": 5.28e-07, @@ -61235,6 +61253,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 3.3e-06, "cache_read_input_token_cost": 3.3e-07, @@ -61397,6 +61416,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 3.3e-06, "cache_read_input_token_cost": 3.3e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ad992ff92eb..0b4a7c08884 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -55278,6 +55278,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/openai.gpt-5.6-terra": { @@ -55317,6 +55318,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/openai.gpt-5.6-cyber": { @@ -55345,6 +55347,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "bedrock_mantle/openai.gpt-daybreak-blue-5.6-sol": { @@ -55377,6 +55380,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-daybreak-blue-56-sol.html" }, @@ -55417,6 +55421,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "us.openai.gpt-5.6-sol": { @@ -55443,6 +55448,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "global.openai.gpt-5.6-sol": { @@ -55469,6 +55475,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "us.openai.gpt-5.6-terra": { @@ -55495,6 +55502,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "global.openai.gpt-5.6-terra": { @@ -55521,6 +55529,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "us.openai.gpt-5.6-luna": { @@ -55547,6 +55556,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "global.openai.gpt-5.6-luna": { @@ -55573,6 +55583,7 @@ "supports_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true }, "bedrock_mantle/openai.gpt-6-astra": { @@ -55606,6 +55617,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html" }, @@ -55634,6 +55646,7 @@ "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html" }, @@ -55662,6 +55675,7 @@ "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html" }, @@ -55699,6 +55713,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/openai.gpt-5.4": { @@ -55735,6 +55750,7 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, + "supports_xhigh_reasoning_effort": true, "supports_web_search": true }, "bedrock_mantle/google.gemma-4-31b": { @@ -61170,6 +61186,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 2.64e-06, "input_cost_per_token_above_272k_tokens": 5.28e-06, @@ -61203,6 +61220,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 2.64e-07, "input_cost_per_token_above_272k_tokens": 5.28e-07, @@ -61235,6 +61253,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 3.3e-06, "cache_read_input_token_cost": 3.3e-07, @@ -61397,6 +61416,7 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, + "supports_xhigh_reasoning_effort": true, "supports_vision": true, "input_cost_per_token": 3.3e-06, "cache_read_input_token_cost": 3.3e-07, diff --git a/tests/test_litellm/test_model_prices_schema.py b/tests/test_litellm/test_model_prices_schema.py index c2c22c25998..7f1db44e71b 100644 --- a/tests/test_litellm/test_model_prices_schema.py +++ b/tests/test_litellm/test_model_prices_schema.py @@ -4,6 +4,7 @@ import importlib.util import json import re from pathlib import Path +from typing import Final import jsonschema import pytest @@ -217,3 +218,21 @@ def test_chat_latest_declares_the_one_effort_openai_accepts(prices: dict): with no declared levels resolves to None, which lets /model_group/info and the dashboard offer levels the upstream will 400 on.""" assert resolve_supported_reasoning_efforts(prices["chat-latest"], deployment_is_mapped=True) == ("medium",) + + +BEDROCK_OPENAI_XHIGH_MARKERS: Final = ("openai.gpt-5.4", "openai.gpt-5.5", "openai.gpt-5.6", "openai.gpt-6-astra") +BEDROCK_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse", "bedrock_mantle")) + + +def test_bedrock_openai_gpt_rows_advertise_xhigh_like_their_openai_twins(prices: dict): + """Bedrock forwards reasoning_effort to these models unchanged, and xhigh is opt-in for the + capability resolver, so a row without the flag drops xhigh from every group it belongs to.""" + missing = [ + name + for name, entry in prices.items() + if isinstance(entry, dict) + and entry.get("litellm_provider") in BEDROCK_PROVIDERS + and any(marker in name for marker in BEDROCK_OPENAI_XHIGH_MARKERS) + and "xhigh" not in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) + ] + assert missing == [] From 4422be0f28e7599859f140a3a2859572b4941a01 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 15:52:43 +0000 Subject: [PATCH 075/157] fix(cost-map): declare minimal unsupported on bedrock-hosted openai gpt rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 20 +++++++++++++++++++ model_prices_and_context_window.json | 20 +++++++++++++++++++ .../test_litellm/test_model_prices_schema.py | 14 ++++++++----- 3 files changed, 49 insertions(+), 5 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 0b4a7c08884..1c86dbaacff 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -55273,6 +55273,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55313,6 +55314,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55343,6 +55345,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55376,6 +55379,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55416,6 +55420,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55446,6 +55451,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55473,6 +55479,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55500,6 +55507,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55527,6 +55535,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55554,6 +55563,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55581,6 +55591,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55613,6 +55624,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55643,6 +55655,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55672,6 +55685,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55708,6 +55722,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55745,6 +55760,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61182,6 +61198,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61216,6 +61233,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61249,6 +61267,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61412,6 +61431,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 0b4a7c08884..1c86dbaacff 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -55273,6 +55273,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55313,6 +55314,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55343,6 +55345,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55376,6 +55379,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55416,6 +55420,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55446,6 +55451,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55473,6 +55479,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55500,6 +55507,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55527,6 +55535,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55554,6 +55563,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55581,6 +55591,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, @@ -55613,6 +55624,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55643,6 +55655,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55672,6 +55685,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55708,6 +55722,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55745,6 +55760,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61182,6 +61198,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61216,6 +61233,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61249,6 +61267,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -61412,6 +61431,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/tests/test_litellm/test_model_prices_schema.py b/tests/test_litellm/test_model_prices_schema.py index 7f1db44e71b..69ee999f066 100644 --- a/tests/test_litellm/test_model_prices_schema.py +++ b/tests/test_litellm/test_model_prices_schema.py @@ -224,15 +224,19 @@ BEDROCK_OPENAI_XHIGH_MARKERS: Final = ("openai.gpt-5.4", "openai.gpt-5.5", "open BEDROCK_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse", "bedrock_mantle")) -def test_bedrock_openai_gpt_rows_advertise_xhigh_like_their_openai_twins(prices: dict): +def test_bedrock_openai_gpt_rows_mirror_their_openai_twins_effort_ladder(prices: dict): """Bedrock forwards reasoning_effort to these models unchanged, and xhigh is opt-in for the - capability resolver, so a row without the flag drops xhigh from every group it belongs to.""" - missing = [ + capability resolver, so a row without the flag drops xhigh from every group it belongs to. + The OpenAI twins reject minimal, and Bedrock forwards reasoning_effort unchanged.""" + mismatched = [ name for name, entry in prices.items() if isinstance(entry, dict) and entry.get("litellm_provider") in BEDROCK_PROVIDERS and any(marker in name for marker in BEDROCK_OPENAI_XHIGH_MARKERS) - and "xhigh" not in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) + and ( + "xhigh" not in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) + or "minimal" in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) + ) ] - assert missing == [] + assert mismatched == [] From 729ea6b8325df376d84b8ed4753e39903b9f7b64 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 09:48:53 -0700 Subject: [PATCH 076/157] perf(proxy): lazy-load provider passthrough routes (#40691) Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/_lazy_features.py | 98 +- litellm/proxy/_lazy_openapi_snapshot.json | 4986 ++++++++++++++++- .../llm_passthrough_endpoints.py | 6 - .../openai_passthrough_endpoints.py | 44 + litellm/proxy/proxy_server.py | 21 +- .../test_llm_pass_through_endpoints.py | 49 +- tests/test_litellm/proxy/test_proxy_server.py | 120 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 387 +- 8 files changed, 5317 insertions(+), 394 deletions(-) create mode 100644 litellm/proxy/pass_through_endpoints/openai_passthrough_endpoints.py diff --git a/litellm/proxy/_lazy_features.py b/litellm/proxy/_lazy_features.py index 50e0a961a49..1d1b736c9fc 100644 --- a/litellm/proxy/_lazy_features.py +++ b/litellm/proxy/_lazy_features.py @@ -8,11 +8,13 @@ omits each feature's routes until the feature is warmed. import asyncio import importlib -from collections.abc import Callable +from collections.abc import Callable, Mapping, Sequence from collections.abc import Set as AbstractSet from dataclasses import dataclass, field +from types import MappingProxyType from typing import TYPE_CHECKING, Final +from starlette.routing import BaseRoute, Match from starlette.types import Receive, Scope, Send from litellm._logging import verbose_proxy_logger @@ -185,6 +187,31 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = ( module_path="litellm.proxy.management_endpoints.config_override_endpoints", path_prefixes=("/config_overrides",), ), + LazyFeature( + name="llm_passthrough", + module_path="litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints", + path_prefixes=( + "/anthropic/", + "/assemblyai/", + "/azure/", + "/azure_ai/", + "/bedrock/", + "/cohere/", + "/comprehendmedical", + "/cursor/", + "/eu.assemblyai/", + "/gemini/", + "/gigachat/", + "/milvus/", + "/mistral/", + "/openai/", + "/openai_passthrough/", + "/vertex-ai/", + "/vertex_ai/", + "/vllm/", + "/watsonx/", + ), + ), LazyFeature( name="realtime", module_path="litellm.proxy.realtime_endpoints.endpoints", @@ -308,14 +335,64 @@ class LazyFeatureMiddleware: if root_path and path.startswith(root_path + "/"): path = path[len(root_path) :] # rebind-ok: local strip after the boundary check above for feat in self._features: - if feat.module_path in self._loaded: + if feat.module_path in self._loaded or not feat.matches(path): continue - if feat.matches(path): - await _force_load(self._fastapi_app, feat) + if _eager_route_wins(self._fastapi_app, feat, scope): + continue + await _force_load(self._fastapi_app, feat, self._features) await self.app(scope, receive, send) -async def _force_load(app: "FastAPI", feat: LazyFeature) -> bool: +def _lazy_slots(app: "FastAPI") -> Mapping[str, int]: + return app.state.lazy_slots if hasattr(app.state, "lazy_slots") else MappingProxyType({}) + + +def reserve_lazy_slot(app: "FastAPI", name: str, features: tuple[LazyFeature, ...] = LAZY_FEATURES) -> None: + """Record the table position the feature's router used to be included at, so its + routes are spliced back in there once it loads and keep the same precedence.""" + feat: Final = next(f for f in features if f.name == name) + app.state.lazy_slots = MappingProxyType({**_lazy_slots(app), feat.module_path: len(app.router.routes)}) + + +def _eager_route_wins(app: "FastAPI", feat: LazyFeature, scope: Scope) -> bool: + """Routes ahead of a feature's reserved slot beat its routes in Starlette's scan, + so a request one of them fully matches never needs the feature loaded.""" + slot: Final = _lazy_slots(app).get(feat.module_path) + if slot is None: + return False + return any(route.matches(scope)[0] is Match.FULL for route in app.router.routes[:slot]) + + +def _in_registry_order( + routes: Sequence[BaseRoute], + lazy_routes: Mapping[str, tuple[BaseRoute, ...]], + features: tuple[LazyFeature, ...], + slots: Mapping[str, int], +) -> tuple[BaseRoute, ...]: + """Lazy routers land in registry order, not first-request order, so overlapping + paths (/openai/{endpoint:path} vs /openai/v1/realtime/calls) resolve the same + way no matter which feature a deployment happens to hit first. Features with a + reserved slot go back where they were eagerly included; the rest follow every + eager route.""" + rank: Final = MappingProxyType({f.module_path: i for i, f in enumerate(features)}) + modules: Final = tuple(sorted(lazy_routes, key=lambda m: rank.get(m, len(rank)))) + lazy_ids: Final = frozenset(id(route) for module_path in modules for route in lazy_routes[module_path]) + eager: Final = tuple(route for route in routes if id(route) not in lazy_ids) + + def slot_of(module_path: str) -> int: + return min(slots.get(module_path, len(eager)), len(eager)) + + return tuple( + route + for index in range(len(eager) + 1) + for route in ( + *(r for module_path in modules if slot_of(module_path) == index for r in lazy_routes[module_path]), + *eager[index : index + 1], + ) + ) + + +async def _force_load(app: "FastAPI", feat: LazyFeature, features: tuple[LazyFeature, ...] = LAZY_FEATURES) -> bool: """Import + register a lazy feature exactly once per (app, module). Shared by the middleware and the /lazy/warm endpoint.""" if not hasattr(app.state, "lazy_loaded"): @@ -330,7 +407,18 @@ async def _force_load(app: "FastAPI", feat: LazyFeature) -> bool: # mutates app.router.routes, so it stays on the loop thread. loop: Final = asyncio.get_running_loop() module: Final = await loop.run_in_executor(None, importlib.import_module, feat.module_path) + before: Final = len(app.router.routes) feat.register_fn(app, module) + previous: Final[Mapping[str, tuple[BaseRoute, ...]]] = ( + app.state.lazy_routes if hasattr(app.state, "lazy_routes") else MappingProxyType({}) + ) + lazy_routes: Final[Mapping[str, tuple[BaseRoute, ...]]] = MappingProxyType( + {**previous, feat.module_path: tuple(app.router.routes[before:])} + ) + app.state.lazy_routes = lazy_routes # rebind-ok: the app owns the record of which routes each feature added + app.router.routes[:] = _in_registry_order( # rebind-ok: the app owns its route table + app.router.routes, lazy_routes, features, _lazy_slots(app) + ) app.state.lazy_loaded.add(feat.module_path) app.openapi_schema = None verbose_proxy_logger.info( diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index f0af17ab818..53af85baac6 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -4335,7 +4335,7 @@ "/anthropic/{endpoint}": { "delete": { "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", - "operationId": "anthropic_proxy_route_anthropic__endpoint__delete", + "operationId": "anthropic_proxy_route_anthropic__endpoint__delete_2", "parameters": [ { "in": "path", @@ -4379,7 +4379,7 @@ }, "get": { "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", - "operationId": "anthropic_proxy_route_anthropic__endpoint__get", + "operationId": "anthropic_proxy_route_anthropic__endpoint__get_2", "parameters": [ { "in": "path", @@ -4423,7 +4423,7 @@ }, "patch": { "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", - "operationId": "anthropic_proxy_route_anthropic__endpoint__patch", + "operationId": "anthropic_proxy_route_anthropic__endpoint__patch_2", "parameters": [ { "in": "path", @@ -4467,7 +4467,7 @@ }, "post": { "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", - "operationId": "anthropic_proxy_route_anthropic__endpoint__post", + "operationId": "anthropic_proxy_route_anthropic__endpoint__post_2", "parameters": [ { "in": "path", @@ -4511,7 +4511,7 @@ }, "put": { "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", - "operationId": "anthropic_proxy_route_anthropic__endpoint__put", + "operationId": "anthropic_proxy_route_anthropic__endpoint__put_2", "parameters": [ { "in": "path", @@ -15836,6 +15836,4976 @@ } } }, + "llm_passthrough": { + "components": { + "schemas": { + "Body_image_edit_api_openai_deployments__model__images_edits_post": { + "properties": { + "image": { + "anyOf": [ + { + "items": { + "contentMediaType": "application/octet-stream", + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Image" + }, + "image[]": { + "anyOf": [ + { + "items": { + "contentMediaType": "application/octet-stream", + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Image[]" + }, + "mask": { + "anyOf": [ + { + "items": { + "contentMediaType": "application/octet-stream", + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Mask" + }, + "mask[]": { + "anyOf": [ + { + "items": { + "contentMediaType": "application/octet-stream", + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Mask[]" + } + }, + "title": "Body_image_edit_api_openai_deployments__model__images_edits_post", + "type": "object" + }, + "ErrorResponse": { + "properties": { + "detail": { + "additionalProperties": true, + "example": { + "error": { + "code": "error_code", + "message": "Error message", + "param": "error_param", + "type": "error_type" + } + }, + "title": "Detail", + "type": "object" + } + }, + "required": [ + "detail" + ], + "title": "ErrorResponse", + "type": "object" + }, + "HTTPValidationError": { + "properties": { + "detail": { + "items": { + "$ref": "#/components/schemas/ValidationError" + }, + "title": "Detail", + "type": "array" + } + }, + "title": "HTTPValidationError", + "type": "object" + }, + "RealtimeClientSecretResponse": { + "description": "Response from POST /v1/realtime/client_secrets.\n\nBoth the top-level `value` and `session.client_secret.value`\nwill contain the encrypted token instead of the raw ephemeral key.\nThe `session` field is kept as a raw dict so unknown fields pass through.", + "properties": { + "expires_at": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Expires At" + }, + "session": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Session" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "value" + ], + "title": "RealtimeClientSecretResponse", + "type": "object" + }, + "RealtimeTranscriptionSessionResponse": { + "additionalProperties": true, + "description": "Response from POST /v1/realtime/transcription_sessions.\n\n`client_secret.value` contains the encrypted token instead of the raw\nephemeral key. Unknown fields pass through unchanged.", + "properties": { + "client_secret": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Client Secret" + } + }, + "title": "RealtimeTranscriptionSessionResponse", + "type": "object" + }, + "ValidationError": { + "properties": { + "ctx": { + "title": "Context", + "type": "object" + }, + "input": { + "title": "Input" + }, + "loc": { + "items": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "integer" + } + ] + }, + "title": "Location", + "type": "array" + }, + "msg": { + "title": "Message", + "type": "string" + }, + "type": { + "title": "Error Type", + "type": "string" + } + }, + "required": [ + "loc", + "msg", + "type" + ], + "title": "ValidationError", + "type": "object" + } + } + }, + "paths": { + "/anthropic/{endpoint}": { + "delete": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", + "operationId": "anthropic_proxy_route_anthropic__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Anthropic Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", + "operationId": "anthropic_proxy_route_anthropic__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Anthropic Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", + "operationId": "anthropic_proxy_route_anthropic__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Anthropic Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", + "operationId": "anthropic_proxy_route_anthropic__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Anthropic Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion)", + "operationId": "anthropic_proxy_route_anthropic__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Anthropic Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/assemblyai/{endpoint}": { + "delete": { + "operationId": "assemblyai_proxy_route_assemblyai__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "operationId": "assemblyai_proxy_route_assemblyai__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "operationId": "assemblyai_proxy_route_assemblyai__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "operationId": "assemblyai_proxy_route_assemblyai__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "operationId": "assemblyai_proxy_route_assemblyai__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/azure/{endpoint}": { + "delete": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/azure_ai/{endpoint}": { + "delete": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure_ai__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure_ai__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure_ai__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure_ai__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Call any azure endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/azure/{endpoint:path}`\n\nChecks if the deployment id in the url is a litellm model name. If so, it will route using the llm_router.allm_passthrough_route.", + "operationId": "azure_proxy_route_azure_ai__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Azure Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/bedrock/{endpoint}": { + "delete": { + "description": "This is the v1 passthrough for Bedrock.\nV2 is handled by the `/bedrock/v2` endpoint.\n[Docs](https://docs.litellm.ai/docs/pass_through/bedrock)", + "operationId": "bedrock_proxy_route_bedrock__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Bedrock Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "This is the v1 passthrough for Bedrock.\nV2 is handled by the `/bedrock/v2` endpoint.\n[Docs](https://docs.litellm.ai/docs/pass_through/bedrock)", + "operationId": "bedrock_proxy_route_bedrock__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Bedrock Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "This is the v1 passthrough for Bedrock.\nV2 is handled by the `/bedrock/v2` endpoint.\n[Docs](https://docs.litellm.ai/docs/pass_through/bedrock)", + "operationId": "bedrock_proxy_route_bedrock__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Bedrock Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "This is the v1 passthrough for Bedrock.\nV2 is handled by the `/bedrock/v2` endpoint.\n[Docs](https://docs.litellm.ai/docs/pass_through/bedrock)", + "operationId": "bedrock_proxy_route_bedrock__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Bedrock Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "This is the v1 passthrough for Bedrock.\nV2 is handled by the `/bedrock/v2` endpoint.\n[Docs](https://docs.litellm.ai/docs/pass_through/bedrock)", + "operationId": "bedrock_proxy_route_bedrock__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Bedrock Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/cohere/{endpoint}": { + "delete": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/cohere)", + "operationId": "cohere_proxy_route_cohere__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cohere Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/cohere)", + "operationId": "cohere_proxy_route_cohere__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cohere Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/cohere)", + "operationId": "cohere_proxy_route_cohere__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cohere Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/cohere)", + "operationId": "cohere_proxy_route_cohere__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cohere Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/cohere)", + "operationId": "cohere_proxy_route_cohere__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cohere Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/comprehendmedical": { + "post": { + "description": "AWS-SDK-shaped pass-through for Amazon Comprehend Medical: point the SDK's\n`endpoint_url` at `/comprehendmedical` and the operation is read from the\n`X-Amz-Target` header, per the AWS JSON 1.1 protocol.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/comprehend_medical)", + "operationId": "comprehend_medical_sdk_proxy_route_comprehendmedical_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Comprehend Medical Sdk Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/comprehendmedical/{operation}": { + "post": { + "description": "Pass-through for Amazon Comprehend Medical, e.g. `POST /comprehendmedical/DetectEntitiesV2`.\n\nThe request body is forwarded as-is to the AWS JSON 1.1 API and signed with SigV4\nusing the proxy's AWS credentials.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/comprehend_medical)", + "operationId": "comprehend_medical_proxy_route_comprehendmedical__operation__post", + "parameters": [ + { + "in": "path", + "name": "operation", + "required": true, + "schema": { + "title": "Operation", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Comprehend Medical Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/cursor/chat/completions": { + "post": { + "description": "Cursor BYOK endpoint. Accepts both request shapes Cursor sends to its OpenAI-compatible\nbase URL and always answers in chat completions format.\n\nCursor agent mode sends Responses API format bodies (`input`, flat tool defs, `reasoning`,\ncustom tools) to the chat/completions path while expecting chat completions responses;\nthose are routed through the Responses API pipeline and converted back. Genuine chat\ncompletions bodies (`messages` present) are routed through the standard chat completions\npipeline, after normalizing each level of the `tools` array and `tool_choice` to the chat\ncompletions shapes OpenAI requires. Cursor mixes Responses API shapes into chat bodies\nper level, independently: a flat tool def (`{\"type\": \"custom\", \"name\": \"ApplyPatch\", ...}`)\ngets nested under `custom`, and a flat grammar format\n(`{\"type\": \"grammar\", \"definition\", \"syntax\"}`) gets wrapped as\n`{\"type\": \"grammar\", \"grammar\": {...}}` wherever it appears, including inside tool defs\nCursor already sent pre-nested.\n\n```bash\ncurl -X POST http://localhost:4000/cursor/chat/completions -H \"Content-Type: application/json\" -H \"Authorization: Bearer sk-1234\" -d '{\n \"model\": \"gpt-4o\",\n \"input\": [{\"role\": \"user\", \"content\": \"Hello\"}]\n}'\nResponds back in chat completions format.\n```", + "operationId": "cursor_chat_completions_cursor_chat_completions_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Chat Completions", + "tags": [ + "llm_passthrough" + ] + } + }, + "/cursor/models": { + "get": { + "description": "OpenAI-compatible model listing for the Cursor BYOK base URL.\n\nClients pointed at `/cursor` as an OpenAI-compatible base URL resolve and\nverify models via `GET {base}/models` (the OpenAI SDK contract). Without this\nroute those requests fall through to the Cursor Cloud Agents passthrough, which\ndemands a Cursor API key and 401s, so key verification silently fails before any\nchat request is ever sent. Delegates to the standard `/v1/models` handler.", + "operationId": "cursor_model_list_cursor_models_get", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Model List", + "tags": [ + "llm_passthrough" + ] + } + }, + "/cursor/v1/models": { + "get": { + "description": "OpenAI-compatible model listing for the Cursor BYOK base URL.\n\nClients pointed at `/cursor` as an OpenAI-compatible base URL resolve and\nverify models via `GET {base}/models` (the OpenAI SDK contract). Without this\nroute those requests fall through to the Cursor Cloud Agents passthrough, which\ndemands a Cursor API key and 401s, so key verification silently fails before any\nchat request is ever sent. Delegates to the standard `/v1/models` handler.", + "operationId": "cursor_model_list_cursor_v1_models_get", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Model List", + "tags": [ + "llm_passthrough" + ] + } + }, + "/cursor/{endpoint}": { + "delete": { + "description": "Pass-through endpoint for the Cursor Cloud Agents API.\n\nSupports all Cursor Cloud Agents endpoints:\n- GET /v0/agents \u2014 List agents\n- POST /v0/agents \u2014 Launch an agent\n- GET /v0/agents/{id} \u2014 Agent status\n- GET /v0/agents/{id}/conversation \u2014 Agent conversation\n- POST /v0/agents/{id}/followup \u2014 Add follow-up\n- POST /v0/agents/{id}/stop \u2014 Stop an agent\n- DELETE /v0/agents/{id} \u2014 Delete an agent\n- GET /v0/me \u2014 API key info\n- GET /v0/models \u2014 List models\n- GET /v0/repositories \u2014 List GitHub repositories\n\nUses Basic Authentication (base64-encoded `API_KEY:`).\n\nCredential lookup order:\n1. passthrough_endpoint_router (config.yaml deployments with use_in_pass_through)\n2. litellm.credential_list (credentials added via UI)\n3. CURSOR_API_KEY environment variable", + "operationId": "cursor_proxy_route_cursor__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Pass-through endpoint for the Cursor Cloud Agents API.\n\nSupports all Cursor Cloud Agents endpoints:\n- GET /v0/agents \u2014 List agents\n- POST /v0/agents \u2014 Launch an agent\n- GET /v0/agents/{id} \u2014 Agent status\n- GET /v0/agents/{id}/conversation \u2014 Agent conversation\n- POST /v0/agents/{id}/followup \u2014 Add follow-up\n- POST /v0/agents/{id}/stop \u2014 Stop an agent\n- DELETE /v0/agents/{id} \u2014 Delete an agent\n- GET /v0/me \u2014 API key info\n- GET /v0/models \u2014 List models\n- GET /v0/repositories \u2014 List GitHub repositories\n\nUses Basic Authentication (base64-encoded `API_KEY:`).\n\nCredential lookup order:\n1. passthrough_endpoint_router (config.yaml deployments with use_in_pass_through)\n2. litellm.credential_list (credentials added via UI)\n3. CURSOR_API_KEY environment variable", + "operationId": "cursor_proxy_route_cursor__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Pass-through endpoint for the Cursor Cloud Agents API.\n\nSupports all Cursor Cloud Agents endpoints:\n- GET /v0/agents \u2014 List agents\n- POST /v0/agents \u2014 Launch an agent\n- GET /v0/agents/{id} \u2014 Agent status\n- GET /v0/agents/{id}/conversation \u2014 Agent conversation\n- POST /v0/agents/{id}/followup \u2014 Add follow-up\n- POST /v0/agents/{id}/stop \u2014 Stop an agent\n- DELETE /v0/agents/{id} \u2014 Delete an agent\n- GET /v0/me \u2014 API key info\n- GET /v0/models \u2014 List models\n- GET /v0/repositories \u2014 List GitHub repositories\n\nUses Basic Authentication (base64-encoded `API_KEY:`).\n\nCredential lookup order:\n1. passthrough_endpoint_router (config.yaml deployments with use_in_pass_through)\n2. litellm.credential_list (credentials added via UI)\n3. CURSOR_API_KEY environment variable", + "operationId": "cursor_proxy_route_cursor__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Pass-through endpoint for the Cursor Cloud Agents API.\n\nSupports all Cursor Cloud Agents endpoints:\n- GET /v0/agents \u2014 List agents\n- POST /v0/agents \u2014 Launch an agent\n- GET /v0/agents/{id} \u2014 Agent status\n- GET /v0/agents/{id}/conversation \u2014 Agent conversation\n- POST /v0/agents/{id}/followup \u2014 Add follow-up\n- POST /v0/agents/{id}/stop \u2014 Stop an agent\n- DELETE /v0/agents/{id} \u2014 Delete an agent\n- GET /v0/me \u2014 API key info\n- GET /v0/models \u2014 List models\n- GET /v0/repositories \u2014 List GitHub repositories\n\nUses Basic Authentication (base64-encoded `API_KEY:`).\n\nCredential lookup order:\n1. passthrough_endpoint_router (config.yaml deployments with use_in_pass_through)\n2. litellm.credential_list (credentials added via UI)\n3. CURSOR_API_KEY environment variable", + "operationId": "cursor_proxy_route_cursor__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Pass-through endpoint for the Cursor Cloud Agents API.\n\nSupports all Cursor Cloud Agents endpoints:\n- GET /v0/agents \u2014 List agents\n- POST /v0/agents \u2014 Launch an agent\n- GET /v0/agents/{id} \u2014 Agent status\n- GET /v0/agents/{id}/conversation \u2014 Agent conversation\n- POST /v0/agents/{id}/followup \u2014 Add follow-up\n- POST /v0/agents/{id}/stop \u2014 Stop an agent\n- DELETE /v0/agents/{id} \u2014 Delete an agent\n- GET /v0/me \u2014 API key info\n- GET /v0/models \u2014 List models\n- GET /v0/repositories \u2014 List GitHub repositories\n\nUses Basic Authentication (base64-encoded `API_KEY:`).\n\nCredential lookup order:\n1. passthrough_endpoint_router (config.yaml deployments with use_in_pass_through)\n2. litellm.credential_list (credentials added via UI)\n3. CURSOR_API_KEY environment variable", + "operationId": "cursor_proxy_route_cursor__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cursor Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/eu.assemblyai/{endpoint}": { + "delete": { + "operationId": "assemblyai_proxy_route_eu_assemblyai__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "operationId": "assemblyai_proxy_route_eu_assemblyai__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "operationId": "assemblyai_proxy_route_eu_assemblyai__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "operationId": "assemblyai_proxy_route_eu_assemblyai__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "operationId": "assemblyai_proxy_route_eu_assemblyai__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Assemblyai Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/gemini/{endpoint}": { + "delete": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/google_ai_studio)", + "operationId": "gemini_proxy_route_gemini__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Gemini Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/google_ai_studio)", + "operationId": "gemini_proxy_route_gemini__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Gemini Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/google_ai_studio)", + "operationId": "gemini_proxy_route_gemini__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Gemini Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/google_ai_studio)", + "operationId": "gemini_proxy_route_gemini__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Gemini Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/google_ai_studio)", + "operationId": "gemini_proxy_route_gemini__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Gemini Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/gigachat/{endpoint}": { + "delete": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/gigachat)", + "operationId": "gigachat_proxy_route_gigachat__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Gigachat Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/gigachat)", + "operationId": "gigachat_proxy_route_gigachat__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Gigachat Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/gigachat)", + "operationId": "gigachat_proxy_route_gigachat__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Gigachat Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/gigachat)", + "operationId": "gigachat_proxy_route_gigachat__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Gigachat Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/gigachat)", + "operationId": "gigachat_proxy_route_gigachat__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Gigachat Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/milvus/{endpoint}": { + "delete": { + "description": "Enable using Milvus `/vectors` endpoint as a pass-through endpoint.", + "operationId": "milvus_proxy_route_milvus__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Milvus Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Enable using Milvus `/vectors` endpoint as a pass-through endpoint.", + "operationId": "milvus_proxy_route_milvus__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Milvus Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Enable using Milvus `/vectors` endpoint as a pass-through endpoint.", + "operationId": "milvus_proxy_route_milvus__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Milvus Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Enable using Milvus `/vectors` endpoint as a pass-through endpoint.", + "operationId": "milvus_proxy_route_milvus__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Milvus Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Enable using Milvus `/vectors` endpoint as a pass-through endpoint.", + "operationId": "milvus_proxy_route_milvus__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Milvus Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/mistral/{endpoint}": { + "delete": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/mistral)", + "operationId": "mistral_proxy_route_mistral__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Mistral Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/mistral)", + "operationId": "mistral_proxy_route_mistral__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Mistral Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/mistral)", + "operationId": "mistral_proxy_route_mistral__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Mistral Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/mistral)", + "operationId": "mistral_proxy_route_mistral__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Mistral Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/mistral)", + "operationId": "mistral_proxy_route_mistral__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Mistral Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/deployments/{model}/chat/completions": { + "post": { + "description": "Follows the exact same API spec as `OpenAI's Chat API https://platform.openai.com/docs/api-reference/chat`\n\n```bash\ncurl -X POST http://localhost:4000/v1/chat/completions \n-H \"Content-Type: application/json\" \n-H \"Authorization: Bearer sk-1234\" \n-d '{\n \"model\": \"gpt-4o\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Hello!\"\n }\n ]\n}'\n```", + "operationId": "chat_completion_openai_deployments__model__chat_completions_post", + "parameters": [ + { + "in": "path", + "name": "model", + "required": true, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful response" + }, + "400": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "ContentPolicyViolationError" + }, + "401": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "AuthenticationError" + }, + "403": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "PermissionDeniedError" + }, + "404": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "NotFoundError" + }, + "408": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "Timeout" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "UnprocessableEntityError" + }, + "429": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "\n Unified rate-limit error.\n\n Every rate-limit condition surfaced by litellm \u2014 whether it originated from\n an upstream LLM provider, a vendor batch endpoint, or one of litellm's own\n proxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\n max-iterations, etc.) \u2014 is raised as an instance of this class.\n\n The :attr:`category` attribute lets callers distinguish the source. See\n :class:`RateLimitErrorCategory` for the available values.\n " + }, + "500": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "JSONSchemaValidationError" + }, + "503": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ErrorResponse" + } + } + }, + "description": "APIConnectionError" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Chat Completion", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/deployments/{model}/completions": { + "post": { + "description": "Follows the exact same API spec as `OpenAI's Completions API https://platform.openai.com/docs/api-reference/completions`\n\n```bash\ncurl -X POST http://localhost:4000/v1/completions \n-H \"Content-Type: application/json\" \n-H \"Authorization: Bearer sk-1234\" \n-d '{\n \"model\": \"gpt-3.5-turbo-instruct\",\n \"prompt\": \"Once upon a time\",\n \"max_tokens\": 50,\n \"temperature\": 0.7\n}'\n```", + "operationId": "completion_openai_deployments__model__completions_post", + "parameters": [ + { + "in": "path", + "name": "model", + "required": true, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Completion", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/deployments/{model}/embeddings": { + "post": { + "description": "Follows the exact same API spec as `OpenAI's Embeddings API https://platform.openai.com/docs/api-reference/embeddings`\n\n```bash\ncurl -X POST http://localhost:4000/v1/embeddings \n-H \"Content-Type: application/json\" \n-H \"Authorization: Bearer sk-1234\" \n-d '{\n \"model\": \"text-embedding-ada-002\",\n \"input\": \"The quick brown fox jumps over the lazy dog\"\n}'\n```", + "operationId": "embeddings_openai_deployments__model__embeddings_post", + "parameters": [ + { + "in": "path", + "name": "model", + "required": true, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Embeddings", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/deployments/{model}/images/edits": { + "post": { + "description": "Follows the OpenAI Images API spec: https://platform.openai.com/docs/api-reference/images/create\n\n```bash\ncurl -s -D >(grep -i x-request-id >&2) -o >(jq -r '.data[0].b64_json' | base64 --decode > gift-basket.png) -X POST \"http://localhost:4000/v1/images/edits\" -H \"Authorization: Bearer sk-1234\" -F \"model=gpt-image-1\" -F \"image[]=@soap.png\" -F 'prompt=Create a studio ghibli image of this'\n```", + "operationId": "image_edit_api_openai_deployments__model__images_edits_post", + "parameters": [ + { + "in": "path", + "name": "model", + "required": true, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + } + } + ], + "requestBody": { + "content": { + "multipart/form-data": { + "schema": { + "$ref": "#/components/schemas/Body_image_edit_api_openai_deployments__model__images_edits_post" + } + } + } + }, + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Image Edit Api", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/deployments/{model}/images/generations": { + "post": { + "operationId": "image_generation_openai_deployments__model__images_generations_post", + "parameters": [ + { + "in": "path", + "name": "model", + "required": true, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Image Generation", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/realtime/calls": { + "post": { + "operationId": "proxy_realtime_calls_openai_v1_realtime_calls_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "summary": "Proxy Realtime Calls", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/realtime/client_secrets": { + "post": { + "operationId": "create_realtime_client_secret_openai_v1_realtime_client_secrets_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/RealtimeClientSecretResponse" + } + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Create Realtime Client Secret", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/realtime/transcription_sessions": { + "post": { + "description": "Create an ephemeral Realtime transcription session\n(POST /v1/realtime/transcription_sessions) for the WebRTC/WebSocket flow.\n\nMirrors the client_secrets route but targets the transcription_sessions\nendpoint and encrypts the ephemeral key returned under `client_secret.value`.", + "operationId": "create_realtime_transcription_session_openai_v1_realtime_transcription_sessions_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/RealtimeTranscriptionSessionResponse" + } + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Create Realtime Transcription Session", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/responses": { + "post": { + "description": "Follows the OpenAI Responses API spec: https://platform.openai.com/docs/api-reference/responses\n\nSupports background mode with polling_via_cache for partial response retrieval.\nWhen background=true and polling_via_cache is enabled, returns a polling_id immediately\nand streams the response in the background, updating Redis cache.\n\n```bash\n# Normal request\ncurl -X POST http://localhost:4000/v1/responses -H \"Content-Type: application/json\" -H \"Authorization: Bearer sk-1234\" -d '{\n \"model\": \"gpt-4o\",\n \"input\": \"Tell me about AI\"\n}'\n\n# Background request with polling\ncurl -X POST http://localhost:4000/v1/responses -H \"Content-Type: application/json\" -H \"Authorization: Bearer sk-1234\" -d '{\n \"model\": \"gpt-4o\",\n \"input\": \"Tell me about AI\",\n \"background\": true\n}'\n```", + "operationId": "responses_api_openai_v1_responses_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Responses Api", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/responses/compact": { + "post": { + "description": "Compact a response by running a compaction pass over a conversation.\n\nReturns encrypted, opaque items that can be used to reduce context size.\n\nFollows the OpenAI Responses API spec: https://platform.openai.com/docs/api-reference/responses/compact\n\n```bash\ncurl -X POST http://localhost:4000/v1/responses/compact -H \"Content-Type: application/json\" -H \"Authorization: Bearer sk-1234\" -d '{\n \"model\": \"gpt-4o\",\n \"input\": [{\"role\": \"user\", \"content\": \"Hello\"}]\n}'\n```", + "operationId": "compact_response_openai_v1_responses_compact_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Compact Response", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/responses/input_tokens": { + "post": { + "description": "Count the input tokens of a Responses API request without calling the model.\n\nFollows the OpenAI Responses API spec: https://platform.openai.com/docs/api-reference/responses/input-tokens\n\n```bash\ncurl -X POST http://localhost:4000/v1/responses/input_tokens -H \"Content-Type: application/json\" -H \"Authorization: Bearer sk-1234\" -d '{\n \"model\": \"gpt-4o\",\n \"input\": \"Hello, how are you?\"\n}'\n```\n\nReturns: `{\"object\": \"response.input_tokens\", \"input_tokens\": }`", + "operationId": "responses_input_tokens_openai_v1_responses_input_tokens_post", + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Responses Input Tokens", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/responses/{response_id}": { + "delete": { + "description": "Delete a response by ID.\n\nSupports both:\n- Polling IDs (litellm_poll_*): Deletes from Redis cache\n- Provider response IDs: Passes through to provider API\n\nFollows the OpenAI Responses API spec: https://platform.openai.com/docs/api-reference/responses/delete\n\n```bash\ncurl -X DELETE http://localhost:4000/v1/responses/resp_abc123 -H \"Authorization: Bearer sk-1234\"\n```", + "operationId": "delete_response_openai_v1_responses__response_id__delete", + "parameters": [ + { + "in": "path", + "name": "response_id", + "required": true, + "schema": { + "title": "Response Id", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Delete Response", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Get a response by ID.\n\nSupports both:\n- Polling IDs (litellm_poll_*): Returns cumulative cached content from background responses\n- Provider response IDs: Passes through to provider API\n\nFollows the OpenAI Responses API spec: https://platform.openai.com/docs/api-reference/responses/get\n\n```bash\n# Get polling response\ncurl -X GET http://localhost:4000/v1/responses/litellm_poll_abc123 -H \"Authorization: Bearer sk-1234\"\n\n# Get provider response\ncurl -X GET http://localhost:4000/v1/responses/resp_abc123 -H \"Authorization: Bearer sk-1234\"\n```", + "operationId": "get_response_openai_v1_responses__response_id__get", + "parameters": [ + { + "in": "path", + "name": "response_id", + "required": true, + "schema": { + "title": "Response Id", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Get Response", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/responses/{response_id}/cancel": { + "post": { + "description": "Cancel a response by ID.\n\nSupports both:\n- Polling IDs (litellm_poll_*): Cancels background response and updates status in Redis\n- Provider response IDs: Passes through to provider API\n\nFollows the OpenAI Responses API spec: https://platform.openai.com/docs/api-reference/responses/cancel\n\n```bash\n# Cancel polling response\ncurl -X POST http://localhost:4000/v1/responses/litellm_poll_abc123/cancel -H \"Authorization: Bearer sk-1234\"\n\n# Cancel provider response\ncurl -X POST http://localhost:4000/v1/responses/resp_abc123/cancel -H \"Authorization: Bearer sk-1234\"\n```", + "operationId": "cancel_response_openai_v1_responses__response_id__cancel_post", + "parameters": [ + { + "in": "path", + "name": "response_id", + "required": true, + "schema": { + "title": "Response Id", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Cancel Response", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/v1/responses/{response_id}/input_items": { + "get": { + "description": "List input items for a response.", + "operationId": "get_response_input_items_openai_v1_responses__response_id__input_items_get", + "parameters": [ + { + "in": "path", + "name": "response_id", + "required": true, + "schema": { + "title": "Response Id", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Get Response Input Items", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai/{endpoint}": { + "delete": { + "description": "Pass-through endpoint for OpenAI API calls.\n\nAvailable on both routes:\n- /openai/{endpoint:path} - Standard OpenAI passthrough route\n- /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API)\n\nUse /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts\nwith LiteLLM's native implementations (e.g., for the Responses API at /v1/responses).\n\nExamples:\n Standard route:\n - /openai/v1/chat/completions\n - /openai/v1/assistants\n - /openai/v1/threads\n\n Dedicated passthrough (for Responses API):\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_proxy_route_openai__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Pass-through endpoint for OpenAI API calls.\n\nAvailable on both routes:\n- /openai/{endpoint:path} - Standard OpenAI passthrough route\n- /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API)\n\nUse /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts\nwith LiteLLM's native implementations (e.g., for the Responses API at /v1/responses).\n\nExamples:\n Standard route:\n - /openai/v1/chat/completions\n - /openai/v1/assistants\n - /openai/v1/threads\n\n Dedicated passthrough (for Responses API):\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_proxy_route_openai__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Pass-through endpoint for OpenAI API calls.\n\nAvailable on both routes:\n- /openai/{endpoint:path} - Standard OpenAI passthrough route\n- /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API)\n\nUse /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts\nwith LiteLLM's native implementations (e.g., for the Responses API at /v1/responses).\n\nExamples:\n Standard route:\n - /openai/v1/chat/completions\n - /openai/v1/assistants\n - /openai/v1/threads\n\n Dedicated passthrough (for Responses API):\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_proxy_route_openai__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Pass-through endpoint for OpenAI API calls.\n\nAvailable on both routes:\n- /openai/{endpoint:path} - Standard OpenAI passthrough route\n- /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API)\n\nUse /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts\nwith LiteLLM's native implementations (e.g., for the Responses API at /v1/responses).\n\nExamples:\n Standard route:\n - /openai/v1/chat/completions\n - /openai/v1/assistants\n - /openai/v1/threads\n\n Dedicated passthrough (for Responses API):\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_proxy_route_openai__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Pass-through endpoint for OpenAI API calls.\n\nAvailable on both routes:\n- /openai/{endpoint:path} - Standard OpenAI passthrough route\n- /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API)\n\nUse /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts\nwith LiteLLM's native implementations (e.g., for the Responses API at /v1/responses).\n\nExamples:\n Standard route:\n - /openai/v1/chat/completions\n - /openai/v1/assistants\n - /openai/v1/threads\n\n Dedicated passthrough (for Responses API):\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_proxy_route_openai__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/openai_passthrough/{endpoint}": { + "delete": { + "description": "Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native\nimplementations (e.g. the Responses API at /v1/responses).\n\nExamples:\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_passthrough_route_openai_passthrough__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Passthrough Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native\nimplementations (e.g. the Responses API at /v1/responses).\n\nExamples:\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_passthrough_route_openai_passthrough__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Passthrough Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native\nimplementations (e.g. the Responses API at /v1/responses).\n\nExamples:\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_passthrough_route_openai_passthrough__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Passthrough Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native\nimplementations (e.g. the Responses API at /v1/responses).\n\nExamples:\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_passthrough_route_openai_passthrough__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Passthrough Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native\nimplementations (e.g. the Responses API at /v1/responses).\n\nExamples:\n - /openai_passthrough/v1/responses\n - /openai_passthrough/v1/responses/{response_id}\n - /openai_passthrough/v1/responses/{response_id}/input_items\n\n[Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough)", + "operationId": "openai_passthrough_route_openai_passthrough__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Openai Passthrough Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/vertex_ai/discovery/{endpoint}": { + "delete": { + "description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`", + "operationId": "vertex_discovery_proxy_route_vertex_ai_discovery__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Vertex Discovery Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`", + "operationId": "vertex_discovery_proxy_route_vertex_ai_discovery__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Vertex Discovery Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`", + "operationId": "vertex_discovery_proxy_route_vertex_ai_discovery__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Vertex Discovery Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`", + "operationId": "vertex_discovery_proxy_route_vertex_ai_discovery__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Vertex Discovery Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`", + "operationId": "vertex_discovery_proxy_route_vertex_ai_discovery__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "summary": "Vertex Discovery Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/vertex_ai/{endpoint}": { + "delete": { + "description": "Call LiteLLM proxy via Vertex AI SDK.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai)", + "operationId": "vertex_proxy_route_vertex_ai__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vertex Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Call LiteLLM proxy via Vertex AI SDK.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai)", + "operationId": "vertex_proxy_route_vertex_ai__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vertex Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Call LiteLLM proxy via Vertex AI SDK.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai)", + "operationId": "vertex_proxy_route_vertex_ai__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vertex Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Call LiteLLM proxy via Vertex AI SDK.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai)", + "operationId": "vertex_proxy_route_vertex_ai__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vertex Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Call LiteLLM proxy via Vertex AI SDK.\n\n[Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai)", + "operationId": "vertex_proxy_route_vertex_ai__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vertex Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/vllm/{endpoint}": { + "delete": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/vllm)", + "operationId": "vllm_proxy_route_vllm__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vllm Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/vllm)", + "operationId": "vllm_proxy_route_vllm__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vllm Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/vllm)", + "operationId": "vllm_proxy_route_vllm__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vllm Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/vllm)", + "operationId": "vllm_proxy_route_vllm__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vllm Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "[Docs](https://docs.litellm.ai/docs/pass_through/vllm)", + "operationId": "vllm_proxy_route_vllm__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Vllm Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + }, + "/watsonx/{endpoint}": { + "delete": { + "description": "Watsonx pass-through endpoint.\nAllows using Watsonx APIs with automatic IAM token management and version parameter injection.\n\nExample:\n POST /watsonx/ml/v1/text/tokenization\n POST /watsonx/ml/v1/text/generation", + "operationId": "watsonx_proxy_route_watsonx__endpoint__delete", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Watsonx Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "get": { + "description": "Watsonx pass-through endpoint.\nAllows using Watsonx APIs with automatic IAM token management and version parameter injection.\n\nExample:\n POST /watsonx/ml/v1/text/tokenization\n POST /watsonx/ml/v1/text/generation", + "operationId": "watsonx_proxy_route_watsonx__endpoint__get", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Watsonx Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "patch": { + "description": "Watsonx pass-through endpoint.\nAllows using Watsonx APIs with automatic IAM token management and version parameter injection.\n\nExample:\n POST /watsonx/ml/v1/text/tokenization\n POST /watsonx/ml/v1/text/generation", + "operationId": "watsonx_proxy_route_watsonx__endpoint__patch", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Watsonx Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "post": { + "description": "Watsonx pass-through endpoint.\nAllows using Watsonx APIs with automatic IAM token management and version parameter injection.\n\nExample:\n POST /watsonx/ml/v1/text/tokenization\n POST /watsonx/ml/v1/text/generation", + "operationId": "watsonx_proxy_route_watsonx__endpoint__post", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Watsonx Proxy Route", + "tags": [ + "llm_passthrough" + ] + }, + "put": { + "description": "Watsonx pass-through endpoint.\nAllows using Watsonx APIs with automatic IAM token management and version parameter injection.\n\nExample:\n POST /watsonx/ml/v1/text/tokenization\n POST /watsonx/ml/v1/text/generation", + "operationId": "watsonx_proxy_route_watsonx__endpoint__put", + "parameters": [ + { + "in": "path", + "name": "endpoint", + "required": true, + "schema": { + "title": "Endpoint", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": {} + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Watsonx Proxy Route", + "tags": [ + "llm_passthrough" + ] + } + } + } + }, "mcp_app": { "components": { "schemas": { @@ -31965,7 +36935,7 @@ "paths": { "/openai/v1/realtime/calls": { "post": { - "operationId": "proxy_realtime_calls_openai_v1_realtime_calls_post", + "operationId": "proxy_realtime_calls_openai_v1_realtime_calls_post_2", "responses": { "200": { "content": { @@ -31984,7 +36954,7 @@ }, "/openai/v1/realtime/client_secrets": { "post": { - "operationId": "create_realtime_client_secret_openai_v1_realtime_client_secrets_post", + "operationId": "create_realtime_client_secret_openai_v1_realtime_client_secrets_post_2", "responses": { "200": { "content": { @@ -32011,7 +36981,7 @@ "/openai/v1/realtime/transcription_sessions": { "post": { "description": "Create an ephemeral Realtime transcription session\n(POST /v1/realtime/transcription_sessions) for the WebRTC/WebSocket flow.\n\nMirrors the client_secrets route but targets the transcription_sessions\nendpoint and encrypts the ephemeral key returned under `client_secret.value`.", - "operationId": "create_realtime_transcription_session_openai_v1_realtime_transcription_sessions_post", + "operationId": "create_realtime_transcription_session_openai_v1_realtime_transcription_sessions_post_2", "responses": { "200": { "content": { diff --git a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py index face515ef88..28a8bab1f24 100644 --- a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py @@ -97,7 +97,6 @@ else: vertex_llm_base: Final = VertexBase() router: Final = APIRouter() -openai_passthrough_router: Final = APIRouter() default_vertex_config: Final = None passthrough_endpoint_router: Final = PassthroughEndpointRouter() @@ -2297,11 +2296,6 @@ async def vertex_proxy_route( ) -@openai_passthrough_router.api_route( - "/openai_passthrough/{endpoint:path}", - methods=["GET", "POST", "PUT", "DELETE", "PATCH"], - tags=["OpenAI Pass-through", "pass-through"], -) @router.api_route( "/openai/{endpoint:path}", methods=["GET", "POST", "PUT", "DELETE", "PATCH"], diff --git a/litellm/proxy/pass_through_endpoints/openai_passthrough_endpoints.py b/litellm/proxy/pass_through_endpoints/openai_passthrough_endpoints.py new file mode 100644 index 00000000000..f56a59dd560 --- /dev/null +++ b/litellm/proxy/pass_through_endpoints/openai_passthrough_endpoints.py @@ -0,0 +1,44 @@ +"""/openai_passthrough must be matched ahead of the native /{provider}/v1/files and +/{provider}/v1/batches routes, so unlike the other provider passthrough routes it is +registered at startup and defers to the lazily loaded handler per call.""" + +from typing import Final + +from fastapi import APIRouter, Depends, Request, Response + +from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + +router: Final = APIRouter() + + +@router.api_route( + "/openai_passthrough/{endpoint:path}", + methods=["GET", "POST", "PUT", "DELETE", "PATCH"], + tags=["OpenAI Pass-through", "pass-through"], +) +async def openai_passthrough_route( + endpoint: str, + request: Request, + fastapi_response: Response, + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +) -> Response: + """ + Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native + implementations (e.g. the Responses API at /v1/responses). + + Examples: + - /openai_passthrough/v1/responses + - /openai_passthrough/v1/responses/{response_id} + - /openai_passthrough/v1/responses/{response_id}/input_items + + [Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough) + """ + from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import openai_proxy_route + + return await openai_proxy_route( + endpoint=endpoint, + request=request, + fastapi_response=fastapi_response, + user_api_key_dict=user_api_key_dict, + ) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 94e74b20297..5d94b65ccba 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -304,7 +304,7 @@ from litellm.litellm_core_utils.sensitive_data_masker import ( ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.llms.vertex_ai.vertex_llm_base import VertexBase -from litellm.proxy._lazy_features import attach_lazy_features +from litellm.proxy._lazy_features import attach_lazy_features, reserve_lazy_slot from litellm.proxy._types import * from litellm.proxy.analytics_endpoints.analytics_endpoints import ( router as analytics_router, @@ -639,13 +639,8 @@ from litellm.proxy.openai_files_endpoints.files_endpoints import ( from litellm.proxy.openai_files_endpoints.files_endpoints import ( set_files_config, ) -from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import ( - openai_passthrough_router, - passthrough_endpoint_router, - vertex_ai_live_websocket_passthrough, -) -from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import ( - router as llm_passthrough_router, +from litellm.proxy.pass_through_endpoints.openai_passthrough_endpoints import ( + router as openai_passthrough_router, ) from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( initialize_pass_through_endpoints, @@ -6047,6 +6042,10 @@ class ProxyConfig: set_files_config(config=files_config) ## default config for vertex ai routes + from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import ( + passthrough_endpoint_router, + ) + default_vertex_config: Final = config.get("default_vertex_config", None) passthrough_endpoint_router.set_default_vertex_config(config=default_vertex_config) @@ -11763,6 +11762,10 @@ async def vertex_ai_live_passthrough_endpoint( This endpoint delegates to the WebSocket function defined in llm_passthrough_endpoints.py """ + from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import ( + vertex_ai_live_websocket_passthrough, + ) + return await vertex_ai_live_websocket_passthrough( websocket=websocket, model=model, @@ -18668,7 +18671,7 @@ app.include_router(credential_router) app.include_router(openai_passthrough_router) app.include_router(batches_router) app.include_router(openai_files_router) -app.include_router(llm_passthrough_router) +reserve_lazy_slot(app, "llm_passthrough") app.include_router(pass_through_router) app.include_router(health_router) app.include_router(key_management_router) diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py index addc952af14..7b285674145 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py @@ -4,7 +4,7 @@ import contextlib import json import os import traceback -from collections.abc import Mapping +from collections.abc import Iterator, Mapping from types import MappingProxyType, SimpleNamespace from typing import Final from unittest import mock @@ -12,7 +12,9 @@ from unittest.mock import AsyncMock, MagicMock, Mock, patch import httpx import pytest +import respx from fastapi import HTTPException, Request, Response +from fastapi.routing import APIRoute from fastapi.responses import StreamingResponse from fastapi.testclient import TestClient from starlette.datastructures import FormData @@ -45,6 +47,7 @@ from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import ( ) from litellm.proxy._types import LitellmUserRoles, SpecialHeaders, UserAPIKeyAuth from litellm.proxy.auth.handle_jwt import JWTHandler +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.types.passthrough_endpoints.vertex_ai import VertexPassThroughCredentials @@ -3339,8 +3342,11 @@ class TestOpenAIPassthroughRoute: def _resolve_route_name(method: str, path: str) -> str | None: from starlette.routing import Match + from litellm.proxy._lazy_features import LAZY_FEATURES, _force_load from litellm.proxy.proxy_server import app + asyncio.run(_force_load(app, next(f for f in LAZY_FEATURES if f.name == "llm_passthrough"))) + scope: Final = { "type": "http", "method": method, @@ -3350,8 +3356,8 @@ def _resolve_route_name(method: str, path: str) -> str | None: "root_path": "", } for route in app.router.routes: - if route.matches(scope)[0] == Match.FULL: - return getattr(route, "name", None) + if isinstance(route, APIRoute) and route.matches(scope)[0] == Match.FULL: + return route.name return None @@ -3376,7 +3382,7 @@ def test_openai_passthrough_prefix_wins_over_native_provider_routes(method, path /{provider}/v1/files and /{provider}/v1/batches routes must never capture it with provider="openai_passthrough" (which 500s on the LlmProviders lookup). """ - assert _resolve_route_name(method, path) == "openai_proxy_route" + assert _resolve_route_name(method, path) == "openai_passthrough_route" @pytest.mark.parametrize( @@ -3393,6 +3399,41 @@ def test_native_provider_routes_are_unchanged(method, path, expected_name): assert _resolve_route_name(method, path) == expected_name +@pytest.fixture +def openai_passthrough_client(monkeypatch: pytest.MonkeyPatch) -> Iterator[TestClient]: + from litellm.proxy.proxy_server import app + + monkeypatch.setenv("OPENAI_API_KEY", "sk-upstream") + monkeypatch.delenv("SERVER_ROOT_PATH", raising=False) + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + litellm.in_memory_llm_clients_cache.flush_cache() + monkeypatch.setitem(app.dependency_overrides, user_api_key_auth, lambda: UserAPIKeyAuth(api_key="sk-virtual")) + yield TestClient(app) + + +@pytest.mark.parametrize( + "method, path, body", + [ + ("POST", "/v1/responses", {"model": "gpt-5.1", "input": "hi"}), + ("GET", "/v1/files", None), + ("POST", "/v1/batches", {"input_file_id": "file-abc123", "endpoint": "/v1/responses"}), + ], +) +def test_openai_passthrough_forwards_verbatim_to_openai( + openai_passthrough_client: TestClient, method: str, path: str, body: dict[str, str] | None +) -> None: + """Every /openai_passthrough request, including the /v1/files and /v1/batches + paths that native provider routes also claim, must reach OpenAI unchanged.""" + with respx.mock(assert_all_called=True) as upstream: + route = upstream.request(method, f"https://api.openai.com{path}").mock( + return_value=httpx.Response(200, json={"id": "upstream_123"}) + ) + response = openai_passthrough_client.request(method, f"/openai_passthrough{path}", json=body) + + assert (response.status_code, response.json()) == (200, {"id": "upstream_123"}) + assert route.calls.last.request.headers["authorization"] == "Bearer sk-upstream" + + class TestCursorProxyRoute: """Tests for the Cursor Cloud Agents pass-through route.""" diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index e058a4f6396..4f4a9e87d18 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -9183,6 +9183,126 @@ class TestLazyFeaturesNotImportedAtStartup: class TestLazyFeatureMiddleware: """Behavior of the middleware itself, exercised in isolation.""" + @pytest.mark.asyncio + async def test_llm_passthrough_loads_on_first_provider_request(self, monkeypatch): + """An app that never registered the provider passthrough routes 404s a + provider request; behind the middleware the same request registers the + routes and is forwarded to the provider with the configured key.""" + import respx + from fastapi import FastAPI + + from litellm.proxy._lazy_features import LAZY_FEATURES, LazyFeatureMiddleware + + monkeypatch.setenv("MISTRAL_API_KEY", "sk-upstream") + monkeypatch.delenv("SERVER_ROOT_PATH", raising=False) + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + litellm.in_memory_llm_clients_cache.flush_cache() + feat = next(f for f in LAZY_FEATURES if f.name == "llm_passthrough") + target_app = FastAPI() + target_app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-virtual") + mw = LazyFeatureMiddleware(target_app, fastapi_app=target_app, features=(feat,)) + + with respx.mock() as upstream: + route = upstream.get("https://api.mistral.ai/v1/models").mock( + return_value=httpx.Response(200, json={"object": "list", "data": []}) + ) + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=target_app), base_url="http://t") as bare: + assert (await bare.get("/mistral/v1/models")).status_code == 404 + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=mw), base_url="http://t") as lazy: + response = await lazy.get("/mistral/v1/models") + + assert (response.status_code, response.json()) == (200, {"object": "list", "data": []}) + assert route.calls.last.request.headers["authorization"] == "Bearer sk-upstream" + + def test_llm_passthrough_prefixes_cover_every_route_the_module_registers(self): + """A route the module registers under a prefix the feature does not claim + would 404 until an unrelated provider request happens to load the module.""" + from litellm.proxy._lazy_features import LAZY_FEATURES + + feat = next(f for f in LAZY_FEATURES if f.name == "llm_passthrough") + paths = [r.path for r in importlib.import_module(feat.module_path).router.routes] + + assert {"/mistral/{endpoint:path}", "/openai/{endpoint:path}"} <= set(paths) + unreachable = [p for p in paths if not feat.matches(p.replace("{endpoint:path}", "x"))] + assert unreachable == [], f"routes the middleware would never load: {unreachable}" + + @pytest.mark.asyncio + @pytest.mark.parametrize("first_hit", ["/v1/realtime/calls", "/openai/v1/models"]) + async def test_lazy_routes_land_in_registry_order_not_first_hit_order(self, first_hit): + """Two lazy features with overlapping paths must answer a request with the + same handler no matter which one a deployment happens to hit first.""" + from fastapi import APIRouter, FastAPI + + from litellm.proxy._lazy_features import LazyFeature, LazyFeatureMiddleware + + def make_register(path, handler): + def register(app, module): + router = APIRouter() + router.add_api_route(path, lambda: {"handler": handler}, methods=["POST"]) + app.include_router(router) + + return register + + catch_all = LazyFeature( + name="catch_all", + module_path="json", + path_prefixes=("/openai/",), + register_fn=make_register("/openai/{endpoint:path}", "catch_all"), + ) + specific = LazyFeature( + name="specific", + module_path="base64", + path_prefixes=("/openai/v1/realtime", "/v1/realtime"), + register_fn=make_register("/openai/v1/realtime/calls", "specific"), + ) + + target_app = FastAPI() + mw = LazyFeatureMiddleware(target_app, fastapi_app=target_app, features=(catch_all, specific)) + + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=mw), base_url="http://t") as client: + await client.post(first_hit) + await client.post("/openai/v1/models") + response = await client.post("/openai/v1/realtime/calls") + + assert response.json() == {"handler": "catch_all"} + + @pytest.mark.asyncio + @pytest.mark.parametrize("root_path", ["", "/api"]) + async def test_reserved_slot_keeps_lazy_catch_all_ahead_of_later_eager_routes(self, root_path): + """/{mcp_server_name}/mcp is registered after the provider passthrough router + at startup, so /mistral/mcp must keep reaching the provider catch-all once + that router loads lazily instead of being swallowed by the MCP route. The + native /mistral/v1/files route sits ahead of it, so that path neither loads + the feature nor changes owner, with or without a SERVER_ROOT_PATH prefix.""" + from fastapi import APIRouter, FastAPI + + from litellm.proxy._lazy_features import LazyFeature, LazyFeatureMiddleware, reserve_lazy_slot + + def register(app, module): + router = APIRouter() + router.add_api_route("/mistral/{endpoint:path}", lambda: {"handler": "passthrough"}, methods=["POST"]) + app.include_router(router) + + passthrough = LazyFeature( + name="llm_passthrough", module_path="json", path_prefixes=("/mistral/",), register_fn=register + ) + target_app = FastAPI(root_path=root_path) + target_app.add_api_route("/mistral/v1/files", lambda: {"handler": "files"}, methods=["POST"]) + reserve_lazy_slot(target_app, "llm_passthrough", features=(passthrough,)) + target_app.add_api_route("/{mcp_server_name}/mcp", lambda: {"handler": "mcp"}, methods=["POST"]) + target_app.add_middleware(LazyFeatureMiddleware, fastapi_app=target_app, features=(passthrough,)) + + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=target_app), base_url="http://t") as client: + files_first = (await client.post(f"{root_path}/mistral/v1/files")).json()["handler"] + loaded_after_files = frozenset(target_app.state.lazy_loaded) + handlers = [ + (await client.post(f"{root_path}{path}")).json()["handler"] + for path in ("/mistral/mcp", "/mistral/v1/files") + ] + + assert (files_first, loaded_after_files) == ("files", frozenset()) + assert handlers == ["passthrough", "files"] + @pytest.mark.asyncio async def test_first_request_triggers_load_subsequent_does_not(self): from fastapi import FastAPI diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 29435c31aee..1b34c6f6a51 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -9547,26 +9547,6 @@ export interface paths { patch?: never; trace?: never; }; - "/openai/": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * WebSocket: openai_websocket_proxy_route - * @description WebSocket connection endpoint - */ - get: operations["websocket_openai_websocket_proxy_route_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/openai/deployments/{model}/chat/completions": { parameters: { query?: never; @@ -10122,26 +10102,6 @@ export interface paths { patch: operations["openai_proxy_route_openai__endpoint__patch"]; trace?: never; }; - "/openai_passthrough/": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * WebSocket: openai_websocket_proxy_route - * @description WebSocket connection endpoint - */ - get: operations["websocket_openai_websocket_proxy_route_get_2"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/openai_passthrough/{endpoint}": { parameters: { query?: never; @@ -10150,132 +10110,72 @@ export interface paths { cookie?: never; }; /** - * Openai Proxy Route - * @description Pass-through endpoint for OpenAI API calls. - * - * Available on both routes: - * - /openai/{endpoint:path} - Standard OpenAI passthrough route - * - /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API) - * - * Use /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts - * with LiteLLM's native implementations (e.g., for the Responses API at /v1/responses). + * Openai Passthrough Route + * @description Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native + * implementations (e.g. the Responses API at /v1/responses). * * Examples: - * Standard route: - * - /openai/v1/chat/completions - * - /openai/v1/assistants - * - /openai/v1/threads - * - * Dedicated passthrough (for Responses API): * - /openai_passthrough/v1/responses * - /openai_passthrough/v1/responses/{response_id} * - /openai_passthrough/v1/responses/{response_id}/input_items * * [Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough) */ - get: operations["openai_proxy_route_openai_passthrough__endpoint__get"]; + get: operations["openai_passthrough_route_openai_passthrough__endpoint__get"]; /** - * Openai Proxy Route - * @description Pass-through endpoint for OpenAI API calls. - * - * Available on both routes: - * - /openai/{endpoint:path} - Standard OpenAI passthrough route - * - /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API) - * - * Use /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts - * with LiteLLM's native implementations (e.g., for the Responses API at /v1/responses). + * Openai Passthrough Route + * @description Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native + * implementations (e.g. the Responses API at /v1/responses). * * Examples: - * Standard route: - * - /openai/v1/chat/completions - * - /openai/v1/assistants - * - /openai/v1/threads - * - * Dedicated passthrough (for Responses API): * - /openai_passthrough/v1/responses * - /openai_passthrough/v1/responses/{response_id} * - /openai_passthrough/v1/responses/{response_id}/input_items * * [Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough) */ - put: operations["openai_proxy_route_openai_passthrough__endpoint__put"]; + put: operations["openai_passthrough_route_openai_passthrough__endpoint__put"]; /** - * Openai Proxy Route - * @description Pass-through endpoint for OpenAI API calls. - * - * Available on both routes: - * - /openai/{endpoint:path} - Standard OpenAI passthrough route - * - /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API) - * - * Use /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts - * with LiteLLM's native implementations (e.g., for the Responses API at /v1/responses). + * Openai Passthrough Route + * @description Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native + * implementations (e.g. the Responses API at /v1/responses). * * Examples: - * Standard route: - * - /openai/v1/chat/completions - * - /openai/v1/assistants - * - /openai/v1/threads - * - * Dedicated passthrough (for Responses API): * - /openai_passthrough/v1/responses * - /openai_passthrough/v1/responses/{response_id} * - /openai_passthrough/v1/responses/{response_id}/input_items * * [Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough) */ - post: operations["openai_proxy_route_openai_passthrough__endpoint__post"]; + post: operations["openai_passthrough_route_openai_passthrough__endpoint__post"]; /** - * Openai Proxy Route - * @description Pass-through endpoint for OpenAI API calls. - * - * Available on both routes: - * - /openai/{endpoint:path} - Standard OpenAI passthrough route - * - /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API) - * - * Use /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts - * with LiteLLM's native implementations (e.g., for the Responses API at /v1/responses). + * Openai Passthrough Route + * @description Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native + * implementations (e.g. the Responses API at /v1/responses). * * Examples: - * Standard route: - * - /openai/v1/chat/completions - * - /openai/v1/assistants - * - /openai/v1/threads - * - * Dedicated passthrough (for Responses API): * - /openai_passthrough/v1/responses * - /openai_passthrough/v1/responses/{response_id} * - /openai_passthrough/v1/responses/{response_id}/input_items * * [Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough) */ - delete: operations["openai_proxy_route_openai_passthrough__endpoint__delete"]; + delete: operations["openai_passthrough_route_openai_passthrough__endpoint__delete"]; options?: never; head?: never; /** - * Openai Proxy Route - * @description Pass-through endpoint for OpenAI API calls. - * - * Available on both routes: - * - /openai/{endpoint:path} - Standard OpenAI passthrough route - * - /openai_passthrough/{endpoint:path} - Dedicated passthrough route (recommended for Responses API) - * - * Use /openai_passthrough/* when you need guaranteed passthrough to OpenAI without conflicts - * with LiteLLM's native implementations (e.g., for the Responses API at /v1/responses). + * Openai Passthrough Route + * @description Dedicated pass-through to the OpenAI API with no overlap with LiteLLM's native + * implementations (e.g. the Responses API at /v1/responses). * * Examples: - * Standard route: - * - /openai/v1/chat/completions - * - /openai/v1/assistants - * - /openai/v1/threads - * - * Dedicated passthrough (for Responses API): * - /openai_passthrough/v1/responses * - /openai_passthrough/v1/responses/{response_id} * - /openai_passthrough/v1/responses/{response_id}/input_items * * [Docs](https://docs.litellm.ai/docs/pass_through/openai_passthrough) */ - patch: operations["openai_proxy_route_openai_passthrough__endpoint__patch"]; + patch: operations["openai_passthrough_route_openai_passthrough__endpoint__patch"]; trace?: never; }; "/organization/daily/activity": { @@ -21891,52 +21791,6 @@ export interface paths { patch?: never; trace?: never; }; - "/vertex-ai/{endpoint}": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Vertex Proxy Route - * @description Call LiteLLM proxy via Vertex AI SDK. - * - * [Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai) - */ - get: operations["vertex_proxy_route_vertex_ai__endpoint__get_2"]; - /** - * Vertex Proxy Route - * @description Call LiteLLM proxy via Vertex AI SDK. - * - * [Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai) - */ - put: operations["vertex_proxy_route_vertex_ai__endpoint__put_2"]; - /** - * Vertex Proxy Route - * @description Call LiteLLM proxy via Vertex AI SDK. - * - * [Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai) - */ - post: operations["vertex_proxy_route_vertex_ai__endpoint__post_2"]; - /** - * Vertex Proxy Route - * @description Call LiteLLM proxy via Vertex AI SDK. - * - * [Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai) - */ - delete: operations["vertex_proxy_route_vertex_ai__endpoint__delete_2"]; - options?: never; - head?: never; - /** - * Vertex Proxy Route - * @description Call LiteLLM proxy via Vertex AI SDK. - * - * [Docs](https://docs.litellm.ai/docs/pass_through/vertex_ai) - */ - patch: operations["vertex_proxy_route_vertex_ai__endpoint__patch_2"]; - trace?: never; - }; "/vertex_ai/discovery/{endpoint}": { parameters: { query?: never; @@ -52513,24 +52367,6 @@ export interface operations { }; }; }; - websocket_openai_websocket_proxy_route_get: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description WebSocket Protocol Switched */ - 101: { - headers: { - [name: string]: unknown; - }; - content?: never; - }; - }; - }; chat_completion_openai_deployments__model__chat_completions_post: { parameters: { query?: never; @@ -53430,25 +53266,7 @@ export interface operations { }; }; }; - websocket_openai_websocket_proxy_route_get_2: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description WebSocket Protocol Switched */ - 101: { - headers: { - [name: string]: unknown; - }; - content?: never; - }; - }; - }; - openai_proxy_route_openai_passthrough__endpoint__get: { + openai_passthrough_route_openai_passthrough__endpoint__get: { parameters: { query?: never; header?: never; @@ -53479,7 +53297,7 @@ export interface operations { }; }; }; - openai_proxy_route_openai_passthrough__endpoint__put: { + openai_passthrough_route_openai_passthrough__endpoint__put: { parameters: { query?: never; header?: never; @@ -53510,7 +53328,7 @@ export interface operations { }; }; }; - openai_proxy_route_openai_passthrough__endpoint__post: { + openai_passthrough_route_openai_passthrough__endpoint__post: { parameters: { query?: never; header?: never; @@ -53541,7 +53359,7 @@ export interface operations { }; }; }; - openai_proxy_route_openai_passthrough__endpoint__delete: { + openai_passthrough_route_openai_passthrough__endpoint__delete: { parameters: { query?: never; header?: never; @@ -53572,7 +53390,7 @@ export interface operations { }; }; }; - openai_proxy_route_openai_passthrough__endpoint__patch: { + openai_passthrough_route_openai_passthrough__endpoint__patch: { parameters: { query?: never; header?: never; @@ -67979,161 +67797,6 @@ export interface operations { }; }; }; - vertex_proxy_route_vertex_ai__endpoint__get_2: { - parameters: { - query?: never; - header?: never; - path: { - endpoint: string; - }; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - vertex_proxy_route_vertex_ai__endpoint__put_2: { - parameters: { - query?: never; - header?: never; - path: { - endpoint: string; - }; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - vertex_proxy_route_vertex_ai__endpoint__post_2: { - parameters: { - query?: never; - header?: never; - path: { - endpoint: string; - }; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - vertex_proxy_route_vertex_ai__endpoint__delete_2: { - parameters: { - query?: never; - header?: never; - path: { - endpoint: string; - }; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - vertex_proxy_route_vertex_ai__endpoint__patch_2: { - parameters: { - query?: never; - header?: never; - path: { - endpoint: string; - }; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; vertex_discovery_proxy_route_vertex_ai_discovery__endpoint__get: { parameters: { query?: never; From 3df127b439442e07927b370227407f53601c6068 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 09:52:09 -0700 Subject: [PATCH 077/157] fix(proxy): give user-key objects their own in-memory cache partition (#40713) Key objects share the 200-entry UserApiKeyCache in-memory store with teams, end users, tags and memberships, so churn in those objects evicts hot keys and forces a LiteLLM_VerificationToken lookup on the next request. Route bare hashed-token keys to a dedicated InMemoryCache inside UserApiKeyCache while keeping Redis, TTL, serialization and invalidation shared Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/dual_cache.py | 4 +- .../auth_cache_invalidation_pubsub.py | 4 +- litellm/proxy/common_utils/debug_utils.py | 25 ++- .../proxy/common_utils/user_api_key_cache.py | 102 ++++++++++- .../key_management_endpoints.py | 2 +- .../mcp_server/test_discoverable_endpoints.py | 2 +- .../test_auth_cache_invalidation_pubsub.py | 31 ++++ .../common_utils/test_user_api_key_cache.py | 158 ++++++++++++++++-- 8 files changed, 296 insertions(+), 32 deletions(-) diff --git a/litellm/caching/dual_cache.py b/litellm/caching/dual_cache.py index be761e1258b..26ad89a70cc 100644 --- a/litellm/caching/dual_cache.py +++ b/litellm/caching/dual_cache.py @@ -376,7 +376,9 @@ class DualCache(BaseCache): ) # async_batch_set_cache - async def async_set_cache_pipeline(self, cache_list: list, local_only: bool = False, **kwargs): + async def async_set_cache_pipeline( + self, cache_list: Sequence[tuple[str, object]], local_only: bool = False, **kwargs + ): """ Batch write values to the cache """ diff --git a/litellm/proxy/common_utils/auth_cache_invalidation_pubsub.py b/litellm/proxy/common_utils/auth_cache_invalidation_pubsub.py index fb2ca6372c0..dbe11882b3c 100644 --- a/litellm/proxy/common_utils/auth_cache_invalidation_pubsub.py +++ b/litellm/proxy/common_utils/auth_cache_invalidation_pubsub.py @@ -196,9 +196,7 @@ class AuthCacheInvalidationSubscriber: for additional_cache in self._additional_in_memory_caches: additional_cache.set_cache(parsed.cache_key, parsed.new_value, ttl=parsed.ttl) return - in_memory_cache: Final = self._user_api_key_cache.in_memory_cache - if in_memory_cache is not None: - in_memory_cache.delete_cache(parsed.cache_key) + self._user_api_key_cache.in_memory_cache_for(parsed.cache_key).delete_cache(parsed.cache_key) for additional_cache in self._additional_in_memory_caches: additional_cache.delete_cache(parsed.cache_key) diff --git a/litellm/proxy/common_utils/debug_utils.py b/litellm/proxy/common_utils/debug_utils.py index 554a6ae8d1a..1d4024bb84e 100644 --- a/litellm/proxy/common_utils/debug_utils.py +++ b/litellm/proxy/common_utils/debug_utils.py @@ -147,8 +147,11 @@ async def memory_usage_in_mem_cache( llm_router.cache.in_memory_cache.ttl_dict ) - num_items_in_user_api_key_cache: Final = len(user_api_key_cache.in_memory_cache.cache_dict) + len( - user_api_key_cache.in_memory_cache.ttl_dict + num_items_in_user_api_key_cache: Final = ( + len(user_api_key_cache.in_memory_cache.cache_dict) + + len(user_api_key_cache.in_memory_cache.ttl_dict) + + len(user_api_key_cache.key_object_cache.in_memory_cache.cache_dict) + + len(user_api_key_cache.key_object_cache.in_memory_cache.ttl_dict) ) num_items_in_proxy_logging_obj_cache: Final = len( @@ -189,6 +192,8 @@ async def memory_usage_in_mem_cache_items( return { "user_api_key_cache": user_api_key_cache.in_memory_cache.cache_dict, "user_api_key_ttl": user_api_key_cache.in_memory_cache.ttl_dict, + "user_key_object_cache": user_api_key_cache.key_object_cache.in_memory_cache.cache_dict, + "user_key_object_ttl": user_api_key_cache.key_object_cache.in_memory_cache.ttl_dict, "llm_router_cache": llm_router_in_memory_cache_dict, "llm_router_ttl": llm_router_in_memory_ttl_dict, "proxy_logging_obj_cache": proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.cache_dict, @@ -294,7 +299,9 @@ async def get_memory_summary( try: # User API key cache - user_cache_items: Final = len(user_api_key_cache.in_memory_cache.cache_dict) + user_cache_items: Final = len(user_api_key_cache.in_memory_cache.cache_dict) + len( + user_api_key_cache.key_object_cache.in_memory_cache.cache_dict + ) total_cache_items += user_cache_items caches["user_api_keys"] = { "count": user_cache_items, @@ -429,10 +436,16 @@ def _get_cache_memory_stats( cache_stats: Final[dict[str, object]] = {} try: # User API key cache - user_cache_size: Final = sys.getsizeof(user_api_key_cache.in_memory_cache.cache_dict) - user_ttl_size: Final = sys.getsizeof(user_api_key_cache.in_memory_cache.ttl_dict) + key_object_in_memory_cache: Final = user_api_key_cache.key_object_cache.in_memory_cache + user_cache_size: Final = sys.getsizeof(user_api_key_cache.in_memory_cache.cache_dict) + sys.getsizeof( + key_object_in_memory_cache.cache_dict + ) + user_ttl_size: Final = sys.getsizeof(user_api_key_cache.in_memory_cache.ttl_dict) + sys.getsizeof( + key_object_in_memory_cache.ttl_dict + ) cache_stats["user_api_key_cache"] = { - "num_items": len(user_api_key_cache.in_memory_cache.cache_dict), + "num_items": len(user_api_key_cache.in_memory_cache.cache_dict) + + len(key_object_in_memory_cache.cache_dict), "cache_dict_size_bytes": user_cache_size, "ttl_dict_size_bytes": user_ttl_size, "total_size_mb": round((user_cache_size + user_ttl_size) / (1024 * 1024), 2), diff --git a/litellm/proxy/common_utils/user_api_key_cache.py b/litellm/proxy/common_utils/user_api_key_cache.py index 76982d30306..cb72088ee4a 100644 --- a/litellm/proxy/common_utils/user_api_key_cache.py +++ b/litellm/proxy/common_utils/user_api_key_cache.py @@ -1,11 +1,15 @@ from __future__ import annotations +import re +from collections.abc import Sequence from typing import TYPE_CHECKING, Any, Final, TypeVar, cast, overload from pydantic import BaseModel from litellm._logging import verbose_proxy_logger from litellm.caching.dual_cache import DualCache +from litellm.caching.in_memory_cache import InMemoryCache +from litellm.caching.redis_cache import RedisCache from litellm.constants import DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL from litellm.proxy.common_utils.cache_pydantic_utils import CacheCodec @@ -14,6 +18,13 @@ if TYPE_CHECKING: T = TypeVar("T", bound=BaseModel) +_HASHED_TOKEN_CACHE_KEY: Final = re.compile(r"[0-9a-f]{64}") + + +def is_user_key_cache_key(key: str) -> bool: + """Only user-key objects are cached under a bare ``hash_token`` digest; every other object uses a prefixed key.""" + return _HASHED_TOKEN_CACHE_KEY.fullmatch(key) is not None + class UserApiKeyCache(DualCache): """ @@ -36,10 +47,50 @@ class UserApiKeyCache(DualCache): ``async_set_cache_pipeline`` applies the same untyped Codec pass as omitting ``model_type`` on ``async_set_cache`` (so ``BaseModel`` rows are dumped before Redis). + User-key objects (see ``is_user_key_cache_key``) live in their own in-memory partition, + ``key_object_cache``, so churn in the other management objects cannot evict them. Both + partitions share the same Redis backend and TTL settings. + ``get_cache`` / ``async_get_cache`` overloads and implementations must be contiguous (no other methods in between) so mypy resolves ``@overload`` + implementation correctly. """ + def __init__( + self, + in_memory_cache: InMemoryCache | None = None, + redis_cache: RedisCache | None = None, + default_in_memory_ttl: float | None = None, + default_redis_ttl: float | None = None, + key_object_in_memory_cache: InMemoryCache | None = None, + ) -> None: + super().__init__( + in_memory_cache=in_memory_cache, + redis_cache=redis_cache, + default_in_memory_ttl=default_in_memory_ttl, + default_redis_ttl=default_redis_ttl, + ) + self.key_object_cache: Final = DualCache( + in_memory_cache=key_object_in_memory_cache or InMemoryCache(), + redis_cache=redis_cache, + default_in_memory_ttl=default_in_memory_ttl, + default_redis_ttl=default_redis_ttl, + ) + + def in_memory_cache_for(self, key: str) -> InMemoryCache: + return self.key_object_cache.in_memory_cache if is_user_key_cache_key(key) else self.in_memory_cache + + def update_cache_ttl(self, default_in_memory_ttl: float | None, default_redis_ttl: float | None) -> None: + super().update_cache_ttl(default_in_memory_ttl=default_in_memory_ttl, default_redis_ttl=default_redis_ttl) + self.key_object_cache.update_cache_ttl( + default_in_memory_ttl=default_in_memory_ttl, default_redis_ttl=default_redis_ttl + ) + + def attach_redis_cache( + self, redis_cache: RedisCache | None = None, *, default_redis_ttl: float | None = None + ) -> None: + super().attach_redis_cache(redis_cache, default_redis_ttl=default_redis_ttl) + self.key_object_cache.attach_redis_cache(redis_cache, default_redis_ttl=default_redis_ttl) + @overload def get_cache( self, @@ -71,7 +122,11 @@ class UserApiKeyCache(DualCache): ) -> object: if model_type is None and "model_type" in kwargs: model_type = cast(type[BaseModel] | None, kwargs.pop("model_type", None)) - cached: Final = super().get_cache(key=key, parent_otel_span=parent_otel_span, local_only=local_only, **kwargs) + cached: Final = ( + self.key_object_cache.get_cache(key=key, parent_otel_span=parent_otel_span, local_only=local_only, **kwargs) + if is_user_key_cache_key(key) + else super().get_cache(key=key, parent_otel_span=parent_otel_span, local_only=local_only, **kwargs) + ) if model_type is None: return cached if cached is None: @@ -117,8 +172,14 @@ class UserApiKeyCache(DualCache): ) -> object: if model_type is None and "model_type" in kwargs: model_type = cast(type[BaseModel] | None, kwargs.pop("model_type", None)) - cached: Final = await super().async_get_cache( - key=key, parent_otel_span=parent_otel_span, local_only=local_only, **kwargs + cached: Final = ( + await self.key_object_cache.async_get_cache( + key=key, parent_otel_span=parent_otel_span, local_only=local_only, **kwargs + ) + if is_user_key_cache_key(key) + else await super().async_get_cache( + key=key, parent_otel_span=parent_otel_span, local_only=local_only, **kwargs + ) ) if model_type is None: return cached @@ -137,20 +198,49 @@ class UserApiKeyCache(DualCache): def set_cache(self, key: str | None, value: object, local_only: bool = False, **kwargs: object): model_type: Final = cast(type[BaseModel] | None, kwargs.pop("model_type", None)) payload: Final[object] = CacheCodec.serialize(value, model_type=model_type) + if key is not None and is_user_key_cache_key(key): + return self.key_object_cache.set_cache(key=key, value=payload, local_only=local_only, **kwargs) return super().set_cache(key=key, value=payload, local_only=local_only, **kwargs) async def async_set_cache(self, key: str | None, value: object, local_only: bool = False, **kwargs: object): model_type: Final = cast(type[BaseModel] | None, kwargs.pop("model_type", None)) payload: Final[object] = CacheCodec.serialize(value, model_type=model_type) + if key is not None and is_user_key_cache_key(key): + return await self.key_object_cache.async_set_cache(key=key, value=payload, local_only=local_only, **kwargs) return await super().async_set_cache(key=key, value=payload, local_only=local_only, **kwargs) - async def async_set_cache_pipeline(self, cache_list: list, local_only: bool = False, **kwargs: object) -> None: + def delete_cache(self, key: str) -> None: + if is_user_key_cache_key(key): + self.key_object_cache.delete_cache(key) + return + super().delete_cache(key) + + async def async_delete_cache(self, key: str) -> None: + if is_user_key_cache_key(key): + await self.key_object_cache.async_delete_cache(key) + return + await super().async_delete_cache(key) + + def flush_cache(self) -> None: + super().flush_cache() + self.key_object_cache.in_memory_cache.flush_cache() + + async def async_set_cache_pipeline( + self, cache_list: Sequence[tuple[str, object]], local_only: bool = False, **kwargs: object + ) -> None: """ Batch writes with the same Codec boundary as ``async_set_cache`` without ``model_type``: ``BaseModel`` values become JSON-safe dicts; dicts/scalars unchanged. """ - normalized: Final = [(key, CacheCodec.serialize(value, model_type=None)) for key, value in cache_list] - return await super().async_set_cache_pipeline(cache_list=normalized, local_only=local_only, **kwargs) + normalized: Final = tuple((key, CacheCodec.serialize(value, model_type=None)) for key, value in cache_list) + key_object_entries: Final = tuple(entry for entry in normalized if is_user_key_cache_key(entry[0])) + other_entries: Final = tuple(entry for entry in normalized if not is_user_key_cache_key(entry[0])) + if key_object_entries: + await self.key_object_cache.async_set_cache_pipeline( + cache_list=key_object_entries, local_only=local_only, **kwargs + ) + if other_entries: + await super().async_set_cache_pipeline(cache_list=other_entries, local_only=local_only, **kwargs) #: Value cached under ``user_object_permission_id_cache_key`` when the user links no permission row, diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index 749a940de0e..db4467dda5b 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -3729,7 +3729,7 @@ async def delete_key_fn( ) verbose_proxy_logger.debug( - "/keys/delete - cache after delete: %s", user_api_key_cache.in_memory_cache.cache_dict + "/keys/delete - cache after delete: %s", user_api_key_cache.key_object_cache.in_memory_cache.cache_dict ) asyncio.create_task( diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py index 5a99139a67f..c53706ac938 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_discoverable_endpoints.py @@ -7150,7 +7150,7 @@ async def test_extract_user_id_rehydrates_cross_replica_dict_cache(proxy_globals key = "sk-alice-key" cache = UserApiKeyCache() - cache.in_memory_cache.set_cache(hash_token(key), {"token": hash_token(key), "user_id": "alice"}) + cache.set_cache(hash_token(key), {"token": hash_token(key), "user_id": "alice"}) proxy_globals.user_api_key_cache = cache proxy_globals.prisma_client = object() diff --git a/tests/test_litellm/proxy/common_utils/test_auth_cache_invalidation_pubsub.py b/tests/test_litellm/proxy/common_utils/test_auth_cache_invalidation_pubsub.py index 7d5fc1a3544..4e2059ac30b 100644 --- a/tests/test_litellm/proxy/common_utils/test_auth_cache_invalidation_pubsub.py +++ b/tests/test_litellm/proxy/common_utils/test_auth_cache_invalidation_pubsub.py @@ -1,4 +1,5 @@ import asyncio +import hashlib import json from typing import Iterable, List, Optional, Tuple from unittest.mock import patch @@ -7,6 +8,7 @@ import pytest from redis.asyncio import Redis from litellm.caching.in_memory_cache import InMemoryCache +from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import ( AUTH_CACHE_INVALIDATION_CHANNEL, AuthCacheInvalidationSubscriber, @@ -145,6 +147,35 @@ async def test_subscriber_deletes_local_cache_entry_on_message() -> None: assert pubsub.subscribed_channels == [AUTH_CACHE_INVALIDATION_CHANNEL] +@pytest.mark.asyncio +async def test_subscriber_deletes_key_object_partition_entry_on_message() -> None: + """ + LIT-7563 moved user-key objects into their own in-memory partition; a key + invalidation broadcast must still evict the hashed-token entry there, or a + deleted key keeps authenticating on other workers until its TTL expires. + """ + hashed_token = hashlib.sha256(b"sk-lit7563-hot-key").hexdigest() + cache = UserApiKeyCache() + cache.set_cache(hashed_token, UserAPIKeyAuth(token=hashed_token), model_type=UserAPIKeyAuth) + assert cache.get_cache(hashed_token, model_type=UserAPIKeyAuth) is not None + + pubsub = _QueuePubSub(initial_messages=[_invalidation_message(hashed_token)]) + subscriber = AuthCacheInvalidationSubscriber( + redis_cache=_FakeRedisCache(client=_ScriptedPubSubRedisClient(pubsubs=[pubsub])), + user_api_key_cache=cache, + ) + subscriber.start() + try: + for _ in range(200): + if cache.get_cache(hashed_token, model_type=UserAPIKeyAuth) is None: + break + await asyncio.sleep(0.01) + finally: + await subscriber.stop() + + assert cache.get_cache(hashed_token, model_type=UserAPIKeyAuth) is None + + @pytest.mark.asyncio async def test_subscriber_deletes_additional_in_memory_cache_entry_on_message() -> None: """ diff --git a/tests/test_litellm/proxy/common_utils/test_user_api_key_cache.py b/tests/test_litellm/proxy/common_utils/test_user_api_key_cache.py index 33e0d8bf38e..2d5d76ed542 100644 --- a/tests/test_litellm/proxy/common_utils/test_user_api_key_cache.py +++ b/tests/test_litellm/proxy/common_utils/test_user_api_key_cache.py @@ -1,3 +1,4 @@ +import hashlib import json from typing import Any @@ -10,10 +11,14 @@ from litellm.constants import DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.common_utils.user_api_key_cache import ( UserApiKeyCache, + end_user_cache_key, get_management_object_ttl, + is_user_key_cache_key, ) from litellm.proxy.proxy_server import UserAPIKeyCacheTTLEnum +HASHED_TOKEN = hashlib.sha256(b"sk-lit7563-hot-key").hexdigest() + class CapturingInMemoryCache(InMemoryCache): """Records ``ttl`` passed into ``set_cache`` (what DualCache injects).""" @@ -204,9 +209,7 @@ class TestUserApiKeyCache: # Bypass UserApiKeyCache.serialize: CacheCodec rejects non-dict cached values # for dict-based models (deserialize returns None). - await cache.in_memory_cache.async_set_cache( - key="k", value="invalid-payload-not-a-dict" - ) + await cache.in_memory_cache.async_set_cache(key="k", value="invalid-payload-not-a-dict") value = await cache.async_get_cache("k", model_type=UserAPIKeyAuth) assert value is None @@ -224,6 +227,141 @@ class TestUserApiKeyCache: fake.set_cache("k2", {"ok": NotSerializable()}) +class TestUserKeyObjectPartition: + """ + Regression for LIT-7563: user-key objects share one 200-entry ``InMemoryCache`` with + every other management object, so end-user / team / tag churn evicts hot keys and + forces a ``LiteLLM_VerificationToken`` lookup on the next request. + """ + + @pytest.mark.parametrize( + ("key", "expected"), + [ + (HASHED_TOKEN, True), + (HASHED_TOKEN.upper(), False), + (f"team_id:{HASHED_TOKEN}", False), + (end_user_cache_key("u1"), False), + ("sk-lit7563-hot-key", False), + ], + ) + def test_is_user_key_cache_key(self, key: str, expected: bool): + assert is_user_key_cache_key(key) is expected + + @pytest.mark.asyncio + async def test_management_object_churn_does_not_evict_key_object(self): + cache = UserApiKeyCache(in_memory_cache=InMemoryCache(max_size_in_memory=2)) + await cache.async_set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth, ttl=100) + for i in range(2): + await cache.async_set_cache(end_user_cache_key(f"u{i}"), {"user_id": f"u{i}"}, ttl=200) + + key_obj = await cache.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) + assert key_obj is not None + assert key_obj.token == HASHED_TOKEN + assert cache.get_cache(end_user_cache_key("u1")) == {"user_id": "u1"} + assert HASHED_TOKEN not in cache.in_memory_cache.cache_dict + + def test_sync_write_and_read_route_to_key_object_partition(self): + cache = UserApiKeyCache(in_memory_cache=InMemoryCache(max_size_in_memory=2)) + cache.set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth, ttl=100) + for i in range(2): + cache.set_cache(end_user_cache_key(f"u{i}"), {"user_id": f"u{i}"}, ttl=200) + + key_obj = cache.get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) + assert key_obj is not None + assert key_obj.token == HASHED_TOKEN + + @pytest.mark.asyncio + async def test_redis_hit_backfills_key_object_partition_with_configured_ttl(self): + redis = FakeRedisCache() + writer = UserApiKeyCache(redis_cache=redis, default_in_memory_ttl=30) + await writer.async_set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth) + + key_partition = CapturingInMemoryCache() + reader = UserApiKeyCache(redis_cache=redis, default_in_memory_ttl=30, key_object_in_memory_cache=key_partition) + key_obj = await reader.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) + + assert key_obj is not None + assert key_obj.token == HASHED_TOKEN + assert key_partition.last_ttl == 30 + assert HASHED_TOKEN not in reader.in_memory_cache.cache_dict + + @pytest.mark.asyncio + async def test_update_cache_ttl_applies_to_key_object_partition(self): + key_partition = CapturingInMemoryCache() + cache = UserApiKeyCache(default_in_memory_ttl=60, key_object_in_memory_cache=key_partition) + cache.update_cache_ttl(default_in_memory_ttl=7, default_redis_ttl=7) + + await cache.async_set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth) + + assert key_partition.last_ttl == 7 + + @pytest.mark.asyncio + async def test_attach_redis_cache_applies_to_key_object_partition(self): + redis = FakeRedisCache() + cache = UserApiKeyCache() + cache.attach_redis_cache(redis) + + await cache.async_set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth) + + other_worker = UserApiKeyCache(redis_cache=redis) + key_obj = await other_worker.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) + assert key_obj is not None + assert key_obj.token == HASHED_TOKEN + + @pytest.mark.asyncio + async def test_delete_removes_key_object_from_partition_and_redis(self): + redis = FakeRedisCache() + cache = UserApiKeyCache(redis_cache=redis) + await cache.async_set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth) + assert await cache.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) is not None + + cache.delete_cache(HASHED_TOKEN) + + assert await cache.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) is None + assert await redis.async_get_cache(HASHED_TOKEN) is None + + @pytest.mark.asyncio + async def test_async_delete_removes_key_object_from_partition_and_redis(self): + redis = FakeRedisCache() + cache = UserApiKeyCache(redis_cache=redis) + await cache.async_set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth) + + await cache.async_delete_cache(HASHED_TOKEN) + + assert await cache.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) is None + assert await redis.async_get_cache(HASHED_TOKEN) is None + + @pytest.mark.asyncio + async def test_pipeline_write_routes_each_entry_to_its_partition(self): + cache = UserApiKeyCache(in_memory_cache=InMemoryCache(max_size_in_memory=2)) + await cache.async_set_cache_pipeline( + [(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN))] + + [(end_user_cache_key(f"u{i}"), {"user_id": f"u{i}"}) for i in range(2)], + ttl=100, + ) + + key_obj = await cache.async_get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) + assert key_obj is not None + assert key_obj.token == HASHED_TOKEN + assert HASHED_TOKEN not in cache.in_memory_cache.cache_dict + assert cache.get_cache(end_user_cache_key("u1")) == {"user_id": "u1"} + + def test_flush_clears_key_object_partition(self): + cache = UserApiKeyCache() + cache.set_cache(HASHED_TOKEN, _make_key_obj(HASHED_TOKEN), model_type=UserAPIKeyAuth) + cache.set_cache(end_user_cache_key("u1"), {"user_id": "u1"}) + + cache.flush_cache() + + assert cache.get_cache(HASHED_TOKEN, model_type=UserAPIKeyAuth) is None + assert cache.get_cache(end_user_cache_key("u1")) is None + + def test_in_memory_cache_for_routes_by_key(self): + cache = UserApiKeyCache() + assert cache.in_memory_cache_for(HASHED_TOKEN) is cache.key_object_cache.in_memory_cache + assert cache.in_memory_cache_for(end_user_cache_key("u1")) is cache.in_memory_cache + + class TestManagementObjectTTL: """ Regression for LIT-3338: ``general_settings.user_api_key_cache_ttl`` (which the @@ -238,19 +376,13 @@ class TestManagementObjectTTL: def test_falls_back_to_constant_when_no_default_configured(self): cache = UserApiKeyCache() assert cache.default_in_memory_ttl is None - assert ( - get_management_object_ttl(cache) - == DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL - ) + assert get_management_object_ttl(cache) == DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL def test_resolves_on_a_plain_dual_cache(self): # Many call sites are typed UserApiKeyCache but exercised in tests with a # bare DualCache; the resolver must work on the base type, not just the subclass. assert get_management_object_ttl(DualCache(default_in_memory_ttl=300)) == 300 - assert ( - get_management_object_ttl(DualCache()) - == DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL - ) + assert get_management_object_ttl(DualCache()) == DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL @pytest.mark.asyncio async def test_management_write_uses_configured_ttl_over_constant(self): @@ -260,9 +392,7 @@ class TestManagementObjectTTL: redis_cache=FakeRedisCache(), default_in_memory_ttl=300, ) - assert get_management_object_ttl(cache) != ( - DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL - ) + assert get_management_object_ttl(cache) != (DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL) await cache.async_set_cache( "team_id:abc", From 47bba14336080810bbc3e0e9b1160438cca1c161 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 09:53:37 -0700 Subject: [PATCH 078/157] fix(passthrough): parse Bedrock stream spend incrementally instead of buffering the whole response (#40724) * fix(passthrough): parse Bedrock stream spend incrementally instead of buffering the whole response Bedrock pass-through streaming kept every relayed chunk in memory until EOF and then decoded, parsed and translated the whole stream again for spend logging. Large or concurrent streams could exhaust proxy worker memory. Sync and async passthrough wrappers now hand each chunk to a provider stream collector as it is relayed. Bedrock decodes event-stream frames incrementally, folds consecutive text deltas, and keeps only what stream_chunk_builder needs for usage, tool calls and metadata. Text deltas are no longer retained in the Bedrock and Anthropic stream decoders either. Providers without a collector keep the previous raw-bytes behavior. Collector failures are isolated so spend tracking can never interrupt the customer stream Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(passthrough): assert the spend payload the collector builds instead of mock internals Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(passthrough): type the Bedrock collector helpers by the collector protocol instead of asserting the class Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/litellm_core_utils/litellm_logging.py | 51 +---- litellm/llms/anthropic/chat/handler.py | 5 +- .../base_llm/passthrough/transformation.py | 41 +++- litellm/llms/bedrock/chat/invoke_handler.py | 2 +- .../bedrock/passthrough/transformation.py | 199 +++++++++++------- litellm/passthrough/main.py | 62 ++++-- ...est_azure_ai_passthrough_transformation.py | 6 +- ...test_bedrock_passthrough_transformation.py | 197 ++++++++++++++++- ...test_streaming_interrupt_spend_tracking.py | 118 ++++++++--- 9 files changed, 507 insertions(+), 174 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 6ee68ab21c5..498d662a906 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -14,7 +14,7 @@ from collections.abc import Awaitable, Callable, Iterator, Mapping, Sequence from datetime import datetime as dt_object from functools import lru_cache from types import MappingProxyType, TracebackType -from typing import TYPE_CHECKING, Any, Final, Literal, Optional, Union, cast +from typing import TYPE_CHECKING, Any, Final, Literal, Union, cast from httpx import Response from pydantic import BaseModel, JsonValue @@ -211,7 +211,7 @@ if TYPE_CHECKING: from litellm.integrations.otel.logger import OpenTelemetryV2 from litellm.integrations.otel.model.config import ExporterSpec, OpenTelemetryV2Config from litellm.litellm_core_utils.llm_cost_calc.utils import BilledTokenRates - from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig, LoggedRelayResponse + from litellm.llms.base_llm.passthrough.transformation import PassthroughStreamCollector try: from litellm_enterprise.enterprise_callbacks.callback_controls import ( EnterpriseCallbackControls, @@ -2396,52 +2396,17 @@ class Logging(LiteLLMLoggingBaseClass): for scope in [key for key in spans_logged if isinstance(key, tuple) and key[-1:] == ("success",)]: del spans_logged[scope] - def _flush_passthrough_collected_chunks_helper( - self, - raw_bytes: list[bytes], - provider_config: "BasePassthroughConfig", - ) -> Optional["LoggedRelayResponse"]: - all_chunks: Final = provider_config._convert_raw_bytes_to_str_lines(raw_bytes) - complete_streaming_response: Final = provider_config.handle_logging_collected_chunks( - all_chunks=all_chunks, - litellm_logging_obj=self, - model=self.model, - custom_llm_provider=self.model_call_details.get("custom_llm_provider", ""), - endpoint=self.model_call_details.get("endpoint", ""), - ) - return complete_streaming_response - - def flush_passthrough_collected_chunks( - self, - raw_bytes: list[bytes], - provider_config: "BasePassthroughConfig", - ): + def flush_passthrough_collected_chunks(self, collector: "PassthroughStreamCollector"): """ - Flush collected chunks from the logging object - This is used to log the collected chunks once streaming is done on passthrough endpoints - - 1. Decode the raw bytes to string lines - 2. Get the complete streaming response from the provider config - 3. Log the complete streaming response (trigger success handler) - This is used for passthrough endpoints + Log the response a passthrough stream collector assembled once streaming is done (trigger success handler) """ - complete_streaming_response: Final = self._flush_passthrough_collected_chunks_helper( - raw_bytes=raw_bytes, - provider_config=provider_config, - ) + complete_streaming_response: Final = collector.build_logged_response(litellm_logging_obj=self) if complete_streaming_response is not None: self.success_handler(result=complete_streaming_response) - async def async_flush_passthrough_collected_chunks( - self, - raw_bytes: list[bytes], - provider_config: "BasePassthroughConfig", - ): - complete_streaming_response: Final = self._flush_passthrough_collected_chunks_helper( - raw_bytes=raw_bytes, - provider_config=provider_config, - ) + async def async_flush_passthrough_collected_chunks(self, collector: "PassthroughStreamCollector"): + complete_streaming_response: Final = collector.build_logged_response(litellm_logging_obj=self) if complete_streaming_response is not None: await self.async_success_handler(result=complete_streaming_response) @@ -6505,7 +6470,7 @@ def _get_traceback_str_for_error(error_str: str) -> str: from decimal import Decimal # used for unit testing -from typing import Any, Optional, Union +from typing import Any, Union def create_dummy_standard_logging_payload() -> StandardLoggingPayload: diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py index d1fe4cadf40..82189461403 100644 --- a/litellm/llms/anthropic/chat/handler.py +++ b/litellm/llms/anthropic/chat/handler.py @@ -724,10 +724,11 @@ class ModelResponseIterator: content_block: Final = ContentBlockDelta(**chunk) thinking_blocks: list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock] = [] - self.content_blocks.append(content_block) if "text" in content_block["delta"]: text = content_block["delta"]["text"] - elif "partial_json" in content_block["delta"]: + return text, tool_use, thinking_blocks, provider_specific_fields, reasoning_content + self.content_blocks.append(content_block) + if "partial_json" in content_block["delta"]: # Only emit tool calls if we're in a tool_use or server_tool_use block # web_search_tool_result blocks also have input_json_delta but should not be treated as tool calls # See: https://github.com/BerriAI/litellm/issues/17254 diff --git a/litellm/llms/base_llm/passthrough/transformation.py b/litellm/llms/base_llm/passthrough/transformation.py index 20180c5cfa2..ec938889b88 100644 --- a/litellm/llms/base_llm/passthrough/transformation.py +++ b/litellm/llms/base_llm/passthrough/transformation.py @@ -4,7 +4,7 @@ import re from abc import abstractmethod from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Final, TypeAlias +from typing import TYPE_CHECKING, Final, Protocol, TypeAlias from pydantic import TypeAdapter, ValidationError @@ -80,6 +80,38 @@ def logged_relay_shape( return parsed +class PassthroughStreamCollector(Protocol): + """Consumes relayed stream bytes as they arrive and builds the response logged for spend tracking.""" + + def add(self, chunk: bytes) -> None: ... + + def build_logged_response(self, litellm_logging_obj: LiteLLMLoggingObj) -> LoggedRelayResponse | None: ... + + +class RawBytesStreamCollector: + def __init__( + self, provider_config: BasePassthroughConfig, model: str, custom_llm_provider: str, endpoint: str + ) -> None: + self._provider_config = provider_config + self._model = model + self._custom_llm_provider = custom_llm_provider + self._endpoint = endpoint + self._raw_bytes: list[bytes] = [] # mutable-ok: instance buffer for streaming chunks + + def add(self, chunk: bytes) -> None: + self._raw_bytes.append(chunk) + + def build_logged_response(self, litellm_logging_obj: LiteLLMLoggingObj) -> LoggedRelayResponse | None: + all_chunks: Final = self._provider_config._convert_raw_bytes_to_str_lines(self._raw_bytes) + return self._provider_config.handle_logging_collected_chunks( + all_chunks=all_chunks, + litellm_logging_obj=litellm_logging_obj, + model=self._model, + custom_llm_provider=self._custom_llm_provider, + endpoint=self._endpoint, + ) + + class BasePassthroughConfig(BaseLLMModelInfo): @abstractmethod def is_streaming_request(self, endpoint: str, request_data: dict) -> bool: @@ -182,6 +214,13 @@ class BasePassthroughConfig(BaseLLMModelInfo): ) -> LoggedRelayResponse | None: return None + def create_stream_collector( + self, model: str, custom_llm_provider: str, endpoint: str + ) -> PassthroughStreamCollector: + return RawBytesStreamCollector( + provider_config=self, model=model, custom_llm_provider=custom_llm_provider, endpoint=endpoint + ) + def _convert_raw_bytes_to_str_lines(self, raw_bytes: list[bytes]) -> list[str]: """ Converts a list of raw bytes into a list of string lines, similar to aiter_lines() diff --git a/litellm/llms/bedrock/chat/invoke_handler.py b/litellm/llms/bedrock/chat/invoke_handler.py index 5f8a5544d65..5c489ecb360 100644 --- a/litellm/llms/bedrock/chat/invoke_handler.py +++ b/litellm/llms/bedrock/chat/invoke_handler.py @@ -490,10 +490,10 @@ class AWSEventStreamDecoder: reasoning_content: str | None = None thinking_blocks: list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock] | None = None - self.content_blocks.append(delta_obj) if "text" in delta_obj: text = delta_obj["text"] elif "toolUse" in delta_obj: + self.content_blocks.append(delta_obj) # When json_mode is True and this is the internal json_tool_call, # convert tool input to text content instead of tool call arguments if self.json_mode is True and self._current_tool_name == RESPONSE_FORMAT_TOOL_NAME: diff --git a/litellm/llms/bedrock/passthrough/transformation.py b/litellm/llms/bedrock/passthrough/transformation.py index fb8bc4f191f..6a120f41cb6 100644 --- a/litellm/llms/bedrock/passthrough/transformation.py +++ b/litellm/llms/bedrock/passthrough/transformation.py @@ -1,23 +1,129 @@ import json -from collections.abc import Mapping +from collections.abc import Callable, Mapping, Sequence from typing import TYPE_CHECKING, Final, Optional, cast import httpx from httpx import Response +from litellm._logging import verbose_logger from litellm.litellm_core_utils.litellm_logging import Logging -from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig +from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig, PassthroughStreamCollector +from litellm.types.utils import ModelResponseStream from ..base_aws_llm import BaseAWSLLM from ..common_utils import BedrockError, BedrockEventStreamDecoderBase, BedrockModelInfo if TYPE_CHECKING: + from botocore.eventstream import EventStreamMessage from httpx import URL from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + from litellm.llms.bedrock.chat.invoke_handler import AWSEventStreamDecoder from litellm.types.utils import CostResponseTypes +_TEXT_ONLY_DELTA_FIELDS: Final = frozenset({"content", "role"}) + + +def _plain_text_delta(chunk: ModelResponseStream) -> str | None: + """Return the delta text when the chunk carries nothing else that stream_chunk_builder reads.""" + if chunk.get("usage") is not None or chunk.provider_specific_fields or len(chunk.choices) != 1: + return None + choice: Final = chunk.choices[0] + if choice.finish_reason or choice.logprobs is not None: + return None + populated: Final = frozenset(key for key, value in choice.delta.model_dump().items() if value is not None) + if not populated <= _TEXT_ONLY_DELTA_FIELDS: + return None + content: Final = choice.delta.get("content") + return content if isinstance(content, str) else None + + +class _CoalescedChunks: + """Retains translated chunks with consecutive text deltas folded into one, so memory tracks the response text, + not the event count.""" + + def __init__(self) -> None: + self._chunks: list[ModelResponseStream] = [] # mutable-ok: instance accumulator for streaming chunks + self._open_text_parts: list[str] = [] # mutable-ok: text deltas pending fold into self._chunks[-1] + + def add(self, chunk: ModelResponseStream) -> None: + text: Final = _plain_text_delta(chunk) + if text is not None and self._open_text_parts: + self._open_text_parts.append(text) + return + self._seal_text_run() + self._chunks.append(chunk) + if text is not None: + self._open_text_parts.append(text) + + def _seal_text_run(self) -> None: + if len(self._open_text_parts) > 1: + self._chunks[-1].choices[0].delta.content = "".join(self._open_text_parts) + self._open_text_parts.clear() + + def chunks(self) -> Sequence[ModelResponseStream]: + self._seal_text_run() + return self._chunks + + +def _translate_message(decoder: "AWSEventStreamDecoder", message: str) -> ModelResponseStream | None: + from litellm.litellm_core_utils.streaming_handler import ( + convert_generic_chunk_to_model_response_stream, + generic_chunk_has_all_required_fields, + ) + from litellm.types.utils import GenericStreamingChunk + + translated_chunk: Final = decoder._chunk_parser(chunk_data=json.loads(message)) + if isinstance(translated_chunk, ModelResponseStream): + return translated_chunk + if generic_chunk_has_all_required_fields(cast(dict, translated_chunk)): + return convert_generic_chunk_to_model_response_stream(cast(GenericStreamingChunk, translated_chunk)) + return None + + +def _build_logged_response( + chunks: Sequence[ModelResponseStream], litellm_logging_obj: "LiteLLMLoggingObj" +) -> Optional["CostResponseTypes"]: + from litellm.main import stream_chunk_builder + + if len(chunks) == 0: + return None + return stream_chunk_builder(chunks=list(chunks), logging_obj=litellm_logging_obj) + + +class BedrockEventStreamCollector: + """Decodes and translates Bedrock event-stream frames as they are relayed instead of buffering the stream.""" + + def __init__( + self, + parse_event: Callable[["EventStreamMessage"], str | None], + decoder: Optional["AWSEventStreamDecoder"], + ) -> None: + from botocore.eventstream import EventStreamBuffer + + self._parse_event = parse_event + self._decoder = decoder + self._event_stream_buffer: Final[EventStreamBuffer] = EventStreamBuffer() + self._chunks: Final = _CoalescedChunks() + + def add(self, chunk: bytes) -> None: + if self._decoder is None: + return + self._event_stream_buffer.add_data(chunk) + for event in self._event_stream_buffer: + self._add_event(self._decoder, event) + + def _add_event(self, decoder: "AWSEventStreamDecoder", event: "EventStreamMessage") -> None: + message: Final = self._parse_event(event) + translated: Final = _translate_message(decoder, message) if message is not None else None + if translated is not None: + self._chunks.add(translated) + + def build_logged_response(self, litellm_logging_obj: "LiteLLMLoggingObj") -> Optional["CostResponseTypes"]: + return _build_logged_response(self._chunks.chunks(), litellm_logging_obj) + + class BedrockPassthroughConfig(BaseAWSLLM, BedrockModelInfo, BedrockEventStreamDecoderBase, BasePassthroughConfig): def get_error_class( self, @@ -168,87 +274,32 @@ class BedrockPassthroughConfig(BaseAWSLLM, BedrockModelInfo, BedrockEventStreamD return litellm_model_response - def _convert_raw_bytes_to_str_lines(self, raw_bytes: list[bytes]) -> list[str]: - from botocore.eventstream import EventStreamBuffer - - all_chunks: Final = [] - event_stream_buffer: Final = EventStreamBuffer() - for chunk in raw_bytes: - event_stream_buffer.add_data(chunk) - for event in event_stream_buffer: - message = self._parse_message_from_event(event) - if message is not None: - all_chunks.append(message) - - return all_chunks - - def handle_logging_collected_chunks( - self, - all_chunks: list[str], - litellm_logging_obj: "LiteLLMLoggingObj", - model: str, - custom_llm_provider: str, - endpoint: str, - ) -> Optional["CostResponseTypes"]: - """ - 1. Convert all_chunks to a ModelResponseStream - 2. combine model_response_stream to model_response - 3. Return the model_response - """ - - from litellm.litellm_core_utils.streaming_handler import ( - convert_generic_chunk_to_model_response_stream, - generic_chunk_has_all_required_fields, + def create_stream_collector( + self, model: str, custom_llm_provider: str, endpoint: str + ) -> PassthroughStreamCollector: + return BedrockEventStreamCollector( + parse_event=self._parse_message_from_event, + decoder=self._get_event_stream_decoder(model=model, endpoint=endpoint), ) + + def _get_event_stream_decoder(self, model: str, endpoint: str) -> Optional["AWSEventStreamDecoder"]: from litellm.llms.bedrock.chat import get_bedrock_event_stream_decoder from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( AmazonInvokeConfig, ) - from litellm.main import stream_chunk_builder - from litellm.types.utils import GenericStreamingChunk, ModelResponseStream - all_translated_chunks: Final = [] if "invoke" in endpoint: invoke_provider: Final = AmazonInvokeConfig.get_bedrock_invoke_provider(model) if invoke_provider is None: - raise ValueError(f"Invalid invoke provider: {invoke_provider}, for model: {model}") - obj = get_bedrock_event_stream_decoder( - invoke_provider=invoke_provider, - model=model, - sync_stream=True, - json_mode=False, - ) - elif "converse" in endpoint: - obj = get_bedrock_event_stream_decoder( - invoke_provider=None, - model=model, - sync_stream=True, - json_mode=False, - ) - else: - return None - - for chunk in all_chunks: - message = json.loads(chunk) - translated_chunk = obj._chunk_parser(chunk_data=message) - - if isinstance(translated_chunk, dict) and generic_chunk_has_all_required_fields( - cast(dict, translated_chunk) - ): - chunk_obj = convert_generic_chunk_to_model_response_stream( - cast(GenericStreamingChunk, translated_chunk) + verbose_logger.warning( + "Bedrock passthrough spend tracking skipped: no invoke provider for model %s", model ) - elif isinstance(translated_chunk, ModelResponseStream): - chunk_obj = translated_chunk - else: - continue - - all_translated_chunks.append(chunk_obj) - - if len(all_translated_chunks) > 0: - model_response: Final = stream_chunk_builder( - chunks=all_translated_chunks, - logging_obj=litellm_logging_obj, + return None + return get_bedrock_event_stream_decoder( + invoke_provider=invoke_provider, model=model, sync_stream=True, json_mode=False + ) + if "converse" in endpoint: + return get_bedrock_event_stream_decoder( + invoke_provider=None, model=model, sync_stream=True, json_mode=False ) - return model_response return None diff --git a/litellm/passthrough/main.py b/litellm/passthrough/main.py index 7076683f294..73d8bab686b 100644 --- a/litellm/passthrough/main.py +++ b/litellm/passthrough/main.py @@ -17,7 +17,7 @@ from httpx._types import CookieTypes, QueryParamTypes, RequestContent, RequestFi from litellm._logging import verbose_logger from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj -from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig +from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig, PassthroughStreamCollector from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from litellm.passthrough.utils import CommonUtils @@ -36,6 +36,35 @@ def _as_generator(iterable: Iterator[bytes]) -> Generator[bytes, bytes, None]: yield from iterable +class _SpendCollection: + """Feeds relayed chunks to the provider's stream collector without letting spend tracking break the relay.""" + + def __init__(self, provider_config: BasePassthroughConfig, litellm_logging_obj: LiteLLMLoggingObj) -> None: + self.collector: Final[PassthroughStreamCollector] = provider_config.create_stream_collector( + model=litellm_logging_obj.model, + custom_llm_provider=litellm_logging_obj.model_call_details.get("custom_llm_provider", ""), + endpoint=litellm_logging_obj.model_call_details.get("endpoint", ""), + ) + self.chunk_count = 0 + self._failed = False + + def add(self, chunk: bytes) -> None: + self.chunk_count += 1 + if self._failed: + return + try: + self.collector.add(chunk) + except Exception as e: # noqa: BLE001 # Safe catch-all: spend tracking must never break the relayed stream + self._failed = True + verbose_logger.exception( + "Passthrough spend-tracking collector failed; spend dropped for this stream: %s", e + ) + + @property + def should_flush(self) -> bool: + return self.chunk_count > 0 and not self._failed + + class AsyncPassthroughStreamingResponse(AsyncGenerator[bytes, bytes]): def __init__( self, @@ -50,8 +79,7 @@ class AsyncPassthroughStreamingResponse(AsyncGenerator[bytes, bytes]): self._response: httpx.Response self._iterator: AsyncGenerator[bytes, bytes] self._litellm_logging_obj = litellm_logging_obj - self._provider_config = provider_config - self._raw_bytes: list[bytes] = [] # mutable-ok: instance buffer for streaming chunks + self._spend = _SpendCollection(provider_config, litellm_logging_obj) self._flush_scheduled = False self._background_tasks: set[asyncio.Task] = set() # mutable-ok: instance set for background task tracking self._hidden_params: dict[str, object] = {} # mutable-ok: router attaches response headers here in place @@ -101,16 +129,13 @@ class AsyncPassthroughStreamingResponse(AsyncGenerator[bytes, bytes]): return _init().__await__() def _start_flush(self) -> None: - if self._flush_scheduled or not self._raw_bytes: + if self._flush_scheduled or not self._spend.should_flush: return self._flush_scheduled = True try: task: Final = asyncio.create_task( - self._litellm_logging_obj.async_flush_passthrough_collected_chunks( - raw_bytes=self._raw_bytes, - provider_config=self._provider_config, - ) + self._litellm_logging_obj.async_flush_passthrough_collected_chunks(collector=self._spend.collector) ) self._background_tasks.add(task) @@ -118,8 +143,8 @@ class AsyncPassthroughStreamingResponse(AsyncGenerator[bytes, bytes]): task.add_done_callback(self._background_tasks.discard) except Exception as e: # noqa: BLE001 # Safe catch-all for verbose logging verbose_logger.exception( - "Failed to schedule passthrough spend-tracking flush; %d buffered chunks dropped: %s", - len(self._raw_bytes), + "Failed to schedule passthrough spend-tracking flush; %d collected chunks dropped: %s", + self._spend.chunk_count, e, ) @@ -134,7 +159,7 @@ class AsyncPassthroughStreamingResponse(AsyncGenerator[bytes, bytes]): await self # pyright: ignore[reportGeneralTypeIssues] # structural type check misses __await__ try: chunk: Final = await anext(self._iterator) - self._raw_bytes.append(chunk) + self._spend.add(chunk) except Exception: # noqa: BLE001 # Safe catch-all for cleanup logic self._start_flush() try: @@ -181,13 +206,12 @@ class PassthroughStreamingResponse(Generator[bytes, bytes, None]): self.headers = response.headers self.status_code = response.status_code self._litellm_logging_obj = litellm_logging_obj - self._provider_config = provider_config self._iterator: Generator[bytes, bytes, None] = _as_generator(response.iter_bytes()) - self._raw_bytes: list[bytes] = [] # mutable-ok: instance buffer for streaming chunks + self._spend = _SpendCollection(provider_config, litellm_logging_obj) self._flush_scheduled = False def _start_flush(self) -> None: - if self._flush_scheduled or not self._raw_bytes: + if self._flush_scheduled or not self._spend.should_flush: return self._flush_scheduled = True @@ -195,14 +219,12 @@ class PassthroughStreamingResponse(Generator[bytes, bytes, None]): try: executor.submit( - self._litellm_logging_obj.flush_passthrough_collected_chunks, - raw_bytes=self._raw_bytes, - provider_config=self._provider_config, + self._litellm_logging_obj.flush_passthrough_collected_chunks, collector=self._spend.collector ) except Exception as e: # noqa: BLE001 # Safe catch-all for verbose logging verbose_logger.exception( - "Failed to schedule passthrough spend-tracking flush; %d buffered chunks dropped: %s", - len(self._raw_bytes), + "Failed to schedule passthrough spend-tracking flush; %d collected chunks dropped: %s", + self._spend.chunk_count, e, ) @@ -212,7 +234,7 @@ class PassthroughStreamingResponse(Generator[bytes, bytes, None]): def __next__(self) -> bytes: try: chunk: Final = next(self._iterator) - self._raw_bytes.append(chunk) + self._spend.add(chunk) except Exception: # noqa: BLE001 # Safe catch-all for cleanup logic self._start_flush() try: diff --git a/tests/test_litellm/llms/azure_ai/passthrough/test_azure_ai_passthrough_transformation.py b/tests/test_litellm/llms/azure_ai/passthrough/test_azure_ai_passthrough_transformation.py index c8007acf70f..f00698a6624 100644 --- a/tests/test_litellm/llms/azure_ai/passthrough/test_azure_ai_passthrough_transformation.py +++ b/tests/test_litellm/llms/azure_ai/passthrough/test_azure_ai_passthrough_transformation.py @@ -608,9 +608,11 @@ async def test_streaming_responses_relay_flush_reaches_the_success_callbacks_wit ) stream = "event: response.completed\ndata: " + json.dumps(RESPONSES_COMPLETED_EVENT) + "\n\n" - await logging_obj.async_flush_passthrough_collected_chunks( - raw_bytes=[stream.encode()], provider_config=AzureAIPassthroughConfig() + collector = AzureAIPassthroughConfig().create_stream_collector( + model="gpt-5.4-mini", custom_llm_provider="azure_ai", endpoint="gpt/openai/responses" ) + collector.add(stream.encode()) + await logging_obj.async_flush_passthrough_collected_chunks(collector=collector) info = litellm.get_model_info("azure_ai/gpt-5.4-mini") assert probe.logged_call_type == "allm_passthrough_route" diff --git a/tests/test_litellm/llms/bedrock/passthrough/test_bedrock_passthrough_transformation.py b/tests/test_litellm/llms/bedrock/passthrough/test_bedrock_passthrough_transformation.py index b005d77ac8b..f2a9af11af7 100644 --- a/tests/test_litellm/llms/bedrock/passthrough/test_bedrock_passthrough_transformation.py +++ b/tests/test_litellm/llms/bedrock/passthrough/test_bedrock_passthrough_transformation.py @@ -1,7 +1,19 @@ +import base64 +import json +import struct +import tracemalloc +from binascii import crc32 +from datetime import datetime from unittest.mock import patch - +from litellm.litellm_core_utils.litellm_logging import Logging +from litellm.llms.base_llm.passthrough.transformation import PassthroughStreamCollector from litellm.llms.bedrock.passthrough.transformation import BedrockPassthroughConfig +from litellm.types.utils import ModelResponse + +CONVERSE_MODEL = "anthropic.claude-sonnet-4-5-20250929-v1:0" +CONVERSE_STREAM_ENDPOINT = f"/model/{CONVERSE_MODEL}/converse-stream" +INVOKE_STREAM_ENDPOINT = f"/model/{CONVERSE_MODEL}/invoke-with-response-stream" def test_bedrock_passthrough_get_complete_url_default_endpoint(): @@ -500,3 +512,186 @@ def test_bedrock_passthrough_model_id_without_arn(): f"https://bedrock-runtime.us-east-1.amazonaws.com/model/{model_id}/converse" ) assert url_str == expected_url + + +def _event_frame(event_type: str, payload: dict) -> bytes: + def header(name: str, value: str) -> bytes: + name_b, value_b = name.encode(), value.encode() + return struct.pack("!B", len(name_b)) + name_b + struct.pack("!B", 7) + struct.pack("!H", len(value_b)) + value_b + + payload_b = json.dumps(payload, separators=(",", ":")).encode() + headers_b = ( + header(":event-type", event_type) + + header(":content-type", "application/json") + + header(":message-type", "event") + ) + prelude = struct.pack("!II", 12 + len(headers_b) + len(payload_b) + 4, len(headers_b)) + prelude_crc = crc32(prelude) & 0xFFFFFFFF + message = struct.pack("!I", prelude_crc) + headers_b + payload_b + return prelude + message + struct.pack("!I", crc32(message, prelude_crc) & 0xFFFFFFFF) + + +def _text_block(index: int, texts: list[str]) -> bytes: + return ( + _event_frame("contentBlockStart", {"contentBlockIndex": index, "start": {}}) + + b"".join( + _event_frame("contentBlockDelta", {"contentBlockIndex": index, "delta": {"text": text}}) for text in texts + ) + + _event_frame("contentBlockStop", {"contentBlockIndex": index}) + ) + + +def _stream_tail(stop_reason: str, output_tokens: int) -> bytes: + return _event_frame("messageStop", {"stopReason": stop_reason}) + _event_frame( + "metadata", + { + "metrics": {"latencyMs": 1234}, + "usage": {"inputTokens": 25, "outputTokens": output_tokens, "totalTokens": 25 + output_tokens}, + }, + ) + + +def _invoke_chunk(payload: dict) -> bytes: + return _event_frame("chunk", {"bytes": base64.b64encode(json.dumps(payload).encode()).decode()}) + + +def _stream_logging_obj(endpoint: str) -> Logging: + logging_obj = Logging( + model=CONVERSE_MODEL, + messages=[], + stream=True, + call_type="pass_through_endpoint", + start_time=datetime.now(), + litellm_call_id="call-1", + function_id="fn-1", + ) + logging_obj.model_call_details["custom_llm_provider"] = "bedrock" + logging_obj.model_call_details["endpoint"] = endpoint + return logging_obj + + +def _converse_stream_logging_obj() -> Logging: + return _stream_logging_obj(CONVERSE_STREAM_ENDPOINT) + + +def _stream_collector(endpoint: str) -> PassthroughStreamCollector: + return BedrockPassthroughConfig().create_stream_collector( + model=CONVERSE_MODEL, custom_llm_provider="bedrock", endpoint=endpoint + ) + + +def _converse_stream_collector() -> PassthroughStreamCollector: + return _stream_collector(CONVERSE_STREAM_ENDPOINT) + + +def _feed(collector: PassthroughStreamCollector, stream: bytes, chunk_size: int = 16384) -> None: + for offset in range(0, len(stream), chunk_size): + collector.add(stream[offset : offset + chunk_size]) + + +def test_converse_stream_collector_keeps_usage_without_retaining_the_stream(): + texts = [f"tok{i} " for i in range(4000)] + stream = _event_frame("messageStart", {"role": "assistant"}) + _text_block(0, texts) + _stream_tail("end_turn", 4000) + _feed(_converse_stream_collector(), stream) + + tracemalloc.start() + try: + base = tracemalloc.get_traced_memory()[0] + collector = _converse_stream_collector() + _feed(collector, stream) + retained = tracemalloc.get_traced_memory()[0] - base + finally: + tracemalloc.stop() + + assert retained < len(stream) // 4 + + response = collector.build_logged_response(_converse_stream_logging_obj()) + assert isinstance(response, ModelResponse) + assert response.choices[0].message.content == "".join(texts) + assert response.choices[0].finish_reason == "stop" + assert (response.usage.prompt_tokens, response.usage.completion_tokens) == (25, 4000) + + +def test_converse_stream_collector_keeps_tool_calls_between_text_runs(): + stream = ( + _event_frame("messageStart", {"role": "assistant"}) + + _text_block(0, ["Let me ", "check."]) + + _event_frame( + "contentBlockStart", + {"contentBlockIndex": 1, "start": {"toolUse": {"toolUseId": "tool-1", "name": "get_weather"}}}, + ) + + _event_frame("contentBlockDelta", {"contentBlockIndex": 1, "delta": {"toolUse": {"input": '{"city": '}}}) + + _event_frame("contentBlockDelta", {"contentBlockIndex": 1, "delta": {"toolUse": {"input": '"Paris"}'}}}) + + _event_frame("contentBlockStop", {"contentBlockIndex": 1}) + + _text_block(2, ["Done", "."]) + + _stream_tail("tool_use", 12) + ) + collector = _converse_stream_collector() + _feed(collector, stream, chunk_size=7) + + response = collector.build_logged_response(_converse_stream_logging_obj()) + assert isinstance(response, ModelResponse) + message = response.choices[0].message + assert message.content == "Let me check.Done." + assert [(call.function.name, call.function.arguments) for call in message.tool_calls] == [ + ("get_weather", '{"city": "Paris"}') + ] + assert response.choices[0].finish_reason == "tool_calls" + assert (response.usage.prompt_tokens, response.usage.completion_tokens) == (25, 12) + + +def test_invoke_stream_collector_keeps_usage_without_retaining_the_stream(): + texts = [f"tok{i} " for i in range(4000)] + stream = ( + _invoke_chunk( + { + "type": "message_start", + "message": { + "id": "msg-1", + "type": "message", + "role": "assistant", + "model": CONVERSE_MODEL, + "content": [], + "stop_reason": None, + "usage": {"input_tokens": 25, "output_tokens": 1}, + }, + } + ) + + _invoke_chunk({"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}) + + b"".join( + _invoke_chunk({"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": text}}) + for text in texts + ) + + _invoke_chunk({"type": "content_block_stop", "index": 0}) + + _invoke_chunk( + {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 4000}} + ) + + _invoke_chunk({"type": "message_stop"}) + ) + _feed(_stream_collector(INVOKE_STREAM_ENDPOINT), stream) + + tracemalloc.start() + try: + base = tracemalloc.get_traced_memory()[0] + collector = _stream_collector(INVOKE_STREAM_ENDPOINT) + _feed(collector, stream) + retained = tracemalloc.get_traced_memory()[0] - base + finally: + tracemalloc.stop() + + assert retained < len(stream) // 4 + + response = collector.build_logged_response(_stream_logging_obj(INVOKE_STREAM_ENDPOINT)) + assert isinstance(response, ModelResponse) + assert response.choices[0].message.content == "".join(texts) + assert response.choices[0].finish_reason == "stop" + assert (response.usage.prompt_tokens, response.usage.completion_tokens) == (25, 4000) + + +def test_stream_collector_logs_nothing_for_an_unrecognized_endpoint(): + collector = BedrockPassthroughConfig().create_stream_collector( + model=CONVERSE_MODEL, custom_llm_provider="bedrock", endpoint=f"/model/{CONVERSE_MODEL}/rerank" + ) + collector.add(_event_frame("messageStart", {"role": "assistant"})) + + assert collector.build_logged_response(_converse_stream_logging_obj()) is None diff --git a/tests/test_litellm/passthrough/test_streaming_interrupt_spend_tracking.py b/tests/test_litellm/passthrough/test_streaming_interrupt_spend_tracking.py index a88b0ef0c4b..5e13db9439b 100644 --- a/tests/test_litellm/passthrough/test_streaming_interrupt_spend_tracking.py +++ b/tests/test_litellm/passthrough/test_streaming_interrupt_spend_tracking.py @@ -34,6 +34,34 @@ class _ImmediateExecutor: fn(*args, **kwargs) +class _RecordingCollector: + def __init__(self) -> None: + self.chunks: List[bytes] = [] + + def add(self, chunk: bytes) -> None: + self.chunks.append(chunk) + + def build_logged_response(self, litellm_logging_obj: MagicMock) -> bytes: + return b"".join(self.chunks) + + +class _FailingCollector(_RecordingCollector): + def add(self, chunk: bytes) -> None: + raise ValueError("bad frame") + + +def _provider_config(collector: _RecordingCollector) -> MagicMock: + provider_config = MagicMock() + provider_config.create_stream_collector.return_value = collector + return provider_config + + +def _spend_payload(flush_mock: MagicMock) -> bytes: + flush_mock.assert_called_once() + collector = flush_mock.call_args.kwargs["collector"] + return collector.build_logged_response(litellm_logging_obj=MagicMock()) + + @pytest.mark.asyncio async def test_asyncpassthroughstreamingresponse_flushes_on_normal_completion(): from litellm.passthrough.main import AsyncPassthroughStreamingResponse @@ -48,13 +76,12 @@ async def test_asyncpassthroughstreamingresponse_flushes_on_normal_completion(): return mock_response mock_logging_obj = _make_logging_obj() - provider_config = MagicMock() received = [] received_response = AsyncPassthroughStreamingResponse( response=response_coro(), litellm_logging_obj=mock_logging_obj, - provider_config=provider_config, + provider_config=_provider_config(_RecordingCollector()), ) async for chunk in received_response: @@ -67,12 +94,7 @@ async def test_asyncpassthroughstreamingresponse_flushes_on_normal_completion(): await asyncio.sleep(0) - mock_logging_obj.async_flush_passthrough_collected_chunks.assert_called_once() - call_kwargs = ( - mock_logging_obj.async_flush_passthrough_collected_chunks.call_args.kwargs - ) - assert call_kwargs["raw_bytes"] == chunks - assert call_kwargs["provider_config"] is provider_config + assert _spend_payload(mock_logging_obj.async_flush_passthrough_collected_chunks) == b"".join(chunks) @pytest.mark.asyncio @@ -93,12 +115,11 @@ async def test_asyncpassthroughstreamingresponse_flushes_on_client_disconnect(): return mock_response mock_logging_obj = _make_logging_obj() - provider_config = MagicMock() gen = AsyncPassthroughStreamingResponse( response=response_coro(), litellm_logging_obj=mock_logging_obj, - provider_config=provider_config, + provider_config=_provider_config(_RecordingCollector()), ) received = [await gen.__anext__()] @@ -108,11 +129,7 @@ async def test_asyncpassthroughstreamingresponse_flushes_on_client_disconnect(): await asyncio.sleep(0) - mock_logging_obj.async_flush_passthrough_collected_chunks.assert_called_once() - call_kwargs = ( - mock_logging_obj.async_flush_passthrough_collected_chunks.call_args.kwargs - ) - assert call_kwargs["raw_bytes"] == [chunks[0]] + assert _spend_payload(mock_logging_obj.async_flush_passthrough_collected_chunks) == chunks[0] @pytest.mark.asyncio @@ -178,14 +195,13 @@ async def test_asyncpassthroughstreamingresponse_flushes_on_upstream_exception_w return mock_response mock_logging_obj = _make_logging_obj() - provider_config = MagicMock() received = [] async def _drain(): async for chunk in AsyncPassthroughStreamingResponse( response=response_coro(), litellm_logging_obj=mock_logging_obj, - provider_config=provider_config, + provider_config=_provider_config(_RecordingCollector()), ): received.append(chunk) @@ -196,11 +212,7 @@ async def test_asyncpassthroughstreamingresponse_flushes_on_upstream_exception_w await asyncio.sleep(0) - mock_logging_obj.async_flush_passthrough_collected_chunks.assert_called_once() - call_kwargs = ( - mock_logging_obj.async_flush_passthrough_collected_chunks.call_args.kwargs - ) - assert call_kwargs["raw_bytes"] == partial_chunks + assert _spend_payload(mock_logging_obj.async_flush_passthrough_collected_chunks) == b"".join(partial_chunks) def test_passthroughstreamingresponse_flushes_on_normal_completion(): @@ -221,12 +233,11 @@ def test_passthroughstreamingresponse_flushes_on_normal_completion(): mock_logging_obj = MagicMock() mock_logging_obj.flush_passthrough_collected_chunks = MagicMock() - provider_config = MagicMock() received_responce = PassthroughStreamingResponse( response=mock_response, litellm_logging_obj=mock_logging_obj, - provider_config=provider_config, + provider_config=_provider_config(_RecordingCollector()), ) with patch("litellm.utils.executor", _ImmediateExecutor()): @@ -237,7 +248,7 @@ def test_passthroughstreamingresponse_flushes_on_normal_completion(): assert received_responce.headers["content-type"] == "application/octet-stream" assert received_responce.headers["x-request-id"] == "req-123" - mock_logging_obj.flush_passthrough_collected_chunks.assert_called_once() + assert _spend_payload(mock_logging_obj.flush_passthrough_collected_chunks) == b"".join(chunks) def test_passthroughstreamingresponse_flushes_on_early_close(): @@ -258,19 +269,66 @@ def test_passthroughstreamingresponse_flushes_on_early_close(): mock_logging_obj = MagicMock() mock_logging_obj.flush_passthrough_collected_chunks = MagicMock() - provider_config = MagicMock() with patch("litellm.utils.executor", _ImmediateExecutor()): gen = PassthroughStreamingResponse( response=mock_response, litellm_logging_obj=mock_logging_obj, - provider_config=provider_config, + provider_config=_provider_config(_RecordingCollector()), ) first = next(gen) gen.close() assert first == chunks[0] - mock_logging_obj.flush_passthrough_collected_chunks.assert_called_once() - call_kwargs = mock_logging_obj.flush_passthrough_collected_chunks.call_args.kwargs - assert call_kwargs["raw_bytes"] == [chunks[0]] + assert _spend_payload(mock_logging_obj.flush_passthrough_collected_chunks) == chunks[0] + + +@pytest.mark.asyncio +async def test_asyncpassthroughstreamingresponse_relays_the_stream_when_spend_parsing_fails(): + from litellm.passthrough.main import AsyncPassthroughStreamingResponse + + chunks = [b"chunk-1", b"chunk-2", b"chunk-3"] + mock_response = _make_streaming_response(chunks) + + async def response_coro(): + return mock_response + + mock_logging_obj = _make_logging_obj() + + received = [ + chunk + async for chunk in AsyncPassthroughStreamingResponse( + response=response_coro(), + litellm_logging_obj=mock_logging_obj, + provider_config=_provider_config(_FailingCollector()), + ) + ] + await asyncio.sleep(0) + + assert received == chunks + mock_logging_obj.async_flush_passthrough_collected_chunks.assert_not_called() + + +def test_passthroughstreamingresponse_relays_the_stream_when_spend_parsing_fails(): + from litellm.passthrough.main import PassthroughStreamingResponse + + chunks = [b"a", b"b", b"c"] + mock_response = MagicMock(spec=httpx.Response) + mock_response.status_code = 200 + mock_response.headers = httpx.Headers({"content-type": "application/octet-stream"}) + mock_response.iter_bytes = lambda: iter(chunks) + + mock_logging_obj = MagicMock() + mock_logging_obj.flush_passthrough_collected_chunks = MagicMock() + + received = list( + PassthroughStreamingResponse( + response=mock_response, + litellm_logging_obj=mock_logging_obj, + provider_config=_provider_config(_FailingCollector()), + ) + ) + + assert received == chunks + mock_logging_obj.flush_passthrough_collected_chunks.assert_not_called() From de79310954935c643389d7a31f07f9cee7ab6292 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 09:54:08 -0700 Subject: [PATCH 079/157] feat(secret_managers): support customer-managed KMS key for virtual keys stored in AWS Secrets Manager (#40475) Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../secret_managers/aws_secret_manager_v2.py | 8 ++- litellm/types/secret_managers/main.py | 3 + .../test_aws_secret_manager_v2.py | 59 ++++++++++++++++++- 3 files changed, 68 insertions(+), 2 deletions(-) diff --git a/litellm/secret_managers/aws_secret_manager_v2.py b/litellm/secret_managers/aws_secret_manager_v2.py index e86c8e7c919..d75375a01cc 100644 --- a/litellm/secret_managers/aws_secret_manager_v2.py +++ b/litellm/secret_managers/aws_secret_manager_v2.py @@ -47,6 +47,7 @@ class AWSSecretsManagerV2(BaseAWSLLM, BaseSecretManager): aws_web_identity_token: str | None = None, aws_sts_endpoint: str | None = None, replica_regions: list[str] | None = None, + kms_key_id: str | None = None, **kwargs, ): BaseSecretManager.__init__(self, **kwargs) @@ -61,6 +62,7 @@ class AWSSecretsManagerV2(BaseAWSLLM, BaseSecretManager): self.aws_web_identity_token = aws_web_identity_token self.aws_sts_endpoint = aws_sts_endpoint self.replica_regions: list[str] = replica_regions or [] + self.kms_key_id = kms_key_id @classmethod def validate_environment(cls): @@ -106,7 +108,8 @@ class AWSSecretsManagerV2(BaseAWSLLM, BaseSecretManager): # Remove None values aws_kwargs = {k: v for k, v in aws_kwargs.items() if v is not None} - litellm.secret_manager_client = cls(**aws_kwargs) + kms_key_id: Final = key_management_settings.kms_key_id if key_management_settings is not None else None + litellm.secret_manager_client = cls(kms_key_id=kms_key_id, **aws_kwargs) litellm._key_management_system = KeyManagementSystem.AWS_SECRET_MANAGER except Exception as e: @@ -275,6 +278,9 @@ class AWSSecretsManagerV2(BaseAWSLLM, BaseSecretManager): if description: data["Description"] = description + if self.kms_key_id: + data["KmsKeyId"] = self.kms_key_id + # ✅ Normalize tags to AWS format if tags: if isinstance(tags, dict): diff --git a/litellm/types/secret_managers/main.py b/litellm/types/secret_managers/main.py index 599e5746dfb..148e680a236 100644 --- a/litellm/types/secret_managers/main.py +++ b/litellm/types/secret_managers/main.py @@ -45,6 +45,9 @@ class KeyManagementSettings(LiteLLMPydanticObjectBase): tags: dict[str, str] | None = None """Optional tags to attach when creating secrets (e.g. {"Environment": "Prod", "Owner": "AI-Platform"}).""" + kms_key_id: str | None = None + """Optional customer-managed KMS key (ID, alias or ARN) used to encrypt secrets created in AWS Secrets Manager.""" + custom_secret_manager: str | None = None """ Path to custom secret manager class (e.g. "my_secret_manager.InMemorySecretManager") diff --git a/tests/test_litellm/secret_managers/test_aws_secret_manager_v2.py b/tests/test_litellm/secret_managers/test_aws_secret_manager_v2.py index 7e655b70756..03422d5433c 100644 --- a/tests/test_litellm/secret_managers/test_aws_secret_manager_v2.py +++ b/tests/test_litellm/secret_managers/test_aws_secret_manager_v2.py @@ -5,11 +5,68 @@ Tests the write/read/delete cycle for JSON and simple string secrets. """ import json -from unittest.mock import AsyncMock, patch +from unittest.mock import AsyncMock, MagicMock, patch import pytest +import respx +import litellm from litellm.secret_managers.aws_secret_manager_v2 import AWSSecretsManagerV2 +from litellm.types.secret_managers.main import KeyManagementSettings + +_STATIC_CREDENTIALS = {"aws_access_key_id": "test-key", "aws_secret_access_key": "test-secret"} +_CMK_ARN = "arn:aws:kms:us-east-1:123456789012:key/11111111-2222-3333-4444-555555555555" + + +async def _create_secret_body_for_settings( + monkeypatch: pytest.MonkeyPatch, respx_mock: respx.MockRouter, settings: KeyManagementSettings +) -> dict[str, object]: + """Boot the manager from settings the way the proxy does and return the CreateSecret body it posts to AWS.""" + monkeypatch.setattr(litellm, "secret_manager_client", None) + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", None) + monkeypatch.setenv("AWS_REGION_NAME", "us-east-1") + AWSSecretsManagerV2.load_aws_secret_manager(use_aws_secret_manager=True, key_management_settings=settings) + manager = litellm.secret_manager_client + assert isinstance(manager, AWSSecretsManagerV2) + + route = respx_mock.post("https://secretsmanager.us-east-1.amazonaws.com/").respond( + json={"ARN": "arn", "Name": "litellm/test-key"} + ) + await manager.async_write_secret( + secret_name="litellm/test-key", + secret_value="sk-test-value", + optional_params=dict(_STATIC_CREDENTIALS), + ) + assert route.call_count == 1 + request = route.calls.last.request + assert request.headers["X-Amz-Target"] == "secretsmanager.CreateSecret" + return json.loads(request.content) + + +@pytest.mark.asyncio +async def test_create_secret_uses_customer_managed_kms_key_from_settings( + monkeypatch: pytest.MonkeyPatch, respx_mock: respx.MockRouter +) -> None: + body = await _create_secret_body_for_settings( + monkeypatch, + respx_mock, + KeyManagementSettings(store_virtual_keys=True, aws_region_name="us-east-1", kms_key_id=_CMK_ARN), + ) + assert body["KmsKeyId"] == _CMK_ARN + assert body["Name"] == "litellm/test-key" + assert body["SecretString"] == "sk-test-value" + + +@pytest.mark.asyncio +async def test_create_secret_omits_kms_key_id_when_not_configured( + monkeypatch: pytest.MonkeyPatch, respx_mock: respx.MockRouter +) -> None: + body = await _create_secret_body_for_settings( + monkeypatch, respx_mock, KeyManagementSettings(store_virtual_keys=True, aws_region_name="us-east-1") + ) + assert "KmsKeyId" not in body + assert body["Name"] == "litellm/test-key" @pytest.mark.asyncio From db3338b2068b6fa6448d9f82540208fae499fcad Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 09:55:30 -0700 Subject: [PATCH 080/157] feat(proxy): make the in-memory management cache capacity configurable (#40725) * feat(proxy): make the in-memory management cache capacity configurable Add general_settings.user_api_key_cache_max_size (positive int, default 200) to resize the in-memory tier of the shared user_api_key_cache at startup and on DB config reloads, expose it in the Admin UI general settings, and cover it with behavioral tests. Prior art: #34726 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): resize the in-memory tier from DualCache so any cache instance honours the cap Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(proxy): wrap the cache capacity field description to the 120 col limit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/dual_cache.py | 5 +- litellm/caching/in_memory_cache.py | 6 +- litellm/proxy/_types.py | 9 ++ litellm/proxy/proxy_server.py | 28 ++++ tests/test_litellm/proxy/test_proxy_server.py | 145 ++++++++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 5 + 6 files changed, 195 insertions(+), 3 deletions(-) diff --git a/litellm/caching/dual_cache.py b/litellm/caching/dual_cache.py index 26ad89a70cc..81e2af45686 100644 --- a/litellm/caching/dual_cache.py +++ b/litellm/caching/dual_cache.py @@ -22,7 +22,7 @@ from litellm._logging import print_verbose, verbose_logger from litellm.constants import DEFAULT_MAX_REDIS_BATCH_CACHE_SIZE from .base_cache import BaseCache -from .in_memory_cache import InMemoryCache +from .in_memory_cache import DEFAULT_MAX_SIZE_IN_MEMORY, InMemoryCache from .redis_cache import RedisCache, RedisCircuitBreakerOpenError, log_redis_failure if TYPE_CHECKING: @@ -83,6 +83,9 @@ class DualCache(BaseCache): if default_redis_ttl is not None: self.default_redis_ttl = default_redis_ttl + def update_in_memory_max_size(self, max_size: int | None) -> None: + self.in_memory_cache.max_size_in_memory = DEFAULT_MAX_SIZE_IN_MEMORY if max_size is None else max_size + def attach_redis_cache( self, redis_cache: RedisCache | None = None, diff --git a/litellm/caching/in_memory_cache.py b/litellm/caching/in_memory_cache.py index 38a9966f9f9..4058a3d72dd 100644 --- a/litellm/caching/in_memory_cache.py +++ b/litellm/caching/in_memory_cache.py @@ -24,11 +24,13 @@ from litellm.constants import MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB from .base_cache import BaseCache +DEFAULT_MAX_SIZE_IN_MEMORY: Final = 200 + class InMemoryCache(BaseCache): def __init__( self, - max_size_in_memory: int | None = 200, + max_size_in_memory: int | None = DEFAULT_MAX_SIZE_IN_MEMORY, default_ttl: int | None = 600, # default ttl is 10 minutes. At maximum litellm rate limiting logic requires objects to be in memory for 1 minute max_size_per_item: int | None = 1024, # 1MB = 1024KB @@ -37,7 +39,7 @@ class InMemoryCache(BaseCache): max_size_in_memory [int]: Maximum number of items in cache. done to prevent memory leaks. Use 200 items as a default """ self.max_size_in_memory = ( - max_size_in_memory if max_size_in_memory is not None else 200 + max_size_in_memory if max_size_in_memory is not None else DEFAULT_MAX_SIZE_IN_MEMORY ) # set an upper bound of 200 items in-memory self.default_ttl = default_ttl or 600 self.max_size_per_item = max_size_per_item or MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB # 1MB = 1024KB diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 2cbff128635..43af18cbef4 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2602,6 +2602,15 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): global_max_parallel_requests: int | None = Field( None, description="global max parallel requests to allow for a proxy instance." ) + user_api_key_cache_max_size: int | None = Field( + None, + gt=0, + description=( + "max number of entries (virtual keys, teams, users, end users, memberships, ...) each worker keeps in " + "its in-memory auth cache. Defaults to 200. Raise this if you have more active keys than that or auth " + "lookups keep hitting the DB" + ), + ) max_request_size_mb: int | None = Field( None, description="max request size in MB, if a request is larger than this size it will be rejected", diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 5d94b65ccba..687fbc0cb9b 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5808,6 +5808,16 @@ class ProxyConfig: default_redis_ttl=ttl, ) + ### USER API KEY CACHE MAX SIZE (in-memory tier shared by keys, teams, users, end users, ...) ### + if "user_api_key_cache_max_size" in general_settings: + user_api_key_cache.update_in_memory_max_size( + ConfigGeneralSettings.model_validate( + MappingProxyType( + {"user_api_key_cache_max_size": general_settings["user_api_key_cache_max_size"]} + ) + ).user_api_key_cache_max_size + ) + ### PKCE MULTI-INSTANCE PREREQUISITE CHECK ### # PKCE verifiers are stored in redis_usage_cache when available so they can # be read back by any instance (not just the one that started the auth flow). @@ -7058,6 +7068,23 @@ class ProxyConfig: "enable_openai_websocket_passthrough" ) + if "user_api_key_cache_max_size" not in self._yaml_general_settings_keys: + db_cache_max_size: Final = _general_settings.get("user_api_key_cache_max_size") + try: + cache_max_size: Final = ConfigGeneralSettings.model_validate( + MappingProxyType({"user_api_key_cache_max_size": db_cache_max_size}) + ).user_api_key_cache_max_size + except ValidationError: + verbose_proxy_logger.warning( + "Ignoring invalid general_settings.user_api_key_cache_max_size=%r from the DB", db_cache_max_size + ) + else: + if cache_max_size is None: + general_settings.pop("user_api_key_cache_max_size", None) + else: + general_settings["user_api_key_cache_max_size"] = cache_max_size + user_api_key_cache.update_in_memory_max_size(cache_max_size) + ## STORE MODEL IN DB ## if "store_model_in_db" in _general_settings: value = _general_settings["store_model_in_db"] @@ -16970,6 +16997,7 @@ _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES: Final[Mapping[str, str]] = MappingPro "cancel_on_disconnect": "Boolean", "disable_auto_add_proxy_admin_to_teams": "Boolean", "apply_user_budget_to_team_keys": "Boolean", + "user_api_key_cache_max_size": "Integer", } ) diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index 4f4a9e87d18..3b3031647dc 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -7313,6 +7313,88 @@ async def test_update_general_settings_apply_user_budget_to_team_keys_yaml_wins( assert ps.general_settings["apply_user_budget_to_team_keys"] is True +def _fill_user_api_key_cache(cache: DualCache, count: int) -> None: + for index in range(count): + cache.set_cache(key=f"key-{index}", value={"token": f"key-{index}"}, local_only=True) + + +@pytest.mark.asyncio +async def test_update_general_settings_user_api_key_cache_max_size_resizes_the_running_cache(monkeypatch): + """The Admin UI writes the capacity to the DB config, so the running cache has + to pick it up on reload; otherwise the knob only works after a restart.""" + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.proxy_server import ProxyConfig + + cache = UserApiKeyCache() + monkeypatch.setattr(proxy_server_module, "general_settings", {}) + monkeypatch.setattr(proxy_server_module, "user_api_key_cache", cache) + await ProxyConfig()._update_general_settings(db_general_settings={"user_api_key_cache_max_size": 300}) + + assert proxy_server_module.general_settings["user_api_key_cache_max_size"] == 300 + + _fill_user_api_key_cache(cache, 250) + assert cache.get_cache(key="key-0", local_only=True) == {"token": "key-0"} + + +@pytest.mark.asyncio +async def test_update_general_settings_clearing_user_api_key_cache_max_size_restores_the_default(monkeypatch): + """Blanking the field in the dashboard deletes the key, so the cache must fall + back to the default capacity rather than keep the last configured size.""" + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.proxy_server import ProxyConfig + + cache = UserApiKeyCache() + cache.update_in_memory_max_size(5000) + monkeypatch.setattr(proxy_server_module, "general_settings", {"user_api_key_cache_max_size": 5000}) + monkeypatch.setattr(proxy_server_module, "user_api_key_cache", cache) + await ProxyConfig()._update_general_settings(db_general_settings={"store_model_in_db": True}) + + assert "user_api_key_cache_max_size" not in proxy_server_module.general_settings + + _fill_user_api_key_cache(cache, 201) + assert cache.get_cache(key="key-0", local_only=True) is None + + +@pytest.mark.asyncio +@pytest.mark.parametrize("db_value", [0, -5, "lots"]) +async def test_update_general_settings_ignores_an_invalid_user_api_key_cache_max_size(db_value, monkeypatch): + """A non-positive capacity would make the eviction loop pop an empty heap on the + next write, so a bad DB value must leave the running cache untouched.""" + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.proxy_server import ProxyConfig + + cache = UserApiKeyCache() + cache.update_in_memory_max_size(300) + monkeypatch.setattr(proxy_server_module, "general_settings", {}) + monkeypatch.setattr(proxy_server_module, "user_api_key_cache", cache) + await ProxyConfig()._update_general_settings(db_general_settings={"user_api_key_cache_max_size": db_value}) + + assert "user_api_key_cache_max_size" not in proxy_server_module.general_settings + + _fill_user_api_key_cache(cache, 250) + assert cache.get_cache(key="key-0", local_only=True) == {"token": "key-0"} + + +@pytest.mark.asyncio +async def test_update_general_settings_user_api_key_cache_max_size_yaml_wins(monkeypatch): + """A DB value must not silently override an explicit YAML capacity on reload.""" + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.proxy_server import ProxyConfig + + proxy_config = ProxyConfig() + proxy_config._yaml_general_settings_keys = {"user_api_key_cache_max_size"} + cache = UserApiKeyCache() + cache.update_in_memory_max_size(300) + monkeypatch.setattr(proxy_server_module, "general_settings", {"user_api_key_cache_max_size": 300}) + monkeypatch.setattr(proxy_server_module, "user_api_key_cache", cache) + await proxy_config._update_general_settings(db_general_settings={"user_api_key_cache_max_size": 10}) + + assert proxy_server_module.general_settings["user_api_key_cache_max_size"] == 300 + + _fill_user_api_key_cache(cache, 250) + assert cache.get_cache(key="key-0", local_only=True) == {"token": "key-0"} + + @pytest.mark.asyncio @pytest.mark.parametrize( "db_value,expected", @@ -10344,6 +10426,27 @@ def test_get_config_list_includes_apply_user_budget_to_team_keys(monkeypatch): app.dependency_overrides.clear() +def test_get_config_list_includes_user_api_key_cache_max_size(monkeypatch): + """The Admin UI General Settings table renders whatever /config/list returns, + so the cache capacity has to be exposed there as an Integer to be editable.""" + mock_prisma = MagicMock() + mock_config_table = MagicMock() + mock_config_table.find_first = AsyncMock(return_value=None) + mock_prisma.db = types.SimpleNamespace(litellm_config=mock_config_table) + monkeypatch.setattr(proxy_server_module, "prisma_client", mock_prisma) + app.dependency_overrides[proxy_server_module.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN + ) + try: + client = TestClient(app) + resp = client.get("/config/list", params={"config_type": "general_settings"}) + assert resp.status_code == 200, resp.text + fields = {item["field_name"]: item for item in resp.json()} + assert fields["user_api_key_cache_max_size"]["field_type"] == "Integer" + finally: + app.dependency_overrides.clear() + + def test_get_config_list_includes_budget_exceeded_throttle_percentage(monkeypatch): """The throttle fraction is a litellm_settings scalar surfaced on the General Settings table as a Float field so it sits with the other global limits; it @@ -12946,6 +13049,48 @@ async def test_load_config_router_authorizes_fallback_targets_against_the_callin assert router.fallback_access_check is router_fallback_access_check +@pytest.mark.asyncio +async def test_load_config_user_api_key_cache_max_size_keeps_more_than_200_entries(tmp_path, monkeypatch): + """The auth cache used to be pinned at InMemoryCache's 200 entry default, so a + deployment with more keys than that evicted constantly and every request + fell through to the DB. The YAML knob has to raise the cap on the live cache.""" + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.proxy_server import ProxyConfig + + config_file = tmp_path / "config.yaml" + config_file.write_text(yaml.dump({"general_settings": {"user_api_key_cache_max_size": "1000"}})) + + cache = UserApiKeyCache() + monkeypatch.setattr(proxy_server_module, "user_api_key_cache", cache) + await ProxyConfig().load_config(router=None, config_file_path=str(config_file)) + + _fill_user_api_key_cache(cache, 999) + assert cache.get_cache(key="key-0", local_only=True) == {"token": "key-0"} + + +@pytest.mark.asyncio +@pytest.mark.parametrize("bad_value", [0, -1, "unbounded"]) +async def test_load_config_rejects_a_non_positive_user_api_key_cache_max_size(tmp_path, bad_value, monkeypatch): + """InMemoryCache treats 0 as 'cache nothing' and a negative cap makes eviction + pop an empty heap, so the proxy must refuse to boot with such a value instead + of silently disabling auth caching.""" + from pydantic import ValidationError + + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.proxy_server import ProxyConfig + + config_file = tmp_path / "config.yaml" + config_file.write_text(yaml.dump({"general_settings": {"user_api_key_cache_max_size": bad_value}})) + + cache = UserApiKeyCache() + monkeypatch.setattr(proxy_server_module, "user_api_key_cache", cache) + with pytest.raises(ValidationError): + await ProxyConfig().load_config(router=None, config_file_path=str(config_file)) + + _fill_user_api_key_cache(cache, 150) + assert cache.get_cache(key="key-0", local_only=True) == {"token": "key-0"} + + def test_docs_redoc_openapi_are_reachable_by_default(): """ LIT-6745: the interactive/machine-readable docs surfaces are on by diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 1b34c6f6a51..7fd4f8e413d 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -26007,6 +26007,11 @@ export interface components { * @description If True and LiteLLM_SpendLogs has been converted to a range-partitioned table (db_scripts/partition_spend_logs.sql), retention cleanup drops expired partitions instead of deleting rows, and pre-creates upcoming partitions. Default is False. */ use_spend_logs_partitioning?: boolean | null; + /** + * User Api Key Cache Max Size + * @description max number of entries (virtual keys, teams, users, end users, memberships, ...) each worker keeps in its in-memory auth cache. Defaults to 200. Raise this if you have more active keys than that or auth lookups keep hitting the DB + */ + user_api_key_cache_max_size?: number | null; /** User Header Mappings */ user_header_mappings?: components["schemas"]["UserHeaderMapping"][] | null; /** From f72b117b21e60a9d96a7deedeb0a00fe35f60ea6 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 10:00:35 -0700 Subject: [PATCH 081/157] fix(vertex_ai): return 400 for invalid reasoning_effort instead of 500 Both reasoning_effort mappers ended their if/elif chain in a bare ValueError. exception_type() has no branch for ValueError, so it fell through to the shared APIConnectionError fallback and the proxy answered a malformed client request with a retryable HTTP 500 carrying no hint of the accepted values. Raise UnsupportedParamsError (400) instead, listing the supported set, matching what the Anthropic and Bedrock transforms already do and what this same file already does at its five other param-validation sites. This also covers 'xhigh' and 'max', which are members of litellm's own REASONING_EFFORT literal but have no Gemini mapping, so callers bridging from OpenAI-shaped code were hitting the 500 without typing anything wrong. Fixes #40474 Claude-Session: https://claude.ai/code/session_01XT1qsbjLwnhiN5sQ2hNUxr --- .../vertex_and_google_ai_studio_gemini.py | 20 +++++- ...test_vertex_and_google_ai_studio_gemini.py | 67 +++++++++++++++++++ 2 files changed, 85 insertions(+), 2 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 69fe5678de9..d113b2b4f6b 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -23,6 +23,7 @@ from litellm.constants import ( DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE, DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO, ) +from litellm.exceptions import UnsupportedParamsError from litellm.litellm_core_utils.json_fragment_accumulator import JSONFragmentAccumulator from litellm.litellm_core_utils.prompt_templates.factory import ( _encode_tool_call_id_with_signature, @@ -108,6 +109,21 @@ else: StreamingChoices = Any +SUPPORTED_REASONING_EFFORTS: Final = ("minimal", "low", "medium", "high", "none", "disable") + + +def _unsupported_reasoning_effort(reasoning_effort: str) -> UnsupportedParamsError: + return UnsupportedParamsError( + message=( + f"Invalid `reasoning_effort`: {reasoning_effort!r}. " + f"Must be one of: {', '.join(repr(effort) for effort in SUPPORTED_REASONING_EFFORTS)}. " + "To drop this param, set `litellm.drop_params = True` or pass in `(.., drop_params=True)` " + "in the request - https://docs.litellm.ai/docs/completion/drop_params" + ), + status_code=400, + ) + + class VertexAIBaseConfig: def get_mapped_special_auth_params(self) -> dict: """ @@ -842,7 +858,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): "includeThoughts": False, } else: - raise ValueError(f"Invalid reasoning effort: {reasoning_effort}") + raise _unsupported_reasoning_effort(reasoning_effort) @staticmethod def _map_reasoning_effort_to_thinking_level( @@ -890,7 +906,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): else: return {"thinkingLevel": "low", "includeThoughts": False} else: - raise ValueError(f"Invalid reasoning effort: {reasoning_effort}") + raise _unsupported_reasoning_effort(reasoning_effort) @staticmethod def _is_thinking_budget_zero(thinking_budget: int | None) -> bool: diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index d2788408e09..101f6e6fa5d 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -5769,3 +5769,70 @@ def test_calculate_web_search_requests_counts_unique_queries(): assert VertexGeminiConfig._calculate_web_search_requests([]) is None assert VertexGeminiConfig._calculate_web_search_requests([{"webSearchQueries": ["", ""]}]) is None + + +@pytest.mark.parametrize("custom_llm_provider", ["gemini", "vertex_ai"]) +@pytest.mark.parametrize( + "model", + ["gemini-2.5-flash", "gemini-3-pro-preview"], + ids=["thinking_budget_mapper", "thinking_level_mapper"], +) +@pytest.mark.parametrize("reasoning_effort", ["banana", "xhigh"]) +def test_invalid_reasoning_effort_is_a_400_not_a_500(custom_llm_provider, model, reasoning_effort): + """Regression for #40474. + + Both reasoning_effort mappers used to end their if/elif chain in a bare `ValueError`, which + `exception_type()` has no branch for, so it fell through to `APIConnectionError` and the proxy + answered a malformed client request with a retryable HTTP 500. `xhigh` is covered alongside the + nonsense value because it is a member of litellm's own `REASONING_EFFORT` literal, so callers + bridging from OpenAI-shaped code reach it without typing anything wrong. + """ + from litellm.utils import get_optional_params + + with pytest.raises(litellm.BadRequestError) as exc_info: + get_optional_params( + model=model, + custom_llm_provider=custom_llm_provider, + reasoning_effort=reasoning_effort, + drop_params=True, + ) + + assert exc_info.value.status_code == 400 + message: Final = str(exc_info.value) + assert reasoning_effort in message + for supported in ("minimal", "low", "medium", "high", "none", "disable"): + assert supported in message + + +@pytest.mark.parametrize("custom_llm_provider", ["gemini", "vertex_ai"]) +def test_invalid_reasoning_effort_surfaces_as_400_through_completion(custom_llm_provider): + """The same request through `completion()` must not come back as a retryable 500. + + Needs no provider credentials: param mapping runs before any network call. + """ + with pytest.raises(litellm.BadRequestError) as exc_info: + completion( + model=f"{custom_llm_provider}/gemini-3-pro-preview", + messages=[{"role": "user", "content": "hi"}], + reasoning_effort="banana", + ) + + assert exc_info.value.status_code == 400 + assert not isinstance(exc_info.value, litellm.APIConnectionError) + + +@pytest.mark.parametrize("model", ["gemini-2.5-flash", "gemini-3-pro-preview"]) +def test_supported_reasoning_efforts_still_map(model): + """Guards the fix against over-rejecting: every advertised value must still produce a config.""" + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + SUPPORTED_REASONING_EFFORTS, + ) + + for effort in SUPPORTED_REASONING_EFFORTS: + result: Final = VertexGeminiConfig().map_openai_params( + non_default_params={"reasoning_effort": effort}, + optional_params={}, + model=model, + drop_params=False, + ) + assert "thinkingConfig" in result From 9316b4194a24edabaa1dd5ae0159af335ed2d316 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 17:10:38 +0000 Subject: [PATCH 082/157] perf(proxy): register liveness and core inference routes first (#40687) Starlette scans the route table in registration order, so a request pays one regex match per route registered ahead of its own. The proxy registers several hundred routes and left the liveness probe near position 280 and the lazy loaded /v1/messages at the very end. Move /health/liveliness, /health/liveness, /v1/chat/completions, /chat/completions and /v1/messages to the front of the route table after startup registration and again after a lazy router loads. Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/_lazy_features.py | 32 ++-- litellm/proxy/proxy_server.py | 2 + litellm/proxy/route_priority.py | 24 +++ .../test_litellm/proxy/test_route_priority.py | 171 ++++++++++++++++++ 4 files changed, 218 insertions(+), 11 deletions(-) create mode 100644 litellm/proxy/route_priority.py create mode 100644 tests/test_litellm/proxy/test_route_priority.py diff --git a/litellm/proxy/_lazy_features.py b/litellm/proxy/_lazy_features.py index 1d1b736c9fc..dd1180b30ad 100644 --- a/litellm/proxy/_lazy_features.py +++ b/litellm/proxy/_lazy_features.py @@ -18,6 +18,7 @@ from starlette.routing import BaseRoute, Match from starlette.types import Receive, Scope, Send from litellm._logging import verbose_proxy_logger +from litellm.proxy.route_priority import hot_routes_first if TYPE_CHECKING: from fastapi import APIRouter, FastAPI @@ -343,31 +344,40 @@ class LazyFeatureMiddleware: await self.app(scope, receive, send) -def _lazy_slots(app: "FastAPI") -> Mapping[str, int]: +def _lazy_slots(app: "FastAPI") -> Mapping[str, BaseRoute | None]: return app.state.lazy_slots if hasattr(app.state, "lazy_slots") else MappingProxyType({}) def reserve_lazy_slot(app: "FastAPI", name: str, features: tuple[LazyFeature, ...] = LAZY_FEATURES) -> None: - """Record the table position the feature's router used to be included at, so its - routes are spliced back in there once it loads and keep the same precedence.""" + """Record the route the feature's router used to be included after, so its routes + are spliced back in there once it loads and keep the same precedence. Anchoring on + the route rather than its index survives later reordering of the table.""" feat: Final = next(f for f in features if f.name == name) - app.state.lazy_slots = MappingProxyType({**_lazy_slots(app), feat.module_path: len(app.router.routes)}) + anchor: Final = app.router.routes[-1] if app.router.routes else None + app.state.lazy_slots = MappingProxyType({**_lazy_slots(app), feat.module_path: anchor}) + + +def _slot_index(routes: Sequence[BaseRoute], anchor: BaseRoute | None) -> int: + if anchor is None: + return 0 + return next((i + 1 for i, route in enumerate(routes) if route is anchor), len(routes)) def _eager_route_wins(app: "FastAPI", feat: LazyFeature, scope: Scope) -> bool: """Routes ahead of a feature's reserved slot beat its routes in Starlette's scan, so a request one of them fully matches never needs the feature loaded.""" - slot: Final = _lazy_slots(app).get(feat.module_path) - if slot is None: + slots: Final = _lazy_slots(app) + if feat.module_path not in slots: return False - return any(route.matches(scope)[0] is Match.FULL for route in app.router.routes[:slot]) + ahead: Final = app.router.routes[: _slot_index(app.router.routes, slots[feat.module_path])] + return any(route.matches(scope)[0] is Match.FULL for route in ahead) def _in_registry_order( routes: Sequence[BaseRoute], lazy_routes: Mapping[str, tuple[BaseRoute, ...]], features: tuple[LazyFeature, ...], - slots: Mapping[str, int], + slots: Mapping[str, BaseRoute | None], ) -> tuple[BaseRoute, ...]: """Lazy routers land in registry order, not first-request order, so overlapping paths (/openai/{endpoint:path} vs /openai/v1/realtime/calls) resolve the same @@ -380,7 +390,7 @@ def _in_registry_order( eager: Final = tuple(route for route in routes if id(route) not in lazy_ids) def slot_of(module_path: str) -> int: - return min(slots.get(module_path, len(eager)), len(eager)) + return _slot_index(eager, slots[module_path]) if module_path in slots else len(eager) return tuple( route @@ -416,8 +426,8 @@ async def _force_load(app: "FastAPI", feat: LazyFeature, features: tuple[LazyFea {**previous, feat.module_path: tuple(app.router.routes[before:])} ) app.state.lazy_routes = lazy_routes # rebind-ok: the app owns the record of which routes each feature added - app.router.routes[:] = _in_registry_order( # rebind-ok: the app owns its route table - app.router.routes, lazy_routes, features, _lazy_slots(app) + app.router.routes[:] = hot_routes_first( # rebind-ok: the app owns its route table + _in_registry_order(app.router.routes, lazy_routes, features, _lazy_slots(app)) ) app.state.lazy_loaded.add(feat.module_path) app.openapi_schema = None diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 687fbc0cb9b..4d8a821bc7a 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -655,6 +655,7 @@ from litellm.proxy.rag_endpoints.endpoints import router as rag_router from litellm.proxy.rerank_endpoints.endpoints import router as rerank_router from litellm.proxy.response_api_endpoints.endpoints import router as response_router from litellm.proxy.route_llm_request import route_request +from litellm.proxy.route_priority import hot_routes_first from litellm.proxy.search_endpoints.endpoints import router as search_router from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start @@ -18739,6 +18740,7 @@ app.include_router(ui_discovery_endpoints_router) app.include_router(google_router) attach_lazy_features(app) +app.router.routes = hot_routes_first(app.router.routes) app.add_middleware( RequestSizeLimitMiddleware, get_max_request_size_mb=lambda: general_settings.get("max_request_size_mb"), diff --git a/litellm/proxy/route_priority.py b/litellm/proxy/route_priority.py new file mode 100644 index 00000000000..77815b678a6 --- /dev/null +++ b/litellm/proxy/route_priority.py @@ -0,0 +1,24 @@ +"""Starlette matches routes in registration order, so the routes that take the most traffic go first.""" + +from collections.abc import Sequence +from typing import Final + +from starlette.routing import BaseRoute, Route + +HOT_ROUTE_PATHS: Final[frozenset[str]] = frozenset( + ( + "/health/liveliness", + "/health/liveness", + "/v1/chat/completions", + "/chat/completions", + "/v1/messages", + ) +) + + +def _is_hot(route: BaseRoute) -> bool: + return isinstance(route, Route) and route.path in HOT_ROUTE_PATHS + + +def hot_routes_first(routes: Sequence[BaseRoute]) -> list[BaseRoute]: # mutable-ok: assigned to Router.routes, a list + return sorted(routes, key=lambda route: not _is_hot(route)) diff --git a/tests/test_litellm/proxy/test_route_priority.py b/tests/test_litellm/proxy/test_route_priority.py new file mode 100644 index 00000000000..dfdc816f4b4 --- /dev/null +++ b/tests/test_litellm/proxy/test_route_priority.py @@ -0,0 +1,171 @@ +import sys +from types import ModuleType + +import httpx +import pytest +from fastapi import APIRouter, FastAPI +from fastapi.testclient import TestClient +from starlette.routing import Match + +from litellm.proxy.route_priority import HOT_ROUTE_PATHS, hot_routes_first + +FILLER_COUNT = 300 + + +def _routes_scanned_before_dispatch(app: FastAPI, method: str, path: str) -> int: + """Number of route.matches() calls Starlette's Router.app makes before it finds a full match.""" + scope = {"type": "http", "method": method, "path": path, "root_path": "", "headers": [], "query_string": b""} + for i, route in enumerate(app.router.routes): + match, _ = route.matches(dict(scope)) + if match == Match.FULL: + return i + 1 + raise AssertionError(f"{method} {path} has no route") + + +def _hot_router() -> APIRouter: + router = APIRouter() + + @router.get("/health/liveliness") + @router.get("/health/liveness") + async def liveliness(): + return "I'm alive!" + + @router.post("/v1/chat/completions") + @router.post("/chat/completions") + async def chat(): + return {"object": "chat.completion"} + + return router + + +def _app_with_filler_then_hot_routes() -> FastAPI: + app = FastAPI() + for i in range(FILLER_COUNT): + + @app.get(f"/filler/{i}") + async def filler(i: int = i): + return {"filler": i} + + app.include_router(_hot_router()) + return app + + +def test_hot_routes_first_puts_hot_routes_ahead_of_everything_else(): + app = _app_with_filler_then_hot_routes() + assert _routes_scanned_before_dispatch(app, "GET", "/health/liveliness") > FILLER_COUNT + + app.router.routes = hot_routes_first(app.router.routes) + + hot_count = sum(1 for r in app.router.routes if getattr(r, "path", None) in HOT_ROUTE_PATHS) + assert _routes_scanned_before_dispatch(app, "GET", "/health/liveliness") <= hot_count + assert _routes_scanned_before_dispatch(app, "GET", "/health/liveness") <= hot_count + assert _routes_scanned_before_dispatch(app, "POST", "/v1/chat/completions") <= hot_count + assert _routes_scanned_before_dispatch(app, "POST", "/chat/completions") <= hot_count + + +def test_hot_routes_first_keeps_the_other_routes_in_order_and_dispatching(): + app = _app_with_filler_then_hot_routes() + before = [r.path for r in app.router.routes if getattr(r, "path", "").startswith("/filler/")] + + app.router.routes = hot_routes_first(app.router.routes) + + after = [r.path for r in app.router.routes if getattr(r, "path", "").startswith("/filler/")] + assert after == before + client = TestClient(app) + assert client.get("/health/liveliness").json() == "I'm alive!" + assert client.get("/filler/7").json() == {"filler": 7} + assert client.post("/v1/chat/completions").json() == {"object": "chat.completion"} + assert client.get("/v1/chat/completions").status_code == 405 + assert client.get("/does/not/exist").status_code == 404 + + +def test_hot_routes_first_is_idempotent(): + app = _app_with_filler_then_hot_routes() + once = hot_routes_first(app.router.routes) + assert hot_routes_first(once) == once + + +@pytest.mark.asyncio +async def test_lazy_loaded_hot_route_moves_to_the_front(monkeypatch): + from litellm.proxy._lazy_features import LazyFeature, LazyFeatureMiddleware + + messages_router = APIRouter() + + @messages_router.post("/v1/messages") + async def messages(): + return {"type": "message"} + + fake_module = ModuleType("fake_anthropic_endpoints") + fake_module.router = messages_router + monkeypatch.setitem(sys.modules, fake_module.__name__, fake_module) + + target_app = _app_with_filler_then_hot_routes() + target_app.router.routes = hot_routes_first(target_app.router.routes) + + async def downstream(scope, receive, send): + await send({"type": "http.response.start", "status": 200, "headers": []}) + await send({"type": "http.response.body", "body": b""}) + + feat = LazyFeature(name="anthropic", module_path=fake_module.__name__, path_prefixes=("/v1/messages",)) + mw = LazyFeatureMiddleware(downstream, fastapi_app=target_app, features=(feat,)) + + async def receive(): + return {"type": "http.request", "body": b"", "more_body": False} + + async def send(message): + pass + + await mw({"type": "http", "path": "/v1/messages", "method": "POST", "headers": []}, receive, send) + + hot_count = sum(1 for r in target_app.router.routes if getattr(r, "path", None) in HOT_ROUTE_PATHS) + assert _routes_scanned_before_dispatch(target_app, "POST", "/v1/messages") <= hot_count + assert TestClient(target_app).post("/v1/messages").json() == {"type": "message"} + + +@pytest.mark.asyncio +async def test_hot_routes_first_keeps_reserved_lazy_slot_ahead_of_later_eager_routes(): + """Liveness is registered after the provider passthrough slot, so pulling it to the + front must not shift where the lazily loaded catch-all is spliced back in.""" + from litellm.proxy._lazy_features import LazyFeature, LazyFeatureMiddleware, reserve_lazy_slot + + def register(app, module): + router = APIRouter() + router.add_api_route("/mistral/{endpoint:path}", lambda: {"handler": "passthrough"}, methods=["POST"]) + app.include_router(router) + + passthrough = LazyFeature( + name="llm_passthrough", module_path="json", path_prefixes=("/mistral/",), register_fn=register + ) + target_app = FastAPI() + target_app.add_api_route("/mistral/v1/files", lambda: {"handler": "files"}, methods=["POST"]) + target_app.add_api_route("/mistral/v1/batches", lambda: {"handler": "batches"}, methods=["POST"]) + reserve_lazy_slot(target_app, "llm_passthrough", features=(passthrough,)) + target_app.include_router(_hot_router()) + target_app.add_api_route("/{mcp_server_name}/mcp", lambda: {"handler": "mcp"}, methods=["POST"]) + target_app.router.routes = hot_routes_first(target_app.router.routes) + target_app.add_middleware(LazyFeatureMiddleware, fastapi_app=target_app, features=(passthrough,)) + + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=target_app), base_url="http://t") as client: + batches_first = (await client.post("/mistral/v1/batches")).json()["handler"] + loaded_after_batches = frozenset(target_app.state.lazy_loaded) + handlers = [ + (await client.post(path)).json()["handler"] + for path in ("/mistral/mcp", "/mistral/v1/files", "/mistral/v1/batches") + ] + + assert (batches_first, loaded_after_batches) == ("batches", frozenset()) + assert handlers == ["passthrough", "files", "batches"] + hot_count = sum(1 for r in target_app.router.routes if getattr(r, "path", None) in HOT_ROUTE_PATHS) + assert _routes_scanned_before_dispatch(target_app, "GET", "/health/liveliness") <= hot_count + + +def test_proxy_app_dispatches_liveness_and_chat_completions_before_the_rest(): + from litellm.proxy.proxy_server import app + + hot_count = sum(1 for r in app.router.routes if getattr(r, "path", None) in HOT_ROUTE_PATHS) + assert hot_count >= 4 + assert len(app.router.routes) > 100 + assert _routes_scanned_before_dispatch(app, "GET", "/health/liveliness") <= hot_count + assert _routes_scanned_before_dispatch(app, "GET", "/health/liveness") <= hot_count + assert _routes_scanned_before_dispatch(app, "POST", "/v1/chat/completions") <= hot_count + assert _routes_scanned_before_dispatch(app, "POST", "/chat/completions") <= hot_count From dfab4794ecd44d1b53b8d8fec85fb2e4375b2e73 Mon Sep 17 00:00:00 2001 From: Anmol Jaiswal <68013660+anmolg1997@users.noreply.github.com> Date: Fri, 11 Sep 2026 22:52:32 +0530 Subject: [PATCH 083/157] docs(router): name both affinity TTL knobs in the _claim_pin docstring (#40663) The docstring cited session_affinity_ttl_seconds as the keepalive bound, but the Router-level knob feeding ttl_seconds is deployment_affinity_ttl_seconds; session_affinity_ttl_seconds is the separate per-request PreRoutingHookResponse override. Anyone grepping the docstring's name to shrink the Router default finds only the override. Name both, scoped correctly. --- .../router_utils/pre_call_checks/deployment_affinity_check.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/router_utils/pre_call_checks/deployment_affinity_check.py b/litellm/router_utils/pre_call_checks/deployment_affinity_check.py index 6f3ea8eb78a..c7eb46046ef 100644 --- a/litellm/router_utils/pre_call_checks/deployment_affinity_check.py +++ b/litellm/router_utils/pre_call_checks/deployment_affinity_check.py @@ -349,7 +349,9 @@ class DeploymentAffinityCheck(CustomLogger): first write instead of the last. Re-claiming with the stored value refreshes its TTL, the same keepalive the complexity router's model pin documents: an active session must not lose its pin mid-conversation just because it outlives the - original write, so `session_affinity_ttl_seconds` bounds idle time, not total + original write, so the affinity TTL (the Router's + `deployment_affinity_ttl_seconds`, or a pre-routing hook's per-request + `session_affinity_ttl_seconds` override) bounds idle time, not total session length. On Redis one Lua script does the get-or-set-or-refresh atomically (same registration seam the rate limiters use) and the in-memory tier is synchronized to the winner; without Redis, and whenever Redis is From 5ddd83943259cf8e9ae5647474777d7e936873a3 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 11 Sep 2026 10:25:43 -0700 Subject: [PATCH 084/157] test(e2e): wait for serving propagation in UI journeys --- tests/e2e/ui/constants.ts | 2 + .../guardrails/presidioUserStory.spec.ts | 114 +++++++++++++----- .../modelsPage/modelHealthStatus.spec.ts | 20 ++- .../ui/tests/proxy-admin/keyBlocking.spec.ts | 7 +- 4 files changed, 107 insertions(+), 36 deletions(-) diff --git a/tests/e2e/ui/constants.ts b/tests/e2e/ui/constants.ts index 71774c95d24..158dea53b71 100644 --- a/tests/e2e/ui/constants.ts +++ b/tests/e2e/ui/constants.ts @@ -17,6 +17,8 @@ export const ARTIFACT_DIR = process.env.E2E_UI_ARTIFACT_DIR || "."; export const MOCK_PRESIDIO_URL = (process.env.E2E_MOCK_PRESIDIO_URL || "http://127.0.0.1:8091").replace(/\/+$/, ""); +export const PROPAGATION_TIMEOUT_MS = 90_000; + const storagePath = (name: string): string => path.join(ARTIFACT_DIR, name); // Storage state paths for each role diff --git a/tests/e2e/ui/tests/guardrails/presidioUserStory.spec.ts b/tests/e2e/ui/tests/guardrails/presidioUserStory.spec.ts index d4ed5308342..b4dc7c4e8da 100644 --- a/tests/e2e/ui/tests/guardrails/presidioUserStory.spec.ts +++ b/tests/e2e/ui/tests/guardrails/presidioUserStory.spec.ts @@ -1,8 +1,8 @@ -import { test, expect, type Page as PlaywrightPage } from "@playwright/test"; -import { ADMIN_STORAGE_PATH, MOCK_PRESIDIO_URL } from "../../constants"; +import { test as base, expect, type Page as PlaywrightPage } from "@playwright/test"; +import { ADMIN_STORAGE_PATH, MOCK_PRESIDIO_URL, PROPAGATION_TIMEOUT_MS } from "../../constants"; import { navigateToPage, dismissFeedbackPopup } from "../../helpers/navigation"; import { Page } from "../../fixtures/pages"; -import { CHAT_MODEL_A, masterKey, rootPath, waitForSpendLogByPrompt } from "../../helpers/traffic"; +import { CHAT_MODEL_A, masterKey, rootPath, uniqueSuffix, waitForSpendLog } from "../../helpers/traffic"; import { openPlayground, selectModel, sendButton, onlyVisible } from "../../helpers/playground"; const RAW_EMAIL = "jane.doe@example.com"; @@ -62,15 +62,63 @@ async function deleteGuardrail(page: PlaywrightPage, guardrailName: string): Pro await expect(page.getByText(`Guardrail "${guardrailName}" deleted successfully`)).toBeVisible({ timeout: 10_000 }); } +const test = base.extend<{ guardrailName: string }>({ + guardrailName: async ({ page }, use) => { + const name = `e2e-presidio-story-${uniqueSuffix()}`; + await createPresidioGuardrail(page, name); + try { + await use(name); + } finally { + await deleteGuardrail(page, name); + } + }, +}); + test.describe("Presidio PII guardrail, end to end from the dashboard", () => { + test.describe.configure({ timeout: 5 * 60_000 }); test.use({ storageState: ADMIN_STORAGE_PATH }); - test("masks PII sent from the Playground and shows the run in Logs", async ({ page, request }) => { - const guardrailName = `e2e-presidio-story-${Date.now()}`; + test("masks PII sent from the Playground and shows the run in Logs", async ({ page, request, guardrailName }) => { const marker = `case-ref-${Math.random().toString(36).slice(2, 10)}`; const prompt = `${marker}. Email me at ${RAW_EMAIL} or call ${RAW_PHONE}.`; - await createPresidioGuardrail(page, guardrailName); + await expect + .poll( + async () => { + const completion = await request.post(`${rootPath()}/v1/chat/completions`, { + headers: { Authorization: `Bearer ${masterKey()}` }, + data: { + model: CHAT_MODEL_A, + messages: [ + { role: "user", content: `readiness-${uniqueSuffix()}. Email ${RAW_EMAIL}; phone ${RAW_PHONE}.` }, + ], + guardrails: [guardrailName], + }, + }); + expect(completion.ok(), `guardrail readiness request failed: ${await completion.text()}`).toBe(true); + const { id }: { id: string } = await completion.json(); + expect(id).toBeTruthy(); + await waitForSpendLog(request, id); + const stored = await request.get(`${rootPath()}/spend/logs`, { + headers: { Authorization: `Bearer ${masterKey()}` }, + params: { request_id: id }, + }); + expect(stored.ok(), `guardrail readiness log read failed: ${stored.status()}`).toBe(true); + const body = await stored.text(); + return { + rawEmail: body.includes(RAW_EMAIL), + rawPhone: body.includes(RAW_PHONE), + maskedEmail: body.includes(""), + maskedPhone: body.includes(""), + }; + }, + { + message: `${guardrailName} never masked email and phone data in a completed request`, + timeout: PROPAGATION_TIMEOUT_MS, + intervals: [2_000], + }, + ) + .toEqual({ rawEmail: false, rawPhone: false, maskedEmail: true, maskedPhone: true }); await openPlayground(page); await selectModel(page, CHAT_MODEL_A); @@ -85,29 +133,31 @@ test.describe("Presidio PII guardrail, end to end from the dashboard", () => { const input = onlyVisible(page.getByPlaceholder("Type your message", { exact: false })); await expect(input).toBeVisible({ timeout: 15_000 }); - await expect - .poll( - async () => { - await input.fill(prompt); - await sendButton(page).click(); - const res = await request.get(`${rootPath()}/spend/logs`, { - headers: { Authorization: `Bearer ${masterKey()}` }, - }); - if (!res.ok()) return false; - const rows: { metadata?: { applied_guardrails?: string[] } }[] = await res.json(); - return (Array.isArray(rows) ? rows : []).some((row) => - (row.metadata?.applied_guardrails ?? []).includes(guardrailName), - ); - }, - { - message: `the playground never produced a request that ran ${guardrailName}`, - timeout: 90_000, - intervals: [5_000], - }, - ) - .toBe(true); - - const requestId = await waitForSpendLogByPrompt(request, marker); + await input.fill(prompt); + const responsePromise = page.waitForResponse( + (response) => + response.request().method() === "POST" && + new URL(response.url()).pathname.endsWith("/chat/completions") && + (response.request().postData()?.includes(marker) ?? false), + ); + await sendButton(page).click(); + const response = await responsePromise; + expect(response.ok(), `Playground completion failed: ${response.status()}`).toBe(true); + const responseBody = await response.text(); + const chunks: { id?: string; error?: unknown }[] = response.headers()["content-type"]?.includes("text/event-stream") + ? responseBody + .split(/\r?\n/) + .filter((line) => line.startsWith("data: ") && line.trim() !== "data: [DONE]") + .map((line) => JSON.parse(line.slice(6))) + : [JSON.parse(responseBody)]; + expect( + chunks.some((chunk) => chunk.error), + "the Playground stream returned an error", + ).toBe(false); + const requestIds = [...new Set(chunks.map((chunk) => chunk.id).filter((id): id is string => !!id))]; + expect(requestIds, "the Playground response identifies exactly one completion").toHaveLength(1); + const [requestId] = requestIds; + await waitForSpendLog(request, requestId); const stored = await request.get(`${rootPath()}/spend/logs?request_id=${requestId}`, { headers: { Authorization: `Bearer ${masterKey()}` }, @@ -138,7 +188,9 @@ test.describe("Presidio PII guardrail, end to end from the dashboard", () => { const drawer = page.getByRole("dialog").first(); await expect(onlyVisible(drawer.getByText("Guardrails & Policy Compliance"))).toBeVisible({ timeout: 20_000 }); - await expect(onlyVisible(drawer.getByText(`Pre-call guardrail: ${guardrailName}`))).toBeVisible({ timeout: 20_000 }); + await expect(onlyVisible(drawer.getByText(`Pre-call guardrail: ${guardrailName}`))).toBeVisible({ + timeout: 20_000, + }); const maskedPrompt = drawer.getByText(`${marker}. Email me at or call .`); await expect(onlyVisible(maskedPrompt)).toBeVisible({ timeout: 20_000 }); @@ -151,7 +203,5 @@ test.describe("Presidio PII guardrail, end to end from the dashboard", () => { await expect(drawer.getByText(RAW_EMAIL)).toHaveCount(0); await expect(drawer.getByText(RAW_PHONE)).toHaveCount(0); - - await deleteGuardrail(page, guardrailName); }); }); diff --git a/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts b/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts index 247cce1b85d..e48128464d2 100644 --- a/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts +++ b/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts @@ -4,7 +4,7 @@ import { type Locator, type Page as PlaywrightPage, } from "@playwright/test"; -import { ADMIN_STORAGE_PATH } from "../../constants"; +import { ADMIN_STORAGE_PATH, PROPAGATION_TIMEOUT_MS } from "../../constants"; import { Page } from "../../fixtures/pages"; import { navigateToPage } from "../../helpers/navigation"; import { readBack } from "../../helpers/roundTrip"; @@ -145,6 +145,24 @@ async function withDeployment( timeout: 60_000, }) .toBe(true); + await expect + .poll( + async () => { + const response = await page.request.get("/health", { + headers: { Authorization: `Bearer ${masterKey()}` }, + params: { model_id: id }, + }); + if (![200, 503].includes(response.status())) return 0; + const body: { healthy_count?: number; unhealthy_count?: number } = await response.json(); + return (body.healthy_count ?? 0) + (body.unhealthy_count ?? 0); + }, + { + message: `deployment ${name} never became available to /health`, + timeout: PROPAGATION_TIMEOUT_MS, + intervals: [2_000], + }, + ) + .toBe(1); await use(name); } finally { await deleteDeployment(page, id); diff --git a/tests/e2e/ui/tests/proxy-admin/keyBlocking.spec.ts b/tests/e2e/ui/tests/proxy-admin/keyBlocking.spec.ts index 99a8065a797..788f4ca1284 100644 --- a/tests/e2e/ui/tests/proxy-admin/keyBlocking.spec.ts +++ b/tests/e2e/ui/tests/proxy-admin/keyBlocking.spec.ts @@ -1,5 +1,5 @@ import { test as base, expect } from "@playwright/test"; -import { ADMIN_STORAGE_PATH } from "../../constants"; +import { ADMIN_STORAGE_PATH, PROPAGATION_TIMEOUT_MS } from "../../constants"; import { Page } from "../../fixtures/pages"; import { dismissFeedbackPopup, navigateToPage, openKeyDetail } from "../../helpers/navigation"; import { @@ -35,6 +35,7 @@ test.describe("Proxy Admin - Key blocking", () => { test.use({ storageState: ADMIN_STORAGE_PATH }); test("blocking a key stops it serving and unblocking restores it", async ({ page, scopedKey }) => { + test.setTimeout(5 * 60_000); const { alias, token, apiKey } = scopedKey; await sendChatCompletion(page.request, { @@ -70,7 +71,7 @@ test.describe("Proxy Admin - Key blocking", () => { }), { message: "a blocked key was still served by /v1/chat/completions", - timeout: 30_000, + timeout: PROPAGATION_TIMEOUT_MS, }, ) .toMatchObject({ status: 401, body: expect.stringContaining("blocked") }); @@ -104,7 +105,7 @@ test.describe("Proxy Admin - Key blocking", () => { }), { message: "an unblocked key is still refused by /v1/chat/completions", - timeout: 30_000, + timeout: PROPAGATION_TIMEOUT_MS, }, ) .toMatchObject({ status: 200, body: expect.stringContaining(MOCK_RESPONSE_TEXT) }); From 057d333d108d355534869a11297b53d949ee773a Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 11 Sep 2026 10:29:10 -0700 Subject: [PATCH 085/157] test(e2e): observe model propagation without pre-running health checks --- .../ui/tests/modelsPage/modelHealthStatus.spec.ts | 13 ++++++------- 1 file changed, 6 insertions(+), 7 deletions(-) diff --git a/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts b/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts index e48128464d2..8260f7a8889 100644 --- a/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts +++ b/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts @@ -148,21 +148,20 @@ async function withDeployment( await expect .poll( async () => { - const response = await page.request.get("/health", { + const response = await page.request.get("/v1/models", { headers: { Authorization: `Bearer ${masterKey()}` }, - params: { model_id: id }, }); - if (![200, 503].includes(response.status())) return 0; - const body: { healthy_count?: number; unhealthy_count?: number } = await response.json(); - return (body.healthy_count ?? 0) + (body.unhealthy_count ?? 0); + expect(response.ok(), `/v1/models failed: ${response.status()}`).toBe(true); + const body: { data: { id: string }[] } = await response.json(); + return body.data.some((model) => model.id === name); }, { - message: `deployment ${name} never became available to /health`, + message: `deployment ${name} never appeared on the serving path`, timeout: PROPAGATION_TIMEOUT_MS, intervals: [2_000], }, ) - .toBe(1); + .toBe(true); await use(name); } finally { await deleteDeployment(page, id); From b01d12154d79019fe0a902427802436e69b9a519 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 11 Sep 2026 10:39:49 -0700 Subject: [PATCH 086/157] test(e2e): budget model health setup and propagation waits --- tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts b/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts index 8260f7a8889..7bd1caa6756 100644 --- a/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts +++ b/tests/e2e/ui/tests/modelsPage/modelHealthStatus.spec.ts @@ -178,6 +178,7 @@ const test = base.extend<{ reachableName: string; unreachableName: string }>({ }); test.describe("Model health status", () => { + test.describe.configure({ timeout: 8 * 60_000 }); test.use({ storageState: ADMIN_STORAGE_PATH }); test("Run Health Check reports a reachable deployment healthy and an unreachable one unhealthy", async ({ From aef8888c0b56e45e6afe9f0179b8d51c43cd2714 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 11:02:38 -0700 Subject: [PATCH 087/157] ci: remove MCP tests from Python 3.10 import smoke --- .github/workflows/test-code-quality.yml | 6 ------ 1 file changed, 6 deletions(-) diff --git a/.github/workflows/test-code-quality.yml b/.github/workflows/test-code-quality.yml index 4abcc47df5e..e6d2264fbf0 100644 --- a/.github/workflows/test-code-quality.yml +++ b/.github/workflows/test-code-quality.yml @@ -187,9 +187,3 @@ jobs: - name: Check litellm CLI run: uv run --no-sync litellm --version - - - name: Verify MCP timeout fallback and auth retries on Python 3.10 - run: >- - uv run --no-sync pytest --noconftest - tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_debug.py - -k error_capture -q From 5f17261534caa773206edc9fa1d85a91fc4435fd Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 11:06:06 -0700 Subject: [PATCH 088/157] test(load): pause Redis outright and budget the chaos phase against the baseline CLIENT PAUSE ALL for the length of the chaos phase instead of CLIENT PAUSE WRITE, so every Redis touchpoint on the request path times out rather than just the writes. The pause is sized to the phase because it freezes the control connection too; teardown's CLIENT UNPAUSE is a safety net for a phase that overran Latency, RSS and CPU are now budgeted as chaos-over-baseline ratios (p50/p90/p99 for latency and RSS, CPU seconds per request once) through a small phase_budget module, replacing the machine-shaped absolutes. The Redis timeout rate is reported but no longer asserted The final /metrics scrape waits for litellm_deployment_failure_responses_total to stop moving, since that counter is bumped from the async logging queue and lagged the load generator by thousands of increments. The model group carries a unique marker so a deployment left behind by an aborted run cannot absorb this run's retries Co-Authored-By: Claude Code --- tests/e2e/CLAUDE.md | 2 +- tests/e2e/conftest.py | 2 +- tests/e2e/load/phase_budget.py | 52 +++++ tests/e2e/load/proxy_usage.py | 8 + tests/e2e/load/test_phase_budget.py | 64 +++++++ tests/e2e/load/test_proxy_usage.py | 12 ++ tests/e2e/load/test_redis_chaos_e2e.py | 254 +++++++++++++++++++------ tests/e2e/pytest.ini | 2 +- 8 files changed, 337 insertions(+), 59 deletions(-) create mode 100644 tests/e2e/load/phase_budget.py create mode 100644 tests/e2e/load/test_phase_budget.py diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index dc197996ac2..bfab4553709 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments with `CLIENT PAUSE WRITE` on the proxy's Redis mid-run, asserting zero failed requests and reporting latency, RSS, and CPU as p50/p90/p99 per phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests and budgeting p50/p90/p99 latency, RSS, and CPU-per-request as ratios against the same run's healthy phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index 99a3f4967c3..700dc822d59 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -60,7 +60,7 @@ def pytest_configure(config: pytest.Config) -> None: ) config.addinivalue_line( "markers", - "redis_chaos: load test that pauses the proxy's Redis writes mid-run; needs a proxy booted from " + "redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from " "gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set", ) diff --git a/tests/e2e/load/phase_budget.py b/tests/e2e/load/phase_budget.py new file mode 100644 index 00000000000..7d354a678bc --- /dev/null +++ b/tests/e2e/load/phase_budget.py @@ -0,0 +1,52 @@ +"""Comparing one load phase against another, for tests that degrade a dependency mid-run. + +A chaos phase's absolute numbers say very little on their own: RSS scales with worker count, +latency with core count, so a ceiling calibrated on one machine is meaningless on the next. +What travels is the ratio against a healthy phase measured on the same machine in the same run. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Final + + +@dataclass(frozen=True, slots=True) +class Budget: + """One metric's healthy value, its degraded value, and how much growth is allowed.""" + + name: str + baseline: float + degraded: float + ratio_ceiling: float + unit: str + decimals: int = 1 + + @property + def ratio(self) -> float | None: + """How many times the baseline the degraded value is, or None if there is no baseline.""" + return self.degraded / self.baseline if self.baseline > 0 else None + + def _rendered(self, value: float) -> str: + return f"{value:.{self.decimals}f}{self.unit}" + + def violation(self) -> str | None: + """Why this metric fails its budget, or None if it passes.""" + ratio: Final = self.ratio + if ratio is None: + return ( + f"{self.name} measured {self._rendered(self.baseline)} in the healthy phase, so there is nothing " + f"to compare the degraded phase against; the measurement did not happen" + ) + if ratio > self.ratio_ceiling: + return ( + f"{self.name} went from {self._rendered(self.baseline)} healthy to " + f"{self._rendered(self.degraded)} degraded, {ratio:.1f}x the baseline and past the " + f"{self.ratio_ceiling:.1f}x allowed" + ) + return None + + +def violations(budgets: tuple[Budget, ...]) -> tuple[str, ...]: + """Every budget the run blew, so one failure reports all of them instead of the first.""" + return tuple(violation for budget in budgets if (violation := budget.violation()) is not None) diff --git a/tests/e2e/load/proxy_usage.py b/tests/e2e/load/proxy_usage.py index b48129f0099..83463c078b8 100644 --- a/tests/e2e/load/proxy_usage.py +++ b/tests/e2e/load/proxy_usage.py @@ -50,6 +50,14 @@ class UsageWindow: return 0.0 return self.samples[-1].cpu_seconds - self.samples[0].cpu_seconds + def cpu_seconds_per_request(self, requests: int) -> float: + """CPU seconds the tree spent per request served. + + The portable cost figure: cores-busy saturates at the worker count under enough load, + so it reads the same whether a request costs 10 ms of CPU or 40 ms. This does not. + """ + return self.cpu_seconds_consumed() / requests if requests else 0.0 + def cpu_utilization_percentiles(self) -> tuple[float, float, float]: """Per-interval CPU utilization (cores busy) at p50, p90 and p99. diff --git a/tests/e2e/load/test_phase_budget.py b/tests/e2e/load/test_phase_budget.py new file mode 100644 index 00000000000..2355fb7cf4c --- /dev/null +++ b/tests/e2e/load/test_phase_budget.py @@ -0,0 +1,64 @@ +from __future__ import annotations + +from typing import Final + +from phase_budget import Budget, violations + + +def _budget(*, baseline: float, degraded: float, ceiling: float = 2.0) -> Budget: + return Budget(name="p99 RSS", baseline=baseline, degraded=degraded, ratio_ceiling=ceiling, unit=" MB", decimals=0) + + +class TestBudget: + def test_growth_within_the_ceiling_is_not_a_violation(self) -> None: + assert _budget(baseline=100, degraded=199).violation() is None + + def test_growth_exactly_at_the_ceiling_is_allowed(self) -> None: + assert _budget(baseline=100, degraded=200).violation() is None + + def test_growth_past_the_ceiling_reports_both_values_and_the_ratio(self) -> None: + violation: Final = _budget(baseline=100, degraded=250).violation() + + assert violation is not None + assert "100 MB" in violation + assert "250 MB" in violation + assert "2.5x" in violation + assert "2.0x allowed" in violation + + def test_shrinking_is_never_a_violation(self) -> None: + assert _budget(baseline=100, degraded=10).violation() is None + + def test_a_missing_baseline_is_a_violation_rather_than_a_silent_pass(self) -> None: + # The trap this guards: 0 as a baseline would make every ratio a division by zero, and + # treating it as "no growth" would pass a run that measured nothing at all. + violation: Final = _budget(baseline=0, degraded=4000).violation() + + assert violation is not None + assert "nothing to compare" in violation + + def test_the_unit_and_decimals_carry_into_the_message(self) -> None: + violation: Final = Budget( + name="p99 latency", baseline=0.16, degraded=9.5, ratio_ceiling=8.0, unit="s", decimals=3 + ).violation() + + assert violation is not None + assert "0.160s" in violation + assert "9.500s" in violation + + +class TestViolations: + def test_every_blown_budget_is_reported_not_just_the_first(self) -> None: + blown: Final = violations( + ( + _budget(baseline=100, degraded=500), + _budget(baseline=100, degraded=120), + Budget(name="CPU per request", baseline=10, degraded=90, ratio_ceiling=6.0, unit=" ms"), + ) + ) + + assert len(blown) == 2 + assert blown[0].startswith("p99 RSS") + assert blown[1].startswith("CPU per request") + + def test_a_run_inside_every_budget_reports_nothing(self) -> None: + assert violations((_budget(baseline=100, degraded=150),)) == () diff --git a/tests/e2e/load/test_proxy_usage.py b/tests/e2e/load/test_proxy_usage.py index 8d9f848825f..915c564de50 100644 --- a/tests/e2e/load/test_proxy_usage.py +++ b/tests/e2e/load/test_proxy_usage.py @@ -50,6 +50,18 @@ class TestCpuUtilization: assert window.cpu_utilization_percentiles() == (0.0, 0.0, 0.0) assert window.cpu_seconds_consumed() == 0.0 + def test_cost_per_request_separates_runs_that_cores_busy_reports_identically(self) -> None: + # Both windows pin 4 cores for 10 seconds, so utilization cannot tell them apart. The + # second one served a tenth of the traffic for the same CPU, which is the regression shape. + window: Final = _window(*((float(i), _MB, 4.0 * i) for i in range(11))) + + assert window.cpu_utilization_percentiles()[0] == 4.0 + assert window.cpu_seconds_per_request(4000) == 0.01 + assert window.cpu_seconds_per_request(400) == 0.1 + + def test_no_requests_reports_zero_cost_rather_than_dividing_by_zero(self) -> None: + assert _window((0.0, _MB, 0.0), (1.0, _MB, 1.0)).cpu_seconds_per_request(0) == 0.0 + def test_summary_reports_every_percentile_in_human_units(self) -> None: window: Final = _window((0.0, 200 * _MB, 0.0), (1.0, 200 * _MB, 1.5), (2.0, 200 * _MB, 3.0)) diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index f91b9c2fa09..950712309dc 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -1,4 +1,4 @@ -"""Live e2e: the proxy under load keeps serving every request while Redis writes time out. +"""Live e2e: the proxy under load keeps serving every request while Redis is down entirely. Runs against a proxy booted from tests/e2e/gateway/redis_chaos_ci_config.yml, which points cache_params at a real Redis with litellm's default socket_timeout. That one client backs all @@ -11,11 +11,13 @@ retries on the failing pair (a 500 is retryable, so retries keep re-picking insi order) and the router's order-based fallback then re-targets order 2. Every request is expected to succeed, and each one carries retry breadcrumbs into cost tracking. -Phase A is a baseline with Redis healthy; phase B holds Redis in CLIENT PAUSE WRITE, so the -spend counter increment times out and the callback stringifies the request metadata, -breadcrumbs included, into a failed-tracking alert. On v1.100.0 that string doubled per request -until the worker hung (LIT-6780), which is what the per-phase RSS and CPU percentiles are here -to catch. +Phase A is a baseline with Redis healthy; phase B holds Redis in CLIENT PAUSE ALL for the +length of the phase, simulating Redis being down outright rather than merely slow to write. +Every touchpoint times out: the auth cache read falls back to Postgres, the response cache +read and write both fail, and the spend counter increment times out and the callback +stringifies the request metadata, breadcrumbs included, into a failed-tracking alert. On +v1.100.0 that string doubled per request until the worker hung (LIT-6780), which is what the +per-phase RSS and CPU percentiles are here to catch. Needs the proxy on the same host, since RSS and CPU come from psutil on its process tree: a multi-worker proxy serves /metrics from the prometheus multiprocess collector, which drops @@ -26,8 +28,10 @@ from __future__ import annotations import os import re +import time from collections.abc import Iterator from dataclasses import dataclass +from itertools import pairwise from typing import Final import pytest @@ -38,12 +42,13 @@ from lifecycle import ResourceManager from load_client import LoadClient from locust_load import LoadResult, run_chat_load from models import KeyGenerateBody, LiteLLMParamsBody +from phase_budget import Budget, violations from proxy_client import ProxyClient from proxy_usage import ProxyUsageSampler, UsageWindow -pytestmark = [pytest.mark.e2e, pytest.mark.redis_chaos] +pytestmark: Final = pytest.mark.e2e -MODEL_GROUP: Final = "redis-chaos-fable" +MODEL_GROUP: Final = f"redis-chaos-fable-{unique_marker()}" MOCK_MODEL: Final = "anthropic/claude-fable-5-1" FAILING_DEPLOYMENTS: Final = 2 SERVING_DEPLOYMENTS: Final = 1 @@ -54,19 +59,42 @@ LOCUST_USERS: Final = 50 LOCUST_SPAWN_RATE: Final = 50.0 BASELINE_SECONDS: Final = 60.0 CHAOS_SECONDS: Final = 90.0 -REDIS_PAUSE_MS: Final = 600_000 -BASELINE_TIMEOUT_RATE_CEILING: Final = 0.05 -CHAOS_TIMEOUT_RATE_FLOOR: Final = 0.20 +REDIS_PAUSE_MS: Final = int(CHAOS_SECONDS * 1000) + +# Chaos-phase ceilings, as a multiple of the same metric in the baseline phase. Ratios rather +# than absolutes because every absolute here is machine-shaped: RSS scales with worker count +# and latency with core count, so a number calibrated on one runner means nothing on another. +# Calibrated from local runs under CLIENT PAUSE ALL that came in around 4x latency at every +# percentile, 1.03x RSS and 4.4x CPU per request, and deliberately loose: the regression these +# guard against grew memory by an order of magnitude, so catching it does not need a tight +# bound, and a tight one would flake on a shared CI runner. Latency gets the most slack because +# it is the metric a Redis outage is legitimately allowed to move, by its socket timeout on +# every call a request attempts. +CHAOS_LATENCY_RATIO_CEILING: Final = 12.0 +CHAOS_RSS_RATIO_CEILING: Final = 1.5 +CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 6.0 + +DRAIN_TIMEOUT_SECONDS: Final = 30.0 +DRAIN_POLL_SECONDS: Final = 1.0 TIMEOUT_FAILURES_RE: Final = re.compile( r'^litellm_redis_circuit_breaker_failures_total\{failure_class="timeout"\} ([0-9.e+]+)$', re.M ) -BREAKER_OPEN_RE: Final = re.compile(r'^litellm_redis_circuit_breaker_state\{state="open"\} ([0-9.e+]+)$', re.M) +# The state gauge carries a pid label under the multiprocess collector, one series per worker, +# so this matches any label order rather than a bare {state="open"} that never appears. +BREAKER_OPEN_RE: Final = re.compile( + r'^litellm_redis_circuit_breaker_state\{[^}]*state="open"[^}]*\} ([0-9.e+]+)$', re.M +) BREAKER_TRANSITIONS_RE: Final = re.compile( r'^litellm_redis_circuit_breaker_transitions_total\{state="[a-z_]+"\} ([0-9.e+]+)$', re.M ) -RETRIES_RE: Final = re.compile(r"^litellm_deployment_failure_responses_total\{[^}]*\} ([0-9.e+]+)$", re.M) -COOLDOWN_RE: Final = re.compile(r"^litellm_deployment_cooled_down_total\{[^}]*\} ([0-9.e+]+)$", re.M) + + +def _deployment_metric_re(name: str, model_ids: tuple[str, ...]) -> re.Pattern[str]: + """A per-deployment counter, narrowed to the deployments one run registered, so traffic + anything else sends the same proxy during the run cannot pad the retry count.""" + ids: Final = "|".join(re.escape(model_id) for model_id in model_ids) + return re.compile(rf'^litellm_{name}\{{[^}}]*model_id="(?:{ids})"[^}}]*\}} ([0-9.e+]+)$', re.M) @dataclass(frozen=True, slots=True) @@ -76,11 +104,22 @@ class Phase: name: str load: LoadResult usage: UsageWindow + redis_timeouts: float + + @property + def timeouts_per_request(self) -> float: + return self.redis_timeouts / self.load.requests if self.load.requests else 0.0 + + @property + def cpu_seconds_per_request(self) -> float: + return self.usage.cpu_seconds_per_request(self.load.requests) def report(self) -> str: return ( f"{self.name}: {self.load.requests} requests, {self.load.failures} failures, " - f"{self.load.requests_per_second:.0f} rps, {self.load.latency_summary()}; {self.usage.summary()}" + f"{self.load.requests_per_second:.0f} rps, {self.load.latency_summary()}; {self.usage.summary()}; " + f"{self.cpu_seconds_per_request * 1000:.1f} ms CPU per request; " + f"{self.timeouts_per_request:.2f} Redis timeouts per request" ) @@ -119,10 +158,12 @@ def proxy_pid() -> int: @pytest.fixture def redis_control() -> Iterator[redis.Redis[bytes]]: - """A control connection to the proxy's Redis, which unpauses writes in teardown. + """A control connection to the proxy's Redis, which unpauses it in teardown as a safety net. - Only writes are paused: CLIENT PAUSE ALL would freeze this connection too, leaving - nothing able to lift the pause. + CLIENT PAUSE ALL freezes every connection including this one, so REDIS_PAUSE_MS is sized + to the chaos phase: by the time teardown runs, the pause has + already lapsed on its own and CLIENT UNPAUSE here returns immediately. It only actually + waits out a lapsed pause if the chaos phase itself overran that duration. """ host: Final = os.environ.get("REDIS_HOST") port: Final = os.environ.get("REDIS_PORT") @@ -135,18 +176,54 @@ def redis_control() -> Iterator[redis.Redis[bytes]]: control.close() -def _metric(proxy: ProxyClient, pattern: re.Pattern[str]) -> float: - body: Final = proxy.probe("/metrics", params=NoBody()).body - return sum(float(match.group(1)) for match in pattern.finditer(body)) +def _scrape(proxy: ProxyClient) -> str: + """One /metrics body, read once per checkpoint so every counter comes from the same instant.""" + scrape: Final = proxy.probe("/metrics", params=NoBody()) + assert scrape.status_code == 200, ( + f"/metrics did not answer ({scrape.status_code}: {scrape.body[:200]}), so no counter can be read; " + f"a silent 0 here would turn every before-and-after difference negative" + ) + return scrape.body -def _register_deployments(proxy: ProxyClient, resources: ResourceManager) -> None: - for _ in range(FAILING_DEPLOYMENTS): - failing_id = proxy.create_model(MODEL_GROUP, _failing_params()) - resources.defer(lambda model_id=failing_id: proxy.delete_model(model_id)) - for _ in range(SERVING_DEPLOYMENTS): - serving_id = proxy.create_model(MODEL_GROUP, _serving_params()) - resources.defer(lambda model_id=serving_id: proxy.delete_model(model_id)) +def _metric(scrape: str, pattern: re.Pattern[str]) -> float: + return sum(float(match.group(1)) for match in pattern.finditer(scrape)) + + +def _scrape_after_drain(proxy: ProxyClient, pattern: re.Pattern[str]) -> str: + """A /metrics body taken once `pattern`'s count has stopped moving. + + `set_llm_deployment_failure_metrics` runs from the async logging callback queue, so a load + generator that just stopped sending traffic can still have thousands of failure increments + in flight, and a scrape taken the instant load stops undercounts them. Settling on the + counter rather than sleeping a fixed duration keeps the wait proportional to how backed up + the queue actually is. + """ + deadline: Final = time.monotonic() + DRAIN_TIMEOUT_SECONDS + + def scrapes() -> Iterator[str]: + yield _scrape(proxy) + while time.monotonic() < deadline: + time.sleep(DRAIN_POLL_SECONDS) + yield _scrape(proxy) + + settled: Final = next( + (later for earlier, later in pairwise(scrapes()) if _metric(earlier, pattern) == _metric(later, pattern)), + None, + ) + return settled if settled is not None else _scrape(proxy) + + +def _register_deployments(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, ...]: + """The model ids this run registered, which scope its per-deployment metric reads.""" + params: Final = ( + *(_failing_params() for _ in range(FAILING_DEPLOYMENTS)), + *(_serving_params() for _ in range(SERVING_DEPLOYMENTS)), + ) + model_ids: Final = tuple(proxy.create_model(MODEL_GROUP, one) for one in params) + for model_id in model_ids: + resources.defer(lambda doomed=model_id: proxy.delete_model(doomed)) + return model_ids def _generate_key_pool(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, ...]: @@ -174,12 +251,68 @@ def _drive(keys: tuple[str, ...], seconds: float) -> LoadResult: ) +def _latency_budget(percentile: str, baseline: float, degraded: float) -> Budget: + return Budget( + name=f"{percentile} latency", + baseline=baseline, + degraded=degraded, + ratio_ceiling=CHAOS_LATENCY_RATIO_CEILING, + unit="s", + decimals=3, + ) + + +def _rss_budget(percentile: str, baseline: UsageWindow, degraded: UsageWindow, fraction: float) -> Budget: + return Budget( + name=f"{percentile} RSS", + baseline=baseline.rss_percentile(fraction) / 2**20, + degraded=degraded.rss_percentile(fraction) / 2**20, + ratio_ceiling=CHAOS_RSS_RATIO_CEILING, + unit=" MB", + decimals=0, + ) + + +def _chaos_budgets(baseline: Phase, chaos: Phase) -> tuple[Budget, ...]: + """What a Redis outage is allowed to cost, measured against the same run's healthy phase. + + Every request still succeeding is the headline assertion, but a proxy can answer every + request while leaking: the v1.100.0 regression (LIT-6780) served traffic the whole way up + to a 61 GB worker. These bound the cost of serving it. + + Latency and RSS are budgeted at p50, p90 and p99 so a regression that only shows up in the + tail (or only in the median) cannot hide behind the other. Latency gets the loosest bound + because a timing-out Redis legitimately adds its socket_timeout to every request that + touches it, several times over on a retried request. RSS gets the tightest: the failure + path has no business allocating more per request. CPU is budgeted once, as CPU seconds per + request rather than per percentile: cores-busy saturates at the worker count under load, so + its percentiles read the same whether a request costs 10 ms of CPU or 40, and cannot budget + anything; seconds per request is the CPU figure that actually moves. + """ + return ( + _latency_budget("p50", baseline.load.p50_seconds, chaos.load.p50_seconds), + _latency_budget("p90", baseline.load.p90_seconds, chaos.load.p90_seconds), + _latency_budget("p99", baseline.load.p99_seconds, chaos.load.p99_seconds), + _rss_budget("p50", baseline.usage, chaos.usage, 0.5), + _rss_budget("p90", baseline.usage, chaos.usage, 0.9), + _rss_budget("p99", baseline.usage, chaos.usage, 0.99), + Budget( + name="CPU per request", + baseline=baseline.cpu_seconds_per_request * 1000, + degraded=chaos.cpu_seconds_per_request * 1000, + ratio_ceiling=CHAOS_CPU_PER_REQUEST_RATIO_CEILING, + unit=" ms", + ), + ) + + +@pytest.mark.redis_chaos class TestRedisChaos: @pytest.mark.covers( "reliability.circuit_breaker.redis_timeout.stays_responsive", - exercised_on=["chat_completions"], + exercised_on=("chat_completions",), ) - def test_load_survives_redis_write_timeouts( + def test_load_survives_redis_being_down( self, client: LoadClient, resources: ResourceManager, @@ -187,20 +320,36 @@ class TestRedisChaos: redis_control: redis.Redis[bytes], ) -> None: proxy: Final = client.proxy - _register_deployments(proxy, resources) + model_ids: Final = _register_deployments(proxy, resources) keys: Final = _generate_key_pool(proxy, resources) - timeouts_at_start: Final = _metric(proxy, TIMEOUT_FAILURES_RE) - retries_before: Final = _metric(proxy, RETRIES_RE) - cooldowns_before: Final = _metric(proxy, COOLDOWN_RE) + retries_re: Final = _deployment_metric_re("deployment_failure_responses_total", model_ids) + cooldown_re: Final = _deployment_metric_re("deployment_cooled_down_total", model_ids) + + at_start: Final = _scrape(proxy) with ProxyUsageSampler(proxy_pid) as sampler: - baseline: Final = Phase(name="baseline", load=_drive(keys, BASELINE_SECONDS), usage=sampler.split()) - timeouts_after_baseline: Final = _metric(proxy, TIMEOUT_FAILURES_RE) + baseline_load: Final = _drive(keys, BASELINE_SECONDS) + baseline_usage: Final = sampler.split() + after_baseline: Final = _scrape(proxy) - redis_control.client_pause(REDIS_PAUSE_MS, all=False) # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any - chaos: Final = Phase(name="chaos", load=_drive(keys, CHAOS_SECONDS), usage=sampler.split()) + redis_control.client_pause(REDIS_PAUSE_MS, all=True) # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any + chaos_load: Final = _drive(keys, CHAOS_SECONDS) + chaos_usage: Final = sampler.split() + at_end: Final = _scrape_after_drain(proxy, retries_re) + baseline: Final = Phase( + name="baseline", + load=baseline_load, + usage=baseline_usage, + redis_timeouts=_metric(after_baseline, TIMEOUT_FAILURES_RE) - _metric(at_start, TIMEOUT_FAILURES_RE), + ) + chaos: Final = Phase( + name="chaos", + load=chaos_load, + usage=chaos_usage, + redis_timeouts=_metric(at_end, TIMEOUT_FAILURES_RE) - _metric(after_baseline, TIMEOUT_FAILURES_RE), + ) report: Final = f"{baseline.report()} | {chaos.report()}" for phase in (baseline, chaos): @@ -215,13 +364,13 @@ class TestRedisChaos: f"a Redis failure reached the response path. {phase.load.diagnosis()}. {report}" ) - cooldowns: Final = _metric(proxy, COOLDOWN_RE) - cooldowns_before + cooldowns: Final = _metric(at_end, cooldown_re) - _metric(at_start, cooldown_re) assert cooldowns == 0, ( f"{cooldowns:.0f} deployments were cooled down during the run; the failing deployments are supposed " f"to stay in rotation so every request keeps exercising the retry path. {report}" ) - retries: Final = _metric(proxy, RETRIES_RE) - retries_before + retries: Final = _metric(at_end, retries_re) - _metric(at_start, retries_re) assert retries >= baseline.load.requests + chaos.load.requests, ( f"only {retries:.0f} deployment failures were counted across " f"{baseline.load.requests + chaos.load.requests} requests; the mock deployments did not fail, so no " @@ -229,24 +378,17 @@ class TestRedisChaos: f"{report}" ) - baseline_timeout_rate: Final = (timeouts_after_baseline - timeouts_at_start) / baseline.load.requests - assert baseline_timeout_rate <= BASELINE_TIMEOUT_RATE_CEILING, ( - f"a healthy Redis timed out on {baseline_timeout_rate:.1%} of baseline requests, over the " - f"{BASELINE_TIMEOUT_RATE_CEILING:.0%} this test tolerates; at litellm's default socket_timeout a " - f"loaded Redis does time out occasionally, but this much means the baseline is already degraded and " - f"the two phases are not comparable. {report}" + transitions: Final = _metric(at_end, BREAKER_TRANSITIONS_RE) - _metric(after_baseline, BREAKER_TRANSITIONS_RE) + breaker_open: Final = _metric(at_end, BREAKER_OPEN_RE) >= 1 + assert transitions >= 1 or breaker_open, ( + f"pausing Redis produced no circuit breaker state transitions and it ended closed; nothing on the " + f"request path ever saw Redis fail, so this run proved nothing. {report}" ) - chaos_timeouts: Final = _metric(proxy, TIMEOUT_FAILURES_RE) - timeouts_after_baseline - chaos_timeout_rate: Final = chaos_timeouts / chaos.load.requests - transitions: Final = _metric(proxy, BREAKER_TRANSITIONS_RE) - breaker_open: Final = _metric(proxy, BREAKER_OPEN_RE) >= 1 - assert chaos_timeout_rate >= CHAOS_TIMEOUT_RATE_FLOOR or transitions >= 1 or breaker_open, ( - f"with writes paused the breaker saw Redis time out on only {chaos_timeout_rate:.1%} of requests " - f"against {baseline_timeout_rate:.1%} at baseline, under the {CHAOS_TIMEOUT_RATE_FLOOR:.0%} a real " - f"outage produces, and it counted {transitions:.0f} state transitions and ended " - f"{'open' if breaker_open else 'closed'}. The spend counter increment never failed, so this run " - f"proved nothing. {report}" + blown: Final = violations(_chaos_budgets(baseline, chaos)) + assert not blown, ( + f"pausing Redis cost the proxy more than the socket timeout on the calls it attempts: " + f"{'; '.join(blown)}. {report}" ) rows: Final = proxy.poll_logs_for_key(keys[0], min_rows=1) diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index 1dcb50a6fab..774d9644497 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -9,4 +9,4 @@ markers = load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set managed_files: needs a proxy running with require_managed_files enabled; deselected unless E2E_MANAGED_FILES_STACK is set - redis_chaos: load test that pauses the proxy's Redis writes mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set + redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set From 1dc0e363b0c62ced71671ed91454a476ef544f90 Mon Sep 17 00:00:00 2001 From: Mateo Wang <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 11:47:05 -0700 Subject: [PATCH 089/157] fix(proxy): authorize every Responses API id, not only the ones the proxy issued (#39548) * fix(proxy): authorize every Responses API id, not only the ones the proxy issued The ownership check on the Responses API only ran when the id arrived in the proxy's own encrypted format. An id in any other shape skipped the check and was forwarded upstream, so a key that did not own the response could retrieve, cancel, delete, or chain off it. Every addressed id now goes through one authorization step shared by retrieve, cancel, delete, list-input-items, and create's previous_response_id. An id the proxy did not issue is refused with 403 unless the deployment opts in with general_settings.allow_unmanaged_response_ids, has responses id security disabled, has no signing key configured, or the caller is a proxy admin. * fix(proxy): re-authorize the retained responses id instead of trusting it --- litellm/proxy/_types.py | 19 ++ litellm/proxy/hooks/responses_id_security.py | 99 +++++--- .../test_responses_id_security.py | 239 ++++++++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 10 + 4 files changed, 338 insertions(+), 29 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 43af18cbef4..382ea384275 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2857,6 +2857,25 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): "UI username/password login. Default is False." ), ) + disable_responses_id_security: bool | None = Field( + None, + description=( + "If True, disables ownership enforcement on Responses API ids. " + "Keys may then retrieve, cancel, delete, and chain from any response id, " + "including ids belonging to another user or team and ids this proxy never issued. " + "WARNING: this removes tenant isolation on /v1/responses" + ), + ) + allow_unmanaged_response_ids: bool | None = Field( + None, + description=( + "If True, lets keys address Responses API ids that this proxy did not issue " + "(raw provider ids, or ids issued before response-id encryption was configured). " + "Such an id carries no owner, so no ownership check can run on it; ids this proxy " + "did issue keep full ownership enforcement. Off by default, in which case an " + "unrecognized response id is rejected with 403" + ), + ) disable_env_credential_login: bool | None = Field( None, description=( diff --git a/litellm/proxy/hooks/responses_id_security.py b/litellm/proxy/hooks/responses_id_security.py index c4c15c40d1e..7e7f70d6f7e 100644 --- a/litellm/proxy/hooks/responses_id_security.py +++ b/litellm/proxy/hooks/responses_id_security.py @@ -32,6 +32,29 @@ if TYPE_CHECKING: _RESPONSES_API_PROVIDER_PREFIX: Final = "/openai" _RESPONSES_API_CREATE_ROUTES: Final = frozenset({"/v1/responses", "/responses"}) +_ADDRESSED_RESPONSE_ID_KEY: Final = "_litellm_addressed_response_id" +_UNMANAGED_RESPONSE_ID_DETAIL: Final = ( + "Forbidden. This response id was not issued by this proxy, so the proxy cannot tell who owns it. " + "To let keys address responses this proxy did not issue, set " + "general_settings::allow_unmanaged_response_ids to True in the config.yaml file." +) +_PROXY_ADMIN_ROLES: Final = frozenset({LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN.value}) + + +def _proxy_general_settings() -> Mapping[str, Any]: + from litellm.proxy.proxy_server import general_settings + + return general_settings + + +def _proxy_signing_key() -> str | None: + import os + + from litellm.proxy.proxy_server import master_key + + salt_key: Final = os.getenv("LITELLM_SALT_KEY", None) + return master_key if salt_key is None else salt_key + _RESPONSE_PAYLOAD_ADAPTER: Final = TypeAdapter(Mapping[str, object]) @@ -83,8 +106,13 @@ def _is_responses_api_create_route(request_route: str | None) -> bool: class ResponsesIDSecurity(CustomLogger): - def __init__(self): - pass + def __init__( + self, + general_settings_reader: Callable[[], Mapping[str, Any]] = _proxy_general_settings, + signing_key_reader: Callable[[], str | None] = _proxy_signing_key, + ) -> None: + self._general_settings_reader: Final = general_settings_reader + self._signing_key_reader: Final = signing_key_reader async def async_pre_call_hook( self, @@ -103,30 +131,51 @@ class ResponsesIDSecurity(CustomLogger): } if call_type not in responses_api_call_types: return None - if call_type == "aresponses": - # check 'previous_response_id' if present in the data - previous_response_id: Final = data.get("previous_response_id") - if previous_response_id and self._is_encrypted_response_id(previous_response_id): - original_response_id, user_id, team_id = self._decrypt_response_id(previous_response_id) - self.check_user_access_to_response_id(user_id, team_id, user_api_key_dict) - data["previous_response_id"] = original_response_id - elif call_type in {"aget_responses", "adelete_responses", "acancel_responses", "alist_input_items"}: - response_id: Final = data.get("response_id") - - if response_id and self._is_encrypted_response_id(response_id): - original_response_id, user_id, team_id = self._decrypt_response_id(response_id) - - self.check_user_access_to_response_id(user_id, team_id, user_api_key_dict) - data["response_id"] = original_response_id + addressed_id_field: Final = "previous_response_id" if call_type == "aresponses" else "response_id" + retained_id: Final = data.get(_ADDRESSED_RESPONSE_ID_KEY) + addressed_id: Final = ( + retained_id if isinstance(retained_id, str) and retained_id else data.get(addressed_id_field) + ) + if not isinstance(addressed_id, str) or not addressed_id: + return data + authorized_id: Final = self._authorize_response_id(addressed_id, user_api_key_dict) + data[addressed_id_field] = authorized_id + data[_ADDRESSED_RESPONSE_ID_KEY] = addressed_id return data + def _authorize_response_id( + self, + response_id: str, + user_api_key_dict: "UserAPIKeyAuth", + ) -> str: + if self._is_encrypted_response_id(response_id): + original_response_id, user_id, team_id = self._decrypt_response_id(response_id) + self.check_user_access_to_response_id(user_id, team_id, user_api_key_dict) + return original_response_id + + if self._unmanaged_response_ids_allowed(user_api_key_dict): + return response_id + + raise HTTPException(status_code=403, detail=_UNMANAGED_RESPONSE_ID_DETAIL) + + def _unmanaged_response_ids_allowed(self, user_api_key_dict: "UserAPIKeyAuth") -> bool: + general_settings: Final = self._general_settings_reader() + + if general_settings.get("disable_responses_id_security", False): + return True + if general_settings.get("allow_unmanaged_response_ids", False): + return True + if self._get_signing_key() is None: + return True + return user_api_key_dict.user_role in _PROXY_ADMIN_ROLES + def check_user_access_to_response_id( self, response_id_user_id: str | None, response_id_team_id: str | None, user_api_key_dict: "UserAPIKeyAuth", ) -> bool: - from litellm.proxy.proxy_server import general_settings + general_settings: Final = self._general_settings_reader() if ( user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value @@ -219,15 +268,7 @@ class ResponsesIDSecurity(CustomLogger): return response_id, None, None def _get_signing_key(self) -> str | None: - """Get the signing key for encryption/decryption.""" - import os - - from litellm.proxy.proxy_server import master_key - - salt_key = os.getenv("LITELLM_SALT_KEY", None) - if salt_key is None: - salt_key = master_key - return salt_key + return self._signing_key_reader() def _encrypt_response_id( self, @@ -274,7 +315,7 @@ class ResponsesIDSecurity(CustomLogger): This method adds response IDs to an in-memory queue, which are then batch-processed by the DBSpendUpdateWriter during regular database update cycles. """ - from litellm.proxy.proxy_server import general_settings + general_settings: Final = self._general_settings_reader() if general_settings.get("disable_responses_id_security", False): return response @@ -288,7 +329,7 @@ class ResponsesIDSecurity(CustomLogger): async def async_post_call_streaming_iterator_hook( self, user_api_key_dict: "UserAPIKeyAuth", response: Any, request_data: dict ) -> AsyncGenerator[BaseLiteLLMOpenAIResponseObject, None]: - from litellm.proxy.proxy_server import general_settings + general_settings: Final = self._general_settings_reader() # Create a request-scoped cache for consistent encryption across streaming chunks. request_encryption_cache: Final[dict[str, str]] = {} diff --git a/tests/test_litellm/test_responses_id_security.py b/tests/test_litellm/test_responses_id_security.py index d35b9563888..a6081670172 100644 --- a/tests/test_litellm/test_responses_id_security.py +++ b/tests/test_litellm/test_responses_id_security.py @@ -855,3 +855,242 @@ class TestAsyncPostCallSuccessHook: ) assert result == mock_response + + + +_FABRICATED_PROVIDER_RESPONSE_ID = "resp_fabricatedprovideridaaaaaaaaaaaaaaaa" +_FABRICATED_UNMANAGED_ID = "resp_fabricatedunmanagedidbbbbbbbbbbbbbbbb" +_UNIT_TEST_SALT_KEY = "lit6837-unit-test-salt-key" +_ADDRESSED_ID_FIELD_BY_CALL_TYPE = { + "aresponses": "previous_response_id", + "aget_responses": "response_id", + "adelete_responses": "response_id", + "acancel_responses": "response_id", + "alist_input_items": "response_id", +} + + +@pytest.fixture +def salt_key_env(monkeypatch): + """Give the encrypt/decrypt helpers a real salt key so ids round-trip for real.""" + monkeypatch.setenv("LITELLM_SALT_KEY", _UNIT_TEST_SALT_KEY) + return _UNIT_TEST_SALT_KEY + + +def _hook(general_settings=None, signing_key=_UNIT_TEST_SALT_KEY): + settings = general_settings if general_settings is not None else {} + return ResponsesIDSecurity( + general_settings_reader=lambda: settings, + signing_key_reader=lambda: signing_key, + ) + + +def _auth(user_id="owner-user", team_id="owner-team", user_role=None): + from litellm.proxy._types import UserAPIKeyAuth + + return UserAPIKeyAuth(user_id=user_id, team_id=team_id, user_role=user_role) + + +def _issue_managed_id(hook, owner, provider_response_id=_FABRICATED_PROVIDER_RESPONSE_ID): + """Mint an id exactly the way the proxy hands one to a client on create.""" + issued = hook._encrypt_response_id( + ResponsesAPIResponse( + id=provider_response_id, created_at=1234567890, output=[], status="completed" + ), + owner, + ) + return issued.id + + +class TestUnrecognizedResponseIdIsRejected: + """An id this proxy never issued carries no owner, so it must not reach the provider.""" + + @pytest.mark.asyncio + @pytest.mark.parametrize("call_type", sorted(_ADDRESSED_ID_FIELD_BY_CALL_TYPE)) + async def test_unmanaged_id_is_rejected_and_not_forwarded(self, mock_cache, salt_key_env, call_type): + field = _ADDRESSED_ID_FIELD_BY_CALL_TYPE[call_type] + data = {field: _FABRICATED_UNMANAGED_ID} + + with pytest.raises(HTTPException) as exc_info: + await _hook().async_pre_call_hook( + user_api_key_dict=_auth(), + cache=mock_cache, + data=data, + call_type=call_type, + ) + + assert exc_info.value.status_code == 403 + assert "allow_unmanaged_response_ids" in exc_info.value.detail + assert data[field] == _FABRICATED_UNMANAGED_ID + + @pytest.mark.asyncio + async def test_owner_can_still_address_the_id_the_proxy_issued_it(self, mock_cache, salt_key_env): + hook = _hook() + owner = _auth() + data = {"response_id": _issue_managed_id(hook, owner)} + + result = await hook.async_pre_call_hook( + user_api_key_dict=owner, + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert result["response_id"] == _FABRICATED_PROVIDER_RESPONSE_ID + + @pytest.mark.asyncio + async def test_stranger_cannot_address_an_id_issued_to_someone_else(self, mock_cache, salt_key_env): + hook = _hook() + issued_id = _issue_managed_id(hook, _auth()) + data = {"response_id": issued_id} + + with pytest.raises(HTTPException) as exc_info: + await hook.async_pre_call_hook( + user_api_key_dict=_auth(user_id="stranger-user", team_id="stranger-team"), + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert exc_info.value.status_code == 403 + assert data["response_id"] == issued_id + + @pytest.mark.asyncio + async def test_unmanaged_previous_response_id_cannot_seed_a_new_response(self, mock_cache, salt_key_env): + data = {"model": "gpt-fake", "previous_response_id": _FABRICATED_UNMANAGED_ID} + + with pytest.raises(HTTPException) as exc_info: + await _hook().async_pre_call_hook( + user_api_key_dict=_auth(), + cache=mock_cache, + data=data, + call_type="aresponses", + ) + + assert exc_info.value.status_code == 403 + assert data["previous_response_id"] == _FABRICATED_UNMANAGED_ID + + @pytest.mark.asyncio + async def test_re_entering_the_hook_on_the_same_request_does_not_reject(self, mock_cache, salt_key_env): + """The rate-limit fallback retry runs pre-call twice over one already-rewritten dict.""" + hook = _hook() + owner = _auth() + data = {"model": "gpt-fake", "previous_response_id": _issue_managed_id(hook, owner)} + + first = await hook.async_pre_call_hook( + user_api_key_dict=owner, cache=mock_cache, data=data, call_type="aresponses" + ) + second = await hook.async_pre_call_hook( + user_api_key_dict=owner, cache=mock_cache, data=first, call_type="aresponses" + ) + + assert second["previous_response_id"] == _FABRICATED_PROVIDER_RESPONSE_ID + + +class TestUnmanagedResponseIdEscapeHatches: + """Deployments that pass provider ids through on purpose must keep working.""" + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "general_settings", + [{"allow_unmanaged_response_ids": True}, {"disable_responses_id_security": True}], + ) + async def test_opted_in_settings_forward_the_id_untouched(self, mock_cache, salt_key_env, general_settings): + data = {"response_id": _FABRICATED_UNMANAGED_ID} + + result = await _hook(general_settings=general_settings).async_pre_call_hook( + user_api_key_dict=_auth(), + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert result["response_id"] == _FABRICATED_UNMANAGED_ID + + @pytest.mark.asyncio + async def test_proxy_without_a_signing_key_forwards_the_id_untouched(self, mock_cache, monkeypatch): + monkeypatch.delenv("LITELLM_SALT_KEY", raising=False) + data = {"response_id": _FABRICATED_UNMANAGED_ID} + + result = await _hook(signing_key=None).async_pre_call_hook( + user_api_key_dict=_auth(), + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert result["response_id"] == _FABRICATED_UNMANAGED_ID + + @pytest.mark.asyncio + async def test_proxy_admin_may_address_an_unmanaged_id(self, mock_cache, salt_key_env): + from litellm.proxy._types import LitellmUserRoles + + data = {"response_id": _FABRICATED_UNMANAGED_ID} + + result = await _hook().async_pre_call_hook( + user_api_key_dict=_auth(user_role=LitellmUserRoles.PROXY_ADMIN), + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert result["response_id"] == _FABRICATED_UNMANAGED_ID + + +class TestClientSuppliedRetainedIdCannotBypassAuthorization: + """The retained-id key travels in the request body, so it is re-authorized, never trusted.""" + + @pytest.mark.asyncio + @pytest.mark.parametrize("call_type", sorted(_ADDRESSED_ID_FIELD_BY_CALL_TYPE)) + async def test_forged_retained_id_is_still_authorized(self, mock_cache, salt_key_env, call_type): + field = _ADDRESSED_ID_FIELD_BY_CALL_TYPE[call_type] + data = { + field: _FABRICATED_UNMANAGED_ID, + "_litellm_addressed_response_id": _FABRICATED_UNMANAGED_ID, + } + + with pytest.raises(HTTPException) as exc_info: + await _hook().async_pre_call_hook( + user_api_key_dict=_auth(), + cache=mock_cache, + data=data, + call_type=call_type, + ) + + assert exc_info.value.status_code == 403 + assert data[field] == _FABRICATED_UNMANAGED_ID + + @pytest.mark.asyncio + @pytest.mark.parametrize("forged", [{"nested": "value"}, ["list"], 42, "", None]) + async def test_non_string_retained_id_falls_back_to_the_addressed_field(self, mock_cache, salt_key_env, forged): + data = {"response_id": _FABRICATED_UNMANAGED_ID, "_litellm_addressed_response_id": forged} + + with pytest.raises(HTTPException) as exc_info: + await _hook().async_pre_call_hook( + user_api_key_dict=_auth(), + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert exc_info.value.status_code == 403 + + @pytest.mark.asyncio + async def test_stranger_forging_their_own_id_never_reaches_someone_elses_response( + self, mock_cache, salt_key_env + ): + hook = _hook() + stranger = _auth(user_id="stranger-user", team_id="stranger-team") + stranger_id = _issue_managed_id(hook, stranger, provider_response_id="resp_strangerownprovideridcccccccc") + victim_provider_id = "resp_victimprovideriddddddddddddddddddddd" + data = {"response_id": victim_provider_id, "_litellm_addressed_response_id": stranger_id} + + result = await hook.async_pre_call_hook( + user_api_key_dict=stranger, + cache=mock_cache, + data=data, + call_type="aget_responses", + ) + + assert result["response_id"] == "resp_strangerownprovideridcccccccc" + assert result["response_id"] != victim_provider_id diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 7fd4f8e413d..6fb0d70dc58 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -25629,6 +25629,11 @@ export interface components { * @description opt-in to RFC 8628 verification_uri_complete for the CLI SSO device flow, pre-filling the user_code in the browser. Off by default; intended for same-host clients where the device that starts the flow and the browser run on the same machine */ allow_cli_sso_verification_uri_complete?: boolean | null; + /** + * Allow Unmanaged Response Ids + * @description If True, lets keys address Responses API ids that this proxy did not issue (raw provider ids, or ids issued before response-id encryption was configured). Such an id carries no owner, so no ownership check can run on it; ids this proxy did issue keep full ownership enforcement. Off by default, in which case an unrecognized response id is rejected with 403 + */ + allow_unmanaged_response_ids?: boolean | null; /** * Allowed Routes * @description Proxy API Endpoints you want users to be able to access @@ -25748,6 +25753,11 @@ export interface components { * @description If True and SSO is configured (MICROSOFT_CLIENT_ID, GOOGLE_CLIENT_ID, GENERIC_CLIENT_ID, or SAML_IDP_METADATA_URL/XML), disables username/password login on /login, /v2/login, and /v3/login so SSO is the only way to reach the Admin UI. An admin locked out of the UI can still administer the proxy over the API with the master key; unset this setting and restart the proxy to restore UI username/password login. Default is False. */ disable_password_login_when_sso_enabled?: boolean | null; + /** + * Disable Responses Id Security + * @description If True, disables ownership enforcement on Responses API ids. Keys may then retrieve, cancel, delete, and chain from any response id, including ids belonging to another user or team and ids this proxy never issued. WARNING: this removes tenant isolation on /v1/responses + */ + disable_responses_id_security?: boolean | null; /** * Enable Openai Websocket Passthrough * @description Serve the OpenAI pass-through WebSocket route, which relays frames to OpenAI under the proxy's own provider credential without reading them. Off by default. From 95b438013ad839f4c39274762bc20da7962caf88 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 11:51:55 -0700 Subject: [PATCH 090/157] fix(router): fall back from unhealthy auto-router tier (#40757) * fix(router): fall back from unhealthy auto-router tier Co-Authored-By: Claude Code (cherry picked from commit 00c7fd8376decfbc4f1908281a4da0d725457249) * fix(router): treat budget and tag exhaustion as a no-capacity verdict The eligibility probe only read typed router errors as "nothing here can serve this". Provider and deployment budget exhaustion, and tag routing with no matching deployment, report it as a bare ValueError carrying a RouterErrors marker, so the probe read a spent tier as live, skipped the peer and default recovery, and failed the request. --------- Co-authored-by: Tin Chi Lo Co-authored-by: Claude Code --- litellm/router.py | 6 +- .../adaptive_router/adaptive_router.py | 2 + .../complexity_router/README.md | 12 + .../complexity_router/complexity_router.py | 251 +++++--- litellm/types/utils.py | 1 + .../adaptive_router/test_adaptive_router.py | 32 + .../router_strategy/test_complexity_router.py | 608 +++++++++++++++++- tests/test_litellm/test_router.py | 57 ++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 2 +- 9 files changed, 875 insertions(+), 96 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 9e6db66db83..597bcfaa20f 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -12669,6 +12669,7 @@ class Router: input: str | list | None = None, specific_deployment: bool | None = False, parent_otel_span: Span | None = None, + health_check_probe: bool = False, ) -> list[dict] | dict: """ Get the healthy deployments for a model. @@ -12718,6 +12719,7 @@ class Router: healthy_deployments = await self._async_filter_health_check_unhealthy_deployments( healthy_deployments=healthy_deployments, parent_otel_span=parent_otel_span, + health_check_probe=health_check_probe, ) cooldown_deployments: Final = await _async_get_cooldown_deployments( @@ -14100,6 +14102,7 @@ class Router: self, healthy_deployments: list[dict], parent_otel_span: Span | None = None, + health_check_probe: bool = False, ) -> list[dict]: """ Filter out deployments marked unhealthy by background health checks. @@ -14136,8 +14139,7 @@ class Router: ] if not filtered: - verbose_router_logger.warning("All deployments marked unhealthy by health checks, bypassing health filter") - return healthy_deployments + return [] if health_check_probe else healthy_deployments # mutable-ok: empty list signals unavailable probe return filtered diff --git a/litellm/router_strategy/adaptive_router/adaptive_router.py b/litellm/router_strategy/adaptive_router/adaptive_router.py index 12ccacbbc1d..7f376b46a8d 100644 --- a/litellm/router_strategy/adaptive_router/adaptive_router.py +++ b/litellm/router_strategy/adaptive_router/adaptive_router.py @@ -461,6 +461,8 @@ class AdaptiveRouter: if d_alpha == 0 and d_beta == 0: continue cell_key = (attribution_type, target_model) + if cell_key not in self._cells: + continue self._cells[cell_key] = apply_delta( self._cells[cell_key], d_alpha, diff --git a/litellm/router_strategy/complexity_router/README.md b/litellm/router_strategy/complexity_router/README.md index 1a4764c291e..4aa342ea59f 100644 --- a/litellm/router_strategy/complexity_router/README.md +++ b/litellm/router_strategy/complexity_router/README.md @@ -270,6 +270,18 @@ change or default takeover records `cause: modality_escalation` with the displac pinned by session affinity, and by default a KEPT session pin bypasses the gate: a session pinned to a text-only model keeps it even when an image arrives. +Context-window and modality recovery take priority over the default model. If a compatible tier +cannot serve, the router checks the remaining compatible recovery tiers before using `default_model`. +A capacity failure without those constraints tries the selected tier's peers, then the default + +The default must fit the context and accept the request's modality. It cannot bypass routing plugins +or a plan-mode floor. Context fit uses the auto-router's existing buffer even when Router-wide pre-call +checks are off. Missing context metadata retains the existing unknown-window behavior + +Health fallback records `cause: health_default_fallback` and `health_displaced:` in `signals`. +It does not replace the session's tier pin. Adaptive feedback retains the model that actually served, +but a default outside the adaptive candidate pool does not become a normal candidate + Add `modality_pin_override: true` to lift that last exemption. The image turn is then re-placed the same way every other decision is, and records `cause: modality_pin_override` whether or not the tier moved, since the model left the pin either way. The pin itself is untouched: the session diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 7519f4c5156..9deccc9a468 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -937,6 +937,7 @@ def _decision_is_pinnable(decision: StandardLoggingRoutingDecision | None) -> bo "modality_escalation", "modality_pin_override", "health_failover", + "health_default_fallback", ) and not decision.get("context_escalated") and _CLASSIFIER_CIRCUIT_OPEN_SIGNAL not in (decision.get("signals") or ()) @@ -1098,6 +1099,15 @@ def _group_provably_fits(facts: tuple[int | None, bool], needed: int, buffer: fl return window is not None and not has_unknown and needed <= int(window * buffer) +class _RequestContextFit(NamedTuple): + facts: Mapping[str, tuple[int | None, bool]] + needed: int | None + buffer: float + + def accepts(self, model: str) -> bool: + return self.needed is None or _window_can_hold(self.facts.get(model, (None, True))[0], self.needed, self.buffer) + + class _ContextWindowPlacement(NamedTuple): """Where the context-window gate placed the request: the placement tier, the subset of its pool the pick may use, and every configured group not provably misfit (the adaptive filter).""" @@ -2681,12 +2691,32 @@ class ComplexityRouter(CustomLogger): verbose_router_logger.debug("ComplexityRouter: context-window token count failed. Got - %s", e) return None + async def _request_context_fit( + self, + resolved_messages: Sequence[Mapping[str, object]] | None, + request_kwargs: Mapping[str, object], + ) -> _RequestContextFit: + if not self.config.enable_context_window_escalation or not resolved_messages: + return _RequestContextFit(EMPTY_MAPPING, None, self.config.context_window_escalation_buffer) + names: Final = frozenset(model for pool in self._tier_pools().values() for model in pool) | frozenset( + (self.config.default_model,) if self.config.default_model else () + ) + facts: Final = MappingProxyType({name: self._group_window_facts(name) for name in names}) + known: Final = tuple(window for window, _ in facts.values() if window is not None) + buffer: Final = self.config.context_window_escalation_buffer + needs_count: Final = known and self._request_byte_upper_bound(resolved_messages, request_kwargs) > int( + min(known) * buffer + ) + needed: Final = await self._counted_request_tokens(resolved_messages, request_kwargs) if needs_count else None + return _RequestContextFit(facts=facts, needed=needed, buffer=buffer) + async def _context_window_placement( self, tier: ComplexityTier | str, resolved_messages: Sequence[Mapping[str, object]] | None, request_kwargs: Mapping[str, object], pool_override: tuple[str, ...] | None = None, + context_fit: _RequestContextFit | None = None, ) -> _ContextWindowPlacement | None: """Correct a decided placement whose models provably cannot hold the prompt, or None (the placement stands). Only a real tokenizer count ever moves a request, escalation @@ -2698,17 +2728,10 @@ class ComplexityRouter(CustomLogger): pool: Final = pool_override if pool_override is not None else tuple(pools.get(_tier_name(tier), ())) if not pool: return None - facts: Final = MappingProxyType({group: self._group_window_facts(group) for group in pool}) - known_windows: Final = tuple(window for window, _ in facts.values() if window is not None) - if not known_windows: + fit: Final = context_fit or await self._request_context_fit(resolved_messages, request_kwargs) + if fit.needed is None: return None - buffer: Final = self.config.context_window_escalation_buffer - if self._request_byte_upper_bound(resolved_messages, request_kwargs) <= int(min(known_windows) * buffer): - return None - needed: Final = await self._counted_request_tokens(resolved_messages, request_kwargs) - if needed is None: - return None - return self._placement_for_tokens(tier=tier, pool=pool, pools=pools, facts=facts, needed=needed) + return self._placement_for_tokens(tier=tier, pool=pool, pools=pools, facts=fit.facts, needed=fit.needed) def _placement_for_tokens( self, @@ -2720,14 +2743,16 @@ class ComplexityRouter(CustomLogger): needed: int, ) -> _ContextWindowPlacement | None: buffer: Final = self.config.context_window_escalation_buffer - in_tier: Final = tuple(group for group in pool if _window_can_hold(facts[group][0], needed, buffer)) + in_tier: Final = tuple( + group for group in pool if _window_can_hold(facts.get(group, (None, True))[0], needed, buffer) + ) if in_tier and len(in_tier) == len(pool): return None holdable: Final = frozenset( group for tier_pool in pools.values() for group in tier_pool - if _window_can_hold(self._group_window_facts(group)[0], needed, buffer) + if _window_can_hold(facts.get(group, (None, True))[0], needed, buffer) ) if in_tier: return _ContextWindowPlacement(tier=tier, allowed_models=in_tier, holdable_models=holdable) @@ -2735,7 +2760,7 @@ class ComplexityRouter(CustomLogger): proven = tuple( group for group in pools.get(name, ()) - if _group_provably_fits(self._group_window_facts(group), needed, buffer) + if _group_provably_fits(facts.get(group, (None, True)), needed, buffer) ) if proven: return _ContextWindowPlacement( @@ -2881,6 +2906,7 @@ class ComplexityRouter(CustomLogger): messages: list[dict[str, Any]] | None, # mutable-ok: forwarded verbatim to the list-typed re-pick resolved_messages: Sequence[Mapping[str, object]] | None, request_kwargs: dict, # mutable-ok: same shape the hook receives + context_fit: _RequestContextFit | None = None, ) -> PreRoutingHookResponse: """Replace a routed model that cannot accept this request's image input. @@ -2911,7 +2937,8 @@ class ComplexityRouter(CustomLogger): or self._model_accepts_image_input(response.model) ): return response - eligible: Final = self._modality_eligible_models() + fit: Final = context_fit or await self._request_context_fit(resolved_messages, request_kwargs) + eligible: Final = frozenset(name for name in self._modality_eligible_models() if fit.accepts(name)) names: Final = self.config.tier_names() pools: Final = self._tier_pools() decided: Final = decision.get("tier") if decision is not None else None @@ -3030,12 +3057,15 @@ class ComplexityRouter(CustomLogger): Every way the owner says "nothing here can serve this" is a negative verdict: no healthy deployment for the group at all (BadRequestError, which ContextWindowExceededError - subclasses), every deployment filtered out (RouterRateLimitError), and every deployment - over its RPM (RouterRateLimitErrorBasic). Anything else is unknown rather than negative, - so it reads as capacity: absent information must never decide the verdict. + subclasses), every deployment filtered out (RouterRateLimitError), every deployment over + its RPM (RouterRateLimitErrorBasic), and every deployment refused by a filter that reports + exhaustion as a bare ValueError naming a RouterErrors marker -- provider and deployment + budgets, and tag routing, which have no typed error of their own. Anything else is unknown + rather than negative, so it reads as capacity: absent information must never decide the + verdict. """ from litellm.exceptions import BadRequestError - from litellm.types.router import RouterRateLimitError, RouterRateLimitErrorBasic + from litellm.types.router import RouterErrors, RouterRateLimitError, RouterRateLimitErrorBasic probe_kwargs: Final = dict(request_kwargs) # mutable-ok: the owner pops routing keys off the dict it is handed try: @@ -3045,10 +3075,15 @@ class ComplexityRouter(CustomLogger): messages=messages, input=input, parent_otel_span=_get_parent_otel_span_from_kwargs(request_kwargs), + health_check_probe=True, ) - except (RouterRateLimitError, RouterRateLimitErrorBasic, BadRequestError): + except (RouterRateLimitError, RouterRateLimitErrorBasic, BadRequestError) as exc: + verbose_router_logger.debug("health probe unavailable model=%s error=%s", model_name, type(exc).__name__) return False except Exception as exc: # noqa: BLE001 # a speculative eligibility read must fail open on unknown faults + if isinstance(exc, ValueError) and any(marker.value in str(exc) for marker in RouterErrors): + verbose_router_logger.debug("health probe exhausted model=%s error=%s", model_name, exc) + return False verbose_router_logger.debug( "ComplexityRouter: eligibility probe for %s failed, treating the group as live: %s", model_name, exc ) @@ -3062,76 +3097,124 @@ class ComplexityRouter(CustomLogger): input: str | list | None, # mutable-ok: mirrors the owner's own input parameter, which this forwards verbatim resolved_messages: Sequence[Mapping[str, object]] | None, request_kwargs: dict, # mutable-ok: same shape the hook receives + context_fit: _RequestContextFit | None = None, ) -> PreRoutingHookResponse: - """Replace a decided model group that has no serving capacity with a live peer in the same tier. - - Applied to the decided response at the hook's exits, so every arm that can place a request - is covered by one owner: a fresh classification, a replayed or escalated session pin, a - plan-mode floor, a context-window escalation, an adaptive pick, and whatever arm is added - next. Peers come from the DECIDED tier only; climbing to another tier is deliberately not - done here, since a higher tier costs more than the classifier asked for. - - Serving capacity is one question asked of one owner (`_model_group_can_serve`), so the - substitute is only ever a group the pipeline would actually accept for this request. The - pick then runs through `_pick_model_for_tier`, so routing plugins decide the substitute - exactly as they decided the original. - - Fails open everywhere it cannot be sure: an unreadable eligibility view, a decision - carrying no tier (default_model), or a tier whose every peer is unusable too. It fails - CLOSED on a plugin that empties the pool, leaving the original decision to fail rather - than serving a model the plugin excluded. - """ + """Try compatible tier recovery before the default, preserving request policy and fit.""" decision: Final = response.routing_decision decided_tier: Final = decision.get("tier") if decision is not None else None if decision is None or not isinstance(decided_tier, str): return response - peers: Final = tuple(self._tier_pools().get(decided_tier, ())) - if len(peers) < 2: - return response - if await self._model_group_can_serve(response.model, messages, input, request_kwargs): + fit: Final = context_fit or await self._request_context_fit(resolved_messages, request_kwargs) + if fit.accepts(response.model) and await self._model_group_can_serve( + response.model, messages, input, request_kwargs + ): return response eligible: Final = ( self._modality_eligible_models() if self.config.modality_routing and resolved_messages and request_contains_image_content(resolved_messages) else None ) - candidates: Final = tuple( - peer for peer in peers if peer != response.model and (eligible is None or peer in eligible) + pools: Final = self._tier_pools() + context_recovery: Final = bool(decision.get("context_escalated")) or any( + not fit.accepts(model) for model in pools.get(decided_tier, ()) ) - if not candidates: - return response - servable: Final = await asyncio.gather( - *(self._model_group_can_serve(peer, messages, input, request_kwargs) for peer in candidates) + modality_recovery: Final = eligible is not None + names: Final = self.config.tier_names() + tiers: Final = ( + tuple(names[names.index(decided_tier) :]) + if (context_recovery or modality_recovery) and decided_tier in names + else (decided_tier,) ) - live: Final = tuple(peer for peer, can_serve in zip(candidates, servable) if can_serve) - if not live: - return response - repick_messages: Final = ( - list(resolved_messages) if resolved_messages else None # mutable-ok: the pick's param is list-typed - ) - try: - new_model: Final = await self._pick_model_for_tier( - decided_tier if self.config.has_custom_tiers else ComplexityTier(decided_tier), - messages, - repick_messages, # pyright: ignore[reportArgumentType] # hook-resolved message dicts; the pick only reads them - request_kwargs, - allowed_models=live, + + async def recover_tier(candidate_tier: str) -> PreRoutingHookResponse | None: + peers: Final = tuple( + model + for model in pools.get(candidate_tier, ()) + if not context_recovery + or candidate_tier == decided_tier + or fit.needed is None + or _group_provably_fits(fit.facts.get(model, (None, True)), fit.needed, fit.buffer) ) - except ValueError as exc: - verbose_router_logger.debug( - "ComplexityRouter: health failover found no candidate the routing plugins allow: %s", exc + candidates: Final = tuple( + peer + for peer in peers + if peer != response.model and fit.accepts(peer) and (eligible is None or peer in eligible) ) + servable: Final = await asyncio.gather( + *(self._model_group_can_serve(peer, messages, input, request_kwargs) for peer in candidates) + ) + live: Final = tuple(peer for peer, can_serve in zip(candidates, servable) if can_serve) + if live: + repick_messages: Final = ( + list(resolved_messages) if resolved_messages else None # mutable-ok: the pick's param is list-typed + ) + try: + new_model: Final = await self._pick_model_for_tier( + candidate_tier if self.config.has_custom_tiers else ComplexityTier(candidate_tier), + messages, + repick_messages, # pyright: ignore[reportArgumentType] # hook-resolved message dicts; the pick only reads them + request_kwargs, + allowed_models=live, + ) + except ValueError as exc: + verbose_router_logger.debug( + "ComplexityRouter: health failover found no candidate the routing plugins allow: %s", exc + ) + else: + self._restamp_adaptive_choice(request_kwargs, response.model, new_model) + verbose_router_logger.info( + "ComplexityRouter: routing decision cause=health_failover, routed_model=%s, displaced=%s", + new_model, + response.model, + ) + new_decision: Final = self._build_routing_decision( + routed_model=new_model, + cause="health_failover", + tier=candidate_tier, + score=decision.get("score"), + signals=(*(decision.get("signals") or ()), f"health_displaced:{response.model}"), + matched_keyword=decision.get("matched_keyword"), + escalation_keyword=decision.get("escalation_keyword"), + escalated=bool(decision.get("escalated", False)), + classifier_model=decision.get("classifier_model"), + classifier_cost=decision.get("classifier_cost"), + conversation_continuing=bool(decision.get("conversation_continuing", True)), + tier_litellm_params=self._litellm_params_for_model(candidate_tier, new_model), + context_escalation_original_tier=decision.get("context_escalation_original_tier"), + ) + return response.model_copy( + update={ # mutable-ok: model_copy types update as a plain dict + "model": new_model, + "litellm_params": self._litellm_params_for_model(candidate_tier, new_model), + "routing_decision": new_decision, + } + ) + return None + + for candidate_tier in tiers: + if (recovered := await recover_tier(candidate_tier)) is not None: + return recovered + default_model: Final = self.config.default_model + plan_mode_active: Final = self._matched_plan_mode_signal(request_kwargs, resolved_messages) is not None + if ( + plan_mode_active + or self.config.plugins + or not default_model + or default_model == response.model + or not fit.accepts(default_model) + or (eligible is not None and default_model not in eligible) + or not await self._model_group_can_serve(default_model, messages, input, request_kwargs) + ): return response - self._restamp_adaptive_choice(request_kwargs, response.model, new_model) + self._restamp_adaptive_choice(request_kwargs, response.model, default_model) verbose_router_logger.info( - "ComplexityRouter: routing decision cause=health_failover, routed_model=%s, displaced=%s", - new_model, + "ComplexityRouter: routing decision cause=health_default_fallback, routed_model=%s, displaced=%s", + default_model, response.model, ) - new_decision: Final = self._build_routing_decision( - routed_model=new_model, - cause="health_failover", - tier=decision.get("tier"), + default_decision: Final = self._build_routing_decision( + routed_model=default_model, + cause="health_default_fallback", score=decision.get("score"), signals=(*(decision.get("signals") or ()), f"health_displaced:{response.model}"), matched_keyword=decision.get("matched_keyword"), @@ -3140,14 +3223,14 @@ class ComplexityRouter(CustomLogger): classifier_model=decision.get("classifier_model"), classifier_cost=decision.get("classifier_cost"), conversation_continuing=bool(decision.get("conversation_continuing", True)), - tier_litellm_params=self._litellm_params_for_model(decided_tier, new_model), + tier_litellm_params=self._litellm_params_for_model(None, default_model), context_escalation_original_tier=decision.get("context_escalation_original_tier"), ) return response.model_copy( update={ # mutable-ok: model_copy types update as a plain dict - "model": new_model, - "litellm_params": self._litellm_params_for_model(decided_tier, new_model), - "routing_decision": new_decision, + "model": default_model, + "litellm_params": self._litellm_params_for_model(None, default_model), + "routing_decision": default_decision, } ) @@ -3462,6 +3545,7 @@ class ComplexityRouter(CustomLogger): # chat-completions messages, so it is real work on every non-chat surface, and # both the conversation shape and the classifier read the same list. resolved_messages: Final = self._resolve_messages(messages, request_kwargs) + context_fit: Final = await self._request_context_fit(resolved_messages, request_kwargs) marker_pairs: Final = self._reminder_markers_for_request(request_kwargs) conversation_continuing: Final = _conversation_is_continuing(resolved_messages) @@ -3512,7 +3596,11 @@ class ComplexityRouter(CustomLogger): pin_source_tier: Final = self._tier_for_model(routed_model) pin_placement: Final = ( await self._context_window_placement( - pin_source_tier, resolved_messages, request_kwargs, pool_override=(routed_model,) + pin_source_tier, + resolved_messages, + request_kwargs, + pool_override=(routed_model,), + context_fit=context_fit, ) if pin_source_tier is not None else None @@ -3582,11 +3670,13 @@ class ComplexityRouter(CustomLogger): messages, resolved_messages, request_kwargs, + context_fit, ), messages, input, resolved_messages, request_kwargs, + context_fit, ) ) @@ -3598,14 +3688,18 @@ class ComplexityRouter(CustomLogger): specific_deployment=specific_deployment, conversation_continuing=conversation_continuing, resolved_messages=resolved_messages, + context_fit=context_fit, ) response: Final = ( await self._gate_response_health( - await self._gate_response_modality(routed_response, messages, resolved_messages, request_kwargs), + await self._gate_response_modality( + routed_response, messages, resolved_messages, request_kwargs, context_fit + ), messages, input, resolved_messages, request_kwargs, + context_fit, ) if routed_response is not None else None @@ -3640,6 +3734,7 @@ class ComplexityRouter(CustomLogger): specific_deployment: bool | None = False, conversation_continuing: bool = True, resolved_messages: Sequence[Mapping[str, object]] | None = None, + context_fit: _RequestContextFit | None = None, ) -> PreRoutingHookResponse | None: """ Classifies the request by complexity and returns the appropriate model. @@ -3811,7 +3906,9 @@ class ComplexityRouter(CustomLogger): plan_floored: Final = tier != pre_floor_tier if plan_floored: signals = (*signals, "plan_mode_floor") - context_placement: Final = await self._context_window_placement(tier, resolved_messages, request_kwargs) + context_placement: Final = await self._context_window_placement( + tier, resolved_messages, request_kwargs, context_fit=context_fit + ) tier, signals, context_original_tier = _apply_context_placement(tier, signals, context_placement) score_repr: Final = f"{score:.3f}" if score is not None else "n/a" fallback_model: Final = self.config.default_model if not self.config.plugins else None diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 58b940227f8..e3ea37dc0c8 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2920,6 +2920,7 @@ RoutingDecisionCause = Literal[ # same tier served instead. The displaced group rides in signals. Reported even on a kept # session pin, since the pinned model did not serve the request. "health_failover", + "health_default_fallback", "session_affinity_pin", "session_affinity_escalation", # classification_mode 'user_turn': the request is an agent loop's continuation turn (no new diff --git a/tests/test_litellm/router_strategy/adaptive_router/test_adaptive_router.py b/tests/test_litellm/router_strategy/adaptive_router/test_adaptive_router.py index f36443db1e5..d717c4e8c89 100644 --- a/tests/test_litellm/router_strategy/adaptive_router/test_adaptive_router.py +++ b/tests/test_litellm/router_strategy/adaptive_router/test_adaptive_router.py @@ -236,6 +236,38 @@ async def test_record_turn_attributes_satisfaction_to_previous_response_model(): assert smart_after.alpha == pytest.approx(smart_before.alpha) +@pytest.mark.asyncio +async def test_external_default_keeps_feedback_history_without_entering_bandit_pool(): + r = _make_router() + before = r._cells[(RequestType.GENERAL, "fast")] + await r.record_turn( + session_id="fallback", + model_name="fast", + request_type=RequestType.GENERAL, + turn=Turn(user_content="fix this retry bug", assistant_content="clear the cache"), + ) + await r.record_turn( + session_id="fallback", + model_name="external-default", + request_type=RequestType.GENERAL, + turn=Turn(user_content="the fix is still broken", assistant_content="keep cache entries"), + ) + assert r._cells[(RequestType.GENERAL, "fast")].beta > before.beta + await r.record_turn( + session_id="fallback", + model_name="smart", + request_type=RequestType.GENERAL, + turn=Turn( + user_content="the fix is still broken", + assistant_content="use the corrected entry", + tool_results=[{"is_error": True, "content": "failure"}], + ), + ) + assert r._feedback_contexts["fallback"].model_name == "smart" + assert all(model != "external-default" for _, model in r._cells) + assert r.config.available_models == ["fast", "smart"] + + @pytest.mark.asyncio async def test_record_turn_bounds_feedback_contexts_and_evicts_least_recent_session(): r = _make_router() diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index dddaee71a63..dba44d1e2e8 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -5,18 +5,19 @@ Tests the rule-based complexity scoring and tier assignment logic. """ import asyncio -from collections.abc import AsyncIterator import json -from copy import deepcopy -from functools import partial import logging import sys import time -from typing import Dict, Final, List +from collections.abc import AsyncIterator, Mapping +from copy import deepcopy +from functools import partial +from typing import Dict, Final, List, Literal from unittest.mock import AsyncMock, MagicMock, patch -import pytest import httpx +import pytest +import respx from pydantic import ValidationError import litellm @@ -69,6 +70,7 @@ from litellm.router_strategy.complexity_router.tier_predictor import ( from litellm.types.router import ( Deployment, LiteLLM_Params, + RouterErrors, TaggedPreRoutingStrategy, ) from litellm.types.llms.openai import ResponsesAPIResponse @@ -2671,7 +2673,8 @@ class TestEncryptedTaskClassifier: assert call["metadata"]["user_api_key_hash"] == "caller-key-hash" assert call["proxy_server_request"]["body"]["input"] == call["input"] assert call["proxy_server_request"]["originating_request_masked"] == { - "input": [task], "metadata": {"authorization": "REDACTED"}, + "input": [task], + "metadata": {"authorization": "REDACTED"}, } assert "source-secret" not in json.dumps(call) assert "originating_request_masked" not in call["proxy_server_request"]["body"] @@ -3315,19 +3318,23 @@ class TestLLMClassifier: ] @pytest.mark.asyncio - @pytest.mark.parametrize("source_body", [ - {"model": "router", "messages": [{"role": "user", "content": "source-only"}]}, - {"model": "router", "system": "source-only", "messages": [{"role": "user", "content": "ask"}]}, - {"model": "router", "instructions": "source-only", "input": "ask"}, - ]) + @pytest.mark.parametrize( + "source_body", + [ + {"model": "router", "messages": [{"role": "user", "content": "source-only"}]}, + {"model": "router", "system": "source-only", "messages": [{"role": "user", "content": "ask"}]}, + {"model": "router", "instructions": "source-only", "input": "ask"}, + ], + ) async def test_classifier_source_is_masked_and_separate_from_provider_input( self, llm_complexity_router, mock_router_instance, source_body ): mock_router_instance.acompletion = AsyncMock(return_value=_llm_response('{"tier": "SIMPLE"}')) outcome = await llm_complexity_router.aclassify( - "classify-this-ask", request_kwargs={"proxy_server_request": { - "body": {**source_body, "metadata": {"authorization": "source-secret"}} - }} + "classify-this-ask", + request_kwargs={ + "proxy_server_request": {"body": {**source_body, "metadata": {"authorization": "source-secret"}}} + }, ) assert outcome.cause == "llm_classifier" call_kwargs = mock_router_instance.acompletion.call_args.kwargs @@ -8138,7 +8145,9 @@ class TestContextAwareClassifier: ), ), ) - def test_only_text_reminder_tails_are_ignored_for_new_asks(self, tail: list[dict[str, object]], expected: bool) -> None: + def test_only_text_reminder_tails_are_ignored_for_new_asks( + self, tail: list[dict[str, object]], expected: bool + ) -> None: from litellm.router_strategy.complexity_router.complexity_router import ( _CODEX_REMINDER_MARKERS, _newest_turn_is_human_ask, @@ -13222,6 +13231,528 @@ class TestModalityRouting: assert cache.async_set_cache.await_args.kwargs["value"] == {"model": "text-cheap", "tier": "SIMPLE"} +@pytest.mark.usefixtures("local_model_cost_map") +class TestHealthFallbackDispatch: + @pytest.fixture(autouse=True) + def httpx_transport(self, monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + + @staticmethod + def _router( + surface: str = "chat", + *, + peer: bool = False, + session: bool = False, + tagged: bool = False, + budgeted: bool = False, + config: Mapping[str, object] | None = None, + ) -> Router: + provider: Final = "anthropic/claude-sonnet-5" if surface == "messages" else "openai/gpt-5.6" + base_suffix: Final = "" if surface == "messages" else "/v1" + return Router( + model_list=[ + { + "model_name": "health-router", + "litellm_params": { + "model": "auto_router/complexity_router", + "complexity_router_default_model": (config or {}).get("default_model", "fallback"), + "complexity_router_config": { + "tiers": {"SIMPLE": ["primary", "peer"] if peer else "primary", "MEDIUM": "primary"}, + "session_affinity": session, + "deployment_affinity": False, + "max_tokens_from_tier_model": False, + **(config or {}), + }, + }, + }, + *[ + { + "model_name": name, + "litellm_params": { + "model": provider, + "api_key": "test-only", + "api_base": f"https://{name}.test{base_suffix}", + **({"tags": [name]} if tagged else {}), + **( + {"max_budget": 1.0, "budget_duration": "1d"} + if budgeted and name == "primary" + else {} + ), + }, + "model_info": {"id": f"{name}-id"}, + } + for name in ("primary", "peer", "fallback") + ], + ], + num_retries=0, + enable_health_check_routing=True, + enable_tag_filtering=tagged, + ) + + @staticmethod + def _unavailable(router: Router, model_id: str, source: Literal["health", "cooldown"]) -> None: + if source == "health": + router.health_state_cache.set_deployment_health_states( + {model_id: {"is_healthy": False, "timestamp": time.time()}} + ) + else: + router.cooldown_cache.add_deployment_to_cooldown( + model_id=model_id, + original_exception=RuntimeError("unavailable"), + exception_status=503, + cooldown_time=60, + ) + + @staticmethod + def _http_response(request: httpx.Request) -> httpx.Response: + body: Final = json.loads(request.content) + text: Final = request.url.host.split(".")[0] + payload: Final[Mapping[str, object]] + events: Final[tuple[Mapping[str, object], ...]] + if request.url.path.endswith("/responses"): + from litellm.responses.main import mock_responses_api_response + + payload = mock_responses_api_response(text).model_dump() + events = ( + {"type": "response.created", "response": {**payload, "status": "in_progress"}, "sequence_number": 0}, + { + "type": "response.output_text.delta", + "delta": text, + "item_id": "msg_test", + "output_index": 0, + "content_index": 0, + "sequence_number": 1, + }, + {"type": "response.completed", "response": payload, "sequence_number": 2}, + ) + elif request.url.path.endswith("/messages"): + payload = { + "id": "msg_test", + "type": "message", + "role": "assistant", + "model": body["model"], + "content": [{"type": "text", "text": text}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 1}, + } + events = ( + {"type": "message_start", "message": {**payload, "content": [], "stop_reason": None}}, + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": text}}, + {"type": "content_block_stop", "index": 0}, + {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 1}}, + {"type": "message_stop"}, + ) + else: + payload = { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1, + "model": body["model"], + "choices": [{"index": 0, "message": {"role": "assistant", "content": text}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 1, "total_tokens": 11}, + } + events = ( + { + **payload, + "object": "chat.completion.chunk", + "choices": [{"index": 0, "delta": {"content": text}, "finish_reason": None}], + }, + { + **payload, + "object": "chat.completion.chunk", + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + }, + ) + if not body.get("stream"): + return httpx.Response(200, json=payload) + wire: Final = "".join( + (f"event: {event['type']}\n" if "type" in event else "") + f"data: {json.dumps(event)}\n\n" + for event in events + ) + return httpx.Response( + 200, + text=wire + ("data: [DONE]\n\n" if "type" not in events[0] else ""), + headers={"content-type": "text/event-stream"}, + ) + + @staticmethod + async def _request(router: Router, surface: str, stream: bool, metadata: dict[str, object]) -> str: + if surface == "responses": + result = await router.aresponses( + model="health-router", input="Hello!", stream=stream, litellm_metadata=metadata + ) + elif surface == "messages": + result = await router.aanthropic_messages( + model="health-router", + messages=[{"role": "user", "content": "Hello!"}], + max_tokens=32, + stream=stream, + litellm_metadata=metadata, + ) + else: + result = await router.acompletion( + model="health-router", + messages=[{"role": "user", "content": "Hello!"}], + stream=stream, + metadata=metadata, + ) + if not stream: + payload = result if isinstance(result, dict) else result.model_dump() + if surface == "responses": + return payload["output"][0]["content"][0]["text"] + if surface == "messages": + return payload["content"][0]["text"] + return payload["choices"][0]["message"]["content"] + if surface == "messages": + wire: Final = b"".join([chunk async for chunk in result]).decode() + events = tuple(json.loads(line[6:]) for line in wire.splitlines() if line.startswith("data: ")) + assert events[-1]["type"] == "message_stop" + return "".join(c["delta"]["text"] for c in events if c["type"] == "content_block_delta") + chunks: Final = [chunk.model_dump() async for chunk in result] + if surface == "responses": + assert chunks[-1]["type"] == "response.completed" + return "".join(c["delta"] for c in chunks if c["type"] == "response.output_text.delta") + assert chunks[-1]["choices"][0]["finish_reason"] == "stop" + return "".join(c["choices"][0]["delta"].get("content") or "" for c in chunks if c["choices"]) + + @pytest.mark.asyncio + @pytest.mark.parametrize("surface", ["chat", "responses", "messages"]) + @pytest.mark.parametrize("stream", [False, True]) + @pytest.mark.parametrize("source", ["health", "cooldown"]) + async def test_public_call_falls_back_and_recovers( + self, surface: str, stream: bool, source: Literal["health", "cooldown"] + ) -> None: + router: Final = self._router(surface, session=True) + self._unavailable(router, "primary-id", source) + metadata: Final[dict[str, object]] = {"session_id": "outage"} + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(primary|peer|fallback)\.test$").mock(side_effect=self._http_response) + assert await self._request(router, surface, stream, metadata) == "fallback" + assert metadata["routing_decision"]["cause"] == "health_default_fallback" + assert "tier" not in metadata["routing_decision"] + assert "health_displaced:primary" in metadata["routing_decision"]["signals"] + assert [c.request.url.host for c in upstream.calls] == ["fallback.test"] + strategy: Final = router.complexity_routers["health-router"][0].strategy + key: Final = strategy._get_session_affinity_cache_key("outage", {}) + assert await router.cache.async_get_cache(key=key) is None + if source == "health": + router.health_state_cache.set_deployment_health_states( + {"primary-id": {"is_healthy": True, "timestamp": time.time()}} + ) + else: + router.cooldown_cache.cooldown_store.delete_cache( + router.cooldown_cache.get_cooldown_cache_key("primary-id") + ) + recovered: Final[dict[str, object]] = {"session_id": "outage"} + assert await self._request(router, surface, stream, recovered) == "primary" + assert recovered["routing_decision"]["routed_model"] == "primary" + assert [c.request.url.host for c in upstream.calls] == ["fallback.test", "primary.test"] + + @pytest.mark.asyncio + @pytest.mark.parametrize("source", ["health", "cooldown"]) + async def test_partial_group_then_peer_then_default(self, source: Literal["health", "cooldown"]) -> None: + router: Final = self._router(peer=True, session=True) + router.add_deployment( + Deployment( + model_name="primary", + litellm_params=LiteLLM_Params( + model="openai/gpt-5.6", api_key="test-only", api_base="https://primary.test/v1" + ), + model_info={"id": "primary-sibling-id"}, + ) + ) + strategy: Final = router.complexity_routers["health-router"][0].strategy + key: Final = strategy._get_session_affinity_cache_key("precedence", {}) + await router.cache.async_set_cache(key=key, value={"model": "primary", "tier": "SIMPLE"}, ttl=600) + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(primary|peer|fallback)\.test$").mock(side_effect=self._http_response) + for model_id, expected, cause in ( + ("primary-id", "primary", "session_affinity_pin"), + ("primary-sibling-id", "peer", "health_failover"), + ("peer-id", "fallback", "health_default_fallback"), + ): + self._unavailable(router, model_id, source) + metadata: Final[dict[str, object]] = {"session_id": "precedence"} + assert await self._request(router, "chat", False, metadata) == expected + assert metadata["routing_decision"]["cause"] == cause + assert await router.cache.async_get_cache(key=key) == {"model": "primary", "tier": "SIMPLE"} + assert [c.request.url.host for c in upstream.calls] == ["primary.test", "peer.test", "fallback.test"] + + @pytest.mark.asyncio + async def test_spent_deployment_budget_falls_back_to_the_default(self, monkeypatch: pytest.MonkeyPatch) -> None: + """A spent budget leaves the tier with nothing that may serve the request, and the budget + filter reports that as a bare ValueError instead of a typed router error. Reading it as + capacity skips the recovery and fails the request the recovery exists for.""" + + async def _no_sync(*args: object, **kwargs: object) -> None: + return None + + monkeypatch.setattr( + "litellm.router_strategy.budget_limiter.RouterBudgetLimiting.periodic_sync_in_memory_spend_with_redis", + _no_sync, + ) + monkeypatch.setattr(litellm, "callbacks", []) + router: Final = self._router(budgeted=True) + limiter: Final = router.router_budget_logger + assert limiter is not None, "a deployment max_budget must install the budget limiter" + await router.cache.async_set_cache(key="deployment_spend:primary-id:1d", value=2.0) + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(primary|fallback)\.test$").mock(side_effect=self._http_response) + metadata: Final[dict[str, object]] = {} + assert await self._request(router, "chat", False, metadata) == "fallback" + assert metadata["routing_decision"]["cause"] == "health_default_fallback" + assert [c.request.url.host for c in upstream.calls] == ["fallback.test"] + + @pytest.mark.asyncio + async def test_concurrent_tag_scopes_keep_fallbacks_request_local(self) -> None: + router: Final = self._router(tagged=True) + router.add_deployment( + Deployment( + model_name="fallback", + litellm_params=LiteLLM_Params( + model="openai/gpt-5.6", api_key="test-only", api_base="https://peer.test/v1", tags=["peer"] + ), + model_info={"id": "fallback-peer-id"}, + ) + ) + self._unavailable(router, "primary-id", "cooldown") + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(peer|fallback)\.test$").mock(side_effect=self._http_response) + scopes: Final = tuple({"tags": [name], "session_id": name} for name in ("peer", "fallback")) + results: Final = await asyncio.gather( + *(self._request(router, "chat", False, metadata) for metadata in scopes) + ) + assert results == ["peer", "fallback"] + assert [m["tags"] for m in scopes] == [["peer"], ["fallback"]] + assert [m["routing_decision"]["routed_model"] for m in scopes] == ["fallback", "fallback"] + assert sorted(c.request.url.host for c in upstream.calls) == ["fallback.test", "peer.test"] + + @pytest.mark.asyncio + async def test_probe_preserves_consumed_request_exclusions(self) -> None: + router: Final = self._router() + self._unavailable(router, "primary-id", "cooldown") + kwargs: Final = {"_excluded_deployment_ids": ["fallback-id"], "_target_order": 1} + strategy: Final = router.complexity_routers["health-router"][0].strategy + response: Final = await strategy.async_pre_routing_hook( + model="health-router", messages=[{"role": "user", "content": "Hello!"}], request_kwargs=kwargs + ) + assert response.model == "primary" + assert kwargs == {"_excluded_deployment_ids": ["fallback-id"], "_target_order": 1} + + @pytest.mark.asyncio + @pytest.mark.parametrize("default_state", ["cooldown", "unconfigured", "same-model"]) + async def test_unavailable_default_preserves_no_deployment_error(self, default_state: str) -> None: + from litellm.types.router import RouterRateLimitError + + router: Final = self._router(config={"default_model": "primary"} if default_state == "same-model" else None) + self._unavailable(router, "primary-id", "cooldown") + if default_state == "unconfigured": + router.delete_deployment(id="fallback-id") + elif default_state == "cooldown": + self._unavailable(router, "fallback-id", "cooldown") + with respx.mock(assert_all_mocked=True) as upstream: + with pytest.raises(RouterRateLimitError, match="No deployments available"): + await self._request(router, "chat", False, {}) + assert not upstream.calls + + @pytest.mark.asyncio + @pytest.mark.parametrize("plan_active", [False, True]) + async def test_plan_floor_outage_cannot_use_untiered_default(self, plan_active: bool) -> None: + from litellm.types.router import RouterRateLimitError + + router: Final = self._router( + config={"tiers": {"SIMPLE": "primary", "MEDIUM": "peer"}, "plan_mode_min_tier": "MEDIUM"} + ) + self._unavailable(router, "primary-id", "cooldown") + self._unavailable(router, "peer-id", "cooldown") + metadata: Final = {} + with respx.mock(assert_all_mocked=True, assert_all_called=False) as upstream: + upstream.post(host="fallback.test").mock(side_effect=self._http_response) + if plan_active: + with pytest.raises(RouterRateLimitError, match="No deployments available"): + await router.acompletion( + model="health-router", + messages=[ + {"role": "system", "content": "Plan mode is active"}, + {"role": "user", "content": "Hello!"}, + ], + metadata=metadata, + ) + assert not upstream.calls + assert metadata["routing_decision"]["routed_model"] == "peer" + assert metadata["routing_decision"]["tier"] == "MEDIUM" + else: + assert await self._request(router, "chat", False, metadata) == "fallback" + + @pytest.mark.asyncio + async def test_default_dispatch_drops_displaced_tier_params(self) -> None: + router: Final = self._router( + config={"tiers": {"SIMPLE": {"model_name": "primary", "litellm_params": {"max_tokens": 9}}}} + ) + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(primary|fallback)\.test$").mock(side_effect=self._http_response) + await router.acompletion( + model="health-router", messages=[{"role": "user", "content": "Hello!"}], max_tokens=32 + ) + assert json.loads(upstream.calls[-1].request.content)["max_completion_tokens"] == 9 + self._unavailable(router, "primary-id", "cooldown") + await router.acompletion( + model="health-router", messages=[{"role": "user", "content": "Hello!"}], max_tokens=32 + ) + assert json.loads(upstream.calls[-1].request.content)["max_completion_tokens"] == 32 + assert upstream.calls[-1].request.url.host == "fallback.test" + + @pytest.mark.asyncio + @pytest.mark.parametrize("source", ["health", "cooldown"]) + async def test_pinned_session_returns_to_primary_after_outage(self, source: Literal["health", "cooldown"]) -> None: + router: Final = self._router(session=True) + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(primary|fallback)\.test$").mock(side_effect=self._http_response) + assert await self._request(router, "chat", False, {"session_id": "pinned"}) == "primary" + self._unavailable(router, "primary-id", source) + outage: Final[dict[str, object]] = {"session_id": "pinned"} + assert await self._request(router, "chat", False, outage) == "fallback" + assert outage["routing_decision"]["cause"] == "health_default_fallback" + if source == "health": + router.health_state_cache.set_deployment_health_states( + {"primary-id": {"is_healthy": True, "timestamp": time.time()}} + ) + else: + router.cooldown_cache.cooldown_store.delete_cache( + router.cooldown_cache.get_cooldown_cache_key("primary-id") + ) + recovered: Final[dict[str, object]] = {"session_id": "pinned"} + assert await self._request(router, "chat", False, recovered) == "primary" + assert recovered["routing_decision"]["cause"] == "session_affinity_pin" + assert [c.request.url.host for c in upstream.calls] == ["primary.test", "fallback.test", "primary.test"] + + @pytest.mark.asyncio + async def test_policy_plugin_does_not_escape_to_live_default(self) -> None: + from litellm.types.router import RouterRateLimitError, RoutingContext + + class PrimaryOnly: + async def run(self, context: RoutingContext) -> RoutingContext: + context.candidate_models = [name for name in context.candidate_models if name == "primary"] + return context + + router: Final = self._router(peer=True, config={"plugins": [PrimaryOnly()]}) + self._unavailable(router, "primary-id", "cooldown") + with respx.mock(assert_all_mocked=True) as upstream: + with pytest.raises(RouterRateLimitError, match="No deployments available"): + await self._request(router, "chat", False, {}) + assert not upstream.calls + + @pytest.mark.asyncio + @pytest.mark.parametrize("live_tier", [True, False]) + @pytest.mark.parametrize("default_fits", [True, False]) + async def test_context_recovery_precedes_default_with_prechecks_off( + self, live_tier: bool, default_fits: bool + ) -> None: + from litellm.types.router import RouterRateLimitError + + router: Final = self._router(config={"tiers": {"SIMPLE": "primary", "MEDIUM": "peer", "COMPLEX": "large"}}) + router.add_deployment( + Deployment( + model_name="large", + litellm_params=LiteLLM_Params( + model="openai/gpt-5.6", api_key="test-only", api_base="https://large.test/v1" + ), + model_info={"id": "large-id", "max_input_tokens": 10000}, + ) + ) + for deployment in router.model_list: + deployment["model_info"]["max_input_tokens"] = ( + 10 + if deployment["model_name"] == "primary" + or (deployment["model_name"] == "fallback" and not default_fits) + else 10000 + ) + self._unavailable(router, "peer-id", "cooldown") + if not live_tier: + self._unavailable(router, "large-id", "cooldown") + assert router.enable_pre_call_checks is False + metadata: Final = {} + messages: Final = [{"role": "user", "content": "hello " * 100}] + with respx.mock(assert_all_mocked=True, assert_all_called=False) as upstream: + upstream.post(host__regex=r"^(large|fallback)\.test$").mock(side_effect=self._http_response) + if not live_tier and not default_fits: + with pytest.raises(RouterRateLimitError, match="No deployments available"): + await router.acompletion(model="health-router", messages=messages, metadata=metadata) + assert not upstream.calls + else: + result: Final = await router.acompletion(model="health-router", messages=messages, metadata=metadata) + expected: Final = "large" if live_tier else "fallback" + assert result.choices[0].message.content == expected + assert upstream.calls[-1].request.url.host == f"{expected}.test" + assert metadata["routing_decision"].get("tier") == ("COMPLEX" if live_tier else None) + + @pytest.mark.asyncio + @pytest.mark.parametrize("live_tier", [True, False]) + async def test_modality_recovery_precedes_default(self, live_tier: bool) -> None: + router: Final = self._router( + config={"modality_routing": True, "tiers": {"SIMPLE": "primary", "MEDIUM": "peer", "COMPLEX": "vision"}} + ) + router.add_deployment( + Deployment( + model_name="vision", + litellm_params=LiteLLM_Params( + model="openai/gpt-5.6", api_key="test-only", api_base="https://vision.test/v1" + ), + model_info={"id": "vision-id", "supports_vision": True}, + ) + ) + for deployment in router.model_list: + deployment["model_info"]["supports_vision"] = deployment["model_name"] != "primary" + self._unavailable(router, "peer-id", "cooldown") + if not live_tier: + self._unavailable(router, "vision-id", "cooldown") + with respx.mock(assert_all_mocked=True) as upstream: + upstream.post(host__regex=r"^(vision|fallback)\.test$").mock(side_effect=self._http_response) + result: Final = await router.acompletion( + model="health-router", + messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": "Hello!"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,aGk="}}, + ], + } + ], + ) + expected: Final = "vision" if live_tier else "fallback" + assert result.choices[0].message.content == expected + assert upstream.calls[-1].request.url.host == f"{expected}.test" + + @pytest.mark.asyncio + @pytest.mark.parametrize("default_fits", [True, False]) + async def test_modality_default_must_also_fit_context(self, default_fits: bool) -> None: + router: Final = self._router(config={"modality_routing": True, "tiers": {"SIMPLE": "primary"}}) + for deployment in router.model_list: + deployment["model_info"]["supports_vision"] = deployment["model_name"] == "fallback" + deployment["model_info"]["max_input_tokens"] = 10000 if default_fits else 10 + with respx.mock(assert_all_mocked=True, assert_all_called=False) as upstream: + upstream.post(host="fallback.test").mock(side_effect=self._http_response) + messages: Final = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "hello " * 100}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,aGk="}}, + ], + } + ] + if default_fits: + result: Final = await router.acompletion(model="health-router", messages=messages) + assert result.choices[0].message.content == "fallback" + else: + with pytest.raises(litellm.BadRequestError, match="modality_routing is enabled"): + await router.acompletion(model="health-router", messages=messages) + assert not upstream.calls + + class TestTierHealthFailover: """A tier whose decided model group is entirely in cooldown falls back to a live peer.""" @@ -13256,7 +13787,7 @@ class TestTierHealthFailover: probed_prompts = [] async def get_healthy_deployments( - model, request_kwargs, messages=None, input=None, parent_otel_span=None, **kwargs + model, request_kwargs, messages=None, input=None, parent_otel_span=None, health_check_probe=False ): probed_kwargs.append(request_kwargs) probed_prompts.append((messages, input)) @@ -13758,6 +14289,51 @@ class TestTierHealthFailover: for _, probed_input in router.litellm_router_instance.probed_prompts ), "the eligibility probe must forward `input` to the owner" + @pytest.mark.asyncio + @pytest.mark.parametrize( + "raised, expected", + [ + (ValueError(f"{RouterErrors.no_deployments_with_tag_routing.value}. Passed model=b"), {"live-c"}), + ( + ValueError(f"{RouterErrors.no_deployments_with_provider_budget_routing.value}: b over budget"), + {"live-c"}, + ), + (ValueError("cannot unpack non-sequence"), {"exhausted-b", "live-c"}), + ], + ) + async def test_a_marked_exhaustion_value_error_is_a_verdict_and_an_unmarked_one_is_not( + self, mock_router_instance, raised, expected + ): + """Budget and tag filters exhaust a group without a typed error, signalling it only by a + RouterErrors marker on a bare ValueError. Those are verdicts; any other ValueError is a + fault, and a fault must still read as capacity rather than silently rerouting.""" + router = self._router( + mock_router_instance, + { + "tiers": { + "SIMPLE": ["dead-a", "exhausted-b", "live-c"], + "MEDIUM": "mid", + "COMPLEX": "big", + "REASONING": "top", + }, + "session_affinity": True, + }, + {"dead-a": ["id-a1"], "exhausted-b": ["id-b1"], "live-c": ["id-c1"]}, + cooling=("id-a1",), + raises_for={"exhausted-b": raised}, + ) + key = router._get_session_affinity_cache_key("sess-exhausted", {}) + await router.litellm_router_instance.cache.async_set_cache( + key=key, value={"model": "dead-a", "tier": "SIMPLE"}, ttl=600 + ) + results = [ + await router.async_pre_routing_hook( + model="m", request_kwargs={"metadata": {"session_id": "sess-exhausted"}}, messages=self.SIMPLE_MESSAGE + ) + for _ in range(20) + ] + assert {r.model for r in results} == expected + @pytest.mark.asyncio async def test_a_group_the_router_has_no_deployment_for_is_not_a_failover_target(self, mock_router_instance): """The owner answers an unconfigured group with BadRequestError. Reading that as live diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index def67ccf88b..2a97e92396a 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -7381,6 +7381,63 @@ async def test_async_get_fully_unhealthy_model_names_marks_name_when_all_unhealt assert await router.async_get_fully_unhealthy_model_names() == {"gpt-4o"} +@pytest.mark.asyncio +@pytest.mark.parametrize("health_check_probe", [False, True]) +@pytest.mark.parametrize( + "state, health_routing, fails_policy, scoped, strict_ids", + [ + ("absent", True, False, False, ("dep-0", "dep-1")), + ("partial", True, False, False, ("dep-1",)), + ("all", True, False, False, ()), + ("stale", True, False, False, ("dep-0", "dep-1")), + ("all", False, False, False, ("dep-0", "dep-1")), + ("all", True, True, False, ("dep-0", "dep-1")), + ("all", True, True, True, ()), + ], +) +async def test_health_probe_preserves_normal_caller_policy( + health_check_probe: bool, + state: str, + health_routing: bool, + fails_policy: bool, + scoped: bool, + strict_ids: tuple[str, ...], +) -> None: + import time + from litellm.types.router import AllowedFailsPolicy, RouterRateLimitError + + router: Final = Router( + model_list=[ + { + "model_name": "health-group", + "litellm_params": {"model": "openai/gpt-5.6", "api_key": "test-only"}, + "model_info": {"id": model_id}, + } + for model_id in ("dep-0", "dep-1") + ], + enable_health_check_routing=health_routing, + allowed_fails_policy=AllowedFailsPolicy(ServiceUnavailableErrorAllowedFails=2) if fails_policy else None, + background_health_check_model_groups=["health-group"] if scoped else None, + ) + if state != "absent": + _seed_unhealthy_states( + router, + ("dep-0",) if state == "partial" else ("dep-0", "dep-1"), + time.time() - router.health_state_cache.staleness_threshold - 10 if state == "stale" else None, + ) + expected: Final = strict_ids if strict_ids or health_check_probe else ("dep-0", "dep-1") + if not expected: + with pytest.raises(RouterRateLimitError, match="No deployments available"): + await router.async_get_healthy_deployments(model="health-group", request_kwargs={}, health_check_probe=True) + else: + deployments: Final = await router.async_get_healthy_deployments( + model="health-group", request_kwargs={}, health_check_probe=health_check_probe + ) + assert {d["model_info"]["id"] for d in deployments} == set(expected) + assert await router.cooldown_cache.async_get_active_cooldowns(["dep-0", "dep-1"], parent_otel_span=None) == [] + + + @pytest.mark.asyncio async def test_async_get_fully_unhealthy_model_names_keeps_name_when_partial(): router = _router_with_two_deployments([False, False]) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 6fb0d70dc58..0664a9f2fdc 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -36329,7 +36329,7 @@ export interface components { * Cause * @enum {string} */ - cause?: "heuristic_scorer" | "heuristic_v2" | "reasoning_override" | "llm_classifier" | "heuristic_first_short_circuit" | "hybrid_short_circuit" | "classifier_plugin" | "classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "housekeeping" | "modality_escalation" | "modality_pin_override" | "health_failover" | "session_affinity_pin" | "session_affinity_escalation" | "user_turn_continuation" | "default_fallback" | "keyword" | "quality_tier" | "bandit"; + cause?: "heuristic_scorer" | "heuristic_v2" | "reasoning_override" | "llm_classifier" | "heuristic_first_short_circuit" | "hybrid_short_circuit" | "classifier_plugin" | "classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "housekeeping" | "modality_escalation" | "modality_pin_override" | "health_failover" | "health_default_fallback" | "session_affinity_pin" | "session_affinity_escalation" | "user_turn_continuation" | "default_fallback" | "keyword" | "quality_tier" | "bandit"; /** Classifier Cost */ classifier_cost?: number; /** Classifier Model */ From 3e23eae24896afa9de102d43ce04928ed7ecff6b Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:22:18 -0700 Subject: [PATCH 091/157] fix(proxy): keep call_type and request start time on failed-request spend logs (#40558) * fix(proxy): keep call_type and request start time on failed-request spend logs post_call_failure_hook pops litellm_logging_obj before the failure callbacks run, so the spend row built from request_data had a blank call_type and used datetime.now() as the start time. A guardrail-blocked MCP tool call therefore showed up in the Logs page as an LLM row with no call type and a 0s duration. Lift call_type and start_time off the logging object alongside the fields already lifted, and have the DB failure hook prefer the lifted start time. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): inject the spend writer into _ProxyDBLogger instead of patching a module global Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/constants.py | 1 + .../proxy/hooks/proxy_track_cost_callback.py | 28 ++++--- .../spend_tracking/spend_tracking_utils.py | 4 +- litellm/proxy/utils.py | 9 ++- .../test_spend_tracking_utils.py | 5 ++ tests/test_litellm/proxy/test_proxy_utils.py | 76 ++++++++++++++++++- 6 files changed, 107 insertions(+), 16 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 028c08a691e..6b984c2673c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -2002,6 +2002,7 @@ NON_INFERENCE_CALL_TYPES: Final[frozenset[str]] = frozenset( UNKNOWN_MODEL_SPEND_LOG_MODEL: Final[str] = "unknown-model" MAX_SPEND_LOG_MODEL_NAME_LENGTH: Final[int] = 256 +MCP_SPEND_LOG_MODEL_PREFIX: Final[str] = "MCP: " # PTU reservation rollup writes rows to LiteLLM_DailyTeamSpend with this # sentinel api_key so PTU flat cost stays distinguishable from real per-request diff --git a/litellm/proxy/hooks/proxy_track_cost_callback.py b/litellm/proxy/hooks/proxy_track_cost_callback.py index 00406ad436e..1e946cc2e23 100644 --- a/litellm/proxy/hooks/proxy_track_cost_callback.py +++ b/litellm/proxy/hooks/proxy_track_cost_callback.py @@ -1,6 +1,6 @@ import asyncio import traceback -from collections.abc import Sequence +from collections.abc import Callable, Sequence from datetime import datetime from typing import TYPE_CHECKING, Any, Final, cast @@ -23,6 +23,7 @@ from litellm.proxy.auth.auth_checks import ( ) from litellm.proxy.auth.route_checks import RouteChecks from litellm.proxy.db.db_spend_update_writer import ( + DBSpendUpdateWriter, debitable_model_access_groups, get_llm_router, ) @@ -81,6 +82,12 @@ _CAPTURED_IDENTITY_CALL_TYPES: Final[frozenset[str]] = frozenset( ) +def _proxy_spend_writer() -> DBSpendUpdateWriter: + from litellm.proxy.proxy_server import proxy_logging_obj + + return proxy_logging_obj.db_spend_update_writer + + class _ProxyDBLogger(CustomLogger): def __init__( self, @@ -88,9 +95,11 @@ class _ProxyDBLogger(CustomLogger): *, turn_off_message_logging: bool = False, message_logging: bool = True, + spend_writer: Callable[[], DBSpendUpdateWriter] = _proxy_spend_writer, ) -> None: super().__init__(turn_off_message_logging=turn_off_message_logging, message_logging=message_logging) self.spend_event_producer = spend_event_producer + self._spend_writer: Final = spend_writer async def async_log_success_event( self, kwargs: ObjectMapping, response_obj: object, start_time: datetime, end_time: datetime @@ -150,8 +159,6 @@ class _ProxyDBLogger(CustomLogger): ): return - from litellm.proxy.proxy_server import proxy_logging_obj - _metadata = dict( LiteLLMProxyRequestSetup.get_sanitized_user_information_from_key(user_api_key_dict=user_api_key_dict) ) @@ -227,13 +234,12 @@ class _ProxyDBLogger(CustomLogger): if request_data.get("litellm_trace_id") is None: request_data["litellm_trace_id"] = getattr(_litellm_logging_obj, "litellm_trace_id", None) - # Use the actual request start time from the logging object so that - # failed requests record the real duration instead of 0. - actual_start_time = datetime.now() - if _litellm_logging_obj is not None: - obj_start: Final = getattr(_litellm_logging_obj, "start_time", None) - if obj_start is not None: - actual_start_time = obj_start + lifted_start_time: Final = request_data.get("start_time") + actual_start_time: Final = ( + lifted_start_time + if isinstance(lifted_start_time, datetime) + else getattr(_litellm_logging_obj, "start_time", None) or datetime.now() + ) # A stream that broke mid-flight still billed the provider for the # chunks already delivered. ``post_call_failure_hook`` lifts that @@ -249,7 +255,7 @@ class _ProxyDBLogger(CustomLogger): existing_metadata.get("standard_logging_guardrail_information") ) - await proxy_logging_obj.db_spend_update_writer.update_database( + await self._spend_writer().update_database( token=user_api_key_dict.api_key, response_cost=recovered_response_cost, user_id=user_api_key_dict.user_id, diff --git a/litellm/proxy/spend_tracking/spend_tracking_utils.py b/litellm/proxy/spend_tracking/spend_tracking_utils.py index f0f38358cf0..01398d38687 100644 --- a/litellm/proxy/spend_tracking/spend_tracking_utils.py +++ b/litellm/proxy/spend_tracking/spend_tracking_utils.py @@ -19,6 +19,7 @@ from litellm.constants import ( LITTELM_CLI_SERVICE_ACCOUNT_NAME, LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME, MAX_SPEND_LOG_MODEL_NAME_LENGTH, + MCP_SPEND_LOG_MODEL_PREFIX, REDACTED_BY_LITELM_STRING, SESSION_ID_OMITTED_METADATA_KEY, UNKNOWN_MODEL_SPEND_LOG_MODEL, @@ -338,7 +339,8 @@ def _sl_attribution_fallback( def _looks_like_model_name(model: str) -> bool: - return len(model) <= MAX_SPEND_LOG_MODEL_NAME_LENGTH and not any(char.isspace() for char in model) + candidate: Final = model.removeprefix(MCP_SPEND_LOG_MODEL_PREFIX) + return len(candidate) <= MAX_SPEND_LOG_MODEL_NAME_LENGTH and not any(char.isspace() for char in candidate) def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogsPayload: diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 253494b02f4..8e6142090a4 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -877,9 +877,10 @@ _EMPTY_LIFT: Final = MappingProxyType({}) def _failure_fields_to_lift(request_data: Mapping[str, object]) -> Mapping[str, object]: """Failure-path callbacks run after ``litellm_logging_obj`` is popped from request_data (it is not serialisable), so the caller merges these fields - onto request_data first: the first-handoff instant for preprocessing - latency, recovered or estimated usage for token counts, and the standard - logging object for deployment attribution on failed-request spend logs.""" + onto request_data first: the request start and first-handoff instants for + duration and preprocessing latency, the call type, recovered or estimated + usage for token counts, and the standard logging object for deployment + attribution on failed-request spend logs.""" _logging_obj: Final = request_data.get("litellm_logging_obj") if _logging_obj is None: return _EMPTY_LIFT @@ -891,7 +892,9 @@ def _failure_fields_to_lift(request_data: Mapping[str, object]) -> Mapping[str, dispatched=_first_handoff is not None, ) _entries: Final = ( + ("start_time", _model_call_details.get("start_time")), ("first_api_call_start_time", _first_handoff), + ("call_type", _model_call_details.get("call_type")), ("combined_usage_object", None if _usage_to_lift is None else _usage_to_lift[0]), ("response_cost", None if _usage_to_lift is None else (_usage_to_lift[1] or 0.0)), ("standard_logging_object", _model_call_details.get("standard_logging_object")), diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py index df113197ec6..dc8a87a97ad 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py @@ -1020,6 +1020,11 @@ _OVERLONG_MODEL: Final = "m" * (MAX_SPEND_LOG_MODEL_NAME_LENGTH + 1) ), ("gpt-5.2", ValueError("provider timed out"), "gpt-5.2"), (_BEDROCK_INFERENCE_PROFILE_ARN, ValueError("provider timed out"), _BEDROCK_INFERENCE_PROFILE_ARN), + ( + "MCP: deepwiki-ask_question", + ValueError("Content blocked: keyword 'confidential' detected"), + "MCP: deepwiki-ask_question", + ), ], ) def test_get_logging_payload_replaces_rejected_or_prompt_shaped_models_with_the_placeholder( diff --git a/tests/test_litellm/proxy/test_proxy_utils.py b/tests/test_litellm/proxy/test_proxy_utils.py index 646596c5b88..9def21c0573 100644 --- a/tests/test_litellm/proxy/test_proxy_utils.py +++ b/tests/test_litellm/proxy/test_proxy_utils.py @@ -581,6 +581,75 @@ class TestPostCallFailureHookLiftsStandardLoggingObject: assert "standard_logging_object" not in request_data +class TestPostCallFailureHookLiftsCallTypeAndStartTime: + """A guardrail-blocked MCP tool call fails before any LLM call. The failure + spend row is built from request_data after ``litellm_logging_obj`` is popped, + so ``call_type`` and the request ``start_time`` must be lifted off the logging + object first, or the Logs page shows the row as an LLM call with a blank call + type and a 0s duration (LIT-7453). + """ + + @pytest.mark.asyncio + async def test_failed_mcp_tool_call_spend_row_keeps_call_type_model_and_duration(self): + import traceback + from types import SimpleNamespace + from unittest.mock import AsyncMock + + from litellm.litellm_core_utils.litellm_logging import Logging + from litellm.proxy.hooks.proxy_track_cost_callback import _ProxyDBLogger + from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload + + request_start = real_datetime.datetime.now() - real_datetime.timedelta(seconds=2) + logging_obj = Logging( + model="MCP: deepwiki-ask_question", + messages=[], + stream=False, + call_type="call_mcp_tool", + start_time=request_start, + litellm_call_id="call-1", + function_id="fn-1", + ) + logging_obj.update_environment_variables( + model="MCP: deepwiki-ask_question", + user="", + optional_params={}, + litellm_params={"metadata": {"user_api_key_hash": "hashed"}}, + ) + blocked = Exception("Content blocked: keyword 'confidential' detected") + logging_obj.failure_handler(blocked, traceback.format_exc(), request_start, real_datetime.datetime.now()) + request_data = { + "name": "deepwiki-ask_question", + "arguments": {"question": "confidential"}, + "litellm_logging_obj": logging_obj, + } + proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache()) + proxy_logging_obj.alert_types = [] + spend_writer = SimpleNamespace(update_database=AsyncMock()) + original_callbacks = list(litellm.callbacks) + litellm.callbacks = [_ProxyDBLogger(spend_writer=lambda: spend_writer)] + try: + with patch.object(proxy_logging_obj, "update_request_status", new=AsyncMock()): + await proxy_logging_obj.post_call_failure_hook( + request_data=request_data, + original_exception=blocked, + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test"), + ) + finally: + litellm.callbacks = original_callbacks + ProxyLogging._callback_capabilities_cache.clear() + + db_call = spend_writer.update_database.call_args.kwargs + payload = get_logging_payload( + kwargs=db_call["kwargs"], + response_obj=db_call["completion_response"], + start_time=db_call["start_time"], + end_time=db_call["end_time"], + ) + assert payload["call_type"] == "call_mcp_tool" + assert payload["model"] == "MCP: deepwiki-ask_question" + assert payload["endTime"] - payload["startTime"] >= real_datetime.timedelta(seconds=2) + + class TestPostCallFailureHookEstimatesDispatchedInputTokens: """A non-stream request that failed after dispatch (timeout, provider error) consumed provider-billed input tokens but recovered no usage. @@ -1848,13 +1917,14 @@ def test_a_failure_with_no_logging_object_lifts_nothing(): assert dict(_failure_fields_to_lift({"litellm_logging_obj": _LoggingObj({})})) == {} -def test_a_dispatched_failure_lifts_the_four_fields_the_spend_log_needs(): +def test_a_dispatched_failure_lifts_the_fields_the_spend_log_needs(): from litellm.proxy.utils import _failure_fields_to_lift lifted = _failure_fields_to_lift( { "litellm_logging_obj": _LoggingObj( { + "start_time": 1699999999.0, "first_api_call_start_time": 1700000000.0, "call_type": "acompletion", "model": FAILURE_USAGE_MODEL, @@ -1866,12 +1936,16 @@ def test_a_dispatched_failure_lifts_the_four_fields_the_spend_log_needs(): ) assert set(lifted) == { + "start_time", "first_api_call_start_time", + "call_type", "combined_usage_object", "response_cost", "standard_logging_object", } + assert lifted["start_time"] == 1699999999.0 assert lifted["first_api_call_start_time"] == 1700000000.0 + assert lifted["call_type"] == "acompletion" assert lifted["response_cost"] == 0.0 assert lifted["combined_usage_object"].prompt_tokens > 0 assert lifted["standard_logging_object"] == {"id": "log-1"} From 1270ecb781ad0b3d5de25250807e26046559a8b9 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 12:28:02 -0700 Subject: [PATCH 092/157] test(load): drive /v1/messages alongside /chat/completions in the Redis chaos test The Anthropic Messages route reaches the same Redis touchpoints and cost-tracking callback through its own request path, so a failure-path regression there would not surface from chat completions alone. Each simulated user now picks one endpoint round robin and stays on it, and the per-endpoint split is asserted and reported so a run that silently drove only one route fails instead of passing. Co-Authored-By: Claude Code --- tests/e2e/CLAUDE.md | 2 +- tests/e2e/coverage_registry/reliability.yaml | 2 +- tests/e2e/load/locust_load.py | 57 +++++++++++++++++--- tests/e2e/load/locustfile.py | 23 ++++++-- tests/e2e/load/test_locust_load.py | 47 ++++++++++++++++ tests/e2e/load/test_redis_chaos_e2e.py | 21 ++++++-- 6 files changed, 135 insertions(+), 17 deletions(-) diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 24bad9f0d0d..80e78dd4cdd 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests and budgeting p50/p90/p99 latency, RSS, and CPU-per-request as ratios against the same run's healthy phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint and budgeting p50/p90/p99 latency, RSS, and CPU-per-request as ratios against the same run's healthy phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index 23163ab666c..cfbedf3d236 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,7 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "Under locust load with every request retrying through failing mock deployments, pausing Redis writes mid-run trips the breaker and every request still succeeds, with latency, RSS, and CPU reported as p50/p90/p99 against the pre-pause baseline; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, messages], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "Under locust load split round robin over /chat/completions and /v1/messages with every request retrying through failing mock deployments, holding Redis in CLIENT PAUSE ALL for the phase trips the breaker and every request still succeeds, with latency, RSS, and CPU reported as p50/p90/p99 against the pre-pause baseline; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/load/locust_load.py b/tests/e2e/load/locust_load.py index a1668c5bfc5..40f9f333db5 100644 --- a/tests/e2e/load/locust_load.py +++ b/tests/e2e/load/locust_load.py @@ -5,6 +5,7 @@ import os import subprocess import sys import tempfile +from collections.abc import Sequence from dataclasses import dataclass from itertools import accumulate from pathlib import Path @@ -19,6 +20,7 @@ _MAX_REPORTED_ERRORS = 5 class LocustStatEntry(BaseModel): + name: str num_requests: int num_failures: int start_time: float @@ -36,6 +38,16 @@ class LoadError: occurrences: int +@dataclass(frozen=True, slots=True) +class EndpointLoad: + """One route's share of a phase, so a run that silently drove only one of them is visible.""" + + name: str + requests: int + failures: int + p50_seconds: float + + @dataclass(frozen=True, slots=True) class LoadResult: requests: int @@ -44,6 +56,7 @@ class LoadResult: p50_seconds: float p90_seconds: float p99_seconds: float + endpoints: tuple[EndpointLoad, ...] errors: tuple[LoadError, ...] generator_warnings: tuple[str, ...] @@ -65,8 +78,15 @@ class LoadResult: def latency_summary(self) -> str: return f"p50 {self.p50_seconds:.3f}s, p90 {self.p90_seconds:.3f}s, p99 {self.p99_seconds:.3f}s" + def endpoint_summary(self) -> str: + return ", ".join( + f"{endpoint.name} {endpoint.requests} requests, {endpoint.failures} failures, " + f"p50 {endpoint.p50_seconds:.3f}s" + for endpoint in self.endpoints + ) -def percentile_seconds(entries: list[LocustStatEntry], fraction: float) -> float: + +def percentile_seconds(entries: Sequence[LocustStatEntry], fraction: float) -> float: """The response time at `fraction` of the merged histograms, in seconds. Locust buckets response times by millisecond, so this reads the first bucket whose @@ -82,13 +102,29 @@ def percentile_seconds(entries: list[LocustStatEntry], fraction: float) -> float return next(milliseconds for (milliseconds, _), seen in zip(samples, running) if seen >= rank) / 1000.0 +def per_endpoint(entries: Sequence[LocustStatEntry]) -> tuple[EndpointLoad, ...]: + """Each locust request name's own totals, in the order the names first appear.""" + names: Final = tuple(dict.fromkeys(entry.name for entry in entries)) + grouped: Final = ((name, tuple(entry for entry in entries if entry.name == name)) for name in names) + return tuple( + EndpointLoad( + name=name, + requests=sum(entry.num_requests for entry in group), + failures=sum(entry.num_failures for entry in group), + p50_seconds=percentile_seconds(group, 0.5), + ) + for name, group in grouped + ) + + def aggregate_stats( - entries: list[LocustStatEntry], + entries: Sequence[LocustStatEntry], errors: tuple[LoadError, ...], generator_warnings: tuple[str, ...], ) -> LoadResult: requests = sum(entry.num_requests for entry in entries) failures = sum(entry.num_failures for entry in entries) + endpoints = per_endpoint(entries) if not entries or requests == 0: return LoadResult( requests=requests, @@ -97,6 +133,7 @@ def aggregate_stats( p50_seconds=0.0, p90_seconds=0.0, p99_seconds=0.0, + endpoints=endpoints, errors=errors, generator_warnings=generator_warnings, ) @@ -108,6 +145,7 @@ def aggregate_stats( p50_seconds=percentile_seconds(entries, 0.5), p90_seconds=percentile_seconds(entries, 0.9), p99_seconds=percentile_seconds(entries, 0.99), + endpoints=endpoints, errors=errors, generator_warnings=generator_warnings, ) @@ -142,19 +180,21 @@ def read_generator_warnings(stderr: str) -> tuple[str, ...]: return tuple(dict.fromkeys(saturated)) -def run_chat_load( +def run_gateway_load( *, base_url: str, api_keys: tuple[str, ...], model: str, + endpoints: tuple[str, ...], users: int, spawn_rate: float, duration_seconds: float, ) -> LoadResult: - """Drive /chat/completions from headless locust and aggregate what it reported. + """Drive `endpoints` from headless locust and aggregate what it reported. Each simulated user picks one of `api_keys`, so auth and budget lookups spread over a - pool of virtual keys instead of keeping one key's cache entry permanently warm. + pool of virtual keys instead of keeping one key's cache entry permanently warm, and one + of `endpoints` round robin, so the run covers every route the caller asked for. """ with tempfile.TemporaryDirectory(prefix="e2e-load-") as report_dir: csv_prefix = Path(report_dir) / _CSV_PREFIX @@ -180,7 +220,12 @@ def run_chat_load( "--exit-code-on-error", "0", ], - env={**os.environ, "LOAD_API_KEYS": ",".join(api_keys), "LOAD_MODEL": model}, + env={ + **os.environ, + "LOAD_API_KEYS": ",".join(api_keys), + "LOAD_MODEL": model, + "LOAD_ENDPOINTS": ",".join(endpoints), + }, capture_output=True, text=True, timeout=duration_seconds + 120, diff --git a/tests/e2e/load/locustfile.py b/tests/e2e/load/locustfile.py index 3d9ef7c83ea..a4e396aa8d7 100644 --- a/tests/e2e/load/locustfile.py +++ b/tests/e2e/load/locustfile.py @@ -3,16 +3,22 @@ from __future__ import annotations import os import random import uuid +from itertools import cycle from typing import Final from locust import FastHttpUser, constant, task _MODEL: Final = os.environ["LOAD_MODEL"] _API_KEYS: Final = tuple(os.environ["LOAD_API_KEYS"].split(",")) +_NEXT_ENDPOINT: Final = cycle(os.environ["LOAD_ENDPOINTS"].split(",")) def _payload() -> dict[str, object]: - """A prompt no other request sent, so the response cache never answers for the deployment.""" + """A prompt no other request sent, so the response cache never answers for the deployment. + + Both endpoints take the same body: /v1/messages requires max_tokens, which /chat/completions + also accepts, so one payload serves the whole round robin. + """ return { "model": _MODEL, "messages": [{"role": "user", "content": f"load test ping {uuid.uuid4().hex}"}], @@ -20,17 +26,24 @@ def _payload() -> dict[str, object]: } -class ChatUser(FastHttpUser): +class GatewayUser(FastHttpUser): + """One simulated user, pinned to one endpoint for its lifetime. + + Endpoints are handed out round robin as users spawn, so a run spreads evenly over them + while each user's traffic stays on a single route, the way a real client behaves. + """ + wait_time = constant(0) def on_start(self) -> None: self.headers = {"Authorization": f"Bearer {random.choice(_API_KEYS)}"} + self.endpoint = next(_NEXT_ENDPOINT) @task - def chat(self) -> None: + def call(self) -> None: self.client.post( # pyright: ignore[reportUnknownMemberType] # locust FastHttpSession.post types json/**kwargs as Any - "/chat/completions", + self.endpoint, json=_payload(), headers=self.headers, - name="/chat/completions", + name=self.endpoint, ) diff --git a/tests/e2e/load/test_locust_load.py b/tests/e2e/load/test_locust_load.py index 40a238ef624..af3e1483099 100644 --- a/tests/e2e/load/test_locust_load.py +++ b/tests/e2e/load/test_locust_load.py @@ -1,6 +1,7 @@ from __future__ import annotations from pathlib import Path +from typing import Final from locust_load import ( LoadError, @@ -18,12 +19,14 @@ _FAILURES_HEADER = "Method,Name,Error,Occurrences,First Seen,Last Seen\n" def _entry( *, num_requests: int, + name: str = "/chat/completions", num_failures: int = 0, start_time: float = 1000.0, last_request_timestamp: float = 1010.0, response_times: dict[int, int] | None = None, ) -> LocustStatEntry: return LocustStatEntry( + name=name, num_requests=num_requests, num_failures=num_failures, start_time=start_time, @@ -44,6 +47,7 @@ def _result( p50_seconds=0.05, p90_seconds=0.08, p99_seconds=0.1, + endpoints=(), errors=errors, generator_warnings=generator_warnings, ) @@ -126,6 +130,49 @@ class TestAggregate: assert result.requests == 0 assert result.requests_per_second == 0.0 assert result.failure_ratio == 1.0 + assert result.endpoints == () + + +class TestPerEndpoint: + def test_each_route_keeps_its_own_requests_failures_and_median(self) -> None: + entries: Final = ( + _entry(name="/chat/completions", num_requests=100, response_times={20: 100}), + _entry(name="/v1/messages", num_requests=40, num_failures=3, response_times={900: 40}), + ) + + result: Final = aggregate_stats(entries, (), ()) + + assert tuple((one.name, one.requests, one.failures, one.p50_seconds) for one in result.endpoints) == ( + ("/chat/completions", 100, 0, 0.02), + ("/v1/messages", 40, 3, 0.9), + ) + + def test_several_stats_entries_for_one_route_fold_into_a_single_row(self) -> None: + entries: Final = ( + _entry(name="/v1/messages", num_requests=10, response_times={30: 10}), + _entry(name="/v1/messages", num_requests=30, num_failures=1, response_times={30: 30}), + ) + + result: Final = aggregate_stats(entries, (), ()) + + assert tuple((one.name, one.requests, one.failures) for one in result.endpoints) == (("/v1/messages", 40, 1),) + + def test_a_route_that_never_ran_is_absent_so_a_one_sided_run_cannot_pass_unnoticed(self) -> None: + result: Final = aggregate_stats((_entry(name="/chat/completions", num_requests=10),), (), ()) + + assert tuple(one.name for one in result.endpoints) == ("/chat/completions",) + + def test_the_summary_names_every_route_with_its_counts(self) -> None: + entries: Final = ( + _entry(name="/chat/completions", num_requests=2, response_times={20: 2}), + _entry(name="/v1/messages", num_requests=1, num_failures=1, response_times={500: 1}), + ) + + result: Final = aggregate_stats(entries, (), ()) + + assert result.endpoint_summary() == ( + "/chat/completions 2 requests, 0 failures, p50 0.020s, /v1/messages 1 requests, 1 failures, p50 0.500s" + ) class TestErrorBreakdown: diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 950712309dc..646429bbdd7 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -11,6 +11,11 @@ retries on the failing pair (a 500 is retryable, so retries keep re-picking insi order) and the router's order-based fallback then re-targets order 2. Every request is expected to succeed, and each one carries retry breadcrumbs into cost tracking. +Traffic is split round robin between /chat/completions and /v1/messages, one endpoint per +simulated user: the Redis touchpoints and the cost-tracking callback are shared by both, but +the Anthropic Messages route reaches them through its own request path, so a regression that +only shows up there would not surface from chat completions alone. + Phase A is a baseline with Redis healthy; phase B holds Redis in CLIENT PAUSE ALL for the length of the phase, simulating Redis being down outright rather than merely slow to write. Every touchpoint times out: the auth cache read falls back to Postgres, the response cache @@ -40,7 +45,7 @@ from e2e_config import PROXY_BASE_URL, unique_marker from e2e_http import NoBody from lifecycle import ResourceManager from load_client import LoadClient -from locust_load import LoadResult, run_chat_load +from locust_load import LoadResult, run_gateway_load from models import KeyGenerateBody, LiteLLMParamsBody from phase_budget import Budget, violations from proxy_client import ProxyClient @@ -55,6 +60,7 @@ SERVING_DEPLOYMENTS: Final = 1 FAILING_ORDER: Final = 1 SERVING_ORDER: Final = 2 KEY_POOL_SIZE: Final = 8 +LOAD_ENDPOINTS: Final = ("/chat/completions", "/v1/messages") LOCUST_USERS: Final = 50 LOCUST_SPAWN_RATE: Final = 50.0 BASELINE_SECONDS: Final = 60.0 @@ -119,7 +125,8 @@ class Phase: f"{self.name}: {self.load.requests} requests, {self.load.failures} failures, " f"{self.load.requests_per_second:.0f} rps, {self.load.latency_summary()}; {self.usage.summary()}; " f"{self.cpu_seconds_per_request * 1000:.1f} ms CPU per request; " - f"{self.timeouts_per_request:.2f} Redis timeouts per request" + f"{self.timeouts_per_request:.2f} Redis timeouts per request; " + f"by endpoint: {self.load.endpoint_summary()}" ) @@ -241,10 +248,11 @@ def _generate_key_pool(proxy: ProxyClient, resources: ResourceManager) -> tuple[ def _drive(keys: tuple[str, ...], seconds: float) -> LoadResult: - return run_chat_load( + return run_gateway_load( base_url=PROXY_BASE_URL, api_keys=keys, model=MODEL_GROUP, + endpoints=LOAD_ENDPOINTS, users=LOCUST_USERS, spawn_rate=LOCUST_SPAWN_RATE, duration_seconds=seconds, @@ -310,7 +318,7 @@ def _chaos_budgets(baseline: Phase, chaos: Phase) -> tuple[Budget, ...]: class TestRedisChaos: @pytest.mark.covers( "reliability.circuit_breaker.redis_timeout.stays_responsive", - exercised_on=("chat_completions",), + exercised_on=("chat_completions", "messages"), ) def test_load_survives_redis_being_down( self, @@ -356,6 +364,11 @@ class TestRedisChaos: assert phase.load.requests > 0, ( f"{phase.name} drove no traffic at all, so it proved nothing: {phase.load.diagnosis()}. {report}" ) + assert frozenset(endpoint.name for endpoint in phase.load.endpoints) == frozenset(LOAD_ENDPOINTS), ( + f"{phase.name} drove {tuple(endpoint.name for endpoint in phase.load.endpoints)} rather than every " + f"endpoint in {LOAD_ENDPOINTS}; the round robin hands one endpoint to each simulated user, so a " + f"missing one means a route never ran and its request path was never exercised. {report}" + ) assert phase.load.failures == 0, ( f"{phase.name} had {phase.load.failures} of {phase.load.requests} requests fail. Every request " f"must succeed: the failing deployments sit at order {FAILING_ORDER} and the serving one at order " From 89f1f9567d068c51ed4c2dfa8e7bbae7676a3684 Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 12:37:19 -0700 Subject: [PATCH 093/157] refactor(ocr): route native requests through core (#40532) * refactor(ocr): route native Mistral through core * fix(ocr): preserve Azure API base resolution * chore(ocr): document bridge boundary casts * fix(ocr): keep Azure environment resolution in Rust * fix(ocr): centralize native execution and isolate request logging * refactor(ocr): narrow native migration to bridge routing --------- Co-authored-by: Stack Plan --- .../crates/python-bridge/src/routes/ocr.rs | 16 ++ .../llms/reducto/test_parse_v3.py | 19 +- .../rust_bridge/native_route_wheel_test.py | 7 +- tests/test_litellm_rust/test_ocr.py | 177 ++++++++++++++++-- 4 files changed, 187 insertions(+), 32 deletions(-) diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index cc2f8e43cea..951caf4eef4 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -2,6 +2,7 @@ use litellm_core::Error; use std::future::Future; use litellm_ai_gateway::io::ocr::{OcrRequest, ocr as run_ocr}; +use litellm_core::ocr::wire::{OcrWireRequest, decode_request, is_supported_request}; use pyo3::prelude::*; use serde_json::Value; @@ -31,6 +32,21 @@ fn prepare_ocr( extra_headers, timeout, } = options; + if is_supported_request(&model, custom_llm_provider.as_deref()) { + let request = decode_request(OcrWireRequest { + model, + document, + api_key, + api_base, + custom_llm_provider, + extra_headers, + optional_params, + timeout_seconds: timeout.map(|value| value.as_secs_f64()), + })?; + return litellm_core::ocr::ocr(request) + .await + .map(|response| response.into_json()); + } run_ocr(OcrRequest { model: &model, document, diff --git a/tests/test_litellm/llms/reducto/test_parse_v3.py b/tests/test_litellm/llms/reducto/test_parse_v3.py index 140b9737dc0..bacd12db58a 100644 --- a/tests/test_litellm/llms/reducto/test_parse_v3.py +++ b/tests/test_litellm/llms/reducto/test_parse_v3.py @@ -1,8 +1,9 @@ import json -import litellm import pytest +import litellm + def _reducto_parse_response() -> dict: return { @@ -68,15 +69,11 @@ def disable_aiohttp_transport(): @pytest.mark.asyncio -async def test_parse_v3_file_upload_and_response_mapping( - disable_aiohttp_transport, respx_mock -): +async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transport, respx_mock): upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond( json={"file_id": "reducto://uploaded.pdf"} ) - parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond( - json=_reducto_parse_response() - ) + parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response()) response = await litellm.aocr( model="reducto/parse-v3", @@ -123,15 +120,11 @@ async def test_parse_v3_file_upload_and_response_mapping( @pytest.mark.asyncio -async def test_parse_v3_reducto_id_passthrough_skips_upload( - disable_aiohttp_transport, respx_mock -): +async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_transport, respx_mock): upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond( json={"file_id": "reducto://should-not-upload.pdf"} ) - parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond( - json=_reducto_parse_response() - ) + parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response()) response = await litellm.aocr( model="reducto/parse-v3", diff --git a/tests/test_litellm/rust_bridge/native_route_wheel_test.py b/tests/test_litellm/rust_bridge/native_route_wheel_test.py index a7f50a82a99..9807febbff4 100644 --- a/tests/test_litellm/rust_bridge/native_route_wheel_test.py +++ b/tests/test_litellm/rust_bridge/native_route_wheel_test.py @@ -227,12 +227,7 @@ async def exercise_async(native: object, api_base: str) -> None: async def exercise_async_concurrency(native: object, api_base: str) -> None: responses: Final = await asyncio.wait_for( - asyncio.gather( - *( - native.amessages(**route_kwargs("messages", api_base, "success")) - for _ in range(32) - ) - ), + asyncio.gather(*(native.amessages(**route_kwargs("messages", api_base, "success")) for _ in range(32))), timeout=15, ) for response in responses: diff --git a/tests/test_litellm_rust/test_ocr.py b/tests/test_litellm_rust/test_ocr.py index 1293de9ee0e..ad1c8c652bb 100644 --- a/tests/test_litellm_rust/test_ocr.py +++ b/tests/test_litellm_rust/test_ocr.py @@ -1,33 +1,48 @@ import json import threading from collections.abc import Generator -from dataclasses import dataclass from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from typing import Final import pytest +import litellm from litellm.rust_bridge import ocr as rust_ocr_bridge pytestmark = pytest.mark.requires_rust_extension -@dataclass(frozen=True, slots=True) -class RecordedOCRRequest: - body: object - - @pytest.fixture -def ocr_server() -> Generator[tuple[ThreadingHTTPServer, list[RecordedOCRRequest]]]: - requests: Final[list[RecordedOCRRequest]] = [] +def ocr_server() -> Generator[tuple[ThreadingHTTPServer, list[dict[str, object]]]]: + requests: Final[list[dict[str, object]]] = [] class Handler(BaseHTTPRequestHandler): def do_POST(self) -> None: requests.append( - RecordedOCRRequest( - body=json.loads(self.rfile.read(int(self.headers["Content-Length"]))), - ) + { + "headers": {name.lower(): value for name, value in self.headers.items()}, + "body": json.loads(self.rfile.read(int(self.headers["Content-Length"]))), + } ) + if self.headers.get("x-test-stall") == "true": + self.connection.settimeout(2) + try: + self.rfile.read(1) + except TimeoutError: + pass + return + if self.headers.get("User-Agent", "").startswith("python-httpx"): + self.send_response(418) + self.end_headers() + return + status = int(self.headers.get("x-test-status", "200")) + if status != 200: + body = b'{"error":"provider unavailable"}' + self.send_response(status) + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return response: Final = json.dumps( { "pages": [{"index": 0, "markdown": "native OCR response", "images": [], "dimensions": None}], @@ -56,7 +71,7 @@ def ocr_server() -> Generator[tuple[ThreadingHTTPServer, list[RecordedOCRRequest def test_native_ocr_with_compiled_rust_extension( - ocr_server: tuple[ThreadingHTTPServer, list[RecordedOCRRequest]], + ocr_server: tuple[ThreadingHTTPServer, list[dict[str, object]]], ) -> None: server, requests = ocr_server address: Final = server.server_address @@ -77,7 +92,143 @@ def test_native_ocr_with_compiled_rust_extension( assert response is not None assert response["pages"][0]["markdown"] == "native OCR response" assert len(requests) == 1 - assert requests[0].body == { + assert not requests[0]["headers"].get("user-agent", "").startswith("python-httpx") + assert requests[0]["body"] == { "model": "mistral-ocr-latest", "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, } + + +@pytest.mark.parametrize("asynchronous", [False, True]) +@pytest.mark.parametrize("model", ["mistral/mistral-ocr-latest", "azure_ai/doc-intelligence/prebuilt-read"]) +@pytest.mark.asyncio +async def test_native_public_ocr_matches_python(model, asynchronous): + import json + from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + from threading import Thread + from typing import Final + from urllib.parse import parse_qsl, urlsplit + + from litellm.rust_bridge import _native + + assert callable(_native.ocr) + calls: Final = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + body: Final = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + target: Final = urlsplit(self.path) + calls.append( + ( + target.path, + parse_qsl(target.query), + self.headers.get("Authorization"), + self.headers.get("Ocp-Apim-Subscription-Key"), + body, + ) + ) + payload: Final = ( + {"status": "succeeded", "analyzeResult": {"pages": []}} + if "doc-intelligence" in model + else {"pages": [{"index": 0, "markdown": "hello"}]} + ) + encoded: Final = json.dumps(payload).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(encoded))) + self.end_headers() + self.wfile.write(encoded) + + def log_message(self, *_args): + pass + + server: Final = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread: Final = Thread(target=server.serve_forever, daemon=True) + thread.start() + responses: Final = [] + try: + for enabled in (False, True): + litellm.rust(enabled) + arguments: Final = { + "model": model, + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "api_key": "test-key", + "api_base": f"http://127.0.0.1:{server.server_port}", + "pages": [0, 2], + "timeout": 3.0, + } + response: Final = await litellm.aocr(**arguments) if asynchronous else litellm.ocr(**arguments) + responses.append(response.model_dump()) + assert len(calls) == 2 + assert calls[0] == calls[1] + for key in ("model", "pages", "object"): + assert responses[0][key] == responses[1][key] + finally: + server.shutdown() + server.server_close() + thread.join(timeout=3) + + +@pytest.mark.parametrize("asynchronous", [False, True]) +@pytest.mark.asyncio +async def test_native_ocr_failures_do_not_retry_on_python(ocr_server, asynchronous): + server, requests = ocr_server + arguments = { + "model": "mistral-ocr-latest", + "custom_llm_provider": "mistral", + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "api_key": "test-key", + "api_base": f"http://127.0.0.1:{server.server_port}", + "extra_headers": {"x-test-status": "503"}, + "num_retries": 0, + } + litellm.rust(True) + with pytest.raises(litellm.ServiceUnavailableError) as caught: + await litellm.aocr(**arguments) if asynchronous else litellm.ocr(**arguments) + assert caught.value.status_code == 503 + assert len(requests) == 1 + assert not requests[0]["headers"].get("user-agent", "").startswith("python-httpx") + + +@pytest.mark.parametrize("custom_provider", ["mistral", "not-a-provider"]) +def test_native_ocr_rejects_invalid_input_before_network(ocr_server, custom_provider): + from litellm.rust_bridge import _native + + server, requests = ocr_server + with pytest.raises(ValueError, match=r"invalid (OCR request field|provider)|invalid request"): + _native.ocr( + model="mistral-ocr-latest", + custom_llm_provider=custom_provider, + document={"type": "document_url"}, + api_key="test-key", + api_base=f"http://127.0.0.1:{server.server_port}", + ) + assert requests == [] + + +@pytest.mark.parametrize("asynchronous", [False, True]) +@pytest.mark.asyncio +async def test_native_ocr_enforces_request_deadline_without_fallback(ocr_server, asynchronous): + import asyncio + import time + + server, requests = ocr_server + litellm.rust(True) + arguments = { + "model": "mistral/mistral-ocr-latest", + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "api_key": "test-key", + "api_base": f"http://127.0.0.1:{server.server_port}", + "extra_headers": {"x-test-stall": "true"}, + "timeout": 0.1, + "num_retries": 0, + } + started = time.monotonic() + with pytest.raises(litellm.APIConnectionError): + await asyncio.wait_for( + litellm.aocr(**arguments) if asynchronous else asyncio.to_thread(litellm.ocr, **arguments), + timeout=3, + ) + assert 0.09 <= time.monotonic() - started < 3 + assert len(requests) == 1 + assert not requests[0]["headers"].get("user-agent", "").startswith("python-httpx") From 4ffd4ecc831871768954975ce787a9424df84481 Mon Sep 17 00:00:00 2001 From: mateo Date: Fri, 11 Sep 2026 19:37:45 +0000 Subject: [PATCH 094/157] fix(model_prices): absorb cerebras/inception PRs, fix vertex/openai/together/openrouter pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 421 +++++++++++++++--- model_prices_and_context_window.json | 421 +++++++++++++++--- .../test_cerebras_chat_transformation.py | 21 + .../test_inception_chat_transformation.py | 22 + 4 files changed, 761 insertions(+), 124 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1c86dbaacff..a55cb6b571e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -13118,6 +13118,22 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "cerebras/qwen-3.8-27b": { + "input_cost_per_token": 9.9e-07, + "litellm_provider": "cerebras", + "max_input_tokens": 65536, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.49e-06, + "source": "https://api.cerebras.ai/public/v1/models/qwen-3.8-27b", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "chatdolphin": { "input_cost_per_token": 5e-07, "litellm_provider": "nlp_cloud", @@ -31328,8 +31344,6 @@ "cache_read_input_token_cost_above_272k_tokens_flex": 5e-07 }, "gpt-5.5-pro": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -31378,8 +31392,6 @@ "supports_low_reasoning_effort": false }, "gpt-5.5-pro-2026-04-23": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -31532,8 +31544,6 @@ "cache_read_input_token_cost_above_272k_tokens_flex": 2.5e-07 }, "gpt-5.4-pro": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -31583,8 +31593,6 @@ "output_cost_per_token_above_272k_tokens_flex": 0.000135 }, "gpt-5.4-pro-2026-03-05": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -33891,6 +33899,20 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "inception/mercury-2.5": { + "input_cost_per_token": 2e-07, + "litellm_provider": "inception", + "max_input_tokens": 260000, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 7.5e-07, + "source": "https://docs.inceptionlabs.ai/get-started/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "text-completion-inception/mercury-edit-2": { "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 2.5e-07, @@ -39127,19 +39149,21 @@ "supports_tool_choice": true }, "openrouter/deepseek/deepseek-chat-v3.1": { - "input_cost_per_token": 2e-07, + "input_cost_per_token": 2.5e-07, "input_cost_per_token_cache_hit": 2e-08, "litellm_provider": "openrouter", "max_input_tokens": 163840, "max_output_tokens": 163840, "max_tokens": 163840, "mode": "chat", - "output_cost_per_token": 8e-07, + "output_cost_per_token": 9.5e-07, "supports_assistant_prefill": true, "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 1.3e-07, + "source": "https://openrouter.ai/deepseek/deepseek-chat-v3.1" }, "openrouter/deepseek/deepseek-v3.2": { "input_cost_per_token": 2.69e-07, @@ -39202,36 +39226,56 @@ "supports_tool_choice": true }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 1.32e-06, + "input_cost_per_token": 8.59908e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 3.96e-06, + "output_cost_per_token": 1.719816e-06, "source": "https://openrouter.ai/deepseek/deepseek-v4-pro", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 7.1659e-08 + }, + "openrouter/deepseek/deepseek-v4.1-flash": { + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 3e-09, + "litellm_provider": "openrouter", + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "source": "https://openrouter.ai/deepseek/deepseek-v4.1-flash", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": false, + "supports_prompt_caching": true }, "openrouter/deepseek/deepseek-v4-pro-0813": { - "input_cost_per_token": 1.32e-06, + "input_cost_per_token": 5.7948e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 3.96e-06, + "output_cost_per_token": 1.73844e-06, "source": "https://openrouter.ai/deepseek/deepseek-v4-pro-0813", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 1.9316e-08 }, "openrouter/google/gemini-2.0-flash-001": { "deprecation_date": "2026-06-01", @@ -40010,6 +40054,28 @@ "supports_tool_choice": true, "supports_vision": true }, + "openrouter/openai/gpt-5.6-sol-pro": { + "input_cost_per_token": 2e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 2e-07, + "cache_creation_input_token_cost": 2.5e-06, + "input_cost_per_token_above_272k_tokens": 4e-06, + "output_cost_per_token_above_272k_tokens": 1.5e-05, + "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-5.6-sol-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/openai/gpt-oss-120b": { "input_cost_per_token": 3.7e-08, "litellm_provider": "openrouter", @@ -40138,13 +40204,13 @@ "supports_tool_choice": true }, "openrouter/qwen/qwen3-235b-a22b-2507": { - "input_cost_per_token": 8.75e-08, + "input_cost_per_token": 2.2e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", - "output_cost_per_token": 3.5e-07, + "output_cost_per_token": 8.8e-07, "source": "https://openrouter.ai/qwen/qwen3-235b-a22b-2507", "supports_function_calling": true, "supports_tool_choice": true @@ -40177,7 +40243,7 @@ "supports_vision": true }, "openrouter/qwen/qwen3.5-35b-a3b": { - "input_cost_per_token": 2.5e-07, + "input_cost_per_token": 3.125e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 65536, @@ -40188,7 +40254,8 @@ "supports_function_calling": true, "supports_reasoning": true, "supports_tool_choice": true, - "supports_vision": true + "supports_vision": true, + "cache_read_input_token_cost": 1.5625e-07 }, "openrouter/qwen/qwen3.5-27b": { "input_cost_per_token": 1.95e-07, @@ -40205,13 +40272,13 @@ "supports_vision": true }, "openrouter/qwen/qwen3.5-122b-a10b": { - "input_cost_per_token": 2.9e-07, + "input_cost_per_token": 2.6e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", - "output_cost_per_token": 2.4e-06, + "output_cost_per_token": 2.08e-06, "source": "https://openrouter.ai/qwen/qwen3.5-122b-a10b", "supports_function_calling": true, "supports_reasoning": true, @@ -40298,18 +40365,19 @@ "supports_web_search": true }, "openrouter/z-ai/glm-4.6": { - "input_cost_per_token": 5.5e-07, + "input_cost_per_token": 4.3e-07, "litellm_provider": "openrouter", "max_input_tokens": 202800, "max_output_tokens": 131000, "max_tokens": 131000, "mode": "chat", - "output_cost_per_token": 2.2e-06, + "output_cost_per_token": 1.75e-06, "source": "https://openrouter.ai/z-ai/glm-4.6", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 8e-08 }, "openrouter/z-ai/glm-4.6:exacto": { "input_cost_per_token": 4.5e-07, @@ -43134,7 +43202,11 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 1.2e-06, + "max_input_tokens": 131072, + "source": "https://api.together.xyz/v1/models" }, "together_ai/Qwen/Qwen2.5-7B-Instruct-Turbo": { "litellm_provider": "together_ai", @@ -43142,7 +43214,11 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 3e-07, + "max_input_tokens": 32768, + "source": "https://api.together.xyz/v1/models" }, "together_ai/Qwen/Qwen3-235B-A22B-Instruct-2507-tput": { "deprecation_date": "2026-07-10", @@ -43354,7 +43430,11 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "max_input_tokens": 32768, + "source": "https://api.together.xyz/v1/models" }, "together_ai/mistralai/Mistral-Small-24B-Instruct-2501": { "deprecation_date": "2026-04-02", @@ -43362,7 +43442,11 @@ "mode": "chat", "supports_function_calling": true, "supports_parallel_function_calling": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "max_input_tokens": 32768, + "source": "https://api.together.xyz/v1/models" }, "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1": { "deprecation_date": "2026-04-16", @@ -48093,27 +48177,29 @@ "supports_tool_choice": true }, "vertex_ai/mistral-small-2503": { - "input_cost_per_token": 1e-06, + "input_cost_per_token": 1e-07, "litellm_provider": "vertex_ai-mistral_models", "max_input_tokens": 128000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 3e-06, + "output_cost_per_token": 3e-07, "supports_function_calling": true, "supports_tool_choice": true, - "supports_vision": true + "supports_vision": true, + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/mistral-small-2503@001": { - "input_cost_per_token": 1e-06, + "input_cost_per_token": 1e-07, "litellm_provider": "vertex_ai-mistral_models", "max_input_tokens": 32000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_token": 3e-06, + "output_cost_per_token": 3e-07, "supports_function_calling": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/mistral-ocr-2505": { "litellm_provider": "vertex_ai", @@ -48163,15 +48249,16 @@ "supports_reasoning": true }, "vertex_ai/openai/gpt-oss-20b-maas": { - "input_cost_per_token": 7.5e-08, + "input_cost_per_token": 7e-08, "litellm_provider": "vertex_ai-openai_models", "max_input_tokens": 131072, "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", - "output_cost_per_token": 3e-07, - "source": "https://console.cloud.google.com/vertex-ai/publishers/openai/model-garden/gpt-oss-120b-maas", - "supports_reasoning": true + "output_cost_per_token": 2.5e-07, + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", + "supports_reasoning": true, + "cache_read_input_token_cost": 7e-09 }, "vertex_ai/xai/grok-4.1-fast-non-reasoning": { "cache_read_input_token_cost": 5e-08, @@ -60695,6 +60782,113 @@ "output_cost_per_token": 4.7e-07, "source": "https://docs.together.ai/docs/serverless-models" }, + "together_ai/moonshotai/Kimi-K2.6": { + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 4.5e-06, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/moonshotai/Kimi-K2.5-fp4": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.8e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/MiniMaxAI/MiniMax-M2.7": { + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 6e-08, + "litellm_provider": "together_ai", + "max_input_tokens": 196608, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/zai-org/GLM-5": { + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/zai-org/GLM-5.1": { + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "cache_read_input_token_cost": 2.6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-0528": { + "input_cost_per_token": 3e-06, + "output_cost_per_token": 7e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 163840, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/Qwen3-Coder-Next-FP8": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-32B-Instruct": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 1.5e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-8B-Instruct": { + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 6.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/mistralai/Ministral-3-14B-Instruct-2512": { + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/nvidia/NVIDIA-Nemotron-Nano-9B-v2": { + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/mistralai/Mistral-7B-Instruct-v0.3": { + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/QwQ-32B": { + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, "cerebras/gemma-4-31b": { "input_cost_per_token": 9.9e-07, "litellm_provider": "cerebras", @@ -62348,6 +62542,28 @@ "cache_read_input_token_cost": 2e-08, "supports_prompt_caching": true }, + "openrouter/openai/gpt-5.6-luna-pro": { + "input_cost_per_token": 2e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 2e-08, + "cache_creation_input_token_cost": 2.5e-07, + "input_cost_per_token_above_272k_tokens": 4e-07, + "output_cost_per_token_above_272k_tokens": 1.8e-06, + "cache_read_input_token_cost_above_272k_tokens": 4e-08, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-5.6-luna-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/openai/gpt-5.6-terra": { "input_cost_per_token": 2e-06, "output_cost_per_token": 1.2e-05, @@ -62367,6 +62583,28 @@ "cache_read_input_token_cost": 2e-07, "supports_prompt_caching": true }, + "openrouter/openai/gpt-5.6-terra-pro": { + "input_cost_per_token": 2e-06, + "output_cost_per_token": 1.2e-05, + "cache_read_input_token_cost": 2e-07, + "cache_creation_input_token_cost": 2.5e-06, + "input_cost_per_token_above_272k_tokens": 4e-06, + "output_cost_per_token_above_272k_tokens": 1.8e-05, + "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-5.6-terra-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/openai/o3": { "input_cost_per_token": 2e-06, "output_cost_per_token": 8e-06, @@ -62600,6 +62838,28 @@ "supports_pdf_input": true, "supports_prompt_caching": true }, + "openrouter/openai/gpt-6-astra-pro": { + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "cache_read_input_token_cost": 1e-06, + "cache_creation_input_token_cost": 1.25e-05, + "input_cost_per_token_above_272k_tokens": 2e-05, + "output_cost_per_token_above_272k_tokens": 7.5e-05, + "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-6-astra-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/qwen/qwen3.8-flash": { "input_cost_per_token": 1.5e-07, "output_cost_per_token": 4.7e-07, @@ -62619,9 +62879,9 @@ "supports_prompt_caching": true }, "openrouter/z-ai/glm-5.3-flash": { - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 2.5e-07, - "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, "litellm_provider": "openrouter", "max_input_tokens": 1310720, "max_output_tokens": 131072, @@ -62736,6 +62996,25 @@ "supports_vision": true, "supports_prompt_caching": true }, + "openrouter/qwen/qwen3.8-max-0902": { + "input_cost_per_token": 2e-06, + "output_cost_per_token": 6e-06, + "cache_read_input_token_cost": 2.5e-07, + "cache_creation_input_token_cost": 2.5e-06, + "litellm_provider": "openrouter", + "max_input_tokens": 1000000, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "source": "https://openrouter.ai/qwen/qwen3.8-max-0902", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": false, + "supports_prompt_caching": true + }, "openrouter/deepseek/deepseek-v4-flash-0731": { "input_cost_per_token": 6.5e-08, "output_cost_per_token": 1.8e-07, @@ -62807,9 +63086,9 @@ "supports_vision": false }, "openrouter/moonshotai/kimi-k3": { - "input_cost_per_token": 3e-06, - "output_cost_per_token": 1.5e-05, - "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 2.1e-06, + "output_cost_per_token": 1.053e-05, + "cache_read_input_token_cost": 2.35e-07, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 943718, @@ -62906,9 +63185,9 @@ "supports_prompt_caching": true }, "openrouter/z-ai/glm-5.2": { - "input_cost_per_token": 9.66e-07, - "output_cost_per_token": 3.036e-06, - "cache_read_input_token_cost": 1.932e-07, + "input_cost_per_token": 6e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 1.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 131072, @@ -62939,9 +63218,9 @@ "supports_vision": false }, "openrouter/moonshotai/kimi-k2.7-code": { - "input_cost_per_token": 6.6e-07, - "output_cost_per_token": 3.4e-06, - "cache_read_input_token_cost": 1.8e-07, + "input_cost_per_token": 7.1e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 1.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 235929, @@ -63189,10 +63468,28 @@ "supports_vision": true, "supports_pdf_input": true }, + "openrouter/openai/gpt-chat-latest": { + "input_cost_per_token": 5e-06, + "output_cost_per_token": 3e-05, + "cache_read_input_token_cost": 5e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-chat-latest", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": false, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 8.778e-08, - "output_cost_per_token": 1.7556e-07, - "cache_read_input_token_cost": 1.7556e-08, + "input_cost_per_token": 8.54e-08, + "output_cost_per_token": 1.708e-07, + "cache_read_input_token_cost": 1.708e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, @@ -63225,8 +63522,8 @@ "supports_prompt_caching": true }, "openrouter/google/gemma-4-26b-a4b-it": { - "input_cost_per_token": 7e-08, - "output_cost_per_token": 3.4e-07, + "input_cost_per_token": 4.2e-08, + "output_cost_per_token": 2.2e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 16384, @@ -63908,7 +64205,7 @@ "supports_vision": false }, "openrouter/qwen/qwen3-next-80b-a3b-instruct": { - "input_cost_per_token": 1e-07, + "input_cost_per_token": 9e-08, "output_cost_per_token": 1.1e-06, "cache_read_input_token_cost": 7e-08, "litellm_provider": "openrouter", @@ -64034,8 +64331,8 @@ "supports_vision": false }, "openrouter/qwen/qwen3-30b-a3b-instruct-2507": { - "input_cost_per_token": 4.815e-08, - "output_cost_per_token": 1.9305e-07, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 3e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 32000, @@ -64233,8 +64530,8 @@ "supports_vision": false }, "openrouter/qwen/qwen3-14b": { - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 2.4e-07, + "input_cost_per_token": 2.275e-07, + "output_cost_per_token": 9.1e-07, "litellm_provider": "openrouter", "max_input_tokens": 131072, "max_output_tokens": 16384, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1c86dbaacff..a55cb6b571e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -13118,6 +13118,22 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "cerebras/qwen-3.8-27b": { + "input_cost_per_token": 9.9e-07, + "litellm_provider": "cerebras", + "max_input_tokens": 65536, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.49e-06, + "source": "https://api.cerebras.ai/public/v1/models/qwen-3.8-27b", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "chatdolphin": { "input_cost_per_token": 5e-07, "litellm_provider": "nlp_cloud", @@ -31328,8 +31344,6 @@ "cache_read_input_token_cost_above_272k_tokens_flex": 5e-07 }, "gpt-5.5-pro": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -31378,8 +31392,6 @@ "supports_low_reasoning_effort": false }, "gpt-5.5-pro-2026-04-23": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -31532,8 +31544,6 @@ "cache_read_input_token_cost_above_272k_tokens_flex": 2.5e-07 }, "gpt-5.4-pro": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -31583,8 +31593,6 @@ "output_cost_per_token_above_272k_tokens_flex": 0.000135 }, "gpt-5.4-pro-2026-03-05": { - "cache_read_input_token_cost": 3e-06, - "cache_read_input_token_cost_above_272k_tokens": 6e-06, "input_cost_per_token": 3e-05, "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, @@ -33891,6 +33899,20 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "inception/mercury-2.5": { + "input_cost_per_token": 2e-07, + "litellm_provider": "inception", + "max_input_tokens": 260000, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 7.5e-07, + "source": "https://docs.inceptionlabs.ai/get-started/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "text-completion-inception/mercury-edit-2": { "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 2.5e-07, @@ -39127,19 +39149,21 @@ "supports_tool_choice": true }, "openrouter/deepseek/deepseek-chat-v3.1": { - "input_cost_per_token": 2e-07, + "input_cost_per_token": 2.5e-07, "input_cost_per_token_cache_hit": 2e-08, "litellm_provider": "openrouter", "max_input_tokens": 163840, "max_output_tokens": 163840, "max_tokens": 163840, "mode": "chat", - "output_cost_per_token": 8e-07, + "output_cost_per_token": 9.5e-07, "supports_assistant_prefill": true, "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 1.3e-07, + "source": "https://openrouter.ai/deepseek/deepseek-chat-v3.1" }, "openrouter/deepseek/deepseek-v3.2": { "input_cost_per_token": 2.69e-07, @@ -39202,36 +39226,56 @@ "supports_tool_choice": true }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 1.32e-06, + "input_cost_per_token": 8.59908e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 3.96e-06, + "output_cost_per_token": 1.719816e-06, "source": "https://openrouter.ai/deepseek/deepseek-v4-pro", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 7.1659e-08 + }, + "openrouter/deepseek/deepseek-v4.1-flash": { + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 3e-09, + "litellm_provider": "openrouter", + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "source": "https://openrouter.ai/deepseek/deepseek-v4.1-flash", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": false, + "supports_prompt_caching": true }, "openrouter/deepseek/deepseek-v4-pro-0813": { - "input_cost_per_token": 1.32e-06, + "input_cost_per_token": 5.7948e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 3.96e-06, + "output_cost_per_token": 1.73844e-06, "source": "https://openrouter.ai/deepseek/deepseek-v4-pro-0813", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 1.9316e-08 }, "openrouter/google/gemini-2.0-flash-001": { "deprecation_date": "2026-06-01", @@ -40010,6 +40054,28 @@ "supports_tool_choice": true, "supports_vision": true }, + "openrouter/openai/gpt-5.6-sol-pro": { + "input_cost_per_token": 2e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 2e-07, + "cache_creation_input_token_cost": 2.5e-06, + "input_cost_per_token_above_272k_tokens": 4e-06, + "output_cost_per_token_above_272k_tokens": 1.5e-05, + "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-5.6-sol-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/openai/gpt-oss-120b": { "input_cost_per_token": 3.7e-08, "litellm_provider": "openrouter", @@ -40138,13 +40204,13 @@ "supports_tool_choice": true }, "openrouter/qwen/qwen3-235b-a22b-2507": { - "input_cost_per_token": 8.75e-08, + "input_cost_per_token": 2.2e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", - "output_cost_per_token": 3.5e-07, + "output_cost_per_token": 8.8e-07, "source": "https://openrouter.ai/qwen/qwen3-235b-a22b-2507", "supports_function_calling": true, "supports_tool_choice": true @@ -40177,7 +40243,7 @@ "supports_vision": true }, "openrouter/qwen/qwen3.5-35b-a3b": { - "input_cost_per_token": 2.5e-07, + "input_cost_per_token": 3.125e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 65536, @@ -40188,7 +40254,8 @@ "supports_function_calling": true, "supports_reasoning": true, "supports_tool_choice": true, - "supports_vision": true + "supports_vision": true, + "cache_read_input_token_cost": 1.5625e-07 }, "openrouter/qwen/qwen3.5-27b": { "input_cost_per_token": 1.95e-07, @@ -40205,13 +40272,13 @@ "supports_vision": true }, "openrouter/qwen/qwen3.5-122b-a10b": { - "input_cost_per_token": 2.9e-07, + "input_cost_per_token": 2.6e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", - "output_cost_per_token": 2.4e-06, + "output_cost_per_token": 2.08e-06, "source": "https://openrouter.ai/qwen/qwen3.5-122b-a10b", "supports_function_calling": true, "supports_reasoning": true, @@ -40298,18 +40365,19 @@ "supports_web_search": true }, "openrouter/z-ai/glm-4.6": { - "input_cost_per_token": 5.5e-07, + "input_cost_per_token": 4.3e-07, "litellm_provider": "openrouter", "max_input_tokens": 202800, "max_output_tokens": 131000, "max_tokens": 131000, "mode": "chat", - "output_cost_per_token": 2.2e-06, + "output_cost_per_token": 1.75e-06, "source": "https://openrouter.ai/z-ai/glm-4.6", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "cache_read_input_token_cost": 8e-08 }, "openrouter/z-ai/glm-4.6:exacto": { "input_cost_per_token": 4.5e-07, @@ -43134,7 +43202,11 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 1.2e-06, + "max_input_tokens": 131072, + "source": "https://api.together.xyz/v1/models" }, "together_ai/Qwen/Qwen2.5-7B-Instruct-Turbo": { "litellm_provider": "together_ai", @@ -43142,7 +43214,11 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 3e-07, + "max_input_tokens": 32768, + "source": "https://api.together.xyz/v1/models" }, "together_ai/Qwen/Qwen3-235B-A22B-Instruct-2507-tput": { "deprecation_date": "2026-07-10", @@ -43354,7 +43430,11 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "max_input_tokens": 32768, + "source": "https://api.together.xyz/v1/models" }, "together_ai/mistralai/Mistral-Small-24B-Instruct-2501": { "deprecation_date": "2026-04-02", @@ -43362,7 +43442,11 @@ "mode": "chat", "supports_function_calling": true, "supports_parallel_function_calling": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "max_input_tokens": 32768, + "source": "https://api.together.xyz/v1/models" }, "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1": { "deprecation_date": "2026-04-16", @@ -48093,27 +48177,29 @@ "supports_tool_choice": true }, "vertex_ai/mistral-small-2503": { - "input_cost_per_token": 1e-06, + "input_cost_per_token": 1e-07, "litellm_provider": "vertex_ai-mistral_models", "max_input_tokens": 128000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 3e-06, + "output_cost_per_token": 3e-07, "supports_function_calling": true, "supports_tool_choice": true, - "supports_vision": true + "supports_vision": true, + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/mistral-small-2503@001": { - "input_cost_per_token": 1e-06, + "input_cost_per_token": 1e-07, "litellm_provider": "vertex_ai-mistral_models", "max_input_tokens": 32000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_token": 3e-06, + "output_cost_per_token": 3e-07, "supports_function_calling": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/mistral-ocr-2505": { "litellm_provider": "vertex_ai", @@ -48163,15 +48249,16 @@ "supports_reasoning": true }, "vertex_ai/openai/gpt-oss-20b-maas": { - "input_cost_per_token": 7.5e-08, + "input_cost_per_token": 7e-08, "litellm_provider": "vertex_ai-openai_models", "max_input_tokens": 131072, "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", - "output_cost_per_token": 3e-07, - "source": "https://console.cloud.google.com/vertex-ai/publishers/openai/model-garden/gpt-oss-120b-maas", - "supports_reasoning": true + "output_cost_per_token": 2.5e-07, + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", + "supports_reasoning": true, + "cache_read_input_token_cost": 7e-09 }, "vertex_ai/xai/grok-4.1-fast-non-reasoning": { "cache_read_input_token_cost": 5e-08, @@ -60695,6 +60782,113 @@ "output_cost_per_token": 4.7e-07, "source": "https://docs.together.ai/docs/serverless-models" }, + "together_ai/moonshotai/Kimi-K2.6": { + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 4.5e-06, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/moonshotai/Kimi-K2.5-fp4": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.8e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/MiniMaxAI/MiniMax-M2.7": { + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 6e-08, + "litellm_provider": "together_ai", + "max_input_tokens": 196608, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/zai-org/GLM-5": { + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/zai-org/GLM-5.1": { + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "cache_read_input_token_cost": 2.6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-0528": { + "input_cost_per_token": 3e-06, + "output_cost_per_token": 7e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 163840, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/Qwen3-Coder-Next-FP8": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-32B-Instruct": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 1.5e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-8B-Instruct": { + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 6.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/mistralai/Ministral-3-14B-Instruct-2512": { + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/nvidia/NVIDIA-Nemotron-Nano-9B-v2": { + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/mistralai/Mistral-7B-Instruct-v0.3": { + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, + "together_ai/Qwen/QwQ-32B": { + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "mode": "chat", + "source": "https://api.together.xyz/v1/models" + }, "cerebras/gemma-4-31b": { "input_cost_per_token": 9.9e-07, "litellm_provider": "cerebras", @@ -62348,6 +62542,28 @@ "cache_read_input_token_cost": 2e-08, "supports_prompt_caching": true }, + "openrouter/openai/gpt-5.6-luna-pro": { + "input_cost_per_token": 2e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 2e-08, + "cache_creation_input_token_cost": 2.5e-07, + "input_cost_per_token_above_272k_tokens": 4e-07, + "output_cost_per_token_above_272k_tokens": 1.8e-06, + "cache_read_input_token_cost_above_272k_tokens": 4e-08, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-5.6-luna-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/openai/gpt-5.6-terra": { "input_cost_per_token": 2e-06, "output_cost_per_token": 1.2e-05, @@ -62367,6 +62583,28 @@ "cache_read_input_token_cost": 2e-07, "supports_prompt_caching": true }, + "openrouter/openai/gpt-5.6-terra-pro": { + "input_cost_per_token": 2e-06, + "output_cost_per_token": 1.2e-05, + "cache_read_input_token_cost": 2e-07, + "cache_creation_input_token_cost": 2.5e-06, + "input_cost_per_token_above_272k_tokens": 4e-06, + "output_cost_per_token_above_272k_tokens": 1.8e-05, + "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-5.6-terra-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/openai/o3": { "input_cost_per_token": 2e-06, "output_cost_per_token": 8e-06, @@ -62600,6 +62838,28 @@ "supports_pdf_input": true, "supports_prompt_caching": true }, + "openrouter/openai/gpt-6-astra-pro": { + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "cache_read_input_token_cost": 1e-06, + "cache_creation_input_token_cost": 1.25e-05, + "input_cost_per_token_above_272k_tokens": 2e-05, + "output_cost_per_token_above_272k_tokens": 7.5e-05, + "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "litellm_provider": "openrouter", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-6-astra-pro", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/qwen/qwen3.8-flash": { "input_cost_per_token": 1.5e-07, "output_cost_per_token": 4.7e-07, @@ -62619,9 +62879,9 @@ "supports_prompt_caching": true }, "openrouter/z-ai/glm-5.3-flash": { - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 2.5e-07, - "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, "litellm_provider": "openrouter", "max_input_tokens": 1310720, "max_output_tokens": 131072, @@ -62736,6 +62996,25 @@ "supports_vision": true, "supports_prompt_caching": true }, + "openrouter/qwen/qwen3.8-max-0902": { + "input_cost_per_token": 2e-06, + "output_cost_per_token": 6e-06, + "cache_read_input_token_cost": 2.5e-07, + "cache_creation_input_token_cost": 2.5e-06, + "litellm_provider": "openrouter", + "max_input_tokens": 1000000, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "source": "https://openrouter.ai/qwen/qwen3.8-max-0902", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": false, + "supports_prompt_caching": true + }, "openrouter/deepseek/deepseek-v4-flash-0731": { "input_cost_per_token": 6.5e-08, "output_cost_per_token": 1.8e-07, @@ -62807,9 +63086,9 @@ "supports_vision": false }, "openrouter/moonshotai/kimi-k3": { - "input_cost_per_token": 3e-06, - "output_cost_per_token": 1.5e-05, - "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 2.1e-06, + "output_cost_per_token": 1.053e-05, + "cache_read_input_token_cost": 2.35e-07, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 943718, @@ -62906,9 +63185,9 @@ "supports_prompt_caching": true }, "openrouter/z-ai/glm-5.2": { - "input_cost_per_token": 9.66e-07, - "output_cost_per_token": 3.036e-06, - "cache_read_input_token_cost": 1.932e-07, + "input_cost_per_token": 6e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 1.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 131072, @@ -62939,9 +63218,9 @@ "supports_vision": false }, "openrouter/moonshotai/kimi-k2.7-code": { - "input_cost_per_token": 6.6e-07, - "output_cost_per_token": 3.4e-06, - "cache_read_input_token_cost": 1.8e-07, + "input_cost_per_token": 7.1e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 1.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 235929, @@ -63189,10 +63468,28 @@ "supports_vision": true, "supports_pdf_input": true }, + "openrouter/openai/gpt-chat-latest": { + "input_cost_per_token": 5e-06, + "output_cost_per_token": 3e-05, + "cache_read_input_token_cost": 5e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://openrouter.ai/openai/gpt-chat-latest", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": false, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 8.778e-08, - "output_cost_per_token": 1.7556e-07, - "cache_read_input_token_cost": 1.7556e-08, + "input_cost_per_token": 8.54e-08, + "output_cost_per_token": 1.708e-07, + "cache_read_input_token_cost": 1.708e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, @@ -63225,8 +63522,8 @@ "supports_prompt_caching": true }, "openrouter/google/gemma-4-26b-a4b-it": { - "input_cost_per_token": 7e-08, - "output_cost_per_token": 3.4e-07, + "input_cost_per_token": 4.2e-08, + "output_cost_per_token": 2.2e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 16384, @@ -63908,7 +64205,7 @@ "supports_vision": false }, "openrouter/qwen/qwen3-next-80b-a3b-instruct": { - "input_cost_per_token": 1e-07, + "input_cost_per_token": 9e-08, "output_cost_per_token": 1.1e-06, "cache_read_input_token_cost": 7e-08, "litellm_provider": "openrouter", @@ -64034,8 +64331,8 @@ "supports_vision": false }, "openrouter/qwen/qwen3-30b-a3b-instruct-2507": { - "input_cost_per_token": 4.815e-08, - "output_cost_per_token": 1.9305e-07, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 3e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 32000, @@ -64233,8 +64530,8 @@ "supports_vision": false }, "openrouter/qwen/qwen3-14b": { - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 2.4e-07, + "input_cost_per_token": 2.275e-07, + "output_cost_per_token": 9.1e-07, "litellm_provider": "openrouter", "max_input_tokens": 131072, "max_output_tokens": 16384, diff --git a/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py b/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py index 09718b1e6e0..6a438888a8a 100644 --- a/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py +++ b/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py @@ -1,3 +1,4 @@ +import litellm from litellm.llms.cerebras.chat import CerebrasConfig @@ -59,3 +60,23 @@ def test_map_openai_params_preserves_max_retries_zero_falsy() -> None: assert "max_retries" in result and result["max_retries"] == 0, ( f"max_retries=0 (falsy) must not be silently omitted; got: {result!r}" ) + + +def test_qwen_3_8_27b_cost_and_tokens(monkeypatch) -> None: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + model = "cerebras/qwen-3.8-27b" + prompt_cost, completion_cost = litellm.cost_per_token( + model=model, + prompt_tokens=1000, + completion_tokens=1000, + ) + assert abs(prompt_cost - 0.00099) < 1e-9 + assert abs(completion_cost - 0.00149) < 1e-9 + + model_info = litellm.get_model_info(model) + assert model_info["max_input_tokens"] == 65536 + assert model_info["max_output_tokens"] == 32768 + assert model_info["supports_vision"] is True + assert model_info["supports_reasoning"] is True + assert model_info["supports_parallel_function_calling"] is True diff --git a/tests/test_litellm/llms/inception/test_inception_chat_transformation.py b/tests/test_litellm/llms/inception/test_inception_chat_transformation.py index 4c0f5969249..e3bc4d7992d 100644 --- a/tests/test_litellm/llms/inception/test_inception_chat_transformation.py +++ b/tests/test_litellm/llms/inception/test_inception_chat_transformation.py @@ -238,6 +238,7 @@ def test_inception_model_list_populated(monkeypatch): litellm.add_known_models() assert "inception/mercury-2" in litellm.inception_models + assert "inception/mercury-2.5" in litellm.inception_models for model in litellm.inception_models: assert model.startswith("inception/") @@ -304,3 +305,24 @@ def test_inception_completion_targets_inception_endpoint(): assert captured["body"]["model"] == "mercury-2" assert captured["body"]["tool_choice"] == "auto" assert response.choices[0].message.content == "hi" + + +def test_inception_mercury_2_5_cost_and_tokens(monkeypatch): + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + model = "inception/mercury-2.5" + prompt_cost, completion_cost = litellm.cost_per_token( + model=model, + prompt_tokens=1000, + completion_tokens=500, + ) + assert abs(prompt_cost - 0.0002) < 1e-9 + assert abs(completion_cost - 0.000375) < 1e-9 + + model_info = litellm.get_model_info(model) + assert model_info["max_input_tokens"] == 260000 + assert model_info["max_output_tokens"] == 65536 + assert model_info["litellm_provider"] == "inception" + assert model_info["mode"] == "chat" + assert model_info["supports_function_calling"] is True + assert model_info["supports_response_schema"] is True From 66980bbb873b4db3d6fd070bc8471feae52efe26 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 12:40:25 -0700 Subject: [PATCH 095/157] test(e2e): address Redis chaos PR review, add log-bytes budget Pad the locust payload to tens of KB so per-request bookkeeping cost scales with body size instead of hiding behind a 40-byte prompt. Turn on use_redis_transaction_buffer in the chaos config and JSON_LOGS in the workflow so the spend buffer, pod lock, and JSON-encoded breaker tracebacks are all part of the measured chaos cost. Add a log-bytes-per-request budget alongside latency, RSS, and CPU, reading the proxy's log file size at each phase split; its ceiling is uncalibrated since no chaos run has measured it yet. Co-Authored-By: Claude Code --- .github/workflows/test-e2e-redis-chaos.yml | 2 + tests/e2e/CLAUDE.md | 2 +- tests/e2e/gateway/redis_chaos_ci_config.yml | 1 + tests/e2e/load/locustfile.py | 7 ++- tests/e2e/load/test_redis_chaos_e2e.py | 59 ++++++++++++++++++--- 5 files changed, 62 insertions(+), 9 deletions(-) diff --git a/.github/workflows/test-e2e-redis-chaos.yml b/.github/workflows/test-e2e-redis-chaos.yml index 2ce20836db7..0880c1fb464 100644 --- a/.github/workflows/test-e2e-redis-chaos.yml +++ b/.github/workflows/test-e2e-redis-chaos.yml @@ -40,6 +40,7 @@ jobs: DATABASE_URL: postgresql://llmproxy:dbpassword9090@localhost:5432/litellm LITELLM_MASTER_KEY: sk-redis-chaos-e2e LITELLM_LOG: WARNING + JSON_LOGS: "true" steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: @@ -73,6 +74,7 @@ jobs: run: | nohup uv run --no-sync litellm --config tests/e2e/gateway/redis_chaos_ci_config.yml --port 4000 --num_workers 4 > proxy.log 2>&1 & echo "E2E_PROXY_PID=$!" >> "$GITHUB_ENV" + echo "E2E_PROXY_LOG=$(pwd)/proxy.log" >> "$GITHUB_ENV" for _ in $(seq 1 90); do if curl -fs http://localhost:4000/health/liveliness > /dev/null; then exit 0 diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 80e78dd4cdd..ed57dd71759 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint and budgeting p50/p90/p99 latency, RSS, and CPU-per-request as ratios against the same run's healthy phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint and budgeting p50/p90/p99 latency, RSS, CPU-per-request, and log-bytes-per-request as ratios against the same run's healthy phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/gateway/redis_chaos_ci_config.yml b/tests/e2e/gateway/redis_chaos_ci_config.yml index 37b024502f7..f7a71c50a71 100644 --- a/tests/e2e/gateway/redis_chaos_ci_config.yml +++ b/tests/e2e/gateway/redis_chaos_ci_config.yml @@ -1,6 +1,7 @@ general_settings: master_key: os.environ/LITELLM_MASTER_KEY store_model_in_db: true + use_redis_transaction_buffer: true litellm_settings: callbacks: ["prometheus"] diff --git a/tests/e2e/load/locustfile.py b/tests/e2e/load/locustfile.py index a4e396aa8d7..9b7bdf2ee1e 100644 --- a/tests/e2e/load/locustfile.py +++ b/tests/e2e/load/locustfile.py @@ -11,17 +11,20 @@ from locust import FastHttpUser, constant, task _MODEL: Final = os.environ["LOAD_MODEL"] _API_KEYS: Final = tuple(os.environ["LOAD_API_KEYS"].split(",")) _NEXT_ENDPOINT: Final = cycle(os.environ["LOAD_ENDPOINTS"].split(",")) +_FILLER: Final = "x" * 40_000 def _payload() -> dict[str, object]: """A prompt no other request sent, so the response cache never answers for the deployment. Both endpoints take the same body: /v1/messages requires max_tokens, which /chat/completions - also accepts, so one payload serves the whole round robin. + also accepts, so one payload serves the whole round robin. Padded to tens of KB so a + per-request bookkeeping cost that scales with body size (string formatting, hashing) shows + up in the CPU and log-size budgets instead of hiding behind a 40-byte prompt. """ return { "model": _MODEL, - "messages": [{"role": "user", "content": f"load test ping {uuid.uuid4().hex}"}], + "messages": [{"role": "user", "content": f"load test ping {uuid.uuid4().hex} {_FILLER}"}], "max_tokens": 16, } diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 646429bbdd7..153be656566 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -22,11 +22,13 @@ Every touchpoint times out: the auth cache read falls back to Postgres, the resp read and write both fail, and the spend counter increment times out and the callback stringifies the request metadata, breadcrumbs included, into a failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung (LIT-6780), which is what the -per-phase RSS and CPU percentiles are here to catch. +per-phase RSS, CPU, and log-bytes budgets are here to catch. Needs the proxy on the same host, since RSS and CPU come from psutil on its process tree: a multi-worker proxy serves /metrics from the prometheus multiprocess collector, which drops -the process collector's memory and CPU series. Deselected unless E2E_REDIS_CHAOS is set. +the process collector's memory and CPU series. Log bytes are read from the file the proxy's +stdout/stderr was redirected to, so the same host requirement covers that too. Deselected +unless E2E_REDIS_CHAOS is set. """ from __future__ import annotations @@ -37,6 +39,7 @@ import time from collections.abc import Iterator from dataclasses import dataclass from itertools import pairwise +from pathlib import Path from typing import Final import pytest @@ -79,6 +82,11 @@ REDIS_PAUSE_MS: Final = int(CHAOS_SECONDS * 1000) CHAOS_LATENCY_RATIO_CEILING: Final = 12.0 CHAOS_RSS_RATIO_CEILING: Final = 1.5 CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 6.0 +# Uncalibrated: no chaos run has measured this yet, since JSON_LOGS and the padded payload +# landed after the last run this file's other ceilings were calibrated from. Deliberately loose +# until a real run tightens it; the failed-tracking alert body that motivates this test already +# logs the full request metadata per timeout, so a JSON-encoded traceback storm should dwarf this. +CHAOS_LOG_BYTES_PER_REQUEST_RATIO_CEILING: Final = 20.0 DRAIN_TIMEOUT_SECONDS: Final = 30.0 DRAIN_POLL_SECONDS: Final = 1.0 @@ -111,6 +119,7 @@ class Phase: load: LoadResult usage: UsageWindow redis_timeouts: float + log_bytes: int @property def timeouts_per_request(self) -> float: @@ -120,11 +129,16 @@ class Phase: def cpu_seconds_per_request(self) -> float: return self.usage.cpu_seconds_per_request(self.load.requests) + @property + def log_bytes_per_request(self) -> float: + return self.log_bytes / self.load.requests if self.load.requests else 0.0 + def report(self) -> str: return ( f"{self.name}: {self.load.requests} requests, {self.load.failures} failures, " f"{self.load.requests_per_second:.0f} rps, {self.load.latency_summary()}; {self.usage.summary()}; " f"{self.cpu_seconds_per_request * 1000:.1f} ms CPU per request; " + f"{self.log_bytes_per_request:.0f} log bytes per request; " f"{self.timeouts_per_request:.2f} Redis timeouts per request; " f"by endpoint: {self.load.endpoint_summary()}" ) @@ -163,6 +177,22 @@ def proxy_pid() -> int: return int(pid) +@pytest.fixture +def proxy_log() -> Path: + """Path to the proxy's stdout/stderr log, which the workflow captures to a file. + + Required rather than discovered for the same reason as proxy_pid: a developer machine may + have more than one proxy log around. + """ + path: Final = os.environ.get("E2E_PROXY_LOG") + assert path, "E2E_PROXY_LOG must hold the path the proxy's stdout/stderr was redirected to" + return Path(path) + + +def _log_bytes(path: Path) -> int: + return path.stat().st_size + + @pytest.fixture def redis_control() -> Iterator[redis.Redis[bytes]]: """A control connection to the proxy's Redis, which unpauses it in teardown as a safety net. @@ -292,10 +322,13 @@ def _chaos_budgets(baseline: Phase, chaos: Phase) -> tuple[Budget, ...]: tail (or only in the median) cannot hide behind the other. Latency gets the loosest bound because a timing-out Redis legitimately adds its socket_timeout to every request that touches it, several times over on a retried request. RSS gets the tightest: the failure - path has no business allocating more per request. CPU is budgeted once, as CPU seconds per - request rather than per percentile: cores-busy saturates at the worker count under load, so - its percentiles read the same whether a request costs 10 ms of CPU or 40, and cannot budget - anything; seconds per request is the CPU figure that actually moves. + path has no business allocating more per request. CPU and log bytes are each budgeted once, + as an amount per request rather than per percentile: cores-busy saturates at the worker + count under load, so its percentiles read the same whether a request costs 10 ms of CPU or + 40, and cannot budget anything; per-request is the figure that actually moves. Log bytes + isolates the cost of the failed-tracking alert's own noisy error handling from the CPU it + burns doing useful retry work, since the two would otherwise be indistinguishable in one + CPU number. """ return ( _latency_budget("p50", baseline.load.p50_seconds, chaos.load.p50_seconds), @@ -311,6 +344,14 @@ def _chaos_budgets(baseline: Phase, chaos: Phase) -> tuple[Budget, ...]: ratio_ceiling=CHAOS_CPU_PER_REQUEST_RATIO_CEILING, unit=" ms", ), + Budget( + name="log bytes per request", + baseline=baseline.log_bytes_per_request, + degraded=chaos.log_bytes_per_request, + ratio_ceiling=CHAOS_LOG_BYTES_PER_REQUEST_RATIO_CEILING, + unit=" B", + decimals=0, + ), ) @@ -325,6 +366,7 @@ class TestRedisChaos: client: LoadClient, resources: ResourceManager, proxy_pid: int, + proxy_log: Path, redis_control: redis.Redis[bytes], ) -> None: proxy: Final = client.proxy @@ -335,28 +377,33 @@ class TestRedisChaos: cooldown_re: Final = _deployment_metric_re("deployment_cooled_down_total", model_ids) at_start: Final = _scrape(proxy) + log_at_start: Final = _log_bytes(proxy_log) with ProxyUsageSampler(proxy_pid) as sampler: baseline_load: Final = _drive(keys, BASELINE_SECONDS) baseline_usage: Final = sampler.split() after_baseline: Final = _scrape(proxy) + log_after_baseline: Final = _log_bytes(proxy_log) redis_control.client_pause(REDIS_PAUSE_MS, all=True) # pyright: ignore[reportUnknownMemberType] # redis-py stubs return Any chaos_load: Final = _drive(keys, CHAOS_SECONDS) chaos_usage: Final = sampler.split() at_end: Final = _scrape_after_drain(proxy, retries_re) + log_at_end: Final = _log_bytes(proxy_log) baseline: Final = Phase( name="baseline", load=baseline_load, usage=baseline_usage, redis_timeouts=_metric(after_baseline, TIMEOUT_FAILURES_RE) - _metric(at_start, TIMEOUT_FAILURES_RE), + log_bytes=log_after_baseline - log_at_start, ) chaos: Final = Phase( name="chaos", load=chaos_load, usage=chaos_usage, redis_timeouts=_metric(at_end, TIMEOUT_FAILURES_RE) - _metric(after_baseline, TIMEOUT_FAILURES_RE), + log_bytes=log_at_end - log_after_baseline, ) report: Final = f"{baseline.report()} | {chaos.report()}" From e3130a87bc0f87a141dd07d84ab4fadea65c23a5 Mon Sep 17 00:00:00 2001 From: mateo Date: Fri, 11 Sep 2026 19:50:27 +0000 Subject: [PATCH 096/157] test(cost-map): type the monkeypatch fixture in cerebras and inception registry tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../llms/cerebras/test_cerebras_chat_transformation.py | 4 +++- .../llms/inception/test_inception_chat_transformation.py | 3 ++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py b/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py index 6a438888a8a..a47180e9511 100644 --- a/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py +++ b/tests/test_litellm/llms/cerebras/test_cerebras_chat_transformation.py @@ -1,3 +1,5 @@ +import pytest + import litellm from litellm.llms.cerebras.chat import CerebrasConfig @@ -62,7 +64,7 @@ def test_map_openai_params_preserves_max_retries_zero_falsy() -> None: ) -def test_qwen_3_8_27b_cost_and_tokens(monkeypatch) -> None: +def test_qwen_3_8_27b_cost_and_tokens(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) model = "cerebras/qwen-3.8-27b" diff --git a/tests/test_litellm/llms/inception/test_inception_chat_transformation.py b/tests/test_litellm/llms/inception/test_inception_chat_transformation.py index e3bc4d7992d..04813143fae 100644 --- a/tests/test_litellm/llms/inception/test_inception_chat_transformation.py +++ b/tests/test_litellm/llms/inception/test_inception_chat_transformation.py @@ -7,6 +7,7 @@ import os from unittest import mock import httpx +import pytest import litellm from litellm.llms.inception.chat.transformation import InceptionChatConfig @@ -307,7 +308,7 @@ def test_inception_completion_targets_inception_endpoint(): assert response.choices[0].message.content == "hi" -def test_inception_mercury_2_5_cost_and_tokens(monkeypatch): +def test_inception_mercury_2_5_cost_and_tokens(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) model = "inception/mercury-2.5" From f22f9bc4614a29e7b498d426e31f737864b4373e Mon Sep 17 00:00:00 2001 From: tin-berri Date: Fri, 11 Sep 2026 12:52:42 -0700 Subject: [PATCH 097/157] feat(auto-router): show routed model and savings in Claude Code and Codex (#40330) --- .../migration.sql | 1 + .../litellm_proxy_extras/schema.prisma | 1 + litellm/litellm_core_utils/private_json.py | 20 + litellm/models/__init__.py | 2 + litellm/models/autorouter_session.py | 39 ++ litellm/proxy/_types.py | 2 + litellm/proxy/client/cli/README.md | 34 +- litellm/proxy/client/cli/commands/agents.py | 83 ++-- litellm/proxy/client/cli/commands/auth.py | 26 +- .../client/cli/commands/autoroute/commands.py | 14 +- .../client/cli/commands/autoroute/config.py | 1 + .../client/cli/commands/claude_settings.py | 171 ++++--- .../proxy/client/cli/commands/configure.py | 62 +-- .../client/cli/commands/statusline_script.py | 392 +++++++++++++++ litellm/proxy/client/cli/commands/up.py | 29 +- litellm/proxy/db/autorouter_session_rollup.py | 22 +- .../auto_router_endpoints.py | 52 +- litellm/proxy/schema.prisma | 1 + litellm/repositories/__init__.py | 2 + .../autorouter_session_repository.py | 34 ++ .../auto_router_endpoints.py | 24 + schema.prisma | 1 + .../spend/test_autorouter_session_rollup.py | 23 + .../litellm_core_utils/test_private_json.py | 35 +- tests/test_litellm/models/test_models.py | 33 ++ .../proxy/auth/test_route_checks.py | 25 + .../client/cli/autoroute/test_commands.py | 29 ++ .../proxy/client/cli/autoroute/test_config.py | 8 +- .../test_litellm/proxy/client/cli/conftest.py | 7 + .../proxy/client/cli/test_agents.py | 185 ++----- .../proxy/client/cli/test_auth_commands.py | 68 ++- .../proxy/client/cli/test_claude_settings.py | 451 +++++++----------- .../client/cli/test_configure_commands.py | 48 +- .../client/cli/test_statusline_script.py | 404 ++++++++++++++++ .../proxy/client/cli/test_up_commands.py | 112 ++--- .../db/test_autorouter_session_rollup.py | 18 +- .../test_auto_router_endpoints.py | 137 ++++++ .../repositories/test_repositories.py | 57 +++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 113 +++++ 39 files changed, 2044 insertions(+), 722 deletions(-) create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260910000000_add_autorouter_session_baseline_models/migration.sql create mode 100644 litellm/models/autorouter_session.py create mode 100644 litellm/proxy/client/cli/commands/statusline_script.py create mode 100644 litellm/repositories/autorouter_session_repository.py create mode 100644 tests/test_litellm/proxy/client/cli/test_statusline_script.py diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260910000000_add_autorouter_session_baseline_models/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260910000000_add_autorouter_session_baseline_models/migration.sql new file mode 100644 index 00000000000..e7ce1a3180b --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260910000000_add_autorouter_session_baseline_models/migration.sql @@ -0,0 +1 @@ +ALTER TABLE "LiteLLM_AutoRouterSession" ADD COLUMN IF NOT EXISTS "baseline_models" JSONB NOT NULL DEFAULT '{}'; diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 817df082d8c..7d521d54791 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -1514,6 +1514,7 @@ model LiteLLM_AutoRouterSession { classifier_cost Float @default(0) classifier_cost_recorded_turns Int @default(0) tier_turns Json @default("{}") + baseline_models Json @default("{}") @@id([api_key, session_id, router_name]) @@index([last_turn_at], map: "idx_autorouter_session_last_turn") diff --git a/litellm/litellm_core_utils/private_json.py b/litellm/litellm_core_utils/private_json.py index 30f64c8fc27..4cd4a9b4f82 100644 --- a/litellm/litellm_core_utils/private_json.py +++ b/litellm/litellm_core_utils/private_json.py @@ -36,6 +36,21 @@ def stage_private_json(path: str, data: Mapping[str, object]) -> str: return tmp_path +def stage_private_bytes(path: str, data: bytes) -> str: + parent: Final = Path(path).parent + parent.mkdir(parents=True, exist_ok=True) + fd, tmp_path = tempfile.mkstemp(dir=str(parent), prefix=".tmp-") + try: + with os.fdopen(fd, "wb") as f: + f.write(data) + f.flush() + os.fsync(f.fileno()) + except BaseException: + Path(tmp_path).unlink(missing_ok=True) + raise + return tmp_path + + def commit_staged_json(staged: str, path: str) -> None: """Move a staged file into place, replacing whatever is there in one step""" try: @@ -68,3 +83,8 @@ def discard_staged_json(staged: str) -> None: def write_private_json(path: str, data: Mapping[str, object]) -> None: """Atomically write JSON to path with owner-only permissions (0600)""" commit_staged_json(stage_private_json(path, data), path) + + +def write_private_bytes(path: str, data: bytes) -> None: + """Atomically write bytes to path with owner-only permissions (0600); a reader holding the old file keeps it whole""" + commit_staged_json(stage_private_bytes(path, data), path) diff --git a/litellm/models/__init__.py b/litellm/models/__init__.py index 07d1ffa743d..50eb6f8af4f 100644 --- a/litellm/models/__init__.py +++ b/litellm/models/__init__.py @@ -3,6 +3,7 @@ Domain models for LiteLLM backend. """ from litellm.models.access_group import LiteLLM_AccessGroupTable +from litellm.models.autorouter_session import LiteLLM_AutoRouterSession from litellm.models.budget import ( LiteLLM_BudgetTable, LiteLLM_BudgetTableFull, @@ -40,6 +41,7 @@ __all__ = [ "CredentialBase", "CredentialItem", "LiteLLM_AccessGroupTable", + "LiteLLM_AutoRouterSession", "LiteLLM_BudgetTable", "LiteLLM_BudgetTableFull", "LiteLLM_Config", diff --git a/litellm/models/autorouter_session.py b/litellm/models/autorouter_session.py new file mode 100644 index 00000000000..c7126236ec3 --- /dev/null +++ b/litellm/models/autorouter_session.py @@ -0,0 +1,39 @@ +""" +Auto-router per-session rollup model. + +Canonical definition for ``litellm_autoroutersession``, the row the spend flush +maintains per (api_key, session_id, router_name). +""" + +from collections.abc import Mapping +from datetime import datetime + +from litellm.types.llms.base import LiteLLMPydanticObjectBase + + +class LiteLLM_AutoRouterSession(LiteLLMPydanticObjectBase): + api_key: str + session_id: str + router_name: str + router_type: str + first_turn_at: datetime + last_turn_at: datetime + last_model: str + turns: int + spend: float + saved_spend: float + classifier_cost: float + tier_turns: Mapping[str, int] + baseline_models: Mapping[str, int] + + @property + def baseline_model(self) -> str | None: + """The baseline most of this session's turns were priced against, or None when no turn recorded one. + + A router reconfigured mid-session leaves turns priced against two baselines; the row keeps both + counts, and the label is the one that priced the most money-carrying turns rather than whatever the + router is configured with now. + """ + if not self.baseline_models: + return None + return max(self.baseline_models, key=lambda model: (self.baseline_models[model], model)) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 382ea384275..7b0e92d81ae 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -880,6 +880,8 @@ class LiteLLMRoutes(enum.Enum): # proxy admin, or team admin naming their own team via team_id "/auto_router/test_routing", "/auto_router/validate_complexity_router_config", + # Per-session auto-router read - the endpoint scopes the row to the caller's own key hash + "/auto_router/session", # Agent registry - reads are role-scoped and writes are proxy-admin-gated # inside agent_endpoints/endpoints.py *agent_management_routes, diff --git a/litellm/proxy/client/cli/README.md b/litellm/proxy/client/cli/README.md index 4045857a237..2071576a943 100644 --- a/litellm/proxy/client/cli/README.md +++ b/litellm/proxy/client/cli/README.md @@ -508,9 +508,9 @@ The credential is short-lived by design (default 24h, configurable via `LITELLM_ ### Route Every Claude Code Session Through the Proxy -`lite claude` wraps a single invocation, but `lite up` goes further: it patches `~/.claude/settings.json`, Claude Code's own config file, so that every Claude Code session started afterward -- from any terminal, launched normally with just `claude`, no wrapper needed -- routes through your LiteLLM proxy. It sets `env.ANTHROPIC_BASE_URL` to the proxy URL, `env.ENABLE_TOOL_SEARCH` to `true` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when those keys are missing, and `apiKeyHelper` to a `lite auth print-token` invocation, drops any stray static `ANTHROPIC_API_KEY` or `ANTHROPIC_AUTH_TOKEN` so the helper-issued token wins, and leaves every other setting in the file untouched. It backs up the original file before patching it. +`lite claude` wraps a single invocation, but `lite up` goes further: it patches `~/.claude/settings.json`, Claude Code's own config file, so that every Claude Code session started afterward -- from any terminal, launched normally with just `claude`, no wrapper needed -- routes through your LiteLLM proxy. It sets `env.ANTHROPIC_BASE_URL` to the proxy URL, `env.ENABLE_TOOL_SEARCH` to `true` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when those keys are missing, writes the key it resolved (your fresh `lite login`, or an explicit `--api-key`) into `env.ANTHROPIC_AUTH_TOKEN` as a static token, drops any stray `ANTHROPIC_API_KEY` or `apiKeyHelper` so nothing fights that token, and leaves every other setting in the file untouched. It backs up the original file before patching it. Nothing here writes an `apiKeyHelper`: Claude Code would spawn `lite` (and its keychain check) on every credential refresh, so the key is copied in instead and `lite up` restores the file when it stops. -Two things need to already be true: you've run `lite login` (or `lite login --pkce`, whose key the helper renews on its own), since the apiKeyHelper depends on that stored token, and the proxy is already reachable, since `lite up` does not start one for you. +Two things need to already be true: you've run `lite login` (or passed a key), and the proxy is already reachable, since `lite up` does not start one for you. ```bash lite login @@ -526,21 +526,19 @@ Cursor is not supported: it has no equivalent file-based config to hot-patch thi #### Making It Permanent at Login -`lite up` holds the patch only for as long as it runs. To wire Claude Code up once and leave it that way, pass `--config-claude` to `lite login`: +`lite up` holds its patch only for as long as it runs. To wire Claude Code up at login and leave it that way, pass `--config-claude` to `lite login`: ```bash lite --base-url https://your-proxy.example.com login --config-claude ``` -It writes the same settings `lite up` does, `env.ANTHROPIC_BASE_URL`, `env.ENABLE_TOOL_SEARCH`, `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY`, and `apiKeyHelper`, but persistently: no foreground process to keep alive, and `lite unconfigure claude` restores what it changed (see below). Every other key in `~/.claude/settings.json` is preserved, the file is created if it does not exist, and it is written atomically with owner-only permissions. Plain `lite login` is unchanged; nothing happens to your Claude Code config unless you pass the flag. +It writes the same settings `lite up` does, `env.ANTHROPIC_BASE_URL`, `env.ENABLE_TOOL_SEARCH`, `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY`, and the key this login minted as `env.ANTHROPIC_AUTH_TOKEN`, but persistently: no foreground process to keep alive, and `lite unconfigure claude` restores what it changed (see below). Every other key in `~/.claude/settings.json` is preserved, the file is created if it does not exist, and it is written atomically with owner-only permissions. Plain `lite login` is unchanged; nothing happens to your Claude Code config unless you pass the flag -Because the credential is reached through `apiKeyHelper` rather than copied into the file, a later `lite login` refreshes it with no further action: Claude Code re-runs the helper on every request and picks up whatever token the most recent login stored. Nothing secret is written to `settings.json`. +The key in the file is the login's own, so it expires with it (24h by default): run `lite login --config-claude` again after that, which rewrites the key in place. Earlier versions wrote an `apiKeyHelper` that ran `lite auth print-token` instead, so a later login refreshed Claude Code by itself; that meant Claude Code spawning a full `lite` start, keychain check included, on every credential refresh, so the helper is no longer written and a stale one is stripped by the next `--config-claude` or `configure claude`. Like `lite up`, the flag refuses to run while a `lite up` session holds a backup, and tells you to run `lite down` first -Run it again to point Claude Code at a different proxy; the base URL and the helper are both rewritten. `lite up` and `--config-claude` manage the same file, so the flag refuses to run while a `lite up` session holds a backup, and tells you to run `lite down` first, rather than writing settings that `lite up` would silently revert when it stops. +#### Configuring Claude Code Once, With a Virtual Key -#### Configuring Claude Code Once, With a Virtual Key or Your Login - -`lite configure claude` wires Claude Code up persistently and `lite unconfigure claude` puts things back. It is what `lite login --config-claude` does, plus a pinned model and an undo, and it also takes a long-lived virtual key when that is what you have: +`lite configure claude` wires Claude Code up persistently with a long-lived virtual key, a pinned model and an undo, and `lite unconfigure claude` puts things back: ```bash curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh @@ -548,11 +546,25 @@ lite --base-url https://your-proxy.example.com configure claude --api-key sk-... claude ``` -With `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) the key is written into `env.ANTHROPIC_AUTH_TOKEN`. Without one, your `lite login` credential is used the way `--config-claude` uses it, through `apiKeyHelper`, so a later `lite login` (or a `--pkce` renewal) picks up on its own and nothing secret lands in the file; a missing or stale login is refreshed first. Either way the command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key, which has to be on `/v1/models` for the key. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute up` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control +The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute up` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control Plain `lite configure`, with no agent named, asks the same things interactively: which agents to wire (Claude Code today) and which of the proxy's models to start on, picked from `/v1/models` with a type-to-filter prompt -What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Like `--config-claude`, both refuse to run while a `lite up` or `lite autoroute up` session holds a backup, and that check comes before any login prompt or request +What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute up` session holds a backup, and that check comes before any request + +#### Routed model and savings in the status line + +`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute up` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline: + +``` +claude-auto · Routed to: claude-haiku-4-5 -63% vs Claude Opus 5 +LiteLLM ████████░░░░░░░░░░░░░░░░ $0.14 +Claude Opus 5 ████████████████████████ $0.38 +``` + +The routed model comes from Claude Code's own transcript, so it only names the tier model when the auto-router deployment sets `return_raw_model_name: true` (the `lite autoroute` wizard does); otherwise it shows the alias you requested. The cost lines come from `GET /auto_router/session?session_id=...`, which any virtual key may call for its own sessions, and are cached for five seconds under a per-user `$TMPDIR/litellm-statusline-` directory. The baseline is the priciest model in the router's hardest tier, the same counterfactual the auto-router's savings reports use. `lite unconfigure claude` removes the `statusLine` entry only while it still points at that script. + +`lite codex` registers the same script as a Codex `Stop` hook for the launch, so after each turn Codex prints the same block as a system message. Codex asks once to trust the hook; the answer is remembered for later launches. ### QA Complexity-Based Auto-Routing Against Your Real Proxy diff --git a/litellm/proxy/client/cli/commands/agents.py b/litellm/proxy/client/cli/commands/agents.py index bf2a784f590..ea1eed65505 100644 --- a/litellm/proxy/client/cli/commands/agents.py +++ b/litellm/proxy/client/cli/commands/agents.py @@ -1,4 +1,6 @@ +import json import os +import re import shutil import subprocess import sys @@ -13,7 +15,7 @@ import requests from pydantic import BaseModel, TypeAdapter, ValidationError from .auth import CliContextObj, context_secret_vault, get_stored_api_key, login -from .claude_settings import claude_settings_path, lite_api_key_helper_configured +from .claude_settings import ClaudeSettingsError, install_statusline_script from .cmd_quoting import quote_for_cmd from .pi import ( LITELLM_PROXY_API_KEY_ENV, @@ -86,8 +88,6 @@ def build_agent_env( base_url: str, api_key: str, profiles: frozenset[str], - *, - export_anthropic_token: bool = True, ) -> dict[str, str]: """Return a copy of base_env wired to route the agent through the proxy. @@ -102,19 +102,12 @@ def build_agent_env( proxy's /v1/models; likewise left alone when already set. pi ignores both base URL variables and instead resolves $LITELLM_PROXY_API_KEY from its synced models.json provider entry. - - With export_anthropic_token=False the bearer is left out (and any inherited - one dropped) so Claude Code asks its configured apiKeyHelper instead; Claude - Code prefers ANTHROPIC_AUTH_TOKEN over the helper and warns when both are set. """ env: Final = dict(base_env) root: Final = base_url.rstrip("/") if PROFILE_ANTHROPIC in profiles: env[ANTHROPIC_BASE_URL_ENV] = root - if export_anthropic_token: - env[ANTHROPIC_AUTH_TOKEN_ENV] = api_key - else: - env.pop(ANTHROPIC_AUTH_TOKEN_ENV, None) + env[ANTHROPIC_AUTH_TOKEN_ENV] = api_key env.pop(ANTHROPIC_API_KEY_ENV, None) if ENABLE_TOOL_SEARCH_ENV not in env: env[ENABLE_TOOL_SEARCH_ENV] = ENABLE_TOOL_SEARCH_VALUE @@ -189,10 +182,52 @@ def prepare_pi( return ("--model", f"{PI_PROVIDER_NAME}/{ids[0]}") +def _warn(message: str) -> None: + click.echo(message, err=True) + + +_CODEX_STOP_HOOKS_DECLARED: Final = re.compile( + r"^\s*(\[\[\s*\"?hooks\"?\s*\.\s*\"?Stop\"?\s*\]\]|\"?hooks\"?(?:\s*\.\s*\"?Stop\"?)?\s*=|\[\s*\"?hooks\"?\s*\])", + re.MULTILINE, +) + + +def codex_config_path(base_env: Mapping[str, str]) -> Path: + return Path(base_env.get("CODEX_HOME") or Path.home() / ".codex") / "config.toml" + + +def codex_declares_stop_hooks(config_path: Path) -> bool: + """A config that cannot be read or decoded declares nothing we can see; Codex reports its own + TOML failure at launch, so the pre-check must not be the thing that stops `lite codex`.""" + try: + return _CODEX_STOP_HOOKS_DECLARED.search(config_path.read_text(encoding="utf-8")) is not None + except (OSError, UnicodeDecodeError): + return False + + +def prepare_codex( + base_url: str, + api_key: str, + base_env: Mapping[str, str], + *, + install: Callable[[], str] = install_statusline_script, + warn: Callable[[str], None] = _warn, +) -> tuple[str, ...]: + """A `-c hooks.Stop=` session flag replaces the user's whole Stop list, so their own hooks win over ours.""" + if codex_declares_stop_hooks(codex_config_path(base_env)): + warn("litellm: your Codex config already declares hooks; not adding the routed-model Stop hook") + return () + try: + command: Final = install() + except ClaudeSettingsError as e: + raise AgentRunError(str(e)) from e + return ("-c", f'hooks.Stop=[{{hooks=[{{type="command",command={json.dumps(command)}}}]}}]') + + _Preparer: TypeAlias = Callable[[str, str, Mapping[str, str]], Sequence[str]] _PREPARERS: Final[Mapping[str, _Preparer]] = MappingProxyType( - {"pi": prepare_pi} # mutable-ok: MappingProxyType freezes the provider registry + {"pi": prepare_pi, "codex": prepare_codex} # mutable-ok: MappingProxyType freezes the provider registry ) @@ -454,10 +489,6 @@ def _restore_controlling_terminal() -> None: os.close(fd) -def _warn(message: str) -> None: - click.echo(message, err=True) - - def run_agent( base_url: str, api_key: str, @@ -474,7 +505,6 @@ def run_agent( launcher: Callable[[str, Sequence[str], Mapping[str, str]], None] = _hand_off, reattach_terminal: Callable[[], None] | None = None, preparers: Mapping[str, _Preparer] = MappingProxyType(_PREPARERS), - export_anthropic_token: bool = True, ) -> None: """Validate, wire the environment, and hand off to the agent. @@ -506,9 +536,7 @@ def run_agent( env: Final = MappingProxyType( { - **build_agent_env( - env_before_sync, base_url, api_key, profiles, export_anthropic_token=export_anthropic_token - ), + **build_agent_env(env_before_sync, base_url, api_key, profiles), **(_NO_EXTRA_ENV if isinstance(synced, ModelSyncSkipped) else synced), } ) @@ -546,26 +574,14 @@ def resolve_api_key(ctx: click.Context) -> str: _SKIP_VERIFY_HELP: Final = "Skip the pre-launch key check against the proxy." -def _helper_supplies_token( - ctx_obj: CliContextObj, base_url: str, profiles: frozenset[str], settings_path: Path -) -> bool: - if PROFILE_ANTHROPIC not in profiles or not ctx_obj.get("api_key_from_token_file"): - return False - return lite_api_key_helper_configured(base_url, settings_path) - - def _launch(ctx: click.Context, binary: str, args: Sequence[str], *, skip_verify: bool) -> None: ctx_obj: Final[CliContextObj] = ctx.obj base_url: Final = ctx_obj["base_url"] started_interactive: Final = _is_interactive() api_key: Final = resolve_api_key(ctx) - display_name, profiles = agent_profile(binary) - settings_path: Final = claude_settings_path(os.environ) - helper_supplies_token: Final = _helper_supplies_token(ctx_obj, base_url, profiles, settings_path) + display_name, _profiles = agent_profile(binary) click.echo(f"litellm: routing {display_name} through proxy at {base_url.rstrip('/')}") - if helper_supplies_token: - click.echo(f"litellm: {display_name} reads its key from the apiKeyHelper in {settings_path}") try: run_agent( @@ -574,7 +590,6 @@ def _launch(ctx: click.Context, binary: str, args: Sequence[str], *, skip_verify [binary, *args], skip_verify=skip_verify, reattach_terminal=(_restore_controlling_terminal if started_interactive else None), - export_anthropic_token=not helper_supplies_token, ) except AgentRunError as e: raise click.ClickException(str(e)) diff --git a/litellm/proxy/client/cli/commands/auth.py b/litellm/proxy/client/cli/commands/auth.py index 4f704afe9d6..98af32fa7aa 100644 --- a/litellm/proxy/client/cli/commands/auth.py +++ b/litellm/proxy/client/cli/commands/auth.py @@ -42,14 +42,13 @@ from litellm.litellm_core_utils.cli_token_utils import ( from .claude_settings import ( STARTING_MODEL_ROLE, - ApiKeyHelper, ClaudeSettingsError, KeepModel, + StaticToken, claude_settings_path, configure_claude_settings, configure_state_path, refuse_while_owned, - resolve_api_key_helper, settings_file_owners, ) from .pkce_login import ( @@ -784,13 +783,16 @@ def _render_and_prompt_for_team_selection(teams: list[CliTeam]) -> str | None: return None -def _configure_claude_code(base_url: str) -> None: - """Point Claude Code at base_url by patching the settings.json it reads, undoable with `lite unconfigure claude`.""" +def _configure_claude_code(base_url: str, api_key: str) -> None: + """Write the key this login just minted into Claude Code's settings.json as a static token, undoable with + `lite unconfigure claude`. The key expires with the login, so the flag is the re-wire step of each login + rather than a one-time setup: no apiKeyHelper is written, since Claude Code would spawn `lite` (and its + keychain probe) on every credential refresh to keep one fresh.""" settings_path: Final = claude_settings_path(os.environ) try: configure_claude_settings( base_url, - ApiKeyHelper(resolve_api_key_helper(base_url)), + StaticToken(api_key), KeepModel(), settings_path, configure_state_path(settings_path), @@ -800,22 +802,28 @@ def _configure_claude_code(base_url: str) -> None: raise click.ClickException(f"Logged in, but could not configure Claude Code: {e}") click.echo(f"\nConfigured Claude Code: {settings_path} now routes through {base_url.rstrip('/')}.") click.echo( + "This login's key is stored in the file, so run `lite login --config-claude` again after it expires. " "Your other Claude Code settings were left untouched. Restart Claude Code to pick this up. " f"Undo with `lite unconfigure claude`; `lite configure claude --model` sets {STARTING_MODEL_ROLE}." ) def _finish_login(base_url: str, api_key: str, config_claude: bool, stored: SecretSave) -> None: + """Claude Code is configured from the key in hand, so it does not wait on the CLI's own store: a login whose + token file or keychain refused it still has a usable key, and `--config-claude` asked for exactly that + key to be written into settings.json.""" from litellm.proxy.client.cli.interface import show_commands click.echo("\nLogin successful!") click.echo(f"JWT Token: {api_key[:20]}...") click.echo(storage_notice(stored)) + if config_claude: + _configure_claude_code(base_url, api_key) if isinstance(stored, (CredentialNotSaved, CredentialNotRecorded)): + if config_claude: + click.echo("Claude Code was configured with this key even though the CLI itself could not keep it.") return click.echo("You can now use the CLI without specifying --api-key") - if config_claude: - _configure_claude_code(base_url) click.echo("\n" + "=" * 60) show_commands() @@ -850,8 +858,8 @@ def _pkce_login(base_url: str, config_claude: bool, vault: SecretVault) -> None: is_flag=True, default=False, help=( - "After logging in, update ~/.claude/settings.json so Claude Code routes through this proxy. " - "Unrelated settings are preserved." + "After logging in, write this login's key into ~/.claude/settings.json so Claude Code routes through " + "this proxy; run it again after the key expires. Unrelated settings are preserved." ), ) @click.option( diff --git a/litellm/proxy/client/cli/commands/autoroute/commands.py b/litellm/proxy/client/cli/commands/autoroute/commands.py index 86701f186fd..5d91fc81350 100644 --- a/litellm/proxy/client/cli/commands/autoroute/commands.py +++ b/litellm/proxy/client/cli/commands/autoroute/commands.py @@ -1,5 +1,4 @@ import atexit -import json import secrets import signal import threading @@ -15,8 +14,10 @@ from ..claude_settings import ( CLAUDE_SETTINGS_PATH, ClaudeSettingsError, StaticToken, + install_statusline_script, load_json_or_empty, merge_claude_settings, + write_claude_settings, ) from ..up import BackupRecord as ClaudeBackupRecord from ..up import restore_claude_settings, write_backup @@ -151,6 +152,7 @@ def up(port: int) -> None: raise click.ClickException(str(e)) try: + status_line: Final = install_statusline_script() original_existed: Final = CLAUDE_SETTINGS_PATH.exists() original_settings: Final = load_json_or_empty(CLAUDE_SETTINGS_PATH) write_backup( @@ -158,11 +160,15 @@ def up(port: int) -> None: AUTOROUTE_BACKUP_PATH, ) merged: Final = merge_claude_settings( - original_settings, base_url, StaticToken(master_key), AUTOROUTER_MODEL_NAME, AUTOROUTER_MODEL_NAME + original_settings, + base_url, + StaticToken(master_key), + AUTOROUTER_MODEL_NAME, + AUTOROUTER_MODEL_NAME, + status_line=status_line, ) CLAUDE_SETTINGS_PATH.parent.mkdir(parents=True, exist_ok=True) - with secure_create(CLAUDE_SETTINGS_PATH) as f: - json.dump(merged, f, indent=2) + write_claude_settings(CLAUDE_SETTINGS_PATH, merged) except ClaudeSettingsError as e: terminate(process.pid) clear_pid_record() diff --git a/litellm/proxy/client/cli/commands/autoroute/config.py b/litellm/proxy/client/cli/commands/autoroute/config.py index 9dfc4ad079b..1f3ad34e3d9 100644 --- a/litellm/proxy/client/cli/commands/autoroute/config.py +++ b/litellm/proxy/client/cli/commands/autoroute/config.py @@ -162,6 +162,7 @@ def build_generated_model_list(config: AutorouteConfig) -> list[JsonValue]: complexity_router_config: Final[dict[str, JsonValue]] = { "tiers": {tier: list(models) for tier, models in config.tiers.items()}, "default_model": config.default_model, + "return_raw_model_name": True, } if isinstance(config.classifier, LLMClassifier): complexity_router_config["classifier_type"] = "llm" diff --git a/litellm/proxy/client/cli/commands/claude_settings.py b/litellm/proxy/client/cli/commands/claude_settings.py index d0ee507f0b2..e6231f3cac9 100644 --- a/litellm/proxy/client/cli/commands/claude_settings.py +++ b/litellm/proxy/client/cli/commands/claude_settings.py @@ -1,16 +1,18 @@ """Shared handling of Claude Code's ~/.claude/settings.json. `lite up` and `lite autoroute up` patch this file temporarily and restore it on -exit; `lite login --config-claude` and `lite configure claude` patch it -persistently and record how to undo it. All of them need the same merge and the -same apiKeyHelper command, and `up` already imports from `auth`, so the shared -parts live here rather than in any one command module. +exit; `lite configure claude` patches it persistently and records how to undo it. +All of them need the same merge, and `up` already imports from `auth`, so the +shared parts live here rather than in any one command module. The credential is +always a static token in `env.ANTHROPIC_AUTH_TOKEN`: Claude Code's `apiKeyHelper` +would spawn a `lite` process on every credential refresh, and that process touches +the keychain, so nothing here writes one; a helper left by an earlier version is +owned like any other key and stripped. """ import hashlib import json import shlex -import shutil import sys from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass @@ -27,13 +29,16 @@ from litellm.litellm_core_utils.private_json import ( discard_staged_json, ensure_private_dir, stage_private_json, + write_private_bytes, ) +from . import statusline_script from .cmd_quoting import quote_for_cmd ENV_KEY: Final = "env" API_KEY_HELPER_KEY: Final = "apiKeyHelper" MODEL_KEY: Final = "model" +STATUS_LINE_KEY: Final = "statusLine" ANTHROPIC_BASE_URL_KEY: Final = "ANTHROPIC_BASE_URL" ANTHROPIC_AUTH_TOKEN_KEY: Final = "ANTHROPIC_AUTH_TOKEN" ANTHROPIC_API_KEY_KEY: Final = "ANTHROPIC_API_KEY" @@ -41,6 +46,7 @@ ENABLE_TOOL_SEARCH_KEY: Final = "ENABLE_TOOL_SEARCH" ENABLE_TOOL_SEARCH_VALUE: Final = "true" ENABLE_GATEWAY_MODEL_DISCOVERY_KEY: Final = "CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY" ENABLE_GATEWAY_MODEL_DISCOVERY_VALUE: Final = "1" +ANTHROPIC_MODEL_KEY: Final = "ANTHROPIC_MODEL" ANTHROPIC_DEFAULT_MODEL_ENV_KEYS: Final = ( "ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL", @@ -53,19 +59,22 @@ OWNED_ENV_KEYS: Final = ( ANTHROPIC_BASE_URL_KEY, ANTHROPIC_AUTH_TOKEN_KEY, ANTHROPIC_API_KEY_KEY, + ANTHROPIC_MODEL_KEY, ) -OWNED_TOP_LEVEL_KEYS: Final = (API_KEY_HELPER_KEY, MODEL_KEY) +OWNED_TOP_LEVEL_KEYS: Final = (API_KEY_HELPER_KEY, MODEL_KEY, STATUS_LINE_KEY) OWNED_PATHS: Final = (*(f"{ENV_KEY}.{key}" for key in OWNED_ENV_KEYS), *OWNED_TOP_LEVEL_KEYS) _CREDENTIAL_ENV_KEYS: Final = frozenset((ANTHROPIC_API_KEY_KEY, ANTHROPIC_AUTH_TOKEN_KEY)) _CREDENTIAL_PATHS: Final = (*(f"{ENV_KEY}.{key}" for key in sorted(_CREDENTIAL_ENV_KEYS)), API_KEY_HELPER_KEY) _BASE_URL_PATH: Final = f"{ENV_KEY}.{ANTHROPIC_BASE_URL_KEY}" -STARTING_MODEL_ROLE: Final = "the /model picker's default row, the model Claude Code starts on" +_MODEL_PATHS: Final = (MODEL_KEY, f"{ENV_KEY}.{ANTHROPIC_MODEL_KEY}") +STARTING_MODEL_ROLE: Final = "the /model picker's default row, the model Claude Code starts and resumes on" CLAUDE_SETTINGS_PATH: Final = Path.home() / ".claude" / "settings.json" CLAUDE_CONFIG_DIR_ENV: Final = "CLAUDE_CONFIG_DIR" BACKUP_PATH: Final = Path.home() / ".litellm" / "claude_settings_backup.json" AUTOROUTE_BACKUP_PATH: Final = Path.home() / ".litellm" / "autorouter" / "claude_settings_backup.json" CONFIGURE_STATE_PATH: Final = Path.home() / ".litellm" / "claude_configure_state.json" +STATUSLINE_SCRIPT_PATH: Final = Path.home() / ".litellm" / "statusline.py" @dataclass(frozen=True, slots=True) @@ -123,16 +132,6 @@ class StaticToken: token: str -@dataclass(frozen=True, slots=True) -class ApiKeyHelper: - """A `lite auth print-token` command Claude Code runs per request, so a login renews in place.""" - - command: str - - -ClaudeCredential: TypeAlias = StaticToken | ApiKeyHelper - - @dataclass(frozen=True, slots=True) class KeepModel: """Leave the top-level `model` as it is, the user's or an earlier configure's (a re-login).""" @@ -145,7 +144,9 @@ class UnpinModel: @dataclass(frozen=True, slots=True) class StartOn: - """Pin the top-level `model`, the row Claude Code starts on.""" + """Pin `model` and `env.ANTHROPIC_MODEL`: the row Claude Code starts on, and the one a resumed session stays + on, since resume otherwise re-sends the transcript's served model, which a raw-model router made a tier + model the key may not reach.""" model: str @@ -259,6 +260,17 @@ def _write_target(settings_path: Path) -> Path: raise ClaudeSettingsError(f"Could not resolve {settings_path}: {e}") from e +def write_claude_settings(settings_path: Path, settings: Mapping[str, JsonValue]) -> None: + """The one way a settings document lands on disk: staged owner-only beside the target and renamed into + place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute up` and the + restores) may be carrying the credential, so none creates the file under the umask or truncates it.""" + target: Final = _write_target(settings_path) + try: + commit_staged_json(stage_private_json(str(target), settings), str(target)) + except OSError as e: + raise ClaudeSettingsError(f"Could not write {settings_path}: {e}") from e + + def _stage(path: Path, document: Mapping[str, object]) -> str: try: return stage_private_json(str(path), document) @@ -287,20 +299,49 @@ def _land( raise ClaudeSettingsError(f"Could not {'remove' if staged is None else 'write'} {path}: {e}") from e +def statusline_command(script_path: Path, platform: str = sys.platform) -> str: + """This interpreter, not a bare `python3`: it is the one the apiKeyHelper already depends on.""" + quote: Final = quote_for_cmd if platform.startswith("win") else shlex.quote + return " ".join(quote(token) for token in (sys.executable, str(script_path))) + + +def install_statusline_script(script_path: Path | None = None) -> str: + target: Final = script_path or STATUSLINE_SCRIPT_PATH + try: + ensure_private_dir(target.parent) + write_private_bytes(str(target), Path(statusline_script.__file__).read_bytes()) + except OSError as e: + raise ClaudeSettingsError(f"Could not install the status line script at {target}: {e}") from e + return statusline_command(target) + + +def with_status_line(settings: Mapping[str, JsonValue], command: str) -> Mapping[str, JsonValue]: + """Ours is recognised by the script it runs, so a re-install under another interpreter is still ours.""" + existing: Final = settings.get(STATUS_LINE_KEY) + existing_command: Final = existing.get("command") if isinstance(existing, dict) else None + ours: Final = existing is None or (isinstance(existing_command, str) and command.split()[-1] in existing_command) + if not ours: + return settings + entry: Final = dict((("type", "command"), ("command", command))) # mutable-ok: JSON document + return dict(chain(settings.items(), ((STATUS_LINE_KEY, entry),))) # mutable-ok: JSON document + + def merge_claude_settings( settings: Mapping[str, JsonValue], base_url: str, - credential: ClaudeCredential, + credential: StaticToken, default_model: str | None = None, tier_model: str | None = None, + *, + status_line: str | None = None, ) -> Mapping[str, JsonValue]: """Return a new settings mapping wired to route Claude Code through the proxy. - A StaticToken lands in env.ANTHROPIC_AUTH_TOKEN, an ApiKeyHelper in the top-level apiKeyHelper; - the other credential slots are removed either way, since Claude Code given two credentials may - send the wrong one. ENABLE_TOOL_SEARCH and CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY get their - defaults only when missing. `default_model` is the top-level `model`, the row Claude Code starts - on; `tier_model` is `lite autoroute up`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one + The token lands in env.ANTHROPIC_AUTH_TOKEN; the other credential slots (a stray ANTHROPIC_API_KEY, + an apiKeyHelper) are removed, since Claude Code given two credentials may send the wrong one. + ENABLE_TOOL_SEARCH and CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY get their defaults only when + missing. `default_model` is the top-level `model` and env.ANTHROPIC_MODEL (see StartOn); + `tier_model` is `lite autoroute up`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one group. Apart from those tier keys, exactly OWNED_PATHS are touched. """ raw_env: Final = settings.get(ENV_KEY, {}) @@ -312,60 +353,24 @@ def merge_claude_settings( (ENABLE_GATEWAY_MODEL_DISCOVERY_KEY, ENABLE_GATEWAY_MODEL_DISCOVERY_VALUE), ), ((key, value) for key, value in current_env.items() if key not in _CREDENTIAL_ENV_KEYS), - ((ANTHROPIC_BASE_URL_KEY, base_url.rstrip("/")),), - ((ANTHROPIC_AUTH_TOKEN_KEY, credential.token),) if isinstance(credential, StaticToken) else (), + ((ANTHROPIC_BASE_URL_KEY, base_url.rstrip("/")), (ANTHROPIC_AUTH_TOKEN_KEY, credential.token)), + ((ANTHROPIC_MODEL_KEY, default_model),) if default_model is not None else (), ((key, tier_model) for key in ANTHROPIC_DEFAULT_MODEL_ENV_KEYS if tier_model is not None), ) ) return dict( # mutable-ok: JSON document handed to json.dump, which rejects a read-only mapping chain( - ((key, value) for key, value in settings.items() if key not in (API_KEY_HELPER_KEY, ENV_KEY)), + ( + (key, value) + for key, value in (with_status_line(settings, status_line) if status_line else settings).items() + if key not in (API_KEY_HELPER_KEY, ENV_KEY) + ), ((ENV_KEY, env),), - ((API_KEY_HELPER_KEY, credential.command),) if isinstance(credential, ApiKeyHelper) else (), ((MODEL_KEY, default_model),) if default_model is not None else (), ) ) -def resolve_api_key_helper(base_url: str, platform: str = sys.platform) -> str: - """Build the shell command Claude Code should run for its apiKeyHelper. - - Claude Code hands the string to the system shell, `sh` on POSIX and cmd.exe - on Windows, so every token is quoted for the shell that will read it. - - Resolves `lite` to an absolute path so the helper works regardless of the - PATH visible to whatever subprocess Claude Code spawns it from. Passing - --base-url explicitly (rather than relying on the bare invocation Claude - Code would otherwise use) makes `print-token` enforce that the cached - token was actually issued for this proxy -- without it, a token minted - for a different, previously-logged-into proxy would be handed to - whichever server the settings currently point at. - - --base-url belongs to the top-level `lite` group, so it has to precede the - subcommand; click rejects it outright after `print-token`. - """ - lite_path: Final = shutil.which("lite") - if lite_path is None: - raise ClaudeSettingsError( - "Could not find `lite` on your PATH. Claude Code's apiKeyHelper needs an absolute path to it." - ) - quote: Final = quote_for_cmd if platform.startswith("win") else shlex.quote - return " ".join(quote(token) for token in (lite_path, "--base-url", base_url, "auth", "print-token")) - - -def lite_api_key_helper_configured(base_url: str, settings_path: Path) -> bool: - """Whether settings_path already carries the apiKeyHelper `lite login --config-claude` writes for base_url. - - Only an exact match counts: a helper for another proxy, a hand-written one, or - settings that cannot be read leave the caller on the env-token path. - """ - try: - configured_helper: Final = load_json_or_empty(settings_path).get(API_KEY_HELPER_KEY) - return configured_helper == resolve_api_key_helper(base_url.rstrip("/")) - except ClaudeSettingsError: - return False - - def _owned(container: Mapping[str, JsonValue], key: str) -> OwnedValue: return OwnedValue(present=key in container, value=container.get(key)) @@ -460,12 +465,13 @@ def read_configure_receipt(state_path: Path) -> ConfigureReceipt | None: def configure_claude_settings( base_url: str, - credential: ClaudeCredential, + credential: StaticToken, model: ModelChoice, settings_path: Path, state_path: Path, owners: Sequence[SettingsFileOwner], commit: Callable[[str, str], None] = commit_staged_json, + script_path: Path | None = None, ) -> None: """Persistently route Claude Code through base_url, recording how to undo it. @@ -474,19 +480,28 @@ def configure_claude_settings( discards the staged settings, and a settings rename that fails after the receipt landed puts the earlier receipt back (or removes the new one), so the receipt on disk never describes settings that were not written. `model`: StartOn pins the starting model, UnpinModel lets go of a pin an - earlier configure made (never of the user's own), KeepModel leaves it alone (a re-login). + earlier configure made (never of the user's own), KeepModel leaves it alone (a re-login). The + status line script is installed and registered under `statusLine` unless the user runs their own; + the receipt owns that key like any other, so unconfigure removes only ours. """ refuse_while_owned(settings_path, owners) current: Final = load_json_or_empty(settings_path) _env_object(current, settings_path) earlier: Final = read_configure_receipt(state_path) - existing: Final = ( - _with(current, MODEL_KEY, earlier.previous[MODEL_KEY]) - if isinstance(model, UnpinModel) and earlier is not None and _ours(current, MODEL_KEY, earlier) - else current + unpinned: Final = MappingProxyType( + { + path: earlier.previous[path] + for path in _MODEL_PATHS + if isinstance(model, UnpinModel) and earlier is not None and _ours(current, path, earlier) + } ) + existing: Final = _with_all(current, unpinned) merged: Final = merge_claude_settings( - existing, base_url, credential, model.model if isinstance(model, StartOn) else None + existing, + base_url, + credential, + model.model if isinstance(model, StartOn) else None, + status_line=install_statusline_script(script_path), ) receipt: Final = _receipt(current, merged, earlier, settings_path.exists()) target: Final = _write_target(settings_path) @@ -581,6 +596,7 @@ __all__ = ( "ANTHROPIC_AUTH_TOKEN_KEY", "ANTHROPIC_BASE_URL_KEY", "ANTHROPIC_DEFAULT_MODEL_ENV_KEYS", + "ANTHROPIC_MODEL_KEY", "API_KEY_HELPER_KEY", "AUTOROUTE_BACKUP_PATH", "BACKUP_PATH", @@ -598,8 +614,8 @@ __all__ = ( "OWNED_TOP_LEVEL_KEYS", "SETTINGS_FILE_OWNERS", "STARTING_MODEL_ROLE", - "ApiKeyHelper", - "ClaudeCredential", + "STATUSLINE_SCRIPT_PATH", + "STATUS_LINE_KEY", "ClaudeSettingsError", "ConfigureReceipt", "KeepModel", @@ -614,12 +630,11 @@ __all__ = ( "claude_settings_path", "configure_claude_settings", "configure_state_path", - "lite_api_key_helper_configured", "load_json_or_empty", "merge_claude_settings", "read_configure_receipt", "refuse_while_owned", - "resolve_api_key_helper", "settings_file_owners", "unconfigure_claude_settings", + "write_claude_settings", ) diff --git a/litellm/proxy/client/cli/commands/configure.py b/litellm/proxy/client/cli/commands/configure.py index 9d329f8d8f2..4acf94e16f9 100644 --- a/litellm/proxy/client/cli/commands/configure.py +++ b/litellm/proxy/client/cli/commands/configure.py @@ -18,11 +18,9 @@ from litellm.proxy.common_utils.model_listing_utils import ( GATEWAY_CLIENT_HEADER, ) -from .auth import CliContextObj, context_secret_vault, get_stored_api_key +from .auth import CliContextObj from .claude_settings import ( STARTING_MODEL_ROLE, - ApiKeyHelper, - ClaudeCredential, ClaudeSettingsError, ModelChoice, StartOn, @@ -33,12 +31,10 @@ from .claude_settings import ( configure_claude_settings, configure_state_path, refuse_while_owned, - resolve_api_key_helper, settings_file_owners, unconfigure_claude_settings, ) from .pi import ListedModel, ListingFailure, PiSyncError, fetch_model_listing -from .up import ensure_fresh_login _LISTED_MODELS_SHOWN: Final = 20 _CLAUDE_TARGET: Final = "claude" @@ -54,24 +50,21 @@ _MODEL_OPTION_HELP: Final = ( ) -def resolve_credential(ctx: click.Context, api_key: str | None) -> tuple[ClaudeCredential, str]: - """The credential to write and the key to check the proxy with. +def resolve_credential(ctx: click.Context, api_key: str | None) -> StaticToken: + """The long-lived key written into settings.json: --api-key, `lite --api-key` or LITELLM_PROXY_API_KEY. - An explicit key (--api-key, `lite --api-key`, LITELLM_PROXY_API_KEY) is long-lived and goes - into settings.json as a static token. Without one, the stored `lite login` credential is used - the way `lite login --config-claude` uses it, through apiKeyHelper, since it expires within a - day and renews in place there; a missing or stale login is refreshed first, as `lite up` does. + A `lite login` credential is never written: it expires within a day, and keeping it fresh would mean + Claude Code running `lite` through `apiKeyHelper` on every credential refresh. """ ctx_obj: Final[CliContextObj] = ctx.obj explicit: Final = api_key or (None if ctx_obj.get("api_key_from_token_file") else ctx_obj.get("api_key")) - if explicit: - return StaticToken(explicit), explicit - base_url: Final = ctx_obj["base_url"] - ensure_fresh_login(ctx) - stored: Final = get_stored_api_key(expected_base_url=base_url, vault=context_secret_vault(ctx)) - if not stored: - raise ClaudeSettingsError("Login did not produce a usable token.") - return ApiKeyHelper(resolve_api_key_helper(base_url)), stored + if not explicit: + raise ClaudeSettingsError( + "`lite configure claude` needs a long-lived virtual key: pass --api-key, `lite --api-key`, or set " + "LITELLM_PROXY_API_KEY. Your `lite login` credential expires within a day, so it is not written " + "into Claude Code's settings." + ) + return StaticToken(explicit) @dataclass(frozen=True, slots=True) @@ -83,22 +76,22 @@ class _Listing: return tuple(model.id for model in self.models) -def _start(ctx: click.Context, api_key: str | None) -> tuple[ClaudeCredential, _Listing]: +def _start(ctx: click.Context, api_key: str | None) -> tuple[StaticToken, _Listing]: """Every configure path begins the same way: the local ownership check first, so a `lite up` - session is refused before any login prompt or request, then the credential, then the listing.""" + session is refused before any request, then the credential, then the listing.""" settings_path: Final = claude_settings_path(os.environ) try: refuse_while_owned(settings_path, settings_file_owners(settings_path)) - credential, key = resolve_credential(ctx, api_key) + credential: Final = resolve_credential(ctx, api_key) except ClaudeSettingsError as e: raise click.ClickException(str(e)) - return credential, _listed_models(ctx.obj["base_url"], key) + return credential, _listed_models(ctx.obj["base_url"], credential.token) def _listing_error(base_url: str, error: PiSyncError) -> str: """The hint that fits how the listing failed: only an unreachable proxy gets the "is it running" question.""" if error.kind is ListingFailure.REJECTED: - return f"LiteLLM rejected your key (HTTP {error.status}). Run `lite login` to refresh it, or pass a valid --api-key." + return f"LiteLLM rejected your key (HTTP {error.status}). Pass a valid --api-key." if error.kind is ListingFailure.UNREACHABLE: return f"{error.message} Is the proxy at {base_url} running, and is --base-url (or LITELLM_PROXY_URL) correct?" if error.kind is ListingFailure.EMPTY: @@ -122,7 +115,7 @@ def _model_choice(model: str | None) -> ModelChoice: return StartOn(model) if model is not None else UnpinModel() -def _apply_claude(ctx: click.Context, credential: ClaudeCredential, listing: _Listing, model: str | None) -> None: +def _apply_claude(ctx: click.Context, credential: StaticToken, listing: _Listing, model: str | None) -> None: ctx_obj: Final[CliContextObj] = ctx.obj base_url: Final = ctx_obj["base_url"] listed: Final = listing.ids @@ -148,16 +141,13 @@ def _apply_claude(ctx: click.Context, credential: ClaudeCredential, listing: _Li in_picker: Final = sum(1 for listed_model in listed if CLAUDE_CODE_PICKER_PATTERN.search(listed_model)) click.echo(f"Configured Claude Code: {settings_path} now routes through {base_url}.") - click.echo( - "Credential: your virtual key, stored in the file as ANTHROPIC_AUTH_TOKEN." - if isinstance(credential, StaticToken) - else "Credential: your `lite login`, read through apiKeyHelper on every request, so a later login renews it." - ) + click.echo("Credential: your virtual key, stored in the file as ANTHROPIC_AUTH_TOKEN.") click.echo( f"Starting model: {starting} ({STARTING_MODEL_ROLE}); switch any time with /model." if starting is not None else "Starting model: not pinned (Claude Code's default, or a model you set yourself); switch with /model, or " - "pass --model to start on a proxy model." + "pass --model to start on a proxy model. Without a pin, a resumed session re-sends the model its transcript " + "recorded, which behind a raw-model auto-router is the tier model." ) click.echo( f"/model will list all {len(listed)} of the proxy's models." @@ -166,7 +156,7 @@ def _apply_claude(ctx: click.Context, credential: ClaudeCredential, listing: _Li "'claude' or 'anthropic', and this proxy does not list the rest under such names." ) click.echo("Start `claude` from any terminal. Undo with `lite unconfigure claude`.") - if isinstance(credential, StaticToken) and settings_path.is_symlink(): + if settings_path.is_symlink(): click.echo( f"Note: {settings_path} is a symlink to {settings_path.resolve()}, so your key now lives in " "that file; keep it out of version control.", @@ -235,16 +225,16 @@ def unconfigure_group() -> None: "api_key", default=None, help="Long-lived LiteLLM virtual key written into Claude Code's settings. Defaults to the `lite --api-key` / " - "LITELLM_PROXY_API_KEY value; with neither, your `lite login` credential is used through apiKeyHelper.", + "LITELLM_PROXY_API_KEY value; required, since a `lite login` credential expires within a day.", ) @click.option("--model", default=None, help=_MODEL_OPTION_HELP) @click.pass_context def configure_claude(ctx: click.Context, api_key: str | None, model: str | None) -> None: """Route every Claude Code session through your LiteLLM proxy until `lite unconfigure claude`. - Patches ~/.claude/settings.json in place: the proxy URL, your credential (a virtual key as a - static token, or your `lite login` through apiKeyHelper), and gateway model discovery so - /model lists the proxy's models; --model picks the one Claude Code starts on. Every other + Patches ~/.claude/settings.json in place: the proxy URL, your virtual key as a static token, + and gateway model discovery so /model lists the proxy's models; --model picks the one Claude + Code starts on and resumes with. Every other setting is kept, and what changed is recorded so `lite unconfigure claude` can put it back. Assumes the proxy is already running. """ diff --git a/litellm/proxy/client/cli/commands/statusline_script.py b/litellm/proxy/client/cli/commands/statusline_script.py new file mode 100644 index 00000000000..a8abeb68978 --- /dev/null +++ b/litellm/proxy/client/cli/commands/statusline_script.py @@ -0,0 +1,392 @@ +"""Claude Code status line and Codex Stop hook for auto-routed sessions. + +`lite` copies this file verbatim to ~/.litellm/statusline.py and registers it as Claude +Code's `statusLine` command and as Codex's `[[hooks.Stop]]` command, so it must stay +standard-library only and must never import litellm. Claude Code re-runs it on every +status refresh (about every 300ms while typing), so the proxy is asked at most once per +TTL per session and every other refresh is served from a small on-disk cache that holds +only the proxy's answer, never the key. + +Claude Code pipes a JSON payload on stdin (session_id, transcript_path, model); the routed +model is the `message.model` of the latest foreground assistant line in the transcript, +which is the proxy's response `model` field. That only names the tier model when the +auto-router deployment sets `return_raw_model_name: true`; otherwise it is the alias the +client requested. Codex pipes its Stop event instead (hook_event_name, session_id) and has +no transcript to read, so the routed model comes from the proxy's session record and the +result is printed as a `systemMessage` for the transcript. The proxy key is read from the +agent's own environment (the static token `lite configure claude` writes); nothing here +spawns a credential helper. + +Cost figures come from GET /auto_router/session on the proxy, which reads the per-session +rollup written by the spend flush. That flush is asynchronous, so a turn's cost lands a +second or two after the turn; the cache TTL absorbs it. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +import tempfile +import time +import urllib.error +import urllib.request +from collections.abc import Callable, Mapping +from pathlib import Path +from types import MappingProxyType +from typing import IO, Final, NamedTuple, Protocol +from urllib.parse import urlencode + +SESSION_ENDPOINT: Final = "/auto_router/session" +CACHE_TTL_SECONDS: Final = 5.0 +FETCH_TIMEOUT_SECONDS: Final = 3 +BAR_WIDTH: Final = 24 +BAR_FULL: Final = "\u2588" +BAR_EMPTY: Final = "\u2591" +SEPARATOR: Final = " \u00b7 " +TRANSCRIPT_SCAN_LIMIT_BYTES: Final = 4 * 1024 * 1024 +CLAUDE_BASE_URL_ENV_KEYS: Final = ("ANTHROPIC_BASE_URL",) +CLAUDE_API_KEY_ENV_KEYS: Final = ("ANTHROPIC_AUTH_TOKEN", "ANTHROPIC_API_KEY") +CODEX_BASE_URL_ENV_KEYS: Final = ("OPENAI_BASE_URL",) +CODEX_API_KEY_ENV_KEYS: Final = ("OPENAI_API_KEY",) +CODEX_STOP_EVENT: Final = "Stop" +SYNTHETIC_MODEL: Final = "" +LITELLM_LABEL: Final = "LiteLLM" +RESET: Final = "\033[0m" +BOLD: Final = "\033[1m" +DIM: Final = "\033[90m" +LITELLM_COLOR: Final = "\033[38;2;79;70;229m" +BASELINE_COLOR: Final = "\033[38;2;217;119;87m" +EMPTY: Final[Mapping[str, object]] = MappingProxyType({}) +EMPTY_ENV: Final[Mapping[str, str]] = MappingProxyType({}) + + +class Session(NamedTuple): + router_name: str + last_model: str + spend: float + baseline_spend: float + baseline_model: str | None + + +class Credentials(NamedTuple): + base_url: str + api_key: str + + @property + def usable(self) -> bool: + return bool(self.base_url and self.api_key) + + +class Fetched(NamedTuple): + session: Session | None + definitive: bool + + +class Fetch(Protocol): + def __call__(self, credentials: Credentials, session_id: str) -> Fetched: ... + + +def as_mapping(value: object) -> Mapping[str, object]: + return value if isinstance(value, dict) else EMPTY + + +def as_str(value: object) -> str: + return value if isinstance(value, str) else "" + + +def printable(value: object) -> str: + """Labels come from the transcript, the proxy, and Claude Code's model cache, none of which this script + controls, and every one is written to a terminal: a control character (ESC, BEL, C1) in a model name + could redraw the screen or set the clipboard, so only printable text survives.""" + return "".join(character for character in as_str(value) if character.isprintable()) + + +def load_json(raw: bytes | str) -> object: + try: + return json.loads(raw) + except ValueError: + return None + + +def resolve_base_url(env: Mapping[str, str], keys: tuple[str, ...]) -> str: + raw: Final = next((env[key] for key in keys if env.get(key)), "").strip().rstrip("/") + return raw.removesuffix("/v1") + + +def resolve_api_key(env: Mapping[str, str], keys: tuple[str, ...]) -> str: + return next((env[key] for key in keys if env.get(key)), "").strip() + + +def claude_credentials(env: Mapping[str, str]) -> Credentials: + """Claude Code's own resolution order, so the key-scoped lookup runs as the principal that wrote the rows: + ANTHROPIC_AUTH_TOKEN, then ANTHROPIC_API_KEY. A `lite` variable such as LITELLM_PROXY_API_KEY is not a + key Claude Code ever sends, so honoring it would ask as someone else. An apiKeyHelper is never run: a + status line refreshes every few hundred milliseconds, and spawning a credential helper that often is + how a keychain prompt ends up on screen a hundred times.""" + return Credentials(resolve_base_url(env, CLAUDE_BASE_URL_ENV_KEYS), resolve_api_key(env, CLAUDE_API_KEY_ENV_KEYS)) + + +def codex_credentials(env: Mapping[str, str]) -> Credentials: + return Credentials(resolve_base_url(env, CODEX_BASE_URL_ENV_KEYS), resolve_api_key(env, CODEX_API_KEY_ENV_KEYS)) + + +def _transcript_line_model(line: bytes) -> str: + """A `` model is Claude Code's own marker for a locally produced message (an API error, a + resume note), not a served model, so it is skipped like a sidechain line.""" + item: Final = as_mapping(load_json(line)) + if item.get("type") != "assistant" or item.get("isSidechain") is True or item.get("agentId"): + return "" + model: Final = printable(as_mapping(item.get("message")).get("model")) + return "" if model == SYNTHETIC_MODEL else model + + +def latest_transcript_model(transcript_path: str) -> str: + if not transcript_path: + return "" + try: + with Path(transcript_path).open("rb") as transcript: + size: Final = transcript.seek(0, os.SEEK_END) + transcript.seek(max(0, size - TRANSCRIPT_SCAN_LIMIT_BYTES)) + tail: Final = transcript.read() + except OSError: + return "" + return next((model for line in reversed(tail.split(b"\n")) if (model := _transcript_line_model(line))), "") + + +def model_label(model: str, config_dir: Path) -> str: + bare: Final = model.rsplit("/", 1)[-1] + try: + raw: Final = (config_dir / "cache" / "gateway-models.json").read_bytes() + except OSError: + return bare + listed: Final = as_mapping(load_json(raw)).get("models") + if not isinstance(listed, list): + return bare + entries: Final = tuple(as_mapping(entry) for entry in listed) + return next( + ( + printable(entry.get("display_name")) + for entry in entries + if entry.get("id") in (model, bare) and printable(entry.get("display_name")) + ), + bare, + ) + + +def baseline_label(model: str, config_dir: Path) -> str: + labelled: Final = model_label(model, config_dir) + if labelled != model.rsplit("/", 1)[-1]: + return labelled + return " ".join(word.capitalize() for word in labelled.replace("-", " ").split()) + + +def fetch_session(credentials: Credentials, session_id: str) -> Fetched: + """Any 4xx is this credential's definite answer (no row, no access, expired login) and is cached for the + TTL; a 5xx or transport failure is not, so the next refresh tries again.""" + query: Final = urlencode((("session_id", session_id),)) + request: Final = urllib.request.Request( + f"{credentials.base_url}{SESSION_ENDPOINT}?{query}", + headers={ # mutable-ok: urllib.request.Request takes a dict + "Authorization": f"Bearer {credentials.api_key}", + "Accept": "application/json", + }, + ) + try: + with urllib.request.urlopen(request, timeout=FETCH_TIMEOUT_SECONDS) as response: + raw: Final[bytes] = response.read() + except urllib.error.HTTPError as error: + return Fetched(session=None, definitive=400 <= error.code < 500) + except (urllib.error.URLError, OSError): + return Fetched(session=None, definitive=False) + session: Final = _session_from_payload(as_mapping(load_json(raw))) + return Fetched(session=session, definitive=session is not None) + + +def _session_from_payload(payload: Mapping[str, object]) -> Session | None: + router_name: Final = printable(payload.get("router_name")) + last_model: Final = printable(payload.get("last_model")) + spend: Final = payload.get("spend") + baseline_spend: Final = payload.get("baseline_spend") + if not router_name or not last_model: + return None + if not isinstance(spend, (int, float)) or not isinstance(baseline_spend, (int, float)): + return None + return Session( + router_name=router_name, + last_model=last_model, + spend=float(spend), + baseline_spend=float(baseline_spend), + baseline_model=printable(payload.get("baseline_model")) or None, + ) + + +def cache_path(cache_dir: Path, credentials: Credentials, session_id: str) -> Path: + identity: Final = "\n".join((credentials.base_url, credentials.api_key, session_id)) + return cache_dir / hashlib.sha256(identity.encode()).hexdigest() + + +def load_session( + credentials: Credentials, + session_id: str, + cache_dir: Path, + fetch: Fetch = fetch_session, + now: Callable[[], float] = time.time, +) -> Session | None: + path: Final = cache_path(cache_dir, credentials, session_id) + cached: Final = _read_cache(path) + fetched_at: Final = cached.get("fetched_at") + if isinstance(fetched_at, (int, float)) and now() - fetched_at < CACHE_TTL_SECONDS: + return _session_from_payload(as_mapping(cached.get("session"))) + fetched: Final = fetch(credentials, session_id) + if fetched.definitive: + _write_cache(path, fetched.session, now()) + return fetched.session + + +NOFOLLOW: Final = getattr(os, "O_NOFOLLOW", 0) + + +def cache_dir_name() -> str: + return f"litellm-statusline-{os.getuid()}" if hasattr(os, "getuid") else "litellm-statusline" + + +def _own_private_dir(directory: Path) -> bool: + """A shared temp root lets another local user pre-create the directory, so it must be ours and private + before anything is read or written under it. Windows has no uids or POSIX mode bits and a per-user temp + directory already, so there it only has to exist and not be a link.""" + try: + directory.mkdir(mode=0o700, parents=True, exist_ok=True) + status: Final = directory.lstat() + except OSError: + return False + if not os.path.isdir(directory) or os.path.islink(directory): + return False + if not hasattr(os, "getuid"): + return True + return status.st_uid == os.getuid() and not status.st_mode & 0o077 + + +def _read_cache(path: Path) -> Mapping[str, object]: + if not _own_private_dir(path.parent): + return EMPTY + try: + descriptor: Final = os.open(path, os.O_RDONLY | NOFOLLOW) + with os.fdopen(descriptor, "rb") as handle: + return as_mapping(load_json(handle.read())) + except OSError: + return EMPTY + + +def _write_cache(path: Path, session: Session | None, fetched_at: float) -> None: + """Staged beside the entry and renamed into place, so a refresh reading the entry never sees a torn write.""" + entry: Final = session._asdict() if session else None + body: Final = json.dumps({"fetched_at": fetched_at, "session": entry}) # mutable-ok: json.dumps takes a dict + if not _own_private_dir(path.parent): + return + try: + descriptor, staged = tempfile.mkstemp(dir=path.parent, prefix=".tmp-") + except OSError: + return + try: + with os.fdopen(descriptor, "w") as handle: + handle.write(body) + os.replace(staged, path) + except OSError: + Path(staged).unlink(missing_ok=True) + + +def _bar(fraction: float, color: str, width: int, use_color: bool) -> str: + filled: Final = round(max(0.0, min(1.0, fraction)) * width) + if not use_color: + return BAR_FULL * filled + BAR_EMPTY * (width - filled) + return f"{color}{BAR_FULL * filled}{DIM}{BAR_EMPTY * (width - filled)}{RESET}" + + +def render(model: str, session: Session | None, config_dir: Path, use_color: bool, bar_width: int = BAR_WIDTH) -> str: + def paint(code: str, text: str) -> str: + return f"{code}{text}{RESET}" if use_color else text + + routed: Final = paint(BOLD, f"Routed to: {model}") + if session is None: + return routed + header: Final = f"{session.router_name}{SEPARATOR}{routed}" + if session.baseline_model is None or session.baseline_spend <= 0: + return header + reference: Final = baseline_label(session.baseline_model, config_dir) + pct: Final = (session.baseline_spend - session.spend) / session.baseline_spend * 100 + delta: Final = paint(LITELLM_COLOR, f"{'-' if pct >= 0 else '+'}{abs(round(pct))}% vs {reference}") + peak: Final = max(session.spend, session.baseline_spend) + label_width: Final = max(len(LITELLM_LABEL), len(reference)) + rows: Final = ( + (LITELLM_LABEL, session.spend, LITELLM_COLOR), + (reference, session.baseline_spend, BASELINE_COLOR), + ) + lines: Final = ( + f"{paint(DIM, label.ljust(label_width))} {_bar(amount / peak, color, bar_width, use_color)} " + f"{paint(DIM, f'${amount:.2f}')}" + for label, amount, color in rows + ) + return "\n".join((f"{header} {delta}", *lines)) + + +def color_enabled(env: Mapping[str, str]) -> bool: + return env.get("NO_COLOR") is None and env.get("TERM", "") not in ("", "dumb") + + +def status_line( + payload: Mapping[str, object], env: Mapping[str, str], config_dir: Path, cache_dir: Path, fetch: Fetch +) -> str: + fallback: Final = printable(as_mapping(payload.get("model")).get("display_name")) + served: Final = latest_transcript_model(as_str(payload.get("transcript_path"))) + if not served: + return fallback or "claude" + label: Final = model_label(served, config_dir) + session_id: Final = as_str(payload.get("session_id")) + credentials: Final = claude_credentials(env) + if not session_id or not credentials.usable: + return render(label, None, config_dir, color_enabled(env)) + session: Final = load_session(credentials, session_id, cache_dir, fetch) + return render(label, session, config_dir, color_enabled(env)) + + +def codex_stop_message( + payload: Mapping[str, object], env: Mapping[str, str], config_dir: Path, cache_dir: Path, fetch: Fetch +) -> str: + """No cache here: the Stop hook runs once per turn, and a first turn's cached absence would hide the record + the next turn finds.""" + session_id: Final = as_str(payload.get("session_id")) + credentials: Final = codex_credentials(env) + if not session_id or not credentials.usable: + return "" + session: Final = fetch(credentials, session_id).session + if session is None: + return "" + text: Final = render(model_label(session.last_model, config_dir), session, config_dir, use_color=False) + return json.dumps({"systemMessage": f"\n{text}"}) # mutable-ok: json.dumps takes a dict + + +def run(stdin: IO[str], stdout: IO[str], env: Mapping[str, str], fetch: Fetch = fetch_session) -> None: + """A failure renders each mode's own quiet fallback: Claude Code gets the label it already knows, Codex gets + nothing at all rather than a bare string it would reject as hook JSON.""" + body: Final = as_mapping(load_json(stdin.read())) + codex: Final = body.get("hook_event_name") == CODEX_STOP_EVENT + config_dir: Final = Path(env.get("CLAUDE_CONFIG_DIR") or Path.home() / ".claude") + cache_dir: Final = ( + Path(env.get("TMPDIR") or env.get("TEMP") or env.get("TMP") or tempfile.gettempdir()) / cache_dir_name() + ) + try: + text: Final = ( + codex_stop_message(body, env, config_dir, cache_dir, fetch) + if codex + else status_line(body, env, config_dir, cache_dir, fetch) + ) + except Exception: # noqa: BLE001 # a status line must never break the agent session + stdout.write("" if codex else printable(as_mapping(body.get("model")).get("display_name")) or "claude") + return + stdout.write(text) + + +if __name__ == "__main__": + run(sys.stdin, sys.stdout, os.environ) diff --git a/litellm/proxy/client/cli/commands/up.py b/litellm/proxy/client/cli/commands/up.py index ffece87ab83..f2624797a5f 100644 --- a/litellm/proxy/client/cli/commands/up.py +++ b/litellm/proxy/client/cli/commands/up.py @@ -23,11 +23,12 @@ from .auth import CliContextObj, context_secret_vault, get_stored_api_key, load_ from .claude_settings import ( BACKUP_PATH, CLAUDE_SETTINGS_PATH, - ApiKeyHelper, ClaudeSettingsError, + StaticToken, + install_statusline_script, load_json_or_empty, merge_claude_settings, - resolve_api_key_helper, + write_claude_settings, ) @@ -98,8 +99,7 @@ def restore_claude_settings(settings_path: Path | None = None, backup_path: Path return None if record.existed and record.content is not None: resolved_settings_path.parent.mkdir(parents=True, exist_ok=True) - with open(resolved_settings_path, "w") as f: - json.dump(record.content, f, indent=2) + write_claude_settings(resolved_settings_path, record.content) elif resolved_settings_path.exists(): resolved_settings_path.unlink() resolved_backup_path.unlink() @@ -134,13 +134,10 @@ def ensure_fresh_login(ctx: click.Context) -> None: pkce: Final = _stored_login_is_pkce(vault) login_command: Final = "lite login --pkce" if pkce else "lite login" if not sys.stdin.isatty(): - raise UpError( - f"No fresh LiteLLM login found for this proxy. Run `{login_command}` first (apiKeyHelper " - "reads this token on every Claude Code request)." - ) + raise UpError(f"No fresh LiteLLM login found for this proxy. Run `{login_command}` first.") click.echo("No fresh LiteLLM login found for this proxy; starting login...") - ctx.invoke(login, pkce=pkce) + ctx.invoke(login, config_claude=False, pkce=pkce) if not _usable_login(get_stored_api_key(expected_base_url=base_url, vault=vault), vault): raise UpError("Login did not produce a usable token.") @@ -162,7 +159,9 @@ def up(ctx: click.Context) -> None: """Route every Claude Code session through your LiteLLM proxy until stopped. Patches ~/.claude/settings.json so Claude Code picks up the proxy on its own - next startup, from any terminal -- no need to launch it through `lite`. + next startup, from any terminal -- no need to launch it through `lite`. The + key written is the one this command resolved (your fresh `lite login`, or an + explicit --api-key), copied in as a static token for as long as `up` runs. Press Ctrl-C to stop and restore your original settings. Assumes the proxy is already running (this does not start one for you). Cursor is not supported: it has no equivalent file-based config to patch. @@ -180,7 +179,7 @@ def up(ctx: click.Context) -> None: "running (or crashed without cleanup). Run `lite down` first." ) - api_key_helper: Final = resolve_api_key_helper(base_url) + status_line: Final = install_statusline_script() original_existed: Final = CLAUDE_SETTINGS_PATH.exists() original_settings: Final = load_json_or_empty(CLAUDE_SETTINGS_PATH) write_backup( @@ -191,9 +190,10 @@ def up(ctx: click.Context) -> None: ) CLAUDE_SETTINGS_PATH.parent.mkdir(exist_ok=True) - merged: Final = merge_claude_settings(original_settings, base_url, ApiKeyHelper(api_key_helper)) - with open(CLAUDE_SETTINGS_PATH, "w") as f: - json.dump(merged, f, indent=2) + merged: Final = merge_claude_settings( + original_settings, base_url, StaticToken(api_key), status_line=status_line + ) + write_claude_settings(CLAUDE_SETTINGS_PATH, merged) except (AgentRunError, ClaudeSettingsError) as e: raise click.ClickException(str(e)) @@ -248,7 +248,6 @@ __all__ = [ "load_json_or_empty", "merge_claude_settings", "read_backup", - "resolve_api_key_helper", "restore_claude_settings", "up", "write_backup", diff --git a/litellm/proxy/db/autorouter_session_rollup.py b/litellm/proxy/db/autorouter_session_rollup.py index b866ecc741f..3a61da164d0 100644 --- a/litellm/proxy/db/autorouter_session_rollup.py +++ b/litellm/proxy/db/autorouter_session_rollup.py @@ -103,6 +103,7 @@ class AutoRouterTurnTransaction: cache_ttl_seconds: int | None cache_touched: bool tier: str | None = None + baseline_model: str | None = None class TurnCacheFacts(NamedTuple): @@ -168,7 +169,7 @@ def _write_ttl_seconds(usage_object: Mapping[str, object] | None) -> int | None: SESSION_ID_MAX_CHARS: Final = 256 -def _bounded_session_id(session_id: str) -> str: +def bounded_session_id(session_id: str) -> str: """The session id as stored, bounded so a caller-chosen identifier cannot exceed Postgres's B-tree index entry limit through the composite primary key. Oversized ids map to a stable digest, so their turns still aggregate into one session.""" @@ -194,6 +195,9 @@ def build_autorouter_turn_transaction( classifier_cost folded into this turn's spend: the excluded classifier row is how it was billed, the decision is how it is attributed. Cache facts are derived from the payload's own usage record through the savings owner, never handed in beside it. + The baseline the turn's saved_spend was priced against travels with the turn, so the + row can name the counterfactual for the money it holds even after the router is + reconfigured or removed. """ if payload.get("status") != "success": return None @@ -216,13 +220,15 @@ def build_autorouter_turn_transaction( usage_object_raw: Final = metadata.get("usage_object") cache: Final = turn_cache_facts(usage_object_raw if isinstance(usage_object_raw, Mapping) else None) tier_raw: Final = routing_decision.get("tier") + baseline_raw: Final = routing_decision.get("savings_baseline_model") classifier_cost: Final = classifier_cost_from_decision(routing_decision) return AutoRouterTurnTransaction( api_key=api_key, - session_id=_bounded_session_id(session_id), + session_id=bounded_session_id(session_id), router_name=router_name, router_type=str(routing_decision.get("router_type") or "unknown"), tier=tier_raw if isinstance(tier_raw, str) and tier_raw else None, + baseline_model=baseline_raw if isinstance(baseline_raw, str) and baseline_raw else None, model=model, turn_at=turn_at, total_tokens=int(payload.get("prompt_tokens") or 0) + int(payload.get("completion_tokens") or 0), @@ -253,6 +259,10 @@ _CACHE_TTL: Final = _p("cache_ttl_seconds") _TOUCHED: Final = _p("cache_touched") _TIER: Final = f"{_p('tier')}::text" _TIER_DELTA: Final = f"(CASE WHEN {_TIER} IS NULL THEN '{{}}'::jsonb ELSE jsonb_build_object({_TIER}, 1) END)" +_BASELINE: Final = f"{_p('baseline_model')}::text" +_BASELINE_DELTA: Final = ( + f"(CASE WHEN {_BASELINE} IS NULL THEN '{{}}'::jsonb ELSE jsonb_build_object({_BASELINE}, 1) END)" +) _IN_ORDER: Final = f"{_TURN_AT}::timestamp >= t.last_turn_at" _SAME: Final = f"{_IN_ORDER} AND t.last_model = {_MODEL}" @@ -270,7 +280,8 @@ INSERT INTO "LiteLLM_AutoRouterSession" AS t ( last_model, models, turns, unordered_turns, covered_turns, cache_hits, same_model_turns, same_model_hits, first_visit_turns, first_visit_hits, return_turns, return_hits, return_expired_misses, return_within_ttl_misses, - ttl_5m_turns, ttl_1h_turns, total_tokens, spend, saved_spend, classifier_cost, classifier_cost_recorded_turns, tier_turns + ttl_5m_turns, ttl_1h_turns, total_tokens, spend, saved_spend, classifier_cost, classifier_cost_recorded_turns, tier_turns, + baseline_models ) VALUES ( {_p("api_key")}, {_p("session_id")}, {_p("router_name")}, {_p("router_type")}, {_TURN_AT}::timestamp, {_TURN_AT}::timestamp, @@ -281,7 +292,7 @@ VALUES ( (CASE WHEN {_CACHE_TTL}::int = {CACHE_TTL_5M_SECONDS} THEN 1 ELSE 0 END), (CASE WHEN {_CACHE_TTL}::int = {CACHE_TTL_1H_SECONDS} THEN 1 ELSE 0 END), {_p("total_tokens")}::bigint, {_p("spend")}::float8, {_p("saved_spend")}::float8, - {_p("classifier_cost")}::float8, 1, {_TIER_DELTA} + {_p("classifier_cost")}::float8, 1, {_TIER_DELTA}, {_BASELINE_DELTA} ) ON CONFLICT (api_key, session_id, router_name) DO UPDATE SET turns = t.turns + 1, @@ -317,6 +328,9 @@ ON CONFLICT (api_key, session_id, router_name) DO UPDATE SET tier_turns = (CASE WHEN {_TIER} IS NOT NULL AND t.router_type = {_p("router_type")} THEN t.tier_turns || jsonb_build_object({_TIER}, COALESCE((t.tier_turns ->> {_TIER})::int, 0) + 1) ELSE t.tier_turns END), + baseline_models = (CASE WHEN {_BASELINE} IS NOT NULL + THEN t.baseline_models || jsonb_build_object({_BASELINE}, COALESCE((t.baseline_models ->> {_BASELINE})::int, 0) + 1) + ELSE t.baseline_models END), first_turn_at = LEAST(t.first_turn_at, EXCLUDED.first_turn_at), last_turn_at = GREATEST(t.last_turn_at, EXCLUDED.last_turn_at) """ diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index bbc914a772a..50716e5d474 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -32,11 +32,15 @@ from litellm.proxy.auth.auth_checks import ( can_key_call_resolved_model, ) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth -from litellm.proxy.db.autorouter_session_rollup import AUTOROUTER_BENCHMARKS_SQL +from litellm.proxy.db.autorouter_session_rollup import ( + AUTOROUTER_BENCHMARKS_SQL, + bounded_session_id, +) from litellm.proxy.litellm_pre_call_utils import ( LiteLLMProxyRequestSetup, refresh_proxy_server_request_body_snapshot, ) +from litellm.repositories.autorouter_session_repository import AutoRouterSessionRepository from litellm.repositories.base_repository import SupportsModelDump from litellm.repositories.team_repository import TeamRepository from litellm.router_strategy.complexity_router import ComplexityRouter @@ -54,6 +58,7 @@ from litellm.types.management_endpoints.auto_router_endpoints import ( AutoRouterCacheStats, AutoRouterRoutingTestRequest, AutoRouterRoutingTestResponse, + AutoRouterSessionResponse, ComplexityRouterConfigValidationRequest, ComplexityRouterConfigValidationResponse, RequestComplexityRouterConfig, @@ -704,6 +709,51 @@ async def get_auto_router_benchmarks( ) +@router.get( + "/auto_router/session", + tags=("auto router",), + dependencies=(Depends(user_api_key_auth),), + response_model=AutoRouterSessionResponse, +) +async def get_auto_router_session( + user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], + session_id: Annotated[ + str, Query(description="The client session id (x-*-session-id header) the turns were sent under") + ], +) -> AutoRouterSessionResponse: + """ + One auto-routed session, for the key that ran it: the model its last turn was routed to and the + session's spend against the router's savings baseline. Built for a coding agent's status line + or stop hook, so any virtual key may call it and only ever sees rows written under its own + key hash. Reads the LiteLLM_AutoRouterSession rollup, which the asynchronous spend flush + fills a moment after each turn; a session with no flushed auto-routed turn yet is a 404. The + id is bounded the way the writer bounded it, so an oversized client id still finds its row. + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value) + row: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key( + user_api_key_dict.api_key, bounded_session_id(session_id) + ) + if row is None: + raise HTTPException( + status_code=404, detail=f"No auto-routed turns recorded for session {session_id!r} under this key" + ) + return AutoRouterSessionResponse( + session_id=session_id, + router_name=row.router_name, + router_type=row.router_type, + turns=row.turns, + last_model=row.last_model, + spend=row.spend, + saved_spend=row.saved_spend, + baseline_spend=row.spend + row.saved_spend, + baseline_model=row.baseline_model, + baseline_models=row.baseline_models, + ) + + # --------------------------------------------------------------------------- # Shadow eval: pre-adoption evaluation of an auto-router against live traffic. # The job row is immutable config plus stopped_at; status, counts, spend, and errors diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 817df082d8c..7d521d54791 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -1514,6 +1514,7 @@ model LiteLLM_AutoRouterSession { classifier_cost Float @default(0) classifier_cost_recorded_turns Int @default(0) tier_turns Json @default("{}") + baseline_models Json @default("{}") @@id([api_key, session_id, router_name]) @@index([last_turn_at], map: "idx_autorouter_session_last_turn") diff --git a/litellm/repositories/__init__.py b/litellm/repositories/__init__.py index 881f7a66cea..dcf9ddfc32a 100644 --- a/litellm/repositories/__init__.py +++ b/litellm/repositories/__init__.py @@ -2,6 +2,7 @@ Repository classes for database operations. """ +from litellm.repositories.autorouter_session_repository import AutoRouterSessionRepository from litellm.repositories.budget_repository import BudgetRepository from litellm.repositories.config_repository import ConfigRepository from litellm.repositories.credentials_repository import CredentialsRepository @@ -92,6 +93,7 @@ __all__ = [ "AdaptiveRouterStateRepository", "AgentsRepository", "AuditLogRepository", + "AutoRouterSessionRepository", "BatchTable", "BudgetCascadeUnitOfWork", "BudgetRepository", diff --git a/litellm/repositories/autorouter_session_repository.py b/litellm/repositories/autorouter_session_repository.py new file mode 100644 index 00000000000..d05ef9421ca --- /dev/null +++ b/litellm/repositories/autorouter_session_repository.py @@ -0,0 +1,34 @@ +""" +Repository for the auto-router per-session rollup (LiteLLM_AutoRouterSession). +""" + +from typing import TYPE_CHECKING, Final + +from litellm.models.autorouter_session import LiteLLM_AutoRouterSession +from litellm.repositories.base_repository import BaseRepository +from litellm.repositories.prisma_protocols import TableActions + +if TYPE_CHECKING: + from prisma import models as prisma_models + + +class AutoRouterSessionRepository(BaseRepository[LiteLLM_AutoRouterSession]): + @property + def table(self) -> TableActions["prisma_models.LiteLLM_AutoRouterSession"]: + return self.prisma_client.db.litellm_autoroutersession + + @property + def model_class(self) -> type[LiteLLM_AutoRouterSession]: + return LiteLLM_AutoRouterSession + + async def find_latest_for_key(self, api_key: str, session_id: str) -> LiteLLM_AutoRouterSession | None: + """The session's most recently active router row under exactly this key hash, or None. + + The key is the row's own partition, not a filter over a wider read: the spend writer keyed the + row under the caller's api_key, so a key can only ever see what it wrote itself. + """ + record: Final = await self.table.find_first( + where={"api_key": api_key, "session_id": session_id}, # mutable-ok: Prisma where filter must be a dict + order={"last_turn_at": "desc"}, # mutable-ok: Prisma order clause must be a dict + ) + return self._to_model(record) diff --git a/litellm/types/management_endpoints/auto_router_endpoints.py b/litellm/types/management_endpoints/auto_router_endpoints.py index 6306658ad0b..9f29f27e41d 100644 --- a/litellm/types/management_endpoints/auto_router_endpoints.py +++ b/litellm/types/management_endpoints/auto_router_endpoints.py @@ -226,6 +226,30 @@ class AutoRouterBenchmarkGroup(AutoRouterBenchmarkTotals): ) +class AutoRouterSessionResponse(BaseModel): + """One auto-routed session as its own key sees it: what the last turn ran on, and what the session cost + against the router's savings baseline (the priciest model in its hardest tier).""" + + session_id: str + router_name: str = Field(description="The auto-router alias the session's requests were sent to") + router_type: str = Field(description="complexity, adaptive or quality") + turns: int = Field(description="Auto-routed turns the rollup has recorded for this session so far") + last_model: str = Field(description="The deployment model the most recent turn was routed to") + spend: float = Field(description="What the session's routed traffic actually cost, classifier calls included") + saved_spend: float = Field(description="Estimated savings against the baseline, net of classifier cost") + baseline_spend: float = Field(description="spend plus saved_spend: the estimated single-model cost") + baseline_model: str | None = Field( + description="The savings baseline most of this session's turns were priced against, recorded turn by " + "turn, so it still names the counterfactual after the router is reconfigured or removed. None when no " + "turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, " + "which derive no baseline and so report no savings" + ) + baseline_models: Mapping[str, int] = Field( + description="Turns priced against each baseline model; more than one entry means the router's " + "baseline changed mid-session and baseline_spend mixes both" + ) + + class AutoRouterBenchmarksResponse(BaseModel): """Benchmarks for the auto-router dashboard, aggregated from the per-session rollup.""" diff --git a/schema.prisma b/schema.prisma index 817df082d8c..7d521d54791 100644 --- a/schema.prisma +++ b/schema.prisma @@ -1514,6 +1514,7 @@ model LiteLLM_AutoRouterSession { classifier_cost Float @default(0) classifier_cost_recorded_turns Int @default(0) tier_turns Json @default("{}") + baseline_models Json @default("{}") @@id([api_key, session_id, router_name]) @@index([last_turn_at], map: "idx_autorouter_session_last_turn") diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py index 9ac6476a03c..e8caa241a53 100644 --- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py +++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py @@ -42,6 +42,7 @@ async def _turn( saved: float = 0.02, classifier_cost: float = 0.0, tier: "str | None" = None, + baseline: "str | None" = None, ) -> None: touched: Final = 1 if (hit or ttl is not None or not covered) else 0 await db.execute_raw( @@ -61,6 +62,7 @@ async def _turn( ttl, touched, tier, + baseline, ) @@ -338,6 +340,27 @@ async def test_a_mid_session_router_type_change_keeps_foreign_tier_names_out_of_ assert row["turns"] == 3 +async def test_baseline_models_count_the_turns_priced_against_each_baseline(db): + key = f"k-{uuid.uuid4()}" + await _turn(db, key, "A", T0, baseline="opus") + await _turn(db, key, "B", T0 + timedelta(seconds=10), baseline="opus") + await _turn(db, key, "A", T0 + timedelta(seconds=20), baseline="sonnet") + + assert (await _row(db, key))["baseline_models"] == {"opus": 2, "sonnet": 1} + + +async def test_a_turn_priced_against_no_baseline_leaves_the_map_alone(db): + key = f"k-{uuid.uuid4()}" + await _turn(db, key, "A", T0, baseline=None) + assert (await _row(db, key))["baseline_models"] == {} + + await _turn(db, key, "A", T0 + timedelta(seconds=10), baseline="opus") + await _turn(db, key, "A", T0 + timedelta(seconds=20), baseline=None) + row = await _row(db, key) + assert row["baseline_models"] == {"opus": 1} + assert row["turns"] == 3 + + async def test_an_out_of_order_turn_still_counts_toward_its_tier(db): key = f"k-{uuid.uuid4()}" await _turn(db, key, "A", T0 + timedelta(seconds=60), tier="simple") diff --git a/tests/test_litellm/litellm_core_utils/test_private_json.py b/tests/test_litellm/litellm_core_utils/test_private_json.py index cedff61959f..3c9f49607f9 100644 --- a/tests/test_litellm/litellm_core_utils/test_private_json.py +++ b/tests/test_litellm/litellm_core_utils/test_private_json.py @@ -4,7 +4,11 @@ import stat import pytest -from litellm.litellm_core_utils.private_json import overwrite_private_json, write_private_json +from litellm.litellm_core_utils.private_json import ( + overwrite_private_json, + write_private_bytes, + write_private_json, +) class TestOverwritePrivateJson: @@ -35,3 +39,32 @@ class TestOverwritePrivateJson: overwrite_private_json(str(path), {"user_id": "u-1"}) assert stat.S_IMODE(path.stat().st_mode) == 0o600 + + +class TestWritePrivateBytes: + def test_replaces_the_file_in_one_step_so_a_reader_holding_the_old_one_keeps_it_whole(self, tmp_path): + path = tmp_path / "script.py" + write_private_bytes(str(path), b"print('one')\n" * 200) + before = path.stat().st_ino + + with path.open("rb") as reader: + write_private_bytes(str(path), b"print('two')\n") + assert reader.read() == b"print('one')\n" * 200 + + assert path.read_bytes() == b"print('two')\n" + assert path.stat().st_ino != before + assert [child.name for child in tmp_path.iterdir()] == ["script.py"] + + @pytest.mark.skipif(os.geteuid() == 0, reason="root ignores file permissions") + def test_lands_owner_only_and_a_refused_stage_leaves_the_previous_file_untouched(self, tmp_path): + path = tmp_path / "script.py" + write_private_bytes(str(path), b"first") + assert stat.S_IMODE(path.stat().st_mode) == 0o600 + + tmp_path.chmod(0o500) + try: + with pytest.raises(PermissionError): + write_private_bytes(str(path), b"second") + finally: + tmp_path.chmod(0o700) + assert path.read_bytes() == b"first" diff --git a/tests/test_litellm/models/test_models.py b/tests/test_litellm/models/test_models.py index 9ae9b732066..aa6449c98dd 100644 --- a/tests/test_litellm/models/test_models.py +++ b/tests/test_litellm/models/test_models.py @@ -8,6 +8,7 @@ import pytest from pydantic import BaseModel, TypeAdapter from litellm.models.access_group import LiteLLM_AccessGroupTable +from litellm.models.autorouter_session import LiteLLM_AutoRouterSession from litellm.models.budget import ( LiteLLM_BudgetTable, LiteLLM_BudgetTableFull, @@ -588,3 +589,35 @@ class TestManagedTables: ) assert table.vector_store_id == "vs1" assert table.custom_llm_provider == "openai" + + +class TestAutoRouterSession: + @staticmethod + def _row(baseline_models: dict) -> LiteLLM_AutoRouterSession: + return LiteLLM_AutoRouterSession( + api_key="k", + session_id="s", + router_name="auto", + router_type="complexity", + first_turn_at=datetime(2026, 9, 1, 12, 0, 0), + last_turn_at=datetime(2026, 9, 1, 12, 5, 0), + last_model="anthropic/claude-sonnet-5", + turns=3, + spend=0.14, + saved_spend=0.24, + classifier_cost=0.0, + tier_turns={}, + baseline_models=baseline_models, + ) + + def test_the_baseline_label_is_the_one_most_turns_were_priced_against(self): + assert self._row({"anthropic/claude-opus-5": 2, "anthropic/claude-sonnet-5": 1}).baseline_model == ( + "anthropic/claude-opus-5" + ) + + def test_a_tie_between_baselines_is_broken_deterministically(self): + assert self._row({"b-model": 1, "a-model": 1}).baseline_model == "b-model" + assert self._row({"a-model": 1, "b-model": 1}).baseline_model == "b-model" + + def test_a_row_whose_turns_recorded_no_baseline_has_no_label(self): + assert self._row({}).baseline_model is None diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py index 38a0e85c1ea..c8b3d789665 100644 --- a/tests/test_litellm/proxy/auth/test_route_checks.py +++ b/tests/test_litellm/proxy/auth/test_route_checks.py @@ -3916,3 +3916,28 @@ def test_claude_code_marketplace_routes_open_to_internal_users(route): """Per-skill visibility is enforced inside the handler, so the route gate must let non-admins through.""" assert RouteChecks.is_llm_api_route(route) is True assert _gate(route, LitellmUserRoles.INTERNAL_USER.value) == "allowed" + + +@pytest.mark.parametrize("user_role", [None, LitellmUserRoles.INTERNAL_USER.value, LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value]) +def test_auto_router_session_is_reachable_by_any_key_but_benchmarks_stays_admin_only(user_role): + valid_token = UserAPIKeyAuth(api_key="hash-of-caller", user_role=user_role) + request = MagicMock(spec=Request) + request.query_params = {"session_id": "sess-1"} + + RouteChecks.non_proxy_admin_allowed_routes_check( + user_obj=None, + _user_role=user_role, + route="/auto_router/session", + request=request, + valid_token=valid_token, + request_data={}, + ) + with pytest.raises(Exception, match="Only proxy admin"): + RouteChecks.non_proxy_admin_allowed_routes_check( + user_obj=None, + _user_role=user_role, + route="/auto_router/benchmarks", + request=request, + valid_token=valid_token, + request_data={}, + ) diff --git a/tests/test_litellm/proxy/client/cli/autoroute/test_commands.py b/tests/test_litellm/proxy/client/cli/autoroute/test_commands.py index 77742ea9f9f..028ab58843f 100644 --- a/tests/test_litellm/proxy/client/cli/autoroute/test_commands.py +++ b/tests/test_litellm/proxy/client/cli/autoroute/test_commands.py @@ -6,6 +6,7 @@ from typing import Optional import yaml from click.testing import CliRunner +from litellm.proxy.client.cli.commands.claude_settings import ClaudeSettingsError from litellm.proxy.client.cli.commands.autoroute import commands as commands_module from litellm.proxy.client.cli.commands.autoroute import process as process_module from litellm.proxy.client.cli.commands.autoroute.commands import down, up @@ -163,6 +164,7 @@ class TestUpCommand: # `lite configure claude --model` or a user pin would 400 on the first message. assert captured["settings"]["model"] == "autorouter" assert captured["settings"]["env"]["ANTHROPIC_DEFAULT_SONNET_MODEL"] == "autorouter" + assert captured["settings"]["statusLine"]["command"].endswith("statusline.py") assert captured["settings_mode"] == 0o600 assert terminate_calls == [99999] @@ -253,6 +255,33 @@ class TestUpCommand: assert not pid_record_path.exists() assert not backup_path.exists() + def test_a_status_line_install_failure_leaves_no_backup_behind(self, monkeypatch, tmp_path): + # The install runs before the backup is written, so a failure cannot strand a backup that + # would make every later `lite configure` / `lite autoroute up` think a session still owns settings.json + config_path, _log_path, claude_settings_path, backup_path, pid_record_path = _patch_paths(monkeypatch, tmp_path) + config_path.write_text(yaml.safe_dump({"model_list": []})) + claude_settings_path.write_text(json.dumps({"theme": "dark"})) + + def boom(): + raise ClaudeSettingsError("disk full") + + fake_process = FakeProcess(pid=778) + terminate_calls = [] + monkeypatch.setattr(commands_module, "launch_proxy", lambda *a, **k: fake_process) + monkeypatch.setattr(commands_module, "poll_liveliness", lambda *a, **k: None) + monkeypatch.setattr(commands_module, "is_port_available", lambda port: True) + monkeypatch.setattr(commands_module, "terminate", lambda pid, **k: terminate_calls.append(pid)) + monkeypatch.setattr(commands_module, "install_statusline_script", boom) + monkeypatch.setattr(commands_module.secrets, "token_urlsafe", lambda n: "fixed-master-key") + + result = self.runner.invoke(up) + + assert result.exit_code != 0 and "disk full" in result.output + assert terminate_calls == [778] + assert not pid_record_path.exists() + assert not backup_path.exists() + assert json.loads(claude_settings_path.read_text()) == {"theme": "dark"} + def test_up_uses_the_same_port_and_master_key_across_runs(self, monkeypatch, tmp_path): """The LIT-4607/LIT-4608 regression: a client configured against one session must keep working in the next, so consecutive runs must patch settings with an identical base URL diff --git a/tests/test_litellm/proxy/client/cli/autoroute/test_config.py b/tests/test_litellm/proxy/client/cli/autoroute/test_config.py index f399b6f957a..47b5459d489 100644 --- a/tests/test_litellm/proxy/client/cli/autoroute/test_config.py +++ b/tests/test_litellm/proxy/client/cli/autoroute/test_config.py @@ -161,7 +161,13 @@ class TestBuildGeneratedModelList: config = _base_config(classifier=HeuristicClassifier(), semantic_matching=NoSemanticMatching()) autorouter = next(m for m in build_generated_model_list(config) if m["model_name"] == "autorouter") router_config = autorouter["litellm_params"]["complexity_router_config"] - assert set(router_config.keys()) == {"tiers", "default_model"} + assert set(router_config.keys()) == {"tiers", "default_model", "return_raw_model_name"} + + def test_the_generated_router_reports_the_tier_model_it_routed_to(self): + # The status line reads the routed model from the response body, which the proxy restamps to the + # requested alias unless the deployment opts out; "autorouter" on every line would tell nothing. + autorouter = next(m for m in build_generated_model_list(_base_config()) if m["model_name"] == "autorouter") + assert autorouter["litellm_params"]["complexity_router_config"]["return_raw_model_name"] is True class TestBuildGeneratedProxyConfig: diff --git a/tests/test_litellm/proxy/client/cli/conftest.py b/tests/test_litellm/proxy/client/cli/conftest.py index 50c76d3f125..c77f516a768 100644 --- a/tests/test_litellm/proxy/client/cli/conftest.py +++ b/tests/test_litellm/proxy/client/cli/conftest.py @@ -5,6 +5,8 @@ from typing import Final import pytest +from litellm.proxy.client.cli.commands import claude_settings + REAL_CLAUDE_SETTINGS: Final = Path(os.path.expanduser("~")) / ".claude" / "settings.json" @@ -12,6 +14,11 @@ def _current_bytes() -> bytes | None: return REAL_CLAUDE_SETTINGS.read_bytes() if REAL_CLAUDE_SETTINGS.exists() else None +@pytest.fixture(autouse=True) +def _statusline_script_under_tmp(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + monkeypatch.setattr(claude_settings, "STATUSLINE_SCRIPT_PATH", tmp_path / "litellm-home" / "statusline.py") + + @pytest.fixture(autouse=True) def isolated_claude_home(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Iterator[Path]: before: Final = _current_bytes() diff --git a/tests/test_litellm/proxy/client/cli/test_agents.py b/tests/test_litellm/proxy/client/cli/test_agents.py index bebea285edc..7804435a60d 100644 --- a/tests/test_litellm/proxy/client/cli/test_agents.py +++ b/tests/test_litellm/proxy/client/cli/test_agents.py @@ -164,27 +164,6 @@ class TestBuildAgentEnv: assert env["PATH"] == "/usr/bin" assert base == {"PATH": "/usr/bin", "ANTHROPIC_API_KEY": "real-key"} - def test_anthropic_profile_leaves_the_bearer_to_the_api_key_helper(self): - env = build_agent_env( - {"ANTHROPIC_AUTH_TOKEN": "stale-token", "ANTHROPIC_API_KEY": "real-key"}, - "http://localhost:4000/", - "sk-key", - frozenset({"anthropic"}), - export_anthropic_token=False, - ) - assert "ANTHROPIC_AUTH_TOKEN" not in env - assert "ANTHROPIC_API_KEY" not in env - assert env["ANTHROPIC_BASE_URL"] == "http://localhost:4000" - assert env["ENABLE_TOOL_SEARCH"] == "true" - assert env["CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY"] == "1" - - def test_helper_mode_still_exports_the_openai_key(self): - env = build_agent_env( - {}, "http://localhost:4000", "sk-key", frozenset({"anthropic", "openai"}), export_anthropic_token=False - ) - assert "ANTHROPIC_AUTH_TOKEN" not in env - assert env["OPENAI_API_KEY"] == "sk-key" - class TestAgentLaunchArgs: def test_claude_and_opencode_get_no_extra_args(self): @@ -531,25 +510,6 @@ class TestRunAgent: assert "ANTHROPIC_API_KEY" not in env assert "OPENAI_BASE_URL" not in env - def test_helper_supplied_token_never_reaches_the_launch_env(self): - calls = {} - verified = [] - - run_agent( - "http://localhost:4000", - "sk-key", - ["claude"], - base_env={"PATH": "/usr/bin", "ANTHROPIC_AUTH_TOKEN": "stale-token"}, - which=lambda name: "/usr/local/bin/claude", - verify=lambda base_url, api_key: verified.append(api_key), - launcher=lambda p, a, e: calls.update(env=dict(e)), - export_anthropic_token=False, - ) - - assert verified == ["sk-key"] - assert "ANTHROPIC_AUTH_TOKEN" not in calls["env"] - assert calls["env"]["ANTHROPIC_BASE_URL"] == "http://localhost:4000" - def test_codex_gets_openai_env(self): calls = {} run_agent( @@ -1093,101 +1053,6 @@ class TestAgentCommands: in result.output ) - def _invoke_claude_with_settings(self, tmp_path, settings, obj, *, default_settings=None): - config_dir = tmp_path / "claude-config" - config_dir.mkdir() - if settings is not None: - (config_dir / "settings.json").write_text(json.dumps(settings)) - default_path = tmp_path / "home-claude" / "settings.json" - default_path.parent.mkdir() - if default_settings is not None: - default_path.write_text(json.dumps(default_settings)) - captured = {} - with ( - patch(f"{CLAUDE_SETTINGS_MODULE}.CLAUDE_SETTINGS_PATH", default_path), - patch(f"{CLAUDE_SETTINGS_MODULE}.shutil.which", return_value="/usr/local/bin/lite"), - patch(f"{AGENTS_MODULE}.run_agent", side_effect=lambda b, k, c, **kw: captured.update(kw)), - ): - result = self.runner.invoke( - _agent_command("claude"), [], obj=obj, env={"CLAUDE_CONFIG_DIR": str(config_dir)} - ) - assert result.exit_code == 0, result.output - return captured, result.output - - def test_helper_is_read_from_the_config_dir_claude_code_uses(self, tmp_path): - captured, output = self._invoke_claude_with_settings( - tmp_path, - {"apiKeyHelper": "/usr/local/bin/lite --base-url http://localhost:4000 auth print-token"}, - {"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": True}, - ) - - assert captured["export_anthropic_token"] is False - assert str(tmp_path / "claude-config" / "settings.json") in output - - def test_helper_only_in_the_default_file_keeps_the_env_token_when_config_dir_points_elsewhere(self, tmp_path): - captured, output = self._invoke_claude_with_settings( - tmp_path, - None, - {"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": True}, - default_settings={"apiKeyHelper": "/usr/local/bin/lite --base-url http://localhost:4000 auth print-token"}, - ) - - assert captured["export_anthropic_token"] is True - assert "apiKeyHelper" not in output - - def test_stored_login_with_a_matching_helper_leaves_the_token_to_the_helper(self, tmp_path): - captured, output = self._invoke_claude_with_settings( - tmp_path, - {"apiKeyHelper": "/usr/local/bin/lite --base-url http://localhost:4000 auth print-token"}, - {"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": True}, - ) - - assert captured["export_anthropic_token"] is False - assert "reads its key from the apiKeyHelper" in output - - def test_explicit_key_is_exported_even_when_a_helper_matches(self, tmp_path): - captured, output = self._invoke_claude_with_settings( - tmp_path, - {"apiKeyHelper": "/usr/local/bin/lite --base-url http://localhost:4000 auth print-token"}, - {"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": False}, - ) - - assert captured["export_anthropic_token"] is True - assert "apiKeyHelper" not in output - - def test_helper_for_another_proxy_keeps_the_env_token(self, tmp_path): - captured, _ = self._invoke_claude_with_settings( - tmp_path, - {"apiKeyHelper": "/usr/local/bin/lite --base-url https://other.example.com auth print-token"}, - {"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": True}, - ) - - assert captured["export_anthropic_token"] is True - - def test_no_claude_settings_keeps_the_env_token(self, tmp_path): - captured, _ = self._invoke_claude_with_settings( - tmp_path, - None, - {"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": True}, - ) - - assert captured["export_anthropic_token"] is True - - def test_codex_never_consults_claude_settings(self): - captured = {} - with ( - patch(f"{AGENTS_MODULE}.lite_api_key_helper_configured", side_effect=AssertionError("consulted")), - patch(f"{AGENTS_MODULE}.run_agent", side_effect=lambda b, k, c, **kw: captured.update(kw)), - ): - result = self.runner.invoke( - _agent_command("codex"), - [], - obj={"base_url": "http://localhost:4000", "api_key": "sk-key", "api_key_from_token_file": True}, - ) - - assert result.exit_code == 0, result.output - assert captured["export_anthropic_token"] is True - def test_codex_shows_friendly_name(self): captured = {} with patch( @@ -1338,3 +1203,53 @@ class TestAgentCommands: ) assert result.exit_code == 0, result.output assert captured["reattach_terminal"] is None + + +class TestPrepareCodex: + def test_registers_the_installed_script_as_a_session_scoped_stop_hook(self): + from litellm.proxy.client.cli.commands.agents import prepare_codex + + args = prepare_codex("http://localhost:4000", "sk-key", {}, install=lambda: "/py /home/me/.litellm/statusline.py") + assert args == ( + "-c", + 'hooks.Stop=[{hooks=[{type="command",command="/py /home/me/.litellm/statusline.py"}]}]', + ) + + def test_a_failed_install_is_an_agent_error_not_a_crash(self): + from litellm.proxy.client.cli.commands.agents import AgentRunError, prepare_codex + from litellm.proxy.client.cli.commands.claude_settings import ClaudeSettingsError + + def boom(): + raise ClaudeSettingsError("disk full") + + with pytest.raises(AgentRunError, match="disk full"): + prepare_codex("http://localhost:4000", "sk-key", {}, install=boom) + + def test_a_config_that_already_declares_hooks_keeps_them_and_skips_ours(self, tmp_path): + from litellm.proxy.client.cli.commands.agents import prepare_codex + + warnings = [] + env = {"CODEX_HOME": str(tmp_path)} + for body in ('[[hooks.Stop]]\nhooks = [{ type = "command", command = "mine" }]\n', 'hooks.Stop = []\n', "[hooks]\n"): + (tmp_path / "config.toml").write_text(body) + assert prepare_codex("http://localhost:4000", "sk", env, install=lambda: "/py /s.py", warn=warnings.append) == () + (tmp_path / "config.toml").write_text('model = "gpt-5.6-sol"\n[projects."/x"]\ntrust_level = "trusted"\n') + assert prepare_codex("http://localhost:4000", "sk", env, install=lambda: "/py /s.py", warn=warnings.append) != () + assert len(warnings) == 3 and "already declares hooks" in warnings[0] + + def test_a_config_that_cannot_be_read_or_decoded_still_lets_codex_launch(self, tmp_path): + # A UTF-16 config.toml (a Windows Notepad save) is Codex's problem to report at launch, not a reason + # for the hook pre-check to abort `lite codex` with a traceback before Codex ever starts. + from litellm.proxy.client.cli.commands.agents import codex_declares_stop_hooks, prepare_codex + + config = tmp_path / "config.toml" + config.write_bytes('[[hooks.Stop]]\nhooks = [{ type = "command", command = "mine" }]\n'.encode("utf-16")) + assert codex_declares_stop_hooks(config) is False + assert codex_declares_stop_hooks(tmp_path / "absent.toml") is False + args = prepare_codex("http://localhost:4000", "sk", {"CODEX_HOME": str(tmp_path)}, install=lambda: "/py /s.py") + assert args[0] == "-c" and "hooks.Stop=" in args[1] + + def test_codex_is_wired_through_the_preparer_registry(self): + from litellm.proxy.client.cli.commands.agents import _PREPARERS, prepare_codex + + assert _PREPARERS["codex"] is prepare_codex diff --git a/tests/test_litellm/proxy/client/cli/test_auth_commands.py b/tests/test_litellm/proxy/client/cli/test_auth_commands.py index 1f314e0c9d8..3a7792db1fe 100644 --- a/tests/test_litellm/proxy/client/cli/test_auth_commands.py +++ b/tests/test_litellm/proxy/client/cli/test_auth_commands.py @@ -17,8 +17,15 @@ from litellm.litellm_core_utils.cli_keyring import ( SecretErased, SecretStored, ) -from litellm.litellm_core_utils.cli_token_utils import CliTokenRecord, save_cli_token +from litellm.litellm_core_utils.cli_token_utils import ( + CliTokenRecord, + CredentialNotRecorded, + CredentialNotSaved, + save_cli_token, +) from litellm.proxy.client.cli import cli +from litellm.proxy.client.cli.commands import claude_settings as claude_settings_module +from litellm.proxy.client.cli.commands.claude_settings import SettingsFileOwner from litellm.proxy.client.cli.commands.auth import ( get_stored_api_key, login, @@ -26,8 +33,6 @@ from litellm.proxy.client.cli.commands.auth import ( print_token, whoami, ) -from litellm.proxy.client.cli.commands import claude_settings as claude_settings_module -from litellm.proxy.client.cli.commands.claude_settings import SettingsFileOwner @pytest.fixture @@ -1394,7 +1399,9 @@ class TestLoginConfigClaude: monkeypatch.setattr(claude_settings_module, "CONFIGURE_STATE_PATH", tmp_path / "claude_configure_state.json") return backup_path - def _run_login(self, tmp_path, monkeypatch, args, base_url="https://test.example.com", *, config_dir_env=None): + def _run_login( + self, tmp_path, monkeypatch, args, base_url="https://test.example.com", *, config_dir_env=None, stored=None + ): settings_path = tmp_path / "claude" / "settings.json" backup_path = self._isolate_default_settings(tmp_path, monkeypatch) env = {"CLAUDE_CONFIG_DIR": str(settings_path.parent)} if config_dir_env is None else config_dir_env @@ -1411,12 +1418,8 @@ class TestLoginConfigClaude: patch("webbrowser.open"), patch("requests.post", return_value=_mock_cli_sso_start_response()), patch("requests.get", return_value=poll_response), - patch("litellm.proxy.client.cli.commands.auth.save_cli_token"), + patch("litellm.proxy.client.cli.commands.auth.save_cli_token", return_value=stored or SecretStored()), patch("litellm.proxy.client.cli.interface.show_commands"), - patch( - "litellm.proxy.client.cli.commands.claude_settings.shutil.which", - return_value="/usr/local/bin/lite", - ), ): result = self.runner.invoke(login, args, obj={"base_url": base_url}, env=env) return result, settings_path, backup_path @@ -1436,10 +1439,13 @@ class TestLoginConfigClaude: written = json.loads(settings_path.read_text()) assert written["env"]["ANTHROPIC_BASE_URL"] == "https://test.example.com" assert written["env"]["ENABLE_TOOL_SEARCH"] == "true" - assert written["apiKeyHelper"] == "/usr/local/bin/lite --base-url https://test.example.com auth print-token" + # The minted key goes in as a static token: an apiKeyHelper would make Claude Code spawn `lite` (and + # its keychain probe) on every credential refresh, which is what this flag used to write. + assert written["env"]["ANTHROPIC_AUTH_TOKEN"] == "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.test.jwt" + assert "apiKeyHelper" not in written assert f"Configured Claude Code: {settings_path} now routes through https://test.example.com." in result.output - assert "pins a proxy model for every tier" not in result.output - assert "the model Claude Code starts on" in result.output + assert "run `lite login --config-claude` again after it expires" in result.output + assert "the model Claude Code starts and resumes on" in result.output def test_flag_preserves_unrelated_settings_on_an_existing_file(self, tmp_path, monkeypatch): settings_path = tmp_path / "claude" / "settings.json" @@ -1490,7 +1496,7 @@ class TestLoginConfigClaude: assert result.exit_code == 0, result.output written = json.loads(settings_path.read_text()) - assert written["apiKeyHelper"] == "/usr/local/bin/lite --base-url https://test.example.com auth print-token" + assert written["env"]["ANTHROPIC_AUTH_TOKEN"] == "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.test.jwt" assert f"Configured Claude Code: {settings_path} now routes through https://test.example.com." in result.output def test_flag_keeps_a_config_dir_receipt_apart_from_the_default_file_receipt(self, tmp_path, monkeypatch): @@ -1503,6 +1509,42 @@ class TestLoginConfigClaude: assert len(receipts) == 1 assert json.loads(receipts[0].read_text())["file_existed"] is False + def test_a_second_login_replaces_the_key_and_unconfigure_still_restores_the_original(self, tmp_path, monkeypatch): + # The stored key expires daily, so the flag is re-run per login; the receipt must keep owning the + # slot across re-logins and hand back what was there before the first one. + from litellm.proxy.client.cli.commands.configure import unconfigure_claude + + settings_path = tmp_path / "claude" / "settings.json" + settings_path.parent.mkdir(parents=True) + settings_path.write_text(json.dumps({"theme": "dark", "env": {"ANTHROPIC_AUTH_TOKEN": "sk-theirs"}})) + self._run_login(tmp_path, monkeypatch, ["--config-claude"]) + first = json.loads(settings_path.read_text())["env"]["ANTHROPIC_AUTH_TOKEN"] + self._run_login(tmp_path, monkeypatch, ["--config-claude"]) + assert json.loads(settings_path.read_text())["env"]["ANTHROPIC_AUTH_TOKEN"] == first != "sk-theirs" + + result = self.runner.invoke(unconfigure_claude, [], env={"CLAUDE_CONFIG_DIR": str(settings_path.parent)}) + assert result.exit_code == 0, result.output + assert json.loads(settings_path.read_text()) == {"theme": "dark", "env": {"ANTHROPIC_AUTH_TOKEN": "sk-theirs"}} + + @pytest.mark.parametrize( + "stored", + [CredentialNotSaved("read-only ~/.litellm"), CredentialNotRecorded()], + ids=["nothing-kept-it", "keychain-took-it-file-refused"], + ) + def test_claude_code_is_configured_even_when_the_cli_could_not_keep_the_credential( + self, tmp_path, monkeypatch, stored + ): + # The key is in hand either way, and --config-claude asked for exactly that key to be written into + # settings.json; whether the CLI's own token file or keychain kept a copy is a separate outcome. + result, settings_path, _backup_path = self._run_login(tmp_path, monkeypatch, ["--config-claude"], stored=stored) + + assert result.exit_code == 0, result.output + written = json.loads(settings_path.read_text()) + assert written["env"]["ANTHROPIC_AUTH_TOKEN"] == "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.test.jwt" + assert f"Configured Claude Code: {settings_path}" in result.output + assert "even though the CLI itself could not keep it" in result.output + assert "You can now use the CLI without specifying --api-key" not in result.output + def test_settings_failure_is_reported_without_claiming_login_failed(self, tmp_path, monkeypatch): settings_path = tmp_path / "claude" / "settings.json" settings_path.parent.mkdir(parents=True) diff --git a/tests/test_litellm/proxy/client/cli/test_claude_settings.py b/tests/test_litellm/proxy/client/cli/test_claude_settings.py index fc2d98f2264..cf52d41e963 100644 --- a/tests/test_litellm/proxy/client/cli/test_claude_settings.py +++ b/tests/test_litellm/proxy/client/cli/test_claude_settings.py @@ -1,7 +1,9 @@ import json import os +import pathlib import shlex import stat +import sys import time from pathlib import Path from unittest.mock import patch @@ -9,8 +11,6 @@ from unittest.mock import patch import pytest from click.testing import CliRunner -from litellm.litellm_core_utils.cli_token_utils import CliTokenRecord -from litellm.proxy.client.cli import cli from litellm.litellm_core_utils.private_json import commit_staged_json from litellm.proxy.client.cli.commands.claude_settings import ( ANTHROPIC_DEFAULT_MODEL_ENV_KEYS, @@ -21,7 +21,6 @@ from litellm.proxy.client.cli.commands.claude_settings import ( OWNED_ENV_KEYS, OWNED_TOP_LEVEL_KEYS, SETTINGS_FILE_OWNERS, - ApiKeyHelper, ClaudeSettingsError, KeepModel, SettingsFileOwner, @@ -30,11 +29,12 @@ from litellm.proxy.client.cli.commands.claude_settings import ( UnpinModel, claude_settings_path, configure_claude_settings, + install_statusline_script, configure_state_path, - lite_api_key_helper_configured, merge_claude_settings, - resolve_api_key_helper, + statusline_command, unconfigure_claude_settings, + with_status_line, ) @@ -45,101 +45,33 @@ def _owners(*backup_paths): CLAUDE_SETTINGS_MODULE = "litellm.proxy.client.cli.commands.claude_settings" AUTH_MODULE = "litellm.proxy.client.cli.commands.auth" -WINDOWS_LITE_EXE = "C:\\Users\\u\\AppData\\Local\\Programs\\Python\\Python313\\Scripts\\lite.EXE" - -CMD_METACHARACTERS = frozenset("&|<>^()") -CMD_PERCENT_GUARD = "%%cd:~,%" - - -def _through_cmd_exe(command): - """The line cmd.exe hands to CreateProcess after reading the apiKeyHelper. - - A `"` toggles cmd's quote state and the metacharacters only act outside it. cmd expands - `%VAR%` even inside quotes, so every `%` has to arrive as the `%%cd:~,%` guard: the first - `%` has no variable name and stays literal, and `%cd:~,%` is a zero length substring of `cd`. - """ - assert not any(CMD_METACHARACTERS & set(run) for run in command.split('"')[::2]), command - assert command.count("%") == 3 * command.count(CMD_PERCENT_GUARD), command - return command.replace(CMD_PERCENT_GUARD, "%") - - -def _through_c_runtime(command_line): - """argv as the Microsoft C runtime builds it for the `lite` executable. - - Outside quotes whitespace ends an argument. A `"` toggles quoting, and inside quotes `""` - is a literal quote. Backslashes are literal unless they run up to a `"`, where each pair - is one backslash and an odd one left over makes the quote literal. - """ - argv = [] - current = None - quoted = False - i = 0 - while i < len(command_line): - ch = command_line[i] - if ch in " \t" and not quoted: - if current is not None: - argv.append(current) - current = None - i += 1 - continue - if current is None: - current = "" - if ch == "\\": - run = len(command_line[i:]) - len(command_line[i:].lstrip("\\")) - before_quote = command_line[i + run : i + run + 1] == '"' - current += "\\" * (run // 2 if before_quote else run) - if before_quote and run % 2: - current += '"' - i += 1 - i += run - elif ch == '"': - if quoted and command_line[i + 1 : i + 2] == '"': - current += '"' - i += 1 - else: - quoted = not quoted - i += 1 - else: - current += ch - i += 1 - return argv if current is None else [*argv, current] - - @pytest.fixture def paths(tmp_path): return tmp_path / "claude" / "settings.json", tmp_path / "backup.json" -@pytest.fixture -def lite_on_path(): - with patch(f"{CLAUDE_SETTINGS_MODULE}.shutil.which", return_value="/usr/local/bin/lite"): - yield - - -def _helper_configure(base_url, settings_path, owners, state_path=None): - """`lite login --config-claude`'s shape: the login credential behind apiKeyHelper, no pinned model.""" +def _static_configure(base_url, settings_path, owners, state_path=None): + """`lite configure claude --api-key`'s shape: a virtual key as a static token, no pinned model.""" state = state_path if state_path is not None else settings_path.parent.parent / "state.json" - root = base_url.rstrip("/") - configure_claude_settings( - root, ApiKeyHelper(resolve_api_key_helper(root)), KeepModel(), settings_path, state, owners - ) + configure_claude_settings(base_url.rstrip("/"), StaticToken("sk-virtual-key"), KeepModel(), settings_path, state, owners) -class TestConfigureWithTheLoginHelper: - def test_creates_the_file_and_its_parent_when_missing(self, paths, lite_on_path): +class TestConfigureClaudeSettings: + def test_creates_the_file_and_its_parent_when_missing(self, paths): settings_path, backup_path = paths assert not settings_path.parent.exists() - _helper_configure("https://proxy.example.com/", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com/", settings_path, _owners(backup_path)) written = json.loads(settings_path.read_text()) assert written["env"]["ANTHROPIC_BASE_URL"] == "https://proxy.example.com" + assert written["env"]["ANTHROPIC_AUTH_TOKEN"] == "sk-virtual-key" assert written["env"]["ENABLE_TOOL_SEARCH"] == "true" assert written["env"]["CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY"] == "1" - assert written["apiKeyHelper"] == "/usr/local/bin/lite --base-url https://proxy.example.com auth print-token" - assert "model" not in written + assert "apiKeyHelper" not in written + assert "model" not in written and "ANTHROPIC_MODEL" not in written["env"] - def test_updates_an_existing_file_preserving_unrelated_settings(self, paths, lite_on_path): + def test_updates_an_existing_file_preserving_unrelated_settings(self, paths): settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.write_text( @@ -148,183 +80,87 @@ class TestConfigureWithTheLoginHelper: "theme": "dark", "permissions": {"allow": ["Bash"]}, "env": {"SOME_OTHER_VAR": "keep-me", "ANTHROPIC_BASE_URL": "https://old.example.com"}, - "apiKeyHelper": "old-helper", } ) ) - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) written = json.loads(settings_path.read_text()) assert written["theme"] == "dark" assert written["permissions"] == {"allow": ["Bash"]} assert written["env"]["SOME_OTHER_VAR"] == "keep-me" assert written["env"]["ANTHROPIC_BASE_URL"] == "https://proxy.example.com" - assert written["apiKeyHelper"] != "old-helper" - def test_rerunning_against_a_new_proxy_refreshes_both_base_url_and_helper(self, paths, lite_on_path): - settings_path, backup_path = paths - - _helper_configure("https://first.example.com", settings_path, _owners(backup_path)) - _helper_configure("https://second.example.com", settings_path, _owners(backup_path)) - - written = json.loads(settings_path.read_text()) - assert written["env"]["ANTHROPIC_BASE_URL"] == "https://second.example.com" - assert "second.example.com" in written["apiKeyHelper"] - assert "first.example.com" not in written["apiKeyHelper"] - - def test_drops_stray_static_credentials_so_the_helper_token_wins(self, paths, lite_on_path): - # Claude Code prefers ANTHROPIC_AUTH_TOKEN over apiKeyHelper, so a virtual key left behind - # by an earlier `lite configure claude --api-key` would silently keep winning. + def test_a_helper_left_by_an_older_lite_is_stripped_so_only_the_static_token_is_sent(self, paths): + # Older `lite` versions wrote `apiKeyHelper: lite auth print-token`; Claude Code would keep spawning + # `lite` (and its keychain probe) on every credential refresh, so configure takes the slot over. settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.write_text( - json.dumps({"env": {"ANTHROPIC_API_KEY": "sk-leaked", "ANTHROPIC_AUTH_TOKEN": "sk-old"}}) + json.dumps({"apiKeyHelper": "/usr/local/bin/lite auth print-token", "env": {"ANTHROPIC_API_KEY": "sk-leaked"}}) ) - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) - env = json.loads(settings_path.read_text())["env"] - assert "ANTHROPIC_API_KEY" not in env and "ANTHROPIC_AUTH_TOKEN" not in env + written = json.loads(settings_path.read_text()) + assert "apiKeyHelper" not in written + assert "ANTHROPIC_API_KEY" not in written["env"] + assert written["env"]["ANTHROPIC_AUTH_TOKEN"] == "sk-virtual-key" - def test_written_file_is_owner_only(self, paths, lite_on_path): + def test_written_file_is_owner_only(self, paths): settings_path, backup_path = paths - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) assert stat.S_IMODE(settings_path.stat().st_mode) == 0o600 - def test_refuses_while_lite_up_holds_a_backup(self, paths, lite_on_path): + def test_refuses_while_lite_up_holds_a_backup(self, paths): settings_path, backup_path = paths backup_path.write_text("{}") with pytest.raises(ClaudeSettingsError, match="lite down"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) assert not settings_path.exists() - def test_refuses_on_corrupt_existing_settings_without_touching_the_file(self, paths, lite_on_path): + def test_refuses_on_corrupt_existing_settings_without_touching_the_file(self, paths): settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.write_text("not json at all {{{") with pytest.raises(ClaudeSettingsError, match="invalid JSON"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) assert settings_path.read_text() == "not json at all {{{" - def test_reports_an_actionable_error_when_lite_is_not_on_path(self, paths): - settings_path, backup_path = paths - with patch(f"{CLAUDE_SETTINGS_MODULE}.shutil.which", return_value=None): - with pytest.raises(ClaudeSettingsError, match="Could not find `lite`"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) - - assert not settings_path.exists() - - def test_reports_an_actionable_error_on_a_non_utf8_file(self, paths, lite_on_path): - """Bytes that are not valid UTF-8 must not escape as UnicodeDecodeError. - - UnicodeDecodeError is a ValueError, not an OSError, so a decode-side catch - is easy to miss; login's broad `except Exception` would then relabel it as - an authentication failure and exit 0. - """ + def test_reports_an_actionable_error_on_a_non_utf8_file(self, paths): + # UnicodeDecodeError is a ValueError, not an OSError, so a decode-side catch is easy to miss. settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.write_bytes(b'{"theme": "\xff\xfe"}') with pytest.raises(ClaudeSettingsError, match="invalid JSON"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) - def test_reports_an_actionable_error_when_the_file_cannot_be_read(self, paths, lite_on_path): - """An unreadable settings file must not surface as "Authentication failed". - - login wraps the whole flow in a broad `except Exception`, so any OSError - escaping this function gets relabelled as an auth failure and sends the - user looking at their SSO config instead of at file permissions. - """ + def test_reports_an_actionable_error_when_the_file_cannot_be_read(self, paths): settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.mkdir() with pytest.raises(ClaudeSettingsError, match="Could not read"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) - def test_reports_an_actionable_error_when_the_file_cannot_be_written(self, paths, lite_on_path): + def test_reports_an_actionable_error_when_the_file_cannot_be_written(self, paths): settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.parent.chmod(0o500) try: with pytest.raises(ClaudeSettingsError, match="Could not write"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) finally: settings_path.parent.chmod(0o700) assert not settings_path.exists() -class TestApiKeyHelperIsActuallyInvocable: - """The helper string is executed verbatim by Claude Code, so it has to parse. - - Asserting only on its text is what let a malformed command (`--base-url`, a - top-level group option, placed after the `print-token` subcommand) ship: click - rejects it with "No such option" and every Claude Code request loses its token. - """ - - def _helper_args(self, base_url): - with patch(f"{CLAUDE_SETTINGS_MODULE}.shutil.which", return_value="/usr/local/bin/lite"): - return shlex.split(resolve_api_key_helper(base_url))[1:] - - def test_the_generated_command_parses(self): - result = CliRunner().invoke(cli, self._helper_args("http://localhost:4000")) - - assert "No such option" not in result.output - assert result.exit_code != 2 - - def test_the_generated_command_reaches_print_token(self): - with patch(f"{AUTH_MODULE}.load_cli_token", return_value=None): - result = CliRunner().invoke(cli, self._helper_args("http://localhost:4000")) - - assert "Not authenticated" in result.output - - def test_the_generated_command_carries_the_base_url_through(self): - stale = CliTokenRecord( - base_url="http://other-proxy.example.com", - key="sk-stale", - timestamp=time.time(), - ) - with patch(f"{AUTH_MODULE}.load_cli_token", return_value=stale): - result = CliRunner().invoke(cli, self._helper_args("http://localhost:4000")) - - assert "Not authenticated for this server" in result.output - - def _windows_argv(self, lite_exe, base_url): - with patch(f"{CLAUDE_SETTINGS_MODULE}.shutil.which", return_value=lite_exe): - helper = resolve_api_key_helper(base_url, platform="win32") - return _through_c_runtime(_through_cmd_exe(helper)) - - @pytest.mark.parametrize( - ("lite_exe", "base_url"), - [ - (WINDOWS_LITE_EXE, "http://localhost:4000"), - ("C:\\Program Files\\LiteLLM\\lite.EXE", "https://gateway.example.com/?a=1&b=2"), - ("C:\\Users\\u\\Scripts\\lite.EXE", "https://gateway.example.com/team%20a/%7Eproxy"), - ('C:\\odd "dir"\\lite.EXE', "http://localhost:4000/x\\"), - ], - ) - def test_the_windows_command_survives_cmd_exe_and_the_c_runtime(self, lite_exe, base_url): - assert self._windows_argv(lite_exe, base_url) == [lite_exe, "--base-url", base_url, "auth", "print-token"] - - def test_the_windows_command_carries_the_base_url_through_cmd_quoting(self): - stale = CliTokenRecord( - base_url="http://other-proxy.example.com", - key="sk-stale", - timestamp=time.time(), - ) - argv = self._windows_argv(WINDOWS_LITE_EXE, "http://localhost:4000") - with patch(f"{AUTH_MODULE}.load_cli_token", return_value=stale): - result = CliRunner().invoke(cli, argv[1:]) - - assert argv[0] == WINDOWS_LITE_EXE - assert "Not authenticated for this server" in result.output - - class TestConflictingOwnersOfTheSettingsFile: """Both `lite up` and `lite autoroute up` restore a backup when they stop. @@ -332,7 +168,7 @@ class TestConflictingOwnersOfTheSettingsFile: write, which is the exact hazard the guard exists to prevent. """ - def test_any_owner_holding_a_backup_blocks_the_write(self, tmp_path, lite_on_path): + def test_any_owner_holding_a_backup_blocks_the_write(self, tmp_path): settings_path = tmp_path / "claude" / "settings.json" for index, owner in enumerate(SETTINGS_FILE_OWNERS): @@ -340,20 +176,20 @@ class TestConflictingOwnersOfTheSettingsFile: backup.write_text("{}") stand_in = SettingsFileOwner(backup, owner.start_command, owner.stop_command) with pytest.raises(ClaudeSettingsError, match="currently managing"): - _helper_configure("https://proxy.example.com", settings_path, (stand_in,)) + _static_configure("https://proxy.example.com", settings_path, (stand_in,)) backup.unlink() assert not settings_path.exists() - def test_the_error_names_the_owner_that_actually_holds_the_file(self, tmp_path, lite_on_path): + def test_the_error_names_the_owner_that_actually_holds_the_file(self, tmp_path): settings_path = tmp_path / "claude" / "settings.json" backup = tmp_path / "auto.json" backup.write_text("{}") autoroute = SettingsFileOwner(backup, "lite autoroute up", "lite autoroute down") with pytest.raises(ClaudeSettingsError, match="`lite autoroute up` is currently managing"): - _helper_configure("https://proxy.example.com", settings_path, (autoroute,)) + _static_configure("https://proxy.example.com", settings_path, (autoroute,)) with pytest.raises(ClaudeSettingsError, match="Run `lite autoroute down` first"): - _helper_configure("https://proxy.example.com", settings_path, (autoroute,)) + _static_configure("https://proxy.example.com", settings_path, (autoroute,)) def test_the_registry_matches_the_paths_the_commands_actually_use(self): """A second definition of the autoroute dir must not drift from this one.""" @@ -365,7 +201,7 @@ class TestConflictingOwnersOfTheSettingsFile: class TestDoesNotDestroyUserOwnedStructure: - def test_writes_through_a_symlinked_settings_file(self, tmp_path, lite_on_path): + def test_writes_through_a_symlinked_settings_file(self, tmp_path): """os.replace() swaps the symlink for a regular file, detaching a dotfiles repo. There is no backup here to undo that, so the link must survive and its @@ -378,20 +214,20 @@ class TestDoesNotDestroyUserOwnedStructure: link.parent.mkdir() link.symlink_to(real) - _helper_configure("https://proxy.example.com", link, ()) + _static_configure("https://proxy.example.com", link, ()) assert link.is_symlink() assert json.loads(real.read_text())["env"]["ANTHROPIC_BASE_URL"] == "https://proxy.example.com" assert json.loads(real.read_text())["theme"] == "dark" - def test_refuses_rather_than_discarding_a_non_object_env(self, paths, lite_on_path): + def test_refuses_rather_than_discarding_a_non_object_env(self, paths): """merge coerces a non-dict env to {}; that is silent data loss on a persistent write.""" settings_path, backup_path = paths settings_path.parent.mkdir(parents=True) settings_path.write_text(json.dumps({"theme": "dark", "env": "not-an-object"})) with pytest.raises(ClaudeSettingsError, match="non-object"): - _helper_configure("https://proxy.example.com", settings_path, _owners(backup_path)) + _static_configure("https://proxy.example.com", settings_path, _owners(backup_path)) assert json.loads(settings_path.read_text())["env"] == "not-an-object" @@ -446,18 +282,13 @@ class TestConfigureStatePath: assert work_state == configure_state_path(tmp_path / "work" / "settings.json") def test_configure_and_unconfigure_under_a_config_dir_leave_the_default_receipt_alone( - self, default_paths, tmp_path, lite_on_path + self, default_paths, tmp_path ): _default_settings, default_state = default_paths work_settings = tmp_path / "work" / "settings.json" work_state = configure_state_path(work_settings) configure_claude_settings( - "https://proxy.example.com", - ApiKeyHelper(resolve_api_key_helper("https://proxy.example.com")), - KeepModel(), - work_settings, - work_state, - (), + "https://proxy.example.com", StaticToken("sk-virtual-key"), KeepModel(), work_settings, work_state, () ) assert work_state.exists() and not default_state.exists() outcome = unconfigure_claude_settings(work_settings, work_state, ()) @@ -465,51 +296,8 @@ class TestConfigureStatePath: assert not work_state.exists() -class TestLiteApiKeyHelperConfigured: - def _settings(self, tmp_path, payload): - settings_path = tmp_path / "settings.json" - settings_path.write_text(payload) - return settings_path - - def test_recognises_the_helper_lite_login_wrote_for_this_proxy(self, tmp_path, lite_on_path): - settings_path = tmp_path / "settings.json" - _helper_configure("https://proxy.example.com/", settings_path, (), tmp_path / "state.json") - - assert lite_api_key_helper_configured("https://proxy.example.com/", settings_path) is True - assert lite_api_key_helper_configured("https://proxy.example.com", settings_path) is True - - def test_a_helper_for_another_proxy_does_not_count(self, tmp_path, lite_on_path): - settings_path = tmp_path / "settings.json" - _helper_configure("https://other.example.com", settings_path, (), tmp_path / "state.json") - - assert lite_api_key_helper_configured("https://proxy.example.com", settings_path) is False - - def test_a_hand_written_helper_does_not_count(self, tmp_path, lite_on_path): - settings_path = self._settings(tmp_path, json.dumps({"apiKeyHelper": "cat ~/.my-proxy-key"})) - - assert lite_api_key_helper_configured("https://proxy.example.com", settings_path) is False - - def test_missing_or_helperless_settings_do_not_count(self, tmp_path, lite_on_path): - assert lite_api_key_helper_configured("https://proxy.example.com", tmp_path / "absent.json") is False - helperless = json.dumps({"env": {"ANTHROPIC_BASE_URL": "https://proxy.example.com"}}) - settings_path = self._settings(tmp_path, helperless) - assert lite_api_key_helper_configured("https://proxy.example.com", settings_path) is False - - def test_unreadable_settings_fall_back_to_false(self, tmp_path, lite_on_path): - settings_path = self._settings(tmp_path, "{not json") - - assert lite_api_key_helper_configured("https://proxy.example.com", settings_path) is False - - def test_lite_missing_from_path_falls_back_to_false(self, tmp_path): - helper = "/usr/local/bin/lite --base-url https://proxy.example.com auth print-token" - settings_path = self._settings(tmp_path, json.dumps({"apiKeyHelper": helper})) - with patch(f"{CLAUDE_SETTINGS_MODULE}.shutil.which", return_value=None): - assert lite_api_key_helper_configured("https://proxy.example.com", settings_path) is False - - class TestMergeClaudeSettings: - """One merge for every way Claude Code gets wired: `lite up`, `lite login --config-claude`, - `lite configure claude` and `lite autoroute up`.""" + """One merge for every way Claude Code gets wired: `lite up`, `lite configure claude` and `lite autoroute up`.""" def test_a_static_token_lands_in_env_and_the_helper_slot_is_cleared(self): settings = {"apiKeyHelper": "/usr/local/bin/lite auth print-token", "env": {"ANTHROPIC_API_KEY": "leaked"}} @@ -523,12 +311,6 @@ class TestMergeClaudeSettings: assert "model" not in merged assert not any(key in merged["env"] for key in ANTHROPIC_DEFAULT_MODEL_ENV_KEYS) - def test_a_helper_lands_top_level_and_the_static_slots_are_cleared(self): - settings = {"env": {"ANTHROPIC_AUTH_TOKEN": "sk-old", "ANTHROPIC_API_KEY": "leaked"}} - merged = merge_claude_settings(settings, "http://127.0.0.1:4000", ApiKeyHelper("lite auth print-token")) - assert merged["apiKeyHelper"] == "lite auth print-token" - assert "ANTHROPIC_AUTH_TOKEN" not in merged["env"] and "ANTHROPIC_API_KEY" not in merged["env"] - def test_keeps_existing_switch_values_and_unrelated_keys_without_mutating_the_input(self): settings = {"theme": "dark", "env": {"SOME_OTHER_VAR": "value", "ENABLE_TOOL_SEARCH": "false"}} merged = merge_claude_settings(settings, "http://127.0.0.1:4000", StaticToken("token-abc")) @@ -537,13 +319,22 @@ class TestMergeClaudeSettings: assert merged["env"]["ENABLE_TOOL_SEARCH"] == "false" assert settings == {"theme": "dark", "env": {"SOME_OTHER_VAR": "value", "ENABLE_TOOL_SEARCH": "false"}} - def test_a_default_model_sets_only_the_row_claude_code_starts_on(self): + def test_a_default_model_pins_the_starting_row_and_the_model_a_resumed_session_keeps(self): + # `model` is only the row Claude Code starts on: a resumed session re-sends the model its transcript + # recorded, which behind a raw-model auto-router is the tier model (403 for a key scoped to the + # router). ANTHROPIC_MODEL outranks the transcript on resume, so the pin has to land there too. merged = merge_claude_settings( {}, "http://127.0.0.1:4000", StaticToken("token-abc"), default_model="claude-auto" ) assert merged["model"] == "claude-auto" + assert merged["env"]["ANTHROPIC_MODEL"] == "claude-auto" assert not any(key in merged["env"] for key in ANTHROPIC_DEFAULT_MODEL_ENV_KEYS) + def test_without_a_default_model_neither_pin_is_written_and_a_users_own_stays(self): + settings = {"model": "mine", "env": {"ANTHROPIC_MODEL": "mine-too"}} + merged = merge_claude_settings(settings, "http://127.0.0.1:4000", StaticToken("token-abc")) + assert merged["model"] == "mine" and merged["env"]["ANTHROPIC_MODEL"] == "mine-too" + def test_a_tier_model_forces_every_claude_code_tier_as_autoroute_needs(self): # Router's auto-router registry is keyed by the literal requested model string with no # wildcard resolution, so `lite autoroute up` overrides the env var each tier reads. @@ -564,7 +355,7 @@ class TestMergeClaudeSettings: "apiKeyHelper": "old-helper", "model": "old-model", } - for credential in (StaticToken("token-abc"), ApiKeyHelper("helper")): + for credential in (StaticToken("token-abc"), StaticToken("token-rotated")): merged = merge_claude_settings(settings, "http://127.0.0.1:4000", credential, default_model="claude-auto") changed_top_level = {key for key in set(settings) | set(merged) if settings.get(key) != merged.get(key)} assert changed_top_level - {"env"} <= set(OWNED_TOP_LEVEL_KEYS) @@ -580,7 +371,7 @@ class TestMergeClaudeSettings: PROXY = "http://127.0.0.1:4000" ANTHROPIC = "https://api.anthropic.com" -HELPER = ApiKeyHelper("lite auth print-token") +RELOGIN = StaticToken("sk-fresh-login") ORIGINAL = { "theme": "dark", "permissions": {"allow": ["Bash"]}, @@ -664,18 +455,19 @@ UNDO_SCENARIOS = { { "restored": { "env.ANTHROPIC_BASE_URL", + "env.ANTHROPIC_AUTH_TOKEN", "env.ENABLE_TOOL_SEARCH", "env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY", - "apiKeyHelper", + "statusLine", }, "kept": (), }, - {"credential": HELPER, "model": KeepModel()}, + {"credential": RELOGIN, "model": KeepModel()}, ), "repeat across credential kinds keeps the first snapshot": ( ORIGINAL, [ - {"credential": HELPER, "model": UnpinModel()}, + {"credential": RELOGIN, "model": UnpinModel()}, {"credential": StaticToken("sk-rotated"), "model": StartOn("claude-sonnet-4-6")}, ], ORIGINAL, @@ -688,13 +480,13 @@ UNDO_SCENARIOS = { {"model": "claude-opus-5"}, {}, ), - "re-login keeps our pin": (None, [{"credential": HELPER, "model": KeepModel()}], None, {"file_removed": True}), + "re-login keeps our pin": (None, [{"credential": RELOGIN, "model": KeepModel()}], None, {"file_removed": True}), "edit between configures survives an unpin repeat": ( ORIGINAL, [ _set("model", "my-favourite"), _set("env.ENABLE_TOOL_SEARCH", "false"), - {"credential": HELPER, "model": UnpinModel()}, + {"credential": RELOGIN, "model": UnpinModel()}, ], {**ORIGINAL, "env": {**ORIGINAL["env"], "ENABLE_TOOL_SEARCH": "false"}, "model": "my-favourite"}, {"kept": {"env.ENABLE_TOOL_SEARCH", "model"}}, @@ -704,14 +496,14 @@ UNDO_SCENARIOS = { [ _set("model", "my-favourite"), _set("env.ENABLE_TOOL_SEARCH", "false"), - {"credential": HELPER, "model": KeepModel()}, + {"credential": RELOGIN, "model": KeepModel()}, ], {**ORIGINAL, "env": {**ORIGINAL["env"], "ENABLE_TOOL_SEARCH": "false"}, "model": "my-favourite"}, {"kept": {"env.ENABLE_TOOL_SEARCH", "model"}}, ), "edit between configures: a same-model repeat displaces it, so it is what comes back": ( ORIGINAL, - [_set("model", "my-favourite"), _set("env.ENABLE_TOOL_SEARCH", "false"), {"credential": HELPER}], + [_set("model", "my-favourite"), _set("env.ENABLE_TOOL_SEARCH", "false"), {"credential": RELOGIN}], {**ORIGINAL, "env": {**ORIGINAL["env"], "ENABLE_TOOL_SEARCH": "false"}, "model": "my-favourite"}, {"kept": {"env.ENABLE_TOOL_SEARCH"}, "restored_includes": {"model"}}, ), @@ -744,12 +536,12 @@ UNDO_SCENARIOS = { None, [ _set("env.ANTHROPIC_API_KEY", "sk-user"), - {"credential": HELPER, "model": KeepModel()}, + {"credential": RELOGIN, "model": KeepModel()}, _set("env.ANTHROPIC_BASE_URL", _ABSENT), ], None, {"withheld": {("env.ANTHROPIC_API_KEY", PROXY)}, "file_removed": True, "receipt_kept": True}, - {"credential": HELPER, "model": KeepModel()}, + {"credential": RELOGIN, "model": KeepModel()}, ), "a credential the user changed is kept, never also withheld": ( ORIGINAL, @@ -832,11 +624,10 @@ class TestConfigureAndUnconfigure: @pytest.mark.parametrize( ("path", "value", "repeat_credential"), [ - ("env.ANTHROPIC_API_KEY", "sk-user-added-later", HELPER), - ("env.ANTHROPIC_AUTH_TOKEN", "sk-users-own-token", HELPER), + ("env.ANTHROPIC_API_KEY", "sk-user-added-later", RELOGIN), ("apiKeyHelper", "/opt/mine/helper", StaticToken("sk-rotated")), ], - ids=["user-adds-api-key", "user-replaces-our-token", "user-sets-own-helper"], + ids=["user-adds-api-key", "user-sets-own-helper"], ) def test_a_credential_the_user_set_between_two_configures_is_what_comes_back( self, tmp_path, path, value, repeat_credential @@ -844,7 +635,7 @@ class TestConfigureAndUnconfigure: # The repeat's merge clears the slot, so the displaced value is snapshotted and is what returns; # it was set while the file pointed at the proxy, so it returns once the file points there again. rig = _Rig(tmp_path, {"theme": "dark"}) - rig.configure(credential=HELPER, model=KeepModel()) + rig.configure(credential=RELOGIN, model=KeepModel()) rig.edit(_set(path, value)) rig.configure(credential=repeat_credential, model=KeepModel()) assert not _lookup(rig.read(), path) @@ -955,3 +746,95 @@ class TestConfigureAndUnconfigure: def _lookup(settings, path): section, _, key = path.rpartition(".") return (settings.get(section) or {}).get(key) if section else settings.get(key) + + +class TestStatusLine: + """Every configure registers the status line, and only while the slot is empty or already ours.""" + + COMMAND = "/opt/lite/bin/python /Users/me/.litellm/statusline.py" + + def test_an_empty_slot_gets_our_status_line(self): + assert with_status_line({}, self.COMMAND)["statusLine"] == {"type": "command", "command": self.COMMAND} + + def test_a_users_own_status_line_is_never_replaced(self): + theirs = {"type": "command", "command": "~/.claude/my-statusline.sh"} + assert with_status_line({"statusLine": theirs}, self.COMMAND)["statusLine"] == theirs + + def test_ours_under_an_older_interpreter_is_refreshed(self): + stale = {"type": "command", "command": "/old/python /Users/me/.litellm/statusline.py"} + assert with_status_line({"statusLine": stale}, self.COMMAND)["statusLine"]["command"] == self.COMMAND + + def test_the_merge_carries_it(self): + merged = merge_claude_settings({}, PROXY, StaticToken("tok"), status_line=self.COMMAND) + assert merged["statusLine"] == {"type": "command", "command": self.COMMAND} + + def test_the_installed_script_is_the_bundled_one_and_the_command_runs_this_interpreter(self, tmp_path): + from litellm.proxy.client.cli.commands import statusline_script + + script = tmp_path / "lite" / "statusline.py" + command = install_statusline_script(script) + assert script.read_bytes() == pathlib.Path(statusline_script.__file__).read_bytes() + assert shlex.split(command) == [sys.executable, str(script)] + assert command == statusline_command(script) + assert stat.S_IMODE(script.stat().st_mode) == 0o600 + assert stat.S_IMODE(script.parent.stat().st_mode) == 0o700 + assert install_statusline_script(script) == command + + def test_a_reinstall_replaces_the_script_in_one_step_and_a_refused_one_leaves_the_old_script_whole(self, tmp_path): + # Claude Code may be running the script at the moment `lite` reinstalls it; the file it has open + # must stay complete, and a reinstall that cannot land must not leave a truncated script behind. + from litellm.proxy.client.cli.commands import statusline_script + + script = tmp_path / "lite" / "statusline.py" + install_statusline_script(script) + bundled = pathlib.Path(statusline_script.__file__).read_bytes() + with script.open("rb") as running: + install_statusline_script(script) + assert running.read() == bundled + assert [child.name for child in script.parent.iterdir()] == ["statusline.py"] + + if os.geteuid() != 0: + script.parent.chmod(0o500) + try: + with pytest.raises(ClaudeSettingsError, match="Could not install the status line script"): + install_statusline_script(script) + finally: + script.parent.chmod(0o700) + assert script.read_bytes() == bundled + + def test_configure_installs_it_and_unconfigure_removes_only_ours(self, tmp_path): + rig = _Rig(tmp_path, {"theme": "dark"}) + script = tmp_path / "statusline.py" + rig.configure(script_path=script) + assert rig.read()["statusLine"]["command"] == statusline_command(script) + assert script.exists() + + outcome = rig.unconfigure() + assert rig.read() == {"theme": "dark"} + assert "statusLine" in outcome.restored + + def test_a_status_line_the_user_replaced_after_configure_survives_unconfigure(self, tmp_path): + rig = _Rig(tmp_path, None) + rig.configure(model=KeepModel(), script_path=tmp_path / "statusline.py") + theirs = {"type": "command", "command": "~/.claude/my-statusline.sh"} + rig.edit(lambda settings: {**settings, "statusLine": theirs}) + + outcome = rig.unconfigure() + assert rig.read()["statusLine"] == theirs + assert "statusLine" in outcome.kept + + def test_a_receipt_from_before_the_status_line_existed_still_unconfigures(self, tmp_path): + # Older receipts never claimed statusLine; a key no configure wrote is never ours, so it stays. + rig = _Rig(tmp_path, None) + script = tmp_path / "statusline.py" + rig.configure(model=KeepModel(), script_path=script) + receipt = json.loads(rig.state.read_text()) + receipt["written"].pop("statusLine") + receipt["previous"].pop("statusLine") + rig.state.write_text(json.dumps(receipt)) + + outcome = rig.unconfigure() + restored = rig.read() + assert "ANTHROPIC_AUTH_TOKEN" not in restored.get("env", {}) + assert restored["statusLine"]["command"] == statusline_command(script) + assert "statusLine" not in outcome.restored diff --git a/tests/test_litellm/proxy/client/cli/test_configure_commands.py b/tests/test_litellm/proxy/client/cli/test_configure_commands.py index ac7408339fe..8ed188af737 100644 --- a/tests/test_litellm/proxy/client/cli/test_configure_commands.py +++ b/tests/test_litellm/proxy/client/cli/test_configure_commands.py @@ -86,6 +86,7 @@ class TestConfigureClaudeWithAVirtualKey: assert "ANTHROPIC_DEFAULT_SONNET_MODEL" not in written["env"] assert state_path.exists() assert VALID_KEY not in result.output + assert written["env"]["ANTHROPIC_MODEL"] == "claude-auto" assert "Starting model: claude-auto" in result.output assert "1 of the proxy's 2 models" in result.output assert "lite unconfigure claude" in result.output @@ -99,7 +100,7 @@ class TestConfigureClaudeWithAVirtualKey: assert result.exit_code == 0, result.output written = json.loads(settings_path.read_text()) assert written["env"]["ANTHROPIC_AUTH_TOKEN"] == VALID_KEY - assert "model" not in written + assert "model" not in written and "ANTHROPIC_MODEL" not in written["env"] assert "Starting model: not pinned" in result.output @responses.activate @@ -159,18 +160,11 @@ class TestConfigureClaudeWithAVirtualKey: assert not settings_path.exists() @responses.activate - @pytest.mark.parametrize("entry", ["virtual-key", "login", "interactive"]) - def test_refuses_while_lite_up_holds_a_backup_before_any_login_or_request( - self, runner, paths, monkeypatch, lite_up_backup, entry - ): + @pytest.mark.parametrize("entry", ["virtual-key", "no-key", "interactive"]) + def test_refuses_while_lite_up_holds_a_backup_before_any_request(self, runner, paths, lite_up_backup, entry): _mock_models() - - def login_must_not_run(ctx): - raise AssertionError("the local precondition must be checked before a login is attempted") - - monkeypatch.setattr(configure_module, "ensure_fresh_login", login_must_not_run) if entry == "interactive": - ctx = click.Context(configure_group, obj={"base_url": PROXY, "api_key": None}) + ctx = click.Context(configure_group, obj={"base_url": PROXY, "api_key": VALID_KEY}) with pytest.raises(click.ClickException, match="lite down"): interactive_configure(ctx, pick_targets=lambda: ("claude",), pick_model=lambda listed: None) else: @@ -195,33 +189,27 @@ class TestConfigureClaudeWithAVirtualKey: assert json.loads(target.read_text())["env"]["ANTHROPIC_AUTH_TOKEN"] == VALID_KEY -class TestConfigureClaudeWithTheLogin: - def _stored_login(self, monkeypatch): - monkeypatch.setattr(configure_module, "ensure_fresh_login", lambda ctx: None) - monkeypatch.setattr(configure_module, "get_stored_api_key", lambda expected_base_url, vault: VALID_KEY) - +class TestConfigureClaudeWithoutAKey: @responses.activate - def test_uses_the_login_through_the_helper_and_writes_no_secret(self, runner, paths, monkeypatch, lite_on_path): + def test_refuses_and_names_the_ways_to_pass_a_key_without_writing_or_logging_in(self, runner, paths): + # A `lite login` credential expires within a day; the old fallback wrote an apiKeyHelper that made + # Claude Code spawn `lite` (and its keychain probe) on every credential refresh. _mock_models() - self._stored_login(monkeypatch) - settings_path, _ = paths + settings_path, state_path = paths result = runner.invoke( configure_claude, ["--model", "claude-auto"], - obj={"base_url": PROXY, "api_key": VALID_KEY, "api_key_from_token_file": True}, + obj={"base_url": PROXY, "api_key": "sk-login-jwt", "api_key_from_token_file": True}, ) - assert result.exit_code == 0, result.output - written = json.loads(settings_path.read_text()) - assert written["apiKeyHelper"] == f"{lite_on_path} --base-url {PROXY} auth print-token" - assert "ANTHROPIC_AUTH_TOKEN" not in written["env"] - assert written["model"] == "claude-auto" - assert VALID_KEY not in settings_path.read_text() - assert "read through apiKeyHelper" in result.output + assert result.exit_code != 0 + assert "--api-key" in result.output and "LITELLM_PROXY_API_KEY" in result.output + assert "apiKeyHelper" not in result.output + assert not settings_path.exists() and not state_path.exists() + assert len(responses.calls) == 0 @responses.activate - def test_an_explicit_key_still_wins_over_a_stored_login(self, runner, paths, monkeypatch, lite_on_path): + def test_an_explicit_key_still_wins_over_a_stored_login(self, runner, paths): _mock_models() - self._stored_login(monkeypatch) settings_path, _ = paths result = runner.invoke( configure_claude, @@ -302,11 +290,13 @@ class TestUnconfigureClaude: edited = json.loads(settings_path.read_text()) edited["env"] = {key: f"{value}-edited" for key, value in edited["env"].items()} edited["model"] = "mine" + edited["statusLine"] = {"type": "command", "command": "~/.claude/my-statusline.sh"} settings_path.write_text(json.dumps(edited)) result = runner.invoke(cli, ["unconfigure", "claude"]) assert result.exit_code == 0, result.output assert "Nothing in" in result.output and "was still ours to restore" in result.output assert "Left as you changed them since:" in result.output and "model" in result.output + assert "statusLine" in result.output @responses.activate def test_names_the_server_a_withheld_credential_was_captured_with_and_keeps_the_receipt(self, runner, paths): diff --git a/tests/test_litellm/proxy/client/cli/test_statusline_script.py b/tests/test_litellm/proxy/client/cli/test_statusline_script.py new file mode 100644 index 00000000000..691764fbef4 --- /dev/null +++ b/tests/test_litellm/proxy/client/cli/test_statusline_script.py @@ -0,0 +1,404 @@ +"""The status line script is copied verbatim to the user's machine, so these drive it the way Claude Code +and Codex do: the documented stdin payload, a transcript on disk, and the proxy behind an injected fetch.""" + +import io +import json +import os +import re +import subprocess +import sys +from pathlib import Path + +import pytest + +from litellm.proxy.client.cli.commands import statusline_script +from litellm.proxy.client.cli.commands.statusline_script import ( + CACHE_TTL_SECONDS, + Credentials, + Fetched, + Session, + cache_dir_name, + cache_path, + claude_credentials, + codex_credentials, + latest_transcript_model, + load_session, + render, + run, +) + +SESSION_ID = "cf712ab8-4c7c-4d48-ba91-eed54bc2956b" +ANSI = re.compile(r"\x1b\[[0-9;]*m") +RECORDED = Session( + router_name="claude-auto", + last_model="anthropic/claude-sonnet-5", + spend=0.14, + baseline_spend=0.38, + baseline_model="anthropic/claude-opus-5", +) + + +def _assistant_line(model: str, **extra: object) -> str: + return json.dumps({"type": "assistant", "message": {"model": model, "role": "assistant"}, **extra}) + + +@pytest.fixture +def transcript(tmp_path: Path) -> Path: + path = tmp_path / "session.jsonl" + path.write_text( + "\n".join( + ( + json.dumps({"type": "user", "message": {"role": "user", "content": "hi"}}), + _assistant_line("claude-haiku-4-5"), + json.dumps({"type": "user", "message": {"role": "user", "content": "harder"}}), + _assistant_line("claude-sonnet-5"), + _assistant_line("claude-haiku-4-5", isSidechain=True), + _assistant_line("claude-haiku-4-5", agentId="agent-1"), + json.dumps({"type": "progress", "data": {}}), + ) + ) + + "\n" + ) + return path + + +@pytest.fixture +def config_dir(tmp_path: Path) -> Path: + directory = tmp_path / "claude" + (directory / "cache").mkdir(parents=True) + (directory / "cache" / "gateway-models.json").write_text( + json.dumps({"models": [{"id": "claude-opus-5", "display_name": "Claude Opus 5"}]}) + ) + return directory + + +def _payload(transcript: Path, session_id: str = SESSION_ID) -> dict: + return { + "session_id": session_id, + "transcript_path": str(transcript), + "model": {"id": "claude-auto", "display_name": "claude-auto"}, + } + + +def _env(tmp_path: Path, config_dir: Path, **extra: str) -> dict[str, str]: + return { + "TMPDIR": str(tmp_path / "tmp"), + "CLAUDE_CONFIG_DIR": str(config_dir), + "TERM": "dumb", + "ANTHROPIC_BASE_URL": "http://127.0.0.1:4000", + "ANTHROPIC_AUTH_TOKEN": "sk-virtual", + **extra, + } + + +def _run(payload: object, env: dict[str, str], fetch) -> str: + out = io.StringIO() + run(io.StringIO(json.dumps(payload)), out, env, fetch) + return out.getvalue() + + +class TestTranscript: + def test_the_latest_foreground_assistant_line_wins_over_later_sidechain_and_agent_lines(self, transcript): + assert latest_transcript_model(str(transcript)) == "claude-sonnet-5" + + def test_a_synthetic_line_is_not_a_served_model(self, tmp_path): + # Claude Code writes `` for messages it produced locally (an API error on resume, for one); + # showing "Routed to: " would name a model no proxy served. + path = tmp_path / "t.jsonl" + path.write_text(_assistant_line("claude-haiku-4-5") + "\n" + _assistant_line("") + "\n") + assert latest_transcript_model(str(path)) == "claude-haiku-4-5" + + def test_a_missing_or_empty_transcript_yields_nothing(self, tmp_path): + empty = tmp_path / "empty.jsonl" + empty.write_text("") + assert latest_transcript_model(str(tmp_path / "missing.jsonl")) == "" + assert latest_transcript_model(str(empty)) == "" + assert latest_transcript_model("") == "" + + +class TestCredentials: + MIXED = { + "ANTHROPIC_BASE_URL": "http://anthropic-side:4000", + "ANTHROPIC_AUTH_TOKEN": "sk-ant", + "OPENAI_BASE_URL": "http://openai-side:4000/v1/", + "OPENAI_API_KEY": "sk-openai", + } + + def test_each_agent_reads_the_pair_it_dials_itself(self): + # A shell that exports both families must not send Codex's hook to the Anthropic proxy. + assert claude_credentials(self.MIXED) == Credentials("http://anthropic-side:4000", "sk-ant") + assert codex_credentials(self.MIXED) == Credentials("http://openai-side:4000", "sk-openai") + + def test_lites_own_shell_variables_are_not_a_credential_either_agent_sends(self): + # A `lite login` shell exports LITELLM_PROXY_*; Claude Code and Codex never read them, so the + # status line must not query the proxy as that principal while the agent used another. + env = {"LITELLM_PROXY_URL": "http://lite:4000/", "LITELLM_PROXY_API_KEY": "sk-lite", **self.MIXED} + assert claude_credentials(env) == Credentials("http://anthropic-side:4000", "sk-ant") + assert codex_credentials(env) == Credentials("http://openai-side:4000", "sk-openai") + assert not claude_credentials({"LITELLM_PROXY_API_KEY": "sk-lite", "ANTHROPIC_BASE_URL": "http://p"}).usable + assert codex_credentials({}) == Credentials("", "") + + def test_claude_code_prefers_the_auth_token_over_a_stray_api_key(self): + env = {"ANTHROPIC_BASE_URL": "http://p", "ANTHROPIC_API_KEY": "sk-stray", "ANTHROPIC_AUTH_TOKEN": "sk-ours"} + assert claude_credentials(env).api_key == "sk-ours" + + def test_an_api_key_helper_in_settings_is_never_run(self, tmp_path, transcript, config_dir): + # `lite` once wrote `apiKeyHelper: lite auth print-token`; running it from a status line that + # refreshes every 300ms spawned `lite` (and a keychain prompt) on every refresh. Without a key + # in the env the proxy is simply not asked. + (config_dir / "settings.json").write_text(json.dumps({"apiKeyHelper": "printf sk-from-helper"})) + asked = [] + env = {k: v for k, v in _env(tmp_path, config_dir).items() if k != "ANTHROPIC_AUTH_TOKEN"} + text = _run(_payload(transcript), env, lambda c, s: asked.append(c) or Fetched(RECORDED, True)) + assert text == "Routed to: claude-sonnet-5" and asked == [] + + +class TestSessionCache: + def test_a_definite_answer_is_served_from_the_cache_within_the_ttl(self, tmp_path): + calls = [] + + def fetch(credentials, session_id): + calls.append(session_id) + return Fetched(RECORDED, definitive=True) + + clock = [100.0] + credentials = Credentials("http://p", "sk") + first = load_session(credentials, SESSION_ID, tmp_path, fetch, now=lambda: clock[0]) + clock[0] = 100.0 + CACHE_TTL_SECONDS - 1 + second = load_session(credentials, SESSION_ID, tmp_path, fetch, now=lambda: clock[0]) + clock[0] = 100.0 + CACHE_TTL_SECONDS + 1 + third = load_session(credentials, SESSION_ID, tmp_path, fetch, now=lambda: clock[0]) + assert first == second == third == RECORDED + assert calls == [SESSION_ID, SESSION_ID] + + def test_a_404_is_cached_as_absence_but_a_transport_failure_is_retried(self, tmp_path): + outcomes = iter((Fetched(None, definitive=False), Fetched(None, definitive=True), Fetched(RECORDED, True))) + calls = [] + + def fetch(credentials, session_id): + calls.append(session_id) + return next(outcomes) + + credentials = Credentials("http://p", "sk") + assert load_session(credentials, SESSION_ID, tmp_path, fetch, now=lambda: 1.0) is None + assert load_session(credentials, SESSION_ID, tmp_path, fetch, now=lambda: 1.0) is None + assert load_session(credentials, SESSION_ID, tmp_path, fetch, now=lambda: 1.0) is None + assert len(calls) == 2 + + def test_a_cache_directory_that_is_not_private_is_never_used(self, tmp_path): + # A shared temp root lets another user pre-create the directory; refuse it rather than write into it. + calls = [] + + def fetch(credentials, session_id): + calls.append(session_id) + return Fetched(RECORDED, definitive=True) + + shared = tmp_path / "litellm-statusline" + shared.mkdir(mode=0o755) + credentials = Credentials("http://p", "sk") + for _ in range(2): + assert load_session(credentials, SESSION_ID, shared, fetch, now=lambda: 1.0) == RECORDED + assert calls == [SESSION_ID, SESSION_ID] + assert list(shared.iterdir()) == [] + + def test_any_client_error_is_a_definite_answer_and_a_server_error_is_not(self, monkeypatch): + import urllib.error + + from litellm.proxy.client.cli.commands.statusline_script import fetch_session + + def fail_with(code): + def opener(request, timeout): + raise urllib.error.HTTPError(request.full_url, code, "x", {}, None) + + return opener + + for code, definitive in ((403, True), (401, True), (404, True), (502, False)): + monkeypatch.setattr("urllib.request.urlopen", fail_with(code)) + assert fetch_session(Credentials("http://127.0.0.1:1", "sk"), SESSION_ID) == Fetched(None, definitive) + + def test_the_cache_file_holds_the_proxy_answer_and_never_the_key(self, tmp_path): + credentials = Credentials("http://p", "sk-secret") + load_session(credentials, SESSION_ID, tmp_path, lambda c, s: Fetched(RECORDED, True)) + path = cache_path(tmp_path, credentials, SESSION_ID) + assert "sk-secret" not in written and SESSION_ID not in written if (written := path.read_text()) else False + assert "sk-secret" not in path.name + assert json.loads(written)["session"]["baseline_model"] == "anthropic/claude-opus-5" + assert (path.stat().st_mode & 0o777) == 0o600 + assert (path.parent.stat().st_mode & 0o777) == 0o700 + + def test_a_refresh_replaces_the_entry_in_one_step_so_a_concurrent_refresh_never_reads_a_torn_one(self, tmp_path): + credentials = Credentials("http://p", "sk") + load_session(credentials, SESSION_ID, tmp_path, lambda c, s: Fetched(RECORDED, True), now=lambda: 1.0) + path = cache_path(tmp_path, credentials, SESSION_ID) + first = path.read_text() + + with path.open() as concurrent_reader: + newer = RECORDED._replace(spend=0.5) + load_session(credentials, SESSION_ID, tmp_path, lambda c, s: Fetched(newer, True), now=lambda: 100.0) + assert concurrent_reader.read() == first + assert json.loads(path.read_text())["session"]["spend"] == 0.5 + assert (path.stat().st_mode & 0o777) == 0o600 + assert [child.name for child in tmp_path.iterdir()] == [path.name] + + def test_the_same_session_id_against_another_proxy_or_key_is_not_served_from_the_cache(self, tmp_path): + answers = iter((Fetched(RECORDED, True), Fetched(RECORDED._replace(spend=9.0), True))) + first = load_session(Credentials("http://p", "sk-a"), SESSION_ID, tmp_path, lambda c, s: next(answers)) + second = load_session(Credentials("http://p", "sk-b"), SESSION_ID, tmp_path, lambda c, s: next(answers)) + assert first == RECORDED and second is not None and second.spend == 9.0 + + +class TestRender: + def test_savings_header_and_bars_against_the_routers_baseline(self, config_dir): + text = render("claude-sonnet-5", RECORDED, config_dir, use_color=False, bar_width=10) + assert text.splitlines() == [ + "claude-auto · Routed to: claude-sonnet-5 -63% vs Claude Opus 5", + "LiteLLM ████░░░░░░ $0.14", + "Claude Opus 5 ██████████ $0.38", + ] + + def test_control_characters_in_any_externally_sourced_label_never_reach_the_terminal(self, tmp_path, config_dir): + # The transcript, the proxy payload and Claude Code's model cache all feed labels straight into a + # terminal, and none is under this script's control. Only the control bytes are dropped (ESC, BEL, + # C1), which is what disarms an OSC-52 clipboard write or a screen clear; the printable remainder of + # such a sequence is inert text and is kept as is. + from litellm.proxy.client.cli.commands.statusline_script import _session_from_payload, model_label + + hostile = "claude-\x1b\x07\x9bsonnet" + path = tmp_path / "t.jsonl" + path.write_text(_assistant_line(hostile) + "\n") + assert latest_transcript_model(str(path)) == "claude-sonnet" + (config_dir / "cache" / "gateway-models.json").write_text( + json.dumps({"models": [{"id": "claude-sonnet", "display_name": "Son\x1b\x07net"}]}) + ) + assert model_label("claude-sonnet", config_dir) == "Sonnet" + session = _session_from_payload( + {"router_name": "auto\x07", "last_model": hostile, "spend": 0.1, "baseline_spend": 0.2, "baseline_model": "op\x1bus"} + ) + assert session == Session("auto", "claude-sonnet", 0.1, 0.2, "opus") + assert latest_transcript_model(str(path)) == "claude-sonnet" + assert "\x1b]52;c;ZXZpbA==" not in render( + latest_transcript_model(str(path)), + _session_from_payload({"router_name": "a", "last_model": "m", "spend": 0.1, "baseline_spend": 0.2, "baseline_model": "\x1b]52;c;ZXZpbA==\x07"}), + config_dir, + use_color=False, + ) + + def test_a_session_that_cost_more_than_its_baseline_reads_as_a_plus(self, config_dir): + dearer = RECORDED._replace(spend=0.50, baseline_spend=0.40) + assert "+25% vs Claude Opus 5" in render("m", dearer, config_dir, use_color=False) + + def test_without_a_baseline_only_the_routed_line_shows(self, config_dir): + assert render("m", RECORDED._replace(baseline_model=None), config_dir, False) == "claude-auto · Routed to: m" + assert render("m", None, config_dir, False) == "Routed to: m" + + def test_color_wraps_the_same_text(self, config_dir): + colored = render("claude-sonnet-5", RECORDED, config_dir, use_color=True, bar_width=10) + assert ANSI.sub("", colored) == render("claude-sonnet-5", RECORDED, config_dir, use_color=False, bar_width=10) + + +class TestClaudeCodeMode: + def test_the_transcript_names_the_routed_model_and_the_proxy_adds_the_savings(self, tmp_path, transcript, config_dir): + seen = [] + + def fetch(credentials, session_id): + seen.append((credentials, session_id)) + return Fetched(RECORDED, definitive=True) + + text = _run(_payload(transcript), _env(tmp_path, config_dir), fetch) + assert text.startswith("claude-auto · Routed to: claude-sonnet-5 -63% vs Claude Opus 5\n") + assert seen == [(Credentials("http://127.0.0.1:4000", "sk-virtual"), SESSION_ID)] + + def test_an_unrecorded_session_degrades_to_the_routed_line(self, tmp_path, transcript, config_dir): + assert _run(_payload(transcript), _env(tmp_path, config_dir), lambda c, s: Fetched(None, True)) == ( + "Routed to: claude-sonnet-5" + ) + + def test_without_credentials_the_proxy_is_never_asked(self, tmp_path, transcript, config_dir): + def fetch(credentials, session_id): + raise AssertionError("must not fetch") + + env = {k: v for k, v in _env(tmp_path, config_dir).items() if k != "ANTHROPIC_AUTH_TOKEN"} + assert _run(_payload(transcript), env, fetch) == "Routed to: claude-sonnet-5" + + def test_before_the_first_response_the_payloads_display_name_shows(self, tmp_path, config_dir): + payload = _payload(tmp_path / "missing.jsonl") + assert _run(payload, _env(tmp_path, config_dir), lambda c, s: Fetched(RECORDED, True)) == "claude-auto" + + def test_a_discovered_display_name_labels_the_routed_model(self, tmp_path, config_dir): + path = tmp_path / "t.jsonl" + path.write_text(_assistant_line("anthropic/claude-opus-5") + "\n") + assert _run(_payload(path), _env(tmp_path, config_dir), lambda c, s: Fetched(None, True)) == ( + "Routed to: Claude Opus 5" + ) + + def test_the_cache_lands_under_the_platforms_temp_dir(self, tmp_path, transcript, config_dir): + env = {k: v for k, v in _env(tmp_path, config_dir).items() if k != "TMPDIR"} + env["TEMP"] = str(tmp_path / "wintemp") + _run(_payload(transcript), env, lambda c, s: Fetched(RECORDED, True)) + assert (tmp_path / "wintemp" / cache_dir_name()).is_dir() + assert cache_dir_name().endswith(str(os.getuid())) + + def test_a_crash_falls_back_to_the_model_label_claude_code_already_knows(self, tmp_path, transcript, config_dir): + def fetch(credentials, session_id): + raise RuntimeError("boom") + + assert _run(_payload(transcript), _env(tmp_path, config_dir), fetch) == "claude-auto" + + def test_garbage_on_stdin_still_prints_something(self, tmp_path, config_dir): + out = io.StringIO() + run(io.StringIO("not json"), out, _env(tmp_path, config_dir), lambda c, s: Fetched(None, True)) + assert out.getvalue() == "claude" + + +class TestCodexMode: + def test_the_stop_hook_prints_a_system_message_from_the_proxys_record(self, tmp_path, config_dir): + env = _env(tmp_path, config_dir, OPENAI_BASE_URL="http://127.0.0.1:4000/v1", OPENAI_API_KEY="sk-codex") + env = {k: v for k, v in env.items() if not k.startswith("ANTHROPIC_")} + seen = [] + + def fetch(credentials, session_id): + seen.append(credentials) + return Fetched(RECORDED, definitive=True) + + out = _run({"hook_event_name": "Stop", "session_id": SESSION_ID, "transcript_path": "/nope"}, env, fetch) + message = json.loads(out)["systemMessage"] + assert message.splitlines()[1] == "claude-auto · Routed to: claude-sonnet-5 -63% vs Claude Opus 5" + assert message.startswith("\n") + assert seen == [Credentials("http://127.0.0.1:4000", "sk-codex")] + + def test_an_unrecorded_session_prints_nothing_so_codex_shows_no_message(self, tmp_path, config_dir): + payload = {"hook_event_name": "Stop", "session_id": SESSION_ID} + assert _run(payload, _env(tmp_path, config_dir), lambda c, s: Fetched(None, True)) == "" + + def test_a_crash_prints_nothing_rather_than_text_codex_would_reject(self, tmp_path, config_dir): + def fetch(credentials, session_id): + raise RuntimeError("boom") + + env = _env(tmp_path, config_dir, OPENAI_BASE_URL="http://127.0.0.1:4000/v1", OPENAI_API_KEY="sk-codex") + assert _run({"hook_event_name": "Stop", "session_id": SESSION_ID}, env, fetch) == "" + + def test_a_turn_right_after_an_unrecorded_one_still_asks_the_proxy(self, tmp_path, config_dir): + # One hook run per turn: a miss on turn one must not be cached across turn two's fetch. + answers = iter((Fetched(None, definitive=True), Fetched(RECORDED, definitive=True))) + payload = {"hook_event_name": "Stop", "session_id": SESSION_ID} + env = _env(tmp_path, config_dir, OPENAI_BASE_URL="http://127.0.0.1:4000/v1", OPENAI_API_KEY="sk-codex") + assert _run(payload, env, lambda c, s: next(answers)) == "" + assert "Routed to: claude-sonnet-5" in json.loads(_run(payload, env, lambda c, s: next(answers)))["systemMessage"] + + +class TestStandalone: + def test_the_file_runs_under_a_bare_interpreter_with_no_litellm_on_the_path(self, tmp_path, transcript, config_dir): + # It is copied verbatim to ~/.litellm/statusline.py, so it must be self-contained. + script = tmp_path / "statusline.py" + script.write_bytes(Path(statusline_script.__file__).read_bytes()) + env = {k: v for k, v in _env(tmp_path, config_dir).items() if k != "ANTHROPIC_AUTH_TOKEN"} + completed = subprocess.run( + [sys.executable, "-I", str(script)], + input=json.dumps(_payload(transcript)), + capture_output=True, + text=True, + env=env, + check=True, + timeout=30, + ) + assert completed.stdout == "Routed to: claude-sonnet-5" diff --git a/tests/test_litellm/proxy/client/cli/test_up_commands.py b/tests/test_litellm/proxy/client/cli/test_up_commands.py index 111d2f3d682..ddd54dd1374 100644 --- a/tests/test_litellm/proxy/client/cli/test_up_commands.py +++ b/tests/test_litellm/proxy/client/cli/test_up_commands.py @@ -11,7 +11,7 @@ from click.testing import CliRunner from litellm.proxy.client.cli.commands import up as up_module from litellm.proxy.client.cli.commands.agents import AgentRunError -from litellm.proxy.client.cli.commands.claude_settings import ApiKeyHelper, ClaudeSettingsError +from litellm.proxy.client.cli.commands.claude_settings import ClaudeSettingsError, StaticToken from litellm.proxy.client.cli.commands.up import ( BackupRecord, UpError, @@ -20,7 +20,6 @@ from litellm.proxy.client.cli.commands.up import ( load_json_or_empty, merge_claude_settings, read_backup, - resolve_api_key_helper, restore_claude_settings, up, write_backup, @@ -40,52 +39,54 @@ def _patch_paths(monkeypatch, tmp_path): class TestMergeClaudeSettings: def test_preserves_unrelated_top_level_keys(self): - merged = merge_claude_settings({"theme": "dark"}, "http://localhost:4000", ApiKeyHelper("helper")) + merged = merge_claude_settings({"theme": "dark"}, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert merged["theme"] == "dark" def test_preserves_unrelated_env_keys(self): settings = {"env": {"SOME_OTHER_VAR": "value"}} - merged = merge_claude_settings(settings, "http://localhost:4000", ApiKeyHelper("helper")) + merged = merge_claude_settings(settings, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert merged["env"]["SOME_OTHER_VAR"] == "value" - def test_overrides_base_url_and_helper(self): + def test_overrides_base_url_and_strips_an_old_helper(self): settings = { "env": {"ANTHROPIC_BASE_URL": "https://old.example.com"}, "apiKeyHelper": "old-helper", } - merged = merge_claude_settings(settings, "http://localhost:4000/", ApiKeyHelper("new-helper")) + merged = merge_claude_settings(settings, "http://localhost:4000/", StaticToken("sk-fresh"), status_line="statusline-cmd") assert merged["env"]["ANTHROPIC_BASE_URL"] == "http://localhost:4000" + assert merged["env"]["ANTHROPIC_AUTH_TOKEN"] == "sk-fresh" assert merged["env"]["ENABLE_TOOL_SEARCH"] == "true" assert merged["env"]["CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY"] == "1" - assert merged["apiKeyHelper"] == "new-helper" + assert "apiKeyHelper" not in merged def test_preserves_existing_gateway_model_discovery(self): settings = {"env": {"CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY": "0"}} - merged = merge_claude_settings(settings, "http://localhost:4000", ApiKeyHelper("helper")) + merged = merge_claude_settings(settings, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert merged["env"]["CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY"] == "0" def test_preserves_existing_tool_search(self): settings = {"env": {"ENABLE_TOOL_SEARCH": "false"}} - merged = merge_claude_settings(settings, "http://localhost:4000", ApiKeyHelper("helper")) + merged = merge_claude_settings(settings, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert merged["env"]["ENABLE_TOOL_SEARCH"] == "false" def test_drops_stray_api_key(self): settings = {"env": {"ANTHROPIC_API_KEY": "leaked-key"}} - merged = merge_claude_settings(settings, "http://localhost:4000", ApiKeyHelper("helper")) + merged = merge_claude_settings(settings, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert "ANTHROPIC_API_KEY" not in merged["env"] def test_works_from_empty_settings(self): - merged = merge_claude_settings({}, "http://localhost:4000", ApiKeyHelper("helper")) + merged = merge_claude_settings({}, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert merged["env"] == { "ANTHROPIC_BASE_URL": "http://localhost:4000", + "ANTHROPIC_AUTH_TOKEN": "sk-fresh", "ENABLE_TOOL_SEARCH": "true", "CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY": "1", } - assert merged["apiKeyHelper"] == "helper" + assert "apiKeyHelper" not in merged def test_does_not_mutate_input(self): settings = {"env": {"FOO": "bar"}} - merge_claude_settings(settings, "http://localhost:4000", ApiKeyHelper("helper")) + merge_claude_settings(settings, "http://localhost:4000", StaticToken("sk-fresh"), status_line="statusline-cmd") assert settings == {"env": {"FOO": "bar"}} @@ -154,6 +155,25 @@ class TestBackupRoundTrip: assert restore_claude_settings() is None assert not settings_path.exists() + def test_restore_writes_owner_only_and_through_a_symlink(self, monkeypatch, tmp_path): + # The backup can hold a token the user had in the file before `up`; a plain open() would put it + # back under the umask, and would replace a dotfiles symlink with a regular file. + target = tmp_path / "dotfiles" / "settings.json" + target.parent.mkdir() + target.write_text("{}") + target.chmod(0o644) + settings_path = tmp_path / "settings.json" + settings_path.symlink_to(target) + monkeypatch.setattr(up_module, "CLAUDE_SETTINGS_PATH", settings_path) + monkeypatch.setattr(up_module, "BACKUP_PATH", tmp_path / "backup.json") + write_backup(BackupRecord(existed=True, content={"env": {"ANTHROPIC_AUTH_TOKEN": "sk-theirs"}})) + + restore_claude_settings() + + assert settings_path.is_symlink() + assert json.loads(target.read_text()) == {"env": {"ANTHROPIC_AUTH_TOKEN": "sk-theirs"}} + assert stat.S_IMODE(target.stat().st_mode) == 0o600 + def test_recreates_claude_dir_if_it_was_deleted_while_up_was_running(self, monkeypatch, tmp_path): """If ~/.claude/ is removed while `lite up` holds it open, restoring must recreate the directory rather than crash with FileNotFoundError and strand the backup file, which @@ -216,49 +236,6 @@ class TestBackupRoundTrip: assert not backup_path.exists() -class TestResolveApiKeyHelper: - def test_returns_helper_command_bound_to_the_selected_proxy(self, monkeypatch): - monkeypatch.setattr(shutil, "which", lambda name: "/usr/local/bin/lite") - helper = resolve_api_key_helper("http://localhost:4000") - assert helper == "/usr/local/bin/lite --base-url http://localhost:4000 auth print-token" - - def test_quotes_a_base_url_containing_shell_metacharacters(self, monkeypatch): - monkeypatch.setattr(shutil, "which", lambda name: "/usr/local/bin/lite") - helper = resolve_api_key_helper("http://example.com/path; rm -rf /") - assert helper == "/usr/local/bin/lite --base-url 'http://example.com/path; rm -rf /' auth print-token" - - def test_raises_when_lite_not_on_path(self, monkeypatch): - monkeypatch.setattr(shutil, "which", lambda name: None) - with pytest.raises(ClaudeSettingsError, match="Could not find `lite`"): - resolve_api_key_helper("http://localhost:4000") - - def test_windows_quotes_for_cmd_exe_instead_of_posix_sh(self, monkeypatch): - """cmd.exe takes a single quote literally, so a POSIX-quoted backslashed path is unrunnable.""" - lite_exe = "C:\\Users\\u\\AppData\\Local\\Programs\\Python\\Python313\\Scripts\\lite.EXE" - monkeypatch.setattr(shutil, "which", lambda name: lite_exe) - - helper = resolve_api_key_helper("https://gateway.example.com", platform="win32") - - assert helper == f'"{lite_exe}" "--base-url" "https://gateway.example.com" "auth" "print-token"' - - def test_windows_keeps_a_spaced_path_and_a_metacharacter_url_as_single_tokens(self, monkeypatch): - monkeypatch.setattr(shutil, "which", lambda name: "C:\\Program Files\\LiteLLM\\lite.EXE") - - helper = resolve_api_key_helper("https://gateway.example.com/?a=1&b=2", platform="win32") - - assert helper == ( - '"C:\\Program Files\\LiteLLM\\lite.EXE" "--base-url" "https://gateway.example.com/?a=1&b=2" ' - '"auth" "print-token"' - ) - - def test_non_windows_platforms_keep_posix_quoting(self, monkeypatch): - monkeypatch.setattr(shutil, "which", lambda name: "/usr/local/bin/lite") - - helper = resolve_api_key_helper("http://example.com/path; rm -rf /", platform="darwin") - - assert helper == "/usr/local/bin/lite --base-url 'http://example.com/path; rm -rf /' auth print-token" - - def _make_ctx(base_url): return click.Context(click.Command("test"), obj={"base_url": base_url}) @@ -307,7 +284,8 @@ def _capture_login(monkeypatch, on_login=lambda: None): login_calls = [] @click.pass_context - def fake_login(ctx, pkce=False): + def fake_login(ctx, config_claude=False, pkce=False): + assert config_claude is False, "`lite up` patches settings itself; the login it starts must not also configure" login_calls.append((ctx.obj["base_url"], pkce)) on_login() @@ -317,8 +295,8 @@ def _capture_login(monkeypatch, on_login=lambda: None): class TestEnsureFreshLogin: """A token that is fresh but was issued for a *different* proxy must not be trusted: without - this check, a user logged into proxy A who runs `up --base-url proxy-b` would silently get an - apiKeyHelper wired up around proxy A's real token, which print-token would then hand to proxy B.""" + this check, a user logged into proxy A who runs `up --base-url proxy-b` would silently get proxy A's + real token written into settings pointed at proxy B.""" def test_reuses_a_fresh_token_issued_for_the_same_proxy(self, monkeypatch): _FakeTokenStore( @@ -500,11 +478,13 @@ class TestUpCommand: settings_path, backup_path = _patch_paths(monkeypatch, tmp_path) original = {"theme": "dark"} settings_path.write_text(json.dumps(original)) + settings_path.chmod(0o644) captured = {} def fake_wait(self, timeout=None): captured["settings"] = json.loads(settings_path.read_text()) + captured["settings_mode"] = stat.S_IMODE(settings_path.stat().st_mode) captured["backup_existed"] = backup_path.exists() return True @@ -514,10 +494,6 @@ class TestUpCommand: patch(f"{UP_MODULE}.is_cli_token_fresh", return_value=True), patch(f"{UP_MODULE}.resolve_api_key", return_value="sk-fresh"), patch(f"{UP_MODULE}.verify_proxy_key"), - patch( - f"{UP_MODULE}.resolve_api_key_helper", - return_value="/usr/local/bin/lite auth print-token", - ), patch(f"{UP_MODULE}.signal.signal"), patch(f"{UP_MODULE}.atexit.register"), patch("threading.Event.wait", new=fake_wait), @@ -529,7 +505,11 @@ class TestUpCommand: assert captured["settings"]["theme"] == "dark" assert captured["settings"]["env"]["ANTHROPIC_BASE_URL"] == "http://localhost:4000" assert captured["settings"]["env"]["ENABLE_TOOL_SEARCH"] == "true" - assert captured["settings"]["apiKeyHelper"] == "/usr/local/bin/lite auth print-token" + assert captured["settings"]["env"]["ANTHROPIC_AUTH_TOKEN"] == "sk-fresh" + # The file now carries the key, so the umask (and the file's earlier 0644) must not decide who reads it. + assert captured["settings_mode"] == 0o600 + assert "apiKeyHelper" not in captured["settings"] + assert captured["settings"]["statusLine"]["command"].endswith("statusline.py") assert json.loads(settings_path.read_text()) == original assert not backup_path.exists() @@ -598,7 +578,7 @@ class TestUpCanInvokeTheRealLoginCommand: @click.pass_context def driver(ctx): ctx.obj = {"base_url": "http://127.0.0.1:9"} - ctx.invoke(real_login, pkce=False) + ctx.invoke(real_login, config_claude=False, pkce=False) with patch( f"{AUTH_MODULE}._start_cli_sso_flow", @@ -618,7 +598,7 @@ class TestUpCanInvokeTheRealLoginCommand: @click.pass_context def driver(ctx): ctx.obj = {"base_url": "http://127.0.0.1:9"} - ctx.invoke(real_login, pkce=False) + ctx.invoke(real_login, config_claude=False, pkce=False) with patch(f"{AUTH_MODULE}._start_cli_sso_flow", side_effect=RuntimeError("stop")): CliRunner().invoke(driver, [], standalone_mode=False, env={"CLAUDE_CONFIG_DIR": str(tmp_path)}) diff --git a/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py b/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py index 4507892bd0f..271751a3ff8 100644 --- a/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py +++ b/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py @@ -59,7 +59,8 @@ class TestBuildTransaction: def test_successful_auto_routed_turn_builds_every_field(self): transaction = _build( metadata=_metadata( - usage_object={"prompt_tokens": 90, "cache_read_input_tokens": 5, "cache_creation_input_tokens": 7} + routing_decision={**ROUTING_DECISION, "savings_baseline_model": "anthropic/claude-opus-5"}, + usage_object={"prompt_tokens": 90, "cache_read_input_tokens": 5, "cache_creation_input_tokens": 7}, ) ) assert transaction == AutoRouterTurnTransaction( @@ -77,6 +78,7 @@ class TestBuildTransaction: cache_hit=True, cache_ttl_seconds=300, cache_touched=True, + baseline_model="anthropic/claude-opus-5", ) @pytest.mark.parametrize( @@ -109,6 +111,17 @@ class TestBuildTransaction: transaction = _build() assert transaction is not None and transaction.tier is None + def test_the_baseline_the_turn_was_priced_against_travels_with_the_turn(self): + decision = {**ROUTING_DECISION, "savings_baseline_model": "anthropic/claude-opus-5"} + transaction = _build(metadata=_metadata(routing_decision=decision)) + assert transaction is not None and transaction.baseline_model == "anthropic/claude-opus-5" + + @pytest.mark.parametrize("baseline", [None, "", 3]) + def test_a_decision_without_a_usable_baseline_records_none(self, baseline: object): + decision = {**ROUTING_DECISION, "savings_baseline_model": baseline} + transaction = _build(metadata=_metadata(routing_decision=decision)) + assert transaction is not None and transaction.baseline_model is None + def test_a_priced_classifier_rides_the_turns_spend(self): """The classifier row is excluded from the rollup, so its charge lands here, folded once into the turn that paid for it (GH #38816).""" @@ -215,6 +228,7 @@ def _transaction( session_id: str = "s1", at: datetime = datetime(2026, 8, 1, 12, 0, 0), tier: str | None = "medium", + baseline_model: str | None = "anthropic/claude-opus-5", ) -> AutoRouterTurnTransaction: return AutoRouterTurnTransaction( api_key="k1", @@ -232,6 +246,7 @@ def _transaction( cache_ttl_seconds=None, cache_touched=False, tier=tier, + baseline_model=baseline_model, ) @@ -265,6 +280,7 @@ class TestFlush: None, 0, "medium", + "anthropic/claude-opus-5", ) def test_a_connect_error_retries_the_same_statement(self): diff --git a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py index dc18e0f7d4a..ef843adad98 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py @@ -844,6 +844,143 @@ from litellm.proxy.management_endpoints.auto_router_endpoints import ( from litellm.types.management_endpoints.auto_router_endpoints import SHADOW_EVAL_TURN_VALVE, StartShadowEvalRequest VIEWER = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, api_key="sk-view", user_id="viewer") + + +class TestAutoRouterSession: + """GET /auto_router/session: a key reads its own session's routed model and savings, nothing else.""" + + ROW = { + "router_name": "claude-auto", + "router_type": "complexity", + "first_turn_at": datetime(2026, 9, 1, 12, 0, 0), + "last_turn_at": datetime(2026, 9, 1, 12, 5, 0), + "turns": 3, + "last_model": "anthropic/claude-sonnet-5", + "spend": 0.14, + "saved_spend": 0.24, + "classifier_cost": 0.0, + "tier_turns": {"simple": 1, "complex": 2}, + "baseline_models": {"anthropic/claude-opus-5": 3}, + } + + @staticmethod + def _rig(monkeypatch: pytest.MonkeyPatch, rows: Sequence[Mapping[str, object]]): + from litellm.proxy import proxy_server + + lookups: list[tuple[Mapping[str, object], Mapping[str, object]]] = [] + + class _Table: + async def find_first(self, where: Mapping[str, object], order: Mapping[str, object]): + lookups.append((where, order)) + matching = [r for r in rows if (r["api_key"], r["session_id"]) == (where["api_key"], where["session_id"])] + return max(matching, key=lambda r: r["last_turn_at"], default=None) + + monkeypatch.setattr( + proxy_server, "prisma_client", type("P", (), {"db": type("D", (), {"litellm_autoroutersession": _Table()})()})() + ) + return lookups + + @pytest.mark.asyncio + async def test_a_key_reads_its_own_session_with_the_baseline_its_turns_were_priced_against( + self, monkeypatch: pytest.MonkeyPatch + ): + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + caller = UserAPIKeyAuth(api_key="sk-caller") + self._rig(monkeypatch, [{**self.ROW, "api_key": caller.api_key, "session_id": "sess-1"}]) + response = await get_auto_router_session(user_api_key_dict=caller, session_id="sess-1") + assert response.model_dump() == { + "session_id": "sess-1", + "router_name": "claude-auto", + "router_type": "complexity", + "turns": 3, + "last_model": "anthropic/claude-sonnet-5", + "spend": 0.14, + "saved_spend": 0.24, + "baseline_spend": pytest.approx(0.38), + "baseline_model": "anthropic/claude-opus-5", + "baseline_models": {"anthropic/claude-opus-5": 3}, + } + + @pytest.mark.asyncio + async def test_another_keys_session_is_a_404_even_for_an_admin(self, monkeypatch: pytest.MonkeyPatch): + # The scope is the caller's own key hash, exactly what the spend writer keyed the row under; + # an admin wanting every key's sessions has /auto_router/benchmarks. + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + other = UserAPIKeyAuth(api_key="sk-other") + lookups = self._rig(monkeypatch, [{**self.ROW, "api_key": other.api_key, "session_id": "sess-1"}]) + with pytest.raises(HTTPException) as err: + await get_auto_router_session(user_api_key_dict=ADMIN, session_id="sess-1") + assert err.value.status_code == 404 + assert lookups == [({"api_key": ADMIN.api_key, "session_id": "sess-1"}, {"last_turn_at": "desc"})] + assert ADMIN.api_key != "sk-test" + + @pytest.mark.asyncio + async def test_the_sessions_most_recently_active_router_is_the_one_reported(self, monkeypatch: pytest.MonkeyPatch): + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + older = {**self.ROW, "api_key": ADMIN.api_key, "session_id": "s", "router_name": "old-auto"} + newer = { + **self.ROW, + "api_key": ADMIN.api_key, + "session_id": "s", + "router_name": "new-auto", + "last_turn_at": datetime(2026, 9, 1, 13, 0, 0), + } + self._rig(monkeypatch, [older, newer]) + response = await get_auto_router_session(user_api_key_dict=ADMIN, session_id="s") + assert response.router_name == "new-auto" + + @pytest.mark.asyncio + async def test_a_reconfigured_router_keeps_the_label_the_money_was_priced_against( + self, monkeypatch: pytest.MonkeyPatch + ): + # The proxy's router now prices against a different baseline, but the row's money was priced + # against opus for two of three turns, and the label says so; the full split is on the response. + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + priced = {"anthropic/claude-opus-5": 2, "anthropic/claude-sonnet-5": 1} + self._rig(monkeypatch, [{**self.ROW, "api_key": ADMIN.api_key, "session_id": "s", "baseline_models": priced}]) + response = await get_auto_router_session(user_api_key_dict=ADMIN, session_id="s") + assert response.baseline_model == "anthropic/claude-opus-5" + assert response.baseline_models == priced + + @pytest.mark.asyncio + async def test_a_session_whose_turns_recorded_no_baseline_reports_the_money_without_a_name( + self, monkeypatch: pytest.MonkeyPatch + ): + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + self._rig(monkeypatch, [{**self.ROW, "api_key": ADMIN.api_key, "session_id": "s", "baseline_models": {}}]) + response = await get_auto_router_session(user_api_key_dict=ADMIN, session_id="s") + assert response.baseline_model is None + assert response.baseline_spend == pytest.approx(0.38) + + @pytest.mark.asyncio + async def test_an_oversized_client_session_id_is_bounded_like_the_writer_bounded_it( + self, monkeypatch: pytest.MonkeyPatch + ): + from litellm.proxy.db.autorouter_session_rollup import bounded_session_id + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + long_id = "s" * 300 + self._rig(monkeypatch, [{**self.ROW, "api_key": ADMIN.api_key, "session_id": bounded_session_id(long_id)}]) + response = await get_auto_router_session(user_api_key_dict=ADMIN, session_id=long_id) + assert response.session_id == long_id + assert response.turns == 3 + + @pytest.mark.asyncio + async def test_without_a_database_the_endpoint_says_so(self, monkeypatch: pytest.MonkeyPatch): + from litellm.proxy import proxy_server + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session + + monkeypatch.setattr(proxy_server, "prisma_client", None) + with pytest.raises(HTTPException) as err: + await get_auto_router_session(user_api_key_dict=ADMIN, session_id="s") + assert err.value.status_code == 500 + + NON_ADMIN = UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER, api_key="sk-user", user_id="user") diff --git a/tests/test_litellm/repositories/test_repositories.py b/tests/test_litellm/repositories/test_repositories.py index a890d7ceed0..c7f6b8ff83a 100644 --- a/tests/test_litellm/repositories/test_repositories.py +++ b/tests/test_litellm/repositories/test_repositories.py @@ -2407,3 +2407,60 @@ class TestCountBillableUsers: client.db.litellm_usertable = _RacyTable() repo = UserRepository(client) assert await repo.count_billable_users() == 0 + + +class TestAutoRouterSessionRepository: + ROW: Final = { + "api_key": "hashed-key", + "session_id": "s1", + "router_name": "claude-auto", + "router_type": "complexity", + "first_turn_at": datetime(2026, 9, 1, 12, 0, 0), + "last_turn_at": datetime(2026, 9, 1, 12, 5, 0), + "last_model": "anthropic/claude-sonnet-5", + "models": {"anthropic/claude-sonnet-5": {"at": 1.0, "ttl": None}}, + "turns": 3, + "spend": 0.14, + "saved_spend": 0.24, + "classifier_cost": 0.01, + "tier_turns": {"complex": 3}, + "baseline_models": {"anthropic/claude-opus-5": 3}, + } + + @staticmethod + def _repo(record: Optional[Dict[str, Any]]): + from litellm.repositories.autorouter_session_repository import AutoRouterSessionRepository + + lookups: List[Dict[str, Any]] = [] + + class _Table: + async def find_first(self, where: Dict[str, Any], order: Dict[str, str]): + lookups.append({"where": where, "order": order}) + return MockRecord(record) if record is not None else None + + client = MagicMock() + client.db.litellm_autoroutersession = _Table() + return AutoRouterSessionRepository(client), lookups + + @pytest.mark.asyncio + async def test_find_latest_for_key_reads_the_keys_own_partition_newest_router_first(self): + repo, lookups = self._repo(dict(self.ROW)) + row = await repo.find_latest_for_key("hashed-key", "s1") + assert lookups == [{"where": {"api_key": "hashed-key", "session_id": "s1"}, "order": {"last_turn_at": "desc"}}] + assert row is not None + assert (row.router_name, row.turns, row.spend, row.saved_spend) == ("claude-auto", 3, 0.14, 0.24) + assert row.baseline_models == {"anthropic/claude-opus-5": 3} + assert row.baseline_model == "anthropic/claude-opus-5" + + @pytest.mark.asyncio + async def test_find_latest_for_key_is_none_when_the_key_wrote_no_such_session(self): + repo, _ = self._repo(None) + assert await repo.find_latest_for_key("hashed-key", "unknown") is None + + def test_table_is_the_session_rollup_and_needs_a_database(self): + from litellm.repositories.autorouter_session_repository import AutoRouterSessionRepository + + client = MagicMock() + assert AutoRouterSessionRepository(client).table is client.db.litellm_autoroutersession + with pytest.raises(RuntimeError, match="No DB Connected"): + _ = AutoRouterSessionRepository(None).table diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 0664a9f2fdc..dd455cba44a 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -1242,6 +1242,31 @@ export interface paths { patch?: never; trace?: never; }; + "/auto_router/session": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Get Auto Router Session + * @description One auto-routed session, for the key that ran it: the model its last turn was routed to and the + * session's spend against the router's savings baseline. Built for a coding agent's status line + * or stop hook, so any virtual key may call it and only ever sees rows written under its own + * key hash. Reads the LiteLLM_AutoRouterSession rollup, which the asynchronous spend flush + * fills a moment after each turn; a session with no flushed auto-routed turn yet is a 404. The + * id is bounded the way the writer bounded it, so an oversized client id still finds its row. + */ + get: operations["get_auto_router_session_auto_router_session_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/auto_router/shadow_eval": { parameters: { query?: never; @@ -23572,6 +23597,62 @@ export interface components { /** @description The decision record this request would have written to its log row */ routing_decision: components["schemas"]["StandardLoggingRoutingDecision"]; }; + /** + * AutoRouterSessionResponse + * @description One auto-routed session as its own key sees it: what the last turn ran on, and what the session cost + * against the router's savings baseline (the priciest model in its hardest tier). + */ + AutoRouterSessionResponse: { + /** + * Baseline Model + * @description The savings baseline most of this session's turns were priced against, recorded turn by turn, so it still names the counterfactual after the router is reconfigured or removed. None when no turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, which derive no baseline and so report no savings + */ + baseline_model: string | null; + /** + * Baseline Models + * @description Turns priced against each baseline model; more than one entry means the router's baseline changed mid-session and baseline_spend mixes both + */ + baseline_models: { + [key: string]: number; + }; + /** + * Baseline Spend + * @description spend plus saved_spend: the estimated single-model cost + */ + baseline_spend: number; + /** + * Last Model + * @description The deployment model the most recent turn was routed to + */ + last_model: string; + /** + * Router Name + * @description The auto-router alias the session's requests were sent to + */ + router_name: string; + /** + * Router Type + * @description complexity, adaptive or quality + */ + router_type: string; + /** + * Saved Spend + * @description Estimated savings against the baseline, net of classifier cost + */ + saved_spend: number; + /** Session Id */ + session_id: string; + /** + * Spend + * @description What the session's routed traffic actually cost, classifier calls included + */ + spend: number; + /** + * Turns + * @description Auto-routed turns the rollup has recorded for this session so far + */ + turns: number; + }; /** AwsSessionTag */ AwsSessionTag: { /** Key */ @@ -41687,6 +41768,38 @@ export interface operations { }; }; }; + get_auto_router_session_auto_router_session_get: { + parameters: { + query: { + /** @description The client session id (x-*-session-id header) the turns were sent under */ + session_id: string; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["AutoRouterSessionResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; list_shadow_eval_jobs_auto_router_shadow_eval_get: { parameters: { query?: { From e073cd3aebee3a5e8a4baecfc845961f934d6f3f Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:54:03 -0700 Subject: [PATCH 098/157] fix(mcp): write failure spend log for guardrail-blocked /mcp-rest/tools/call (#40555) * fix(mcp): write failure spend log for guardrail-blocked /mcp-rest/tools/call call_tool_rest_api only translated exceptions to HTTP responses, so a pre_mcp_call guardrail block never reached failure_handler / async_failure_handler / post_call_failure_hook and no LiteLLM_SpendLogs failure row was written. Extract the failure logging from call_mcp_tool into _fire_mcp_tool_call_failure_logging and run it in the REST route for anything raised between common_processing_pre_call_logic and execute_mcp_tool Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(mcp): keep the original REST tool error when failure logging raises Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(mcp): log virtual mcp_tool_call failures and keep REST success latency scoped to tool execution Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../mcp_server/rest_endpoints.py | 184 ++++++++----- .../proxy/_experimental/mcp_server/server.py | 71 ++--- .../mcp_server/test_rest_endpoints.py | 260 +++++++++++++++++- 3 files changed, 407 insertions(+), 108 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py index 8ddfd63bcb6..7a97e995570 100644 --- a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py +++ b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py @@ -191,6 +191,7 @@ if MCP_AVAILABLE: execute_mcp_tool, filter_tools_by_allowed_tools, filter_tools_by_key_team_permissions, + fire_mcp_tool_call_failure_logging, ) ######################################################## @@ -232,6 +233,20 @@ if MCP_AVAILABLE: return result return outcome + async def _safe_fire_mcp_tool_call_failure_logging( + logging_obj: "LiteLLMLoggingObj | None", + exception: Exception, + start_time: datetime, + user_api_key_auth: UserAPIKeyAuth, + request_data: Mapping[str, object], + ) -> None: + try: + await fire_mcp_tool_call_failure_logging( + logging_obj, exception, start_time, user_api_key_auth, request_data + ) + except Exception as logging_error: + verbose_logger.warning("MCP tool call failure logging failed (continuing): %s", logging_error) + def _relay_upstream_auth_http_exception(e: MCPUpstreamAuthError, request: Request) -> HTTPException: """Convert a client-forwarded pass-through upstream 401 into an HTTPException that preserves the upstream WWW-Authenticate, so a standards-compliant MCP client can run the upstream OAuth flow @@ -310,26 +325,39 @@ if MCP_AVAILABLE: ) # MCP_TOOL_CALL_TOOL_NAME: run the same pre-call pipeline as the normal path so the tool # execution is spend-logged and guardrail-checked. - (_, virtual_logging_obj) = await ProxyBaseLLMRequestProcessing(data=data).common_processing_pre_call_logic( - request=request, - user_api_key_dict=user_api_key_dict, - proxy_config=proxy_config, - route_type=CallTypes.call_mcp_tool.value, - proxy_logging_obj=proxy_logging_obj, - general_settings=general_settings, - ) - _tool_start_time: Final = datetime.now() - result: Final = await handle_mcp_tool_call( - tool_name=tool_arguments.get("tool_name", ""), - arguments=tool_arguments.get("arguments") or {}, - user_api_key_dict=user_api_key_dict, - client_ip=rest_client_ip, - mcp_auth_header=virtual_mcp_auth_header, - mcp_server_auth_headers=virtual_mcp_server_auth_headers, - oauth2_headers=virtual_oauth2_headers, - raw_headers=virtual_raw_headers, - litellm_logging_obj=virtual_logging_obj, - ) + virtual_processor: Final = ProxyBaseLLMRequestProcessing(data=data) + _request_start_time: Final = datetime.now() # noqa: DTZ005 # naive to match the tool start time below + try: + (_, virtual_logging_obj) = await virtual_processor.common_processing_pre_call_logic( + request=request, + user_api_key_dict=user_api_key_dict, + proxy_config=proxy_config, + route_type=CallTypes.call_mcp_tool.value, + proxy_logging_obj=proxy_logging_obj, + general_settings=general_settings, + ) + _tool_start_time: Final = datetime.now() + result: Final = await handle_mcp_tool_call( + tool_name=tool_arguments.get("tool_name", ""), + arguments=tool_arguments.get("arguments") or {}, + user_api_key_dict=user_api_key_dict, + client_ip=rest_client_ip, + mcp_auth_header=virtual_mcp_auth_header, + mcp_server_auth_headers=virtual_mcp_server_auth_headers, + oauth2_headers=virtual_oauth2_headers, + raw_headers=virtual_raw_headers, + litellm_logging_obj=virtual_logging_obj, + ) + except Exception as e: + virtual_request_data: Final = virtual_processor.data + await _safe_fire_mcp_tool_call_failure_logging( + virtual_request_data.get("litellm_logging_obj"), + e, + _request_start_time, + user_api_key_dict, + virtual_request_data, + ) + raise return await _safe_fire_mcp_tool_call_logging( virtual_logging_obj, result, @@ -1081,65 +1109,73 @@ if MCP_AVAILABLE: ) proxy_base_llm_response_processor: Final = ProxyBaseLLMRequestProcessing(data=data) - ( - data, - logging_obj, - ) = await proxy_base_llm_response_processor.common_processing_pre_call_logic( - request=request, - user_api_key_dict=user_api_key_dict, - proxy_config=proxy_config, - route_type=CallTypes.call_mcp_tool.value, - proxy_logging_obj=proxy_logging_obj, - general_settings=general_settings, - ) + _request_start_time: Final = datetime.now() # noqa: DTZ005 # naive to match the tool start time below + try: + ( + data, + logging_obj, + ) = await proxy_base_llm_response_processor.common_processing_pre_call_logic( + request=request, + user_api_key_dict=user_api_key_dict, + proxy_config=proxy_config, + route_type=CallTypes.call_mcp_tool.value, + proxy_logging_obj=proxy_logging_obj, + general_settings=general_settings, + ) - # Extract MCP auth headers from request and add to data dict - ( - mcp_auth_header, - mcp_server_auth_headers, - raw_headers_from_request, - ) = _extract_mcp_headers_from_request(request, MCPRequestHandler) - if mcp_auth_header: - data["mcp_auth_header"] = mcp_auth_header - if mcp_server_auth_headers: - data["mcp_server_auth_headers"] = mcp_server_auth_headers - data["raw_headers"] = raw_headers_from_request + # Extract MCP auth headers from request and add to data dict + ( + mcp_auth_header, + mcp_server_auth_headers, + raw_headers_from_request, + ) = _extract_mcp_headers_from_request(request, MCPRequestHandler) + if mcp_auth_header: + data["mcp_auth_header"] = mcp_auth_header + if mcp_server_auth_headers: + data["mcp_server_auth_headers"] = mcp_server_auth_headers + data["raw_headers"] = raw_headers_from_request - # Extract user_api_key_auth from metadata and add to top level - # call_mcp_tool expects user_api_key_auth as a top-level parameter - if "metadata" in data and "user_api_key_auth" in data["metadata"]: - data["user_api_key_auth"] = data["metadata"]["user_api_key_auth"] + # Extract user_api_key_auth from metadata and add to top level + # call_mcp_tool expects user_api_key_auth as a top-level parameter + if "metadata" in data and "user_api_key_auth" in data["metadata"]: + data["user_api_key_auth"] = data["metadata"]["user_api_key_auth"] - # Resolve allowed MCP servers with IP filtering - ( - allowed_mcp_servers, - canonical_server_id, - ) = await _resolve_allowed_mcp_servers_with_ip_filter(request, user_api_key_dict, server_id) + # Resolve allowed MCP servers with IP filtering + ( + allowed_mcp_servers, + canonical_server_id, + ) = await _resolve_allowed_mcp_servers_with_ip_filter(request, user_api_key_dict, server_id) - # Look up per-user OAuth headers for this server (mirrors list_tool_rest_api). - user_oauth_extra_headers: dict[str, str] | None = None - target_server: Final = next( - (s for s in allowed_mcp_servers if s.server_id == canonical_server_id), - None, - ) - if target_server is not None: - user_oauth_extra_headers = await _get_user_oauth_extra_headers(target_server, user_api_key_dict) + # Look up per-user OAuth headers for this server (mirrors list_tool_rest_api). + user_oauth_extra_headers: dict[str, str] | None = None + target_server: Final = next( + (s for s in allowed_mcp_servers if s.server_id == canonical_server_id), + None, + ) + if target_server is not None: + user_oauth_extra_headers = await _get_user_oauth_extra_headers(target_server, user_api_key_dict) - # Call execute_mcp_tool directly (permission checks already done) - _tool_start_time: Final = datetime.now() - result: Final = await execute_mcp_tool( - name=tool_name, - arguments=tool_arguments, - allowed_mcp_servers=allowed_mcp_servers, - start_time=_tool_start_time, - user_api_key_auth=data.get("user_api_key_auth"), - mcp_auth_header=data.get("mcp_auth_header"), - mcp_server_auth_headers=data.get("mcp_server_auth_headers"), - oauth2_headers=user_oauth_extra_headers or data.get("oauth2_headers"), - raw_headers=data.get("raw_headers"), - litellm_logging_obj=data.get("litellm_logging_obj"), - requested_server_id=canonical_server_id, - ) + # Call execute_mcp_tool directly (permission checks already done) + _tool_start_time: Final = datetime.now() + result: Final = await execute_mcp_tool( + name=tool_name, + arguments=tool_arguments, + allowed_mcp_servers=allowed_mcp_servers, + start_time=_tool_start_time, + user_api_key_auth=data.get("user_api_key_auth"), + mcp_auth_header=data.get("mcp_auth_header"), + mcp_server_auth_headers=data.get("mcp_server_auth_headers"), + oauth2_headers=user_oauth_extra_headers or data.get("oauth2_headers"), + raw_headers=data.get("raw_headers"), + litellm_logging_obj=data.get("litellm_logging_obj"), + requested_server_id=canonical_server_id, + ) + except Exception as e: + request_data: Final = proxy_base_llm_response_processor.data + await _safe_fire_mcp_tool_call_failure_logging( + request_data.get("litellm_logging_obj"), e, _request_start_time, user_api_key_dict, request_data + ) + raise return await _safe_fire_mcp_tool_call_logging( logging_obj, result, diff --git a/litellm/proxy/_experimental/mcp_server/server.py b/litellm/proxy/_experimental/mcp_server/server.py index 95129d9aeed..43a7f894db8 100644 --- a/litellm/proxy/_experimental/mcp_server/server.py +++ b/litellm/proxy/_experimental/mcp_server/server.py @@ -3339,6 +3339,43 @@ if MCP_AVAILABLE: ) return result + async def fire_mcp_tool_call_failure_logging( + logging_obj: LiteLLMLoggingObj | None, + exception: Exception, + start_time: datetime, + user_api_key_auth: UserAPIKeyAuth | None, + request_data: Mapping[str, object], + ) -> None: + """Failure logging shared by the ``/mcp`` path and the REST endpoint. Call from + inside the ``except`` block so the traceback is still available. + + The failure handlers run first because ``_ProxyDBLogger.async_post_call_failure_hook`` + builds the failure spend-log row from the ``standard_logging_object`` they produce; + both gate on ``should_run_logging``, so the ``@client`` wrapper does not log twice. + A relayed upstream 401 (``MCPUpstreamAuthError``) is an expected caller-must-reauth + signal and skips ``post_call_failure_hook``, which fires the ``llm_exceptions`` alert. + """ + from litellm.proxy.proxy_server import proxy_logging_obj + + traceback_str: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG) + if logging_obj is not None: + end_time: Final = datetime.now() # noqa: DTZ005 # naive to match `start_time`, which it is subtracted from + logging_obj.failure_handler(exception, traceback_str, start_time, end_time) + await logging_obj.async_failure_handler(exception, traceback_str, start_time, end_time) + + if isinstance(exception, MCPUpstreamAuthError) or not proxy_logging_obj or user_api_key_auth is None: + return + sanitized_request_data: Final = { + key: value for key, value in request_data.items() if key not in _MCP_CREDENTIAL_REQUEST_FIELDS + } + await proxy_logging_obj.post_call_failure_hook( + request_data=sanitized_request_data, + original_exception=exception, + user_api_key_dict=user_api_key_auth, + route="/mcp/call_tool", + traceback_str=traceback_str, + ) + @client async def call_mcp_tool( name: str, @@ -3405,40 +3442,8 @@ if MCP_AVAILABLE: raw_headers=raw_headers, **kwargs, ) - except MCPUpstreamAuthError: - # A client-forwarded pass-through upstream 401 is an expected caller-must-reauth signal, so - # re-raise it without post_call_failure_hook, which fires the proxy's llm_exceptions alert. - # mcp_server_tool_call then downgrades it to an informational isError result for the - # streamable client. Note: this function is @client-decorated, so the decorator's standard - # failure logging still records the event (spend log / OTel); only the extra alert sink is - # skipped here. - raise except Exception as e: - traceback_str: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG) - from litellm.proxy.proxy_server import proxy_logging_obj - - # Ordering is load-bearing. ``_ProxyDBLogger.async_post_call_failure_hook``, - # reached below, writes the failure spend-log row from this logger's - # ``standard_logging_object``, which only exists once the failure handlers - # have run. Flush them first or the row lands with - # ``guardrail_information=None`` and a guardrail block is never counted. - # - # Not double-logged: both handlers gate on ``should_run_logging`` and then - # mark it, so the ``@client`` wrapper's own post-raise logging no-ops on this - # logger, same as ``_fire_mcp_tool_call_logging`` does for ``isError=True``. - if litellm_logging_obj is not None: - end_time: Final = datetime.now() # noqa: DTZ005 # naive to match `start_time`, which it is subtracted from - litellm_logging_obj.failure_handler(e, traceback_str, start_time, end_time) - await litellm_logging_obj.async_failure_handler(e, traceback_str, start_time, end_time) - - if proxy_logging_obj and user_api_key_auth: - await proxy_logging_obj.post_call_failure_hook( - request_data=kwargs, - original_exception=e, - user_api_key_dict=user_api_key_auth, - route="/mcp/call_tool", - traceback_str=traceback_str, - ) + await fire_mcp_tool_call_failure_logging(litellm_logging_obj, e, start_time, user_api_key_auth, kwargs) raise if litellm_logging_obj: diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py index 16c6aa128d0..31ccd5c9817 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py @@ -2595,6 +2595,79 @@ class TestCallToolRestAPI: assert result == masked_result + async def test_success_logging_start_time_excludes_pre_call_processing(self, monkeypatch): + """Pre-call hook latency (guardrails, header resolution) must not inflate the tool call's + logged duration on success.""" + from litellm.proxy import proxy_server + + async def fake_contexts(user_api_key_auth): + return [user_api_key_auth] + + async def fake_get_allowed_mcp_servers(*args, **kwargs): + return ["server-1"] + + class StubServer: + server_id = "server-1" + alias = "server-1" + server_name = "server-1" + name = "stub" + allowed_tools = None + mcp_info = {"server_name": "stub"} + available_on_public_internet = True + auth_type = None + + stub_server = StubServer() + + async def fake_add_litellm_data_to_request(**kwargs): + return kwargs.get("data", {}) + + pre_call_finished_at = {} + + async def slow_pre_call_hook(user_api_key_dict, data, call_type): + await asyncio.sleep(0.05) + pre_call_finished_at["value"] = datetime.now() + return data + + captured = {} + + async def fake_execute_mcp_tool(**kwargs): + captured.update(kwargs) + return {"result": "ok"} + + fire_logging = AsyncMock(return_value={"result": "ok"}) + monkeypatch.setattr(rest_endpoints, "build_effective_auth_contexts", fake_contexts, raising=False) + monkeypatch.setattr( + rest_endpoints.global_mcp_server_manager, + "get_allowed_mcp_servers", + fake_get_allowed_mcp_servers, + raising=False, + ) + monkeypatch.setattr( + rest_endpoints.global_mcp_server_manager, + "get_mcp_server_by_id", + lambda server_id: stub_server if server_id == "server-1" else None, + raising=False, + ) + monkeypatch.setattr( + proxy_server, "add_litellm_data_to_request", fake_add_litellm_data_to_request, raising=False + ) + monkeypatch.setattr(proxy_server, "proxy_config", {}, raising=False) + monkeypatch.setattr(proxy_server.proxy_logging_obj, "pre_call_hook", slow_pre_call_hook) + monkeypatch.setattr(rest_endpoints, "execute_mcp_tool", fake_execute_mcp_tool, raising=False) + monkeypatch.setattr(rest_endpoints, "_fire_mcp_tool_call_logging", fire_logging, raising=False) + + request = _build_request( + path="/mcp-rest/tools/call", + method="POST", + json_body={"server_id": "server-1", "name": "demo-tool", "arguments": {}}, + ) + + await rest_endpoints.call_tool_rest_api(request, user_api_key_dict=UserAPIKeyAuth()) + + logged_start_time = fire_logging.await_args.args[2] + assert captured["start_time"] >= pre_call_finished_at["value"] + assert logged_start_time == captured["start_time"] + async def test_success_logging_guardrail_rejection_propagates(self, monkeypatch): """A guardrail rejecting the tool result must not be swallowed as a logging failure, otherwise the unguarded result would still be returned to the caller.""" @@ -2765,6 +2838,139 @@ class TestCallToolRestAPI: info_messages = [_rendered_log_message(c) for c in mock_logger.info.call_args_list if c.args] assert not any("relaying upstream" in m for m in info_messages) + @pytest.mark.parametrize("raise_site", ["pre_call_hook", "execute_mcp_tool"]) + async def test_guardrail_block_runs_failure_logging_before_http_translation(self, monkeypatch, raise_site): + """A pre_mcp_call guardrail block, whether raised by the pre-call hook or from inside + execute_mcp_tool, must reach proxy_logging_obj.post_call_failure_hook (the only path that + writes the failure spend-log row) with the logging object's failure payload already built, + and the REST caller must still get the same 400 it got before.""" + from litellm.proxy import proxy_server + + async def fake_contexts(user_api_key_auth): + return [user_api_key_auth] + + async def fake_get_allowed_mcp_servers(*args, **kwargs): + return ["server-1"] + + class StubServer: + server_id = "server-1" + alias = "server-1" + server_name = "server-1" + name = "stub" + allowed_tools = None + mcp_info = {"server_name": "stub"} + available_on_public_internet = True + auth_type = None + + async def fake_add_litellm_data_to_request(**kwargs): + return kwargs.get("data", {}) + + guardrail_error = HTTPException( + status_code=400, + detail={"error": "Content blocked: keyword 'confidential' detected", "keyword": "confidential"}, + ) + + async def passthrough_pre_call_hook(user_api_key_dict, data, call_type): + return data + + async def blocking_pre_call_hook(user_api_key_dict, data, call_type): + raise guardrail_error + + async def fake_execute_mcp_tool(**kwargs): + raise guardrail_error + + async def passthrough_execute_mcp_tool(**kwargs): + return [] + + monkeypatch.setattr(rest_endpoints, "build_effective_auth_contexts", fake_contexts, raising=False) + monkeypatch.setattr( + rest_endpoints.global_mcp_server_manager, + "get_allowed_mcp_servers", + fake_get_allowed_mcp_servers, + raising=False, + ) + monkeypatch.setattr( + rest_endpoints.global_mcp_server_manager, + "get_mcp_server_by_id", + lambda server_id: StubServer() if server_id == "server-1" else None, + raising=False, + ) + monkeypatch.setattr( + proxy_server, "add_litellm_data_to_request", fake_add_litellm_data_to_request, raising=False + ) + monkeypatch.setattr(proxy_server, "proxy_config", {}, raising=False) + monkeypatch.setattr( + proxy_server.proxy_logging_obj, + "pre_call_hook", + blocking_pre_call_hook if raise_site == "pre_call_hook" else passthrough_pre_call_hook, + ) + monkeypatch.setattr( + rest_endpoints, + "execute_mcp_tool", + fake_execute_mcp_tool if raise_site == "execute_mcp_tool" else passthrough_execute_mcp_tool, + raising=False, + ) + post_call_failure_hook = AsyncMock(return_value=None) + monkeypatch.setattr(proxy_server.proxy_logging_obj, "post_call_failure_hook", post_call_failure_hook) + + user_api_key_dict = UserAPIKeyAuth(api_key="hashed-key", request_route="/mcp-rest/tools/call") + request = _build_request( + headers={"x-mcp-deepwiki-authorization": "Bearer upstream-secret"}, + path="/mcp-rest/tools/call", + method="POST", + json_body={"server_id": "server-1", "name": "demo-tool", "arguments": {"q": "confidential"}}, + ) + + with pytest.raises(HTTPException) as exc_info: + await rest_endpoints.call_tool_rest_api(request, user_api_key_dict=user_api_key_dict) + + assert exc_info.value is guardrail_error + + post_call_failure_hook.assert_awaited_once() + hook_kwargs = post_call_failure_hook.await_args.kwargs + assert hook_kwargs["original_exception"] is guardrail_error + assert hook_kwargs["user_api_key_dict"] is user_api_key_dict + assert hook_kwargs["route"] == "/mcp/call_tool" + request_data = hook_kwargs["request_data"] + assert "raw_headers" not in request_data + assert "mcp_server_auth_headers" not in request_data + standard_logging_object = request_data["litellm_logging_obj"].model_call_details["standard_logging_object"] + assert standard_logging_object["status"] == "failure" + assert standard_logging_object["error_str"] == str(guardrail_error) + + async def test_failure_logging_error_does_not_replace_guardrail_error(self, monkeypatch): + from litellm.proxy import proxy_server + + guardrail_error = HTTPException(status_code=400, detail={"error": "Content blocked"}) + + async def fake_add_litellm_data_to_request(**kwargs): + return kwargs.get("data", {}) + + async def blocking_pre_call_hook(user_api_key_dict, data, call_type): + raise guardrail_error + + failure_logging = AsyncMock(side_effect=RuntimeError("spend log db down")) + monkeypatch.setattr( + proxy_server, "add_litellm_data_to_request", fake_add_litellm_data_to_request, raising=False + ) + monkeypatch.setattr(proxy_server, "proxy_config", {}, raising=False) + monkeypatch.setattr(proxy_server.proxy_logging_obj, "pre_call_hook", blocking_pre_call_hook) + monkeypatch.setattr(rest_endpoints, "fire_mcp_tool_call_failure_logging", failure_logging, raising=False) + + request = _build_request( + path="/mcp-rest/tools/call", + method="POST", + json_body={"server_id": "server-1", "name": "demo-tool", "arguments": {"q": "confidential"}}, + ) + + with pytest.raises(HTTPException) as exc_info: + await rest_endpoints.call_tool_rest_api( + request, user_api_key_dict=UserAPIKeyAuth(api_key="hashed-key", request_route="/mcp-rest/tools/call") + ) + + assert exc_info.value is guardrail_error + failure_logging.assert_awaited_once() + async def test_success_logging_cancellation_propagates(self, monkeypatch): fire_logging = AsyncMock(side_effect=asyncio.CancelledError()) monkeypatch.setattr( @@ -2801,7 +3007,7 @@ class TestCallToolRestAPI: class _FakePreCall: def __init__(self, data): - pass + self.data = data async def common_processing_pre_call_logic(self, **kwargs): return None, MagicMock() @@ -2835,6 +3041,58 @@ class TestCallToolRestAPI: assert exc_info.value.headers is not None assert exc_info.value.headers.get("www-authenticate") == challenge + async def test_virtual_mcp_tool_call_guardrail_block_runs_failure_logging(self, monkeypatch): + """A pre_mcp_call guardrail block on the virtual mcp_tool_call branch must write a failure + spend log, same as the direct tool call branch, and still raise the original error.""" + from litellm.proxy import proxy_server + from litellm.proxy._types import LiteLLM_ObjectPermissionTable + + guardrail_error = HTTPException(status_code=400, detail={"error": "Content blocked"}) + + async def fake_contexts(user_api_key_auth): + return [user_api_key_auth] + + async def fake_add_litellm_data_to_request(**kwargs): + return kwargs.get("data", {}) + + async def blocking_pre_call_hook(user_api_key_dict, data, call_type): + raise guardrail_error + + failure_logging = AsyncMock() + monkeypatch.setattr(rest_endpoints, "build_effective_auth_contexts", fake_contexts, raising=False) + monkeypatch.setattr( + proxy_server, "add_litellm_data_to_request", fake_add_litellm_data_to_request, raising=False + ) + monkeypatch.setattr(proxy_server, "proxy_config", {}, raising=False) + monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False) + monkeypatch.setattr(proxy_server.proxy_logging_obj, "pre_call_hook", blocking_pre_call_hook) + monkeypatch.setattr(rest_endpoints, "fire_mcp_tool_call_failure_logging", failure_logging, raising=False) + + user_api_key_dict = UserAPIKeyAuth( + api_key="hashed-key", + request_route="/mcp-rest/tools/call", + object_permission=LiteLLM_ObjectPermissionTable( + object_permission_id="search-scope", + mcp_tool_search_enabled=True, + ), + ) + request = _build_request( + path="/mcp-rest/tools/call", + method="POST", + json_body={"name": "mcp_tool_call", "arguments": {"tool_name": "x", "arguments": {"q": "confidential"}}}, + ) + + with pytest.raises(HTTPException) as exc_info: + await rest_endpoints.call_tool_rest_api(request, user_api_key_dict=user_api_key_dict) + + assert exc_info.value is guardrail_error + failure_logging.assert_awaited_once() + logging_obj, exception, _start_time, user_api_key_auth, request_data = failure_logging.await_args.args + assert exception is guardrail_error + assert user_api_key_auth is user_api_key_dict + assert logging_obj is request_data.get("litellm_logging_obj") + assert logging_obj is not None + class TestGetToolsForSingleServer: """Test _get_tools_for_single_server with object_permission filtering""" From 02279bc9921e61bccb7b3f4b36b4637abb97995c Mon Sep 17 00:00:00 2001 From: mateo Date: Fri, 11 Sep 2026 19:59:23 +0000 Subject: [PATCH 099/157] test(cost-map): gpt-5.5-pro has no published cached input rate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../llm_cost_calc/test_llm_cost_calc_utils.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index fbb9d178390..cbe6fe198c9 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1715,7 +1715,7 @@ def test_generic_cost_per_token_gpt55(_local_model_cost_map): def test_generic_cost_per_token_gpt55_pro(_local_model_cost_map): - """gpt-5.5-pro: responses-only model — $30/1M input, $180/1M output, $3/1M cached input.""" + """gpt-5.5-pro: responses-only model, $30/1M input, $180/1M output, no cached input rate published.""" model = "gpt-5.5-pro" custom_llm_provider = "openai" @@ -1724,7 +1724,7 @@ def test_generic_cost_per_token_gpt55_pro(_local_model_cost_map): # Sanity-check the map values match OpenAI's published pricing. assert model_cost_map["input_cost_per_token"] == 3e-5 assert model_cost_map["output_cost_per_token"] == 1.8e-4 - assert model_cost_map["cache_read_input_token_cost"] == 3e-6 + assert "cache_read_input_token_cost" not in model_cost_map assert model_cost_map["litellm_provider"] == "openai" # gpt-5.5-pro is a responses-only model (no /v1/chat/completions endpoint). assert model_cost_map["mode"] == "responses" From ae6a4a2f2aeba45a70b2201c74c9c0ff8367322e Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 13:03:06 -0700 Subject: [PATCH 100/157] feat(ocr): add Azure Mistral adapter and document fetching (#40533) * feat(ocr): add Azure Mistral adapter and document fetching * fix(ocr): decline missing Azure credentials * fix(ocr): map Azure credentials in gateway errors * refactor(ocr): preserve Azure Mistral extra params * refactor(ocr): adopt request preparation contract --- litellm-rust/Cargo.lock | 7 + .../src/audio_transcription/hooks.rs | 5 +- .../crates/ai-gateway/src/ocr/hooks.rs | 5 +- litellm-rust/crates/ai-gateway/src/ocr/mod.rs | 12 + .../ai-gateway/src/routes/messages/mod.rs | 4 +- .../crates/ai-gateway/tests/ocr_lifecycle.rs | 51 ++ litellm-rust/crates/core/Cargo.toml | 2 +- litellm-rust/crates/core/src/constants.rs | 7 + litellm-rust/crates/core/src/error.rs | 28 + litellm-rust/crates/core/src/lib.rs | 1 + litellm-rust/crates/core/src/media.rs | 528 ++++++++++++++++++ .../core/src/ocr/adapters/azure_mistral.rs | 146 +++++ .../crates/core/src/ocr/adapters/mod.rs | 3 + litellm-rust/crates/core/src/ocr/client.rs | 19 +- litellm-rust/crates/core/src/ocr/document.rs | 214 +++++++ litellm-rust/crates/core/src/ocr/error.rs | 16 + litellm-rust/crates/core/src/ocr/mod.rs | 4 + litellm-rust/crates/core/src/ocr/registry.rs | 12 + litellm-rust/crates/core/src/ocr/types.rs | 24 + litellm-rust/crates/core/src/ocr/wire.rs | 1 + .../providers/azure_ai/ocr/transformation.rs | 7 +- .../crates/core/tests/azure_ai_ocr.rs | 75 +++ litellm-rust/crates/core/tests/ocr/support.rs | 6 +- .../crates/python-bridge/src/errors.rs | 2 + .../crates/python-bridge/src/routes/ocr.rs | 16 + .../rust_bridge/native_route_wheel_test.py | 26 +- 26 files changed, 1205 insertions(+), 16 deletions(-) create mode 100644 litellm-rust/crates/core/src/media.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs create mode 100644 litellm-rust/crates/core/src/ocr/document.rs create mode 100644 litellm-rust/crates/core/tests/azure_ai_ocr.rs diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock index 0e8e6e09a21..a7c514270da 100644 --- a/litellm-rust/Cargo.lock +++ b/litellm-rust/Cargo.lock @@ -878,6 +878,12 @@ version = "2.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" +[[package]] +name = "data-url" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "be1e0bca6c3637f992fc1cc7cbc52a78c1ef6db076dbf1059c4323d6a2048376" + [[package]] name = "deranged" version = "0.5.8" @@ -1608,6 +1614,7 @@ dependencies = [ "aws-smithy-runtime-api", "aws-types", "base64 0.22.1", + "data-url", "rand 0.8.7", "reqwest", "rstest", diff --git a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs index 098ce071efc..6f4a573e6ff 100644 --- a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs @@ -269,7 +269,10 @@ fn guardrail_error_to_core_error(error: GuardrailError) -> Error { fn core_error_kind(error: &Error) -> &'static str { match error { - Error::Auth(_) | Error::MissingApiKey { .. } => "AuthError", + Error::Auth(_) + | Error::MissingApiKey { .. } + | Error::MissingAzureAiCredentials + | Error::MissingAzureAiCredentialsOrAdToken => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", Error::InvalidRequest(_) => "InvalidRequest", Error::InvalidType { .. } => "InvalidType", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs index d1dd811f7ab..c7e8344aafc 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs @@ -386,7 +386,10 @@ fn guardrail_error_to_core_error(error: GuardrailError) -> Error { fn core_error_kind(error: &Error) -> &'static str { match error { - Error::Auth(_) | Error::MissingApiKey { .. } => "AuthError", + Error::Auth(_) + | Error::MissingApiKey { .. } + | Error::MissingAzureAiCredentials + | Error::MissingAzureAiCredentialsOrAdToken => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", Error::InvalidRequest(_) => "InvalidRequest", Error::InvalidType { .. } => "InvalidType", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs index 2acdd232c80..9ed93d779f1 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs @@ -31,6 +31,18 @@ mod tests { use super::{OcrRequest, ocr}; use crate::integrations::types::RequestMetadata; + use litellm_core::ocr::wire::is_supported_request; + + #[test] + fn core_activation_excludes_unmigrated_azure_document_intelligence() { + assert!(is_supported_request("model", Some("mistral"))); + assert!(is_supported_request("pixtral-12b", Some("azure_ai"))); + assert!(!is_supported_request( + "doc-intelligence/prebuilt-layout", + Some("azure_ai") + )); + assert!(!is_supported_request("parse-v3", Some("reducto"))); + } async fn read_http_request(socket: &mut TcpStream) -> String { let mut request = Vec::new(); diff --git a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs index c22d05f5726..adfbf2b5910 100644 --- a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs @@ -115,7 +115,9 @@ impl IntoResponse for MessagesRouteError { | Error::InvalidResponse(_) | Error::InvalidType { .. } | Error::MissingField(_) - | Error::MissingApiKey { .. } => ( + | Error::MissingApiKey { .. } + | Error::MissingAzureAiCredentials + | Error::MissingAzureAiCredentialsOrAdToken => ( StatusCode::BAD_GATEWAY, "messages provider request failed".to_string(), ), diff --git a/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs b/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs index 60e90ed2a7c..5502a467511 100644 --- a/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs +++ b/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs @@ -236,6 +236,57 @@ fn base_ocr_request(model: &str) -> OcrRequest<'_> { } } +#[tokio::test] +async fn azure_mistral_uses_prepared_authorization_through_gateway() { + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let api_base = format!("http://{}", listener.local_addr().unwrap()); + let server = tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.unwrap(); + let request = read_http_request(&mut socket).await; + let body = br#"{"pages":[]}"#; + socket + .write_all( + format!( + "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n", + body.len() + ) + .as_bytes(), + ) + .await + .unwrap(); + socket.write_all(body).await.unwrap(); + request + }); + let request = OcrRequest { + model: "mistral-ocr-2505", + document: json!({ + "type":"document_url", + "document_url":"data:application/pdf;base64,YWJj" + }), + api_key: None, + api_base: Some(&api_base), + custom_llm_provider: Some("azure_ai"), + extra_headers: Some(Map::from_iter([( + "Authorization".into(), + json!("Bearer python-prepared-token"), + )])), + optional_params: Map::new(), + timeout: None, + callbacks: Vec::new(), + guardrails: Vec::new(), + request_metadata: RequestMetadata::default(), + litellm_call_id: None, + }; + + ocr(request).await.unwrap(); + let sent = server.await.unwrap(); + assert!(sent.starts_with("POST /providers/mistral/azure/ocr ")); + assert!( + sent.to_ascii_lowercase() + .contains("authorization: bearer python-prepared-token\r\n") + ); +} + #[tokio::test] async fn reducto_during_call_guardrail_blocks_before_upload() { let listener = TcpListener::bind("127.0.0.1:0") diff --git a/litellm-rust/crates/core/Cargo.toml b/litellm-rust/crates/core/Cargo.toml index b4ca88cf16f..6d0a2fa775b 100644 --- a/litellm-rust/crates/core/Cargo.toml +++ b/litellm-rust/crates/core/Cargo.toml @@ -12,6 +12,7 @@ path = "tests/workspace_crate_allowlist.rs" [dependencies] base64.workspace = true +data-url = "0.3.2" rand.workspace = true reqwest.workspace = true serde.workspace = true @@ -44,5 +45,4 @@ observability = ["dep:tracing-subscriber"] [dev-dependencies] rstest.workspace = true -tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } tracing-subscriber.workspace = true diff --git a/litellm-rust/crates/core/src/constants.rs b/litellm-rust/crates/core/src/constants.rs index 108d2a48e30..8a12e186197 100644 --- a/litellm-rust/crates/core/src/constants.rs +++ b/litellm-rust/crates/core/src/constants.rs @@ -43,6 +43,13 @@ pub const EMPTY_TEXT_PLACEHOLDER: &str = "[System: Empty message content sanitised to satisfy protocol]"; pub const FUNCTION_TRACE_TARGET: &str = "litellm::function_trace"; + +pub(crate) const MEDIA_CONNECT_TIMEOUT_SECS: u64 = 10; + pub(crate) const OCR_HTTP_TIMEOUT_SECS: u64 = 600; pub(crate) const OCR_CONNECT_TIMEOUT_SECS: u64 = 10; +pub(crate) const OCR_INLINE_MAX_BYTES: usize = 50 * 1024 * 1024; +pub(crate) const OCR_DOWNLOAD_MAX_BYTES: u64 = 50 * 1024 * 1024; +pub(crate) const OCR_MAX_FETCH_REDIRECTS: usize = 10; +pub(crate) const AZURE_AI_OCR_PATH: &str = "/providers/mistral/azure/ocr"; pub(crate) const MISTRAL_OCR_API_BASE: &str = "https://api.mistral.ai/v1"; diff --git a/litellm-rust/crates/core/src/error.rs b/litellm-rust/crates/core/src/error.rs index 0382314057f..eefa7d606d8 100644 --- a/litellm-rust/crates/core/src/error.rs +++ b/litellm-rust/crates/core/src/error.rs @@ -21,6 +21,12 @@ pub enum Error { "Missing {provider} API Key - A call is being made to {provider} but no key is set either in the environment variables or via params" )] MissingApiKey { provider: &'static str }, + #[error( + "Missing Azure AI credentials - set AZURE_AI_API_KEY or provide an Authorization header" + )] + MissingAzureAiCredentials, + #[error("Missing Azure AI credentials - set AZURE_AI_API_KEY or provide azure_ad_token")] + MissingAzureAiCredentialsOrAdToken, #[error("upstream request failed with status {status}: {body}")] Http { status: u16, body: String }, #[error("upstream network error: {0}")] @@ -40,6 +46,28 @@ pub enum Error { Unsupported(&'static str), } +#[derive(Debug, ThisError)] +pub(crate) enum MediaError { + #[error("media URL rejected by network policy")] + BlockedUrl, + #[error("media download is disabled")] + DownloadDisabled, + #[error("media download exceeds the maximum size")] + DownloadTooLarge, + #[error("too many redirects while fetching media")] + TooManyRedirects, + #[error("media redirect is missing a Location header")] + MissingRedirectLocation, + #[error("invalid media redirect")] + InvalidRedirect, + #[error("media download failed with status {0}")] + Http(u16), + #[error("media download timed out")] + Timeout, + #[error("{0}")] + Transport(#[from] TransportError), +} + #[derive(Clone, Debug, ThisError, PartialEq, Eq)] pub enum TransportError { #[error("upstream request failed with status {status}: {body}")] diff --git a/litellm-rust/crates/core/src/lib.rs b/litellm-rust/crates/core/src/lib.rs index 3a0896a4d5a..7e81b292441 100644 --- a/litellm-rust/crates/core/src/lib.rs +++ b/litellm-rust/crates/core/src/lib.rs @@ -5,6 +5,7 @@ pub mod chat_completions; pub mod constants; pub mod error; pub mod http_utils; +mod media; pub mod messages; #[cfg(any(feature = "observability", test))] pub mod observability; diff --git a/litellm-rust/crates/core/src/media.rs b/litellm-rust/crates/core/src/media.rs new file mode 100644 index 00000000000..5f9a43794c2 --- /dev/null +++ b/litellm-rust/crates/core/src/media.rs @@ -0,0 +1,528 @@ +use std::future::Future; +use std::io; +use std::net::{IpAddr, SocketAddr}; +use std::pin::Pin; +use std::sync::Arc; +use std::time::Duration; + +use reqwest::Url; +use reqwest::dns::{Addrs, Name, Resolve, Resolving}; + +use crate::constants::MEDIA_CONNECT_TIMEOUT_SECS; +use crate::error::{MediaError, TransportError}; + +#[derive(Clone)] +pub(crate) struct MediaFetcher { + client: reqwest::Client, + address_resolver: Arc, + allow_private_network: bool, +} + +type AddressResolution<'a> = Pin>> + Send + 'a>>; + +trait AddressResolver: Send + Sync { + fn resolve<'a>(&'a self, host: &'a str, port: u16) -> AddressResolution<'a>; +} + +#[derive(Clone, Copy)] +pub(crate) struct DownloadPolicy { + pub(crate) timeout: Duration, + pub(crate) max_bytes: u64, + pub(crate) max_redirects: usize, +} + +#[derive(Debug)] +pub(crate) struct DownloadedMedia { + pub(crate) bytes: Vec, + pub(crate) content_type: String, +} + +impl MediaFetcher { + pub(crate) fn new() -> Result { + Self::with_resolvers(Arc::new(PublicDnsResolver), Arc::new(SystemAddressResolver)) + } + + fn with_resolvers( + transport_resolver: Arc, + address_resolver: Arc, + ) -> Result + where + R: Resolve + 'static, + { + let client = reqwest::Client::builder() + .connect_timeout(Duration::from_secs(MEDIA_CONNECT_TIMEOUT_SECS)) + .redirect(reqwest::redirect::Policy::none()) + .no_proxy() + .dns_resolver(transport_resolver) + .build()?; + Ok(Self { + client, + address_resolver, + allow_private_network: false, + }) + } + + #[cfg(test)] + pub(crate) fn for_test(client: reqwest::Client) -> Self { + Self { + client, + address_resolver: Arc::new(AllowPrivateResolver), + allow_private_network: true, + } + } + + pub(crate) async fn fetch( + &self, + url: Url, + policy: DownloadPolicy, + ) -> Result { + if policy.max_bytes == 0 { + return Err(MediaError::DownloadDisabled); + } + tokio::time::timeout(policy.timeout, self.fetch_before_deadline(url, policy)) + .await + .map_err(|_| MediaError::Timeout)? + } + + async fn fetch_before_deadline( + &self, + mut url: Url, + policy: DownloadPolicy, + ) -> Result { + let mut redirects_followed = 0; + loop { + self.validate_url(&url).await?; + let mut response = self + .client + .get(url.clone()) + .send() + .await + .map_err(TransportError::from)?; + if response.status().is_redirection() { + if redirects_followed == policy.max_redirects { + return Err(MediaError::TooManyRedirects); + } + let location = response + .headers() + .get(reqwest::header::LOCATION) + .and_then(|value| value.to_str().ok()) + .ok_or(MediaError::MissingRedirectLocation)?; + url = url + .join(location) + .map_err(|_| MediaError::InvalidRedirect)?; + redirects_followed += 1; + continue; + } + if !response.status().is_success() { + return Err(MediaError::Http(response.status().as_u16())); + } + enforce_download_size(response.content_length().unwrap_or(0), policy.max_bytes)?; + let content_type = response + .headers() + .get(reqwest::header::CONTENT_TYPE) + .and_then(|value| value.to_str().ok()) + .and_then(|value| value.split(';').next()) + .map(str::trim) + .filter(|value| !value.is_empty()) + .unwrap_or("application/octet-stream") + .to_string(); + let mut bytes = Vec::new(); + while let Some(chunk) = response.chunk().await.map_err(TransportError::from)? { + enforce_download_size(bytes.len() as u64 + chunk.len() as u64, policy.max_bytes)?; + bytes.extend_from_slice(&chunk); + } + return Ok(DownloadedMedia { + bytes, + content_type, + }); + } + } + + async fn validate_url(&self, url: &Url) -> Result<(), MediaError> { + if !matches!(url.scheme(), "http" | "https") + || !url.username().is_empty() + || url.password().is_some() + { + return Err(MediaError::BlockedUrl); + } + let host = url.host_str().ok_or(MediaError::BlockedUrl)?; + if self.allow_private_network { + return Ok(()); + } + if let Ok(ip) = host.parse::() { + return (!is_blocked_ip(ip)) + .then_some(()) + .ok_or(MediaError::BlockedUrl); + } + let port = url.port_or_known_default().ok_or(MediaError::BlockedUrl)?; + let addresses = self + .address_resolver + .resolve(host, port) + .await + .map_err(|error| TransportError::Network(error.to_string()))?; + validate_addresses(&addresses) + } +} + +fn enforce_download_size(length: u64, max_bytes: u64) -> Result<(), MediaError> { + if length > max_bytes { + return Err(MediaError::DownloadTooLarge); + } + Ok(()) +} + +fn validate_addresses(addresses: &[SocketAddr]) -> Result<(), MediaError> { + if addresses.is_empty() || addresses.iter().any(|address| is_blocked_ip(address.ip())) { + return Err(MediaError::BlockedUrl); + } + Ok(()) +} + +fn is_blocked_ip(ip: IpAddr) -> bool { + match ip { + IpAddr::V4(ip) => { + let [first, second, third, _] = ip.octets(); + first == 0 + || first == 10 + || first == 127 + || (first == 100 && (64..=127).contains(&second)) + || (first == 169 && second == 254) + || (first == 172 && (16..=31).contains(&second)) + || (first == 192 && second == 0 && (third == 0 || third == 2)) + || (first == 192 && second == 168) + || (first == 192 && second == 88 && third == 99) + || (first == 198 && (second == 18 || second == 19)) + || (first == 198 && second == 51 && third == 100) + || (first == 203 && second == 0 && third == 113) + || first >= 224 + } + IpAddr::V6(ip) => { + let segments = ip.segments(); + ip.is_loopback() + || ip.is_unspecified() + || ip.is_multicast() + || (segments[0] & 0xfe00) == 0xfc00 + || (segments[0] & 0xffc0) == 0xfe80 + || (segments[0] & 0xffc0) == 0xfec0 + || (segments[0] == 0x2001 && segments[1] == 0x0db8) + || ip + .to_ipv4_mapped() + .or_else(|| ip.to_ipv4()) + .map(|ipv4| is_blocked_ip(IpAddr::V4(ipv4))) + .unwrap_or(false) + } + } +} + +#[derive(Default)] +struct PublicDnsResolver; + +struct SystemAddressResolver; + +impl AddressResolver for SystemAddressResolver { + fn resolve<'a>(&'a self, host: &'a str, port: u16) -> AddressResolution<'a> { + Box::pin(async move { + Ok(tokio::net::lookup_host((host, port)) + .await? + .collect::>()) + }) + } +} + +#[cfg(test)] +struct AllowPrivateResolver; + +#[cfg(test)] +impl AddressResolver for AllowPrivateResolver { + fn resolve<'a>(&'a self, _host: &'a str, port: u16) -> AddressResolution<'a> { + Box::pin(async move { Ok(vec![SocketAddr::from(([8, 8, 8, 8], port))]) }) + } +} + +impl Resolve for PublicDnsResolver { + fn resolve(&self, name: Name) -> Resolving { + let host = name.as_str().to_string(); + Box::pin(async move { + let addresses = tokio::net::lookup_host((host.as_str(), 0)) + .await + .map_err(|error| Box::new(error) as Box)? + .collect::>(); + validate_addresses(&addresses).map_err(|_| { + Box::new(io::Error::other("destination rejected by network policy")) + as Box + })?; + Ok(Box::new(addresses.into_iter()) as Addrs) + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::HashSet; + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + use tokio::net::TcpListener; + + async fn serve(response: &'static [u8]) -> (Url, tokio::task::JoinHandle<()>) { + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("test listener binds"); + let address = listener.local_addr().expect("listener has address"); + let task = tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.expect("accepts request"); + let mut request = [0_u8; 1024]; + let bytes_read = socket.read(&mut request).await.expect("reads request"); + assert!(bytes_read > 0); + socket.write_all(response).await.expect("writes response"); + }); + ( + Url::parse(&format!("http://{address}/document")).expect("valid test URL"), + task, + ) + } + + async fn serve_named( + host: &str, + responses: Vec<&'static [u8]>, + ) -> (Url, tokio::task::JoinHandle>, SocketAddr) { + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("test listener binds"); + let address = listener.local_addr().expect("listener has address"); + let task = tokio::spawn(async move { + let mut requests = Vec::with_capacity(responses.len()); + for response in responses { + let (mut socket, _) = listener.accept().await.expect("accepts request"); + let mut request = [0_u8; 4096]; + let bytes_read = socket.read(&mut request).await.expect("reads request"); + requests.push(String::from_utf8_lossy(&request[..bytes_read]).into_owned()); + socket.write_all(response).await.expect("writes response"); + } + requests + }); + ( + Url::parse(&format!("http://{host}:{}/document", address.port())) + .expect("valid test URL"), + task, + address, + ) + } + + struct LoopbackDnsResolver(SocketAddr); + + impl Resolve for LoopbackDnsResolver { + fn resolve(&self, _name: Name) -> Resolving { + let address = self.0; + Box::pin(async move { Ok(Box::new(vec![address].into_iter()) as Addrs) }) + } + } + + struct TestAddressResolver { + blocked_hosts: HashSet<&'static str>, + } + + impl AddressResolver for TestAddressResolver { + fn resolve<'a>(&'a self, host: &'a str, port: u16) -> AddressResolution<'a> { + let blocked = self.blocked_hosts.contains(host); + Box::pin(async move { + let ip = if blocked { + IpAddr::from([127, 0, 0, 1]) + } else { + IpAddr::from([8, 8, 8, 8]) + }; + Ok(vec![SocketAddr::new(ip, port)]) + }) + } + } + + fn policy_checked_fetcher( + address: SocketAddr, + blocked_hosts: HashSet<&'static str>, + ) -> MediaFetcher { + MediaFetcher::with_resolvers( + Arc::new(LoopbackDnsResolver(address)), + Arc::new(TestAddressResolver { blocked_hosts }), + ) + .expect("test fetcher builds") + } + + fn policy(max_bytes: u64, max_redirects: usize) -> DownloadPolicy { + DownloadPolicy { + timeout: Duration::from_secs(1), + max_bytes, + max_redirects, + } + } + + #[test] + fn blocks_non_public_addresses() { + for address in [ + "0.0.0.1", + "10.0.0.1", + "100.64.0.1", + "127.0.0.1", + "169.254.1.1", + "172.16.0.1", + "192.168.0.1", + "198.18.0.1", + "198.51.100.1", + "203.0.113.1", + "224.0.0.1", + "::1", + "fc00::1", + "fe80::1", + "2001:db8::1", + "::ffff:127.0.0.1", + ] { + assert!(is_blocked_ip(address.parse().expect("valid test address"))); + } + assert!(!is_blocked_ip( + "8.8.8.8".parse().expect("valid public address") + )); + } + + #[tokio::test] + async fn fetches_exact_limit_and_normalizes_content_type() { + let (url, server) = serve( + b"HTTP/1.1 200 OK\r\nContent-Type: application/pdf; charset=binary\r\nContent-Length: 3\r\nConnection: close\r\n\r\nabc", + ) + .await; + let client = reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .build() + .expect("test client builds"); + let media = MediaFetcher::for_test(client) + .fetch(url, policy(3, 0)) + .await + .expect("download succeeds at exact limit"); + server.await.expect("server completes"); + assert_eq!(media.bytes, b"abc"); + assert_eq!(media.content_type, "application/pdf"); + } + + #[tokio::test] + async fn rejects_declared_oversize_body() { + let (url, server) = serve( + b"HTTP/1.1 200 OK\r\nContent-Type: application/pdf\r\nContent-Length: 3\r\nConnection: close\r\n\r\nabc", + ) + .await; + let client = reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .build() + .expect("test client builds"); + let error = MediaFetcher::for_test(client) + .fetch(url, policy(2, 0)) + .await + .expect_err("oversize body is rejected"); + server.await.expect("server completes"); + assert!(matches!(error, MediaError::DownloadTooLarge)); + } + + #[tokio::test] + async fn rejects_streamed_oversize_body() { + let (url, server) = serve( + b"HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\nConnection: close\r\n\r\n2\r\nab\r\n2\r\ncd\r\n0\r\n\r\n", + ) + .await; + let client = reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .build() + .expect("test client builds"); + let error = MediaFetcher::for_test(client) + .fetch(url, policy(3, 0)) + .await + .expect_err("stream crossing limit is rejected"); + server.await.expect("server completes"); + assert!(matches!(error, MediaError::DownloadTooLarge)); + } + + #[tokio::test] + async fn follows_allowed_redirects_and_revalidates_each_destination() { + let (url, server, address) = serve_named( + "public.test", + vec![ + b"HTTP/1.1 302 Found\r\nLocation: /final\r\nContent-Length: 0\r\n\r\n", + b"HTTP/1.1 200 OK\r\nContent-Type: text/plain\r\nContent-Length: 2\r\nConnection: close\r\n\r\nok", + ], + ) + .await; + let media = policy_checked_fetcher(address, HashSet::new()) + .fetch(url, policy(2, 1)) + .await + .expect("redirected fetch succeeds"); + let requests = server.await.expect("server completes"); + assert_eq!(requests.len(), 2); + assert!(requests[1].starts_with("GET /final ")); + assert_eq!(media.bytes, b"ok"); + } + + #[tokio::test] + async fn blocks_redirected_private_destination_before_second_request() { + let (url, server, address) = serve_named( + "public.test", + vec![b"HTTP/1.1 302 Found\r\nLocation: http://blocked.test/document\r\nContent-Length: 0\r\n\r\n"], + ) + .await; + let error = policy_checked_fetcher(address, HashSet::from(["blocked.test"])) + .fetch(url, policy(10, 1)) + .await + .expect_err("private redirect is rejected"); + let requests = server.await.expect("server completes"); + assert_eq!(requests.len(), 1); + assert!(matches!(error, MediaError::BlockedUrl)); + } + + #[tokio::test] + async fn enforces_total_timeout() { + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("test listener binds"); + let address = listener.local_addr().expect("listener has address"); + let server = tokio::spawn(async move { + let (_socket, _) = listener.accept().await.expect("accepts request"); + tokio::time::sleep(Duration::from_millis(100)).await; + }); + let url = Url::parse(&format!("http://public.test:{}/document", address.port())) + .expect("valid test URL"); + let error = policy_checked_fetcher(address, HashSet::new()) + .fetch( + url, + DownloadPolicy { + timeout: Duration::from_millis(20), + max_bytes: 10, + max_redirects: 0, + }, + ) + .await + .expect_err("fetch times out"); + server.await.expect("server completes"); + assert!(matches!(error, MediaError::Timeout)); + } + + #[tokio::test] + async fn document_client_does_not_send_ambient_credentials() { + let (url, server, address) = serve_named( + "public.test", + vec![b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\nConnection: close\r\n\r\nok"], + ) + .await; + policy_checked_fetcher(address, HashSet::new()) + .fetch(url, policy(2, 0)) + .await + .expect("fetch succeeds"); + let requests = server.await.expect("server completes"); + assert!(!requests[0].to_ascii_lowercase().contains("authorization:")); + assert!(!requests[0].to_ascii_lowercase().contains("api-key:")); + } + + #[tokio::test] + async fn rejects_url_credentials_before_network_access() { + let fetcher = MediaFetcher::new().expect("media fetcher builds"); + let url = + Url::parse("https://user:password@8.8.8.8/document").expect("credentialed URL parses"); + assert!(matches!( + fetcher.validate_url(&url).await, + Err(MediaError::BlockedUrl) + )); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs b/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs new file mode 100644 index 00000000000..468c883a1dd --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs @@ -0,0 +1,146 @@ +use super::OcrAdapter; +use crate::Error; +use crate::constants::AZURE_AI_OCR_PATH; +use crate::ocr::OcrClient; +use crate::ocr::codecs::mistral::{self, MistralOcrParams, MistralOcrResponse}; +use crate::ocr::document::{inline_remote_document, validate_inline_document}; +use crate::ocr::error::{OcrError, OcrRequestError, OcrResponseError}; +use crate::ocr::prepare::{ + _prepare_ocr_request, ParsedProviderParams, credential_env, transform_request_body, +}; +use crate::ocr::registry::OcrProvider; +use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrConnection}; +use crate::url_utils::ApiUrl; + +const AZURE_AI_API_KEY_ENV: &str = "AZURE_AI_API_KEY"; +const AZURE_AI_API_BASE_ENV: &str = "AZURE_AI_API_BASE"; + +#[derive(Clone, Debug)] +pub(crate) struct AzureMistralAdapter; + +impl OcrAdapter for AzureMistralAdapter { + type ProviderResponse = MistralOcrResponse; + const PROVIDER: OcrProvider = OcrProvider::AzureAi; + + async fn prepare_request( + &self, + request: &LiteLLMOcrRequest, + client: &OcrClient, + ) -> Result { + let ParsedProviderParams { + known: params, + extra_params: _extra_params, + } = _prepare_ocr_request::(request)?; + let headers = authenticate(&request.connection, &credential_env)?; + let url = get_complete_url(request.connection.api_base.as_deref(), &credential_env)?; + let document = inline_remote_document( + client.document_fetcher(), + request.document.clone(), + &request.connection, + ) + .await?; + let body = mistral::transform_ocr_request(&request.model, document, ¶ms)?; + transform_request_body(client, request, &url, &headers, body, |body| { + validate_inline_document(&body.document) + }) + .await + } + + fn transform_ocr_response( + &self, + request: &LiteLLMOcrRequest, + response: Self::ProviderResponse, + ) -> Result { + mistral::transform_ocr_response(&request.model, response) + } +} + +fn get_complete_url( + api_base: Option<&str>, + env_lookup: &dyn Fn(&str) -> Option, +) -> Result { + let base = nonblank(api_base.map(str::to_string)) + .or_else(|| nonblank(env_lookup(AZURE_AI_API_BASE_ENV))) + .ok_or_else(|| Error::Auth( + "Missing Azure AI API Base - Set AZURE_AI_API_BASE environment variable or pass api_base parameter".into(), + ))?; + let path: Vec<&str> = AZURE_AI_OCR_PATH.trim_matches('/').split('/').collect(); + ApiUrl::parse(&base) + .and_then(|url| url.complete_path(&path)) + .map(|url| url.into_string()) + .map_err(|_| { + OcrRequestError::RequestField { + path: "api_base".into(), + } + .into() + }) +} + +fn authenticate( + connection: &OcrConnection, + env_lookup: &dyn Fn(&str) -> Option, +) -> Result, OcrError> { + if crate::http_utils::has_header(&connection.extra_headers, "authorization") { + return Ok(connection.extra_headers.clone()); + } + let key = nonblank(connection.api_key.clone()) + .or_else(|| nonblank(env_lookup(AZURE_AI_API_KEY_ENV))) + .ok_or(Error::MissingAzureAiCredentials)?; + Ok( + std::iter::once(("Authorization".into(), format!("Bearer {key}"))) + .chain(connection.extra_headers.clone()) + .collect(), + ) +} + +fn nonblank(value: Option) -> Option { + value + .map(|value| value.trim().to_string()) + .filter(|value| !value.is_empty()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn completes_azure_path_and_preserves_query() { + assert_eq!( + get_complete_url(Some("https://example.com/?tenant=a"), &|_| None).unwrap(), + "https://example.com/providers/mistral/azure/ocr?tenant=a" + ); + assert_eq!( + get_complete_url( + Some("https://example.com/providers/mistral/azure/ocr"), + &|_| None + ) + .unwrap(), + "https://example.com/providers/mistral/azure/ocr" + ); + } + + #[test] + fn supplied_authorization_precedes_keys() { + let connection = OcrConnection { + api_key: Some("request-key".into()), + extra_headers: vec![("authorization".into(), "Bearer prepared".into())], + ..Default::default() + }; + assert_eq!( + authenticate(&connection, &|_| Some("environment-key".into())).unwrap(), + connection.extra_headers + ); + } + + #[test] + fn request_key_precedes_environment_key() { + let connection = OcrConnection { + api_key: Some("request-key".into()), + ..Default::default() + }; + assert_eq!( + authenticate(&connection, &|_| Some("environment-key".into())).unwrap()[0], + ("Authorization".into(), "Bearer request-key".into()) + ); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/mod.rs index 7dfb08b4dc2..bbd6feb6c7b 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/mod.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/mod.rs @@ -8,8 +8,10 @@ use super::registry::OcrProvider; use super::types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrResponseFormat}; use super::wire::DecodedOcrResponse; +mod azure_mistral; mod mistral; +pub(crate) use azure_mistral::AzureMistralAdapter; pub(crate) use mistral::MistralAdapter; /// Converts a complete LiteLLM OCR call to provider HTTP and normalizes its response. @@ -62,6 +64,7 @@ macro_rules! for_each_ocr_adapter { ($callback:ident) => { $callback! { Mistral, $crate::ocr::adapters::MistralAdapter, $crate::ocr::adapters::MistralAdapter, Mistral; + AzureMistral, $crate::ocr::adapters::AzureMistralAdapter, $crate::ocr::adapters::AzureMistralAdapter, AzureAi; } }; } diff --git a/litellm-rust/crates/core/src/ocr/client.rs b/litellm-rust/crates/core/src/ocr/client.rs index 36bfc036bae..699e44412ac 100644 --- a/litellm-rust/crates/core/src/ocr/client.rs +++ b/litellm-rust/crates/core/src/ocr/client.rs @@ -10,15 +10,21 @@ use super::wire::{DecodedOcrResponse, decode_response}; use crate::Error; use crate::constants::OCR_CONNECT_TIMEOUT_SECS; use crate::error::TransportError; +use crate::media::MediaFetcher; #[derive(Clone)] pub struct OcrClient { provider_http: reqwest::Client, + document_fetcher: MediaFetcher, } impl OcrClient { pub fn new(provider_http: reqwest::Client) -> Result { - Ok(Self { provider_http }) + let document_fetcher = MediaFetcher::new().map_err(TransportError::from)?; + Ok(Self { + provider_http, + document_fetcher, + }) } #[tracing::instrument( @@ -35,9 +41,16 @@ impl OcrClient { &self.provider_http } + pub(crate) fn document_fetcher(&self) -> &MediaFetcher { + &self.document_fetcher + } + #[cfg(test)] - pub(crate) fn for_test(provider_http: reqwest::Client) -> Self { - Self { provider_http } + pub(crate) fn for_test(provider_http: reqwest::Client, document_http: reqwest::Client) -> Self { + Self { + provider_http, + document_fetcher: MediaFetcher::for_test(document_http), + } } } diff --git a/litellm-rust/crates/core/src/ocr/document.rs b/litellm-rust/crates/core/src/ocr/document.rs new file mode 100644 index 00000000000..11a6e612a3c --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/document.rs @@ -0,0 +1,214 @@ +use base64::{Engine, engine::general_purpose::STANDARD}; +#[cfg(test)] +use data_url::mime::Mime; +use data_url::{DataUrl, DataUrlError, forgiving_base64::DecodeError}; +use reqwest::Url; + +use super::error::{OcrError, OcrRequestError, OcrResponseError}; +use super::types::{OcrConnection, OcrDocument}; +use crate::constants::OCR_MAX_FETCH_REDIRECTS; +use crate::error::{MediaError, TransportError}; +use crate::media::{DownloadPolicy, MediaFetcher}; + +pub(crate) struct InlineDocument<'a>(DataUrl<'a>); + +impl<'a> InlineDocument<'a> { + pub(crate) fn parse(source: &'a str) -> Result, OcrRequestError> { + match DataUrl::process(source) { + Ok(url) => Ok(Some(Self(url))), + Err(DataUrlError::NotADataUrl) => Ok(None), + Err(DataUrlError::NoComma) => Err(OcrRequestError::InvalidDataUri), + } + } + + #[cfg(test)] + pub(crate) fn mime_type(&self) -> &Mime { + self.0.mime_type() + } + + pub(crate) fn decode(&self, max_bytes: usize) -> Result, OcrRequestError> { + let mut body = Vec::new(); + self.0 + .decode(|bytes| { + if bytes.len() > max_bytes.saturating_sub(body.len()) { + return Err(OcrRequestError::InlineDocumentTooLarge); + } + body.extend_from_slice(bytes); + Ok(()) + }) + .map_err(|error| match error { + DecodeError::InvalidBase64(_) => OcrRequestError::InvalidDataUri, + DecodeError::WriteError(error) => error, + })?; + Ok(body) + } +} + +pub(crate) fn validate_inline_document(document: &OcrDocument) -> Result<(), OcrRequestError> { + let inline = + InlineDocument::parse(document.source())?.ok_or(OcrRequestError::InvalidDataUri)?; + inline.decode(crate::constants::OCR_INLINE_MAX_BYTES)?; + Ok(()) +} + +pub(crate) async fn inline_remote_document( + fetcher: &MediaFetcher, + document: OcrDocument, + connection: &OcrConnection, +) -> Result { + let source = document.source(); + if !source.starts_with("http://") && !source.starts_with("https://") { + validate_inline_document(&document)?; + return Ok(document); + } + let url = Url::parse(source).map_err(|_| OcrRequestError::RequestField { + path: "document URL".into(), + })?; + let downloaded = fetcher + .fetch( + url, + DownloadPolicy { + timeout: connection.timeout, + max_bytes: connection.max_download_bytes, + max_redirects: OCR_MAX_FETCH_REDIRECTS, + }, + ) + .await + .map_err(map_media_error)?; + let result = document.with_source(format!( + "data:{};base64,{}", + downloaded.content_type, + STANDARD.encode(downloaded.bytes) + )); + validate_inline_document(&result)?; + Ok(result) +} + +fn map_media_error(error: MediaError) -> OcrError { + match error { + MediaError::BlockedUrl => OcrRequestError::BlockedDocumentUrl.into(), + MediaError::DownloadDisabled => OcrRequestError::DownloadDisabled.into(), + MediaError::DownloadTooLarge => OcrRequestError::DownloadTooLarge.into(), + MediaError::TooManyRedirects => OcrRequestError::TooManyRedirects.into(), + MediaError::MissingRedirectLocation => OcrResponseError::MissingRedirectLocation.into(), + MediaError::InvalidRedirect => OcrResponseError::InvalidRedirect.into(), + MediaError::Http(status) => TransportError::Http { + status, + body: "OCR document download failed".into(), + } + .into(), + MediaError::Timeout => { + TransportError::Network("OCR document download timed out".into()).into() + } + MediaError::Transport(error) => error.into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::Map; + + fn document(source: &str) -> OcrDocument { + OcrDocument::DocumentUrl { + document_url: source.into(), + extra_fields: Map::new(), + } + } + + #[test] + fn decodes_data_urls_and_limits_decoded_size() { + for (source, expected) in [ + ("data:application/pdf;base64,YWJj", b"abc".as_slice()), + ("DATA:application/pdf;BASE64,YWI", b"ab".as_slice()), + ("data:,a%20b%00%FF", b"a b\0\xff".as_slice()), + ] { + let inline = InlineDocument::parse(source).unwrap().unwrap(); + assert_eq!(inline.decode(expected.len()).unwrap(), expected); + assert_eq!( + inline.decode(expected.len() - 1), + Err(OcrRequestError::InlineDocumentTooLarge) + ); + } + } + + #[test] + fn preserves_mime_parameters_and_standard_default() { + let inline = InlineDocument::parse("data:application/pdf;version=1.7;base64,YQ==") + .unwrap() + .unwrap(); + assert!(inline.mime_type().matches("application", "pdf")); + assert_eq!(inline.mime_type().get_parameter("version"), Some("1.7")); + let default = InlineDocument::parse("data:,a").unwrap().unwrap(); + assert!(default.mime_type().matches("text", "plain")); + assert_eq!( + default.mime_type().get_parameter("charset"), + Some("US-ASCII") + ); + } + + #[test] + fn rejects_invalid_inline_documents() { + for source in [ + "https://example.com/document.pdf", + "data:application/pdf;base64", + "data:application/pdf;base64,INVALID!", + ] { + assert!(validate_inline_document(&document(source)).is_err()); + } + } + + #[tokio::test] + async fn remote_conversion_preserves_kind_and_isolates_provider_credentials() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + use tokio::net::TcpListener; + + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + let server = tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.unwrap(); + let mut request = vec![0_u8; 2048]; + let count = socket.read(&mut request).await.unwrap(); + socket + .write_all(b"HTTP/1.1 200 OK\r\nContent-Type: image/png; charset=binary\r\nContent-Length: 3\r\nConnection: close\r\n\r\nabc") + .await + .unwrap(); + String::from_utf8_lossy(&request[..count]).into_owned() + }); + let mut provider_headers = reqwest::header::HeaderMap::new(); + provider_headers.insert( + reqwest::header::AUTHORIZATION, + reqwest::header::HeaderValue::from_static("Bearer provider-secret"), + ); + let provider_http = reqwest::Client::builder() + .default_headers(provider_headers) + .build() + .unwrap(); + let document_http = reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .build() + .unwrap(); + let client = super::super::OcrClient::for_test(provider_http, document_http); + let converted = inline_remote_document( + client.document_fetcher(), + OcrDocument::ImageUrl { + image_url: format!("http://{address}/image"), + extra_fields: Map::from_iter([("detail".into(), serde_json::json!("high"))]), + }, + &OcrConnection::default(), + ) + .await + .unwrap(); + let request = server.await.unwrap(); + + assert_eq!( + converted, + OcrDocument::ImageUrl { + image_url: "data:image/png;base64,YWJj".into(), + extra_fields: Map::from_iter([("detail".into(), serde_json::json!("high"))]), + } + ); + assert!(!request.to_ascii_lowercase().contains("authorization")); + assert!(!request.contains("provider-secret")); + } +} diff --git a/litellm-rust/crates/core/src/ocr/error.rs b/litellm-rust/crates/core/src/ocr/error.rs index 50793aad566..395bb60000c 100644 --- a/litellm-rust/crates/core/src/ocr/error.rs +++ b/litellm-rust/crates/core/src/ocr/error.rs @@ -10,12 +10,28 @@ pub enum OcrRequestError { RequestField { path: String }, #[error("missing required field: {0}")] MissingField(&'static str), + #[error("invalid OCR document data URI")] + InvalidDataUri, + #[error("inline OCR document exceeds the size limit")] + InlineDocumentTooLarge, + #[error("OCR document URL is blocked by network policy")] + BlockedDocumentUrl, + #[error("OCR document downloads are disabled")] + DownloadDisabled, + #[error("OCR document download exceeds the size limit")] + DownloadTooLarge, + #[error("OCR document download exceeded the redirect limit")] + TooManyRedirects, } #[derive(Debug, Clone, PartialEq, Eq, Error)] pub enum OcrResponseError { #[error("invalid OCR response field: {path}")] ResponseField { path: String }, + #[error("OCR document redirect is missing a location")] + MissingRedirectLocation, + #[error("OCR document redirect location is invalid")] + InvalidRedirect, } #[derive(Debug, Error)] diff --git a/litellm-rust/crates/core/src/ocr/mod.rs b/litellm-rust/crates/core/src/ocr/mod.rs index 78c86aeaa0e..aa11d0ab3cf 100644 --- a/litellm-rust/crates/core/src/ocr/mod.rs +++ b/litellm-rust/crates/core/src/ocr/mod.rs @@ -1,6 +1,7 @@ mod adapters; pub mod client; mod codecs; +mod document; pub mod error; mod handler; pub mod hooks; @@ -13,6 +14,9 @@ pub mod wire; pub use client::{OcrClient, ocr}; pub use types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrConnection, OcrDocument}; +#[cfg(test)] +#[path = "../../tests/azure_ai_ocr.rs"] +mod azure_ai_tests; #[cfg(test)] #[path = "../../tests/ocr/support.rs"] pub(crate) mod test_support; diff --git a/litellm-rust/crates/core/src/ocr/registry.rs b/litellm-rust/crates/core/src/ocr/registry.rs index 9bae8153ee8..e70bd7314c0 100644 --- a/litellm-rust/crates/core/src/ocr/registry.rs +++ b/litellm-rust/crates/core/src/ocr/registry.rs @@ -24,12 +24,14 @@ super::adapters::for_each_ocr_adapter!(define_adapter_types); #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub(crate) enum OcrProvider { Mistral, + AzureAi, } impl OcrProvider { pub(crate) const fn as_str(self) -> &'static str { match self { Self::Mistral => "mistral", + Self::AzureAi => "azure_ai", } } } @@ -45,9 +47,19 @@ pub(crate) fn resolve_wire_adapter( }); let typed_provider = match provider.custom_llm_provider { "mistral" => OcrProvider::Mistral, + "azure_ai" => OcrProvider::AzureAi, value => return Err(Error::InvalidProvider(value.to_string())), }; match typed_provider { OcrProvider::Mistral => Ok((provider.model.to_string(), OcrAdapterKind::Mistral)), + OcrProvider::AzureAi if is_document_intelligence_model(provider.model) => { + Err(Error::InvalidProvider("azure_ai".into())) + } + OcrProvider::AzureAi => Ok((provider.model.to_string(), OcrAdapterKind::AzureMistral)), } } + +fn is_document_intelligence_model(model: &str) -> bool { + let model = model.to_ascii_lowercase(); + model.contains("doc-intelligence") || model.contains("documentintelligence") +} diff --git a/litellm-rust/crates/core/src/ocr/types.rs b/litellm-rust/crates/core/src/ocr/types.rs index 02b5330188c..eeda94738a0 100644 --- a/litellm-rust/crates/core/src/ocr/types.rs +++ b/litellm-rust/crates/core/src/ocr/types.rs @@ -32,6 +32,28 @@ pub enum OcrDocument { }, } +impl OcrDocument { + pub(crate) fn source(&self) -> &str { + match self { + Self::DocumentUrl { document_url, .. } => document_url, + Self::ImageUrl { image_url, .. } => image_url, + } + } + + pub(crate) fn with_source(self, source: String) -> Self { + match self { + Self::DocumentUrl { extra_fields, .. } => Self::DocumentUrl { + document_url: source, + extra_fields, + }, + Self::ImageUrl { extra_fields, .. } => Self::ImageUrl { + image_url: source, + extra_fields, + }, + } + } +} + #[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "lowercase")] pub enum OcrResponseFormat { @@ -46,6 +68,7 @@ pub struct OcrConnection { pub api_base: Option, pub extra_headers: Vec<(String, String)>, pub timeout: Duration, + pub max_download_bytes: u64, } impl Default for OcrConnection { @@ -55,6 +78,7 @@ impl Default for OcrConnection { api_base: None, extra_headers: Vec::new(), timeout: Duration::from_secs(OCR_HTTP_TIMEOUT_SECS), + max_download_bytes: crate::constants::OCR_DOWNLOAD_MAX_BYTES, } } } diff --git a/litellm-rust/crates/core/src/ocr/wire.rs b/litellm-rust/crates/core/src/ocr/wire.rs index f37c06a01ac..db8d91905c4 100644 --- a/litellm-rust/crates/core/src/ocr/wire.rs +++ b/litellm-rust/crates/core/src/ocr/wire.rs @@ -70,6 +70,7 @@ pub fn decode_request(wire: OcrWireRequest) -> Result api_base: nonblank(wire.api_base), extra_headers: headers, timeout: timeout.unwrap_or(defaults.timeout), + max_download_bytes: defaults.max_download_bytes, }; Ok(LiteLLMOcrRequest { connection, diff --git a/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs b/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs index 2af25a5639a..2dbec8e2187 100644 --- a/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs +++ b/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs @@ -125,12 +125,7 @@ pub fn validate_azure_ai_environment( } non_empty(azure_ad_token) .map(|token| prepend_auth_header(headers, "Authorization", format!("Bearer {token}"))) - .ok_or_else(|| { - Error::Auth( - "Missing Azure AI credentials - set AZURE_AI_API_KEY or provide azure_ad_token" - .to_string(), - ) - }) + .ok_or(Error::MissingAzureAiCredentialsOrAdToken) } pub fn validate_document_intelligence_environment( diff --git a/litellm-rust/crates/core/tests/azure_ai_ocr.rs b/litellm-rust/crates/core/tests/azure_ai_ocr.rs new file mode 100644 index 00000000000..3d7fb2e54a3 --- /dev/null +++ b/litellm-rust/crates/core/tests/azure_ai_ocr.rs @@ -0,0 +1,75 @@ +use std::sync::Arc; + +use serde_json::{Value, json}; + +use super::hooks::{OcrDuringCallRequest, OcrHookFuture, OcrHooks}; +use super::test_support::{MockResponse, mock_server, perform_ocr, wire_request}; + +#[tokio::test] +async fn facade_executes_azure_mistral_with_prepared_auth() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "pages":[{"index":0,"markdown":"hello"}], + "usage_info":{"pages_processed":1} + }))]) + .await; + let mut request = wire_request( + "azure_ai/model", + &base, + json!({"include_image_base64":true}), + ); + request.connection.api_key = None; + request.connection.extra_headers = vec![( + "Authorization".into(), + "Bearer python-prepared-token".into(), + )]; + + let result = perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert_eq!(result.pages[0]["markdown"], "hello"); + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(requests[0].starts_with("POST /providers/mistral/azure/ocr ")); + assert!( + requests[0] + .to_ascii_lowercase() + .contains("authorization: bearer python-prepared-token\r\n") + ); + let body: Value = serde_json::from_str(requests[0].split_once("\r\n\r\n").unwrap().1).unwrap(); + assert_eq!( + body, + json!({ + "model":"model", + "document":{"type":"document_url","document_url":"data:application/pdf;base64,YWJj"}, + "include_image_base64":true + }) + ); +} + +struct ReplaceBodyDocument; + +impl OcrHooks for ReplaceBodyDocument { + fn has_guardrails(&self) -> bool { + true + } + + fn during_call( + &self, + mut request: OcrDuringCallRequest, + ) -> OcrHookFuture<'_, OcrDuringCallRequest> { + Box::pin(async move { + request.body["document"] = json!({ + "type":"document_url", + "document_url":"https://example.com/not-inline.pdf" + }); + Ok(request) + }) + } +} + +#[tokio::test] +async fn rejects_non_inline_body_after_guardrails() { + let mut request = wire_request("azure_ai/model", "http://127.0.0.1:1", json!({})); + request.hooks = Arc::new(ReplaceBodyDocument); + let error = perform_ocr(request).await.unwrap_err(); + assert!(error.to_string().contains("data URI")); +} diff --git a/litellm-rust/crates/core/tests/ocr/support.rs b/litellm-rust/crates/core/tests/ocr/support.rs index d9720019419..45047a0e62a 100644 --- a/litellm-rust/crates/core/tests/ocr/support.rs +++ b/litellm-rust/crates/core/tests/ocr/support.rs @@ -8,7 +8,11 @@ use crate::ocr::wire::{OcrWireRequest, decode_request}; use crate::ocr::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrClient}; pub(crate) fn ocr_client() -> OcrClient { - OcrClient::for_test(reqwest::Client::new()) + let document_http = reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .build() + .expect("test document client builds"); + OcrClient::for_test(reqwest::Client::new(), document_http) } pub(crate) async fn perform_ocr( diff --git a/litellm-rust/crates/python-bridge/src/errors.rs b/litellm-rust/crates/python-bridge/src/errors.rs index 407475dedce..0c30eab8112 100644 --- a/litellm-rust/crates/python-bridge/src/errors.rs +++ b/litellm-rust/crates/python-bridge/src/errors.rs @@ -42,6 +42,8 @@ pub(crate) fn chat_completions_error_to_pyerr(err: Error) -> PyErr { | Error::InvalidType { .. } | Error::MissingField(_) | Error::MissingApiKey { .. } + | Error::MissingAzureAiCredentials + | Error::MissingAzureAiCredentialsOrAdToken | Error::Routing(_) // Nothing reached the provider, so serving it on Python cannot double // bill and is the only way the caller gets an answer at all. diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index 951caf4eef4..2e6900f784f 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -87,3 +87,19 @@ bridge_route! { prepare = prepare_ocr, errors = ocr_error_to_pyerr, } + +#[cfg(test)] +mod tests { + use litellm_core::ocr::wire::is_supported_request; + + #[test] + fn native_activation_excludes_unmigrated_azure_document_intelligence() { + assert!(is_supported_request("model", Some("mistral"))); + assert!(is_supported_request("pixtral-12b", Some("azure_ai"))); + assert!(!is_supported_request( + "documentintelligence/prebuilt-read", + Some("azure_ai") + )); + assert!(!is_supported_request("mistral-ocr", Some("vertex_ai"))); + } +} diff --git a/tests/test_litellm/rust_bridge/native_route_wheel_test.py b/tests/test_litellm/rust_bridge/native_route_wheel_test.py index 9807febbff4..5b2af3cf02e 100644 --- a/tests/test_litellm/rust_bridge/native_route_wheel_test.py +++ b/tests/test_litellm/rust_bridge/native_route_wheel_test.py @@ -73,7 +73,7 @@ def assert_native_request( headers: HTTPMessage, body: object, ) -> None: - if route not in {"ocr", "transcription", "messages", "chat_completions"}: + if route not in {"ocr", "azure_ocr", "transcription", "messages", "chat_completions"}: raise AssertionError(f"unexpected route marker: {route!r}") if outcome not in {"success", "429", "hang"}: raise AssertionError(f"unexpected outcome marker: {outcome!r}") @@ -86,6 +86,12 @@ def assert_native_request( assert body["document"]["document_url"] == "https://example.com/document.pdf" assert body["include_image_base64"] is True return + if route == "azure_ocr": + assert path == "/providers/mistral/azure/ocr" + assert headers.get("authorization") == "Bearer prepared-azure-token" + assert body["model"] == "mistral-ocr-2505" + assert body["document"]["document_url"] == "data:application/pdf;base64,YWJj" + return if route == "transcription": assert path == "/model/mistral.voxtral-mini-3b-2507/converse" assert headers.get("authorization", "").startswith("AWS4-HMAC-SHA256 ") @@ -107,7 +113,7 @@ def assert_native_request( def native_response(status: int, route: str | None) -> bytes: if status == 429: return b'{"error":"native-rate-limit"}' - if route == "ocr": + if route in {"ocr", "azure_ocr"}: return b'{"pages":[{"index":0,"markdown":"native-ocr"}]}' if route == "transcription": return b'{"output":{"message":{"content":[{"text":"native-transcription"}]}}}' @@ -181,6 +187,20 @@ def assert_success(route: str, response: object) -> None: raise AssertionError(f"{route} returned {actual!r}, expected {expected!r}") +def azure_ocr_kwargs(api_base: str) -> dict[str, object]: + return { + "model": "mistral-ocr-2505", + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "api_base": api_base, + "custom_llm_provider": "azure_ai", + "extra_headers": { + "Authorization": "Bearer prepared-azure-token", + "x-test-outcome": "success", + "x-test-route": "azure_ocr", + }, + } + + def success_value(route: str, response: dict[object, object]) -> object: if route == "ocr": return response["pages"][0]["markdown"] @@ -211,6 +231,7 @@ def exercise_sync(native: object, api_base: str) -> None: assert_rate_limit(native, route, error) else: raise AssertionError(f"{route} accepted a 429 response") + assert_success("ocr", native.ocr(**azure_ocr_kwargs(api_base))) async def exercise_async(native: object, api_base: str) -> None: @@ -223,6 +244,7 @@ async def exercise_async(native: object, api_base: str) -> None: assert_rate_limit(native, route, error) else: raise AssertionError(f"a{route} accepted a 429 response") + assert_success("ocr", await native.aocr(**azure_ocr_kwargs(api_base))) async def exercise_async_concurrency(native: object, api_base: str) -> None: From 09b694894d4b71fab0b6a194ada64689b0f59e08 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 13:12:14 -0700 Subject: [PATCH 101/157] fix(datadog_llm_obs): keep tool call and result structure under redaction and emit tool output tokens (#40666) * fix(datadog_llm_obs): keep tool call and result structure under redaction and emit tool output tokens Under datadog_llm_observability_params.turn_off_message_logging the span kept only one role plus "redacted-by-litellm" per message, so Datadog showed Tool Call 0, Tool Result 0 and no tool output token data. The shared CustomLogger hook collapsed the messages before the callback ran, and the Datadog redaction then dropped tool_calls and tool_results. The Datadog callback now opts out of the shared message collapse (redacts_messages_itself) and redacts its own normalized messages, keeping roles, tool names, ids and types while replacing content, arguments and results. Tool result tokens are counted with litellm.token_counter before redaction and shipped as the tool_output_tokens metric. Other callbacks keep the inherited behavior. Resolves LIT-7545 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(datadog_llm_obs): drop explanatory docstrings from the redaction change Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(ui): regenerate schema.d.ts for the classifier descriptions changed in #40655 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/custom_logger.py | 8 +- .../integrations/datadog/datadog_llm_obs.py | 65 +++++-- litellm/types/integrations/datadog_llm_obs.py | 1 + .../datadog/test_datadog_llm_obs.py | 164 ++++++++++++++++-- .../test_redact_messages.py | 35 ++++ 5 files changed, 249 insertions(+), 24 deletions(-) diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py index d445a3adf14..62ca6b0254e 100644 --- a/litellm/integrations/custom_logger.py +++ b/litellm/integrations/custom_logger.py @@ -886,12 +886,16 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac return LITELLM_METADATA_FIELD return OLD_LITELLM_METADATA_FIELD + def redacts_messages_itself(self) -> bool: + return False + def redact_standard_logging_payload_from_model_call_details(self, model_call_details: dict) -> dict: """ Redacts or excludes fields from StandardLoggingPayload before callbacks receive it. This method handles two features: - 1. turn_off_message_logging: When True, redacts messages and responses + 1. turn_off_message_logging: When True, redacts messages and responses (unless the callback + redacts them itself, see `redacts_messages_itself`) 2. standard_logging_payload_excluded_fields: Removes specified fields entirely Return a modified copy of the provided logging payload. @@ -921,7 +925,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac } # Handle turn_off_message_logging - redact messages and responses (if not already excluded) - if turn_off_message_logging: + if turn_off_message_logging and not self.redacts_messages_itself(): redacted_str: Final = "redacted-by-litellm" if "messages" not in (excluded_fields or ()) and standard_logging_object_copy.get("messages") is not None: diff --git a/litellm/integrations/datadog/datadog_llm_obs.py b/litellm/integrations/datadog/datadog_llm_obs.py index 728bf41856f..c64a12c6d75 100644 --- a/litellm/integrations/datadog/datadog_llm_obs.py +++ b/litellm/integrations/datadog/datadog_llm_obs.py @@ -165,18 +165,54 @@ def _metadata_without_prompt_carriers(standard_logging_metadata: Mapping[str, An ) -def _redact_messages(messages: Sequence[Message]) -> tuple[Message, ...]: - """Each message's shape with its content replaced and tool payloads dropped; no message is invented.""" - return tuple( - { - "role": role if isinstance(role, str) and role in _SAFE_REDACTED_MESSAGE_ROLES else "", - "content": REDACTED_BY_LITELLM, - } - for message in messages - for role in (message.get("role", ""),) +def _safe_identifier(value: object) -> str: + return value if isinstance(value, str) else "" + + +def _redact_tool_call(tool_call: ToolCall) -> ToolCall: + return ToolCall( + name=_safe_identifier(tool_call.get("name")), + arguments=REDACTED_BY_LITELLM, + tool_id=_safe_identifier(tool_call.get("tool_id")), + type=_safe_identifier(tool_call.get("type")), ) +def _redact_tool_result(tool_result: ToolResult) -> ToolResult: + return ToolResult( + name=_safe_identifier(tool_result.get("name")), + result=REDACTED_BY_LITELLM, + tool_id=_safe_identifier(tool_result.get("tool_id")), + type=_safe_identifier(tool_result.get("type")), + ) + + +def _redact_message(message: Message) -> Message: + role: Final = message.get("role", "") + tool_calls: Final = message.get("tool_calls", ()) + tool_results: Final = message.get("tool_results", ()) + redacted: Final[Message] = { + "role": role if isinstance(role, str) and role in _SAFE_REDACTED_MESSAGE_ROLES else "", + "content": REDACTED_BY_LITELLM, + **({"tool_calls": tuple(_redact_tool_call(call) for call in tool_calls)} if tool_calls else {}), + **({"tool_results": tuple(_redact_tool_result(result) for result in tool_results)} if tool_results else {}), + } + return redacted + + +def _redact_messages(messages: Sequence[Message]) -> tuple[Message, ...]: + return tuple(_redact_message(message) for message in messages) + + +def _tool_output_tokens(messages: Sequence[Message], model: str) -> float | None: + results: Final = tuple( + result.get("result", "") for message in messages for result in message.get("tool_results", ()) + ) + if not results: + return None + return float(sum(litellm.token_counter(model=model, text=result) for result in results)) + + def _cost_dimension_tags( standard_logging_payload: StandardLoggingPayload, router_fields: Mapping[str, object] ) -> tuple[str, ...]: @@ -583,6 +619,7 @@ class DataDogLLMObsLogger(CustomBatchLogger): standard_logging_payload=standard_logging_payload, call_type=standard_logging_payload.get("call_type"), ) + tool_output_tokens: Final = _tool_output_tokens(input_messages, standard_logging_payload.get("model") or "") input_meta: Final = InputMeta(messages=_redact_messages(input_messages) if redact_payload else input_messages) output_meta: Final = OutputMeta( messages=_redact_messages(output_messages) if redact_payload else output_messages @@ -618,7 +655,7 @@ class DataDogLLMObsLogger(CustomBatchLogger): **({"tool_definitions": tool_definitions} if tool_definitions else {}), } - metrics: Final = self._assemble_metrics(standard_logging_payload) + metrics: Final = self._assemble_metrics(standard_logging_payload, tool_output_tokens) payload: Final[LLMObsPayload] = LLMObsPayload( parent_id=metadata_parent_id if metadata_parent_id else "undefined", @@ -676,6 +713,9 @@ class DataDogLLMObsLogger(CustomBatchLogger): ) return error_info + def redacts_messages_itself(self) -> bool: + return True + def _payload_logging_is_off(self, kwargs: Mapping[str, Any]) -> bool: return ( bool(self.turn_off_message_logging) @@ -683,7 +723,9 @@ class DataDogLLMObsLogger(CustomBatchLogger): or should_redact_message_logging(dict(kwargs)) ) - def _assemble_metrics(self, standard_logging_payload: StandardLoggingPayload) -> LLMMetrics: + def _assemble_metrics( + self, standard_logging_payload: StandardLoggingPayload, tool_output_tokens: float | None + ) -> LLMMetrics: """ Build the span metrics, including the prompt-cache counts LLM Obs charts cache savings from. @@ -721,6 +763,7 @@ class DataDogLLMObsLogger(CustomBatchLogger): else {} ), **({"reasoning_output_tokens": reasoning_output_tokens} if reasoning_output_tokens else {}), + **({"tool_output_tokens": tool_output_tokens} if tool_output_tokens is not None else {}), } return metrics diff --git a/litellm/types/integrations/datadog_llm_obs.py b/litellm/types/integrations/datadog_llm_obs.py index 17cf5831c96..f4fcabf53ed 100644 --- a/litellm/types/integrations/datadog_llm_obs.py +++ b/litellm/types/integrations/datadog_llm_obs.py @@ -87,6 +87,7 @@ class LLMMetrics(TypedDict, total=False): cache_write_input_tokens: ReadOnly[float] non_cached_input_tokens: ReadOnly[float] reasoning_output_tokens: ReadOnly[float] + tool_output_tokens: ReadOnly[float] class LLMObsPayload(TypedDict, total=False): diff --git a/tests/test_litellm/integrations/datadog/test_datadog_llm_obs.py b/tests/test_litellm/integrations/datadog/test_datadog_llm_obs.py index 0555447e34f..a2f81091893 100644 --- a/tests/test_litellm/integrations/datadog/test_datadog_llm_obs.py +++ b/tests/test_litellm/integrations/datadog/test_datadog_llm_obs.py @@ -541,24 +541,166 @@ def _span_json(logger_under_test: DataDogLLMObsLogger, payload: dict[str, Any]) return json.loads(safe_dumps(span)) +SECRET_TOOL_RESULT: Final = '{"city": "Paris", "temp_c": 18, "account_secret": "SECRET-7545"}' +TOOL_CONVERSATION: Final[list[dict[str, Any]]] = [ + {"role": "user", "content": "secret prompt"}, + {"role": "assistant", "content": None, "tool_calls": [ASSISTANT_TOOL_CALL]}, + {"role": "tool", "tool_call_id": "call_abc123", "content": SECRET_TOOL_RESULT}, +] + + +def _redacted_span_as_the_proxy_builds_it(payload: dict[str, Any]) -> dict[str, Any]: + logger_under_test = _redacting_logger(turn_off_message_logging=True) + return _span_json( + logger_under_test, logger_under_test.redact_standard_logging_payload_from_model_call_details(payload) + ) + + def test_redaction_keeps_the_conversation_shape_without_its_content() -> None: - """Roles and message count survive so the trace stays legible; contents and tool payloads do not.""" - result = _span_json( - _redacting_logger(turn_off_message_logging=True), + result = _redacted_span_as_the_proxy_builds_it( + build_payload( + messages=TOOL_CONVERSATION, + response_message={"role": "assistant", "content": "secret response", "tool_calls": [ASSISTANT_TOOL_CALL]}, + ) + ) + + redacted_call = { + "name": "get_weather", + "arguments": "redacted-by-litellm", + "tool_id": "call_abc123", + "type": "function", + } + assert result["meta"]["input"]["messages"] == [ + {"role": "user", "content": "redacted-by-litellm"}, + {"role": "assistant", "content": "redacted-by-litellm", "tool_calls": [redacted_call]}, + { + "role": "tool", + "content": "redacted-by-litellm", + "tool_results": [ + {"name": "get_weather", "result": "redacted-by-litellm", "tool_id": "call_abc123", "type": "function"} + ], + }, + ] + assert result["meta"]["output"]["messages"] == [ + {"role": "assistant", "content": "redacted-by-litellm", "tool_calls": [redacted_call]} + ] + serialized = safe_dumps(result) + assert "SECRET-7545" not in serialized + assert "Paris" not in serialized + assert "secret" not in serialized + + +def test_redaction_counts_tool_result_tokens_before_replacing_them() -> None: + payload = build_payload(messages=TOOL_CONVERSATION) + payload["standard_logging_object"]["model"] = "claude-sonnet-5" + + result = _redacted_span_as_the_proxy_builds_it(payload) + + expected_tokens = litellm.token_counter(model="claude-sonnet-5", text=SECRET_TOOL_RESULT) + assert expected_tokens > 0 + assert result["metrics"]["tool_output_tokens"] == float(expected_tokens) + assert result["metrics"]["input_tokens"] == 4447.0 + + +def test_tool_output_tokens_sum_every_result_in_the_request(logger: DataDogLLMObsLogger) -> None: + payload = build( + logger, + messages=[ + {"role": "tool", "tool_call_id": "call_1", "content": "one two three"}, + {"role": "tool", "tool_call_id": "call_2", "content": "four five six seven"}, + ], + ) + + assert payload["metrics"]["tool_output_tokens"] == float( + litellm.token_counter(text="one two three") + litellm.token_counter(text="four five six seven") + ) + + +def test_a_request_without_tool_results_reports_no_tool_output_tokens(logger: DataDogLLMObsLogger) -> None: + payload = build(logger, messages=[{"role": "user", "content": "hi"}]) + + assert "tool_output_tokens" not in payload["metrics"] + assert "tool_output_tokens" not in _redacted_span_as_the_proxy_builds_it(build_payload())["metrics"] + + +def test_a_tool_that_returned_nothing_still_counts_as_zero_tool_output_tokens(logger: DataDogLLMObsLogger) -> None: + payload = build(logger, messages=[{"role": "tool", "tool_call_id": "call_1", "content": ""}]) + + assert payload["metrics"]["tool_output_tokens"] == 0.0 + + +def test_redaction_keeps_anthropic_tool_blocks_as_structure_only() -> None: + result = _redacted_span_as_the_proxy_builds_it( build_payload( messages=[ - {"role": "user", "content": "secret prompt"}, - {"role": "assistant", "content": None, "tool_calls": [ASSISTANT_TOOL_CALL]}, - ], - response_message={"role": "assistant", "content": "secret response"}, - ), + { + "role": "assistant", + "content": [ + {"type": "tool_use", "id": "toolu_1", "name": "get_weather", "input": {"city": "Paris"}} + ], + }, + { + "role": "user", + "content": [{"type": "tool_result", "tool_use_id": "toolu_1", "content": SECRET_TOOL_RESULT}], + }, + ] + ) ) assert result["meta"]["input"]["messages"] == [ - {"role": "user", "content": "redacted-by-litellm"}, - {"role": "assistant", "content": "redacted-by-litellm"}, + { + "role": "assistant", + "content": "redacted-by-litellm", + "tool_calls": [ + {"name": "get_weather", "arguments": "redacted-by-litellm", "tool_id": "toolu_1", "type": "tool_use"} + ], + }, + { + "role": "user", + "content": "redacted-by-litellm", + "tool_results": [ + {"name": "get_weather", "result": "redacted-by-litellm", "tool_id": "toolu_1", "type": "function"} + ], + }, ] - assert result["meta"]["output"]["messages"] == [{"role": "assistant", "content": "redacted-by-litellm"}] + assert "Paris" not in safe_dumps(result) + + +def test_redaction_blanks_tool_identifiers_that_are_not_strings() -> None: + result = _redacted_span_as_the_proxy_builds_it( + build_payload( + messages=[ + { + "role": "assistant", + "content": None, + "tool_calls": [ + {"id": {"leak": "SECRET-7545"}, "type": ["SECRET-7545"], "function": {"name": ["SECRET-7545"]}} + ], + } + ] + ) + ) + + assert result["meta"]["input"]["messages"][0]["tool_calls"] == [ + {"name": "", "arguments": "redacted-by-litellm", "tool_id": "", "type": ""} + ] + assert "SECRET-7545" not in safe_dumps(result) + + +def test_the_shared_hook_still_strips_what_redaction_governs_besides_messages() -> None: + payload = build_payload(messages=TOOL_CONVERSATION) + payload["standard_logging_object"]["classifier_input"] = {"system": "SECRET-7545"} + logger_under_test = _redacting_logger(turn_off_message_logging=True) + + with patch.object( # test-quality-ok: the hook reads this module global with no injection seam + litellm, "standard_logging_payload_excluded_fields", ["response"] + ): + redacted = logger_under_test.redact_standard_logging_payload_from_model_call_details(payload) + + assert "classifier_input" not in redacted["standard_logging_object"] + assert "response" not in redacted["standard_logging_object"] + assert redacted["standard_logging_object"]["messages"] == TOOL_CONVERSATION + assert payload["standard_logging_object"]["classifier_input"] == {"system": "SECRET-7545"} def test_redaction_drops_unrecognized_and_malformed_message_roles() -> None: diff --git a/tests/test_litellm/litellm_core_utils/test_redact_messages.py b/tests/test_litellm/litellm_core_utils/test_redact_messages.py index 76092e96307..584a3ac471c 100644 --- a/tests/test_litellm/litellm_core_utils/test_redact_messages.py +++ b/tests/test_litellm/litellm_core_utils/test_redact_messages.py @@ -926,3 +926,38 @@ def test_classifier_callback_redaction_preserves_exclusions(monkeypatch: pytest. assert failure_payload["response"]["choices"][0]["message"]["content"] == "redacted-by-litellm" assert payload["classifier_input"] == {"system": "private rubric"} assert payload["response"]["choices"][0]["message"]["content"] == "private answer" + + +class _SelfRedactingLogger(CustomLogger): + def redacts_messages_itself(self) -> bool: + return True + + +@pytest.mark.parametrize("logger", [CustomLogger(), _SelfRedactingLogger()], ids=["default", "redacts_itself"]) +def test_field_exclusion_alone_leaves_messages_and_responses_intact(monkeypatch: pytest.MonkeyPatch, logger: CustomLogger) -> None: + monkeypatch.setattr(litellm, "standard_logging_payload_excluded_fields", ["model"]) + payload: Final = { + "messages": [{"role": "user", "content": "private prompt"}], + "response": {"choices": [{"message": {"content": "private answer"}}]}, + "model": "classifier", + } + stored: Final = logger.redact_standard_logging_payload_from_model_call_details({"standard_logging_object": payload})[ + "standard_logging_object" + ] + assert stored == {"messages": payload["messages"], "response": payload["response"]} + + +def test_a_callback_that_redacts_itself_keeps_its_messages_but_not_the_classifier_audit() -> None: + payload: Final = { + "classifier_input": {"system": "private rubric"}, + "messages": [{"role": "user", "content": "private prompt"}], + "response": {"choices": [{"message": {"content": "private answer"}}]}, + } + logger: Final = _SelfRedactingLogger() + logger.turn_off_message_logging = True + stored: Final = logger.redact_standard_logging_payload_from_model_call_details({"standard_logging_object": payload})[ + "standard_logging_object" + ] + assert "classifier_input" not in stored + assert stored["messages"] == payload["messages"] + assert stored["response"] == payload["response"] From dbc57c13d4311105561f4153e11536c8ce0d1ab1 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 13:37:50 -0700 Subject: [PATCH 102/157] fix(cost-map): match Bedrock GPT effort flags to what Bedrock accepts Live calls to Bedrock Mantle and Converse on 2026-09-11: the gpt-5.6 luna, sol, and terra rows and gpt-6-astra return 200 on reasoning_effort=max, gpt-6-astra returns 400 on none, and Mantle gpt-5.4 and gpt-5.5 return 200 on minimal. The commercial Bedrock rows now carry exactly those flags, and the schema test asserts the measured ladder per row instead of a blanket mirror of the direct OpenAI rows --- ...odel_prices_and_context_window_backup.json | 17 ++++++- model_prices_and_context_window.json | 17 ++++++- .../test_litellm/test_model_prices_schema.py | 49 ++++++++++++++----- 3 files changed, 67 insertions(+), 16 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a55cb6b571e..be4badedee4 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -55360,6 +55360,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55401,6 +55402,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55507,6 +55509,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55538,6 +55541,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55566,6 +55570,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55594,6 +55599,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55622,6 +55628,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55650,6 +55657,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55678,6 +55686,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55711,7 +55720,9 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, + "supports_none_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55742,7 +55753,9 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, + "supports_none_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55772,7 +55785,9 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, + "supports_none_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55809,7 +55824,6 @@ "text" ], "supports_function_calling": true, - "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55847,7 +55861,6 @@ "text" ], "supports_function_calling": true, - "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a55cb6b571e..be4badedee4 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -55360,6 +55360,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55401,6 +55402,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55507,6 +55509,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55538,6 +55541,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55566,6 +55570,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55594,6 +55599,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55622,6 +55628,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55650,6 +55657,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55678,6 +55686,7 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, "supports_tool_choice": true, "supports_reasoning": true, @@ -55711,7 +55720,9 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, + "supports_none_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55742,7 +55753,9 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, + "supports_none_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55772,7 +55785,9 @@ "text" ], "supports_function_calling": true, + "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, + "supports_none_reasoning_effort": false, "supports_tool_choice": true, "supports_prompt_caching": true, "supports_reasoning": true, @@ -55809,7 +55824,6 @@ "text" ], "supports_function_calling": true, - "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55847,7 +55861,6 @@ "text" ], "supports_function_calling": true, - "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/tests/test_litellm/test_model_prices_schema.py b/tests/test_litellm/test_model_prices_schema.py index 69ee999f066..fcdb89003ac 100644 --- a/tests/test_litellm/test_model_prices_schema.py +++ b/tests/test_litellm/test_model_prices_schema.py @@ -4,6 +4,7 @@ import importlib.util import json import re from pathlib import Path +from types import MappingProxyType from typing import Final import jsonschema @@ -220,23 +221,47 @@ def test_chat_latest_declares_the_one_effort_openai_accepts(prices: dict): assert resolve_supported_reasoning_efforts(prices["chat-latest"], deployment_is_mapped=True) == ("medium",) -BEDROCK_OPENAI_XHIGH_MARKERS: Final = ("openai.gpt-5.4", "openai.gpt-5.5", "openai.gpt-5.6", "openai.gpt-6-astra") +BEDROCK_OPENAI_GPT_MARKERS: Final = ("openai.gpt-5.4", "openai.gpt-5.5", "openai.gpt-5.6", "openai.gpt-6-astra") BEDROCK_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse", "bedrock_mantle")) +BEDROCK_ROW_PREFIXES: Final = ("bedrock_mantle/", "us.", "global.") +GPT_5_4_BEDROCK_LADDER: Final = ("none", "minimal", "low", "medium", "high", "xhigh") +GPT_5_6_BEDROCK_LADDER: Final = ("none", "low", "medium", "high", "xhigh", "max") +GPT_6_ASTRA_BEDROCK_LADDER: Final = ("low", "medium", "high", "xhigh", "max") +BEDROCK_OPENAI_GPT_LADDERS: Final = MappingProxyType( + { + "bedrock_mantle/openai.gpt-5.4": GPT_5_4_BEDROCK_LADDER, + "bedrock_mantle/openai.gpt-5.5": GPT_5_4_BEDROCK_LADDER, + **{ + f"{prefix}openai.gpt-5.6-{variant}": GPT_5_6_BEDROCK_LADDER + for prefix in BEDROCK_ROW_PREFIXES + for variant in ("luna", "sol", "terra") + }, + **{f"{prefix}openai.gpt-6-astra": GPT_6_ASTRA_BEDROCK_LADDER for prefix in BEDROCK_ROW_PREFIXES}, + } +) -def test_bedrock_openai_gpt_rows_mirror_their_openai_twins_effort_ladder(prices: dict): - """Bedrock forwards reasoning_effort to these models unchanged, and xhigh is opt-in for the - capability resolver, so a row without the flag drops xhigh from every group it belongs to. - The OpenAI twins reject minimal, and Bedrock forwards reasoning_effort unchanged.""" - mismatched = [ +@pytest.mark.parametrize( + ("name", "ladder"), tuple(BEDROCK_OPENAI_GPT_LADDERS.items()), ids=tuple(BEDROCK_OPENAI_GPT_LADDERS) +) +def test_bedrock_openai_gpt_rows_advertise_the_ladder_bedrock_accepts(prices: dict, name: str, ladder: tuple[str, ...]): + """Each ladder is the set of levels the Bedrock Mantle and Converse endpoints answered 200 to + for that row on 2026-09-11 (PR #40740), which differs from the direct OpenAI rows in three + places: Bedrock gpt-5.6 and gpt-6-astra take max, Bedrock gpt-5.4 and gpt-5.5 take minimal, and + gpt-6-astra refuses none. xhigh and max are opt-in for the resolver, so a row missing either + flag silently drops that level from every group it belongs to.""" + assert resolve_supported_reasoning_efforts(prices[name], deployment_is_mapped=True) == ladder + + +def test_every_bedrock_openai_gpt_row_advertises_xhigh(prices: dict): + """The GovCloud and gpt-5.6-cyber rows cannot be called from our account, so they carry the + family's xhigh flag rather than a measured ladder.""" + missing: Final = [ name for name, entry in prices.items() if isinstance(entry, dict) and entry.get("litellm_provider") in BEDROCK_PROVIDERS - and any(marker in name for marker in BEDROCK_OPENAI_XHIGH_MARKERS) - and ( - "xhigh" not in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) - or "minimal" in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) - ) + and any(marker in name for marker in BEDROCK_OPENAI_GPT_MARKERS) + and "xhigh" not in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ()) ] - assert mismatched == [] + assert missing == [] From cc1d2c66c835df8ee25fb02031ed415c2e936634 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 13:38:01 -0700 Subject: [PATCH 103/157] test(e2e): bound chaos latency and log volume with flat ceilings A ratio against the healthy phase cannot bound either metric. Once the Redis circuit breaker opens, a request skips Redis instead of waiting on its socket timeout, so the chaos phase can measure cheaper than the baseline it is compared against: local runs came in at 0.61x baseline p90 while a log-bytes ratio read 724x. Splitting Budget into RatioBudget and AbsoluteBudget lets RSS and CPU keep the ratio they need, since both are machine-shaped, while latency and log volume get the wall-clock ceiling a user actually cares about. Co-Authored-By: Claude Code --- tests/e2e/CLAUDE.md | 2 +- tests/e2e/load/phase_budget.py | 55 +++++++++++++----- tests/e2e/load/test_phase_budget.py | 53 +++++++++++++++-- tests/e2e/load/test_redis_chaos_e2e.py | 79 ++++++++++++-------------- 4 files changed, 126 insertions(+), 63 deletions(-) diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index ed57dd71759..419cfd6dedd 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint and budgeting p50/p90/p99 latency, RSS, CPU-per-request, and log-bytes-per-request as ratios against the same run's healthy phase; needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/load/phase_budget.py b/tests/e2e/load/phase_budget.py index 7d354a678bc..066e2579da8 100644 --- a/tests/e2e/load/phase_budget.py +++ b/tests/e2e/load/phase_budget.py @@ -1,18 +1,27 @@ """Comparing one load phase against another, for tests that degrade a dependency mid-run. -A chaos phase's absolute numbers say very little on their own: RSS scales with worker count, -latency with core count, so a ceiling calibrated on one machine is meaningless on the next. -What travels is the ratio against a healthy phase measured on the same machine in the same run. +Two shapes of ceiling, because the metrics divide into two kinds. RSS and CPU are +machine-shaped: RSS scales with worker count and CPU with core count, so an absolute number +calibrated on one runner means nothing on the next, and what travels is the ratio against a +healthy phase measured on the same machine in the same run. Latency and log volume are not: +a ratio there is actively misleading, because a dependency that fails fast once its breaker +opens can make the degraded phase look cheaper than the healthy one while still being far +slower or noisier than a user should ever see. Those get a flat ceiling, which is the promise +the test is actually making. """ from __future__ import annotations from dataclasses import dataclass -from typing import Final +from typing import Final, TypeAlias + + +def _rendered(value: float, unit: str, decimals: int) -> str: + return f"{value:.{decimals}f}{unit}" @dataclass(frozen=True, slots=True) -class Budget: +class RatioBudget: """One metric's healthy value, its degraded value, and how much growth is allowed.""" name: str @@ -27,26 +36,46 @@ class Budget: """How many times the baseline the degraded value is, or None if there is no baseline.""" return self.degraded / self.baseline if self.baseline > 0 else None - def _rendered(self, value: float) -> str: - return f"{value:.{self.decimals}f}{self.unit}" - def violation(self) -> str | None: """Why this metric fails its budget, or None if it passes.""" ratio: Final = self.ratio if ratio is None: return ( - f"{self.name} measured {self._rendered(self.baseline)} in the healthy phase, so there is nothing " - f"to compare the degraded phase against; the measurement did not happen" + f"{self.name} measured {_rendered(self.baseline, self.unit, self.decimals)} in the healthy phase, " + f"so there is nothing to compare the degraded phase against; the measurement did not happen" ) if ratio > self.ratio_ceiling: return ( - f"{self.name} went from {self._rendered(self.baseline)} healthy to " - f"{self._rendered(self.degraded)} degraded, {ratio:.1f}x the baseline and past the " - f"{self.ratio_ceiling:.1f}x allowed" + f"{self.name} went from {_rendered(self.baseline, self.unit, self.decimals)} healthy to " + f"{_rendered(self.degraded, self.unit, self.decimals)} degraded, {ratio:.1f}x the baseline and past " + f"the {self.ratio_ceiling:.1f}x allowed" ) return None +@dataclass(frozen=True, slots=True) +class AbsoluteBudget: + """One metric's degraded value against a flat ceiling, for metrics a ratio cannot bound.""" + + name: str + measured: float + ceiling: float + unit: str + decimals: int = 1 + + def violation(self) -> str | None: + """Why this metric fails its budget, or None if it passes.""" + if self.measured > self.ceiling: + return ( + f"{self.name} measured {_rendered(self.measured, self.unit, self.decimals)} in the degraded phase, " + f"past the {_rendered(self.ceiling, self.unit, self.decimals)} allowed" + ) + return None + + +Budget: TypeAlias = RatioBudget | AbsoluteBudget + + def violations(budgets: tuple[Budget, ...]) -> tuple[str, ...]: """Every budget the run blew, so one failure reports all of them instead of the first.""" return tuple(violation for budget in budgets if (violation := budget.violation()) is not None) diff --git a/tests/e2e/load/test_phase_budget.py b/tests/e2e/load/test_phase_budget.py index 2355fb7cf4c..ea9e56afb0d 100644 --- a/tests/e2e/load/test_phase_budget.py +++ b/tests/e2e/load/test_phase_budget.py @@ -2,14 +2,16 @@ from __future__ import annotations from typing import Final -from phase_budget import Budget, violations +from phase_budget import AbsoluteBudget, RatioBudget, violations -def _budget(*, baseline: float, degraded: float, ceiling: float = 2.0) -> Budget: - return Budget(name="p99 RSS", baseline=baseline, degraded=degraded, ratio_ceiling=ceiling, unit=" MB", decimals=0) +def _budget(*, baseline: float, degraded: float, ceiling: float = 2.0) -> RatioBudget: + return RatioBudget( + name="p99 RSS", baseline=baseline, degraded=degraded, ratio_ceiling=ceiling, unit=" MB", decimals=0 + ) -class TestBudget: +class TestRatioBudget: def test_growth_within_the_ceiling_is_not_a_violation(self) -> None: assert _budget(baseline=100, degraded=199).violation() is None @@ -37,7 +39,7 @@ class TestBudget: assert "nothing to compare" in violation def test_the_unit_and_decimals_carry_into_the_message(self) -> None: - violation: Final = Budget( + violation: Final = RatioBudget( name="p99 latency", baseline=0.16, degraded=9.5, ratio_ceiling=8.0, unit="s", decimals=3 ).violation() @@ -46,13 +48,40 @@ class TestBudget: assert "9.500s" in violation +class TestAbsoluteBudget: + def test_a_value_under_the_ceiling_is_not_a_violation(self) -> None: + assert AbsoluteBudget(name="p99 latency", measured=1.2, ceiling=5.0, unit="s", decimals=3).violation() is None + + def test_a_value_exactly_at_the_ceiling_is_allowed(self) -> None: + assert AbsoluteBudget(name="p99 latency", measured=5.0, ceiling=5.0, unit="s", decimals=3).violation() is None + + def test_a_value_past_the_ceiling_reports_the_measurement_and_the_ceiling(self) -> None: + violation: Final = AbsoluteBudget( + name="p99 latency", measured=9.5, ceiling=5.0, unit="s", decimals=3 + ).violation() + + assert violation is not None + assert "9.500s" in violation + assert "5.000s allowed" in violation + + def test_a_flat_ceiling_fails_a_degraded_phase_that_is_cheaper_than_its_baseline(self) -> None: + # The whole reason this shape exists: once the breaker opens, requests skip Redis instead + # of waiting on its socket timeout, so the chaos phase can measure faster than the healthy + # one. A ratio against that baseline passes; the user still waited 9.5s. + assert _budget(baseline=20.0, degraded=9.5, ceiling=2.0).violation() is None + assert AbsoluteBudget(name="p99 latency", measured=9.5, ceiling=5.0, unit="s").violation() is not None + + def test_a_zero_measurement_is_not_a_violation(self) -> None: + assert AbsoluteBudget(name="log bytes per request", measured=0, ceiling=12_000, unit=" B").violation() is None + + class TestViolations: def test_every_blown_budget_is_reported_not_just_the_first(self) -> None: blown: Final = violations( ( _budget(baseline=100, degraded=500), _budget(baseline=100, degraded=120), - Budget(name="CPU per request", baseline=10, degraded=90, ratio_ceiling=6.0, unit=" ms"), + RatioBudget(name="CPU per request", baseline=10, degraded=90, ratio_ceiling=6.0, unit=" ms"), ) ) @@ -60,5 +89,17 @@ class TestViolations: assert blown[0].startswith("p99 RSS") assert blown[1].startswith("CPU per request") + def test_both_budget_shapes_report_together(self) -> None: + blown: Final = violations( + ( + _budget(baseline=100, degraded=500), + AbsoluteBudget(name="p99 latency", measured=9.5, ceiling=5.0, unit="s", decimals=3), + ) + ) + + assert len(blown) == 2 + assert blown[0].startswith("p99 RSS") + assert blown[1].startswith("p99 latency") + def test_a_run_inside_every_budget_reports_nothing(self) -> None: assert violations((_budget(baseline=100, degraded=150),)) == () diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 153be656566..94c879d29cb 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -50,7 +50,7 @@ from lifecycle import ResourceManager from load_client import LoadClient from locust_load import LoadResult, run_gateway_load from models import KeyGenerateBody, LiteLLMParamsBody -from phase_budget import Budget, violations +from phase_budget import AbsoluteBudget, Budget, RatioBudget, violations from proxy_client import ProxyClient from proxy_usage import ProxyUsageSampler, UsageWindow @@ -70,23 +70,25 @@ BASELINE_SECONDS: Final = 60.0 CHAOS_SECONDS: Final = 90.0 REDIS_PAUSE_MS: Final = int(CHAOS_SECONDS * 1000) -# Chaos-phase ceilings, as a multiple of the same metric in the baseline phase. Ratios rather -# than absolutes because every absolute here is machine-shaped: RSS scales with worker count -# and latency with core count, so a number calibrated on one runner means nothing on another. -# Calibrated from local runs under CLIENT PAUSE ALL that came in around 4x latency at every -# percentile, 1.03x RSS and 4.4x CPU per request, and deliberately loose: the regression these -# guard against grew memory by an order of magnitude, so catching it does not need a tight -# bound, and a tight one would flake on a shared CI runner. Latency gets the most slack because -# it is the metric a Redis outage is legitimately allowed to move, by its socket timeout on -# every call a request attempts. -CHAOS_LATENCY_RATIO_CEILING: Final = 12.0 +# RSS and CPU are budgeted as a multiple of the same metric in the baseline phase, because both +# are machine-shaped: RSS scales with worker count and CPU with core count, so a number +# calibrated on one runner means nothing on another. Calibrated from local runs under CLIENT +# PAUSE ALL that came in around 1.03x RSS and 4.4x CPU per request, and deliberately loose: the +# regression these guard against grew memory by an order of magnitude, so catching it does not +# need a tight bound, and a tight one would flake on a shared CI runner. CHAOS_RSS_RATIO_CEILING: Final = 1.5 CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 6.0 -# Uncalibrated: no chaos run has measured this yet, since JSON_LOGS and the padded payload -# landed after the last run this file's other ceilings were calibrated from. Deliberately loose -# until a real run tightens it; the failed-tracking alert body that motivates this test already -# logs the full request metadata per timeout, so a JSON-encoded traceback storm should dwarf this. -CHAOS_LOG_BYTES_PER_REQUEST_RATIO_CEILING: Final = 20.0 + +# Latency and log volume get flat ceilings instead, because a ratio cannot bound either one. Once +# the breaker opens, a request skips Redis rather than waiting on its socket timeout, so the chaos +# phase can come in faster than baseline (local runs measured p90 at 0.61x) and a ratio passes on a +# phase that was never slow. What a user actually cares about is the wall-clock number, which these +# hold directly. Calibrated from local runs whose worst chaos phase was p50 0.66s, p90 0.74s, p99 +# 1.20s and 3.4 KB of log per request, then left roughly 3x loose for a shared CI runner. +CHAOS_P50_LATENCY_CEILING_SECONDS: Final = 2.0 +CHAOS_P90_LATENCY_CEILING_SECONDS: Final = 3.0 +CHAOS_P99_LATENCY_CEILING_SECONDS: Final = 5.0 +CHAOS_LOG_BYTES_PER_REQUEST_CEILING: Final = 12_000.0 DRAIN_TIMEOUT_SECONDS: Final = 30.0 DRAIN_POLL_SECONDS: Final = 1.0 @@ -289,19 +291,12 @@ def _drive(keys: tuple[str, ...], seconds: float) -> LoadResult: ) -def _latency_budget(percentile: str, baseline: float, degraded: float) -> Budget: - return Budget( - name=f"{percentile} latency", - baseline=baseline, - degraded=degraded, - ratio_ceiling=CHAOS_LATENCY_RATIO_CEILING, - unit="s", - decimals=3, - ) +def _latency_budget(percentile: str, measured: float, ceiling: float) -> Budget: + return AbsoluteBudget(name=f"{percentile} latency", measured=measured, ceiling=ceiling, unit="s", decimals=3) def _rss_budget(percentile: str, baseline: UsageWindow, degraded: UsageWindow, fraction: float) -> Budget: - return Budget( + return RatioBudget( name=f"{percentile} RSS", baseline=baseline.rss_percentile(fraction) / 2**20, degraded=degraded.rss_percentile(fraction) / 2**20, @@ -312,18 +307,18 @@ def _rss_budget(percentile: str, baseline: UsageWindow, degraded: UsageWindow, f def _chaos_budgets(baseline: Phase, chaos: Phase) -> tuple[Budget, ...]: - """What a Redis outage is allowed to cost, measured against the same run's healthy phase. + """What a Redis outage is allowed to cost. Every request still succeeding is the headline assertion, but a proxy can answer every request while leaking: the v1.100.0 regression (LIT-6780) served traffic the whole way up - to a 61 GB worker. These bound the cost of serving it. + to a 61 GB worker. These bound the cost of serving it. RSS and CPU are bounded against the + same run's healthy phase, latency and log bytes against a flat ceiling; see phase_budget + for why the two kinds of metric cannot share one shape. Latency and RSS are budgeted at p50, p90 and p99 so a regression that only shows up in the - tail (or only in the median) cannot hide behind the other. Latency gets the loosest bound - because a timing-out Redis legitimately adds its socket_timeout to every request that - touches it, several times over on a retried request. RSS gets the tightest: the failure - path has no business allocating more per request. CPU and log bytes are each budgeted once, - as an amount per request rather than per percentile: cores-busy saturates at the worker + tail (or only in the median) cannot hide behind the other. RSS gets the tightest bound: the + failure path has no business allocating more per request. CPU and log bytes are each budgeted + once, as an amount per request rather than per percentile: cores-busy saturates at the worker count under load, so its percentiles read the same whether a request costs 10 ms of CPU or 40, and cannot budget anything; per-request is the figure that actually moves. Log bytes isolates the cost of the failed-tracking alert's own noisy error handling from the CPU it @@ -331,24 +326,23 @@ def _chaos_budgets(baseline: Phase, chaos: Phase) -> tuple[Budget, ...]: CPU number. """ return ( - _latency_budget("p50", baseline.load.p50_seconds, chaos.load.p50_seconds), - _latency_budget("p90", baseline.load.p90_seconds, chaos.load.p90_seconds), - _latency_budget("p99", baseline.load.p99_seconds, chaos.load.p99_seconds), + _latency_budget("p50", chaos.load.p50_seconds, CHAOS_P50_LATENCY_CEILING_SECONDS), + _latency_budget("p90", chaos.load.p90_seconds, CHAOS_P90_LATENCY_CEILING_SECONDS), + _latency_budget("p99", chaos.load.p99_seconds, CHAOS_P99_LATENCY_CEILING_SECONDS), _rss_budget("p50", baseline.usage, chaos.usage, 0.5), _rss_budget("p90", baseline.usage, chaos.usage, 0.9), _rss_budget("p99", baseline.usage, chaos.usage, 0.99), - Budget( + RatioBudget( name="CPU per request", baseline=baseline.cpu_seconds_per_request * 1000, degraded=chaos.cpu_seconds_per_request * 1000, ratio_ceiling=CHAOS_CPU_PER_REQUEST_RATIO_CEILING, unit=" ms", ), - Budget( + AbsoluteBudget( name="log bytes per request", - baseline=baseline.log_bytes_per_request, - degraded=chaos.log_bytes_per_request, - ratio_ceiling=CHAOS_LOG_BYTES_PER_REQUEST_RATIO_CEILING, + measured=chaos.log_bytes_per_request, + ceiling=CHAOS_LOG_BYTES_PER_REQUEST_CEILING, unit=" B", decimals=0, ), @@ -447,8 +441,7 @@ class TestRedisChaos: blown: Final = violations(_chaos_budgets(baseline, chaos)) assert not blown, ( - f"pausing Redis cost the proxy more than the socket timeout on the calls it attempts: " - f"{'; '.join(blown)}. {report}" + f"pausing Redis cost the proxy more than a Redis outage is allowed to: {'; '.join(blown)}. {report}" ) rows: Final = proxy.poll_logs_for_key(keys[0], min_rows=1) From 1d71006567311f6608688873db5d670e8fae21af Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 13:43:18 -0700 Subject: [PATCH 104/157] test(e2e): tighten chaos RSS and CPU ceilings to what runs actually measured RSS moved 0.91x-1.40x across three identical local runs, so it stays loose at 2x rather than the arbitrary 1.5x carried over from the pre-padding-payload calibration. CPU per request held steady at 1.33x-1.36x across the same runs, so 4x replaces the looser 6x it inherited from stale numbers. Co-Authored-By: Claude Code --- tests/e2e/load/test_redis_chaos_e2e.py | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 94c879d29cb..c1da41aeb16 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -72,12 +72,11 @@ REDIS_PAUSE_MS: Final = int(CHAOS_SECONDS * 1000) # RSS and CPU are budgeted as a multiple of the same metric in the baseline phase, because both # are machine-shaped: RSS scales with worker count and CPU with core count, so a number -# calibrated on one runner means nothing on another. Calibrated from local runs under CLIENT -# PAUSE ALL that came in around 1.03x RSS and 4.4x CPU per request, and deliberately loose: the -# regression these guard against grew memory by an order of magnitude, so catching it does not -# need a tight bound, and a tight one would flake on a shared CI runner. -CHAOS_RSS_RATIO_CEILING: Final = 1.5 -CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 6.0 +# calibrated on one runner means nothing on another. RSS moved 0.91x-1.40x across three otherwise +# identical local runs, so it stays loose; CPU per request held steady at 1.33x-1.36x across the +# same runs, so it can sit closer to what is actually measured. +CHAOS_RSS_RATIO_CEILING: Final = 2.0 +CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 4.0 # Latency and log volume get flat ceilings instead, because a ratio cannot bound either one. Once # the breaker opens, a request skips Redis rather than waiting on its socket timeout, so the chaos From e81a7887762fff8c4e9b31d38bd77a9c4546a2a4 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 13:43:53 -0700 Subject: [PATCH 105/157] test(e2e): hold chaos CPU per request to 2x Three local runs measured 1.33x-1.36x, so 2x is the tightest bound the data supports and still catches a regression far smaller than 4x would. Noted in the comment that this is the ceiling to loosen first if a weekly run trips it, since core count shifts how much of baseline CPU is fixed per-request work. Co-Authored-By: Claude Code --- tests/e2e/load/test_redis_chaos_e2e.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index c1da41aeb16..c800ccb0e51 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -74,9 +74,11 @@ REDIS_PAUSE_MS: Final = int(CHAOS_SECONDS * 1000) # are machine-shaped: RSS scales with worker count and CPU with core count, so a number # calibrated on one runner means nothing on another. RSS moved 0.91x-1.40x across three otherwise # identical local runs, so it stays loose; CPU per request held steady at 1.33x-1.36x across the -# same runs, so it can sit closer to what is actually measured. +# same runs, so it sits close to what is actually measured. That makes CPU the likeliest of these +# to flake first on a runner whose core count shifts how much of baseline CPU is fixed per-request +# work: loosen it rather than widening the others if a weekly run trips it without a real cause. CHAOS_RSS_RATIO_CEILING: Final = 2.0 -CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 4.0 +CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 2.0 # Latency and log volume get flat ceilings instead, because a ratio cannot bound either one. Once # the breaker opens, a request skips Redis rather than waiting on its socket timeout, so the chaos From a1f9f4cbe839a8bd4bcf605bbcf44effddd3724f Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 13:45:11 -0700 Subject: [PATCH 106/157] test(e2e): tighten chaos latency ceilings to 1s/2s/3s Local runs measured p50 0.19s, p90 0.23s, p99 0.69s, so 2s/3s/5s left several times that as slack. 1s/2s/3s keeps a comfortable margin while catching a smaller regression than the looser ceilings would have. Co-Authored-By: Claude Code --- tests/e2e/load/test_redis_chaos_e2e.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index c800ccb0e51..518bbd262a9 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -84,11 +84,11 @@ CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 2.0 # the breaker opens, a request skips Redis rather than waiting on its socket timeout, so the chaos # phase can come in faster than baseline (local runs measured p90 at 0.61x) and a ratio passes on a # phase that was never slow. What a user actually cares about is the wall-clock number, which these -# hold directly. Calibrated from local runs whose worst chaos phase was p50 0.66s, p90 0.74s, p99 -# 1.20s and 3.4 KB of log per request, then left roughly 3x loose for a shared CI runner. -CHAOS_P50_LATENCY_CEILING_SECONDS: Final = 2.0 -CHAOS_P90_LATENCY_CEILING_SECONDS: Final = 3.0 -CHAOS_P99_LATENCY_CEILING_SECONDS: Final = 5.0 +# hold directly. Calibrated from local runs whose worst chaos phase was p50 0.19s, p90 0.23s, p99 +# 0.69s and 3.5 KB of log per request, with several times that left as slack for a shared CI runner. +CHAOS_P50_LATENCY_CEILING_SECONDS: Final = 1.0 +CHAOS_P90_LATENCY_CEILING_SECONDS: Final = 2.0 +CHAOS_P99_LATENCY_CEILING_SECONDS: Final = 3.0 CHAOS_LOG_BYTES_PER_REQUEST_CEILING: Final = 12_000.0 DRAIN_TIMEOUT_SECONDS: Final = 30.0 From 0ea18f28af5d2320fa06aac960d06ce9f6898a4f Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 13:45:42 -0700 Subject: [PATCH 107/157] test(e2e): tighten chaos log-bytes ceiling to 10 KB per request Local runs measured 3.5 KB per request, so 10 KB keeps close to 3x headroom while tightening from the earlier 12 KB. Co-Authored-By: Claude Code --- tests/e2e/load/test_redis_chaos_e2e.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 518bbd262a9..0732f725f34 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -89,7 +89,7 @@ CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 2.0 CHAOS_P50_LATENCY_CEILING_SECONDS: Final = 1.0 CHAOS_P90_LATENCY_CEILING_SECONDS: Final = 2.0 CHAOS_P99_LATENCY_CEILING_SECONDS: Final = 3.0 -CHAOS_LOG_BYTES_PER_REQUEST_CEILING: Final = 12_000.0 +CHAOS_LOG_BYTES_PER_REQUEST_CEILING: Final = 10_000.0 DRAIN_TIMEOUT_SECONDS: Final = 30.0 DRAIN_POLL_SECONDS: Final = 1.0 From 71f45683d73d741db4e8c0801045b75296ee7b54 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 13:46:20 -0700 Subject: [PATCH 108/157] fix(cost-map): keep minimal withheld on Bedrock gpt-5.4 and gpt-5.5 LiteLLM sends the Bedrock Mantle GPT rows through Bedrock's Responses endpoint, which refuses minimal on gpt-5.4 and gpt-5.5 like every other Bedrock GPT row. The earlier commit measured the raw chat endpoint, which accepts it, and dropped the flag by mistake. The ladder test now matches what the proxy path returns --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ tests/test_litellm/test_model_prices_schema.py | 13 +++++++------ 3 files changed, 11 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index be4badedee4..aadd9bd3028 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -55824,6 +55824,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55861,6 +55862,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index be4badedee4..aadd9bd3028 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -55824,6 +55824,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, @@ -55861,6 +55862,7 @@ "text" ], "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/tests/test_litellm/test_model_prices_schema.py b/tests/test_litellm/test_model_prices_schema.py index fcdb89003ac..0b9dbd23097 100644 --- a/tests/test_litellm/test_model_prices_schema.py +++ b/tests/test_litellm/test_model_prices_schema.py @@ -224,7 +224,7 @@ def test_chat_latest_declares_the_one_effort_openai_accepts(prices: dict): BEDROCK_OPENAI_GPT_MARKERS: Final = ("openai.gpt-5.4", "openai.gpt-5.5", "openai.gpt-5.6", "openai.gpt-6-astra") BEDROCK_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse", "bedrock_mantle")) BEDROCK_ROW_PREFIXES: Final = ("bedrock_mantle/", "us.", "global.") -GPT_5_4_BEDROCK_LADDER: Final = ("none", "minimal", "low", "medium", "high", "xhigh") +GPT_5_4_BEDROCK_LADDER: Final = ("none", "low", "medium", "high", "xhigh") GPT_5_6_BEDROCK_LADDER: Final = ("none", "low", "medium", "high", "xhigh", "max") GPT_6_ASTRA_BEDROCK_LADDER: Final = ("low", "medium", "high", "xhigh", "max") BEDROCK_OPENAI_GPT_LADDERS: Final = MappingProxyType( @@ -245,11 +245,12 @@ BEDROCK_OPENAI_GPT_LADDERS: Final = MappingProxyType( ("name", "ladder"), tuple(BEDROCK_OPENAI_GPT_LADDERS.items()), ids=tuple(BEDROCK_OPENAI_GPT_LADDERS) ) def test_bedrock_openai_gpt_rows_advertise_the_ladder_bedrock_accepts(prices: dict, name: str, ladder: tuple[str, ...]): - """Each ladder is the set of levels the Bedrock Mantle and Converse endpoints answered 200 to - for that row on 2026-09-11 (PR #40740), which differs from the direct OpenAI rows in three - places: Bedrock gpt-5.6 and gpt-6-astra take max, Bedrock gpt-5.4 and gpt-5.5 take minimal, and - gpt-6-astra refuses none. xhigh and max are opt-in for the resolver, so a row missing either - flag silently drops that level from every group it belongs to.""" + """Each ladder is the set of levels Bedrock answered 200 to for that row through the proxy on + 2026-09-11 (PR #40740): the Mantle rows go out over its Responses endpoint and the Converse rows + over inference profiles. Bedrock differs from the direct OpenAI rows in two places, gpt-5.6 and + gpt-6-astra take max there, and gpt-6-astra refuses none; minimal is refused on every row. + xhigh and max are opt-in for the resolver, so a row missing either flag silently drops that + level from every group it belongs to.""" assert resolve_supported_reasoning_efforts(prices[name], deployment_is_mapped=True) == ladder From 4423876857443a6b32731713053f9c3491b3e1f3 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 11 Sep 2026 13:56:27 -0700 Subject: [PATCH 109/157] test(responses): fix stale Anthropic smoke request --- .../test_e2e_openai_responses_api.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/openai_endpoints_tests/test_e2e_openai_responses_api.py b/tests/openai_endpoints_tests/test_e2e_openai_responses_api.py index 7e338dafb86..4a392042d63 100644 --- a/tests/openai_endpoints_tests/test_e2e_openai_responses_api.py +++ b/tests/openai_endpoints_tests/test_e2e_openai_responses_api.py @@ -127,14 +127,14 @@ def test_bad_request_bad_param_error(): ) -def test_anthropic_with_responses_api(): - client = get_test_client() - response = client.responses.create( - model="anthropic/claude-sonnet-4-5-20250929", +def test_anthropic_with_responses_api() -> None: + client: Final = get_test_client() + response: Final = client.responses.create( + model="anthropic/claude-sonnet-5", input="just respond with the word 'ping'", - previous_response_id="hi", ) - print("anthropic response=", response) + assert response.status == "completed" + assert response.output_text.strip() def test_cancel_response(): From d51a7af655e2cde00bc8ae25e70efd2555f63db7 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Fri, 11 Sep 2026 14:07:32 -0700 Subject: [PATCH 110/157] fix(search): propagate GET provider HTTP errors (#40779) --- .../llms/base_llm/search/transformation.py | 7 ++ litellm/llms/custom_httpx/llm_http_handler.py | 6 ++ .../llms/tinyfish/search/transformation.py | 10 +- .../custom_httpx/test_llm_http_handler.py | 93 +++++++++++++++++++ 4 files changed, 114 insertions(+), 2 deletions(-) diff --git a/litellm/llms/base_llm/search/transformation.py b/litellm/llms/base_llm/search/transformation.py index 7668c6132d6..9eca3e69909 100644 --- a/litellm/llms/base_llm/search/transformation.py +++ b/litellm/llms/base_llm/search/transformation.py @@ -264,6 +264,13 @@ class BaseSearchConfig: """ raise NotImplementedError("transform_search_response must be implemented by provider") + def get_http_error_class(self, error: httpx.HTTPStatusError) -> Exception: + return self.get_error_class( + error_message=error.response.text, + status_code=error.response.status_code, + headers=dict(error.response.headers), # mutable-ok: provider error factories require dict headers + ) + def get_error_class( self, error_message: str, diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index e720428847d..7109e6942d1 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -1918,6 +1918,7 @@ class BaseLLMHTTPHandler: url=complete_url, headers=signed_headers, ) + response.raise_for_status() else: # A signed body must be sent verbatim, re-serializing it would break the signature response = client.post( @@ -1927,6 +1928,8 @@ class BaseLLMHTTPHandler: json=data if signed_json_body is None else None, timeout=timeout, ) + except httpx.HTTPStatusError as e: + raise provider_config.get_http_error_class(e) except Exception as e: raise self._handle_error(e=e, provider_config=provider_config) @@ -2019,6 +2022,7 @@ class BaseLLMHTTPHandler: url=complete_url, headers=signed_headers, ) + response.raise_for_status() else: # A signed body must be sent verbatim, re-serializing it would break the signature response = await async_httpx_client.post( @@ -2028,6 +2032,8 @@ class BaseLLMHTTPHandler: json=data if signed_json_body is None else None, timeout=timeout, ) + except httpx.HTTPStatusError as e: + raise provider_config.get_http_error_class(e) except Exception as e: raise self._handle_error(e=e, provider_config=provider_config) diff --git a/litellm/llms/tinyfish/search/transformation.py b/litellm/llms/tinyfish/search/transformation.py index b688dc2cd01..460394c6f2d 100644 --- a/litellm/llms/tinyfish/search/transformation.py +++ b/litellm/llms/tinyfish/search/transformation.py @@ -247,6 +247,13 @@ class TinyfishSearchConfig(BaseSearchConfig): hidden["additional_headers"] = process_response_headers(raw_headers) return parsed + def get_http_error_class(self, error: httpx.HTTPStatusError) -> Exception: + return self._wrap_error( + error_message=error.response.text, + status_code=error.response.status_code, + headers=dict(error.response.headers), # mutable-ok: existing error wrapper requires dict headers + ) + def _wrap_error( self, error_message: str, @@ -256,8 +263,7 @@ class TinyfishSearchConfig(BaseSearchConfig): """ Build an attributed ``BaseLLMException`` from a TinyFish error body. - Used only at the call sites we control inside - ``transform_search_response`` (non-2xx, JSONDecodeError, ValidationError). + Used for HTTP status errors and response transformation errors. Not an override of ``BaseSearchConfig.get_error_class``: that path is left to inherit from the base so it auto-picks-up any future LiteLLM improvements. Trade-off: network failures (routed through LiteLLM diff --git a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py index cea2d439198..c39779972c0 100644 --- a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py +++ b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py @@ -3,6 +3,7 @@ import json import logging import threading import time +from typing import Final from unittest.mock import AsyncMock, Mock, patch import httpx @@ -20,7 +21,9 @@ from litellm.llms.base_llm.audio_transcription.transformation import ( BaseAudioTranscriptionConfig, ) from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException +from litellm.llms.base_llm.search.transformation import BaseSearchConfig, SearchResponse from litellm.llms.bedrock.base_aws_llm import SignsRequestsWithAWS +from litellm.llms.brave.search.transformation import BraveSearchConfig from litellm.llms.base_llm.image_edit.transformation import BaseImageEditConfig from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.llms.custom_httpx.llm_http_handler import ( @@ -36,6 +39,7 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran ) from litellm.llms.mistral.ocr.transformation import MistralOCRConfig from litellm.llms.openai.videos.transformation import OpenAIVideoConfig +from litellm.llms.tinyfish.search.transformation import TinyfishSearchConfig from litellm.types.llms.openai import ResponsesAPIResponse from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import ImageObject, ImageResponse, ModelResponse, TranscriptionResponse @@ -44,6 +48,95 @@ from tests.test_litellm.llms.bedrock.event_loop_probe import EventLoopProbe _ACTIVE_KEY = "_code_interpreter_interception_active" _SANDBOX_KEY = "_code_interpreter_interception_sandbox_key" + +async def _get_search_with_client( + client: HTTPHandler | AsyncHTTPHandler, provider_config: BaseSearchConfig | None = None +) -> SearchResponse: + result: Final = BaseLLMHTTPHandler().search( + query="test", + optional_params={}, + timeout=5, + logging_obj=Mock(), + api_key="test-key", + api_base="https://search.example.test/", + custom_llm_provider="tinyfish" if isinstance(provider_config, TinyfishSearchConfig) else "brave", + client=client, + asearch=isinstance(client, AsyncHTTPHandler), + provider_config=provider_config or BraveSearchConfig(), + ) + return await result if asyncio.iscoroutine(result) else result + + +@pytest.mark.asyncio +@pytest.mark.parametrize("is_async", (False, True)) +@pytest.mark.parametrize("status_code", (400, 401, 403, 422, 429, 500)) +async def test_get_search_raises_provider_http_errors(is_async: bool, status_code: int) -> None: + upstream_response: Final = httpx.Response( + status_code, json={"error": "rejected request"}, headers={"retry-after": "7"} + ) + transport: Final = httpx.MockTransport(lambda request: upstream_response) + async with httpx.AsyncClient(transport=transport) as async_client: + with httpx.Client(transport=transport) as sync_client: + client: Final = AsyncHTTPHandler() if is_async else HTTPHandler(client=sync_client) + if isinstance(client, AsyncHTTPHandler): + await client.close() + client.client = async_client + with pytest.raises(BaseLLMException) as error: + await _get_search_with_client(client) + assert error.value.status_code == status_code + assert "rejected request" in error.value.message + assert error.value.headers is not None + assert error.value.headers["retry-after"] == "7" + + +@pytest.mark.asyncio +@pytest.mark.parametrize("is_async", (False, True)) +@pytest.mark.parametrize("has_results", (False, True)) +async def test_get_search_preserves_successful_results(is_async: bool, has_results: bool) -> None: + results: Final = ( + [{"title": "Example", "url": "https://example.com", "description": "Example snippet"}] if has_results else [] + ) + transport: Final = httpx.MockTransport(lambda request: httpx.Response(200, json={"web": {"results": results}})) + async with httpx.AsyncClient(transport=transport) as async_client: + with httpx.Client(transport=transport) as sync_client: + client: Final = AsyncHTTPHandler() if is_async else HTTPHandler(client=sync_client) + if isinstance(client, AsyncHTTPHandler): + await client.close() + client.client = async_client + response: Final = await _get_search_with_client(client) + assert response.object == "search" + assert len(response.results) == int(has_results) + if has_results: + assert response.results[0].title == "Example" + assert response.results[0].url == "https://example.com" + assert response.results[0].snippet == "Example snippet" + + +@pytest.mark.asyncio +@pytest.mark.parametrize("is_async", (False, True)) +async def test_get_search_preserves_tinyfish_http_error_formatting(is_async: bool) -> None: + upstream_response: Final = httpx.Response( + 429, + json={"error": {"code": "RATE_LIMIT_EXCEEDED", "message": "rate limit exceeded"}}, + headers={"retry-after": "7"}, + ) + transport: Final = httpx.MockTransport(lambda request: upstream_response) + async with httpx.AsyncClient(transport=transport) as async_client: + with httpx.Client(transport=transport) as sync_client: + client: Final = AsyncHTTPHandler() if is_async else HTTPHandler(client=sync_client) + if isinstance(client, AsyncHTTPHandler): + await client.close() + client.client = async_client + with pytest.raises(BaseLLMException) as error: + await _get_search_with_client(client, TinyfishSearchConfig()) + assert error.value.status_code == 429 + assert error.value.message == ( + "TinyFish Search: rate limit exceeded. See https://docs.tinyfish.ai/search-api for details." + ) + assert error.value.headers is not None + assert error.value.headers["retry-after"] == "7" + + OCR_RESPONSE = { "pages": [{"index": 0, "markdown": "OCR output", "images": []}], "model": "mistral-ocr-latest", From 4295bf823ae77deb2b04763e38029e6519b9d2fa Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 14:16:43 -0700 Subject: [PATCH 111/157] fix(proxy): bound the cursorless grouped log page offset A page starting at or past SPEND_LOGS_PAGINATION_COUNT_CAP lies outside the total the client is given, so it now returns no rows without running the page query and the grouped top-N sort bound stays capped. Rewrites the offset test to page a fake session store instead of asserting on the generated SQL, and covers the last page inside the cap next to the first page past it. Claude-Session: https://claude.ai/code/session_01ESi9JwaXDww1vP3Qsrr4Mz --- .../spend_management_endpoints.py | 14 ++- .../test_spend_management_endpoints.py | 89 ++++++++++++++++--- 2 files changed, 89 insertions(+), 14 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 4a5995167f7..68181488c8d 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -2947,7 +2947,9 @@ async def _ui_session_grouped_spend_logs( page depth does not degrade the query plan. A request for ``page > 1`` without a cursor (the UI jumping straight to the last page, or back to a page it never walked through) falls back to ``OFFSET (page - 1) * - page_size``, bounded by the capped total. Each session is represented + page_size``; a page starting at or past ``SPEND_LOGS_PAGINATION_COUNT_CAP`` + lies outside the capped total the client is given, so it returns no rows + without running the query and the sort bound stays capped. Each session is represented by its newest non-MCP row, enriched by ``_build_ui_spend_logs_response`` exactly like the flat listing, and the response carries ``next_session_cursor`` / ``has_more`` while ``total`` counts sessions @@ -2966,7 +2968,9 @@ async def _ui_session_grouped_spend_logs( ) cursor_params: Final[tuple[object, ...]] = cursor if cursor else () limit_index: Final = next_param_index + len(cursor_params) - offset_params: Final[tuple[int, ...]] = ((page - 1) * page_size,) if cursor is None and page > 1 else () + offset: Final = (page - 1) * page_size if cursor is None else 0 + beyond_capped_window: Final = offset >= SPEND_LOGS_PAGINATION_COUNT_CAP + offset_params: Final[tuple[int, ...]] = (offset,) if offset and not beyond_capped_window else () offset_clause: Final = f"OFFSET ${limit_index + 1}" if offset_params else "" page_query: Final = f""" @@ -2980,8 +2984,10 @@ async def _ui_session_grouped_spend_logs( ORDER BY MAX("startTime") {direction}, {_SESSION_KEY_EXPR} {direction}, api_key {direction} LIMIT ${limit_index} {offset_clause} """ - page_rows: Final[Sequence[_SessionPageRow]] = await _query_raw( - prisma_client, page_query, *sql_params, *cursor_params, page_size + 1, *offset_params + page_rows: Final[Sequence[_SessionPageRow]] = ( + () + if beyond_capped_window + else await _query_raw(prisma_client, page_query, *sql_params, *cursor_params, page_size + 1, *offset_params) ) has_more: Final = len(page_rows) > page_size diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 5052cf3c085..eb65bb3714b 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -6629,6 +6629,30 @@ def _session_page_row(session_key, last_activity): return {"session_key": session_key, "api_key": "hashed-key", "last_activity": last_activity} +def _session_grouped_paginating_prisma(sessions): + """Mock prisma serving the grouped page query out of ``sessions``, honoring the LIMIT and OFFSET it asks for.""" + + async def mock_query_raw(sql_query, *params): + if "COUNT(*) AS total_count" in sql_query: + return [{"total_count": min(len(sessions), params[-1])}] + if "DISTINCT ON" in sql_query: + return [_session_representative_row(f"req-{session_key}", session_key) for session_key in params[-2]] + if "COALESCE(SUM(spend)" in sql_query: + return [] + bounds = re.search(r"LIMIT \$(\d+)(?: OFFSET \$(\d+))?", sql_query) + limit = params[int(bounds.group(1)) - 1] + offset = params[int(bounds.group(2)) - 1] if bounds.group(2) else 0 + return [ + _session_page_row(session_key, last_activity) + for session_key, last_activity in sessions[offset : offset + limit] + ] + + mock_prisma = MagicMock() + mock_prisma.db = MagicMock() + mock_prisma.db.query_raw = AsyncMock(side_effect=mock_query_raw) + return mock_prisma + + @pytest.mark.asyncio async def test_ui_view_spend_logs_group_by_session_first_page(client, monkeypatch): """One row per (session, api_key), session-count total, and a keyset cursor for the next page.""" @@ -6744,10 +6768,9 @@ async def test_ui_view_spend_logs_group_by_session_cursor_page(client, monkeypat @pytest.mark.asyncio async def test_ui_view_spend_logs_group_by_session_jumps_to_page_without_cursor(client, monkeypatch): - """page > 1 with no session_cursor (the UI's last-page jump) skips (page - 1) * page_size sessions by OFFSET.""" - page_rows = [_session_page_row("sess-3", "2026-08-29 06:00:00")] - reps = [_session_representative_row("req-3", "sess-3")] - mock_prisma = _session_grouped_mock_prisma(page_rows, 60, reps) + """page > 1 with no session_cursor (the UI's last-page jump) serves the sessions that page starts at.""" + sessions = tuple((f"sess-{index:02d}", f"2026-08-29 10:{59 - index:02d}:00") for index in range(60)) + mock_prisma = _session_grouped_paginating_prisma(sessions) monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma) monkeypatch.setattr( "litellm.proxy.spend_tracking.spend_management_endpoints._is_admin_view_safe", @@ -6772,14 +6795,60 @@ async def test_ui_view_spend_logs_group_by_session_jumps_to_page_without_cursor( assert response.status_code == 200, response.text data = response.json() assert data["page"] == 3 + assert data["total"] == 60 assert data["has_more"] is False - assert [row["request_id"] for row in data["data"]] == ["req-3"] + assert data["next_session_cursor"] is None + assert [row["request_id"] for row in data["data"]] == [f"req-sess-{index:02d}" for index in range(50, 60)] + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) - page_query_call = mock_prisma.db.query_raw.await_args_list[0] - page_query_sql = page_query_call.args[0] - assert "HAVING" not in page_query_sql - assert "OFFSET" in page_query_sql - assert page_query_call.args[-2:] == (26, 50), "LIMIT page_size + 1 then OFFSET (page - 1) * page_size" + +@pytest.mark.asyncio +async def test_ui_view_spend_logs_group_by_session_page_past_count_cap_is_empty(client, monkeypatch): + """The last page inside the capped total still lists sessions; the page after it is empty and costs no query.""" + cap = spend_management_endpoints.SPEND_LOGS_PAGINATION_COUNT_CAP + sessions = tuple((f"sess-{index:06d}", "2026-08-29 10:00:00") for index in range(cap + 50)) + mock_prisma = _session_grouped_paginating_prisma(sessions) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma) + monkeypatch.setattr( + "litellm.proxy.spend_tracking.spend_management_endpoints._is_admin_view_safe", + lambda user_api_key_dict: True, + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_user" + ) + try: + start_date, end_date = _default_date_range() + params = { + "start_date": start_date, + "end_date": end_date, + "group_by_session": "true", + "page_size": 25, + } + last_page = client.get( + "/spend/logs/ui", + params={**params, "page": cap // 25}, + headers={"Authorization": "Bearer sk-test"}, + ) + assert last_page.status_code == 200, last_page.text + last_page_data = last_page.json() + assert last_page_data["total"] == cap + assert last_page_data["total_is_capped"] is True + assert last_page_data["data"][0]["request_id"] == f"req-sess-{cap - 25:06d}" + assert len(last_page_data["data"]) == 25 + + mock_prisma.db.query_raw.reset_mock() + past_cap = client.get( + "/spend/logs/ui", + params={**params, "page": cap // 25 + 1}, + headers={"Authorization": "Bearer sk-test"}, + ) + assert past_cap.status_code == 200, past_cap.text + past_cap_data = past_cap.json() + assert past_cap_data["data"] == [] + assert past_cap_data["has_more"] is False + assert past_cap_data["total"] == cap + assert mock_prisma.db.query_raw.await_count == 1, "only the bounded count query runs past the capped window" finally: app.dependency_overrides.pop(ps.user_api_key_auth, None) From a426dc43cb4ec02e1908cbfd2e813d8c511bed9f Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 14:20:10 -0700 Subject: [PATCH 112/157] fix(policy_engine): run global policy pipelines before scoped ones (#39697) * fix(policy_engine): run global policy pipelines before scoped ones Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(policy_engine): rank duplicate attachments by broadest scope Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(policy_engine): rank combined-scope attachments below single-scope ones Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: shivam Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/litellm_pre_call_utils.py | 26 ++++---- .../policy_engine/attachment_registry.py | 60 ++++++++++------- .../policy_engine/test_attachment_registry.py | 65 +++++++++++++++++++ .../proxy/test_litellm_pre_call_utils.py | 25 +++++++ 4 files changed, 141 insertions(+), 35 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index ad4687e95db..c2805d00c2e 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -173,6 +173,7 @@ _ENABLE_TEAM_STALE_ALIAS_BYPASS: bool | None = None if TYPE_CHECKING: from litellm.integrations.otel.model.destination import OtelDestination + from litellm.proxy.policy_engine.attachment_registry import AttachmentRegistry from litellm.proxy.proxy_server import ProxyConfig as _ProxyConfig from litellm.types.proxy.policy_engine import Policy, PolicyMatchContext @@ -3141,6 +3142,7 @@ def _match_and_track_policies( context: "PolicyMatchContext", request_body_policies: Sequence[str], policies_override: dict[str, "Policy"] | None = None, + attachment_registry_override: "AttachmentRegistry | None" = None, ) -> tuple[list[str], dict[str, str]]: """ Match policies via attachments and request body, track them in metadata. @@ -3157,7 +3159,9 @@ def _match_and_track_policies( from litellm.proxy.policy_engine.policy_matcher import PolicyMatcher # Get matching policies via attachments (with match reasons for attribution) - attachment_registry: Final = get_attachment_registry() + attachment_registry: Final = ( + attachment_registry_override if attachment_registry_override is not None else get_attachment_registry() + ) matches_with_reasons: Final = attachment_registry.get_attached_policies_with_reasons(context) matching_policy_names: Final = [m["policy_name"] for m in matches_with_reasons] policy_reasons: Final = {m["policy_name"]: m["matched_via"] for m in matches_with_reasons} @@ -3165,9 +3169,11 @@ def _match_and_track_policies( verbose_proxy_logger.debug("Policy engine: matched policies via attachments: %s", matching_policy_names) # Combine attachment-based policies with dynamic request body policies - all_policy_names: Final = set(matching_policy_names) - if request_body_policies and isinstance(request_body_policies, list): - all_policy_names.update(request_body_policies) + request_body_policies_list: Final = ( + tuple(request_body_policies) if request_body_policies and isinstance(request_body_policies, list) else () + ) + all_policy_names: Final = tuple(dict.fromkeys((*matching_policy_names, *request_body_policies_list))) + if request_body_policies_list: verbose_proxy_logger.debug("Policy engine: added dynamic policies from request body: %s", request_body_policies) if not all_policy_names: @@ -3238,16 +3244,14 @@ def _apply_resolved_guardrails_to_metadata( if not resolved_guardrails and not pipelines: return - existing_guardrails = data[metadata_variable_name].get("guardrails", []) - if not isinstance(existing_guardrails, list): - existing_guardrails = [] + existing_guardrails: Final = data[metadata_variable_name].get("guardrails", []) + existing_guardrails_list: Final = existing_guardrails if isinstance(existing_guardrails, list) else [] # Combine existing guardrails with policy-resolved guardrails (no duplicates) - combined = set(existing_guardrails) - combined.update(resolved_guardrails) - data[metadata_variable_name]["guardrails"] = list(combined) + combined: Final = list(dict.fromkeys((*existing_guardrails_list, *resolved_guardrails))) + data[metadata_variable_name]["guardrails"] = combined - verbose_proxy_logger.debug("Policy engine: added guardrails to request metadata: %s", list(combined)) + verbose_proxy_logger.debug("Policy engine: added guardrails to request metadata: %s", combined) async def add_guardrails_from_policy_engine( diff --git a/litellm/proxy/policy_engine/attachment_registry.py b/litellm/proxy/policy_engine/attachment_registry.py index 05df44242aa..a8ead86ac36 100644 --- a/litellm/proxy/policy_engine/attachment_registry.py +++ b/litellm/proxy/policy_engine/attachment_registry.py @@ -30,6 +30,23 @@ class PolicyAttachmentMatch(TypedDict): matched_via: str +def _attachment_specificity(attachment: PolicyAttachment) -> tuple[int, int]: + if attachment.is_global(): + return (0, 0) + + dims: Final = tuple( + specificity + for values, specificity in ( + (attachment.teams, 1), + (attachment.keys, 2), + (attachment.tags, 3), + (attachment.models, 4), + ) + if values + ) + return (max(dims, default=0), len(dims)) + + class AttachmentRegistry: """ In-memory registry for storing and managing policy attachments. @@ -116,31 +133,26 @@ class AttachmentRegistry: """ from litellm.proxy.policy_engine.policy_matcher import PolicyMatcher - results: Final[list[PolicyAttachmentMatch]] = [] - seen_policies: Final[set[str]] = set() + matching_attachments: Final = sorted( + ( + attachment + for attachment in self._attachments + if PolicyMatcher.scope_matches(scope=attachment.to_policy_scope(), context=context) + ), + key=_attachment_specificity, + ) + unique_attachments: Final = tuple( + next(attachment for attachment in matching_attachments if attachment.policy == policy_name) + for policy_name in dict.fromkeys(attachment.policy for attachment in matching_attachments) + ) - for attachment in self._attachments: - scope = attachment.to_policy_scope() - if PolicyMatcher.scope_matches(scope=scope, context=context): - if attachment.policy not in seen_policies: - seen_policies.add(attachment.policy) - matched_via = self._describe_match_reason(attachment, context) - results.append( - { - "policy_name": attachment.policy, - "matched_via": matched_via, - } - ) - verbose_proxy_logger.debug( - "Attachment matched: policy=%s, matched_via=%s, context=(team=%s, key=%s, model=%s)", - attachment.policy, - matched_via, - context.team_alias, - context.key_alias, - context.model, - ) - - return results + return [ + { + "policy_name": attachment.policy, + "matched_via": self._describe_match_reason(attachment, context), + } + for attachment in unique_attachments + ] @staticmethod def _describe_match_reason(attachment: PolicyAttachment, context: PolicyMatchContext) -> str: diff --git a/tests/test_litellm/proxy/policy_engine/test_attachment_registry.py b/tests/test_litellm/proxy/policy_engine/test_attachment_registry.py index cc231e383a3..87b1bc56659 100644 --- a/tests/test_litellm/proxy/policy_engine/test_attachment_registry.py +++ b/tests/test_litellm/proxy/policy_engine/test_attachment_registry.py @@ -139,6 +139,71 @@ class TestGetAttachedPolicies: assert "gpt4-policy" in attached assert len(attached) == 3 + def test_matches_are_ordered_from_broadest_to_narrowest_scope(self): + registry = AttachmentRegistry() + registry.load_attachments( + [ + {"policy": "model-policy", "models": ["gpt-4"]}, + {"policy": "team-policy", "teams": ["t1"]}, + {"policy": "global-policy", "scope": "*"}, + ] + ) + + context = PolicyMatchContext(team_alias="t1", model="gpt-4") + + assert registry.get_attached_policies(context) == [ + "global-policy", + "team-policy", + "model-policy", + ] + + def test_combined_team_and_model_attachment_uses_model_specificity(self): + registry = AttachmentRegistry() + registry.load_attachments( + [ + {"policy": "team-policy", "teams": ["t1"]}, + {"policy": "team-model-policy", "teams": ["t1"], "models": ["gpt-4"]}, + ] + ) + + context = PolicyMatchContext(team_alias="t1", model="gpt-4") + + assert registry.get_attached_policies(context) == [ + "team-policy", + "team-model-policy", + ] + + def test_duplicate_policy_uses_broadest_matching_attachment(self): + registry = AttachmentRegistry() + registry.load_attachments( + [ + {"policy": "shared-policy", "models": ["gpt-4"]}, + {"policy": "model-policy", "models": ["gpt-4"]}, + {"policy": "shared-policy", "scope": "*"}, + ] + ) + + context = PolicyMatchContext(model="gpt-4") + + assert registry.get_attached_policies(context) == [ + "shared-policy", + "model-policy", + ] + assert registry.get_attached_policies_with_reasons(context)[0]["matched_via"] == "scope:*" + + def test_duplicate_policy_prefers_single_scope_over_combined_scope(self): + registry = AttachmentRegistry() + registry.load_attachments( + [ + {"policy": "shared-policy", "teams": ["t1"], "models": ["gpt-4"]}, + {"policy": "shared-policy", "models": ["gpt-4"]}, + ] + ) + + context = PolicyMatchContext(team_alias="t1", model="gpt-4") + + assert registry.get_attached_policies_with_reasons(context)[0]["matched_via"] == "model:gpt-4" + def test_same_policy_multiple_attachments_no_duplicates(self): """Test same policy attached multiple ways doesn't duplicate.""" registry = AttachmentRegistry() diff --git a/tests/test_litellm/proxy/test_litellm_pre_call_utils.py b/tests/test_litellm/proxy/test_litellm_pre_call_utils.py index 94aaa32519e..ec9025a5220 100644 --- a/tests/test_litellm/proxy/test_litellm_pre_call_utils.py +++ b/tests/test_litellm/proxy/test_litellm_pre_call_utils.py @@ -23,6 +23,7 @@ from litellm.proxy.litellm_pre_call_utils import ( _get_dynamic_logging_metadata, _get_enforced_params, _get_metadata_variable_name, + _match_and_track_policies, _promoted_trace_control_fields, _resolve_credential_from_model_config, _resolve_provider_from_deployment, @@ -4149,6 +4150,30 @@ async def test_add_guardrails_from_policy_engine(): attachment_registry._initialized = False +def test_match_and_track_policies_preserves_attachment_and_request_body_order(): + from litellm.proxy.policy_engine.attachment_registry import AttachmentRegistry + from litellm.types.proxy.policy_engine import Policy, PolicyMatchContext + + attachment_policy_names = [f"attachment-policy-{index}" for index in range(8)] + request_body_policy_names = ["body-policy-1", "body-policy-2"] + policy_names = [*attachment_policy_names, *request_body_policy_names] + policies = {policy_name: Policy() for policy_name in policy_names} + attachment_registry = AttachmentRegistry() + attachment_registry.load_attachments( + [{"policy": policy_name, "scope": "*"} for policy_name in attachment_policy_names] + ) + + applied_policy_names, _ = _match_and_track_policies( + data={"metadata": {}}, + context=PolicyMatchContext(model="gpt-4"), + request_body_policies=request_body_policy_names, + policies_override=policies, + attachment_registry_override=attachment_registry, + ) + + assert applied_policy_names == policy_names + + @pytest.mark.asyncio async def test_add_guardrails_from_policy_engine_keeps_a_policy_added_guardrail_its_pipeline_also_steps(): from litellm.proxy.policy_engine.attachment_registry import get_attachment_registry From cb434742c6dc4e06c9f756b24bc18ef9cc136315 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 14:27:36 -0700 Subject: [PATCH 113/157] fix(proxy): end the cursorless grouped log page at the capped total A page size that does not divide SPEND_LOGS_PAGINATION_COUNT_CAP left the last page starting inside the capped window and reading past it, so the rows disagreed with the total reported next to them. The page limit now stops at the end of that window, and has_more plus next_session_cursor still hand back a cursor for walking further. Claude-Session: https://claude.ai/code/session_01ESi9JwaXDww1vP3Qsrr4Mz --- .../spend_management_endpoints.py | 18 ++++----- .../test_spend_management_endpoints.py | 39 +++++++++++++++++++ 2 files changed, 48 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 68181488c8d..2105d682d9a 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -2947,9 +2947,9 @@ async def _ui_session_grouped_spend_logs( page depth does not degrade the query plan. A request for ``page > 1`` without a cursor (the UI jumping straight to the last page, or back to a page it never walked through) falls back to ``OFFSET (page - 1) * - page_size``; a page starting at or past ``SPEND_LOGS_PAGINATION_COUNT_CAP`` - lies outside the capped total the client is given, so it returns no rows - without running the query and the sort bound stays capped. Each session is represented + page_size``, trimmed to the end of the ``SPEND_LOGS_PAGINATION_COUNT_CAP`` + window the capped ``total`` promises, so a page never runs past that total + and one starting at or past it returns no rows without a query. Each session is represented by its newest non-MCP row, enriched by ``_build_ui_spend_logs_response`` exactly like the flat listing, and the response carries ``next_session_cursor`` / ``has_more`` while ``total`` counts sessions @@ -2969,8 +2969,8 @@ async def _ui_session_grouped_spend_logs( cursor_params: Final[tuple[object, ...]] = cursor if cursor else () limit_index: Final = next_param_index + len(cursor_params) offset: Final = (page - 1) * page_size if cursor is None else 0 - beyond_capped_window: Final = offset >= SPEND_LOGS_PAGINATION_COUNT_CAP - offset_params: Final[tuple[int, ...]] = (offset,) if offset and not beyond_capped_window else () + page_limit: Final = min(page_size, SPEND_LOGS_PAGINATION_COUNT_CAP - offset) + offset_params: Final[tuple[int, ...]] = (offset,) if offset and page_limit > 0 else () offset_clause: Final = f"OFFSET ${limit_index + 1}" if offset_params else "" page_query: Final = f""" @@ -2986,12 +2986,12 @@ async def _ui_session_grouped_spend_logs( """ page_rows: Final[Sequence[_SessionPageRow]] = ( () - if beyond_capped_window - else await _query_raw(prisma_client, page_query, *sql_params, *cursor_params, page_size + 1, *offset_params) + if page_limit <= 0 + else await _query_raw(prisma_client, page_query, *sql_params, *cursor_params, page_limit + 1, *offset_params) ) - has_more: Final = len(page_rows) > page_size - visible_rows: Final = page_rows[:page_size] + has_more: Final = len(page_rows) > page_limit + visible_rows: Final = page_rows[:page_limit] next_cursor: Final = ( f"{visible_rows[-1]['last_activity']}|{visible_rows[-1]['api_key']}|{visible_rows[-1]['session_key']}" if has_more and visible_rows diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index eb65bb3714b..5920a984239 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -6853,6 +6853,45 @@ async def test_ui_view_spend_logs_group_by_session_page_past_count_cap_is_empty( app.dependency_overrides.pop(ps.user_api_key_auth, None) +@pytest.mark.asyncio +async def test_ui_view_spend_logs_group_by_session_last_page_stops_at_the_capped_total(client, monkeypatch): + """A page size that does not divide the cap still ends the last page at the capped total it reports.""" + cap = spend_management_endpoints.SPEND_LOGS_PAGINATION_COUNT_CAP + sessions = tuple((f"sess-{index:06d}", "2026-08-29 10:00:00") for index in range(cap + 50)) + mock_prisma = _session_grouped_paginating_prisma(sessions) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma) + monkeypatch.setattr( + "litellm.proxy.spend_tracking.spend_management_endpoints._is_admin_view_safe", + lambda user_api_key_dict: True, + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_user" + ) + try: + start_date, end_date = _default_date_range() + response = client.get( + "/spend/logs/ui", + params={ + "start_date": start_date, + "end_date": end_date, + "group_by_session": "true", + "page": cap // 7 + 1, + "page_size": 7, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + assert response.status_code == 200, response.text + data = response.json() + assert data["total"] == cap + assert [row["request_id"] for row in data["data"]] == [ + f"req-sess-{index:06d}" for index in range(cap - cap % 7, cap) + ] + assert data["has_more"] is True + assert data["next_session_cursor"] is not None + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + @pytest.mark.asyncio async def test_ui_view_spend_logs_group_by_session_offset_for_non_starttime_sort( client, monkeypatch From c64e746b7187137e5fcf1663b5b4e6fd584670ab Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 14:34:21 -0700 Subject: [PATCH 114/157] fix(content_filter): log only scan time as streaming post_call guardrail duration (#40760) The streaming iterator hook timed the whole provider stream and logged that as the guardrail duration, so PrometheusLogger added LLM generation time to litellm_overhead_with_guardrails_latency_metric. The hook now accumulates the time spent inside _filter_single_text per chunk and logs that sum, keeping start_time and end_time as the wall-clock window. Resolves LIT-7589 Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../litellm_content_filter/content_filter.py | 10 +++- .../content_filter/test_content_filter.py | 58 +++++++++++++++++++ 2 files changed, 67 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py index 722f96ef814..1e684c514de 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py @@ -9,6 +9,7 @@ import asyncio import json import os import re +import time from collections.abc import AsyncGenerator, Coroutine, Mapping, Sequence from datetime import datetime from re import Pattern @@ -1702,6 +1703,7 @@ class ContentFilterGuardrail(CustomGuardrail): start_time: datetime, masked_entity_count: dict[str, int], exception_str: str, + duration: float | None = None, ) -> None: """ Log guardrail information to request_data metadata. @@ -1713,6 +1715,7 @@ class ContentFilterGuardrail(CustomGuardrail): start_time: Start time of guardrail execution masked_entity_count: Count of masked entities by type exception_str: Exception string if guardrail failed + duration: Seconds spent inside the guardrail; defaults to the wall clock since start_time """ # Convert TypedDict detections to regular dicts for JSON serialization guardrail_json_response: Exception | str | dict | list[dict] = [dict(detection) for detection in detections] @@ -1741,7 +1744,7 @@ class ContentFilterGuardrail(CustomGuardrail): guardrail_status=status, start_time=start_time.timestamp(), end_time=datetime.now().timestamp(), - duration=(datetime.now() - start_time).total_seconds(), + duration=(datetime.now() - start_time).total_seconds() if duration is None else duration, masked_entity_count=masked_entity_count, tracing_detail=GuardrailTracingDetail(**tracing_kw), ) @@ -1971,6 +1974,7 @@ class ContentFilterGuardrail(CustomGuardrail): buffer_size: Final = 50 # Increased buffer to catch patterns split across many chunks start_time: Final = datetime.now() + scan_seconds: float = 0.0 # rebind-ok: accumulates per-chunk scan time across the stream detections: list[ContentFilterDetection] = [] masked_entity_count: Final[dict[str, int]] = {} status: GuardrailStatus = "success" @@ -2007,6 +2011,7 @@ class ContentFilterGuardrail(CustomGuardrail): # Add a space at the end if it's the final chunk to trigger word boundaries (\b) text_to_scan = text_to_check + (" " if is_final else "") choice_detections: list[ContentFilterDetection] = [] + scan_started = time.perf_counter() try: # _filter_single_text scans the whole accumulated @@ -2024,6 +2029,8 @@ class ContentFilterGuardrail(CustomGuardrail): except Exception as e: verbose_proxy_logger.error("ContentFilterGuardrail: Error in masking: %s", e) masked_text = text_to_scan # Fallback to current text + finally: + scan_seconds += time.perf_counter() - scan_started # Determine how much can be safely yielded if is_final: @@ -2074,6 +2081,7 @@ class ContentFilterGuardrail(CustomGuardrail): start_time=start_time, masked_entity_count=masked_entity_count, exception_str=exception_str, + duration=scan_seconds, ) @staticmethod diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter.py index be55ac47bde..130b0da000b 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter.py @@ -626,6 +626,64 @@ class TestContentFilterGuardrail: assert entry["guardrail_status"] == "success" assert entry["guardrail_response"] == [] + @pytest.mark.asyncio + async def test_streaming_hook_duration_excludes_provider_wait(self): + """ + Streaming post-call: the logged guardrail duration must only cover the + per-chunk scans, not the time spent waiting on the provider between + chunks. PrometheusLogger adds post_call guardrail duration to + litellm_overhead_with_guardrails_latency_metric, so a duration spanning + the whole stream reports LLM generation time as guardrail overhead. + """ + import asyncio + + from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices + + guardrail = ContentFilterGuardrail( + guardrail_name="test-streaming-duration", + patterns=[ + ContentFilterPattern( + pattern_type="prebuilt", + pattern_name="email", + action=ContentFilterAction.MASK, + ), + ], + event_hook=GuardrailEventHooks.post_call, + ) + + provider_wait_per_chunk = 0.15 + chunks = ("Hello ", "world, reach me at ", "test@example.com ") + + async def slow_stream(): + for i, text in enumerate(chunks): + await asyncio.sleep(provider_wait_per_chunk) + yield ModelResponseStream( + id=f"chunk{i}", + choices=[ + StreamingChoices( + delta=Delta(content=text), + index=0, + finish_reason="stop" if i == len(chunks) - 1 else None, + ) + ], + model="gpt-4", + ) + + request_data = {"messages": [{"role": "user", "content": "Hi"}], "model": "gpt-4o", "metadata": {}} + + async for _ in guardrail.async_post_call_streaming_iterator_hook( + user_api_key_dict=MagicMock(), + response=slow_stream(), + request_data=request_data, + ): + pass + + entry = request_data["metadata"]["standard_logging_guardrail_information"][0] + stream_wall_clock = entry["end_time"] - entry["start_time"] + assert stream_wall_clock >= provider_wait_per_chunk * len(chunks) + assert entry["masked_entity_count"].get("email", 0) >= 1 + assert 0 < entry["duration"] < provider_wait_per_chunk, entry["duration"] + @pytest.mark.asyncio async def test_streaming_hook_logs_guardrail_information_mask(self): """ From a8d017180079aff46a4e3b9612bdefa5a3e323ac Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:38:10 -0700 Subject: [PATCH 115/157] ci: gate stable and RC releases on the Redis chaos load test The chaos test only ran on a weekly cron, so a release could be cut from a commit it had never covered. Making it callable lets create-release.yml run it against the exact commit being tagged and refuse to tag if it fails. Dev, nightly, alpha and beta tags skip the gate: they are cut far more often than stable and RC tags, and the weekly schedule already covers the default branch. Input validation moves into the gate job so a malformed tag or SHA fails before spending a multi-minute chaos run. Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 42 +++++++++++++++++++--- .github/workflows/test-e2e-redis-chaos.yml | 7 ++++ 2 files changed, 45 insertions(+), 4 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 0ad84cd3ceb..bdc1810f204 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -15,11 +15,14 @@ on: permissions: {} jobs: - release: - name: Create Release + # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta + # tags are cut too often to spend a multi-minute chaos run on each one, and the weekly + # schedule already covers the default branch. + gate: + name: Validate inputs and decide whether this tag is gated runs-on: ubuntu-latest - permissions: - contents: write + outputs: + gated: ${{ steps.decide.outputs.gated }} steps: - name: Validate inputs env: @@ -35,6 +38,37 @@ jobs: exit 1 fi + - name: Decide + id: decide + env: + TAG: ${{ inputs.tag }} + run: | + if echo "${TAG}" | grep -qiE '(nightly|alpha|beta|[-.]dev)'; then + echo "gated=false" >> "$GITHUB_OUTPUT" + echo "${TAG} is a pre-release that skips the chaos gate" + else + echo "gated=true" >> "$GITHUB_OUTPUT" + echo "${TAG} is a stable or RC tag and must pass the chaos gate" + fi + + redis-chaos-gate: + name: Redis Chaos E2E + needs: gate + if: needs.gate.outputs.gated == 'true' + permissions: + contents: read + uses: ./.github/workflows/test-e2e-redis-chaos.yml + with: + ref: ${{ inputs.commit_hash }} + + release: + name: Create Release + needs: [gate, redis-chaos-gate] + if: always() && needs.gate.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') + runs-on: ubuntu-latest + permissions: + contents: write + steps: - name: Create release env: TAG: ${{ inputs.tag }} diff --git a/.github/workflows/test-e2e-redis-chaos.yml b/.github/workflows/test-e2e-redis-chaos.yml index 0880c1fb464..9deca013603 100644 --- a/.github/workflows/test-e2e-redis-chaos.yml +++ b/.github/workflows/test-e2e-redis-chaos.yml @@ -4,6 +4,12 @@ on: schedule: - cron: "0 12 * * 6" workflow_dispatch: + workflow_call: + inputs: + ref: + description: "Commit SHA or ref to test. Defaults to the ref the workflow was triggered on" + required: false + type: string permissions: contents: read @@ -45,6 +51,7 @@ jobs: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: persist-credentials: false + ref: ${{ inputs.ref || github.sha }} - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 From 93c8ef1beb446a864eb74ef49ffd0f3229c8cc80 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:40:08 -0700 Subject: [PATCH 116/157] docs(e2e): note the release gate in the load/ harness guide Co-Authored-By: Claude Code --- tests/e2e/CLAUDE.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 419cfd6dedd..b8e9c8ccde3 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`, which `create-release.yml` also calls to gate stable and RC tags on the commit being released), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke From 9d91aff249b71b9f8e08b00e8d729e3fb1985c89 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:43:47 -0700 Subject: [PATCH 117/157] ci(release): rename the tag-decision job to stable-release-gate Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index bdc1810f204..56f336c258d 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -18,7 +18,7 @@ jobs: # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta # tags are cut too often to spend a multi-minute chaos run on each one, and the weekly # schedule already covers the default branch. - gate: + stable-release-gate: name: Validate inputs and decide whether this tag is gated runs-on: ubuntu-latest outputs: @@ -53,8 +53,8 @@ jobs: redis-chaos-gate: name: Redis Chaos E2E - needs: gate - if: needs.gate.outputs.gated == 'true' + needs: stable-release-gate + if: needs.stable-release-gate.outputs.gated == 'true' permissions: contents: read uses: ./.github/workflows/test-e2e-redis-chaos.yml @@ -63,8 +63,8 @@ jobs: release: name: Create Release - needs: [gate, redis-chaos-gate] - if: always() && needs.gate.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') + needs: [stable-release-gate, redis-chaos-gate] + if: always() && needs.stable-release-gate.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') runs-on: ubuntu-latest permissions: contents: write From cdf8a8cab8b56b0c70a5eee437e9e97eb002fdf6 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:47:57 -0700 Subject: [PATCH 118/157] ci(release): rename the tag-classification job to prepare Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 56f336c258d..9100349440e 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -18,7 +18,7 @@ jobs: # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta # tags are cut too often to spend a multi-minute chaos run on each one, and the weekly # schedule already covers the default branch. - stable-release-gate: + prepare: name: Validate inputs and decide whether this tag is gated runs-on: ubuntu-latest outputs: @@ -53,8 +53,8 @@ jobs: redis-chaos-gate: name: Redis Chaos E2E - needs: stable-release-gate - if: needs.stable-release-gate.outputs.gated == 'true' + needs: prepare + if: needs.prepare.outputs.gated == 'true' permissions: contents: read uses: ./.github/workflows/test-e2e-redis-chaos.yml @@ -63,8 +63,8 @@ jobs: release: name: Create Release - needs: [stable-release-gate, redis-chaos-gate] - if: always() && needs.stable-release-gate.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') + needs: [prepare, redis-chaos-gate] + if: always() && needs.prepare.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') runs-on: ubuntu-latest permissions: contents: write From 7ebf60f33e16cefe6a9cbf500870436059015c37 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:49:59 -0700 Subject: [PATCH 119/157] ci(release): rename the tag-classification job to classify-if-stable-release Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 9100349440e..7fd95badb0a 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -18,7 +18,7 @@ jobs: # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta # tags are cut too often to spend a multi-minute chaos run on each one, and the weekly # schedule already covers the default branch. - prepare: + classify-if-stable-release: name: Validate inputs and decide whether this tag is gated runs-on: ubuntu-latest outputs: @@ -53,8 +53,8 @@ jobs: redis-chaos-gate: name: Redis Chaos E2E - needs: prepare - if: needs.prepare.outputs.gated == 'true' + needs: classify-if-stable-release + if: needs.classify-if-stable-release.outputs.gated == 'true' permissions: contents: read uses: ./.github/workflows/test-e2e-redis-chaos.yml @@ -63,8 +63,8 @@ jobs: release: name: Create Release - needs: [prepare, redis-chaos-gate] - if: always() && needs.prepare.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') + needs: [classify-if-stable-release, redis-chaos-gate] + if: always() && needs.classify-if-stable-release.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') runs-on: ubuntu-latest permissions: contents: write From 4bcd60e72b60414ff4f5b7ba08b1ef4eb3f60452 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 14:51:05 -0700 Subject: [PATCH 120/157] perf(proxy): total a short grouped log page from the page itself A cursorless page that comes back without its lookahead row is the end of the list, so the total is offset + len(page) and the bounded grouped COUNT over the whole spend-log table is skipped. First pages on small deployments and every offset last page now cost one query less. Moves the count into _count_grouped_sessions and reworks the query-optimization test that asserted the count always runs second onto a full page, where it does. Claude-Session: https://claude.ai/code/session_01ESi9JwaXDww1vP3Qsrr4Mz --- .../spend_management_endpoints.py | 50 +++++++++++++------ .../test_spend_management_endpoints.py | 40 ++++++++++++++- .../test_spend_query_optimization.py | 14 +++--- 3 files changed, 79 insertions(+), 25 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 2105d682d9a..582622cfb50 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -2925,6 +2925,32 @@ async def _fetch_session_representatives( return [rep_by_key[key] for key in session_keys if key in rep_by_key] # mutable-ok: rows are enriched in place +async def _count_grouped_sessions( + prisma_client: "PrismaClient", + where_clause: str, + sql_params: Sequence[object], + next_param_index: int, +) -> tuple[int, bool]: + """Count the sessions matching the filter, returning ``(total, total_is_capped)`` bounded by the count cap.""" + count_query: Final = f""" + SELECT COUNT(*) AS total_count + FROM ( + SELECT 1 + FROM "LiteLLM_SpendLogs" + WHERE {where_clause} + GROUP BY {_SESSION_GROUP_KEY_SQL} + LIMIT ${next_param_index} + ) AS bounded_sessions + """ + count_rows: Final[Sequence[_SpendLogsCountRow]] = await _query_raw( + prisma_client, count_query, *sql_params, SPEND_LOGS_PAGINATION_COUNT_CAP + 1 + ) + raw_total: Final = int(count_rows[0]["total_count"]) if count_rows else 0 + return ( + (SPEND_LOGS_PAGINATION_COUNT_CAP, True) if raw_total > SPEND_LOGS_PAGINATION_COUNT_CAP else (raw_total, False) + ) + + async def _ui_session_grouped_spend_logs( prisma_client: "PrismaClient", sql_conditions: Sequence[str], @@ -2953,7 +2979,9 @@ async def _ui_session_grouped_spend_logs( by its newest non-MCP row, enriched by ``_build_ui_spend_logs_response`` exactly like the flat listing, and the response carries ``next_session_cursor`` / ``has_more`` while ``total`` counts sessions - (capped like the flat total). + (capped like the flat total). A page that runs out of sessions is itself + the end of the list, so its ``total`` is ``offset + len(page)`` and the + grouped count query is skipped. """ where_clause: Final = " AND ".join(sql_conditions) if sql_conditions else "TRUE" cmp_op: Final = "<" if sort_desc else ">" @@ -2998,22 +3026,12 @@ async def _ui_session_grouped_spend_logs( else None ) - count_query: Final = f""" - SELECT COUNT(*) AS total_count - FROM ( - SELECT 1 - FROM "LiteLLM_SpendLogs" - WHERE {where_clause} - GROUP BY {_SESSION_GROUP_KEY_SQL} - LIMIT ${next_param_index} - ) AS bounded_sessions - """ - count_rows: Final[Sequence[_SpendLogsCountRow]] = await _query_raw( - prisma_client, count_query, *sql_params, SPEND_LOGS_PAGINATION_COUNT_CAP + 1 + page_ends_the_list: Final = cursor is None and page_limit > 0 and not has_more + total_records, total_is_capped = ( + (offset + len(page_rows), False) + if page_ends_the_list + else await _count_grouped_sessions(prisma_client, where_clause, sql_params, next_param_index) ) - raw_total: Final = int(count_rows[0]["total_count"]) if count_rows else 0 - total_is_capped: Final = raw_total > SPEND_LOGS_PAGINATION_COUNT_CAP - total_records: Final = SPEND_LOGS_PAGINATION_COUNT_CAP if total_is_capped else raw_total session_keys: Final = tuple((row["session_key"], row["api_key"]) for row in visible_rows) data: Final[list[dict[str, object]]] = ( # mutable-ok: _build_ui_spend_logs_response writes onto each row diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 5920a984239..c709ac77a0b 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -6629,12 +6629,12 @@ def _session_page_row(session_key, last_activity): return {"session_key": session_key, "api_key": "hashed-key", "last_activity": last_activity} -def _session_grouped_paginating_prisma(sessions): +def _session_grouped_paginating_prisma(sessions, counted_total=None): """Mock prisma serving the grouped page query out of ``sessions``, honoring the LIMIT and OFFSET it asks for.""" async def mock_query_raw(sql_query, *params): if "COUNT(*) AS total_count" in sql_query: - return [{"total_count": min(len(sessions), params[-1])}] + return [{"total_count": min(len(sessions) if counted_total is None else counted_total, params[-1])}] if "DISTINCT ON" in sql_query: return [_session_representative_row(f"req-{session_key}", session_key) for session_key in params[-2]] if "COALESCE(SUM(spend)" in sql_query: @@ -6803,6 +6803,42 @@ async def test_ui_view_spend_logs_group_by_session_jumps_to_page_without_cursor( app.dependency_overrides.pop(ps.user_api_key_auth, None) +@pytest.mark.asyncio +async def test_ui_view_spend_logs_group_by_session_short_page_totals_itself(client, monkeypatch): + """A page that runs out of sessions is the end of the list, so the total comes from it and nothing is counted.""" + sessions = tuple((f"sess-{index:02d}", f"2026-08-29 10:{59 - index:02d}:00") for index in range(10)) + mock_prisma = _session_grouped_paginating_prisma(sessions, counted_total=999) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma) + monkeypatch.setattr( + "litellm.proxy.spend_tracking.spend_management_endpoints._is_admin_view_safe", + lambda user_api_key_dict: True, + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_user" + ) + try: + start_date, end_date = _default_date_range() + response = client.get( + "/spend/logs/ui", + params={ + "start_date": start_date, + "end_date": end_date, + "group_by_session": "true", + "page": 1, + "page_size": 25, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + assert response.status_code == 200, response.text + data = response.json() + assert data["total"] == 10, "the count query's 999 would have won if it had been asked" + assert data["total_is_capped"] is False + assert data["total_pages"] == 1 + assert len(data["data"]) == 10 + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + @pytest.mark.asyncio async def test_ui_view_spend_logs_group_by_session_page_past_count_cap_is_empty(client, monkeypatch): """The last page inside the capped total still lists sessions; the page after it is empty and costs no query.""" diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_query_optimization.py b/tests/test_litellm/proxy/spend_tracking/test_spend_query_optimization.py index a7de3f1d8d6..54e5a6d5385 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_query_optimization.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_query_optimization.py @@ -528,8 +528,8 @@ async def test_spend_logs_ui_group_by_session_paginates_sessions(monkeypatch): group_key = "COALESCE(NULLIF(session_id, ''), request_id), api_key" session_rows = [ - {"session_key": "req-1", "api_key": "k", "last_activity": "2026-02-16 10:00:00"}, - {"session_key": "req-2", "api_key": "k", "last_activity": "2026-02-16 09:00:00"}, + {"session_key": f"req-{index}", "api_key": "k", "last_activity": f"2026-02-16 10:{59 - index:02d}:00"} + for index in range(51) ] representative_rows = [ {"request_id": "req-1", "api_key": "k", "metadata": "{}", "session_id": None}, @@ -538,7 +538,7 @@ async def test_spend_logs_ui_group_by_session_paginates_sessions(monkeypatch): async def mock_query_raw(sql_query, *params): if "COUNT(*) AS total_count" in sql_query: - return [{"total_count": 12}] + return [{"total_count": 60}] if "DISTINCT ON" in sql_query: return representative_rows return session_rows @@ -590,11 +590,11 @@ async def test_spend_logs_ui_group_by_session_paginates_sessions(monkeypatch): assert "COUNT(*) OVER ()" not in rep_sql assert [row["request_id"] for row in response["data"]] == ["req-1", "req-2"] - assert response["total"] == 12 + assert response["total"] == 60 assert response["total_is_capped"] is False - assert response["total_pages"] == 1 - assert response["has_more"] is False - assert response["next_session_cursor"] is None + assert response["total_pages"] == 2 + assert response["has_more"] is True + assert response["next_session_cursor"] == "2026-02-16 10:10:00|k|req-49" @pytest.mark.asyncio From 445b45f261184ac58c7ddad0ef3749cd7bb6d220 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:55:59 -0700 Subject: [PATCH 121/157] ci(release): rename the tag-classification job to run-stable-release-checks Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 7fd95badb0a..8d971001348 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -18,7 +18,7 @@ jobs: # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta # tags are cut too often to spend a multi-minute chaos run on each one, and the weekly # schedule already covers the default branch. - classify-if-stable-release: + run-stable-release-checks: name: Validate inputs and decide whether this tag is gated runs-on: ubuntu-latest outputs: @@ -53,8 +53,8 @@ jobs: redis-chaos-gate: name: Redis Chaos E2E - needs: classify-if-stable-release - if: needs.classify-if-stable-release.outputs.gated == 'true' + needs: run-stable-release-checks + if: needs.run-stable-release-checks.outputs.gated == 'true' permissions: contents: read uses: ./.github/workflows/test-e2e-redis-chaos.yml @@ -63,8 +63,8 @@ jobs: release: name: Create Release - needs: [classify-if-stable-release, redis-chaos-gate] - if: always() && needs.classify-if-stable-release.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') + needs: [run-stable-release-checks, redis-chaos-gate] + if: always() && needs.run-stable-release-checks.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') runs-on: ubuntu-latest permissions: contents: write From 06fe2691c3fd151db136d4061766816359bc9e3c Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 14:59:18 -0700 Subject: [PATCH 122/157] fix(proxy): count when a grouped log page starts past the last one An out-of-range cursorless page returns nothing, and reading its total off the offset reported more sessions than exist (page 4 of 100 sessions at page size 50 claimed 150). Only a page that holds rows, or the first page, ends the list; anything past it falls back to the bounded count. Claude-Session: https://claude.ai/code/session_01ESi9JwaXDww1vP3Qsrr4Mz --- .../spend_management_endpoints.py | 10 +++--- .../test_spend_management_endpoints.py | 35 +++++++++++++++++++ 2 files changed, 41 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 582622cfb50..8cfb6354dd0 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -2979,9 +2979,10 @@ async def _ui_session_grouped_spend_logs( by its newest non-MCP row, enriched by ``_build_ui_spend_logs_response`` exactly like the flat listing, and the response carries ``next_session_cursor`` / ``has_more`` while ``total`` counts sessions - (capped like the flat total). A page that runs out of sessions is itself - the end of the list, so its ``total`` is ``offset + len(page)`` and the - grouped count query is skipped. + (capped like the flat total). A page that runs out of sessions while still + holding some is itself the end of the list, so its ``total`` is + ``offset + len(page)`` and the grouped count query is skipped; a page that + starts past the end says nothing about the total, so that one is counted. """ where_clause: Final = " AND ".join(sql_conditions) if sql_conditions else "TRUE" cmp_op: Final = "<" if sort_desc else ">" @@ -3026,7 +3027,8 @@ async def _ui_session_grouped_spend_logs( else None ) - page_ends_the_list: Final = cursor is None and page_limit > 0 and not has_more + page_starts_inside_the_list: Final = offset == 0 or len(page_rows) > 0 + page_ends_the_list: Final = cursor is None and page_limit > 0 and not has_more and page_starts_inside_the_list total_records, total_is_capped = ( (offset + len(page_rows), False) if page_ends_the_list diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index c709ac77a0b..6e43ac4a12b 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -6839,6 +6839,41 @@ async def test_ui_view_spend_logs_group_by_session_short_page_totals_itself(clie app.dependency_overrides.pop(ps.user_api_key_auth, None) +@pytest.mark.asyncio +async def test_ui_view_spend_logs_group_by_session_page_past_the_end_keeps_the_real_total(client, monkeypatch): + """An empty page past the last one says nothing about the total, so it is counted rather than inferred.""" + sessions = tuple((f"sess-{index:02d}", f"2026-08-29 10:{59 - index:02d}:00") for index in range(100)) + mock_prisma = _session_grouped_paginating_prisma(sessions) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma) + monkeypatch.setattr( + "litellm.proxy.spend_tracking.spend_management_endpoints._is_admin_view_safe", + lambda user_api_key_dict: True, + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_user" + ) + try: + start_date, end_date = _default_date_range() + response = client.get( + "/spend/logs/ui", + params={ + "start_date": start_date, + "end_date": end_date, + "group_by_session": "true", + "page": 4, + "page_size": 50, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + assert response.status_code == 200, response.text + data = response.json() + assert data["data"] == [] + assert data["total"] == 100, "the empty page's offset is not a total" + assert data["total_pages"] == 2 + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + @pytest.mark.asyncio async def test_ui_view_spend_logs_group_by_session_page_past_count_cap_is_empty(client, monkeypatch): """The last page inside the capped total still lists sessions; the page after it is empty and costs no query.""" From 39324dc96038ff8a9f0b43fb07da9cec26d315b1 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 14:59:35 -0700 Subject: [PATCH 123/157] ci(release): rename redis-chaos-gate to redis-chaos-check Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 8d971001348..c51bf02bdc7 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -51,7 +51,7 @@ jobs: echo "${TAG} is a stable or RC tag and must pass the chaos gate" fi - redis-chaos-gate: + redis-chaos-check: name: Redis Chaos E2E needs: run-stable-release-checks if: needs.run-stable-release-checks.outputs.gated == 'true' @@ -63,8 +63,8 @@ jobs: release: name: Create Release - needs: [run-stable-release-checks, redis-chaos-gate] - if: always() && needs.run-stable-release-checks.result == 'success' && (needs.redis-chaos-gate.result == 'success' || needs.redis-chaos-gate.result == 'skipped') + needs: [run-stable-release-checks, redis-chaos-check] + if: always() && needs.run-stable-release-checks.result == 'success' && (needs.redis-chaos-check.result == 'success' || needs.redis-chaos-check.result == 'skipped') runs-on: ubuntu-latest permissions: contents: write From 7d8f2c9ad3fa9c4ed0409ffb1a36663ab6f0e97a Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 15:00:57 -0700 Subject: [PATCH 124/157] ci(e2e): drop the weekly cron for the Redis chaos test Now that create-release.yml gates stable and RC releases on this test directly, the weekly schedule is redundant: every release gets a run against its own commit instead of whatever happened to be on the default branch that Saturday. workflow_dispatch stays for manual runs. Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 3 +-- .github/workflows/test-e2e-redis-chaos.yml | 5 +---- tests/e2e/CLAUDE.md | 2 +- tests/e2e/CONTRIBUTING.md | 2 +- tests/e2e/load/test_redis_chaos_e2e.py | 2 +- 5 files changed, 5 insertions(+), 9 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index c51bf02bdc7..5bf10ef324f 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -16,8 +16,7 @@ permissions: {} jobs: # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta - # tags are cut too often to spend a multi-minute chaos run on each one, and the weekly - # schedule already covers the default branch. + # tags are cut too often to spend a multi-minute chaos run on each one. run-stable-release-checks: name: Validate inputs and decide whether this tag is gated runs-on: ubuntu-latest diff --git a/.github/workflows/test-e2e-redis-chaos.yml b/.github/workflows/test-e2e-redis-chaos.yml index 9deca013603..c7412a63334 100644 --- a/.github/workflows/test-e2e-redis-chaos.yml +++ b/.github/workflows/test-e2e-redis-chaos.yml @@ -1,8 +1,6 @@ -name: "Weekly Redis Chaos E2E" +name: "Redis Chaos E2E" on: - schedule: - - cron: "0 12 * * 6" workflow_dispatch: workflow_call: inputs: @@ -16,7 +14,6 @@ permissions: jobs: redis-chaos-e2e: - if: github.event_name != 'schedule' || github.repository == 'BerriAI/litellm' runs-on: ubuntu-latest-16-cores timeout-minutes: 30 services: diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index b8e9c8ccde3..20cd0c35465 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven weekly by `.github/workflows/test-e2e-redis-chaos.yml`, which `create-release.yml` also calls to gate stable and RC tags on the commit being released), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven by `.github/workflows/test-e2e-redis-chaos.yml`, which `create-release.yml` calls to gate stable and RC tags on the commit being released), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index b1331de190a..b6faebed867 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -65,7 +65,7 @@ A couple of logging destinations are configured on the proxy rather than by the ### The pull request check -Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots on a weekly schedule +Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots when `create-release.yml` calls it to gate a stable or RC release Every selected file must execute at least one passing test in each pass, and any test failure, collection error, or entirely skipped or deselected file fails the check. A file whose tests are all marked skip therefore cannot pass this check, so unskip at least one of them, or add the file to `UNSUPPORTED` in `select_tests.py` with the reason, before changing one. A failed pass stops the run. The public log prints pytest's one-line summary for each pass, including the rerun count, and names each failed or errored test as `classname::name`, so a retried network error or a failing test is visible without the raw output. The final `e2e-changed-tests` job succeeds only when no supported test files changed or the approved run completed all three passes. Fork PRs with selected tests fail this gate until a maintainer brings the reviewed change onto a same-repository branch diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 0732f725f34..9a9e8082e5b 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -76,7 +76,7 @@ REDIS_PAUSE_MS: Final = int(CHAOS_SECONDS * 1000) # identical local runs, so it stays loose; CPU per request held steady at 1.33x-1.36x across the # same runs, so it sits close to what is actually measured. That makes CPU the likeliest of these # to flake first on a runner whose core count shifts how much of baseline CPU is fixed per-request -# work: loosen it rather than widening the others if a weekly run trips it without a real cause. +# work: loosen it rather than widening the others if a CI run trips it without a real cause. CHAOS_RSS_RATIO_CEILING: Final = 2.0 CHAOS_CPU_PER_REQUEST_RATIO_CEILING: Final = 2.0 From d4e083348ca5edfbd45142fd23dbe08f3f3cb9d0 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 15:14:19 -0700 Subject: [PATCH 125/157] revert: drop the create-release.yml gating and E2E_REDIS_CHAOS opt-in create-release.yml is back to calling the chaos test through no mechanism at all; it never called it. Also drops the E2E_REDIS_CHAOS opt-in gate itself: the redis_chaos marker still exists for -m selection and is still excluded from the per-PR selector by path (tests/e2e/(ui|claude_code|load)/), but the test no longer needs an env var to run once its file is targeted. Co-Authored-By: Claude Code --- .github/workflows/create-release.yml | 41 +++------------------- .github/workflows/test-e2e-redis-chaos.yml | 1 - tests/e2e/CLAUDE.md | 2 +- tests/e2e/CONTRIBUTING.md | 2 +- tests/e2e/conftest.py | 2 +- tests/e2e/e2e_config.py | 1 - tests/e2e/load/conftest.py | 8 ++--- tests/e2e/load/test_redis_chaos_e2e.py | 3 +- tests/e2e/pytest.ini | 2 +- 9 files changed, 11 insertions(+), 51 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 5bf10ef324f..0ad84cd3ceb 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -15,13 +15,11 @@ on: permissions: {} jobs: - # Stable and RC tags are gated on the Redis chaos load test; dev, nightly, alpha and beta - # tags are cut too often to spend a multi-minute chaos run on each one. - run-stable-release-checks: - name: Validate inputs and decide whether this tag is gated + release: + name: Create Release runs-on: ubuntu-latest - outputs: - gated: ${{ steps.decide.outputs.gated }} + permissions: + contents: write steps: - name: Validate inputs env: @@ -37,37 +35,6 @@ jobs: exit 1 fi - - name: Decide - id: decide - env: - TAG: ${{ inputs.tag }} - run: | - if echo "${TAG}" | grep -qiE '(nightly|alpha|beta|[-.]dev)'; then - echo "gated=false" >> "$GITHUB_OUTPUT" - echo "${TAG} is a pre-release that skips the chaos gate" - else - echo "gated=true" >> "$GITHUB_OUTPUT" - echo "${TAG} is a stable or RC tag and must pass the chaos gate" - fi - - redis-chaos-check: - name: Redis Chaos E2E - needs: run-stable-release-checks - if: needs.run-stable-release-checks.outputs.gated == 'true' - permissions: - contents: read - uses: ./.github/workflows/test-e2e-redis-chaos.yml - with: - ref: ${{ inputs.commit_hash }} - - release: - name: Create Release - needs: [run-stable-release-checks, redis-chaos-check] - if: always() && needs.run-stable-release-checks.result == 'success' && (needs.redis-chaos-check.result == 'success' || needs.redis-chaos-check.result == 'skipped') - runs-on: ubuntu-latest - permissions: - contents: write - steps: - name: Create release env: TAG: ${{ inputs.tag }} diff --git a/.github/workflows/test-e2e-redis-chaos.yml b/.github/workflows/test-e2e-redis-chaos.yml index c7412a63334..6b066739c95 100644 --- a/.github/workflows/test-e2e-redis-chaos.yml +++ b/.github/workflows/test-e2e-redis-chaos.yml @@ -91,7 +91,6 @@ jobs: - name: Run the Redis chaos load test env: - E2E_REDIS_CHAOS: "1" LITELLM_PROXY_URL: http://localhost:4000 REDIS_HOST: 127.0.0.1 REDIS_PORT: "6379" diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 20cd0c35465..f9ad73b4508 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set, driven by `.github/workflows/test-e2e-redis-chaos.yml`, which `create-release.yml` calls to gate stable and RC tags on the commit being released), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index b6faebed867..a82e8ea43f5 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -65,7 +65,7 @@ A couple of logging destinations are configured on the proxy rather than by the ### The pull request check -Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots when `create-release.yml` calls it to gate a stable or RC release +Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots Every selected file must execute at least one passing test in each pass, and any test failure, collection error, or entirely skipped or deselected file fails the check. A file whose tests are all marked skip therefore cannot pass this check, so unskip at least one of them, or add the file to `UNSUPPORTED` in `select_tests.py` with the reason, before changing one. A failed pass stops the run. The public log prints pytest's one-line summary for each pass, including the rerun count, and names each failed or errored test as `classname::name`, so a retried network error or a failing test is visible without the raw output. The final `e2e-changed-tests` job succeeds only when no supported test files changed or the approved run completed all three passes. Fork PRs with selected tests fail this gate until a maintainer brings the reviewed change onto a same-repository branch diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index 700dc822d59..72448201f8b 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -61,7 +61,7 @@ def pytest_configure(config: pytest.Config) -> None: config.addinivalue_line( "markers", "redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from " - "gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set", + "gateway/redis_chaos_ci_config.yml on the same host", ) diff --git a/tests/e2e/e2e_config.py b/tests/e2e/e2e_config.py index 7344a2b6cec..86f0c616c04 100644 --- a/tests/e2e/e2e_config.py +++ b/tests/e2e/e2e_config.py @@ -143,7 +143,6 @@ LOAD_MIN_CONCURRENCY_EFFICIENCY = float(os.environ.get("E2E_LOAD_MIN_CONCURRENCY WEEKLY_ANOMALY_OPT_IN_ENV = "E2E_WEEKLY_ANOMALY" MANAGED_FILES_OPT_IN_ENV = "E2E_MANAGED_FILES_STACK" -REDIS_CHAOS_OPT_IN_ENV = "E2E_REDIS_CHAOS" ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6")) ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6")) ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3")) diff --git a/tests/e2e/load/conftest.py b/tests/e2e/load/conftest.py index e6f3c7aa538..7c749837225 100644 --- a/tests/e2e/load/conftest.py +++ b/tests/e2e/load/conftest.py @@ -3,15 +3,11 @@ from __future__ import annotations import os import pytest - -from e2e_config import REDIS_CHAOS_OPT_IN_ENV, WEEKLY_ANOMALY_OPT_IN_ENV +from e2e_config import WEEKLY_ANOMALY_OPT_IN_ENV from load_client import LoadClient, build_client from proxy_client import ProxyClient -_OPT_IN_MARKERS = ( - ("weekly", WEEKLY_ANOMALY_OPT_IN_ENV), - ("redis_chaos", REDIS_CHAOS_OPT_IN_ENV), -) +_OPT_IN_MARKERS = (("weekly", WEEKLY_ANOMALY_OPT_IN_ENV),) def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None: diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 9a9e8082e5b..2e1e2a92bd3 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -27,8 +27,7 @@ per-phase RSS, CPU, and log-bytes budgets are here to catch. Needs the proxy on the same host, since RSS and CPU come from psutil on its process tree: a multi-worker proxy serves /metrics from the prometheus multiprocess collector, which drops the process collector's memory and CPU series. Log bytes are read from the file the proxy's -stdout/stderr was redirected to, so the same host requirement covers that too. Deselected -unless E2E_REDIS_CHAOS is set. +stdout/stderr was redirected to, so the same host requirement covers that too. """ from __future__ import annotations diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index 774d9644497..28f7fc011f7 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -9,4 +9,4 @@ markers = load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set managed_files: needs a proxy running with require_managed_files enabled; deselected unless E2E_MANAGED_FILES_STACK is set - redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set + redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host From ffc16a4b0e467005b8e4387fb0cc449f57cf009c Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 11 Sep 2026 22:46:31 +0000 Subject: [PATCH 126/157] test(e2e): restore the E2E_REDIS_CHAOS opt-in for the redis chaos test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .github/workflows/test-e2e-redis-chaos.yml | 1 + tests/e2e/CLAUDE.md | 2 +- tests/e2e/CONTRIBUTING.md | 2 +- tests/e2e/conftest.py | 2 +- tests/e2e/e2e_config.py | 1 + tests/e2e/load/conftest.py | 7 +++++-- tests/e2e/load/test_redis_chaos_e2e.py | 3 ++- tests/e2e/pytest.ini | 2 +- 8 files changed, 13 insertions(+), 7 deletions(-) diff --git a/.github/workflows/test-e2e-redis-chaos.yml b/.github/workflows/test-e2e-redis-chaos.yml index 6b066739c95..c7412a63334 100644 --- a/.github/workflows/test-e2e-redis-chaos.yml +++ b/.github/workflows/test-e2e-redis-chaos.yml @@ -91,6 +91,7 @@ jobs: - name: Run the Redis chaos load test env: + E2E_REDIS_CHAOS: "1" LITELLM_PROXY_URL: http://localhost:4000 REDIS_HOST: 127.0.0.1 REDIS_PORT: "6379" diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index f9ad73b4508..8be2e7b4ce2 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -18,7 +18,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `logging/` - logging-integration delivery (datadog and friends) - `security/` - secret handling and log-leak protection - `router/` - routing and reliability behavior (fallbacks, cooldowns) -- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml`), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic +- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index a82e8ea43f5..f1f7b17d86b 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -65,7 +65,7 @@ A couple of logging destinations are configured on the proxy rather than by the ### The pull request check -Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots +Every same-repository PR that adds, modifies, or renames a `tests/e2e/**/test_*.py` file runs those changed files three times. A change to the harness itself, meaning a root-level `tests/e2e/*.py` file or `pytest.ini`, `tests/e2e/gateway/`, `.github/e2e-stack/`, or the workflow, also runs the `access_control` suite as a canary, because those files have no test of their own that exercises the stack. `.github/e2e-stack/select_tests.py` applies both rules. The stack config at `tests/e2e/gateway/stage_mirror_ci_config.yml` must declare every model the selected suites use; a missing one shows up as a failed test id in the public log. The suite's own single rerun for network errors and 5xx responses (see `pytest.ini`) applies on every pass, so a transport blip does not fail the check while a race inside a test still does. The stage-mirror stack has a control-plane backend, two gateways behind nginx, Postgres, Jaeger, and TLS cluster-mode Valkey. The stack exports every gateway address in `LITELLM_PROXY_REPLICA_URLS`, so model registration waits until each gateway lists the new model rather than whichever one the load balancer answered from. Documentation, deleted-file, and application-only changes do not start the stack or request environment approval. The `ui/`, `claude_code/`, and `load/` directories, `batches/test_managed_files_enforcement_e2e.py`, `llm_translation/realtime/test_realtime_pipecat_audio_e2e.py`, and `guardrails/test_presidio_masking_e2e.py` remain outside this check because they use separate tooling or need a differently configured stack: the pipecat audio suite skips itself at import time unless the NLTK `punkt_tab` data is installed, and the presidio suite fails without the analyzer and anonymizer services this stack does not start. The Redis chaos test under `load/` needs a proxy it can pause the Redis of on the same host (`gateway/redis_chaos_ci_config.yml`), which `.github/workflows/test-e2e-redis-chaos.yml` boots, and which the Buildkite `e2e-redis-chaos` step in project-releaser runs co-located with Postgres and Valkey in one pod; it is deselected unless `E2E_REDIS_CHAOS` is set Every selected file must execute at least one passing test in each pass, and any test failure, collection error, or entirely skipped or deselected file fails the check. A file whose tests are all marked skip therefore cannot pass this check, so unskip at least one of them, or add the file to `UNSUPPORTED` in `select_tests.py` with the reason, before changing one. A failed pass stops the run. The public log prints pytest's one-line summary for each pass, including the rerun count, and names each failed or errored test as `classname::name`, so a retried network error or a failing test is visible without the raw output. The final `e2e-changed-tests` job succeeds only when no supported test files changed or the approved run completed all three passes. Fork PRs with selected tests fail this gate until a maintainer brings the reviewed change onto a same-repository branch diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index 72448201f8b..700dc822d59 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -61,7 +61,7 @@ def pytest_configure(config: pytest.Config) -> None: config.addinivalue_line( "markers", "redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from " - "gateway/redis_chaos_ci_config.yml on the same host", + "gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set", ) diff --git a/tests/e2e/e2e_config.py b/tests/e2e/e2e_config.py index 86f0c616c04..7344a2b6cec 100644 --- a/tests/e2e/e2e_config.py +++ b/tests/e2e/e2e_config.py @@ -143,6 +143,7 @@ LOAD_MIN_CONCURRENCY_EFFICIENCY = float(os.environ.get("E2E_LOAD_MIN_CONCURRENCY WEEKLY_ANOMALY_OPT_IN_ENV = "E2E_WEEKLY_ANOMALY" MANAGED_FILES_OPT_IN_ENV = "E2E_MANAGED_FILES_STACK" +REDIS_CHAOS_OPT_IN_ENV = "E2E_REDIS_CHAOS" ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6")) ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6")) ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3")) diff --git a/tests/e2e/load/conftest.py b/tests/e2e/load/conftest.py index 7c749837225..fa608d157cc 100644 --- a/tests/e2e/load/conftest.py +++ b/tests/e2e/load/conftest.py @@ -3,11 +3,14 @@ from __future__ import annotations import os import pytest -from e2e_config import WEEKLY_ANOMALY_OPT_IN_ENV +from e2e_config import REDIS_CHAOS_OPT_IN_ENV, WEEKLY_ANOMALY_OPT_IN_ENV from load_client import LoadClient, build_client from proxy_client import ProxyClient -_OPT_IN_MARKERS = (("weekly", WEEKLY_ANOMALY_OPT_IN_ENV),) +_OPT_IN_MARKERS = ( + ("weekly", WEEKLY_ANOMALY_OPT_IN_ENV), + ("redis_chaos", REDIS_CHAOS_OPT_IN_ENV), +) def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]) -> None: diff --git a/tests/e2e/load/test_redis_chaos_e2e.py b/tests/e2e/load/test_redis_chaos_e2e.py index 2e1e2a92bd3..9a9e8082e5b 100644 --- a/tests/e2e/load/test_redis_chaos_e2e.py +++ b/tests/e2e/load/test_redis_chaos_e2e.py @@ -27,7 +27,8 @@ per-phase RSS, CPU, and log-bytes budgets are here to catch. Needs the proxy on the same host, since RSS and CPU come from psutil on its process tree: a multi-worker proxy serves /metrics from the prometheus multiprocess collector, which drops the process collector's memory and CPU series. Log bytes are read from the file the proxy's -stdout/stderr was redirected to, so the same host requirement covers that too. +stdout/stderr was redirected to, so the same host requirement covers that too. Deselected +unless E2E_REDIS_CHAOS is set. """ from __future__ import annotations diff --git a/tests/e2e/pytest.ini b/tests/e2e/pytest.ini index 28f7fc011f7..774d9644497 100644 --- a/tests/e2e/pytest.ini +++ b/tests/e2e/pytest.ini @@ -9,4 +9,4 @@ markers = load: heavy throughput/load test; collected last so it never perturbs latency-sensitive suites weekly: real-provider anomaly load test that spends real money; deselected unless E2E_WEEKLY_ANOMALY is set managed_files: needs a proxy running with require_managed_files enabled; deselected unless E2E_MANAGED_FILES_STACK is set - redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host + redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set From 5e23db8e03d770d1dee78a5d7b3e4903619cb95e Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 16:22:55 -0700 Subject: [PATCH 127/157] feat(ocr): add Azure Mistral adapter with native authentication (#40502) * feat(ocr): move Azure credential resolution to Rust * fix(auth): keep shared primitives warning-free * fix(auth): preserve missing key provider errors * fix(auth): enforce Azure input provenance * fix(ocr): preserve proxy credential provenance --- .github/scripts/verify_linux_native_wheel.py | 4 +- litellm-rust/Cargo.lock | 605 ++++++++++++++- litellm-rust/Cargo.toml | 5 + .../crates/ai-gateway/src/io/realtime.rs | 6 +- .../crates/ai-gateway/src/io/responses_ws.rs | 6 +- litellm-rust/crates/core/Cargo.toml | 6 + .../crates/core/src/auth/credential.rs | 168 +++++ litellm-rust/crates/core/src/auth/error.rs | 120 +++ litellm-rust/crates/core/src/auth/http.rs | 86 +++ litellm-rust/crates/core/src/auth/mod.rs | 55 ++ litellm-rust/crates/core/src/auth/policy.rs | 114 +++ litellm-rust/crates/core/src/auth/secret.rs | 41 + litellm-rust/crates/core/src/auth/token.rs | 47 ++ litellm-rust/crates/core/src/error.rs | 19 +- litellm-rust/crates/core/src/lib.rs | 2 + .../core/src/ocr/adapters/azure_mistral.rs | 129 +++- litellm-rust/crates/core/src/ocr/types.rs | 10 + litellm-rust/crates/core/src/ocr/wire.rs | 15 + .../anthropic/messages/transformation.rs | 9 +- .../auth/credential_provider_cache.rs | 43 ++ .../core/src/providers/azure_ai/auth/mod.rs | 7 + .../src/providers/azure_ai/auth/native.rs | 702 ++++++++++++++++++ .../src/providers/azure_ai/auth/resolve.rs | 683 +++++++++++++++++ .../core/src/providers/azure_ai/auth/types.rs | 195 +++++ .../azure_ai/messages/transformation.rs | 16 +- .../crates/core/src/providers/azure_ai/mod.rs | 1 + .../crates/core/tests/azure_ai_ocr.rs | 22 + litellm-rust/crates/core/tests/ocr.rs | 2 + litellm-rust/crates/core/tests/ocr/support.rs | 1 + .../python-bridge/src/routes/definition.rs | 2 +- .../crates/python-bridge/src/routes/ocr.rs | 9 + litellm/ocr/main.py | 454 ++++++----- litellm/proxy/litellm_pre_call_utils.py | 1 + litellm/rust_bridge/ocr.py | 22 +- ...cr_azure_document_intelligence_api_base.py | 24 +- .../ocr/test_ocr_native_format.py | 41 +- tests/test_litellm/ocr/test_rust_bridge.py | 431 +++++++++-- .../proxy/test_litellm_pre_call_utils.py | 3 + .../rust_bridge/native_route_wheel_test.py | 2 +- 39 files changed, 3767 insertions(+), 341 deletions(-) create mode 100644 litellm-rust/crates/core/src/auth/credential.rs create mode 100644 litellm-rust/crates/core/src/auth/error.rs create mode 100644 litellm-rust/crates/core/src/auth/http.rs create mode 100644 litellm-rust/crates/core/src/auth/mod.rs create mode 100644 litellm-rust/crates/core/src/auth/policy.rs create mode 100644 litellm-rust/crates/core/src/auth/secret.rs create mode 100644 litellm-rust/crates/core/src/auth/token.rs create mode 100644 litellm-rust/crates/core/src/providers/azure_ai/auth/credential_provider_cache.rs create mode 100644 litellm-rust/crates/core/src/providers/azure_ai/auth/mod.rs create mode 100644 litellm-rust/crates/core/src/providers/azure_ai/auth/native.rs create mode 100644 litellm-rust/crates/core/src/providers/azure_ai/auth/resolve.rs create mode 100644 litellm-rust/crates/core/src/providers/azure_ai/auth/types.rs diff --git a/.github/scripts/verify_linux_native_wheel.py b/.github/scripts/verify_linux_native_wheel.py index 899e2a211c0..4fb8f068eb0 100644 --- a/.github/scripts/verify_linux_native_wheel.py +++ b/.github/scripts/verify_linux_native_wheel.py @@ -205,7 +205,7 @@ def main( native_module: Final = load_native_module(native_path) native_module_loads: Final = native_module is not None panic_test_hook_absent: Final = native_module is not None and not hasattr(native_module, "_panic_for_test") - native_size_limit: Final = 20_000_000 + native_size_limit: Final = 25_000_000 native_size_within_limit: Final = native_member.file_size <= native_size_limit validations: Final = ( (f"Python tag is {EXPECTED_PYTHON_TAG}", python_tag == EXPECTED_PYTHON_TAG), @@ -222,7 +222,7 @@ def main( ("Python extension entry point is present", extension_entry_point_present), ("Native module loads", native_module_loads), ("Production module omits the panic test hook", panic_test_hook_absent), - ("Native extension does not exceed 20 MB", native_size_within_limit), + ("Native extension does not exceed 25 MB", native_size_within_limit), ("Wheel contents are valid", not unexpected_members), ) diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock index a7c514270da..9c0a8cb7fe7 100644 --- a/litellm-rust/Cargo.lock +++ b/litellm-rust/Cargo.lock @@ -2,6 +2,12 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "adler2" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" + [[package]] name = "ahash" version = "0.8.12" @@ -55,6 +61,29 @@ dependencies = [ "rustversion", ] +[[package]] +name = "async-compression" +version = "0.4.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4f10dafd0c8d2e51ae9a748805777613ed0bbe17bf586b76c8311f45c020a32f" +dependencies = [ + "compression-codecs", + "compression-core", + "pin-project-lite", + "tokio", +] + +[[package]] +name = "async-lock" +version = "3.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "290f7f2596bd5b78a9fec8088ccd89180d7f9f55b94b0576823bbbdc72ee8311" +dependencies = [ + "event-listener", + "event-listener-strategy", + "pin-project-lite", +] + [[package]] name = "async-trait" version = "0.1.91" @@ -482,6 +511,58 @@ dependencies = [ "tracing", ] +[[package]] +name = "azure_core" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e41cbd819986ba41904c207d8ffc4106f8f8352a548d773e9554906379bb2fb" +dependencies = [ + "async-lock", + "async-trait", + "azure_core_macros", + "bytes", + "futures", + "pin-project", + "rustc_version", + "serde", + "serde_json", + "tokio", + "tracing", + "typespec", + "typespec_client_core", +] + +[[package]] +name = "azure_core_macros" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9b52dba6a345f3ad2d42ff8d0d63df9d0994cfa29657bf18ffdbf149f78a4f5" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "tracing", +] + +[[package]] +name = "azure_identity" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32edf96b356ca7c51d7590c4925cc36efc3947a5da4468e8e0b25c56ecbb3de5" +dependencies = [ + "async-lock", + "async-trait", + "azure_core", + "futures", + "pin-project", + "serde", + "serde_json", + "time", + "tokio", + "tracing", + "url", +] + [[package]] name = "base64" version = "0.13.1" @@ -494,6 +575,12 @@ version = "0.22.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" +[[package]] +name = "base64" +version = "0.23.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" + [[package]] name = "base64-simd" version = "0.8.0" @@ -673,6 +760,16 @@ version = "0.5.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0c9ea0ac24bc397ab3c98583a3c9ba74fa56b09a4449bbe172b9b1ddb016027a" +[[package]] +name = "combine" +version = "4.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfc320937d09e6de266b31b9afb480f197d7a861be86be7cb2ea7e5d1bfffc5e" +dependencies = [ + "bytes", + "memchr", +] + [[package]] name = "compact_str" version = "0.9.1" @@ -688,6 +785,23 @@ dependencies = [ "static_assertions", ] +[[package]] +name = "compression-codecs" +version = "0.4.41" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "58a6d0db8759036a783bc7c3f7a07f8cef3bf9470eb1db3bc86e8bcd1c5d0fe8" +dependencies = [ + "compression-core", + "flate2", + "memchr", +] + +[[package]] +name = "compression-core" +version = "0.4.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e8ccc4ea9f6acc32d102c0f6d471d11d913ad15f20c04de743374861fa1d414" + [[package]] name = "const-oid" version = "0.10.2" @@ -728,6 +842,15 @@ dependencies = [ "libc", ] +[[package]] +name = "crc32fast" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" +dependencies = [ + "cfg-if", +] + [[package]] name = "criterion" version = "0.8.2" @@ -763,6 +886,15 @@ dependencies = [ "itertools 0.13.0", ] +[[package]] +name = "crossbeam-channel" +version = "0.5.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "98b0cc327b5bc766e7fda9c9260cc0fa81b43a8e240440422dff70788e3f9ef1" +dependencies = [ + "crossbeam-utils", +] + [[package]] name = "crossbeam-deque" version = "0.8.7" @@ -889,6 +1021,9 @@ name = "deranged" version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" +dependencies = [ + "serde_core", +] [[package]] name = "derive_builder" @@ -960,6 +1095,12 @@ version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" +[[package]] +name = "dyn-clone" +version = "1.0.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" + [[package]] name = "either" version = "1.16.0" @@ -972,12 +1113,42 @@ version = "1.0.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + [[package]] name = "esaxx-rs" version = "0.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d817e038c30374a4bcb22f94d0a8a0e216958d4c3dcde369b1439fec4bdda6e6" +[[package]] +name = "event-listener" +version = "5.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" +dependencies = [ + "parking", + "pin-project-lite", +] + +[[package]] +name = "event-listener-strategy" +version = "0.5.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8be9f3dfaaffdae2972880079a491a1a8bb7cbed0b8dd7a347f668b4150a3b93" +dependencies = [ + "event-listener", + "pin-project-lite", +] + [[package]] name = "fastrand" version = "2.5.0" @@ -990,6 +1161,17 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +[[package]] +name = "flate2" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +dependencies = [ + "crc32fast", + "miniz_oxide", + "zlib-rs", +] + [[package]] name = "fnv" version = "1.0.7" @@ -1011,6 +1193,21 @@ version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" +[[package]] +name = "futures" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a88cf1f829d945f548cf8fec32c61b1f202b6d93b45848602fc02af4b12ad218" +dependencies = [ + "futures-channel", + "futures-core", + "futures-executor", + "futures-io", + "futures-sink", + "futures-task", + "futures-util", +] + [[package]] name = "futures-channel" version = "0.3.33" @@ -1027,6 +1224,17 @@ version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" +[[package]] +name = "futures-executor" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6754879cc9f2c66f88c6e5c35344bb0bdb0708b0352b1201815667c7eabc7458" +dependencies = [ + "futures-core", + "futures-task", + "futures-util", +] + [[package]] name = "futures-io" version = "0.3.33" @@ -1068,6 +1276,7 @@ version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" dependencies = [ + "futures-channel", "futures-core", "futures-io", "futures-macro", @@ -1537,6 +1746,55 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "jni" +version = "0.22.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5efd9a482cf3a427f00d6b35f14332adc7902ce91efb778580e180ff90fa3498" +dependencies = [ + "cfg-if", + "combine", + "jni-macros", + "jni-sys", + "log", + "simd_cesu8", + "thiserror 2.0.19", + "walkdir", + "windows-link", +] + +[[package]] +name = "jni-macros" +version = "0.22.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a00109accc170f0bdb141fed3e393c565b6f5e072365c3bd58f5b062591560a3" +dependencies = [ + "proc-macro2", + "quote", + "rustc_version", + "simd_cesu8", + "syn 2.0.119", +] + +[[package]] +name = "jni-sys" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6377a88cb3910bee9b0fa88d4f42e1d2da8e79915598f65fb0c7ee14c878af2" +dependencies = [ + "jni-sys-macros", +] + +[[package]] +name = "jni-sys-macros" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" +dependencies = [ + "quote", + "syn 2.0.119", +] + [[package]] name = "jobserver" version = "0.1.35" @@ -1580,7 +1838,7 @@ dependencies = [ "futures-util", "litellm-config", "litellm-core", - "reqwest", + "reqwest 0.12.28", "rustls 0.23.42", "rustls-native-certs", "serde", @@ -1613,20 +1871,26 @@ dependencies = [ "aws-sigv4", "aws-smithy-runtime-api", "aws-types", + "azure_core", + "azure_identity", "base64 0.22.1", "data-url", + "moka", "rand 0.8.7", - "reqwest", + "reqwest 0.12.28", "rstest", "serde", "serde_json", "serde_path_to_error", "sha2 0.10.9", + "strum", + "subtle", "thiserror 2.0.19", "tokio", "tracing", "tracing-subscriber", "url", + "veil", ] [[package]] @@ -1681,6 +1945,15 @@ version = "0.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +[[package]] +name = "lock_api" +version = "0.4.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965" +dependencies = [ + "scopeguard", +] + [[package]] name = "log" version = "0.4.33" @@ -1743,6 +2016,16 @@ version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a" +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + [[package]] name = "mio" version = "1.2.2" @@ -1754,6 +2037,26 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "moka" +version = "0.12.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4293f18e7567a1caf3c584855554377025c65e0aa445344d04171f5ad63d19b9" +dependencies = [ + "async-lock", + "crossbeam-channel", + "crossbeam-epoch", + "crossbeam-utils", + "equivalent", + "event-listener", + "futures-util", + "parking_lot", + "portable-atomic", + "smallvec", + "tagptr", + "uuid", +] + [[package]] name = "monostate" version = "0.1.18" @@ -1866,6 +2169,35 @@ dependencies = [ "winapi", ] +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + +[[package]] +name = "parking_lot" +version = "0.12.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a" +dependencies = [ + "lock_api", + "parking_lot_core", +] + +[[package]] +name = "parking_lot_core" +version = "0.9.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1" +dependencies = [ + "cfg-if", + "libc", + "redox_syscall", + "smallvec", + "windows-link", +] + [[package]] name = "paste" version = "1.0.15" @@ -1884,6 +2216,26 @@ version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" +[[package]] +name = "pin-project" +version = "1.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2466b2336ed02bcdca6b294417127b90ec92038d1d5c4fbeac971a922e0e0924" +dependencies = [ + "pin-project-internal", +] + +[[package]] +name = "pin-project-internal" +version = "1.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c96395f0a926bc13b1c17622aaddda1ecb55d49c8f1bf9777e4d877800a43f8b" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "pin-project-lite" version = "0.2.17" @@ -2085,6 +2437,7 @@ version = "0.11.16" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" dependencies = [ + "aws-lc-rs", "bytes", "getrandom 0.4.3", "lru-slab", @@ -2252,6 +2605,15 @@ dependencies = [ "crossbeam-utils", ] +[[package]] +name = "redox_syscall" +version = "0.5.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" +dependencies = [ + "bitflags", +] + [[package]] name = "regex" version = "1.13.1" @@ -2332,11 +2694,49 @@ dependencies = [ "url", "wasm-bindgen", "wasm-bindgen-futures", - "wasm-streams", + "wasm-streams 0.4.2", "web-sys", "webpki-roots", ] +[[package]] +name = "reqwest" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "16a1cfa75cc186dd73d5818e510e042e40927bccc9c236b061cea97e1eb08029" +dependencies = [ + "base64 0.23.1", + "bytes", + "futures-core", + "futures-util", + "http 1.4.2", + "http-body 1.1.0", + "http-body-util", + "hyper 1.10.1", + "hyper-rustls 0.27.9", + "hyper-util", + "js-sys", + "log", + "percent-encoding", + "pin-project-lite", + "quinn", + "rustls 0.23.42", + "rustls-pki-types", + "rustls-platform-verifier", + "sync_wrapper", + "tokio", + "tokio-rustls 0.26.4", + "tokio-util", + "tower", + "tower-http", + "tower-service", + "url", + "wasm-bindgen", + "wasm-bindgen-futures", + "wasm-streams 0.5.0", + "web-sys", +] + [[package]] name = "ring" version = "0.17.14" @@ -2444,6 +2844,33 @@ dependencies = [ "zeroize", ] +[[package]] +name = "rustls-platform-verifier" +version = "0.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "26d1e2536ce4f35f4846aa13bff16bd0ff40157cdb14cc056c7b14ba41233ba0" +dependencies = [ + "core-foundation", + "core-foundation-sys", + "jni", + "log", + "once_cell", + "rustls 0.23.42", + "rustls-native-certs", + "rustls-platform-verifier-android", + "rustls-webpki 0.103.13", + "security-framework", + "security-framework-sys", + "webpki-root-certs", + "windows-sys 0.61.2", +] + +[[package]] +name = "rustls-platform-verifier-android" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f87165f0995f63a9fbeea62b64d10b4d9d8e78ec6d7d51fb2125fda7bb36788f" + [[package]] name = "rustls-webpki" version = "0.101.7" @@ -2496,6 +2923,12 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "scopeguard" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" + [[package]] name = "sct" version = "0.7.1" @@ -2649,6 +3082,38 @@ version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" +[[package]] +name = "signal-hook-registry" +version = "1.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4db69cba1110affc0e9f7bcd48bbf87b3f4fc7c61fc9155afd4c469eb3d6c1b" +dependencies = [ + "errno", + "libc", +] + +[[package]] +name = "simd-adler32" +version = "0.3.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" + +[[package]] +name = "simd_cesu8" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520" +dependencies = [ + "rustc_version", + "simdutf8", +] + +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + [[package]] name = "slab" version = "0.4.12" @@ -2711,6 +3176,27 @@ version = "0.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" +[[package]] +name = "strum" +version = "0.28.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9628de9b8791db39ceda2b119bbe13134770b56c138ec1d3af810d045c04f9bd" +dependencies = [ + "strum_macros", +] + +[[package]] +name = "strum_macros" +version = "0.28.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ab85eea0270ee17587ed4156089e10b9e6880ee688791d45a905f5b1ca36f664" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "subtle" version = "2.6.1" @@ -2759,6 +3245,12 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "tagptr" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b2093cf4c8eb1e67749a6762251bc9cd836b6fc171623bd0a9d324d37af2417" + [[package]] name = "target-lexicon" version = "0.13.5" @@ -2922,6 +3414,7 @@ dependencies = [ "libc", "mio", "pin-project-lite", + "signal-hook-registry", "socket2 0.6.5", "tokio-macros", "windows-sys 0.61.2", @@ -3039,12 +3532,17 @@ version = "0.6.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" dependencies = [ + "async-compression", "bitflags", "bytes", + "futures-core", "futures-util", "http 1.4.2", "http-body 1.1.0", + "http-body-util", "pin-project-lite", + "tokio", + "tokio-util", "tower", "tower-layer", "tower-service", @@ -3138,6 +3636,57 @@ version = "1.20.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" +[[package]] +name = "typespec" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "753a2fe021e407d4fc9ee6f4f0a33403cc306d5c54c4e4ebe1b8cbde0ca052b9" +dependencies = [ + "base64 0.22.1", + "bytes", + "futures", + "serde", + "serde_json", + "url", +] + +[[package]] +name = "typespec_client_core" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0373af0f9d4f580b3a1a9d9639cedaabe015ed262b35bfbe13941bfb14fe1ea6" +dependencies = [ + "async-trait", + "base64 0.22.1", + "bytes", + "dyn-clone", + "futures", + "pin-project", + "rand 0.10.2", + "reqwest 0.13.5", + "serde", + "serde_json", + "time", + "tokio", + "tracing", + "typespec", + "typespec_macros", + "url", + "uuid", +] + +[[package]] +name = "typespec_macros" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2c608f4427943f8adb211abc95c87672b1b98847152783507d54e3246e502f60" +dependencies = [ + "proc-macro2", + "quote", + "rustc_version", + "syn 2.0.119", +] + [[package]] name = "unicase" version = "2.9.0" @@ -3213,10 +3762,32 @@ version = "1.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" dependencies = [ + "getrandom 0.4.3", "js-sys", "wasm-bindgen", ] +[[package]] +name = "veil" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c7352f0bbf3ab98911b0c0277065094c1b1ec79bbc85fa3b7d16bf1859c3d96f" +dependencies = [ + "once_cell", + "veil-macros", +] + +[[package]] +name = "veil-macros" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47a3f4f06d904eb789b935253752ba6bcc1dfa61349f8d5341c66abe070b44e5" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "version_check" version = "0.9.5" @@ -3331,6 +3902,19 @@ dependencies = [ "web-sys", ] +[[package]] +name = "wasm-streams" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d1ec4f6517c9e11ae630e200b2b65d193279042e28edd4a2cda233e46670bbb" +dependencies = [ + "futures-util", + "js-sys", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", +] + [[package]] name = "web-sys" version = "0.3.103" @@ -3351,6 +3935,15 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "webpki-root-certs" +version = "1.0.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b" +dependencies = [ + "rustls-pki-types", +] + [[package]] name = "webpki-roots" version = "1.0.9" @@ -3609,6 +4202,12 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "zlib-rs" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" + [[package]] name = "zmij" version = "1.0.23" diff --git a/litellm-rust/Cargo.toml b/litellm-rust/Cargo.toml index f3e54e5b2aa..0f30ac5cf7d 100644 --- a/litellm-rust/Cargo.toml +++ b/litellm-rust/Cargo.toml @@ -41,8 +41,13 @@ tokio = { version = "1", features = ["rt-multi-thread", "macros", "time", "net"] tokio-tungstenite = { version = "0.24", default-features = false, features = ["connect", "rustls-tls-native-roots"] } futures-util = { version = "0.3", default-features = false, features = ["sink", "std"] } base64 = "0.22" +azure_core = "1.0.0" +azure_identity = { version = "1.0.0", features = ["tokio"] } +moka = { version = "0.12.16", features = ["future"] } +strum = { version = "0.28.0", features = ["derive"] } url = "2.5.8" criterion = "0.8.2" +veil = "0.3.0" [profile.release] opt-level = 3 diff --git a/litellm-rust/crates/ai-gateway/src/io/realtime.rs b/litellm-rust/crates/ai-gateway/src/io/realtime.rs index 207c31dffa0..1aa31adcc38 100644 --- a/litellm-rust/crates/ai-gateway/src/io/realtime.rs +++ b/litellm-rust/crates/ai-gateway/src/io/realtime.rs @@ -15,6 +15,8 @@ use std::time::Duration; use futures_util::stream::{SplitSink, SplitStream}; use futures_util::{Sink, SinkExt, Stream, StreamExt}; +use litellm_core::AuthError; +use litellm_core::auth::error::MissingCredential; use litellm_core::error::Error; use litellm_core::realtime::transformation::RealtimeProviderConfig; use litellm_core::realtime::types::RealtimeEvent; @@ -32,8 +34,6 @@ use crate::io::tls::connect_upstream; /// Environment variable holding the OpenAI API key (last-resort fallback). const OPENAI_API_KEY_ENV: &str = "OPENAI_API_KEY"; -const MISSING_KEY_MESSAGE: &str = "Missing OpenAI API Key - a realtime call is being made but no key was passed via params or the OPENAI_API_KEY environment variable"; - /// Default **idle** timeout: if neither side sends a frame for this long, the /// session is reaped. It resets on any activity, so it does not cap a healthy /// (continuously streaming) session — it only frees a stalled one (e.g. a @@ -59,7 +59,7 @@ pub(crate) fn resolve_api_key(api_key: Option<&str>) -> Result { .ok() .filter(|key| !key.trim().is_empty()) }) - .ok_or_else(|| Error::Auth(MISSING_KEY_MESSAGE.to_string())) + .ok_or_else(|| Error::from(AuthError::from(MissingCredential::OpenAiRealtimeApiKey))) } /// Open the upstream WebSocket to OpenAI for `(model, api_key, api_base)`. diff --git a/litellm-rust/crates/ai-gateway/src/io/responses_ws.rs b/litellm-rust/crates/ai-gateway/src/io/responses_ws.rs index 9df3d0c6cc5..7f3b6b0650f 100644 --- a/litellm-rust/crates/ai-gateway/src/io/responses_ws.rs +++ b/litellm-rust/crates/ai-gateway/src/io/responses_ws.rs @@ -4,7 +4,9 @@ use std::time::Duration; use futures_util::stream::{SplitSink, SplitStream}; use futures_util::{Sink, SinkExt, Stream, StreamExt}; +use litellm_core::AuthError; use litellm_core::Error; +use litellm_core::auth::error::MissingCredential; use litellm_core::providers::openai::responses::transformation::OPENAI_RESPONSES_WS_CONFIG; use litellm_core::responses::types::ResponsesWsEvent; use litellm_core::responses::websocket::ResponsesWebSocketProviderConfig; @@ -23,8 +25,6 @@ use crate::constants::{ }; const OPENAI_API_KEY_ENV: &str = "OPENAI_API_KEY"; -const MISSING_KEY_MESSAGE: &str = "Missing OpenAI API Key - a Responses WebSocket call is being made but no key was passed via params or the OPENAI_API_KEY environment variable"; - pub type ResponsesUpstreamWs = WebSocketStream>; type UpstreamTx = SplitSink; type UpstreamRx = SplitStream; @@ -120,7 +120,7 @@ pub(crate) fn resolve_api_key(api_key: Option<&str>) -> Result { .ok() .filter(|value| !value.trim().is_empty()) }) - .ok_or_else(|| Error::Auth(MISSING_KEY_MESSAGE.to_string())) + .ok_or_else(|| Error::from(AuthError::from(MissingCredential::OpenAiResponsesApiKey))) } async fn dial_upstream( diff --git a/litellm-rust/crates/core/Cargo.toml b/litellm-rust/crates/core/Cargo.toml index 6d0a2fa775b..dc4a1acea16 100644 --- a/litellm-rust/crates/core/Cargo.toml +++ b/litellm-rust/crates/core/Cargo.toml @@ -12,18 +12,24 @@ path = "tests/workspace_crate_allowlist.rs" [dependencies] base64.workspace = true +azure_core.workspace = true +azure_identity.workspace = true data-url = "0.3.2" +moka.workspace = true rand.workspace = true reqwest.workspace = true serde.workspace = true serde_json.workspace = true serde_path_to_error = "0.1" +strum.workspace = true +subtle.workspace = true tokio.workspace = true thiserror.workspace = true tracing.workspace = true tracing-subscriber = { workspace = true, optional = true } sha2.workspace = true url.workspace = true +veil.workspace = true aws-config = { version = "1.9.0", default-features = false, features = ["rustls", "rt-tokio"], optional = true } aws-credential-types = { version = "1.3.0", features = ["hardcoded-credentials"], optional = true } aws-sdk-sts = { version = "1.108.0", default-features = false, features = ["rustls", "rt-tokio"], optional = true } diff --git a/litellm-rust/crates/core/src/auth/credential.rs b/litellm-rust/crates/core/src/auth/credential.rs new file mode 100644 index 00000000000..b5235b6780c --- /dev/null +++ b/litellm-rust/crates/core/src/auth/credential.rs @@ -0,0 +1,168 @@ +use std::future::Future; +use std::path::PathBuf; +use std::pin::Pin; +use std::sync::Arc; + +use veil::Redact; + +use crate::AuthError; + +use super::{ResolvedCredential, SecretValue, TokenProviderHandle}; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CredentialFileRef { + Path(PathBuf), + EnvironmentVariable(String), +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CredentialRef { + Explicit(SecretValue), + Env(String), + File(CredentialFileRef), + Request(String), + Host(String), + None, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CredentialLookup { + Found(SecretValue), + Missing, + Declined, +} + +pub type CredentialLookupFuture<'a> = + Pin> + Send + 'a>>; + +pub trait CredentialResolver: std::fmt::Debug + Send + Sync { + fn resolve<'a>(&'a self, reference: &'a CredentialRef) -> CredentialLookupFuture<'a>; +} + +#[derive(Clone, Redact)] +pub struct CredentialResolverHandle(#[redact(with = "[REDACTED]")] Arc); + +impl CredentialResolverHandle { + pub fn new(resolver: Arc) -> Self { + Self(resolver) + } + + pub async fn resolve(&self, reference: &CredentialRef) -> Result { + self.0.resolve(reference).await + } +} + +#[derive(Clone, Debug)] +pub enum CredentialPlan { + Static(CredentialRef), + Caller(TokenProviderHandle), + None, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CredentialPlanResolution { + Resolved(ResolvedCredential), + Unavailable, +} + +impl CredentialPlan { + pub async fn resolve( + &self, + resolver: &CredentialResolverHandle, + ) -> Result { + match self { + Self::Static(CredentialRef::Explicit(secret)) => Ok( + CredentialPlanResolution::Resolved(ResolvedCredential::Static(secret.clone())), + ), + Self::Static(CredentialRef::None) | Self::None => { + Ok(CredentialPlanResolution::Unavailable) + } + Self::Static(reference) => match resolver.resolve(reference).await? { + CredentialLookup::Found(secret) => Ok(CredentialPlanResolution::Resolved( + ResolvedCredential::Static(secret), + )), + CredentialLookup::Missing | CredentialLookup::Declined => { + Ok(CredentialPlanResolution::Unavailable) + } + }, + Self::Caller(caller) => { + let credential = caller.acquire().await?; + if credential.secret().expose().is_empty() { + return Err(AuthError::EmptyCallerCredential); + } + Ok(CredentialPlanResolution::Resolved(credential)) + } + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::{ + CredentialLookup, CredentialLookupFuture, CredentialPlan, CredentialPlanResolution, + CredentialRef, CredentialResolver, CredentialResolverHandle, + }; + use crate::AuthError; + use crate::auth::SecretValue; + + #[derive(Debug)] + struct HostResolver; + + impl CredentialResolver for HostResolver { + fn resolve<'a>(&'a self, reference: &'a CredentialRef) -> CredentialLookupFuture<'a> { + Box::pin(async move { + Ok(match reference { + CredentialRef::Host(name) if name == "rotating-token" => { + CredentialLookup::Found(SecretValue::new("resolved")) + } + _ => CredentialLookup::Declined, + }) + }) + } + } + + #[tokio::test] + async fn static_host_reference_resolves_at_acquisition_time() { + let resolver = CredentialResolverHandle::new(Arc::new(HostResolver)); + let plan = CredentialPlan::Static(CredentialRef::Host("rotating-token".to_string())); + + let resolved = plan.resolve(&resolver).await.unwrap(); + + assert!(matches!(resolved, CredentialPlanResolution::Resolved(_))); + } + + #[tokio::test] + async fn declined_reference_is_available_for_pre_acquisition_fallback() { + let resolver = CredentialResolverHandle::new(Arc::new(HostResolver)); + let plan = CredentialPlan::Static(CredentialRef::Request("api-key".to_string())); + + assert_eq!( + plan.resolve(&resolver).await.unwrap(), + CredentialPlanResolution::Unavailable + ); + } + + #[derive(Debug)] + struct FailingResolver; + + impl CredentialResolver for FailingResolver { + fn resolve<'a>(&'a self, _reference: &'a CredentialRef) -> CredentialLookupFuture<'a> { + Box::pin(async { Err(AuthError::UnresolvedOidcReference) }) + } + } + + #[tokio::test] + async fn acquisition_failure_is_terminal() { + let resolver = CredentialResolverHandle::new(Arc::new(FailingResolver)); + let plan = CredentialPlan::Static(CredentialRef::Host("token".to_string())); + + let error = plan + .resolve(&resolver) + .await + .expect_err("acquisition errors cannot become fallback"); + + assert_eq!(error, AuthError::UnresolvedOidcReference); + } +} diff --git a/litellm-rust/crates/core/src/auth/error.rs b/litellm-rust/crates/core/src/auth/error.rs new file mode 100644 index 00000000000..ddd4a6d016e --- /dev/null +++ b/litellm-rust/crates/core/src/auth/error.rs @@ -0,0 +1,120 @@ +use thiserror::Error; + +#[derive(Clone, Debug, Error, PartialEq, Eq)] +pub enum AuthError { + #[error("invalid authentication configuration: {0}")] + Configuration(#[from] AuthConfigurationError), + #[error("credential acquisition failed: {0}")] + AzureTokenAcquisition(String), + #[error("credential acquisition failed: {}", .0.iter().map(ToString::to_string).collect::>().join("; "))] + CredentialChain(Vec), + #[error("credential caller failed: credential caller returned an empty credential")] + EmptyCallerCredential, + #[error("credential caller failed: Azure AD token provider returned an empty token")] + EmptyAzureToken, + #[error("credential acquisition failed: Azure OIDC reference did not resolve to a value")] + UnresolvedOidcReference, + #[error( + "Missing {provider} API Key - A call is being made to {provider} but no key is set either in the environment variables or via params" + )] + MissingApiKey { provider: &'static str }, + #[error( + "Missing {provider} API Base - Set {environment_variable} environment variable or pass api_base parameter" + )] + MissingApiBase { + provider: &'static str, + environment_variable: &'static str, + }, + #[error("{0}")] + MissingCredential(#[from] MissingCredential), + #[error("{0}")] + Aws(#[from] AwsAuthError), + #[error("invalid authentication header")] + InvalidHeader, +} + +#[derive(Clone, Debug, Error, PartialEq, Eq)] +pub enum AuthConfigurationError { + #[error("credential header already exists")] + ExistingCredentialHeader, + #[error("credential plan is not allowed by the provider auth policy")] + DisallowedCredentialPlan, + #[error("credential cannot be empty")] + EmptyCredential, + #[error("invalid Azure credential selector")] + InvalidAzureSelector, + #[error("ClientSecretCredential requires tenant_id, client_id, and client_secret")] + MissingClientSecretFields, + #[error("WorkloadIdentityCredential requires tenant_id")] + MissingWorkloadTenant, + #[error("WorkloadIdentityCredential requires client_id")] + MissingWorkloadClient, + #[error("WorkloadIdentityCredential requires azure_federated_token_file")] + MissingWorkloadTokenFile, + #[error("credential reference requires a host credential resolver")] + MissingHostResolver, + #[error("caller credential plan requires provider-specific inputs")] + MissingCallerInputs, + #[error("credential header {0} already exists")] + DuplicateHeader(&'static str), + #[error("{0} must be a string or null")] + InvalidFieldType(String), + #[error("unsupported OIDC reference")] + UnsupportedOidcReference, + #[error("{0} cannot be empty")] + EmptyReference(String), + #[error("Azure credential initialization failed: {0}")] + AzureCredentialInitialization(String), + #[error("Azure authority must be an HTTPS origin without credentials, query, or fragment")] + InvalidAzureAuthority, + #[error("request-controlled Azure auth inputs cannot be combined with host credentials")] + MixedAzureCredentialSources, + #[error("request-controlled Azure credential references are not allowed")] + RequestAzureCredentialReference, + #[error("host credentials cannot be sent to a request-controlled Azure endpoint")] + RequestAzureCredentialDestination, +} + +#[derive(Clone, Debug, Error, PartialEq, Eq)] +pub enum MissingCredential { + #[error( + "Missing Anthropic API Key - Set `api_key` or the ANTHROPIC_API_KEY environment variable" + )] + AnthropicApiKey, + #[error("Missing Azure API Key - Set `api_key` or the AZURE_API_KEY environment variable")] + AzureApiKey, + #[error( + "Missing Azure API Base - Set `api_base` or the AZURE_API_BASE environment variable. Expected format: https://.services.ai.azure.com/anthropic" + )] + AzureApiBase, + #[error( + "Missing OpenAI API Key - a realtime call is being made but no key was passed via params or the OPENAI_API_KEY environment variable" + )] + OpenAiRealtimeApiKey, + #[error( + "Missing OpenAI API Key - a Responses WebSocket call is being made but no key was passed via params or the OPENAI_API_KEY environment variable" + )] + OpenAiResponsesApiKey, +} + +#[derive(Clone, Debug, Error, PartialEq, Eq)] +pub enum AwsAuthError { + #[error("AWS profile credentials failed: {0}")] + Profile(String), + #[error("AWS default credentials failed: {0}")] + DefaultChain(String), + #[error("AWS role credentials failed: {0}")] + AssumeRole(String), + #[error("AWS web identity credentials failed: {0}")] + WebIdentity(String), + #[error("AWS web identity expiration was invalid: {0}")] + WebIdentityExpiration(String), + #[error("AWS signing parameters failed: {0}")] + SigningParameters(String), + #[error("AWS signable request failed: {0}")] + SignableRequest(String), + #[error("AWS request signing failed: {0}")] + Signing(String), + #[error("AWS web identity response had no credentials")] + MissingWebIdentityCredentials, +} diff --git a/litellm-rust/crates/core/src/auth/http.rs b/litellm-rust/crates/core/src/auth/http.rs new file mode 100644 index 00000000000..83931311550 --- /dev/null +++ b/litellm-rust/crates/core/src/auth/http.rs @@ -0,0 +1,86 @@ +use crate::AuthError; +use crate::auth::error::AuthConfigurationError; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CredentialPlacement { + Bearer, + Header(&'static str), +} + +impl CredentialPlacement { + pub fn header_name(self) -> &'static str { + match self { + Self::Bearer => "Authorization", + Self::Header(name) => name, + } + } +} + +pub(crate) fn apply_credential( + headers: Vec<(String, String)>, + credential: &str, + placement: CredentialPlacement, +) -> Result, AuthError> { + if credential.trim().is_empty() { + return Err(AuthError::Configuration( + AuthConfigurationError::EmptyCredential, + )); + } + if headers + .iter() + .any(|(name, _)| name.eq_ignore_ascii_case(placement.header_name())) + { + return Err(AuthError::Configuration( + AuthConfigurationError::DuplicateHeader(placement.header_name()), + )); + } + let value = match placement { + CredentialPlacement::Bearer => format!("Bearer {credential}"), + CredentialPlacement::Header(_) => credential.to_string(), + }; + Ok( + std::iter::once((placement.header_name().to_string(), value)) + .chain(headers) + .collect(), + ) +} + +/// How the upstream call is authenticated. API-key strategies are resolved in +/// `prepare`; SigV4 needs the serialized body, so the handler signs it. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum RequestAuth { + Header { name: &'static str, value: String }, + Bearer { token: String }, + AwsSigV4 { region: String }, +} + +#[cfg(test)] +mod tests { + use super::{CredentialPlacement, apply_credential}; + + #[test] + fn bearer_uses_authorization_header() { + let headers = apply_credential(Vec::new(), "key", CredentialPlacement::Bearer) + .expect("credential applies"); + + assert_eq!( + headers, + vec![("Authorization".to_string(), "Bearer key".to_string())] + ); + } + + #[test] + fn named_header_rejects_existing_value() { + let error = apply_credential( + vec![( + "ocp-apim-subscription-key".to_string(), + "caller-key".to_string(), + )], + "configured-key", + CredentialPlacement::Header("Ocp-Apim-Subscription-Key"), + ) + .expect_err("provider policy must handle existing credentials"); + + assert!(error.to_string().contains("already exists")); + } +} diff --git a/litellm-rust/crates/core/src/auth/mod.rs b/litellm-rust/crates/core/src/auth/mod.rs new file mode 100644 index 00000000000..2ca2f3c3016 --- /dev/null +++ b/litellm-rust/crates/core/src/auth/mod.rs @@ -0,0 +1,55 @@ +mod credential; +pub mod error; +pub use error::AuthError; +pub(crate) mod http; +mod policy; +mod secret; +mod token; + +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Copy, Debug, Default, Deserialize, Eq, Hash, PartialEq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum InputSource { + Request, + #[default] + Deployment, + Environment, +} + +#[derive(Clone, Debug, Default, Eq, PartialEq)] +pub struct Sourced { + value: T, + source: InputSource, +} + +impl Sourced { + pub fn new(value: T, source: InputSource) -> Self { + Self { value, source } + } + + pub fn value(&self) -> &T { + &self.value + } + + pub fn source(&self) -> InputSource { + self.source + } + + pub fn into_value(self) -> T { + self.value + } + + pub fn map(self, map: impl FnOnce(T) -> U) -> Sourced { + Sourced::new(map(self.value), self.source) + } +} + +pub use credential::{ + CredentialFileRef, CredentialLookup, CredentialLookupFuture, CredentialPlan, + CredentialPlanResolution, CredentialRef, CredentialResolver, CredentialResolverHandle, +}; +pub use http::{CredentialPlacement, RequestAuth}; +pub use policy::{CredentialPlanKind, CredentialRule, ExistingHeaderBehavior, ProviderAuthPolicy}; +pub use secret::SecretValue; +pub use token::{ResolvedCredential, TokenFuture, TokenProvider, TokenProviderHandle}; diff --git a/litellm-rust/crates/core/src/auth/policy.rs b/litellm-rust/crates/core/src/auth/policy.rs new file mode 100644 index 00000000000..b796dedf0d8 --- /dev/null +++ b/litellm-rust/crates/core/src/auth/policy.rs @@ -0,0 +1,114 @@ +use crate::AuthError; +use crate::auth::error::AuthConfigurationError; + +use super::http::apply_credential; +use super::{CredentialPlacement, ResolvedCredential}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CredentialPlanKind { + Static, + Entra, + Caller, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CredentialRule { + pub kind: CredentialPlanKind, + pub placement: CredentialPlacement, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ExistingHeaderBehavior { + Preserve, + Reject, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ProviderAuthPolicy { + pub rules: &'static [CredentialRule], + pub accepted_existing_headers: &'static [&'static str], + pub existing_header_behavior: ExistingHeaderBehavior, + pub scope: Option<&'static str>, + pub audience: Option<&'static str>, +} + +impl ProviderAuthPolicy { + pub fn has_existing_credential(&self, headers: &[(String, String)]) -> bool { + headers.iter().any(|(name, _)| { + self.accepted_existing_headers + .iter() + .any(|accepted| name.eq_ignore_ascii_case(accepted)) + }) + } + + pub fn apply( + &self, + headers: Vec<(String, String)>, + kind: CredentialPlanKind, + credential: &ResolvedCredential, + ) -> Result, AuthError> { + if self.has_existing_credential(&headers) { + return match self.existing_header_behavior { + ExistingHeaderBehavior::Preserve => Ok(headers), + ExistingHeaderBehavior::Reject => Err(AuthError::Configuration( + AuthConfigurationError::ExistingCredentialHeader, + )), + }; + } + let rule = + self.rules + .iter() + .find(|rule| rule.kind == kind) + .ok_or(AuthError::Configuration( + AuthConfigurationError::DisallowedCredentialPlan, + ))?; + apply_credential(headers, credential.secret().expose(), rule.placement) + } +} + +#[cfg(test)] +mod tests { + use super::{CredentialPlanKind, CredentialRule, ExistingHeaderBehavior, ProviderAuthPolicy}; + use crate::auth::{CredentialPlacement, ResolvedCredential, SecretValue}; + + const RULES: &[CredentialRule] = &[CredentialRule { + kind: CredentialPlanKind::Static, + placement: CredentialPlacement::Header("x-api-key"), + }]; + const POLICY: ProviderAuthPolicy = ProviderAuthPolicy { + rules: RULES, + accepted_existing_headers: &["x-api-key"], + existing_header_behavior: ExistingHeaderBehavior::Preserve, + scope: None, + audience: None, + }; + + #[test] + fn rules_define_allowed_plans_and_credential_placement() { + let headers = POLICY + .apply( + Vec::new(), + CredentialPlanKind::Static, + &ResolvedCredential::Static(SecretValue::new("secret")), + ) + .unwrap(); + + assert_eq!( + headers, + vec![("x-api-key".to_string(), "secret".to_string())] + ); + } + + #[test] + fn unsupported_plan_is_rejected() { + let error = POLICY + .apply( + Vec::new(), + CredentialPlanKind::Entra, + &ResolvedCredential::Static(SecretValue::new("secret")), + ) + .unwrap_err(); + + assert!(error.to_string().contains("not allowed")); + } +} diff --git a/litellm-rust/crates/core/src/auth/secret.rs b/litellm-rust/crates/core/src/auth/secret.rs new file mode 100644 index 00000000000..3ecb0a835ee --- /dev/null +++ b/litellm-rust/crates/core/src/auth/secret.rs @@ -0,0 +1,41 @@ +use veil::Redact; + +#[derive(Redact, Clone)] +pub struct SecretValue(#[redact(with = "[REDACTED]")] String); + +impl SecretValue { + pub fn new(value: impl Into) -> Self { + Self(value.into()) + } + + pub fn expose(&self) -> &str { + &self.0 + } +} + +impl PartialEq for SecretValue { + fn eq(&self, other: &Self) -> bool { + subtle::ConstantTimeEq::ct_eq(self.0.as_bytes(), other.0.as_bytes()).into() + } +} + +impl Eq for SecretValue {} + +#[cfg(test)] +mod tests { + use super::SecretValue; + + #[test] + fn debug_redacts_plaintext() { + let debug = format!("{:?}", SecretValue::new("credential-value")); + + assert!(!debug.contains("credential-value")); + assert!(debug.contains("REDACTED")); + } + + #[test] + fn equality_compares_plaintext_values() { + assert_eq!(SecretValue::new("same"), SecretValue::new("same")); + assert_ne!(SecretValue::new("same"), SecretValue::new("different")); + } +} diff --git a/litellm-rust/crates/core/src/auth/token.rs b/litellm-rust/crates/core/src/auth/token.rs new file mode 100644 index 00000000000..cfc6b8f0d6b --- /dev/null +++ b/litellm-rust/crates/core/src/auth/token.rs @@ -0,0 +1,47 @@ +use std::future::Future; +use std::pin::Pin; +use std::sync::Arc; +use std::time::SystemTime; + +use veil::Redact; + +use crate::AuthError; + +use super::secret::SecretValue; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ResolvedCredential { + Static(SecretValue), + AccessToken { + token: SecretValue, + expires_on: Option, + }, +} + +impl ResolvedCredential { + pub fn secret(&self) -> &SecretValue { + match self { + Self::Static(secret) | Self::AccessToken { token: secret, .. } => secret, + } + } +} + +pub type TokenFuture<'a> = + Pin> + Send + 'a>>; + +pub trait TokenProvider: std::fmt::Debug + Send + Sync { + fn acquire(&self) -> TokenFuture<'_>; +} + +#[derive(Clone, Redact)] +pub struct TokenProviderHandle(#[redact(with = "[REDACTED]")] Arc); + +impl TokenProviderHandle { + pub fn new(caller: Arc) -> Self { + Self(caller) + } + + pub async fn acquire(&self) -> Result { + self.0.acquire().await + } +} diff --git a/litellm-rust/crates/core/src/error.rs b/litellm-rust/crates/core/src/error.rs index eefa7d606d8..b171c0c4274 100644 --- a/litellm-rust/crates/core/src/error.rs +++ b/litellm-rust/crates/core/src/error.rs @@ -22,7 +22,7 @@ pub enum Error { )] MissingApiKey { provider: &'static str }, #[error( - "Missing Azure AI credentials - set AZURE_AI_API_KEY or provide an Authorization header" + "invalid authentication configuration: Missing Azure AI credentials - set AZURE_AI_API_KEY or configure Entra ID" )] MissingAzureAiCredentials, #[error("Missing Azure AI credentials - set AZURE_AI_API_KEY or provide azure_ad_token")] @@ -121,6 +121,15 @@ impl From for Error { } } +impl From for Error { + fn from(error: crate::AuthError) -> Self { + match error { + crate::AuthError::MissingApiKey { provider } => Self::MissingApiKey { provider }, + error => Self::Auth(error.to_string()), + } + } +} + pub fn json_type_name(value: &serde_json::Value) -> &'static str { match value { serde_json::Value::Null => "null", @@ -136,6 +145,14 @@ pub fn json_type_name(value: &serde_json::Value) -> &'static str { mod transport_tests { use super::*; + #[test] + fn missing_auth_key_preserves_provider_in_public_error() { + assert_eq!( + Error::from(crate::AuthError::MissingApiKey { provider: "Vertex" }), + Error::MissingApiKey { provider: "Vertex" } + ); + } + #[tokio::test] async fn transport_errors_remove_urls_and_keep_dispatch_context() { let error = reqwest::Client::builder() diff --git a/litellm-rust/crates/core/src/lib.rs b/litellm-rust/crates/core/src/lib.rs index 7e81b292441..0b3573deab2 100644 --- a/litellm-rust/crates/core/src/lib.rs +++ b/litellm-rust/crates/core/src/lib.rs @@ -1,4 +1,5 @@ pub mod audio_transcription; +pub mod auth; pub mod caching; pub mod call_lifecycle; pub mod chat_completions; @@ -17,4 +18,5 @@ pub mod router; pub mod routing_utils; mod url_utils; +pub use auth::AuthError; pub use error::Error; diff --git a/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs b/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs index 468c883a1dd..0e52bc61249 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs @@ -1,5 +1,9 @@ +use std::sync::OnceLock; + use super::OcrAdapter; use crate::Error; +use crate::auth::error::AuthConfigurationError; +use crate::auth::{InputSource, Sourced}; use crate::constants::AZURE_AI_OCR_PATH; use crate::ocr::OcrClient; use crate::ocr::codecs::mistral::{self, MistralOcrParams, MistralOcrResponse}; @@ -10,6 +14,7 @@ use crate::ocr::prepare::{ }; use crate::ocr::registry::OcrProvider; use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrConnection}; +use crate::providers::azure_ai::auth::{AzureAuthInputs, AzureAuthService}; use crate::url_utils::ApiUrl; const AZURE_AI_API_KEY_ENV: &str = "AZURE_AI_API_KEY"; @@ -31,7 +36,12 @@ impl OcrAdapter for AzureMistralAdapter { known: params, extra_params: _extra_params, } = _prepare_ocr_request::(request)?; - let headers = authenticate(&request.connection, &credential_env)?; + let config = AzureAuthInputs::from_sourced_optional_params( + &request.optional_params, + &request.input_sources, + ) + .map_err(Error::from)?; + let headers = validate_environment(&request.connection, &config, &credential_env).await?; let url = get_complete_url(request.connection.api_base.as_deref(), &credential_env)?; let document = inline_remote_document( client.document_fetcher(), @@ -76,21 +86,61 @@ fn get_complete_url( }) } -fn authenticate( +async fn validate_environment( connection: &OcrConnection, - env_lookup: &dyn Fn(&str) -> Option, + config: &AzureAuthInputs, + env_lookup: &(dyn Fn(&str) -> Option + Sync), ) -> Result, OcrError> { if crate::http_utils::has_header(&connection.extra_headers, "authorization") { + validate_destination(connection, connection.extra_headers_source)?; return Ok(connection.extra_headers.clone()); } let key = nonblank(connection.api_key.clone()) - .or_else(|| nonblank(env_lookup(AZURE_AI_API_KEY_ENV))) + .map(|value| Sourced::new(value, connection.api_key_source)) + .or_else(|| { + nonblank(env_lookup(AZURE_AI_API_KEY_ENV)) + .map(|value| Sourced::new(value, InputSource::Environment)) + }); + if let Some(key) = key { + validate_destination(connection, key.source())?; + return Ok(bearer_headers(connection, key.value())); + } + static SERVICE: OnceLock = OnceLock::new(); + let key = SERVICE + .get_or_init(AzureAuthService::default) + .get_azure_ad_token(config, env_lookup) + .await + .map_err(Error::from)? + .map(|credential| { + let source = credential.source(); + let value = credential.value().secret().expose().to_string(); + Sourced::new(value, source) + }) .ok_or(Error::MissingAzureAiCredentials)?; - Ok( - std::iter::once(("Authorization".into(), format!("Bearer {key}"))) - .chain(connection.extra_headers.clone()) - .collect(), - ) + validate_destination(connection, key.source())?; + Ok(bearer_headers(connection, key.value())) +} + +fn validate_destination( + connection: &OcrConnection, + credential_source: InputSource, +) -> Result<(), OcrError> { + if connection.api_base.is_some() + && connection.api_base_source == InputSource::Request + && credential_source != InputSource::Request + { + return Err(Error::from(crate::AuthError::Configuration( + AuthConfigurationError::RequestAzureCredentialDestination, + )) + .into()); + } + Ok(()) +} + +fn bearer_headers(connection: &OcrConnection, key: &str) -> Vec<(String, String)> { + std::iter::once(("Authorization".into(), format!("Bearer {key}"))) + .chain(connection.extra_headers.clone()) + .collect() } fn nonblank(value: Option) -> Option { @@ -119,27 +169,76 @@ mod tests { ); } - #[test] - fn supplied_authorization_precedes_keys() { + #[tokio::test] + async fn supplied_authorization_precedes_keys() { let connection = OcrConnection { api_key: Some("request-key".into()), extra_headers: vec![("authorization".into(), "Bearer prepared".into())], ..Default::default() }; assert_eq!( - authenticate(&connection, &|_| Some("environment-key".into())).unwrap(), + validate_environment(&connection, &Default::default(), &|_| Some( + "environment-key".into() + )) + .await + .unwrap(), connection.extra_headers ); } - #[test] - fn request_key_precedes_environment_key() { + #[tokio::test] + async fn request_key_precedes_environment_key() { let connection = OcrConnection { api_key: Some("request-key".into()), ..Default::default() }; assert_eq!( - authenticate(&connection, &|_| Some("environment-key".into())).unwrap()[0], + validate_environment(&connection, &Default::default(), &|_| Some( + "environment-key".into() + )) + .await + .unwrap()[0], + ("Authorization".into(), "Bearer request-key".into()) + ); + } + + #[tokio::test] + async fn request_endpoint_cannot_receive_environment_key() { + let connection = OcrConnection { + api_base: Some("https://request.example".into()), + api_base_source: InputSource::Request, + ..Default::default() + }; + + let error = validate_environment(&connection, &Default::default(), &|name| { + (name == AZURE_AI_API_KEY_ENV).then(|| "environment-key".into()) + }) + .await + .unwrap_err(); + + assert!( + error + .to_string() + .contains("request-controlled Azure endpoint") + ); + } + + #[tokio::test] + async fn request_endpoint_accepts_request_owned_key() { + let connection = OcrConnection { + api_key: Some("request-key".into()), + api_key_source: InputSource::Request, + api_base: Some("https://request.example".into()), + api_base_source: InputSource::Request, + ..Default::default() + }; + + let headers = validate_environment(&connection, &Default::default(), &|_| None) + .await + .unwrap(); + + assert_eq!( + headers[0], ("Authorization".into(), "Bearer request-key".into()) ); } diff --git a/litellm-rust/crates/core/src/ocr/types.rs b/litellm-rust/crates/core/src/ocr/types.rs index eeda94738a0..dec474876fb 100644 --- a/litellm-rust/crates/core/src/ocr/types.rs +++ b/litellm-rust/crates/core/src/ocr/types.rs @@ -1,3 +1,4 @@ +use std::collections::BTreeMap; use std::sync::Arc; use std::time::Duration; @@ -7,6 +8,7 @@ use serde_json::{Map, Value}; use super::hooks::{NoopOcrHooks, OcrHooks}; use super::registry::{OcrAdapterKind, resolve_wire_adapter}; use crate::Error; +use crate::auth::InputSource; use crate::constants::OCR_HTTP_TIMEOUT_SECS; #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] @@ -65,8 +67,11 @@ pub enum OcrResponseFormat { #[derive(Clone)] pub struct OcrConnection { pub api_key: Option, + pub api_key_source: InputSource, pub api_base: Option, + pub api_base_source: InputSource, pub extra_headers: Vec<(String, String)>, + pub extra_headers_source: InputSource, pub timeout: Duration, pub max_download_bytes: u64, } @@ -75,8 +80,11 @@ impl Default for OcrConnection { fn default() -> Self { Self { api_key: None, + api_key_source: InputSource::Deployment, api_base: None, + api_base_source: InputSource::Deployment, extra_headers: Vec::new(), + extra_headers_source: InputSource::Deployment, timeout: Duration::from_secs(OCR_HTTP_TIMEOUT_SECS), max_download_bytes: crate::constants::OCR_DOWNLOAD_MAX_BYTES, } @@ -90,6 +98,7 @@ pub struct LiteLLMOcrRequest { pub hooks: Arc, pub litellm_call_id: Option, pub optional_params: Map, + pub input_sources: BTreeMap, pub(crate) adapter: OcrAdapterKind, } @@ -109,6 +118,7 @@ impl LiteLLMOcrRequest { hooks: Arc::new(NoopOcrHooks), litellm_call_id: None, optional_params, + input_sources: BTreeMap::new(), adapter: adapter_kind, }) } diff --git a/litellm-rust/crates/core/src/ocr/wire.rs b/litellm-rust/crates/core/src/ocr/wire.rs index db8d91905c4..d0fe32378b9 100644 --- a/litellm-rust/crates/core/src/ocr/wire.rs +++ b/litellm-rust/crates/core/src/ocr/wire.rs @@ -1,10 +1,12 @@ use crate::ocr::error::OcrRequestError; use crate::ocr::error::OcrResponseError; +use std::collections::BTreeMap; use std::time::Duration; use super::hooks::{OcrDuringCallRequest, OcrPreCallRequest}; use super::types::{LiteLLMOcrRequest, OcrConnection, OcrDocument}; use crate::Error; +use crate::auth::InputSource; use serde::{ Deserialize, de::{DeserializeOwned, IntoDeserializer}, @@ -28,6 +30,8 @@ pub struct OcrWireRequest { pub extra_headers: Option>, #[serde(default)] pub optional_params: Map, + #[serde(default)] + pub input_sources: BTreeMap, pub timeout_seconds: Option, } @@ -36,6 +40,9 @@ pub fn is_supported_request(model: &str, custom_llm_provider: Option<&str>) -> b } pub fn decode_request(wire: OcrWireRequest) -> Result { + let api_key_source = source_for(&wire.input_sources, "api_key"); + let api_base_source = source_for(&wire.input_sources, "api_base"); + let extra_headers_source = source_for(&wire.input_sources, "extra_headers"); let document = decode_request_value(wire.document, "document")?; let headers = wire .extra_headers @@ -67,17 +74,25 @@ pub fn decode_request(wire: OcrWireRequest) -> Result )?; let connection = OcrConnection { api_key: nonblank(wire.api_key), + api_key_source, api_base: nonblank(wire.api_base), + api_base_source, extra_headers: headers, + extra_headers_source, timeout: timeout.unwrap_or(defaults.timeout), max_download_bytes: defaults.max_download_bytes, }; Ok(LiteLLMOcrRequest { connection, + input_sources: wire.input_sources, ..request }) } +fn source_for(sources: &BTreeMap, name: &str) -> InputSource { + sources.get(name).copied().unwrap_or_default() +} + fn nonblank(value: Option) -> Option { value .map(|s| s.trim().to_string()) diff --git a/litellm-rust/crates/core/src/providers/anthropic/messages/transformation.rs b/litellm-rust/crates/core/src/providers/anthropic/messages/transformation.rs index f31b961e78a..3ed00b7cc5f 100644 --- a/litellm-rust/crates/core/src/providers/anthropic/messages/transformation.rs +++ b/litellm-rust/crates/core/src/providers/anthropic/messages/transformation.rs @@ -1,3 +1,4 @@ +use crate::auth::error::MissingCredential; use crate::error::Error; use crate::messages::transformation::{AnthropicMessagesProviderConfig, MessagesAuthStrategy}; @@ -21,13 +22,7 @@ pub fn resolve_anthropic_api_key( non_empty(api_key) .map(str::to_string) .or_else(|| env_lookup(ANTHROPIC_API_KEY_ENV).filter(|value| !value.trim().is_empty())) - .ok_or_else(|| { - Error::Auth( - "Missing Anthropic API Key - Set `api_key` or the ANTHROPIC_API_KEY \ - environment variable" - .to_string(), - ) - }) + .ok_or_else(|| Error::from(crate::AuthError::from(MissingCredential::AnthropicApiKey))) } pub fn complete_anthropic_url( diff --git a/litellm-rust/crates/core/src/providers/azure_ai/auth/credential_provider_cache.rs b/litellm-rust/crates/core/src/providers/azure_ai/auth/credential_provider_cache.rs new file mode 100644 index 00000000000..297e4cc6502 --- /dev/null +++ b/litellm-rust/crates/core/src/providers/azure_ai/auth/credential_provider_cache.rs @@ -0,0 +1,43 @@ +use std::future::Future; +use std::sync::Arc; + +use azure_core::credentials::TokenCredential; +use moka::future::Cache; + +use crate::AuthError; + +#[derive(Clone, Debug, PartialEq, Eq, Hash)] +pub(crate) struct AzureCredentialProviderCacheKey { + pub(crate) mechanism: &'static str, + pub(crate) authority: String, + pub(crate) tenant_id: String, + pub(crate) client_id: String, + pub(crate) scope: String, + pub(crate) secret_identity: String, +} + +pub(crate) struct AzureCredentialProviderCache { + entries: Cache>, +} + +impl AzureCredentialProviderCache { + pub(crate) fn new(capacity: u64) -> Self { + Self { + entries: Cache::builder().max_capacity(capacity).build(), + } + } + + pub(crate) async fn get_or_create( + &self, + key: AzureCredentialProviderCacheKey, + create: F, + ) -> Result, AuthError> + where + F: Future, AuthError>>, + { + self.entries + .try_get_with(key, create) + .await + .map_err(|error| (*error).clone()) + } +} diff --git a/litellm-rust/crates/core/src/providers/azure_ai/auth/mod.rs b/litellm-rust/crates/core/src/providers/azure_ai/auth/mod.rs new file mode 100644 index 00000000000..33d007c1945 --- /dev/null +++ b/litellm-rust/crates/core/src/providers/azure_ai/auth/mod.rs @@ -0,0 +1,7 @@ +mod credential_provider_cache; +mod native; +mod resolve; +mod types; + +pub(crate) use resolve::AzureAuthService; +pub(crate) use types::AzureAuthInputs; diff --git a/litellm-rust/crates/core/src/providers/azure_ai/auth/native.rs b/litellm-rust/crates/core/src/providers/azure_ai/auth/native.rs new file mode 100644 index 00000000000..b8f19818d16 --- /dev/null +++ b/litellm-rust/crates/core/src/providers/azure_ai/auth/native.rs @@ -0,0 +1,702 @@ +use crate::auth::error::AuthConfigurationError; +use std::sync::Arc; +use std::time::{Duration, UNIX_EPOCH}; + +use azure_core::cloud::{CloudConfiguration, CustomConfiguration}; +use azure_core::credentials::{Secret, TokenCredential}; +use azure_core::http::ClientOptions; +use azure_identity::{ + ClientAssertion, ClientAssertionCredential, ClientAssertionCredentialOptions, + ClientSecretCredential, ClientSecretCredentialOptions, DeveloperToolsCredential, + ManagedIdentityCredential, ManagedIdentityCredentialOptions, UserAssignedId, + WorkloadIdentityCredential, WorkloadIdentityCredentialOptions, +}; +use sha2::{Digest, Sha256}; + +use crate::AuthError; +use crate::auth::{InputSource, ResolvedCredential, SecretValue, Sourced}; + +use super::credential_provider_cache::{ + AzureCredentialProviderCache, AzureCredentialProviderCacheKey, +}; + +#[derive(Clone, Debug)] +pub(crate) enum NativeAzureRequest { + ClientSecret { + tenant_id: Sourced, + client_id: Sourced, + client_secret: Sourced, + scope: Sourced, + authority: Option>, + }, + ClientAssertion { + tenant_id: Sourced, + client_id: Sourced, + assertion: Sourced, + assertion_identity: String, + scope: Sourced, + authority: Option>, + }, + WorkloadIdentity { + tenant_id: Sourced, + client_id: Sourced, + token_file_path: Sourced, + scope: Sourced, + authority: Option>, + }, + ManagedIdentity { + client_id: Option>, + scope: Sourced, + selection_source: InputSource, + }, + DeveloperTools { + scope: Sourced, + selection_source: InputSource, + }, +} + +#[derive(Clone, Debug)] +pub(crate) struct ValidatedAzureRequest { + request: NativeAzureRequest, + credential_source: InputSource, +} + +impl ValidatedAzureRequest { + pub(crate) fn new(request: NativeAzureRequest) -> Result { + validate_authority(&request)?; + let credential_source = validate_sources(&request)?; + Ok(Self { + request, + credential_source, + }) + } + + pub(crate) fn credential_source(&self) -> InputSource { + self.credential_source + } + + #[cfg(test)] + pub(super) fn kind(&self) -> &'static str { + match self.request { + NativeAzureRequest::ClientSecret { .. } => "client-secret", + NativeAzureRequest::ClientAssertion { .. } => "client-assertion", + NativeAzureRequest::WorkloadIdentity { .. } => "workload-identity", + NativeAzureRequest::ManagedIdentity { .. } => "managed-identity", + NativeAzureRequest::DeveloperTools { .. } => "developer-tools", + } + } +} + +pub(crate) struct NativeAzureTokenAcquirer { + cache: AzureCredentialProviderCache, + transport: Option, +} + +impl Default for NativeAzureTokenAcquirer { + fn default() -> Self { + Self::new(64) + } +} + +impl NativeAzureTokenAcquirer { + pub(crate) fn new(cache_capacity: u64) -> Self { + Self { + cache: AzureCredentialProviderCache::new(cache_capacity), + transport: None, + } + } + + #[cfg(test)] + pub(super) fn with_transport( + cache_capacity: u64, + transport: azure_core::http::Transport, + ) -> Self { + Self { + cache: AzureCredentialProviderCache::new(cache_capacity), + transport: Some(transport), + } + } + + pub(crate) async fn acquire( + &self, + request: ValidatedAzureRequest, + ) -> Result { + let scope = request.request.scope().to_string(); + let key = request.request.cache_key(); + let transport = self.transport.clone(); + let credential = self + .cache + .get_or_create( + key, + async move { build_credential(request.request, transport) }, + ) + .await?; + let token = credential + .get_token(&[scope.as_str()], None) + .await + .map_err(|error| AuthError::AzureTokenAcquisition(error.to_string()))?; + let expires_on = u64::try_from(token.expires_on.unix_timestamp()) + .ok() + .map(|seconds| UNIX_EPOCH + Duration::from_secs(seconds)); + + Ok(ResolvedCredential::AccessToken { + token: SecretValue::new(token.token.secret()), + expires_on, + }) + } +} + +impl NativeAzureRequest { + fn scope(&self) -> &str { + match self { + Self::ClientSecret { scope, .. } + | Self::ClientAssertion { scope, .. } + | Self::WorkloadIdentity { scope, .. } + | Self::ManagedIdentity { scope, .. } + | Self::DeveloperTools { scope, .. } => scope.value(), + } + } + + fn cache_key(&self) -> AzureCredentialProviderCacheKey { + match self { + Self::ClientSecret { + tenant_id, + client_id, + client_secret, + scope, + authority, + } => AzureCredentialProviderCacheKey { + mechanism: "client-secret", + authority: authority + .as_ref() + .map(|value| value.value().clone()) + .unwrap_or_default(), + tenant_id: tenant_id.value().clone(), + client_id: client_id.value().clone(), + scope: scope.value().clone(), + secret_identity: secret_digest(client_secret.value().expose()), + }, + Self::ClientAssertion { + tenant_id, + client_id, + assertion, + assertion_identity, + scope, + authority, + } => AzureCredentialProviderCacheKey { + mechanism: "client-assertion", + authority: authority + .as_ref() + .map(|value| value.value().clone()) + .unwrap_or_default(), + tenant_id: tenant_id.value().clone(), + client_id: client_id.value().clone(), + scope: scope.value().clone(), + secret_identity: format!( + "{assertion_identity}:{}", + secret_digest(assertion.value().expose()) + ), + }, + Self::WorkloadIdentity { + tenant_id, + client_id, + token_file_path, + scope, + authority, + } => AzureCredentialProviderCacheKey { + mechanism: "workload-identity", + authority: authority + .as_ref() + .map(|value| value.value().clone()) + .unwrap_or_default(), + tenant_id: tenant_id.value().clone(), + client_id: client_id.value().clone(), + scope: scope.value().clone(), + secret_identity: token_file_path.value().clone(), + }, + Self::ManagedIdentity { + client_id, scope, .. + } => AzureCredentialProviderCacheKey { + mechanism: "managed-identity", + authority: String::new(), + tenant_id: String::new(), + client_id: client_id + .as_ref() + .map(|value| value.value().clone()) + .unwrap_or_default(), + scope: scope.value().clone(), + secret_identity: String::new(), + }, + Self::DeveloperTools { scope, .. } => AzureCredentialProviderCacheKey { + mechanism: "developer-tools", + authority: String::new(), + tenant_id: String::new(), + client_id: String::new(), + scope: scope.value().clone(), + secret_identity: String::new(), + }, + } + } +} + +fn validate_authority(request: &NativeAzureRequest) -> Result<(), AuthError> { + let authority = match request { + NativeAzureRequest::ClientSecret { authority, .. } + | NativeAzureRequest::ClientAssertion { authority, .. } + | NativeAzureRequest::WorkloadIdentity { authority, .. } => authority.as_ref(), + NativeAzureRequest::ManagedIdentity { .. } | NativeAzureRequest::DeveloperTools { .. } => { + None + } + }; + let Some(authority) = authority else { + return Ok(()); + }; + let url = url::Url::parse(authority.value()) + .map_err(|_| AuthError::Configuration(AuthConfigurationError::InvalidAzureAuthority))?; + if url.scheme() != "https" + || url.host_str().is_none() + || !url.username().is_empty() + || url.password().is_some() + || url.query().is_some() + || url.fragment().is_some() + || !matches!(url.path(), "" | "/") + { + return Err(AuthError::Configuration( + AuthConfigurationError::InvalidAzureAuthority, + )); + } + Ok(()) +} + +fn validate_sources(request: &NativeAzureRequest) -> Result { + match request { + NativeAzureRequest::ClientSecret { + tenant_id, + client_id, + client_secret, + scope, + authority, + } => { + let identity_sources = [ + tenant_id.source(), + client_id.source(), + client_secret.source(), + ]; + let request_identity = identity_sources.contains(&InputSource::Request); + if request_identity + && !identity_sources + .iter() + .all(|source| *source == InputSource::Request) + { + return mixed_sources(); + } + if !request_identity && is_request_controlled(scope, authority.as_ref()) { + return mixed_sources(); + } + Ok(if request_identity { + InputSource::Request + } else { + trusted_source(&identity_sources) + }) + } + NativeAzureRequest::ClientAssertion { + tenant_id, + client_id, + assertion, + scope, + authority, + .. + } => trusted_only(&[ + tenant_id.source(), + client_id.source(), + assertion.source(), + scope.source(), + authority + .as_ref() + .map(Sourced::source) + .unwrap_or(InputSource::Environment), + ]), + NativeAzureRequest::WorkloadIdentity { + tenant_id, + client_id, + token_file_path, + scope, + authority, + } => trusted_only(&[ + tenant_id.source(), + client_id.source(), + token_file_path.source(), + scope.source(), + authority + .as_ref() + .map(Sourced::source) + .unwrap_or(InputSource::Environment), + ]), + NativeAzureRequest::ManagedIdentity { + client_id, + scope, + selection_source, + } => trusted_only(&[ + client_id + .as_ref() + .map(Sourced::source) + .unwrap_or(InputSource::Environment), + scope.source(), + *selection_source, + ]), + NativeAzureRequest::DeveloperTools { + scope, + selection_source, + } => trusted_only(&[scope.source(), *selection_source]), + } +} + +fn is_request_controlled(value: &Sourced, optional: Option<&Sourced>) -> bool { + value.source() == InputSource::Request + || optional.is_some_and(|value| value.source() == InputSource::Request) +} + +fn trusted_only(sources: &[InputSource]) -> Result { + if sources.contains(&InputSource::Request) { + return mixed_sources(); + } + Ok(trusted_source(sources)) +} + +fn trusted_source(sources: &[InputSource]) -> InputSource { + if sources.contains(&InputSource::Deployment) { + InputSource::Deployment + } else { + InputSource::Environment + } +} + +fn mixed_sources() -> Result { + Err(AuthError::Configuration( + AuthConfigurationError::MixedAzureCredentialSources, + )) +} + +fn build_credential( + request: NativeAzureRequest, + transport: Option, +) -> Result, AuthError> { + match request { + NativeAzureRequest::ClientSecret { + tenant_id, + client_id, + client_secret, + authority, + .. + } => ClientSecretCredential::new( + tenant_id.value(), + client_id.into_value(), + Secret::new(client_secret.value().expose().to_string()), + Some(ClientSecretCredentialOptions { + client_options: client_options(authority.map(Sourced::into_value), transport), + }), + ) + .map(|credential| credential as Arc), + NativeAzureRequest::ClientAssertion { + tenant_id, + client_id, + assertion, + authority, + .. + } => ClientAssertionCredential::new( + tenant_id.into_value(), + client_id.into_value(), + StaticAssertion(assertion.into_value()), + Some(ClientAssertionCredentialOptions { + client_options: client_options(authority.map(Sourced::into_value), transport), + }), + ) + .map(|credential| credential as Arc), + NativeAzureRequest::WorkloadIdentity { + tenant_id, + client_id, + token_file_path, + authority, + .. + } => WorkloadIdentityCredential::new(Some(WorkloadIdentityCredentialOptions { + credential_options: azure_identity::ClientAssertionCredentialOptions { + client_options: client_options(authority.map(Sourced::into_value), transport), + }, + client_id: Some(client_id.into_value()), + tenant_id: Some(tenant_id.into_value()), + token_file_path: Some(token_file_path.into_value().into()), + })) + .map(|credential| credential as Arc), + NativeAzureRequest::ManagedIdentity { client_id, .. } => { + ManagedIdentityCredential::new(Some(ManagedIdentityCredentialOptions { + user_assigned_id: client_id + .map(Sourced::into_value) + .map(UserAssignedId::ClientId), + client_options: client_options(None, transport), + })) + .map(|credential| credential as Arc) + } + NativeAzureRequest::DeveloperTools { .. } => DeveloperToolsCredential::new(None) + .map(|credential| credential as Arc), + } + .map_err(|error| { + AuthError::Configuration(AuthConfigurationError::AzureCredentialInitialization( + error.to_string(), + )) + }) +} + +fn client_options( + authority: Option, + transport: Option, +) -> ClientOptions { + let cloud = authority.map(|authority_host| { + let mut custom = CustomConfiguration::default(); + custom.authority_host = authority_host; + Arc::new(CloudConfiguration::from(custom)) + }); + ClientOptions { + cloud, + transport, + ..Default::default() + } +} + +fn secret_digest(secret: &str) -> String { + format!("{:x}", Sha256::digest(secret.as_bytes())) +} + +#[derive(Debug)] +struct StaticAssertion(SecretValue); + +impl ClientAssertion for StaticAssertion { + fn secret<'life0, 'life1, 'async_trait>( + &'life0 self, + _options: Option>, + ) -> std::pin::Pin< + Box> + Send + 'async_trait>, + > + where + 'life0: 'async_trait, + 'life1: 'async_trait, + Self: 'async_trait, + { + Box::pin(async move { Ok(self.0.expose().to_string()) }) + } +} + +#[cfg(test)] +mod tests { + use std::sync::{Arc, Mutex}; + + use azure_core::http::headers::Headers; + use azure_core::http::{AsyncRawResponse, HttpClient, Request, StatusCode, Transport}; + use azure_core::{Bytes, Result}; + + use super::{NativeAzureRequest, NativeAzureTokenAcquirer, ValidatedAzureRequest}; + use crate::auth::{InputSource, SecretValue, Sourced}; + + fn deployment(value: T) -> Sourced { + Sourced::new(value, InputSource::Deployment) + } + + fn sourced_client_secret( + credential_source: InputSource, + authority_source: InputSource, + authority: &str, + ) -> NativeAzureRequest { + NativeAzureRequest::ClientSecret { + tenant_id: Sourced::new("tenant".to_string(), credential_source), + client_id: Sourced::new("client".to_string(), credential_source), + client_secret: Sourced::new(SecretValue::new("secret"), credential_source), + scope: Sourced::new("scope".to_string(), InputSource::Environment), + authority: Some(Sourced::new(authority.to_string(), authority_source)), + } + } + + fn client_secret_request( + tenant: &str, + client: &str, + secret: &str, + scope: &str, + authority: &str, + ) -> ValidatedAzureRequest { + ValidatedAzureRequest::new(NativeAzureRequest::ClientSecret { + tenant_id: deployment(tenant.to_string()), + client_id: deployment(client.to_string()), + client_secret: deployment(SecretValue::new(secret)), + scope: deployment(scope.to_string()), + authority: Some(deployment(authority.to_string())), + }) + .unwrap() + } + + #[derive(Debug, Default)] + struct RecordingTokenClient { + requests: Mutex>, + } + + impl HttpClient for RecordingTokenClient { + fn execute_request<'life0, 'life1, 'async_trait>( + &'life0 self, + request: &'life1 Request, + ) -> std::pin::Pin< + Box> + Send + 'async_trait>, + > + where + 'life0: 'async_trait, + 'life1: 'async_trait, + Self: 'async_trait, + { + Box::pin(async move { + let body = Bytes::from(request.body()); + self.requests.lock().unwrap().push(( + request.url().to_string(), + String::from_utf8(body.to_vec()).unwrap(), + )); + Ok(AsyncRawResponse::from_bytes( + StatusCode::Ok, + Headers::new(), + r#"{"token_type":"Bearer","expires_in":3600,"ext_expires_in":3600,"access_token":"native-token"}"#, + )) + }) + } + } + + #[tokio::test] + async fn client_secret_uses_sdk_protocol_and_reuses_cached_credential() { + let transport = Arc::new(RecordingTokenClient::default()); + let acquirer = + NativeAzureTokenAcquirer::with_transport(4, Transport::new(transport.clone())); + let request = client_secret_request( + "tenant", + "client", + "secret", + "https://service.test/.default", + "https://login.test", + ); + + let first = acquirer.acquire(request.clone()).await.unwrap(); + let second = acquirer.acquire(request).await.unwrap(); + + assert_eq!(first.secret().expose(), "native-token"); + assert_eq!(second.secret().expose(), "native-token"); + let requests = transport.requests.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert_eq!(requests[0].0, "https://login.test/tenant/oauth2/v2.0/token"); + assert!(requests[0].1.contains("client_id=client")); + assert!(requests[0].1.contains("client_secret=secret")); + assert!( + requests[0] + .1 + .contains("scope=https%3A%2F%2Fservice.test%2F.default") + ); + } + + #[tokio::test] + async fn credential_provider_cache_isolates_every_client_secret_identity_field() { + let transport = Arc::new(RecordingTokenClient::default()); + let acquirer = + NativeAzureTokenAcquirer::with_transport(16, Transport::new(transport.clone())); + let request = client_secret_request; + let base = request("tenant", "client", "secret", "scope", "https://login.test"); + let variants = [ + base.clone(), + request( + "other-tenant", + "client", + "secret", + "scope", + "https://login.test", + ), + request( + "tenant", + "other-client", + "secret", + "scope", + "https://login.test", + ), + request( + "tenant", + "client", + "other-secret", + "scope", + "https://login.test", + ), + request( + "tenant", + "client", + "secret", + "other-scope", + "https://login.test", + ), + request( + "tenant", + "client", + "secret", + "scope", + "https://other-login.test", + ), + ]; + + acquirer.acquire(base.clone()).await.unwrap(); + acquirer.acquire(base).await.unwrap(); + for request in variants.into_iter().skip(1) { + acquirer.acquire(request).await.unwrap(); + } + + assert_eq!(transport.requests.lock().unwrap().len(), 6); + } + + #[test] + fn request_authority_requires_request_owned_client_secret_identity() { + let error = ValidatedAzureRequest::new(sourced_client_secret( + InputSource::Deployment, + InputSource::Request, + "https://login.example", + )) + .unwrap_err(); + + assert!(matches!( + error, + crate::AuthError::Configuration( + crate::auth::error::AuthConfigurationError::MixedAzureCredentialSources + ) + )); + } + + #[test] + fn request_owned_client_secret_identity_can_select_custom_authority() { + let request = ValidatedAzureRequest::new(sourced_client_secret( + InputSource::Request, + InputSource::Request, + "https://login.example", + )) + .unwrap(); + + assert_eq!(request.credential_source(), InputSource::Request); + } + + #[test] + fn authority_is_restricted_to_an_https_origin() { + for authority in [ + "http://login.example", + "https://user@login.example", + "https://login.example/tenant", + "https://login.example?target=other", + ] { + let error = ValidatedAzureRequest::new(sourced_client_secret( + InputSource::Deployment, + InputSource::Deployment, + authority, + )) + .unwrap_err(); + assert!(matches!( + error, + crate::AuthError::Configuration( + crate::auth::error::AuthConfigurationError::InvalidAzureAuthority + ) + )); + } + } +} diff --git a/litellm-rust/crates/core/src/providers/azure_ai/auth/resolve.rs b/litellm-rust/crates/core/src/providers/azure_ai/auth/resolve.rs new file mode 100644 index 00000000000..025dd4f8740 --- /dev/null +++ b/litellm-rust/crates/core/src/providers/azure_ai/auth/resolve.rs @@ -0,0 +1,683 @@ +use crate::AuthError; +use crate::auth::error::AuthConfigurationError; +use crate::auth::{ + CredentialFileRef, CredentialLookup, CredentialRef, InputSource, ResolvedCredential, + SecretValue, Sourced, TokenProviderHandle, +}; +use std::future::Future; +use std::pin::Pin; +use std::sync::Arc; + +use super::native::{NativeAzureRequest, NativeAzureTokenAcquirer, ValidatedAzureRequest}; +use super::types::{AzureAuthInputs, AzureCredentialType, ConfigValue, DEFAULT_AZURE_SCOPE}; + +const AZURE_AD_TOKEN_ENV: &str = "AZURE_AD_TOKEN"; +const AZURE_TENANT_ID_ENV: &str = "AZURE_TENANT_ID"; +const AZURE_CLIENT_ID_ENV: &str = "AZURE_CLIENT_ID"; +const AZURE_CLIENT_SECRET_ENV: &str = "AZURE_CLIENT_SECRET"; +const AZURE_SCOPE_ENV: &str = "AZURE_SCOPE"; +const AZURE_AUTHORITY_HOST_ENV: &str = "AZURE_AUTHORITY_HOST"; +const AZURE_CREDENTIAL_ENV: &str = "AZURE_CREDENTIAL"; +const AZURE_FEDERATED_TOKEN_FILE_ENV: &str = "AZURE_FEDERATED_TOKEN_FILE"; + +#[derive(Clone, Debug)] +pub(crate) enum AzureCredentialPlan { + Supplied(Sourced), + Caller(TokenProviderHandle), + Oidc { + reference: Sourced, + tenant_id: Sourced, + client_id: Sourced, + scope: Sourced, + authority: Option>, + }, + Native(ValidatedAzureRequest), + Chain(Vec), + Missing, +} + +/// Rust counterpart to Python's `get_azure_ad_token`, not `BaseAzureLLM`. +pub(crate) struct AzureAuthService { + native: Arc, +} + +trait AzureTokenAcquirer: Send + Sync { + fn acquire( + &self, + request: ValidatedAzureRequest, + ) -> Pin> + Send + '_>>; +} + +impl AzureTokenAcquirer for NativeAzureTokenAcquirer { + fn acquire( + &self, + request: ValidatedAzureRequest, + ) -> Pin> + Send + '_>> { + Box::pin(NativeAzureTokenAcquirer::acquire(self, request)) + } +} + +impl Default for AzureAuthService { + fn default() -> Self { + Self { + native: Arc::new(NativeAzureTokenAcquirer::default()), + } + } +} + +impl AzureAuthService { + #[cfg(test)] + fn with_acquirer(native: Arc) -> Self { + Self { native } + } + + pub(crate) async fn get_azure_ad_token( + &self, + inputs: &AzureAuthInputs, + env_lookup: &(dyn Fn(&str) -> Option + Sync), + ) -> Result>, AuthError> { + match select_auth_plan(inputs, env_lookup)? { + AzureCredentialPlan::Supplied(credential) => Ok(Some(credential)), + AzureCredentialPlan::Caller(caller) => { + let credential = caller.acquire().await?; + if credential.secret().expose().is_empty() { + return Err(AuthError::EmptyAzureToken); + } + Ok(Some(Sourced::new(credential, InputSource::Deployment))) + } + AzureCredentialPlan::Oidc { + reference, + tenant_id, + client_id, + scope, + authority, + } => { + let assertion = resolve_reference(inputs, env_lookup, reference.value()) + .await? + .ok_or(AuthError::UnresolvedOidcReference)?; + let request = ValidatedAzureRequest::new(NativeAzureRequest::ClientAssertion { + tenant_id, + client_id, + assertion: Sourced::new(assertion, reference.source()), + assertion_identity: format!("{:?}", reference.value()), + scope, + authority, + })?; + let source = request.credential_source(); + self.native + .acquire(request) + .await + .map(|credential| Sourced::new(credential, source)) + .map(Some) + } + AzureCredentialPlan::Native(request) => { + let source = request.credential_source(); + self.native + .acquire(request) + .await + .map(|credential| Some(Sourced::new(credential, source))) + } + AzureCredentialPlan::Chain(requests) => { + let mut failures = Vec::new(); + for request in requests { + let source = request.credential_source(); + match self.native.acquire(request).await { + Ok(credential) => return Ok(Some(Sourced::new(credential, source))), + Err(error) => failures.push(error), + } + } + Err(AuthError::CredentialChain(failures)) + } + AzureCredentialPlan::Missing => Ok(None), + } + } +} + +pub(crate) fn select_auth_plan( + inputs: &AzureAuthInputs, + env_lookup: &dyn Fn(&str) -> Option, +) -> Result { + let token = configured_secret(&inputs.azure_ad_token, AZURE_AD_TOKEN_ENV, env_lookup); + let tenant_id = configured_string(&inputs.tenant_id, AZURE_TENANT_ID_ENV, env_lookup); + let client_id = configured_string(&inputs.client_id, AZURE_CLIENT_ID_ENV, env_lookup); + let client_secret = + configured_secret(&inputs.client_secret, AZURE_CLIENT_SECRET_ENV, env_lookup); + let scope = configured_string(&inputs.azure_scope, AZURE_SCOPE_ENV, env_lookup) + .unwrap_or_else(|| Sourced::new(DEFAULT_AZURE_SCOPE.to_string(), InputSource::Environment)); + let authority = configured_string( + &inputs.azure_authority_host, + AZURE_AUTHORITY_HOST_ENV, + env_lookup, + ); + let selector = configured_string(&inputs.azure_credential, AZURE_CREDENTIAL_ENV, env_lookup) + .map(|value| { + value + .value() + .parse::() + .map(|selector| Sourced::new(selector, value.source())) + }) + .transpose() + .map_err(|_| AuthError::Configuration(AuthConfigurationError::InvalidAzureSelector))?; + let federated_token_file = configured_string( + &inputs.federated_token_file, + AZURE_FEDERATED_TOKEN_FILE_ENV, + env_lookup, + ); + + if inputs.azure_ad_token_provider.is_none() + && let (Some(tenant_id), Some(client_id), Some(client_secret)) = + (tenant_id.clone(), client_id.clone(), client_secret) + { + return Ok(AzureCredentialPlan::Native(ValidatedAzureRequest::new( + NativeAzureRequest::ClientSecret { + tenant_id, + client_id, + client_secret, + scope, + authority, + }, + )?)); + } + + if let (Some(reference), Some(tenant_id), Some(client_id)) = ( + oidc_reference(&token)?, + tenant_id.clone(), + client_id.clone(), + ) { + return Ok(AzureCredentialPlan::Oidc { + reference, + tenant_id, + client_id, + scope, + authority, + }); + } + + if let Some(caller) = &inputs.azure_ad_token_provider { + return Ok(AzureCredentialPlan::Caller(caller.clone())); + } + + if let Some(token) = token { + return Ok(AzureCredentialPlan::Supplied(token.map(|token| { + ResolvedCredential::AccessToken { + token, + expires_on: None, + } + }))); + } + + if !*inputs.enable_azure_ad_token_refresh.value() && selector.is_none() { + return Ok(AzureCredentialPlan::Missing); + } + + select_native_plan( + selector, + tenant_id, + client_id, + federated_token_file, + scope, + authority, + inputs.enable_azure_ad_token_refresh.source(), + ) +} + +fn select_native_plan( + selector: Option>, + tenant_id: Option>, + client_id: Option>, + federated_token_file: Option>, + scope: Sourced, + authority: Option>, + refresh_source: InputSource, +) -> Result { + let selected = selector.unwrap_or_else(|| { + Sourced::new( + { + if federated_token_file.is_some() { + AzureCredentialType::DefaultAzureCredential + } else if client_id.is_some() { + AzureCredentialType::ManagedIdentityCredential + } else { + AzureCredentialType::DefaultAzureCredential + } + }, + refresh_source, + ) + }); + let selection_source = selected.source(); + + match selected.into_value() { + AzureCredentialType::ClientSecretCredential => Err(AuthError::Configuration( + AuthConfigurationError::MissingClientSecretFields, + )), + AzureCredentialType::WorkloadIdentityCredential => { + Ok(AzureCredentialPlan::Native(ValidatedAzureRequest::new( + workload_request(tenant_id, client_id, federated_token_file, scope, authority)?, + )?)) + } + AzureCredentialType::ManagedIdentityCredential => Ok(AzureCredentialPlan::Native( + ValidatedAzureRequest::new(NativeAzureRequest::ManagedIdentity { + client_id, + scope, + selection_source, + })?, + )), + AzureCredentialType::DefaultAzureCredential => { + let workload = match (tenant_id, client_id.clone(), federated_token_file) { + (Some(tenant_id), Some(client_id), Some(token_file_path)) => { + Some(NativeAzureRequest::WorkloadIdentity { + tenant_id, + client_id, + token_file_path, + scope: scope.clone(), + authority, + }) + } + _ => None, + }; + Ok(AzureCredentialPlan::Chain( + workload + .into_iter() + .chain(std::iter::once(NativeAzureRequest::ManagedIdentity { + client_id, + scope: scope.clone(), + selection_source, + })) + .chain(std::iter::once(NativeAzureRequest::DeveloperTools { + scope, + selection_source, + })) + .map(ValidatedAzureRequest::new) + .collect::, _>>()?, + )) + } + AzureCredentialType::DeploymentIdentityCredential => { + let workload = match (tenant_id, client_id.clone(), federated_token_file) { + (Some(tenant_id), Some(client_id), Some(token_file_path)) => { + Some(NativeAzureRequest::WorkloadIdentity { + tenant_id, + client_id, + token_file_path, + scope: scope.clone(), + authority, + }) + } + _ => None, + }; + let user_assigned = client_id.map(|client_id| NativeAzureRequest::ManagedIdentity { + client_id: Some(client_id), + scope: scope.clone(), + selection_source, + }); + Ok(AzureCredentialPlan::Chain( + workload + .into_iter() + .chain(user_assigned) + .chain(std::iter::once(NativeAzureRequest::ManagedIdentity { + client_id: None, + scope, + selection_source, + })) + .map(ValidatedAzureRequest::new) + .collect::, _>>()?, + )) + } + } +} + +fn workload_request( + tenant_id: Option>, + client_id: Option>, + token_file_path: Option>, + scope: Sourced, + authority: Option>, +) -> Result { + Ok(NativeAzureRequest::WorkloadIdentity { + tenant_id: tenant_id.ok_or(AuthError::Configuration( + AuthConfigurationError::MissingWorkloadTenant, + ))?, + client_id: client_id.ok_or(AuthError::Configuration( + AuthConfigurationError::MissingWorkloadClient, + ))?, + token_file_path: token_file_path.ok_or(AuthError::Configuration( + AuthConfigurationError::MissingWorkloadTokenFile, + ))?, + scope, + authority, + }) +} + +fn configured_string( + configured: &ConfigValue, + environment_name: &str, + env_lookup: &dyn Fn(&str) -> Option, +) -> Option> { + configured + .as_value() + .filter(|value| !value.value().is_empty()) + .cloned() + .or_else(|| { + env_lookup(environment_name) + .filter(|value| !value.is_empty()) + .map(|value| Sourced::new(value, InputSource::Environment)) + }) +} + +fn configured_secret( + configured: &ConfigValue, + environment_name: &str, + env_lookup: &dyn Fn(&str) -> Option, +) -> Option> { + configured + .as_value() + .filter(|value| !value.value().expose().is_empty()) + .cloned() + .or_else(|| { + env_lookup(environment_name) + .filter(|value| !value.is_empty()) + .map(|value| Sourced::new(SecretValue::new(value), InputSource::Environment)) + }) +} + +async fn resolve_reference( + inputs: &AzureAuthInputs, + env_lookup: &(dyn Fn(&str) -> Option + Sync), + reference: &CredentialRef, +) -> Result, AuthError> { + let lookup = match reference { + CredentialRef::Explicit(secret) => return Ok(Some(secret.clone())), + CredentialRef::Env(name) => env_lookup(name) + .filter(|value| !value.is_empty()) + .map(SecretValue::new) + .map_or(CredentialLookup::Missing, CredentialLookup::Found), + CredentialRef::None => return Ok(None), + CredentialRef::File(_) | CredentialRef::Request(_) | CredentialRef::Host(_) => { + let resolver = inputs + .credential_resolver + .as_ref() + .ok_or(AuthError::Configuration( + AuthConfigurationError::MissingHostResolver, + ))?; + resolver.resolve(reference).await? + } + }; + Ok(match lookup { + CredentialLookup::Found(secret) => Some(secret), + CredentialLookup::Missing | CredentialLookup::Declined => None, + }) +} + +fn oidc_reference( + token: &Option>, +) -> Result>, AuthError> { + let Some(token) = token.as_ref() else { + return Ok(None); + }; + let value = token.value().expose(); + if token.source() == InputSource::Request && value.starts_with("oidc/") { + return Err(AuthError::Configuration( + AuthConfigurationError::RequestAzureCredentialReference, + )); + } + if let Some(name) = value.strip_prefix("oidc/env/") { + return non_empty_reference(name, "OIDC environment reference") + .map(CredentialRef::Env) + .map(|reference| Sourced::new(reference, token.source())) + .map(Some); + } + if let Some(name) = value.strip_prefix("oidc/env_path/") { + return non_empty_reference(name, "OIDC environment path reference") + .map(|name| CredentialRef::File(CredentialFileRef::EnvironmentVariable(name))) + .map(|reference| Sourced::new(reference, token.source())) + .map(Some); + } + if let Some(path) = value.strip_prefix("oidc/file/") { + let path = non_empty_reference(path, "OIDC file reference")?; + return Ok(Some(Sourced::new( + CredentialRef::File(CredentialFileRef::Path(path.into())), + token.source(), + ))); + } + if value.starts_with("oidc/") { + return Err(AuthError::Configuration( + AuthConfigurationError::UnsupportedOidcReference, + )); + } + Ok(None) +} + +fn non_empty_reference(value: &str, kind: &str) -> Result { + if value.is_empty() { + return Err(AuthError::Configuration( + AuthConfigurationError::EmptyReference(kind.to_string()), + )); + } + Ok(value.to_string()) +} + +#[cfg(test)] +mod tests { + use std::future::Future; + use std::sync::{Arc, Mutex}; + + use serde_json::json; + + use super::{ + AzureAuthService, AzureCredentialPlan, AzureTokenAcquirer, oidc_reference, + resolve_reference, select_auth_plan, + }; + use crate::AuthError; + use crate::auth::ResolvedCredential; + use crate::auth::{ + CredentialFileRef, CredentialLookup, CredentialLookupFuture, CredentialRef, + CredentialResolver, CredentialResolverHandle, InputSource, SecretValue, Sourced, + }; + use crate::providers::azure_ai::auth::native::ValidatedAzureRequest; + use crate::providers::azure_ai::auth::types::AzureAuthInputs; + + #[derive(Debug)] + struct FileResolver; + + struct ChainAcquirer { + requests: Mutex>, + succeed_on: Option<&'static str>, + } + + impl AzureTokenAcquirer for ChainAcquirer { + fn acquire( + &self, + request: ValidatedAzureRequest, + ) -> std::pin::Pin< + Box> + Send + '_>, + > { + let kind = request.kind(); + self.requests.lock().unwrap().push(kind); + Box::pin(async move { + if self.succeed_on == Some(kind) { + Ok(ResolvedCredential::AccessToken { + token: SecretValue::new("chain-token"), + expires_on: None, + }) + } else { + Err(AuthError::AzureTokenAcquisition(format!("{kind} failed"))) + } + }) + } + } + + impl CredentialResolver for FileResolver { + fn resolve<'a>(&'a self, reference: &'a CredentialRef) -> CredentialLookupFuture<'a> { + Box::pin(async move { + Ok(match reference { + CredentialRef::File(CredentialFileRef::Path(path)) + if path == std::path::Path::new("/run/secrets/assertion") => + { + CredentialLookup::Found(SecretValue::new("rotated-assertion")) + } + _ => CredentialLookup::Declined, + }) + }) + } + } + + #[test] + fn null_and_empty_values_fall_back_to_environment() { + let params = json!({"tenant_id": null, "client_id": "", "client_secret": null}); + let inputs = AzureAuthInputs::from_optional_params(params.as_object().unwrap()).unwrap(); + let plan = select_auth_plan(&inputs, &|name| match name { + "AZURE_TENANT_ID" => Some("tenant".to_string()), + "AZURE_CLIENT_ID" => Some("client".to_string()), + "AZURE_CLIENT_SECRET" => Some("secret".to_string()), + _ => None, + }) + .unwrap(); + + assert!(matches!(plan, AzureCredentialPlan::Native(_))); + } + + #[test] + fn supplied_token_does_not_require_refresh() { + let params = json!({"azure_ad_token": "token"}); + let inputs = AzureAuthInputs::from_optional_params(params.as_object().unwrap()).unwrap(); + + assert!(matches!( + select_auth_plan(&inputs, &|_| None).unwrap(), + AzureCredentialPlan::Supplied(_) + )); + } + + #[test] + fn oidc_reference_is_deferred() { + let params = json!({ + "azure_ad_token": "oidc/env/ASSERTION", + "tenant_id": "tenant", + "client_id": "client" + }); + let inputs = AzureAuthInputs::from_optional_params(params.as_object().unwrap()).unwrap(); + + assert!(matches!( + select_auth_plan(&inputs, &|_| None).unwrap(), + AzureCredentialPlan::Oidc { + reference, + .. + } if reference.value() == &CredentialRef::Env("ASSERTION".to_string()) + )); + } + + #[test] + fn oidc_file_location_is_typed_before_resolution() { + assert_eq!( + oidc_reference(&Some(Sourced::new( + SecretValue::new("oidc/file//run/secrets/assertion"), + InputSource::Deployment, + ))) + .unwrap() + .map(Sourced::into_value), + Some(CredentialRef::File(CredentialFileRef::Path( + "/run/secrets/assertion".into() + ))) + ); + } + + #[test] + fn unsupported_oidc_reference_is_rejected_during_plan_creation() { + let error = oidc_reference(&Some(Sourced::new( + SecretValue::new("oidc/vault/assertion"), + InputSource::Deployment, + ))) + .expect_err("unsupported backend must fail validation"); + + assert!(error.to_string().contains("unsupported OIDC reference")); + } + + #[test] + fn request_oidc_reference_is_rejected_before_lookup() { + let params = json!({ + "azure_ad_token": "oidc/env/ASSERTION", + "tenant_id": "tenant", + "client_id": "client" + }); + let sources = std::collections::BTreeMap::from([ + ("azure_ad_token".to_string(), InputSource::Request), + ("tenant_id".to_string(), InputSource::Request), + ("client_id".to_string(), InputSource::Request), + ]); + let inputs = + AzureAuthInputs::from_sourced_optional_params(params.as_object().unwrap(), &sources) + .unwrap(); + + let error = select_auth_plan(&inputs, &|name| { + assert_ne!(name, "ASSERTION"); + None + }) + .unwrap_err(); + + assert!(matches!( + error, + AuthError::Configuration( + crate::auth::error::AuthConfigurationError::RequestAzureCredentialReference + ) + )); + } + + #[tokio::test] + async fn host_resolver_owns_file_access() { + let inputs = AzureAuthInputs { + credential_resolver: Some(CredentialResolverHandle::new(Arc::new(FileResolver))), + ..AzureAuthInputs::default() + }; + let reference = + CredentialRef::File(CredentialFileRef::Path("/run/secrets/assertion".into())); + + let resolved = resolve_reference(&inputs, &|_| None, &reference) + .await + .unwrap(); + + assert_eq!(resolved, Some(SecretValue::new("rotated-assertion"))); + } + + #[tokio::test] + async fn default_chain_uses_declared_order_and_stops_after_success() { + let acquirer = Arc::new(ChainAcquirer { + requests: Mutex::new(Vec::new()), + succeed_on: Some("developer-tools"), + }); + let service = AzureAuthService::with_acquirer(acquirer.clone()); + let inputs = AzureAuthInputs { + enable_azure_ad_token_refresh: Sourced::new(true, InputSource::Deployment), + ..Default::default() + }; + + let credential = service + .get_azure_ad_token(&inputs, &|_| None) + .await + .unwrap() + .unwrap(); + + assert_eq!(credential.value().secret().expose(), "chain-token"); + assert_eq!( + *acquirer.requests.lock().unwrap(), + ["managed-identity", "developer-tools"] + ); + } + + #[tokio::test] + async fn chain_reports_each_acquisition_failure() { + let acquirer = Arc::new(ChainAcquirer { + requests: Mutex::new(Vec::new()), + succeed_on: None, + }); + let service = AzureAuthService::with_acquirer(acquirer); + let inputs = AzureAuthInputs { + enable_azure_ad_token_refresh: Sourced::new(true, InputSource::Deployment), + ..Default::default() + }; + + let error = service + .get_azure_ad_token(&inputs, &|_| None) + .await + .unwrap_err(); + + assert!(matches!(error, AuthError::CredentialChain(errors) if errors.len() == 2)); + } +} diff --git a/litellm-rust/crates/core/src/providers/azure_ai/auth/types.rs b/litellm-rust/crates/core/src/providers/azure_ai/auth/types.rs new file mode 100644 index 00000000000..f15d526d945 --- /dev/null +++ b/litellm-rust/crates/core/src/providers/azure_ai/auth/types.rs @@ -0,0 +1,195 @@ +use crate::auth::error::AuthConfigurationError; +use serde_json::{Map, Value}; +use std::collections::BTreeMap; +use strum::EnumString; + +use crate::AuthError; +use crate::auth::{ + CredentialResolverHandle, InputSource, SecretValue, Sourced, TokenProviderHandle, +}; + +pub const DEFAULT_AZURE_SCOPE: &str = "https://cognitiveservices.azure.com/.default"; + +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub enum ConfigValue { + #[default] + Absent, + ExplicitNone(InputSource), + Value(Sourced), +} + +impl ConfigValue { + pub fn as_value(&self) -> Option<&Sourced> { + match self { + Self::Value(value) => Some(value), + Self::Absent | Self::ExplicitNone(_) => None, + } + } +} + +#[derive(Clone, Copy, Debug, EnumString, PartialEq, Eq, Hash)] +#[allow(clippy::enum_variant_names)] +pub enum AzureCredentialType { + ClientSecretCredential, + ManagedIdentityCredential, + DefaultAzureCredential, + DeploymentIdentityCredential, + WorkloadIdentityCredential, +} + +#[derive(Clone, Debug, Default)] +pub struct AzureAuthInputs { + pub azure_ad_token: ConfigValue, + pub azure_ad_token_provider: Option, + pub credential_resolver: Option, + pub tenant_id: ConfigValue, + pub client_id: ConfigValue, + pub client_secret: ConfigValue, + pub azure_scope: ConfigValue, + pub azure_authority_host: ConfigValue, + pub azure_credential: ConfigValue, + pub federated_token_file: ConfigValue, + pub enable_azure_ad_token_refresh: Sourced, +} + +impl AzureAuthInputs { + #[cfg(test)] + pub fn from_optional_params(params: &Map) -> Result { + Self::from_sourced_optional_params(params, &BTreeMap::new()) + } + + pub fn from_sourced_optional_params( + params: &Map, + sources: &BTreeMap, + ) -> Result { + Ok(Self { + azure_ad_token: secret_config(params, sources, "azure_ad_token")?, + azure_ad_token_provider: None, + credential_resolver: None, + tenant_id: string_config(params, sources, "tenant_id")?, + client_id: string_config(params, sources, "client_id")?, + client_secret: secret_config(params, sources, "client_secret")?, + azure_scope: string_config(params, sources, "azure_scope")?, + azure_authority_host: string_config(params, sources, "azure_authority_host")?, + azure_credential: string_config(params, sources, "azure_credential")?, + federated_token_file: string_config(params, sources, "azure_federated_token_file")?, + enable_azure_ad_token_refresh: Sourced::new( + params + .get("enable_azure_ad_token_refresh") + .and_then(Value::as_bool) + .unwrap_or(false), + source_for(sources, "enable_azure_ad_token_refresh"), + ), + }) + } +} + +fn string_config( + params: &Map, + sources: &BTreeMap, + name: &str, +) -> Result, AuthError> { + let source = source_for(sources, name); + match params.get(name) { + None => Ok(ConfigValue::Absent), + Some(Value::Null) => Ok(ConfigValue::ExplicitNone(source)), + Some(Value::String(value)) => Ok(ConfigValue::Value(Sourced::new(value.clone(), source))), + Some(_) => Err(AuthError::Configuration( + AuthConfigurationError::InvalidFieldType(name.to_string()), + )), + } +} + +fn secret_config( + params: &Map, + sources: &BTreeMap, + name: &str, +) -> Result, AuthError> { + Ok(match string_config(params, sources, name)? { + ConfigValue::Absent => ConfigValue::Absent, + ConfigValue::ExplicitNone(source) => ConfigValue::ExplicitNone(source), + ConfigValue::Value(value) => ConfigValue::Value(value.map(SecretValue::new)), + }) +} + +fn source_for(sources: &BTreeMap, name: &str) -> InputSource { + sources.get(name).copied().unwrap_or_default() +} + +#[cfg(test)] +mod tests { + use serde_json::json; + + use std::collections::BTreeMap; + + use super::{AzureAuthInputs, AzureCredentialType, ConfigValue}; + use crate::auth::{InputSource, Sourced}; + + #[test] + fn selector_parsing_is_exact() { + assert_eq!( + "ClientSecretCredential".parse::(), + Ok(AzureCredentialType::ClientSecretCredential) + ); + assert!( + "clientsecretcredential" + .parse::() + .is_err() + ); + } + + #[test] + fn defaults_preserve_absence() { + let inputs = AzureAuthInputs::default(); + + assert_eq!(inputs.tenant_id, ConfigValue::Absent); + assert_eq!(inputs.azure_ad_token, ConfigValue::Absent); + } + + #[test] + fn parsing_distinguishes_null_empty_and_absent() { + let params = json!({"tenant_id": null, "client_id": ""}); + let inputs = AzureAuthInputs::from_optional_params(params.as_object().unwrap()).unwrap(); + + assert_eq!( + inputs.tenant_id, + ConfigValue::ExplicitNone(InputSource::Deployment) + ); + assert_eq!( + inputs.client_id, + ConfigValue::Value(Sourced::new(String::new(), InputSource::Deployment)) + ); + assert_eq!(inputs.client_secret, ConfigValue::Absent); + } + + #[test] + fn parsing_preserves_trusted_input_sources() { + let params = json!({"tenant_id": "tenant", "client_secret": null}); + let sources = BTreeMap::from([ + ("tenant_id".to_string(), InputSource::Request), + ("client_secret".to_string(), InputSource::Request), + ]); + let inputs = + AzureAuthInputs::from_sourced_optional_params(params.as_object().unwrap(), &sources) + .unwrap(); + + assert_eq!( + inputs.tenant_id, + ConfigValue::Value(Sourced::new("tenant".to_string(), InputSource::Request)) + ); + assert_eq!( + inputs.client_secret, + ConfigValue::ExplicitNone(InputSource::Request) + ); + } + + #[test] + fn debug_does_not_expose_secrets() { + let params = json!({"azure_ad_token": "token-value", "client_secret": "secret-value"}); + let inputs = AzureAuthInputs::from_optional_params(params.as_object().unwrap()).unwrap(); + let debug = format!("{inputs:?}"); + + assert!(!debug.contains("token-value")); + assert!(!debug.contains("secret-value")); + } +} diff --git a/litellm-rust/crates/core/src/providers/azure_ai/messages/transformation.rs b/litellm-rust/crates/core/src/providers/azure_ai/messages/transformation.rs index b8ca10461fb..585b34f393f 100644 --- a/litellm-rust/crates/core/src/providers/azure_ai/messages/transformation.rs +++ b/litellm-rust/crates/core/src/providers/azure_ai/messages/transformation.rs @@ -1,3 +1,4 @@ +use crate::auth::error::MissingCredential; use crate::error::Error; use crate::messages::transformation::{AnthropicMessagesProviderConfig, MessagesAuthStrategy}; use crate::messages::types::{ @@ -32,12 +33,7 @@ pub fn resolve_azure_api_key( non_empty(api_key) .map(str::to_string) .or_else(|| env_lookup(AZURE_API_KEY_ENV).filter(|value| !value.trim().is_empty())) - .ok_or_else(|| { - Error::Auth( - "Missing Azure API Key - Set `api_key` or the AZURE_API_KEY environment variable" - .to_string(), - ) - }) + .ok_or_else(|| Error::from(crate::AuthError::from(MissingCredential::AzureApiKey))) } pub fn complete_azure_anthropic_url( @@ -47,13 +43,7 @@ pub fn complete_azure_anthropic_url( let api_base = non_empty(api_base) .map(str::to_string) .or_else(|| env_lookup(AZURE_API_BASE_ENV).filter(|value| !value.trim().is_empty())) - .ok_or_else(|| { - Error::Auth( - "Missing Azure API Base - Set `api_base` or the AZURE_API_BASE environment variable. \ - Expected format: https://.services.ai.azure.com/anthropic" - .to_string(), - ) - })?; + .ok_or_else(|| Error::from(crate::AuthError::from(MissingCredential::AzureApiBase)))?; let api_base = api_base.trim_end_matches('/'); diff --git a/litellm-rust/crates/core/src/providers/azure_ai/mod.rs b/litellm-rust/crates/core/src/providers/azure_ai/mod.rs index 5d13fa93e00..f2d5b679aee 100644 --- a/litellm-rust/crates/core/src/providers/azure_ai/mod.rs +++ b/litellm-rust/crates/core/src/providers/azure_ai/mod.rs @@ -1,2 +1,3 @@ +pub(crate) mod auth; pub mod messages; pub mod ocr; diff --git a/litellm-rust/crates/core/tests/azure_ai_ocr.rs b/litellm-rust/crates/core/tests/azure_ai_ocr.rs index 3d7fb2e54a3..d7d532cfef1 100644 --- a/litellm-rust/crates/core/tests/azure_ai_ocr.rs +++ b/litellm-rust/crates/core/tests/azure_ai_ocr.rs @@ -45,6 +45,28 @@ async fn facade_executes_azure_mistral_with_prepared_auth() { ); } +#[tokio::test] +async fn facade_acquires_supplied_entra_token_for_final_request() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({"pages":[]}))]).await; + let mut request = wire_request( + "azure_ai/model", + &base, + json!({"azure_ad_token":"rust-owned-token"}), + ); + request.connection.api_key = None; + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!( + requests[0] + .to_ascii_lowercase() + .contains("authorization: bearer rust-owned-token\r\n") + ); +} + struct ReplaceBodyDocument; impl OcrHooks for ReplaceBodyDocument { diff --git a/litellm-rust/crates/core/tests/ocr.rs b/litellm-rust/crates/core/tests/ocr.rs index d1828dfb816..cecd8869741 100644 --- a/litellm-rust/crates/core/tests/ocr.rs +++ b/litellm-rust/crates/core/tests/ocr.rs @@ -21,6 +21,7 @@ fn request_boundary_selects_mistral_and_rejects_unknown_providers() { .as_object() .unwrap() .clone(), + input_sources: Default::default(), timeout_seconds: None, }; assert!(decode_request(request).is_ok()); @@ -33,6 +34,7 @@ fn request_boundary_selects_mistral_and_rejects_unknown_providers() { custom_llm_provider: Some("unknown".into()), extra_headers: None, optional_params: serde_json::Map::new(), + input_sources: Default::default(), timeout_seconds: None, }) .is_err() diff --git a/litellm-rust/crates/core/tests/ocr/support.rs b/litellm-rust/crates/core/tests/ocr/support.rs index 45047a0e62a..a2e67dffc7d 100644 --- a/litellm-rust/crates/core/tests/ocr/support.rs +++ b/litellm-rust/crates/core/tests/ocr/support.rs @@ -30,6 +30,7 @@ pub(crate) fn wire_request(model: &str, base: &str, options: Value) -> LiteLLMOc custom_llm_provider: None, extra_headers: None, optional_params: options.as_object().unwrap().clone(), + input_sources: Default::default(), timeout_seconds: Some(2.0), }) .unwrap() diff --git a/litellm-rust/crates/python-bridge/src/routes/definition.rs b/litellm-rust/crates/python-bridge/src/routes/definition.rs index bc51647cbad..97313651011 100644 --- a/litellm-rust/crates/python-bridge/src/routes/definition.rs +++ b/litellm-rust/crates/python-bridge/src/routes/definition.rs @@ -225,7 +225,7 @@ mod tests { ( "ocr", "aocr", - "(model, document, api_key=None, api_base=None, custom_llm_provider=None, extra_headers=None, optional_params=None, timeout_seconds=None)", + "(model, document, api_key=None, api_base=None, custom_llm_provider=None, extra_headers=None, optional_params=None, input_sources=None, timeout_seconds=None)", ), ( "transcription", diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index 2e6900f784f..50095e3ebf2 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -22,6 +22,12 @@ fn prepare_ocr( timeout_seconds: inputs.timeout_seconds, })?; let optional_params = object_or_empty("optional_params", inputs.optional_params)?; + let input_sources = inputs + .input_sources + .map(serde_json::from_value) + .transpose() + .map_err(|error| pyo3::exceptions::PyValueError::new_err(error.to_string()))? + .unwrap_or_default(); Ok(async move { let RouteOptions { @@ -41,6 +47,7 @@ fn prepare_ocr( custom_llm_provider, extra_headers, optional_params, + input_sources, timeout_seconds: timeout.map(|value| value.as_secs_f64()), })?; return litellm_core::ocr::ocr(request) @@ -82,6 +89,8 @@ bridge_route! { extra_headers: Option, #[pyo3(from_py_with = litellm_python_interop::from_py)] optional_params: Option, + #[pyo3(from_py_with = litellm_python_interop::from_py)] + input_sources: Option, timeout_seconds: Option, }, prepare = prepare_ocr, diff --git a/litellm/ocr/main.py b/litellm/ocr/main.py index df3f9d2096b..56bfd98895d 100644 --- a/litellm/ocr/main.py +++ b/litellm/ocr/main.py @@ -10,6 +10,7 @@ import re from collections.abc import Callable, Coroutine, Mapping from dataclasses import dataclass from io import IOBase +from types import MappingProxyType from typing import Any, Final, cast import httpx @@ -19,6 +20,7 @@ from litellm._logging import verbose_logger from litellm.constants import request_timeout from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.azure_ai.ocr.common_utils import ( + is_azure_cohere_parse_model, is_azure_document_intelligence_model, ) from litellm.llms.base_llm.ocr.transformation import ( @@ -52,21 +54,32 @@ class _PreparedOCRRequest: litellm_params: dict[str, object] effective_timeout: float | httpx.Timeout litellm_logging_obj: LiteLLMLoggingObj + caller_supplied_api_key: bool = True + caller_supplied_api_base: bool = True -@dataclass -class _PreparedRustOCRCall: - api_key: str | None - api_base: str | None - headers: dict[str, object] - optional_params: dict[str, object] - - -_RUST_OCR_PROVIDERS: Final = { - "mistral", - "azure_ai", - "vertex_ai", -} +_RUST_OCR_PROVIDERS: Final = frozenset({"mistral", "azure_ai", "vertex_ai"}) +_RUST_OCR_CONFIG_FIELDS: Final = frozenset( + { + "azure_ad_token", + "tenant_id", + "client_id", + "client_secret", + "azure_scope", + "azure_authority_host", + "azure_credential", + "azure_federated_token_file", + "vertex_credentials", + "vertex_ai_credentials", + "vertex_project", + "vertex_ai_project", + "vertex_location", + "vertex_ai_location", + } +) +_RUST_OCR_SECRET_FIELDS: Final = frozenset( + {"azure_ad_token", "client_secret", "azure_federated_token_file", "vertex_credentials", "vertex_ai_credentials"} +) def _prepare_ocr_request( @@ -94,6 +107,7 @@ def _prepare_ocr_request( if doc_type not in ["document_url", "image_url"]: raise ValueError(f"Invalid document type: {doc_type}. Must be 'document_url', 'image_url', or 'file'") + caller_supplied_api_key: Final = api_key is not None caller_supplied_api_base: Final = api_base is not None ( @@ -187,182 +201,256 @@ def _prepare_ocr_request( litellm_params=dict(litellm_params), effective_timeout=effective_timeout, litellm_logging_obj=litellm_logging_obj, + caller_supplied_api_key=caller_supplied_api_key, + caller_supplied_api_base=caller_supplied_api_base, ) -def _rust_ocr_supported(prepared_request: _PreparedOCRRequest) -> bool: - if prepared_request.optional_params.get(OCR_REQUEST_FORMAT_PARAM) == "native": - return False - if not prepared_request.provider_config.supports_rust_bridge(): - return False - return prepared_request.custom_llm_provider in _RUST_OCR_PROVIDERS - - -def _rust_bridge_optional_params( - prepared_request: _PreparedOCRRequest, - resolve_secret: Callable[[str], str | None], -) -> dict[str, object]: - optional_params: Final = dict(prepared_request.optional_params) - if prepared_request.custom_llm_provider == "vertex_ai": - vertex_project: Final = ( - prepared_request.litellm_params.get("vertex_project") - or prepared_request.litellm_params.get("vertex_ai_project") - or litellm.vertex_project - or resolve_secret("VERTEXAI_PROJECT") - ) - vertex_location: Final = ( - prepared_request.litellm_params.get("vertex_location") - or prepared_request.litellm_params.get("vertex_ai_location") - or litellm.vertex_location - or resolve_secret("VERTEXAI_LOCATION") - or resolve_secret("VERTEX_LOCATION") - ) - if vertex_project is not None: - optional_params["vertex_project"] = vertex_project - if vertex_location is not None: - optional_params["vertex_location"] = vertex_location - return optional_params - - -def _rust_bridge_api_base( - prepared_request: _PreparedOCRRequest, - resolve_secret: Callable[[str], str | None], -) -> str | None: - if prepared_request.api_base is not None: - return prepared_request.api_base - if prepared_request.custom_llm_provider == "azure_ai": - if is_azure_document_intelligence_model(prepared_request.model): - return resolve_secret("AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT") - return resolve_secret("AZURE_AI_API_BASE") +def _rust_ocr_provider(request: rust_ocr_bridge.LiteLLMOcrRequest) -> str | None: + if request.custom_llm_provider is not None: + return request.custom_llm_provider + prefix: Final = request.model.partition("/")[0] + if prefix in _RUST_OCR_PROVIDERS: + return prefix + if request.model.startswith("mistral-ocr"): + return "mistral" return None -def _prepare_rust_ocr_call( - prepared_request: _PreparedOCRRequest, - resolve_api_key: Callable[[str], str | None], -) -> _PreparedRustOCRCall: - provider_config: Final = prepared_request.provider_config - api_key_env_var: Final = provider_config.get_api_key_env_var() - resolved_api_key: Final = prepared_request.api_key or ( - resolve_api_key(api_key_env_var) if api_key_env_var is not None else None +def _rust_ocr_supported(request: rust_ocr_bridge.LiteLLMOcrRequest) -> bool: + provider: Final = _rust_ocr_provider(request) + if provider not in _RUST_OCR_PROVIDERS or request.kwargs.get(OCR_REQUEST_FORMAT_PARAM) == "native": + return False + if provider == "azure_ai": + return ( + not is_azure_cohere_parse_model(request.model) + and not callable(request.kwargs.get("azure_ad_token_provider")) + and request.kwargs.get("azure_username") is None + and request.kwargs.get("azure_password") is None + ) + return True + + +def _rust_bridge_optional_params( + request: rust_ocr_bridge.LiteLLMOcrRequest, + resolve_secret: Callable[[str], str | None], +) -> Mapping[str, object]: + optional_params: Final = MappingProxyType( + { + name: value + for name, value in request.kwargs.items() + if (name not in GenericLiteLLMParams.model_fields or name in _RUST_OCR_CONFIG_FIELDS) + and name not in {"litellm_logging_obj", "aocr", "litellm_call_id", "proxy_server_request"} + } ) - resolved_headers: Final = provider_config.validate_environment( - headers=prepared_request.extra_headers or {}, - model=prepared_request.model, - api_key=resolved_api_key, - api_base=prepared_request.api_base, - litellm_params=prepared_request.litellm_params, + provider: Final = _rust_ocr_provider(request) + if provider == "azure_ai" and litellm.enable_azure_ad_token_refresh is True: + return MappingProxyType({**optional_params, "enable_azure_ad_token_refresh": True}) + if provider != "vertex_ai": + return optional_params + project: Final = ( + request.kwargs.get("vertex_project") + or request.kwargs.get("vertex_ai_project") + or litellm.vertex_project + or resolve_secret("VERTEXAI_PROJECT") ) - resolved_complete_url: Final = provider_config.get_complete_url( - api_base=prepared_request.api_base, - model=prepared_request.model, - optional_params=prepared_request.optional_params, - litellm_params=prepared_request.litellm_params, + location: Final = ( + request.kwargs.get("vertex_location") + or request.kwargs.get("vertex_ai_location") + or litellm.vertex_location + or resolve_secret("VERTEXAI_LOCATION") + or resolve_secret("VERTEX_LOCATION") ) - rust_api_base: Final = _rust_bridge_api_base(prepared_request, resolve_api_key) - rust_optional_params: Final = _rust_bridge_optional_params(prepared_request, resolve_api_key) - prepared_request.litellm_logging_obj.pre_call( + credentials: Final = ( + request.kwargs.get("vertex_credentials") + or request.kwargs.get("vertex_ai_credentials") + or resolve_secret("VERTEXAI_CREDENTIALS") + ) + vertex_params: Final = MappingProxyType( + { + name: value + for name, value in ( + ("vertex_project", project), + ("vertex_location", location), + ("vertex_credentials", credentials), + ) + if value is not None + } + ) + return MappingProxyType({**optional_params, **vertex_params}) + + +def _rust_bridge_input_sources( + request: rust_ocr_bridge.LiteLLMOcrRequest, + optional_params: Mapping[str, object], +) -> Mapping[str, str]: + proxy_request: Final = request.kwargs.get("proxy_server_request") + if not isinstance(proxy_request, Mapping): + return MappingProxyType({}) + proxy_request_mapping: Final = cast( # cast-ok: runtime Mapping check loses generic key and value types + Mapping[object, object], proxy_request + ) + body_value: Final = proxy_request_mapping.get("body") + if not isinstance(body_value, Mapping): + return MappingProxyType({}) + body: Final = cast( # cast-ok: runtime Mapping check loses generic key and value types + Mapping[object, object], body_value + ) + credential_fields_value: Final = proxy_request_mapping.get("credential_fields", ()) + credential_fields: Final = ( + frozenset(name for name in credential_fields_value if isinstance(name, str)) + if isinstance(credential_fields_value, (list, tuple, set, frozenset)) + else frozenset() + ) + names: Final = frozenset(optional_params) | frozenset({"api_key", "api_base", "extra_headers"}) + request_sources: Final = MappingProxyType( + {name: "request" for name in names if name in body or name in credential_fields} + ) + if litellm.enable_azure_ad_token_refresh is True and "enable_azure_ad_token_refresh" in optional_params: + return MappingProxyType({**request_sources, "enable_azure_ad_token_refresh": "deployment"}) + return request_sources + + +def _marshal_rust_ocr_request( + request: rust_ocr_bridge.LiteLLMOcrRequest, + resolve_secret: Callable[[str], str | None], +) -> rust_ocr_bridge.LiteLLMOcrRequest: + if not isinstance(request.document, dict): + raise TypeError(f"document must be a dict with 'type' and URL/file field, got {type(request.document)}") + document: Final = ( + convert_file_document_to_url_document(request.document) + if request.document.get("type") == "file" + else request.document + ) + provider: Final = _rust_ocr_provider(request) + api_key: Final = request.api_key or resolve_secret("MISTRAL_API_KEY") if provider == "mistral" else request.api_key + optional_params: Final = _rust_bridge_optional_params(request, resolve_secret) + input_sources: Final = _rust_bridge_input_sources(request, optional_params) + logged_optional_params: Final = MappingProxyType( + {name: "****" if name in _RUST_OCR_SECRET_FIELDS else value for name, value in optional_params.items()} + ) + logged_kwargs: Final = MappingProxyType( + { + name: "****" if name in _RUST_OCR_SECRET_FIELDS else value + for name, value in request.kwargs.items() + if name != "proxy_server_request" + } + ) + logging_obj: Final = cast( # cast-ok: bridge kwargs carry the prepared logging object + LiteLLMLoggingObj, request.kwargs["litellm_logging_obj"] + ) + logging_obj.update_from_kwargs( + kwargs=dict(logged_kwargs), # mutable-ok: logging API requires an owned dict + model=request.model, + optional_params=dict(logged_optional_params), # mutable-ok: logging API requires an owned dict + litellm_params={ + "litellm_call_id": request.kwargs.get("litellm_call_id"), + "api_base": request.api_base, + }, # mutable-ok: legacy logging requires a concrete params dict + custom_llm_provider=provider, + ) + logging_obj.pre_call( input="OCR document processing", - api_key=resolved_api_key, - additional_args={ + api_key=api_key, + additional_args={ # mutable-ok: pre_call mutates the additional_args dict "complete_input_dict": { - "model": prepared_request.model, - "document": prepared_request.document, - **rust_optional_params, - }, - "api_base": resolved_complete_url, - "headers": resolved_headers, + "model": request.model, + "document": document, + **logged_optional_params, + }, # mutable-ok: callbacks consume a JSON-serializable request dict + "api_base": request.api_base or "", + "headers": request.extra_headers or {}, # mutable-ok: logging callbacks consume a concrete headers dict }, ) - return _PreparedRustOCRCall( - api_key=resolved_api_key, - api_base=rust_api_base, - headers=cast(dict[str, object], resolved_headers), - optional_params=rust_optional_params, + return rust_ocr_bridge.LiteLLMOcrRequest( + model=request.model, + document=document, + api_key=api_key, + api_base=request.api_base, + timeout=request.timeout if request.timeout is not None else request_timeout, + custom_llm_provider=request.custom_llm_provider, + extra_headers=request.extra_headers, + kwargs=optional_params, + input_sources=input_sources, ) def _map_rust_ocr_error( error: Exception, - prepared_request: _PreparedOCRRequest, + request: rust_ocr_bridge.LiteLLMOcrRequest, exception_types: tuple[type[BaseException], type[BaseException]] | None, ) -> Exception: - if exception_types is None: + if exception_types is None or not isinstance(error, exception_types[1]): return error - _, upstream_error = exception_types - if not isinstance(error, upstream_error): + provider: Final = _rust_ocr_provider(request) + if provider is None: return error - error_args: Final = cast( # cast-ok: BaseException.args is typed with Any in the standard library stubs + provider_config: Final = ProviderConfigManager.get_provider_ocr_config( + model=request.model.removeprefix(f"{provider}/"), provider=litellm.LlmProviders(provider) + ) + if provider_config is None: + return error + error_args: Final = cast( # cast-ok: Python exceptions expose positional args as a tuple tuple[object, ...], error.args ) - status_value: Final = error_args[0] if error_args else 0 - message_value: Final = error_args[1] if len(error_args) > 1 else str(error) - status: Final = status_value if isinstance(status_value, int) else 0 - message: Final = message_value if isinstance(message_value, str) else str(message_value) - error_factory: Final = cast( # cast-ok: the legacy provider interface leaves callable parameters untyped - Callable[..., Exception], prepared_request.provider_config.get_error_class + status: Final = error_args[0] if error_args and isinstance(error_args[0], int) else 500 + message: Final = str(error_args[1]) if len(error_args) > 1 else str(error) + error_factory: Final = cast( # cast-ok: provider configs expose heterogeneous exception factories + Callable[..., Exception], provider_config.get_error_class ) return error_factory( - error_message=message, - status_code=status or 500, - headers={}, # mutable-ok: provider error factories require a concrete header dict - ) + error_message=message, status_code=status or 500, headers={} + ) # mutable-ok: provider error factories require a concrete headers dict def _run_rust_ocr( - prepared_request: _PreparedOCRRequest, + request: rust_ocr_bridge.LiteLLMOcrRequest, resolve_api_key: Callable[[str], str | None], ) -> OCRResponse | None: if rust_ocr_bridge.load_rust_ocr() is None: return None - prepared: Final = _prepare_rust_ocr_call( - prepared_request=prepared_request, - resolve_api_key=resolve_api_key, - ) + marshalled: Final = _marshal_rust_ocr_request(request, resolve_api_key) + input_sources: Final = marshalled.input_sources try: - rust_response: Final = rust_ocr_bridge.ocr( - model=prepared_request.model, - document=prepared_request.document, - api_key=prepared.api_key, - api_base=prepared.api_base, - custom_llm_provider=prepared_request.custom_llm_provider, - extra_headers=prepared.headers, - optional_params=prepared.optional_params, - timeout=prepared_request.effective_timeout, + response: Final = rust_ocr_bridge.ocr( + model=marshalled.model, + document=dict(marshalled.document), # mutable-ok: PyO3 OCR binding requires a concrete dict + api_key=marshalled.api_key, + api_base=marshalled.api_base, + custom_llm_provider=marshalled.custom_llm_provider, + extra_headers=marshalled.extra_headers, + optional_params=dict(marshalled.kwargs), # mutable-ok: PyO3 OCR binding requires a concrete dict + input_sources=input_sources, + timeout=marshalled.timeout, ) except Exception as error: - raise _map_rust_ocr_error(error, prepared_request, native_exception_types()) from error - if rust_response is None: - return None - return OCRResponse.model_validate(rust_response) + raise _map_rust_ocr_error(error, request, native_exception_types()) from error + return OCRResponse.model_validate(response) if response is not None else None async def _run_rust_aocr( - prepared_request: _PreparedOCRRequest, + request: rust_ocr_bridge.LiteLLMOcrRequest, resolve_api_key: Callable[[str], str | None], ) -> OCRResponse | None: if rust_ocr_bridge.load_rust_aocr() is None: return None - prepared: Final = _prepare_rust_ocr_call( - prepared_request=prepared_request, - resolve_api_key=resolve_api_key, - ) + marshalled: Final = _marshal_rust_ocr_request(request, resolve_api_key) + input_sources: Final = marshalled.input_sources try: - rust_response: Final = await rust_ocr_bridge.aocr( - model=prepared_request.model, - document=prepared_request.document, - api_key=prepared.api_key, - api_base=prepared.api_base, - custom_llm_provider=prepared_request.custom_llm_provider, - extra_headers=prepared.headers, - optional_params=prepared.optional_params, - timeout=prepared_request.effective_timeout, + response: Final = await rust_ocr_bridge.aocr( + model=marshalled.model, + document=dict(marshalled.document), # mutable-ok: PyO3 OCR binding requires a concrete dict + api_key=marshalled.api_key, + api_base=marshalled.api_base, + custom_llm_provider=marshalled.custom_llm_provider, + extra_headers=marshalled.extra_headers, + optional_params=dict(marshalled.kwargs), # mutable-ok: PyO3 OCR binding requires a concrete dict + input_sources=input_sources, + timeout=marshalled.timeout, ) except Exception as error: - raise _map_rust_ocr_error(error, prepared_request, native_exception_types()) from error - if rust_response is None: - return None - return OCRResponse.model_validate(rust_response) + raise _map_rust_ocr_error(error, request, native_exception_types()) from error + return OCRResponse.model_validate(response) if response is not None else None @client @@ -444,7 +532,29 @@ async def aocr( "extra_headers": extra_headers, "kwargs": kwargs, } + request: Final = rust_ocr_bridge.LiteLLMOcrRequest( + model=model, + document=document, + api_key=api_key, + api_base=api_base, + timeout=timeout, + custom_llm_provider=custom_llm_provider, + extra_headers=extra_headers, + kwargs=kwargs, + ) try: + if rust_enabled() and _rust_ocr_supported(request): + from litellm.secret_managers.main import get_secret_str + + rust_response: Final = await _run_rust_aocr( + request=request, + resolve_api_key=get_secret_str, + ) + if rust_response is None: + verbose_logger.debug("Async Rust OCR bridge unavailable; falling back to Python path") + else: + return rust_response + prepared: Final = _prepare_ocr_request( model=model, document=document, @@ -459,18 +569,6 @@ async def aocr( custom_llm_provider = prepared.custom_llm_provider completion_kwargs.update({"model": model, "custom_llm_provider": custom_llm_provider}) - if _rust_ocr_supported(prepared) and rust_enabled(): - from litellm.secret_managers.main import get_secret_str - - rust_response: Final = await _run_rust_aocr( - prepared_request=prepared, - resolve_api_key=get_secret_str, - ) - if rust_response is None: - verbose_logger.debug("Async Rust OCR bridge unavailable; falling back to Python path") - else: - return rust_response - response = base_llm_http_handler.ocr( model=prepared.model, document=prepared.document, @@ -494,9 +592,11 @@ async def aocr( return response except Exception as e: + error_provider: Final = custom_llm_provider or _rust_ocr_provider(request) + error_model: Final = model.removeprefix(f"{error_provider}/") if error_provider else model raise litellm.exception_type( - model=model, - custom_llm_provider=custom_llm_provider, + model=error_model, + custom_llm_provider=error_provider, original_exception=e, completion_kwargs=completion_kwargs, extra_kwargs=kwargs, @@ -714,9 +814,31 @@ def ocr( "extra_headers": extra_headers, "kwargs": kwargs, } + request: Final = rust_ocr_bridge.LiteLLMOcrRequest( + model=model, + document=document, + api_key=api_key, + api_base=api_base, + timeout=timeout, + custom_llm_provider=custom_llm_provider, + extra_headers=extra_headers, + kwargs=kwargs, + ) try: _is_async: Final = kwargs.pop("aocr", False) is True completion_kwargs["aocr"] = _is_async + if rust_enabled() and _rust_ocr_supported(request): + from litellm.secret_managers.main import get_secret_str + + rust_response: Final = _run_rust_ocr( + request=request, + resolve_api_key=get_secret_str, + ) + if rust_response is None: + verbose_logger.debug("Rust OCR bridge unavailable; falling back to Python path") + else: + return rust_response + prepared: Final = _prepare_ocr_request( model=model, document=document, @@ -731,18 +853,6 @@ def ocr( custom_llm_provider = prepared.custom_llm_provider completion_kwargs.update({"model": model, "custom_llm_provider": custom_llm_provider}) - if _rust_ocr_supported(prepared) and rust_enabled(): - from litellm.secret_managers.main import get_secret_str - - rust_response: Final = _run_rust_ocr( - prepared_request=prepared, - resolve_api_key=get_secret_str, - ) - if rust_response is None: - verbose_logger.debug("Rust OCR bridge unavailable; falling back to Python path") - else: - return rust_response - response: Final = base_llm_http_handler.ocr( model=prepared.model, document=prepared.document, @@ -760,9 +870,11 @@ def ocr( return response except Exception as e: + error_provider: Final = custom_llm_provider or _rust_ocr_provider(request) + error_model: Final = model.removeprefix(f"{error_provider}/") if error_provider else model raise litellm.exception_type( - model=model, - custom_llm_provider=custom_llm_provider, + model=error_model, + custom_llm_provider=error_provider, original_exception=e, completion_kwargs=completion_kwargs, extra_kwargs=kwargs, diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index c2805d00c2e..8f7b515c22a 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -2016,6 +2016,7 @@ async def add_litellm_data_to_request( "method": request.method, "headers": _logging_safe_headers, "body": None, # filled in post-strip; see below + "credential_fields": tuple(sorted(name for name in _TRANSPORT_ONLY_CREDENTIAL_KEYS if name in data)), "arrival_time": arrival_time, # Track when request arrived at proxy } diff --git a/litellm/rust_bridge/ocr.py b/litellm/rust_bridge/ocr.py index b7fdb5a98ef..db959e76f7c 100644 --- a/litellm/rust_bridge/ocr.py +++ b/litellm/rust_bridge/ocr.py @@ -2,7 +2,8 @@ from __future__ import annotations -from collections.abc import Awaitable +from collections.abc import Awaitable, Mapping +from dataclasses import dataclass from typing import Final, Protocol, cast # noqa: TID251 # native extension exposes dynamically typed callables import httpx @@ -11,6 +12,19 @@ from litellm.rust_bridge.bindings import NativeBinding from litellm.rust_bridge.timeouts import timeout_to_seconds as _timeout_to_seconds +@dataclass(frozen=True, slots=True) +class LiteLLMOcrRequest: + model: str + document: Mapping[str, object] + api_key: str | None + api_base: str | None + timeout: float | httpx.Timeout | None + custom_llm_provider: str | None + extra_headers: dict[str, object] | None + kwargs: Mapping[str, object] + input_sources: Mapping[str, str] | None = None + + class RustOcr(Protocol): def __call__( self, @@ -21,6 +35,7 @@ class RustOcr(Protocol): custom_llm_provider: str | None, extra_headers: dict[str, object] | None, optional_params: dict[str, object], + input_sources: dict[str, str], timeout_seconds: float | None, ) -> dict[str, object]: raise NotImplementedError @@ -36,6 +51,7 @@ class RustAocr(Protocol): custom_llm_provider: str | None, extra_headers: dict[str, object] | None, optional_params: dict[str, object], + input_sources: dict[str, str], timeout_seconds: float | None, ) -> Awaitable[dict[str, object]]: raise NotImplementedError @@ -71,6 +87,7 @@ def ocr( extra_headers: dict[str, object] | None, optional_params: dict[str, object], timeout: float | httpx.Timeout | None, + input_sources: Mapping[str, str] | None = None, ) -> dict[str, object] | None: rust_ocr: Final = load_rust_ocr() if rust_ocr is None: @@ -83,6 +100,7 @@ def ocr( custom_llm_provider=custom_llm_provider, extra_headers=extra_headers, optional_params=optional_params, + input_sources=dict(input_sources or {}), # mutable-ok: native boundary requires a concrete dict timeout_seconds=_timeout_to_seconds(timeout), ) @@ -97,6 +115,7 @@ async def aocr( extra_headers: dict[str, object] | None, optional_params: dict[str, object], timeout: float | httpx.Timeout | None, + input_sources: Mapping[str, str] | None = None, ) -> dict[str, object] | None: rust_aocr: Final = load_rust_aocr() if rust_aocr is None: @@ -109,5 +128,6 @@ async def aocr( custom_llm_provider=custom_llm_provider, extra_headers=extra_headers, optional_params=optional_params, + input_sources=dict(input_sources or {}), # mutable-ok: native boundary requires a concrete dict timeout_seconds=_timeout_to_seconds(timeout), ) diff --git a/tests/test_litellm/ocr/test_ocr_azure_document_intelligence_api_base.py b/tests/test_litellm/ocr/test_ocr_azure_document_intelligence_api_base.py index 0c8b1cc2836..460aff3e8d1 100644 --- a/tests/test_litellm/ocr/test_ocr_azure_document_intelligence_api_base.py +++ b/tests/test_litellm/ocr/test_ocr_azure_document_intelligence_api_base.py @@ -1,20 +1,18 @@ """ -Regression tests for Azure Document Intelligence api_base resolution in OCR. +Regression tests for Azure Document Intelligence api_base ownership in OCR. `azure_ai` exposes two OCR services on one provider; the `doc-intelligence` -sub-route must resolve to `AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT`, not to the -generic `AZURE_AI_API_BASE` fallback that `get_llm_provider` injects. These tests -pin that routing and guard the backwards-compatibility contract that an explicitly -supplied api_base is always honoured. +sub-route must defer environment resolution to Rust, not accept the generic +`AZURE_AI_API_BASE` fallback that `get_llm_provider` injects. An explicitly +supplied api_base is still always honoured. """ from litellm.llms.azure_ai.ocr.common_utils import ( is_azure_document_intelligence_model, ) -from litellm.ocr.main import _prepare_ocr_request, _rust_bridge_api_base +from litellm.ocr.main import _prepare_ocr_request _DOC = {"type": "document_url", "document_url": "https://example.com/doc.pdf"} -_DOC_INTELLIGENCE_ENDPOINT = "https://di.cognitiveservices.azure.com" _AZURE_AI_API_BASE = "https://generic-azure-ai.example.com" @@ -23,13 +21,6 @@ class _FakeLogging: return None -def _resolve_secret(name: str) -> str | None: - return { - "AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT": _DOC_INTELLIGENCE_ENDPOINT, - "AZURE_AI_API_BASE": _AZURE_AI_API_BASE, - }.get(name) - - def _prepare(model: str, api_base: str | None): return _prepare_ocr_request( model=model, @@ -56,15 +47,13 @@ class TestIsAzureDocumentIntelligenceModel: class TestDocIntelligenceApiBaseResolution: def test_generic_azure_ai_base_does_not_hijack_doc_intelligence(self, monkeypatch): - """Without an explicit api_base, the AZURE_AI_API_BASE fallback must not - overwrite the endpoint, so it resolves to the Document Intelligence one.""" + """The generic Azure base must not overwrite Rust-owned DI resolution.""" monkeypatch.setenv("AZURE_AI_API_BASE", _AZURE_AI_API_BASE) monkeypatch.delenv("AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT", raising=False) prepared = _prepare("azure_ai/doc-intelligence/prebuilt-layout", None) assert prepared.api_base is None - assert _rust_bridge_api_base(prepared, _resolve_secret) == _DOC_INTELLIGENCE_ENDPOINT def test_explicit_api_base_is_honoured_for_doc_intelligence(self, monkeypatch): """A caller-supplied api_base must always win, even for doc-intelligence.""" @@ -74,7 +63,6 @@ class TestDocIntelligenceApiBaseResolution: prepared = _prepare("azure_ai/doc-intelligence/prebuilt-layout", custom) assert prepared.api_base == custom - assert _rust_bridge_api_base(prepared, _resolve_secret) == custom def test_generic_azure_ai_base_still_applies_to_mistral_ocr(self, monkeypatch): """Non doc-intelligence azure_ai models keep using AZURE_AI_API_BASE.""" diff --git a/tests/test_litellm/ocr/test_ocr_native_format.py b/tests/test_litellm/ocr/test_ocr_native_format.py index 249fbda713e..5f69708fe91 100644 --- a/tests/test_litellm/ocr/test_ocr_native_format.py +++ b/tests/test_litellm/ocr/test_ocr_native_format.py @@ -4,49 +4,42 @@ providers that don't support a native response must reject it, and the Rust bridge (which only returns the normalized shape) must not serve native requests. """ -import dataclasses -from unittest.mock import MagicMock - import pytest import litellm -from litellm.llms.azure_ai.ocr.cohere_parse_transformation import AzureAICohereParseConfig -from litellm.llms.cohere.ocr.transformation import CohereParseConfig -from litellm.ocr.main import _PreparedOCRRequest, _rust_ocr_supported +from litellm.ocr.main import _rust_ocr_supported +from litellm.rust_bridge.ocr import LiteLLMOcrRequest DOCUMENT = {"type": "document_url", "document_url": "https://example.com/doc.pdf"} -def _prepared(optional_params: dict[str, object]) -> _PreparedOCRRequest: - return _PreparedOCRRequest( - model="doc-intelligence/prebuilt-layout", - document=dict(DOCUMENT), +def _request( + optional_params: dict[str, object], model: str = "azure_ai/doc-intelligence/prebuilt-layout" +) -> LiteLLMOcrRequest: + return LiteLLMOcrRequest( + model=model, + document=DOCUMENT, api_key="fake-key", - api_base="https://example.cognitiveservices.azure.com", - custom_llm_provider="azure_ai", + api_base=None, + custom_llm_provider=None, extra_headers=None, - provider_config=MagicMock(), - optional_params=optional_params, - litellm_params={}, - effective_timeout=60.0, - litellm_logging_obj=MagicMock(), + timeout=60.0, + kwargs=optional_params, ) @pytest.mark.parametrize("optional_params", [{}, {"req_format": "litellm"}]) def test_rust_ocr_serves_default_format(optional_params): - assert _rust_ocr_supported(_prepared(optional_params)) is True + assert _rust_ocr_supported(_request(optional_params)) is True def test_rust_ocr_skipped_for_native_format(): - assert _rust_ocr_supported(_prepared({"req_format": "native"})) is False + assert _rust_ocr_supported(_request({"req_format": "native"})) is False -@pytest.mark.parametrize("provider_config", [CohereParseConfig(), AzureAICohereParseConfig()]) -def test_rust_ocr_skipped_for_configs_without_bridge_support(provider_config): - prepared = dataclasses.replace(_prepared({}), provider_config=provider_config) - - assert _rust_ocr_supported(prepared) is False +@pytest.mark.parametrize("model", ["cohere/cohere-parse", "azure_ai/cohere-parse"]) +def test_rust_ocr_skipped_for_unsupported_models(model): + assert _rust_ocr_supported(_request({}, model)) is False @pytest.mark.asyncio diff --git a/tests/test_litellm/ocr/test_rust_bridge.py b/tests/test_litellm/ocr/test_rust_bridge.py index c34833221cc..3441bd4de34 100644 --- a/tests/test_litellm/ocr/test_rust_bridge.py +++ b/tests/test_litellm/ocr/test_rust_bridge.py @@ -3,7 +3,6 @@ import builtins import importlib import types -from typing import Any import httpx import pytest @@ -59,6 +58,7 @@ class RecordingBridge: custom_llm_provider: str | None, extra_headers: dict[str, object] | None, optional_params: dict[str, object], + input_sources: dict[str, str], timeout_seconds: float | None, ) -> dict[str, object]: self.calls.append( @@ -70,6 +70,7 @@ class RecordingBridge: "custom_llm_provider": custom_llm_provider, "extra_headers": extra_headers, "optional_params": optional_params, + "input_sources": input_sources, "timeout_seconds": timeout_seconds, } ) @@ -91,6 +92,7 @@ class RecordingAsyncBridge: custom_llm_provider: str | None, extra_headers: dict[str, object] | None, optional_params: dict[str, object], + input_sources: dict[str, str], timeout_seconds: float | None, ) -> dict[str, object]: self.calls.append( @@ -102,6 +104,7 @@ class RecordingAsyncBridge: "custom_llm_provider": custom_llm_provider, "extra_headers": extra_headers, "optional_params": optional_params, + "input_sources": input_sources, "timeout_seconds": timeout_seconds, } ) @@ -118,6 +121,7 @@ class RaisingBridge: custom_llm_provider: str | None, extra_headers: dict[str, object] | None, optional_params: dict[str, object], + input_sources: dict[str, str], timeout_seconds: float | None, ) -> dict[str, object]: raise RuntimeError("bridge failed") @@ -133,6 +137,7 @@ class RaisingAsyncBridge: custom_llm_provider: str | None, extra_headers: dict[str, object] | None, optional_params: dict[str, object], + input_sources: dict[str, str], timeout_seconds: float | None, ) -> dict[str, object]: raise RuntimeError("bridge failed") @@ -144,6 +149,9 @@ class RecordingLogging: def __init__(self) -> None: self.pre_call_kwargs: dict[str, object] | None = None + def update_from_kwargs(self, **kwargs: object) -> None: + self.update_kwargs = kwargs + def pre_call( self, *, @@ -158,66 +166,32 @@ class RecordingLogging: } -class FakeOCRConfig: - """A stand-in ``BaseOCRConfig`` that echoes the request it would build.""" - - def __init__(self, api_key_env_var: str = "MISTRAL_API_KEY") -> None: - self.api_key_env_var = api_key_env_var - - def get_api_key_env_var(self) -> str: - return self.api_key_env_var - - def validate_environment( - self, - *, - headers: dict[str, object], - model: str, - api_key: str | None, - api_base: str | None, - litellm_params: dict[str, object], - ) -> dict[str, object]: - return {"Authorization": f"Bearer {api_key}", **headers} - - def get_complete_url( - self, - *, - api_base: str | None, - model: str, - optional_params: dict[str, object], - litellm_params: dict[str, object], - ) -> str: - return f"{api_base or 'https://api.mistral.ai/v1'}/ocr" - - def get_error_class(self, error_message: str, status_code: int, headers: dict[str, str]) -> BaseLLMException: - return BaseLLMException(status_code=status_code, message=error_message, headers=headers) - - -def build_prepared_request( +def build_request( *, logging_obj: RecordingLogging | None = None, - provider_config: FakeOCRConfig | None = None, model: str = "mistral-ocr-latest", document: dict[str, object] = DOCUMENT, api_key: str | None = "sk-test", api_base: str | None = None, - custom_llm_provider: str = "mistral", + custom_llm_provider: str | None = "mistral", extra_headers: dict[str, object] | None = None, optional_params: dict[str, object] | None = None, litellm_params: dict[str, object] | None = None, timeout: float | httpx.Timeout | None = 12.5, -) -> Any: - return ocr_main._PreparedOCRRequest( +) -> rust_bridge.LiteLLMOcrRequest: + return rust_bridge.LiteLLMOcrRequest( model=model, document=document, api_key=api_key, api_base=api_base, custom_llm_provider=custom_llm_provider, extra_headers=extra_headers, - provider_config=provider_config or FakeOCRConfig(), - optional_params=optional_params or {}, - litellm_params=litellm_params or {}, - effective_timeout=timeout, - litellm_logging_obj=logging_obj or RecordingLogging(), + timeout=timeout, + kwargs={ + **(optional_params or {}), + **(litellm_params or {}), + "litellm_logging_obj": logging_obj or RecordingLogging(), + }, ) @@ -425,6 +399,7 @@ def test_bridge_wrapper_forwards_prepared_args_and_wraps_response(): "x-trace-id": "trace-1", }, "optional_params": {"include_image_base64": True, "pages": [0]}, + "input_sources": {}, "timeout_seconds": 12.5, } @@ -456,6 +431,7 @@ async def test_bridge_wrapper_forwards_prepared_async_args_and_wraps_response(): "custom_llm_provider": "vertex_ai", "extra_headers": None, "optional_params": {"vertex_project": "project-1"}, + "input_sources": {}, "timeout_seconds": 42.0, } @@ -467,7 +443,7 @@ def test_run_rust_ocr_prepares_request_and_wraps_response(): rust_bridge._OCR.override(bridge) response = ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( logging_obj=logging_obj, api_base="https://proxy.internal", extra_headers={"x-trace-id": "trace-1"}, @@ -486,10 +462,10 @@ def test_run_rust_ocr_prepares_request_and_wraps_response(): "api_base": "https://proxy.internal", "custom_llm_provider": "mistral", "extra_headers": { - "Authorization": "Bearer sk-test", "x-trace-id": "trace-1", }, "optional_params": {"include_image_base64": True}, + "input_sources": {}, "timeout_seconds": 12.5, } @@ -499,7 +475,7 @@ def test_rust_upstream_error_uses_ocr_provider_error_mapping(): mapped = ocr_main._map_rust_ocr_error( error, - build_prepared_request(), + build_request(), (RuntimeError, RustUpstreamError), ) @@ -514,7 +490,7 @@ def test_run_rust_ocr_resolves_key_via_secret_manager_when_missing(): rust_bridge._OCR.override(bridge) ocr_main._run_rust_ocr( - prepared_request=build_prepared_request(api_key=None, timeout=None), + request=build_request(api_key=None, timeout=None), resolve_api_key=lambda name: "sk-from-vault" if name == "MISTRAL_API_KEY" else None, ) @@ -530,7 +506,7 @@ def test_run_rust_ocr_prefers_explicit_key_over_resolver(): raise AssertionError(f"resolver should not be called for {name}") ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( api_key="sk-explicit", timeout=None, ), @@ -540,7 +516,7 @@ def test_run_rust_ocr_prefers_explicit_key_over_resolver(): assert bridge.calls[0]["api_key"] == "sk-explicit" -def test_run_rust_ocr_uses_provider_api_key_env_var(): +def test_run_rust_ocr_uses_mistral_secret_manager_without_provider_config(): bridge = RecordingBridge() resolver_calls = [] litellm.rust(True) @@ -551,16 +527,15 @@ def test_run_rust_ocr_uses_provider_api_key_env_var(): return "sk-provider-env" ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( - provider_config=FakeOCRConfig(api_key_env_var="PROVIDER_OCR_API_KEY"), - model="provider-ocr-model", + request=build_request( + model="mistral-ocr-latest", api_key=None, timeout=None, ), resolve_api_key=_resolver, ) - assert resolver_calls == ["PROVIDER_OCR_API_KEY"] + assert resolver_calls == ["MISTRAL_API_KEY"] assert bridge.calls[0]["api_key"] == "sk-provider-env" @@ -570,7 +545,7 @@ def test_prepare_rust_ocr_call_forwards_vertex_routing_metadata(): rust_bridge._OCR.override(bridge) ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( custom_llm_provider="vertex_ai", model="mistral-ocr-maas", litellm_params={ @@ -588,6 +563,7 @@ def test_prepare_rust_ocr_call_forwards_vertex_routing_metadata(): "include_image_base64": True, "vertex_project": "project-1", "vertex_location": "us-central1", + "vertex_credentials": "redacted", } @@ -603,7 +579,7 @@ def test_prepare_rust_ocr_call_resolves_vertex_routing_metadata_from_secret_mana }.get(name) ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( custom_llm_provider="vertex_ai", model="mistral-ocr-maas", timeout=None, @@ -615,42 +591,189 @@ def test_prepare_rust_ocr_call_resolves_vertex_routing_metadata_from_secret_mana assert bridge.calls[0]["optional_params"]["vertex_location"] == "us-east5" -def test_prepare_rust_ocr_call_resolves_azure_ai_api_base_from_secret_manager(): +def test_prepare_rust_ocr_call_defers_azure_environment_resolution_to_rust(): bridge = RecordingBridge() litellm.rust(True) rust_bridge._OCR.override(bridge) ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( custom_llm_provider="azure_ai", model="pixtral-12b-2409", + api_key=None, api_base=None, timeout=None, ), - resolve_api_key=lambda name: "https://azure.example.com" if name == "AZURE_AI_API_BASE" else None, + resolve_api_key=lambda name: pytest.fail(f"Python resolved Azure secret {name}"), ) - assert bridge.calls[0]["api_base"] == "https://azure.example.com" + assert bridge.calls[0]["api_base"] is None + assert bridge.calls[0]["api_key"] is None + assert bridge.calls[0]["extra_headers"] is None -def test_prepare_rust_ocr_call_resolves_document_intelligence_endpoint(): +def test_prepare_rust_ocr_call_defers_document_intelligence_environment_to_rust(): bridge = RecordingBridge() litellm.rust(True) rust_bridge._OCR.override(bridge) ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( custom_llm_provider="azure_ai", model="doc-intelligence/prebuilt-layout", api_base=None, timeout=None, ), - resolve_api_key=lambda name: ( - "https://document-intelligence.example.com" if name == "AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT" else None - ), + resolve_api_key=lambda name: pytest.fail(f"Python resolved Azure secret {name}"), ) - assert bridge.calls[0]["api_base"] == "https://document-intelligence.example.com" + assert bridge.calls[0]["api_base"] is None + + +def test_prepare_rust_ocr_call_forwards_raw_azure_auth_inputs(): + bridge = RecordingBridge() + litellm.rust(True) + rust_bridge._OCR.override(bridge) + + ocr_main._run_rust_ocr( + request=build_request( + custom_llm_provider="azure_ai", + model="pixtral-12b-2409", + api_key=None, + api_base="https://azure.example.com", + extra_headers={"x-trace-id": "trace-1"}, + litellm_params={ + "azure_ad_token": "entra-token", + "tenant_id": "tenant", + "client_id": "client", + "client_secret": "secret", + "azure_scope": "scope", + "azure_authority_host": "https://login.example.com", + "azure_credential": "ClientSecretCredential", + "azure_federated_token_file": "/token", + }, + timeout=None, + ), + resolve_api_key=lambda name: pytest.fail(f"Python resolved Azure secret {name}"), + ) + + call = bridge.calls[0] + assert call["api_key"] is None + assert call["api_base"] == "https://azure.example.com" + assert call["extra_headers"] == {"x-trace-id": "trace-1"} + assert call["optional_params"] == { + "azure_ad_token": "entra-token", + "tenant_id": "tenant", + "client_id": "client", + "client_secret": "secret", + "azure_scope": "scope", + "azure_authority_host": "https://login.example.com", + "azure_credential": "ClientSecretCredential", + "azure_federated_token_file": "/token", + } + assert call["input_sources"] == {} + + +def test_prepare_rust_ocr_call_preserves_proxy_input_sources(): + bridge = RecordingBridge() + litellm.rust(True) + rust_bridge._OCR.override(bridge) + request_values = { + "tenant_id": "tenant", + "client_id": "client", + "client_secret": "secret", + "azure_authority_host": "https://login.example.com", + "api_base": "https://azure.example.com", + } + + ocr_main._run_rust_ocr( + request=build_request( + custom_llm_provider="azure_ai", + model="pixtral-12b-2409", + api_key="request-key", + api_base="https://azure.example.com", + litellm_params={ + "tenant_id": "tenant", + "client_id": "client", + "client_secret": "secret", + "azure_authority_host": "https://login.example.com", + "proxy_server_request": {"body": request_values, "credential_fields": ("api_key",)}, + }, + ), + resolve_api_key=lambda _name: None, + ) + + assert bridge.calls[0]["input_sources"] == { + **{name: "request" for name in request_values}, + "api_key": "request", + } + + +def test_rust_ocr_logging_redacts_azure_credentials(): + bridge = RecordingBridge() + logging_obj = RecordingLogging() + litellm.rust(True) + rust_bridge._OCR.override(bridge) + + ocr_main._run_rust_ocr( + request=build_request( + logging_obj=logging_obj, + custom_llm_provider="azure_ai", + model="pixtral-12b-2409", + api_key=None, + litellm_params={"azure_ad_token": "token", "client_secret": "secret"}, + ), + resolve_api_key=lambda _name: None, + ) + + assert logging_obj.update_kwargs["optional_params"] == { + "azure_ad_token": "****", + "client_secret": "****", + } + assert logging_obj.pre_call_kwargs is not None + additional_args = logging_obj.pre_call_kwargs["additional_args"] + assert isinstance(additional_args, dict) + complete_input = additional_args["complete_input_dict"] + assert isinstance(complete_input, dict) + assert complete_input["azure_ad_token"] == "****" + assert complete_input["client_secret"] == "****" + + +def test_rust_eligibility_rejects_python_only_azure_auth_modes(): + for params in ( + {"azure_ad_token_provider": lambda: "token"}, + {"azure_username": "user"}, + {"azure_password": "password"}, + ): + assert not ocr_main._rust_ocr_supported( + build_request( + custom_llm_provider="azure_ai", + model="pixtral-12b-2409", + litellm_params=params, + ) + ) + + +def test_prepare_rust_ocr_call_forwards_global_azure_refresh(monkeypatch: pytest.MonkeyPatch): + bridge = RecordingBridge() + litellm.rust(True) + rust_bridge._OCR.override(bridge) + monkeypatch.setattr(litellm, "enable_azure_ad_token_refresh", True) + + ocr_main._run_rust_ocr( + request=build_request( + custom_llm_provider="azure_ai", + model="pixtral-12b-2409", + api_key=None, + api_base="https://azure.example.com", + litellm_params={"proxy_server_request": {"body": {"enable_azure_ad_token_refresh": True}}}, + timeout=None, + ), + resolve_api_key=lambda _name: None, + ) + + assert bridge.calls[0]["optional_params"] == {"enable_azure_ad_token_refresh": True} + assert bridge.calls[0]["input_sources"] == {"enable_azure_ad_token_refresh": "deployment"} def test_run_rust_ocr_runs_pre_call_logging(): @@ -660,7 +783,7 @@ def test_run_rust_ocr_runs_pre_call_logging(): rust_bridge._OCR.override(bridge) ocr_main._run_rust_ocr( - prepared_request=build_prepared_request( + request=build_request( logging_obj=logging_obj, api_base="https://api.mistral.ai/v1", extra_headers={"x-trace-id": "trace-1"}, @@ -676,9 +799,8 @@ def test_run_rust_ocr_runs_pre_call_logging(): complete_input = additional_args["complete_input_dict"] assert complete_input["document"] == DOCUMENT assert complete_input["include_image_base64"] is True - assert additional_args["api_base"] == "https://api.mistral.ai/v1/ocr" + assert additional_args["api_base"] == "https://api.mistral.ai/v1" assert additional_args["headers"] == { - "Authorization": "Bearer sk-test", "x-trace-id": "trace-1", } @@ -696,12 +818,11 @@ def test_ocr_routes_to_rust_when_enabled(fake_bridge): assert response.pages[0].markdown == "hello world" assert len(fake_bridge.calls) == 1 call = fake_bridge.calls[0] - assert call["model"] == "mistral-ocr-latest" + assert call["model"] == MODEL assert call["document"] == DOCUMENT assert call["api_key"] == "sk-test" - assert call["custom_llm_provider"] == "mistral" + assert call["custom_llm_provider"] is None assert call["extra_headers"] == { - "Authorization": "Bearer sk-test", "x-trace-id": "trace-1", } assert call["optional_params"].get("include_image_base64") is True @@ -717,8 +838,29 @@ def test_ocr_routes_azure_ai_to_rust_when_enabled(fake_bridge): assert isinstance(response, OCRResponse) assert len(fake_bridge.calls) == 1 - assert fake_bridge.calls[0]["model"] == "pixtral-12b-2409" - assert fake_bridge.calls[0]["custom_llm_provider"] == "azure_ai" + assert fake_bridge.calls[0]["model"] == "azure_ai/pixtral-12b-2409" + assert fake_bridge.calls[0]["custom_llm_provider"] is None + assert fake_bridge.calls[0]["extra_headers"] is None + + +def test_ocr_routes_azure_entra_inputs_to_rust_without_python_auth(fake_bridge): + response = litellm.ocr( + model="azure_ai/pixtral-12b-2409", + document=DOCUMENT, + api_base="https://example.services.ai.azure.com", + azure_ad_token="entra-token", + tenant_id="tenant", + client_id="client", + ) + + assert isinstance(response, OCRResponse) + assert fake_bridge.calls[0]["api_key"] is None + assert fake_bridge.calls[0]["extra_headers"] is None + assert fake_bridge.calls[0]["optional_params"] == { + "azure_ad_token": "entra-token", + "tenant_id": "tenant", + "client_id": "client", + } def test_ocr_rust_path_converts_file_document_before_bridge(fake_bridge): @@ -768,12 +910,11 @@ async def test_aocr_routes_to_async_rust_when_enabled(fake_async_bridge): assert response.pages[0].markdown == "hello world" assert len(fake_async_bridge.calls) == 1 call = fake_async_bridge.calls[0] - assert call["model"] == "mistral-ocr-latest" + assert call["model"] == MODEL assert call["document"] == DOCUMENT assert call["api_key"] == "sk-test" - assert call["custom_llm_provider"] == "mistral" + assert call["custom_llm_provider"] is None assert call["extra_headers"] == { - "Authorization": "Bearer sk-test", "x-trace-id": "trace-1", } assert call["optional_params"].get("include_image_base64") is True @@ -864,3 +1005,137 @@ def test_ocr_provider_configs_expose_api_key_env_vars(): assert AzureDocumentIntelligenceOCRConfig().get_api_key_env_var() == "AZURE_DOCUMENT_INTELLIGENCE_API_KEY" assert VertexAIOCRConfig().get_api_key_env_var() == "VERTEX_AI_API_KEY" assert VertexAIDeepSeekOCRConfig().get_api_key_env_var() == "VERTEX_AI_API_KEY" + + +@pytest.mark.parametrize("asynchronous", [False, True]) +@pytest.mark.asyncio +async def test_rust_receives_unmapped_azure_options(asynchronous, fake_bridge, fake_async_bridge): + from typing import Final + + arguments: Final = { + "model": "azure_ai/doc-intelligence/prebuilt-layout", + "document": DOCUMENT, + "api_key": "test-key", + "pages": [0, 2], + "features": ["languages", "style"], + "provider_extension": {"enabled": True}, + } + if asynchronous: + await litellm.aocr(**arguments) + else: + litellm.ocr(**arguments) + call: Final = (fake_async_bridge if asynchronous else fake_bridge).calls[0] + assert call["model"] == arguments["model"] + assert call["custom_llm_provider"] is None + assert call["extra_headers"] is None + assert call["optional_params"] == { + "pages": [0, 2], + "features": ["languages", "style"], + "provider_extension": {"enabled": True}, + } + + +@pytest.mark.parametrize("enabled", [False, True]) +@pytest.mark.asyncio +async def test_python_fallback_maps_original_options_once(enabled, monkeypatch): + from io import BytesIO + from typing import Final + + class PythonHandler: + def __init__(self): + self.calls = [] + + def ocr(self, **kwargs): + self.calls.append(kwargs) + return OCRResponse(pages=[], model=kwargs["model"]) + + handler: Final = PythonHandler() + monkeypatch.setattr(ocr_main, "base_llm_http_handler", handler) + litellm.rust(enabled) + rust_bridge._OCR.override(None) + rust_bridge._AOCR.override(None) + for asynchronous in (False, True): + file: Final = BytesIO(b"test document") + arguments: Final = { + "model": "azure_ai/doc-intelligence/prebuilt-layout", + "document": {"type": "file", "file": file}, + "api_key": "test-key", + "pages": [0, 2], + } + if asynchronous: + await litellm.aocr(**arguments) + else: + litellm.ocr(**arguments) + assert handler.calls[-1]["optional_params"]["pages"] == "1,3" + assert handler.calls[-1]["document"]["document_url"].endswith("dGVzdCBkb2N1bWVudA==") + assert len(handler.calls) == 2 + + +@pytest.mark.parametrize("asynchronous", [False, True]) +@pytest.mark.parametrize("model", ["mistral/mistral-ocr-latest", "azure_ai/doc-intelligence/prebuilt-read"]) +@pytest.mark.asyncio +async def test_native_public_ocr_matches_python(model, asynchronous): + import json + from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + from threading import Thread + from typing import Final + from urllib.parse import parse_qsl, urlsplit + + native: Final = rust_bridge_loader.get_native_bridge() + if native is None: + pytest.skip("requires the compiled Rust extension") + calls: Final = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + body: Final = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + target: Final = urlsplit(self.path) + calls.append( + ( + target.path, + parse_qsl(target.query), + self.headers.get("Authorization"), + self.headers.get("Ocp-Apim-Subscription-Key"), + body, + ) + ) + payload: Final = ( + {"status": "succeeded", "analyzeResult": {"pages": []}} + if "doc-intelligence" in model + else {"pages": [{"index": 0, "markdown": "hello"}]} + ) + encoded: Final = json.dumps(payload).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(encoded))) + self.end_headers() + self.wfile.write(encoded) + + def log_message(self, *_args): + pass + + server: Final = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread: Final = Thread(target=server.serve_forever, daemon=True) + thread.start() + responses: Final = [] + try: + for enabled in (False, True): + litellm.rust(enabled) + arguments: Final = { + "model": model, + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "api_key": "test-key", + "api_base": f"http://127.0.0.1:{server.server_port}", + "pages": [0, 2], + "timeout": 3.0, + } + response: Final = await litellm.aocr(**arguments) if asynchronous else litellm.ocr(**arguments) + responses.append(response.model_dump()) + assert len(calls) == 2 + assert calls[0] == calls[1] + for key in ("model", "pages", "object"): + assert responses[0][key] == responses[1][key] + finally: + server.shutdown() + server.server_close() + thread.join(timeout=3) diff --git a/tests/test_litellm/proxy/test_litellm_pre_call_utils.py b/tests/test_litellm/proxy/test_litellm_pre_call_utils.py index ec9025a5220..37b983d709a 100644 --- a/tests/test_litellm/proxy/test_litellm_pre_call_utils.py +++ b/tests/test_litellm/proxy/test_litellm_pre_call_utils.py @@ -769,6 +769,7 @@ async def test_add_litellm_data_to_request_body_snapshot_excludes_proxy_server_r data = { "model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "hello"}], + "api_key": "request-key", } user_api_key_dict = UserAPIKeyAuth( @@ -796,6 +797,8 @@ async def test_add_litellm_data_to_request_body_snapshot_excludes_proxy_server_r assert "proxy_server_request" not in snapshot_body, ( "proxy_server_request must be excluded from its own body snapshot to prevent the body from self-referencing" ) + assert "api_key" not in snapshot_body + assert updated["proxy_server_request"]["credential_fields"] == ("api_key",) def test_refresh_proxy_server_request_body_snapshot_picks_up_guardrail_masking(): diff --git a/tests/test_litellm/rust_bridge/native_route_wheel_test.py b/tests/test_litellm/rust_bridge/native_route_wheel_test.py index 5b2af3cf02e..3b70043fada 100644 --- a/tests/test_litellm/rust_bridge/native_route_wheel_test.py +++ b/tests/test_litellm/rust_bridge/native_route_wheel_test.py @@ -194,10 +194,10 @@ def azure_ocr_kwargs(api_base: str) -> dict[str, object]: "api_base": api_base, "custom_llm_provider": "azure_ai", "extra_headers": { - "Authorization": "Bearer prepared-azure-token", "x-test-outcome": "success", "x-test-route": "azure_ocr", }, + "optional_params": {"azure_ad_token": "prepared-azure-token"}, } From b544f2244b38f4361c789ead93880000e070a5c4 Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 16:22:55 -0700 Subject: [PATCH 128/157] feat(ocr): add Azure Document Intelligence adapter (#40534) * feat(ocr): add Azure Document Intelligence * fix(ocr): decline missing Document Intelligence credentials * fix(ocr): map Document Intelligence credentials * test(ocr): expose Azure transport to adapter tests * fix(ocr): preserve native responses through Rust bridge * feat(core): add URL query pair completion * fix(ocr): declare Document Intelligence native responses * fix(auth): preserve Azure credential provenance in OCR adapters * refactor(ocr): use shared native response handling * refactor(ocr): preserve Document Intelligence extra params * refactor(ocr): adopt request preparation contract * refactor(ocr): keep native response handling behind bridge * fix(ocr): prevent credential-bearing polling redirects * fix(ocr): update Azure auth imports * fix(ocr): bound Document Intelligence polling rate * fix(ocr): preserve proxy credential provenance * test(ocr): assert native bridge format support --- .../src/audio_transcription/hooks.rs | 3 +- .../crates/ai-gateway/src/ocr/hooks.rs | 3 +- litellm-rust/crates/ai-gateway/src/ocr/mod.rs | 4 +- .../ai-gateway/src/routes/messages/mod.rs | 3 +- litellm-rust/crates/core/src/constants.rs | 7 + litellm-rust/crates/core/src/error.rs | 4 + .../azure/document_intelligence/mod.rs | 214 ++++++++++ .../azure/document_intelligence/polling.rs | 100 +++++ .../{azure_mistral.rs => azure/mistral.rs} | 54 +-- .../crates/core/src/ocr/adapters/azure/mod.rs | 49 +++ .../crates/core/src/ocr/adapters/mod.rs | 5 +- litellm-rust/crates/core/src/ocr/client.rs | 15 + .../ocr/codecs/document_intelligence/mod.rs | 9 + .../codecs/document_intelligence/params.rs | 195 +++++++++ .../document_intelligence/transformation.rs | 111 +++++ .../ocr/codecs/document_intelligence/types.rs | 138 ++++++ .../crates/core/src/ocr/codecs/mod.rs | 1 + litellm-rust/crates/core/src/ocr/error.rs | 23 + litellm-rust/crates/core/src/ocr/mod.rs | 3 + litellm-rust/crates/core/src/ocr/registry.rs | 7 +- litellm-rust/crates/core/src/ocr/types.rs | 2 + litellm-rust/crates/core/src/ocr/wire.rs | 1 + litellm-rust/crates/core/src/url_utils.rs | 23 + .../tests/azure_document_intelligence_ocr.rs | 395 ++++++++++++++++++ .../crates/python-bridge/src/errors.rs | 1 + .../crates/python-bridge/src/routes/ocr.rs | 4 +- litellm/rust_bridge/ocr.py | 303 +++++++++++++- .../ocr/test_ocr_native_format.py | 31 +- tests/test_litellm/ocr/test_rust_bridge.py | 18 + .../rust_bridge/native_route_wheel_test.py | 27 +- 30 files changed, 1689 insertions(+), 64 deletions(-) create mode 100644 litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/mod.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/polling.rs rename litellm-rust/crates/core/src/ocr/adapters/{azure_mistral.rs => azure/mistral.rs} (83%) create mode 100644 litellm-rust/crates/core/src/ocr/adapters/azure/mod.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/document_intelligence/mod.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/document_intelligence/params.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/document_intelligence/transformation.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/document_intelligence/types.rs create mode 100644 litellm-rust/crates/core/tests/azure_document_intelligence_ocr.rs diff --git a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs index 6f4a573e6ff..873c6429ff8 100644 --- a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs @@ -272,7 +272,8 @@ fn core_error_kind(error: &Error) -> &'static str { Error::Auth(_) | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken => "AuthError", + | Error::MissingAzureAiCredentialsOrAdToken + | Error::MissingAzureDocumentIntelligenceCredentials => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", Error::InvalidRequest(_) => "InvalidRequest", Error::InvalidType { .. } => "InvalidType", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs index c7e8344aafc..bca54bebd4a 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs @@ -389,7 +389,8 @@ fn core_error_kind(error: &Error) -> &'static str { Error::Auth(_) | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken => "AuthError", + | Error::MissingAzureAiCredentialsOrAdToken + | Error::MissingAzureDocumentIntelligenceCredentials => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", Error::InvalidRequest(_) => "InvalidRequest", Error::InvalidType { .. } => "InvalidType", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs index 9ed93d779f1..2cdc9f9a714 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs @@ -34,10 +34,10 @@ mod tests { use litellm_core::ocr::wire::is_supported_request; #[test] - fn core_activation_excludes_unmigrated_azure_document_intelligence() { + fn core_activation_includes_azure_document_intelligence() { assert!(is_supported_request("model", Some("mistral"))); assert!(is_supported_request("pixtral-12b", Some("azure_ai"))); - assert!(!is_supported_request( + assert!(is_supported_request( "doc-intelligence/prebuilt-layout", Some("azure_ai") )); diff --git a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs index adfbf2b5910..a64ad1a1376 100644 --- a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs @@ -117,7 +117,8 @@ impl IntoResponse for MessagesRouteError { | Error::MissingField(_) | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken => ( + | Error::MissingAzureAiCredentialsOrAdToken + | Error::MissingAzureDocumentIntelligenceCredentials => ( StatusCode::BAD_GATEWAY, "messages provider request failed".to_string(), ), diff --git a/litellm-rust/crates/core/src/constants.rs b/litellm-rust/crates/core/src/constants.rs index 8a12e186197..73f2f5d284b 100644 --- a/litellm-rust/crates/core/src/constants.rs +++ b/litellm-rust/crates/core/src/constants.rs @@ -51,5 +51,12 @@ pub(crate) const OCR_CONNECT_TIMEOUT_SECS: u64 = 10; pub(crate) const OCR_INLINE_MAX_BYTES: usize = 50 * 1024 * 1024; pub(crate) const OCR_DOWNLOAD_MAX_BYTES: u64 = 50 * 1024 * 1024; pub(crate) const OCR_MAX_FETCH_REDIRECTS: usize = 10; +pub(crate) const OCR_POLL_TIMEOUT_SECS: u64 = 120; +pub(crate) const OCR_POLL_RETRY_SECS: u64 = 2; +pub(crate) const AZURE_DI_API_VERSION: &str = "2024-11-30"; +pub(crate) const AZURE_DI_SUBSCRIPTION_HEADER: &str = "Ocp-Apim-Subscription-Key"; +pub(crate) const AZURE_DI_DEFAULT_DPI: i64 = 96; +pub(crate) const AZURE_DI_DEFAULT_WIDTH: f64 = 8.5; +pub(crate) const AZURE_DI_DEFAULT_HEIGHT: f64 = 11.0; pub(crate) const AZURE_AI_OCR_PATH: &str = "/providers/mistral/azure/ocr"; pub(crate) const MISTRAL_OCR_API_BASE: &str = "https://api.mistral.ai/v1"; diff --git a/litellm-rust/crates/core/src/error.rs b/litellm-rust/crates/core/src/error.rs index b171c0c4274..bbeb23daa13 100644 --- a/litellm-rust/crates/core/src/error.rs +++ b/litellm-rust/crates/core/src/error.rs @@ -27,6 +27,10 @@ pub enum Error { MissingAzureAiCredentials, #[error("Missing Azure AI credentials - set AZURE_AI_API_KEY or provide azure_ad_token")] MissingAzureAiCredentialsOrAdToken, + #[error( + "invalid authentication configuration: Missing Azure Document Intelligence credentials - set AZURE_DOCUMENT_INTELLIGENCE_API_KEY or configure Entra ID" + )] + MissingAzureDocumentIntelligenceCredentials, #[error("upstream request failed with status {status}: {body}")] Http { status: u16, body: String }, #[error("upstream network error: {0}")] diff --git a/litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/mod.rs new file mode 100644 index 00000000000..71ca69ddc58 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/mod.rs @@ -0,0 +1,214 @@ +use super::super::OcrAdapter; +use crate::Error; +use crate::auth::{InputSource, Sourced}; +use crate::constants::{AZURE_DI_API_VERSION, AZURE_DI_SUBSCRIPTION_HEADER}; +use crate::ocr::OcrClient; +use crate::ocr::codecs::document_intelligence::{ + self, AzureDocumentIntelligenceOperation, DocumentIntelligenceParams, +}; +use crate::ocr::error::{OcrError, OcrRequestError, OcrResponseError}; +use crate::ocr::prepare::{credential_env, transform_request_body}; +use crate::ocr::registry::OcrProvider; +use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrConnection, OcrResponseFormat}; +use crate::ocr::wire::DecodedOcrResponse; +use crate::providers::azure_ai::auth::AzureAuthInputs; +use crate::url_utils::ApiUrl; + +mod polling; + +const AZURE_DI_API_KEY_ENV: &str = "AZURE_DOCUMENT_INTELLIGENCE_API_KEY"; +const AZURE_DI_ENDPOINT_ENV: &str = "AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"; + +#[derive(Clone, Debug)] +pub(crate) struct AzureDocumentIntelligenceAdapter; + +impl OcrAdapter for AzureDocumentIntelligenceAdapter { + type ProviderResponse = AzureDocumentIntelligenceOperation; + const PROVIDER: OcrProvider = OcrProvider::AzureAi; + + async fn prepare_request( + &self, + request: &LiteLLMOcrRequest, + client: &OcrClient, + ) -> Result { + let params = map_ocr_params(request)?; + let config = AzureAuthInputs::from_sourced_optional_params( + &request.optional_params, + &request.input_sources, + ) + .map_err(Error::from)?; + let headers = validate_environment(&request.connection, &config, &credential_env).await?; + let endpoint = nonblank(request.connection.api_base.clone()) + .or_else(|| nonblank(credential_env(AZURE_DI_ENDPOINT_ENV))) + .ok_or_else(|| Error::Auth("Missing Azure Document Intelligence API Base - Set AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT or pass api_base".into()))?; + let url = get_complete_url(&endpoint, &request.model, ¶ms)?; + let body = document_intelligence::transform_ocr_request(request.document.clone())?; + transform_request_body(client, request, &url, &headers, body, |_| Ok(())).await + } + + fn transform_ocr_response( + &self, + request: &LiteLLMOcrRequest, + response: Self::ProviderResponse, + ) -> Result { + document_intelligence::transform_ocr_response(&request.model, response) + } + + async fn read_response( + &self, + client: &OcrClient, + response: reqwest::Response, + url: &str, + headers: &[(String, String)], + request: &LiteLLMOcrRequest, + ) -> Result, OcrError> { + polling::read_operation_response( + client.polling_http(), + response, + url, + headers, + &request.connection, + request.response_format()? == OcrResponseFormat::Native, + ) + .await + } +} + +#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] +fn map_ocr_params( + request: &LiteLLMOcrRequest, +) -> Result { + let params = document_intelligence::decode_input_params( + request.optional_params.clone(), + "optional_params", + )?; + let crate::ocr::prepare::ParsedProviderParams { + known: params, + extra_params: _extra_params, + } = params; + document_intelligence::map_ocr_params(params) +} + +fn get_complete_url( + endpoint: &str, + model: &str, + params: &DocumentIntelligenceParams, +) -> Result { + let model = format!("{}:analyze", model_id(model)?); + ApiUrl::parse(endpoint) + .and_then(|url| url.complete_path(&["documentintelligence", "documentModels", &model])) + .map(|url| { + url.append_query_pairs( + [("api-version", AZURE_DI_API_VERSION)] + .into_iter() + .chain(params.pages.iter().map(|pages| ("pages", pages.as_str()))) + .chain( + params + .features + .iter() + .map(|features| ("features", features.as_str())), + ), + ) + .into_string() + }) + .map_err(|_| OcrRequestError::RequestField { + path: "api_base".into(), + }) + .map_err(OcrError::from) +} + +async fn validate_environment( + connection: &OcrConnection, + config: &AzureAuthInputs, + env_lookup: &(dyn Fn(&str) -> Option + Sync), +) -> Result, OcrError> { + if crate::http_utils::has_header(&connection.extra_headers, "authorization") + || crate::http_utils::has_header(&connection.extra_headers, AZURE_DI_SUBSCRIPTION_HEADER) + { + super::validate_destination(connection, connection.extra_headers_source)?; + return Ok(connection.extra_headers.clone()); + } + let key = nonblank(connection.api_key.clone()) + .map(|value| Sourced::new(value, connection.api_key_source)) + .or_else(|| { + nonblank(env_lookup(AZURE_DI_API_KEY_ENV)) + .map(|value| Sourced::new(value, InputSource::Environment)) + }); + if let Some(key) = key { + super::validate_destination(connection, key.source())?; + return Ok( + std::iter::once((AZURE_DI_SUBSCRIPTION_HEADER.into(), key.into_value())) + .chain(connection.extra_headers.clone()) + .collect(), + ); + } + let token = super::resolve_entra(config, env_lookup) + .await? + .ok_or(Error::MissingAzureDocumentIntelligenceCredentials)?; + super::validate_destination(connection, token.source())?; + Ok( + std::iter::once(("Authorization".into(), format!("Bearer {}", token.value()))) + .chain(connection.extra_headers.clone()) + .collect(), + ) +} + +fn model_id(model: &str) -> Result<&str, OcrRequestError> { + let model = model.rsplit('/').next().unwrap_or(model); + if matches!(model, "." | "..") { + return Err(OcrRequestError::DotModel); + } + Ok(model) +} + +fn nonblank(value: Option) -> Option { + value + .map(|value| value.trim().to_string()) + .filter(|value| !value.is_empty()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn request_endpoint_cannot_receive_environment_key() { + let connection = OcrConnection { + api_base: Some("https://request.example".into()), + api_base_source: InputSource::Request, + ..Default::default() + }; + + let error = validate_environment(&connection, &Default::default(), &|name| { + (name == AZURE_DI_API_KEY_ENV).then(|| "environment-key".into()) + }) + .await + .unwrap_err(); + + assert!( + error + .to_string() + .contains("request-controlled Azure endpoint") + ); + } + + #[tokio::test] + async fn request_endpoint_accepts_request_owned_key() { + let connection = OcrConnection { + api_key: Some("request-key".into()), + api_key_source: InputSource::Request, + api_base: Some("https://request.example".into()), + api_base_source: InputSource::Request, + ..Default::default() + }; + + let headers = validate_environment(&connection, &Default::default(), &|_| None) + .await + .unwrap(); + + assert_eq!( + headers[0], + (AZURE_DI_SUBSCRIPTION_HEADER.into(), "request-key".into()) + ); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/polling.rs b/litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/polling.rs new file mode 100644 index 00000000000..1bddea0da4f --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/azure/document_intelligence/polling.rs @@ -0,0 +1,100 @@ +use std::time::Duration; + +use reqwest::Url; +use tokio::time::Instant; + +use crate::constants::{AZURE_DI_SUBSCRIPTION_HEADER, OCR_POLL_RETRY_SECS}; +use crate::ocr::client::read_json_response; +use crate::ocr::codecs::document_intelligence::{ + AzureDocumentIntelligenceOperation, OperationStatus, +}; +use crate::ocr::error::{OcrError, OcrPollingError, OcrResponseError}; +use crate::ocr::types::OcrConnection; +use crate::ocr::wire::DecodedOcrResponse; + +pub(super) async fn read_operation_response( + http_client: &reqwest::Client, + response: reqwest::Response, + original_url: &str, + headers: &[(String, String)], + connection: &OcrConnection, + native: bool, +) -> Result, OcrError> { + if response.status() != reqwest::StatusCode::ACCEPTED { + return read_json_response(response, native).await; + } + let location = response + .headers() + .get("operation-location") + .and_then(|value| value.to_str().ok()) + .ok_or(OcrPollingError::PollLocation)?; + let original = Url::parse(original_url).map_err(|_| OcrPollingError::PollOrigin)?; + let operation = Url::parse(location).map_err(|_| OcrPollingError::PollOrigin)?; + if original.origin() != operation.origin() + || !operation.username().is_empty() + || operation.password().is_some() + { + return Err(OcrPollingError::PollOrigin.into()); + } + poll_operation(http_client, operation, headers, connection, native).await +} + +async fn poll_operation( + http_client: &reqwest::Client, + url: Url, + headers: &[(String, String)], + connection: &OcrConnection, + native: bool, +) -> Result, OcrError> { + let deadline = Instant::now() + .checked_add(connection.poll_timeout) + .ok_or(OcrPollingError::PollTimeout)?; + loop { + let remaining = deadline + .checked_duration_since(Instant::now()) + .filter(|remaining| !remaining.is_zero()) + .ok_or(OcrPollingError::PollTimeout)?; + let builder = http_client + .get(url.clone()) + .timeout(remaining.min(connection.timeout)); + let builder = crate::http_utils::with_headers( + builder, + headers, + crate::http_utils::HeaderPolicy::Only(&[AZURE_DI_SUBSCRIPTION_HEADER, "authorization"]), + ); + let response = tokio::time::timeout_at(deadline, crate::http_utils::http_request(builder)) + .await + .map_err(|_| OcrPollingError::PollTimeout)? + .map_err(crate::error::TransportError::from)?; + let retry = response + .headers() + .get(reqwest::header::RETRY_AFTER) + .and_then(|value| value.to_str().ok()) + .and_then(|value| value.parse::().ok()) + .unwrap_or(OCR_POLL_RETRY_SECS) + .max(1); + let decoded = tokio::time::timeout_at( + deadline, + read_json_response::(response, native), + ) + .await + .map_err(|_| OcrPollingError::PollTimeout)??; + match &decoded.data.status { + Some(OperationStatus::Succeeded) => return Ok(decoded), + Some(OperationStatus::Running | OperationStatus::NotStarted) => { + tokio::time::timeout_at(deadline, tokio::time::sleep(Duration::from_secs(retry))) + .await + .map_err(|_| OcrPollingError::PollTimeout)?; + } + status => { + return Err(OcrResponseError::OperationStatus( + status + .as_ref() + .map(ToString::to_string) + .unwrap_or_else(|| "None".into()), + ) + .into()); + } + } + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs b/litellm-rust/crates/core/src/ocr/adapters/azure/mistral.rs similarity index 83% rename from litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs rename to litellm-rust/crates/core/src/ocr/adapters/azure/mistral.rs index 0e52bc61249..3107494d39e 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/azure_mistral.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/azure/mistral.rs @@ -1,8 +1,5 @@ -use std::sync::OnceLock; - -use super::OcrAdapter; +use super::super::OcrAdapter; use crate::Error; -use crate::auth::error::AuthConfigurationError; use crate::auth::{InputSource, Sourced}; use crate::constants::AZURE_AI_OCR_PATH; use crate::ocr::OcrClient; @@ -14,7 +11,7 @@ use crate::ocr::prepare::{ }; use crate::ocr::registry::OcrProvider; use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrConnection}; -use crate::providers::azure_ai::auth::{AzureAuthInputs, AzureAuthService}; +use crate::providers::azure_ai::auth::AzureAuthInputs; use crate::url_utils::ApiUrl; const AZURE_AI_API_KEY_ENV: &str = "AZURE_AI_API_KEY"; @@ -92,7 +89,7 @@ async fn validate_environment( env_lookup: &(dyn Fn(&str) -> Option + Sync), ) -> Result, OcrError> { if crate::http_utils::has_header(&connection.extra_headers, "authorization") { - validate_destination(connection, connection.extra_headers_source)?; + super::validate_destination(connection, connection.extra_headers_source)?; return Ok(connection.extra_headers.clone()); } let key = nonblank(connection.api_key.clone()) @@ -102,41 +99,16 @@ async fn validate_environment( .map(|value| Sourced::new(value, InputSource::Environment)) }); if let Some(key) = key { - validate_destination(connection, key.source())?; + super::validate_destination(connection, key.source())?; return Ok(bearer_headers(connection, key.value())); } - static SERVICE: OnceLock = OnceLock::new(); - let key = SERVICE - .get_or_init(AzureAuthService::default) - .get_azure_ad_token(config, env_lookup) - .await - .map_err(Error::from)? - .map(|credential| { - let source = credential.source(); - let value = credential.value().secret().expose().to_string(); - Sourced::new(value, source) - }) + let key = super::resolve_entra(config, env_lookup) + .await? .ok_or(Error::MissingAzureAiCredentials)?; - validate_destination(connection, key.source())?; + super::validate_destination(connection, key.source())?; Ok(bearer_headers(connection, key.value())) } -fn validate_destination( - connection: &OcrConnection, - credential_source: InputSource, -) -> Result<(), OcrError> { - if connection.api_base.is_some() - && connection.api_base_source == InputSource::Request - && credential_source != InputSource::Request - { - return Err(Error::from(crate::AuthError::Configuration( - AuthConfigurationError::RequestAzureCredentialDestination, - )) - .into()); - } - Ok(()) -} - fn bearer_headers(connection: &OcrConnection, key: &str) -> Vec<(String, String)> { std::iter::once(("Authorization".into(), format!("Bearer {key}"))) .chain(connection.extra_headers.clone()) @@ -177,9 +149,9 @@ mod tests { ..Default::default() }; assert_eq!( - validate_environment(&connection, &Default::default(), &|_| Some( - "environment-key".into() - )) + validate_environment(&connection, &Default::default(), &|_| { + Some("environment-key".into()) + }) .await .unwrap(), connection.extra_headers @@ -193,9 +165,9 @@ mod tests { ..Default::default() }; assert_eq!( - validate_environment(&connection, &Default::default(), &|_| Some( - "environment-key".into() - )) + validate_environment(&connection, &Default::default(), &|_| { + Some("environment-key".into()) + }) .await .unwrap()[0], ("Authorization".into(), "Bearer request-key".into()) diff --git a/litellm-rust/crates/core/src/ocr/adapters/azure/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/azure/mod.rs new file mode 100644 index 00000000000..9c02a7471c9 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/azure/mod.rs @@ -0,0 +1,49 @@ +mod document_intelligence; +mod mistral; + +use std::sync::OnceLock; + +use crate::Error; +use crate::auth::error::AuthConfigurationError; +use crate::auth::{InputSource, Sourced}; +use crate::ocr::error::OcrError; +use crate::ocr::types::OcrConnection; +use crate::providers::azure_ai::auth::{AzureAuthInputs, AzureAuthService}; + +pub(crate) use document_intelligence::AzureDocumentIntelligenceAdapter; +pub(crate) use mistral::AzureMistralAdapter; + +async fn resolve_entra( + config: &AzureAuthInputs, + env_lookup: &(dyn Fn(&str) -> Option + Sync), +) -> Result>, Error> { + static SERVICE: OnceLock = OnceLock::new(); + SERVICE + .get_or_init(AzureAuthService::default) + .get_azure_ad_token(config, env_lookup) + .await + .map(|credential| { + credential.map(|credential| { + let source = credential.source(); + let value = credential.value().secret().expose().to_string(); + Sourced::new(value, source) + }) + }) + .map_err(Error::from) +} + +fn validate_destination( + connection: &OcrConnection, + credential_source: InputSource, +) -> Result<(), OcrError> { + if connection.api_base.is_some() + && connection.api_base_source == InputSource::Request + && credential_source != InputSource::Request + { + return Err(Error::from(crate::AuthError::Configuration( + AuthConfigurationError::RequestAzureCredentialDestination, + )) + .into()); + } + Ok(()) +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/mod.rs index bbd6feb6c7b..530b28aeadd 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/mod.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/mod.rs @@ -8,10 +8,10 @@ use super::registry::OcrProvider; use super::types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrResponseFormat}; use super::wire::DecodedOcrResponse; -mod azure_mistral; +mod azure; mod mistral; -pub(crate) use azure_mistral::AzureMistralAdapter; +pub(crate) use azure::{AzureDocumentIntelligenceAdapter, AzureMistralAdapter}; pub(crate) use mistral::MistralAdapter; /// Converts a complete LiteLLM OCR call to provider HTTP and normalizes its response. @@ -65,6 +65,7 @@ macro_rules! for_each_ocr_adapter { $callback! { Mistral, $crate::ocr::adapters::MistralAdapter, $crate::ocr::adapters::MistralAdapter, Mistral; AzureMistral, $crate::ocr::adapters::AzureMistralAdapter, $crate::ocr::adapters::AzureMistralAdapter, AzureAi; + AzureDocumentIntelligence, $crate::ocr::adapters::AzureDocumentIntelligenceAdapter, $crate::ocr::adapters::AzureDocumentIntelligenceAdapter, AzureAi; } }; } diff --git a/litellm-rust/crates/core/src/ocr/client.rs b/litellm-rust/crates/core/src/ocr/client.rs index 699e44412ac..61b7ae8d995 100644 --- a/litellm-rust/crates/core/src/ocr/client.rs +++ b/litellm-rust/crates/core/src/ocr/client.rs @@ -15,6 +15,7 @@ use crate::media::MediaFetcher; #[derive(Clone)] pub struct OcrClient { provider_http: reqwest::Client, + polling_http: reqwest::Client, document_fetcher: MediaFetcher, } @@ -23,6 +24,7 @@ impl OcrClient { let document_fetcher = MediaFetcher::new().map_err(TransportError::from)?; Ok(Self { provider_http, + polling_http: no_redirect_http()?, document_fetcher, }) } @@ -41,6 +43,10 @@ impl OcrClient { &self.provider_http } + pub(crate) fn polling_http(&self) -> &reqwest::Client { + &self.polling_http + } + pub(crate) fn document_fetcher(&self) -> &MediaFetcher { &self.document_fetcher } @@ -49,11 +55,20 @@ impl OcrClient { pub(crate) fn for_test(provider_http: reqwest::Client, document_http: reqwest::Client) -> Self { Self { provider_http, + polling_http: no_redirect_http().expect("test polling client builds"), document_fetcher: MediaFetcher::for_test(document_http), } } } +fn no_redirect_http() -> Result { + reqwest::Client::builder() + .connect_timeout(Duration::from_secs(OCR_CONNECT_TIMEOUT_SECS)) + .redirect(reqwest::redirect::Policy::none()) + .build() + .map_err(TransportError::from) +} + pub async fn ocr(request: LiteLLMOcrRequest) -> Result { static CLIENT: OnceLock> = OnceLock::new(); let client = CLIENT diff --git a/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/mod.rs b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/mod.rs new file mode 100644 index 00000000000..8031f2124a3 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/mod.rs @@ -0,0 +1,9 @@ +mod params; +mod transformation; +mod types; + +pub(crate) use params::{decode_input_params, map_ocr_params}; +pub(crate) use transformation::{transform_ocr_request, transform_ocr_response}; +pub(crate) use types::{ + AzureDocumentIntelligenceOperation, DocumentIntelligenceParams, OperationStatus, +}; diff --git a/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/params.rs b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/params.rs new file mode 100644 index 00000000000..85d1dafa542 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/params.rs @@ -0,0 +1,195 @@ +use std::collections::BTreeSet; + +use serde_json::{Map, Value}; + +use super::types::{ + DocumentIntelligenceInputParams, DocumentIntelligenceParams, FeaturesInput, PagesInput, +}; +use crate::ocr::error::OcrRequestError; +use crate::ocr::prepare::ParsedProviderParams; + +pub(crate) fn decode_input_params( + params: Map, + prefix: &str, +) -> Result, OcrRequestError> { + if let Some(Value::Array(pages)) = params.get("pages") { + if pages.iter().any(Value::is_boolean) { + return Err(OcrRequestError::Pages("boolean page index".into())); + } + if pages + .iter() + .any(|page| page.is_number() && page.as_i64().is_none()) + { + return Err(OcrRequestError::Pages("page index is out of range".into())); + } + if !pages.iter().all(Value::is_i64) && !pages.iter().all(Value::is_string) { + return Err(OcrRequestError::Pages("mixed page element types".into())); + } + } + crate::ocr::wire::decode_request_value(Value::Object(params), prefix) +} + +pub(crate) fn map_ocr_params( + params: DocumentIntelligenceInputParams, +) -> Result { + Ok(DocumentIntelligenceParams { + pages: params.pages.map(normalize_pages).transpose()?.flatten(), + features: params + .features + .map(normalize_features) + .transpose()? + .flatten(), + }) +} + +fn normalize_pages(pages: PagesInput) -> Result, OcrRequestError> { + let normalized = match pages { + PagesInput::ZeroBasedIndices(indices) => { + if indices.is_empty() { + return Ok(None); + } + indices + .into_iter() + .map(|page| { + if page < 0 { + return Err(OcrRequestError::Pages("negative page index".into())); + } + page.checked_add(1) + .ok_or_else(|| OcrRequestError::Pages("page index is out of range".into())) + }) + .collect::, _>>()? + .into_iter() + .map(|page| page.to_string()) + .collect::>() + .join(",") + } + PagesInput::NativeTokens(tokens) => { + if tokens.is_empty() { + return Ok(None); + } + tokens + .iter() + .map(|token| token.trim()) + .collect::>() + .join(",") + } + PagesInput::NativeRange(range) => range + .split(',') + .map(str::trim) + .collect::>() + .join(","), + }; + if !normalized.split(',').all(valid_page_token) { + return Err(OcrRequestError::Pages("invalid native page range".into())); + } + Ok(Some(normalized)) +} + +fn valid_page_token(token: &str) -> bool { + let mut parts = token.split('-'); + let start = parts.next().unwrap_or_default(); + if start.is_empty() || !start.chars().all(|character| character.is_ascii_digit()) { + return false; + } + match parts.next() { + None => true, + Some(end) => { + !end.is_empty() + && end.chars().all(|character| character.is_ascii_digit()) + && parts.next().is_none() + } + } +} + +fn normalize_features(features: FeaturesInput) -> Result, OcrRequestError> { + let tokens = match features { + FeaturesInput::Names(names) => names, + FeaturesInput::CommaSeparated(names) => names.split(',').map(str::to_string).collect(), + }; + if tokens.is_empty() { + return Ok(None); + } + let normalized = tokens.iter().map(|token| token.trim()).collect::>(); + if !normalized.iter().all(|token| { + let Some((first, rest)) = token.as_bytes().split_first() else { + return false; + }; + first.is_ascii_alphabetic() && rest.iter().all(u8::is_ascii_alphanumeric) + }) { + return Err(OcrRequestError::Features); + } + Ok(Some(normalized.join(","))) +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + use serde_json::{Value, json}; + + use super::*; + + fn map(value: Value) -> Result { + let fields = value.as_object().unwrap().clone(); + map_ocr_params(decode_input_params(fields, "optional_params")?.known) + } + + #[test] + fn input_params_retain_unknown_fields() { + let parsed = decode_input_params( + json!({ + "pages": [0], + "future_ocr_option": true, + "extra_body": {"provider_option": "value"} + }) + .as_object() + .unwrap() + .clone(), + "optional_params", + ) + .unwrap(); + + assert_eq!( + parsed.known.pages, + Some(PagesInput::ZeroBasedIndices(vec![0])) + ); + assert_eq!(parsed.extra_params["future_ocr_option"], true); + assert_eq!( + parsed.extra_params["extra_body"], + json!({"provider_option": "value"}) + ); + assert_eq!( + serde_json::to_value(map_ocr_params(parsed.known).unwrap()).unwrap(), + json!({"pages": "1", "features": null}) + ); + } + + #[rstest] + #[case(json!(["keyValuePairs"]), "keyValuePairs")] + #[case(json!(["keyValuePairs", "languages"]), "keyValuePairs,languages")] + #[case(json!("keyValuePairs"), "keyValuePairs")] + #[case(json!("keyValuePairs,languages"), "keyValuePairs,languages")] + #[case(json!("keyValuePairs, languages"), "keyValuePairs,languages")] + fn feature_mapping_matches_python(#[case] input: Value, #[case] expected: &str) { + assert_eq!( + map(json!({"features": input})).unwrap().features.as_deref(), + Some(expected) + ); + } + + #[rstest] + #[case(json!("keyValuePairs&pages=9"))] + #[case(json!("key value pairs"))] + #[case(json!(""))] + #[case(json!([1, 2]))] + #[case(json!([["keyValuePairs"]]))] + #[case(json!({"feature":"keyValuePairs"}))] + #[case(json!(5))] + fn invalid_feature_mapping_matches_python(#[case] input: Value) { + assert!(map(json!({"features": input})).is_err()); + } + + #[test] + fn empty_feature_list_is_omitted() { + assert_eq!(map(json!({"features": []})).unwrap().features, None); + } +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/transformation.rs b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/transformation.rs new file mode 100644 index 00000000000..2b848fcfb7a --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/transformation.rs @@ -0,0 +1,111 @@ +use base64::{Engine, engine::general_purpose::STANDARD}; +use serde_json::{Map, Value, json}; + +use super::types::*; +use crate::constants::{AZURE_DI_DEFAULT_DPI, AZURE_DI_DEFAULT_HEIGHT, AZURE_DI_DEFAULT_WIDTH}; +use crate::ocr::document::InlineDocument; +use crate::ocr::error::{OcrRequestError, OcrResponseError}; +use crate::ocr::types::{LiteLLMOcrResponse, OcrDocument}; + +#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] +pub(crate) fn transform_ocr_request( + document: OcrDocument, +) -> Result { + let source = document.source(); + if source.is_empty() { + return Err(OcrRequestError::MissingField("document URL")); + } + Ok(if let Some(document) = InlineDocument::parse(source)? { + DocumentIntelligenceRequest::Base64Source( + STANDARD.encode(document.decode(crate::constants::OCR_INLINE_MAX_BYTES)?), + ) + } else { + DocumentIntelligenceRequest::UrlSource(source.to_string()) + }) +} + +pub(crate) fn transform_ocr_response( + model: &str, + response: AzureDocumentIntelligenceOperation, +) -> Result { + if response.status != Some(OperationStatus::Succeeded) { + return Err(OcrResponseError::OperationStatus( + response + .status + .map(|status| status.to_string()) + .unwrap_or_else(|| "None".into()), + )); + } + let result = response.analyze_result.unwrap_or_default(); + let pages = result + .pages + .into_iter() + .map(normalize_page) + .collect::, _>>()?; + let pages_processed = pages.len(); + let mut extra_fields = Map::new(); + extra_fields.insert("content".into(), option_value(result.content)); + extra_fields.insert("tables".into(), option_value(result.tables)); + extra_fields.insert( + "key_value_pairs".into(), + option_value(result.key_value_pairs), + ); + Ok(LiteLLMOcrResponse { + pages, + model: model.into(), + document_annotation: None, + usage_info: Some(json!({"pages_processed":pages_processed})), + object: "ocr".into(), + extra_fields, + provider_native_response: None, + }) +} + +fn normalize_page(page: AzureDocumentIntelligencePage) -> Result { + let index = page + .page_number + .unwrap_or(1) + .checked_sub(1) + .ok_or(OcrResponseError::NumericRange("page.pageNumber"))?; + let scale = if page.unit.as_deref().unwrap_or("inch") == "inch" { + AZURE_DI_DEFAULT_DPI as f64 + } else { + 1.0 + }; + let width = pixel_dimension( + page.width.unwrap_or(AZURE_DI_DEFAULT_WIDTH), + scale, + "page.width", + )?; + let height = pixel_dimension( + page.height.unwrap_or(AZURE_DI_DEFAULT_HEIGHT), + scale, + "page.height", + )?; + let markdown = page + .lines + .iter() + .map(|line| line.content.as_deref().unwrap_or_default()) + .collect::>() + .join("\n"); + Ok(json!({ + "index":index, + "markdown":markdown, + "images":null, + "dimensions":{"width":width,"height":height,"dpi":AZURE_DI_DEFAULT_DPI} + })) +} + +fn pixel_dimension(value: f64, scale: f64, field: &'static str) -> Result { + let value = value * scale; + if !value.is_finite() || value < i64::MIN as f64 || value > i64::MAX as f64 { + return Err(OcrResponseError::NumericRange(field)); + } + Ok(value.trunc() as i64) +} + +fn option_value(value: Option) -> Value { + value + .and_then(|value| serde_json::to_value(value).ok()) + .unwrap_or(Value::Null) +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/types.rs b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/types.rs new file mode 100644 index 00000000000..793f4547e99 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/document_intelligence/types.rs @@ -0,0 +1,138 @@ +use serde::{Deserialize, Deserializer, Serialize}; +use serde_json::{Map, Value}; + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +#[serde(untagged)] +pub(crate) enum PagesInput { + ZeroBasedIndices(Vec), + NativeTokens(Vec), + NativeRange(String), +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +#[serde(untagged)] +pub(crate) enum FeaturesInput { + Names(Vec), + CommaSeparated(String), +} + +#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] +pub(crate) struct DocumentIntelligenceInputParams { + pub pages: Option, + pub features: Option, +} + +#[derive(Clone, Debug, PartialEq, Serialize)] +pub(crate) struct DocumentIntelligenceParams { + pub pages: Option, + pub features: Option, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) enum DocumentIntelligenceRequest { + #[serde(rename = "urlSource")] + UrlSource(String), + #[serde(rename = "base64Source")] + Base64Source(String), +} + +#[derive(Clone, Debug, PartialEq)] +pub(crate) enum OperationStatus { + Succeeded, + Running, + NotStarted, + Failed, + Unknown(String), +} + +impl<'de> Deserialize<'de> for OperationStatus { + fn deserialize>(deserializer: D) -> Result { + Ok(match String::deserialize(deserializer)?.as_str() { + "succeeded" => Self::Succeeded, + "running" => Self::Running, + "notStarted" => Self::NotStarted, + "failed" => Self::Failed, + value => Self::Unknown(value.to_string()), + }) + } +} + +impl std::fmt::Display for OperationStatus { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter.write_str(match self { + Self::Succeeded => "succeeded", + Self::Running => "running", + Self::NotStarted => "notStarted", + Self::Failed => "failed", + Self::Unknown(value) => value, + }) + } +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct AzureDocumentIntelligenceOperation { + pub status: Option, + #[serde(rename = "analyzeResult")] + pub analyze_result: Option, +} + +#[derive(Clone, Debug, Default, Deserialize)] +pub(crate) struct AzureDocumentIntelligenceAnalyzeResult { + pub content: Option, + #[serde(default)] + pub pages: Vec, + pub tables: Option>>, + #[serde(rename = "keyValuePairs")] + pub key_value_pairs: Option>>, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct AzureDocumentIntelligencePage { + #[serde(rename = "pageNumber", default, deserialize_with = "optional_i64")] + pub page_number: Option, + #[serde(default, deserialize_with = "optional_f64")] + pub width: Option, + #[serde(default, deserialize_with = "optional_f64")] + pub height: Option, + pub unit: Option, + #[serde(default)] + pub lines: Vec, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct AzureDocumentIntelligenceLine { + pub content: Option, +} + +fn optional_i64<'de, D: Deserializer<'de>>(deserializer: D) -> Result, D::Error> { + match Option::::deserialize(deserializer)? { + None | Some(Value::Null) => Ok(None), + Some(Value::Number(number)) => number + .as_i64() + .map(Some) + .ok_or_else(|| serde::de::Error::custom("expected an integer")), + Some(Value::String(value)) => value + .parse::() + .map(Some) + .map_err(|_| serde::de::Error::custom("expected an integer")), + Some(_) => Err(serde::de::Error::custom("expected an integer")), + } +} + +fn optional_f64<'de, D: Deserializer<'de>>(deserializer: D) -> Result, D::Error> { + match Option::::deserialize(deserializer)? { + None | Some(Value::Null) => Ok(None), + Some(Value::Number(number)) => number + .as_f64() + .filter(|value| value.is_finite()) + .map(Some) + .ok_or_else(|| serde::de::Error::custom("expected a finite number")), + Some(Value::String(value)) => value + .parse::() + .ok() + .filter(|value| value.is_finite()) + .map(Some) + .ok_or_else(|| serde::de::Error::custom("expected a finite number")), + Some(_) => Err(serde::de::Error::custom("expected a number")), + } +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/mod.rs b/litellm-rust/crates/core/src/ocr/codecs/mod.rs index 170ef5f68a7..ef525e3692f 100644 --- a/litellm-rust/crates/core/src/ocr/codecs/mod.rs +++ b/litellm-rust/crates/core/src/ocr/codecs/mod.rs @@ -1 +1,2 @@ +pub(crate) mod document_intelligence; pub(crate) mod mistral; diff --git a/litellm-rust/crates/core/src/ocr/error.rs b/litellm-rust/crates/core/src/ocr/error.rs index 395bb60000c..2278c6ba948 100644 --- a/litellm-rust/crates/core/src/ocr/error.rs +++ b/litellm-rust/crates/core/src/ocr/error.rs @@ -22,6 +22,12 @@ pub enum OcrRequestError { DownloadTooLarge, #[error("OCR document download exceeded the redirect limit")] TooManyRedirects, + #[error("invalid OCR pages: {0}")] + Pages(String), + #[error("invalid OCR features")] + Features, + #[error("OCR model cannot be a dot segment")] + DotModel, } #[derive(Debug, Clone, PartialEq, Eq, Error)] @@ -32,6 +38,20 @@ pub enum OcrResponseError { MissingRedirectLocation, #[error("OCR document redirect location is invalid")] InvalidRedirect, + #[error("OCR operation ended with status {0}")] + OperationStatus(String), + #[error("OCR response numeric value is out of range: {0}")] + NumericRange(&'static str), +} + +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum OcrPollingError { + #[error("OCR accepted response is missing a valid operation-location")] + PollLocation, + #[error("OCR operation-location must use the submission origin without credentials")] + PollOrigin, + #[error("OCR polling timed out")] + PollTimeout, } #[derive(Debug, Error)] @@ -43,6 +63,8 @@ pub enum OcrError { #[error("{0}")] Transport(#[from] TransportError), #[error("{0}")] + Polling(#[from] OcrPollingError), + #[error("{0}")] Public(#[from] crate::Error), } @@ -52,6 +74,7 @@ impl From for crate::Error { OcrError::Request(error) => error.into(), OcrError::Response(error) => error.into(), OcrError::Transport(error) => error.into(), + OcrError::Polling(error) => crate::Error::InvalidResponse(error.to_string()), OcrError::Public(error) => error, } } diff --git a/litellm-rust/crates/core/src/ocr/mod.rs b/litellm-rust/crates/core/src/ocr/mod.rs index aa11d0ab3cf..f7cda0009b0 100644 --- a/litellm-rust/crates/core/src/ocr/mod.rs +++ b/litellm-rust/crates/core/src/ocr/mod.rs @@ -18,6 +18,9 @@ pub use types::{LiteLLMOcrRequest, LiteLLMOcrResponse, OcrConnection, OcrDocumen #[path = "../../tests/azure_ai_ocr.rs"] mod azure_ai_tests; #[cfg(test)] +#[path = "../../tests/azure_document_intelligence_ocr.rs"] +mod azure_document_intelligence_tests; +#[cfg(test)] #[path = "../../tests/ocr/support.rs"] pub(crate) mod test_support; #[cfg(test)] diff --git a/litellm-rust/crates/core/src/ocr/registry.rs b/litellm-rust/crates/core/src/ocr/registry.rs index e70bd7314c0..676c54579bc 100644 --- a/litellm-rust/crates/core/src/ocr/registry.rs +++ b/litellm-rust/crates/core/src/ocr/registry.rs @@ -52,9 +52,10 @@ pub(crate) fn resolve_wire_adapter( }; match typed_provider { OcrProvider::Mistral => Ok((provider.model.to_string(), OcrAdapterKind::Mistral)), - OcrProvider::AzureAi if is_document_intelligence_model(provider.model) => { - Err(Error::InvalidProvider("azure_ai".into())) - } + OcrProvider::AzureAi if is_document_intelligence_model(provider.model) => Ok(( + provider.model.to_string(), + OcrAdapterKind::AzureDocumentIntelligence, + )), OcrProvider::AzureAi => Ok((provider.model.to_string(), OcrAdapterKind::AzureMistral)), } } diff --git a/litellm-rust/crates/core/src/ocr/types.rs b/litellm-rust/crates/core/src/ocr/types.rs index dec474876fb..0e92b0b6868 100644 --- a/litellm-rust/crates/core/src/ocr/types.rs +++ b/litellm-rust/crates/core/src/ocr/types.rs @@ -74,6 +74,7 @@ pub struct OcrConnection { pub extra_headers_source: InputSource, pub timeout: Duration, pub max_download_bytes: u64, + pub poll_timeout: Duration, } impl Default for OcrConnection { @@ -87,6 +88,7 @@ impl Default for OcrConnection { extra_headers_source: InputSource::Deployment, timeout: Duration::from_secs(OCR_HTTP_TIMEOUT_SECS), max_download_bytes: crate::constants::OCR_DOWNLOAD_MAX_BYTES, + poll_timeout: Duration::from_secs(crate::constants::OCR_POLL_TIMEOUT_SECS), } } } diff --git a/litellm-rust/crates/core/src/ocr/wire.rs b/litellm-rust/crates/core/src/ocr/wire.rs index d0fe32378b9..34d0a7d7b86 100644 --- a/litellm-rust/crates/core/src/ocr/wire.rs +++ b/litellm-rust/crates/core/src/ocr/wire.rs @@ -81,6 +81,7 @@ pub fn decode_request(wire: OcrWireRequest) -> Result extra_headers_source, timeout: timeout.unwrap_or(defaults.timeout), max_download_bytes: defaults.max_download_bytes, + poll_timeout: defaults.poll_timeout, }; Ok(LiteLLMOcrRequest { connection, diff --git a/litellm-rust/crates/core/src/url_utils.rs b/litellm-rust/crates/core/src/url_utils.rs index 982dca0dbe3..1150f93a5c7 100644 --- a/litellm-rust/crates/core/src/url_utils.rs +++ b/litellm-rust/crates/core/src/url_utils.rs @@ -60,6 +60,14 @@ impl ApiUrl { } impl ApiUrl { + pub(crate) fn append_query_pairs<'a>( + mut self, + pairs: impl IntoIterator, + ) -> Self { + self.url.query_pairs_mut().extend_pairs(pairs); + self + } + pub(crate) fn into_string(self) -> String { self.url.into() } @@ -92,4 +100,19 @@ mod tests { .expect("url builds"); assert_eq!(actual, "https://example.test/v1/ocr?tenant=a"); } + + #[test] + fn appended_query_pairs_are_encoded() { + let actual = ApiUrl::parse("https://example.test") + .and_then(|url| url.complete_path(&["analyze"])) + .map(|url| { + url.append_query_pairs([("model", "name with spaces")]) + .into_string() + }) + .expect("url builds"); + assert_eq!( + actual, + "https://example.test/analyze?model=name+with+spaces" + ); + } } diff --git a/litellm-rust/crates/core/tests/azure_document_intelligence_ocr.rs b/litellm-rust/crates/core/tests/azure_document_intelligence_ocr.rs new file mode 100644 index 00000000000..e4c81dea5a7 --- /dev/null +++ b/litellm-rust/crates/core/tests/azure_document_intelligence_ocr.rs @@ -0,0 +1,395 @@ +use serde_json::{Value, json}; + +use super::test_support::{MockResponse, mock_server, perform_ocr, wire_request}; +use super::wire::{OcrWireRequest, decode_request}; + +fn query_value(url: &str, key: &str) -> Option { + url::Url::parse(url) + .unwrap() + .query_pairs() + .find_map(|(name, value)| (name == key).then(|| value.into_owned())) +} + +#[tokio::test] +async fn facade_maps_pages_features_and_url_document() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "status":"succeeded", + "analyzeResult":{"pages":[]} + }))]) + .await; + let mut request = wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({"pages":[2,0,0,1],"features":["keyValuePairs","languages"]}), + ); + request.document = serde_json::from_value(json!({ + "type":"document_url", + "document_url":"https://example.com/document.pdf" + })) + .unwrap(); + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + let request = &seen.lock().unwrap()[0]; + let target = request.split_whitespace().nth(1).unwrap(); + let url = format!("{base}{target}"); + assert_eq!(query_value(&url, "pages").as_deref(), Some("1,2,3")); + assert_eq!( + query_value(&url, "features").as_deref(), + Some("keyValuePairs,languages") + ); + let body: Value = serde_json::from_str(request.split_once("\r\n\r\n").unwrap().1).unwrap(); + assert_eq!( + body, + json!({"urlSource":"https://example.com/document.pdf"}) + ); +} + +#[tokio::test] +async fn rejects_invalid_pages_features_and_format() { + for options in [ + json!({"pages":[true]}), + json!({"pages":[1,"2"]}), + json!({"pages":[-1]}), + json!({"pages":"1&&features=bad"}), + json!({"features":"languages&pages=1"}), + json!({"req_format":"azure"}), + ] { + let result = decode_request(OcrWireRequest { + model: "azure_ai/doc-intelligence/prebuilt-read".into(), + document: json!({"type":"document_url","document_url":"https://example.com/a.pdf"}), + api_key: Some("key".into()), + api_base: Some("http://127.0.0.1:1".into()), + custom_llm_provider: None, + extra_headers: None, + optional_params: options.as_object().unwrap().clone(), + input_sources: Default::default(), + timeout_seconds: None, + }); + let rejected = match result { + Ok(request) => perform_ocr(request).await.is_err(), + Err(_) => true, + }; + assert!(rejected, "accepted {options}"); + } +} + +#[tokio::test] +async fn inline_document_decodes_to_base64_source() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "status":"succeeded" + }))]) + .await; + let request = wire_request("azure_ai/doc-intelligence/prebuilt-read", &base, json!({})); + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + let request = &seen.lock().unwrap()[0]; + let body: Value = serde_json::from_str(request.split_once("\r\n\r\n").unwrap().1).unwrap(); + assert_eq!(body, json!({"base64Source":"YWJj"})); +} + +#[tokio::test] +async fn immediate_response_normalizes_pages_and_preserves_native() { + let operation = json!({ + "status":"succeeded", + "operationExtension":42, + "analyzeResult":{ + "content":"A\n\nB", + "tables":[{"cells":[]}], + "keyValuePairs":[{"key":{"content":"A"}}], + "pages":[{ + "pageNumber":"2", + "width":"8.5", + "height":11, + "unit":"inch", + "lines":[{"content":"A"},{"content":null},{"content":"B"}] + }] + } + }); + let (base, _, server) = mock_server(vec![MockResponse::json(operation.clone())]).await; + let result = perform_ocr(wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({"req_format":"native"}), + )) + .await + .unwrap(); + server.await.unwrap(); + + assert_eq!(result.pages[0]["index"], 1); + assert_eq!(result.pages[0]["markdown"], "A\n\nB"); + assert_eq!( + result.pages[0]["dimensions"], + json!({"width":816,"height":1056,"dpi":96}) + ); + assert_eq!(result.usage_info, Some(json!({"pages_processed":1}))); + assert_eq!(result.provider_native_response, Some(operation)); +} + +#[tokio::test] +async fn accepted_response_polls_to_success_with_only_credentials() { + let operation = json!({"status":"succeeded","analyzeResult":{"pages":[]}}); + let (base, seen, server) = mock_server(vec![ + MockResponse { + status: 202, + headers: vec![("Operation-Location", "{base}/operation".into())], + body: json!({}), + }, + MockResponse { + status: 200, + headers: vec![("Retry-After", "0".into())], + body: json!({"status":"running"}), + }, + MockResponse::json(operation.clone()), + ]) + .await; + let mut request = wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({"req_format":"native"}), + ); + request + .connection + .extra_headers + .push(("X-Trace".into(), "initial-only".into())); + + let result = perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert_eq!(result.provider_native_response, Some(operation)); + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 3); + assert!(requests[0].to_ascii_lowercase().contains("x-trace:")); + for poll in &requests[1..] { + assert!(!poll.to_ascii_lowercase().contains("x-trace:")); + assert!( + poll.to_ascii_lowercase() + .contains("ocp-apim-subscription-key: test-key") + ); + } +} + +#[tokio::test] +async fn polling_forwards_bearer_credentials() { + let (base, seen, server) = mock_server(vec![ + MockResponse { + status: 202, + headers: vec![("Operation-Location", "{base}/operation".into())], + body: json!({}), + }, + MockResponse::json(json!({"status":"succeeded"})), + ]) + .await; + let mut request = wire_request("azure_ai/doc-intelligence/prebuilt-read", &base, json!({})); + request.connection.api_key = None; + request.connection.extra_headers = vec![("Authorization".into(), "Bearer token".into())]; + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + let requests = seen.lock().unwrap(); + assert!( + requests[1] + .to_ascii_lowercase() + .contains("authorization: bearer token") + ); +} + +#[tokio::test] +async fn polling_does_not_follow_redirects() { + let (base, seen, server) = mock_server(vec![ + MockResponse { + status: 202, + headers: vec![("Operation-Location", "{base}/operation".into())], + body: json!({}), + }, + MockResponse { + status: 302, + headers: vec![("Location", "{base}/redirected".into())], + body: json!({}), + }, + MockResponse::json(json!({"status":"succeeded"})), + ]) + .await; + + let error = perform_ocr(wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({}), + )) + .await + .unwrap_err(); + + assert!(error.to_string().contains("status 302"), "{error}"); + assert_eq!(seen.lock().unwrap().len(), 2); + server.abort(); +} + +#[tokio::test] +async fn polling_rejects_terminal_failure() { + let (base, _, server) = mock_server(vec![ + MockResponse { + status: 202, + headers: vec![("Operation-Location", "{base}/operation".into())], + body: json!({}), + }, + MockResponse::json(json!({"status":"failed"})), + ]) + .await; + + let error = perform_ocr(wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({}), + )) + .await + .unwrap_err(); + server.await.unwrap(); + assert!(error.to_string().contains("status failed")); +} + +#[tokio::test] +async fn malformed_provider_pages_report_response_paths() { + for (analysis, path) in [ + (json!({"pages":null}), "pages"), + (json!({"pages":[null]}), "pages[0]"), + (json!({"pages":[{"lines":null}]}), "lines"), + (json!({"pages":[{"width":"bad"}]}), "width"), + ] { + let (base, _, server) = mock_server(vec![MockResponse::json(json!({ + "status":"succeeded", + "analyzeResult":analysis + }))]) + .await; + let error = perform_ocr(wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({}), + )) + .await + .unwrap_err(); + server.await.unwrap(); + assert!(error.to_string().contains(path), "{error}"); + } +} + +#[tokio::test] +async fn rejects_missing_invalid_and_cross_origin_operation_locations() { + for headers in [ + Vec::new(), + vec![("Operation-Location", "/relative".into())], + vec![("Operation-Location", "http://example.com/operation".into())], + vec![( + "Operation-Location", + "http://user:password@127.0.0.1/operation".into(), + )], + ] { + let (base, _, server) = mock_server(vec![MockResponse { + status: 202, + headers, + body: json!({}), + }]) + .await; + let error = perform_ocr(wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({}), + )) + .await + .unwrap_err(); + server.await.unwrap(); + assert!(error.to_string().contains("operation-location")); + } +} + +#[tokio::test] +async fn polling_deadline_bounds_retry_delay() { + let (base, _, server) = mock_server(vec![ + MockResponse { + status: 202, + headers: vec![("Operation-Location", "{base}/operation".into())], + body: json!({}), + }, + MockResponse { + status: 200, + headers: vec![("Retry-After", "9999".into())], + body: json!({"status":"notStarted"}), + }, + ]) + .await; + let mut request = wire_request("azure_ai/doc-intelligence/prebuilt-read", &base, json!({})); + request.connection.poll_timeout = std::time::Duration::from_millis(100); + + let error = tokio::time::timeout(std::time::Duration::from_secs(1), perform_ocr(request)) + .await + .unwrap() + .unwrap_err(); + server.await.unwrap(); + assert!(error.to_string().contains("timed out")); +} + +#[tokio::test] +async fn model_id_is_encoded_and_dot_segments_are_rejected() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "status":"succeeded" + }))]) + .await; + perform_ocr(wire_request( + "azure_ai/doc-intelligence/a ?#é", + &base, + json!({}), + )) + .await + .unwrap(); + server.await.unwrap(); + assert!(seen.lock().unwrap()[0].contains("a%20%3F%23%C3%A9:analyze")); + + for model in [ + "azure_ai/doc-intelligence/.", + "azure_ai/doc-intelligence/..", + ] { + let error = perform_ocr(wire_request(model, "http://127.0.0.1:1", json!({}))) + .await + .unwrap_err(); + assert!(error.to_string().contains("dot segment")); + } +} + +#[tokio::test] +async fn pre_call_guardrail_receives_caller_pages_before_mapping() { + use crate::ocr::hooks::{OcrHookFuture, OcrHooks, OcrPreCallRequest}; + use std::sync::Arc; + + struct RewritePages; + impl OcrHooks for RewritePages { + fn has_guardrails(&self) -> bool { + true + } + + fn pre_call(&self, request: OcrPreCallRequest) -> OcrHookFuture<'_, OcrPreCallRequest> { + Box::pin(async move { + assert_eq!(request.optional_params["pages"], json!([0, 2])); + Ok(OcrPreCallRequest { + optional_params: json!({"pages": [1]}), + ..request + }) + }) + } + } + let (base, seen, server) = + mock_server(vec![MockResponse::json(json!({"status": "succeeded"}))]).await; + let request = wire_request( + "azure_ai/doc-intelligence/prebuilt-read", + &base, + json!({"pages": [0, 2]}), + ) + .with_host_hooks(Arc::new(RewritePages), None); + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + let requests = seen.lock().unwrap(); + let target = requests[0].split_whitespace().nth(1).unwrap(); + assert_eq!( + query_value(&format!("{base}{target}"), "pages").as_deref(), + Some("2") + ); + assert_eq!(requests.len(), 1); +} diff --git a/litellm-rust/crates/python-bridge/src/errors.rs b/litellm-rust/crates/python-bridge/src/errors.rs index 0c30eab8112..9ae25e790a6 100644 --- a/litellm-rust/crates/python-bridge/src/errors.rs +++ b/litellm-rust/crates/python-bridge/src/errors.rs @@ -44,6 +44,7 @@ pub(crate) fn chat_completions_error_to_pyerr(err: Error) -> PyErr { | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials | Error::MissingAzureAiCredentialsOrAdToken + | Error::MissingAzureDocumentIntelligenceCredentials | Error::Routing(_) // Nothing reached the provider, so serving it on Python cannot double // bill and is the only way the caller gets an answer at all. diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index 50095e3ebf2..e960a354d9f 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -102,10 +102,10 @@ mod tests { use litellm_core::ocr::wire::is_supported_request; #[test] - fn native_activation_excludes_unmigrated_azure_document_intelligence() { + fn native_activation_includes_azure_document_intelligence() { assert!(is_supported_request("model", Some("mistral"))); assert!(is_supported_request("pixtral-12b", Some("azure_ai"))); - assert!(!is_supported_request( + assert!(is_supported_request( "documentintelligence/prebuilt-read", Some("azure_ai") )); diff --git a/litellm/rust_bridge/ocr.py b/litellm/rust_bridge/ocr.py index db959e76f7c..68bbb186f9b 100644 --- a/litellm/rust_bridge/ocr.py +++ b/litellm/rust_bridge/ocr.py @@ -2,14 +2,44 @@ from __future__ import annotations -from collections.abc import Awaitable, Mapping +from collections.abc import Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass +from types import MappingProxyType from typing import Final, Protocol, cast # noqa: TID251 # native extension exposes dynamically typed callables import httpx -from litellm.rust_bridge.bindings import NativeBinding +import litellm +from litellm.constants import request_timeout +from litellm.llms.azure_ai.ocr.common_utils import is_azure_cohere_parse_model +from litellm.llms.base_llm.ocr.transformation import PROVIDER_NATIVE_RESPONSE_KEY, OCRResponse +from litellm.rust_bridge.bindings import NativeBinding, native_exception_types from litellm.rust_bridge.timeouts import timeout_to_seconds as _timeout_to_seconds +from litellm.types.router import GenericLiteLLMParams +from litellm.utils import ProviderConfigManager + +_RUST_OCR_PROVIDERS: Final = frozenset({"mistral", "azure_ai", "vertex_ai"}) +_RUST_OCR_CONFIG_FIELDS: Final = frozenset( + { + "azure_ad_token", + "tenant_id", + "client_id", + "client_secret", + "azure_scope", + "azure_authority_host", + "azure_credential", + "azure_federated_token_file", + "vertex_credentials", + "vertex_ai_credentials", + "vertex_project", + "vertex_ai_project", + "vertex_location", + "vertex_ai_location", + } +) +_RUST_OCR_SECRET_FIELDS: Final = frozenset( + {"azure_ad_token", "client_secret", "azure_federated_token_file", "vertex_credentials", "vertex_ai_credentials"} +) @dataclass(frozen=True, slots=True) @@ -57,6 +87,26 @@ class RustAocr(Protocol): raise NotImplementedError +class _OCRLogging(Protocol): + def update_from_kwargs( + self, + *, + kwargs: dict[str, object], + model: str, + optional_params: dict[str, object], + litellm_params: dict[str, object], + custom_llm_provider: str | None, + ) -> None: ... + + def pre_call( + self, + *, + input: str, + api_key: str | None, + additional_args: dict[str, object], + ) -> None: ... + + def _as_ocr(value: object) -> RustOcr | None: return cast(RustOcr, value) if callable(value) else None @@ -77,6 +127,255 @@ def load_rust_aocr() -> RustAocr | None: return _AOCR.load() +def provider(request: LiteLLMOcrRequest) -> str | None: + if request.custom_llm_provider is not None: + return request.custom_llm_provider + prefix: Final = request.model.partition("/")[0] + if prefix in _RUST_OCR_PROVIDERS: + return prefix + if request.model.startswith("mistral-ocr"): + return "mistral" + return None + + +def supported(request: LiteLLMOcrRequest) -> bool: + request_provider: Final = provider(request) + if request_provider not in _RUST_OCR_PROVIDERS: + return False + if request_provider == "azure_ai": + return ( + not is_azure_cohere_parse_model(request.model) + and not callable(request.kwargs.get("azure_ad_token_provider")) + and request.kwargs.get("azure_username") is None + and request.kwargs.get("azure_password") is None + ) + return True + + +def _optional_params(request: LiteLLMOcrRequest, resolve_secret: Callable[[str], str | None]) -> Mapping[str, object]: + optional_params: Final = MappingProxyType( + { + name: value + for name, value in request.kwargs.items() + if (name not in GenericLiteLLMParams.model_fields or name in _RUST_OCR_CONFIG_FIELDS) + and name not in ("litellm_logging_obj", "aocr", "litellm_call_id", "proxy_server_request") + } + ) + request_provider: Final = provider(request) + if request_provider == "azure_ai" and litellm.enable_azure_ad_token_refresh is True: + return MappingProxyType({**optional_params, "enable_azure_ad_token_refresh": True}) + if request_provider != "vertex_ai": + return optional_params + project: Final = ( + request.kwargs.get("vertex_project") + or request.kwargs.get("vertex_ai_project") + or litellm.vertex_project + or resolve_secret("VERTEXAI_PROJECT") + ) + location: Final = ( + request.kwargs.get("vertex_location") + or request.kwargs.get("vertex_ai_location") + or litellm.vertex_location + or resolve_secret("VERTEXAI_LOCATION") + or resolve_secret("VERTEX_LOCATION") + ) + vertex_params: Final = MappingProxyType( + { + name: value + for name, value in (("vertex_project", project), ("vertex_location", location)) + if value is not None + } + ) + return MappingProxyType({**optional_params, **vertex_params}) + + +def _input_sources(request: LiteLLMOcrRequest, optional_params: Mapping[str, object]) -> Mapping[str, str]: + proxy_request_value: Final = request.kwargs.get("proxy_server_request") + if not isinstance(proxy_request_value, Mapping): + return MappingProxyType({}) + proxy_request: Final = cast( # cast-ok: runtime Mapping check narrows metadata with unknown key and value types + Mapping[object, object], proxy_request_value + ) + credential_fields_value: Final = proxy_request.get("credential_fields", ()) + credential_fields: Final = ( + frozenset(name for name in credential_fields_value if isinstance(name, str)) + if isinstance(credential_fields_value, (list, tuple, set, frozenset)) + else frozenset() + ) + request_fields_value: Final = proxy_request.get("body_fields") + request_fields: Sequence[object] + if isinstance(request_fields_value, Sequence) and not isinstance(request_fields_value, (str, bytes)): + request_fields = cast( # cast-ok: runtime Sequence check excludes scalar strings and bytes + Sequence[object], request_fields_value + ) + else: + body_value: Final = proxy_request.get("body") + request_fields = ( + tuple(cast(Mapping[object, object], body_value)) # cast-ok: runtime Mapping check establishes iterable keys + if isinstance(body_value, Mapping) + else () + ) + names: Final = frozenset(optional_params) | frozenset({"api_key", "api_base", "extra_headers"}) + request_sources: Final = MappingProxyType( + {name: "request" for name in names if name in request_fields or name in credential_fields} + ) + if litellm.enable_azure_ad_token_refresh is True and "enable_azure_ad_token_refresh" in optional_params: + return MappingProxyType({**request_sources, "enable_azure_ad_token_refresh": "deployment"}) + return request_sources + + +def _marshal( + request: LiteLLMOcrRequest, + resolve_secret: Callable[[str], str | None], + convert_file_document: Callable[[dict[str, object]], dict[str, str]], +) -> LiteLLMOcrRequest: + if not isinstance(request.document, dict): + raise TypeError(f"document must be a dict with 'type' and URL/file field, got {type(request.document)}") + document: Final = ( + convert_file_document(request.document) if request.document.get("type") == "file" else request.document + ) + request_provider: Final = provider(request) + api_key: Final = ( + request.api_key or resolve_secret("MISTRAL_API_KEY") if request_provider == "mistral" else request.api_key + ) + optional_params: Final = _optional_params(request, resolve_secret) + input_sources: Final = _input_sources(request, optional_params) + logged_optional_params: Final = MappingProxyType( + {name: "****" if name in _RUST_OCR_SECRET_FIELDS else value for name, value in optional_params.items()} + ) + logged_kwargs: Final = MappingProxyType( + { + name: "****" if name in _RUST_OCR_SECRET_FIELDS else value + for name, value in request.kwargs.items() + if name != "proxy_server_request" + } + ) + logging_obj: Final = cast( # cast-ok: client decorator injects the logging object through untyped kwargs + _OCRLogging, request.kwargs["litellm_logging_obj"] + ) + logging_obj.update_from_kwargs( + kwargs=dict(logged_kwargs), # mutable-ok: legacy logging mutates its kwargs copy + model=request.model, + optional_params=dict(logged_optional_params), # mutable-ok: legacy logging requires concrete dict params + litellm_params={ # mutable-ok: legacy logging requires a concrete params dict + "litellm_call_id": request.kwargs.get("litellm_call_id"), + "api_base": request.api_base, + }, + custom_llm_provider=request_provider, + ) + logging_obj.pre_call( + input="OCR document processing", + api_key=api_key, + additional_args={ # mutable-ok: pre_call mutates the additional_args dict + "complete_input_dict": { # mutable-ok: callbacks consume a JSON-serializable request dict + "model": request.model, + "document": document, + **logged_optional_params, + }, + "api_base": request.api_base or "", + "headers": request.extra_headers or {}, # mutable-ok: logging callbacks consume a concrete headers dict + }, + ) + return LiteLLMOcrRequest( + model=request.model, + document=document, + api_key=api_key, + api_base=request.api_base, + timeout=request.timeout if request.timeout is not None else request_timeout, + custom_llm_provider=request.custom_llm_provider, + extra_headers=request.extra_headers, + kwargs=optional_params, + input_sources=input_sources, + ) + + +def _map_error(error: Exception, request: LiteLLMOcrRequest) -> Exception: + exception_types: Final = native_exception_types() + if exception_types is None or not isinstance(error, exception_types[1]): + return error + request_provider: Final = provider(request) + if request_provider is None: + return error + provider_config: Final = ProviderConfigManager.get_provider_ocr_config( + model=request.model.removeprefix(f"{request_provider}/"), provider=litellm.LlmProviders(request_provider) + ) + if provider_config is None: + return error + error_args: Final = cast( # cast-ok: BaseException.args exposes Any while native errors carry scalar args + tuple[object, ...], error.args + ) + status: Final = error_args[0] if error_args and isinstance(error_args[0], int) else 500 + message: Final = str(error_args[1]) if len(error_args) > 1 else str(error) + error_factory: Final = cast( # cast-ok: legacy provider error factories have untyped callable parameters + Callable[..., Exception], provider_config.get_error_class + ) + return error_factory( + error_message=message, + status_code=status or 500, + headers={}, # mutable-ok: provider error factories require a concrete headers dict + ) + + +def _response(response: Mapping[str, object]) -> OCRResponse: + provider_native_response: Final = response.get(PROVIDER_NATIVE_RESPONSE_KEY) + normalized: Final = OCRResponse.model_validate( + MappingProxyType({key: value for key, value in response.items() if key != PROVIDER_NATIVE_RESPONSE_KEY}) + ) + if isinstance(provider_native_response, Mapping): + normalized.set_provider_native_response(provider_native_response) + return normalized + + +def run( + request: LiteLLMOcrRequest, + resolve_secret: Callable[[str], str | None], + convert_file_document: Callable[[dict[str, object]], dict[str, str]], +) -> OCRResponse | None: + if load_rust_ocr() is None: + return None + marshalled: Final = _marshal(request, resolve_secret, convert_file_document) + try: + response: Final = ocr( + model=marshalled.model, + document=dict(marshalled.document), # mutable-ok: PyO3 OCR binding requires a concrete dict + api_key=marshalled.api_key, + api_base=marshalled.api_base, + custom_llm_provider=marshalled.custom_llm_provider, + extra_headers=marshalled.extra_headers, + optional_params=dict(marshalled.kwargs), # mutable-ok: PyO3 OCR binding requires a concrete dict + input_sources=marshalled.input_sources, + timeout=marshalled.timeout, + ) + except Exception as error: + raise _map_error(error, request) from error + return _response(response) if response is not None else None + + +async def arun( + request: LiteLLMOcrRequest, + resolve_secret: Callable[[str], str | None], + convert_file_document: Callable[[dict[str, object]], dict[str, str]], +) -> OCRResponse | None: + if load_rust_aocr() is None: + return None + marshalled: Final = _marshal(request, resolve_secret, convert_file_document) + try: + response: Final = await aocr( + model=marshalled.model, + document=dict(marshalled.document), # mutable-ok: PyO3 OCR binding requires a concrete dict + api_key=marshalled.api_key, + api_base=marshalled.api_base, + custom_llm_provider=marshalled.custom_llm_provider, + extra_headers=marshalled.extra_headers, + optional_params=dict(marshalled.kwargs), # mutable-ok: PyO3 OCR binding requires a concrete dict + input_sources=marshalled.input_sources, + timeout=marshalled.timeout, + ) + except Exception as error: + raise _map_error(error, request) from error + return _response(response) if response is not None else None + + def ocr( *, model: str, diff --git a/tests/test_litellm/ocr/test_ocr_native_format.py b/tests/test_litellm/ocr/test_ocr_native_format.py index 5f69708fe91..46e9a4d3729 100644 --- a/tests/test_litellm/ocr/test_ocr_native_format.py +++ b/tests/test_litellm/ocr/test_ocr_native_format.py @@ -1,13 +1,11 @@ """ -Tests for the OCR `req_format` option in the SDK request path: -providers that don't support a native response must reject it, and the Rust -bridge (which only returns the normalized shape) must not serve native requests. +Tests for the OCR `req_format` option in the SDK request path. """ import pytest import litellm -from litellm.ocr.main import _rust_ocr_supported +from litellm.rust_bridge import ocr as rust_ocr_bridge from litellm.rust_bridge.ocr import LiteLLMOcrRequest DOCUMENT = {"type": "document_url", "document_url": "https://example.com/doc.pdf"} @@ -30,16 +28,33 @@ def _request( @pytest.mark.parametrize("optional_params", [{}, {"req_format": "litellm"}]) def test_rust_ocr_serves_default_format(optional_params): - assert _rust_ocr_supported(_request(optional_params)) is True + assert rust_ocr_bridge.supported(_request(optional_params)) is True -def test_rust_ocr_skipped_for_native_format(): - assert _rust_ocr_supported(_request({"req_format": "native"})) is False +def test_rust_ocr_serves_native_format_for_document_intelligence(): + assert rust_ocr_bridge.supported(_request({"req_format": "native"})) is True + + +def test_rust_ocr_response_retains_provider_native_response(): + provider_response = {"status": "succeeded", "analyzeResult": {"content": "native"}} + response = rust_ocr_bridge._response( + { + "pages": [], + "model": "prebuilt-layout", + "document_annotation": None, + "usage_info": {"pages_processed": 0}, + "object": "ocr", + "provider_native_response": provider_response, + } + ) + + assert response.get_provider_native_response() == provider_response + assert response.model_dump().get("provider_native_response") is None @pytest.mark.parametrize("model", ["cohere/cohere-parse", "azure_ai/cohere-parse"]) def test_rust_ocr_skipped_for_unsupported_models(model): - assert _rust_ocr_supported(_request({}, model)) is False + assert rust_ocr_bridge.supported(_request({}, model)) is False @pytest.mark.asyncio diff --git a/tests/test_litellm/ocr/test_rust_bridge.py b/tests/test_litellm/ocr/test_rust_bridge.py index 3441bd4de34..9c1cae6a551 100644 --- a/tests/test_litellm/ocr/test_rust_bridge.py +++ b/tests/test_litellm/ocr/test_rust_bridge.py @@ -708,6 +708,24 @@ def test_prepare_rust_ocr_call_preserves_proxy_input_sources(): "api_key": "request", } + marshaled = rust_bridge._marshal( + build_request( + custom_llm_provider="azure_ai", + model="pixtral-12b-2409", + api_key="request-key", + api_base="https://azure.example.com", + litellm_params={ + "proxy_server_request": { + "body": {"api_base": "https://azure.example.com"}, + "credential_fields": ("api_key",), + } + }, + ), + lambda _name: None, + lambda document: document, + ) + assert marshaled.input_sources == {"api_base": "request", "api_key": "request"} + def test_rust_ocr_logging_redacts_azure_credentials(): bridge = RecordingBridge() diff --git a/tests/test_litellm/rust_bridge/native_route_wheel_test.py b/tests/test_litellm/rust_bridge/native_route_wheel_test.py index 3b70043fada..8d83f4ca8a6 100644 --- a/tests/test_litellm/rust_bridge/native_route_wheel_test.py +++ b/tests/test_litellm/rust_bridge/native_route_wheel_test.py @@ -73,7 +73,7 @@ def assert_native_request( headers: HTTPMessage, body: object, ) -> None: - if route not in {"ocr", "azure_ocr", "transcription", "messages", "chat_completions"}: + if route not in {"ocr", "azure_ocr", "azure_di", "transcription", "messages", "chat_completions"}: raise AssertionError(f"unexpected route marker: {route!r}") if outcome not in {"success", "429", "hang"}: raise AssertionError(f"unexpected outcome marker: {outcome!r}") @@ -92,6 +92,13 @@ def assert_native_request( assert body["model"] == "mistral-ocr-2505" assert body["document"]["document_url"] == "data:application/pdf;base64,YWJj" return + if route == "azure_di": + assert path.startswith("/documentintelligence/documentModels/prebuilt-read:analyze?") + assert "api-version=2024-11-30" in path + assert "pages=1%2C3" in path + assert headers.get("ocp-apim-subscription-key") == "di-key" + assert body == {"base64Source": "YWJj"} + return if route == "transcription": assert path == "/model/mistral.voxtral-mini-3b-2507/converse" assert headers.get("authorization", "").startswith("AWS4-HMAC-SHA256 ") @@ -115,6 +122,8 @@ def native_response(status: int, route: str | None) -> bytes: return b'{"error":"native-rate-limit"}' if route in {"ocr", "azure_ocr"}: return b'{"pages":[{"index":0,"markdown":"native-ocr"}]}' + if route == "azure_di": + return b'{"status":"succeeded","analyzeResult":{"pages":[]}}' if route == "transcription": return b'{"output":{"message":{"content":[{"text":"native-transcription"}]}}}' return ANTHROPIC_RESPONSE @@ -201,6 +210,18 @@ def azure_ocr_kwargs(api_base: str) -> dict[str, object]: } +def azure_di_kwargs(api_base: str) -> dict[str, object]: + return { + "model": "doc-intelligence/prebuilt-read", + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "api_key": "di-key", + "api_base": api_base, + "custom_llm_provider": "azure_ai", + "extra_headers": {"x-test-outcome": "success", "x-test-route": "azure_di"}, + "optional_params": {"req_format": "native", "pages": [0, 2]}, + } + + def success_value(route: str, response: dict[object, object]) -> object: if route == "ocr": return response["pages"][0]["markdown"] @@ -232,6 +253,8 @@ def exercise_sync(native: object, api_base: str) -> None: else: raise AssertionError(f"{route} accepted a 429 response") assert_success("ocr", native.ocr(**azure_ocr_kwargs(api_base))) + di_response: Final = native.ocr(**azure_di_kwargs(api_base)) + assert di_response["provider_native_response"]["status"] == "succeeded" async def exercise_async(native: object, api_base: str) -> None: @@ -245,6 +268,8 @@ async def exercise_async(native: object, api_base: str) -> None: else: raise AssertionError(f"a{route} accepted a 429 response") assert_success("ocr", await native.aocr(**azure_ocr_kwargs(api_base))) + di_response: Final = await native.aocr(**azure_di_kwargs(api_base)) + assert di_response["provider_native_response"]["status"] == "succeeded" async def exercise_async_concurrency(native: object, api_base: str) -> None: From 0dd5e6e289f242acaca1bd44f6c93edcd76be1b5 Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 16:22:56 -0700 Subject: [PATCH 129/157] feat(ocr): add Reducto legacy and v3 adapters (#40535) * feat(ocr): add Reducto adapters * fix(ocr): decline missing Reducto credentials * fix(ocr): map Reducto credentials in gateway errors * test(ocr): keep Reducto coverage at SDK boundary * test(ocr): remove stale gateway Reducto cases * fix(ocr): stop retaining Reducto responses by default * refactor(ocr): preserve Reducto extra params * refactor(ocr): adopt request preparation contract * fix(ocr): preserve provider model passthrough * fix(ocr): reject unknown Reducto models * fix(ocr): preserve Reducto provider options --- .../src/audio_transcription/hooks.rs | 3 +- .../crates/ai-gateway/src/ocr/common_utils.rs | 2 - .../crates/ai-gateway/src/ocr/hooks.rs | 89 +--- litellm-rust/crates/ai-gateway/src/ocr/mod.rs | 150 +------ .../ai-gateway/src/routes/messages/mod.rs | 3 +- .../crates/ai-gateway/tests/ocr_lifecycle.rs | 89 ---- litellm-rust/crates/core/src/constants.rs | 3 + litellm-rust/crates/core/src/error.rs | 4 + .../crates/core/src/ocr/adapters/mod.rs | 4 + .../core/src/ocr/adapters/reducto/legacy.rs | 45 ++ .../core/src/ocr/adapters/reducto/mod.rs | 148 +++++++ .../core/src/ocr/adapters/reducto/v3.rs | 45 ++ .../crates/core/src/ocr/codecs/mod.rs | 1 + .../crates/core/src/ocr/codecs/reducto/mod.rs | 9 + .../src/ocr/codecs/reducto/transformation.rs | 115 +++++ .../core/src/ocr/codecs/reducto/types.rs | 128 ++++++ litellm-rust/crates/core/src/ocr/document.rs | 2 - litellm-rust/crates/core/src/ocr/error.rs | 2 + litellm-rust/crates/core/src/ocr/mod.rs | 3 + litellm-rust/crates/core/src/ocr/prepare.rs | 58 ++- litellm-rust/crates/core/src/ocr/registry.rs | 71 ++- litellm-rust/crates/core/src/providers/mod.rs | 1 - .../crates/core/src/providers/reducto/mod.rs | 1 - .../core/src/providers/reducto/ocr/mod.rs | 4 - .../core/src/providers/reducto/ocr/tests.rs | 202 --------- .../providers/reducto/ocr/transformation.rs | 407 ------------------ litellm-rust/crates/core/tests/reducto_ocr.rs | 232 ++++++++++ .../crates/python-bridge/src/errors.rs | 1 + .../crates/python-bridge/src/routes/ocr.rs | 4 +- litellm/utils.py | 4 +- .../llms/reducto/test_parse_v3.py | 25 ++ 31 files changed, 902 insertions(+), 953 deletions(-) create mode 100644 litellm-rust/crates/core/src/ocr/adapters/reducto/legacy.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/reducto/mod.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/reducto/v3.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/reducto/mod.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/reducto/transformation.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/reducto/types.rs delete mode 100644 litellm-rust/crates/core/src/providers/reducto/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/reducto/ocr/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/reducto/ocr/tests.rs delete mode 100644 litellm-rust/crates/core/src/providers/reducto/ocr/transformation.rs create mode 100644 litellm-rust/crates/core/tests/reducto_ocr.rs diff --git a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs index 873c6429ff8..17f5591d1fc 100644 --- a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs @@ -273,7 +273,8 @@ fn core_error_kind(error: &Error) -> &'static str { | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials | Error::MissingAzureAiCredentialsOrAdToken - | Error::MissingAzureDocumentIntelligenceCredentials => "AuthError", + | Error::MissingAzureDocumentIntelligenceCredentials + | Error::MissingReductoApiKey => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", Error::InvalidRequest(_) => "InvalidRequest", Error::InvalidType { .. } => "InvalidType", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs b/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs index d2be17260a3..10e08253e1a 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs @@ -12,7 +12,6 @@ use litellm_core::providers::azure_ai::ocr::transformation::{ AZURE_AI_OCR_CONFIG, AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG, }; use litellm_core::providers::mistral::ocr::transformation::MISTRAL_OCR_CONFIG; -use litellm_core::providers::reducto::ocr::transformation as reducto; use litellm_core::providers::vertex_ai::ocr::transformation as vertex_ai; use litellm_core::providers::vertex_ai::ocr::transformation::{ VERTEX_AI_DEEPSEEK_OCR_CONFIG, VERTEX_AI_OCR_CONFIG, @@ -40,7 +39,6 @@ pub(super) fn ocr_provider_config( ) -> Option<&'static dyn OcrProviderConfig> { match provider { "mistral" => Some(&MISTRAL_OCR_CONFIG), - "reducto" => reducto::config_for_model(model), "azure_ai" if is_azure_document_intelligence_model(model) => { Some(&AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG) } diff --git a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs index bca54bebd4a..3d8246af3f5 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs @@ -1,15 +1,11 @@ use litellm_core::call_lifecycle::{CallLifecycleContext, CallLifecycleHooks, CallLifecycleTiming}; use litellm_core::error::Error; -use litellm_core::providers::reducto::ocr::transformation::{ - build_upload_request, extract_document_source, extract_upload_file_id, -}; use serde_json::{Map, Value, json}; use std::future::Future; use std::pin::Pin; -use super::common_utils::{convert_document_url_to_data_uri, string_headers, truncate_error_body}; +use super::common_utils::{convert_document_url_to_data_uri, string_headers}; use super::types::{PreparedOcrRequest, ProviderOcrRequest}; -use crate::client::http_client; use crate::integrations::custom_guardrail::{ CustomGuardrailRunner, GuardrailContext, GuardrailError, GuardrailRequest, }; @@ -93,19 +89,7 @@ impl OcrLifecycleHooks { )?; let model = request.model.clone(); let custom_llm_provider = request.custom_llm_provider.clone(); - let is_reducto = custom_llm_provider == "reducto"; - let document = if is_reducto { - let guarded_document = self - .run_during_call_guardrails(&model, &custom_llm_provider, &url, request.document) - .await?; - upload_reducto_document( - &guarded_document, - request.api_base.as_deref(), - request.timeout, - &upstream_headers, - ) - .await? - } else if config.requires_data_uri_document() { + let document = if config.requires_data_uri_document() { convert_document_url_to_data_uri(request.document).await? } else { request.document @@ -114,12 +98,9 @@ impl OcrLifecycleHooks { let body = config .transform_ocr_request(&request.model, document, optional_params.clone())? .data; - let body = if is_reducto { - body - } else { - self.run_during_call_guardrails(&model, &custom_llm_provider, &url, body) - .await? - }; + let body = self + .run_during_call_guardrails(&model, &custom_llm_provider, &url, body) + .await?; Ok(ProviderOcrRequest { model, config, @@ -186,63 +167,6 @@ impl OcrLifecycleHooks { } } -async fn upload_reducto_document( - document: &Value, - api_base: Option<&str>, - timeout: Option, - upstream_headers: &[(String, String)], -) -> Result { - let source = extract_document_source(document)?; - let Some(authorization) = upstream_headers - .iter() - .find(|(name, _)| name.eq_ignore_ascii_case("authorization")) - .map(|(_, value)| value.as_str()) - else { - return Err(Error::Auth( - "Reducto upload requires an Authorization header".to_string(), - )); - }; - let Some(upload) = build_upload_request(source, authorization, api_base) else { - return Ok(document.clone()); - }; - let part = reqwest::multipart::Part::bytes(upload.bytes) - .file_name(upload.file_name) - .mime_str(&upload.mime_type) - .map_err(|error| Error::InvalidRequest(error.to_string()))?; - let form = reqwest::multipart::Form::new().part("file", part); - let mut request_builder = http_client().post(upload.url).multipart(form); - for (name, value) in upstream_headers { - if !name.eq_ignore_ascii_case("content-type") - && !name.eq_ignore_ascii_case("content-length") - { - request_builder = request_builder.header(name, value); - } - } - if let Some(timeout) = timeout { - request_builder = request_builder.timeout(timeout); - } - let response = request_builder - .send() - .await - .map_err(|error| Error::Network(error.to_string()))?; - let status = response.status(); - let body = response - .text() - .await - .map_err(|error| Error::Network(error.to_string()))?; - if !status.is_success() { - return Err(Error::Http { - status: status.as_u16(), - body: truncate_error_body(&body), - }); - } - let response_json: Value = serde_json::from_str(&body).map_err(|error| { - Error::InvalidResponse(format!("invalid Reducto upload response JSON: {error}")) - })?; - let file_id = extract_upload_file_id(&response_json)?; - Ok(json!({"type": "document_url", "document_url": file_id})) -} - impl CallLifecycleHooks for OcrLifecycleHooks { type PreCallFuture<'a> = OcrFuture<'a, PreparedOcrRequest>; type DuringCallFuture<'a> = OcrFuture<'a, PreparedOcrRequest>; @@ -390,7 +314,8 @@ fn core_error_kind(error: &Error) -> &'static str { | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials | Error::MissingAzureAiCredentialsOrAdToken - | Error::MissingAzureDocumentIntelligenceCredentials => "AuthError", + | Error::MissingAzureDocumentIntelligenceCredentials + | Error::MissingReductoApiKey => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", Error::InvalidRequest(_) => "InvalidRequest", Error::InvalidType { .. } => "InvalidType", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs index 2cdc9f9a714..cdf4a15125f 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs @@ -25,162 +25,16 @@ pub async fn ocr(request: OcrRequest<'_>) -> Result { #[cfg(test)] mod tests { - use serde_json::{Map, json}; - use tokio::io::{AsyncReadExt, AsyncWriteExt}; - use tokio::net::{TcpListener, TcpStream}; - - use super::{OcrRequest, ocr}; - use crate::integrations::types::RequestMetadata; use litellm_core::ocr::wire::is_supported_request; #[test] - fn core_activation_includes_azure_document_intelligence() { + fn core_activation_includes_migrated_providers() { assert!(is_supported_request("model", Some("mistral"))); assert!(is_supported_request("pixtral-12b", Some("azure_ai"))); assert!(is_supported_request( "doc-intelligence/prebuilt-layout", Some("azure_ai") )); - assert!(!is_supported_request("parse-v3", Some("reducto"))); - } - - async fn read_http_request(socket: &mut TcpStream) -> String { - let mut request = Vec::new(); - let mut buffer = [0_u8; 1024]; - let header_end = loop { - let n = socket.read(&mut buffer).await.expect("reads request"); - if n == 0 { - break request.len(); - } - request.extend_from_slice(&buffer[..n]); - if let Some(position) = request.windows(4).position(|window| window == b"\r\n\r\n") { - break position + 4; - } - }; - let headers = String::from_utf8_lossy(&request[..header_end]); - let content_length = headers - .lines() - .find_map(|line| { - let (name, value) = line.split_once(':')?; - name.eq_ignore_ascii_case("content-length") - .then(|| value.trim().parse::().ok()) - .flatten() - }) - .unwrap_or(0); - while request.len().saturating_sub(header_end) < content_length { - let n = socket.read(&mut buffer).await.expect("reads body"); - if n == 0 { - break; - } - request.extend_from_slice(&buffer[..n]); - } - String::from_utf8(request).expect("request is utf8") - } - - fn base_ocr_request(model: &str) -> OcrRequest<'_> { - OcrRequest { - model, - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-test"), - api_base: None, - custom_llm_provider: None, - extra_headers: None, - optional_params: Map::new(), - timeout: None, - callbacks: Vec::new(), - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: None, - } - } - - #[tokio::test] - async fn reducto_file_upload_then_parse_maps_response() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let address = listener.local_addr().expect("listener has local address"); - let server = tokio::spawn(async move { - let (mut upload_socket, _) = listener.accept().await.expect("accepts upload request"); - let upload_request = read_http_request(&mut upload_socket).await; - let upload_body = r#"{"file_id":"reducto://uploaded.pdf"}"#; - let upload_response = format!( - "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - upload_body.len(), - upload_body - ); - upload_socket - .write_all(upload_response.as_bytes()) - .await - .expect("writes upload response"); - - let (mut parse_socket, _) = listener.accept().await.expect("accepts parse request"); - let parse_request = read_http_request(&mut parse_socket).await; - let parse_body = r#"{"job_id":"job_123","usage":{"num_pages":3,"credits":3},"result":{"chunks":[{"content":"Page 1 block A","blocks":[{"content":"Page 1 block A","bbox":{"page":1},"kind":"text"}]},{"content":"Page 2 block A","blocks":[{"content":"Page 2 block A","bbox":{"page":2},"kind":"table"}]},{"content":"Page 1 block B","blocks":[{"content":"Page 1 block B","bbox":{"page":1},"kind":"text"}]},{"content":"Page 3 block A","blocks":[{"content":"Page 3 block A","bbox":{"page":3},"kind":"figure"}]}]}}"#; - let parse_response = format!( - "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - parse_body.len(), - parse_body - ); - parse_socket - .write_all(parse_response.as_bytes()) - .await - .expect("writes parse response"); - (upload_request, parse_request) - }); - let api_base = format!("http://{address}"); - let mut request = base_ocr_request("reducto/parse-v3"); - request.api_base = Some(&api_base); - request.api_key = None; - request.extra_headers = Some(Map::from_iter([ - ("Authorization".to_string(), json!("Bearer test-key")), - ("x-trace-id".to_string(), json!("trace-1")), - ])); - request.document = json!({ - "type": "document_url", - "document_url": "data:application/pdf;base64,JVBERi0xLjQ=" - }); - request.optional_params = Map::from_iter([ - ( - "formatting".to_string(), - json!({"table_output_format": "html"}), - ), - ("retrieval".to_string(), json!({"chunk_mode": "section"})), - ("settings".to_string(), json!({"ocr_system": "standard"})), - ]); - - let response = ocr(request).await.expect("Reducto OCR succeeds"); - - assert_eq!(response["pages"].as_array().map(Vec::len), Some(3)); - assert_eq!( - response["pages"][0]["markdown"], - "Page 1 block A\n\nPage 1 block B" - ); - assert_eq!(response["pages"][1]["markdown"], "Page 2 block A"); - assert_eq!(response["pages"][2]["markdown"], "Page 3 block A"); - assert_eq!(response["usage_info"]["pages_processed"], 3); - assert_eq!(response["usage_info"]["credits"], 3); - assert_eq!(response["provider_native_response"]["job_id"], "job_123"); - let (upload_request, parse_request) = server.await.expect("server task completes"); - assert!( - upload_request - .to_ascii_lowercase() - .contains("authorization: bearer test-key") - ); - assert!(upload_request.contains("application/pdf")); - assert!(upload_request.contains("%PDF-1.4")); - assert!(upload_request.contains("x-trace-id: trace-1")); - assert!( - parse_request - .to_ascii_lowercase() - .contains("authorization: bearer test-key") - ); - assert!(parse_request.contains(r#""input":"reducto://uploaded.pdf""#)); - assert!(parse_request.contains(r#""table_output_format":"html""#)); - assert!(parse_request.contains(r#""chunk_mode":"section""#)); - assert!(parse_request.contains(r#""ocr_system":"standard""#)); + assert!(is_supported_request("parse-v3", Some("reducto"))); } } diff --git a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs index a64ad1a1376..9707e9f2611 100644 --- a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs @@ -118,7 +118,8 @@ impl IntoResponse for MessagesRouteError { | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials | Error::MissingAzureAiCredentialsOrAdToken - | Error::MissingAzureDocumentIntelligenceCredentials => ( + | Error::MissingAzureDocumentIntelligenceCredentials + | Error::MissingReductoApiKey => ( StatusCode::BAD_GATEWAY, "messages provider request failed".to_string(), ), diff --git a/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs b/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs index 5502a467511..2fbd25d986f 100644 --- a/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs +++ b/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs @@ -158,15 +158,6 @@ impl RecordingOcrGuardrail { } } - fn blocking_during_call() -> Self { - Self { - hooks: vec![GuardrailEventHook::DuringCall], - events: Mutex::new(Vec::new()), - block_pre_call: false, - block_during_call: true, - } - } - fn events(&self) -> Vec<&'static str> { self.events.lock().unwrap().clone() } @@ -216,26 +207,6 @@ impl CustomGuardrail for RecordingOcrGuardrail { } } -fn base_ocr_request(model: &str) -> OcrRequest<'_> { - OcrRequest { - model, - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-test"), - api_base: None, - custom_llm_provider: None, - extra_headers: None, - optional_params: Map::new(), - timeout: None, - callbacks: Vec::new(), - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: None, - } -} - #[tokio::test] async fn azure_mistral_uses_prepared_authorization_through_gateway() { let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); @@ -287,66 +258,6 @@ async fn azure_mistral_uses_prepared_authorization_through_gateway() { ); } -#[tokio::test] -async fn reducto_during_call_guardrail_blocks_before_upload() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let address = listener.local_addr().expect("listener has local address"); - let api_base = format!("http://{address}"); - let guardrail = Arc::new(RecordingOcrGuardrail::blocking_during_call()); - let mut request = base_ocr_request("reducto/parse-v3"); - request.api_base = Some(&api_base); - request.document = json!({ - "type": "document_url", - "document_url": "data:application/pdf;base64,JVBERi0xLjQ=" - }); - request.guardrails = vec![guardrail.clone()]; - - let error = ocr(request).await.expect_err("guardrail blocks upload"); - - assert!(matches!(error, Error::InvalidRequest(_))); - assert_eq!(guardrail.events(), vec!["async_moderation_hook"]); - let accepted = tokio::time::timeout(Duration::from_millis(100), listener.accept()).await; - assert!(accepted.is_err(), "upload socket should not be touched"); -} - -#[tokio::test] -async fn reducto_upload_error_body_is_truncated() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let address = listener.local_addr().expect("listener has local address"); - let server = tokio::spawn(async move { - let (mut socket, _) = listener.accept().await.expect("accepts upload request"); - let _request = read_http_request(&mut socket).await; - let body = "x".repeat(300); - let response = format!( - "HTTP/1.1 500 Internal Server Error\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - body.len(), - body - ); - socket - .write_all(response.as_bytes()) - .await - .expect("writes upload response"); - }); - let api_base = format!("http://{address}"); - let mut request = base_ocr_request("reducto/parse-v3"); - request.api_base = Some(&api_base); - request.document = json!({ - "type": "document_url", - "document_url": "data:application/pdf;base64,JVBERi0xLjQ=" - }); - - let error = ocr(request).await.expect_err("upload should fail"); - - assert!( - matches!(error, Error::Http { status: 500, body } if body.chars().count() < 300 && body.ends_with("... (truncated)")) - ); - server.await.expect("server task completes"); -} - #[tokio::test] async fn ocr_lifecycle_runs_pre_during_and_success_hooks() { let listener = TcpListener::bind("127.0.0.1:0") diff --git a/litellm-rust/crates/core/src/constants.rs b/litellm-rust/crates/core/src/constants.rs index 73f2f5d284b..9469d379462 100644 --- a/litellm-rust/crates/core/src/constants.rs +++ b/litellm-rust/crates/core/src/constants.rs @@ -58,5 +58,8 @@ pub(crate) const AZURE_DI_SUBSCRIPTION_HEADER: &str = "Ocp-Apim-Subscription-Key pub(crate) const AZURE_DI_DEFAULT_DPI: i64 = 96; pub(crate) const AZURE_DI_DEFAULT_WIDTH: f64 = 8.5; pub(crate) const AZURE_DI_DEFAULT_HEIGHT: f64 = 11.0; +pub(crate) const REDUCTO_API_BASE: &str = "https://platform.reducto.ai"; +pub(crate) const REDUCTO_API_KEY_ENV: &str = "REDUCTO_API_KEY"; +pub(crate) const REDUCTO_ID_PREFIX: &str = "reducto://"; pub(crate) const AZURE_AI_OCR_PATH: &str = "/providers/mistral/azure/ocr"; pub(crate) const MISTRAL_OCR_API_BASE: &str = "https://api.mistral.ai/v1"; diff --git a/litellm-rust/crates/core/src/error.rs b/litellm-rust/crates/core/src/error.rs index bbeb23daa13..2a4cbad96c0 100644 --- a/litellm-rust/crates/core/src/error.rs +++ b/litellm-rust/crates/core/src/error.rs @@ -31,6 +31,10 @@ pub enum Error { "invalid authentication configuration: Missing Azure Document Intelligence credentials - set AZURE_DOCUMENT_INTELLIGENCE_API_KEY or configure Entra ID" )] MissingAzureDocumentIntelligenceCredentials, + #[error( + "Missing REDUCTO_API_KEY - set it in the environment or pass api_key to litellm.ocr()/litellm.aocr()" + )] + MissingReductoApiKey, #[error("upstream request failed with status {status}: {body}")] Http { status: u16, body: String }, #[error("upstream network error: {0}")] diff --git a/litellm-rust/crates/core/src/ocr/adapters/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/mod.rs index 530b28aeadd..c089ca90605 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/mod.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/mod.rs @@ -10,9 +10,11 @@ use super::wire::DecodedOcrResponse; mod azure; mod mistral; +mod reducto; pub(crate) use azure::{AzureDocumentIntelligenceAdapter, AzureMistralAdapter}; pub(crate) use mistral::MistralAdapter; +pub(crate) use reducto::{ReductoLegacyAdapter, ReductoV3Adapter}; /// Converts a complete LiteLLM OCR call to provider HTTP and normalizes its response. pub(crate) trait OcrAdapter: Send + Sync + Sized + 'static { @@ -66,6 +68,8 @@ macro_rules! for_each_ocr_adapter { Mistral, $crate::ocr::adapters::MistralAdapter, $crate::ocr::adapters::MistralAdapter, Mistral; AzureMistral, $crate::ocr::adapters::AzureMistralAdapter, $crate::ocr::adapters::AzureMistralAdapter, AzureAi; AzureDocumentIntelligence, $crate::ocr::adapters::AzureDocumentIntelligenceAdapter, $crate::ocr::adapters::AzureDocumentIntelligenceAdapter, AzureAi; + ReductoLegacy, $crate::ocr::adapters::ReductoLegacyAdapter, $crate::ocr::adapters::ReductoLegacyAdapter, Reducto; + ReductoV3, $crate::ocr::adapters::ReductoV3Adapter, $crate::ocr::adapters::ReductoV3Adapter, Reducto; } }; } diff --git a/litellm-rust/crates/core/src/ocr/adapters/reducto/legacy.rs b/litellm-rust/crates/core/src/ocr/adapters/reducto/legacy.rs new file mode 100644 index 00000000000..062a0071a34 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/reducto/legacy.rs @@ -0,0 +1,45 @@ +use super::super::OcrAdapter; +use crate::ocr::OcrClient; +use crate::ocr::codecs::reducto::{self, ReductoLegacyParams, ReductoResponse}; +use crate::ocr::error::{OcrError, OcrResponseError}; +use crate::ocr::prepare::{ + _prepare_ocr_request, ParsedProviderParams, build_http_request, credential_env, + guardrail_document, merge_extra_params, +}; +use crate::ocr::registry::OcrProvider; +use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse}; + +#[derive(Clone, Debug)] +pub(crate) struct ReductoLegacyAdapter; + +impl OcrAdapter for ReductoLegacyAdapter { + type ProviderResponse = ReductoResponse; + const PROVIDER: OcrProvider = OcrProvider::Reducto; + + async fn prepare_request( + &self, + request: &LiteLLMOcrRequest, + client: &OcrClient, + ) -> Result { + let ParsedProviderParams { + known: params, + extra_params, + } = _prepare_ocr_request::(request)?; + let headers = super::validate_environment(&request.connection, &credential_env)?; + let url = super::get_complete_url(request.connection.api_base.as_deref(), "parse")?; + let document = guardrail_document(request, &url).await?; + let document = + super::prepare_document(client, document, &request.connection, &headers).await?; + let body = reducto::transform_legacy_ocr_request(&request.model, document, ¶ms)?; + let body = merge_extra_params(&body, extra_params)?; + build_http_request(client, request, &url, &headers, &body) + } + + fn transform_ocr_response( + &self, + request: &LiteLLMOcrRequest, + response: Self::ProviderResponse, + ) -> Result { + reducto::transform_ocr_response(&request.model, response) + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/reducto/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/reducto/mod.rs new file mode 100644 index 00000000000..7621d0d326a --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/reducto/mod.rs @@ -0,0 +1,148 @@ +mod legacy; +mod v3; + +use crate::Error; +use crate::constants::{REDUCTO_API_BASE, REDUCTO_API_KEY_ENV, REDUCTO_ID_PREFIX}; +use crate::ocr::document::InlineDocument; +use crate::ocr::error::{OcrError, OcrRequestError, OcrResponseError}; +use crate::ocr::types::{OcrConnection, OcrDocument}; +use crate::url_utils::ApiUrl; + +pub(crate) use legacy::ReductoLegacyAdapter; +pub(crate) use v3::ReductoV3Adapter; + +pub(super) fn get_complete_url(api_base: Option<&str>, path: &str) -> Result { + let base = api_base + .map(str::trim) + .filter(|base| !base.is_empty()) + .unwrap_or(REDUCTO_API_BASE); + ApiUrl::parse(base) + .and_then(|url| url.complete_path(&[path])) + .map(|url| url.into_string()) + .map_err(|_| { + OcrRequestError::RequestField { + path: "api_base".into(), + } + .into() + }) +} + +pub(super) fn validate_environment( + connection: &OcrConnection, + env_lookup: &(dyn Fn(&str) -> Option + Sync), +) -> Result, OcrError> { + if crate::http_utils::has_header(&connection.extra_headers, "authorization") { + return Ok(connection.extra_headers.clone()); + } + let api_key = connection + .api_key + .as_deref() + .map(str::trim) + .filter(|key| !key.is_empty()) + .map(str::to_string) + .or_else(|| { + env_lookup(REDUCTO_API_KEY_ENV) + .map(|key| key.trim().to_string()) + .filter(|key| !key.is_empty()) + }) + .ok_or(Error::MissingReductoApiKey)?; + Ok( + std::iter::once(("Authorization".into(), format!("Bearer {api_key}"))) + .chain(connection.extra_headers.clone()) + .collect(), + ) +} + +pub(super) async fn prepare_document( + client: &crate::ocr::OcrClient, + document: OcrDocument, + connection: &OcrConnection, + headers: &[(String, String)], +) -> Result { + if document.source().starts_with(REDUCTO_ID_PREFIX) { + if document.source()[REDUCTO_ID_PREFIX.len()..] + .trim() + .is_empty() + { + return Err(OcrRequestError::RequestField { + path: "document file id".into(), + } + .into()); + } + return Ok(document); + } + let inline = InlineDocument::parse(document.source())?.ok_or(OcrRequestError::ReductoSource)?; + let mime = inline.mime_type().to_string(); + let bytes = inline.decode(crate::constants::OCR_INLINE_MAX_BYTES)?; + let part = reqwest::multipart::Part::bytes(bytes) + .file_name("document") + .mime_str(&mime) + .map_err(|_| OcrRequestError::InvalidDataUri)?; + let builder = client + .provider_http() + .post(get_complete_url(connection.api_base.as_deref(), "upload")?) + .multipart(reqwest::multipart::Form::new().part("file", part)) + .timeout(connection.timeout); + let builder = crate::http_utils::with_headers( + builder, + headers, + crate::http_utils::HeaderPolicy::Except(&["content-type", "content-length"]), + ); + let response = crate::http_utils::http_request(builder) + .await + .map_err(crate::error::TransportError::from)?; + let uploaded = crate::ocr::client::read_json_response::< + crate::ocr::codecs::reducto::ReductoUploadResponse, + >(response, false) + .await? + .data; + let file_id = uploaded + .file_id + .as_deref() + .map(str::trim) + .filter(|id| !id.is_empty()); + let Some(file_id) = file_id else { + return Err(OcrResponseError::ResponseField { + path: "file_id".into(), + } + .into()); + }; + Ok(document.with_source(file_id.to_string())) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn explicit_key_precedes_environment_key() { + let connection = OcrConnection { + api_key: Some("passed-key".into()), + ..Default::default() + }; + let headers = validate_environment(&connection, &|_| Some("env-key".into())).unwrap(); + assert_eq!(headers[0].1, "Bearer passed-key"); + } + + #[test] + fn blank_explicit_key_uses_environment_key() { + let connection = OcrConnection { + api_key: Some(" ".into()), + ..Default::default() + }; + let headers = validate_environment(&connection, &|_| Some(" env-key ".into())).unwrap(); + assert_eq!(headers[0].1, "Bearer env-key"); + } + + #[test] + fn existing_authorization_skips_key_lookup() { + let connection = OcrConnection { + extra_headers: vec![("authorization".into(), "Bearer existing".into())], + ..Default::default() + }; + assert_eq!( + validate_environment(&connection, &|_| None).unwrap(), + connection.extra_headers + ); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/reducto/v3.rs b/litellm-rust/crates/core/src/ocr/adapters/reducto/v3.rs new file mode 100644 index 00000000000..a49f8105e26 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/reducto/v3.rs @@ -0,0 +1,45 @@ +use super::super::OcrAdapter; +use crate::ocr::OcrClient; +use crate::ocr::codecs::reducto::{self, ReductoResponse, ReductoV3Params}; +use crate::ocr::error::{OcrError, OcrResponseError}; +use crate::ocr::prepare::{ + _prepare_ocr_request, ParsedProviderParams, build_http_request, credential_env, + guardrail_document, merge_extra_params, +}; +use crate::ocr::registry::OcrProvider; +use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse}; + +#[derive(Clone, Debug)] +pub(crate) struct ReductoV3Adapter; + +impl OcrAdapter for ReductoV3Adapter { + type ProviderResponse = ReductoResponse; + const PROVIDER: OcrProvider = OcrProvider::Reducto; + + async fn prepare_request( + &self, + request: &LiteLLMOcrRequest, + client: &OcrClient, + ) -> Result { + let ParsedProviderParams { + known: params, + extra_params, + } = _prepare_ocr_request::(request)?; + let headers = super::validate_environment(&request.connection, &credential_env)?; + let url = super::get_complete_url(request.connection.api_base.as_deref(), "parse")?; + let document = guardrail_document(request, &url).await?; + let document = + super::prepare_document(client, document, &request.connection, &headers).await?; + let body = reducto::transform_v3_ocr_request(&request.model, document, ¶ms)?; + let body = merge_extra_params(&body, extra_params)?; + build_http_request(client, request, &url, &headers, &body) + } + + fn transform_ocr_response( + &self, + request: &LiteLLMOcrRequest, + response: Self::ProviderResponse, + ) -> Result { + reducto::transform_ocr_response(&request.model, response) + } +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/mod.rs b/litellm-rust/crates/core/src/ocr/codecs/mod.rs index ef525e3692f..79dcd150f5f 100644 --- a/litellm-rust/crates/core/src/ocr/codecs/mod.rs +++ b/litellm-rust/crates/core/src/ocr/codecs/mod.rs @@ -1,2 +1,3 @@ pub(crate) mod document_intelligence; pub(crate) mod mistral; +pub(crate) mod reducto; diff --git a/litellm-rust/crates/core/src/ocr/codecs/reducto/mod.rs b/litellm-rust/crates/core/src/ocr/codecs/reducto/mod.rs new file mode 100644 index 00000000000..3fff40451c6 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/reducto/mod.rs @@ -0,0 +1,9 @@ +mod transformation; +mod types; + +pub(crate) use transformation::{ + transform_legacy_ocr_request, transform_ocr_response, transform_v3_ocr_request, +}; +pub(crate) use types::{ + ReductoLegacyParams, ReductoResponse, ReductoUploadResponse, ReductoV3Params, +}; diff --git a/litellm-rust/crates/core/src/ocr/codecs/reducto/transformation.rs b/litellm-rust/crates/core/src/ocr/codecs/reducto/transformation.rs new file mode 100644 index 00000000000..7073643f6b6 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/reducto/transformation.rs @@ -0,0 +1,115 @@ +use std::collections::BTreeMap; + +use serde_json::{Value, json}; + +use super::types::*; +use crate::ocr::error::{OcrRequestError, OcrResponseError}; +use crate::ocr::types::{LiteLLMOcrResponse, OcrDocument}; + +#[tracing::instrument( + name = "transform_ocr_request", + target = "litellm::function_trace", + level = "trace", + skip_all +)] +pub(crate) fn transform_v3_ocr_request( + _model: &str, + document: OcrDocument, + params: &ReductoV3Params, +) -> Result { + Ok(ReductoV3Request { + input: document.source().to_string(), + params: params.clone(), + }) +} + +#[tracing::instrument( + name = "transform_ocr_request", + target = "litellm::function_trace", + level = "trace", + skip_all +)] +pub(crate) fn transform_legacy_ocr_request( + _model: &str, + document: OcrDocument, + params: &ReductoLegacyParams, +) -> Result { + Ok(ReductoLegacyRequest { + document_url: document.source().to_string(), + options: params.enhance.as_ref().map(|_| params.clone()), + }) +} + +pub(crate) fn transform_ocr_response( + model: &str, + response: ReductoResponse, +) -> Result { + let result = match response.result { + Some(result) => result.unwrap_or_default(), + None => ReductoResult { + chunks: response.chunks, + }, + }; + let usage = response.usage.unwrap_or_default(); + Ok(LiteLLMOcrResponse { + pages: build_pages(result.chunks.unwrap_or_default()), + model: model.to_string(), + document_annotation: None, + usage_info: Some(json!({ + "pages_processed": usage.num_pages, + "credits": usage.credits, + })), + object: "ocr".to_string(), + extra_fields: serde_json::Map::new(), + provider_native_response: None, + }) +} + +fn build_pages(chunks: Vec) -> Vec { + let blocks_by_page = chunks + .iter() + .flat_map(|chunk| chunk.blocks.iter().flatten()) + .filter_map(|block| block.bbox.as_ref()?.page.map(|page| (page, block))) + .fold( + BTreeMap::>::new(), + |mut pages, (page, block)| { + pages.entry(page).or_default().push(block); + pages + }, + ); + if blocks_by_page.is_empty() { + let markdown = join_content(chunks.iter().map(|chunk| chunk.content.as_deref())); + return if markdown.is_empty() { + Vec::new() + } else { + vec![page(0, markdown, None)] + }; + } + blocks_by_page + .into_iter() + .map(|(index, blocks)| { + let markdown = join_content(blocks.iter().map(|block| block.content.as_deref())); + page( + index.saturating_sub(1).max(0), + markdown, + Some(json!(blocks)), + ) + }) + .collect() +} + +fn join_content<'a>(content: impl Iterator>) -> String { + content + .flatten() + .filter(|text| !text.is_empty()) + .collect::>() + .join("\n\n") +} + +fn page(index: i64, markdown: String, blocks: Option) -> Value { + let mut result = json!({"index":index,"markdown":markdown,"images":null}); + if let (Value::Object(fields), Some(blocks)) = (&mut result, blocks) { + fields.insert("blocks".into(), blocks); + } + result +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/reducto/types.rs b/litellm-rust/crates/core/src/ocr/codecs/reducto/types.rs new file mode 100644 index 00000000000..c03720cc8ae --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/reducto/types.rs @@ -0,0 +1,128 @@ +use serde::{Deserialize, Deserializer, Serialize}; +use serde_json::{Map, Value}; + +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub(crate) struct ReductoV3Params { + #[serde(skip_serializing_if = "Option::is_none")] + pub formatting: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub retrieval: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub settings: Option>, +} + +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub(crate) struct ReductoLegacyParams { + #[serde(skip_serializing_if = "Option::is_none")] + pub enhance: Option>, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct ReductoV3Request { + pub input: String, + #[serde(flatten)] + pub params: ReductoV3Params, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct ReductoLegacyRequest { + pub document_url: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub options: Option, +} + +#[derive(Deserialize)] +pub(crate) struct ReductoUploadResponse { + pub file_id: Option, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct ReductoResponse { + #[serde(default, deserialize_with = "present_nullable")] + pub result: Option>, + pub usage: Option, + #[serde(default)] + pub chunks: Option>, +} + +fn present_nullable<'de, D: Deserializer<'de>, T: Deserialize<'de>>( + deserializer: D, +) -> Result>, D::Error> { + Option::::deserialize(deserializer).map(Some) +} + +#[derive(Clone, Debug, Default, Deserialize)] +pub(crate) struct ReductoResult { + pub chunks: Option>, +} + +#[derive(Clone, Debug, Default, Deserialize)] +pub(crate) struct ReductoUsage { + #[serde(default, deserialize_with = "optional_i64")] + pub num_pages: Option, + #[serde(default, deserialize_with = "optional_f64")] + pub credits: Option, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct ReductoChunk { + pub content: Option, + pub blocks: Option>, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct ReductoBlock { + #[serde(skip_serializing_if = "Option::is_none")] + pub content: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub bbox: Option, + #[serde(flatten)] + pub extra_fields: Map, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct ReductoBoundingBox { + #[serde(default, deserialize_with = "optional_i64")] + pub page: Option, + #[serde(flatten)] + pub extra_fields: Map, +} + +fn optional_i64<'de, D: Deserializer<'de>>(deserializer: D) -> Result, D::Error> { + match Option::::deserialize(deserializer)? { + None | Some(Value::Null) => Ok(None), + Some(Value::Number(number)) => number + .as_i64() + .or_else(|| number.as_f64().and_then(checked_truncated_i64)) + .map(Some) + .ok_or_else(|| serde::de::Error::custom("expected an integer")), + Some(Value::String(value)) => value + .trim() + .parse::() + .map(Some) + .map_err(|_| serde::de::Error::custom("expected an integer")), + Some(Value::Bool(value)) => Ok(Some(i64::from(value))), + Some(_) => Ok(None), + } +} + +fn optional_f64<'de, D: Deserializer<'de>>(deserializer: D) -> Result, D::Error> { + match Option::::deserialize(deserializer)? { + None | Some(Value::Null) => Ok(None), + Some(Value::Number(number)) => number + .as_f64() + .map(Some) + .ok_or_else(|| serde::de::Error::custom("expected a number")), + Some(Value::String(value)) => value + .trim() + .parse::() + .map(Some) + .map_err(|_| serde::de::Error::custom("expected a number")), + Some(_) => Ok(None), + } +} + +fn checked_truncated_i64(value: f64) -> Option { + (value.is_finite() && value >= i64::MIN as f64 && value <= i64::MAX as f64) + .then(|| value.trunc() as i64) +} diff --git a/litellm-rust/crates/core/src/ocr/document.rs b/litellm-rust/crates/core/src/ocr/document.rs index 11a6e612a3c..e89b1c5c569 100644 --- a/litellm-rust/crates/core/src/ocr/document.rs +++ b/litellm-rust/crates/core/src/ocr/document.rs @@ -1,5 +1,4 @@ use base64::{Engine, engine::general_purpose::STANDARD}; -#[cfg(test)] use data_url::mime::Mime; use data_url::{DataUrl, DataUrlError, forgiving_base64::DecodeError}; use reqwest::Url; @@ -21,7 +20,6 @@ impl<'a> InlineDocument<'a> { } } - #[cfg(test)] pub(crate) fn mime_type(&self) -> &Mime { self.0.mime_type() } diff --git a/litellm-rust/crates/core/src/ocr/error.rs b/litellm-rust/crates/core/src/ocr/error.rs index 2278c6ba948..f42ac2ceb18 100644 --- a/litellm-rust/crates/core/src/ocr/error.rs +++ b/litellm-rust/crates/core/src/ocr/error.rs @@ -12,6 +12,8 @@ pub enum OcrRequestError { MissingField(&'static str), #[error("invalid OCR document data URI")] InvalidDataUri, + #[error("Reducto requires a reducto:// id or a data URI")] + ReductoSource, #[error("inline OCR document exceeds the size limit")] InlineDocumentTooLarge, #[error("OCR document URL is blocked by network policy")] diff --git a/litellm-rust/crates/core/src/ocr/mod.rs b/litellm-rust/crates/core/src/ocr/mod.rs index f7cda0009b0..5f274186ed0 100644 --- a/litellm-rust/crates/core/src/ocr/mod.rs +++ b/litellm-rust/crates/core/src/ocr/mod.rs @@ -21,6 +21,9 @@ mod azure_ai_tests; #[path = "../../tests/azure_document_intelligence_ocr.rs"] mod azure_document_intelligence_tests; #[cfg(test)] +#[path = "../../tests/reducto_ocr.rs"] +mod reducto_tests; +#[cfg(test)] #[path = "../../tests/ocr/support.rs"] pub(crate) mod test_support; #[cfg(test)] diff --git a/litellm-rust/crates/core/src/ocr/prepare.rs b/litellm-rust/crates/core/src/ocr/prepare.rs index cff40b7de50..363f963a66c 100644 --- a/litellm-rust/crates/core/src/ocr/prepare.rs +++ b/litellm-rust/crates/core/src/ocr/prepare.rs @@ -4,7 +4,7 @@ use serde_json::{Map, Value}; use super::OcrClient; use super::error::{OcrError, OcrRequestError}; use super::hooks::OcrDuringCallRequest; -use super::types::LiteLLMOcrRequest; +use super::types::{LiteLLMOcrRequest, OcrDocument}; #[derive(Debug, Deserialize)] pub(crate) struct ParsedProviderParams { @@ -24,6 +24,39 @@ pub(crate) fn _prepare_ocr_request( ) } +pub(crate) fn merge_extra_params( + body: &B, + extra_params: Map, +) -> Result { + let Value::Object(fields) = + serde_json::to_value(body).map_err(|_| OcrRequestError::RequestField { + path: "body".into(), + })? + else { + return Err(OcrRequestError::RequestField { + path: "body".into(), + }); + }; + let extra_body = extra_params + .get("extra_body") + .and_then(Value::as_object) + .cloned() + .unwrap_or_default() + .into_iter() + .collect::>(); + Ok(Value::Object( + fields + .into_iter() + .chain( + extra_params + .into_iter() + .filter(|(name, _)| name != "extra_body"), + ) + .chain(extra_body) + .collect(), + )) +} + pub(crate) async fn transform_request_body( client: &OcrClient, request: &LiteLLMOcrRequest, @@ -77,6 +110,29 @@ pub(crate) fn build_http_request( .map_err(OcrError::from) } +pub(crate) async fn guardrail_document( + request: &LiteLLMOcrRequest, + url: &str, +) -> Result { + if !request.hooks.has_guardrails() { + return Ok(request.document.clone()); + } + let changed = request + .hooks + .during_call(OcrDuringCallRequest { + model: request.model.clone(), + custom_llm_provider: request.adapter.provider().as_str().into(), + url: url.into(), + body: serde_json::to_value(&request.document).map_err(|_| { + OcrRequestError::RequestField { + path: "document".into(), + } + })?, + }) + .await?; + super::wire::decode_request_value(changed.body, "guardrail.document").map_err(OcrError::from) +} + #[derive(Serialize)] struct OcrWireBody { #[serde(flatten)] diff --git a/litellm-rust/crates/core/src/ocr/registry.rs b/litellm-rust/crates/core/src/ocr/registry.rs index 676c54579bc..6b100795fc8 100644 --- a/litellm-rust/crates/core/src/ocr/registry.rs +++ b/litellm-rust/crates/core/src/ocr/registry.rs @@ -25,6 +25,7 @@ super::adapters::for_each_ocr_adapter!(define_adapter_types); pub(crate) enum OcrProvider { Mistral, AzureAi, + Reducto, } impl OcrProvider { @@ -32,6 +33,7 @@ impl OcrProvider { match self { Self::Mistral => "mistral", Self::AzureAi => "azure_ai", + Self::Reducto => "reducto", } } } @@ -48,19 +50,72 @@ pub(crate) fn resolve_wire_adapter( let typed_provider = match provider.custom_llm_provider { "mistral" => OcrProvider::Mistral, "azure_ai" => OcrProvider::AzureAi, + "reducto" => OcrProvider::Reducto, value => return Err(Error::InvalidProvider(value.to_string())), }; - match typed_provider { - OcrProvider::Mistral => Ok((provider.model.to_string(), OcrAdapterKind::Mistral)), - OcrProvider::AzureAi if is_document_intelligence_model(provider.model) => Ok(( - provider.model.to_string(), - OcrAdapterKind::AzureDocumentIntelligence, - )), - OcrProvider::AzureAi => Ok((provider.model.to_string(), OcrAdapterKind::AzureMistral)), - } + let adapter = match typed_provider { + OcrProvider::Mistral => OcrAdapterKind::Mistral, + OcrProvider::AzureAi if is_document_intelligence_model(provider.model) => { + OcrAdapterKind::AzureDocumentIntelligence + } + OcrProvider::AzureAi => OcrAdapterKind::AzureMistral, + OcrProvider::Reducto if provider.model.eq_ignore_ascii_case("parse-legacy") => { + OcrAdapterKind::ReductoLegacy + } + OcrProvider::Reducto if provider.model.eq_ignore_ascii_case("parse-v3") => { + OcrAdapterKind::ReductoV3 + } + OcrProvider::Reducto => { + return Err(Error::InvalidRequest(format!( + "unsupported Reducto OCR model: {}", + provider.model + ))); + } + }; + Ok((provider.model.to_string(), adapter)) } fn is_document_intelligence_model(model: &str) -> bool { let model = model.to_ascii_lowercase(); model.contains("doc-intelligence") || model.contains("documentintelligence") } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn provider_models_are_preserved_without_a_local_allowlist() { + let cases = [ + ("mistral/future-ocr-model", OcrAdapterKind::Mistral), + ("azure_ai/future-ocr-model", OcrAdapterKind::AzureMistral), + ]; + + for (qualified_model, expected_adapter) in cases { + let expected_model = qualified_model.split_once('/').unwrap().1; + let (model, adapter) = resolve_wire_adapter(qualified_model, None).unwrap(); + assert_eq!(model, expected_model); + assert_eq!(adapter, expected_adapter); + } + } + + #[test] + fn unknown_reducto_models_are_rejected() { + assert!(matches!( + resolve_wire_adapter("reducto/future-parse-model", None), + Err(Error::InvalidRequest(_)) + )); + } + + #[test] + fn known_protocol_models_still_select_specialized_adapters() { + let (model, adapter) = resolve_wire_adapter("reducto/parse-legacy", None).unwrap(); + assert_eq!(model, "parse-legacy"); + assert_eq!(adapter, OcrAdapterKind::ReductoLegacy); + + let (model, adapter) = + resolve_wire_adapter("azure_ai/doc-intelligence/prebuilt-layout", None).unwrap(); + assert_eq!(model, "doc-intelligence/prebuilt-layout"); + assert_eq!(adapter, OcrAdapterKind::AzureDocumentIntelligence); + } +} diff --git a/litellm-rust/crates/core/src/providers/mod.rs b/litellm-rust/crates/core/src/providers/mod.rs index c0c2c69831b..805600d6dbe 100644 --- a/litellm-rust/crates/core/src/providers/mod.rs +++ b/litellm-rust/crates/core/src/providers/mod.rs @@ -4,5 +4,4 @@ pub mod azure_ai; pub mod bedrock; pub mod mistral; pub mod openai; -pub mod reducto; pub mod vertex_ai; diff --git a/litellm-rust/crates/core/src/providers/reducto/mod.rs b/litellm-rust/crates/core/src/providers/reducto/mod.rs deleted file mode 100644 index 3621ff6a2fd..00000000000 --- a/litellm-rust/crates/core/src/providers/reducto/mod.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod ocr; diff --git a/litellm-rust/crates/core/src/providers/reducto/ocr/mod.rs b/litellm-rust/crates/core/src/providers/reducto/ocr/mod.rs deleted file mode 100644 index 8acee8f770c..00000000000 --- a/litellm-rust/crates/core/src/providers/reducto/ocr/mod.rs +++ /dev/null @@ -1,4 +0,0 @@ -pub mod transformation; - -#[cfg(test)] -mod tests; diff --git a/litellm-rust/crates/core/src/providers/reducto/ocr/tests.rs b/litellm-rust/crates/core/src/providers/reducto/ocr/tests.rs deleted file mode 100644 index 2b66d058b5d..00000000000 --- a/litellm-rust/crates/core/src/providers/reducto/ocr/tests.rs +++ /dev/null @@ -1,202 +0,0 @@ -use rstest::{fixture, rstest}; -use serde_json::{Value, json}; - -use super::transformation::*; -use crate::ocr::transformation::OcrProviderConfig; - -#[fixture] -fn parse_response() -> Value { - json!({ - "job_id": "job_123", - "usage": {"num_pages": 3, "credits": 3}, - "result": { - "chunks": [ - { - "content": "Page 1 block A", - "blocks": [{ - "content": "Page 1 block A", - "bbox": {"page": 1}, - "kind": "text", - }], - }, - { - "content": "Page 2 block A", - "blocks": [{ - "content": "Page 2 block A", - "bbox": {"page": 2}, - "kind": "table", - }], - }, - { - "content": "Page 1 block B", - "blocks": [{ - "content": "Page 1 block B", - "bbox": {"page": 1}, - "kind": "text", - }], - }, - { - "content": "Page 3 block A", - "blocks": [{ - "content": "Page 3 block A", - "bbox": {"page": 3}, - "kind": "figure", - }], - }, - ], - }, - }) -} - -#[rstest] -fn test_parse_v3_file_upload_and_response_mapping(parse_response: Value) { - let source = classify_document_source("data:application/pdf;base64,JVBERi0xLjQ=") - .expect("PDF data URI should be valid"); - let upload = build_upload_request( - source, - "Bearer test-key", - Some("https://platform.reducto.ai"), - ) - .expect("data URI should require upload"); - assert_eq!(upload.url, "https://platform.reducto.ai/upload"); - assert_eq!(upload.authorization, "Bearer test-key"); - assert_eq!(upload.file_name, "document"); - assert_eq!(upload.mime_type, "application/pdf"); - assert_eq!(upload.bytes, b"%PDF-1.4"); - - let optional_params = json!({ - "formatting": {"table_output_format": "html"}, - "retrieval": {"chunk_mode": "section"}, - "settings": {"ocr_system": "standard"}, - }) - .as_object() - .expect("params should be an object") - .clone(); - let request = build_parse_v3_request("reducto://uploaded.pdf", optional_params); - assert_eq!( - request.data, - json!({ - "input": "reducto://uploaded.pdf", - "formatting": {"table_output_format": "html"}, - "retrieval": {"chunk_mode": "section"}, - "settings": {"ocr_system": "standard"}, - }) - ); - - let transformed = transform_reducto_response("parse-v3", parse_response.clone()) - .expect("response should transform"); - assert_eq!( - transformed.usage_info, - Some(json!({"pages_processed": 3, "credits": 3})) - ); - assert_eq!(transformed.pages.len(), 3); - assert_eq!( - transformed.pages[0], - json!({ - "index": 0, - "markdown": "Page 1 block A\n\nPage 1 block B", - "blocks": [ - {"content": "Page 1 block A", "bbox": {"page": 1}, "kind": "text"}, - {"content": "Page 1 block B", "bbox": {"page": 1}, "kind": "text"}, - ], - }) - ); - assert_eq!(transformed.pages[1]["markdown"], "Page 2 block A"); - assert_eq!(transformed.pages[2]["markdown"], "Page 3 block A"); - assert_eq!(transformed.provider_native_response, Some(parse_response)); -} - -#[rstest] -fn test_parse_v3_reducto_id_passthrough_skips_upload(parse_response: Value) { - let document = json!({ - "type": "document_url", - "document_url": "reducto://already-uploaded.pdf", - }); - let source = extract_document_source(&document).expect("Reducto ID should be valid"); - assert!(build_upload_request(source.clone(), "Bearer test-key", None).is_none()); - assert_eq!( - source, - ReductoDocumentSource::FileId("reducto://already-uploaded.pdf".to_string()) - ); - - let request = REDUCTO_PARSE_V3_CONFIG - .transform_ocr_request( - "parse-v3", - document, - json!({"retrieval": {"chunk_mode": "section"}}) - .as_object() - .expect("params should be object") - .clone(), - ) - .expect("direct ID should transform"); - assert_eq!(request.data["input"], "reducto://already-uploaded.pdf"); - assert_eq!(request.data["retrieval"]["chunk_mode"], "section"); - - let response = REDUCTO_PARSE_V3_CONFIG - .transform_ocr_response("parse-v3", parse_response) - .expect("response should transform"); - assert!( - response.pages[0]["markdown"] - .as_str() - .expect("markdown should be string") - .starts_with("Page 1 block A") - ); -} - -#[rstest] -fn test_parse_legacy_wraps_enhance_under_options() { - let request = build_parse_legacy_request( - "reducto://legacy.pdf", - json!({"enhance": {"agentic": [{"type": "table"}]}}) - .as_object() - .expect("params should be object"), - ); - assert_eq!( - request.data, - json!({ - "document_url": "reducto://legacy.pdf", - "options": {"enhance": {"agentic": [{"type": "table"}]}}, - }) - ); -} - -#[rstest] -fn test_parse_v3_image_data_uri_upload_uses_image_mime() { - let source = classify_document_source("data:image/png;base64,iVBORw0KGgo=") - .expect("PNG data URI should be valid"); - let upload = build_upload_request( - source, - "Bearer programmatic-key", - Some("https://custom.reducto.test/"), - ) - .expect("data URI should require upload"); - assert_eq!(upload.url, "https://custom.reducto.test/upload"); - assert_eq!(upload.authorization, "Bearer programmatic-key"); - assert_eq!(upload.mime_type, "image/png"); - assert_eq!(upload.bytes, b"\x89PNG\r\n\x1a\n"); -} - -#[rstest] -#[case::http("http://example.com/document.pdf")] -#[case::https("https://example.com/document.pdf")] -fn test_parse_v3_rejects_plain_http_urls(#[case] source: &str) { - let error = classify_document_source(source).expect_err("plain URL should be rejected"); - assert!(error.to_string().contains("upload the file first")); -} - -#[rstest] -fn test_parse_v3_uses_programmatic_api_key_over_env() { - let key = resolve_api_key(Some("passed-key"), &|_| Some("env-reducto-key".to_string())) - .expect("explicit key should resolve"); - assert_eq!(key, "passed-key"); - - let headers = REDUCTO_PARSE_V3_CONFIG - .validate_environment(Vec::new(), Some("passed-key"), &|_| { - Some("env-reducto-key".to_string()) - }) - .expect("headers should validate"); - assert_eq!( - headers, - vec![("Authorization".to_string(), "Bearer passed-key".to_string())] - ); -} diff --git a/litellm-rust/crates/core/src/providers/reducto/ocr/transformation.rs b/litellm-rust/crates/core/src/providers/reducto/ocr/transformation.rs deleted file mode 100644 index f025887846f..00000000000 --- a/litellm-rust/crates/core/src/providers/reducto/ocr/transformation.rs +++ /dev/null @@ -1,407 +0,0 @@ -use std::collections::BTreeMap; - -use base64::Engine; -use base64::engine::general_purpose::STANDARD as BASE64_STANDARD; -use serde_json::{Map, Value, json}; - -use crate::error::{Error, json_type_name}; -use crate::ocr::transformation::OcrProviderConfig; -use crate::ocr::types::{LiteLLMOcrResponse, OcrRequestData}; - -pub const REDUCTO_API_BASE: &str = "https://platform.reducto.ai"; -pub const REDUCTO_API_KEY_ENV: &str = "REDUCTO_API_KEY"; -pub const REDUCTO_ID_PREFIX: &str = "reducto://"; - -const PARSE_V3_SUPPORTED_OCR_PARAMS: &[&str] = &["formatting", "retrieval", "settings"]; -const PARSE_LEGACY_SUPPORTED_OCR_PARAMS: &[&str] = &["enhance"]; -const MISSING_KEY_MESSAGE: &str = "Missing REDUCTO_API_KEY - set it in the environment or pass api_key to litellm.ocr()/litellm.aocr()"; -const DATA_URI_UPLOAD_REQUIRED: &str = - "Reducto data URI upload must complete before OCR request transformation"; - -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum ReductoDocumentSource { - FileId(String), - Upload { bytes: Vec, mime_type: String }, -} - -#[derive(Clone, PartialEq, Eq)] -pub struct ReductoUploadRequest { - pub url: String, - pub authorization: String, - pub file_name: &'static str, - pub bytes: Vec, - pub mime_type: String, -} - -pub struct ReductoParseV3Config; -pub struct ReductoParseLegacyConfig; - -pub const REDUCTO_PARSE_V3_CONFIG: ReductoParseV3Config = ReductoParseV3Config; -pub const REDUCTO_PARSE_LEGACY_CONFIG: ReductoParseLegacyConfig = ReductoParseLegacyConfig; - -pub fn config_for_model(model: &str) -> Option<&'static dyn OcrProviderConfig> { - match model { - "parse-v3" => Some(&REDUCTO_PARSE_V3_CONFIG), - "parse-legacy" => Some(&REDUCTO_PARSE_LEGACY_CONFIG), - _ => None, - } -} - -pub fn normalize_api_base(api_base: Option<&str>) -> String { - api_base - .map(str::trim) - .filter(|base| !base.is_empty()) - .unwrap_or(REDUCTO_API_BASE) - .trim_end_matches('/') - .to_string() -} - -pub fn parse_url(api_base: Option<&str>) -> String { - format!("{}/parse", normalize_api_base(api_base)) -} - -pub fn upload_url(api_base: Option<&str>) -> String { - format!("{}/upload", normalize_api_base(api_base)) -} - -pub fn resolve_api_key( - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - api_key - .map(str::trim) - .filter(|key| !key.is_empty()) - .map(str::to_string) - .or_else(|| { - env_lookup(REDUCTO_API_KEY_ENV) - .map(|key| key.trim().to_string()) - .filter(|key| !key.is_empty()) - }) - .ok_or_else(|| Error::Auth(MISSING_KEY_MESSAGE.to_string())) -} - -pub fn extract_document_source(document: &Value) -> Result { - let document = document.as_object().ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(document), - })?; - let source = document - .get("document_url") - .and_then(Value::as_str) - .filter(|source| !source.is_empty()) - .or_else(|| document.get("image_url").and_then(Value::as_str)) - .ok_or_else(|| { - Error::InvalidRequest( - "Reducto expected OCR preprocessing to produce document_url or image_url" - .to_string(), - ) - })?; - classify_document_source(source) -} - -pub fn classify_document_source(source: &str) -> Result { - if source.starts_with(REDUCTO_ID_PREFIX) { - return Ok(ReductoDocumentSource::FileId(source.to_string())); - } - if source.starts_with("http://") || source.starts_with("https://") { - return Err(Error::InvalidRequest( - "Reducto requires type='file' (auto-uploaded) or a reducto:// id. Plain http(s) URLs are not supported; upload the file first." - .to_string(), - )); - } - if !source.starts_with("data:") { - return Err(Error::InvalidRequest( - "Reducto requires a reducto:// id or a base64 data URI after OCR preprocessing." - .to_string(), - )); - } - - let (header, encoded) = source - .split_once(',') - .ok_or_else(|| Error::InvalidRequest("Invalid Reducto data URI provided.".to_string()))?; - if !header.split(';').any(|part| part == "base64") { - return Err(Error::InvalidRequest( - "Reducto only supports base64-encoded data URIs.".to_string(), - )); - } - - let mime_type = header - .strip_prefix("data:") - .and_then(|header| header.split(';').next()) - .filter(|mime| !mime.is_empty()) - .unwrap_or("application/octet-stream") - .to_string(); - let bytes = BASE64_STANDARD.decode(encoded).map_err(|_| { - Error::InvalidRequest("Invalid Reducto base64 payload provided.".to_string()) - })?; - - Ok(ReductoDocumentSource::Upload { bytes, mime_type }) -} - -pub fn build_upload_request( - source: ReductoDocumentSource, - authorization: &str, - api_base: Option<&str>, -) -> Option { - let ReductoDocumentSource::Upload { bytes, mime_type } = source else { - return None; - }; - - Some(ReductoUploadRequest { - url: upload_url(api_base), - authorization: authorization.to_string(), - file_name: "document", - bytes, - mime_type, - }) -} - -pub fn extract_upload_file_id(response_json: &Value) -> Result<&str, Error> { - response_json - .as_object() - .and_then(|response| response.get("file_id")) - .and_then(Value::as_str) - .filter(|file_id| !file_id.is_empty()) - .ok_or_else(|| { - Error::InvalidResponse(format!( - "Reducto /upload returned 200 without a file_id; got payload={response_json}" - )) - }) -} - -pub fn build_parse_v3_request( - file_id: &str, - optional_params: Map, -) -> OcrRequestData { - let data = std::iter::once(("input".to_string(), Value::String(file_id.to_string()))) - .chain(optional_params) - .collect(); - OcrRequestData { - data: Value::Object(data), - files: None, - } -} - -pub fn build_parse_legacy_request( - file_id: &str, - optional_params: &Map, -) -> OcrRequestData { - let options = optional_params - .get("enhance") - .filter(|enhance| !enhance.is_null()) - .map(|enhance| json!({"options": {"enhance": enhance}})); - let data = match options { - Some(Value::Object(options)) => std::iter::once(( - "document_url".to_string(), - Value::String(file_id.to_string()), - )) - .chain(options) - .collect(), - _ => Map::from_iter([( - "document_url".to_string(), - Value::String(file_id.to_string()), - )]), - }; - OcrRequestData { - data: Value::Object(data), - files: None, - } -} - -fn source_file_id(document: &Value) -> Result { - match extract_document_source(document)? { - ReductoDocumentSource::FileId(file_id) => Ok(file_id), - ReductoDocumentSource::Upload { .. } => Err(Error::Unsupported(DATA_URI_UPLOAD_REQUIRED)), - } -} - -fn page_number(block: &Map) -> Option { - let page = block.get("bbox")?.as_object()?.get("page")?; - page.as_i64() - .or_else(|| page.as_u64().and_then(|page| i64::try_from(page).ok())) - .or_else(|| page.as_str().and_then(|page| page.parse().ok())) -} - -fn chunks(result: &Map) -> &[Value] { - result - .get("chunks") - .and_then(Value::as_array) - .map(Vec::as_slice) - .unwrap_or_default() -} - -fn build_pages(result: &Map) -> Vec { - let blocks_by_page = chunks(result) - .iter() - .filter_map(Value::as_object) - .filter_map(|chunk| chunk.get("blocks").and_then(Value::as_array)) - .flatten() - .filter_map(|block| block.as_object().map(|object| (block, object))) - .filter_map(|(block, object)| page_number(object).map(|page| (page, block.clone()))) - .fold( - BTreeMap::>::new(), - |mut pages, (page, block)| { - pages.entry(page).or_default().push(block); - pages - }, - ); - - if blocks_by_page.is_empty() { - let markdown = chunks(result) - .iter() - .filter_map(Value::as_object) - .filter_map(|chunk| chunk.get("content").and_then(Value::as_str)) - .filter(|content| !content.is_empty()) - .collect::>() - .join("\n\n"); - return if markdown.is_empty() { - Vec::new() - } else { - vec![json!({"index": 0, "markdown": markdown})] - }; - } - - blocks_by_page - .into_iter() - .map(|(page, blocks)| { - let markdown = blocks - .iter() - .filter_map(Value::as_object) - .filter_map(|block| block.get("content").and_then(Value::as_str)) - .filter(|content| !content.is_empty()) - .collect::>() - .join("\n\n"); - json!({ - "index": page.saturating_sub(1).max(0), - "markdown": markdown, - "blocks": blocks, - }) - }) - .collect() -} - -pub fn transform_reducto_response( - model: &str, - response_json: Value, -) -> Result { - let response = response_json - .as_object() - .ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(&response_json), - })?; - let empty_result = Map::new(); - let result = match response.get("result") { - Some(Value::Object(result)) => result, - Some(Value::Null) => &empty_result, - Some(_) => { - return Err(Error::InvalidResponse( - "Reducto result must be an object".to_string(), - )); - } - None => response, - }; - let usage = response - .get("usage") - .and_then(Value::as_object) - .cloned() - .unwrap_or_default(); - let usage_info = Some(json!({ - "pages_processed": usage.get("num_pages").cloned().unwrap_or(Value::Null), - "credits": usage.get("credits").cloned().unwrap_or(Value::Null), - })); - - Ok(LiteLLMOcrResponse { - pages: build_pages(result), - model: model.to_string(), - document_annotation: None, - usage_info, - object: "ocr".to_string(), - extra_fields: Map::new(), - provider_native_response: Some(response_json), - }) -} - -impl OcrProviderConfig for ReductoParseV3Config { - fn supported_ocr_params(&self) -> &'static [&'static str] { - PARSE_V3_SUPPORTED_OCR_PARAMS - } - - fn transform_ocr_request( - &self, - _model: &str, - document: Value, - optional_params: Map, - ) -> Result { - let file_id = source_file_id(&document)?; - Ok(build_parse_v3_request(&file_id, optional_params)) - } - - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - transform_reducto_response(model, response_json) - } - - fn complete_url( - &self, - api_base: Option<&str>, - _model: &str, - _optional_params: &Map, - _env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - Ok(parse_url(api_base)) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_api_key(api_key, env_lookup) - } -} - -impl OcrProviderConfig for ReductoParseLegacyConfig { - fn supported_ocr_params(&self) -> &'static [&'static str] { - PARSE_LEGACY_SUPPORTED_OCR_PARAMS - } - - fn transform_ocr_request( - &self, - _model: &str, - document: Value, - optional_params: Map, - ) -> Result { - let file_id = source_file_id(&document)?; - Ok(build_parse_legacy_request(&file_id, &optional_params)) - } - - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - transform_reducto_response(model, response_json) - } - - fn complete_url( - &self, - api_base: Option<&str>, - _model: &str, - _optional_params: &Map, - _env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - Ok(parse_url(api_base)) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_api_key(api_key, env_lookup) - } -} diff --git a/litellm-rust/crates/core/tests/reducto_ocr.rs b/litellm-rust/crates/core/tests/reducto_ocr.rs new file mode 100644 index 00000000000..8e86e4713ef --- /dev/null +++ b/litellm-rust/crates/core/tests/reducto_ocr.rs @@ -0,0 +1,232 @@ +use std::sync::Arc; + +use rstest::rstest; +use serde_json::{Value, json}; + +use super::hooks::{OcrDuringCallRequest, OcrHookFuture, OcrHooks}; +use super::test_support::{MockResponse, mock_server, perform_ocr, wire_request}; + +fn request_body(request: &str) -> Value { + serde_json::from_str(request.split_once("\r\n\r\n").unwrap().1).unwrap() +} + +#[rstest] +#[case( + "reducto/parse-v3", + json!({ + "formatting":{"table_output_format":"html"}, + "retrieval":{"chunk_mode":"section"}, + "settings":{"ocr_system":"standard"}, + "future_ocr_option":true, + "extra_body":{"provider_option":"value"} + }), + "reducto://already.pdf", + json!({ + "input":"reducto://already.pdf", + "formatting":{"table_output_format":"html"}, + "retrieval":{"chunk_mode":"section"}, + "settings":{"ocr_system":"standard"}, + "future_ocr_option":true, + "provider_option":"value" + }) +)] +#[case( + "reducto/parse-legacy", + json!({ + "enhance":{"agentic":[{"type":"table"}]}, + "future_ocr_option":true, + "extra_body":{"provider_option":"value"} + }), + "reducto://legacy.pdf", + json!({ + "document_url":"reducto://legacy.pdf", + "options":{"enhance":{"agentic":[{"type":"table"}]}}, + "future_ocr_option":true, + "provider_option":"value" + }) +)] +#[tokio::test] +async fn request_mapping_matches_python( + #[case] model: &str, + #[case] options: Value, + #[case] source: &str, + #[case] expected: Value, +) { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "result":{"chunks":[]} + }))]) + .await; + let mut request = wire_request(model, &base, options); + request.document = request.document.with_source(source.into()); + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(requests[0].starts_with("POST /parse ")); + assert_eq!(request_body(&requests[0]), expected); +} + +#[rstest] +#[case("parse-v3")] +#[case("parse-legacy")] +#[tokio::test] +async fn data_uri_upload_preserves_multipart_headers(#[case] model: &str) { + let (base, seen, server) = mock_server(vec![ + MockResponse::json(json!({"file_id":"reducto://uploaded.pdf"})), + MockResponse::json(json!({"result":{"chunks":[{"content":"hello"}]}})), + ]) + .await; + let mut request = wire_request(&format!("reducto/{model}"), &base, json!({})); + request.connection.extra_headers = vec![ + ("Content-Type".into(), "application/json".into()), + ("X-Trace".into(), "upload-test".into()), + ]; + + let response = perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert_eq!(response.pages[0]["markdown"], "hello"); + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 2); + assert!(requests[0].starts_with("POST /upload ")); + assert!( + requests[0] + .to_ascii_lowercase() + .contains("content-type: multipart/form-data; boundary=") + ); + assert!(requests[0].contains("x-trace: upload-test")); + assert!(requests[0].contains("application/pdf")); + assert!(requests[0].contains("abc")); + assert!(requests[1].starts_with("POST /parse ")); +} + +#[rstest] +#[case(json!({"file_id":""}))] +#[case(json!({}))] +#[case(json!({"file_id":null}))] +#[tokio::test] +async fn invalid_upload_ids_stop_before_parse(#[case] response: Value) { + let (base, seen, server) = mock_server(vec![MockResponse::json(response)]).await; + let error = perform_ocr(wire_request("reducto/parse-v3", &base, json!({}))) + .await + .unwrap_err(); + server.await.unwrap(); + assert!(error.to_string().contains("file_id")); + assert_eq!(seen.lock().unwrap().len(), 1); +} + +#[tokio::test] +async fn upload_failure_stops_before_parse() { + let (base, seen, server) = mock_server(vec![MockResponse { + status: 503, + headers: vec![], + body: json!({"error":"unavailable"}), + }]) + .await; + assert!( + perform_ocr(wire_request("reducto/parse-v3", &base, json!({}))) + .await + .is_err() + ); + server.await.unwrap(); + assert_eq!(seen.lock().unwrap().len(), 1); +} + +#[rstest] +#[case("https://example.com/a.pdf")] +#[case("reducto://")] +#[case("data:application/pdf;base64")] +#[case("data:application/pdf;base64,INVALID!")] +#[tokio::test] +async fn rejects_invalid_document_sources_before_network(#[case] source: &str) { + let mut request = wire_request("reducto/parse-v3", "http://127.0.0.1:1", json!({})); + request.document = request.document.with_source(source.into()); + assert!(perform_ocr(request).await.is_err()); +} + +#[test] +fn response_normalization_groups_blocks_and_distinguishes_null_result() { + use crate::ocr::codecs::reducto::{ReductoResponse, transform_ocr_response}; + + let raw = json!({"usage":{"num_pages":"2","credits":"3"},"result":{"chunks":[ + {"blocks":[{"content":"B","bbox":{"page":2},"kind":"table"}]}, + {"blocks":[{"content":"A","bbox":{"page":1},"kind":"text"},{"content":"C","bbox":{"page":1}}]} + ]}}); + let response: ReductoResponse = serde_json::from_value(raw).unwrap(); + let normalized = transform_ocr_response("parse-v3", response) + .unwrap() + .into_json(); + assert_eq!(normalized["pages"][0]["markdown"], "A\n\nC"); + assert_eq!(normalized["pages"][1]["markdown"], "B"); + assert_eq!(normalized["pages"][1]["blocks"][0]["kind"], "table"); + assert_eq!(normalized["usage_info"]["pages_processed"], 2); + assert_eq!(normalized["usage_info"]["credits"], 3.0); + + let missing: ReductoResponse = + serde_json::from_value(json!({"chunks":[{"content":"text"}]})).unwrap(); + let missing = transform_ocr_response("parse-v3", missing).unwrap(); + assert_eq!(missing.pages[0]["markdown"], "text"); + let null: ReductoResponse = serde_json::from_value( + json!({"result":null,"chunks":[{"content":"ignored"}],"usage":null}), + ) + .unwrap(); + let null = transform_ocr_response("parse-v3", null).unwrap(); + assert!(null.pages.is_empty()); +} + +#[tokio::test] +async fn facade_omits_native_response_by_default_and_preserves_auth_priority() { + let raw = json!({"job_id":"job-1","result":{"chunks":[]}}); + let (base, seen, server) = mock_server(vec![MockResponse::json(raw)]).await; + let mut request = wire_request("reducto/parse-v3", &base, json!({})); + request.document = request.document.with_source("reducto://ready.pdf".into()); + request.connection.extra_headers = vec![("authorization".into(), "Bearer existing".into())]; + + let response = perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert_eq!(response.provider_native_response, None); + assert!( + seen.lock().unwrap()[0] + .to_ascii_lowercase() + .contains("authorization: bearer existing") + ); +} + +struct RewriteDocument; + +impl OcrHooks for RewriteDocument { + fn has_guardrails(&self) -> bool { + true + } + + fn during_call( + &self, + request: OcrDuringCallRequest, + ) -> OcrHookFuture<'_, OcrDuringCallRequest> { + Box::pin(async move { + assert_eq!( + request.body["document_url"], + "data:application/pdf;base64,YWJj" + ); + Ok(OcrDuringCallRequest { + body: json!({"type":"document_url","document_url":"reducto://guarded.pdf"}), + ..request + }) + }) + } +} + +#[tokio::test] +async fn guardrail_rewrites_document_before_upload() { + let (base, seen, server) = + mock_server(vec![MockResponse::json(json!({"result":{"chunks":[]}}))]).await; + let mut request = wire_request("reducto/parse-v3", &base, json!({})); + request.hooks = Arc::new(RewriteDocument); + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(requests[0].starts_with("POST /parse ")); + assert!(requests[0].contains("reducto://guarded.pdf")); +} diff --git a/litellm-rust/crates/python-bridge/src/errors.rs b/litellm-rust/crates/python-bridge/src/errors.rs index 9ae25e790a6..864f30db6a9 100644 --- a/litellm-rust/crates/python-bridge/src/errors.rs +++ b/litellm-rust/crates/python-bridge/src/errors.rs @@ -45,6 +45,7 @@ pub(crate) fn chat_completions_error_to_pyerr(err: Error) -> PyErr { | Error::MissingAzureAiCredentials | Error::MissingAzureAiCredentialsOrAdToken | Error::MissingAzureDocumentIntelligenceCredentials + | Error::MissingReductoApiKey | Error::Routing(_) // Nothing reached the provider, so serving it on Python cannot double // bill and is the only way the caller gets an answer at all. diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index e960a354d9f..d1d4a70007d 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -102,13 +102,15 @@ mod tests { use litellm_core::ocr::wire::is_supported_request; #[test] - fn native_activation_includes_azure_document_intelligence() { + fn native_activation_includes_migrated_providers() { assert!(is_supported_request("model", Some("mistral"))); assert!(is_supported_request("pixtral-12b", Some("azure_ai"))); assert!(is_supported_request( "documentintelligence/prebuilt-read", Some("azure_ai") )); + assert!(is_supported_request("parse-v3", Some("reducto"))); + assert!(is_supported_request("parse-legacy", Some("reducto"))); assert!(!is_supported_request("mistral-ocr", Some("vertex_ai"))); } } diff --git a/litellm/utils.py b/litellm/utils.py index a765e1b1246..bb1bce66d9b 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -9400,11 +9400,9 @@ class ProviderConfigManager: ReductoParseV3Config, ) - if model == "parse-v3": - return ReductoParseV3Config() if model == "parse-legacy": return ReductoParseLegacyConfig() - return None + return ReductoParseV3Config() MistralOCRConfig: Final = litellm_utils.MistralOCRConfig PROVIDER_TO_CONFIG_MAP: Final = { diff --git a/tests/test_litellm/llms/reducto/test_parse_v3.py b/tests/test_litellm/llms/reducto/test_parse_v3.py index bacd12db58a..1d0c826ef8b 100644 --- a/tests/test_litellm/llms/reducto/test_parse_v3.py +++ b/tests/test_litellm/llms/reducto/test_parse_v3.py @@ -143,3 +143,28 @@ async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_tran assert parse_request_body["input"] == "reducto://already-uploaded.pdf" assert parse_request_body["retrieval"]["chunk_mode"] == "section" assert response.pages[0].markdown.startswith("Page 1 block A") + + +@pytest.mark.asyncio +async def test_unknown_model_uses_current_protocol_without_local_rejection( + disable_aiohttp_transport, respx_mock +): + parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond( + json=_reducto_parse_response() + ) + + response = await litellm.aocr( + model="reducto/future-parse-model", + document={ + "type": "document_url", + "document_url": "reducto://already-uploaded.pdf", + }, + api_key="test-key", + api_base="https://platform.reducto.ai", + ) + + assert parse_route.called + assert json.loads(parse_route.calls[0].request.read()) == { + "input": "reducto://already-uploaded.pdf" + } + assert response.model == "future-parse-model" From b8928170e93b4937857e947e29224ba1dc30880e Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 16:22:56 -0700 Subject: [PATCH 130/157] feat(ocr): add Vertex Mistral adapter (#40507) * feat(ocr): add Vertex Mistral adapter * test(ocr): validate Vertex credentials at adapter boundary * refactor(ocr): preserve Vertex Mistral extra params * refactor(ocr): align Vertex authentication lifecycle * refactor(ocr): keep Vertex preparation behind bridge * fix(ocr): protect Vertex credential destinations * fix(auth): restrict request Vertex token endpoints --- litellm-rust/Cargo.lock | 138 ++++ litellm-rust/Cargo.toml | 1 + .../crates/ai-gateway/src/ocr/common_utils.rs | 6 +- litellm-rust/crates/ai-gateway/src/ocr/mod.rs | 2 + litellm-rust/crates/core/Cargo.toml | 1 + litellm-rust/crates/core/src/auth/error.rs | 8 + litellm-rust/crates/core/src/auth/mod.rs | 1 + litellm-rust/crates/core/src/auth/vertex.rs | 592 ++++++++++++++++++ .../crates/core/src/ocr/adapters/mod.rs | 3 + .../core/src/ocr/adapters/vertex/mistral.rs | 154 +++++ .../core/src/ocr/adapters/vertex/mod.rs | 19 + litellm-rust/crates/core/src/ocr/client.rs | 8 + litellm-rust/crates/core/src/ocr/mod.rs | 3 + litellm-rust/crates/core/src/ocr/registry.rs | 7 + .../providers/vertex_ai/ocr/transformation.rs | 106 ---- .../crates/core/tests/vertex_ai_ocr.rs | 110 ++++ .../crates/python-bridge/src/routes/ocr.rs | 3 +- litellm/rust_bridge/ocr.py | 11 +- tests/test_litellm/ocr/test_rust_bridge.py | 2 + 19 files changed, 1063 insertions(+), 112 deletions(-) create mode 100644 litellm-rust/crates/core/src/auth/vertex.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/vertex/mistral.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs create mode 100644 litellm-rust/crates/core/tests/vertex_ai_ocr.rs diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock index 9c0a8cb7fe7..cf000e85d68 100644 --- a/litellm-rust/Cargo.lock +++ b/litellm-rust/Cargo.lock @@ -40,6 +40,15 @@ dependencies = [ "cc", ] +[[package]] +name = "android_system_properties" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" +dependencies = [ + "libc", +] + [[package]] name = "anes" version = "0.1.6" @@ -693,6 +702,20 @@ dependencies = [ "rand_core 0.10.1", ] +[[package]] +name = "chrono" +version = "0.4.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" +dependencies = [ + "iana-time-zone", + "js-sys", + "num-traits", + "serde", + "wasm-bindgen", + "windows-link", +] + [[package]] name = "ciborium" version = "0.2.2" @@ -1287,6 +1310,33 @@ dependencies = [ "slab", ] +[[package]] +name = "gcp_auth" +version = "0.12.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "26d27dbcc645b60b8e7f6e2868a9d7102ece97d1bb49c1288b5321fcc67f7260" +dependencies = [ + "async-trait", + "base64 0.22.1", + "bytes", + "chrono", + "http 1.4.2", + "http-body-util", + "hyper 1.10.1", + "hyper-rustls 0.27.9", + "hyper-util", + "ring", + "rustls 0.23.42", + "rustls-pki-types", + "serde", + "serde_json", + "thiserror 2.0.19", + "tokio", + "tracing", + "tracing-futures", + "url", +] + [[package]] name = "generic-array" version = "0.14.7" @@ -1595,6 +1645,30 @@ dependencies = [ "tracing", ] +[[package]] +name = "iana-time-zone" +version = "0.1.65" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" +dependencies = [ + "android_system_properties", + "core-foundation-sys", + "iana-time-zone-haiku", + "js-sys", + "log", + "wasm-bindgen", + "windows-core", +] + +[[package]] +name = "iana-time-zone-haiku" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" +dependencies = [ + "cc", +] + [[package]] name = "icu_collections" version = "2.2.0" @@ -1875,6 +1949,7 @@ dependencies = [ "azure_identity", "base64 0.22.1", "data-url", + "gcp_auth", "moka", "rand 0.8.7", "reqwest 0.12.28", @@ -3593,6 +3668,16 @@ dependencies = [ "once_cell", ] +[[package]] +name = "tracing-futures" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "97d095ae15e245a057c8e8451bab9b3ee1e1f68e9ba2b4fbc18d0ac5237835f2" +dependencies = [ + "pin-project", + "tracing", +] + [[package]] name = "tracing-subscriber" version = "0.3.23" @@ -3984,12 +4069,65 @@ version = "0.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link", + "windows-result", + "windows-strings", +] + +[[package]] +name = "windows-implement" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-interface" +version = "0.59.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "windows-link" version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link", +] + [[package]] name = "windows-sys" version = "0.52.0" diff --git a/litellm-rust/Cargo.toml b/litellm-rust/Cargo.toml index 0f30ac5cf7d..5f25e69a1f8 100644 --- a/litellm-rust/Cargo.toml +++ b/litellm-rust/Cargo.toml @@ -41,6 +41,7 @@ tokio = { version = "1", features = ["rt-multi-thread", "macros", "time", "net"] tokio-tungstenite = { version = "0.24", default-features = false, features = ["connect", "rustls-tls-native-roots"] } futures-util = { version = "0.3", default-features = false, features = ["sink", "std"] } base64 = "0.22" +gcp_auth = "0.12.7" azure_core = "1.0.0" azure_identity = { version = "1.0.0", features = ["tokio"] } moka = { version = "0.12.16", features = ["future"] } diff --git a/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs b/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs index 10e08253e1a..8305cc80a1d 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs @@ -13,9 +13,7 @@ use litellm_core::providers::azure_ai::ocr::transformation::{ }; use litellm_core::providers::mistral::ocr::transformation::MISTRAL_OCR_CONFIG; use litellm_core::providers::vertex_ai::ocr::transformation as vertex_ai; -use litellm_core::providers::vertex_ai::ocr::transformation::{ - VERTEX_AI_DEEPSEEK_OCR_CONFIG, VERTEX_AI_OCR_CONFIG, -}; +use litellm_core::providers::vertex_ai::ocr::transformation::VERTEX_AI_DEEPSEEK_OCR_CONFIG; use crate::client::http_client; @@ -44,7 +42,7 @@ pub(super) fn ocr_provider_config( } "azure_ai" => Some(&AZURE_AI_OCR_CONFIG), "vertex_ai" if vertex_ai::is_deepseek_model(model) => Some(&VERTEX_AI_DEEPSEEK_OCR_CONFIG), - "vertex_ai" => Some(&VERTEX_AI_OCR_CONFIG), + "vertex_ai" => None, _ => None, } } diff --git a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs index cdf4a15125f..116c0f4a5e9 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs @@ -36,5 +36,7 @@ mod tests { Some("azure_ai") )); assert!(is_supported_request("parse-v3", Some("reducto"))); + assert!(is_supported_request("mistral-ocr", Some("vertex_ai"))); + assert!(!is_supported_request("deepseek-ocr", Some("vertex_ai"))); } } diff --git a/litellm-rust/crates/core/Cargo.toml b/litellm-rust/crates/core/Cargo.toml index dc4a1acea16..a2433435e34 100644 --- a/litellm-rust/crates/core/Cargo.toml +++ b/litellm-rust/crates/core/Cargo.toml @@ -15,6 +15,7 @@ base64.workspace = true azure_core.workspace = true azure_identity.workspace = true data-url = "0.3.2" +gcp_auth.workspace = true moka.workspace = true rand.workspace = true reqwest.workspace = true diff --git a/litellm-rust/crates/core/src/auth/error.rs b/litellm-rust/crates/core/src/auth/error.rs index ddd4a6d016e..e7027c0df10 100644 --- a/litellm-rust/crates/core/src/auth/error.rs +++ b/litellm-rust/crates/core/src/auth/error.rs @@ -6,6 +6,8 @@ pub enum AuthError { Configuration(#[from] AuthConfigurationError), #[error("credential acquisition failed: {0}")] AzureTokenAcquisition(String), + #[error("credential acquisition failed: Vertex AI credentials: {0}")] + VertexTokenAcquisition(String), #[error("credential acquisition failed: {}", .0.iter().map(ToString::to_string).collect::>().join("; "))] CredentialChain(Vec), #[error("credential caller failed: credential caller returned an empty credential")] @@ -73,6 +75,12 @@ pub enum AuthConfigurationError { RequestAzureCredentialReference, #[error("host credentials cannot be sent to a request-controlled Azure endpoint")] RequestAzureCredentialDestination, + #[error("credentials cannot be sent to a request-controlled Vertex AI endpoint")] + RequestVertexCredentialDestination, + #[error( + "request-controlled Vertex credentials must use the canonical Google OAuth token endpoint" + )] + RequestVertexTokenEndpoint, } #[derive(Clone, Debug, Error, PartialEq, Eq)] diff --git a/litellm-rust/crates/core/src/auth/mod.rs b/litellm-rust/crates/core/src/auth/mod.rs index 2ca2f3c3016..35d9c676f65 100644 --- a/litellm-rust/crates/core/src/auth/mod.rs +++ b/litellm-rust/crates/core/src/auth/mod.rs @@ -1,5 +1,6 @@ mod credential; pub mod error; +pub(crate) mod vertex; pub use error::AuthError; pub(crate) mod http; mod policy; diff --git a/litellm-rust/crates/core/src/auth/vertex.rs b/litellm-rust/crates/core/src/auth/vertex.rs new file mode 100644 index 00000000000..00a0a7ea7ee --- /dev/null +++ b/litellm-rust/crates/core/src/auth/vertex.rs @@ -0,0 +1,592 @@ +use std::collections::BTreeMap; +use std::future::Future; +use std::path::Path; +use std::pin::Pin; +use std::sync::Arc; + +use gcp_auth::{CustomServiceAccount, TokenProvider}; +use moka::future::Cache; +use serde_json::{Map, Value}; +use sha2::{Digest, Sha256}; + +use crate::auth::error::AuthConfigurationError; +use crate::auth::http::apply_credential; +use crate::auth::{AuthError, CredentialPlacement, InputSource, SecretValue, Sourced}; + +const CLOUD_PLATFORM_SCOPE: &str = "https://www.googleapis.com/auth/cloud-platform"; +const GOOGLE_OAUTH_TOKEN_ENDPOINT: &str = "https://oauth2.googleapis.com/token"; +const GOOGLE_APPLICATION_CREDENTIALS_ENV: &str = "GOOGLE_APPLICATION_CREDENTIALS"; +const VERTEX_AI_API_KEY_ENV: &str = "VERTEX_AI_API_KEY"; +const VERTEXAI_API_KEY_ENV: &str = "VERTEXAI_API_KEY"; +const VERTEXAI_CREDENTIALS_ENV: &str = "VERTEXAI_CREDENTIALS"; +const VERTEXAI_PROJECT_ENV: &str = "VERTEXAI_PROJECT"; +const VERTEXAI_LOCATION_ENV: &str = "VERTEXAI_LOCATION"; +const VERTEX_LOCATION_ENV: &str = "VERTEX_LOCATION"; + +#[derive(Clone, Debug, Default)] +pub(crate) struct VertexConfig { + credentials: Option>, + project_id: Option, + location: Option, +} + +impl VertexConfig { + pub(crate) fn from_sourced_optional_params( + params: &Map, + sources: &BTreeMap, + ) -> Result { + Ok(Self { + credentials: optional_credentials( + params, + sources, + &["vertex_credentials", "vertex_ai_credentials"], + )?, + project_id: optional_string(params, &["vertex_project", "vertex_ai_project"])?, + location: optional_string(params, &["vertex_location", "vertex_ai_location"])?, + }) + } + + pub(crate) fn project_id(&self) -> Option<&str> { + self.project_id.as_deref() + } + + pub(crate) fn location(&self) -> Option<&str> { + self.location.as_deref() + } +} + +pub(crate) struct VertexEnvironment { + pub headers: Vec<(String, String)>, + pub project_id: String, +} + +struct VertexAccessToken { + token: String, + project_id: String, +} + +pub(crate) fn get_vertex_ai_project( + config: &VertexConfig, + env_lookup: &dyn Fn(&str) -> Option, +) -> Option { + config + .project_id() + .map(str::to_string) + .or_else(|| non_empty_env(env_lookup, VERTEXAI_PROJECT_ENV)) +} + +pub(crate) fn get_vertex_ai_location( + config: &VertexConfig, + env_lookup: &dyn Fn(&str) -> Option, +) -> Option { + config + .location() + .map(str::to_string) + .or_else(|| non_empty_env(env_lookup, VERTEXAI_LOCATION_ENV)) + .or_else(|| non_empty_env(env_lookup, VERTEX_LOCATION_ENV)) +} + +#[derive(Clone)] +pub(crate) struct VertexAuth { + providers: Cache>, + loader: Arc, +} + +impl Default for VertexAuth { + fn default() -> Self { + Self::new(Arc::new(GcpProviderLoader)) + } +} + +impl VertexAuth { + fn new(loader: Arc) -> Self { + Self { + providers: Cache::builder().max_capacity(64).build(), + loader, + } + } + + #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] + pub(crate) async fn validate_environment( + &self, + headers: Vec<(String, String)>, + api_key: Option<&str>, + config: &VertexConfig, + env_lookup: &(dyn Fn(&str) -> Option + Sync), + ) -> Result { + let has_authorization = headers + .iter() + .any(|(name, _)| name.eq_ignore_ascii_case("Authorization")); + let static_token = api_key + .map(str::trim) + .filter(|value| !value.is_empty()) + .map(str::to_string) + .or_else(|| non_empty_env(env_lookup, VERTEX_AI_API_KEY_ENV)) + .or_else(|| non_empty_env(env_lookup, VERTEXAI_API_KEY_ENV)); + let project_id = get_vertex_ai_project(config, env_lookup); + + if !has_authorization && static_token.is_none() { + let access = self.get_access_token(config, env_lookup).await?; + return Ok(VertexEnvironment { + headers: apply_credential(headers, &access.token, CredentialPlacement::Bearer)?, + project_id: project_id.unwrap_or(access.project_id), + }); + } + + let project_id = match project_id { + Some(project_id) => project_id, + None => { + self.load_provider(config, env_lookup) + .await? + .project_id() + .await? + } + }; + let headers = if has_authorization { + headers + } else { + apply_credential( + headers, + static_token.as_deref().expect("static token was checked"), + CredentialPlacement::Bearer, + )? + }; + Ok(VertexEnvironment { + headers, + project_id, + }) + } + + async fn get_access_token( + &self, + config: &VertexConfig, + env_lookup: &(dyn Fn(&str) -> Option + Sync), + ) -> Result { + let provider = self.load_provider(config, env_lookup).await?; + let (token, project_id) = tokio::try_join!(provider.token(), provider.project_id())?; + Ok(VertexAccessToken { token, project_id }) + } + + async fn load_provider( + &self, + config: &VertexConfig, + env_lookup: &(dyn Fn(&str) -> Option + Sync), + ) -> Result, AuthError> { + let source = credential_source(config, env_lookup); + let key = source.cache_key(); + self.providers + .try_get_with(key, self.loader.load(source)) + .await + .map_err(|error| (*error).clone()) + } +} + +trait VertexTokenSource: Send + Sync { + fn project_id(&self) -> VertexAuthFuture<'_, String>; + fn token(&self) -> VertexAuthFuture<'_, String>; +} + +trait VertexProviderLoader: Send + Sync { + fn load(&self, source: CredentialSource) -> VertexAuthFuture<'_, Arc>; +} + +type VertexAuthFuture<'a, T> = Pin> + Send + 'a>>; + +struct GcpTokenSource(Arc); + +impl VertexTokenSource for GcpTokenSource { + fn project_id(&self) -> VertexAuthFuture<'_, String> { + Box::pin(async move { + self.0 + .project_id() + .await + .map(|project| project.to_string()) + .map_err(auth_acquisition_error) + }) + } + + fn token(&self) -> VertexAuthFuture<'_, String> { + Box::pin(async move { + self.0 + .token(&[CLOUD_PLATFORM_SCOPE]) + .await + .map(|token| token.as_str().to_string()) + .map_err(auth_acquisition_error) + }) + } +} + +struct GcpProviderLoader; + +impl VertexProviderLoader for GcpProviderLoader { + fn load(&self, source: CredentialSource) -> VertexAuthFuture<'_, Arc> { + Box::pin(async move { + let provider: Arc = match source { + CredentialSource::Inline(configured) => Arc::new( + CustomServiceAccount::from_json(validate_request_credentials( + configured.expose(), + )?) + .map_err(auth_acquisition_error)?, + ), + CredentialSource::Trusted(configured) => { + let configured = configured.expose(); + let service_account = if Path::new(configured).is_file() { + CustomServiceAccount::from_file(configured) + } else { + CustomServiceAccount::from_json(configured) + } + .map_err(auth_acquisition_error)?; + Arc::new(service_account) + } + CredentialSource::ApplicationCredentials(path) => { + Arc::new(CustomServiceAccount::from_file(path).map_err(auth_acquisition_error)?) + } + CredentialSource::Adc => { + gcp_auth::provider().await.map_err(auth_acquisition_error)? + } + }; + Ok(Arc::new(GcpTokenSource(provider)) as Arc) + }) + } +} + +fn validate_request_credentials(configured: &str) -> Result<&str, AuthError> { + let token_uri = serde_json::from_str::(configured) + .ok() + .and_then(|credentials| { + credentials + .get("token_uri") + .and_then(Value::as_str) + .map(str::to_string) + }); + if token_uri.as_deref() != Some(GOOGLE_OAUTH_TOKEN_ENDPOINT) { + return Err(AuthConfigurationError::RequestVertexTokenEndpoint.into()); + } + Ok(configured) +} + +#[derive(Clone, Debug)] +enum CredentialSource { + Inline(SecretValue), + Trusted(SecretValue), + ApplicationCredentials(String), + Adc, +} + +impl CredentialSource { + fn cache_key(&self) -> CredentialCacheKey { + match self { + Self::Inline(configured) => { + CredentialCacheKey::Inline(Sha256::digest(configured.expose()).into()) + } + Self::Trusted(configured) => { + CredentialCacheKey::Trusted(Sha256::digest(configured.expose()).into()) + } + Self::ApplicationCredentials(path) => { + CredentialCacheKey::ApplicationCredentials(path.clone()) + } + Self::Adc => CredentialCacheKey::Adc, + } + } +} + +#[derive(Clone, Debug, Hash, PartialEq, Eq)] +enum CredentialCacheKey { + Inline([u8; 32]), + Trusted([u8; 32]), + ApplicationCredentials(String), + Adc, +} + +fn credential_source( + config: &VertexConfig, + env_lookup: &dyn Fn(&str) -> Option, +) -> CredentialSource { + if let Some(configured) = config.credentials.clone() { + return match configured.source() { + InputSource::Request => CredentialSource::Inline(configured.into_value()), + InputSource::Deployment | InputSource::Environment => { + CredentialSource::Trusted(configured.into_value()) + } + }; + } + if let Some(configured) = non_empty_env(env_lookup, VERTEXAI_CREDENTIALS_ENV) { + return CredentialSource::Trusted(SecretValue::new(configured)); + } + non_empty_env(env_lookup, GOOGLE_APPLICATION_CREDENTIALS_ENV) + .map(CredentialSource::ApplicationCredentials) + .unwrap_or(CredentialSource::Adc) +} + +fn optional_credentials( + params: &Map, + sources: &BTreeMap, + names: &[&str], +) -> Result>, AuthError> { + for name in names { + let source = source_for(sources, name); + match params.get(*name) { + None | Some(Value::Null) => continue, + Some(Value::String(value)) if value.trim().is_empty() => continue, + Some(Value::String(value)) => { + return Ok(Some(Sourced::new(SecretValue::new(value), source))); + } + Some(Value::Object(value)) if value.is_empty() => continue, + Some(Value::Object(value)) => { + return serde_json::to_string(value) + .map(SecretValue::new) + .map(|value| Sourced::new(value, source)) + .map(Some) + .map_err(|error| { + AuthError::Configuration(AuthConfigurationError::InvalidFieldType(format!( + "{}: {error}", + names[0] + ))) + }); + } + Some(_) => { + return Err(AuthError::Configuration( + AuthConfigurationError::InvalidFieldType(names[0].to_string()), + )); + } + } + } + Ok(None) +} + +fn source_for(sources: &BTreeMap, name: &str) -> InputSource { + sources.get(name).copied().unwrap_or_default() +} + +fn optional_string( + params: &Map, + names: &[&str], +) -> Result, AuthError> { + for name in names { + match params.get(*name) { + None | Some(Value::Null) => continue, + Some(Value::String(value)) if value.trim().is_empty() => continue, + Some(Value::String(value)) => return Ok(Some(value.clone())), + Some(_) => { + return Err(AuthError::Configuration( + AuthConfigurationError::InvalidFieldType(names[0].to_string()), + )); + } + } + } + Ok(None) +} + +fn non_empty_env(env_lookup: &dyn Fn(&str) -> Option, name: &str) -> Option { + env_lookup(name) + .map(|value| value.trim().to_string()) + .filter(|value| !value.is_empty()) +} + +fn auth_acquisition_error(error: gcp_auth::Error) -> AuthError { + AuthError::VertexTokenAcquisition(error.to_string()) +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicUsize, Ordering}; + + use serde_json::json; + + use super::*; + + struct FakeProvider { + calls: Arc, + } + + impl VertexTokenSource for FakeProvider { + fn project_id(&self) -> VertexAuthFuture<'_, String> { + self.calls.fetch_add(1, Ordering::SeqCst); + Box::pin(async { Ok("adc-project".into()) }) + } + + fn token(&self) -> VertexAuthFuture<'_, String> { + self.calls.fetch_add(1, Ordering::SeqCst); + Box::pin(async { Ok("adc-token".into()) }) + } + } + + struct FakeLoader { + loads: Arc, + provider: Arc, + } + + impl VertexProviderLoader for FakeLoader { + fn load( + &self, + _source: CredentialSource, + ) -> VertexAuthFuture<'_, Arc> { + let loads = self.loads.clone(); + let provider = self.provider.clone(); + Box::pin(async move { + loads.fetch_add(1, Ordering::SeqCst); + Ok(provider) + }) + } + } + + fn config(value: Value) -> VertexConfig { + VertexConfig::from_sourced_optional_params(value.as_object().unwrap(), &BTreeMap::new()) + .unwrap() + } + + fn auth(calls: Arc, loads: Arc) -> VertexAuth { + let provider: Arc = Arc::new(FakeProvider { calls }); + VertexAuth::new(Arc::new(FakeLoader { loads, provider })) + } + + #[test] + fn config_is_typed_and_secrets_are_redacted() { + let config = config(json!({ + "vertex_credentials":{"private_key":"secret-key"}, + "vertex_project":"project-1", + "vertex_location":"europe-west4" + })); + assert_eq!(config.project_id(), Some("project-1")); + assert_eq!(config.location(), Some("europe-west4")); + assert!(!format!("{config:?}").contains("secret-key")); + assert!( + VertexConfig::from_sourced_optional_params( + json!({"vertex_credentials":true}).as_object().unwrap(), + &BTreeMap::new() + ) + .is_err() + ); + } + + #[test] + fn empty_primary_values_fall_back_to_python_aliases() { + let config = config(json!({ + "vertex_credentials": null, + "vertex_ai_credentials": "alias-credentials", + "vertex_project": " ", + "vertex_ai_project": "alias-project", + "vertex_location": null, + "vertex_ai_location": "alias-location" + })); + assert_eq!( + config.credentials.as_ref().unwrap().value().expose(), + "alias-credentials" + ); + assert_eq!(config.project_id(), Some("alias-project")); + assert_eq!(config.location(), Some("alias-location")); + } + + #[test] + fn project_and_location_prefer_input_then_environment() { + let configured = + config(json!({"vertex_project":"input-project","vertex_location":"input-location"})); + let env = |name: &str| Some(format!("env-{name}")); + assert_eq!( + get_vertex_ai_project(&configured, &env).as_deref(), + Some("input-project") + ); + assert_eq!( + get_vertex_ai_location(&configured, &env).as_deref(), + Some("input-location") + ); + let empty = VertexConfig::default(); + assert_eq!( + get_vertex_ai_project(&empty, &|_| Some("env-project".into())).as_deref(), + Some("env-project") + ); + assert_eq!( + get_vertex_ai_location(&empty, &|name| (name == VERTEX_LOCATION_ENV) + .then(|| "fallback-location".into())) + .as_deref(), + Some("fallback-location") + ); + } + + #[test] + fn credential_discovery_prefers_input_then_environment_then_adc() { + let params = json!({"vertex_credentials":"input-json"}); + let sources = BTreeMap::from([("vertex_credentials".to_string(), InputSource::Request)]); + let configured = + VertexConfig::from_sourced_optional_params(params.as_object().unwrap(), &sources) + .unwrap(); + assert!( + matches!(credential_source(&configured, &|_| Some("environment-value".into())), CredentialSource::Inline(value) if value.expose() == "input-json") + ); + let empty = VertexConfig::default(); + assert!( + matches!(credential_source(&empty, &|name| (name == VERTEXAI_CREDENTIALS_ENV).then(|| "environment-json".into())), CredentialSource::Trusted(value) if value.expose() == "environment-json") + ); + assert!( + matches!(credential_source(&empty, &|name| (name == GOOGLE_APPLICATION_CREDENTIALS_ENV).then(|| "adc.json".into())), CredentialSource::ApplicationCredentials(path) if path == "adc.json") + ); + assert!(matches!( + credential_source(&empty, &|_| None), + CredentialSource::Adc + )); + assert_ne!( + CredentialSource::Inline(SecretValue::new("same-value")).cache_key(), + CredentialSource::Trusted(SecretValue::new("same-value")).cache_key() + ); + } + + #[test] + fn request_credentials_require_canonical_token_endpoint() { + assert!( + validate_request_credentials(r#"{"token_uri":"https://oauth2.googleapis.com/token"}"#) + .is_ok() + ); + assert!(matches!( + validate_request_credentials(r#"{"token_uri":"http://127.0.0.1/token"}"#), + Err(AuthError::Configuration( + AuthConfigurationError::RequestVertexTokenEndpoint + )) + )); + assert!(matches!( + validate_request_credentials("{}"), + Err(AuthError::Configuration( + AuthConfigurationError::RequestVertexTokenEndpoint + )) + )); + } + + #[tokio::test] + async fn explicit_token_and_header_do_not_acquire_adc() { + let loads = Arc::new(AtomicUsize::new(0)); + let auth = auth(Arc::new(AtomicUsize::new(0)), loads.clone()); + let configured = config(json!({"vertex_project":"project-1"})); + let explicit = auth + .validate_environment(Vec::new(), Some("access-token"), &configured, &|_| None) + .await + .unwrap(); + assert_eq!(explicit.headers[0].1, "Bearer access-token"); + let existing = auth + .validate_environment( + vec![("authorization".into(), "Bearer existing".into())], + None, + &configured, + &|_| None, + ) + .await + .unwrap(); + assert_eq!(existing.headers[0].1, "Bearer existing"); + assert_eq!(loads.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn provider_is_reused_across_authentication_calls() { + let calls = Arc::new(AtomicUsize::new(0)); + let loads = Arc::new(AtomicUsize::new(0)); + let auth = auth(calls.clone(), loads.clone()); + for _ in 0..2 { + let environment = auth + .validate_environment(Vec::new(), None, &VertexConfig::default(), &|_| None) + .await + .unwrap(); + assert_eq!(environment.project_id, "adc-project"); + assert_eq!(environment.headers[0].1, "Bearer adc-token"); + } + assert_eq!(loads.load(Ordering::SeqCst), 1); + assert_eq!(calls.load(Ordering::SeqCst), 4); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/mod.rs index c089ca90605..a96a7fcdf38 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/mod.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/mod.rs @@ -11,10 +11,12 @@ use super::wire::DecodedOcrResponse; mod azure; mod mistral; mod reducto; +mod vertex; pub(crate) use azure::{AzureDocumentIntelligenceAdapter, AzureMistralAdapter}; pub(crate) use mistral::MistralAdapter; pub(crate) use reducto::{ReductoLegacyAdapter, ReductoV3Adapter}; +pub(crate) use vertex::VertexMistralAdapter; /// Converts a complete LiteLLM OCR call to provider HTTP and normalizes its response. pub(crate) trait OcrAdapter: Send + Sync + Sized + 'static { @@ -70,6 +72,7 @@ macro_rules! for_each_ocr_adapter { AzureDocumentIntelligence, $crate::ocr::adapters::AzureDocumentIntelligenceAdapter, $crate::ocr::adapters::AzureDocumentIntelligenceAdapter, AzureAi; ReductoLegacy, $crate::ocr::adapters::ReductoLegacyAdapter, $crate::ocr::adapters::ReductoLegacyAdapter, Reducto; ReductoV3, $crate::ocr::adapters::ReductoV3Adapter, $crate::ocr::adapters::ReductoV3Adapter, Reducto; + VertexMistral, $crate::ocr::adapters::VertexMistralAdapter, $crate::ocr::adapters::VertexMistralAdapter, VertexAi; } }; } diff --git a/litellm-rust/crates/core/src/ocr/adapters/vertex/mistral.rs b/litellm-rust/crates/core/src/ocr/adapters/vertex/mistral.rs new file mode 100644 index 00000000000..f3335bf497c --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/vertex/mistral.rs @@ -0,0 +1,154 @@ +use super::super::OcrAdapter; +use super::validate_destination; +use crate::Error; +use crate::auth::vertex::{self, VertexConfig}; +use crate::ocr::OcrClient; +use crate::ocr::codecs::mistral::{self, MistralOcrParams, MistralOcrResponse}; +use crate::ocr::document::{inline_remote_document, validate_inline_document}; +use crate::ocr::error::{OcrError, OcrRequestError, OcrResponseError}; +use crate::ocr::prepare::{ + _prepare_ocr_request, ParsedProviderParams, credential_env, transform_request_body, +}; +use crate::ocr::registry::OcrProvider; +use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse}; +use crate::url_utils::ApiUrl; +const DEFAULT_LOCATION: &str = "us-central1"; + +#[derive(Clone, Debug)] +pub(crate) struct VertexMistralAdapter; + +impl OcrAdapter for VertexMistralAdapter { + type ProviderResponse = MistralOcrResponse; + const PROVIDER: OcrProvider = OcrProvider::VertexAi; + + async fn prepare_request( + &self, + request: &LiteLLMOcrRequest, + client: &OcrClient, + ) -> Result { + validate_destination(&request.connection)?; + let ParsedProviderParams { + known: params, + extra_params: _extra_params, + } = _prepare_ocr_request::(request)?; + let config = VertexConfig::from_sourced_optional_params( + &request.optional_params, + &request.input_sources, + ) + .map_err(Error::from)?; + let authentication = client + .vertex_auth() + .validate_environment( + request.connection.extra_headers.clone(), + request.connection.api_key.as_deref(), + &config, + &credential_env, + ) + .await + .map_err(Error::from)?; + let location = vertex::get_vertex_ai_location(&config, &credential_env) + .unwrap_or_else(|| DEFAULT_LOCATION.to_string()); + let url = get_complete_url( + request.connection.api_base.as_deref(), + &authentication.project_id, + &location, + &request.model, + )?; + let document = inline_remote_document( + client.document_fetcher(), + request.document.clone(), + &request.connection, + ) + .await?; + let body = mistral::transform_ocr_request(&request.model, document, ¶ms)?; + transform_request_body( + client, + request, + &url, + &authentication.headers, + body, + |body| validate_inline_document(&body.document), + ) + .await + } + + fn transform_ocr_response( + &self, + request: &LiteLLMOcrRequest, + response: Self::ProviderResponse, + ) -> Result { + mistral::transform_ocr_response(&request.model, response) + } +} + +fn get_complete_url( + api_base: Option<&str>, + project: &str, + location: &str, + model: &str, +) -> Result { + validate_location(location)?; + let default_base = format!("https://{location}-aiplatform.googleapis.com"); + let base = api_base + .map(str::trim) + .filter(|base| !base.is_empty()) + .unwrap_or(&default_base); + let prediction = format!("{model}:rawPredict"); + ApiUrl::parse(base) + .and_then(|url| { + url.complete_path(&[ + "v1", + "projects", + project, + "locations", + location, + "publishers", + "mistralai", + "models", + &prediction, + ]) + }) + .map(|url| url.into_string()) + .map_err(|_| { + OcrRequestError::RequestField { + path: "api_base".into(), + } + .into() + }) +} + +fn validate_location(location: &str) -> Result<(), OcrError> { + let valid = !location.is_empty() + && location + .bytes() + .all(|value| value.is_ascii_lowercase() || value.is_ascii_digit() || value == b'-') + && location + .as_bytes() + .first() + .is_some_and(u8::is_ascii_alphanumeric) + && location + .as_bytes() + .last() + .is_some_and(u8::is_ascii_alphanumeric); + if valid { + return Ok(()); + } + Err(OcrRequestError::RequestField { + path: "vertex_location".into(), + } + .into()) +} + +#[cfg(test)] +mod tests { + use super::get_complete_url; + + #[test] + fn endpoint_uses_location_project_and_model() { + assert_eq!( + get_complete_url(None, "proj-1", "europe-west4", "mistral-ocr-maas").unwrap(), + "https://europe-west4-aiplatform.googleapis.com/v1/projects/proj-1/locations/europe-west4/publishers/mistralai/models/mistral-ocr-maas:rawPredict" + ); + assert!(get_complete_url(None, "proj-1", "attacker.example/path", "model").is_err()); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs new file mode 100644 index 00000000000..ce6f884b41d --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs @@ -0,0 +1,19 @@ +mod mistral; + +use crate::Error; +use crate::auth::InputSource; +use crate::auth::error::AuthConfigurationError; +use crate::ocr::error::OcrError; +use crate::ocr::types::OcrConnection; + +pub(crate) use mistral::VertexMistralAdapter; + +fn validate_destination(connection: &OcrConnection) -> Result<(), OcrError> { + if connection.api_base.is_some() && connection.api_base_source == InputSource::Request { + return Err(Error::from(crate::AuthError::Configuration( + AuthConfigurationError::RequestVertexCredentialDestination, + )) + .into()); + } + Ok(()) +} diff --git a/litellm-rust/crates/core/src/ocr/client.rs b/litellm-rust/crates/core/src/ocr/client.rs index 61b7ae8d995..ab2d098d0bb 100644 --- a/litellm-rust/crates/core/src/ocr/client.rs +++ b/litellm-rust/crates/core/src/ocr/client.rs @@ -8,6 +8,7 @@ use super::handler::perform_ocr_request; use super::types::{LiteLLMOcrRequest, LiteLLMOcrResponse}; use super::wire::{DecodedOcrResponse, decode_response}; use crate::Error; +use crate::auth::vertex::VertexAuth; use crate::constants::OCR_CONNECT_TIMEOUT_SECS; use crate::error::TransportError; use crate::media::MediaFetcher; @@ -17,6 +18,7 @@ pub struct OcrClient { provider_http: reqwest::Client, polling_http: reqwest::Client, document_fetcher: MediaFetcher, + vertex_auth: VertexAuth, } impl OcrClient { @@ -26,6 +28,7 @@ impl OcrClient { provider_http, polling_http: no_redirect_http()?, document_fetcher, + vertex_auth: VertexAuth::default(), }) } @@ -51,12 +54,17 @@ impl OcrClient { &self.document_fetcher } + pub(crate) fn vertex_auth(&self) -> &VertexAuth { + &self.vertex_auth + } + #[cfg(test)] pub(crate) fn for_test(provider_http: reqwest::Client, document_http: reqwest::Client) -> Self { Self { provider_http, polling_http: no_redirect_http().expect("test polling client builds"), document_fetcher: MediaFetcher::for_test(document_http), + vertex_auth: VertexAuth::default(), } } } diff --git a/litellm-rust/crates/core/src/ocr/mod.rs b/litellm-rust/crates/core/src/ocr/mod.rs index 5f274186ed0..69b2958483a 100644 --- a/litellm-rust/crates/core/src/ocr/mod.rs +++ b/litellm-rust/crates/core/src/ocr/mod.rs @@ -29,3 +29,6 @@ pub(crate) mod test_support; #[cfg(test)] #[path = "../../tests/ocr.rs"] pub(crate) mod tests; +#[cfg(test)] +#[path = "../../tests/vertex_ai_ocr.rs"] +mod vertex_ai_tests; diff --git a/litellm-rust/crates/core/src/ocr/registry.rs b/litellm-rust/crates/core/src/ocr/registry.rs index 6b100795fc8..1097d102a20 100644 --- a/litellm-rust/crates/core/src/ocr/registry.rs +++ b/litellm-rust/crates/core/src/ocr/registry.rs @@ -26,6 +26,7 @@ pub(crate) enum OcrProvider { Mistral, AzureAi, Reducto, + VertexAi, } impl OcrProvider { @@ -34,6 +35,7 @@ impl OcrProvider { Self::Mistral => "mistral", Self::AzureAi => "azure_ai", Self::Reducto => "reducto", + Self::VertexAi => "vertex_ai", } } } @@ -51,6 +53,7 @@ pub(crate) fn resolve_wire_adapter( "mistral" => OcrProvider::Mistral, "azure_ai" => OcrProvider::AzureAi, "reducto" => OcrProvider::Reducto, + "vertex_ai" => OcrProvider::VertexAi, value => return Err(Error::InvalidProvider(value.to_string())), }; let adapter = match typed_provider { @@ -71,6 +74,10 @@ pub(crate) fn resolve_wire_adapter( provider.model ))); } + OcrProvider::VertexAi if provider.model.to_ascii_lowercase().contains("deepseek") => { + return Err(Error::Unsupported("Vertex DeepSeek OCR")); + } + OcrProvider::VertexAi => OcrAdapterKind::VertexMistral, }; Ok((provider.model.to_string(), adapter)) } diff --git a/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs b/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs index b0e5a278d0b..c2d810822ae 100644 --- a/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs +++ b/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs @@ -3,8 +3,6 @@ use crate::ocr::transformation::OcrProviderConfig; use crate::ocr::types::{LiteLLMOcrResponse, OcrRequestData}; use serde_json::{Map, Value, json}; -use crate::providers::mistral::ocr::transformation::MISTRAL_OCR_CONFIG; - const VERTEX_DEFAULT_LOCATION: &str = "us-central1"; const VERTEX_DEFAULT_DEEPSEEK_API_BASE: &str = "https://aiplatform.googleapis.com"; const VERTEX_AI_API_KEY_ENV: &str = "VERTEX_AI_API_KEY"; @@ -23,10 +21,8 @@ const DEEPSEEK_SUPPORTED_OCR_PARAMS: &[&str] = &[ "stop", ]; -pub struct VertexAiOcrConfig; pub struct VertexAiDeepSeekOcrConfig; -pub const VERTEX_AI_OCR_CONFIG: VertexAiOcrConfig = VertexAiOcrConfig; pub const VERTEX_AI_DEEPSEEK_OCR_CONFIG: VertexAiDeepSeekOcrConfig = VertexAiDeepSeekOcrConfig; fn string_param<'a>(params: &'a Map, keys: &[&str]) -> Option<&'a str> { @@ -84,30 +80,6 @@ fn vertex_location( .unwrap_or_else(|| VERTEX_DEFAULT_LOCATION.to_string()) } -fn vertex_mistral_api_base(api_base: Option<&str>, location: &str) -> String { - api_base - .map(str::trim) - .filter(|value| !value.is_empty()) - .map(str::to_string) - .unwrap_or_else(|| format!("https://{location}-aiplatform.googleapis.com")) - .trim_end_matches('/') - .to_string() -} - -pub fn complete_vertex_mistral_url( - api_base: Option<&str>, - model: &str, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - let project = vertex_project(optional_params, env_lookup)?; - let location = vertex_location(optional_params, env_lookup); - let base = vertex_mistral_api_base(api_base, &location); - Ok(format!( - "{base}/v1/projects/{project}/locations/{location}/publishers/mistralai/models/{model}:rawPredict" - )) -} - pub fn complete_vertex_deepseek_url( api_base: Option<&str>, optional_params: &Map, @@ -207,53 +179,6 @@ fn ocr_data_from_content(content: Value, usage: Option, model: &str) -> V } } -impl OcrProviderConfig for VertexAiOcrConfig { - fn supported_ocr_params(&self) -> &'static [&'static str] { - MISTRAL_OCR_CONFIG.supported_ocr_params() - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_request( - &self, - model: &str, - document: Value, - optional_params: Map, - ) -> Result { - MISTRAL_OCR_CONFIG.transform_ocr_request(model, document, optional_params) - } - - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - MISTRAL_OCR_CONFIG.transform_ocr_response(model, response_json) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn complete_url( - &self, - api_base: Option<&str>, - model: &str, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - complete_vertex_mistral_url(api_base, model, optional_params, env_lookup) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_vertex_api_key(api_key, env_lookup) - } - - fn requires_data_uri_document(&self) -> bool { - true - } -} - impl OcrProviderConfig for VertexAiDeepSeekOcrConfig { #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] fn supported_ocr_params(&self) -> &'static [&'static str] { @@ -379,37 +304,6 @@ mod tests { use super::*; use rstest::rstest; - #[test] - fn vertex_mistral_url_uses_project_location_and_model() { - let params = Map::from_iter([ - ("vertex_project".to_string(), json!("proj-1")), - ("vertex_location".to_string(), json!("europe-west4")), - ]); - - let url = complete_vertex_mistral_url(None, "mistral-ocr-maas", ¶ms, &|_| None) - .expect("url builds"); - - assert_eq!( - url, - "https://europe-west4-aiplatform.googleapis.com/v1/projects/proj-1/locations/europe-west4/publishers/mistralai/models/mistral-ocr-maas:rawPredict" - ); - } - - #[test] - fn vertex_mistral_reuses_mistral_body_transform() { - let body = VERTEX_AI_OCR_CONFIG - .transform_ocr_request( - "mistral-ocr-maas", - json!({"type": "image_url", "image_url": "data:image/png;base64,abc"}), - Map::new(), - ) - .expect("request transforms") - .data; - - assert_eq!(body["model"], "mistral-ocr-maas"); - assert_eq!(body["document"]["image_url"], "data:image/png;base64,abc"); - } - #[test] fn vertex_deepseek_request_uses_ocr_endpoint_shape() { let body = VERTEX_AI_DEEPSEEK_OCR_CONFIG diff --git a/litellm-rust/crates/core/tests/vertex_ai_ocr.rs b/litellm-rust/crates/core/tests/vertex_ai_ocr.rs new file mode 100644 index 00000000000..2358552c742 --- /dev/null +++ b/litellm-rust/crates/core/tests/vertex_ai_ocr.rs @@ -0,0 +1,110 @@ +use serde_json::{Value, json}; + +use super::test_support::{MockResponse, mock_server, perform_ocr, wire_request}; +use crate::auth::InputSource; +use crate::ocr::wire::{OcrWireRequest, decode_request}; + +fn request_body(request: &str) -> Value { + serde_json::from_str(request.split_once("\r\n\r\n").unwrap().1).unwrap() +} + +#[tokio::test] +async fn facade_executes_vertex_mistral_with_resolved_project_and_location() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "pages":[{"index":0,"markdown":"hello"}], + "usage_info":{"pages_processed":1} + }))]) + .await; + let request = wire_request( + "vertex_ai/mistral-ocr-maas", + &base, + json!({ + "vertex_project":"project-1", + "vertex_location":"europe-west4", + "extract_footer":true + }), + ); + + let response = perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert_eq!(response.pages[0]["markdown"], "hello"); + let requests = seen.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert!(requests[0].starts_with( + "POST /v1/projects/project-1/locations/europe-west4/publishers/mistralai/models/mistral-ocr-maas:rawPredict " + )); + assert!( + requests[0] + .to_ascii_lowercase() + .contains("authorization: bearer test-key") + ); + assert_eq!( + request_body(&requests[0]), + json!({ + "model":"mistral-ocr-maas", + "document":{"type":"document_url","document_url":"data:application/pdf;base64,YWJj"}, + "extract_footer":true + }) + ); +} + +#[tokio::test] +async fn supplied_authorization_is_forwarded_without_a_static_token() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({"pages":[]}))]).await; + let mut request = wire_request( + "vertex_ai/model", + &base, + json!({"vertex_project":"project-1"}), + ); + request.connection.api_key = None; + request.connection.extra_headers = vec![("authorization".into(), "Bearer supplied".into())]; + + perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert!( + seen.lock().unwrap()[0] + .to_ascii_lowercase() + .contains("authorization: bearer supplied") + ); +} + +#[tokio::test] +async fn invalid_credentials_fail_before_provider_http() { + let request = wire_request( + "vertex_ai/model", + "http://127.0.0.1:1", + json!({"vertex_credentials": true}), + ); + let error = perform_ocr(request).await.unwrap_err(); + assert!(error.to_string().contains("vertex_credentials")); +} + +#[tokio::test] +async fn request_controlled_api_base_is_rejected_before_vertex_auth() { + let request = decode_request(OcrWireRequest { + model: "vertex_ai/model".into(), + document: json!({"type":"document_url","document_url":"data:application/pdf;base64,YWJj"}), + api_key: Some("test-key".into()), + api_base: Some("https://attacker.example".into()), + custom_llm_provider: None, + extra_headers: None, + optional_params: json!({"vertex_project":"project-1"}) + .as_object() + .unwrap() + .clone(), + input_sources: std::collections::BTreeMap::from([( + "api_base".to_string(), + InputSource::Request, + )]), + timeout_seconds: Some(2.0), + }) + .unwrap(); + + let error = perform_ocr(request).await.unwrap_err(); + + assert!( + error + .to_string() + .contains("request-controlled Vertex AI endpoint") + ); +} diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index d1d4a70007d..0aab11e3cfc 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -111,6 +111,7 @@ mod tests { )); assert!(is_supported_request("parse-v3", Some("reducto"))); assert!(is_supported_request("parse-legacy", Some("reducto"))); - assert!(!is_supported_request("mistral-ocr", Some("vertex_ai"))); + assert!(is_supported_request("mistral-ocr", Some("vertex_ai"))); + assert!(!is_supported_request("deepseek-ocr", Some("vertex_ai"))); } } diff --git a/litellm/rust_bridge/ocr.py b/litellm/rust_bridge/ocr.py index 68bbb186f9b..89eab71ccba 100644 --- a/litellm/rust_bridge/ocr.py +++ b/litellm/rust_bridge/ocr.py @@ -179,10 +179,19 @@ def _optional_params(request: LiteLLMOcrRequest, resolve_secret: Callable[[str], or resolve_secret("VERTEXAI_LOCATION") or resolve_secret("VERTEX_LOCATION") ) + credentials: Final = ( + request.kwargs.get("vertex_credentials") + or request.kwargs.get("vertex_ai_credentials") + or resolve_secret("VERTEXAI_CREDENTIALS") + ) vertex_params: Final = MappingProxyType( { name: value - for name, value in (("vertex_project", project), ("vertex_location", location)) + for name, value in ( + ("vertex_project", project), + ("vertex_location", location), + ("vertex_credentials", credentials), + ) if value is not None } ) diff --git a/tests/test_litellm/ocr/test_rust_bridge.py b/tests/test_litellm/ocr/test_rust_bridge.py index 9c1cae6a551..dbb4f822d0b 100644 --- a/tests/test_litellm/ocr/test_rust_bridge.py +++ b/tests/test_litellm/ocr/test_rust_bridge.py @@ -576,6 +576,7 @@ def test_prepare_rust_ocr_call_resolves_vertex_routing_metadata_from_secret_mana return { "VERTEXAI_PROJECT": "project-from-secret", "VERTEXAI_LOCATION": "us-east5", + "VERTEXAI_CREDENTIALS": "credentials-from-secret", }.get(name) ocr_main._run_rust_ocr( @@ -589,6 +590,7 @@ def test_prepare_rust_ocr_call_resolves_vertex_routing_metadata_from_secret_mana assert bridge.calls[0]["optional_params"]["vertex_project"] == "project-from-secret" assert bridge.calls[0]["optional_params"]["vertex_location"] == "us-east5" + assert bridge.calls[0]["optional_params"]["vertex_credentials"] == "credentials-from-secret" def test_prepare_rust_ocr_call_defers_azure_environment_resolution_to_rust(): From 83ab0113f0b083b97e66385fff8adb903432db3b Mon Sep 17 00:00:00 2001 From: yujonglee Date: Fri, 11 Sep 2026 16:22:57 -0700 Subject: [PATCH 131/157] feat(ocr): add Vertex DeepSeek adapter and remove legacy OCR pipeline (#40509) * feat(ocr): add Vertex DeepSeek adapter * fix(ocr): restore stacked CI coverage * style(ocr): apply workspace rustfmt * fix(ocr): deduplicate stacked gateway error mapping * test(ocr): keep response format checks at dispatch * fix(ocr): initialize gateway input provenance * refactor(ocr): preserve DeepSeek extra params * refactor(ocr): align Vertex DeepSeek preparation * fix(gateway): drop removed OCR credential error variant * fix(ocr): fail closed for deferred hooks and Vertex destinations * fix(ocr): preserve DeepSeek credential provenance --- .../src/audio_transcription/hooks.rs | 1 - .../crates/ai-gateway/src/ocr/common_utils.rs | 525 ------- .../crates/ai-gateway/src/ocr/handler.rs | 84 - .../crates/ai-gateway/src/ocr/hooks.rs | 330 ---- litellm-rust/crates/ai-gateway/src/ocr/mod.rs | 113 +- .../crates/ai-gateway/src/ocr/prepare.rs | 163 -- .../crates/ai-gateway/src/ocr/types.rs | 36 - .../ai-gateway/src/routes/messages/mod.rs | 13 +- .../crates/ai-gateway/tests/ocr_lifecycle.rs | 603 -------- litellm-rust/crates/core/src/error.rs | 2 - .../crates/core/src/ocr/adapters/mod.rs | 3 +- .../core/src/ocr/adapters/vertex/deepseek.rs | 134 ++ .../core/src/ocr/adapters/vertex/mod.rs | 2 + .../core/src/ocr/codecs/deepseek/mod.rs | 5 + .../src/ocr/codecs/deepseek/transformation.rs | 98 ++ .../core/src/ocr/codecs/deepseek/types.rs | 95 ++ .../src/ocr/codecs/mistral/transformation.rs | 178 ++- .../crates/core/src/ocr/codecs/mod.rs | 1 + litellm-rust/crates/core/src/ocr/error.rs | 2 + litellm-rust/crates/core/src/ocr/mod.rs | 7 +- litellm-rust/crates/core/src/ocr/prepare.rs | 1 - litellm-rust/crates/core/src/ocr/registry.rs | 2 +- .../crates/core/src/ocr/transformation.rs | 107 -- litellm-rust/crates/core/src/ocr/types.rs | 6 - .../crates/core/src/providers/azure_ai/mod.rs | 1 - .../core/src/providers/azure_ai/ocr/mod.rs | 1 - .../providers/azure_ai/ocr/transformation.rs | 1376 ----------------- .../crates/core/src/providers/mistral/mod.rs | 1 - .../core/src/providers/mistral/ocr/mod.rs | 1 - .../providers/mistral/ocr/transformation.rs | 436 ------ litellm-rust/crates/core/src/providers/mod.rs | 2 - .../core/src/providers/vertex_ai/mod.rs | 1 - .../core/src/providers/vertex_ai/ocr/mod.rs | 1 - .../providers/vertex_ai/ocr/transformation.rs | 361 ----- .../crates/core/tests/deepseek_ocr.rs | 95 ++ .../core/tests/vertex_ai_deepseek_ocr.rs | 83 + .../crates/core/tests/vertex_ai_ocr.rs | 91 +- .../crates/python-bridge/src/errors.rs | 1 - .../crates/python-bridge/src/routes/ocr.rs | 2 +- 39 files changed, 853 insertions(+), 4111 deletions(-) delete mode 100644 litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs delete mode 100644 litellm-rust/crates/ai-gateway/src/ocr/handler.rs delete mode 100644 litellm-rust/crates/ai-gateway/src/ocr/hooks.rs delete mode 100644 litellm-rust/crates/ai-gateway/src/ocr/prepare.rs delete mode 100644 litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs create mode 100644 litellm-rust/crates/core/src/ocr/adapters/vertex/deepseek.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/deepseek/mod.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/deepseek/transformation.rs create mode 100644 litellm-rust/crates/core/src/ocr/codecs/deepseek/types.rs delete mode 100644 litellm-rust/crates/core/src/ocr/transformation.rs delete mode 100644 litellm-rust/crates/core/src/providers/azure_ai/ocr/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs delete mode 100644 litellm-rust/crates/core/src/providers/mistral/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/mistral/ocr/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/mistral/ocr/transformation.rs delete mode 100644 litellm-rust/crates/core/src/providers/vertex_ai/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/vertex_ai/ocr/mod.rs delete mode 100644 litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs create mode 100644 litellm-rust/crates/core/tests/deepseek_ocr.rs create mode 100644 litellm-rust/crates/core/tests/vertex_ai_deepseek_ocr.rs diff --git a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs index 17f5591d1fc..6f48f38c9f6 100644 --- a/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs +++ b/litellm-rust/crates/ai-gateway/src/audio_transcription/hooks.rs @@ -272,7 +272,6 @@ fn core_error_kind(error: &Error) -> &'static str { Error::Auth(_) | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken | Error::MissingAzureDocumentIntelligenceCredentials | Error::MissingReductoApiKey => "AuthError", Error::InvalidProvider(_) => "InvalidProvider", diff --git a/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs b/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs deleted file mode 100644 index 8305cc80a1d..00000000000 --- a/litellm-rust/crates/ai-gateway/src/ocr/common_utils.rs +++ /dev/null @@ -1,525 +0,0 @@ -use std::net::IpAddr; -use std::time::{Duration, Instant}; - -use base64::Engine; -use base64::engine::general_purpose::STANDARD as BASE64_STANDARD; -use litellm_core::error::Error; -use litellm_core::ocr::transformation::OcrProviderConfig; -use reqwest::Url; -use serde_json::{Map, Value}; - -use litellm_core::providers::azure_ai::ocr::transformation::{ - AZURE_AI_OCR_CONFIG, AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG, -}; -use litellm_core::providers::mistral::ocr::transformation::MISTRAL_OCR_CONFIG; -use litellm_core::providers::vertex_ai::ocr::transformation as vertex_ai; -use litellm_core::providers::vertex_ai::ocr::transformation::VERTEX_AI_DEEPSEEK_OCR_CONFIG; - -use crate::client::http_client; - -const ERROR_BODY_MAX_CHARS: usize = 256; -const AZURE_DOCUMENT_INTELLIGENCE_POLL_TIMEOUT_SECS: u64 = 120; -const DEFAULT_MAX_IMAGE_URL_DOWNLOAD_SIZE_MB: f64 = 50.0; -const MAX_SAFE_FETCH_REDIRECTS: usize = 10; - -pub(super) fn truncate_error_body(body: &str) -> String { - if body.chars().count() <= ERROR_BODY_MAX_CHARS { - return body.to_string(); - } - let truncated: String = body.chars().take(ERROR_BODY_MAX_CHARS).collect(); - format!("{truncated}... (truncated)") -} - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub(super) fn ocr_provider_config( - provider: &str, - model: &str, -) -> Option<&'static dyn OcrProviderConfig> { - match provider { - "mistral" => Some(&MISTRAL_OCR_CONFIG), - "azure_ai" if is_azure_document_intelligence_model(model) => { - Some(&AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG) - } - "azure_ai" => Some(&AZURE_AI_OCR_CONFIG), - "vertex_ai" if vertex_ai::is_deepseek_model(model) => Some(&VERTEX_AI_DEEPSEEK_OCR_CONFIG), - "vertex_ai" => None, - _ => None, - } -} - -fn is_azure_document_intelligence_model(model: &str) -> bool { - let model = model.to_ascii_lowercase(); - model.contains("doc-intelligence") || model.contains("documentintelligence") -} - -pub(super) fn string_headers( - extra_headers: Option>, -) -> Result, Error> { - extra_headers - .unwrap_or_default() - .into_iter() - .map(|(key, value)| { - value - .as_str() - .map(|value| (key.clone(), value.to_string())) - .ok_or_else(|| { - Error::InvalidRequest(format!( - "OCR extra_headers.{key} must be a string, got {}", - litellm_core::error::json_type_name(&value) - )) - }) - }) - .collect() -} - -fn document_url_field(document: &Value) -> Result, Error> { - let Some(object) = document.as_object() else { - return Ok(None); - }; - let Some(doc_type) = object.get("type").and_then(Value::as_str) else { - return Ok(None); - }; - let field = match doc_type { - "document_url" => "document_url", - "image_url" => "image_url", - _ => return Ok(None), - }; - let Some(url) = object.get(field).and_then(Value::as_str) else { - return Ok(None); - }; - Ok(Some((field, url))) -} - -fn is_url_requiring_fetch(url: &str) -> bool { - !url.starts_with("data:") && (url.starts_with("http://") || url.starts_with("https://")) -} - -fn max_document_download_bytes() -> u64 { - let max_size_mb = std::env::var("MAX_IMAGE_URL_DOWNLOAD_SIZE_MB") - .ok() - .and_then(|value| value.parse::().ok()) - .unwrap_or(DEFAULT_MAX_IMAGE_URL_DOWNLOAD_SIZE_MB); - (max_size_mb.max(0.0) * 1024.0 * 1024.0) as u64 -} - -fn is_blocked_ip(ip: IpAddr) -> bool { - match ip { - IpAddr::V4(ip) => { - ip.is_private() - || ip.is_loopback() - || ip.is_link_local() - || ip.is_broadcast() - || ip.is_multicast() - || ip.is_unspecified() - } - IpAddr::V6(ip) => { - let first_segment = ip.segments()[0]; - let is_unique_local = (first_segment & 0xfe00) == 0xfc00; - let is_link_local = (first_segment & 0xffc0) == 0xfe80; - ip.is_loopback() - || ip.is_unspecified() - || ip.is_multicast() - || is_unique_local - || is_link_local - || ip - .to_ipv4_mapped() - .or_else(|| ip.to_ipv4()) - .map(|v4| is_blocked_ip(IpAddr::V4(v4))) - .unwrap_or(false) - } - } -} - -fn blocked_url_error(url: &Url) -> Error { - Error::InvalidRequest(format!( - "OCR document URL rejected by SSRF protection: {url}" - )) -} - -async fn validate_safe_fetch_url(url: &Url) -> Result<(), Error> { - if !matches!(url.scheme(), "http" | "https") { - return Err(blocked_url_error(url)); - } - - let host = url.host_str().ok_or_else(|| blocked_url_error(url))?; - if let Ok(ip) = host.parse::() { - if is_blocked_ip(ip) { - return Err(blocked_url_error(url)); - } - return Ok(()); - } - - let port = url - .port_or_known_default() - .ok_or_else(|| blocked_url_error(url))?; - let addresses = tokio::net::lookup_host((host, port)) - .await - .map_err(|err| Error::Network(err.to_string()))?; - let mut saw_address = false; - for address in addresses { - saw_address = true; - if is_blocked_ip(address.ip()) { - return Err(blocked_url_error(url)); - } - } - if !saw_address { - return Err(blocked_url_error(url)); - } - Ok(()) -} - -fn redirect_location(response: &reqwest::Response, url: &Url) -> Result { - let location = response - .headers() - .get(reqwest::header::LOCATION) - .and_then(|value| value.to_str().ok()) - .ok_or_else(|| { - Error::InvalidResponse("OCR document redirect missing Location header".to_string()) - })?; - url.join(location) - .map_err(|err| Error::InvalidResponse(format!("invalid OCR document redirect: {err}"))) -} - -async fn safe_get_document_url(url: &str) -> Result<(Url, reqwest::Response), Error> { - let client = reqwest::Client::builder() - .redirect(reqwest::redirect::Policy::none()) - .build() - .map_err(|err| Error::Network(err.to_string()))?; - let mut current_url = Url::parse(url) - .map_err(|err| Error::InvalidRequest(format!("invalid OCR document URL: {err}")))?; - - for _ in 0..MAX_SAFE_FETCH_REDIRECTS { - validate_safe_fetch_url(¤t_url).await?; - let response = client - .get(current_url.clone()) - .send() - .await - .map_err(|err| Error::Network(err.to_string()))?; - if !response.status().is_redirection() { - return Ok((current_url, response)); - } - current_url = redirect_location(&response, ¤t_url)?; - } - - Err(Error::InvalidRequest( - "Too many redirects while fetching OCR document URL".to_string(), - )) -} - -fn enforce_download_size(content_length: u64, max_bytes: u64, url: &Url) -> Result<(), Error> { - if max_bytes == 0 { - return Err(Error::InvalidRequest(format!( - "OCR document URL download is disabled (MAX_IMAGE_URL_DOWNLOAD_SIZE_MB=0). url={url}" - ))); - } - if content_length > max_bytes { - let size_mb = content_length as f64 / (1024.0 * 1024.0); - let max_size_mb = max_bytes as f64 / (1024.0 * 1024.0); - return Err(Error::InvalidRequest(format!( - "OCR document size ({size_mb:.2}MB) exceeds maximum allowed size ({max_size_mb:.2}MB). url={url}" - ))); - } - Ok(()) -} - -async fn read_response_with_limit( - mut response: reqwest::Response, - url: &Url, -) -> Result, Error> { - let max_bytes = max_document_download_bytes(); - if let Some(content_length) = response.content_length() { - enforce_download_size(content_length, max_bytes, url)?; - } else { - enforce_download_size(0, max_bytes, url)?; - } - - let mut bytes = Vec::new(); - let mut bytes_downloaded: u64 = 0; - while let Some(chunk) = response - .chunk() - .await - .map_err(|err| Error::Network(err.to_string()))? - { - bytes_downloaded += chunk.len() as u64; - enforce_download_size(bytes_downloaded, max_bytes, url)?; - bytes.extend_from_slice(&chunk); - } - Ok(bytes) -} - -pub(super) async fn convert_document_url_to_data_uri(document: Value) -> Result { - let Some((field, url)) = document_url_field(&document)? else { - return Ok(document); - }; - if !is_url_requiring_fetch(url) { - return Ok(document); - } - - let (final_url, response) = safe_get_document_url(url).await?; - let status = response.status(); - if !status.is_success() { - let body = response.text().await.unwrap_or_default(); - return Err(Error::Http { - status: status.as_u16(), - body: truncate_error_body(&body), - }); - } - let content_type = response - .headers() - .get(reqwest::header::CONTENT_TYPE) - .and_then(|value| value.to_str().ok()) - .and_then(|value| value.split(';').next()) - .map(str::trim) - .filter(|value| !value.is_empty()) - .unwrap_or("application/octet-stream") - .to_string(); - let bytes = read_response_with_limit(response, &final_url).await?; - let data_uri = format!( - "data:{content_type};base64,{}", - BASE64_STANDARD.encode(bytes) - ); - - let mut transformed = document - .as_object() - .cloned() - .ok_or_else(|| Error::InvalidRequest("OCR document must be an object".to_string()))?; - transformed.insert(field.to_string(), Value::String(data_uri)); - Ok(Value::Object(transformed)) -} - -fn same_origin(left: &str, right: &str) -> bool { - let Ok(left) = reqwest::Url::parse(left) else { - return false; - }; - let Ok(right) = reqwest::Url::parse(right) else { - return false; - }; - left.scheme() == right.scheme() - && left.host_str() == right.host_str() - && left.port_or_known_default() == right.port_or_known_default() -} - -fn retry_after_secs(response: &reqwest::Response) -> u64 { - response - .headers() - .get(reqwest::header::RETRY_AFTER) - .and_then(|value| value.to_str().ok()) - .and_then(|value| value.parse::().ok()) - .unwrap_or(2) -} - -fn operation_status(response_json: &Value) -> Result<&str, Error> { - let status = response_json - .get("status") - .and_then(Value::as_str) - .ok_or(Error::MissingField("status"))?; - match status { - "succeeded" => Ok("succeeded"), - "running" | "notStarted" => Ok("running"), - "failed" => { - let message = response_json - .get("error") - .and_then(|error| error.get("message")) - .and_then(Value::as_str) - .unwrap_or("Unknown error"); - Err(Error::InvalidResponse(format!( - "Azure Document Intelligence analysis failed: {message}" - ))) - } - other => Err(Error::InvalidResponse(format!( - "Unknown operation status: {other}" - ))), - } -} - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub(super) async fn poll_document_intelligence( - operation_url: &str, - original_url: &str, - headers: &[(String, String)], - timeout: Option, -) -> Result { - if !same_origin(operation_url, original_url) { - return Err(Error::InvalidResponse( - "Azure Document Intelligence: rejected cross-origin polling URL".to_string(), - )); - } - - let start = Instant::now(); - let timeout = timeout.unwrap_or(Duration::from_secs( - AZURE_DOCUMENT_INTELLIGENCE_POLL_TIMEOUT_SECS, - )); - loop { - if start.elapsed() > timeout { - return Err(Error::Network(format!( - "Azure Document Intelligence operation polling timed out after {} seconds", - timeout.as_secs() - ))); - } - - let mut request_builder = http_client().get(operation_url); - for (key, value) in headers { - if key.eq_ignore_ascii_case("ocp-apim-subscription-key") { - request_builder = request_builder.header(key, value); - } - } - let response = request_builder - .send() - .await - .map_err(|err| Error::Network(err.to_string()))?; - let retry_after = retry_after_secs(&response); - let status = response.status(); - let text = response - .text() - .await - .map_err(|err| Error::Network(err.to_string()))?; - if !status.is_success() { - return Err(Error::Http { - status: status.as_u16(), - body: truncate_error_body(&text), - }); - } - let response_json: Value = serde_json::from_str(&text).map_err(|err| { - Error::InvalidResponse(format!("invalid Azure DI poll response JSON: {err}")) - })?; - if operation_status(&response_json)? == "succeeded" { - return Ok(response_json); - } - tokio::time::sleep(Duration::from_secs(retry_after)).await; - } -} - -#[cfg(test)] -mod tests { - use litellm_core::ocr::transformation::OcrResponseHandling; - use serde_json::json; - - use super::*; - - #[test] - fn blocks_private_and_metadata_ips() { - assert!(is_blocked_ip("127.0.0.1".parse().unwrap())); - assert!(is_blocked_ip("10.0.0.1".parse().unwrap())); - assert!(is_blocked_ip("169.254.169.254".parse().unwrap())); - assert!(is_blocked_ip("::1".parse().unwrap())); - assert!(is_blocked_ip("fd00::1".parse().unwrap())); - assert!(is_blocked_ip("fe80::1".parse().unwrap())); - assert!(is_blocked_ip("::ffff:169.254.169.254".parse().unwrap())); - assert!(is_blocked_ip("::ffff:10.0.0.1".parse().unwrap())); - assert!(!is_blocked_ip("8.8.8.8".parse().unwrap())); - assert!(!is_blocked_ip("::ffff:8.8.8.8".parse().unwrap())); - } - - #[tokio::test] - async fn convert_document_url_rejects_loopback_fetch() { - let error = convert_document_url_to_data_uri(json!({ - "type": "image_url", - "image_url": "http://127.0.0.1/image.png" - })) - .await - .unwrap_err(); - - assert!(matches!( - error, - Error::InvalidRequest(message) - if message.contains("SSRF protection") - )); - } - - #[tokio::test] - async fn convert_document_url_leaves_data_uri_untouched() { - let document = json!({ - "type": "image_url", - "image_url": "data:image/png;base64,abcd" - }); - - let transformed = convert_document_url_to_data_uri(document.clone()) - .await - .unwrap(); - - assert_eq!(transformed, document); - } - - #[test] - fn truncate_error_body_passes_short_strings_through() { - let body = "Unauthorized"; - assert_eq!(truncate_error_body(body), "Unauthorized"); - } - - #[test] - fn truncate_error_body_caps_long_payloads() { - let body = "x".repeat(306); - let truncated = truncate_error_body(&body); - - assert!(truncated.ends_with("... (truncated)")); - let prefix_chars = truncated - .strip_suffix("... (truncated)") - .expect("truncated marker present") - .chars() - .count(); - assert_eq!(prefix_chars, 256); - } - - #[test] - fn truncate_error_body_does_not_split_multibyte_chars() { - let body = "é".repeat(266); - let truncated = truncate_error_body(&body); - assert!(truncated.is_char_boundary(truncated.len())); - } - - #[test] - fn ocr_dispatch_supports_migrated_providers() { - assert!(ocr_provider_config("mistral", "mistral-ocr-latest").is_some()); - assert!( - ocr_provider_config("azure_ai", "pixtral-12b-2409") - .expect("azure ai config resolves") - .requires_data_uri_document() - ); - assert_eq!( - ocr_provider_config("azure_ai", "doc-intelligence/prebuilt-read") - .expect("document intelligence config resolves") - .response_handling(), - OcrResponseHandling::AzureDocumentIntelligencePoll - ); - assert!( - ocr_provider_config("vertex_ai", "deepseek-ocr-maas") - .expect("vertex deepseek config resolves") - .supported_ocr_params() - .contains(&"temperature") - ); - assert!(ocr_provider_config("openai", "gpt-4o").is_none()); - } - - #[test] - fn string_headers_accepts_string_values() { - let headers = json!({ - "x-trace-id": "trace-1" - }) - .as_object() - .unwrap() - .clone(); - - assert_eq!( - string_headers(Some(headers)).expect("string headers accepted"), - vec![("x-trace-id".to_string(), "trace-1".to_string())] - ); - } - - #[test] - fn string_headers_rejects_non_string_values() { - let headers = json!({ - "x-retry-count": 3 - }) - .as_object() - .unwrap() - .clone(); - - let err = string_headers(Some(headers)).expect_err("non-string header rejected"); - assert_eq!( - err, - Error::InvalidRequest( - "OCR extra_headers.x-retry-count must be a string, got number".to_string() - ) - ); - } -} diff --git a/litellm-rust/crates/ai-gateway/src/ocr/handler.rs b/litellm-rust/crates/ai-gateway/src/ocr/handler.rs deleted file mode 100644 index 6c6e12724cd..00000000000 --- a/litellm-rust/crates/ai-gateway/src/ocr/handler.rs +++ /dev/null @@ -1,84 +0,0 @@ -use litellm_core::error::Error; -use litellm_core::http_utils::http_request; -use litellm_core::ocr::transformation::OcrResponseHandling; -use serde_json::Value; - -use super::common_utils::{poll_document_intelligence, truncate_error_body}; -use super::hooks::OcrLifecycleHooks; -use super::types::PreparedOcrRequest; -use crate::client::http_client; - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub(crate) async fn execute_ocr_provider_call( - request: PreparedOcrRequest, - hooks: &OcrLifecycleHooks, -) -> Result { - let request = hooks.prepare_provider_request(request).await?; - let mut request_builder = http_client().post(&request.url).json(&request.body); - for (key, value) in &request.upstream_headers { - request_builder = request_builder.header(key, value); - } - if let Some(duration) = request.timeout { - request_builder = request_builder.timeout(duration); - } - - let response = http_request(request_builder) - .await - .map_err(|err| Error::Network(err.to_string()))?; - - let status = response.status(); - if request.config.response_handling() == OcrResponseHandling::AzureDocumentIntelligencePoll - && status.as_u16() == 202 - { - let operation_url = response - .headers() - .get("operation-location") - .and_then(|value| value.to_str().ok()) - .map(str::to_string) - .ok_or_else(|| { - Error::InvalidResponse( - "Azure Document Intelligence returned 202 but no Operation-Location header found" - .to_string(), - ) - })?; - let response_json = poll_document_intelligence( - &operation_url, - &request.url, - &request.upstream_headers, - request.timeout, - ) - .await?; - return Ok(request - .config - .transform_ocr_response_with_params( - &request.model, - response_json, - &request.optional_params, - )? - .into_json()); - } - - let text = response - .text() - .await - .map_err(|err| Error::Network(err.to_string()))?; - - if !status.is_success() { - return Err(Error::Http { - status: status.as_u16(), - body: truncate_error_body(&text), - }); - } - - let response_json: Value = serde_json::from_str(&text) - .map_err(|err| Error::InvalidResponse(format!("invalid OCR response JSON: {err}")))?; - - Ok(request - .config - .transform_ocr_response_with_params( - &request.model, - response_json, - &request.optional_params, - )? - .into_json()) -} diff --git a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs b/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs deleted file mode 100644 index 3d8246af3f5..00000000000 --- a/litellm-rust/crates/ai-gateway/src/ocr/hooks.rs +++ /dev/null @@ -1,330 +0,0 @@ -use litellm_core::call_lifecycle::{CallLifecycleContext, CallLifecycleHooks, CallLifecycleTiming}; -use litellm_core::error::Error; -use serde_json::{Map, Value, json}; -use std::future::Future; -use std::pin::Pin; - -use super::common_utils::{convert_document_url_to_data_uri, string_headers}; -use super::types::{PreparedOcrRequest, ProviderOcrRequest}; -use crate::integrations::custom_guardrail::{ - CustomGuardrailRunner, GuardrailContext, GuardrailError, GuardrailRequest, -}; -use crate::integrations::custom_logger::{ - CallType, CallbackTiming, CallbackValue, CustomLoggerRunner, LoggingError, ModelCallDetails, -}; -use crate::integrations::types::{ - RequestMetadata, StandardLoggingMetadata, StandardLoggingPayload, -}; - -pub(crate) struct OcrLifecycleHooks { - logger_runner: CustomLoggerRunner, - guardrail_runner: CustomGuardrailRunner, - request_metadata: RequestMetadata, -} - -type OcrFuture<'a, T> = Pin> + Send + 'a>>; -type OcrLogFuture<'a> = Pin + Send + 'a>>; - -impl OcrLifecycleHooks { - pub(crate) fn new( - logger_runner: CustomLoggerRunner, - guardrail_runner: CustomGuardrailRunner, - request_metadata: RequestMetadata, - ) -> Self { - Self { - logger_runner, - guardrail_runner, - request_metadata, - } - } - - async fn run_pre_call_guardrails( - &self, - request: PreparedOcrRequest, - ) -> Result { - if self.guardrail_runner.is_empty() { - return Ok(request); - } - - let context = guardrail_context(&self.request_metadata); - let guardrail_request = GuardrailRequest::new(json!({ - "model": request.model, - "custom_llm_provider": request.custom_llm_provider, - "document": request.document, - "optional_params": request.optional_params, - })); - let (guardrail_request, _) = self - .guardrail_runner - .run_pre_call(&context, guardrail_request) - .await - .map_err(guardrail_error_to_core_error)?; - let (document, optional_params) = parse_ocr_pre_call_guardrail_request(guardrail_request)?; - let optional_params = match &request.config { - Ok(config) => config.map_ocr_params(&optional_params), - Err(_) => optional_params, - }; - Ok(PreparedOcrRequest { - document, - optional_params, - ..request - }) - } - - pub(crate) async fn prepare_provider_request( - &self, - request: PreparedOcrRequest, - ) -> Result { - let config = request.config?; - let env_lookup = |key: &str| std::env::var(key).ok(); - let upstream_headers = config.validate_environment( - string_headers(request.extra_headers)?, - request.api_key.as_deref(), - &env_lookup, - )?; - let url = config.complete_url( - request.api_base.as_deref(), - &request.model, - &request.optional_params, - &env_lookup, - )?; - let model = request.model.clone(); - let custom_llm_provider = request.custom_llm_provider.clone(); - let document = if config.requires_data_uri_document() { - convert_document_url_to_data_uri(request.document).await? - } else { - request.document - }; - let optional_params = request.optional_params; - let body = config - .transform_ocr_request(&request.model, document, optional_params.clone())? - .data; - let body = self - .run_during_call_guardrails(&model, &custom_llm_provider, &url, body) - .await?; - Ok(ProviderOcrRequest { - model, - config, - url, - body, - optional_params, - upstream_headers, - timeout: request.timeout, - }) - } - - async fn run_during_call_guardrails( - &self, - model: &str, - custom_llm_provider: &str, - url: &str, - body: Value, - ) -> Result { - if self.guardrail_runner.is_empty() { - return Ok(body); - } - - let context = guardrail_context(&self.request_metadata); - let guardrail_request = GuardrailRequest::new(json!({ - "model": model, - "custom_llm_provider": custom_llm_provider, - "url": url, - "body": body, - })); - let (guardrail_request, _) = self - .guardrail_runner - .run_during_call(&context, guardrail_request) - .await - .map_err(guardrail_error_to_core_error)?; - parse_ocr_during_call_guardrail_request(guardrail_request) - } - - fn standard_logging_payload( - &self, - context: &CallLifecycleContext, - timing: &CallLifecycleTiming, - ) -> StandardLoggingPayload { - StandardLoggingPayload { - id: context.litellm_call_id.clone(), - litellm_call_id: context.litellm_call_id.clone(), - call_type: context.call_type.clone(), - model: context.model.clone(), - custom_llm_provider: context.custom_llm_provider.clone(), - response_cost: 0.0, - prompt_tokens: 0, - completion_tokens: 0, - total_tokens: 0, - start_time: timing.start_time, - end_time: timing.end_time, - stream: false, - metadata: StandardLoggingMetadata { - user_api_key_hash: self.request_metadata.user_api_key_hash.clone(), - user_api_key_user_id: self.request_metadata.user_api_key_user_id.clone(), - user_api_key_team_id: self.request_metadata.user_api_key_team_id.clone(), - ..Default::default() - }, - messages: None, - } - } -} - -impl CallLifecycleHooks for OcrLifecycleHooks { - type PreCallFuture<'a> = OcrFuture<'a, PreparedOcrRequest>; - type DuringCallFuture<'a> = OcrFuture<'a, PreparedOcrRequest>; - type SuccessFuture<'a> = OcrLogFuture<'a>; - type FailureFuture<'a> = OcrLogFuture<'a>; - - fn async_pre_call_hook<'a>( - &'a self, - _context: &'a CallLifecycleContext, - request: PreparedOcrRequest, - ) -> Self::PreCallFuture<'a> { - Box::pin(async move { self.run_pre_call_guardrails(request).await }) - } - - fn async_during_call_hook<'a>( - &'a self, - _context: &'a CallLifecycleContext, - request: PreparedOcrRequest, - ) -> Self::DuringCallFuture<'a> { - Box::pin(async move { Ok(request) }) - } - - #[tracing::instrument( - name = "success_callback", - target = "litellm::function_trace", - level = "trace", - skip_all - )] - fn async_log_success_event<'a>( - &'a self, - context: &'a CallLifecycleContext, - response: &'a Value, - timing: &'a CallLifecycleTiming, - ) -> Self::SuccessFuture<'a> { - Box::pin(async move { - if self.logger_runner.is_empty() { - return; - } - let response_obj = CallbackValue::new("ocr", response.clone()); - self.logger_runner - .async_log_success_event( - &ModelCallDetails::from_standard_logging_payload( - self.standard_logging_payload(context, timing), - ), - &response_obj, - CallbackTiming::new(timing.start_time, timing.end_time), - ) - .await; - }) - } - - #[tracing::instrument( - name = "failure_callback", - target = "litellm::function_trace", - level = "trace", - skip_all - )] - fn async_log_failure_event<'a>( - &'a self, - context: &'a CallLifecycleContext, - error: &'a Error, - timing: &'a CallLifecycleTiming, - ) -> Self::FailureFuture<'a> { - Box::pin(async move { - if self.logger_runner.is_empty() { - return; - } - let logging_error = LoggingError { - message: error.to_string(), - kind: core_error_kind(error).to_string(), - }; - let response_obj = CallbackValue::new( - "error", - json!({ - "message": logging_error.message, - "kind": logging_error.kind, - }), - ); - self.logger_runner - .async_log_failure_event( - &ModelCallDetails::from_standard_logging_payload( - self.standard_logging_payload(context, timing), - ) - .with_failure_error(logging_error), - Some(&response_obj), - CallbackTiming::new(timing.start_time, timing.end_time), - ) - .await; - }) - } -} - -fn guardrail_context(metadata: &RequestMetadata) -> GuardrailContext { - GuardrailContext { - call_type: CallType::Ocr, - selected_guardrails: Vec::new(), - metadata: std::collections::HashMap::new(), - user_api_key_hash: metadata.user_api_key_hash.clone(), - user_api_key_user_id: metadata.user_api_key_user_id.clone(), - user_api_key_team_id: metadata.user_api_key_team_id.clone(), - trace_parent: None, - } -} - -fn parse_ocr_pre_call_guardrail_request( - request: GuardrailRequest, -) -> Result<(Value, Map), Error> { - let Value::Object(mut data) = request.data else { - return Err(Error::InvalidRequest( - "OCR pre_call guardrail must return an object".to_string(), - )); - }; - let document = data.remove("document").ok_or_else(|| { - Error::InvalidRequest("OCR pre_call guardrail removed document".to_string()) - })?; - let optional_params = match data.remove("optional_params") { - Some(Value::Object(params)) => params, - Some(_) => { - return Err(Error::InvalidRequest( - "OCR pre_call guardrail optional_params must be an object".to_string(), - )); - } - None => Map::new(), - }; - Ok((document, optional_params)) -} - -fn parse_ocr_during_call_guardrail_request(request: GuardrailRequest) -> Result { - let Value::Object(mut data) = request.data else { - return Err(Error::InvalidRequest( - "OCR during_call guardrail must return an object".to_string(), - )); - }; - data.remove("body") - .ok_or_else(|| Error::InvalidRequest("OCR during_call guardrail removed body".to_string())) -} - -fn guardrail_error_to_core_error(error: GuardrailError) -> Error { - Error::InvalidRequest(format!("{}: {}", error.kind, error.message)) -} - -fn core_error_kind(error: &Error) -> &'static str { - match error { - Error::Auth(_) - | Error::MissingApiKey { .. } - | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken - | Error::MissingAzureDocumentIntelligenceCredentials - | Error::MissingReductoApiKey => "AuthError", - Error::InvalidProvider(_) => "InvalidProvider", - Error::InvalidRequest(_) => "InvalidRequest", - Error::InvalidType { .. } => "InvalidType", - Error::MissingField(_) => "MissingField", - Error::Http { .. } => "HttpError", - Error::InvalidResponse(_) => "InvalidResponse", - Error::Network(_) => "NetworkError", - Error::Connect(_) => "ConnectError", - Error::Routing(_) => "RoutingError", - Error::Unsupported(_) => "UnsupportedRequest", - } -} diff --git a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs index 116c0f4a5e9..fb63a02f7ad 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/mod.rs @@ -1,31 +1,96 @@ use litellm_core::Error; -use litellm_core::call_lifecycle::CallLifecycle; +use litellm_core::ocr::{ + OcrClient, + wire::{OcrWireRequest, decode_request}, +}; use serde_json::Value; -mod common_utils; -mod handler; -mod hooks; -mod prepare; mod types; pub use types::OcrRequest; -use handler::execute_ocr_provider_call; -use prepare::{PreparedOcrCall, prepare_ocr_call}; - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] pub async fn ocr(request: OcrRequest<'_>) -> Result { - let PreparedOcrCall { request, hooks } = prepare_ocr_call(request); - CallLifecycle::default() - .run_request(request, &hooks, |request| { - execute_ocr_provider_call(request, &hooks) - }) + core_ocr(request).await +} + +async fn core_ocr(request: OcrRequest<'_>) -> Result { + validate_host_hooks(&request)?; + let client = OcrClient::new(crate::client::http_client().clone())?; + let core_request = decode_request(OcrWireRequest { + model: request.model.to_string(), + document: request.document, + api_key: request.api_key.map(str::to_string), + api_base: request.api_base.map(str::to_string), + custom_llm_provider: request.custom_llm_provider.map(str::to_string), + extra_headers: request.extra_headers, + optional_params: request.optional_params, + input_sources: Default::default(), + timeout_seconds: request.timeout.map(|timeout| timeout.as_secs_f64()), + })?; + client + .perform(core_request) .await + .map(|response| response.into_json()) +} + +fn validate_host_hooks(request: &OcrRequest<'_>) -> Result<(), Error> { + if !request.guardrails.is_empty() { + return Err(Error::Unsupported( + "OCR host guardrails are not wired to the core path", + )); + } + if !request.callbacks.is_empty() { + return Err(Error::Unsupported( + "OCR host callbacks are not wired to the core path", + )); + } + Ok(()) } #[cfg(test)] mod tests { + use std::sync::Arc; + use litellm_core::ocr::wire::is_supported_request; + use serde_json::{Map, json}; + + use super::{OcrRequest, validate_host_hooks}; + use crate::integrations::custom_guardrail::{CustomGuardrail, GuardrailEventHook}; + use crate::integrations::custom_logger::CustomLogger; + + struct TestGuardrail; + + impl CustomGuardrail for TestGuardrail { + fn guardrail_name(&self) -> &str { + "test" + } + + fn supported_event_hooks(&self) -> &[GuardrailEventHook] { + &[] + } + } + + struct TestLogger; + + impl CustomLogger for TestLogger {} + + fn request() -> OcrRequest<'static> { + OcrRequest { + model: "model", + document: json!({"type":"image_url","image_url":"data:image/png;base64,YQ=="}), + api_key: None, + api_base: None, + custom_llm_provider: Some("mistral"), + extra_headers: None, + optional_params: Map::new(), + timeout: None, + callbacks: Vec::new(), + guardrails: Vec::new(), + request_metadata: Default::default(), + litellm_call_id: None, + } + } #[test] fn core_activation_includes_migrated_providers() { @@ -37,6 +102,26 @@ mod tests { )); assert!(is_supported_request("parse-v3", Some("reducto"))); assert!(is_supported_request("mistral-ocr", Some("vertex_ai"))); - assert!(!is_supported_request("deepseek-ocr", Some("vertex_ai"))); + assert!(is_supported_request("deepseek-ocr", Some("vertex_ai"))); + } + + #[test] + fn core_path_rejects_unwired_guardrails() { + let request = OcrRequest { + guardrails: vec![Arc::new(TestGuardrail)], + ..request() + }; + let error = validate_host_hooks(&request).unwrap_err(); + assert!(error.to_string().contains("guardrails are not wired")); + } + + #[test] + fn core_path_rejects_unwired_callbacks() { + let request = OcrRequest { + callbacks: vec![Arc::new(TestLogger)], + ..request() + }; + let error = validate_host_hooks(&request).unwrap_err(); + assert!(error.to_string().contains("callbacks are not wired")); } } diff --git a/litellm-rust/crates/ai-gateway/src/ocr/prepare.rs b/litellm-rust/crates/ai-gateway/src/ocr/prepare.rs deleted file mode 100644 index fa9ca1a193e..00000000000 --- a/litellm-rust/crates/ai-gateway/src/ocr/prepare.rs +++ /dev/null @@ -1,163 +0,0 @@ -use std::sync::atomic::{AtomicU64, Ordering}; -use std::time::{SystemTime, UNIX_EPOCH}; - -use litellm_core::routing_utils::provider::{CustomLlmProvider, get_custom_llm_provider}; -use serde_json::{Map, Value}; - -use super::common_utils::ocr_provider_config; -use super::hooks::OcrLifecycleHooks; -use super::types::{OcrRequest, PreparedOcrRequest}; -use crate::integrations::custom_guardrail::CustomGuardrailRunner; -use crate::integrations::custom_logger::CustomLoggerRunner; - -pub(crate) struct PreparedOcrCall { - pub(crate) request: PreparedOcrRequest, - pub(crate) hooks: OcrLifecycleHooks, -} - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub(crate) fn prepare_ocr_call(request: OcrRequest<'_>) -> PreparedOcrCall { - let call_id = request - .litellm_call_id - .map(str::to_string) - .unwrap_or_else(new_ocr_call_id); - let provider_info = get_custom_llm_provider(request.model, request.custom_llm_provider) - .unwrap_or(CustomLlmProvider { - model: request.model, - custom_llm_provider: "mistral", - }); - let model = provider_info.model.to_string(); - let custom_llm_provider = provider_info.custom_llm_provider.to_string(); - let config = ocr_provider_config(&custom_llm_provider, &model) - .ok_or_else(|| litellm_core::Error::InvalidProvider(custom_llm_provider.clone())) - .and_then(|config| { - validate_request_format(config, &request.optional_params, &custom_llm_provider)?; - Ok(config) - }); - let optional_params = match &config { - Ok(config) => { - let supported = config.supported_ocr_params(); - let mut mapped = config.map_ocr_params( - &request - .optional_params - .iter() - .filter(|(name, _)| supported.contains(&name.as_str())) - .map(|(name, value)| (name.clone(), value.clone())) - .collect(), - ); - for name in [ - "vertex_project", - "vertex_ai_project", - "vertex_location", - "vertex_ai_location", - ] { - if let Some(value) = request.optional_params.get(name) { - mapped.insert(name.to_string(), value.clone()); - } - } - mapped - } - Err(_) => request.optional_params, - }; - - PreparedOcrCall { - request: PreparedOcrRequest { - config, - model, - custom_llm_provider, - litellm_call_id: call_id, - document: request.document, - api_key: request.api_key.map(str::to_string), - api_base: request.api_base.map(str::to_string), - extra_headers: request.extra_headers, - optional_params, - timeout: request.timeout, - }, - hooks: OcrLifecycleHooks::new( - CustomLoggerRunner::new(request.callbacks), - CustomGuardrailRunner::new(request.guardrails), - request.request_metadata, - ), - } -} - -fn validate_request_format( - config: &'static dyn litellm_core::ocr::transformation::OcrProviderConfig, - optional_params: &Map, - provider: &str, -) -> Result<(), litellm_core::Error> { - let Some(format) = optional_params.get("req_format") else { - return Ok(()); - }; - match format.as_str() { - Some("litellm") => Ok(()), - Some("native") if config.supported_ocr_params().contains(&"req_format") => Ok(()), - Some("native") => Err(litellm_core::Error::InvalidRequest(format!( - "`req_format=native` is not supported for provider {provider}" - ))), - _ => Err(litellm_core::Error::InvalidRequest(format!( - "Invalid `req_format`: {format}. Expected `litellm` or `native`" - ))), - } -} - -fn new_ocr_call_id() -> String { - static COUNTER: AtomicU64 = AtomicU64::new(1); - let sequence = COUNTER.fetch_add(1, Ordering::Relaxed); - let timestamp = SystemTime::now() - .duration_since(UNIX_EPOCH) - .map(|duration| duration.as_nanos()) - .unwrap_or(0); - format!("ocr-{timestamp}-{sequence}") -} - -#[cfg(test)] -mod tests { - use litellm_core::error::Error; - use serde_json::{Map, json}; - - use super::{OcrRequest, prepare_ocr_call}; - use crate::integrations::types::RequestMetadata; - - fn base_ocr_request(model: &str) -> OcrRequest<'_> { - OcrRequest { - model, - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-test"), - api_base: None, - custom_llm_provider: None, - extra_headers: None, - optional_params: Map::new(), - timeout: None, - callbacks: Vec::new(), - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: None, - } - } - - fn request_with_format(format: &str) -> OcrRequest<'_> { - let mut request = base_ocr_request("mistral/mistral-ocr-latest"); - request.optional_params = Map::from_iter([("req_format".to_string(), json!(format))]); - request - } - - #[test] - fn native_format_rejected_for_provider_without_support_as_bad_request() { - let prepared = prepare_ocr_call(request_with_format("native")); - assert!( - matches!(prepared.request.config, Err(Error::InvalidRequest(message)) if message.contains("not supported for provider")) - ); - } - - #[test] - fn unknown_format_rejected_for_provider_without_support_as_bad_request() { - let prepared = prepare_ocr_call(request_with_format("raw")); - assert!( - matches!(prepared.request.config, Err(Error::InvalidRequest(message)) if message.contains("Invalid `req_format`")) - ); - } -} diff --git a/litellm-rust/crates/ai-gateway/src/ocr/types.rs b/litellm-rust/crates/ai-gateway/src/ocr/types.rs index 75a8e61ddbf..e96d2df1adb 100644 --- a/litellm-rust/crates/ai-gateway/src/ocr/types.rs +++ b/litellm-rust/crates/ai-gateway/src/ocr/types.rs @@ -1,8 +1,6 @@ use std::sync::Arc; use std::time::Duration; -use litellm_core::call_lifecycle::{CallLifecycleContext, CallLifecycleRequest}; -use litellm_core::ocr::transformation::OcrProviderConfig; use serde_json::{Map, Value}; use crate::integrations::custom_guardrail::CustomGuardrail; @@ -23,37 +21,3 @@ pub struct OcrRequest<'a> { pub request_metadata: RequestMetadata, pub litellm_call_id: Option<&'a str>, } - -pub(crate) struct PreparedOcrRequest { - pub(crate) config: Result<&'static dyn OcrProviderConfig, litellm_core::Error>, - pub(crate) model: String, - pub(crate) custom_llm_provider: String, - pub(crate) litellm_call_id: String, - pub(crate) document: Value, - pub(crate) api_key: Option, - pub(crate) api_base: Option, - pub(crate) extra_headers: Option>, - pub(crate) optional_params: Map, - pub(crate) timeout: Option, -} - -impl CallLifecycleRequest for PreparedOcrRequest { - fn lifecycle_context(&self) -> CallLifecycleContext { - CallLifecycleContext::new( - "ocr", - self.model.clone(), - self.custom_llm_provider.clone(), - self.litellm_call_id.clone(), - ) - } -} - -pub(crate) struct ProviderOcrRequest { - pub(crate) model: String, - pub(crate) config: &'static dyn OcrProviderConfig, - pub(crate) url: String, - pub(crate) body: Value, - pub(crate) optional_params: Map, - pub(crate) upstream_headers: Vec<(String, String)>, - pub(crate) timeout: Option, -} diff --git a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs index 9707e9f2611..39465e28e84 100644 --- a/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs +++ b/litellm-rust/crates/ai-gateway/src/routes/messages/mod.rs @@ -105,7 +105,11 @@ impl IntoResponse for MessagesRouteError { StatusCode::NOT_FOUND, "no messages deployment is configured for this model".to_string(), ), - Error::Auth(_) => ( + Error::Auth(_) + | Error::MissingApiKey { .. } + | Error::MissingAzureAiCredentials + | Error::MissingAzureDocumentIntelligenceCredentials + | Error::MissingReductoApiKey => ( StatusCode::BAD_GATEWAY, "messages provider authentication failed".to_string(), ), @@ -114,12 +118,7 @@ impl IntoResponse for MessagesRouteError { | Error::Connect(_) | Error::InvalidResponse(_) | Error::InvalidType { .. } - | Error::MissingField(_) - | Error::MissingApiKey { .. } - | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken - | Error::MissingAzureDocumentIntelligenceCredentials - | Error::MissingReductoApiKey => ( + | Error::MissingField(_) => ( StatusCode::BAD_GATEWAY, "messages provider request failed".to_string(), ), diff --git a/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs b/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs deleted file mode 100644 index 2fbd25d986f..00000000000 --- a/litellm-rust/crates/ai-gateway/tests/ocr_lifecycle.rs +++ /dev/null @@ -1,603 +0,0 @@ -use std::sync::{Arc, Mutex}; -use std::time::Duration; - -use litellm_ai_gateway::integrations::custom_guardrail::{ - CustomGuardrail, GuardrailContext, GuardrailDecision, GuardrailError, GuardrailEventHook, - GuardrailFuture, GuardrailRequest, -}; -use litellm_ai_gateway::integrations::custom_logger::{ - CallbackTiming, CallbackValue, CustomLogger, LogFuture, ModelCallDetails, -}; -use litellm_ai_gateway::integrations::types::RequestMetadata; -use litellm_ai_gateway::ocr::{OcrRequest, ocr}; -use litellm_core::error::Error; -#[cfg(feature = "trace-parity")] -use litellm_core::observability::FunctionTrace; -use serde_json::{Map, Value, json}; -use tokio::io::{AsyncReadExt, AsyncWriteExt}; -use tokio::net::{TcpListener, TcpStream}; -#[cfg(feature = "trace-parity")] -use tracing::instrument::WithSubscriber; - -async fn read_http_headers(socket: &mut TcpStream) -> String { - let mut request = Vec::new(); - let mut buffer = [0_u8; 1024]; - loop { - let n = socket.read(&mut buffer).await.expect("reads request"); - if n == 0 { - break; - } - request.extend_from_slice(&buffer[..n]); - if request.windows(4).any(|window| window == b"\r\n\r\n") { - break; - } - } - String::from_utf8(request).expect("request is utf8") -} - -async fn read_http_request(socket: &mut TcpStream) -> String { - let mut request = Vec::new(); - let mut buffer = [0_u8; 1024]; - let header_end = loop { - let n = socket.read(&mut buffer).await.expect("reads request"); - if n == 0 { - break request.len(); - } - request.extend_from_slice(&buffer[..n]); - if let Some(position) = request.windows(4).position(|window| window == b"\r\n\r\n") { - break position + 4; - } - }; - let headers = String::from_utf8_lossy(&request[..header_end]); - let content_length = headers - .lines() - .find_map(|line| { - let (name, value) = line.split_once(':')?; - name.eq_ignore_ascii_case("content-length") - .then(|| value.trim().parse::().ok()) - .flatten() - }) - .unwrap_or(0); - while request.len().saturating_sub(header_end) < content_length { - let n = socket.read(&mut buffer).await.expect("reads body"); - if n == 0 { - break; - } - request.extend_from_slice(&buffer[..n]); - } - String::from_utf8(request).expect("request is utf8") -} - -#[derive(Clone, Debug, PartialEq)] -struct RecordedLogEvent { - hook: &'static str, - model: String, - call_type: String, - user_id: Option, - response_object: Option, - error_kind: Option, -} - -#[derive(Default)] -struct RecordingOcrLogger { - events: Mutex>, -} - -impl RecordingOcrLogger { - fn events(&self) -> Vec { - self.events.lock().unwrap().clone() - } -} - -impl CustomLogger for RecordingOcrLogger { - fn async_log_success_event<'a>( - &'a self, - model_call_details: &'a ModelCallDetails, - response_obj: &'a CallbackValue, - _timing: CallbackTiming, - ) -> LogFuture<'a> { - Box::pin(async move { - self.events.lock().unwrap().push(RecordedLogEvent { - hook: "async_log_success_event", - model: model_call_details.model.clone(), - call_type: model_call_details.call_type.to_string(), - user_id: model_call_details.metadata.user_api_key_user_id.clone(), - response_object: Some(response_obj.object.clone()), - error_kind: None, - }); - Ok(()) - }) - } - - fn async_log_failure_event<'a>( - &'a self, - model_call_details: &'a ModelCallDetails, - response_obj: Option<&'a CallbackValue>, - _timing: CallbackTiming, - ) -> LogFuture<'a> { - Box::pin(async move { - self.events.lock().unwrap().push(RecordedLogEvent { - hook: "async_log_failure_event", - model: model_call_details.model.clone(), - call_type: model_call_details.call_type.to_string(), - user_id: model_call_details.metadata.user_api_key_user_id.clone(), - response_object: response_obj.map(|value| value.object.clone()), - error_kind: model_call_details - .failure_error - .as_ref() - .map(|error| error.kind.clone()), - }); - Ok(()) - }) - } -} - -struct RecordingOcrGuardrail { - hooks: Vec, - events: Mutex>, - block_pre_call: bool, - block_during_call: bool, -} - -impl RecordingOcrGuardrail { - fn new(hooks: Vec) -> Self { - Self { - hooks, - events: Mutex::new(Vec::new()), - block_pre_call: false, - block_during_call: false, - } - } - - fn blocking_pre_call() -> Self { - Self { - hooks: vec![GuardrailEventHook::PreCall], - events: Mutex::new(Vec::new()), - block_pre_call: true, - block_during_call: false, - } - } - - fn events(&self) -> Vec<&'static str> { - self.events.lock().unwrap().clone() - } -} - -impl CustomGuardrail for RecordingOcrGuardrail { - fn guardrail_name(&self) -> &str { - "recording-ocr-guardrail" - } - - fn supported_event_hooks(&self) -> &[GuardrailEventHook] { - &self.hooks - } - - fn async_pre_call_hook<'a>( - &'a self, - _context: &'a GuardrailContext, - mut request: GuardrailRequest, - ) -> GuardrailFuture<'a> { - Box::pin(async move { - self.events.lock().unwrap().push("async_pre_call_hook"); - if self.block_pre_call { - return Ok(GuardrailDecision::Block(GuardrailError::blocked( - "blocked before provider", - ))); - } - request.data["document"]["guarded_pre"] = json!(true); - Ok(GuardrailDecision::Mask(request)) - }) - } - - fn async_moderation_hook<'a>( - &'a self, - _context: &'a GuardrailContext, - mut request: GuardrailRequest, - ) -> GuardrailFuture<'a> { - Box::pin(async move { - self.events.lock().unwrap().push("async_moderation_hook"); - if self.block_during_call { - return Ok(GuardrailDecision::Block(GuardrailError::blocked( - "blocked before provider", - ))); - } - request.data["body"]["guarded_during"] = json!(true); - Ok(GuardrailDecision::Mask(request)) - }) - } -} - -#[tokio::test] -async fn azure_mistral_uses_prepared_authorization_through_gateway() { - let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let api_base = format!("http://{}", listener.local_addr().unwrap()); - let server = tokio::spawn(async move { - let (mut socket, _) = listener.accept().await.unwrap(); - let request = read_http_request(&mut socket).await; - let body = br#"{"pages":[]}"#; - socket - .write_all( - format!( - "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n", - body.len() - ) - .as_bytes(), - ) - .await - .unwrap(); - socket.write_all(body).await.unwrap(); - request - }); - let request = OcrRequest { - model: "mistral-ocr-2505", - document: json!({ - "type":"document_url", - "document_url":"data:application/pdf;base64,YWJj" - }), - api_key: None, - api_base: Some(&api_base), - custom_llm_provider: Some("azure_ai"), - extra_headers: Some(Map::from_iter([( - "Authorization".into(), - json!("Bearer python-prepared-token"), - )])), - optional_params: Map::new(), - timeout: None, - callbacks: Vec::new(), - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: None, - }; - - ocr(request).await.unwrap(); - let sent = server.await.unwrap(); - assert!(sent.starts_with("POST /providers/mistral/azure/ocr ")); - assert!( - sent.to_ascii_lowercase() - .contains("authorization: bearer python-prepared-token\r\n") - ); -} - -#[tokio::test] -async fn ocr_lifecycle_runs_pre_during_and_success_hooks() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let addr = listener.local_addr().expect("listener has local addr"); - - let server = tokio::spawn(async move { - let (mut socket, _) = listener.accept().await.expect("accepts one request"); - let request = read_http_request(&mut socket).await; - let response_body = r#"{"pages":[{"index":0,"markdown":"ok"}],"model":"mistral-ocr-latest","usage_info":{"pages_processed":1}}"#; - let response = format!( - "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - response_body.len(), - response_body - ); - socket - .write_all(response.as_bytes()) - .await - .expect("writes response"); - request - }); - - let logger = Arc::new(RecordingOcrLogger::default()); - let guardrail = Arc::new(RecordingOcrGuardrail::new(vec![ - GuardrailEventHook::PreCall, - GuardrailEventHook::DuringCall, - ])); - #[cfg(feature = "trace-parity")] - let trace = FunctionTrace::default(); - let api_base = format!("http://{addr}"); - let call = ocr(OcrRequest { - model: "mistral-ocr-latest", - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-test"), - api_base: Some(&api_base), - custom_llm_provider: Some("mistral"), - extra_headers: None, - optional_params: Map::new(), - timeout: Some(Duration::from_secs(5)), - callbacks: vec![logger.clone()], - guardrails: vec![guardrail.clone()], - request_metadata: RequestMetadata { - user_api_key_user_id: Some("user-1".to_string()), - ..Default::default() - }, - litellm_call_id: Some("ocr-call-1"), - }); - #[cfg(feature = "trace-parity")] - let call = call.with_subscriber(trace.dispatcher()); - let response = call.await.expect("ocr request succeeds"); - - assert_eq!(response["pages"][0]["markdown"], "ok"); - assert_eq!( - guardrail.events(), - vec!["async_pre_call_hook", "async_moderation_hook"] - ); - assert_eq!( - logger.events(), - vec![RecordedLogEvent { - hook: "async_log_success_event", - model: "mistral-ocr-latest".to_string(), - call_type: "ocr".to_string(), - user_id: Some("user-1".to_string()), - response_object: Some("ocr".to_string()), - error_kind: None, - }] - ); - #[cfg(feature = "trace-parity")] - assert_eq!( - trace - .events() - .iter() - .filter(|event| event.function.ends_with("_callback")) - .map(|event| event.function) - .collect::>(), - vec!["success_callback"] - ); - - let request = server.await.expect("server task completes"); - assert!(request.contains(r#""guarded_pre":true"#), "{request}"); - assert!(request.contains(r#""guarded_during":true"#), "{request}"); -} - -#[tokio::test] -async fn ocr_lifecycle_runs_failure_hook_on_provider_error() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let addr = listener.local_addr().expect("listener has local addr"); - - let server = tokio::spawn(async move { - let (mut socket, _) = listener.accept().await.expect("accepts one request"); - let _request = read_http_request(&mut socket).await; - let response_body = "provider failed"; - let response = format!( - "HTTP/1.1 500 Internal Server Error\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - response_body.len(), - response_body - ); - socket - .write_all(response.as_bytes()) - .await - .expect("writes response"); - }); - - let logger = Arc::new(RecordingOcrLogger::default()); - #[cfg(feature = "trace-parity")] - let trace = FunctionTrace::default(); - let api_base = format!("http://{addr}"); - let call = ocr(OcrRequest { - model: "mistral-ocr-latest", - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-test"), - api_base: Some(&api_base), - custom_llm_provider: Some("mistral"), - extra_headers: None, - optional_params: Map::new(), - timeout: Some(Duration::from_secs(5)), - callbacks: vec![logger.clone()], - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: Some("ocr-call-2"), - }); - #[cfg(feature = "trace-parity")] - let call = call.with_subscriber(trace.dispatcher()); - let err = call.await.expect_err("provider error propagates"); - - assert!(matches!(err, Error::Http { status: 500, .. })); - server.await.expect("server task completes"); - assert_eq!( - logger.events(), - vec![RecordedLogEvent { - hook: "async_log_failure_event", - model: "mistral-ocr-latest".to_string(), - call_type: "ocr".to_string(), - user_id: None, - response_object: Some("error".to_string()), - error_kind: Some("HttpError".to_string()), - }] - ); - #[cfg(feature = "trace-parity")] - assert_eq!( - trace - .events() - .iter() - .filter(|event| event.function.ends_with("_callback")) - .map(|event| event.function) - .collect::>(), - vec!["failure_callback"] - ); -} - -#[tokio::test] -async fn ocr_lifecycle_pre_call_block_skips_provider_socket() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let addr = listener.local_addr().expect("listener has local addr"); - let logger = Arc::new(RecordingOcrLogger::default()); - let guardrail = Arc::new(RecordingOcrGuardrail::blocking_pre_call()); - - let err = ocr(OcrRequest { - model: "mistral-ocr-latest", - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-test"), - api_base: Some(&format!("http://{addr}")), - custom_llm_provider: Some("mistral"), - extra_headers: None, - optional_params: Map::new(), - timeout: Some(Duration::from_millis(100)), - callbacks: vec![logger.clone()], - guardrails: vec![guardrail.clone()], - request_metadata: RequestMetadata::default(), - litellm_call_id: Some("ocr-call-3"), - }) - .await - .expect_err("guardrail blocks request"); - - assert!(matches!(err, Error::InvalidRequest(_))); - assert_eq!(guardrail.events(), vec!["async_pre_call_hook"]); - assert_eq!( - logger.events(), - vec![RecordedLogEvent { - hook: "async_log_failure_event", - model: "mistral-ocr-latest".to_string(), - call_type: "ocr".to_string(), - user_id: None, - response_object: Some("error".to_string()), - error_kind: Some("InvalidRequest".to_string()), - }] - ); - let accepted = tokio::time::timeout(Duration::from_millis(100), listener.accept()).await; - assert!(accepted.is_err(), "provider socket should not be touched"); -} - -#[tokio::test] -async fn ocr_does_not_duplicate_authorization_header_when_header_is_supplied() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let addr = listener.local_addr().expect("listener has local addr"); - - let server = tokio::spawn(async move { - let (mut socket, _) = listener.accept().await.expect("accepts one request"); - let request = read_http_headers(&mut socket).await; - let response_body = r#"{"pages":[{"index":0,"markdown":"ok"}],"model":"mistral-ocr-latest","usage_info":{"pages_processed":1}}"#; - let response = format!( - "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - response_body.len(), - response_body - ); - socket - .write_all(response.as_bytes()) - .await - .expect("writes response"); - request - }); - - let mut headers = Map::new(); - headers.insert( - "Authorization".to_string(), - Value::String("Bearer sk-from-python".to_string()), - ); - headers.insert( - "x-trace-id".to_string(), - Value::String("trace-1".to_string()), - ); - - let response = ocr(OcrRequest { - model: "mistral-ocr-latest", - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("sk-for-rust-fallback"), - api_base: Some(&format!("http://{addr}")), - custom_llm_provider: Some("mistral"), - extra_headers: Some(headers), - optional_params: Map::new(), - timeout: Some(Duration::from_secs(5)), - callbacks: Vec::new(), - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: None, - }) - .await - .expect("ocr request succeeds"); - - assert_eq!(response["pages"][0]["markdown"], "ok"); - - let request = server.await.expect("server task completes"); - let authorization_count = request - .lines() - .filter(|line| line.to_ascii_lowercase().starts_with("authorization:")) - .count(); - assert_eq!(authorization_count, 1, "{request}"); - assert!( - request.contains("authorization: Bearer sk-from-python") - || request.contains("Authorization: Bearer sk-from-python"), - "{request}" - ); -} - -#[tokio::test] -async fn document_intelligence_poll_uses_resolved_subscription_key() { - let listener = TcpListener::bind("127.0.0.1:0") - .await - .expect("test listener binds"); - let addr = listener.local_addr().expect("listener has local addr"); - let operation_url = format!("http://{addr}/operations/1"); - - let server = tokio::spawn(async move { - let (mut post_socket, _) = listener.accept().await.expect("accepts post request"); - let post_request = read_http_headers(&mut post_socket).await; - let post_response = format!( - "HTTP/1.1 202 Accepted\r\noperation-location: {operation_url}\r\ncontent-length: 0\r\nconnection: close\r\n\r\n" - ); - post_socket - .write_all(post_response.as_bytes()) - .await - .expect("writes post response"); - - let (mut poll_socket, _) = listener.accept().await.expect("accepts poll request"); - let poll_request = read_http_headers(&mut poll_socket).await; - let response_body = r#"{"status":"succeeded","analyzeResult":{"pages":[{"pageNumber":1,"lines":[{"content":"ok"}]}]}}"#; - let poll_response = format!( - "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{}", - response_body.len(), - response_body - ); - poll_socket - .write_all(poll_response.as_bytes()) - .await - .expect("writes poll response"); - (post_request, poll_request) - }); - - let response = ocr(OcrRequest { - model: "doc-intelligence/prebuilt-read", - document: json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }), - api_key: Some("di-key"), - api_base: Some(&format!("http://{addr}")), - custom_llm_provider: Some("azure_ai"), - extra_headers: None, - optional_params: Map::new(), - timeout: Some(Duration::from_secs(5)), - callbacks: Vec::new(), - guardrails: Vec::new(), - request_metadata: RequestMetadata::default(), - litellm_call_id: None, - }) - .await - .expect("document intelligence request succeeds"); - - assert_eq!(response["pages"][0]["markdown"], "ok"); - - let (post_request, poll_request) = server.await.expect("server task completes"); - assert!( - post_request - .to_ascii_lowercase() - .contains("ocp-apim-subscription-key: di-key"), - "{post_request}" - ); - assert!( - poll_request - .to_ascii_lowercase() - .contains("ocp-apim-subscription-key: di-key"), - "{poll_request}" - ); -} diff --git a/litellm-rust/crates/core/src/error.rs b/litellm-rust/crates/core/src/error.rs index 2a4cbad96c0..fa4a9d36e03 100644 --- a/litellm-rust/crates/core/src/error.rs +++ b/litellm-rust/crates/core/src/error.rs @@ -25,8 +25,6 @@ pub enum Error { "invalid authentication configuration: Missing Azure AI credentials - set AZURE_AI_API_KEY or configure Entra ID" )] MissingAzureAiCredentials, - #[error("Missing Azure AI credentials - set AZURE_AI_API_KEY or provide azure_ad_token")] - MissingAzureAiCredentialsOrAdToken, #[error( "invalid authentication configuration: Missing Azure Document Intelligence credentials - set AZURE_DOCUMENT_INTELLIGENCE_API_KEY or configure Entra ID" )] diff --git a/litellm-rust/crates/core/src/ocr/adapters/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/mod.rs index a96a7fcdf38..9171d11836c 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/mod.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/mod.rs @@ -16,7 +16,7 @@ mod vertex; pub(crate) use azure::{AzureDocumentIntelligenceAdapter, AzureMistralAdapter}; pub(crate) use mistral::MistralAdapter; pub(crate) use reducto::{ReductoLegacyAdapter, ReductoV3Adapter}; -pub(crate) use vertex::VertexMistralAdapter; +pub(crate) use vertex::{VertexDeepSeekAdapter, VertexMistralAdapter}; /// Converts a complete LiteLLM OCR call to provider HTTP and normalizes its response. pub(crate) trait OcrAdapter: Send + Sync + Sized + 'static { @@ -73,6 +73,7 @@ macro_rules! for_each_ocr_adapter { ReductoLegacy, $crate::ocr::adapters::ReductoLegacyAdapter, $crate::ocr::adapters::ReductoLegacyAdapter, Reducto; ReductoV3, $crate::ocr::adapters::ReductoV3Adapter, $crate::ocr::adapters::ReductoV3Adapter, Reducto; VertexMistral, $crate::ocr::adapters::VertexMistralAdapter, $crate::ocr::adapters::VertexMistralAdapter, VertexAi; + VertexDeepSeek, $crate::ocr::adapters::VertexDeepSeekAdapter, $crate::ocr::adapters::VertexDeepSeekAdapter, VertexAi; } }; } diff --git a/litellm-rust/crates/core/src/ocr/adapters/vertex/deepseek.rs b/litellm-rust/crates/core/src/ocr/adapters/vertex/deepseek.rs new file mode 100644 index 00000000000..ef188f8b9ac --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/adapters/vertex/deepseek.rs @@ -0,0 +1,134 @@ +use super::super::OcrAdapter; +use super::validate_destination; +use crate::Error; +use crate::auth::vertex::{self, VertexConfig}; +use crate::ocr::OcrClient; +use crate::ocr::codecs::deepseek::{self, DeepSeekOcrParams, DeepSeekOcrResponse}; +use crate::ocr::error::{OcrError, OcrRequestError, OcrResponseError}; +use crate::ocr::prepare::{ + _prepare_ocr_request, ParsedProviderParams, credential_env, transform_request_body, +}; +use crate::ocr::registry::OcrProvider; +use crate::ocr::types::{LiteLLMOcrRequest, LiteLLMOcrResponse}; +use crate::url_utils::ApiUrl; +const DEFAULT_API_BASE: &str = "https://aiplatform.googleapis.com"; +const MODEL_NAMESPACE: &str = "deepseek-ai"; +const DEFAULT_LOCATION: &str = "us-central1"; + +#[derive(Clone, Debug)] +pub(crate) struct VertexDeepSeekAdapter; + +impl OcrAdapter for VertexDeepSeekAdapter { + type ProviderResponse = DeepSeekOcrResponse; + const PROVIDER: OcrProvider = OcrProvider::VertexAi; + + async fn prepare_request( + &self, + request: &LiteLLMOcrRequest, + client: &OcrClient, + ) -> Result { + validate_destination(&request.connection)?; + let ParsedProviderParams { + known: params, + extra_params: _extra_params, + } = _prepare_ocr_request::(request)?; + let config = VertexConfig::from_sourced_optional_params( + &request.optional_params, + &request.input_sources, + ) + .map_err(Error::from)?; + let authentication = client + .vertex_auth() + .validate_environment( + request.connection.extra_headers.clone(), + request.connection.api_key.as_deref(), + &config, + &credential_env, + ) + .await + .map_err(Error::from)?; + let location = vertex::get_vertex_ai_location(&config, &credential_env) + .unwrap_or_else(|| DEFAULT_LOCATION.to_string()); + let url = get_complete_url( + request.connection.api_base.as_deref(), + &authentication.project_id, + &location, + )?; + let document = request.document.clone(); + let body = + deepseek::transform_ocr_request(&provider_model(&request.model), document, ¶ms)?; + transform_request_body(client, request, &url, &authentication.headers, body, |_| { + Ok(()) + }) + .await + } + + fn transform_ocr_response( + &self, + request: &LiteLLMOcrRequest, + response: Self::ProviderResponse, + ) -> Result { + deepseek::transform_ocr_response(&request.model, response) + } +} + +fn provider_model(model: &str) -> String { + if model.starts_with(&format!("{MODEL_NAMESPACE}/")) { + model.to_string() + } else { + format!("{MODEL_NAMESPACE}/{model}") + } +} + +fn get_complete_url( + api_base: Option<&str>, + project: &str, + location: &str, +) -> Result { + let base = api_base + .map(str::trim) + .filter(|base| !base.is_empty()) + .unwrap_or(DEFAULT_API_BASE); + ApiUrl::parse(base) + .and_then(|url| { + url.complete_path(&[ + "v1", + "projects", + project, + "locations", + location, + "endpoints", + "openapi", + "chat", + "completions", + ]) + }) + .map(|url| url.into_string()) + .map_err(|_| { + OcrRequestError::RequestField { + path: "api_base".into(), + } + .into() + }) +} + +#[cfg(test)] +mod tests { + use super::{get_complete_url, provider_model}; + + #[test] + fn adapter_owns_model_namespace_and_endpoint() { + assert_eq!( + provider_model("deepseek-ocr-maas"), + "deepseek-ai/deepseek-ocr-maas" + ); + assert_eq!( + provider_model("deepseek-ai/deepseek-ocr-maas"), + "deepseek-ai/deepseek-ocr-maas" + ); + assert_eq!( + get_complete_url(None, "proj-1", "europe-west4").unwrap(), + "https://aiplatform.googleapis.com/v1/projects/proj-1/locations/europe-west4/endpoints/openapi/chat/completions" + ); + } +} diff --git a/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs b/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs index ce6f884b41d..270c41e647d 100644 --- a/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs +++ b/litellm-rust/crates/core/src/ocr/adapters/vertex/mod.rs @@ -1,3 +1,4 @@ +mod deepseek; mod mistral; use crate::Error; @@ -6,6 +7,7 @@ use crate::auth::error::AuthConfigurationError; use crate::ocr::error::OcrError; use crate::ocr::types::OcrConnection; +pub(crate) use deepseek::VertexDeepSeekAdapter; pub(crate) use mistral::VertexMistralAdapter; fn validate_destination(connection: &OcrConnection) -> Result<(), OcrError> { diff --git a/litellm-rust/crates/core/src/ocr/codecs/deepseek/mod.rs b/litellm-rust/crates/core/src/ocr/codecs/deepseek/mod.rs new file mode 100644 index 00000000000..682b3addde7 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/deepseek/mod.rs @@ -0,0 +1,5 @@ +mod transformation; +mod types; + +pub(crate) use transformation::{transform_ocr_request, transform_ocr_response}; +pub(crate) use types::{DeepSeekOcrParams, DeepSeekOcrResponse}; diff --git a/litellm-rust/crates/core/src/ocr/codecs/deepseek/transformation.rs b/litellm-rust/crates/core/src/ocr/codecs/deepseek/transformation.rs new file mode 100644 index 00000000000..98cfc0db78d --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/deepseek/transformation.rs @@ -0,0 +1,98 @@ +use serde::de::IntoDeserializer; +use serde_json::{Value, json}; + +use super::types::*; +use crate::ocr::error::{OcrRequestError, OcrResponseError}; +use crate::ocr::types::{LiteLLMOcrResponse, OcrDocument}; + +#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] +pub(crate) fn transform_ocr_request( + provider_model: &str, + document: OcrDocument, + params: &DeepSeekOcrParams, +) -> Result { + if document.source().is_empty() { + return Err(OcrRequestError::MissingField("document URL")); + } + Ok(DeepSeekOcrRequest { + model: provider_model.to_string(), + messages: vec![DeepSeekOcrMessage { + role: UserRole::User, + content: vec![document], + }], + params: params.clone(), + }) +} + +pub(crate) fn transform_ocr_response( + model: &str, + response: DeepSeekOcrResponse, +) -> Result { + let content = response + .choices + .into_iter() + .next() + .and_then(|choice| choice.message.content) + .ok_or(OcrResponseError::EmptyContent)?; + let decoded = decode_content(content)?; + let pages = match decoded.result.pages { + Some(pages) if !pages.is_empty() => pages + .into_iter() + .map(|page| serde_json::to_value(page).expect("DeepSeek page serializes")) + .collect(), + _ => vec![json!({ + "index":0, + "markdown":decoded.fallback_markdown, + "images":null + })], + }; + Ok(LiteLLMOcrResponse { + pages, + model: decoded.result.model.unwrap_or_else(|| model.to_string()), + document_annotation: decoded.result.document_annotation, + usage_info: decoded.result.usage_info.or(response.usage), + object: "ocr".into(), + extra_fields: decoded.result.extra_fields, + provider_native_response: None, + }) +} + +struct DecodedContent { + result: DeepSeekOcrResult, + fallback_markdown: String, +} + +fn decode_content(content: DeepSeekContent) -> Result { + let (result, fallback_markdown) = match content { + DeepSeekContent::Text(text) if text.is_empty() => { + return Err(OcrResponseError::EmptyContent); + } + DeepSeekContent::Text(text) => (decode_json_content(&text)?, text), + DeepSeekContent::Object(object) => { + let fallback = + serde_json::to_string(&object).map_err(|_| OcrResponseError::ResponseField { + path: "choices[0].message.content".into(), + })?; + (Some(object), fallback) + } + }; + Ok(DecodedContent { + result: result.unwrap_or_default(), + fallback_markdown, + }) +} + +fn decode_json_content(text: &str) -> Result, OcrResponseError> { + if !text.trim_start().starts_with('{') { + return Ok(None); + } + let value = match serde_json::from_str::(text) { + Ok(value) => value, + Err(_) => return Ok(None), + }; + serde_path_to_error::deserialize(value.into_deserializer()) + .map(Some) + .map_err(|error| OcrResponseError::ResponseField { + path: format!("choices[0].message.content.{}", error.path()), + }) +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/deepseek/types.rs b/litellm-rust/crates/core/src/ocr/codecs/deepseek/types.rs new file mode 100644 index 00000000000..0ce2d9913f7 --- /dev/null +++ b/litellm-rust/crates/core/src/ocr/codecs/deepseek/types.rs @@ -0,0 +1,95 @@ +use serde::{Deserialize, Serialize}; +use serde_json::{Map, Value}; + +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub(crate) struct DeepSeekOcrParams { + #[serde(skip_serializing_if = "Option::is_none")] + pub stream: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub temperature: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub max_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub top_p: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub n: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub stop: Option, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(untagged)] +pub(crate) enum StopSequences { + One(String), + Many(Vec), +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct DeepSeekOcrRequest { + pub model: String, + pub messages: Vec, + #[serde(flatten)] + pub params: DeepSeekOcrParams, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct DeepSeekOcrMessage { + pub role: UserRole, + pub content: Vec, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub(crate) enum UserRole { + User, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct DeepSeekOcrResponse { + #[serde(default)] + pub choices: Vec, + pub usage: Option, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct DeepSeekChoice { + pub message: DeepSeekResponseMessage, +} + +#[derive(Clone, Debug, Deserialize)] +pub(crate) struct DeepSeekResponseMessage { + pub content: Option, +} + +#[derive(Clone, Debug, Deserialize)] +#[serde(untagged)] +pub(crate) enum DeepSeekContent { + Text(String), + Object(DeepSeekOcrResult), +} + +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub(crate) struct DeepSeekOcrResult { + #[serde(skip_serializing_if = "Option::is_none")] + pub pages: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub model: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub usage_info: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub document_annotation: Option, + #[serde(flatten)] + pub extra_fields: Map, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub(crate) struct DeepSeekPage { + #[serde(default)] + pub index: i64, + #[serde(default)] + pub markdown: String, + pub images: Option, + pub dimensions: Option, + #[serde(flatten)] + pub extra_fields: Map, +} diff --git a/litellm-rust/crates/core/src/ocr/codecs/mistral/transformation.rs b/litellm-rust/crates/core/src/ocr/codecs/mistral/transformation.rs index cd0a1dc6b17..5bd7e555a1e 100644 --- a/litellm-rust/crates/core/src/ocr/codecs/mistral/transformation.rs +++ b/litellm-rust/crates/core/src/ocr/codecs/mistral/transformation.rs @@ -36,6 +36,101 @@ mod tests { use rstest::rstest; use serde_json::{Value, json}; + fn mapped_params(value: Value) -> Value { + serde_json::to_value(serde_json::from_value::(value).unwrap()).unwrap() + } + + fn document() -> OcrDocument { + serde_json::from_value( + json!({"type":"document_url","document_url":"https://example.com/a.pdf"}), + ) + .unwrap() + } + + #[rstest] + fn extract_header_is_a_supported_ocr_param() { + assert_eq!( + mapped_params(json!({"extract_header":true}))["extract_header"], + true + ); + } + + #[rstest] + fn extract_footer_is_a_supported_ocr_param() { + assert_eq!( + mapped_params(json!({"extract_footer":false}))["extract_footer"], + false + ); + } + + #[rstest] + fn existing_ocr_params_remain_supported() { + let mapped = mapped_params(json!({ + "pages":[0,2], + "include_image_base64":true, + "image_limit":2, + "image_min_size":100, + "bbox_annotation_format":{"type":"json_schema"}, + "document_annotation_format":{"type":"json_schema"} + })); + assert_eq!(mapped["pages"], json!([0, 2])); + assert_eq!(mapped["include_image_base64"], true); + assert_eq!(mapped["image_limit"], 2); + assert_eq!(mapped["image_min_size"], 100); + assert_eq!(mapped["bbox_annotation_format"]["type"], "json_schema"); + assert_eq!(mapped["document_annotation_format"]["type"], "json_schema"); + } + + #[rstest] + fn map_ocr_params_forwards_extract_header() { + assert_eq!( + mapped_params(json!({"extract_header":true}))["extract_header"], + true + ); + } + + #[rstest] + fn map_ocr_params_forwards_extract_footer() { + assert_eq!( + mapped_params(json!({"extract_footer":true}))["extract_footer"], + true + ); + } + + #[rstest] + fn map_ocr_params_forwards_extract_header_and_footer() { + let mapped = mapped_params(json!({"extract_header":true,"extract_footer":false})); + assert_eq!(mapped["extract_header"], true); + assert_eq!(mapped["extract_footer"], false); + } + + #[rstest] + fn map_ocr_params_drops_unknown_params() { + let mapped = mapped_params(json!({"extract_header":true,"unsupported_param":"value"})); + assert_eq!(mapped["extract_header"], true); + assert!(mapped.get("unsupported_param").is_none()); + } + + #[rstest] + #[case("table_format", json!("html"))] + #[case("confidence_scores_granularity", json!("word"))] + #[case("document_annotation_prompt", json!("extract"))] + #[case("include_blocks", json!(true))] + #[case("id", json!("req-123"))] + fn new_ocr_params_are_supported(#[case] name: &str, #[case] value: Value) { + assert_eq!(mapped_params(json!({name:value.clone()}))[name], value); + } + + #[rstest] + #[case("table_format", json!("html"))] + #[case("confidence_scores_granularity", json!("word"))] + #[case("document_annotation_prompt", json!("extract"))] + #[case("include_blocks", json!(true))] + #[case("id", json!("req-123"))] + fn map_ocr_params_forwards_new_ocr_params(#[case] name: &str, #[case] value: Value) { + assert_eq!(mapped_params(json!({name:value.clone()}))[name], value); + } + #[rstest] #[case("pages", json!([0, 2]))] #[case("include_image_base64", json!(true))] @@ -53,50 +148,81 @@ mod tests { fn request_mapping_matches_python(#[case] name: &str, #[case] value: Value) { let params: MistralOcrParams = serde_json::from_value(json!({name: value.clone()})).unwrap(); - let document: OcrDocument = serde_json::from_value( - json!({"type":"document_url","document_url":"https://example.com/a.pdf"}), - ) - .unwrap(); let result = - serde_json::to_value(transform_ocr_request("model", document, ¶ms).unwrap()) + serde_json::to_value(transform_ocr_request("model", document(), ¶ms).unwrap()) .unwrap(); assert_eq!(result["model"], "model"); assert_eq!(result[name], value); } - #[test] - fn request_mapping_filters_unknown_fields() { - let params: MistralOcrParams = serde_json::from_value(json!({"unknown": true})).unwrap(); - let document: OcrDocument = serde_json::from_value( - json!({"type":"document_url","document_url":"https://example.com/a.pdf"}), + #[rstest] + #[case("table_format", json!("html"))] + #[case("confidence_scores_granularity", json!("word"))] + #[case("document_annotation_prompt", json!("extract"))] + #[case("id", json!("req-123"))] + #[case("extract_header", json!(true))] + #[case("include_blocks", json!(true))] + #[case("pages", json!([0,1]))] + fn transform_ocr_request_includes_each_optional_param( + #[case] name: &str, + #[case] value: Value, + ) { + let params: MistralOcrParams = serde_json::from_value(json!({name:value.clone()})).unwrap(); + let result = serde_json::to_value( + transform_ocr_request("mistral-ocr-latest", document(), ¶ms).unwrap(), ) .unwrap(); - let result = - serde_json::to_value(transform_ocr_request("model", document, ¶ms).unwrap()) - .unwrap(); - assert!(result.get("unknown").is_none()); + assert_eq!(result[name], value); + assert_eq!(result["model"], "mistral-ocr-latest"); } - #[test] - fn response_preserves_provider_fields() { + #[rstest] + fn transform_ocr_request_includes_multiple_new_params() { + let params: MistralOcrParams = serde_json::from_value(json!({ + "table_format":"html", + "confidence_scores_granularity":"page", + "extract_header":true + })) + .unwrap(); + let result = serde_json::to_value( + transform_ocr_request("mistral-ocr-latest", document(), ¶ms).unwrap(), + ) + .unwrap(); + assert_eq!(result["table_format"], "html"); + assert_eq!(result["confidence_scores_granularity"], "page"); + assert_eq!(result["extract_header"], true); + } + + #[rstest] + fn transform_ocr_response_preserves_blocks_and_confidence_scores() { let response: MistralOcrResponse = serde_json::from_value(json!({ - "pages":[{"index":0,"markdown":"hello","header":"head","confidence_scores":{"mean":0.99}}], + "pages":[{"index":0,"markdown":"hello","blocks":[{"type":"title"}],"confidence_scores":{"mean":0.99}}], "model":"returned-model", - "usage_info":{"pages_processed":1,"future_counter":5}, - "future_response_field":"kept" + "usage_info":{"pages_processed":1} })) .unwrap(); let result = transform_ocr_response("model", response) .unwrap() .into_json(); - assert_eq!(result["pages"][0]["header"], "head"); - assert_eq!(result["usage_info"]["future_counter"], 5); - assert_eq!(result["future_response_field"], "kept"); - assert_eq!(result["model"], "returned-model"); + assert_eq!(result["pages"][0]["blocks"][0]["type"], "title"); + assert_eq!(result["pages"][0]["confidence_scores"]["mean"], 0.99); } - #[test] - fn response_rejects_null_pages() { - assert!(serde_json::from_value::(json!({"pages":null})).is_err()); + #[rstest] + fn transform_ocr_response_preserves_ocr4_page_fields() { + let page = json!({ + "index":0, + "markdown":"table page", + "tables":[{"rows":2,"cols":3}], + "hyperlinks":["https://example.com"], + "header":"header", + "footer":"footer" + }); + let response: MistralOcrResponse = + serde_json::from_value(json!({"pages":[page.clone()]})).unwrap(); + let result = transform_ocr_response("model", response) + .unwrap() + .into_json(); + assert_eq!(result["pages"][0], page); } } diff --git a/litellm-rust/crates/core/src/ocr/codecs/mod.rs b/litellm-rust/crates/core/src/ocr/codecs/mod.rs index 79dcd150f5f..7c752749901 100644 --- a/litellm-rust/crates/core/src/ocr/codecs/mod.rs +++ b/litellm-rust/crates/core/src/ocr/codecs/mod.rs @@ -1,3 +1,4 @@ +pub(crate) mod deepseek; pub(crate) mod document_intelligence; pub(crate) mod mistral; pub(crate) mod reducto; diff --git a/litellm-rust/crates/core/src/ocr/error.rs b/litellm-rust/crates/core/src/ocr/error.rs index f42ac2ceb18..522d059ec48 100644 --- a/litellm-rust/crates/core/src/ocr/error.rs +++ b/litellm-rust/crates/core/src/ocr/error.rs @@ -36,6 +36,8 @@ pub enum OcrRequestError { pub enum OcrResponseError { #[error("invalid OCR response field: {path}")] ResponseField { path: String }, + #[error("OCR response is missing non-empty content")] + EmptyContent, #[error("OCR document redirect is missing a location")] MissingRedirectLocation, #[error("OCR document redirect location is invalid")] diff --git a/litellm-rust/crates/core/src/ocr/mod.rs b/litellm-rust/crates/core/src/ocr/mod.rs index 69b2958483a..1e975c3f521 100644 --- a/litellm-rust/crates/core/src/ocr/mod.rs +++ b/litellm-rust/crates/core/src/ocr/mod.rs @@ -7,7 +7,6 @@ mod handler; pub mod hooks; mod prepare; mod registry; -pub mod transformation; pub mod types; pub mod wire; @@ -21,6 +20,9 @@ mod azure_ai_tests; #[path = "../../tests/azure_document_intelligence_ocr.rs"] mod azure_document_intelligence_tests; #[cfg(test)] +#[path = "../../tests/deepseek_ocr.rs"] +mod deepseek_tests; +#[cfg(test)] #[path = "../../tests/reducto_ocr.rs"] mod reducto_tests; #[cfg(test)] @@ -30,5 +32,8 @@ pub(crate) mod test_support; #[path = "../../tests/ocr.rs"] pub(crate) mod tests; #[cfg(test)] +#[path = "../../tests/vertex_ai_deepseek_ocr.rs"] +mod vertex_ai_deepseek_tests; +#[cfg(test)] #[path = "../../tests/vertex_ai_ocr.rs"] mod vertex_ai_tests; diff --git a/litellm-rust/crates/core/src/ocr/prepare.rs b/litellm-rust/crates/core/src/ocr/prepare.rs index 363f963a66c..bf6f924088c 100644 --- a/litellm-rust/crates/core/src/ocr/prepare.rs +++ b/litellm-rust/crates/core/src/ocr/prepare.rs @@ -163,7 +163,6 @@ impl OcrWireBody { pub(crate) fn credential_env(name: &str) -> Option { std::env::var(name).ok() } - #[cfg(test)] mod tests { use serde_json::json; diff --git a/litellm-rust/crates/core/src/ocr/registry.rs b/litellm-rust/crates/core/src/ocr/registry.rs index 1097d102a20..1b20a91143b 100644 --- a/litellm-rust/crates/core/src/ocr/registry.rs +++ b/litellm-rust/crates/core/src/ocr/registry.rs @@ -75,7 +75,7 @@ pub(crate) fn resolve_wire_adapter( ))); } OcrProvider::VertexAi if provider.model.to_ascii_lowercase().contains("deepseek") => { - return Err(Error::Unsupported("Vertex DeepSeek OCR")); + OcrAdapterKind::VertexDeepSeek } OcrProvider::VertexAi => OcrAdapterKind::VertexMistral, }; diff --git a/litellm-rust/crates/core/src/ocr/transformation.rs b/litellm-rust/crates/core/src/ocr/transformation.rs deleted file mode 100644 index ac4f10bf15b..00000000000 --- a/litellm-rust/crates/core/src/ocr/transformation.rs +++ /dev/null @@ -1,107 +0,0 @@ -use crate::Error; -use serde_json::{Map, Value}; - -use super::types::{LiteLLMOcrResponse, OcrRequestData}; - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum OcrAuthStrategy { - Bearer, - Header(&'static str), -} - -impl OcrAuthStrategy { - pub fn header_name(self) -> &'static str { - match self { - Self::Bearer => "authorization", - Self::Header(header_name) => header_name, - } - } -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum OcrResponseHandling { - Json, - AzureDocumentIntelligencePoll, -} - -pub trait OcrProviderConfig: Sync { - fn supported_ocr_params(&self) -> &'static [&'static str]; - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn map_ocr_params(&self, non_default_params: &Map) -> Map { - let mut mapped_params = Map::new(); - for (param, value) in non_default_params { - if self.supported_ocr_params().contains(¶m.as_str()) { - mapped_params.insert(param.clone(), value.clone()); - } - } - mapped_params - } - - fn transform_ocr_request( - &self, - model: &str, - document: Value, - optional_params: Map, - ) -> Result; - - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result; - - fn transform_ocr_response_with_params( - &self, - model: &str, - response_json: Value, - _optional_params: &Map, - ) -> Result { - self.transform_ocr_response(model, response_json) - } - - fn complete_url( - &self, - api_base: Option<&str>, - model: &str, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result; - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result; - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn validate_environment( - &self, - headers: Vec<(String, String)>, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result, Error> { - let strategy = self.auth_strategy(); - if crate::http_utils::has_header(&headers, strategy.header_name()) { - return Ok(headers); - } - let api_key = self.resolve_api_key(api_key, env_lookup)?; - let auth_header = match strategy { - OcrAuthStrategy::Bearer => ("Authorization".to_string(), format!("Bearer {api_key}")), - OcrAuthStrategy::Header(name) => (name.to_string(), api_key), - }; - Ok(std::iter::once(auth_header).chain(headers).collect()) - } - - fn auth_strategy(&self) -> OcrAuthStrategy { - OcrAuthStrategy::Bearer - } - - fn requires_data_uri_document(&self) -> bool { - false - } - - fn response_handling(&self) -> OcrResponseHandling { - OcrResponseHandling::Json - } -} diff --git a/litellm-rust/crates/core/src/ocr/types.rs b/litellm-rust/crates/core/src/ocr/types.rs index 0e92b0b6868..06519f86c91 100644 --- a/litellm-rust/crates/core/src/ocr/types.rs +++ b/litellm-rust/crates/core/src/ocr/types.rs @@ -11,12 +11,6 @@ use crate::Error; use crate::auth::InputSource; use crate::constants::OCR_HTTP_TIMEOUT_SECS; -#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] -pub struct OcrRequestData { - pub data: Value, - pub files: Option, -} - #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] #[serde(tag = "type")] pub enum OcrDocument { diff --git a/litellm-rust/crates/core/src/providers/azure_ai/mod.rs b/litellm-rust/crates/core/src/providers/azure_ai/mod.rs index f2d5b679aee..4f41d1d6abb 100644 --- a/litellm-rust/crates/core/src/providers/azure_ai/mod.rs +++ b/litellm-rust/crates/core/src/providers/azure_ai/mod.rs @@ -1,3 +1,2 @@ pub(crate) mod auth; pub mod messages; -pub mod ocr; diff --git a/litellm-rust/crates/core/src/providers/azure_ai/ocr/mod.rs b/litellm-rust/crates/core/src/providers/azure_ai/ocr/mod.rs deleted file mode 100644 index f239b6921fa..00000000000 --- a/litellm-rust/crates/core/src/providers/azure_ai/ocr/mod.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod transformation; diff --git a/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs b/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs deleted file mode 100644 index 2dbec8e2187..00000000000 --- a/litellm-rust/crates/core/src/providers/azure_ai/ocr/transformation.rs +++ /dev/null @@ -1,1376 +0,0 @@ -use std::collections::BTreeSet; - -use crate::error::{Error, json_type_name}; -use crate::ocr::transformation::{OcrAuthStrategy, OcrProviderConfig, OcrResponseHandling}; -use crate::ocr::types::{LiteLLMOcrResponse, OcrRequestData}; -use serde_json::{Map, Value, json}; - -use crate::providers::mistral::ocr::transformation::MISTRAL_OCR_CONFIG; - -const AZURE_AI_API_KEY_ENV: &str = "AZURE_AI_API_KEY"; -const AZURE_AI_API_BASE_ENV: &str = "AZURE_AI_API_BASE"; -const AZURE_DOCUMENT_INTELLIGENCE_API_KEY_ENV: &str = "AZURE_DOCUMENT_INTELLIGENCE_API_KEY"; -const AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT_ENV: &str = "AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"; -const AZURE_DOCUMENT_INTELLIGENCE_API_VERSION: &str = "2024-11-30"; -const AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI: i64 = 96; - -const AZURE_DOCUMENT_INTELLIGENCE_SUPPORTED_OCR_PARAMS: &[&str] = - &["pages", "features", "req_format"]; - -pub struct AzureAiOcrConfig; -pub struct AzureDocumentIntelligenceOcrConfig; - -pub const AZURE_AI_OCR_CONFIG: AzureAiOcrConfig = AzureAiOcrConfig; -pub const AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG: AzureDocumentIntelligenceOcrConfig = - AzureDocumentIntelligenceOcrConfig; - -fn non_empty(value: Option<&str>) -> Option<&str> { - value.map(str::trim).filter(|value| !value.is_empty()) -} - -fn resolve_value( - explicit: Option<&str>, - env_name: &str, - env_lookup: &dyn Fn(&str) -> Option, - missing_message: &str, -) -> Result { - non_empty(explicit) - .map(str::to_string) - .or_else(|| env_lookup(env_name).filter(|value| !value.trim().is_empty())) - .ok_or_else(|| Error::Auth(missing_message.to_string())) -} - -pub fn resolve_azure_ai_api_key( - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - resolve_value( - api_key, - AZURE_AI_API_KEY_ENV, - env_lookup, - "Missing Azure AI API Key - A call is being made to Azure AI but no key is set either in the environment variables or via params", - ) -} - -pub fn resolve_azure_ai_api_base( - api_base: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - resolve_value( - api_base, - AZURE_AI_API_BASE_ENV, - env_lookup, - "Missing Azure AI API Base - Set AZURE_AI_API_BASE environment variable or pass api_base parameter", - ) -} - -pub fn complete_azure_ai_url( - api_base: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - let base = resolve_azure_ai_api_base(api_base, env_lookup)?; - Ok(format!( - "{}/providers/mistral/azure/ocr", - base.trim_end_matches('/') - )) -} - -pub fn resolve_document_intelligence_api_key( - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - resolve_value( - api_key, - AZURE_DOCUMENT_INTELLIGENCE_API_KEY_ENV, - env_lookup, - "Missing Azure Document Intelligence API Key - Set AZURE_DOCUMENT_INTELLIGENCE_API_KEY environment variable or pass api_key parameter", - ) -} - -pub fn resolve_document_intelligence_endpoint( - api_base: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - resolve_value( - api_base, - AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT_ENV, - env_lookup, - "Missing Azure Document Intelligence Endpoint - Set AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT environment variable or pass api_base parameter", - ) -} - -fn prepend_auth_header( - headers: Vec<(String, String)>, - name: &str, - value: String, -) -> Vec<(String, String)> { - std::iter::once((name.to_string(), value)) - .chain(headers) - .collect() -} - -pub fn validate_azure_ai_environment( - headers: Vec<(String, String)>, - api_key: Option<&str>, - azure_ad_token: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result, Error> { - if crate::http_utils::has_header(&headers, "Authorization") - || crate::http_utils::has_header(&headers, "Api-Key") - { - return Ok(headers); - } - if let Ok(api_key) = resolve_azure_ai_api_key(api_key, env_lookup) { - return Ok(prepend_auth_header(headers, "Api-Key", api_key)); - } - non_empty(azure_ad_token) - .map(|token| prepend_auth_header(headers, "Authorization", format!("Bearer {token}"))) - .ok_or(Error::MissingAzureAiCredentialsOrAdToken) -} - -pub fn validate_document_intelligence_environment( - headers: Vec<(String, String)>, - api_key: Option<&str>, - azure_ad_token: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result, Error> { - if crate::http_utils::has_header(&headers, "Authorization") - || crate::http_utils::has_header(&headers, "Ocp-Apim-Subscription-Key") - { - return Ok(headers); - } - if let Ok(api_key) = resolve_document_intelligence_api_key(api_key, env_lookup) { - return Ok(prepend_auth_header( - headers, - "Ocp-Apim-Subscription-Key", - api_key, - )); - } - non_empty(azure_ad_token) - .map(|token| prepend_auth_header(headers, "Authorization", format!("Bearer {token}"))) - .ok_or_else(|| { - Error::Auth( - "Missing Azure Document Intelligence credentials - set AZURE_DOCUMENT_INTELLIGENCE_API_KEY or provide azure_ad_token" - .to_string(), - ) - }) -} - -fn encode_model_id(model: &str) -> Result { - let model_id = model.rsplit('/').next().unwrap_or(model); - if matches!(model_id, "." | "..") { - return Err(Error::InvalidRequest( - "model_id cannot be a dot path segment".to_string(), - )); - } - Ok(model_id - .bytes() - .flat_map(|byte| match byte { - b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { - vec![byte as char] - } - _ => format!("%{byte:02X}").chars().collect(), - }) - .collect()) -} - -fn pages_token_is_valid(token: &str) -> bool { - let mut parts = token.split('-'); - let Some(start) = parts.next() else { - return false; - }; - if start.is_empty() || !start.chars().all(|ch| ch.is_ascii_digit()) { - return false; - } - match parts.next() { - None => true, - Some(end) => { - !end.is_empty() && end.chars().all(|ch| ch.is_ascii_digit()) && parts.next().is_none() - } - } -} - -fn normalize_pages_param(pages: &Value) -> Result, Error> { - match pages { - Value::String(value) => { - let normalized = value - .split(',') - .map(str::trim) - .collect::>() - .join(","); - if normalized.split(',').all(pages_token_is_valid) { - Ok(Some(normalized)) - } else { - Err(Error::InvalidRequest(format!( - "Invalid `pages` string for Azure Document Intelligence: {value:?}. Expected format like '1-3,5,7-9'." - ))) - } - } - Value::Array(values) => { - if values.is_empty() { - return Ok(None); - } - if values.iter().any(Value::is_boolean) { - return Err(Error::InvalidRequest( - "`pages` must be integers, not booleans".to_string(), - )); - } - if values.iter().all(Value::is_i64) { - let mut pages = BTreeSet::new(); - for value in values { - let page = value.as_i64().expect("checked is_i64"); - if page < 0 { - return Err(Error::InvalidRequest( - "`pages` integers must be >= 0 (Mistral 0-based indices)".to_string(), - )); - } - pages.insert(page + 1); - } - return Ok(Some( - pages - .into_iter() - .map(|page| page.to_string()) - .collect::>() - .join(","), - )); - } - if values.iter().all(Value::is_string) { - let normalized = values - .iter() - .filter_map(Value::as_str) - .map(str::trim) - .collect::>() - .join(","); - if normalized.split(',').all(pages_token_is_valid) { - return Ok(Some(normalized)); - } - return Err(Error::InvalidRequest(format!( - "Invalid `pages` list for Azure Document Intelligence: {values:?}. Expected tokens like '1' or '3-5'." - ))); - } - Err(Error::InvalidRequest( - "`pages` must be a list[int] (0-based, Mistral-style) or a string like '1-3,5,7-9'." - .to_string(), - )) - } - _ => Err(Error::InvalidRequest( - "`pages` must be a list[int] (0-based, Mistral-style) or a string like '1-3,5,7-9'." - .to_string(), - )), - } -} - -fn feature_token_is_valid(token: &str) -> bool { - let Some((first, rest)) = token.as_bytes().split_first() else { - return false; - }; - first.is_ascii_alphabetic() && rest.iter().all(u8::is_ascii_alphanumeric) -} - -fn invalid_features_error(features: &Value) -> Error { - Error::InvalidRequest(format!( - "Invalid `features` for Azure Document Intelligence: {features:?}. Expected a list of feature names or a comma-separated string like 'keyValuePairs' or 'keyValuePairs,languages'." - )) -} - -fn normalize_features_param(features: &Value) -> Result, Error> { - let normalized = match features { - Value::String(value) => value - .split(',') - .map(str::trim) - .collect::>() - .join(","), - Value::Array(values) if values.is_empty() => return Ok(None), - Value::Array(values) => values - .iter() - .map(Value::as_str) - .collect::>>() - .ok_or_else(|| invalid_features_error(features))? - .into_iter() - .map(str::trim) - .collect::>() - .join(","), - _ => return Err(invalid_features_error(features)), - }; - - if normalized.split(',').all(feature_token_is_valid) { - Ok(Some(normalized)) - } else { - Err(invalid_features_error(features)) - } -} - -fn normalize_req_format(req_format: &Value) -> Result { - match req_format.as_str() { - Some(value @ ("native" | "litellm")) => Ok(value.to_string()), - _ => Err(Error::InvalidRequest(format!( - "Invalid `req_format` for Azure Document Intelligence: {req_format:?}. Expected 'native' or 'litellm'." - ))), - } -} - -pub fn map_document_intelligence_ocr_params( - non_default_params: &Map, -) -> Result, Error> { - let mut mapped = Map::new(); - if let Some(pages) = non_default_params.get("pages") - && let Some(normalized) = normalize_pages_param(pages)? - { - mapped.insert("pages".to_string(), Value::String(normalized)); - } - if let Some(features) = non_default_params.get("features") - && let Some(normalized) = normalize_features_param(features)? - { - mapped.insert("features".to_string(), Value::String(normalized)); - } - if let Some(req_format) = non_default_params.get("req_format") { - mapped.insert( - "req_format".to_string(), - Value::String(normalize_req_format(req_format)?), - ); - } - Ok(mapped) -} - -pub fn complete_document_intelligence_url( - api_base: Option<&str>, - model: &str, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - let endpoint = resolve_document_intelligence_endpoint(api_base, env_lookup)?; - let mut url = format!( - "{}/documentintelligence/documentModels/{}:analyze?api-version={}", - endpoint.trim_end_matches('/'), - encode_model_id(model)?, - AZURE_DOCUMENT_INTELLIGENCE_API_VERSION - ); - - if let Some(pages) = optional_params.get("pages") - && let Some(normalized) = normalize_pages_param(pages)? - { - url.push_str("&pages="); - url.push_str(&normalized); - } - - if let Some(features) = optional_params.get("features") - && let Some(normalized) = normalize_features_param(features)? - { - url.push_str("&features="); - url.push_str(&normalized); - } - - if let Some(req_format) = optional_params.get("req_format") { - normalize_req_format(req_format)?; - } - - Ok(url) -} - -fn document_url_from_mistral_document(document: &Value) -> Result<&str, Error> { - let object = document.as_object().ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(document), - })?; - let doc_type = object - .get("type") - .and_then(Value::as_str) - .ok_or(Error::MissingField("document.type"))?; - let field_name = match doc_type { - "document_url" => "document_url", - "image_url" => "image_url", - other => { - return Err(Error::InvalidRequest(format!( - "Invalid document type: {other}. Must be 'document_url' or 'image_url'" - ))); - } - }; - object - .get(field_name) - .and_then(Value::as_str) - .filter(|value| !value.is_empty()) - .ok_or(Error::MissingField(field_name)) -} - -fn extract_base64_from_data_uri(data_uri: &str) -> &str { - data_uri - .split_once(',') - .map(|(_, data)| data) - .unwrap_or(data_uri) -} - -fn page_markdown(page: &Map) -> String { - page.get("lines") - .and_then(Value::as_array) - .map(|lines| { - lines - .iter() - .filter_map(|line| line.get("content").and_then(Value::as_str)) - .collect::>() - .join("\n") - }) - .unwrap_or_default() -} - -fn page_dimensions(page: &Map) -> Value { - let width = page.get("width").and_then(Value::as_f64).unwrap_or(8.5); - let height = page.get("height").and_then(Value::as_f64).unwrap_or(11.0); - let unit = page.get("unit").and_then(Value::as_str).unwrap_or("inch"); - let (width, height) = if unit == "inch" { - ( - (width * AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI as f64) as i64, - (height * AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI as f64) as i64, - ) - } else { - (width as i64, height as i64) - }; - json!({ - "width": width, - "height": height, - "dpi": AZURE_DOCUMENT_INTELLIGENCE_DEFAULT_DPI, - }) -} - -fn transform_document_intelligence_response( - model: &str, - response_json: Value, - preserve_native_response: bool, -) -> Result { - let response = response_json - .as_object() - .ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(&response_json), - })?; - let status = response - .get("status") - .and_then(Value::as_str) - .ok_or(Error::MissingField("status"))?; - if status != "succeeded" { - return Err(Error::InvalidResponse(format!( - "Azure Document Intelligence analysis failed with status: {status}" - ))); - } - - let analyze_result = response.get("analyzeResult").and_then(Value::as_object); - let azure_pages = analyze_result - .and_then(|result| result.get("pages")) - .and_then(Value::as_array) - .cloned() - .unwrap_or_default(); - let pages = azure_pages - .iter() - .filter_map(Value::as_object) - .map(|page| { - let page_number = page.get("pageNumber").and_then(Value::as_i64).unwrap_or(1); - json!({ - "index": page_number - 1, - "markdown": page_markdown(page), - "dimensions": page_dimensions(page), - }) - }) - .collect::>(); - let extra_fields = ["content", "tables", "keyValuePairs"] - .into_iter() - .map(|field| { - ( - field.to_string(), - analyze_result - .and_then(|result| result.get(field)) - .cloned() - .unwrap_or(Value::Null), - ) - }) - .collect(); - - Ok(LiteLLMOcrResponse { - usage_info: Some(json!({ - "pages_processed": pages.len(), - "doc_size_bytes": null, - })), - pages, - model: model.to_string(), - document_annotation: None, - object: "ocr".to_string(), - extra_fields, - provider_native_response: preserve_native_response.then_some(response_json), - }) -} - -impl OcrProviderConfig for AzureAiOcrConfig { - fn supported_ocr_params(&self) -> &'static [&'static str] { - MISTRAL_OCR_CONFIG.supported_ocr_params() - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_request( - &self, - model: &str, - document: Value, - optional_params: Map, - ) -> Result { - MISTRAL_OCR_CONFIG.transform_ocr_request(model, document, optional_params) - } - - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - MISTRAL_OCR_CONFIG.transform_ocr_response(model, response_json) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn complete_url( - &self, - api_base: Option<&str>, - _model: &str, - _optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - complete_azure_ai_url(api_base, env_lookup) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_azure_ai_api_key(api_key, env_lookup) - } - - fn requires_data_uri_document(&self) -> bool { - true - } -} - -impl OcrProviderConfig for AzureDocumentIntelligenceOcrConfig { - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn supported_ocr_params(&self) -> &'static [&'static str] { - AZURE_DOCUMENT_INTELLIGENCE_SUPPORTED_OCR_PARAMS - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn map_ocr_params(&self, non_default_params: &Map) -> Map { - map_document_intelligence_ocr_params(non_default_params).unwrap_or_else(|_| { - non_default_params - .iter() - .filter(|(name, _)| { - AZURE_DOCUMENT_INTELLIGENCE_SUPPORTED_OCR_PARAMS.contains(&name.as_str()) - }) - .map(|(name, value)| (name.clone(), value.clone())) - .collect() - }) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_request( - &self, - _model: &str, - document: Value, - _optional_params: Map, - ) -> Result { - let document_url = document_url_from_mistral_document(&document)?; - let mut data = Map::new(); - if document_url.starts_with("data:") { - data.insert( - "base64Source".to_string(), - Value::String(extract_base64_from_data_uri(document_url).to_string()), - ); - } else { - data.insert( - "urlSource".to_string(), - Value::String(document_url.to_string()), - ); - } - Ok(OcrRequestData { - data: Value::Object(data), - files: None, - }) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - transform_document_intelligence_response(model, response_json, false) - } - - fn transform_ocr_response_with_params( - &self, - model: &str, - response_json: Value, - optional_params: &Map, - ) -> Result { - transform_document_intelligence_response( - model, - response_json, - optional_params.get("req_format").and_then(Value::as_str) == Some("native"), - ) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn complete_url( - &self, - api_base: Option<&str>, - model: &str, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - complete_document_intelligence_url(api_base, model, optional_params, env_lookup) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_document_intelligence_api_key(api_key, env_lookup) - } - - fn auth_strategy(&self) -> OcrAuthStrategy { - OcrAuthStrategy::Header("Ocp-Apim-Subscription-Key") - } - - fn response_handling(&self) -> OcrResponseHandling { - OcrResponseHandling::AzureDocumentIntelligencePoll - } -} - -#[cfg(test)] -mod tests { - use super::*; - use rstest::{fixture, rstest}; - - const ENDPOINT: &str = "https://example.cognitiveservices.azure.com"; - - #[fixture] - fn document_intelligence_config() -> AzureDocumentIntelligenceOcrConfig { - AzureDocumentIntelligenceOcrConfig - } - - fn header_value<'a>(headers: &'a [(String, String)], name: &str) -> Option<&'a str> { - headers - .iter() - .find(|(header_name, _)| header_name.eq_ignore_ascii_case(name)) - .map(|(_, value)| value.as_str()) - } - - #[fixture] - fn native_operation() -> Value { - json!({ - "status": "succeeded", - "createdDateTime": "2026-07-02T00:00:00Z", - "lastUpdatedDateTime": "2026-07-02T00:00:05Z", - "analyzeResult": { - "content": "Invoice\nInvoice No: INV-12345\nTotal: $100.00", - "pages": [{ - "pageNumber": 1, - "width": 8.5, - "height": 11, - "unit": "inch", - "angle": 0.13, - "lines": [ - {"content": "Invoice"}, - {"content": "Invoice No: INV-12345"}, - {"content": "Total: $100.00"} - ], - "words": [{"content": "Invoice", "confidence": 0.994}] - }], - "tables": [ - { - "rowCount": 2, - "columnCount": 2, - "cells": [ - {"kind": "columnHeader", "rowIndex": 0, "columnIndex": 0, "content": "Item"}, - {"kind": "columnHeader", "rowIndex": 0, "columnIndex": 1, "content": "Price"}, - {"rowIndex": 1, "columnIndex": 0, "content": "Widget"}, - {"rowIndex": 1, "columnIndex": 1, "content": "$100.00"} - ] - }, - { - "rowCount": 1, - "columnCount": 1, - "cells": [{"rowIndex": 0, "columnIndex": 0, "content": "Totals"}] - } - ], - "keyValuePairs": [ - { - "key": {"content": "Invoice No"}, - "value": {"content": "INV-12345"}, - "confidence": 0.98 - }, - { - "key": {"content": "Total"}, - "value": {"content": "$100.00"}, - "confidence": 0.95 - } - ], - "paragraphs": [{"content": "Invoice"}] - } - }) - } - - fn assert_native_fields_preserved(response: &LiteLLMOcrResponse, operation: &Value) { - let analyze_result = &operation["analyzeResult"]; - - assert_eq!(response.extra_fields["content"], analyze_result["content"]); - assert_eq!(response.extra_fields["tables"], analyze_result["tables"]); - assert_eq!( - response.extra_fields["keyValuePairs"], - analyze_result["keyValuePairs"] - ); - assert_eq!(response.object, "ocr"); - assert_eq!( - response.usage_info, - Some(json!({"pages_processed": 1, "doc_size_bytes": null})) - ); - assert_eq!(response.pages[0]["index"], 0); - assert_eq!( - response.pages[0]["markdown"], - "Invoice\nInvoice No: INV-12345\nTotal: $100.00" - ); - assert_eq!( - response.pages[0]["dimensions"], - json!({"width": 816, "height": 1056, "dpi": 96}) - ); - } - - #[test] - fn azure_ai_reuses_mistral_body_transform() { - let body = AZURE_AI_OCR_CONFIG - .transform_ocr_request( - "pixtral-12b-2409", - json!({"type": "document_url", "document_url": "data:application/pdf;base64,abc"}), - serde_json::Map::from_iter([("include_image_base64".to_string(), json!(true))]), - ) - .expect("request transforms") - .data; - - assert_eq!(body["model"], "pixtral-12b-2409"); - assert_eq!(body["include_image_base64"], true); - assert_eq!( - body["document"]["document_url"], - "data:application/pdf;base64,abc" - ); - } - - #[test] - fn document_intelligence_url_normalizes_zero_based_pages() { - let params = serde_json::Map::from_iter([("pages".to_string(), json!([2, 0, 2]))]); - let url = complete_document_intelligence_url( - Some("https://example.cognitiveservices.azure.com/"), - "azure_ai/doc-intelligence/prebuilt-layout", - ¶ms, - &|_| None, - ) - .expect("url builds"); - - assert_eq!( - url, - "https://example.cognitiveservices.azure.com/documentintelligence/documentModels/prebuilt-layout:analyze?api-version=2024-11-30&pages=1,3" - ); - } - - #[test] - fn document_intelligence_url_normalizes_features() { - let params = serde_json::Map::from_iter([( - "features".to_string(), - json!("keyValuePairs, languages"), - )]); - let url = complete_document_intelligence_url( - Some("https://example.cognitiveservices.azure.com"), - "prebuilt-layout", - ¶ms, - &|_| None, - ) - .expect("url builds"); - - assert_eq!( - url, - "https://example.cognitiveservices.azure.com/documentintelligence/documentModels/prebuilt-layout:analyze?api-version=2024-11-30&features=keyValuePairs,languages" - ); - } - - #[test] - fn document_intelligence_url_combines_pages_and_feature_list() { - let params = serde_json::Map::from_iter([ - ("pages".to_string(), json!([0, 1, 2])), - ( - "features".to_string(), - json!([" keyValuePairs ", "languages"]), - ), - ]); - let url = complete_document_intelligence_url( - Some("https://example.cognitiveservices.azure.com"), - "prebuilt-layout", - ¶ms, - &|_| None, - ) - .expect("url builds"); - - assert_eq!( - url, - "https://example.cognitiveservices.azure.com/documentintelligence/documentModels/prebuilt-layout:analyze?api-version=2024-11-30&pages=1,2,3&features=keyValuePairs,languages" - ); - } - - #[test] - fn document_intelligence_url_omits_empty_feature_list() { - let params = serde_json::Map::from_iter([("features".to_string(), json!([]))]); - assert!( - map_document_intelligence_ocr_params(¶ms) - .expect("empty features map") - .is_empty() - ); - let url = complete_document_intelligence_url( - Some("https://example.cognitiveservices.azure.com"), - "prebuilt-layout", - ¶ms, - &|_| None, - ) - .expect("url builds"); - - assert_eq!( - url, - "https://example.cognitiveservices.azure.com/documentintelligence/documentModels/prebuilt-layout:analyze?api-version=2024-11-30" - ); - } - - #[rstest] - #[case::query_injection(json!("keyValuePairs&pages=9"))] - #[case::spaces(json!("key value pairs"))] - #[case::empty_string(json!(""))] - #[case::integer_list(json!([1, 2]))] - #[case::nested_list(json!([["keyValuePairs"]]))] - #[case::object(json!({"feature": "keyValuePairs"}))] - #[case::number(json!(5))] - fn document_intelligence_mapping_rejects_invalid_features(#[case] features: Value) { - let params = serde_json::Map::from_iter([("features".to_string(), features)]); - let error = - map_document_intelligence_ocr_params(¶ms).expect_err("invalid features must fail"); - - assert!(matches!( - error, - Error::InvalidRequest(message) if message.contains("Invalid `features`") - )); - } - - #[rstest] - #[case::single_list(json!(["keyValuePairs"]), "keyValuePairs")] - #[case::multiple_list( - json!(["keyValuePairs", "languages"]), - "keyValuePairs,languages" - )] - #[case::single_string(json!("keyValuePairs"), "keyValuePairs")] - #[case::comma_separated(json!("keyValuePairs,languages"), "keyValuePairs,languages")] - #[case::spaces(json!("keyValuePairs, languages"), "keyValuePairs,languages")] - fn document_intelligence_maps_features(#[case] features: Value, #[case] expected: &str) { - let params = Map::from_iter([ - ("features".to_string(), features), - ("unsupported".to_string(), json!(true)), - ]); - - assert_eq!( - AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG.map_ocr_params(¶ms), - Map::from_iter([("features".to_string(), json!(expected))]) - ); - } - - #[test] - fn document_intelligence_request_uses_base64_source_for_data_uri() { - let body = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_request( - "prebuilt-read", - json!({"type": "document_url", "document_url": "data:application/pdf;base64,abc123"}), - Map::new(), - ) - .expect("request transforms") - .data; - - assert_eq!(body, json!({"base64Source": "abc123"})); - } - - #[rstest] - fn document_intelligence_response_normalizes_pages(native_operation: Value) { - let response = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response("prebuilt-layout", native_operation.clone()) - .expect("response transforms"); - - assert_native_fields_preserved(&response, &native_operation); - } - - #[test] - fn azure_document_intelligence_model_id_is_encoded() { - let url = complete_document_intelligence_url( - Some(ENDPOINT), - "prebuilt-layout?x=1#frag", - &Map::new(), - &|_| None, - ) - .expect("url builds"); - - assert_eq!( - url, - "https://example.cognitiveservices.azure.com/documentintelligence/documentModels/prebuilt-layout%3Fx%3D1%23frag:analyze?api-version=2024-11-30" - ); - } - - #[test] - fn azure_document_intelligence_dot_segment_model_id_is_rejected() { - let error = complete_document_intelligence_url( - Some(ENDPOINT), - "azure_ai/doc-intelligence/..", - &Map::new(), - &|_| None, - ) - .expect_err("dot segment must fail"); - - assert_eq!( - error, - Error::InvalidRequest("model_id cannot be a dot path segment".to_string()) - ); - } - - #[rstest] - fn document_intelligence_async_response_preserves_normalized_fields(native_operation: Value) { - let response = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response( - "azure_ai/doc-intelligence/prebuilt-layout", - native_operation.clone(), - ) - .expect("response transforms"); - - assert_native_fields_preserved(&response, &native_operation); - } - - #[test] - fn document_intelligence_response_tolerates_missing_native_fields() { - let response = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response( - "azure_ai/doc-intelligence/prebuilt-read", - json!({ - "status": "succeeded", - "analyzeResult": { - "pages": [{ - "pageNumber": 1, - "width": 8.5, - "height": 11, - "unit": "inch", - "lines": [{"content": "hello"}] - }] - } - }), - ) - .expect("missing optional fields are allowed"); - - assert_eq!(response.pages[0]["markdown"], "hello"); - assert_eq!(response.extra_fields["content"], Value::Null); - assert_eq!(response.extra_fields["tables"], Value::Null); - assert_eq!(response.extra_fields["keyValuePairs"], Value::Null); - } - - #[test] - fn document_intelligence_non_succeeded_status_is_rejected() { - let error = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response( - "azure_ai/doc-intelligence/prebuilt-layout", - json!({"status": "failed"}), - ) - .expect_err("failed status must fail"); - - assert_eq!( - error, - Error::InvalidResponse( - "Azure Document Intelligence analysis failed with status: failed".to_string() - ) - ); - } - - #[test] - fn document_intelligence_supported_params_include_features() { - assert_eq!( - AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG.supported_ocr_params(), - &["pages", "features", "req_format"] - ); - } - - #[rstest] - fn document_intelligence_native_format_carries_raw_operation(native_operation: Value) { - let response = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response_with_params( - "azure_ai/doc-intelligence/prebuilt-layout", - native_operation.clone(), - &Map::from_iter([("req_format".to_string(), json!("native"))]), - ) - .expect("native response transforms"); - - assert_eq!( - response.provider_native_response, - Some(native_operation.clone()) - ); - assert_native_fields_preserved(&response, &native_operation); - } - - #[rstest] - fn document_intelligence_async_native_format_carries_raw_operation(native_operation: Value) { - let response = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response_with_params( - "azure_ai/doc-intelligence/prebuilt-layout", - native_operation.clone(), - &Map::from_iter([("req_format".to_string(), json!("native"))]), - ) - .expect("native response transforms"); - - assert_eq!( - response.provider_native_response, - Some(native_operation.clone()) - ); - assert_native_fields_preserved(&response, &native_operation); - } - - #[rstest] - #[case::default(Map::new())] - #[case::litellm(Map::from_iter([("req_format".to_string(), json!("litellm"))]))] - fn document_intelligence_default_format_omits_raw_operation( - #[case] optional_params: Map, - native_operation: Value, - ) { - let response = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_response_with_params( - "azure_ai/doc-intelligence/prebuilt-layout", - native_operation.clone(), - &optional_params, - ) - .expect("response transforms"); - - assert_eq!(response.provider_native_response, None); - assert_native_fields_preserved(&response, &native_operation); - } - - #[rstest] - #[case::native("native")] - #[case::litellm("litellm")] - fn document_intelligence_maps_req_format(#[case] req_format: &str) { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "req_format".to_string(), - json!(req_format), - )])) - .expect("req_format maps"); - - assert_eq!( - mapped, - Map::from_iter([("req_format".to_string(), json!(req_format))]) - ); - } - - #[test] - fn document_intelligence_rejects_unknown_req_format() { - let error = map_document_intelligence_ocr_params(&Map::from_iter([( - "req_format".to_string(), - json!("azure"), - )])) - .expect_err("unknown req_format must fail"); - - assert!( - matches!(error, Error::InvalidRequest(message) if message.contains("Invalid `req_format`")) - ); - } - - #[test] - fn document_intelligence_url_omits_req_format() { - let url = complete_document_intelligence_url( - Some(ENDPOINT), - "prebuilt-layout", - &Map::from_iter([("req_format".to_string(), json!("native"))]), - &|_| None, - ) - .expect("url builds"); - - assert!(!url.contains("req_format")); - } - - #[test] - fn document_intelligence_validate_environment_uses_subscription_key() { - let headers = - validate_document_intelligence_environment(Vec::new(), Some("my-key"), None, &|_| None) - .expect("api key authenticates"); - - assert_eq!( - header_value(&headers, "Ocp-Apim-Subscription-Key"), - Some("my-key") - ); - } - - #[test] - fn document_intelligence_validate_environment_falls_back_to_entra_token() { - let headers = validate_document_intelligence_environment( - Vec::new(), - None, - Some("entra-token"), - &|_| None, - ) - .expect("Entra token authenticates"); - - assert_eq!( - header_value(&headers, "Authorization"), - Some("Bearer entra-token") - ); - assert_eq!(header_value(&headers, "Ocp-Apim-Subscription-Key"), None); - } - - #[test] - fn document_intelligence_supported_params_include_pages_features_and_req_format() { - assert_eq!( - AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG.supported_ocr_params(), - &["pages", "features", "req_format"] - ); - } - - #[test] - fn document_intelligence_maps_zero_based_page_list() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!([0, 1, 2]), - )])) - .expect("pages map"); - - assert_eq!( - mapped, - Map::from_iter([("pages".to_string(), json!("1,2,3"))]) - ); - } - - #[test] - fn document_intelligence_page_mapping_dedupes_and_sorts() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!([2, 0, 0, 1]), - )])) - .expect("pages map"); - - assert_eq!(mapped["pages"], "1,2,3"); - } - - #[test] - fn document_intelligence_page_mapping_omits_empty_list() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!([]), - )])) - .expect("empty pages map"); - - assert!(mapped.is_empty()); - } - - #[test] - fn document_intelligence_page_mapping_accepts_native_range() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!("3-9"), - )])) - .expect("range maps"); - - assert_eq!(mapped["pages"], "3-9"); - } - - #[test] - fn document_intelligence_page_mapping_strips_spaces() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!("1-3, 5"), - )])) - .expect("range maps"); - - assert_eq!(mapped["pages"], "1-3,5"); - } - - #[test] - fn document_intelligence_page_mapping_accepts_string_tokens() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!(["1", "3-5"]), - )])) - .expect("tokens map"); - - assert_eq!(mapped["pages"], "1,3-5"); - } - - #[test] - fn document_intelligence_page_mapping_rejects_invalid_string() { - let error = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!("a,b"), - )])) - .expect_err("invalid pages must fail"); - - assert!( - matches!(error, Error::InvalidRequest(message) if message.contains("Invalid `pages` string")) - ); - } - - #[test] - fn document_intelligence_page_mapping_rejects_negative_index() { - let error = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!([-1]), - )])) - .expect_err("negative pages must fail"); - - assert!( - matches!(error, Error::InvalidRequest(message) if message.contains("must be >= 0")) - ); - } - - #[test] - fn document_intelligence_page_mapping_rejects_bool_list() { - let error = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!([true, false]), - )])) - .expect_err("boolean pages must fail"); - - assert!( - matches!(error, Error::InvalidRequest(message) if message.contains("integers, not booleans")) - ); - } - - #[test] - fn document_intelligence_page_mapping_rejects_unsupported_type() { - let error = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!(5), - )])) - .expect_err("unsupported pages must fail"); - - assert!( - matches!(error, Error::InvalidRequest(message) if message.contains("Mistral-style")) - ); - } - - #[test] - fn document_intelligence_url_appends_pages_query() { - let url = complete_document_intelligence_url( - Some("https://example.cognitiveservices.azure.com/"), - "azure_ai/doc-intelligence/prebuilt-layout", - &Map::from_iter([("pages".to_string(), json!("1-3,5"))]), - &|_| None, - ) - .expect("url builds"); - - assert!(url.contains("api-version=2024-11-30")); - assert!(url.contains("pages=1-3,5")); - assert!(url.contains("/documentintelligence/documentModels/prebuilt-layout:analyze")); - } - - #[test] - fn document_intelligence_url_has_no_pages_when_params_are_empty() { - let url = complete_document_intelligence_url( - Some(ENDPOINT), - "prebuilt-layout", - &Map::new(), - &|_| None, - ) - .expect("url builds"); - - assert!(!url.contains("pages=")); - } - - #[rstest] - fn document_intelligence_request_keeps_pages_out_of_body( - document_intelligence_config: AzureDocumentIntelligenceOcrConfig, - ) { - let request = document_intelligence_config - .transform_ocr_request( - "prebuilt-layout", - json!({"type": "document_url", "document_url": "https://example.com/x.pdf"}), - Map::from_iter([("pages".to_string(), json!("1,2,3"))]), - ) - .expect("request transforms"); - - assert_eq!( - request.data, - json!({"urlSource": "https://example.com/x.pdf"}) - ); - } - - #[test] - fn document_intelligence_mistral_pages_flow_to_query_only() { - let mapped = map_document_intelligence_ocr_params(&Map::from_iter([( - "pages".to_string(), - json!([2, 3, 4, 5, 6, 7, 8]), - )])) - .expect("pages map"); - let url = - complete_document_intelligence_url(Some(ENDPOINT), "prebuilt-layout", &mapped, &|_| { - None - }) - .expect("url builds"); - let request = AZURE_DOCUMENT_INTELLIGENCE_OCR_CONFIG - .transform_ocr_request( - "prebuilt-layout", - json!({"type": "document_url", "document_url": "https://example.com/x.pdf"}), - mapped, - ) - .expect("request transforms"); - - assert!(url.contains("pages=3,4,5,6,7,8,9")); - assert_eq!( - request.data, - json!({"urlSource": "https://example.com/x.pdf"}) - ); - } - - #[test] - fn document_intelligence_endpoint_ignores_generic_azure_ai_base() { - let resolved = resolve_document_intelligence_endpoint(None, &|name| match name { - AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT_ENV => Some(ENDPOINT.to_string()), - AZURE_AI_API_BASE_ENV => Some("https://generic.example.com".to_string()), - _ => None, - }) - .expect("endpoint resolves"); - - assert_eq!(resolved, ENDPOINT); - } - - #[test] - fn document_intelligence_endpoint_honors_explicit_api_base() { - let resolved = resolve_document_intelligence_endpoint( - Some("https://my-di.cognitiveservices.azure.com"), - &|name| match name { - AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT_ENV => Some(ENDPOINT.to_string()), - AZURE_AI_API_BASE_ENV => Some("https://generic.example.com".to_string()), - _ => None, - }, - ) - .expect("endpoint resolves"); - - assert_eq!(resolved, "https://my-di.cognitiveservices.azure.com"); - } - - #[test] - fn azure_ai_mistral_ocr_uses_generic_api_base() { - let resolved = resolve_azure_ai_api_base(None, &|name| match name { - AZURE_AI_API_BASE_ENV => Some("https://generic-azure-ai.example.com".to_string()), - AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT_ENV => Some(ENDPOINT.to_string()), - _ => None, - }) - .expect("api base resolves"); - - assert_eq!(resolved, "https://generic-azure-ai.example.com"); - } - - #[test] - fn azure_ai_ocr_authenticates_with_entra_token() { - let headers = - validate_azure_ai_environment(Vec::new(), None, Some("entra-token"), &|_| None) - .expect("Entra token authenticates"); - - assert_eq!( - header_value(&headers, "Authorization"), - Some("Bearer entra-token") - ); - } -} diff --git a/litellm-rust/crates/core/src/providers/mistral/mod.rs b/litellm-rust/crates/core/src/providers/mistral/mod.rs deleted file mode 100644 index 3621ff6a2fd..00000000000 --- a/litellm-rust/crates/core/src/providers/mistral/mod.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod ocr; diff --git a/litellm-rust/crates/core/src/providers/mistral/ocr/mod.rs b/litellm-rust/crates/core/src/providers/mistral/ocr/mod.rs deleted file mode 100644 index f239b6921fa..00000000000 --- a/litellm-rust/crates/core/src/providers/mistral/ocr/mod.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod transformation; diff --git a/litellm-rust/crates/core/src/providers/mistral/ocr/transformation.rs b/litellm-rust/crates/core/src/providers/mistral/ocr/transformation.rs deleted file mode 100644 index 044fc587c22..00000000000 --- a/litellm-rust/crates/core/src/providers/mistral/ocr/transformation.rs +++ /dev/null @@ -1,436 +0,0 @@ -use crate::error::{Error, json_type_name}; -use crate::ocr::transformation::OcrProviderConfig; -use crate::ocr::types::{LiteLLMOcrResponse, OcrRequestData}; -use serde_json::{Map, Value}; - -const SUPPORTED_OCR_PARAMS: &[&str] = &[ - "pages", - "include_image_base64", - "image_limit", - "image_min_size", - "bbox_annotation_format", - "document_annotation_format", - "document_annotation_prompt", - "extract_header", - "extract_footer", - "table_format", - "confidence_scores_granularity", - "include_blocks", - "id", -]; - -/// Default Mistral API base, used when the caller does not override `api_base`. -pub const MISTRAL_DEFAULT_API_BASE: &str = "https://api.mistral.ai/v1"; - -/// Environment variable holding the Mistral API key. -pub const MISTRAL_API_KEY_ENV: &str = "MISTRAL_API_KEY"; - -/// Error message raised when no Mistral API key can be resolved. -pub const MISSING_KEY_MESSAGE: &str = "Missing Mistral API Key - A call is being made to Mistral but no key is set either in the environment variables or via params"; - -/// Build the complete OCR endpoint URL, de-duplicating a trailing `/v1`. -/// -/// Blank/whitespace `api_base` is treated as absent (guard at resolution time). -pub fn complete_url(api_base: Option<&str>) -> String { - let base = api_base - .map(str::trim) - .filter(|base| !base.is_empty()) - .unwrap_or(MISTRAL_DEFAULT_API_BASE) - .trim_end_matches('/'); - - if base.ends_with("/v1") { - format!("{base}/ocr") - } else { - format!("{base}/v1/ocr") - } -} - -/// Resolve the Mistral API key from the explicit param or the environment. -/// -/// Blank/whitespace values are treated as absent. Returns `Error::Auth` -/// when no usable key is available. -/// -/// Note: the env fallback only reads the process environment. Secret-manager -/// backends (AWS/Azure/GCP/Vault) are resolved on the Python side and passed in -/// via `api_key`; this fallback is a last resort for direct/standalone use. -pub fn resolve_api_key( - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - api_key - .map(str::trim) - .filter(|key| !key.is_empty()) - .map(str::to_string) - .or_else(|| env_lookup(MISTRAL_API_KEY_ENV).filter(|key| !key.trim().is_empty())) - .ok_or_else(|| Error::Auth(MISSING_KEY_MESSAGE.to_string())) -} - -pub struct MistralOcrConfig; - -pub const MISTRAL_OCR_CONFIG: MistralOcrConfig = MistralOcrConfig; - -impl OcrProviderConfig for MistralOcrConfig { - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn supported_ocr_params(&self) -> &'static [&'static str] { - SUPPORTED_OCR_PARAMS - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_request( - &self, - model: &str, - document: Value, - optional_params: Map, - ) -> Result { - if !document.is_object() { - return Err(Error::InvalidType { - expected: "object", - actual: json_type_name(&document), - }); - } - - let mut data = Map::new(); - data.insert("model".to_string(), Value::String(model.to_string())); - data.insert("document".to_string(), document); - for (param, value) in optional_params { - data.insert(param, value); - } - - Ok(OcrRequestData { - data: Value::Object(data), - files: None, - }) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - let response_object = response_json - .as_object() - .ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(&response_json), - })?; - - let pages = response_object - .get("pages") - .and_then(Value::as_array) - .cloned() - .unwrap_or_default(); - let model = response_object - .get("model") - .and_then(Value::as_str) - .unwrap_or(model) - .to_string(); - let document_annotation = response_object.get("document_annotation").cloned(); - let usage_info = response_object.get("usage_info").cloned(); - - Ok(LiteLLMOcrResponse { - pages, - model, - document_annotation, - usage_info, - object: "ocr".to_string(), - extra_fields: Map::new(), - provider_native_response: None, - }) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn complete_url( - &self, - api_base: Option<&str>, - _model: &str, - _optional_params: &Map, - _env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - Ok(complete_url(api_base)) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_api_key(api_key, env_lookup) - } -} - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub fn supported_ocr_params() -> &'static [&'static str] { - MISTRAL_OCR_CONFIG.supported_ocr_params() -} - -pub fn map_ocr_params(non_default_params: &Map) -> Map { - MISTRAL_OCR_CONFIG.map_ocr_params(non_default_params) -} - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub fn transform_ocr_request( - model: &str, - document: Value, - optional_params: Map, -) -> Result { - MISTRAL_OCR_CONFIG.transform_ocr_request(model, document, optional_params) -} - -#[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] -pub fn transform_ocr_response( - model: &str, - response_json: Value, -) -> Result { - MISTRAL_OCR_CONFIG.transform_ocr_response(model, response_json) -} - -#[cfg(test)] -mod tests { - use super::*; - use serde_json::json; - - #[test] - fn extract_header_is_a_supported_ocr_param() { - assert!(supported_ocr_params().contains(&"extract_header")); - } - - #[test] - fn extract_footer_is_a_supported_ocr_param() { - assert!(supported_ocr_params().contains(&"extract_footer")); - } - - #[test] - fn existing_ocr_params_remain_supported() { - for param in [ - "pages", - "include_image_base64", - "image_limit", - "image_min_size", - "bbox_annotation_format", - "document_annotation_format", - ] { - assert!(supported_ocr_params().contains(¶m)); - } - } - - #[test] - fn map_ocr_params_forwards_extract_header() { - let params = json!({"extract_header": true}); - assert_eq!( - map_ocr_params(params.as_object().unwrap()), - params.as_object().unwrap().clone() - ); - } - - #[test] - fn map_ocr_params_forwards_extract_footer() { - let params = json!({"extract_footer": true}); - assert_eq!( - map_ocr_params(params.as_object().unwrap()), - params.as_object().unwrap().clone() - ); - } - - #[test] - fn map_ocr_params_forwards_extract_header_and_footer() { - let params = json!({"extract_header": true, "extract_footer": false}); - assert_eq!( - map_ocr_params(params.as_object().unwrap()), - params.as_object().unwrap().clone() - ); - } - - #[test] - fn map_ocr_params_drops_unknown_params() { - let params = json!({"extract_header": true, "unsupported_param": "value"}); - let mapped = map_ocr_params(params.as_object().unwrap()); - assert_eq!(mapped.get("extract_header"), Some(&json!(true))); - assert!(!mapped.contains_key("unsupported_param")); - } - - #[test] - fn new_ocr_params_are_supported() { - for param in [ - "table_format", - "confidence_scores_granularity", - "document_annotation_prompt", - "include_blocks", - "id", - ] { - assert!(supported_ocr_params().contains(¶m)); - } - } - - #[test] - fn map_ocr_params_forwards_new_ocr_params() { - for (param, value) in [ - ("table_format", json!("html")), - ("confidence_scores_granularity", json!("word")), - ( - "document_annotation_prompt", - json!("Extract all invoice line items"), - ), - ("include_blocks", json!(true)), - ("id", json!("req-123")), - ] { - let params = json!({param: value}); - assert_eq!( - map_ocr_params(params.as_object().unwrap()), - params.as_object().unwrap().clone() - ); - } - } - - #[test] - fn transform_ocr_request_includes_each_optional_param() { - let document = json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }); - for (param, value) in [ - ("table_format", json!("html")), - ("confidence_scores_granularity", json!("word")), - ( - "document_annotation_prompt", - json!("Extract all invoice line items"), - ), - ("id", json!("req-123")), - ("extract_header", json!(true)), - ("include_blocks", json!(true)), - ("pages", json!([0, 1])), - ] { - let result = transform_ocr_request( - "mistral-ocr-latest", - document.clone(), - json!({param: value}).as_object().unwrap().clone(), - ) - .expect("request should transform"); - assert_eq!(result.data.get(param), Some(&value)); - assert_eq!(result.data.get("model"), Some(&json!("mistral-ocr-latest"))); - assert_eq!(result.data.get("document"), Some(&document)); - assert_eq!(result.files, None); - } - } - - #[test] - fn transform_ocr_request_includes_multiple_new_params() { - let document = json!({ - "type": "document_url", - "document_url": "https://example.com/doc.pdf" - }); - let optional_params = json!({ - "table_format": "html", - "confidence_scores_granularity": "page", - "extract_header": true - }) - .as_object() - .unwrap() - .clone(); - let result = transform_ocr_request("mistral-ocr-latest", document, optional_params) - .expect("request should transform"); - assert_eq!(result.data.get("table_format"), Some(&json!("html"))); - assert_eq!( - result.data.get("confidence_scores_granularity"), - Some(&json!("page")) - ); - assert_eq!(result.data.get("extract_header"), Some(&json!(true))); - } - - #[test] - fn transform_ocr_response_preserves_blocks_and_confidence_scores() { - let blocks = json!([{"type": "title", "content": "Invoice"}]); - let confidence_scores = json!({"page": 0.98}); - let response = json!({ - "pages": [{"index": 0, "markdown": "# Invoice", "blocks": blocks, "confidence_scores": confidence_scores}], - "model": "mistral-ocr-4-0", - "usage_info": {"pages_processed": 1} - }); - let result = - transform_ocr_response("mistral-ocr-4-0", response).expect("response should transform"); - assert_eq!(result.pages[0].get("blocks"), Some(&blocks)); - assert_eq!( - result.pages[0].get("confidence_scores"), - Some(&confidence_scores) - ); - } - - #[test] - fn transform_ocr_response_preserves_ocr4_page_fields() { - let response = json!({ - "pages": [{"index": 0, "markdown": "table page", "tables": [{"rows": 2, "cols": 3}], "hyperlinks": ["https://example.com"], "header": "Acme Corp", "footer": "Page 1"}], - "model": "mistral-ocr-4-0", - "usage_info": {"pages_processed": 1} - }); - let result = transform_ocr_response("mistral-ocr-4-0", response.clone()) - .expect("response should transform"); - assert_eq!(result.pages[0], response["pages"][0]); - } - - #[test] - fn transform_ocr_request_rejects_non_object_document() { - let err = transform_ocr_request("mistral-ocr-latest", json!("bad"), Map::new()) - .expect_err("string document should be rejected"); - - assert_eq!( - err, - Error::InvalidType { - expected: "object", - actual: "string", - } - ); - } - - #[test] - fn transform_ocr_response_normalizes_mistral_json() { - let response = json!({ - "pages": [{"index": 0, "markdown": "hello"}], - "model": "mistral-ocr-2505-completion", - "document_annotation": null, - "usage_info": {"pages_processed": 1} - }); - - let result = transform_ocr_response("mistral-ocr-latest", response) - .expect("response should transform"); - - assert_eq!(result.pages, vec![json!({"index": 0, "markdown": "hello"})]); - assert_eq!(result.model, "mistral-ocr-2505-completion"); - assert_eq!(result.document_annotation, Some(Value::Null)); - assert_eq!(result.usage_info, Some(json!({"pages_processed": 1}))); - assert_eq!(result.object, "ocr"); - } - - #[test] - fn complete_url_defaults_and_dedupes_v1() { - assert_eq!(complete_url(None), "https://api.mistral.ai/v1/ocr"); - assert_eq!(complete_url(Some(" ")), "https://api.mistral.ai/v1/ocr"); - assert_eq!( - complete_url(Some("https://proxy.internal")), - "https://proxy.internal/v1/ocr" - ); - assert_eq!( - complete_url(Some("https://proxy.internal/v1/")), - "https://proxy.internal/v1/ocr" - ); - } - - #[test] - fn resolve_api_key_prefers_param_then_env() { - let no_env = |_: &str| None; - assert_eq!( - resolve_api_key(Some("sk-param"), &no_env).unwrap(), - "sk-param" - ); - - let with_env = |key: &str| (key == MISTRAL_API_KEY_ENV).then(|| "sk-env".to_string()); - assert_eq!(resolve_api_key(None, &with_env).unwrap(), "sk-env"); - // Blank param falls through to the environment. - assert_eq!(resolve_api_key(Some(" "), &with_env).unwrap(), "sk-env"); - } - - #[test] - fn resolve_api_key_errors_when_absent() { - let err = resolve_api_key(None, &|_| None).expect_err("missing key should error"); - assert_eq!(err, Error::Auth(MISSING_KEY_MESSAGE.to_string())); - } -} diff --git a/litellm-rust/crates/core/src/providers/mod.rs b/litellm-rust/crates/core/src/providers/mod.rs index 805600d6dbe..1aeb75063d6 100644 --- a/litellm-rust/crates/core/src/providers/mod.rs +++ b/litellm-rust/crates/core/src/providers/mod.rs @@ -2,6 +2,4 @@ pub mod anthropic; pub mod azure_ai; #[cfg(feature = "bedrock-auth")] pub mod bedrock; -pub mod mistral; pub mod openai; -pub mod vertex_ai; diff --git a/litellm-rust/crates/core/src/providers/vertex_ai/mod.rs b/litellm-rust/crates/core/src/providers/vertex_ai/mod.rs deleted file mode 100644 index 3621ff6a2fd..00000000000 --- a/litellm-rust/crates/core/src/providers/vertex_ai/mod.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod ocr; diff --git a/litellm-rust/crates/core/src/providers/vertex_ai/ocr/mod.rs b/litellm-rust/crates/core/src/providers/vertex_ai/ocr/mod.rs deleted file mode 100644 index f239b6921fa..00000000000 --- a/litellm-rust/crates/core/src/providers/vertex_ai/ocr/mod.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod transformation; diff --git a/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs b/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs deleted file mode 100644 index c2d810822ae..00000000000 --- a/litellm-rust/crates/core/src/providers/vertex_ai/ocr/transformation.rs +++ /dev/null @@ -1,361 +0,0 @@ -use crate::error::{Error, json_type_name}; -use crate::ocr::transformation::OcrProviderConfig; -use crate::ocr::types::{LiteLLMOcrResponse, OcrRequestData}; -use serde_json::{Map, Value, json}; - -const VERTEX_DEFAULT_LOCATION: &str = "us-central1"; -const VERTEX_DEFAULT_DEEPSEEK_API_BASE: &str = "https://aiplatform.googleapis.com"; -const VERTEX_AI_API_KEY_ENV: &str = "VERTEX_AI_API_KEY"; -const VERTEXAI_API_KEY_ENV: &str = "VERTEXAI_API_KEY"; -const VERTEXAI_PROJECT_ENV: &str = "VERTEXAI_PROJECT"; -const VERTEXAI_LOCATION_ENV: &str = "VERTEXAI_LOCATION"; -const VERTEX_LOCATION_ENV: &str = "VERTEX_LOCATION"; - -#[rustfmt::skip] -const DEEPSEEK_SUPPORTED_OCR_PARAMS: &[&str] = &[ - "stream", - "temperature", - "max_tokens", - "top_p", - "n", - "stop", -]; - -pub struct VertexAiDeepSeekOcrConfig; - -pub const VERTEX_AI_DEEPSEEK_OCR_CONFIG: VertexAiDeepSeekOcrConfig = VertexAiDeepSeekOcrConfig; - -fn string_param<'a>(params: &'a Map, keys: &[&str]) -> Option<&'a str> { - keys.iter() - .find_map(|key| params.get(*key).and_then(Value::as_str)) - .map(str::trim) - .filter(|value| !value.is_empty()) -} - -pub fn is_deepseek_model(model: &str) -> bool { - model.to_ascii_lowercase().contains("deepseek") -} - -pub fn resolve_vertex_api_key( - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - api_key - .map(str::trim) - .filter(|key| !key.is_empty()) - .map(str::to_string) - .or_else(|| env_lookup(VERTEX_AI_API_KEY_ENV).filter(|key| !key.trim().is_empty())) - .or_else(|| env_lookup(VERTEXAI_API_KEY_ENV).filter(|key| !key.trim().is_empty())) - .ok_or_else(|| { - Error::Auth( - "Missing Vertex AI access token - pass api_key or provide Authorization via extra_headers" - .to_string(), - ) - }) -} - -fn vertex_project( - params: &Map, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - string_param(params, &["vertex_project", "vertex_ai_project"]) - .map(str::to_string) - .or_else(|| env_lookup(VERTEXAI_PROJECT_ENV).filter(|value| !value.trim().is_empty())) - .ok_or_else(|| { - Error::InvalidRequest( - "Missing vertex_project - Set VERTEXAI_PROJECT environment variable or pass vertex_project parameter" - .to_string(), - ) - }) -} - -fn vertex_location( - params: &Map, - env_lookup: &dyn Fn(&str) -> Option, -) -> String { - string_param(params, &["vertex_location", "vertex_ai_location"]) - .map(str::to_string) - .or_else(|| env_lookup(VERTEXAI_LOCATION_ENV).filter(|value| !value.trim().is_empty())) - .or_else(|| env_lookup(VERTEX_LOCATION_ENV).filter(|value| !value.trim().is_empty())) - .unwrap_or_else(|| VERTEX_DEFAULT_LOCATION.to_string()) -} - -pub fn complete_vertex_deepseek_url( - api_base: Option<&str>, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result { - let project = vertex_project(optional_params, env_lookup)?; - let location = vertex_location(optional_params, env_lookup); - let base = api_base - .map(str::trim) - .filter(|value| !value.is_empty()) - .unwrap_or(VERTEX_DEFAULT_DEEPSEEK_API_BASE) - .trim_end_matches('/'); - Ok(format!( - "{base}/v1/projects/{project}/locations/{location}/endpoints/openapi/chat/completions" - )) -} - -fn document_content_item(document: &Value) -> Result { - let object = document.as_object().ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(document), - })?; - let doc_type = object - .get("type") - .and_then(Value::as_str) - .ok_or(Error::MissingField("document.type"))?; - let url_field = match doc_type { - "image_url" => "image_url", - "document_url" => "document_url", - other => { - return Err(Error::InvalidRequest(format!( - "Unsupported document type: {other}. Expected 'image_url' or 'document_url'" - ))); - } - }; - let url = object - .get(url_field) - .and_then(Value::as_str) - .filter(|value| !value.is_empty()) - .ok_or(Error::MissingField(url_field))?; - - Ok(json!({ - "type": "image_url", - "image_url": url, - })) -} - -fn deepseek_model_name(model: &str) -> String { - if model.starts_with("deepseek-ai/") { - model.to_string() - } else { - format!("deepseek-ai/{model}") - } -} - -fn first_choice_content(response: &Value) -> Result { - response - .get("choices") - .and_then(Value::as_array) - .and_then(|choices| choices.first()) - .and_then(|choice| choice.get("message")) - .and_then(|message| message.get("content")) - .cloned() - .filter(|content| match content { - Value::String(value) => !value.is_empty(), - Value::Object(_) => true, - _ => false, - }) - .ok_or_else(|| Error::InvalidResponse("No content in DeepSeek OCR response".to_string())) -} - -fn ocr_data_from_content(content: Value, usage: Option, model: &str) -> Value { - match content { - Value::String(content) => { - if content.trim_start().starts_with('{') { - serde_json::from_str(&content).unwrap_or_else(|_| { - json!({ - "pages": [{"index": 0, "markdown": content}], - "model": model, - "usage_info": usage.unwrap_or_else(|| json!({})), - }) - }) - } else { - json!({ - "pages": [{"index": 0, "markdown": content}], - "model": model, - "usage_info": usage.unwrap_or_else(|| json!({})), - }) - } - } - Value::Object(_) => content, - other => json!({ - "pages": [{"index": 0, "markdown": other.to_string()}], - "model": model, - "usage_info": usage.unwrap_or_else(|| json!({})), - }), - } -} - -impl OcrProviderConfig for VertexAiDeepSeekOcrConfig { - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn supported_ocr_params(&self) -> &'static [&'static str] { - DEEPSEEK_SUPPORTED_OCR_PARAMS - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn map_ocr_params(&self, non_default_params: &Map) -> Map { - non_default_params - .iter() - .filter(|(name, _)| DEEPSEEK_SUPPORTED_OCR_PARAMS.contains(&name.as_str())) - .map(|(name, value)| (name.clone(), value.clone())) - .collect() - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_request( - &self, - model: &str, - document: Value, - optional_params: Map, - ) -> Result { - let mut data = Map::new(); - data.insert( - "model".to_string(), - Value::String(deepseek_model_name(model)), - ); - data.insert( - "messages".to_string(), - json!([{"role": "user", "content": [document_content_item(&document)?]}]), - ); - for (key, value) in optional_params { - if DEEPSEEK_SUPPORTED_OCR_PARAMS.contains(&key.as_str()) { - data.insert(key, value); - } - } - Ok(OcrRequestData { - data: Value::Object(data), - files: None, - }) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn transform_ocr_response( - &self, - model: &str, - response_json: Value, - ) -> Result { - let response = response_json - .as_object() - .ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(&response_json), - })?; - let usage = response.get("usage").cloned(); - let content = first_choice_content(&response_json)?; - let mut ocr_data = ocr_data_from_content(content.clone(), usage.clone(), model); - - if !ocr_data.get("pages").is_some_and(Value::is_array) { - ocr_data = json!({ - "pages": [{ - "index": 0, - "markdown": match content { - Value::String(value) => value, - other => other.to_string(), - } - }], - "model": ocr_data.get("model").and_then(Value::as_str).unwrap_or(model), - "usage_info": ocr_data.get("usage_info").cloned().or(usage).unwrap_or_else(|| json!({})), - }); - } - - let object = ocr_data.as_object().ok_or_else(|| Error::InvalidType { - expected: "object", - actual: json_type_name(&ocr_data), - })?; - let pages = object - .get("pages") - .and_then(Value::as_array) - .cloned() - .unwrap_or_default(); - let usage_info = object - .get("usage_info") - .cloned() - .or_else(|| response.get("usage").cloned()); - Ok(LiteLLMOcrResponse { - pages, - model: object - .get("model") - .and_then(Value::as_str) - .unwrap_or(model) - .to_string(), - document_annotation: object.get("document_annotation").cloned(), - usage_info, - object: "ocr".to_string(), - extra_fields: Map::new(), - provider_native_response: None, - }) - } - - #[tracing::instrument(target = "litellm::function_trace", level = "trace", skip_all)] - fn complete_url( - &self, - api_base: Option<&str>, - _model: &str, - optional_params: &Map, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - complete_vertex_deepseek_url(api_base, optional_params, env_lookup) - } - - fn resolve_api_key( - &self, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, - ) -> Result { - resolve_vertex_api_key(api_key, env_lookup) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use rstest::rstest; - - #[test] - fn vertex_deepseek_request_uses_ocr_endpoint_shape() { - let body = VERTEX_AI_DEEPSEEK_OCR_CONFIG - .transform_ocr_request( - "deepseek-ocr-maas", - json!({"type": "document_url", "document_url": "gs://bucket/doc.pdf"}), - Map::from_iter([("temperature".to_string(), json!(0.1))]), - ) - .expect("request transforms") - .data; - - assert_eq!(body["model"], "deepseek-ai/deepseek-ocr-maas"); - assert_eq!(body["temperature"], 0.1); - assert_eq!( - body["messages"][0]["content"][0], - json!({"type": "image_url", "image_url": "gs://bucket/doc.pdf"}) - ); - } - - #[rstest] - #[case::bare_model("deepseek-ocr-maas")] - #[case::namespaced_model("deepseek-ai/deepseek-ocr-maas")] - fn vertex_deepseek_request_uses_single_provider_namespace(#[case] model: &str) { - let body = VERTEX_AI_DEEPSEEK_OCR_CONFIG - .transform_ocr_request( - model, - json!({"type": "image_url", "image_url": "data:image/png;base64,AA=="}), - Map::new(), - ) - .expect("request transforms") - .data; - - assert_eq!(body["model"], "deepseek-ai/deepseek-ocr-maas"); - } - - #[test] - fn vertex_deepseek_response_wraps_markdown_content() { - let response = VERTEX_AI_DEEPSEEK_OCR_CONFIG - .transform_ocr_response( - "deepseek-ocr-maas", - json!({ - "choices": [{"message": {"content": "# OCR text"}}], - "usage": {"prompt_tokens": 1} - }), - ) - .expect("response transforms"); - - assert_eq!( - response.pages, - vec![json!({"index": 0, "markdown": "# OCR text"})] - ); - assert_eq!(response.model, "deepseek-ocr-maas"); - assert_eq!(response.usage_info, Some(json!({"prompt_tokens": 1}))); - } -} diff --git a/litellm-rust/crates/core/tests/deepseek_ocr.rs b/litellm-rust/crates/core/tests/deepseek_ocr.rs new file mode 100644 index 00000000000..875fc9e3dc6 --- /dev/null +++ b/litellm-rust/crates/core/tests/deepseek_ocr.rs @@ -0,0 +1,95 @@ +use rstest::rstest; +use serde_json::{Value, json}; + +use crate::ocr::codecs::deepseek::{ + DeepSeekOcrParams, DeepSeekOcrResponse, transform_ocr_request, transform_ocr_response, +}; +use crate::ocr::types::OcrDocument; + +fn document() -> OcrDocument { + serde_json::from_value(json!({"type":"image_url","image_url":"gs://bucket/a.png"})).unwrap() +} + +#[rstest] +#[case("stream", json!(true))] +#[case("temperature", json!(0.1))] +#[case("max_tokens", json!(1024))] +#[case("top_p", json!(0.9))] +#[case("n", json!(2))] +#[case("stop", json!("done"))] +#[case("stop", json!(["done", "stop"]))] +fn request_mapping_matches_python(#[case] name: &str, #[case] value: Value) { + let params: DeepSeekOcrParams = + serde_json::from_value(json!({name: value.clone(), "ignored": true})).unwrap(); + let result = serde_json::to_value( + transform_ocr_request("deepseek-ai/deepseek-ocr-maas", document(), ¶ms).unwrap(), + ) + .unwrap(); + assert_eq!(result["model"], "deepseek-ai/deepseek-ocr-maas"); + assert_eq!( + result["messages"][0]["content"][0], + json!({"type":"image_url","image_url":"gs://bucket/a.png"}) + ); + assert_eq!(result[name], value); + assert!(result.get("ignored").is_none()); +} + +#[rstest] +#[case(json!("# hello"), "# hello")] +#[case(json!("{broken"), "{broken")] +#[case(json!(" {\"pages\":[]} "), " {\"pages\":[]} ")] +#[case(json!({"pages":[]}), "{\"pages\":[]}")] +#[case(json!({}), "{}")] +#[case(json!("[]"), "[]")] +#[case(json!("{\"pages\":[{\"markdown\":\"json text\"}]}"), "json text")] +#[case(json!({"pages":[{"markdown":"object"}]}), "object")] +fn response_codec_handles_text_json_and_objects(#[case] content: Value, #[case] expected: &str) { + let response: DeepSeekOcrResponse = serde_json::from_value( + json!({"choices":[{"message":{"content":content}}],"usage":{"prompt_tokens":1}}), + ) + .unwrap(); + let result = transform_ocr_response("model", response) + .unwrap() + .into_json(); + assert_eq!(result["pages"][0]["markdown"], expected); + assert_eq!(result["pages"][0]["index"], 0); + assert_eq!(result["usage_info"]["prompt_tokens"], 1); +} + +#[test] +fn structured_result_maps_pages_usage_model_and_annotation() { + let response: DeepSeekOcrResponse = serde_json::from_value(json!({ + "choices":[{"message":{"content":{ + "pages":[{"index":2,"markdown":"page","images":[{"id":"one"}],"dimensions":{"width":10}}], + "model":"provider-model", + "usage_info":{"pages_processed":1}, + "document_annotation":{"language":"en"}, + "future":"kept" + }}}] + })) + .unwrap(); + let result = transform_ocr_response("requested", response) + .unwrap() + .into_json(); + assert_eq!(result["pages"][0]["index"], 2); + assert_eq!(result["pages"][0]["images"][0]["id"], "one"); + assert_eq!(result["model"], "provider-model"); + assert_eq!(result["usage_info"]["pages_processed"], 1); + assert_eq!(result["document_annotation"]["language"], "en"); + assert_eq!(result["future"], "kept"); +} + +#[test] +fn response_codec_rejects_missing_empty_and_malformed_content() { + for value in [ + json!({"choices":[]}), + json!({"choices":[{"message":{"content":""}}]}), + json!({"choices":[{"message":{"content":"{\"pages\":[{\"markdown\":42}]}"}}]}), + json!({"choices":[{"message":{"content":{"pages":[{"markdown":42}]}}}]}), + ] { + let result = serde_json::from_value::(value) + .map_err(|_| ()) + .and_then(|response| transform_ocr_response("model", response).map_err(|_| ())); + assert!(result.is_err()); + } +} diff --git a/litellm-rust/crates/core/tests/vertex_ai_deepseek_ocr.rs b/litellm-rust/crates/core/tests/vertex_ai_deepseek_ocr.rs new file mode 100644 index 00000000000..6d3061d8f5d --- /dev/null +++ b/litellm-rust/crates/core/tests/vertex_ai_deepseek_ocr.rs @@ -0,0 +1,83 @@ +use serde_json::{Value, json}; + +use super::test_support::{MockResponse, mock_server, perform_ocr, wire_request}; +use crate::auth::InputSource; + +fn request_body(request: &str) -> Value { + serde_json::from_str(request.split_once("\r\n\r\n").unwrap().1).unwrap() +} + +#[tokio::test] +async fn facade_executes_vertex_deepseek_at_the_openai_endpoint() { + let (base, seen, server) = mock_server(vec![MockResponse::json(json!({ + "choices":[{"message":{"content":"recognized"}}], + "usage":{"prompt_tokens":1} + }))]) + .await; + let mut request = wire_request( + "vertex_ai/deepseek-ocr-maas", + &base, + json!({ + "vertex_project":"project-1", + "vertex_location":"europe-west4", + "temperature":0.1, + "future_ocr_option":true, + "extra_body":{"provider_option":"value"} + }), + ); + request.document = request + .document + .with_source("gs://bucket/document.pdf".into()); + + let response = perform_ocr(request).await.unwrap(); + server.await.unwrap(); + assert_eq!(response.pages[0]["markdown"], "recognized"); + assert_eq!(response.usage_info.unwrap()["prompt_tokens"], 1); + let requests = seen.lock().unwrap(); + assert!(requests[0].starts_with( + "POST /v1/projects/project-1/locations/europe-west4/endpoints/openapi/chat/completions " + )); + assert!( + requests[0] + .to_ascii_lowercase() + .contains("authorization: bearer test-key") + ); + let body = request_body(&requests[0]); + assert_eq!(body["model"], "deepseek-ai/deepseek-ocr-maas"); + assert_eq!(body["temperature"], 0.1); + assert!(body.get("future_ocr_option").is_none()); + assert!(body.get("extra_body").is_none()); + assert_eq!( + body["messages"][0]["content"][0], + json!({"type":"document_url","document_url":"gs://bucket/document.pdf"}) + ); +} + +#[test] +fn host_registration_selects_deepseek_without_affecting_mistral() { + assert!(crate::ocr::wire::is_supported_request( + "deepseek-ocr-maas", + Some("vertex_ai") + )); + assert!(crate::ocr::wire::is_supported_request( + "mistral-ocr-maas", + Some("vertex_ai") + )); +} + +#[tokio::test] +async fn request_controlled_api_base_is_rejected_before_vertex_auth() { + let mut request = wire_request( + "vertex_ai/deepseek-ocr-maas", + "https://caller.example", + json!({"vertex_project":"project-1"}), + ); + request.connection.api_base_source = InputSource::Request; + + let error = perform_ocr(request).await.unwrap_err(); + assert!( + error + .to_string() + .contains("request-controlled Vertex AI endpoint") + ); +} diff --git a/litellm-rust/crates/core/tests/vertex_ai_ocr.rs b/litellm-rust/crates/core/tests/vertex_ai_ocr.rs index 2358552c742..96a19dd62b4 100644 --- a/litellm-rust/crates/core/tests/vertex_ai_ocr.rs +++ b/litellm-rust/crates/core/tests/vertex_ai_ocr.rs @@ -2,7 +2,6 @@ use serde_json::{Value, json}; use super::test_support::{MockResponse, mock_server, perform_ocr, wire_request}; use crate::auth::InputSource; -use crate::ocr::wire::{OcrWireRequest, decode_request}; fn request_body(request: &str) -> Value { serde_json::from_str(request.split_once("\r\n\r\n").unwrap().1).unwrap() @@ -81,30 +80,82 @@ async fn invalid_credentials_fail_before_provider_http() { #[tokio::test] async fn request_controlled_api_base_is_rejected_before_vertex_auth() { - let request = decode_request(OcrWireRequest { - model: "vertex_ai/model".into(), - document: json!({"type":"document_url","document_url":"data:application/pdf;base64,YWJj"}), - api_key: Some("test-key".into()), - api_base: Some("https://attacker.example".into()), - custom_llm_provider: None, - extra_headers: None, - optional_params: json!({"vertex_project":"project-1"}) - .as_object() - .unwrap() - .clone(), - input_sources: std::collections::BTreeMap::from([( - "api_base".to_string(), - InputSource::Request, - )]), - timeout_seconds: Some(2.0), - }) - .unwrap(); + let mut request = wire_request( + "vertex_ai/mistral-ocr-maas", + "https://caller.example", + json!({"vertex_project":"project-1"}), + ); + request.connection.api_base_source = InputSource::Request; let error = perform_ocr(request).await.unwrap_err(); - assert!( error .to_string() .contains("request-controlled Vertex AI endpoint") ); } + +#[tokio::test] +async fn adapters_build_complete_requests_and_share_mistral_normalization() { + use std::time::Duration; + + use crate::ocr::adapters::{MistralAdapter, OcrAdapter, VertexMistralAdapter}; + use crate::ocr::test_support::ocr_client; + + let client = ocr_client(); + let options = json!({ + "pages": [0, 2], + "include_image_base64": true, + "vertex_project": "project-1", + "vertex_location": "us-central1", + "unknown": "ignored" + }); + let direct = wire_request( + "mistral/mistral-ocr-maas", + "https://mistral.test", + options.clone(), + ); + let vertex = wire_request("vertex_ai/mistral-ocr-maas", "https://vertex.test", options); + let direct_http = MistralAdapter + .prepare_request(&direct, &client) + .await + .unwrap(); + let vertex_http = VertexMistralAdapter + .prepare_request(&vertex, &client) + .await + .unwrap(); + assert_eq!(direct_http.url().as_str(), "https://mistral.test/v1/ocr"); + assert_eq!( + vertex_http.url().as_str(), + "https://vertex.test/v1/projects/project-1/locations/us-central1/publishers/mistralai/models/mistral-ocr-maas:rawPredict" + ); + for http in [&direct_http, &vertex_http] { + assert_eq!(http.method(), reqwest::Method::POST); + assert_eq!(http.headers()["authorization"], "Bearer test-key"); + assert_eq!(http.headers()["content-type"], "application/json"); + assert_eq!(http.timeout(), Some(&Duration::from_secs(2))); + let body: Value = serde_json::from_slice(http.body().unwrap().as_bytes().unwrap()).unwrap(); + assert_eq!( + body, + json!({ + "model": "mistral-ocr-maas", + "document": {"type": "document_url", "document_url": "data:application/pdf;base64,YWJj"}, + "pages": [0, 2], + "include_image_base64": true + }) + ); + } + let payload = json!({"pages": [{"index": 0, "markdown": "hello"}], "extra": "preserved"}); + let direct_response = MistralAdapter + .transform_ocr_response(&direct, serde_json::from_value(payload.clone()).unwrap()) + .unwrap() + .into_json(); + let vertex_response = VertexMistralAdapter + .transform_ocr_response(&vertex, serde_json::from_value(payload).unwrap()) + .unwrap() + .into_json(); + assert_eq!(direct_response, vertex_response); + assert_eq!(direct_response["model"], "mistral-ocr-maas"); + assert_eq!(direct_response["object"], "ocr"); + assert_eq!(direct_response["extra"], "preserved"); +} diff --git a/litellm-rust/crates/python-bridge/src/errors.rs b/litellm-rust/crates/python-bridge/src/errors.rs index 864f30db6a9..e1f458ea0bc 100644 --- a/litellm-rust/crates/python-bridge/src/errors.rs +++ b/litellm-rust/crates/python-bridge/src/errors.rs @@ -43,7 +43,6 @@ pub(crate) fn chat_completions_error_to_pyerr(err: Error) -> PyErr { | Error::MissingField(_) | Error::MissingApiKey { .. } | Error::MissingAzureAiCredentials - | Error::MissingAzureAiCredentialsOrAdToken | Error::MissingAzureDocumentIntelligenceCredentials | Error::MissingReductoApiKey | Error::Routing(_) diff --git a/litellm-rust/crates/python-bridge/src/routes/ocr.rs b/litellm-rust/crates/python-bridge/src/routes/ocr.rs index 0aab11e3cfc..c5def64c2f1 100644 --- a/litellm-rust/crates/python-bridge/src/routes/ocr.rs +++ b/litellm-rust/crates/python-bridge/src/routes/ocr.rs @@ -112,6 +112,6 @@ mod tests { assert!(is_supported_request("parse-v3", Some("reducto"))); assert!(is_supported_request("parse-legacy", Some("reducto"))); assert!(is_supported_request("mistral-ocr", Some("vertex_ai"))); - assert!(!is_supported_request("deepseek-ocr", Some("vertex_ai"))); + assert!(is_supported_request("deepseek-ocr", Some("vertex_ai"))); } } From d78861bb290852c774b290ed503aee1d2e696f29 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 16:44:56 -0700 Subject: [PATCH 132/157] fix(guardrails): stop logging the request payload as guardrail_response on pre_call hooks (#39699) * fix(guardrails): stop logging the request payload as guardrail_response on pre_call hooks Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(guardrails): snapshot pre_call request before the hook so in-place edits log as mask Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(guardrails): log non-mapping pre_call hook results as mask instead of raising Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(guardrails): treat legacy functions and tool_choice edits as mask in pre_call logging Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(guardrails): log a pre_call rejection string as is instead of "mask" Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: shivam Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: yucheng --- litellm/integrations/custom_guardrail.py | 92 ++++++++----- .../guardrail_hooks/azure/prompt_shield.py | 7 +- .../integrations/test_custom_guardrail.py | 128 ++++++++++++++++++ 3 files changed, 187 insertions(+), 40 deletions(-) diff --git a/litellm/integrations/custom_guardrail.py b/litellm/integrations/custom_guardrail.py index 77bf4820a1a..bb54767edef 100644 --- a/litellm/integrations/custom_guardrail.py +++ b/litellm/integrations/custom_guardrail.py @@ -1130,7 +1130,7 @@ class CustomGuardrail(CustomLogger): def add_standard_logging_guardrail_information_to_request_data( self, - guardrail_json_response: Exception | str | dict | list[dict], + guardrail_json_response: object, request_data: dict, guardrail_status: GuardrailStatus, start_time: float | None = None, @@ -1275,17 +1275,10 @@ class CustomGuardrail(CustomLogger): This gets logged on downsteam Langfuse, DataDog, etc. """ - # Convert None to empty dict to satisfy type requirements - guardrail_response: dict[str, object] | str = {} if response is None else response - - # For apply_guardrail functions in custom_code_guardrail scenario, - # simplify the logged response to "allow", "deny", or "mask" - if original_inputs is not None and isinstance(response, dict): - # Check if inputs were modified by comparing them - if self._inputs_were_modified(original_inputs, response): - guardrail_response = "mask" - else: - guardrail_response = "allow" + guardrail_response: Final = self._summarize_guardrail_response( + response=response, + original_inputs=original_inputs, + ) verbose_logger.debug("Guardrail response: %s", response) @@ -1300,6 +1293,27 @@ class CustomGuardrail(CustomLogger): ) return response + def _summarize_guardrail_response( + self, + response: object, + original_inputs: Mapping[str, object] | None, + ) -> object: + """Reduce a hook's return value to what is safe to log as ``guardrail_response``. + + ``apply_guardrail`` returns the (possibly masked) inputs and ``async_pre_call_hook`` + returns the (possibly modified) request payload. Neither is a provider verdict, and + logging them verbatim ships the user's prompt to every logging sink (OTEL spans, + Datadog, spend logs), so both collapse to ``"allow"`` / ``"mask"`` by comparing + against ``original_inputs``, a copy taken before the hook ran. A string result is the + hook's own rejection message (the proxy turns it into a 400), not user input, so it is + logged as is. + """ + if response is None: + return {} + if original_inputs is None or not isinstance(response, Mapping): + return response + return "mask" if self._inputs_were_modified(original_inputs, response) else "allow" + @staticmethod def _is_guardrail_intervention(e: Exception) -> bool: """Retained spelling for existing callers; prefer ``is_guardrail_intervention``.""" @@ -1339,24 +1353,9 @@ class CustomGuardrail(CustomLogger): ) raise e - def _inputs_were_modified(self, original_inputs: dict, response: dict) -> bool: - """ - Compare original inputs with response to determine if content was modified. - - Returns True if the inputs were modified (mask scenario), False otherwise (allow scenario). - """ - # Get all keys from both dictionaries - all_keys: Final = set(original_inputs.keys()) | set(response.keys()) - - # Compare each key's value - for key in all_keys: - original_value = original_inputs.get(key) - response_value = response.get(key) - if original_value != response_value: - return True - - # No modifications detected - return False + def _inputs_were_modified(self, original_inputs: Mapping[str, object], response: Mapping[str, object]) -> bool: + """True when any baseline key's value differs in ``response`` (mask), False otherwise (allow).""" + return any(response.get(key) != value for key, value in original_inputs.items()) def mask_content_in_string( self, @@ -1463,6 +1462,31 @@ def _sync_guardrail_info_to_logging_obj(request_data: dict, logging_obj: object) _append_slg_to_litellm_params(mcd.get("litellm_params"), entries) +_PRE_CALL_CONTENT_KEYS: Final = frozenset( + {"messages", "input", "prompt", "system", "instructions", "tools", "functions", "function_call", "tool_choice"} +) + + +def _original_inputs_for( + func_name: str, + kwargs: Mapping[str, object], + request_data: Mapping[str, object], + event_type: GuardrailEventHooks | None, +) -> dict | None: # mutable-ok: matches _process_response(original_inputs=) signature + """Baseline the hook's return value is compared against to decide "allow" vs "mask". + + ``apply_guardrail`` masks a fresh ``inputs`` dict, so that dict is the baseline. Pre-call + hooks edit the request in place and return it, so the baseline is a deep copy of the + prompt-bearing keys taken before the hook runs. + """ + if func_name == "apply_guardrail": + inputs: Final = kwargs.get("inputs") + return inputs if isinstance(inputs, dict) else None + if event_type != GuardrailEventHooks.pre_call: + return None + return {key: copy.deepcopy(value) for key, value in request_data.items() if key in _PRE_CALL_CONTENT_KEYS} + + def log_guardrail_information(func): """ Decorator to add standard logging guardrail information to any function @@ -1521,9 +1545,7 @@ def log_guardrail_information(func): event_type: Final = _infer_event_type_from_function_name(func.__name__) # Store original inputs for comparison (for apply_guardrail functions) - original_inputs = None - if func.__name__ == "apply_guardrail" and "inputs" in kwargs: - original_inputs = kwargs.get("inputs") + original_inputs: Final = _original_inputs_for(func.__name__, kwargs, request_data, event_type) logging_obj: Final = kwargs.get("logging_obj") or request_data.get("litellm_logging_obj") self_recorded_token: Final = _guardrail_self_recorded.set(False) @@ -1563,9 +1585,7 @@ def log_guardrail_information(func): event_type: Final = _infer_event_type_from_function_name(func.__name__) # Store original inputs for comparison (for apply_guardrail functions) - original_inputs = None - if func.__name__ == "apply_guardrail" and "inputs" in kwargs: - original_inputs = kwargs.get("inputs") + original_inputs: Final = _original_inputs_for(func.__name__, kwargs, request_data, event_type) logging_obj: Final = kwargs.get("logging_obj") or request_data.get("litellm_logging_obj") self_recorded_token: Final = _guardrail_self_recorded.set(False) diff --git a/litellm/proxy/guardrails/guardrail_hooks/azure/prompt_shield.py b/litellm/proxy/guardrails/guardrail_hooks/azure/prompt_shield.py index 6e29d44662e..de9618a44a1 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/azure/prompt_shield.py +++ b/litellm/proxy/guardrails/guardrail_hooks/azure/prompt_shield.py @@ -338,10 +338,9 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai estimated cost) and the ``azure`` provider label to the recorded guardrail information. Follows the OpenAI moderation override pattern (openai/moderations.py).""" - guardrail_response: Final[dict | str] = ( # mutable-ok: mirrors CustomGuardrail._process_response - ("mask" if self._inputs_were_modified(original_inputs, response) else "allow") - if original_inputs is not None and isinstance(response, dict) - else ({} if response is None else response) # mutable-ok: empty placeholder, never mutated + guardrail_response: Final = self._summarize_guardrail_response( + response=response, + original_inputs=original_inputs, ) self.add_standard_logging_guardrail_information_to_request_data( guardrail_json_response=guardrail_response, diff --git a/tests/test_litellm/integrations/test_custom_guardrail.py b/tests/test_litellm/integrations/test_custom_guardrail.py index ddc8439a83a..1644d78ae37 100644 --- a/tests/test_litellm/integrations/test_custom_guardrail.py +++ b/tests/test_litellm/integrations/test_custom_guardrail.py @@ -2829,3 +2829,131 @@ class TestCustomGuardrailPostCallSuccessDeploymentHook: assert response.choices[0].message.content == "filtered response" assert "guardrail_to_apply" not in request_data assert len(_guardrail_entries(request_data)) == 1 + + +class TestPreCallHookResponseIsNotLoggedVerbatim: + """Regression for LIT-6935: a pre_call hook returning the request payload leaked the prompt + into ``guardrail_response`` and from there onto OTEL guardrail spans.""" + + @staticmethod + def _logged_response(request_data: dict[str, object]) -> object: + metadata = request_data["litellm_metadata"] + assert isinstance(metadata, dict) + entries = metadata["standard_logging_guardrail_information"] + assert len(entries) == 1 + return entries[0]["guardrail_response"] + + @staticmethod + def _request() -> dict[str, object]: + return { + "model": "gpt-4.1-mini", + "input": "SECRET_PROMPT", + "messages": [{"role": "user", "content": "SECRET_PROMPT"}], + "litellm_metadata": {}, + } + + @pytest.mark.asyncio + async def test_pre_call_hook_returning_request_logs_allow(self): + class PassthroughGuardrail(CustomGuardrail): + @log_guardrail_information + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: object, + data: dict[str, object], + call_type: str, + ) -> dict[str, object]: + return data + + data = self._request() + await PassthroughGuardrail(guardrail_name="g").async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(), cache=None, data=data, call_type="aresponses" + ) + + assert self._logged_response(data) == "allow" + + @pytest.mark.asyncio + async def test_pre_call_hook_returning_modified_copy_logs_mask(self): + class MaskingGuardrail(CustomGuardrail): + @log_guardrail_information + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: object, + data: dict[str, object], + call_type: str, + ) -> dict[str, object]: + return {**data, "input": "[MASKED]"} + + data = self._request() + await MaskingGuardrail(guardrail_name="g").async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(), cache=None, data=data, call_type="aresponses" + ) + + assert self._logged_response(data) == "mask" + + @pytest.mark.asyncio + async def test_pre_call_hook_mutating_request_in_place_logs_mask(self): + class InPlaceMaskingGuardrail(CustomGuardrail): + @log_guardrail_information + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: object, + data: dict[str, object], + call_type: str, + ) -> dict[str, object]: + messages = data["messages"] + assert isinstance(messages, list) + messages[0]["content"] = "[MASKED]" + return data + + data = self._request() + await InPlaceMaskingGuardrail(guardrail_name="g").async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(), cache=None, data=data, call_type="acompletion" + ) + + assert self._logged_response(data) == "mask" + + @pytest.mark.asyncio + async def test_pre_call_hook_returning_rejection_string_logs_that_string(self): + class RejectingGuardrail(CustomGuardrail): + @log_guardrail_information + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: object, + data: dict[str, object], + call_type: str, + ) -> str: + return "Blocked by policy" + + data = self._request() + result = await RejectingGuardrail(guardrail_name="g").async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(), cache=None, data=data, call_type="acompletion" + ) + + assert result == "Blocked by policy" + assert self._logged_response(data) == "Blocked by policy" + + @pytest.mark.asyncio + async def test_pre_call_hook_removing_legacy_functions_in_place_logs_mask(self): + class FunctionStrippingGuardrail(CustomGuardrail): + @log_guardrail_information + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: object, + data: dict[str, object], + call_type: str, + ) -> dict[str, object]: + data["functions"] = [] + data["function_call"] = "none" + return data + + data = {**self._request(), "functions": [{"name": "delete_db"}], "function_call": "auto"} + await FunctionStrippingGuardrail(guardrail_name="g").async_pre_call_hook( + user_api_key_dict=UserAPIKeyAuth(), cache=None, data=data, call_type="acompletion" + ) + + assert self._logged_response(data) == "mask" From 359b7a848900c0b30ea2001dff8c8c230295cfb4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 23:47:05 +0000 Subject: [PATCH 133/157] feat(rust): count tiktoken cl100k_base admission tokens in Rust (#40777) * feat(rust): count tiktoken cl100k_base admission tokens in Rust The Rust admission token counter only had the Anthropic tokenizer, so every other model (OpenAI gpt-4 family, Azure, Gemini, Bedrock non-Claude, Mistral) tokenized with tiktoken on the Python inference worker. Add an exact cl100k_base counter to litellm-token-counter: the vendored rank file (base64 token / rank lines, the bytes Python's tiktoken uses) is parsed into a byte-level BPE model and the cl100k split pattern is a handwritten scanner over the shared Unicode classes, so no regex engine runs per request. Both tokenizers share the message, tool and reply-priming accounting. The PyO3 TokenCounter gains a from_cl100k_ranks constructor; Python reads the rank file and passes it in, the way claude_json_str already works. The bridge selects the counter through the same predicates litellm.token_counter uses (huggingface_tokenizer_kind, openai_tokenizer_encoding), declines o200k_base, downloaded HuggingFace and custom tokenizers to Python, and budget reservation counts once per distinct tokenizer a request names. The legacy gpt-3.5-turbo-0301 message accounting (4 per message, -1 per name) stays in Python: the selector declines it through the predicate token_counter itself uses. * feat(rust): count tiktoken o200k_base admission tokens in Rust (#40794) Add a handwritten o200k_base split scanner and TokenCounter::from_o200k_ranks next to the cl100k_base counter, sharing MergeRanks and the request accounting. The Python bridge selects it when openai_tokenizer_encoding names o200k_base, so gpt-4o, gpt-4.1, gpt-5, o1/o3/o4 and chatgpt-4o requests stop tokenizing on the Python worker under LITELLM_RUST=true Co-authored-by: yassin --------- Co-authored-by: yassin Co-authored-by: devin-ai-integration[bot] <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm-rust/Cargo.lock | 2 + .../crates/python-bridge/src/token_counter.rs | 32 +- litellm-rust/crates/token-counter/Cargo.toml | 2 + .../crates/token-counter/src/byte_level.rs | 29 +- .../crates/token-counter/src/cl100k.rs | 125 + .../crates/token-counter/src/counter.rs | 69 +- .../crates/token-counter/src/error.rs | 4 + litellm-rust/crates/token-counter/src/lib.rs | 4 + .../crates/token-counter/src/o200k.rs | 217 + .../crates/token-counter/src/scanner.rs | 107 + .../crates/token-counter/src/tiktoken.rs | 215 + .../token-counter/src/unicode_classes.rs | 97 +- .../tests/fixtures/cl100k/requests.jsonl | 10 + .../tests/fixtures/cl100k/texts.jsonl | 4053 +++++++++++++++++ .../token-counter/tests/fixtures/generate.py | 422 ++ .../tests/fixtures/o200k/requests.jsonl | 10 + .../tests/fixtures/o200k/texts.jsonl | 4053 +++++++++++++++++ .../token-counter/tests/token_counter.rs | 145 + .../litellm_core_utils/default_encoding.py | 15 + litellm/litellm_core_utils/token_counter.py | 33 +- .../spend_tracking/budget_reservation.py | 36 +- litellm/rust_bridge/token_counter.py | 63 +- litellm/utils.py | 48 +- .../spend_tracking/test_budget_reservation.py | 176 +- .../rust_bridge/test_token_counter.py | 312 +- 25 files changed, 10084 insertions(+), 195 deletions(-) create mode 100644 litellm-rust/crates/token-counter/src/cl100k.rs create mode 100644 litellm-rust/crates/token-counter/src/o200k.rs create mode 100644 litellm-rust/crates/token-counter/src/scanner.rs create mode 100644 litellm-rust/crates/token-counter/src/tiktoken.rs create mode 100644 litellm-rust/crates/token-counter/tests/fixtures/cl100k/requests.jsonl create mode 100644 litellm-rust/crates/token-counter/tests/fixtures/cl100k/texts.jsonl create mode 100644 litellm-rust/crates/token-counter/tests/fixtures/generate.py create mode 100644 litellm-rust/crates/token-counter/tests/fixtures/o200k/requests.jsonl create mode 100644 litellm-rust/crates/token-counter/tests/fixtures/o200k/texts.jsonl diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock index cf000e85d68..7b0b593b70f 100644 --- a/litellm-rust/Cargo.lock +++ b/litellm-rust/Cargo.lock @@ -2002,11 +2002,13 @@ dependencies = [ name = "litellm-token-counter" version = "0.1.0" dependencies = [ + "base64 0.22.1", "criterion", "indexmap", "itoa", "rand 0.8.7", "rstest", + "rustc-hash", "serde", "serde_json", "thiserror 2.0.19", diff --git a/litellm-rust/crates/python-bridge/src/token_counter.rs b/litellm-rust/crates/python-bridge/src/token_counter.rs index ee82c170d55..b4de50c5f1a 100644 --- a/litellm-rust/crates/python-bridge/src/token_counter.rs +++ b/litellm-rust/crates/python-bridge/src/token_counter.rs @@ -30,12 +30,17 @@ struct TokenCounter { impl TokenCounter { #[new] fn new(py: Python<'_>, tokenizer_json: &str) -> PyResult { - let inner = release_gil(py, || CoreTokenCounter::from_json(tokenizer_json)) - .map_err(token_count_error_to_pyerr)?; - Ok(Self { - inner: Arc::new(inner), - encode_slots: Arc::new(Semaphore::new(encode_parallelism())), - }) + Self::load(py, || CoreTokenCounter::from_json(tokenizer_json)) + } + + #[staticmethod] + fn from_cl100k_ranks(py: Python<'_>, rank_file: &str) -> PyResult { + Self::load(py, || CoreTokenCounter::from_cl100k_ranks(rank_file)) + } + + #[staticmethod] + fn from_o200k_ranks(py: Python<'_>, rank_file: &str) -> PyResult { + Self::load(py, || CoreTokenCounter::from_o200k_ranks(rank_file)) } fn acount_request<'py>(&self, py: Python<'py>, body: &[u8]) -> PyResult> { @@ -58,6 +63,19 @@ impl TokenCounter { } } +impl TokenCounter { + fn load( + py: Python<'_>, + load: impl FnOnce() -> Result + Send, + ) -> PyResult { + let inner = release_gil(py, load).map_err(token_count_error_to_pyerr)?; + Ok(Self { + inner: Arc::new(inner), + encode_slots: Arc::new(Semaphore::new(encode_parallelism())), + }) + } +} + fn encode_parallelism() -> usize { available_parallelism().map_or(TOKEN_COUNT_FALLBACK_PARALLELISM, NonZero::get) } @@ -70,7 +88,7 @@ fn count_body(counter: &CoreTokenCounter, body: &[u8]) -> Result PyErr { let message = error.to_string(); match error { - Error::Load(_) => PyValueError::new_err(message), + Error::Load(_) | Error::Ranks(_) | Error::UnicodeClasses => PyValueError::new_err(message), Error::RequestParse(_) | Error::MissingInput | Error::FloatText diff --git a/litellm-rust/crates/token-counter/Cargo.toml b/litellm-rust/crates/token-counter/Cargo.toml index 61c9cf6e991..d0369631682 100644 --- a/litellm-rust/crates/token-counter/Cargo.toml +++ b/litellm-rust/crates/token-counter/Cargo.toml @@ -6,8 +6,10 @@ license.workspace = true repository.workspace = true [dependencies] +base64.workspace = true indexmap = { version = "2.14.0", features = ["serde"] } itoa = "1.0" +rustc-hash = "2.1.3" serde.workspace = true serde_json.workspace = true thiserror.workspace = true diff --git a/litellm-rust/crates/token-counter/src/byte_level.rs b/litellm-rust/crates/token-counter/src/byte_level.rs index 2b435eb39bb..ec6134a252e 100644 --- a/litellm-rust/crates/token-counter/src/byte_level.rs +++ b/litellm-rust/crates/token-counter/src/byte_level.rs @@ -12,7 +12,7 @@ use tokenizers::pre_tokenizers::PreTokenizerWrapper; use tokenizers::{Model, Tokenizer}; use unicode_normalization_alignments::{IsNormalized, UnicodeNormalization, is_nfkc_quick}; -use super::unicode_classes::UnicodeClasses; +use super::unicode_classes::{Class, UnicodeClasses, class, run_len}; const CONTRACTIONS: [&str; 7] = ["'s", "'t", "'re", "'ve", "'m", "'ll", "'d"]; @@ -110,27 +110,6 @@ fn mapped_len(piece: &str) -> usize { .count() } -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Class { - Letter, - Number, - Space, - Other, -} - -fn class(character: char, unicode_classes: &UnicodeClasses) -> Class { - match character { - 'A'..='Z' | 'a'..='z' => Class::Letter, - '0'..='9' => Class::Number, - '\t'..='\r' | ' ' => Class::Space, - _ if character.is_ascii() => Class::Other, - _ if unicode_classes.is_letter(character) => Class::Letter, - _ if unicode_classes.is_number(character) => Class::Number, - _ if unicode_classes.is_space(character) => Class::Space, - _ => Class::Other, - } -} - /// The regex matches every character, so the pieces tile the text. fn pieces<'a>( text: &'a str, @@ -169,12 +148,6 @@ fn piece_len(text: &str, first: char, unicode_classes: &UnicodeClasses) -> usize } } -fn run_len(text: &str, run_class: Class, unicode_classes: &UnicodeClasses) -> usize { - text.char_indices() - .find(|(_, character)| class(*character, unicode_classes) != run_class) - .map_or(text.len(), |(index, _)| index) -} - /// `\s+(?!\S)|\s+`: whitespace followed by a non-space leaves its last /// character to start the next piece (` ?` on the following alternatives). fn space_run_len(text: &str, unicode_classes: &UnicodeClasses) -> usize { diff --git a/litellm-rust/crates/token-counter/src/cl100k.rs b/litellm-rust/crates/token-counter/src/cl100k.rs new file mode 100644 index 00000000000..2b2811afc70 --- /dev/null +++ b/litellm-rust/crates/token-counter/src/cl100k.rs @@ -0,0 +1,125 @@ +//! Scanner for tiktoken's `cl100k_base` split regex, +//! `'(?i:[sdmt]|ll|ve|re)|[^\r\n\p{L}\p{N}]?+\p{L}++|\p{N}{1,3}+| ?[^\s\p{L}\p{N}]++[\r\n]*+|\s++$|\s*[\r\n]|\s+(?!\S)|\s`. + +use super::scanner::{contraction_len, digit_run_len, is_newline}; +use super::unicode_classes::{Class, UnicodeClasses, class, run_len}; + +/// The alternatives in regex order; the possessive quantifiers mean an +/// alternative that starts matching and runs out of input fails as a whole. +pub(super) fn piece_len(text: &str, first: char, unicode_classes: &UnicodeClasses) -> usize { + if let Some(len) = contraction_len(text) { + return len; + } + let first_class = class(first, unicode_classes); + match first_class { + Class::Letter => return run_len(text, Class::Letter, unicode_classes), + Class::Number => return digit_run_len(text, unicode_classes), + Class::Space | Class::Other => {} + } + let rest = &text[first.len_utf8()..]; + let second_class = rest + .chars() + .next() + .map(|character| class(character, unicode_classes)); + if !is_newline(first) && second_class == Some(Class::Letter) { + return first.len_utf8() + run_len(rest, Class::Letter, unicode_classes); + } + if first_class == Class::Other { + return symbol_run_len(text, unicode_classes); + } + if first == ' ' && second_class == Some(Class::Other) { + return 1 + symbol_run_len(rest, unicode_classes); + } + space_run_len(text, unicode_classes) +} + +/// `[^\s\p{L}\p{N}]++[\r\n]*+` +fn symbol_run_len(text: &str, unicode_classes: &UnicodeClasses) -> usize { + let symbols = run_len(text, Class::Other, unicode_classes); + symbols + + text[symbols..] + .bytes() + .take_while(|byte| matches!(byte, b'\r' | b'\n')) + .count() +} + +/// `\s++$|\s*[\r\n]|\s+(?!\S)|\s`: whitespace to the end of the text is one +/// piece; otherwise the piece ends at the last newline of the run, or leaves +/// the run's last character for the next piece's optional leading space. +fn space_run_len(text: &str, unicode_classes: &UnicodeClasses) -> usize { + let run = run_len(text, Class::Space, unicode_classes); + if run == text.len() { + return run; + } + if let Some(newline) = text[..run].rfind(['\r', '\n']) { + return newline + 1; + } + let last = text[..run].chars().next_back().map_or(0, char::len_utf8); + match run - last { + 0 => run, + shorter => shorter, + } +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + + use super::*; + use crate::scanner::pieces; + + fn split(text: &str) -> Vec<&str> { + pieces( + text, + piece_len, + UnicodeClasses::get().expect("Oniguruma exposes Unicode classes"), + ) + .collect() + } + + #[rstest] + #[case("", &[])] + #[case("Hello world", &["Hello", " world"])] + #[case("don't I'LL you'Ve we'RE he'd I'm", &["don", "'t", " I", "'LL", " you", "'Ve", " we", "'RE", " he", "'d", " I", "'m"])] + #[case("IT'SOK it'Dbe 'Sx 'Tx", &["IT", "'S", "OK", " it", "'D", "be", " '", "Sx", " '", "Tx"])] + #[case("'Sx'Tx'Mx'LLx'VEx'REx'Dx", &["'S", "x", "'T", "x", "'M", "x", "'LL", "x", "'VE", "x", "'RE", "x", "'D", "x"])] + #[case("'ſ 'lx", &["'ſ", " '", "lx"])] + #[case("12345 6", &["123", "45", " ", "6"])] + #[case("!abc !!abc", &["!abc", " !!", "abc"])] + #[case(" !!!\r\n\r\nx", &[" !!!\r\n\r\n", "x"])] + #[case("a b \n\n c", &["a", " ", " b", " \n\n", " ", " c"])] + #[case("a\nb\r\nc\n\nd \n e", &["a", "\n", "b", "\r\n", "c", "\n\n", "d", " \n", " e"])] + #[case("x \t\n \t y\n", &["x", " \t\n", " \t", " y", "\n"])] + #[case("end ", &["end", " "])] + #[case("\u{a0}abc\u{a0}!", &["\u{a0}abc", "\u{a0}", "!"])] + #[case("<|endoftext|>", &["<|", "endoftext", "|>"])] + #[case("e\u{301}a", &["e", "\u{301}a"])] + #[case("日本語 ١٢٣٤", &["日本語", " ", "١٢٣", "٤"])] + fn scanner_splits_like_the_regex(#[case] text: &str, #[case] expected: &[&str]) { + assert_eq!(split(text), expected); + } + + #[derive(serde::Deserialize)] + struct TextFixture { + text: String, + pieces: Vec, + } + + #[test] + fn scanner_splits_the_fixture_corpus_like_tiktoken_regex() { + let fixtures: Vec = include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/fixtures/cl100k/texts.jsonl" + )) + .lines() + .map(|line| serde_json::from_str(line).expect("fixture line is json")) + .collect(); + assert!(fixtures.len() > 3000); + let mismatches: Vec<_> = fixtures + .iter() + .filter(|fixture| split(&fixture.text) != fixture.pieces) + .map(|fixture| (&fixture.text, split(&fixture.text), &fixture.pieces)) + .collect(); + assert!(mismatches.is_empty(), "{mismatches:#?}"); + } +} diff --git a/litellm-rust/crates/token-counter/src/counter.rs b/litellm-rust/crates/token-counter/src/counter.rs index a3943515a64..7eedc449dd1 100644 --- a/litellm-rust/crates/token-counter/src/counter.rs +++ b/litellm-rust/crates/token-counter/src/counter.rs @@ -3,6 +3,7 @@ use serde::Serialize; use crate::Error; use crate::byte_level::ByteLevelCounter; use crate::python_json; +use crate::scanner::{SplitPattern, TiktokenCounter}; use crate::tools::format_function_definitions; use crate::types::{ ContentBlock, ContentItem, CountableRequest, Message, MessageContent, TextValue, ToolChoice, @@ -23,12 +24,19 @@ pub struct InputTokenCount { pub input_tokens: usize, } -/// A loaded HuggingFace tokenizer plus the message accounting Python applies on -/// top of it. Encoding is CPU-bound and synchronous; hosts run it off their -/// event loop. +enum Encoder { + HuggingFace { + tokenizer: Box, + byte_level: Option, + }, + Tiktoken(TiktokenCounter), +} + +/// A loaded tokenizer plus the message accounting Python applies on top of +/// it. Encoding is CPU-bound and synchronous; hosts run it off their event +/// loop. pub struct TokenCounter { - tokenizer: tokenizers::Tokenizer, - byte_level: Option, + encoder: Encoder, } impl TokenCounter { @@ -39,23 +47,50 @@ impl TokenCounter { .map_err(Error::Load)?; let byte_level = ByteLevelCounter::detect(&tokenizer); Ok(Self { - tokenizer, - byte_level, + encoder: Encoder::HuggingFace { + tokenizer: Box::new(tokenizer), + byte_level, + }, + }) + } + + /// Load tiktoken's `cl100k_base` rank file (`base64(token) rank` lines). + /// The host reads the file. + pub fn from_cl100k_ranks(rank_file: &str) -> Result { + Self::from_tiktoken_ranks(SplitPattern::Cl100k, rank_file) + } + + /// Load tiktoken's `o200k_base` rank file (`base64(token) rank` lines). + /// The host reads the file. + pub fn from_o200k_ranks(rank_file: &str) -> Result { + Self::from_tiktoken_ranks(SplitPattern::O200k, rank_file) + } + + fn from_tiktoken_ranks(split: SplitPattern, rank_file: &str) -> Result { + Ok(Self { + encoder: Encoder::Tiktoken(TiktokenCounter::from_ranks(split, rank_file)?), }) } pub fn count_text(&self, text: &str) -> Result { - if let Some(count) = self - .byte_level - .as_ref() - .and_then(|counter| counter.count(&self.tokenizer, text)) - { - return Ok(count); + match &self.encoder { + Encoder::Tiktoken(counter) => Ok(counter.count(text)), + Encoder::HuggingFace { + tokenizer, + byte_level, + } => { + if let Some(count) = byte_level + .as_ref() + .and_then(|counter| counter.count(tokenizer, text)) + { + return Ok(count); + } + tokenizer + .encode_fast(text, true) + .map(|encoding| encoding.len()) + .map_err(Error::Encode) + } } - self.tokenizer - .encode_fast(text, true) - .map(|encoding| encoding.len()) - .map_err(Error::Encode) } /// Mirrors the host's key precedence: `messages`, then `prompt`, then diff --git a/litellm-rust/crates/token-counter/src/error.rs b/litellm-rust/crates/token-counter/src/error.rs index dc590093f64..6b8668fe182 100644 --- a/litellm-rust/crates/token-counter/src/error.rs +++ b/litellm-rust/crates/token-counter/src/error.rs @@ -6,6 +6,10 @@ use thiserror::Error as ThisError; pub enum Error { #[error("failed to load tokenizer: {0}")] Load(#[source] tokenizers::Error), + #[error("failed to load tokenizer: tiktoken rank file: {0}")] + Ranks(String), + #[error("failed to load tokenizer: Unicode character classes are unavailable")] + UnicodeClasses, #[error("unsupported by the rust token counter: request body could not be parsed: {0}")] RequestParse(#[source] serde_json::Error), #[error("unsupported by the rust token counter: request has no countable input")] diff --git a/litellm-rust/crates/token-counter/src/lib.rs b/litellm-rust/crates/token-counter/src/lib.rs index eb602b3cade..fa0014e2bad 100644 --- a/litellm-rust/crates/token-counter/src/lib.rs +++ b/litellm-rust/crates/token-counter/src/lib.rs @@ -5,9 +5,13 @@ #![forbid(unsafe_code)] mod byte_level; +mod cl100k; mod counter; mod error; +mod o200k; mod python_json; +mod scanner; +mod tiktoken; mod tools; mod types; mod unicode_classes; diff --git a/litellm-rust/crates/token-counter/src/o200k.rs b/litellm-rust/crates/token-counter/src/o200k.rs new file mode 100644 index 00000000000..c0d85c45e78 --- /dev/null +++ b/litellm-rust/crates/token-counter/src/o200k.rs @@ -0,0 +1,217 @@ +//! Scanner for tiktoken's `o200k_base` split regex, +//! `[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]*[\p{Ll}\p{Lm}\p{Lo}\p{M}]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?|[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+[\p{Ll}\p{Lm}\p{Lo}\p{M}]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n/]*|\s*[\r\n]+|\s+(?!\S)|\s+`. +//! tiktoken runs it with a backtracking engine, so the letter alternatives +//! below reproduce where the greedy quantifiers settle, not only what the +//! classes say. + +use super::scanner::{contraction_len, digit_run_len, is_newline}; +use super::unicode_classes::{Case, Class, UnicodeClasses, case, case_run_len, class, run_len}; + +/// The alternatives in regex order: a number is never a letter piece, a +/// letter always is, and only whitespace and symbols reach the last three. +pub(super) fn piece_len(text: &str, first: char, unicode_classes: &UnicodeClasses) -> usize { + let first_class = class(first, unicode_classes); + if first_class == Class::Number { + return digit_run_len(text, unicode_classes); + } + if let Some(len) = letter_piece_len(text, first, first_class, unicode_classes) { + return len; + } + if first_class == Class::Other { + return symbol_run_len(text, unicode_classes); + } + let rest = &text[first.len_utf8()..]; + if first == ' ' + && rest + .chars() + .next() + .is_some_and(|character| class(character, unicode_classes) == Class::Other) + { + return 1 + symbol_run_len(rest, unicode_classes); + } + space_run_len(text, unicode_classes) +} + +/// The two letter alternatives, each first with then without the optional +/// `[^\r\n\p{L}\p{N}]` prefix: the order the engine tries them in. +fn letter_piece_len( + text: &str, + first: char, + first_class: Class, + unicode_classes: &UnicodeClasses, +) -> Option { + let prefix = (!is_newline(first) && matches!(first_class, Class::Space | Class::Other)) + .then(|| first.len_utf8()); + let after_prefix = |shape: fn(&str, &UnicodeClasses) -> Option| { + prefix.and_then(|prefix| shape(&text[prefix..], unicode_classes).map(|len| prefix + len)) + }; + after_prefix(upper_then_lower_len) + .or_else(|| upper_then_lower_len(text, unicode_classes)) + .or_else(|| after_prefix(upper_run_len)) + .or_else(|| upper_run_len(text, unicode_classes)) +} + +/// `[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]*[\p{Ll}\p{Lm}\p{Lo}\p{M}]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?`. +/// The upper run is greedy; when no lower character follows it, the engine +/// gives characters back until the one it just gave back is lower too, and +/// that single character is the lower run. +fn upper_then_lower_len(text: &str, unicode_classes: &UnicodeClasses) -> Option { + let upper = case_run_len(text, Case::is_upper, unicode_classes); + let lower = case_run_len(&text[upper..], Case::is_lower, unicode_classes); + let letters = if lower > 0 { + upper + lower + } else { + let (index, last_both) = text[..upper] + .char_indices() + .rev() + .find(|(_, character)| case(*character, unicode_classes).is_lower())?; + index + last_both.len_utf8() + }; + Some(letters + contraction_len(&text[letters..]).unwrap_or(0)) +} + +/// `[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+[\p{Ll}\p{Lm}\p{Lo}\p{M}]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?` +fn upper_run_len(text: &str, unicode_classes: &UnicodeClasses) -> Option { + let upper = case_run_len(text, Case::is_upper, unicode_classes); + if upper == 0 { + return None; + } + let letters = upper + case_run_len(&text[upper..], Case::is_lower, unicode_classes); + Some(letters + contraction_len(&text[letters..]).unwrap_or(0)) +} + +/// `[^\s\p{L}\p{N}]+[\r\n/]*` +fn symbol_run_len(text: &str, unicode_classes: &UnicodeClasses) -> usize { + let symbols = run_len(text, Class::Other, unicode_classes); + symbols + + text[symbols..] + .bytes() + .take_while(|byte| matches!(byte, b'\r' | b'\n' | b'/')) + .count() +} + +/// `\s*[\r\n]+|\s+(?!\S)|\s+`: a run with a newline ends at its last newline, +/// even at the end of the text; otherwise whitespace to the end of the text +/// is one piece, or the run leaves its last character for the next piece's +/// optional leading space. +fn space_run_len(text: &str, unicode_classes: &UnicodeClasses) -> usize { + let run = run_len(text, Class::Space, unicode_classes); + if let Some(newline) = text[..run].rfind(['\r', '\n']) { + return newline + 1; + } + if run == text.len() { + return run; + } + let last = text[..run].chars().next_back().map_or(0, char::len_utf8); + match run - last { + 0 => run, + shorter => shorter, + } +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + use tokenizers::utils::SysRegex; + + use super::*; + use crate::scanner::pieces; + + fn split(text: &str) -> Vec<&str> { + pieces( + text, + piece_len, + UnicodeClasses::get().expect("Oniguruma exposes Unicode classes"), + ) + .collect() + } + + #[rstest] + #[case("", &[])] + #[case("Hello world", &["Hello", " world"])] + #[case("camelCase PascalCase ABCdef ABCdeF ABC", &["camel", "Case", " Pascal", "Case", " ABCdef", " ABCde", "F", " ABC"])] + #[case("日本ABC ABC日本 日本語abc abc日本語", &["日本", "ABC", " ABC日本", " 日本語abc", " abc日本語"])] + #[case("\u{301}ABC \u{301}abc \u{301}\u{301}A A\u{301}\u{301} E\u{301}A aE\u{301}", &["\u{301}", "ABC", " \u{301}abc", " \u{301}\u{301}", "A", " A\u{301}\u{301}", " E\u{301}", "A", " a", "E\u{301}"])] + #[case("ᵃbc ᵃBC Aᵃbc Aᵃ ᵃ' ᵃ's", &["ᵃbc", " ᵃ", "BC", " Aᵃbc", " Aᵃ", " ᵃ", "'", " ᵃ's"])] + #[case("Džungla aDžB ADžB ADžb", &["Džungla", " a", "DžB", " ADžB", " ADžb"])] + #[case("don'tx ABC's abc'S abc'ſ ABC'ſx IT'SOK it'Dbe", &["don't", "x", " ABC's", " abc'S", " abc'ſ", " ABC'ſ", "x", " IT'S", "OK", " it'D", "be"])] + #[case("'sabc x's 's 'Sx'Tx 9'9 a'9 ' s", &["'sabc", " x's", " '", "s", " '", "Sx'T", "x", " ", "9", "'", "9", " a", "'", "9", " '", " s"])] + #[case("!ABC !AbC !!abc !!\u{301}a \u{a0}\u{301}A", &["!ABC", " !", "Ab", "C", " !!", "abc", " !!\u{301}", "a", " ", "\u{a0}\u{301}", "A"])] + #[case("!!/\n/x a/b !!\n/x /x //", &["!!/\n/", "x", " a", "/b", " !!\n/", "x", " ", " /", "x", " ", " //"])] + #[case("12345 6 1abc abc1", &["123", "45", " ", "6", " ", "1", "abc", " abc", "1"])] + #[case("x \n x \r\n \r\n y", &["x", " \n", " x", " \r\n \r\n", " y"])] + #[case("x \n ", &["x", " \n", " "])] + #[case("a b \n\n c", &["a", " ", " b", " \n\n", " ", " c"])] + #[case("x\t\ty x\t\t", &["x", "\t", "\ty", " x", "\t\t"])] + #[case("end ", &["end", " "])] + #[case("\u{a0}abc\u{a0}!", &["\u{a0}abc", "\u{a0}", "!"])] + #[case("<|endoftext|>", &["<|", "endoftext", "|>"])] + #[case("İstanbul ΣΊΣΥΦΟΣ Ελληνικά Русский", &["İstanbul", " ΣΊΣΥΦΟΣ", " Ελληνικά", " Русский"])] + #[case("日本語 ١٢٣٤", &["日本語", " ", "١٢٣", "٤"])] + fn scanner_splits_like_the_regex(#[case] text: &str, #[case] expected: &[&str]) { + assert_eq!(split(text), expected); + } + + #[test] + fn every_scalar_alone_is_one_piece() { + for character in (0..=0x10FFFFu32).filter_map(char::from_u32) { + let text = character.to_string(); + assert_eq!( + split(&text), + [text.as_str()], + "U+{:04X}", + u32::from(character) + ); + } + } + + #[test] + fn cases_match_oniguruma() { + let unicode_classes = UnicodeClasses::get().expect("Oniguruma exposes Unicode classes"); + let upper = SysRegex::new(r"[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]").expect("regex"); + let lower = SysRegex::new(r"[\p{Ll}\p{Lm}\p{Lo}\p{M}]").expect("regex"); + let whole = + |regex: &SysRegex, text: &str| regex.find_iter(text).next() == Some((0, text.len())); + let mut text = String::new(); + for character in (0..=0x10FFFFu32).filter_map(char::from_u32) { + text.clear(); + text.push(character); + let expected = match (whole(&upper, &text), whole(&lower, &text)) { + (true, true) => Case::Both, + (true, false) => Case::Upper, + (false, true) => Case::Lower, + (false, false) => Case::Neither, + }; + assert_eq!( + case(character, unicode_classes), + expected, + "U+{:04X}", + u32::from(character) + ); + } + } + + #[derive(serde::Deserialize)] + struct TextFixture { + text: String, + pieces: Vec, + } + + #[test] + fn scanner_splits_the_fixture_corpus_like_tiktoken_regex() { + let fixtures: Vec = include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/fixtures/o200k/texts.jsonl" + )) + .lines() + .map(|line| serde_json::from_str(line).expect("fixture line is json")) + .collect(); + assert!(fixtures.len() > 3000); + let mismatches: Vec<_> = fixtures + .iter() + .filter(|fixture| split(&fixture.text) != fixture.pieces) + .map(|fixture| (&fixture.text, split(&fixture.text), &fixture.pieces)) + .collect(); + assert!(mismatches.is_empty(), "{mismatches:#?}"); + } +} diff --git a/litellm-rust/crates/token-counter/src/scanner.rs b/litellm-rust/crates/token-counter/src/scanner.rs new file mode 100644 index 00000000000..c2c3057aeeb --- /dev/null +++ b/litellm-rust/crates/token-counter/src/scanner.rs @@ -0,0 +1,107 @@ +//! Exact token counting for tiktoken encodings. A hand-written scanner +//! reproduces the piece boundaries of the encoding's split regex, and each +//! piece is merged with the rank file. Special tokens are ordinary text, as +//! with `encode(text, disallowed_special=())`. + +use std::iter; + +use super::tiktoken::{MergeRanks, MergeScratch}; +use super::unicode_classes::{Class, UnicodeClasses, class}; +use super::{cl100k, o200k}; +use crate::Error; + +const MAX_DIGITS_PER_PIECE: usize = 3; + +/// Byte length of the piece the split regex matches at the start of the +/// text, given the text's first character. +pub(super) type PieceLen = fn(&str, char, &UnicodeClasses) -> usize; + +#[derive(Clone, Copy, Debug)] +pub(super) enum SplitPattern { + Cl100k, + O200k, +} + +impl SplitPattern { + fn piece_len(self) -> PieceLen { + match self { + Self::Cl100k => cl100k::piece_len, + Self::O200k => o200k::piece_len, + } + } +} + +pub(super) struct TiktokenCounter { + ranks: MergeRanks, + piece_len: PieceLen, + unicode_classes: &'static UnicodeClasses, +} + +impl TiktokenCounter { + pub(super) fn from_ranks(split: SplitPattern, rank_file: &str) -> Result { + Ok(Self { + ranks: MergeRanks::parse(rank_file)?, + piece_len: split.piece_len(), + unicode_classes: UnicodeClasses::get().ok_or(Error::UnicodeClasses)?, + }) + } + + pub(super) fn count(&self, text: &str) -> usize { + let mut scratch = MergeScratch::default(); + pieces(text, self.piece_len, self.unicode_classes) + .map(|piece| self.ranks.count_piece(piece.as_bytes(), &mut scratch)) + .sum() + } +} + +/// The regex matches every character, so the pieces tile the text. +pub(super) fn pieces<'a>( + text: &'a str, + piece_len: PieceLen, + unicode_classes: &'static UnicodeClasses, +) -> impl Iterator { + iter::successors( + split_piece(text, piece_len, unicode_classes), + move |(_, rest)| split_piece(rest, piece_len, unicode_classes), + ) + .map(|(piece, _)| piece) +} + +fn split_piece<'a>( + text: &'a str, + piece_len: PieceLen, + unicode_classes: &UnicodeClasses, +) -> Option<(&'a str, &'a str)> { + let first = text.chars().next()?; + Some(text.split_at(piece_len(text, first, unicode_classes))) +} + +/// `'(?i:s|t|re|ve|m|ll|d)`, the contraction both encodings spell out. Simple +/// case folding also maps U+017F (long s) onto `s`. +pub(super) fn contraction_len(text: &str) -> Option { + let mut characters = text.chars(); + if characters.next()? != '\'' { + return None; + } + let first = characters.next()?; + let len = match first { + 's' | 'S' | '\u{17F}' | 'd' | 'D' | 'm' | 'M' | 't' | 'T' => first.len_utf8(), + 'l' | 'L' => matches!(characters.next(), Some('l' | 'L')).then_some(2)?, + 'v' | 'V' | 'r' | 'R' => matches!(characters.next(), Some('e' | 'E')).then_some(2)?, + _ => return None, + }; + Some(1 + len) +} + +pub(super) fn is_newline(character: char) -> bool { + matches!(character, '\r' | '\n') +} + +/// `\p{N}{1,3}` +pub(super) fn digit_run_len(text: &str, unicode_classes: &UnicodeClasses) -> usize { + text.chars() + .take(MAX_DIGITS_PER_PIECE) + .take_while(|character| class(*character, unicode_classes) == Class::Number) + .map(char::len_utf8) + .sum() +} diff --git a/litellm-rust/crates/token-counter/src/tiktoken.rs b/litellm-rust/crates/token-counter/src/tiktoken.rs new file mode 100644 index 00000000000..c479ae01be9 --- /dev/null +++ b/litellm-rust/crates/token-counter/src/tiktoken.rs @@ -0,0 +1,215 @@ +//! tiktoken's byte-level BPE: a rank file of `base64(token) rank` lines and +//! the merge loop that turns one regex piece into tokens. The merge order is +//! tiktoken's (lowest rank first, leftmost pair on ties) so the token count is +//! identical, but pairs are tracked in a heap so a long piece costs +//! `O(n log n)` instead of tiktoken's `O(n^2)`. + +use std::cmp::Reverse; +use std::collections::BinaryHeap; + +use base64::Engine; +use base64::engine::general_purpose::STANDARD; +use rustc_hash::FxHashMap; + +use crate::Error; + +type Rank = u32; + +const NO_RANK: Rank = Rank::MAX; +const END: usize = usize::MAX; + +pub(super) struct MergeRanks(FxHashMap, Rank>); + +impl MergeRanks { + pub(super) fn parse(text: &str) -> Result { + let ranks = text + .lines() + .filter(|line| !line.is_empty()) + .map(parse_line) + .collect::, _>>()?; + if let Some(byte) = (0..=u8::MAX).find(|byte| !ranks.contains_key(&[*byte][..])) { + return Err(Error::Ranks(format!("byte 0x{byte:02X} has no token"))); + } + Ok(Self(ranks)) + } + + fn rank(&self, bytes: &[u8]) -> Rank { + self.0.get(bytes).copied().unwrap_or(NO_RANK) + } + + /// Token count of one regex piece, as `encode_ordinary` would produce. + pub(super) fn count_piece(&self, piece: &[u8], scratch: &mut MergeScratch) -> usize { + if piece.len() < 2 || self.0.contains_key(piece) { + return 1; + } + scratch.reset(piece.len()); + for start in 0..piece.len() - 1 { + scratch.set_rank(start, self.rank(&piece[start..start + 2])); + } + let mut parts = piece.len(); + while let Some(Reverse((rank, start))) = scratch.heap.pop() { + if scratch.next[start] == END || scratch.rank[start] != rank { + continue; + } + let merged = scratch.next[start]; + let after = scratch.next[merged]; + scratch.next[merged] = END; + scratch.next[start] = after; + parts -= 1; + if after < piece.len() { + scratch.prev[after] = start; + scratch.set_rank(start, self.rank(&piece[start..scratch.end(after)])); + } else { + scratch.rank[start] = NO_RANK; + } + let before = scratch.prev[start]; + if before != END { + scratch.set_rank(before, self.rank(&piece[before..scratch.end(start)])); + } + } + parts + } +} + +fn parse_line(line: &str) -> Result<(Box<[u8]>, Rank), Error> { + let (token, rank) = line + .split_once(' ') + .ok_or_else(|| Error::Ranks(format!("line without a rank: {line:?}")))?; + let bytes = STANDARD + .decode(token) + .map_err(|error| Error::Ranks(format!("token is not base64: {error}")))?; + let rank = rank + .parse() + .map_err(|error| Error::Ranks(format!("rank is not an integer: {error}")))?; + Ok((bytes.into_boxed_slice(), rank)) +} + +/// Buffers reused across the pieces of one text. Parts are addressed by the +/// byte offset they start at, which also gives the leftmost-pair tie break. +#[derive(Default)] +pub(super) struct MergeScratch { + next: Vec, + prev: Vec, + rank: Vec, + heap: BinaryHeap>, +} + +impl MergeScratch { + fn reset(&mut self, len: usize) { + self.next.clear(); + self.next.extend(1..=len); + self.prev.clear(); + self.prev.push(END); + self.prev.extend(0..len - 1); + self.rank.clear(); + self.rank.resize(len, NO_RANK); + self.heap.clear(); + } + + fn end(&self, start: usize) -> usize { + self.next[start] + } + + fn set_rank(&mut self, start: usize, rank: Rank) { + self.rank[start] = rank; + if rank != NO_RANK { + self.heap.push(Reverse((rank, start))); + } + } +} + +#[cfg(test)] +mod tests { + use rand::rngs::StdRng; + use rand::{Rng, SeedableRng}; + + use super::*; + + fn ranks() -> MergeRanks { + let path = concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../../litellm/litellm_core_utils/tokenizers/9b5ad71b2ce5302211f9c61530b329a4922fc6a4" + ); + MergeRanks::parse(&std::fs::read_to_string(path).expect("cl100k rank file is in the repo")) + .expect("rank file parses") + } + + /// tiktoken's `_byte_pair_merge`, transcribed, as the reference. + fn reference_count(ranks: &MergeRanks, piece: &[u8]) -> usize { + if piece.len() < 2 || ranks.0.contains_key(piece) { + return 1; + } + let mut parts: Vec<(usize, Rank)> = (0..piece.len() - 1) + .map(|index| (index, ranks.rank(&piece[index..index + 2]))) + .chain([(piece.len() - 1, NO_RANK), (piece.len(), NO_RANK)]) + .collect(); + let get_rank = |parts: &[(usize, Rank)], index: usize| { + if index + 3 < parts.len() { + ranks.rank(&piece[parts[index].0..parts[index + 3].0]) + } else { + NO_RANK + } + }; + loop { + let Some(index) = parts[..parts.len() - 1] + .iter() + .enumerate() + .filter(|(_, (_, rank))| *rank != NO_RANK) + .min_by_key(|(index, (_, rank))| (*rank, *index)) + .map(|(index, _)| index) + else { + return parts.len() - 1; + }; + if index > 0 { + parts[index - 1].1 = get_rank(&parts, index - 1); + } + parts[index].1 = get_rank(&parts, index); + parts.remove(index + 1); + } + } + + #[test] + fn every_byte_is_a_token() { + let ranks = ranks(); + assert_eq!(ranks.0.len(), 100_256); + assert!((0..=u8::MAX).all(|byte| ranks.rank(&[byte]) != NO_RANK)); + } + + #[test] + fn heap_merge_matches_tiktokens_merge_loop() { + let ranks = ranks(); + let mut scratch = MergeScratch::default(); + let mut rng = StdRng::seed_from_u64(99); + let alphabet = b" abcdeorstn.,'\n\xc3\xa9\xe2\x82\xac0123"; + for _ in 0..20_000 { + let piece: Vec = (0..rng.gen_range(1..24)) + .map(|_| alphabet[rng.gen_range(0..alphabet.len())]) + .collect(); + assert_eq!( + ranks.count_piece(&piece, &mut scratch), + reference_count(&ranks, &piece), + "piece {:?}", + String::from_utf8_lossy(&piece) + ); + } + } + + #[test] + fn long_repeated_runs_stay_cheap() { + let ranks = ranks(); + let mut scratch = MergeScratch::default(); + let piece = vec![b' '; 1 << 20]; + let started = std::time::Instant::now(); + let count = ranks.count_piece(&piece, &mut scratch); + assert!(count > 0); + assert!(started.elapsed().as_secs() < 5, "{:?}", started.elapsed()); + } + + #[test] + fn malformed_rank_files_are_rejected() { + assert!(MergeRanks::parse("IQ==").is_err()); + assert!(MergeRanks::parse("IQ== x").is_err()); + assert!(MergeRanks::parse("!!! 1").is_err()); + assert!(MergeRanks::parse("IQ== 1").is_err()); + } +} diff --git a/litellm-rust/crates/token-counter/src/unicode_classes.rs b/litellm-rust/crates/token-counter/src/unicode_classes.rs index 405cee34949..6aa34047adf 100644 --- a/litellm-rust/crates/token-counter/src/unicode_classes.rs +++ b/litellm-rust/crates/token-counter/src/unicode_classes.rs @@ -9,6 +9,8 @@ pub(super) struct UnicodeClasses { letters: Ranges, numbers: Ranges, spaces: Ranges, + uppers: Ranges, + lowers: Ranges, } static CLASSES: LazyLock> = LazyLock::new(|| { @@ -19,6 +21,8 @@ static CLASSES: LazyLock> = LazyLock::new(|| { letters: Ranges::load(r"\p{L}+", &scalars)?, numbers: Ranges::load(r"\p{N}+", &scalars)?, spaces: Ranges::load(r"\s+", &scalars)?, + uppers: Ranges::load(r"[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+", &scalars)?, + lowers: Ranges::load(r"[\p{Ll}\p{Lm}\p{Lo}\p{M}]+", &scalars)?, }) }); @@ -59,15 +63,102 @@ impl UnicodeClasses { CLASSES.as_ref() } - pub(super) fn is_letter(&self, character: char) -> bool { + fn is_letter(&self, character: char) -> bool { self.letters.contains(character) } - pub(super) fn is_number(&self, character: char) -> bool { + fn is_number(&self, character: char) -> bool { self.numbers.contains(character) } - pub(super) fn is_space(&self, character: char) -> bool { + fn is_space(&self, character: char) -> bool { self.spaces.contains(character) } + + fn is_upper(&self, character: char) -> bool { + self.uppers.contains(character) + } + + fn is_lower(&self, character: char) -> bool { + self.lowers.contains(character) + } +} + +/// `\p{L}`, `\p{N}`, `\s` and everything else, the character classes the +/// split regexes are written in. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum Class { + Letter, + Number, + Space, + Other, +} + +pub(super) fn class(character: char, unicode_classes: &UnicodeClasses) -> Class { + match character { + 'A'..='Z' | 'a'..='z' => Class::Letter, + '0'..='9' => Class::Number, + '\t'..='\r' | ' ' => Class::Space, + _ if character.is_ascii() => Class::Other, + _ if unicode_classes.is_letter(character) => Class::Letter, + _ if unicode_classes.is_number(character) => Class::Number, + _ if unicode_classes.is_space(character) => Class::Space, + _ => Class::Other, + } +} + +/// Byte length of the leading run of `run_class` characters. +pub(super) fn run_len(text: &str, run_class: Class, unicode_classes: &UnicodeClasses) -> usize { + text.char_indices() + .find(|(_, character)| class(*character, unicode_classes) != run_class) + .map_or(text.len(), |(index, _)| index) +} + +/// Membership in the two letter classes of the o200k split regex, +/// `[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]` and `[\p{Ll}\p{Lm}\p{Lo}\p{M}]`; `Lm`, +/// `Lo` and `M` are in both. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(super) enum Case { + Upper, + Lower, + Both, + Neither, +} + +impl Case { + pub(super) fn is_upper(self) -> bool { + matches!(self, Case::Upper | Case::Both) + } + + pub(super) fn is_lower(self) -> bool { + matches!(self, Case::Lower | Case::Both) + } +} + +pub(super) fn case(character: char, unicode_classes: &UnicodeClasses) -> Case { + match character { + 'A'..='Z' => Case::Upper, + 'a'..='z' => Case::Lower, + _ if character.is_ascii() => Case::Neither, + _ => match ( + unicode_classes.is_upper(character), + unicode_classes.is_lower(character), + ) { + (true, true) => Case::Both, + (true, false) => Case::Upper, + (false, true) => Case::Lower, + (false, false) => Case::Neither, + }, + } +} + +/// Byte length of the leading run of characters whose case passes `in_class`. +pub(super) fn case_run_len( + text: &str, + in_class: fn(Case) -> bool, + unicode_classes: &UnicodeClasses, +) -> usize { + text.char_indices() + .find(|(_, character)| !in_class(case(*character, unicode_classes))) + .map_or(text.len(), |(index, _)| index) } diff --git a/litellm-rust/crates/token-counter/tests/fixtures/cl100k/requests.jsonl b/litellm-rust/crates/token-counter/tests/fixtures/cl100k/requests.jsonl new file mode 100644 index 00000000000..c2a7b718cc3 --- /dev/null +++ b/litellm-rust/crates/token-counter/tests/fixtures/cl100k/requests.jsonl @@ -0,0 +1,10 @@ +{"body": "{\"model\": \"gpt-4\", \"messages\": [{\"role\": \"user\", \"content\": \"Hello, how are you today?\"}]}", "input_tokens": 14} +{"body": "{\"model\": \"gpt-4\", \"messages\": [{\"role\": \"system\", \"content\": \"You are a terse assistant.\"}, {\"role\": \"user\", \"name\": \"alice\", \"content\": [{\"type\": \"text\", \"text\": \"Summarise this paragraph about ships and harbours.\"}, \"plain string item\"]}, {\"role\": \"assistant\", \"content\": [{\"type\": \"text\", \"text\": \"Sure.\"}]}]}", "input_tokens": 39} +{"body": "{\"model\": \"gpt-4\", \"messages\": [{\"role\": \"user\", \"content\": \"weather?\"}], \"tools\": [{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Get weather\", \"parameters\": {\"type\": \"object\", \"properties\": {\"location\": {\"type\": \"string\", \"description\": \"City name\"}, \"unit\": {\"type\": \"string\", \"enum\": [\"celsius\", \"fahrenheit\"]}, \"days\": {\"type\": \"integer\"}, \"tags\": {\"type\": \"array\", \"items\": {\"type\": \"string\"}}, \"opts\": {\"type\": \"object\", \"properties\": {\"verbose\": {\"type\": \"boolean\"}, \"level\": {\"type\": \"integer\", \"enum\": [1, 2]}}, \"required\": [\"verbose\"]}, \"anything\": {}}, \"required\": [\"location\"]}}}, {\"type\": \"function\", \"function\": {\"name\": \"noop\"}}], \"tool_choice\": {\"type\": \"function\", \"function\": {\"name\": \"get_weather\"}}}", "input_tokens": 106} +{"body": "{\"model\": \"gpt-4\", \"messages\": [{\"role\": \"system\", \"content\": \"sys\"}, {\"role\": \"user\", \"content\": \"weather?\"}], \"tools\": [{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Get weather\", \"parameters\": {\"type\": \"object\", \"properties\": {\"location\": {\"type\": \"string\", \"description\": \"City name\"}, \"unit\": {\"type\": \"string\", \"enum\": [\"celsius\", \"fahrenheit\"]}, \"days\": {\"type\": \"integer\"}, \"tags\": {\"type\": \"array\", \"items\": {\"type\": \"string\"}}, \"opts\": {\"type\": \"object\", \"properties\": {\"verbose\": {\"type\": \"boolean\"}, \"level\": {\"type\": \"integer\", \"enum\": [1, 2]}}, \"required\": [\"verbose\"]}, \"anything\": {}}, \"required\": [\"location\"]}}}, {\"type\": \"function\", \"function\": {\"name\": \"noop\"}}], \"tool_choice\": \"none\"}", "input_tokens": 99} +{"body": "{\"model\": \"gpt-4\", \"prompt\": \"Write a haiku about ships.\"}", "input_tokens": 7} +{"body": "{\"model\": \"gpt-4\", \"prompt\": [\"first prompt\", \"second prompt\"]}", "input_tokens": 4} +{"body": "{\"model\": \"gpt-4\", \"input\": [{\"role\": \"user\", \"content\": [{\"type\": \"input_text\", \"text\": \"Summarise caf\\u00e9 menus, na\\u00efve \\u2014 ok? \\\"quoted\\\"\\n\"}]}, {\"role\": \"assistant\", \"content\": \"Sure.\"}], \"instructions\": \"be terse\"}", "input_tokens": 60} +{"body": "{\"model\": \"gpt-4\", \"input\": [[101, 2023, 5], [7]], \"encoding_format\": \"float\"}", "input_tokens": 5} +{"body": "{\"model\": \"gpt-4\", \"query\": \"best harbour\", \"documents\": [\"doc one\", {\"text\": \"doc two\", \"title\": \"T\", \"n\": 3, \"ok\": true, \"none\": null, \"tags\": [\"a\", \"b\"]}]}", "input_tokens": 42} +{"body": "{\"model\": \"gpt-4\", \"messages\": [{\"role\": \"system\", \"content\": \"You are a helpful assistant. Answer precisely and cite sources.\"}, {\"role\": \"user\", \"content\": \"\\ud83d\\ude42 a every WON'T They'RE involved counting caf\\u00e9 backtracking boundaries Z\\u00fcrich WON'T 100% \\\"quotes\\\" caf\\u00e9 tiktoken's budget They'RE request request regex v1.2.3 hand hand fox tiktoken's on 3.14159 mirrors that don't WON'T admission before budget the admission WON'T no 1999 100% no admission budget mirrors way caf\\u00e9 dog a quick mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 $1,234.56 hand gateway regex on jumps because {braces} 'single' the the the {braces} piece hand scanned reservation [brackets] we'll 3.14159 request don't \\\"quotes\\\" na\\u00efve C++ caf\\u00e9 caf\\u00e9 for https://example.com/a/b?c=d we'll no over the node.js for while over way\"}, {\"role\": \"assistant\", \"content\": \"it tiktoken's node.js it scanned v1.2.3 boundaries (parens) and while reservation lazy the that https://example.com/a/b?c=d quick don't budget engine boundaries F# budget every we'll before jumps scanned jumps there's counting the don't jumps a once admission 100% 3.14159 brown brown that because jumps They'RE 100% caf\\u00e9 WON'T They'RE boundaries \\u6771\\u4eac exactly scanner caf\\u00e9 fox (parens) body dog a 1999 boundaries 'single' {braces} reservation mirrors while {braces} {braces} so admission the F# https://example.com/a/b?c=d brown a caf\\u00e9 on admission that on counting that we'll over lazy https://example.com/a/b?c=d way over engine reservation gateway body I'M for \\\"quotes\\\" 42 and F# there's 'single' quick \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T scanner written every (parens) so\"}, {\"role\": \"user\", \"content\": \"I'M over faster counting https://example.com/a/b?c=d way while it over way mirrors boundaries keep body hand na\\u00efve once because that node.js for every the 'single' every caf\\u00e9 caf\\u00e9 piece there's v1.2.3 v1.2.3 tiktoken's brown so counting there's for we'll boundaries 'single' no https://example.com/a/b?c=d node.js \\u6771\\u4eac written with budget (parens) because \\\"quotes\\\" \\u6771\\u4eac while \\u6771\\u4eac counting scanned 1999 \\u6771\\u4eac hand keep so that gateway caf\\u00e9 for scanned it request keep counting user@example.com admission request \\\"quotes\\\" v1.2.3 caf\\u00e9 over that F# we'll regex a faster with that They'RE piece because tiktoken's engine hand 100% WON'T reservation (parens) caf\\u00e9 on Z\\u00fcrich dog \\u6771\\u4eac that They'RE 'single' the 'single' na\\u00efve involved the {braces} there's scanned no engine dog backtracking there's\"}, {\"role\": \"assistant\", \"content\": \"every node.js C++ for 3.14159 \\u6771\\u4eac 'single' gateway once the user@example.com a 100% every every tokens written and every brown na\\u00efve the body the because quick brown engine fox 3.14159 (parens) F# while with a 100% C++ with the written WON'T on request Z\\u00fcrich hand on https://example.com/a/b?c=d don't way way 42 that we'll \\ud83d\\ude42 [brackets] admission tiktoken's request hand caf\\u00e9 every every (parens) They'RE that jumps the regex 100% \\\"quotes\\\" is budget\"}, {\"role\": \"user\", \"content\": \"engine with I'M backtracking while F# quick 'single' mirrors \\\"quotes\\\" 'single' regex caf\\u00e9 It's Z\\u00fcrich exactly a counting tiktoken's we'll once \\u0645\\u0631\\u062d\\u0628\\u0627 42 v1.2.3 scanner WON'T \\u6771\\u4eac there's node.js It's budget that budget {braces} because [brackets] It's a $1,234.56 that v1.2.3 42 gateway backtracking it C++ scanned keep \\ud83d\\ude42 I'M hand a counting scanner reservation \\u6771\\u4eac written 3.14159 there's once fox I'M C++ lazy the \\u0645\\u0631\\u062d\\u0628\\u0627 before don't we'll gateway exactly \\ud83d\\ude42 Z\\u00fcrich 3.14159 with WON'T there's node.js \\ud83d\\ude42 quick written faster counting 1999 backtracking 'single' counting we'll engine counting don't \\\"quotes\\\" engine \\\"quotes\\\" way I'M budget [brackets] backtracking \\\"quotes\\\" 3.14159 don't written written \\u0645\\u0631\\u062d\\u0628\\u0627 body I'M 'single' while on and reservation \\ud83d\\ude42 regex and no budget regex (parens) we'll They'RE na\\u00efve v1.2.3\"}, {\"role\": \"assistant\", \"content\": \"hand lazy dog budget lazy involved because scanned piece quick that involved 'single' dog quick na\\u00efve budget scanner exactly $1,234.56 fox quick I'M hand (parens) node.js while jumps boundaries \\ud83d\\ude42 They'RE way request engine 'single' fox backtracking regex \\u0645\\u0631\\u062d\\u0628\\u0627 because and over jumps 3.14159 WON'T counting for and mirrors admission 'single' caf\\u00e9 https://example.com/a/b?c=d that \\ud83d\\ude42 faster \\u0645\\u0631\\u062d\\u0628\\u0627 admission jumps quick for $1,234.56 exactly exactly $1,234.56 scanned keep 3.14159 backtracking piece tiktoken's is is hand reservation before regex budget 3.14159 tiktoken's I'M it no budget user@example.com budget written budget over that https://example.com/a/b?c=d no written so quick tokens C++\"}, {\"role\": \"user\", \"content\": \"once body involved every brown every it that hand engine with scanned reservation \\u6771\\u4eac backtracking and {braces} over [brackets] brown over for is \\ud83d\\ude42 100% before quick no counting no v1.2.3 counting \\\"quotes\\\" mirrors 3.14159 1999 there's way \\\"quotes\\\" piece on na\\u00efve They'RE no every exactly node.js 42 because over there's 1999 C++ we'll mirrors F# it scanner jumps It's scanner mirrors mirrors It's tokens backtracking body brown $1,234.56 keep admission 100% the exactly we'll caf\\u00e9 jumps over 3.14159 we'll so 42 \\u6771\\u4eac I'M WON'T lazy brown It's a na\\u00efve \\\"quotes\\\" https://example.com/a/b?c=d (parens) \\u6771\\u4eac tiktoken's so dog body 'single' scanned piece body 3.14159 scanned body the so \\\"quotes\\\" counting\"}, {\"role\": \"assistant\", \"content\": \"boundaries request the that 100% gateway the there's hand the They'RE caf\\u00e9 fox C++ scanner written \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE scanner WON'T keep before for it so counting that (parens) {braces} They'RE scanner every {braces} no node.js keep I'M jumps backtracking gateway https://example.com/a/b?c=d don't tiktoken's a is 1999 don't F# v1.2.3 involved hand scanner They'RE backtracking exactly and for exactly for $1,234.56 counting v1.2.3 request gateway mirrors that no \\ud83d\\ude42 every dog once They'RE 100% reservation It's a with and 100% because so It's admission there's gateway 42 on over gateway is every 3.14159 boundaries no for admission quick lazy lazy fox node.js because while $1,234.56 quick I'M lazy involved WON'T {braces} na\\u00efve {braces} there's with caf\\u00e9 https://example.com/a/b?c=d \\\"quotes\\\" [brackets] lazy lazy it 100% caf\\u00e9 way lazy tokens C++ that it \\u6771\\u4eac lazy\"}, {\"role\": \"user\", \"content\": \"Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors budget reservation 'single' it scanner F# is exactly They'RE \\ud83d\\ude42 I'M boundaries F# {braces} https://example.com/a/b?c=d over backtracking node.js is \\ud83d\\ude42 \\ud83d\\ude42 $1,234.56 so and counting It's Z\\u00fcrich once node.js once v1.2.3 on that we'll 'single' mirrors I'M \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking They'RE tokens hand counting 1999 user@example.com Z\\u00fcrich They'RE tokens reservation exactly for https://example.com/a/b?c=d mirrors and boundaries regex the brown while keep \\\"quotes\\\" we'll the involved 42 {braces} scanner reservation Z\\u00fcrich no we'll [brackets] caf\\u00e9 written that hand user@example.com $1,234.56 42 while budget every 3.14159 a exactly way body scanned admission C++ (parens) tiktoken's body for WON'T hand no dog 1999 (parens) on don't 1999 we'll I'M v1.2.3 that WON'T fox scanned 1999\"}, {\"role\": \"assistant\", \"content\": \"gateway $1,234.56 $1,234.56 \\\"quotes\\\" scanner there's scanner $1,234.56 100% it budget \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) admission admission with I'M admission 'single' hand jumps scanned It's \\\"quotes\\\" is F# before engine counting dog because admission tiktoken's backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 that 42 written it body quick Z\\u00fcrich I'M F# I'M that brown It's request caf\\u00e9 is boundaries jumps don't written written faster for admission Z\\u00fcrich engine reservation and Z\\u00fcrich scanned written keep scanned is keep jumps the so F# 'single' user@example.com keep because They'RE over because C++ written faster 42 mirrors na\\u00efve 42 tiktoken's [brackets] na\\u00efve hand on no written so $1,234.56 there's 100% 100% $1,234.56 is \\ud83d\\ude42 scanner dog lazy 100% https://example.com/a/b?c=d tiktoken's They'RE [brackets] user@example.com while hand \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d we'll 42 WON'T it a mirrors v1.2.3\"}, {\"role\": \"user\", \"content\": \"that involved dog C++ way that regex with https://example.com/a/b?c=d because 1999 jumps and and caf\\u00e9 F# backtracking that 3.14159 the quick v1.2.3 engine piece request brown (parens) because tokens na\\u00efve we'll and \\ud83d\\ude42 user@example.com that na\\u00efve \\u6771\\u4eac na\\u00efve tokens jumps 3.14159 tiktoken's na\\u00efve brown It's jumps before and a the user@example.com Z\\u00fcrich WON'T mirrors It's fox It's involved don't \\u6771\\u4eac 'single' written that written fox and the 3.14159 Z\\u00fcrich way on once na\\u00efve \\u0645\\u0631\\u062d\\u0628\\u0627 $1,234.56 request fox so fox 'single' admission admission node.js mirrors backtracking it the\"}, {\"role\": \"assistant\", \"content\": \"engine so $1,234.56 don't caf\\u00e9 lazy I'M while 'single' budget scanned \\ud83d\\ude42 Z\\u00fcrich piece tokens exactly budget boundaries admission written tokens every \\u0645\\u0631\\u062d\\u0628\\u0627 before with backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 engine regex faster brown \\ud83d\\ude42 before we'll reservation na\\u00efve regex there's faster counting na\\u00efve reservation quick Z\\u00fcrich tiktoken's that gateway we'll C++ They'RE scanned for engine budget 100% na\\u00efve $1,234.56 written admission no we'll WON'T scanned body dog node.js while\"}, {\"role\": \"user\", \"content\": \"once exactly scanner boundaries scanned every there's mirrors $1,234.56 there's so dog dog They'RE lazy fox because 'single' so gateway Z\\u00fcrich faster v1.2.3 quick I'M \\u6771\\u4eac exactly written tokens node.js na\\u00efve brown [brackets] mirrors while on \\u6771\\u4eac $1,234.56 once hand over exactly quick backtracking over exactly {braces} it don't regex 3.14159 way the 100% $1,234.56 is I'M admission admission [brackets] scanned while boundaries piece counting that node.js reservation \\u6771\\u4eac body jumps while node.js I'M Z\\u00fcrich with fox C++ reservation F# \\\"quotes\\\" They'RE {braces} (parens) caf\\u00e9 1999 there's {braces} every WON'T no dog WON'T 1999 admission [brackets] that body 'single' gateway while mirrors with with scanner hand no over scanned hand na\\u00efve It's the while with once \\\"quotes\\\" boundaries \\\"quotes\\\" It's tokens and\"}, {\"role\": \"assistant\", \"content\": \"jumps 100% before body there's a C++ $1,234.56 on exactly exactly every hand lazy dog don't every that so F# Z\\u00fcrich jumps caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 written scanned tokens admission WON'T backtracking scanned 1999 engine brown exactly counting node.js 'single' 1999 counting that keep keep $1,234.56 Z\\u00fcrich They'RE before scanned They'RE \\u6771\\u4eac It's over tiktoken's once hand request tiktoken's while gateway node.js no \\ud83d\\ude42 it https://example.com/a/b?c=d It's 100% written there's hand {braces} there's admission $1,234.56 F# [brackets]\"}, {\"role\": \"user\", \"content\": \"hand mirrors exactly 42 They'RE [brackets] before \\u0645\\u0631\\u062d\\u0628\\u0627 that while written caf\\u00e9 body because fox tokens no WON'T dog faster and fox keep don't caf\\u00e9 'single' node.js (parens) reservation C++ I'M fox F# \\u6771\\u4eac mirrors over 1999 written keep na\\u00efve gateway a 3.14159 keep once body \\\"quotes\\\" F# we'll $1,234.56 https://example.com/a/b?c=d regex fox backtracking is admission request involved so over engine caf\\u00e9 way backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac They'RE It's a mirrors \\\"quotes\\\" jumps They'RE way way the engine C++ request because backtracking with with written the scanned boundaries https://example.com/a/b?c=d (parens) no for gateway dog \\ud83d\\ude42 boundaries node.js\"}, {\"role\": \"assistant\", \"content\": \"and tiktoken's They'RE caf\\u00e9 exactly the quick gateway caf\\u00e9 budget hand node.js the node.js we'll faster fox 'single' involved scanner na\\u00efve reservation https://example.com/a/b?c=d {braces} way we'll backtracking keep while and no with dog exactly the WON'T dog \\u0645\\u0631\\u062d\\u0628\\u0627 regex [brackets] written 1999 keep v1.2.3 I'M mirrors admission \\u6771\\u4eac before we'll \\\"quotes\\\" (parens) 100% over mirrors 42 a request F# WON'T a counting \\ud83d\\ude42 scanner no we'll fox gateway boundaries 1999 WON'T I'M Z\\u00fcrich faster there's caf\\u00e9 They'RE that the with for faster quick that body reservation 42 don't I'M I'M scanned faster hand for\"}, {\"role\": \"user\", \"content\": \"and the admission caf\\u00e9 admission once F# written reservation budget written https://example.com/a/b?c=d scanner a that Z\\u00fcrich once the Z\\u00fcrich faster C++ no I'M counting is hand caf\\u00e9 before engine on C++ 'single' \\\"quotes\\\" I'M Z\\u00fcrich quick that boundaries node.js lazy 3.14159 while na\\u00efve Z\\u00fcrich it reservation request that before F# boundaries once written the WON'T https://example.com/a/b?c=d the written piece mirrors 100% mirrors https://example.com/a/b?c=d over before regex scanner \\u0645\\u0631\\u062d\\u0628\\u0627 before node.js WON'T \\u0645\\u0631\\u062d\\u0628\\u0627 way jumps for scanned that involved for every (parens) lazy C++ boundaries backtracking it user@example.com no that C++ faster engine I'M piece with faster for that brown 3.14159 user@example.com it while so on involved that 42 once quick written we'll a budget\"}, {\"role\": \"assistant\", \"content\": \"admission \\ud83d\\ude42 100% tiktoken's on jumps we'll hand body is tiktoken's and 100% \\ud83d\\ude42 the it 3.14159 $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 scanner over 100% scanned \\ud83d\\ude42 so way engine that scanner 100% so so tokens boundaries node.js we'll jumps user@example.com that tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 once don't way is body so user@example.com on request is lazy and every engine tokens no lazy admission jumps reservation v1.2.3 scanned \\u0645\\u0631\\u062d\\u0628\\u0627 the \\ud83d\\ude42 node.js so \\\"quotes\\\" every it \\u6771\\u4eac F# regex don't faster every $1,234.56 on way engine regex every keep $1,234.56 no that engine written don't involved {braces} \\u6771\\u4eac v1.2.3 while request 42 (parens)\"}, {\"role\": \"user\", \"content\": \"keep fox we'll backtracking once it engine with lazy (parens) faster caf\\u00e9 F# with It's (parens) user@example.com for over counting we'll it https://example.com/a/b?c=d fox the (parens) way over because https://example.com/a/b?c=d before written fox involved written [brackets] before it tokens \\\"quotes\\\" there's once [brackets] 'single' tiktoken's C++ v1.2.3 quick 100% user@example.com dog \\u6771\\u4eac I'M 'single' keep a before node.js F# exactly request the Z\\u00fcrich exactly tokens so faster admission lazy and every counting $1,234.56 brown\"}, {\"role\": \"assistant\", \"content\": \"'single' every budget 42 It's quick way dog {braces} because don't \\u0645\\u0631\\u062d\\u0628\\u0627 way brown involved counting body don't (parens) body tiktoken's scanned \\u0645\\u0631\\u062d\\u0628\\u0627 dog it (parens) the for once once engine written there's that node.js every \\u0645\\u0631\\u062d\\u0628\\u0627 1999 $1,234.56 caf\\u00e9 written body dog gateway for It's WON'T 42 'single' 3.14159 that Z\\u00fcrich jumps C++ while WON'T admission so involved It's \\u0645\\u0631\\u062d\\u0628\\u0627 node.js on with It's it They'RE lazy is engine way scanner $1,234.56 and lazy boundaries 'single' before dog that faster 3.14159 tokens It's tokens we'll once and reservation over They'RE\"}, {\"role\": \"user\", \"content\": \"I'M node.js C++ na\\u00efve once engine with \\ud83d\\ude42 involved 100% brown scanned and is scanner scanned 'single' counting don't fox way way C++ before every faster piece hand They'RE so brown piece because while on a WON'T 1999 backtracking budget \\u6771\\u4eac piece and engine every we'll dog user@example.com na\\u00efve https://example.com/a/b?c=d brown Z\\u00fcrich it way and so hand on \\\"quotes\\\" (parens) before we'll\"}, {\"role\": \"assistant\", \"content\": \"They'RE there's node.js mirrors there's the there's reservation engine is boundaries fox a body fox 42 user@example.com (parens) 100% on (parens) written request https://example.com/a/b?c=d {braces} It's Z\\u00fcrich every counting $1,234.56 scanned once user@example.com C++ so 100% exactly exactly {braces} body jumps on no involved 100% and admission every scanned reservation tokens exactly keep there's Z\\u00fcrich v1.2.3 keep 42 the 42 the with boundaries request involved there's faster keep on https://example.com/a/b?c=d user@example.com \\u0645\\u0631\\u062d\\u0628\\u0627 on (parens) na\\u00efve (parens) WON'T \\u6771\\u4eac with engine na\\u00efve body 1999 the engine quick counting boundaries there's counting {braces} for 100% involved there's involved so WON'T lazy a over way keep They'RE tokens keep piece na\\u00efve node.js exactly reservation body every faster backtracking\"}, {\"role\": \"user\", \"content\": \"and scanned C++ jumps They'RE scanned boundaries C++ that hand budget tokens \\\"quotes\\\" scanned I'M 100% 'single' counting because (parens) regex (parens) na\\u00efve F# (parens) I'M and with so once once for regex piece dog quick They'RE while keep exactly before 1999 is 100% keep caf\\u00e9 before 42 hand that reservation 3.14159 because fox that regex body gateway don't once gateway and engine with once don't I'M piece way 3.14159 admission the don't it body piece \\\"quotes\\\" 42 mirrors on body don't 100% that quick backtracking {braces} is we'll and body we'll budget every every v1.2.3 exactly way I'M \\\"quotes\\\" involved gateway scanned once \\u0645\\u0631\\u062d\\u0628\\u0627 keep jumps \\u6771\\u4eac backtracking dog engine \\u6771\\u4eac tiktoken's that 42 user@example.com don't scanner (parens) I'M\"}, {\"role\": \"assistant\", \"content\": \"the piece 1999 over v1.2.3 C++ It's https://example.com/a/b?c=d because request that fox \\ud83d\\ude42 way \\u6771\\u4eac tokens tokens brown (parens) way v1.2.3 that na\\u00efve mirrors \\\"quotes\\\" admission dog 100% 100% regex backtracking reservation 1999 user@example.com \\\"quotes\\\" 3.14159 I'M budget mirrors 1999 lazy admission a on that user@example.com They'RE They'RE \\\"quotes\\\" user@example.com node.js 100% so na\\u00efve and It's \\\"quotes\\\" with the piece boundaries we'll 42 42 while https://example.com/a/b?c=d user@example.com budget faster na\\u00efve and 1999 a admission fox and 'single' for They'RE boundaries scanned\"}, {\"role\": \"user\", \"content\": \"quick gateway They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 fox the (parens) mirrors because while we'll we'll every mirrors counting Z\\u00fcrich 42 on gateway WON'T written backtracking no user@example.com caf\\u00e9 \\u6771\\u4eac that boundaries while so fox backtracking involved They'RE exactly 1999 {braces} [brackets] exactly regex mirrors \\u6771\\u4eac 1999 over reservation way involved \\u0645\\u0631\\u062d\\u0628\\u0627 42 3.14159 while quick so a (parens) hand there's F# a They'RE body engine F# v1.2.3 faster hand over dog backtracking while brown that before I'M 1999 way before piece counting request that budget boundaries v1.2.3 exactly that They'RE faster involved \\\"quotes\\\" scanner C++ It's 100% $1,234.56 written\"}, {\"role\": \"assistant\", \"content\": \"https://example.com/a/b?c=d way \\\"quotes\\\" fox fox C++ is admission It's \\ud83d\\ude42 tiktoken's WON'T tokens so caf\\u00e9 way admission a scanned $1,234.56 3.14159 that F# don't fox counting every engine scanner scanned don't 'single' tiktoken's https://example.com/a/b?c=d I'M keep there's piece \\u6771\\u4eac is while way hand over fox WON'T \\u6771\\u4eac scanner v1.2.3 \\u6771\\u4eac don't written \\\"quotes\\\" keep the body and brown that counting before budget request 3.14159 boundaries {braces}\"}, {\"role\": \"user\", \"content\": \"exactly It's 3.14159 hand body They'RE tokens don't v1.2.3 backtracking on quick (parens) 100% body quick so scanner so \\ud83d\\ude42 backtracking is once exactly regex https://example.com/a/b?c=d na\\u00efve before there's every C++ [brackets] tokens They'RE with the 1999 piece a reservation request caf\\u00e9 it 'single' while \\\"quotes\\\" boundaries because lazy \\u0645\\u0631\\u062d\\u0628\\u0627 for Z\\u00fcrich It's (parens) a 100% there's dog caf\\u00e9 \\ud83d\\ude42 (parens) because quick with node.js https://example.com/a/b?c=d engine piece \\ud83d\\ude42 counting brown way It's user@example.com user@example.com \\u0645\\u0631\\u062d\\u0628\\u0627 scanned reservation 'single' don't request admission\"}, {\"role\": \"assistant\", \"content\": \"reservation it so before no scanned backtracking a 1999 every exactly jumps 100% over It's 1999 we'll boundaries don't scanned once and before \\u6771\\u4eac faster backtracking regex backtracking every 'single' admission caf\\u00e9 brown 3.14159 \\ud83d\\ude42 there's \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T 42 $1,234.56 is before reservation it admission backtracking before over $1,234.56 {braces} mirrors It's {braces} before admission quick hand lazy scanned \\ud83d\\ude42 exactly faster while reservation C++ They'RE it exactly gateway it body for quick so quick 3.14159 1999 for user@example.com involved {braces} https://example.com/a/b?c=d quick while I'M lazy brown backtracking keep\"}, {\"role\": \"user\", \"content\": \"quick keep v1.2.3 request budget a and regex keep body gateway so It's before \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 I'M 3.14159 and written counting {braces} v1.2.3 for brown engine user@example.com mirrors mirrors over \\u6771\\u4eac (parens) regex jumps keep written and Z\\u00fcrich a no dog and jumps piece C++ the counting with quick on the user@example.com request is jumps boundaries quick 'single' 100% 'single' over 100% the over there's na\\u00efve 1999 engine quick https://example.com/a/b?c=d 3.14159 it jumps way no hand \\ud83d\\ude42 [brackets] {braces} admission F# scanner 3.14159 F# it 3.14159 boundaries budget keep don't that quick 'single' reservation C++ the gateway and dog exactly body so https://example.com/a/b?c=d \\\"quotes\\\" 100%\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 budget \\\"quotes\\\" quick once body I'M that way no once a user@example.com boundaries admission gateway engine caf\\u00e9 $1,234.56 They'RE They'RE reservation user@example.com They'RE They'RE budget for \\u0645\\u0631\\u062d\\u0628\\u0627 dog is WON'T faster involved lazy so while faster backtracking 1999 user@example.com we'll engine reservation v1.2.3 it on tokens counting \\ud83d\\ude42 every every over written it every tiktoken's no before (parens) body the request quick every admission the \\ud83d\\ude42 body don't admission regex there's \\u6771\\u4eac \\\"quotes\\\" dog don't no Z\\u00fcrich $1,234.56 and scanned scanned https://example.com/a/b?c=d request while way written engine reservation there's scanned the request that jumps it caf\\u00e9 and v1.2.3 scanned na\\u00efve involved brown no user@example.com tokens because counting there's \\\"quotes\\\" on node.js quick exactly every that written\"}, {\"role\": \"user\", \"content\": \"counting {braces} \\ud83d\\ude42 C++ admission boundaries C++ it 'single' that mirrors jumps the while that don't They'RE while 42 Z\\u00fcrich 1999 scanner so quick v1.2.3 there's don't keep Z\\u00fcrich no keep faster reservation request is (parens) body we'll 100% don't na\\u00efve (parens) for backtracking fox boundaries backtracking v1.2.3 $1,234.56 we'll 42 tokens faster counting once keep budget fox v1.2.3 'single' mirrors no engine exactly fox 42 way the C++ quick tiktoken's counting 1999 It's we'll counting because hand \\ud83d\\ude42 user@example.com is request quick 42 it keep don't They'RE WON'T once and fox v1.2.3 and with na\\u00efve way lazy user@example.com written is 'single' brown scanned request is caf\\u00e9 we'll node.js a so while we'll WON'T way admission They'RE fox C++ once node.js WON'T that v1.2.3 every\"}, {\"role\": \"assistant\", \"content\": \"C++ don't no boundaries scanner https://example.com/a/b?c=d node.js backtracking hand [brackets] keep F# counting caf\\u00e9 scanner v1.2.3 https://example.com/a/b?c=d counting no https://example.com/a/b?c=d the backtracking it I'M I'M F# involved WON'T with exactly (parens) brown body v1.2.3 body https://example.com/a/b?c=d keep it It's 1999 F# \\u6771\\u4eac so \\ud83d\\ude42 scanner for user@example.com v1.2.3 I'M don't fox written request engine once regex counting tokens the admission (parens) admission reservation faster C++ 3.14159 tokens hand \\\"quotes\\\" counting caf\\u00e9 [brackets] It's don't Z\\u00fcrich 3.14159 the tiktoken's I'M v1.2.3 admission C++ They'RE hand tokens dog no $1,234.56 engine while boundaries so caf\\u00e9 100% \\u0645\\u0631\\u062d\\u0628\\u0627 F# scanned jumps faster while na\\u00efve lazy\"}, {\"role\": \"user\", \"content\": \"WON'T hand keep brown \\u0645\\u0631\\u062d\\u0628\\u0627 because counting faster with that involved is because because 42 Z\\u00fcrich that engine with the the brown {braces} tiktoken's [brackets] I'M It's over (parens) C++ It's over faster \\u6771\\u4eac user@example.com user@example.com 100% on it on with mirrors 3.14159 42 budget It's WON'T node.js keep tiktoken's \\u6771\\u4eac https://example.com/a/b?c=d 1999 involved body scanned hand involved engine faster every is and that tokens every 42 on dog admission (parens) body for caf\\u00e9 once before a \\u6771\\u4eac before with don't tokens It's because \\\"quotes\\\" node.js na\\u00efve it engine is backtracking [brackets] regex brown They'RE so is is involved engine so scanner gateway scanned They'RE keep na\\u00efve body reservation 42 scanned WON'T faster there's backtracking it caf\\u00e9 v1.2.3 brown regex {braces} lazy node.js C++ don't written WON'T with user@example.com piece over we'll v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich jumps (parens)\"}, {\"role\": \"assistant\", \"content\": \"na\\u00efve scanned every reservation no we'll faster before it {braces} admission 1999 scanned budget written there's WON'T $1,234.56 faster brown reservation (parens) 'single' every (parens) $1,234.56 It's is counting way jumps no scanned I'M the node.js that na\\u00efve over faster [brackets] 1999 there's keep user@example.com 100% user@example.com before once reservation brown don't C++ dog it over backtracking every \\\"quotes\\\" quick a budget \\ud83d\\ude42 budget because v1.2.3 dog 100% once v1.2.3 and and every don't 3.14159 once once quick gateway before for 3.14159 v1.2.3 \\u6771\\u4eac tokens that on because it budget a involved scanned scanner lazy 'single' caf\\u00e9 They'RE engine request while Z\\u00fcrich engine WON'T\"}, {\"role\": \"user\", \"content\": \"100% over written reservation https://example.com/a/b?c=d WON'T scanner F# before https://example.com/a/b?c=d we'll that while counting is admission 42 caf\\u00e9 C++ body tiktoken's body 3.14159 the I'M with node.js It's \\u6771\\u4eac [brackets] don't https://example.com/a/b?c=d C++ before on [brackets] 'single' piece brown before 1999 {braces} written {braces} way reservation request F# so https://example.com/a/b?c=d 100% $1,234.56 piece piece keep https://example.com/a/b?c=d way budget 100% boundaries node.js lazy because F# tokens (parens) and C++ backtracking tiktoken's piece a 3.14159 WON'T scanned backtracking node.js because regex backtracking so keep 1999 the is because WON'T node.js we'll lazy quick with once and once \\\"quotes\\\" we'll keep It's the on It's so is faster\"}, {\"role\": \"assistant\", \"content\": \"hand every for written mirrors backtracking $1,234.56 It's WON'T because faster Z\\u00fcrich don't don't {braces} jumps 'single' regex gateway user@example.com once there's a on over \\\"quotes\\\" 100% node.js regex a body that F# 100% once on 3.14159 while backtracking v1.2.3 hand \\u6771\\u4eac tokens admission \\\"quotes\\\" 100% admission and It's Z\\u00fcrich that so while and $1,234.56 caf\\u00e9 every F# involved fox backtracking the \\u0645\\u0631\\u062d\\u0628\\u0627 and https://example.com/a/b?c=d exactly node.js before mirrors 'single' exactly tokens that scanned body https://example.com/a/b?c=d na\\u00efve once it the the faster v1.2.3 scanned fox that jumps v1.2.3 1999 gateway F# caf\\u00e9 'single' fox brown [brackets] 100% over user@example.com on na\\u00efve https://example.com/a/b?c=d node.js They'RE and scanner 'single' involved the because hand scanner fox user@example.com\"}, {\"role\": \"user\", \"content\": \"1999 $1,234.56 that 'single' no counting (parens) because Z\\u00fcrich on (parens) a fox user@example.com admission over is there's \\\"quotes\\\" {braces} exactly piece that with I'M F# over for counting and \\ud83d\\ude42 scanner over every \\ud83d\\ude42 100% admission before \\u0645\\u0631\\u062d\\u0628\\u0627 reservation WON'T brown scanner on faster 100% caf\\u00e9 piece [brackets] counting scanner It's written {braces} C++ that is boundaries exactly \\u0645\\u0631\\u062d\\u0628\\u0627 once 'single' quick F# hand 'single' user@example.com reservation jumps it F# reservation request that 'single' (parens) the way WON'T (parens) a backtracking backtracking that \\\"quotes\\\" C++ I'M C++ C++ gateway with body body\"}, {\"role\": \"assistant\", \"content\": \"budget that 42 It's body backtracking caf\\u00e9 1999 because hand hand F# piece is (parens) and Z\\u00fcrich jumps user@example.com don't I'M They'RE regex body reservation once (parens) with F# piece is every node.js no piece because {braces} with 42 Z\\u00fcrich tiktoken's keep I'M is scanner no body 'single' 1999 that request so \\u0645\\u0631\\u062d\\u0628\\u0627 3.14159 \\u6771\\u4eac and https://example.com/a/b?c=d engine tiktoken's on while $1,234.56 over budget 1999 1999 backtracking is \\ud83d\\ude42 the tiktoken's involved over don't Z\\u00fcrich I'M so gateway written dog before written v1.2.3 backtracking gateway keep dog [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9\"}, {\"role\": \"user\", \"content\": \"exactly involved user@example.com fox keep boundaries and 42 reservation {braces} mirrors the It's budget C++ tiktoken's budget mirrors faster [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 over scanner way quick while \\\"quotes\\\" exactly 'single' F# no faster [brackets] backtracking scanner WON'T WON'T tokens quick that the node.js it regex user@example.com involved while boundaries regex while hand mirrors way na\\u00efve a hand request that before v1.2.3 1999 3.14159 caf\\u00e9 because on over scanner so body tokens quick because counting exactly hand user@example.com that It's\"}, {\"role\": \"assistant\", \"content\": \"so quick dog (parens) Z\\u00fcrich backtracking for the \\ud83d\\ude42 because regex involved tokens dog we'll 'single' we'll Z\\u00fcrich once 100% regex we'll [brackets] 42 while with 1999 Z\\u00fcrich fox admission while written the C++ over every mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 on it https://example.com/a/b?c=d scanned gateway no 1999 dog https://example.com/a/b?c=d we'll 1999 before the every budget on 1999 with piece 1999 faster user@example.com I'M body keep while the once scanned I'M\"}, {\"role\": \"user\", \"content\": \"tiktoken's quick budget on budget the \\u0645\\u0631\\u062d\\u0628\\u0627 for Z\\u00fcrich I'M don't WON'T we'll user@example.com It's jumps \\u6771\\u4eac we'll involved gateway it that jumps for for counting that $1,234.56 so while 'single' WON'T quick the brown we'll once mirrors [brackets] \\u6771\\u4eac \\\"quotes\\\" boundaries way for because {braces} it user@example.com faster $1,234.56 no a 'single' body backtracking before the 3.14159 scanner {braces} backtracking hand because so body [brackets] because that involved scanned WON'T there's \\u0645\\u0631\\u062d\\u0628\\u0627 it hand we'll dog a before the and faster \\ud83d\\ude42 hand 100% on body [brackets] 42 the node.js it counting and jumps we'll a fox 3.14159 once (parens) \\ud83d\\ude42 hand every don't way jumps body over involved \\u6771\\u4eac is budget way scanned $1,234.56 because request gateway involved the that dog the reservation while \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"assistant\", \"content\": \"counting written brown on once it They'RE involved we'll every hand \\ud83d\\ude42 it They'RE dog for reservation on every no tokens regex piece on (parens) tiktoken's before gateway it every the while that we'll gateway a the a v1.2.3 the over lazy node.js gateway budget WON'T \\ud83d\\ude42 is exactly jumps over lazy piece backtracking written with 3.14159 jumps involved lazy scanner keep keep [brackets] boundaries that no that because quick F# v1.2.3 reservation involved\"}, {\"role\": \"user\", \"content\": \"caf\\u00e9 engine request na\\u00efve jumps involved lazy regex scanned {braces} 3.14159 engine mirrors \\\"quotes\\\" Z\\u00fcrich on there's (parens) They'RE regex for over regex 1999 tiktoken's scanner that once admission F# mirrors 42 the body WON'T the WON'T dog 100% 42 is lazy \\ud83d\\ude42 that request once 100% na\\u00efve dog tokens budget jumps \\ud83d\\ude42 $1,234.56 user@example.com $1,234.56 don't admission that brown (parens) it node.js tokens na\\u00efve we'll it v1.2.3 once so is written admission faster way we'll request jumps v1.2.3 fox the lazy gateway $1,234.56 42 counting Z\\u00fcrich the 'single' exactly once way I'M \\u6771\\u4eac They'RE before caf\\u00e9 boundaries over counting piece brown faster while so counting na\\u00efve hand \\ud83d\\ude42 quick the every Z\\u00fcrich {braces} scanned \\\"quotes\\\" with regex keep while once engine before fox\"}, {\"role\": \"assistant\", \"content\": \"mirrors {braces} because \\u6771\\u4eac mirrors scanned exactly budget way mirrors https://example.com/a/b?c=d boundaries involved 100% 3.14159 exactly engine brown They'RE is keep that hand lazy exactly tokens It's keep every \\ud83d\\ude42 the F# written {braces} user@example.com the counting $1,234.56 keep regex tiktoken's and mirrors fox the gateway \\ud83d\\ude42 keep They'RE written I'M there's we'll na\\u00efve WON'T is brown C++ involved brown and piece \\ud83d\\ude42 reservation [brackets] reservation I'M \\\"quotes\\\" while exactly way https://example.com/a/b?c=d don't \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac over keep scanner \\u6771\\u4eac scanned over tiktoken's boundaries mirrors once for quick faster a \\u6771\\u4eac lazy (parens) na\\u00efve gateway \\u6771\\u4eac \\u6771\\u4eac brown They'RE there's on keep once dog that 1999 [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 while it It's no\"}, {\"role\": \"user\", \"content\": \"that body over Z\\u00fcrich before that dog keep and \\\"quotes\\\" and regex tiktoken's way piece is scanner is quick C++ node.js written dog regex the It's dog https://example.com/a/b?c=d scanned engine we'll counting it 'single' counting a boundaries is \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 brown budget fox the Z\\u00fcrich jumps {braces} that boundaries piece because the 'single' v1.2.3 {braces} once that mirrors \\ud83d\\ude42 backtracking no no mirrors on on scanned F# They'RE regex scanned the every for faster v1.2.3 \\ud83d\\ude42 I'M we'll exactly that is once for because it caf\\u00e9 It's caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 keep budget admission\"}, {\"role\": \"assistant\", \"content\": \"and node.js is so admission body They'RE https://example.com/a/b?c=d the 1999 we'll It's keep mirrors request so piece gateway engine way hand exactly is C++ 100% exactly for body piece with jumps WON'T the no keep WON'T so F# don't $1,234.56 {braces} and \\ud83d\\ude42 request reservation written tiktoken's that written way written the tokens 3.14159 WON'T admission \\u0645\\u0631\\u062d\\u0628\\u0627 C++ budget gateway every before brown WON'T piece 1999 [brackets] a dog [brackets] while scanner \\ud83d\\ude42 we'll v1.2.3 scanner quick dog a while admission hand so there's every 42 node.js that engine faster na\\u00efve is \\\"quotes\\\"\"}, {\"role\": \"user\", \"content\": \"scanner hand boundaries admission it \\u6771\\u4eac WON'T na\\u00efve user@example.com Z\\u00fcrich F# F# They'RE a is caf\\u00e9 \\\"quotes\\\" before on request $1,234.56 boundaries (parens) don't brown the every 1999 (parens) [brackets] so 1999 boundaries 100% the hand 42 tokens every over that the the F# that quick They'RE before because tokens caf\\u00e9 we'll exactly boundaries because brown keep no 'single' there's brown v1.2.3 user@example.com 42 1999 \\u6771\\u4eac {braces} for mirrors on tiktoken's WON'T \\\"quotes\\\" na\\u00efve They'RE {braces} v1.2.3 brown written involved tokens jumps boundaries budget jumps boundaries way 42 body written na\\u00efve written is request every admission counting before backtracking with They'RE fox\"}, {\"role\": \"assistant\", \"content\": \"\\ud83d\\ude42 we'll backtracking involved counting request node.js caf\\u00e9 lazy $1,234.56 so I'M while is every it 42 exactly node.js that They'RE there's I'M I'M a and counting admission no 'single' admission v1.2.3 request boundaries 'single' tiktoken's budget gateway 100% caf\\u00e9 jumps hand no the \\u6771\\u4eac I'M we'll fox 1999 piece and body {braces} na\\u00efve hand Z\\u00fcrich v1.2.3 over quick for 100% while backtracking scanned scanner scanner request for hand fox the that hand scanned It's while (parens) written\"}, {\"role\": \"user\", \"content\": \"body there's scanned na\\u00efve quick It's a request lazy hand 1999 https://example.com/a/b?c=d jumps and caf\\u00e9 every It's before we'll on body no lazy there's 'single' mirrors we'll tokens over (parens) while \\u0645\\u0631\\u062d\\u0628\\u0627 C++ v1.2.3 keep gateway it because request mirrors for there's $1,234.56 keep over gateway https://example.com/a/b?c=d brown before They'RE 42 [brackets] tokens once once while scanner scanned and I'M Z\\u00fcrich it involved na\\u00efve I'M hand exactly before node.js 'single' because no lazy 3.14159 It's scanner F# 'single' while exactly It's (parens) \\ud83d\\ude42 counting while for tokens backtracking fox (parens) v1.2.3 once quick because is there's reservation because It's node.js that regex for don't\"}, {\"role\": \"assistant\", \"content\": \"with with so written budget is jumps admission before that I'M \\\"quotes\\\" because 3.14159 on WON'T \\\"quotes\\\" caf\\u00e9 is for scanned engine so engine on admission no body before once node.js node.js [brackets] They'RE Z\\u00fcrich written the \\ud83d\\ude42 reservation once so it a budget the request \\u6771\\u4eac https://example.com/a/b?c=d request \\u0645\\u0631\\u062d\\u0628\\u0627 a the 3.14159 with mirrors because body [brackets] exactly \\ud83d\\ude42 the a 1999 every lazy 'single' the request keep once dog user@example.com the scanned \\\"quotes\\\" is regex scanner every hand keep faster request quick WON'T while 3.14159 It's 3.14159 mirrors reservation written the a It's don't reservation C++ It's 42 v1.2.3 involved reservation lazy keep \\\"quotes\\\" the body \\ud83d\\ude42 while 3.14159 node.js dog no user@example.com budget [brackets] hand over\"}, {\"role\": \"user\", \"content\": \"before 1999 \\ud83d\\ude42 on that don't Z\\u00fcrich keep a way for request \\u6771\\u4eac every \\\"quotes\\\" https://example.com/a/b?c=d 3.14159 once for every hand that WON'T 'single' request written jumps regex F# once quick \\u6771\\u4eac dog na\\u00efve the node.js way [brackets] gateway (parens) piece exactly \\ud83d\\ude42 v1.2.3 jumps They'RE written fox the scanner exactly WON'T na\\u00efve so body faster brown the that is with exactly Z\\u00fcrich [brackets] Z\\u00fcrich budget it lazy and the engine budget 3.14159 counting \\ud83d\\ude42 https://example.com/a/b?c=d it WON'T $1,234.56 {braces} a because we'll exactly it with request the that don't scanner jumps involved and tiktoken's the quick so \\ud83d\\ude42 [brackets] regex engine involved is\"}, {\"role\": \"assistant\", \"content\": \"They'RE body because engine 100% WON'T {braces} the tiktoken's before node.js it every na\\u00efve lazy the don't caf\\u00e9 admission 1999 faster dog 3.14159 {braces} once tokens we'll before the admission WON'T the it and It's because engine with user@example.com it \\u6771\\u4eac C++ over is tokens piece exactly it piece backtracking gateway node.js no the 3.14159 exactly regex fox \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors a 1999 v1.2.3 request 100% scanned reservation node.js I'M the 100% so quick before piece brown caf\\u00e9 gateway 42 involved It's involved admission Z\\u00fcrich and {braces} It's quick \\ud83d\\ude42 body fox counting\"}, {\"role\": \"user\", \"content\": \"I'M admission regex that involved engine it \\ud83d\\ude42 counting every dog hand before scanned the 100% we'll exactly Z\\u00fcrich \\u6771\\u4eac v1.2.3 engine na\\u00efve node.js body don't lazy tiktoken's gateway don't node.js request F# scanner there's 100% scanner fox the a 42 quick na\\u00efve admission while 1999 way regex there's \\\"quotes\\\" scanned is so before tiktoken's involved [brackets] quick 1999 \\u6771\\u4eac 1999 for lazy v1.2.3 regex lazy body F# hand boundaries once mirrors so brown \\\"quotes\\\" dog $1,234.56 tokens that once request hand tokens lazy lazy [brackets] quick don't mirrors quick don't lazy It's body boundaries mirrors F# [brackets] there's while 100% body\"}, {\"role\": \"assistant\", \"content\": \"budget before over They'RE I'M request is that while written before $1,234.56 {braces} F# that admission and with for [brackets] fox \\ud83d\\ude42 42 body \\\"quotes\\\" the F# reservation dog 3.14159 WON'T It's tiktoken's exactly regex 3.14159 there's involved backtracking don't 1999 with faster dog body before tiktoken's with and over 42 written there's $1,234.56 hand 42 \\\"quotes\\\" lazy gateway backtracking 100% because \\\"quotes\\\" hand caf\\u00e9 faster request on over \\\"quotes\\\" once scanned over 100% engine mirrors \\u6771\\u4eac gateway user@example.com fox every before [brackets] body admission WON'T engine no dog for node.js gateway node.js na\\u00efve piece the way scanner a the 'single' gateway user@example.com They'RE is don't \\u6771\\u4eac before scanned\"}, {\"role\": \"user\", \"content\": \"It's way that \\u0645\\u0631\\u062d\\u0628\\u0627 written tiktoken's over faster It's with WON'T it because faster because with backtracking Z\\u00fcrich the while because scanned mirrors that scanned They'RE fox exactly 100% It's the tiktoken's hand so user@example.com request \\u6771\\u4eac faster v1.2.3 quick on once over dog piece counting node.js They'RE the because (parens) hand F# \\u6771\\u4eac with hand dog over budget written Z\\u00fcrich jumps there's budget C++ backtracking v1.2.3 'single' don't fox They'RE for \\ud83d\\ude42 admission involved v1.2.3 3.14159 the while 42 quick regex 100% on node.js tokens scanner written every with They'RE \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"100% because user@example.com caf\\u00e9 way 1999 Z\\u00fcrich over written 'single' 3.14159 $1,234.56 tokens I'M C++ [brackets] na\\u00efve $1,234.56 we'll $1,234.56 node.js we'll it (parens) {braces} v1.2.3 with written admission 100% exactly {braces} reservation there's {braces} fox with engine na\\u00efve for 1999 brown hand while dog backtracking backtracking na\\u00efve user@example.com tokens \\u0645\\u0631\\u062d\\u0628\\u0627 body brown don't C++ piece over scanner (parens) 100% no F# keep there's v1.2.3 [brackets] we'll don't hand dog every engine 'single' v1.2.3\"}, {\"role\": \"user\", \"content\": \"before and every $1,234.56 caf\\u00e9 the mirrors reservation https://example.com/a/b?c=d 3.14159 the quick gateway exactly it $1,234.56 1999 v1.2.3 $1,234.56 it on node.js backtracking (parens) there's \\ud83d\\ude42 no quick regex \\ud83d\\ude42 before once \\u0645\\u0631\\u062d\\u0628\\u0627 involved gateway caf\\u00e9 admission $1,234.56 caf\\u00e9 backtracking scanner na\\u00efve a engine [brackets] a way written is caf\\u00e9 1999 gateway 100% boundaries [brackets] backtracking there's tokens while 42 na\\u00efve counting it C++ v1.2.3 100% v1.2.3 scanned \\u0645\\u0631\\u062d\\u0628\\u0627 'single' \\u6771\\u4eac WON'T there's keep It's so WON'T\"}, {\"role\": \"assistant\", \"content\": \"keep 42 because quick and because faster is 'single' scanned fox regex {braces} C++ I'M it 100% written tiktoken's engine so scanned reservation \\ud83d\\ude42 3.14159 \\\"quotes\\\" before the that exactly for with there's 100% over with WON'T It's quick Z\\u00fcrich v1.2.3 'single' the tiktoken's (parens) while we'll written [brackets] reservation the on engine way na\\u00efve caf\\u00e9 exactly $1,234.56 https://example.com/a/b?c=d before written 'single' (parens) piece They'RE Z\\u00fcrich written counting the every that written na\\u00efve hand admission over piece Z\\u00fcrich we'll no quick https://example.com/a/b?c=d keep we'll while 'single' request quick on we'll tiktoken's on regex Z\\u00fcrich jumps written that a because mirrors\"}, {\"role\": \"user\", \"content\": \"\\\"quotes\\\" reservation na\\u00efve there's 100% dog reservation [brackets] 3.14159 F# node.js \\u0645\\u0631\\u062d\\u0628\\u0627 the fox boundaries C++ na\\u00efve before It's so written brown C++ lazy 3.14159 100% [brackets] tokens lazy tiktoken's {braces} with and and fox I'M for the exactly gateway keep backtracking that fox Z\\u00fcrich quick lazy every \\u0645\\u0631\\u062d\\u0628\\u0627 hand Z\\u00fcrich \\u6771\\u4eac caf\\u00e9 C++ there's keep the over v1.2.3 mirrors user@example.com It's They'RE tiktoken's 1999 because piece jumps it dog a budget boundaries brown exactly piece lazy 100% so scanned https://example.com/a/b?c=d there's that don't the engine fox tiktoken's 42 $1,234.56 \\u6771\\u4eac keep the dog admission boundaries piece It's mirrors fox user@example.com boundaries I'M faster reservation It's for no fox \\u0645\\u0631\\u062d\\u0628\\u0627 and because boundaries budget involved written They'RE v1.2.3 admission while brown F# so 'single' body that\"}, {\"role\": \"assistant\", \"content\": \"body tokens exactly over dog Z\\u00fcrich WON'T exactly F# no engine and \\ud83d\\ude42 scanner way {braces} scanner jumps tokens piece [brackets] hand that na\\u00efve \\\"quotes\\\" dog for node.js the v1.2.3 that They'RE once gateway so {braces} with keep that regex no [brackets] admission on caf\\u00e9 fox scanner tokens don't before quick They'RE budget dog C++ once jumps user@example.com a node.js (parens) because WON'T \\ud83d\\ude42 while request 3.14159 with because there's regex don't Z\\u00fcrich written user@example.com with while request keep They'RE mirrors 1999 over involved \\u6771\\u4eac $1,234.56 C++ counting involved that tokens way for \\ud83d\\ude42 regex jumps we'll because na\\u00efve lazy with I'M don't before regex reservation fox that so\"}, {\"role\": \"user\", \"content\": \"involved because dog 1999 exactly 1999 keep boundaries node.js tiktoken's once Z\\u00fcrich [brackets] way na\\u00efve v1.2.3 faster C++ scanner mirrors \\u6771\\u4eac \\u6771\\u4eac [brackets] hand \\ud83d\\ude42 node.js tiktoken's don't counting exactly tiktoken's keep WON'T while with 42 brown it way Z\\u00fcrich over because keep before way {braces} the before way boundaries that \\u6771\\u4eac 1999 request mirrors over It's gateway with scanner It's tiktoken's Z\\u00fcrich gateway faster we'll Z\\u00fcrich admission\"}, {\"role\": \"assistant\", \"content\": \"once They'RE node.js we'll \\u0645\\u0631\\u062d\\u0628\\u0627 don't Z\\u00fcrich admission {braces} It's \\ud83d\\ude42 \\u6771\\u4eac gateway brown is so scanner [brackets] {braces} there's tiktoken's a \\u0645\\u0631\\u062d\\u0628\\u0627 node.js and [brackets] don't counting \\\"quotes\\\" They'RE every mirrors reservation body on piece on regex node.js a 'single' \\u6771\\u4eac tiktoken's \\\"quotes\\\" for scanned there's faster gateway They'RE mirrors jumps https://example.com/a/b?c=d faster tokens every WON'T WON'T that request budget It's counting we'll scanner faster tokens that [brackets] tiktoken's there's F# lazy the reservation on the because with lazy user@example.com $1,234.56 jumps gateway C++ [brackets]\"}, {\"role\": \"user\", \"content\": \"over on regex lazy piece don't (parens) don't so https://example.com/a/b?c=d WON'T v1.2.3 brown node.js 100% \\u0645\\u0631\\u062d\\u0628\\u0627 over admission {braces} https://example.com/a/b?c=d scanned hand It's tiktoken's C++ quick don't They'RE {braces} WON'T scanner there's admission the body $1,234.56 dog \\u6771\\u4eac Z\\u00fcrich involved (parens) for so hand 3.14159 so scanned \\u0645\\u0631\\u062d\\u0628\\u0627 42 100% \\u6771\\u4eac counting tiktoken's 1999 fox keep reservation caf\\u00e9 there's tokens quick F# no is v1.2.3 body \\ud83d\\ude42 brown dog way boundaries scanned node.js quick over admission user@example.com boundaries faster regex on 42 admission na\\u00efve engine lazy WON'T that user@example.com 100% I'M with because faster exactly way for faster C++ because scanner written the \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"there's for I'M on reservation once once mirrors the body body 3.14159 fox dog \\\"quotes\\\" written we'll \\ud83d\\ude42 regex with user@example.com fox 1999 Z\\u00fcrich na\\u00efve is hand hand before on for over piece exactly that the exactly 100% caf\\u00e9 that brown counting way admission 100% 1999 v1.2.3 don't \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} mirrors it once piece Z\\u00fcrich gateway caf\\u00e9 and the budget that because hand scanner tokens request regex for exactly over piece F# we'll [brackets] I'M while and fox scanner \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} on fox quick mirrors It's scanner the scanner no it because node.js 'single' regex written caf\\u00e9 v1.2.3 https://example.com/a/b?c=d {braces} the don't jumps It's hand 'single' scanned 3.14159 involved every fox over we'll keep there's mirrors written on I'M WON'T F# that F# that 3.14159 1999 keep 'single'\"}, {\"role\": \"user\", \"content\": \"exactly involved caf\\u00e9 that 1999 over a there's it F# exactly \\ud83d\\ude42 tokens we'll don't regex fox because that while involved backtracking involved reservation there's over They'RE admission 1999 regex counting keep $1,234.56 for engine it 'single' regex the that body v1.2.3 WON'T once It's 42 tiktoken's written no 'single' piece fox and before it tiktoken's a caf\\u00e9 v1.2.3 user@example.com quick user@example.com scanned admission the scanner on is over reservation request \\u6771\\u4eac we'll 3.14159 with before \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking 3.14159 counting reservation don't way with WON'T involved before It's admission keep C++ written with counting 'single' brown the 100% They'RE 42 before while every and \\\"quotes\\\" for They'RE there's 42 tokens v1.2.3 https://example.com/a/b?c=d before involved there's body once\"}, {\"role\": \"assistant\", \"content\": \"Z\\u00fcrich backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 the 3.14159 involved \\\"quotes\\\" 42 node.js once scanner mirrors {braces} na\\u00efve I'M brown the admission over They'RE piece it faster so request don't over way na\\u00efve (parens) once mirrors 100% don't scanner Z\\u00fcrich hand budget fox there's there's boundaries Z\\u00fcrich don't \\u6771\\u4eac scanner 42 v1.2.3 \\\"quotes\\\" na\\u00efve 3.14159 100% every we'll 3.14159 quick 1999 backtracking {braces} don't lazy dog lazy a dog the tokens with body that quick brown tokens lazy \\u6771\\u4eac piece piece every dog every 1999 \\\"quotes\\\" mirrors faster fox the 3.14159 way piece boundaries user@example.com there's caf\\u00e9 faster on F# https://example.com/a/b?c=d we'll admission \\u0645\\u0631\\u062d\\u0628\\u0627 that written piece scanned counting and admission we'll Z\\u00fcrich dog 3.14159 node.js every engine Z\\u00fcrich body involved backtracking mirrors 100% 100%\"}, {\"role\": \"user\", \"content\": \"(parens) that on 3.14159 \\u6771\\u4eac over is 42 \\\"quotes\\\" 'single' body faster exactly mirrors F# way (parens) no exactly mirrors that \\ud83d\\ude42 $1,234.56 faster brown Z\\u00fcrich a brown tiktoken's \\ud83d\\ude42 tokens regex while They'RE I'M budget node.js https://example.com/a/b?c=d 42 tiktoken's https://example.com/a/b?c=d over over every 'single' 100% tiktoken's scanner gateway for don't tiktoken's a hand v1.2.3 {braces} fox that quick every that fox faster exactly 'single' (parens) v1.2.3 request dog over no is jumps for Z\\u00fcrich WON'T so body once scanner 3.14159 every that \\u6771\\u4eac so\"}, {\"role\": \"assistant\", \"content\": \"no while faster the because there's \\\"quotes\\\" piece tiktoken's a reservation Z\\u00fcrich because caf\\u00e9 gateway tokens gateway over Z\\u00fcrich They'RE that 3.14159 quick They'RE na\\u00efve mirrors faster a 100% reservation I'M on F# lazy 1999 and budget and $1,234.56 exactly budget written fox 100% body and there's don't once body the on while budget \\ud83d\\ude42 100% \\\"quotes\\\" request \\u0645\\u0631\\u062d\\u0628\\u0627 keep lazy 1999 before piece that before involved \\ud83d\\ude42 written don't gateway the once user@example.com (parens) that backtracking budget request involved the v1.2.3 42 hand tokens mirrors I'M written \\u6771\\u4eac \\\"quotes\\\" body [brackets] written 'single' brown so request 3.14159 node.js budget node.js dog (parens) scanner {braces} a WON'T v1.2.3 because there's that engine tiktoken's every no over before on mirrors mirrors don't \\u0645\\u0631\\u062d\\u0628\\u0627 budget for quick 42 quick over lazy over v1.2.3 gateway \\u0645\\u0631\\u062d\\u0628\\u0627 and\"}, {\"role\": \"user\", \"content\": \"that caf\\u00e9 user@example.com hand over we'll no that Z\\u00fcrich brown 42 so fox counting quick every regex https://example.com/a/b?c=d brown once body 'single' reservation v1.2.3 $1,234.56 I'M that on \\ud83d\\ude42 F# scanner faster over F# we'll dog because mirrors \\ud83d\\ude42 we'll scanned regex budget on and I'M admission is written It's fox na\\u00efve with WON'T involved 'single' user@example.com WON'T the [brackets] exactly 42 'single' {braces} involved user@example.com They'RE written before keep tokens dog It's over on \\ud83d\\ude42 a that 1999 reservation I'M for that fox it boundaries no hand for $1,234.56 They'RE jumps hand WON'T https://example.com/a/b?c=d faster over [brackets] 3.14159 Z\\u00fcrich\"}, {\"role\": \"assistant\", \"content\": \"100% the quick scanner is with node.js [brackets] 100% keep scanned because brown every that $1,234.56 \\\"quotes\\\" because 3.14159 it 42 tiktoken's because once 1999 user@example.com node.js the admission it quick scanner involved It's $1,234.56 user@example.com 1999 3.14159 exactly https://example.com/a/b?c=d na\\u00efve dog counting exactly quick v1.2.3 piece involved once v1.2.3 exactly 42 scanner gateway (parens) request \\u6771\\u4eac fox keep 42 node.js while keep with the gateway with F# over no involved caf\\u00e9 [brackets] request They'RE scanner on with v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 every the 42 It's every jumps a scanned involved the brown scanned quick the tokens I'M I'M over no over faster v1.2.3 caf\\u00e9 \\\"quotes\\\" body request that fox fox before node.js brown dog a dog body engine faster Z\\u00fcrich https://example.com/a/b?c=d before I'M we'll hand WON'T admission and counting while on quick and\"}, {\"role\": \"user\", \"content\": \"na\\u00efve \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 mirrors mirrors while tiktoken's tiktoken's hand Z\\u00fcrich there's backtracking brown brown \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors is (parens) it that tokens v1.2.3 way admission that caf\\u00e9 so exactly engine with scanner budget WON'T \\u6771\\u4eac 'single' hand na\\u00efve there's 'single' \\u6771\\u4eac piece na\\u00efve $1,234.56 is before faster admission request \\ud83d\\ude42 \\u6771\\u4eac counting backtracking written \\\"quotes\\\" there's involved lazy quick faster that request scanned v1.2.3 $1,234.56 every dog with for regex once with [brackets] v1.2.3 so caf\\u00e9 It's written it Z\\u00fcrich lazy caf\\u00e9 scanner counting there's no caf\\u00e9 every F# 100% the C++ every faster we'll on tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 engine $1,234.56 faster faster 1999 that exactly mirrors for admission budget 'single' 3.14159 no engine\"}, {\"role\": \"assistant\", \"content\": \"https://example.com/a/b?c=d there's gateway mirrors body mirrors written for brown node.js for brown node.js 'single' scanner na\\u00efve request is every 100% user@example.com dog 3.14159 no we'll that Z\\u00fcrich don't involved with https://example.com/a/b?c=d fox dog F# reservation C++ a v1.2.3 v1.2.3 backtracking WON'T F# no piece C++ {braces} a mirrors way $1,234.56 that WON'T body They'RE no we'll jumps F# once caf\\u00e9 reservation WON'T that piece 1999 They'RE counting piece They'RE request tiktoken's a engine na\\u00efve scanned body lazy faster It's over It's that tokens Z\\u00fcrich C++ scanner with \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"user\", \"content\": \"scanner while tiktoken's fox on boundaries tiktoken's tokens before tiktoken's it body gateway and over (parens) the (parens) 'single' on C++ faster budget every request once on exactly every that it Z\\u00fcrich quick regex and user@example.com jumps it user@example.com quick 100% it exactly every 'single' mirrors body jumps counting caf\\u00e9 tokens admission that piece scanner {braces} tokens is \\ud83d\\ude42 request admission C++ is node.js body for body They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking WON'T\"}, {\"role\": \"assistant\", \"content\": \"dog C++ (parens) It's with 'single' we'll [brackets] request is there's 3.14159 scanned 100% for request on admission $1,234.56 over there's scanned engine scanner They'RE that lazy \\u6771\\u4eac before budget user@example.com once C++ $1,234.56 that scanner exactly tokens is engine \\ud83d\\ude42 dog user@example.com na\\u00efve They'RE we'll \\ud83d\\ude42 that 'single' jumps the Z\\u00fcrich on jumps involved hand involved tiktoken's It's C++ \\ud83d\\ude42 user@example.com that before user@example.com $1,234.56 don't once scanner and a user@example.com user@example.com exactly and user@example.com $1,234.56 admission backtracking budget written 100% we'll 1999 scanned mirrors every \\u0645\\u0631\\u062d\\u0628\\u0627 a brown dog mirrors reservation 1999 regex hand budget \\u6771\\u4eac user@example.com it before 'single' hand over with {braces} no the a keep quick tiktoken's 3.14159 caf\\u00e9 every Z\\u00fcrich F# \\u6771\\u4eac the I'M Z\\u00fcrich na\\u00efve no way gateway na\\u00efve https://example.com/a/b?c=d request {braces} backtracking\"}, {\"role\": \"user\", \"content\": \"\\\"quotes\\\" \\\"quotes\\\" scanned is lazy user@example.com v1.2.3 regex a It's tokens piece They'RE engine over Z\\u00fcrich 3.14159 {braces} 'single' (parens) a boundaries node.js request jumps body before for hand dog boundaries budget on the boundaries we'll every because {braces} Z\\u00fcrich way https://example.com/a/b?c=d 1999 body Z\\u00fcrich there's budget 1999 before so v1.2.3 WON'T user@example.com F# it written lazy with once dog faster faster once we'll \\ud83d\\ude42 exactly user@example.com faster 42 \\u0645\\u0631\\u062d\\u0628\\u0627 a admission over \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T lazy\"}, {\"role\": \"assistant\", \"content\": \"na\\u00efve the (parens) for the brown https://example.com/a/b?c=d and 42 Z\\u00fcrich on {braces} \\ud83d\\ude42 regex over regex na\\u00efve F# over C++ before with the for don't 'single' counting \\u6771\\u4eac tiktoken's before mirrors fox gateway faster so regex \\ud83d\\ude42 keep $1,234.56 reservation so \\ud83d\\ude42 42 https://example.com/a/b?c=d tokens user@example.com for the for exactly \\\"quotes\\\" it written for F# dog I'M engine faster \\u6771\\u4eac gateway dog na\\u00efve is {braces} tokens over request backtracking 1999 body request written 'single' They'RE tokens F# tokens brown C++ boundaries admission every budget we'll no budget it keep don't https://example.com/a/b?c=d every exactly caf\\u00e9 'single' there's C++ 'single' dog v1.2.3 backtracking They'RE engine (parens) 'single' gateway every it $1,234.56 1999 lazy once reservation over fox that {braces} 1999 budget node.js written C++ way [brackets] before \\ud83d\\ude42 gateway on tiktoken's and (parens) the written \\u6771\\u4eac\"}, {\"role\": \"user\", \"content\": \"user@example.com scanned 42 user@example.com piece backtracking every engine that \\u0645\\u0631\\u062d\\u0628\\u0627 request $1,234.56 lazy Z\\u00fcrich mirrors fox no that so It's {braces} It's piece mirrors v1.2.3 regex quick counting for They'RE F# 3.14159 once brown no that no \\u6771\\u4eac so 100% scanner 1999 counting fox there's once engine caf\\u00e9 brown [brackets] piece hand mirrors budget lazy over the counting fox counting is for I'M faster a with node.js we'll \\u6771\\u4eac it 1999 user@example.com that It's mirrors written lazy lazy the before faster piece na\\u00efve fox piece 3.14159 fox na\\u00efve dog regex [brackets] [brackets] \\u6771\\u4eac over body user@example.com 42 node.js while and https://example.com/a/b?c=d na\\u00efve piece faster brown (parens) \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 \\\"quotes\\\"\"}, {\"role\": \"assistant\", \"content\": \"faster written 100% $1,234.56 engine 'single' gateway faster the boundaries exactly on C++ so so mirrors once backtracking caf\\u00e9 quick 1999 while we'll WON'T a Z\\u00fcrich boundaries involved piece scanner {braces} written 1999 and 3.14159 there's \\ud83d\\ude42 that the because quick scanner 3.14159 that it way counting na\\u00efve reservation for before $1,234.56 over quick engine scanner no the we'll scanned backtracking that no don't before 1999 \\u6771\\u4eac we'll before 3.14159 $1,234.56 budget counting tokens jumps while before tokens node.js tokens gateway scanner that \\ud83d\\ude42 3.14159 'single' WON'T so way (parens) backtracking \\\"quotes\\\" the v1.2.3 {braces} body body C++ it 'single'\"}, {\"role\": \"user\", \"content\": \"because request that that dog counting 100% mirrors I'M fox there's the piece with budget regex while before tokens with over there's \\u0645\\u0631\\u062d\\u0628\\u0627 reservation body on reservation tiktoken's and \\u0645\\u0631\\u062d\\u0628\\u0627 while so quick that budget quick brown and user@example.com \\u6771\\u4eac 100% dog keep there's It's WON'T caf\\u00e9 quick there's $1,234.56 way user@example.com budget for v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 quick way boundaries is reservation 42 no that \\ud83d\\ude42 the hand \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) na\\u00efve $1,234.56 3.14159 quick 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors na\\u00efve hand once every F# the \\ud83d\\ude42 once piece I'M 42 with involved the {braces} [brackets] a 'single' caf\\u00e9 that once before once 'single' counting {braces} every reservation budget jumps WON'T\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 {braces} it They'RE scanned for \\u6771\\u4eac hand mirrors written 42 with node.js quick admission I'M counting \\u0645\\u0631\\u062d\\u0628\\u0627 brown \\u6771\\u4eac 100% over \\u6771\\u4eac that \\ud83d\\ude42 \\u6771\\u4eac It's \\ud83d\\ude42 scanner budget so dog that regex keep hand piece lazy no is user@example.com counting tiktoken's request admission we'll mirrors hand gateway boundaries no https://example.com/a/b?c=d on engine reservation no that {braces} mirrors brown Z\\u00fcrich admission piece user@example.com scanned and with 100% the tiktoken's fox WON'T written\"}, {\"role\": \"user\", \"content\": \"Z\\u00fcrich is over 'single' C++ It's gateway [brackets] no over way quick admission no faster written node.js 42 v1.2.3 involved {braces} 3.14159 because [brackets] reservation body before the exactly every [brackets] tokens so fox piece boundaries boundaries v1.2.3 caf\\u00e9 I'M F# 1999 backtracking user@example.com with with don't body a C++ brown don't It's caf\\u00e9 regex They'RE way https://example.com/a/b?c=d scanned 'single' boundaries that reservation \\ud83d\\ude42 brown fox node.js backtracking gateway na\\u00efve (parens)\"}, {\"role\": \"assistant\", \"content\": \"piece WON'T body it $1,234.56 that 3.14159 1999 it reservation \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's boundaries 'single' It's reservation gateway C++ body budget mirrors is the backtracking and \\u0645\\u0631\\u062d\\u0628\\u0627 gateway a 'single' is and while 42 hand it They'RE caf\\u00e9 backtracking counting gateway exactly F# because na\\u00efve scanned that C++ is 100% involved admission piece 3.14159 quick it the lazy involved WON'T a before It's tokens They'RE it mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 reservation it \\ud83d\\ude42 na\\u00efve so is a and that request while that written body keep reservation the on every They'RE WON'T budget scanned reservation $1,234.56 fox is there's $1,234.56 brown involved admission\"}, {\"role\": \"user\", \"content\": \"way F# don't brown v1.2.3 the regex na\\u00efve involved brown budget because dog that WON'T and v1.2.3 for it counting because because because brown \\u6771\\u4eac backtracking (parens) I'M that dog reservation keep that scanner request WON'T They'RE backtracking admission every 42 because 100% [brackets] https://example.com/a/b?c=d admission way the so don't I'M \\ud83d\\ude42 3.14159 that C++ 3.14159 https://example.com/a/b?c=d na\\u00efve \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com node.js way I'M over https://example.com/a/b?c=d engine once mirrors Z\\u00fcrich F# fox on \\u0645\\u0631\\u062d\\u0628\\u0627 na\\u00efve mirrors piece with the \\ud83d\\ude42 WON'T counting involved don't written boundaries gateway keep 1999 boundaries so Z\\u00fcrich way piece every user@example.com exactly They'RE piece boundaries written brown I'M C++ gateway keep the tiktoken's reservation counting Z\\u00fcrich a involved\"}, {\"role\": \"assistant\", \"content\": \"way is a for scanned dog exactly don't body na\\u00efve [brackets] fox na\\u00efve {braces} that body for scanner reservation user@example.com quick fox hand faster reservation mirrors (parens) counting no we'll a exactly reservation user@example.com tiktoken's hand tokens don't C++ body na\\u00efve hand with backtracking for engine for They'RE admission {braces} fox boundaries keep quick it with once body the body the budget \\\"quotes\\\" caf\\u00e9 a admission $1,234.56 we'll way v1.2.3 3.14159 reservation fox WON'T 3.14159 once that written Z\\u00fcrich for na\\u00efve once piece tiktoken's over engine \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 F# 3.14159 lazy \\ud83d\\ude42 user@example.com na\\u00efve \\u6771\\u4eac don't because 'single' They'RE \\ud83d\\ude42 \\u6771\\u4eac body because lazy a for exactly request tiktoken's gateway no C++ before with every {braces} keep v1.2.3 faster mirrors (parens) (parens) and quick scanned I'M budget engine with\"}, {\"role\": \"user\", \"content\": \"regex with exactly \\u0645\\u0631\\u062d\\u0628\\u0627 dog budget that 42 {braces} tokens https://example.com/a/b?c=d tiktoken's that I'M we'll 100% engine counting F# no regex node.js no boundaries way tiktoken's because \\\"quotes\\\" exactly 3.14159 and [brackets] mirrors written keep that F# backtracking scanned They'RE it gateway over {braces} node.js 100% budget scanner budget \\\"quotes\\\" we'll v1.2.3 we'll caf\\u00e9 \\ud83d\\ude42 tokens don't once because mirrors backtracking na\\u00efve every faster once budget caf\\u00e9 it I'M caf\\u00e9 a backtracking regex fox while engine faster scanned I'M tiktoken's don't \\\"quotes\\\" 3.14159 scanner It's na\\u00efve request boundaries hand lazy no $1,234.56 every for jumps written 1999 mirrors {braces} WON'T lazy engine and\"}, {\"role\": \"assistant\", \"content\": \"and reservation caf\\u00e9 it (parens) before piece backtracking and WON'T brown counting that 1999 $1,234.56 tokens user@example.com user@example.com it faster It's the body and request brown caf\\u00e9 keep scanner \\u0645\\u0631\\u062d\\u0628\\u0627 for tokens caf\\u00e9 'single' request boundaries 'single' request They'RE user@example.com tokens quick there's keep exactly a scanned is regex it hand quick hand {braces} so 'single' gateway and\"}, {\"role\": \"user\", \"content\": \"na\\u00efve is request that It's the way caf\\u00e9 the 42 there's quick user@example.com the over that $1,234.56 with user@example.com quick They'RE budget the it F# gateway It's fox we'll involved tiktoken's na\\u00efve over while https://example.com/a/b?c=d that na\\u00efve tiktoken's 42 scanned we'll na\\u00efve and tiktoken's (parens) piece 'single' exactly Z\\u00fcrich budget WON'T (parens) It's engine we'll body involved and keep\"}, {\"role\": \"assistant\", \"content\": \"way v1.2.3 it so once before \\\"quotes\\\" WON'T WON'T written and over F# the with no boundaries v1.2.3 body that dog exactly Z\\u00fcrich backtracking there's on \\u6771\\u4eac mirrors request on They'RE hand regex engine reservation it scanned \\\"quotes\\\" hand boundaries for reservation before brown WON'T once hand \\u0645\\u0631\\u062d\\u0628\\u0627 body 42 \\u0645\\u0631\\u062d\\u0628\\u0627 once every admission body $1,234.56 before $1,234.56 WON'T {braces} na\\u00efve mirrors before the that 3.14159 way brown It's 1999 gateway reservation F# for user@example.com exactly jumps https://example.com/a/b?c=d is and node.js 100% boundaries v1.2.3 'single' we'll no gateway They'RE with F# [brackets] C++ mirrors for tokens written every\"}, {\"role\": \"user\", \"content\": \"WON'T na\\u00efve 'single' It's 3.14159 \\ud83d\\ude42 don't 'single' \\\"quotes\\\" involved C++ caf\\u00e9 with mirrors 100% It's that no user@example.com WON'T before is gateway once node.js gateway while on quick \\ud83d\\ude42 boundaries \\u6771\\u4eac budget 1999 admission They'RE mirrors engine I'M is \\u0645\\u0631\\u062d\\u0628\\u0627 I'M budget for 'single' on \\u6771\\u4eac with WON'T and because brown regex hand reservation scanner 100% 100% user@example.com there's it\"}, {\"role\": \"assistant\", \"content\": \"there's exactly [brackets] faster budget while $1,234.56 (parens) engine caf\\u00e9 They'RE no 100% for because $1,234.56 regex Z\\u00fcrich https://example.com/a/b?c=d tokens jumps because counting brown user@example.com request lazy \\u0645\\u0631\\u062d\\u0628\\u0627 request scanned WON'T $1,234.56 exactly brown every mirrors with 'single' engine 42 \\ud83d\\ude42 engine reservation hand lazy 'single' counting hand (parens) gateway that counting user@example.com 42 fox request \\\"quotes\\\" hand lazy 100% 42 I'M It's WON'T https://example.com/a/b?c=d engine while\"}, {\"role\": \"user\", \"content\": \"C++ \\\"quotes\\\" we'll gateway \\ud83d\\ude42 counting written piece na\\u00efve \\u6771\\u4eac WON'T over caf\\u00e9 backtracking engine https://example.com/a/b?c=d the v1.2.3 that so quick counting na\\u00efve \\u6771\\u4eac don't WON'T It's we'll that the 100% request I'M brown na\\u00efve tokens engine budget before the scanned mirrors dog engine {braces} {braces} 42 don't WON'T It's piece \\\"quotes\\\" that no quick 3.14159 F# once fox don't for scanner \\u6771\\u4eac it brown backtracking tokens $1,234.56 reservation written They'RE the \\\"quotes\\\" caf\\u00e9 backtracking C++ reservation\"}, {\"role\": \"assistant\", \"content\": \"regex (parens) Z\\u00fcrich engine https://example.com/a/b?c=d written tiktoken's C++ dog It's lazy dog exactly with it that admission way no scanner a 3.14159 on They'RE It's involved caf\\u00e9 brown on dog that https://example.com/a/b?c=d don't \\u6771\\u4eac jumps \\ud83d\\ude42 and \\u6771\\u4eac so dog so don't (parens) scanned engine budget the exactly F# tokens over \\u0645\\u0631\\u062d\\u0628\\u0627 jumps backtracking 3.14159 budget 100% boundaries gateway backtracking don't scanner dog with [brackets] gateway 3.14159 caf\\u00e9 faster budget na\\u00efve caf\\u00e9 way for every reservation backtracking request They'RE It's \\u6771\\u4eac written\"}, {\"role\": \"user\", \"content\": \"42 1999 so boundaries $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} budget backtracking [brackets] we'll admission piece node.js admission Z\\u00fcrich fox a involved request is it node.js node.js so \\u6771\\u4eac [brackets] no \\u6771\\u4eac brown node.js quick na\\u00efve scanner C++ user@example.com admission 'single' involved 1999 user@example.com I'M v1.2.3 It's 100% there's gateway F# node.js gateway scanned that caf\\u00e9 3.14159 Z\\u00fcrich no admission {braces} tiktoken's over It's I'M 1999 [brackets] body written scanned that scanner on the on with that scanned hand\"}, {\"role\": \"assistant\", \"content\": \"{braces} regex [brackets] piece \\ud83d\\ude42 mirrors there's tiktoken's It's lazy mirrors every v1.2.3 I'M that boundaries 'single' node.js on C++ user@example.com budget user@example.com no user@example.com every engine while a I'M over [brackets] WON'T piece over there's lazy I'M that $1,234.56 budget https://example.com/a/b?c=d it a request exactly 42 keep They'RE caf\\u00e9 tokens \\u0645\\u0631\\u062d\\u0628\\u0627 keep scanner while dog F# mirrors backtracking we'll gateway scanner so we'll request there's budget it lazy scanner for exactly faster we'll reservation scanner no WON'T request (parens) tiktoken's exactly before so while hand involved with $1,234.56 mirrors 3.14159 on the [brackets] boundaries 100% with request way engine keep the https://example.com/a/b?c=d \\u6771\\u4eac the WON'T that so\"}, {\"role\": \"user\", \"content\": \"body hand backtracking (parens) counting na\\u00efve They'RE \\ud83d\\ude42 WON'T boundaries v1.2.3 budget na\\u00efve admission once the that because tokens dog tiktoken's scanner scanner (parens) I'M \\\"quotes\\\" budget 100% there's once WON'T keep {braces} the before fox (parens) the counting because lazy 1999 \\u0645\\u0631\\u062d\\u0628\\u0627 admission 1999 and there's while admission 3.14159 a counting C++ with exactly don't don't lazy \\ud83d\\ude42 42 \\\"quotes\\\" backtracking C++ there's keep user@example.com request is node.js piece before brown budget 42 100% because admission and with tiktoken's no the 100%\"}, {\"role\": \"assistant\", \"content\": \"with 1999 Z\\u00fcrich that is with mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 the brown $1,234.56 way scanner https://example.com/a/b?c=d lazy written scanner quick v1.2.3 \\\"quotes\\\" regex \\\"quotes\\\" They'RE 3.14159 \\ud83d\\ude42 [brackets] Z\\u00fcrich the counting for so $1,234.56 It's WON'T the quick hand that mirrors the 42 caf\\u00e9 exactly node.js we'll backtracking while I'M and fox is before written it 1999 no for involved \\u0645\\u0631\\u062d\\u0628\\u0627 the \\\"quotes\\\" piece the way involved that no 100% before there's 100% we'll is [brackets] C++ the we'll the node.js reservation is while [brackets] engine scanner we'll no C++ don't dog v1.2.3 that so tiktoken's before \\ud83d\\ude42 mirrors 'single' it once caf\\u00e9 there's backtracking gateway body over quick way and that once \\u6771\\u4eac 'single' we'll (parens) and before for quick They'RE 1999 regex there's \\ud83d\\ude42\"}, {\"role\": \"user\", \"content\": \"engine They'RE https://example.com/a/b?c=d we'll quick tokens written WON'T don't regex backtracking It's it lazy 1999 brown budget on [brackets] node.js {braces} They'RE involved a mirrors exactly tokens with $1,234.56 F# 42 backtracking involved engine It's and {braces} \\u6771\\u4eac Z\\u00fcrich it it [brackets] gateway keep na\\u00efve request scanned don't regex we'll C++ counting tokens so request while scanned tokens scanner for v1.2.3 with tokens \\ud83d\\ude42 They'RE na\\u00efve 'single' keep node.js v1.2.3 'single' it way gateway It's keep reservation backtracking body fox with that [brackets] the and 1999 request no engine counting is [brackets] lazy on request quick involved \\ud83d\\ude42 brown v1.2.3 involved fox exactly hand {braces} we'll every regex\"}, {\"role\": \"assistant\", \"content\": \"(parens) faster tiktoken's with the every mirrors boundaries that caf\\u00e9 backtracking involved once 42 \\u6771\\u4eac Z\\u00fcrich {braces} boundaries that [brackets] 1999 it admission before a jumps and 3.14159 way scanner don't {braces} budget for a 3.14159 mirrors boundaries while caf\\u00e9 counting \\u6771\\u4eac a written tiktoken's and we'll https://example.com/a/b?c=d jumps while so [brackets] F# 3.14159 mirrors that {braces} request request I'M keep the over hand engine keep node.js dog na\\u00efve [brackets] every body $1,234.56 \\\"quotes\\\" $1,234.56 we'll \\u0645\\u0631\\u062d\\u0628\\u0627 that 42 budget scanned it They'RE no with the {braces} we'll tiktoken's a \\ud83d\\ude42 tiktoken's https://example.com/a/b?c=d It's exactly It's for over that 100% a \\\"quotes\\\" caf\\u00e9 no 3.14159 hand gateway quick budget over user@example.com I'M\"}, {\"role\": \"user\", \"content\": \"mirrors engine so Z\\u00fcrich and reservation caf\\u00e9 1999 na\\u00efve exactly involved caf\\u00e9 and 1999 over tiktoken's user@example.com over once that It's jumps involved C++ and is node.js brown gateway 100% 100% keep hand backtracking 3.14159 involved hand 'single' once because budget They'RE brown 3.14159 every faster request it regex while \\ud83d\\ude42 1999 exactly way we'll 'single' request scanner boundaries on (parens) we'll\"}, {\"role\": \"assistant\", \"content\": \"once 1999 fox that Z\\u00fcrich \\u6771\\u4eac that we'll 42 'single' dog \\u6771\\u4eac mirrors (parens) because na\\u00efve every and piece C++ 3.14159 https://example.com/a/b?c=d 42 once so gateway is scanned no user@example.com node.js caf\\u00e9 faster node.js tiktoken's It's F# brown it admission body don't don't before scanner before F# 3.14159 na\\u00efve backtracking user@example.com F# that (parens) (parens) there's exactly a reservation it before backtracking It's piece tiktoken's lazy written is {braces} 1999 scanned keep body because mirrors no written that piece boundaries request once scanner \\\"quotes\\\" with engine hand v1.2.3 we'll v1.2.3 the we'll $1,234.56 the quick and tokens backtracking is quick 'single' 42 once reservation admission fox scanner a scanner engine 1999 quick [brackets] \\u6771\\u4eac budget for quick engine keep user@example.com 100% C++ [brackets] hand on Z\\u00fcrich hand Z\\u00fcrich regex 1999\"}, {\"role\": \"user\", \"content\": \"\\u0645\\u0631\\u062d\\u0628\\u0627 scanner They'RE budget C++ exactly {braces} is \\ud83d\\ude42 because way body brown counting with regex it \\\"quotes\\\" we'll body node.js over a node.js that while it v1.2.3 3.14159 quick request that dog node.js They'RE 1999 that gateway 3.14159 gateway there's backtracking lazy C++ admission node.js 42 request They'RE we'll gateway so once 1999 brown na\\u00efve 3.14159 budget faster \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" backtracking that 100% because a fox once involved on involved v1.2.3 WON'T mirrors 'single' a \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" is They'RE that involved reservation They'RE (parens) v1.2.3 Z\\u00fcrich WON'T Z\\u00fcrich fox that don't boundaries engine regex don't exactly na\\u00efve don't caf\\u00e9 request budget [brackets] F# tokens exactly It's request it\"}, {\"role\": \"assistant\", \"content\": \"scanned (parens) reservation {braces} engine tokens v1.2.3 node.js scanner {braces} It's with admission the 100% F# \\u0645\\u0631\\u062d\\u0628\\u0627 body na\\u00efve with 'single' boundaries regex (parens) because mirrors there's keep once lazy it brown involved boundaries request engine before on It's \\ud83d\\ude42 na\\u00efve fox WON'T we'll backtracking jumps exactly scanner it admission mirrors don't that \\u6771\\u4eac faster body mirrors reservation (parens) \\\"quotes\\\" regex mirrors keep exactly They'RE 3.14159 there's a \\u0645\\u0631\\u062d\\u0628\\u0627 and with admission written C++ https://example.com/a/b?c=d the piece jumps that is lazy tokens before 100% boundaries tokens dog tiktoken's it $1,234.56 I'M 100% once (parens) I'M way brown keep\"}, {\"role\": \"user\", \"content\": \"no no hand F# jumps so faster every don't so tiktoken's 'single' I'M [brackets] \\\"quotes\\\" v1.2.3 over admission exactly scanner node.js 100% quick gateway \\ud83d\\ude42 faster budget over faster over and engine request faster fox 'single' backtracking body brown no na\\u00efve is it \\\"quotes\\\" piece with \\ud83d\\ude42 for tokens I'M scanned while admission way faster na\\u00efve mirrors tokens boundaries \\u6771\\u4eac jumps tiktoken's it there's faster\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 no It's piece request \\u6771\\u4eac Z\\u00fcrich don't \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE 3.14159 quick Z\\u00fcrich once tiktoken's with engine jumps over I'M because caf\\u00e9 faster tiktoken's that 'single' we'll dog 42 \\u6771\\u4eac way once counting it C++ on that the that backtracking we'll scanned before 100% na\\u00efve request dog every request 1999 request quick over 1999 gateway https://example.com/a/b?c=d because 3.14159 that reservation mirrors that a over I'M with regex Z\\u00fcrich mirrors piece regex caf\\u00e9 engine Z\\u00fcrich the written v1.2.3 \"}, {\"role\": \"user\", \"content\": \"hand is It's scanned regex 1999 way on It's we'll tokens with reservation while way mirrors regex no over 42 is user@example.com scanner scanned keep budget {braces} brown budget while is involved [brackets] body jumps F# dog no 3.14159 exactly over 100% (parens) It's jumps so we'll \\u6771\\u4eac faster admission exactly admission that for tiktoken's 100% once engine and every a before budget regex They'RE counting\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 involved backtracking regex na\\u00efve way tokens we'll 100% gateway the gateway the body lazy I'M budget is regex body (parens) C++ that body no the jumps [brackets] exactly 1999 budget request brown \\ud83d\\ude42 jumps keep scanned user@example.com for exactly regex dog node.js scanner that node.js 42 engine fox \\ud83d\\ude42 tiktoken's WON'T mirrors written faster while 42 100% \\u6771\\u4eac user@example.com dog engine lazy [brackets] exactly quick node.js written written 100% every \\ud83d\\ude42 body 3.14159 hand because once gateway faster before with It's a 1999 42 boundaries jumps that {braces} 100% caf\\u00e9 a They'RE 3.14159 [brackets] reservation 1999 faster \\ud83d\\ude42 hand 1999 [brackets] scanner\"}, {\"role\": \"user\", \"content\": \"https://example.com/a/b?c=d dog for C++ v1.2.3 because every 100% \\u6771\\u4eac and brown is request piece boundaries on before because that body WON'T [brackets] that keep tokens lazy WON'T keep on for WON'T body quick {braces} gateway body written tokens quick the mirrors is is with I'M 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 don't \\u0645\\u0631\\u062d\\u0628\\u0627 piece involved user@example.com written tiktoken's fox written and no 1999 don't on because counting \\\"quotes\\\" They'RE 42 a gateway \\u6771\\u4eac scanner admission user@example.com 3.14159 there's on over 'single' written that a {braces} that because gateway the\"}, {\"role\": \"assistant\", \"content\": \"WON'T backtracking \\ud83d\\ude42 node.js 3.14159 keep it \\u0645\\u0631\\u062d\\u0628\\u0627 there's is so user@example.com on there's 3.14159 mirrors involved keep \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's {braces} no engine brown brown there's dog way regex that 'single' for and 3.14159 node.js every there's a every They'RE WON'T exactly scanner tiktoken's They'RE regex the dog counting over the for request scanner with and mirrors dog the counting tokens gateway https://example.com/a/b?c=d for https://example.com/a/b?c=d backtracking\"}, {\"role\": \"user\", \"content\": \"don't because that fox Z\\u00fcrich with Z\\u00fcrich way piece (parens) every (parens) user@example.com {braces} keep so a that boundaries 100% keep C++ node.js 1999 piece $1,234.56 \\ud83d\\ude42 backtracking {braces} that involved gateway boundaries no 42 hand 3.14159 \\ud83d\\ude42 na\\u00efve na\\u00efve that for Z\\u00fcrich $1,234.56 engine lazy fox scanned before once hand no https://example.com/a/b?c=d admission WON'T admission It's no a because \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac exactly written \\u0645\\u0631\\u062d\\u0628\\u0627 faster that is user@example.com v1.2.3 'single' boundaries gateway faster backtracking no engine fox $1,234.56 lazy the $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 that WON'T before I'M scanned\"}, {\"role\": \"assistant\", \"content\": \"it is admission admission brown the They'RE we'll we'll user@example.com [brackets] way tokens $1,234.56 on caf\\u00e9 I'M gateway 3.14159 written 'single' on involved mirrors node.js that fox 'single' quick once \\u6771\\u4eac gateway $1,234.56 no exactly body Z\\u00fcrich that 'single' engine and Z\\u00fcrich don't no so tiktoken's I'M that over hand dog we'll (parens) 'single' \\ud83d\\ude42 while 100% is node.js no quick over while on budget that \\\"quotes\\\" budget It's don't regex on C++ 1999 [brackets] scanned 1999 that there's $1,234.56 way\"}, {\"role\": \"user\", \"content\": \"budget while tiktoken's backtracking gateway and hand no while \\ud83d\\ude42 request node.js we'll and don't while na\\u00efve quick WON'T don't [brackets] quick there's fox that that I'M keep \\ud83d\\ude42 tiktoken's v1.2.3 the body the \\u0645\\u0631\\u062d\\u0628\\u0627 reservation written F# jumps admission boundaries user@example.com while I'M brown lazy It's regex fox user@example.com {braces} is on 'single' written request there's engine exactly tiktoken's WON'T caf\\u00e9 the tokens on and brown exactly scanner with involved regex on v1.2.3 jumps that written F# on exactly before over once body \\u6771\\u4eac \\ud83d\\ude42 lazy that \\ud83d\\ude42\"}, {\"role\": \"assistant\", \"content\": \"don't every there's scanned while They'RE a tokens is scanned mirrors so is request counting once way (parens) {braces} counting 100% \\\"quotes\\\" before boundaries brown tiktoken's that way engine quick tiktoken's over involved 100% before for the C++ It's that node.js [brackets] because https://example.com/a/b?c=d engine 100% budget we'll with tokens 42 budget it dog 1999 user@example.com 42 C++ keep the for a 42 keep no caf\\u00e9 node.js 'single' the for don't It's C++ admission a involved 3.14159 'single' don't They'RE WON'T quick so backtracking because once body \\\"quotes\\\" WON'T user@example.com caf\\u00e9 v1.2.3 engine 'single' scanner hand I'M 'single' $1,234.56 [brackets] {braces} counting it don't there's no that\"}, {\"role\": \"user\", \"content\": \"so jumps dog hand no 100% while request tiktoken's there's that so regex $1,234.56 don't 'single' boundaries \\\"quotes\\\" that 100% the boundaries keep backtracking (parens) no \\u6771\\u4eac fox F# mirrors 3.14159 we'll there's \\\"quotes\\\" scanner backtracking $1,234.56 100% request the WON'T boundaries caf\\u00e9 tiktoken's node.js backtracking counting user@example.com C++ exactly is reservation 100% we'll quick regex \\ud83d\\ude42 'single' I'M request written https://example.com/a/b?c=d boundaries They'RE there's It's piece F# v1.2.3 boundaries [brackets] a 100% that scanner v1.2.3 42 because https://example.com/a/b?c=d [brackets] that I'M once body faster because with engine 'single' scanned on engine the 1999 \\u6771\\u4eac quick written no piece involved boundaries [brackets] counting while I'M Z\\u00fcrich scanner for the written $1,234.56 (parens) that boundaries the because\"}, {\"role\": \"assistant\", \"content\": \"(parens) https://example.com/a/b?c=d 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich Z\\u00fcrich \\u6771\\u4eac involved don't reservation there's we'll \\\"quotes\\\" reservation They'RE tokens because lazy there's a exactly It's jumps every $1,234.56 v1.2.3 exactly 42 a quick https://example.com/a/b?c=d caf\\u00e9 tiktoken's piece reservation every C++ caf\\u00e9 v1.2.3 scanned backtracking na\\u00efve 1999 piece engine while that I'M involved on fox 42 https://example.com/a/b?c=d (parens) tiktoken's for dog once the involved a dog once there's https://example.com/a/b?c=d tokens na\\u00efve (parens) once 1999 over na\\u00efve caf\\u00e9 hand They'RE every tiktoken's involved we'll node.js\"}, {\"role\": \"user\", \"content\": \"that request while jumps faster exactly written 3.14159 scanner that \\\"quotes\\\" with exactly dog that keep with piece piece boundaries no [brackets] backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 F# hand $1,234.56 engine written (parens) and It's involved node.js Z\\u00fcrich boundaries with backtracking na\\u00efve and body counting node.js keep Z\\u00fcrich that while \\u6771\\u4eac because there's \\u6771\\u4eac boundaries no They'RE counting jumps v1.2.3 with don't a \\u0645\\u0631\\u062d\\u0628\\u0627 fox user@example.com no that it F# once it so every 100% a caf\\u00e9 the \\\"quotes\\\" [brackets] that gateway while we'll scanned caf\\u00e9 keep involved \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"the while I'M gateway WON'T tiktoken's 'single' over 'single' request faster (parens) faster faster \\u0645\\u0631\\u062d\\u0628\\u0627 hand tiktoken's keep while v1.2.3 $1,234.56 that written body WON'T way \\u0645\\u0631\\u062d\\u0628\\u0627 brown a F# scanned \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE keep lazy caf\\u00e9 100% 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} we'll exactly the C++ body Z\\u00fcrich Z\\u00fcrich way with reservation mirrors C++ backtracking involved WON'T exactly with involved WON'T way a counting the admission 1999 hand because once They'RE the backtracking 1999 no a 42 brown\"}, {\"role\": \"user\", \"content\": \"node.js so jumps is piece a way on because the that backtracking request the reservation {braces} {braces} the https://example.com/a/b?c=d it tokens reservation the lazy They'RE reservation a 1999 I'M boundaries no with no C++ is way 3.14159 WON'T hand 'single' counting Z\\u00fcrich on F# with 1999 admission every (parens) we'll request mirrors that the 'single' it faster dog \\\"quotes\\\" jumps that faster WON'T faster every 100% involved a there's \\ud83d\\ude42 way way 'single' na\\u00efve no on it (parens) WON'T user@example.com na\\u00efve boundaries mirrors [brackets] it regex regex 42 engine engine and faster \\\"quotes\\\" gateway \\u0645\\u0631\\u062d\\u0628\\u0627 over request $1,234.56 They'RE user@example.com that node.js I'M https://example.com/a/b?c=d involved on (parens) it a that 3.14159 way over no\"}, {\"role\": \"assistant\", \"content\": \"boundaries so budget involved exactly written node.js [brackets] with jumps caf\\u00e9 node.js (parens) [brackets] 100% tiktoken's 42 it that piece (parens) engine user@example.com involved no gateway \\\"quotes\\\" exactly It's budget exactly once It's a involved (parens) budget we'll fox node.js tokens Z\\u00fcrich no body \\\"quotes\\\" na\\u00efve while it is I'M the hand mirrors WON'T the caf\\u00e9 \\u6771\\u4eac mirrors https://example.com/a/b?c=d counting before way because \\\"quotes\\\" a admission lazy dog keep the budget quick mirrors there's [brackets] hand Z\\u00fcrich https://example.com/a/b?c=d mirrors user@example.com 42 regex request scanned \\\"quotes\\\" engine because the It's Z\\u00fcrich F# 42 boundaries there's keep engine \\ud83d\\ude42 I'M They'RE because over $1,234.56 admission for fox because exactly because piece regex on \\u6771\\u4eac 'single' backtracking a\"}, {\"role\": \"user\", \"content\": \"They'RE backtracking \\u6771\\u4eac boundaries mirrors that 100% {braces} 42 gateway boundaries tokens so on counting gateway tiktoken's tiktoken's tokens tiktoken's there's while tokens It's C++ mirrors the tokens budget hand we'll over over [brackets] jumps the \\ud83d\\ude42 way and that boundaries 100% counting reservation there's [brackets] 3.14159 They'RE keep the regex It's while budget request Z\\u00fcrich it https://example.com/a/b?c=d C++ admission quick engine keep quick \\u6771\\u4eac $1,234.56 engine budget hand \\u6771\\u4eac body 100% don't scanned $1,234.56 dog the \\u6771\\u4eac piece over 1999 $1,234.56 $1,234.56 the dog we'll gateway is dog there's fox Z\\u00fcrich over They'RE so a engine 'single' regex and lazy na\\u00efve tokens tokens \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 \\ud83d\\ude42 (parens)\"}, {\"role\": \"assistant\", \"content\": \"\\\"quotes\\\" {braces} so with brown (parens) that the because keep boundaries before because fox exactly every because F# so mirrors way C++ request gateway there's that on request tokens it \\\"quotes\\\" regex written dog scanner budget while it that $1,234.56 scanned every no we'll is because while reservation They'RE don't dog scanner dog WON'T fox no scanned \\ud83d\\ude42 budget no dog request it scanned is hand over no 3.14159 so regex backtracking exactly backtracking the 42 100% faster quick scanner before 'single' reservation engine it we'll caf\\u00e9 every quick for no faster 100% $1,234.56 on Z\\u00fcrich written keep\"}, {\"role\": \"user\", \"content\": \"on tokens don't F# counting keep exactly every \\ud83d\\ude42 keep Z\\u00fcrich for fox counting no caf\\u00e9 reservation na\\u00efve mirrors They'RE It's F# the that piece \\\"quotes\\\" \\u6771\\u4eac \\u6771\\u4eac engine no $1,234.56 \\\"quotes\\\" because dog 100% na\\u00efve user@example.com tokens [brackets] no with that boundaries I'M Z\\u00fcrich before (parens) while and scanned gateway user@example.com faster request boundaries \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d written C++ I'M while quick [brackets] before 1999 and They'RE C++ v1.2.3 for no {braces} request C++ They'RE over WON'T on caf\\u00e9 {braces} 100% the body quick keep C++ \\ud83d\\ude42 tokens scanned admission keep hand a node.js WON'T keep 100% $1,234.56 hand $1,234.56 (parens) 1999 because the \\ud83d\\ude42 is 42 body Z\\u00fcrich way [brackets] it once that dog C++ jumps fox tiktoken's piece \\\"quotes\\\" v1.2.3 exactly for 42 \\ud83d\\ude42 'single' na\\u00efve admission \\u6771\\u4eac They'RE fox mirrors regex\"}, {\"role\": \"assistant\", \"content\": \"They'RE scanner tiktoken's quick (parens) https://example.com/a/b?c=d budget quick for engine (parens) hand tiktoken's piece written https://example.com/a/b?c=d counting counting over no once engine I'M because fox exactly regex no once keep and written regex faster 100% the admission counting a scanner C++ quick involved gateway brown piece Z\\u00fcrich regex written is faster \\u6771\\u4eac body brown no tokens scanned every request it I'M \\ud83d\\ude42 for (parens) Z\\u00fcrich written 3.14159 'single' I'M no na\\u00efve quick regex gateway it counting (parens) the over so hand gateway is body way dog v1.2.3 faster so request while before counting body F# faster [brackets] engine involved a $1,234.56 dog that that budget mirrors user@example.com the brown that no C++ it every na\\u00efve exactly that body Z\\u00fcrich \\u6771\\u4eac with don't so It's \\u0645\\u0631\\u062d\\u0628\\u0627 that gateway a the\"}, {\"role\": \"user\", \"content\": \"dog is once budget \\ud83d\\ude42 quick regex fox (parens) [brackets] gateway \\ud83d\\ude42 no regex because lazy request with that written we'll It's that request user@example.com is \\\"quotes\\\" \\\"quotes\\\" faster the don't quick node.js \\u0645\\u0631\\u062d\\u0628\\u0627 faster na\\u00efve They'RE the 100% written that the 100% backtracking \\ud83d\\ude42 written the fox scanner na\\u00efve it piece It's no so It's before na\\u00efve gateway there's because gateway reservation \\ud83d\\ude42 quick involved \\u0645\\u0631\\u062d\\u0628\\u0627 with WON'T request F# scanned exactly F# reservation scanned engine jumps v1.2.3 we'll 'single' it quick WON'T jumps na\\u00efve tokens brown node.js backtracking reservation 100% v1.2.3 once exactly\"}, {\"role\": \"assistant\", \"content\": \"fox user@example.com it (parens) we'll 100% involved dog and brown 'single' exactly mirrors 100% WON'T na\\u00efve 1999 F# faster 'single' tokens tiktoken's admission \\\"quotes\\\" the WON'T request that on we'll I'M piece there's body Z\\u00fcrich boundaries because jumps the They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 boundaries \\u6771\\u4eac the https://example.com/a/b?c=d caf\\u00e9 piece don't brown exactly node.js the budget every so admission that and 100% fox 'single' {braces} 100% $1,234.56 C++ counting WON'T brown don't https://example.com/a/b?c=d boundaries I'M 100% [brackets] lazy quick It's brown \\u6771\\u4eac for is \\u6771\\u4eac for is and request\"}, {\"role\": \"user\", \"content\": \"{braces} counting I'M request fox involved boundaries with WON'T WON'T $1,234.56 is faster don't request https://example.com/a/b?c=d while It's every body the a brown engine 42 for we'll request (parens) involved is scanned backtracking jumps C++ gateway keep scanner for counting 42 reservation hand \\\"quotes\\\" hand C++ every reservation caf\\u00e9 that don't with $1,234.56 there's WON'T {braces} that 3.14159 F# before over exactly caf\\u00e9 budget \\u6771\\u4eac the before and body gateway hand admission It's C++ that F# It's faster no that we'll written (parens) fox mirrors we'll user@example.com no https://example.com/a/b?c=d node.js that brown request\"}, {\"role\": \"assistant\", \"content\": \"I'M admission counting Z\\u00fcrich user@example.com faster a \\ud83d\\ude42 mirrors once lazy is the involved \\ud83d\\ude42 \\u6771\\u4eac every scanned every because budget hand and it jumps there's while there's [brackets] na\\u00efve on \\\"quotes\\\" \\\"quotes\\\" a \\ud83d\\ude42 hand \\u6771\\u4eac we'll and It's with the {braces} piece WON'T 'single' for \\u6771\\u4eac no backtracking lazy fox a [brackets] fox 42 \\ud83d\\ude42 (parens) 1999 https://example.com/a/b?c=d 1999 lazy and 1999 reservation way (parens) 100% it na\\u00efve 100%\"}, {\"role\": \"user\", \"content\": \"regex on scanner once $1,234.56 1999 (parens) gateway budget dog F# before user@example.com once brown jumps 100% $1,234.56 that budget Z\\u00fcrich the reservation because a on \\\"quotes\\\" backtracking hand dog because backtracking a is reservation F# F# brown gateway for the we'll the node.js piece I'M keep tokens once body WON'T there's tokens piece 'single' They'RE node.js so 100% body over a is user@example.com no \\\"quotes\\\" boundaries F# They'RE the C++ backtracking and \\ud83d\\ude42 over WON'T\"}, {\"role\": \"assistant\", \"content\": \"is 'single' scanned \\u6771\\u4eac 3.14159 mirrors {braces} there's scanner every reservation mirrors reservation backtracking involved that tiktoken's {braces} node.js we'll F# jumps v1.2.3 v1.2.3 exactly keep \\u0645\\u0631\\u062d\\u0628\\u0627 1999 scanner with tokens once counting so for https://example.com/a/b?c=d piece the while request Z\\u00fcrich it [brackets] dog involved the involved the admission and (parens) engine F# \\u0645\\u0631\\u062d\\u0628\\u0627 gateway quick \\ud83d\\ude42 v1.2.3 tiktoken's https://example.com/a/b?c=d faster \\ud83d\\ude42 that 42 (parens) jumps caf\\u00e9 so hand budget admission [brackets] gateway brown I'M quick admission C++ C++ caf\\u00e9 \\u6771\\u4eac 'single' user@example.com hand that don't C++ once\"}, {\"role\": \"user\", \"content\": \"there's na\\u00efve reservation na\\u00efve before gateway 100% 1999 request that jumps tiktoken's for admission on quick on scanned boundaries jumps \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors we'll dog because a budget \\u6771\\u4eac it na\\u00efve 1999 that the caf\\u00e9 involved counting jumps tokens with \\u0645\\u0631\\u062d\\u0628\\u0627 involved before the over admission I'M dog v1.2.3 engine tiktoken's we'll scanner every tiktoken's They'RE once It's exactly request jumps node.js WON'T 3.14159 over that exactly engine \\u0645\\u0631\\u062d\\u0628\\u0627 once budget reservation because it 42 keep 'single' boundaries so \\u0645\\u0631\\u062d\\u0628\\u0627 node.js scanner budget 3.14159 engine node.js C++ I'M budget [brackets] a caf\\u00e9 user@example.com jumps tokens every 1999 It's\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 scanner (parens) fox gateway mirrors 42 They'RE reservation we'll fox written brown \\\"quotes\\\" mirrors so 42 hand because budget 100% I'M counting request quick 42 caf\\u00e9 C++ C++ 3.14159 request https://example.com/a/b?c=d jumps jumps scanned brown keep brown and \\ud83d\\ude42 once tokens admission fox F# Z\\u00fcrich user@example.com the gateway tiktoken's https://example.com/a/b?c=d \\\"quotes\\\" tokens \\ud83d\\ude42 \\ud83d\\ude42 [brackets] while node.js [brackets] 1999 on \\\"quotes\\\" caf\\u00e9 1999 jumps 3.14159 request over body admission tiktoken's \\ud83d\\ude42 regex that the once scanner because They'RE regex 'single' admission 42\"}, {\"role\": \"user\", \"content\": \"gateway \\u0645\\u0631\\u062d\\u0628\\u0627 keep quick https://example.com/a/b?c=d F# involved hand It's written request {braces} over mirrors there's gateway every so it caf\\u00e9 involved {braces} quick that gateway over 42 fox lazy involved every keep don't (parens) It's request every for \\u6771\\u4eac admission piece on 3.14159 backtracking written quick gateway and jumps F# (parens) 100% the budget quick keep it 100% \\ud83d\\ude42 for (parens) is \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"assistant\", \"content\": \"way boundaries F# (parens) that It's \\\"quotes\\\" scanned \\ud83d\\ude42 \\u6771\\u4eac that brown regex is over na\\u00efve 3.14159 because Z\\u00fcrich 1999 (parens) \\u0645\\u0631\\u062d\\u0628\\u0627 C++ over way so that user@example.com and user@example.com tiktoken's scanned \\u6771\\u4eac [brackets] engine lazy 1999 involved it 100% a so \\u0645\\u0631\\u062d\\u0628\\u0627 on every body na\\u00efve faster regex engine fox that tokens caf\\u00e9 na\\u00efve 3.14159 and hand engine with\"}, {\"role\": \"user\", \"content\": \"keep there's that every [brackets] $1,234.56 \\\"quotes\\\" on body https://example.com/a/b?c=d involved https://example.com/a/b?c=d tokens scanner 1999 https://example.com/a/b?c=d https://example.com/a/b?c=d node.js 1999 backtracking user@example.com that [brackets] don't body lazy WON'T 42 written hand we'll WON'T no \\u0645\\u0631\\u062d\\u0628\\u0627 $1,234.56 that $1,234.56 way piece https://example.com/a/b?c=d admission admission C++ boundaries we'll reservation scanner the C++ \\u0645\\u0631\\u062d\\u0628\\u0627 hand engine na\\u00efve I'M backtracking while regex \\ud83d\\ude42 budget quick v1.2.3 dog v1.2.3 quick [brackets] fox for node.js [brackets] admission \\ud83d\\ude42 They'RE that on boundaries every that regex 42 a {braces} way every scanned backtracking boundaries budget we'll \\\"quotes\\\" it no caf\\u00e9 scanner jumps backtracking while involved and it 1999 with way {braces} \\\"quotes\\\" 'single'\"}, {\"role\": \"assistant\", \"content\": \"42 is \\\"quotes\\\" budget It's They'RE dog fox exactly fox written regex backtracking written with \\\"quotes\\\" written WON'T that backtracking for a every a engine lazy 42 the once budget on because dog we'll and node.js it $1,234.56 mirrors (parens) It's 42 na\\u00efve it scanned https://example.com/a/b?c=d v1.2.3 C++ 'single' boundaries caf\\u00e9 1999 before once 3.14159 is [brackets] don't tiktoken's scanner fox [brackets] \\u6771\\u4eac over once [brackets] They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 we'll with 42 counting \\u0645\\u0631\\u062d\\u0628\\u0627 C++ 3.14159 it the quick\"}, {\"role\": \"user\", \"content\": \" (parens) with before so is that node.js C++ we'll on admission the \\u6771\\u4eac faster the caf\\u00e9 jumps Z\\u00fcrich (parens) fox \\u6771\\u4eac {braces} every {braces} fox mirrors They'RE They'RE fox counting C++ keep \\\"quotes\\\" 1999 once 100% C++ don't once way dog They'RE once involved exactly \\ud83d\\ude42 we'll piece request reservation quick the don't (parens) scanner I'M \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 Z\\u00fcrich once admission no regex tiktoken's exactly 1999 over engine regex there's involved 42 admission no 'single' over there's (parens) caf\\u00e9 because request keep \\u6771\\u4eac keep \\\"quotes\\\" budget piece piece faster $1,234.56 quick WON'T it piece hand\"}, {\"role\": \"assistant\", \"content\": \"involved scanner WON'T {braces} {braces} It's F# piece it \\\"quotes\\\" I'M because body before quick counting piece na\\u00efve reservation hand WON'T hand the request 1999 involved because {braces} so don't F# and [brackets] faster na\\u00efve way for brown brown hand it Z\\u00fcrich 3.14159 the user@example.com with 1999 while every scanner lazy (parens) tokens [brackets] for tokens body exactly I'M \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 over caf\\u00e9 exactly counting a tiktoken's while no https://example.com/a/b?c=d \\u0645\\u0631\\u062d\\u0628\\u0627 lazy lazy (parens) backtracking that involved keep and a tiktoken's is body They'RE it it 100% scanned jumps jumps (parens) na\\u00efve so for v1.2.3 F# 'single' way no exactly 'single' tiktoken's there's dog caf\\u00e9 engine $1,234.56 is tokens\"}, {\"role\": \"user\", \"content\": \"a C++ 100% faster the node.js gateway every It's budget reservation there's scanner before Z\\u00fcrich \\ud83d\\ude42 written written tokens faster tokens $1,234.56 counting quick Z\\u00fcrich for lazy tokens node.js before before backtracking WON'T It's jumps before Z\\u00fcrich $1,234.56 with boundaries mirrors the request once node.js faster that jumps backtracking 1999 boundaries node.js fox admission tiktoken's mirrors way the we'll involved backtracking request 3.14159 the jumps na\\u00efve na\\u00efve \\\"quotes\\\" mirrors 3.14159 na\\u00efve scanner F# because node.js 3.14159 that caf\\u00e9 100% and and (parens) \\\"quotes\\\" WON'T $1,234.56 \\u6771\\u4eac quick $1,234.56 reservation while the na\\u00efve exactly once written backtracking written the na\\u00efve\"}, {\"role\": \"assistant\", \"content\": \"tokens \\u0645\\u0631\\u062d\\u0628\\u0627 scanned [brackets] every admission [brackets] scanned 'single' exactly scanner tokens that [brackets] tokens scanned the before before They'RE every because \\\"quotes\\\" C++ scanner with with brown exactly is for user@example.com while caf\\u00e9 boundaries it WON'T scanner that admission reservation 100% is admission before gateway reservation caf\\u00e9 100% request tiktoken's 3.14159 before backtracking while so https://example.com/a/b?c=d 42 faster and and faster It's we'll mirrors 42 tiktoken's regex node.js there's scanned\"}, {\"role\": \"user\", \"content\": \"written v1.2.3 exactly reservation piece every F# reservation node.js node.js \\u6771\\u4eac that scanner https://example.com/a/b?c=d 42 {braces} it and keep It's no regex jumps with user@example.com body over It's \\\"quotes\\\" and 'single' counting I'M that \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) engine the way $1,234.56 scanned They'RE for caf\\u00e9 \\\"quotes\\\" lazy don't backtracking tokens \\\"quotes\\\" quick {braces} \\\"quotes\\\" involved involved engine scanner because hand tiktoken's v1.2.3 F# regex $1,234.56 engine {braces} gateway counting that (parens) that the before admission that every backtracking \\\"quotes\\\" [brackets] way gateway engine lazy fox 42 don't [brackets] once na\\u00efve \\\"quotes\\\" They'RE They'RE is jumps every 100% written backtracking body caf\\u00e9 that backtracking v1.2.3 (parens) F# keep exactly piece once v1.2.3 F# They'RE backtracking 100% the fox the Z\\u00fcrich involved F#\"}, {\"role\": \"assistant\", \"content\": \"{braces} 42 engine over na\\u00efve hand involved They'RE request It's Z\\u00fcrich before {braces} 42 dog every node.js is over https://example.com/a/b?c=d na\\u00efve engine fox is [brackets] is on over \\ud83d\\ude42 (parens) \\ud83d\\ude42 regex They'RE They'RE \\ud83d\\ude42 (parens) C++ node.js 100% jumps once that scanner on mirrors regex that faster tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's while F# exactly 100% 3.14159 engine \\u6771\\u4eac scanner {braces} na\\u00efve I'M the over brown written don't tokens user@example.com is F# once faster on It's a there's C++ 1999 we'll [brackets] over the dog there's request dog C++ the brown regex way na\\u00efve written before Z\\u00fcrich engine caf\\u00e9 WON'T 1999 with engine and scanned $1,234.56 keep I'M every user@example.com 'single' a and 1999 don't I'M written while and that\"}, {\"role\": \"user\", \"content\": \"budget {braces} hand so and (parens) fox for that no way is counting tokens WON'T tokens tokens lazy tokens piece on that request dog mirrors I'M mirrors scanned regex on na\\u00efve C++ once user@example.com before lazy user@example.com \\\"quotes\\\" I'M once over involved that request written backtracking \\\"quotes\\\" is no brown there's {braces} because \\ud83d\\ude42 keep reservation written 100% caf\\u00e9 v1.2.3 tiktoken's body F# \\u6771\\u4eac counting quick every faster It's while \\u6771\\u4eac every backtracking lazy They'RE tokens fox {braces} tokens \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" 1999 piece involved [brackets] tokens They'RE the a don't hand caf\\u00e9 quick before 'single' so \\u6771\\u4eac that [brackets] quick \\\"quotes\\\" counting gateway it that is It's scanner the 'single' the \\u0645\\u0631\\u062d\\u0628\\u0627 keep 100% it fox way na\\u00efve exactly mirrors caf\\u00e9 mirrors there's because piece every a\"}, {\"role\": \"assistant\", \"content\": \"counting no tiktoken's hand dog {braces} They'RE $1,234.56 involved \\\"quotes\\\" on \\u0645\\u0631\\u062d\\u0628\\u0627 the because tiktoken's 3.14159 that because counting a 3.14159 so node.js counting backtracking https://example.com/a/b?c=d C++ quick involved (parens) request (parens) F# no gateway exactly every backtracking hand on user@example.com It's written scanned before reservation {braces} there's [brackets] $1,234.56 I'M engine involved brown reservation Z\\u00fcrich request F# brown a way I'M written exactly faster \\u6771\\u4eac Z\\u00fcrich mirrors regex 42 tiktoken's caf\\u00e9 42 100% F# way 1999 no node.js way involved while and {braces} that 1999 They'RE node.js admission I'M the is there's 'single' scanner na\\u00efve scanned tiktoken's \\u6771\\u4eac before backtracking 'single'\"}, {\"role\": \"user\", \"content\": \"na\\u00efve tiktoken's boundaries admission 42 100% {braces} 'single' (parens) and and way I'M 3.14159 budget jumps with gateway the fox engine because on scanner \\\"quotes\\\" 1999 user@example.com that written 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 quick [brackets] WON'T 42 \\u6771\\u4eac \\\"quotes\\\" jumps so budget scanned 'single' boundaries and F# quick every https://example.com/a/b?c=d regex tiktoken's lazy involved we'll \\ud83d\\ude42 it admission v1.2.3 reservation counting scanner that \\u0645\\u0631\\u062d\\u0628\\u0627 over there's counting because 100% keep F# lazy $1,234.56 so that is once request is \\ud83d\\ude42 before once na\\u00efve keep\"}, {\"role\": \"assistant\", \"content\": \"with Z\\u00fcrich a 3.14159 faster faster faster jumps 'single' before over dog It's $1,234.56 node.js 'single' na\\u00efve body way we'll It's once regex the counting 3.14159 [brackets] gateway C++ on piece jumps body is node.js lazy user@example.com don't with v1.2.3 exactly user@example.com [brackets] once It's it Z\\u00fcrich [brackets] {braces} faster \\u0645\\u0631\\u062d\\u0628\\u0627 1999 that C++ involved on quick C++ scanner reservation reservation regex node.js faster because jumps It's mirrors that F# brown the request with every reservation backtracking (parens) v1.2.3 I'M {braces} once tokens reservation 42 a 3.14159 F# we'll (parens) way caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 1999 F# piece 1999 quick for scanned\"}, {\"role\": \"user\", \"content\": \"scanned piece F# reservation because regex regex counting written na\\u00efve budget 42 the before keep keep way It's It's exactly node.js because 3.14159 tokens keep with \\u0645\\u0631\\u062d\\u0628\\u0627 body exactly that reservation v1.2.3 42 [brackets] keep scanner dog piece it F# [brackets] regex that tokens exactly body node.js that once faster hand with lazy exactly boundaries jumps that scanned that written\"}, {\"role\": \"assistant\", \"content\": \"brown They'RE gateway the Z\\u00fcrich gateway user@example.com over quick and \\\"quotes\\\" boundaries regex \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 the tokens 'single' is every tokens 42 reservation (parens) request I'M backtracking regex tokens boundaries on is hand scanner boundaries caf\\u00e9 engine and jumps It's the over user@example.com user@example.com I'M on lazy while 1999 WON'T brown that reservation It's v1.2.3 WON'T exactly gateway with quick F# F# the I'M gateway I'M don't way boundaries \\\"quotes\\\" reservation Z\\u00fcrich and 3.14159 on faster na\\u00efve 42 v1.2.3 v1.2.3 3.14159 jumps hand we'll once (parens) C++ the faster keep They'RE node.js fox scanned 42 jumps faster way gateway hand 1999 it that request keep \\u6771\\u4eac $1,234.56 tokens engine that we'll F# we'll written \\\"quotes\\\" a is no keep tiktoken's node.js node.js on counting user@example.com \\\"quotes\\\" {braces} 1999 user@example.com hand fox I'M node.js is\"}, {\"role\": \"user\", \"content\": \"'single' \\u6771\\u4eac and scanned v1.2.3 dog \\u6771\\u4eac with budget $1,234.56 a exactly and scanned body way na\\u00efve on brown mirrors a 3.14159 dog fox jumps {braces} (parens) jumps the way over and we'll involved faster user@example.com fox body there's gateway body brown jumps there's for 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking It's mirrors Z\\u00fcrich 'single' tokens 'single' for user@example.com 100% faster there's keep C++ and don't gateway budget {braces} faster hand don't is involved \\u6771\\u4eac regex regex reservation scanner\"}, {\"role\": \"assistant\", \"content\": \"F# body is regex jumps WON'T reservation Z\\u00fcrich tokens piece that and \\u0645\\u0631\\u062d\\u0628\\u0627 [brackets] for 'single' we'll engine F# there's \\\"quotes\\\" so counting \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d [brackets] piece [brackets] involved exactly fox {braces} \\ud83d\\ude42 3.14159 way F# 3.14159 scanned while keep every request \\u6771\\u4eac way C++ na\\u00efve engine engine user@example.com because admission I'M gateway v1.2.3 v1.2.3 It's it Z\\u00fcrich https://example.com/a/b?c=d 3.14159 Z\\u00fcrich hand [brackets] [brackets] dog that 'single' exactly \\u6771\\u4eac regex brown reservation involved and WON'T F# na\\u00efve before It's a tiktoken's I'M we'll with written budget engine caf\\u00e9 tokens 100% scanned It's brown every a for\"}, {\"role\": \"user\", \"content\": \"involved because caf\\u00e9 so there's I'M 42 no na\\u00efve na\\u00efve exactly tiktoken's once scanner tiktoken's counting exactly dog the a and faster WON'T so They'RE 'single' faster for written C++ 3.14159 tiktoken's $1,234.56 C++ way exactly https://example.com/a/b?c=d every way node.js budget 42 before before gateway it mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 way faster budget (parens) budget every that don't over so v1.2.3 request we'll because with \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 don't 'single' reservation fox tokens there's request it They'RE [brackets] it node.js gateway 'single' node.js body once way quick 3.14159\"}, {\"role\": \"assistant\", \"content\": \"3.14159 while reservation body before on once (parens) request because https://example.com/a/b?c=d involved 42 user@example.com and no request They'RE dog F# na\\u00efve They'RE exactly [brackets] regex engine mirrors scanner engine It's there's counting \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich way 3.14159 over counting dog gateway no we'll for \\\"quotes\\\" no lazy 42 involved counting involved written counting 1999 a $1,234.56 tokens regex quick and no once \\u6771\\u4eac every so brown is \\u0645\\u0631\\u062d\\u0628\\u0627 fox https://example.com/a/b?c=d caf\\u00e9 brown \\u6771\\u4eac 42 exactly a scanner over \\ud83d\\ude42 we'll budget dog {braces} for before faster\"}, {\"role\": \"user\", \"content\": \" faster the over keep while \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's once the because involved scanner regex that \\u6771\\u4eac there's brown WON'T dog \\u0645\\u0631\\u062d\\u0628\\u0627 budget lazy v1.2.3 v1.2.3 caf\\u00e9 counting so don't the scanner I'M caf\\u00e9 1999 don't once engine lazy (parens) request keep user@example.com the keep fox the Z\\u00fcrich https://example.com/a/b?c=d 3.14159 caf\\u00e9 (parens) It's piece I'M written \\\"quotes\\\" 'single' It's that {braces} reservation piece\"}, {\"role\": \"assistant\", \"content\": \"fox quick C++ brown before no lazy faster the that the every 100% na\\u00efve that 100% {braces} scanner we'll involved na\\u00efve before that {braces} faster mirrors because 'single' it faster v1.2.3 involved is so with request piece piece while that engine admission tokens gateway scanned before F# every $1,234.56 hand 1999 request is over v1.2.3 over quick regex v1.2.3 v1.2.3 boundaries Z\\u00fcrich and\"}, {\"role\": \"user\", \"content\": \"node.js WON'T faster \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 dog tokens and the no because They'RE and WON'T WON'T caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 42 na\\u00efve written keep don't the WON'T is admission $1,234.56 and every the I'M is C++ budget node.js tokens (parens) node.js caf\\u00e9 'single' with engine https://example.com/a/b?c=d v1.2.3 written node.js {braces} tiktoken's boundaries tiktoken's lazy scanned user@example.com quick body regex backtracking involved jumps faster quick They'RE request na\\u00efve body on written \\ud83d\\ude42 regex a It's a a no \\\"quotes\\\" \\ud83d\\ude42 while involved the jumps and \\ud83d\\ude42 node.js https://example.com/a/b?c=d boundaries na\\u00efve node.js brown hand 42 na\\u00efve for before C++ while https://example.com/a/b?c=d 42 because that for reservation \\\"quotes\\\" gateway budget is a so It's Z\\u00fcrich budget regex once scanned piece regex budget before over na\\u00efve regex\"}, {\"role\": \"assistant\", \"content\": \"WON'T on scanned user@example.com They'RE exactly faster the for a boundaries faster faster $1,234.56 I'M with don't {braces} jumps \\u0645\\u0631\\u062d\\u0628\\u0627 jumps so is keep while Z\\u00fcrich 42 so tiktoken's [brackets] written {braces} request engine no faster node.js counting \\u6771\\u4eac lazy counting faster is mirrors it there's jumps tiktoken's piece [brackets] \\u6771\\u4eac so brown admission brown Z\\u00fcrich way mirrors reservation user@example.com v1.2.3 $1,234.56 budget because the \"}, {\"role\": \"user\", \"content\": \"every jumps 'single' engine so dog scanned F# there's \\ud83d\\ude42 https://example.com/a/b?c=d user@example.com exactly backtracking node.js admission is na\\u00efve v1.2.3 tokens the fox involved gateway that admission don't reservation hand engine C++ so for we'll over written dog https://example.com/a/b?c=d while way dog and written with brown F# budget backtracking so tiktoken's \\u6771\\u4eac exactly written involved They'RE because reservation a caf\\u00e9 over boundaries regex is 'single' admission keep admission I'M WON'T admission counting scanned dog {braces} admission admission every F# 1999 scanner They'RE and the na\\u00efve 42 there's with involved 42 exactly 100% it [brackets] way They'RE node.js \\ud83d\\ude42 keep we'll I'M so I'M tiktoken's reservation WON'T caf\\u00e9 boundaries C++\"}, {\"role\": \"assistant\", \"content\": \"node.js the It's so C++ scanned so 1999 the budget request F# fox body body keep is [brackets] 42 over scanner there's \\u0645\\u0631\\u062d\\u0628\\u0627 node.js \\u0645\\u0631\\u062d\\u0628\\u0627 don't involved F# hand na\\u00efve scanned with involved engine jumps WON'T {braces} mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 is counting C++ fox it (parens) 3.14159 regex engine {braces} it keep 'single' the \\u0645\\u0631\\u062d\\u0628\\u0627 gateway engine WON'T user@example.com keep [brackets] once before budget the reservation regex hand keep user@example.com It's written reservation no fox \\\"quotes\\\" on \\u6771\\u4eac (parens) Z\\u00fcrich backtracking \\ud83d\\ude42 [brackets] mirrors gateway scanner boundaries v1.2.3 brown admission 1999 because way tokens regex admission so \\u0645\\u0631\\u062d\\u0628\\u0627 with counting fox tiktoken's na\\u00efve tokens counting gateway\"}, {\"role\": \"user\", \"content\": \"v1.2.3 dog counting 42 1999 with (parens) 42 scanner involved the it 'single' jumps over and user@example.com once I'M I'M engine before mirrors a faster brown quick Z\\u00fcrich v1.2.3 \\u6771\\u4eac body it mirrors na\\u00efve fox budget 'single' with mirrors WON'T user@example.com faster so Z\\u00fcrich scanned It's that written reservation is that 42 v1.2.3 over 1999 on 1999 that over na\\u00efve brown there's the 100% \\u6771\\u4eac request \\\"quotes\\\" $1,234.56 budget regex piece na\\u00efve 3.14159 the scanner \\u6771\\u4eac scanner scanner It's v1.2.3 on tokens engine 1999 42\"}, {\"role\": \"assistant\", \"content\": \"faster $1,234.56 piece user@example.com for \\u0645\\u0631\\u062d\\u0628\\u0627 while so on Z\\u00fcrich backtracking no It's \\\"quotes\\\" hand node.js before engine it tokens jumps reservation dog jumps [brackets] boundaries scanner for (parens) {braces} and request before scanned because before the for admission mirrors every piece admission keep and gateway we'll brown 1999 is I'M They'RE over https://example.com/a/b?c=d 3.14159 while hand once exactly way counting caf\\u00e9 3.14159 hand na\\u00efve every scanned \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors because They'RE backtracking \\ud83d\\ude42 \\\"quotes\\\" over admission piece on don't scanned exactly reservation and (parens)\"}, {\"role\": \"user\", \"content\": \"every quick that that while na\\u00efve fox {braces} while brown that v1.2.3 dog a C++ F# over for regex while every jumps it is 1999 caf\\u00e9 it and there's C++ we'll over counting boundaries brown 42 I'M involved and reservation boundaries so before I'M with scanned every is involved https://example.com/a/b?c=d lazy regex piece It's a while v1.2.3 user@example.com the 'single' I'M a is is \\\"quotes\\\" user@example.com because node.js faster F# Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 that na\\u00efve because brown involved the so involved WON'T body backtracking quick engine backtracking gateway that a user@example.com body [brackets] tokens every boundaries while \\ud83d\\ude42 is on boundaries scanner node.js Z\\u00fcrich is there's for regex it 100% piece tiktoken's\"}, {\"role\": \"assistant\", \"content\": \"reservation hand before F# counting $1,234.56 caf\\u00e9 faster \\u6771\\u4eac [brackets] tokens Z\\u00fcrich tiktoken's on written {braces} It's keep \\\"quotes\\\" while tokens na\\u00efve the no because keep https://example.com/a/b?c=d because counting request \\\"quotes\\\" it a because hand involved reservation node.js and backtracking 42 on piece I'M gateway 'single' mirrors the Z\\u00fcrich quick lazy brown {braces} admission Z\\u00fcrich C++ before https://example.com/a/b?c=d It's engine backtracking don't \\u6771\\u4eac piece dog no keep Z\\u00fcrich piece and 42 regex C++ caf\\u00e9 with we'll 3.14159 'single' mirrors counting Z\\u00fcrich it dog scanner a node.js piece {braces} we'll we'll because They'RE user@example.com fox caf\\u00e9 1999 42 every counting tiktoken's It's before backtracking scanner 42 the \\u6771\\u4eac admission lazy F# gateway mirrors It's C++ tiktoken's is dog no written https://example.com/a/b?c=d \\u6771\\u4eac exactly \\\"quotes\\\" body exactly\"}, {\"role\": \"user\", \"content\": \"that https://example.com/a/b?c=d the over boundaries scanner don't [brackets] quick every because because I'M because so once dog mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" quick on 'single' (parens) 42 and that before while while engine while over tiktoken's counting exactly {braces} tiktoken's piece gateway user@example.com It's that admission scanner It's request \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com that 42 for so \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 user@example.com for F# exactly that 100% backtracking boundaries it fox tiktoken's piece counting once engine admission piece Z\\u00fcrich a budget keep and {braces} is reservation scanner with https://example.com/a/b?c=d gateway no user@example.com na\\u00efve so user@example.com written brown\"}, {\"role\": \"assistant\", \"content\": \"piece They'RE body v1.2.3 \\ud83d\\ude42 C++ na\\u00efve It's we'll backtracking on because the counting a involved counting that there's request WON'T jumps that node.js Z\\u00fcrich quick https://example.com/a/b?c=d They'RE Z\\u00fcrich 3.14159 engine budget \\ud83d\\ude42 \\\"quotes\\\" scanner don't {braces} F# request don't $1,234.56 \\\"quotes\\\" tokens over while admission every with quick I'M \\\"quotes\\\" {braces} fox [brackets] gateway dog keep Z\\u00fcrich every [brackets] every mirrors C++ \\ud83d\\ude42 They'RE is [brackets] 3.14159 scanner hand and request 3.14159 with scanner no the regex 1999 They'RE every dog there's the admission that tiktoken's 'single' engine\"}, {\"role\": \"user\", \"content\": \"boundaries It's way dog I'M body lazy https://example.com/a/b?c=d budget faster for I'M regex every body piece with \\ud83d\\ude42 regex before for quick 'single' exactly with over brown there's jumps the don't {braces} and [brackets] it jumps on while don't no exactly every tokens and the 1999 \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 no [brackets] keep \\ud83d\\ude42 with Z\\u00fcrich gateway while backtracking the [brackets] every is \\u6771\\u4eac while faster for while budget [brackets] \\ud83d\\ude42 don't once quick so hand 'single' It's on dog exactly with once engine 'single' no written involved over tokens way over [brackets] with so regex caf\\u00e9 3.14159 \\ud83d\\ude42 a node.js every because faster reservation budget v1.2.3 with lazy I'M engine because and for before before engine mirrors \\\"quotes\\\" that node.js node.js counting a admission counting on WON'T no piece I'M once boundaries\"}, {\"role\": \"assistant\", \"content\": \"piece counting quick \\u0645\\u0631\\u062d\\u0628\\u0627 brown the there's mirrors regex faster involved tokens backtracking F# backtracking \\\"quotes\\\" \\ud83d\\ude42 with \\u6771\\u4eac body with gateway counting lazy exactly request [brackets] 'single' because while keep is I'M jumps body {braces} mirrors 1999 $1,234.56 faster it so {braces} v1.2.3 keep 42 admission admission It's that involved https://example.com/a/b?c=d na\\u00efve \\u6771\\u4eac \\u6771\\u4eac 42 fox node.js the (parens) (parens) [brackets] on faster hand node.js engine before \\u0645\\u0631\\u062d\\u0628\\u0627 jumps body we'll engine regex Z\\u00fcrich body admission 3.14159 They'RE Z\\u00fcrich scanner before scanner reservation \\u6771\\u4eac admission involved don't lazy $1,234.56 $1,234.56 that lazy scanned F# for request https://example.com/a/b?c=d there's before scanner reservation They'RE on \\u0645\\u0631\\u062d\\u0628\\u0627 quick we'll 100% lazy while budget 'single' once quick brown admission node.js gateway 1999 v1.2.3 piece because https://example.com/a/b?c=d reservation $1,234.56 for na\\u00efve \\ud83d\\ude42 exactly gateway that exactly on node.js and\"}, {\"role\": \"user\", \"content\": \"hand counting request while reservation there's body regex tokens don't regex every no a engine it a 1999 F# reservation with on mirrors for before Z\\u00fcrich v1.2.3 on the \\u0645\\u0631\\u062d\\u0628\\u0627 quick involved budget every node.js the hand it with node.js 1999 quick faster backtracking quick 42 while counting mirrors over so reservation involved [brackets] no written and tokens fox tiktoken's It's written \\ud83d\\ude42 over admission with over don't \\u6771\\u4eac way counting \\ud83d\\ude42 $1,234.56 https://example.com/a/b?c=d body hand over request tiktoken's that budget https://example.com/a/b?c=d exactly scanned gateway regex the while It's there's written for lazy F# counting body caf\\u00e9 jumps that counting jumps\"}, {\"role\": \"assistant\", \"content\": \"faster Z\\u00fcrich boundaries faster don't counting before dog engine I'M Z\\u00fcrich 3.14159 C++ \\u0645\\u0631\\u062d\\u0628\\u0627 for scanned involved engine backtracking brown \\u6771\\u4eac mirrors is (parens) backtracking tokens admission that Z\\u00fcrich don't C++ tiktoken's user@example.com that mirrors lazy scanner (parens) piece the body every \\\"quotes\\\" the request tokens body C++ (parens) a don't They'RE exactly the 3.14159 is budget user@example.com it \\\"quotes\\\" on \\ud83d\\ude42 backtracking admission written \\ud83d\\ude42 C++ quick fox 1999 I'M for it don't 1999 because way there's body reservation 100% WON'T dog 1999 tokens once a over once so $1,234.56 jumps They'RE scanner 3.14159 regex request 'single' exactly $1,234.56 exactly\"}, {\"role\": \"user\", \"content\": \"we'll 1999 lazy over 42 user@example.com over scanned gateway lazy piece so lazy exactly 'single' caf\\u00e9 node.js there's and once v1.2.3 while faster gateway every way 1999 gateway while 1999 1999 on caf\\u00e9 so written backtracking \\\"quotes\\\" backtracking scanner https://example.com/a/b?c=d body admission dog piece way brown fox 3.14159 with WON'T \\u6771\\u4eac \\\"quotes\\\" and exactly lazy na\\u00efve 1999 (parens) and involved keep that 3.14159 F# keep lazy no na\\u00efve request https://example.com/a/b?c=d jumps 'single' fox 'single' [brackets] tiktoken's counting once there's na\\u00efve admission faster F# we'll \\\"quotes\\\" https://example.com/a/b?c=d \\u0645\\u0631\\u062d\\u0628\\u0627 exactly \\u6771\\u4eac hand over https://example.com/a/b?c=d F# fox scanned 1999\"}, {\"role\": \"assistant\", \"content\": \"1999 no there's is 'single' C++ na\\u00efve that over admission \\\"quotes\\\" F# scanner gateway piece 100% while for no {braces} with fox scanned counting \\\"quotes\\\" there's gateway user@example.com keep body involved 3.14159 Z\\u00fcrich Z\\u00fcrich \\\"quotes\\\" [brackets] quick admission request a is 42 scanned caf\\u00e9 \\\"quotes\\\" I'M piece C++ mirrors 'single' every $1,234.56 3.14159 regex quick dog 42 WON'T brown no {braces} piece body quick before Z\\u00fcrich for for request {braces} engine while \\u6771\\u4eac Z\\u00fcrich brown gateway counting backtracking before don't $1,234.56 the request so {braces} \\u6771\\u4eac [brackets] because a node.js a 3.14159 boundaries every quick we'll jumps na\\u00efve that on we'll scanner tokens \\u6771\\u4eac https://example.com/a/b?c=d tokens 'single' keep\"}, {\"role\": \"user\", \"content\": \"fox admission exactly dog \\u6771\\u4eac backtracking {braces} node.js \\\"quotes\\\" way gateway on 1999 engine scanner keep quick way written boundaries because $1,234.56 don't there's way so over tokens F# that we'll v1.2.3 scanned F# for \\u0645\\u0631\\u062d\\u0628\\u0627 involved there's while {braces} \\u0645\\u0631\\u062d\\u0628\\u0627 admission admission [brackets] They'RE \\ud83d\\ude42 regex body quick that v1.2.3 dog quick user@example.com over before admission hand is budget brown admission we'll body node.js na\\u00efve boundaries hand (parens) there's we'll v1.2.3 100% caf\\u00e9 for [brackets] 100% dog \\\"quotes\\\" piece boundaries I'M admission backtracking gateway engine no C++ there's we'll tiktoken's regex \\u0645\\u0631\\u062d\\u0628\\u0627 way WON'T {braces} once counting 1999 it before quick request WON'T we'll I'M admission lazy tokens involved 42 gateway lazy faster written that reservation exactly \\u6771\\u4eac 100% piece 100% scanner admission\"}, {\"role\": \"assistant\", \"content\": \"exactly because body and we'll so backtracking over fox F# I'M that jumps don't don't lazy scanned I'M lazy way piece It's They'RE scanner gateway 1999 so fox no because gateway 1999 that boundaries engine They'RE on admission scanner over Z\\u00fcrich that Z\\u00fcrich written v1.2.3 WON'T [brackets] tiktoken's scanner 3.14159 don't \\u6771\\u4eac tiktoken's user@example.com [brackets] before \\\"quotes\\\" no C++ node.js scanned 42\"}, {\"role\": \"user\", \"content\": \"once so 3.14159 because 100% with $1,234.56 faster no jumps once written gateway written tokens [brackets] before brown a counting \\\"quotes\\\" tiktoken's They'RE the quick (parens) exactly request jumps before involved mirrors dog budget \\ud83d\\ude42 every scanned \\ud83d\\ude42 fox https://example.com/a/b?c=d gateway body node.js lazy na\\u00efve v1.2.3 exactly way [brackets] gateway scanned WON'T [brackets] https://example.com/a/b?c=d piece \\u6771\\u4eac C++ 3.14159 fox v1.2.3 tokens mirrors caf\\u00e9 engine brown Z\\u00fcrich They'RE hand quick $1,234.56 WON'T tokens lazy written na\\u00efve 'single' dog lazy it that way \\ud83d\\ude42\"}, {\"role\": \"assistant\", \"content\": \"dog a They'RE It's so They'RE counting user@example.com jumps F# request no node.js there's na\\u00efve mirrors \\u6771\\u4eac tiktoken's scanner backtracking hand boundaries 1999 brown written mirrors written faster that budget because hand with 3.14159 and over WON'T every that way regex boundaries C++ \\u0645\\u0631\\u062d\\u0628\\u0627 piece Z\\u00fcrich because it They'RE involved https://example.com/a/b?c=d body 42 a fox that every and request {braces} with node.js F# [brackets] $1,234.56 on no budget way on backtracking node.js so 100%\"}, {\"role\": \"user\", \"content\": \"They'RE scanned dog body on that don't 1999 \\ud83d\\ude42 scanner 1999 that exactly C++ They'RE every caf\\u00e9 brown so \\ud83d\\ude42 caf\\u00e9 mirrors and while request WON'T budget scanner na\\u00efve Z\\u00fcrich the fox 3.14159 100% 'single' before \\ud83d\\ude42 Z\\u00fcrich scanner before that 1999 v1.2.3 body every boundaries that and v1.2.3 1999 100% because user@example.com dog engine keep the we'll I'M scanned na\\u00efve \\\"quotes\\\" there's that lazy https://example.com/a/b?c=d {braces} Z\\u00fcrich exactly \\\"quotes\\\" budget reservation involved that mirrors boundaries na\\u00efve that backtracking \\ud83d\\ude42 once \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 WON'T WON'T I'M node.js involved don't {braces} a and lazy piece backtracking for hand regex\"}, {\"role\": \"assistant\", \"content\": \"WON'T admission admission F# because don't [brackets] tiktoken's before the so na\\u00efve don't hand no na\\u00efve request on mirrors dog \\u0645\\u0631\\u062d\\u0628\\u0627 and before while body $1,234.56 v1.2.3 mirrors piece reservation so backtracking user@example.com every 100% F# regex 1999 brown [brackets] gateway 42 and on with 42 regex (parens) {braces} WON'T v1.2.3 {braces} a there's every I'M the 'single' 42 F# involved \\ud83d\\ude42 piece keep scanner the before I'M reservation written 3.14159 before there's piece once It's \\\"quotes\\\" dog written brown (parens) because while fox quick exactly user@example.com $1,234.56 boundaries Z\\u00fcrich is admission WON'T scanner we'll $1,234.56 hand [brackets] written {braces} {braces} {braces} the\"}, {\"role\": \"user\", \"content\": \"counting once C++ it the and involved keep it so there's I'M C++ 1999 no keep piece It's no user@example.com 'single' I'M WON'T node.js there's 100% quick 100% dog \\ud83d\\ude42 faster Z\\u00fcrich is before there's on don't F# {braces} keep that it na\\u00efve faster counting faster is over hand hand before involved tokens v1.2.3 {braces} boundaries backtracking so on before regex hand backtracking for while keep so (parens) tokens is with exactly backtracking (parens) admission $1,234.56 once no 42 C++ brown reservation Z\\u00fcrich \\\"quotes\\\"\"}, {\"role\": \"assistant\", \"content\": \"so {braces} It's involved body $1,234.56 lazy scanner caf\\u00e9 na\\u00efve v1.2.3 mirrors 3.14159 na\\u00efve the fox It's before we'll over no involved counting C++ brown quick hand \\u6771\\u4eac every exactly 42 boundaries $1,234.56 dog WON'T quick scanned na\\u00efve dog na\\u00efve boundaries quick backtracking body [brackets] tokens piece 100% exactly F# on admission the WON'T budget scanner \\\"quotes\\\" and WON'T we'll boundaries on budget scanned with \\ud83d\\ude42 $1,234.56 before we'll tiktoken's https://example.com/a/b?c=d https://example.com/a/b?c=d over no the there's It's don't over way keep [brackets] with body They'RE the once jumps keep admission They'RE tokens tokens jumps no 1999 fox 3.14159 100% reservation (parens) lazy [brackets] because https://example.com/a/b?c=d \\u0645\\u0631\\u062d\\u0628\\u0627 reservation 3.14159 keep while They'RE I'M brown quick They'RE admission exactly F# regex gateway and v1.2.3 while (parens) because with \\\"quotes\\\"\"}, {\"role\": \"user\", \"content\": \"scanner while node.js scanner 42 hand tokens tiktoken's They'RE engine is 'single' 'single' piece and v1.2.3 while it faster don't brown don't body na\\u00efve (parens) \\ud83d\\ude42 node.js the \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d caf\\u00e9 that $1,234.56 mirrors admission \\ud83d\\ude42 scanner that \\\"quotes\\\" C++ (parens) WON'T Z\\u00fcrich They'RE It's \\ud83d\\ude42 caf\\u00e9 that with WON'T it hand that over \\\"quotes\\\" 'single' na\\u00efve 1999 piece keep {braces} quick the no regex \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 piece na\\u00efve the \\u0645\\u0631\\u062d\\u0628\\u0627 'single' no that a body with\"}, {\"role\": \"assistant\", \"content\": \"that C++ over caf\\u00e9 tiktoken's scanned \\u0645\\u0631\\u062d\\u0628\\u0627 fox \\\"quotes\\\" for node.js once keep C++ counting once body so for jumps every before reservation user@example.com budget [brackets] boundaries 42 1999 written jumps $1,234.56 that the https://example.com/a/b?c=d faster Z\\u00fcrich request and once involved [brackets] so over over na\\u00efve caf\\u00e9 for for \\\"quotes\\\" while that scanned tokens scanner no no It's F# request request reservation Z\\u00fcrich jumps the na\\u00efve (parens) the backtracking jumps na\\u00efve scanner caf\\u00e9 keep I'M \\u6771\\u4eac quick scanner body hand scanner I'M way https://example.com/a/b?c=d it caf\\u00e9 mirrors (parens) F# keep {braces} brown WON'T regex engine user@example.com reservation user@example.com brown don't engine brown hand They'RE written https://example.com/a/b?c=d way (parens) dog\"}, {\"role\": \"user\", \"content\": \"\\u6771\\u4eac F# scanner that regex It's lazy na\\u00efve 100% involved keep the admission na\\u00efve user@example.com a for Z\\u00fcrich {braces} exactly lazy counting while tokens with hand written a is counting I'M [brackets] [brackets] 100% tokens body \\\"quotes\\\" \\\"quotes\\\" tiktoken's v1.2.3 https://example.com/a/b?c=d counting $1,234.56 $1,234.56 don't keep that before scanned lazy before gateway na\\u00efve scanner fox $1,234.56 tiktoken's faster so I'M involved over (parens) admission lazy dog caf\\u00e9 don't jumps 100% node.js request because caf\\u00e9 https://example.com/a/b?c=d 42 before 1999 [brackets] {braces} once [brackets] caf\\u00e9 scanner quick there's\"}, {\"role\": \"assistant\", \"content\": \"scanner caf\\u00e9 (parens) keep so {braces} on F# it because so over brown It's exactly I'M I'M F# faster $1,234.56 tokens exactly $1,234.56 caf\\u00e9 the na\\u00efve engine user@example.com \\\"quotes\\\" written C++ so regex don't backtracking engine the caf\\u00e9 C++ They'RE fox lazy written WON'T boundaries keep It's caf\\u00e9 while {braces} C++ keep jumps no node.js 3.14159 there's faster no 'single' so It's so a once scanner \\u0645\\u0631\\u062d\\u0628\\u0627 engine every once before backtracking request because F# scanned jumps I'M exactly we'll we'll and once [brackets] there's exactly 100%\"}, {\"role\": \"user\", \"content\": \"C++ tiktoken's $1,234.56 body scanner dog I'M regex that https://example.com/a/b?c=d mirrors every before user@example.com \\\"quotes\\\" with involved scanner budget I'M reservation hand that for 42 fox 42 tokens scanned They'RE They'RE \\\"quotes\\\" keep reservation piece Z\\u00fcrich the 3.14159 counting counting boundaries F# every while body boundaries 42 don't hand faster the reservation no \\u0645\\u0631\\u062d\\u0628\\u0627 that \\u6771\\u4eac while involved lazy involved caf\\u00e9 that on \\u6771\\u4eac dog 'single' on over once keep Z\\u00fcrich lazy https://example.com/a/b?c=d backtracking scanned 100% 1999 C++ body \\u0645\\u0631\\u062d\\u0628\\u0627 scanner https://example.com/a/b?c=d {braces} don't 3.14159 node.js regex fox faster I'M we'll lazy because backtracking \\ud83d\\ude42 fox F# \\\"quotes\\\" mirrors tokens It's and the for a WON'T regex reservation 42 dog They'RE request on node.js 'single' tokens node.js that admission 1999 on WON'T budget \\ud83d\\ude42 admission exactly 1999 that we'll \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich\"}, {\"role\": \"assistant\", \"content\": \"it 1999 engine \\\"quotes\\\" v1.2.3 every body boundaries over fox before I'M faster that fox \\u0645\\u0631\\u062d\\u0628\\u0627 involved node.js keep They'RE fox the with WON'T jumps hand caf\\u00e9 (parens) exactly user@example.com engine Z\\u00fcrich for request that involved na\\u00efve boundaries faster before no way 'single' piece don't tokens \\ud83d\\ude42 regex $1,234.56 \\ud83d\\ude42 keep [brackets] over 100% once that exactly node.js (parens) Z\\u00fcrich v1.2.3 scanned the admission there's F# and is tiktoken's on \\u0645\\u0631\\u062d\\u0628\\u0627 no involved I'M C++ 42 there's while before v1.2.3 piece over brown we'll 1999 lazy that user@example.com budget C++ on quick engine 1999 {braces} faster and that node.js budget body\"}, {\"role\": \"user\", \"content\": \"user@example.com no tiktoken's involved jumps exactly \\ud83d\\ude42 They'RE \\ud83d\\ude42 once because faster node.js I'M engine hand lazy we'll [brackets] fox budget quick every the tokens C++ a quick They'RE {braces} faster caf\\u00e9 for mirrors caf\\u00e9 it user@example.com node.js written on that engine regex https://example.com/a/b?c=d there's tokens don't na\\u00efve counting \\ud83d\\ude42 user@example.com mirrors engine $1,234.56 3.14159 backtracking hand request brown keep tiktoken's \\u6771\\u4eac a there's it scanner keep engine because fox scanned no I'M no scanned faster 42 [brackets] hand \\u0645\\u0631\\u062d\\u0628\\u0627 that budget scanned that tokens admission C++ and [brackets] that 'single'\"}, {\"role\": \"assistant\", \"content\": \"node.js regex while and WON'T body 100% gateway there's https://example.com/a/b?c=d WON'T keep brown 100% budget with budget \\u0645\\u0631\\u062d\\u0628\\u0627 and so don't while don't exactly backtracking faster tokens gateway node.js lazy over written counting 100% node.js tiktoken's brown dog v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 we'll C++ the with don't we'll tokens 100% no no written scanned counting involved hand user@example.com \\\"quotes\\\" lazy \\u0645\\u0631\\u062d\\u0628\\u0627 body counting for [brackets] 'single' backtracking piece It's regex don't admission once boundaries over lazy F# \\u6771\\u4eac tokens it \\u6771\\u4eac user@example.com fox that while tiktoken's Z\\u00fcrich and (parens) every [brackets] keep counting 100% 42 3.14159 counting it {braces} piece 'single' involved https://example.com/a/b?c=d involved quick \\ud83d\\ude42 tokens written that request tiktoken's lazy admission 42 I'M tokens \\\"quotes\\\" Z\\u00fcrich engine reservation tiktoken's once so a scanned {braces} scanner dog before piece written keep https://example.com/a/b?c=d while It's {braces} Z\\u00fcrich it\"}, {\"role\": \"user\", \"content\": \"Z\\u00fcrich 3.14159 They'RE They'RE there's budget we'll gateway a brown [brackets] jumps once fox there's F# boundaries node.js for once 42 F# dog once so jumps 42 and 100% while it user@example.com exactly brown so [brackets] Z\\u00fcrich 'single' request with tiktoken's Z\\u00fcrich every reservation hand boundaries keep a \\ud83d\\ude42 it once admission piece and we'll budget dog na\\u00efve 3.14159 backtracking so gateway 42 Z\\u00fcrich faster (parens) node.js 100% lazy jumps fox scanner engine 'single' counting request \\u6771\\u4eac a regex brown that 1999 https://example.com/a/b?c=d user@example.com They'RE regex I'M \\ud83d\\ude42 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 scanner budget tokens [brackets] WON'T that quick budget v1.2.3 way way user@example.com request the [brackets] $1,234.56 gateway They'RE quick scanned $1,234.56 \\ud83d\\ude42 na\\u00efve don't admission \\u0645\\u0631\\u062d\\u0628\\u0627 budget \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"assistant\", \"content\": \"WON'T admission piece \\u6771\\u4eac gateway no \\ud83d\\ude42 WON'T tokens 'single' piece admission 3.14159 It's while I'M Z\\u00fcrich the keep boundaries Z\\u00fcrich WON'T It's [brackets] brown tokens They'RE exactly boundaries WON'T admission piece (parens) so tiktoken's backtracking is \\u0645\\u0631\\u062d\\u0628\\u0627 before mirrors 3.14159 dog backtracking written caf\\u00e9 hand scanned way that \\ud83d\\ude42 tiktoken's quick way They'RE {braces} counting na\\u00efve once reservation fox admission $1,234.56 hand and fox we'll hand because quick the is \\u6771\\u4eac {braces} \\\"quotes\\\" exactly and dog so scanner involved \\u6771\\u4eac [brackets]\"}, {\"role\": \"user\", \"content\": \"a \\u0645\\u0631\\u062d\\u0628\\u0627 I'M we'll before backtracking [brackets] scanner don't tokens admission tokens https://example.com/a/b?c=d C++ $1,234.56 is \\ud83d\\ude42 exactly dog piece no tiktoken's admission 'single' the quick that every before before dog It's engine so tiktoken's the boundaries https://example.com/a/b?c=d 'single' 1999 written body {braces} there's scanned budget admission don't for we'll \\\"quotes\\\" body quick regex quick there's we'll it https://example.com/a/b?c=d user@example.com involved na\\u00efve 3.14159 on we'll the over na\\u00efve na\\u00efve \\u6771\\u4eac gateway I'M v1.2.3 it way the scanner jumps budget C++ C++ \\u6771\\u4eac it with quick F# F# with piece budget involved request before backtracking the tiktoken's scanner $1,234.56 backtracking body admission over They'RE on engine request node.js while that They'RE backtracking request v1.2.3 42 3.14159 and 100% scanned a caf\\u00e9 every 100% exactly request tokens is gateway before a exactly caf\\u00e9 we'll node.js because \\ud83d\\ude42 admission\"}, {\"role\": \"assistant\", \"content\": \"F# \\u0645\\u0631\\u062d\\u0628\\u0627 the we'll user@example.com mirrors F# \\u6771\\u4eac faster over with C++ no because piece scanned because 42 piece (parens) scanned admission 100% Z\\u00fcrich It's 42 for 1999 the regex the dog faster budget (parens) brown engine It's on (parens) so https://example.com/a/b?c=d piece over v1.2.3 scanned don't It's piece regex admission https://example.com/a/b?c=d every don't dog F# no regex hand is dog piece no counting v1.2.3 don't https://example.com/a/b?c=d the 3.14159 because once https://example.com/a/b?c=d \\u6771\\u4eac\"}, {\"role\": \"user\", \"content\": \"we'll body tiktoken's F# lazy is is request we'll because dog \\u6771\\u4eac boundaries lazy caf\\u00e9 exactly quick jumps They'RE 3.14159 brown \\u6771\\u4eac 3.14159 'single' on keep faster 'single' because I'M no 'single' tiktoken's exactly F# v1.2.3 tokens hand because faster that over and gateway don't 42 \\u0645\\u0631\\u062d\\u0628\\u0627 lazy before that once for keep body lazy Z\\u00fcrich before faster piece gateway caf\\u00e9 exactly involved scanner the \\ud83d\\ude42 node.js WON'T way before C++ written 1999 It's with that is every na\\u00efve 1999 keep \\u6771\\u4eac quick request regex budget It's 1999 before a hand lazy engine WON'T don't while brown admission \\u0645\\u0631\\u062d\\u0628\\u0627 1999 caf\\u00e9 na\\u00efve WON'T boundaries counting reservation \\\"quotes\\\" there's the once over counting request\"}, {\"role\": \"assistant\", \"content\": \"gateway reservation once F# C++ written 42 on piece https://example.com/a/b?c=d [brackets] that \\u0645\\u0631\\u062d\\u0628\\u0627 we'll reservation They'RE node.js the fox piece https://example.com/a/b?c=d user@example.com every F# $1,234.56 \\\"quotes\\\" admission body [brackets] 'single' that and because regex we'll \\u6771\\u4eac 42 keep na\\u00efve F# user@example.com keep admission we'll user@example.com 1999 {braces} 100% because no $1,234.56 counting the 42 there's boundaries F# no admission 42 so so v1.2.3 for mirrors while the regex piece budget 42 node.js {braces} na\\u00efve regex with jumps that counting a involved $1,234.56 that admission tiktoken's na\\u00efve involved before $1,234.56 budget 1999 because v1.2.3 that backtracking \\u6771\\u4eac mirrors [brackets] node.js hand {braces} don't brown counting Z\\u00fcrich scanned over \\ud83d\\ude42 with I'M lazy tokens\"}, {\"role\": \"user\", \"content\": \"quick keep it body so every C++ gateway It's tokens 'single' 100% we'll quick regex so scanner v1.2.3 the involved It's while {braces} $1,234.56 that written that C++ scanned dog 100% 100% boundaries (parens) gateway exactly scanner regex gateway \\u6771\\u4eac mirrors brown 1999 regex v1.2.3 $1,234.56 caf\\u00e9 scanned brown there's caf\\u00e9 1999 \\\"quotes\\\" no F# for that 100% backtracking exactly admission the quick (parens) 'single' [brackets] {braces} that written boundaries reservation https://example.com/a/b?c=d request on quick 1999 hand so brown regex while exactly piece hand piece reservation tiktoken's scanner \\\"quotes\\\" caf\\u00e9 way because na\\u00efve (parens) tiktoken's 3.14159 admission so don't on involved na\\u00efve C++ dog reservation It's boundaries admission it (parens) gateway\"}, {\"role\": \"assistant\", \"content\": \"keep there's with scanner that budget no on node.js budget scanner dog before exactly quick 1999 WON'T reservation node.js it F# $1,234.56 caf\\u00e9 1999 scanned engine and \\ud83d\\ude42 [brackets] with scanned and admission \\u6771\\u4eac \\ud83d\\ude42 faster {braces} WON'T on I'M F# scanned while dog no exactly written C++ because tiktoken's request reservation written don't 100% gateway budget \\u6771\\u4eac that jumps counting \\u6771\\u4eac every engine scanned They'RE \\ud83d\\ude42 for engine F# and {braces} exactly exactly \\\"quotes\\\" the hand before\"}, {\"role\": \"user\", \"content\": \"jumps $1,234.56 we'll we'll {braces} user@example.com way hand fox way It's brown 3.14159 https://example.com/a/b?c=d regex jumps don't once dog written quick lazy tiktoken's WON'T way it budget the that quick It's WON'T They'RE brown keep because so keep \\u6771\\u4eac don't written because way with is I'M https://example.com/a/b?c=d 3.14159 scanner counting v1.2.3 gateway the it 42 while WON'T tokens caf\\u00e9 jumps once regex budget v1.2.3 C++ backtracking brown exactly involved every regex boundaries the mirrors F# F# $1,234.56 [brackets] WON'T \\u6771\\u4eac $1,234.56 jumps every that It's \\u0645\\u0631\\u062d\\u0628\\u0627 the keep while keep hand regex once jumps v1.2.3 caf\\u00e9 don't brown (parens) 'single'\"}, {\"role\": \"assistant\", \"content\": \"quick because faster $1,234.56 reservation F# don't the boundaries v1.2.3 on user@example.com reservation WON'T body 1999 brown \\u6771\\u4eac jumps \\\"quotes\\\" quick WON'T that (parens) the keep $1,234.56 C++ written {braces} so tokens on {braces} dog https://example.com/a/b?c=d \\ud83d\\ude42 3.14159 written backtracking gateway the way 'single' is gateway that jumps it fox piece backtracking tokens that faster backtracking there's a request request [brackets] and tokens $1,234.56 for user@example.com \\u6771\\u4eac so 'single' 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T don't 100% [brackets] for lazy the node.js hand because It's faster\"}, {\"role\": \"user\", \"content\": \"is faster admission budget that 1999 counting quick engine no while exactly while the is written v1.2.3 faster v1.2.3 mirrors once They'RE [brackets] tiktoken's 3.14159 quick with because fox while so [brackets] a dog while involved keep 1999 that involved \\ud83d\\ude42 keep is v1.2.3 100% once C++ node.js piece backtracking I'M 3.14159 that (parens) They'RE admission budget fox lazy scanned boundaries on on once that keep https://example.com/a/b?c=d lazy exactly because brown with exactly written \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"that 1999 node.js C++ \\u0645\\u0631\\u062d\\u0628\\u0627 no tiktoken's boundaries request so engine before 42 user@example.com a faster piece involved regex body for before lazy hand faster 'single' and hand tiktoken's lazy that node.js jumps budget fox way [brackets] v1.2.3 [brackets] 'single' brown piece It's counting mirrors every 3.14159 \\\"quotes\\\" https://example.com/a/b?c=d na\\u00efve we'll F# 100% once budget because scanned tokens Z\\u00fcrich engine scanned\"}, {\"role\": \"user\", \"content\": \"'single' It's over lazy piece 100% gateway caf\\u00e9 caf\\u00e9 \\\"quotes\\\" a way the request written the budget reservation counting $1,234.56 with Z\\u00fcrich counting I'M keep tokens once we'll the while https://example.com/a/b?c=d a on the tokens involved caf\\u00e9 gateway backtracking before fox na\\u00efve with on 3.14159 written lazy {braces} \\\"quotes\\\" there's don't quick 'single' that for They'RE faster because I'M is scanner \\u6771\\u4eac counting the every Z\\u00fcrich body $1,234.56 quick keep backtracking scanner 'single' lazy and 42 counting written so tokens engine 'single' because scanned They'RE v1.2.3 tokens \\u6771\\u4eac exactly 3.14159 on jumps jumps reservation tokens \\u0645\\u0631\\u062d\\u0628\\u0627 budget They'RE na\\u00efve na\\u00efve request\"}, {\"role\": \"assistant\", \"content\": \"lazy is reservation budget it jumps quick {braces} body 'single' is no jumps \\u6771\\u4eac Z\\u00fcrich no admission 100% and WON'T admission 'single' piece is we'll v1.2.3 is the we'll admission engine request body that body WON'T user@example.com 42 and piece I'M backtracking {braces} WON'T boundaries [brackets] engine request so backtracking budget {braces} caf\\u00e9 3.14159 {braces} It's hand once $1,234.56 They'RE budget It's quick brown regex 42 (parens) and 1999 regex Z\\u00fcrich Z\\u00fcrich Z\\u00fcrich exactly https://example.com/a/b?c=d way tokens no tiktoken's counting that hand backtracking jumps way over WON'T na\\u00efve {braces} piece it that 100% (parens) exactly and is over the WON'T user@example.com 1999 request written counting brown while on because 42 tokens no because (parens) Z\\u00fcrich lazy reservation admission once counting 42 exactly They'RE is dog\"}, {\"role\": \"user\", \"content\": \"that They'RE the admission F# (parens) https://example.com/a/b?c=d tiktoken's 3.14159 scanned reservation 'single' boundaries written exactly faster F# na\\u00efve while and [brackets] It's brown \\ud83d\\ude42 caf\\u00e9 jumps with \\\"quotes\\\" backtracking on reservation C++ written over regex [brackets] counting [brackets] lazy scanned $1,234.56 faster 'single' {braces} (parens) They'RE \\\"quotes\\\" on request F# while the dog faster https://example.com/a/b?c=d brown tokens 100% because WON'T way counting don't scanner a scanner 'single' dog because hand admission tiktoken's {braces} hand caf\\u00e9 that a \\ud83d\\ude42 fox \\ud83d\\ude42 \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors $1,234.56 once tiktoken's node.js \\u6771\\u4eac exactly \\\"quotes\\\" Z\\u00fcrich WON'T (parens) the hand It's once every no They'RE F# C++ lazy every don't 1999 user@example.com piece C++ admission 1999 because I'M budget\"}, {\"role\": \"assistant\", \"content\": \"dog over caf\\u00e9 the $1,234.56 before is that 'single' piece It's jumps exactly na\\u00efve every budget hand tokens $1,234.56 \\ud83d\\ude42 we'll (parens) hand for (parens) exactly way na\\u00efve body gateway brown keep It's before every lazy and while quick budget 100% (parens) no 1999 the 3.14159 regex that fox user@example.com 1999 it quick $1,234.56 there's no that exactly written there's \\u0645\\u0631\\u062d\\u0628\\u0627 request mirrors on involved involved C++ no [brackets] exactly na\\u00efve https://example.com/a/b?c=d 3.14159 gateway the jumps once scanned with that admission budget They'RE 'single' caf\\u00e9 mirrors WON'T \\u0645\\u0631\\u062d\\u0628\\u0627 reservation request It's the body admission reservation scanned we'll node.js regex counting before with \\ud83d\\ude42 backtracking [brackets] https://example.com/a/b?c=d 3.14159 the once It's request a written WON'T mirrors 1999\"}, {\"role\": \"user\", \"content\": \"1999 WON'T so user@example.com jumps fox \\ud83d\\ude42 {braces} there's budget before written quick mirrors {braces} for with 42 and gateway counting we'll request it so a for engine {braces} WON'T Z\\u00fcrich user@example.com because https://example.com/a/b?c=d the node.js is fox I'M 3.14159 tiktoken's faster is once 42 involved no 'single' I'M caf\\u00e9 WON'T I'M They'RE [brackets] $1,234.56 we'll I'M so 42 It's every (parens) we'll once \\ud83d\\ude42 once mirrors jumps budget body {braces} \\ud83d\\ude42 F# fox {braces} exactly \\ud83d\\ude42 WON'T that on 100% budget engine reservation C++ fox F# 42 no (parens) counting mirrors v1.2.3 for fox [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 while way 1999 boundaries piece 100% involved every \\ud83d\\ude42 'single' brown once v1.2.3 before 'single' (parens) v1.2.3 Z\\u00fcrich we'll involved 100% {braces}\"}, {\"role\": \"assistant\", \"content\": \"engine backtracking budget on na\\u00efve mirrors dog reservation don't jumps budget gateway counting dog tokens brown for way Z\\u00fcrich dog piece 'single' that quick it there's quick budget 42 It's that dog They'RE F# written dog piece \\u0645\\u0631\\u062d\\u0628\\u0627 F# regex every no hand body once (parens) every lazy 100% over quick once mirrors don't and v1.2.3 They'RE faster brown 1999 a \\ud83d\\ude42 42 gateway https://example.com/a/b?c=d Z\\u00fcrich for over na\\u00efve it lazy admission admission brown hand \\\"quotes\\\" gateway reservation 1999 keep and {braces} keep F# request written tiktoken's every budget every regex there's exactly I'M \\\"quotes\\\" \\u0645\\u0631\\u062d\\u0628\\u0627 there's backtracking and don't boundaries F# caf\\u00e9 3.14159 (parens) 3.14159 user@example.com https://example.com/a/b?c=d tiktoken's keep every\"}, {\"role\": \"user\", \"content\": \"gateway 'single' v1.2.3 request that hand tiktoken's over v1.2.3 'single' so \\ud83d\\ude42 1999 once once na\\u00efve admission na\\u00efve piece that \\\"quotes\\\" 3.14159 way exactly engine F# the \\u0645\\u0631\\u062d\\u0628\\u0627 3.14159 user@example.com on https://example.com/a/b?c=d It's https://example.com/a/b?c=d F# there's written with {braces} admission {braces} a on user@example.com dog dog [brackets] that caf\\u00e9 over for admission v1.2.3 100% node.js boundaries scanner 42 piece because is (parens) (parens) body no is {braces} that user@example.com \\u0645\\u0631\\u062d\\u0628\\u0627 hand \\\"quotes\\\" user@example.com on that \\\"quotes\\\" backtracking \\\"quotes\\\" on no hand 3.14159 brown faster 100% (parens)\"}, {\"role\": \"assistant\", \"content\": \"there's user@example.com They'RE It's lazy user@example.com before before https://example.com/a/b?c=d request WON'T WON'T fox WON'T and faster for body involved 3.14159 keep written admission on gateway written I'M with quick v1.2.3 C++ there's jumps quick admission scanned jumps node.js before while 1999 over node.js with reservation every once node.js brown don't there's mirrors a [brackets] so faster before \\ud83d\\ude42 tiktoken's admission F# don't written na\\u00efve admission is fox we'll way dog tokens scanned the \\\"quotes\\\" hand no 'single' \\ud83d\\ude42 over we'll that regex so every for (parens) is with the \\\"quotes\\\" exactly piece regex exactly brown for for hand tokens C++ user@example.com tiktoken's engine I'M the It's faster regex is dog 'single' C++ user@example.com \\ud83d\\ude42 Z\\u00fcrich {braces} node.js for \\u6771\\u4eac brown regex the with tiktoken's 42 $1,234.56 (parens)\"}, {\"role\": \"user\", \"content\": \"request it reservation that https://example.com/a/b?c=d so engine user@example.com jumps that and the 3.14159 because 3.14159 brown tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 scanner \\u0645\\u0631\\u062d\\u0628\\u0627 fox counting there's dog lazy node.js tiktoken's I'M we'll 42 it tokens faster faster before \\u0645\\u0631\\u062d\\u0628\\u0627 reservation so with the regex it once backtracking for It's hand regex way there's 100% exactly dog no we'll 42 brown lazy scanned budget na\\u00efve WON'T because and we'll 42 way tiktoken's WON'T gateway we'll I'M dog https://example.com/a/b?c=d scanned and that mirrors \\ud83d\\ude42 C++ They'RE with counting on keep na\\u00efve 3.14159 once request keep admission \\\"quotes\\\" C++ \\ud83d\\ude42 once quick boundaries user@example.com tokens budget budget scanner caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 exactly node.js hand\"}, {\"role\": \"assistant\", \"content\": \"fox for user@example.com there's jumps quick fox I'M boundaries mirrors before involved over node.js admission that is lazy It's 1999 1999 fox involved user@example.com \\\"quotes\\\" boundaries 100% \\u0645\\u0631\\u062d\\u0628\\u0627 that \\u6771\\u4eac backtracking 100% there's jumps because 42 tiktoken's caf\\u00e9 that v1.2.3 tokens brown \\ud83d\\ude42 backtracking user@example.com F# 100% 3.14159 no boundaries It's quick https://example.com/a/b?c=d Z\\u00fcrich tokens 42 100% over WON'T They'RE for lazy body involved every on on user@example.com \\u6771\\u4eac (parens) quick once faster quick over backtracking don't once They'RE engine caf\\u00e9 tokens fox because there's every dog we'll \\u6771\\u4eac exactly 3.14159 \\ud83d\\ude42 boundaries no every the F# 100% faster faster once boundaries a I'M {braces} 1999 over \\\"quotes\\\" mirrors that fox 3.14159 lazy na\\u00efve admission that reservation no for written because scanned caf\\u00e9 body\"}, {\"role\": \"user\", \"content\": \"and involved body jumps \\\"quotes\\\" {braces} it we'll regex it na\\u00efve counting the reservation keep admission once before while admission v1.2.3 I'M reservation 1999 scanned tokens counting don't boundaries lazy exactly we'll $1,234.56 and \\u0645\\u0631\\u062d\\u0628\\u0627 \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 engine I'M 42 F# keep every (parens) budget 100% written tiktoken's (parens) don't na\\u00efve I'M body because the is (parens) admission brown hand piece 42 every request don't $1,234.56 once there's 1999 with we'll that backtracking F# with fox mirrors [brackets] for WON'T with jumps request fox scanned a request while no that faster way I'M \\u6771\\u4eac boundaries that brown dog so (parens) written\"}, {\"role\": \"assistant\", \"content\": \" 42 exactly and \\\"quotes\\\" mirrors admission F# faster F# a on faster with quick brown reservation piece (parens) fox \\u0645\\u0631\\u062d\\u0628\\u0627 budget C++ v1.2.3 so over node.js Z\\u00fcrich WON'T tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 F# 1999 42 the involved once 100% is quick (parens) before jumps dog don't over because $1,234.56 reservation 3.14159 that WON'T caf\\u00e9 backtracking request the \\u0645\\u0631\\u062d\\u0628\\u0627 piece every involved 100% budget there's boundaries counting once written we'll na\\u00efve 42 \\ud83d\\ude42 I'M 3.14159 regex scanner reservation tiktoken's on I'M \\u6771\\u4eac boundaries on brown there's that 'single' the exactly scanned no admission request we'll tiktoken's 'single' because \\\"quotes\\\" \\u6771\\u4eac no tiktoken's caf\\u00e9 jumps it tiktoken's lazy faster exactly body tiktoken's fox budget\"}, {\"role\": \"user\", \"content\": \"so fox a backtracking Z\\u00fcrich admission budget fox scanned 42 request don't because reservation way admission gateway v1.2.3 the so \\u6771\\u4eac involved 'single' with a 100% 'single' v1.2.3 user@example.com dog backtracking body C++ piece the so mirrors written once $1,234.56 while fox 42 [brackets] F# we'll Z\\u00fcrich piece budget while for we'll the we'll written exactly $1,234.56 mirrors counting \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors on 'single' na\\u00efve is node.js while is a on\"}, {\"role\": \"assistant\", \"content\": \"on involved 42 'single' They'RE piece while [brackets] dog that and there's before once jumps counting that so is \\u6771\\u4eac [brackets] over and exactly with involved it once on 1999 42 3.14159 is keep \\u6771\\u4eac jumps that there's They'RE piece lazy over with lazy request \\ud83d\\ude42 written node.js admission backtracking way the over backtracking node.js and while $1,234.56 [brackets] while 42 \\u0645\\u0631\\u062d\\u0628\\u0627 I'M 'single' while user@example.com WON'T \\\"quotes\\\" before admission the faster budget mirrors counting we'll once jumps involved WON'T 3.14159 v1.2.3 [brackets] keep over for 'single' \\u6771\\u4eac regex written na\\u00efve way (parens) user@example.com request there's every involved I'M for scanned It's WON'T https://example.com/a/b?c=d C++ caf\\u00e9 once user@example.com [brackets] a 1999 it tiktoken's 'single' a over mirrors jumps request there's dog reservation mirrors tokens jumps tiktoken's is user@example.com tokens\"}, {\"role\": \"user\", \"content\": \"backtracking once over user@example.com body scanner piece faster \\ud83d\\ude42 over for C++ exactly https://example.com/a/b?c=d no engine 'single' piece 42 over hand for for backtracking quick way 1999 regex scanned every reservation no before over \\\"quotes\\\" so $1,234.56 [brackets] tokens na\\u00efve request a budget fox budget user@example.com tokens brown I'M \\\"quotes\\\" hand I'M \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac engine once engine backtracking while there's no [brackets] jumps the for [brackets] there's \\u0645\\u0631\\u062d\\u0628\\u0627 dog 42 admission so na\\u00efve F# there's \\u6771\\u4eac reservation F# keep piece keep user@example.com caf\\u00e9 the that It's $1,234.56 v1.2.3 It's na\\u00efve engine lazy They'RE counting faster 'single' piece scanned $1,234.56 user@example.com boundaries [brackets] 'single' \\u6771\\u4eac brown so scanned 'single' scanner boundaries 'single' admission admission \\\"quotes\\\" caf\\u00e9 piece\"}, {\"role\": \"assistant\", \"content\": \"Z\\u00fcrich counting while the the no 100% \\u0645\\u0631\\u062d\\u0628\\u0627 body 3.14159 before the while 1999 once that on caf\\u00e9 1999 on it tiktoken's piece scanned 'single' Z\\u00fcrich and [brackets] regex with regex \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com counting admission way keep scanner scanner \\u0645\\u0631\\u062d\\u0628\\u0627 every so [brackets] tokens $1,234.56 every budget boundaries because reservation and keep 1999 \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d tiktoken's https://example.com/a/b?c=d scanner admission caf\\u00e9 $1,234.56 F# faster so engine so it regex 'single' before hand with that admission fox keep Z\\u00fcrich no scanned I'M tiktoken's the it admission a jumps brown that there's it that (parens) 100% (parens) gateway for\"}, {\"role\": \"user\", \"content\": \"so 100% once it there's (parens) that with 1999 lazy written written keep na\\u00efve brown caf\\u00e9 regex the tiktoken's that a while brown It's dog and It's 'single' $1,234.56 42 \\u6771\\u4eac I'M v1.2.3 once na\\u00efve I'M {braces} node.js scanned It's hand piece no caf\\u00e9 backtracking admission with we'll is on node.js fox on tokens brown brown regex scanner admission $1,234.56 lazy regex piece regex node.js (parens) once on while quick dog gateway body that admission for written it scanned I'M it mirrors because the v1.2.3 every 3.14159 boundaries request na\\u00efve It's because faster hand dog tokens v1.2.3 hand that tiktoken's keep caf\\u00e9 keep tokens we'll it jumps no \\ud83d\\ude42 node.js exactly na\\u00efve na\\u00efve every involved tiktoken's \\u6771\\u4eac counting dog over user@example.com https://example.com/a/b?c=d 1999 3.14159 node.js is hand F# written a\"}, {\"role\": \"assistant\", \"content\": \"we'll fox there's 3.14159 C++ so budget because we'll Z\\u00fcrich the on before we'll that Z\\u00fcrich fox keep scanned dog once that \\\"quotes\\\" while scanned brown engine for fox while there's (parens) 42 every we'll caf\\u00e9 https://example.com/a/b?c=d budget exactly gateway F# https://example.com/a/b?c=d I'M dog way before brown with it 3.14159 don't no engine the piece keep user@example.com Z\\u00fcrich It's once written and They'RE the admission scanner with written that fox engine body on scanner 1999 the admission WON'T quick user@example.com I'M admission tiktoken's https://example.com/a/b?c=d while https://example.com/a/b?c=d every with involved the admission gateway there's 'single' gateway a scanner and request that is engine counting dog user@example.com that hand counting lazy admission boundaries engine while way WON'T brown \\ud83d\\ude42 F# I'M F# scanner \\u6771\\u4eac 3.14159 a backtracking scanner I'M brown\"}, {\"role\": \"user\", \"content\": \"I'M lazy It's over on faster fox tiktoken's while that It's for that reservation https://example.com/a/b?c=d keep node.js the \\u6771\\u4eac the admission body 1999 exactly and way dog (parens) node.js \\u6771\\u4eac on They'RE quick that faster F# boundaries tokens boundaries reservation reservation backtracking tokens is brown 100% {braces} quick engine we'll node.js 100% scanned over F# hand na\\u00efve that request jumps quick because regex\"}, {\"role\": \"assistant\", \"content\": \"with 100% $1,234.56 boundaries we'll so no \\\"quotes\\\" $1,234.56 Z\\u00fcrich we'll brown on exactly 1999 WON'T caf\\u00e9 every dog we'll regex before tokens before [brackets] lazy over faster once Z\\u00fcrich request lazy They'RE tokens https://example.com/a/b?c=d [brackets] don't It's with keep hand hand (parens) I'M budget {braces} faster it way is there's every dog gateway \\\"quotes\\\" it Z\\u00fcrich way the over quick tokens C++ hand mirrors It's It's (parens) F# 100% boundaries caf\\u00e9 F# we'll scanner admission\"}, {\"role\": \"user\", \"content\": \"it node.js caf\\u00e9 'single' quick C++ $1,234.56 3.14159 hand 42 $1,234.56 I'M every exactly 100% lazy the \\u6771\\u4eac tiktoken's WON'T jumps with \\u0645\\u0631\\u062d\\u0628\\u0627 I'M counting written piece and keep it regex on F# keep 3.14159 'single' dog and boundaries tokens \\\"quotes\\\" engine with hand that 3.14159 brown WON'T don't scanner keep I'M WON'T fox F# exactly 'single' piece engine and so on request and It's reservation written that boundaries is caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 on every counting a written so jumps with scanner with quick don't jumps way \\u6771\\u4eac keep mirrors regex (parens) we'll it scanned node.js $1,234.56 reservation They'RE caf\\u00e9 node.js for while scanner it faster [brackets] gateway it scanner for mirrors na\\u00efve admission dog on\"}, {\"role\": \"assistant\", \"content\": \"that v1.2.3 written node.js I'M keep user@example.com tiktoken's tiktoken's way backtracking They'RE quick while scanned way we'll user@example.com way 1999 way caf\\u00e9 https://example.com/a/b?c=d hand faster is involved 1999 (parens) quick is (parens) on brown scanner with dog don't admission involved exactly F# They'RE is with exactly way before counting \\\"quotes\\\" backtracking once gateway faster 3.14159 [brackets] way [brackets] a keep \\ud83d\\ude42 hand (parens) counting boundaries budget I'M mirrors v1.2.3 F# once way 1999 WON'T reservation [brackets] for it scanned because 100% reservation 100% tiktoken's They'RE engine a it that so mirrors written over mirrors node.js F# \\ud83d\\ude42 brown dog it v1.2.3 user@example.com the 1999 reservation counting [brackets] budget way mirrors C++ F# and so there's so They'RE I'M scanner request request WON'T and written budget Z\\u00fcrich faster written lazy C++ brown user@example.com \\ud83d\\ude42 lazy reservation caf\\u00e9 node.js caf\\u00e9 brown\"}, {\"role\": \"user\", \"content\": \"budget budget and \\u6771\\u4eac tiktoken's tiktoken's don't that boundaries that 1999 backtracking with (parens) tiktoken's there's fox counting because exactly the every [brackets] engine Z\\u00fcrich admission way scanned boundaries \\ud83d\\ude42 user@example.com counting na\\u00efve F# fox 100% so faster 1999 because so quick engine way fox that engine lazy quick 1999 way $1,234.56 quick while body that 'single' Z\\u00fcrich They'RE na\\u00efve keep 3.14159 https://example.com/a/b?c=d the involved I'M jumps\"}, {\"role\": \"assistant\", \"content\": \"every jumps Z\\u00fcrich piece 'single' F# we'll 'single' over exactly written C++ 3.14159 with scanner exactly while user@example.com a body engine there's engine way 100% request [brackets] exactly way jumps Z\\u00fcrich budget once gateway before I'M \\\"quotes\\\" https://example.com/a/b?c=d user@example.com written admission exactly I'M {braces} \\ud83d\\ude42 They'RE with WON'T on backtracking node.js a with a lazy request na\\u00efve that scanner that once scanned quick 'single'\"}, {\"role\": \"user\", \"content\": \"v1.2.3 budget brown every [brackets] that na\\u00efve there's the and (parens) is brown They'RE F# there's every there's because the so because \\u0645\\u0631\\u062d\\u0628\\u0627 and over 3.14159 brown piece F# involved (parens) so regex while Z\\u00fcrich (parens) counting user@example.com while on $1,234.56 it v1.2.3 {braces} engine v1.2.3 while 1999 keep regex while https://example.com/a/b?c=d once engine WON'T involved \\u6771\\u4eac I'M it with for on hand \\ud83d\\ude42 v1.2.3 on a na\\u00efve regex there's na\\u00efve before v1.2.3 dog (parens) for brown 3.14159 v1.2.3 [brackets] exactly exactly no user@example.com on mirrors It's with that {braces} faster\"}, {\"role\": \"assistant\", \"content\": \"for once admission backtracking engine because once for I'M brown fox the quick quick backtracking there's lazy engine mirrors every don't caf\\u00e9 before mirrors engine 3.14159 on user@example.com with don't keep it node.js before hand regex reservation while scanner na\\u00efve body body no with F# admission Z\\u00fcrich F# WON'T no I'M \\ud83d\\ude42 way scanned $1,234.56 with a no it caf\\u00e9 lazy the dog https://example.com/a/b?c=d involved caf\\u00e9 jumps we'll scanner caf\\u00e9 42 na\\u00efve is we'll\"}, {\"role\": \"user\", \"content\": \"'single' scanner 100% we'll it a {braces} while it v1.2.3 while mirrors there's exactly caf\\u00e9 It's gateway (parens) fox every dog (parens) a written while while before hand involved admission written counting It's mirrors C++ 1999 scanned fox don't written don't reservation exactly mirrors it https://example.com/a/b?c=d there's no scanned that I'M involved $1,234.56 lazy fox scanned written I'M 100% is we'll every don't F# brown every It's keep while scanned jumps tokens 'single' hand Z\\u00fcrich request admission It's 'single' quick backtracking user@example.com request tokens budget exactly gateway jumps {braces} 'single' once tokens a Z\\u00fcrich before admission admission way jumps fox piece hand They'RE we'll exactly reservation on it written mirrors every once $1,234.56 scanner scanner Z\\u00fcrich jumps is over (parens) written piece Z\\u00fcrich $1,234.56 before request 'single' is 'single' keep\"}, {\"role\": \"assistant\", \"content\": \"scanned $1,234.56 gateway over because It's because tokens (parens) C++ admission is hand C++ brown v1.2.3 user@example.com brown \\\"quotes\\\" once backtracking lazy [brackets] with scanner dog lazy 'single' budget 1999 we'll reservation exactly on exactly the counting the [brackets] engine They'RE scanner caf\\u00e9 (parens) involved na\\u00efve on dog (parens) before don't while fox F# request once the lazy \\ud83d\\ude42 gateway user@example.com once that for 42 involved body user@example.com on regex the for lazy backtracking node.js body over involved v1.2.3 \\u6771\\u4eac regex mirrors there's involved regex faster that before request 3.14159 3.14159 that body 1999 faster for 42 'single' user@example.com request user@example.com piece scanned 1999 keep 3.14159 every fox F# scanned before jumps that admission counting They'RE that tokens I'M\"}, {\"role\": \"user\", \"content\": \"exactly quick and https://example.com/a/b?c=d a brown is dog hand regex It's tiktoken's the every with on scanner every keep scanned backtracking because budget keep fox there's so with so that on exactly counting once it exactly (parens) 3.14159 and that while fox admission 42 reservation It's the na\\u00efve 42 regex we'll piece hand so written regex faster fox backtracking that They'RE while 1999 \\u6771\\u4eac \\\"quotes\\\" request \\u6771\\u4eac They'RE and 100% the written for over way 1999 tokens https://example.com/a/b?c=d brown I'M brown brown keep while with $1,234.56 hand user@example.com {braces} [brackets] that faster na\\u00efve scanner on it caf\\u00e9 counting They'RE \\\"quotes\\\" {braces} counting there's body don't jumps budget fox reservation faster \\u0645\\u0631\\u062d\\u0628\\u0627 faster we'll with gateway [brackets] over counting It's there's \\u6771\\u4eac tokens that backtracking scanner that it so\"}, {\"role\": \"assistant\", \"content\": \"boundaries body fox counting that F# on there's caf\\u00e9 scanned jumps 42 dog hand jumps lazy \\\"quotes\\\" for WON'T is 42 on while 1999 scanner over there's engine 100% \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" jumps before don't [brackets] is counting involved It's I'M boundaries the (parens) with I'M engine 100% for exactly 'single' piece C++ na\\u00efve {braces} we'll every a no a \\\"quotes\\\" user@example.com body the C++ written https://example.com/a/b?c=d counting I'M a v1.2.3 dog 'single' Z\\u00fcrich keep scanner written before over lazy scanned the with keep over 'single' lazy 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) dog scanned the the dog no $1,234.56 with a so because quick Z\\u00fcrich fox v1.2.3 once faster \\ud83d\\ude42 faster boundaries it 1999 mirrors way a 3.14159 so it written \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d because the before involved Z\\u00fcrich budget while v1.2.3 scanned boundaries regex we'll backtracking\"}, {\"role\": \"user\", \"content\": \"involved $1,234.56 while jumps \\ud83d\\ude42 \\\"quotes\\\" piece hand written fox jumps v1.2.3 hand piece a jumps [brackets] brown involved Z\\u00fcrich engine that faster 'single' way faster budget F# 1999 so that tiktoken's involved regex WON'T budget mirrors mirrors F# before gateway I'M They'RE https://example.com/a/b?c=d lazy counting gateway regex boundaries \\\"quotes\\\" \\\"quotes\\\" 100% {braces} keep written 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 boundaries before written na\\u00efve 3.14159 that v1.2.3 no before request brown dog for $1,234.56 we'll node.js written mirrors scanned WON'T because we'll we'll way F# piece They'RE and while\"}, {\"role\": \"assistant\", \"content\": \"and that scanner brown They'RE we'll Z\\u00fcrich C++ the it 1999 hand F# admission \\u6771\\u4eac is mirrors so once the it admission counting is boundaries gateway that tiktoken's gateway lazy \\u6771\\u4eac na\\u00efve the we'll fox we'll is 42 that hand 3.14159 while reservation piece there's 1999 (parens) so involved we'll because involved written it caf\\u00e9 before once because https://example.com/a/b?c=d hand fox $1,234.56 admission is hand $1,234.56 boundaries for backtracking WON'T 1999 gateway the They'RE [brackets] budget na\\u00efve because that 42 They'RE C++ is don't \\ud83d\\ude42 C++ I'M $1,234.56 once 3.14159 1999 Z\\u00fcrich boundaries reservation quick exactly gateway $1,234.56 {braces}\"}, {\"role\": \"user\", \"content\": \"42 written (parens) backtracking jumps 1999 a \\\"quotes\\\" involved scanned admission They'RE gateway so backtracking 42 100% keep written It's WON'T user@example.com scanner that we'll WON'T Z\\u00fcrich don't na\\u00efve \\ud83d\\ude42 lazy mirrors (parens) before dog 'single' backtracking tokens faster 100% the body with 42 while {braces} while lazy brown I'M way exactly involved $1,234.56 'single' tokens engine counting tiktoken's $1,234.56 it user@example.com so reservation a 3.14159 dog that brown body fox mirrors quick before don't lazy gateway 100% \\u6771\\u4eac tokens the 42 dog 100% the gateway lazy because request there's $1,234.56 It's and regex 100% It's 'single' {braces} that keep involved caf\\u00e9 with {braces} They'RE faster the the $1,234.56 I'M backtracking I'M reservation 42 1999 before lazy request keep mirrors [brackets] 3.14159 42 {braces} $1,234.56 body 42 It's request 3.14159 {braces}\"}, {\"role\": \"assistant\", \"content\": \"and dog body user@example.com fox quick faster scanned the it faster $1,234.56 $1,234.56 with user@example.com over gateway fox tokens piece quick Z\\u00fcrich and counting engine no dog dog reservation on it it 100% every regex {braces} node.js mirrors before the Z\\u00fcrich v1.2.3 mirrors \\ud83d\\ude42 scanned WON'T brown there's \\u6771\\u4eac that mirrors for I'M it mirrors it node.js is \\u6771\\u4eac body \\\"quotes\\\" dog WON'T tiktoken's regex the It's the It's keep 3.14159 quick regex budget WON'T don't and counting over piece scanned C++ scanned is no \\u0645\\u0631\\u062d\\u0628\\u0627 1999 we'll it https://example.com/a/b?c=d keep https://example.com/a/b?c=d user@example.com admission backtracking once user@example.com piece request that (parens) budget written faster we'll budget scanned the user@example.com regex for WON'T piece counting\"}, {\"role\": \"user\", \"content\": \"{braces} 100% for It's the 1999 \\\"quotes\\\" scanned the 'single' quick caf\\u00e9 \\ud83d\\ude42 once once there's C++ It's we'll I'M engine no It's it na\\u00efve involved fox we'll written jumps 3.14159 dog exactly there's It's gateway tiktoken's every It's C++ WON'T we'll na\\u00efve reservation user@example.com 3.14159 every that with backtracking hand every v1.2.3 before so tiktoken's brown It's 3.14159 v1.2.3 caf\\u00e9 because \\\"quotes\\\" faster user@example.com It's before It's (parens) for I'M request written written dog \\u6771\\u4eac counting faster boundaries because tiktoken's on and gateway fox C++ admission It's {braces} v1.2.3 [brackets] 3.14159 exactly that budget that C++ (parens) so 1999 regex involved budget because scanner {braces} dog because I'M reservation exactly [brackets] reservation don't\"}, {\"role\": \"assistant\", \"content\": \"scanner They'RE \\\"quotes\\\" gateway 100% 3.14159 $1,234.56 so request 1999 C++ na\\u00efve quick faster we'll WON'T dog before engine 'single' tiktoken's node.js \\u0645\\u0631\\u062d\\u0628\\u0627 scanner it and na\\u00efve \\ud83d\\ude42 regex so once there's (parens) before 3.14159 dog [brackets] hand 100% backtracking over don't is once jumps caf\\u00e9 https://example.com/a/b?c=d so lazy that 3.14159 reservation on admission \\ud83d\\ude42 (parens) exactly budget v1.2.3 backtracking with brown request It's it with tokens (parens) scanner 1999 backtracking a fox faster v1.2.3 $1,234.56 I'M node.js the reservation piece [brackets] so \\u6771\\u4eac lazy counting the $1,234.56 I'M backtracking over 3.14159 fox fox keep the backtracking with It's [brackets] 3.14159 (parens) that They'RE so [brackets] piece [brackets] {braces} counting written boundaries mirrors budget (parens) gateway \\ud83d\\ude42 exactly hand engine WON'T 3.14159 faster WON'T fox request\"}, {\"role\": \"user\", \"content\": \"while quick \\u0645\\u0631\\u062d\\u0628\\u0627 so every 42 \\ud83d\\ude42 boundaries counting gateway that \\u0645\\u0631\\u062d\\u0628\\u0627 'single' no 100% and regex because before request and 'single' the once F# na\\u00efve $1,234.56 quick $1,234.56 reservation don't node.js scanner 42 node.js 1999 $1,234.56 https://example.com/a/b?c=d jumps lazy 'single' WON'T They'RE WON'T brown {braces} that scanner because Z\\u00fcrich written dog 3.14159 budget \\\"quotes\\\" engine 3.14159 3.14159 for before exactly tiktoken's jumps user@example.com is (parens) scanner lazy piece request brown Z\\u00fcrich node.js admission It's\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 C++ node.js because it 42 dog boundaries mirrors piece don't C++ 3.14159 https://example.com/a/b?c=d scanner there's is keep \\ud83d\\ude42 once before because backtracking over $1,234.56 [brackets] quick we'll gateway because we'll exactly once every gateway with over because 3.14159 keep every [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 the because 42 1999 no 100% backtracking 1999 \\u6771\\u4eac faster (parens) faster 42 the F# budget dog 1999 regex before\"}, {\"role\": \"user\", \"content\": \"no it \\ud83d\\ude42 mirrors https://example.com/a/b?c=d reservation C++ on budget \\u6771\\u4eac the for {braces} [brackets] and tiktoken's written 42 so body \\\"quotes\\\" involved C++ budget is written WON'T jumps It's {braces} counting reservation tokens brown piece hand before no It's gateway gateway [brackets] piece mirrors there's \\ud83d\\ude42 while a caf\\u00e9 \\u6771\\u4eac caf\\u00e9 counting counting it 3.14159 don't caf\\u00e9 so written for tiktoken's boundaries tokens 1999 fox keep quick that boundaries that no scanner Z\\u00fcrich budget it written that backtracking involved quick mirrors backtracking brown jumps the lazy They'RE and 42 backtracking 100% brown tokens gateway boundaries caf\\u00e9 reservation way https://example.com/a/b?c=d $1,234.56 100% regex \\u0645\\u0631\\u062d\\u0628\\u0627 lazy we'll \\u0645\\u0631\\u062d\\u0628\\u0627 involved jumps \\u0645\\u0631\\u062d\\u0628\\u0627 while node.js Z\\u00fcrich Z\\u00fcrich counting 100% (parens) that 100% on {braces} counting once user@example.com faster gateway \\u0645\\u0631\\u062d\\u0628\\u0627 that with\"}, {\"role\": \"assistant\", \"content\": \"faster on that boundaries on They'RE F# https://example.com/a/b?c=d that node.js quick for 100% the regex v1.2.3 don't with 42 C++ gateway is before keep is I'M the (parens) They'RE It's \\\"quotes\\\" so mirrors every backtracking tiktoken's v1.2.3 the 3.14159 They'RE scanner over no so that and WON'T that every the don't \\\"quotes\\\" 42 3.14159 node.js budget every 'single' caf\\u00e9 node.js before counting engine\"}, {\"role\": \"user\", \"content\": \"3.14159 piece tiktoken's with request regex on body [brackets] Z\\u00fcrich every gateway the that tiktoken's https://example.com/a/b?c=d no so budget node.js 1999 Z\\u00fcrich we'll They'RE keep and because piece every \\ud83d\\ude42 reservation faster written while request 3.14159 because {braces} fox WON'T written \\\"quotes\\\" (parens) over jumps mirrors body that jumps hand keep 3.14159 a the the it because so we'll gateway tiktoken's regex the scanned 1999 Z\\u00fcrich and don't and\"}, {\"role\": \"assistant\", \"content\": \"quick v1.2.3 100% node.js the na\\u00efve na\\u00efve involved once hand that \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors that request I'M tokens C++ don't na\\u00efve it mirrors there's with tokens https://example.com/a/b?c=d a there's so I'M mirrors $1,234.56 hand mirrors $1,234.56 admission C++ \\u0645\\u0631\\u062d\\u0628\\u0627 scanned user@example.com piece fox admission node.js https://example.com/a/b?c=d caf\\u00e9 It's \\ud83d\\ude42 scanned before https://example.com/a/b?c=d (parens) way https://example.com/a/b?c=d because F# I'M is the It's is reservation exactly user@example.com It's https://example.com/a/b?c=d lazy while the \\ud83d\\ude42 42 because for tiktoken's\"}, {\"role\": \"user\", \"content\": \"fox over {braces} is because caf\\u00e9 hand {braces} backtracking scanner F# engine engine user@example.com It's 42 quick jumps C++ \\\"quotes\\\" Z\\u00fcrich involved C++ na\\u00efve don't we'll admission exactly I'M a {braces} regex $1,234.56 exactly for [brackets] piece F# tokens \\\"quotes\\\" na\\u00efve dog lazy na\\u00efve there's on that with request \\u0645\\u0631\\u062d\\u0628\\u0627 quick a the gateway \\\"quotes\\\" and user@example.com 1999 (parens) over dog involved on engine over so hand over gateway Z\\u00fcrich counting keep that and $1,234.56 tokens scanned [brackets] reservation Z\\u00fcrich it node.js\"}, {\"role\": \"assistant\", \"content\": \"while on because fox \\u6771\\u4eac mirrors quick lazy I'M F# while a the because {braces} \\\"quotes\\\" once mirrors 100% a that on before body over request tokens we'll jumps that for 1999 mirrors hand lazy we'll over dog \\\"quotes\\\" because over way scanned that brown gateway over brown Z\\u00fcrich don't \\ud83d\\ude42 involved mirrors that once v1.2.3 \\ud83d\\ude42 v1.2.3 counting (parens) quick 'single' $1,234.56 scanner https://example.com/a/b?c=d no It's boundaries tokens hand caf\\u00e9 request body scanned 42 a counting with engine there's v1.2.3 fox gateway hand so counting scanned Z\\u00fcrich written https://example.com/a/b?c=d that {braces} v1.2.3 involved there's hand it there's is the exactly \\u0645\\u0631\\u062d\\u0628\\u0627 that scanner\"}, {\"role\": \"user\", \"content\": \"\\u0645\\u0631\\u062d\\u0628\\u0627 piece way scanned we'll hand written jumps fox https://example.com/a/b?c=d lazy $1,234.56 Z\\u00fcrich na\\u00efve WON'T scanner brown \\\"quotes\\\" mirrors It's caf\\u00e9 I'M \\\"quotes\\\" while admission fox engine brown tokens hand over while no scanner 3.14159 reservation budget \\\"quotes\\\" 100% hand while the It's faster tokens hand admission no and F# because \\u6771\\u4eac It's 'single' body that tiktoken's reservation tiktoken's is They'RE every (parens) Z\\u00fcrich scanner request the that F# caf\\u00e9 it They'RE way because\"}, {\"role\": \"assistant\", \"content\": \"brown \\\"quotes\\\" and don't regex [brackets] before on a that is every don't mirrors (parens) over lazy with F# on mirrors boundaries body jumps dog the It's \\u0645\\u0631\\u062d\\u0628\\u0627 so way before jumps WON'T request lazy v1.2.3 caf\\u00e9 it https://example.com/a/b?c=d $1,234.56 tiktoken's and {braces} admission \\u0645\\u0631\\u062d\\u0628\\u0627 fox on jumps jumps faster node.js user@example.com user@example.com so https://example.com/a/b?c=d counting It's \\\"quotes\\\" admission scanner on that while I'M piece written node.js way dog dog 100% before with \\u6771\\u4eac node.js hand mirrors involved 1999 a mirrors\"}, {\"role\": \"user\", \"content\": \"hand tiktoken's [brackets] 42 before 'single' F# F# for 'single' with body way every gateway {braces} brown keep scanner I'M the tiktoken's https://example.com/a/b?c=d for dog 100% tokens body counting over v1.2.3 https://example.com/a/b?c=d no 1999 They'RE {braces} tokens exactly scanner a for \\u0645\\u0631\\u062d\\u0628\\u0627 node.js way $1,234.56 gateway backtracking the brown a tiktoken's dog na\\u00efve I'M tokens for node.js boundaries a It's it C++ that tokens way C++\"}, {\"role\": \"assistant\", \"content\": \"a a counting don't WON'T no tokens quick we'll engine that written \\u0645\\u0631\\u062d\\u0628\\u0627 so 'single' every I'M Z\\u00fcrich there's v1.2.3 lazy there's caf\\u00e9 that scanned keep is fox 1999 100% tokens (parens) jumps 1999 v1.2.3 3.14159 WON'T {braces} for dog [brackets] the involved brown fox [brackets] (parens) They'RE (parens) gateway 1999 before a is scanned keep piece don't I'M exactly na\\u00efve we'll don't that 42 every gateway (parens) the na\\u00efve with Z\\u00fcrich backtracking lazy 100% is node.js faster while quick 42 engine tokens with \\u0645\\u0631\\u062d\\u0628\\u0627 C++ scanner\"}, {\"role\": \"user\", \"content\": \"that hand every exactly \\u6771\\u4eac $1,234.56 tokens {braces} body we'll engine quick once \\u6771\\u4eac on F# v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 I'M over over scanner tiktoken's tokens na\\u00efve that piece tiktoken's once and scanned scanned request with it jumps counting {braces} scanner 100% WON'T scanned exactly is piece 'single' request \\u6771\\u4eac regex for quick written 'single' every F# we'll way na\\u00efve and we'll tokens v1.2.3 regex piece v1.2.3 scanned is with counting with dog gateway admission 100% lazy because that \\ud83d\\ude42 3.14159 F# way hand exactly $1,234.56 while user@example.com It's jumps faster reservation It's don't it mirrors scanned $1,234.56 it while keep $1,234.56 fox $1,234.56 backtracking involved admission 3.14159 mirrors while with on over mirrors C++ for there's fox budget a keep\"}, {\"role\": \"assistant\", \"content\": \"100% scanner user@example.com backtracking every na\\u00efve quick hand jumps body tokens for v1.2.3 exactly \\u0645\\u0631\\u062d\\u0628\\u0627 that \\\"quotes\\\" \\ud83d\\ude42 gateway no caf\\u00e9 that written https://example.com/a/b?c=d the engine piece They'RE budget https://example.com/a/b?c=d involved the https://example.com/a/b?c=d the for involved once exactly every written exactly while 3.14159 user@example.com backtracking request fox is engine way budget counting 1999 piece so 3.14159 tokens F# involved brown a \\u6771\\u4eac scanned keep v1.2.3 for tiktoken's 1999 They'RE and on backtracking counting boundaries na\\u00efve regex\"}, {\"role\": \"user\", \"content\": \"fox is regex https://example.com/a/b?c=d is node.js mirrors user@example.com 100% involved request piece with \\u6771\\u4eac admission the on quick way \\u6771\\u4eac backtracking keep tiktoken's because F# lazy 100% WON'T tiktoken's Z\\u00fcrich involved quick reservation [brackets] written don't {braces} na\\u00efve caf\\u00e9 a $1,234.56 {braces} regex hand 100% 100% so reservation is there's engine \\u0645\\u0631\\u062d\\u0628\\u0627 is 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking v1.2.3 we'll so quick is before\"}, {\"role\": \"assistant\", \"content\": \"[brackets] request lazy C++ 100% scanned F# a because budget They'RE 1999 that is so \\u0645\\u0631\\u062d\\u0628\\u0627 I'M hand involved 3.14159 admission node.js faster once admission tiktoken's jumps is hand quick tokens mirrors lazy It's 42 fox gateway because request once $1,234.56 mirrors there's so backtracking a over engine we'll \\ud83d\\ude42 body engine fox {braces} {braces} counting hand we'll (parens) that 'single' \\\"quotes\\\" lazy budget na\\u00efve lazy WON'T body v1.2.3 the (parens) \"}, {\"role\": \"user\", \"content\": \"for before quick v1.2.3 scanner tokens faster is hand gateway lazy backtracking C++ budget \\u6771\\u4eac brown (parens) They'RE \\u6771\\u4eac It's \\ud83d\\ude42 3.14159 tokens na\\u00efve backtracking (parens) don't WON'T request WON'T that \\u6771\\u4eac piece v1.2.3 mirrors engine the over mirrors na\\u00efve gateway Z\\u00fcrich C++ written 3.14159 dog involved written \\\"quotes\\\" keep \\\"quotes\\\" They'RE fox written 'single' for backtracking gateway written 1999 the a for F# we'll F# brown faster once engine https://example.com/a/b?c=d it quick piece {braces} \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors [brackets] boundaries $1,234.56 {braces} request keep piece involved body quick once for for They'RE that involved scanned on for 100% mirrors involved budget 3.14159 They'RE faster with \\u6771\\u4eac backtracking reservation the 1999 for \\u0645\\u0631\\u062d\\u0628\\u0627 \\u0645\\u0631\\u062d\\u0628\\u0627 42 \\ud83d\\ude42 keep no caf\\u00e9 engine\"}, {\"role\": \"assistant\", \"content\": \"that tokens dog exactly piece Z\\u00fcrich quick the request 100% every it user@example.com that https://example.com/a/b?c=d way the reservation 1999 F# while a that They'RE backtracking It's Z\\u00fcrich admission budget C++ They'RE we'll keep counting \\ud83d\\ude42 \\\"quotes\\\" that \\ud83d\\ude42 lazy the engine They'RE \\u6771\\u4eac quick involved while no jumps the https://example.com/a/b?c=d is that 3.14159 100% 42 the no while tiktoken's over while written I'M backtracking node.js tokens F# $1,234.56 caf\\u00e9 it while keep (parens) boundaries admission so over no that fox involved F# brown na\\u00efve $1,234.56 reservation lazy body mirrors every $1,234.56 v1.2.3 that scanned user@example.com before quick body with [brackets] brown tiktoken's v1.2.3 fox budget \\\"quotes\\\" 42 every tokens budget don't counting counting no tiktoken's request C++ It's I'M and \\\"quotes\\\" tiktoken's every (parens) user@example.com the the we'll no I'M boundaries F# over on over tokens\"}, {\"role\": \"user\", \"content\": \"tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 It's counting piece $1,234.56 42 once counting there's faster [brackets] 100% piece counting quick \\ud83d\\ude42 caf\\u00e9 exactly written node.js is way on lazy counting it 42 scanned lazy It's v1.2.3 dog dog and because tiktoken's lazy \\ud83d\\ude42 don't involved and once 3.14159 1999 reservation boundaries caf\\u00e9 written written reservation we'll regex {braces} it the https://example.com/a/b?c=d lazy a once body na\\u00efve 1999 tiktoken's caf\\u00e9 caf\\u00e9 for body WON'T \\ud83d\\ude42 no because regex lazy 3.14159 with dog request scanned and involved involved scanned keep that fox https://example.com/a/b?c=d on we'll we'll budget boundaries exactly exactly admission lazy backtracking that F# while fox 1999 boundaries request \\u6771\\u4eac mirrors brown brown 100% Z\\u00fcrich {braces} They'RE 3.14159 v1.2.3 there's v1.2.3 once 1999 faster and WON'T there's we'll on written C++ v1.2.3\"}, {\"role\": \"assistant\", \"content\": \"user@example.com \\u6771\\u4eac because before so keep written for WON'T the They'RE is hand \\ud83d\\ude42 tiktoken's on 42 F# we'll written body user@example.com \\u6771\\u4eac 1999 admission with reservation 42 and that before 3.14159 piece request exactly 1999 for node.js gateway 100% no hand hand it and admission body scanner https://example.com/a/b?c=d involved user@example.com dog boundaries on while 42 body written don't once keep that node.js on reservation \\u0645\\u0631\\u062d\\u0628\\u0627 lazy \\\"quotes\\\" They'RE that with regex while Z\\u00fcrich na\\u00efve that lazy caf\\u00e9 don't jumps before {braces} don't for https://example.com/a/b?c=d keep user@example.com once lazy because counting brown written that 'single' F# \\u6771\\u4eac regex tiktoken's admission I'M a that It's exactly while with so hand don't we'll on\"}, {\"role\": \"user\", \"content\": \"{braces} (parens) fox on over tiktoken's budget lazy for \\\"quotes\\\" 100% so because 100% boundaries Z\\u00fcrich gateway boundaries don't jumps faster don't F# the \\\"quotes\\\" {braces} tokens backtracking They'RE regex every tokens the \\\"quotes\\\" once user@example.com {braces} every \\\"quotes\\\" counting budget It's gateway there's exactly regex is node.js request for F# piece so don't quick 'single' is user@example.com once I'M because user@example.com before user@example.com it v1.2.3 piece a lazy scanned 100% lazy counting dog and we'll we'll for tiktoken's exactly no fox once 1999 \\u6771\\u4eac we'll every the because regex $1,234.56 100% scanner so \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 on while that na\\u00efve body brown way 3.14159 scanner scanner on They'RE don't over because the 42 involved node.js that a They'RE caf\\u00e9 for \\ud83d\\ude42 user@example.com user@example.com I'M so faster\"}, {\"role\": \"assistant\", \"content\": \"a don't It's gateway fox for \\u6771\\u4eac exactly that exactly I'M piece https://example.com/a/b?c=d \\ud83d\\ude42 faster so once jumps reservation piece boundaries v1.2.3 scanned a while that user@example.com [brackets] keep no there's brown the 1999 backtracking They'RE brown node.js v1.2.3 tiktoken's with is v1.2.3 that hand so I'M and engine boundaries C++ user@example.com tokens keep na\\u00efve {braces} gateway because backtracking [brackets] gateway don't we'll for we'll 100% boundaries counting 3.14159 is the F# it fox node.js for is tiktoken's no backtracking engine and 'single' and involved before fox tokens It's https://example.com/a/b?c=d mirrors written boundaries (parens) reservation the tiktoken's (parens) fox hand body (parens) (parens) it is It's and way because {braces} dog tiktoken's the counting while \\ud83d\\ude42 42 scanned over the\"}, {\"role\": \"user\", \"content\": \"it brown brown tokens backtracking jumps over boundaries with quick counting faster gateway {braces} \\u6771\\u4eac 42 mirrors \\ud83d\\ude42 engine hand \\u6771\\u4eac reservation v1.2.3 quick no no that involved over the scanner faster every mirrors for admission I'M we'll and scanned node.js tokens brown budget 42 written gateway lazy 3.14159 engine it we'll it 3.14159 $1,234.56 the that it (parens) over \\u0645\\u0631\\u062d\\u0628\\u0627 node.js for that {braces} \\\"quotes\\\" way before that no quick tokens and scanner the WON'T boundaries dog C++ boundaries while F# once backtracking 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 a while counting node.js Z\\u00fcrich engine fox faster it F# 1999 gateway written It's 'single' na\\u00efve there's 3.14159 na\\u00efve Z\\u00fcrich brown 3.14159 \\u6771\\u4eac user@example.com WON'T exactly Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 node.js 100% gateway regex while user@example.com and no quick lazy with\"}, {\"role\": \"assistant\", \"content\": \"1999 100% dog so for mirrors tiktoken's backtracking on that fox I'M every for involved involved \\\"quotes\\\" 100% so while 42 and it for F# node.js that gateway backtracking $1,234.56 fox boundaries on budget backtracking mirrors body Z\\u00fcrich for on 42 hand 1999 counting way body brown with scanned 42 [brackets] a jumps I'M written there's no They'RE faster backtracking I'M the don't once admission scanner 'single' \\ud83d\\ude42 way on\"}, {\"role\": \"user\", \"content\": \"every regex \\ud83d\\ude42 3.14159 we'll fox it tiktoken's Z\\u00fcrich \\\"quotes\\\" user@example.com na\\u00efve quick \\u0645\\u0631\\u062d\\u0628\\u0627 fox fox it C++ fox \\u0645\\u0631\\u062d\\u0628\\u0627 body gateway 3.14159 100% fox there's keep They'RE so reservation it lazy with lazy gateway request keep scanned hand every fox budget for quick $1,234.56 piece piece faster hand we'll 42 user@example.com written 1999 \\u6771\\u4eac every na\\u00efve request It's involved caf\\u00e9 [brackets] faster scanned keep 42 It's \\u6771\\u4eac every way is 3.14159 written mirrors brown WON'T once WON'T exactly the They'RE there's scanner because [brackets] gateway v1.2.3 fox tiktoken's 'single' Z\\u00fcrich {braces} jumps caf\\u00e9 I'M https://example.com/a/b?c=d C++ boundaries \\\"quotes\\\" with written we'll quick a with the every way involved scanned so scanner request scanner $1,234.56 exactly the caf\\u00e9 so there's is lazy that jumps user@example.com so with don't faster 1999 request admission They'RE while boundaries reservation\"}, {\"role\": \"assistant\", \"content\": \"jumps no that dog \\u0645\\u0631\\u062d\\u0628\\u0627 the lazy request user@example.com dog with lazy Z\\u00fcrich and tokens \\u6771\\u4eac boundaries engine keep boundaries 'single' so C++ \\ud83d\\ude42 gateway \\ud83d\\ude42 mirrors F# on {braces} once the before body user@example.com boundaries na\\u00efve and keep counting \\\"quotes\\\" while hand request body tiktoken's request we'll so I'M \\\"quotes\\\" on {braces} https://example.com/a/b?c=d They'RE is regex don't dog They'RE (parens) once caf\\u00e9 involved written faster keep before scanner while They'RE boundaries admission once gateway reservation every caf\\u00e9 dog tokens regex 'single' 3.14159 there's counting that tokens dog https://example.com/a/b?c=d piece F# before engine scanner exactly \\\"quotes\\\" don't while\"}, {\"role\": \"user\", \"content\": \"WON'T node.js v1.2.3 involved mirrors on brown keep every tiktoken's F# 100% 'single' They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 na\\u00efve don't 3.14159 \\\"quotes\\\" there's once I'M $1,234.56 dog tokens that the Z\\u00fcrich \\\"quotes\\\" exactly C++ 3.14159 (parens) v1.2.3 no It's no keep user@example.com 42 Z\\u00fcrich a over written tiktoken's keep tiktoken's written with no WON'T $1,234.56 scanner keep keep that fox regex admission every [brackets] WON'T \\u6771\\u4eac 42 engine (parens) C++ we'll tokens a it while [brackets] counting involved the \\u0645\\u0631\\u062d\\u0628\\u0627 lazy 'single' WON'T 3.14159 1999 $1,234.56 regex that 3.14159 because caf\\u00e9 mirrors \\\"quotes\\\" counting 3.14159 with exactly admission every They'RE because 100% body piece na\\u00efve regex counting I'M that admission lazy \\ud83d\\ude42 scanner quick fox we'll brown admission C++\"}, {\"role\": \"assistant\", \"content\": \"\\ud83d\\ude42 the that \\u6771\\u4eac exactly while engine counting \\ud83d\\ude42 42 admission admission once caf\\u00e9 for tokens v1.2.3 jumps piece piece scanner caf\\u00e9 is \\\"quotes\\\" boundaries so jumps hand so while exactly involved written lazy (parens) quick F# with 1999 C++ jumps Z\\u00fcrich faster because quick the scanned C++ because because F# for no there's keep caf\\u00e9 jumps while is tokens quick lazy C++ boundaries 'single' exactly once tiktoken's node.js Z\\u00fcrich https://example.com/a/b?c=d before exactly {braces} boundaries written no caf\\u00e9 100% dog regex WON'T counting dog jumps F# for tokens no budget scanner that {braces} user@example.com C++ scanned \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 for every way and 'single' mirrors\"}, {\"role\": \"user\", \"content\": \"gateway because tokens jumps \\u6771\\u4eac exactly 3.14159 I'M 'single' It's WON'T body 42 and way is tiktoken's dog that admission on (parens) \\u6771\\u4eac C++ node.js over boundaries piece piece backtracking node.js dog hand on there's C++ 'single' so mirrors (parens) that (parens) the 1999 scanner hand node.js keep (parens) {braces} every [brackets] user@example.com that They'RE over tokens is with every so is quick request piece so reservation 42 brown on \\u6771\\u4eac fox They'RE $1,234.56 They'RE na\\u00efve boundaries jumps 42 quick exactly \\u0645\\u0631\\u062d\\u0628\\u0627 the 3.14159 over na\\u00efve is reservation mirrors it tokens no \\u6771\\u4eac on \\u6771\\u4eac mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 dog for engine before before node.js $1,234.56 100% every no 100% a budget backtracking jumps there's faster there's boundaries tokens every tiktoken's way F# $1,234.56 that every exactly piece\"}, {\"role\": \"assistant\", \"content\": \"over 3.14159 body 100% admission is F# tiktoken's user@example.com once with \\u0645\\u0631\\u062d\\u0628\\u0627 no with quick \\\"quotes\\\" the and $1,234.56 with faster exactly is involved keep exactly before 'single' there's body regex for involved counting https://example.com/a/b?c=d every over user@example.com C++ there's there's They'RE I'M scanned https://example.com/a/b?c=d scanned hand \\ud83d\\ude42 every counting don't fox 'single' WON'T na\\u00efve 100% gateway 'single' I'M no once involved node.js 1999 every admission caf\\u00e9 regex with regex written mirrors $1,234.56 tokens while backtracking we'll on it dog na\\u00efve that I'M every [brackets] a tokens I'M engine \\u0645\\u0631\\u062d\\u0628\\u0627 F# body don't request admission mirrors WON'T keep Z\\u00fcrich engine $1,234.56 request $1,234.56\"}, {\"role\": \"user\", \"content\": \"It's way body mirrors involved written 100% WON'T [brackets] no boundaries piece brown and while over before is there's we'll jumps scanned so \\ud83d\\ude42 backtracking keep don't dog and 42 that \\u0645\\u0631\\u062d\\u0628\\u0627 over and before and \\u6771\\u4eac on for (parens) https://example.com/a/b?c=d the request user@example.com I'M \\u0645\\u0631\\u062d\\u0628\\u0627 [brackets] 3.14159 so Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 involved the hand once there's caf\\u00e9 jumps is body gateway v1.2.3 budget They'RE that it I'M 'single' gateway scanner that brown brown They'RE {braces} caf\\u00e9 hand admission budget scanner $1,234.56 mirrors hand 1999 scanned 3.14159 a It's\"}, {\"role\": \"assistant\", \"content\": \"regex Z\\u00fcrich https://example.com/a/b?c=d lazy engine keep regex admission with before body \\\"quotes\\\" boundaries once tiktoken's gateway 'single' and the way I'M $1,234.56 backtracking the engine dog Z\\u00fcrich 3.14159 Z\\u00fcrich user@example.com fox that keep while over once {braces} \\ud83d\\ude42 admission tokens regex tokens budget no I'M that on caf\\u00e9 faster I'M dog 'single' dog we'll with quick no because tiktoken's that counting tiktoken's we'll involved node.js involved with C++ body (parens) dog over before v1.2.3 100% na\\u00efve while request F# every jumps gateway over a that for it body exactly Z\\u00fcrich \\u6771\\u4eac a mirrors caf\\u00e9 I'M that Z\\u00fcrich brown that https://example.com/a/b?c=d no 'single' boundaries \\u0645\\u0631\\u062d\\u0628\\u0627 we'll They'RE exactly request once lazy It's 3.14159 before \\\"quotes\\\" \\u6771\\u4eac scanned scanned It's jumps C++ They'RE C++ and\"}, {\"role\": \"user\", \"content\": \"the fox They'RE budget faster with brown 3.14159 because backtracking \\ud83d\\ude42 that because reservation the so {braces} is WON'T request 3.14159 100% over a \\u0645\\u0631\\u062d\\u0628\\u0627 brown while admission because there's and once tokens [brackets] is no with a over there's over written scanned don't exactly I'M gateway is boundaries it tokens faster exactly tokens 42 lazy node.js https://example.com/a/b?c=d is engine with\"}, {\"role\": \"assistant\", \"content\": \"3.14159 I'M gateway scanner tokens C++ node.js 1999 engine for jumps user@example.com It's request (parens) so [brackets] \\ud83d\\ude42 that it admission 3.14159 budget WON'T caf\\u00e9 that involved budget v1.2.3 [brackets] node.js They'RE mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 request with v1.2.3 WON'T we'll admission brown don't 1999 \\u6771\\u4eac \\u6771\\u4eac F# for the scanner is 'single' is it piece tokens body body Z\\u00fcrich tiktoken's we'll involved \\u6771\\u4eac on no tokens on {braces} It's mirrors I'M [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 lazy Z\\u00fcrich a while caf\\u00e9 involved request $1,234.56 tiktoken's once lazy\"}, {\"role\": \"user\", \"content\": \"3.14159 lazy piece fox on lazy v1.2.3 dog while piece 'single' involved written 100% jumps hand scanner v1.2.3 don't 'single' scanned with piece don't They'RE piece (parens) for with F# scanned don't no backtracking written 100% 100% lazy backtracking involved user@example.com scanner way we'll 42 'single' reservation 'single' lazy the written don't 'single' every and before reservation so it Z\\u00fcrich before a node.js before F# involved jumps counting involved that scanned regex so that user@example.com it with \\u0645\\u0631\\u062d\\u0628\\u0627 involved admission caf\\u00e9 the reservation [brackets] over mirrors that hand {braces} \\u0645\\u0631\\u062d\\u0628\\u0627 don't that we'll [brackets] hand faster I'M regex boundaries piece keep reservation WON'T hand backtracking counting piece $1,234.56 \\\"quotes\\\" Z\\u00fcrich tokens jumps it body before don't scanner {braces} body It's 42 caf\\u00e9 $1,234.56 scanner tokens\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 engine involved faster involved keep with $1,234.56 while piece F# Z\\u00fcrich 100% tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 is budget that tiktoken's admission scanned node.js 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 that on on body dog with regex admission 3.14159 piece na\\u00efve so 'single' mirrors boundaries boundaries keep on no while \\u0645\\u0631\\u062d\\u0628\\u0627 reservation the node.js F# 42 piece reservation It's tiktoken's reservation $1,234.56 hand tiktoken's that I'M it \\ud83d\\ude42 (parens) counting Z\\u00fcrich is [brackets] https://example.com/a/b?c=d reservation boundaries lazy (parens) that with over Z\\u00fcrich don't counting\"}, {\"role\": \"user\", \"content\": \"body dog backtracking that we'll don't \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com keep quick before scanned caf\\u00e9 It's body quick na\\u00efve 100% before quick jumps \\\"quotes\\\" once regex 42 I'M \\ud83d\\ude42 and keep a faster brown caf\\u00e9 regex \\\"quotes\\\" we'll engine exactly so 42 written They'RE every caf\\u00e9 the the once na\\u00efve piece no hand that boundaries scanned before there's there's reservation https://example.com/a/b?c=d admission I'M v1.2.3 dog They'RE dog scanned engine It's user@example.com no reservation \\u0645\\u0631\\u062d\\u0628\\u0627 regex scanned Z\\u00fcrich is it a and \\\"quotes\\\" involved exactly exactly way hand jumps \\ud83d\\ude42 reservation quick tokens mirrors over user@example.com boundaries 3.14159 I'M node.js piece F# https://example.com/a/b?c=d before tiktoken's I'M piece brown is na\\u00efve and {braces} WON'T It's don't it is {braces} on so so \\u6771\\u4eac it it user@example.com brown \\\"quotes\\\" I'M 100% scanned C++ body tokens \\u6771\\u4eac [brackets]\"}, {\"role\": \"assistant\", \"content\": \"mirrors F# $1,234.56 for dog (parens) keep that [brackets] (parens) 3.14159 v1.2.3 WON'T is don't the a It's https://example.com/a/b?c=d faster 'single' it user@example.com is dog dog counting so $1,234.56 v1.2.3 boundaries every on regex backtracking before quick brown quick C++ quick v1.2.3 scanner written \\ud83d\\ude42 user@example.com before is 42 3.14159 on engine dog dog once request https://example.com/a/b?c=d lazy on caf\\u00e9 faster don't before is exactly written brown C++ piece a 3.14159 no WON'T faster it https://example.com/a/b?c=d and boundaries \\ud83d\\ude42 reservation that engine we'll fox while the dog v1.2.3 because boundaries scanned It's \\u6771\\u4eac \\\"quotes\\\" scanned that so is tiktoken's scanned {braces} gateway exactly regex quick over is fox It's we'll {braces} faster\"}, {\"role\": \"user\", \"content\": \"Summarise the conversation so far in three sentences.\"}]}", "input_tokens": 50354} diff --git a/litellm-rust/crates/token-counter/tests/fixtures/cl100k/texts.jsonl b/litellm-rust/crates/token-counter/tests/fixtures/cl100k/texts.jsonl new file mode 100644 index 00000000000..3dd681d43f9 --- /dev/null +++ b/litellm-rust/crates/token-counter/tests/fixtures/cl100k/texts.jsonl @@ -0,0 +1,4053 @@ +{"text": "", "tokens": 0, "pieces": []} +{"text": "Hello, how are you today?", "tokens": 7, "pieces": ["Hello", ",", " how", " are", " you", " today", "?"]} +{"text": "I'm sure they're right, we'll see. WE'LL SEE, I'M SURE THEY'RE RIGHT, IT'S HERS AND IT'D BE 'D", "tokens": 34, "pieces": ["I", "'m", " sure", " they", "'re", " right", ",", " we", "'ll", " see", ".", " WE", "'LL", " SEE", ",", " I", "'M", " SURE", " THEY", "'RE", " RIGHT", ",", " IT", "'S", " HERS", " AND", " IT", "'D", " BE", " '", "D"]} +{"text": "don't Don'T DON'T won'T i've I'VE i'Ve you'RE 'S 'T 'M 'D 'LL 'VE 'RE 'ſ 'x", "tokens": 37, "pieces": ["don", "'t", " Don", "'T", " DON", "'T", " won", "'T", " i", "'ve", " I", "'VE", " i", "'Ve", " you", "'RE", " '", "S", " '", "T", " '", "M", " '", "D", " '", "LL", " '", "VE", " '", "RE", " '", "ſ", " '", "x"]} +{"text": "1234567890 123 12 1 0000000 ٣٤٥٦٧٨ ३४५६ 1,234,567.89 2026-09-11T18:00:00Z", "tokens": 58, "pieces": ["123", "456", "789", "0", " ", "123", " ", "12", " ", "1", " ", "000", "000", "0", " ", "٣٤٥", "٦٧٨", " ", "३४५", "६", " ", "1", ",", "234", ",", "567", ".", "89", " ", "202", "6", "-", "09", "-", "11", "T", "18", ":", "00", ":", "00", "Z"]} +{"text": "$abc %def &ghi @jkl _mno #pqr ~stu ^vwx |yz \\a /b :c ;d ?e !f (g )h [i ]j {k }l n =o +p *q", "tokens": 56, "pieces": ["$abc", " %", "def", " &", "ghi", " @", "jkl", " _", "mno", " #", "pqr", " ~", "stu", " ^", "vwx", " |", "yz", " \\", "a", " /", "b", " :", "c", " ;", "d", " ?", "e", " !", "f", " (", "g", " )", "h", " [", "i", " ]", "j", " {", "k", " }", "l", " <", "m", " >", "n", " =", "o", " +", "p", " *", "q"]} +{"text": "foo bar baz \t qux\t\tquux \n\nline\r\nline\r\n\r\n \n\t\r\n x ", "tokens": 22, "pieces": ["foo", " ", " bar", " ", " baz", " \t", " qux", "\t", "\tquux", " \n\n", "line", "\r\n", "line", "\r\n\r\n \n\t\r\n", " ", " x", " "]} +{"text": "trailing spaces ", "tokens": 4, "pieces": ["trailing", " spaces", " "]} +{"text": "trailing tabs\t\t", "tokens": 4, "pieces": ["trailing", " tabs", "\t\t"]} +{"text": "trailing newline\n", "tokens": 4, "pieces": ["trailing", " newline", "\n"]} +{"text": "\n\n\n", "tokens": 1, "pieces": ["\n\n\n"]} +{"text": "\r\n\r\n\r\n", "tokens": 1, "pieces": ["\r\n\r\n\r\n"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "😀😃😄 👍🏽 🇺🇸 👨‍👩‍👧‍👦 ✈️ ❤️‍🔥 ٭ ※ ⌘ ⏎", "tokens": 54, "pieces": ["😀😃😄", " 👍🏽", " 🇺🇸", " 👨‍👩‍👧‍👦", " ✈️", " ❤️‍🔥", " ٭", " ※", " ⌘", " ⏎"]} +{"text": "漢字かな交じり文、東京都千代田区。日本語のテキストです。中文测试。한국어 텍스트", "tokens": 42, "pieces": ["漢字かな交じり文", "、東京都千代田区", "。日本語のテキストです", "。中文测试", "。한국어", " 텍스트"]} +{"text": "مرحبا بالعالم، هذا نص عربي مع أرقام ١٢٣٤٥٦٧ و علامات ترقيم!", "tokens": 52, "pieces": ["مرحبا", " بالعالم", "،", " هذا", " نص", " عربي", " مع", " أرقام", " ", "١٢٣", "٤٥٦", "٧", " و", " علامات", " ترقيم", "!"]} +{"text": "Zürich, façade, naïve, Ærøskøbing, Ελληνικά, Русский текст, עברית, हिन्दी, ไทย", "tokens": 52, "pieces": ["Zürich", ",", " façade", ",", " naïve", ",", " Ærøskøbing", ",", " Ελληνικά", ",", " Русский", " текст", ",", " עברית", ",", " ह", "िन", "्द", "ी,", " ไทย"]} +{"text": "é å ḍ̇ ́́ combining̈ markś!", "tokens": 19, "pieces": ["e", "́", " a", "̊", " ḋ", "̣", " ́́", " combining", "̈", " marks", "́!"]} +{"text": "ΣΊΣΥΦΟΣ Džungla İstanbul file flow Abc ㍿ ㋿ ꟲ 𐞁", "tokens": 50, "pieces": ["ΣΊΣΥΦΟΣ", " Džungla", " İstanbul", " file", " flow", " Abc", " ㍿", " ㋿", " ꟲ", " 𐞁"]} +{"text": "<|endoftext|> <|fim_prefix|>code<|fim_middle|>more<|fim_suffix|> <|endofprompt|> <|im_start|>", "tokens": 40, "pieces": ["<|", "endoftext", "|>", " <|", "fim", "_prefix", "|>", "code", "<|", "fim", "_middle", "|>", "more", "<|", "fim", "_suffix", "|>", " <|", "endofprompt", "|>", " <|", "im", "_start", "|>"]} +{"text": " [INST] [/INST] <>", "tokens": 22, "pieces": ["", " <", "META", "_START", ">", " <", "s", ">", " ", " [", "INST", "]", " [/", "INST", "]", " <<", "SYS", ">>"]} +{"text": "def f(x):\n return {'a': x ** 2, \"b\": [1, 2, 3]} # comment\n\nprint(f(10))\n", "tokens": 35, "pieces": ["def", " f", "(x", "):\n", " ", " return", " {'", "a", "':", " x", " **", " ", "2", ",", " \"", "b", "\":", " [", "1", ",", " ", "2", ",", " ", "3", "]}", " ", " #", " comment", "\n\n", "print", "(f", "(", "10", "))\n"]} +{"text": "{\"model\":\"gpt-4\",\"messages\":[{\"role\":\"user\",\"content\":\"hi\\n\"}],\"temperature\":0.7}", "tokens": 26, "pieces": ["{\"", "model", "\":\"", "gpt", "-", "4", "\",\"", "messages", "\":[{\"", "role", "\":\"", "user", "\",\"", "content", "\":\"", "hi", "\\n", "\"}],\"", "temperature", "\":", "0", ".", "7", "}"]} +{"text": "https://example.com/path?query=1&other=two#fragment user@example.com 192.168.0.1", "tokens": 26, "pieces": ["https", "://", "example", ".com", "/path", "?query", "=", "1", "&other", "=two", "#fragment", " user", "@example", ".com", " ", "192", ".", "168", ".", "0", ".", "1"]} +{"text": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", "tokens": 375, "pieces": ["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"]} +{"text": " ", "tokens": 24, "pieces": [" "]} +{"text": "........................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................", "tokens": 48, "pieces": ["........................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................"]} +{"text": "abababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababab", "tokens": 1500, "pieces": ["abababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababab"]} +{"text": "\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n", "tokens": 95, "pieces": ["\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n"]} +{"text": "000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000", "tokens": 1000, "pieces": ["000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000"]} +{"text": "!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!", "tokens": 375, "pieces": ["!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!"]} +{"text": "😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀", "tokens": 2000, "pieces": ["😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀"]} +{"text": "漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢", "tokens": 2000, "pieces": ["漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢"]} +{"text": " abc ! 
x  y ​z ‍‍ q", "tokens": 18, "pieces": [" abc", " ", "!", " ", "
x", " ", " y", " ​", "z", " ‍‍", " q"]} +{"text": "x…y \u000b\f z", "tokens": 8, "pieces": ["x", "…y", " \u000b\f", " z"]} +{"text": "\u0000\u0001\u0002  �", "tokens": 6, "pieces": ["\u0000\u0001\u0002", " ", " �"]} +{"text": "tab\tseparated\tvalues\n1\t2\t3\n", "tokens": 11, "pieces": ["tab", "\tseparated", "\tvalues", "\n", "1", "\t", "2", "\t", "3", "\n"]} +{"text": "MiXeD cAsE wOrDs AND ACRONYMS like NASA, HTTP/2, gRPC, iOS, macOS", "tokens": 28, "pieces": ["MiXeD", " cAsE", " wOrDs", " AND", " ACRONYMS", " like", " NASA", ",", " HTTP", "/", "2", ",", " gRPC", ",", " iOS", ",", " macOS"]} +{"text": "snake_case_identifier camelCaseIdentifier PascalCaseIdentifier SCREAMING_SNAKE_CASE kebab-case", "tokens": 19, "pieces": ["snake", "_case", "_identifier", " camelCaseIdentifier", " PascalCaseIdentifier", " SCREAMING", "_SNAKE", "_CASE", " kebab", "-case"]} +{"text": "x'sy x'ty x'rey x'vey x'my x'lly x'dy x'S x'T x'RE x'VE x'M x'LL x'D x'sS x'llL", "tokens": 43, "pieces": ["x", "'s", "y", " x", "'t", "y", " x", "'re", "y", " x", "'ve", "y", " x", "'m", "y", " x", "'ll", "y", " x", "'d", "y", " x", "'S", " x", "'T", " x", "'RE", " x", "'VE", " x", "'M", " x", "'LL", " x", "'D", " x", "'s", "S", " x", "'ll", "L"]} +{"text": "IT'SOK it'Dbe x'Sy x'Ty x'My x'Dy x'LLy x'VEy x'REy x'Ly x'Vy x'Ry 'Sx'Tx'Mx'LLx'VEx'REx'Dx", "tokens": 55, "pieces": ["IT", "'S", "OK", " it", "'D", "be", " x", "'S", "y", " x", "'T", "y", " x", "'M", "y", " x", "'D", "y", " x", "'LL", "y", " x", "'VE", "y", " x", "'RE", "y", " x", "'Ly", " x", "'Vy", " x", "'Ry", " '", "Sx", "'T", "x", "'M", "x", "'LL", "x", "'VE", "x", "'RE", "x", "'D", "x"]} +{"text": "'s't're've'm'll'd 'S'T'RE'VE'M'LL'D ''s '''s", "tokens": 21, "pieces": ["'s", "'t", "'re", "'ve", "'m", "'ll", "'d", " '", "S", "'T", "'RE", "'VE", "'M", "'LL", "'D", " ''", "s", " '''", "s"]} +{"text": "9'9 9's a'9 '9 ' 's' ' 's", "tokens": 18, "pieces": ["9", "'", "9", " ", "9", "'s", " a", "'", "9", " '", "9", " '", " '", "s", "'", " '", " '", "s"]} +{"text": "١٢٣٤ ½⅓¼ ⅣⅤ 𝟘𝟙𝟚𝟛𝟜𝟝𝟞𝟟𝟠𝟡 ①②③", "tokens": 56, "pieces": ["١٢٣", "٤", " ", "½⅓¼", " ", "ⅣⅤ", " ", "𝟘𝟙𝟚", "𝟛𝟜𝟝", "𝟞𝟟𝟠", "𝟡", " ", "①②③"]} +{"text": "camelCase PascalCase ABCdef ABCdeF ABC aB Ab ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyzABC", "tokens": 21, "pieces": ["camelCase", " PascalCase", " ABCdef", " ABCdeF", " ABC", " aB", " Ab", " ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyzABC"]} +{"text": "日本ABC ABC日本 日本語abc abc日本語 漢字Kanji kanji漢字 KANJI漢字kanji مرحباABC ABCمرحبا abcمرحبا", "tokens": 53, "pieces": ["日本ABC", " ABC日本", " 日本語abc", " abc日本語", " 漢字Kanji", " kanji漢字", " KANJI漢字kanji", " مرحباABC", " ABCمرحبا", " abcمرحبا"]} +{"text": "́ABC ́abc ́́A Á́ ÉA aÉ !!́a  ́A ẍY Ẍy", "tokens": 33, "pieces": ["́ABC", " ́", "abc", " ́́", "A", " A", "́́", " E", "́A", " aE", "́", " !!́", "a", " ", " ", "́A", " x", "̈Y", " X", "̈y"]} +{"text": "ᵃbc ᵃBC Aᵃbc Aᵃ ᵃ' ᵃ's Džungla aDžB ADžB ADžb DžDž Ljx İi ΣΊΣΥΦΟΣσ ΣσΣ", "tokens": 73, "pieces": ["ᵃbc", " ᵃBC", " Aᵃbc", " Aᵃ", " ᵃ", "'", " ᵃ", "'s", " Džungla", " aDžB", " ADžB", " ADžb", " DžDž", " Ljx", " İi", " ΣΊΣΥΦΟΣσ", " ΣσΣ"]} +{"text": "don'tx ABC's abc'S abc'ſ ABC'ſx IT'SOK it'Dbe 'sabc x's 's 'Sx'Tx x’s X'LLx X'Ll", "tokens": 43, "pieces": ["don", "'t", "x", " ABC", "'s", " abc", "'S", " abc", "'ſ", " ABC", "'ſ", "x", " IT", "'S", "OK", " it", "'D", "be", " '", "sabc", " x", "'s", " '", "s", " '", "Sx", "'T", "x", " x", "’s", " X", "'LL", "x", " X", "'Ll"]} +{"text": "!ABC !AbC !!abc #camelCase (ABCdef)  ABC abc Abc \tABC\tabc", "tokens": 27, "pieces": ["!ABC", " !", "AbC", " !!", "abc", " #", "camelCase", " (", "ABCdef", ")", " ", " ABC", " abc", " Abc", " ", "\tABC", "\tabc"]} +{"text": "!!/\n/x a/b !!\n/x /x // path/to/file.rs http://x.y/z?a=b/c \\/\\/ //\r\n//\n", "tokens": 29, "pieces": ["!!/\n", "/x", " a", "/b", " !!\n", "/x", " ", " /", "x", " ", " //", " path", "/to", "/file", ".rs", " http", "://", "x", ".y", "/z", "?a", "=b", "/c", " \\/\\/", " //\r\n", "//\n"]} +{"text": "x \n x \r\n \r\n y x \n a b \n\n c x\t\ty x\t\t end \n \n", "tokens": 22, "pieces": ["x", " \n", " x", " \r\n \r\n", " y", " x", " \n", " ", " a", " ", " b", " \n\n", " ", " c", " x", "\t", "\ty", " x", "\t\t", " end", " \n \n"]} +{"text": "12345 6 1abc abc1 ABC123abc 123ABC ١٢٣٤٥abc", "tokens": 27, "pieces": ["123", "45", " ", "6", " ", "1", "abc", " abc", "1", " ABC", "123", "abc", " ", "123", "ABC", " ", "١٢٣", "٤٥", "abc"]} +{"text": "Ⅳ٣٤٥٦<|endoftext|>9
Dž#$%", "tokens": 24, "pieces": ["Ⅳ٣٤", "٥٦", "<|", "endoftext", "|>", "9", "
Dž", "#$%"]} +{"text": "́!!ſİ'D​a'll 字0'MZſⅣ ḍ̇éfi㍿𐞁<|endoftext|>'reA'S#$%", "tokens": 46, "pieces": ["́!!", "ſİ", "'D", "​a", "'ll", " 字", "0", "'M", "Zſ", "Ⅳ", " ḋ", "̣éfi", "㍿𐞁", "<|", "endoftext", "|>'", "reA", "'S", "#$%"]} +{"text": "ع字'T\r\ń½sꟲ㋿'VE'S<😀🏽!!12345678 ٣٤٥٦Džſḍ̇\réEOT­'ſ<|endoftext|><|fim_prefix|>ś
\tm", "tokens": 77, "pieces": ["ع字", "'T", "\r\n", "́", "½", "sꟲ", "㋿'", "VE", "'", "S", "<😀🏽!!", "123", "456", "78", " <", "EOT", ">", "٣٤٥", "٦", "Džſḋ", "̣\r", "e", "́EOT", "­'", "ſ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "s", "́", "
", "\tm"]} +{"text": "9'Re\r\n'ſ  'T'Re \né#$%<½!!٣٤٥٦'ſ<\"عßZt\u000bDž'T<|fim_prefix|>ſ㋿\"éſ", "tokens": 61, "pieces": ["9", "'", "Re", "\r\n", "'ſ", "  ", " '", "T", "'Re", " \n", "é", "#$%<", "½", "!!", "٣٤٥", "٦", "'ſ", "<<", "META", "_START", ">\"", "عßZt", "\u000bDž", "'T", "<|", "fim", "_prefix", "|>", "ſ", "㋿\"", "e", "́ſ"]} +{"text": "<|endoftext|>12345678#$%tعⅣ'T0'D<|endoftext|>é'M-'ſ'sß12345678ꟲ0(>\r\n'MZ'Sa'M", "tokens": 57, "pieces": ["<|", "endoftext", "|>", "123", "456", "78", "#$%", "tع", "Ⅳ", "'T", "0", "'D", "<|", "endoftext", "|>", "e", "́'", "M", "-<", "EOT", ">'", "ſ", "'s", "ß", "123", "456", "78", "ꟲ", "0", "(>\r\n", "'M", "Z", "'S", "a", "'M"]} +{"text": " \n'll\"‍ d \nß㍿\u000baⅣ😀🏽́ſ\n \n
'Reİ\tDž٣٤٥٦ع'llEOT.\nݽ٣٤٥٦>å\u000b<|fim_prefix|>\"𐞁", "tokens": 70, "pieces": [" \n", "'ll", "\"‍", " ", " d", " \n", "ß", "㍿", "\u000ba", "Ⅳ", "😀🏽́", "ſ", "\n \n", "
", "'Re", "İ", "\tDž", "٣٤٥", "٦", "ع", "'ll", "EOT", ".\n", "İ", "½٣٤", "٥٦", ">a", "̊", "\u000b", "<|", "fim", "_prefix", "|>\"", "𐞁"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "å'remfi㍿s", "tokens": 11, "pieces": ["a", "̊'", "remfi", "㍿s"]} +{"text": "<|endoftext|>", "tokens": 7, "pieces": ["<|", "endoftext", "|>"]} +{"text": "🙂३'M🙂 12345678'VÉ,\u000bİ<|fim_prefix|> A'TDž 's!!", " ", "A", "'T", "Dž", " ", "'s", "!!<", "t", "\"!", "d", " \n", "m", "'M", "('", "ſ"]} +{"text": "\t", "tokens": 1, "pieces": ["\t"]} +{"text": "…<|fim_prefix|>\n.-åß\r\n\r\nEOT'llå½-fi!é'VE12345678EOT字'VE!🙂<|fim_prefix|>'ſ'D \r\r漢0…", "tokens": 61, "pieces": ["…", "<|", "fim", "_prefix", "|>\n", ".-", "a", "̊ß", "\r\n\r\n", "EOT", "'ll", "a", "̊", "½", "-fi", "!e", "́'", "VE", "123", "456", "78", "EOT字", "'VE", "!🙂<|", "fim", "_prefix", "|>'", "ſ", "'D", " \r\r", "漢", "0", "…"]} +{"text": "a'S'T👍🏽> \n-s㍿.㍿#$%́\r\n\r\n", "tokens": 23, "pieces": ["a", "'S", "'T", "👍🏽>", " \n", "-s", "㍿.㍿#$%́\r\n\r\n"]} +{"text": " 漢#$%­…­ع­ß#$%\t0é", "tokens": 18, "pieces": [" 漢", "#$%­", "…", "­ع", "­ß", "#$%", "\t", "0", "e", "́"]} +{"text": ",9'㍿-", "tokens": 10, "pieces": [",", "9", "'㍿-"]} +{"text": "٣٤٥٦Ⅳ漢İ'sع", "tokens": 15, "pieces": ["٣٤٥", "٦Ⅳ", "漢İ", "'s", "ع"]} +{"text": ".éé'sZ>éEOTZ'0​e½ 0#$%́fiⅣ", "tokens": 25, "pieces": [".e", "́é", "'s", "Z", ">éEOTZ", "'", "0", "​e", "½", " ", " ", "0", "#$%́", "fi", "Ⅳ"]} +{"text": "t<Ⅳ…Dž'VE ꟲ٣٤٥٦éåع㍿'s字👍🏽ع ३EOTⅣ😀🏽e \nA\"", "tokens": 63, "pieces": ["t", "<", "Ⅳ", "…Dž", "'VE", "", " ", " ꟲ", "٣٤٥", "٦", "éa", "̊ع", "㍿'<", "META", "_START", ">s字", "👍🏽", "ع", " ", " ", "३", "EOT", "Ⅳ", "😀🏽", "e", " \n", "A", "\""]} +{"text": "' . t\t<|fim_prefix|>㍿Dž३s😀🏽\t㍿EOTå𐞁​EOT
\t- \ns \n#$%ḍ̇é\r\n ee漢", "tokens": 64, "pieces": ["'", " .", " ", " t", "\t", "<|", "fim", "_prefix", "|>㍿", "Dž", "३", "s", "😀🏽", "\t", "㍿EOTa", "̊𐞁", "​EOT", "
", "\t", "-", " \n", "s", " \n", "#$%", "ḋ", "̣é", "\r\n", " ee漢"]} +{"text": "'ſEOTéⅣ\nİ'S!!!t<|endoftext|>éA<|fim_prefix|>Ⅳ'Z‍'Re'
<|endoftext|>\u000b<㋿ #$%漢A\"ꟲ㍿'T'T", "tokens": 72, "pieces": ["'ſ", "EOTe", "́", "Ⅳ", "\n", "İ", "'S", "!!!", "t", "<|", "endoftext", "|>", "e", "́A", "<|", "fim", "_prefix", "|>", "Ⅳ", "'Z", "‍'", "Re", "'", "
", "<|", "endoftext", "|>", "\u000b", "<㋿", " ", "#$%", "漢A", "\"ꟲ", "㍿'", "T", "'T"]} +{"text": "'Re😀🏽½(.>\u000bſa㍿<|fim_prefix|>>t!'ll३ꟲ \né\n0e\r\n\r\n…😀🏽½́dḍ̇𐞁\r\n\r\n<|fim_prefix|>.<|endoftext|>9", "tokens": 73, "pieces": ["'Re", "😀🏽", "½", "(.>", "\u000bſa", "㍿<|", "fim", "_prefix", "|>>", "t", "!'", "ll", "३", "ꟲ", " \n", "e", "́\n", "0", "e", "\r\n\r\n", "…", "😀🏽", "½", "́dḋ", "̣𐞁", "\r\n\r\n", "<|", "fim", "_prefix", "|>.<|", "endoftext", "|>", "9"]} +{"text": "\r\nå👍🏽!!é", "tokens": 13, "pieces": ["\r\n", "a", "̊👍🏽!!", "e", "́"]} +{"text": "dİ(.عع \n字㍿\nå(Z'ſ㍿\r\n\r\n,<|fim_prefix|>", "tokens": 33, "pieces": ["dİ", "(.", "عع", " \n", "字", "㍿\n", "a", "̊(", "Z", "'ſ", "㍿\r\n\r\n", ",<|", "fim", "_prefix", "|>"]} +{"text": "!", "tokens": 1, "pieces": ["!"]} +{"text": " \n\r\n.s <|endoftext|>aꟲ'sꟲ३…\r\n\r\n\u000bDž‍\t9🙂ſ", "tokens": 33, "pieces": [" \n\r\n", ".s", " <|", "endoftext", "|>", "aꟲ", "'s", "ꟲ", "३", "…\r\n\r\n", "\u000bDž", "‍", "\t", "9", "🙂ſ"]} +{"text": "'re𐞁é३ fi", "tokens": 12, "pieces": ["'re", "𐞁e", "́", "३", " ", " fi"]} +{"text": "12345678 #$%<|fim_prefix|>‍㍿'T😀🏽fi'll's'S12345678½é
,🙂٣٤٥٦#$%👍🏽🙂12345678<Ⅳ!\"'VE
", "tokens": 74, "pieces": ["123", "456", "78", " ", " #$%<|", "fim", "_prefix", "|>‍㍿'", "T", "😀🏽", "fi", "'ll", "'s", "'S", "123", "456", "78½", "e", "́", "
", ",🙂<", "META", "_START", ">", "٣٤٥", "٦", "#$%👍🏽🙂", "123", "456", "78", "<", "Ⅳ", "!\"'", "VE", "
"]} +{"text": "fi>İ𐞁…\u000b'D­\tß👍🏽 Ⅳß'Dé\r\n \nß 👍🏽", "tokens": 43, "pieces": ["fi", ">İ𐞁", "…", "\u000b", "'D", "­", "\t", "ß", "👍🏽", " ", " ", "Ⅳ", "ß", "'D", "e", "́\r\n", " \n", "ß", " ", "👍🏽"]} +{"text": "#$% ßع́'T<A012345678 \n<|fim_prefix|> ㍿👍🏽'", "tokens": 79, "pieces": ["👍🏽>", "A", "012", "345", "678", " \n", "<|", "fim", "_prefix", "|>", " ", "㍿👍🏽'"]} +{"text": "ꟲ'll#$%(\r\n\r\nZ0\u000b👍🏽
'VE ,½'ll­EOT", "tokens": 27, "pieces": ["ꟲ", "'ll", "#$%(\r\n\r\n", "Z", "0", "\u000b", "👍🏽", "
", "'VE", " ", ",", "½", "'ll", "­EOT"]} +{"text": "İ#$%\n9sßEOTd-!!<|endoftext|> 'ReAAfiⅣ'ſéſ🙂étſ\ne'ſ㋿é'VE\"ꟲ漢", "tokens": 56, "pieces": ["İ", "#$%\n", "9", "sßEOTd", "-!!<|", "endoftext", "|>", " ", "'Re", "AAfi", "Ⅳ", "'ſ", "éſ", "🙂étſ", "\n", "e", "'ſ", "㋿é", "'VE", "\"ꟲ漢"]} +{"text": "<३…😀🏽m>'s<|endoftext|>\r\n\r\n३'ſ'S<|endoftext|><|fim_prefix|>Dž🙂ſⅣA㋿-'re#$%!é\r<|fim_prefix|>å9…s'VE", "tokens": 74, "pieces": ["<", "३", "…", "😀🏽", "m", ">'", "s", "<|", "endoftext", "|>\r\n\r\n", "३", "'ſ", "'S", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "Dž", "🙂ſ", "Ⅳ", "A", "㋿-'", "re", "#$%!", "é", "\r", "<|", "fim", "_prefix", "|>", "a", "̊", "9", "…s", "'VE"]} +{"text": " <|endoftext|>\r\t<|endoftext|>…ßſ\n#$%🙂㋿ḍ̇\r\n\r\n", "tokens": 38, "pieces": [" ", "<|", "endoftext", "|>\r", "\t", "<|", "endoftext", "|>", "…ßſ", "\n", "#$%🙂㋿", "ḋ", "̣\r\n\r\n"]} +{"text": " \n(\u000b!!\u000b\r\n\t́'t漢३!\nß \n㍿\t'T,-m-\u000b>ſ \ns३
<|endoftext|>㋿ \n'S'VE>\u000b🙂\r\n", "tokens": 54, "pieces": [" \n", "(", "\u000b", "!!", "\u000b\r\n", "\t", "́'", "t漢", "३", "!\n", "ß", " \n", "㍿", "\t", "'T", ",-", "m", "-", "\u000b", ">ſ", " \n", "s", "३", "
", "<|", "endoftext", "|>㋿", " \n", "'S", "'VE", ">", "\u000b", "🙂\r\n"]} +{"text": "d'Rea'0㍿İé👍🏽s.٣٤٥٦('S\",​
12345678…ꟲ.\r\n\r\n'T㋿ ½'D", "tokens": 59, "pieces": ["d", "'Re", "a", "'", "0", "㍿İé", "👍🏽", "s", ".", "٣٤٥", "٦", "('", "S", "<", "META", "_START", ">\",​", "
", "123", "456", "78", "…ꟲ", ".\r\n\r\n", "<", "EOT", ">'", "T", "㋿", " ", "½", "'D"]} +{"text": "'ſ'S…\" 'll㍿. ! (s!!\r\nß字㋿ḍ̇Z'ſ<|fim_prefix|>'Tß㍿ſ\n9\r\n#$%
㍿'T", "tokens": 65, "pieces": ["'ſ", "'S", "…", "\"", " '", "ll", "㍿.", " !", " ", "(<", "EOT", ">s", "!!\r\n", "ß字", "㋿ḋ", "̣Z", "'ſ", "<|", "fim", "_prefix", "|>'", "Tß", "㍿ſ", "\n", "9", "\r\n", "#$%", "
", "㍿'", "T"]} +{"text": "İm#$%🙂é'll'VE'VEfi \n\r\r0漢عt'llİd's\r'M­9", "tokens": 30, "pieces": ["İm", "#$%🙂", "é", "'ll", "'VE", "'VE", "fi", " \n\r\r", "0", "漢عt", "'ll", "İd", "'s", "\r", "'M", "­", "9"]} +{"text": "́'T'ſ \ne\u000bⅣ\ns9ḍ̇'S,\r\n\r\néed're<|fim_prefix|> 👍🏽\u000b½'Re", "tokens": 41, "pieces": ["́'", "T", "'ſ", " \n", "e", "\u000b", "Ⅳ", "\n", "s", "9", "ḋ", "̣'", "S", ",\r\n\r\n", "éed", "'re", "<|", "fim", "_prefix", "|>", " ", " 👍🏽", "\u000b", "½", "'Re"]} +{"text": "🙂'VE‍<​<|endoftext|>ⅣEOTdſ‍Dž-'D(", "tokens": 29, "pieces": ["🙂'", "VE", "‍<​<|", "endoftext", "|>", "Ⅳ", "EOTdſ", "‍Dž", "-'", "D", "("]} +{"text": "İ'Reİ𐞁'ſ\re", "tokens": 12, "pieces": ["İ", "'Re", "İ𐞁", "'ſ", "\r", "e"]} +{"text": "…åt\r'VE\nع​<|fim_prefix|>Ⅳß😀🏽s㍿<|fim_prefix|>,\r ­ꟲ…İ're!!,'T< 9EOT", "tokens": 60, "pieces": ["…a", "̊t", "\r", "'VE", "\n", "ع", "​<|", "fim", "_prefix", "|>", "Ⅳ", "ß", "😀🏽", "s", "㍿<|", "fim", "_prefix", "|>,\r", " ", " ­", "ꟲ", "…İ", "'re", "!!,'", "T", "<", " ", "9", "EOT"]} +{"text": "a‍\"mé,\rßꟲ'llé,t…#$% 'M!!t㍿'VE<|endoftext|>t('ſ", "tokens": 39, "pieces": ["a", "‍\"", "me", "́,\r", "ßꟲ", "'ll", "é", ",t", "…", "#$%", " '", "M", "!!", "t", "㍿'", "VE", "<|", "endoftext", "|>", "t", "('", "ſ"]} +{"text": "ع'S,", "tokens": 3, "pieces": ["ع", "'S", ","]} +{"text": "漢 a字", "tokens": 5, "pieces": ["漢", " a字"]} +{"text": "d'Reé're \n,!!<|fim_prefix|>😀🏽 \n​12345678m \n㍿ \r\n\r\nḍ̇'VE'S‍tZ>å#$%'S'D!,​,#$%\"٣٤٥٦A<漢,", "tokens": 71, "pieces": ["d", "'Re", "e", "́'", "re", " \n", ",!!<|", "fim", "_prefix", "|>😀🏽", " \n", "​", "123", "456", "78", "m", " \n", "㍿", " \r\n\r\n", "ḋ", "̣'", "VE", "'S", "‍tZ", ">a", "̊#$%'", "S", "'D", "!,​,#$%\"", "٣٤٥", "٦", "A", "<漢", ","]} +{"text": "'Sefi're\t­<|fim_prefix|>‍'Re\u000b🙂!12345678!! \na𐞁'S12345678EOT­A<|endoftext|>㍿'llİeé", "tokens": 56, "pieces": ["'S", "efi", "'re", "\t", "­<|", "fim", "_prefix", "|>‍'", "Re", "\u000b", "🙂!", "123", "456", "78", "!!", " \n", "a𐞁", "'S", "123", "456", "78", "EOT", "­A", "<|", "endoftext", "|>㍿'", "llİeé"]} +{"text": " 𐞁‍३Dž́­!½\r\n\r\nZsA!'T", "tokens": 22, "pieces": [" 𐞁", "‍", "३", "Dž", "́­!", "½", "\r\n\r\n", "ZsA", "!'", "T"]} +{"text": "0‍'Re.٣٤٥٦'ſ 's\ta\r½\r\n>ée'Dع\u000b𐞁a'Dİ 0 🙂'D'så漢'D'D३é'M>", "tokens": 56, "pieces": ["0", "‍'", "Re", ".", "٣٤٥", "٦", "'ſ", " '", "s", "\ta", "\r", "½", "\r\n", ">ée", "'D", "ع", "\u000b𐞁a", "'D", "İ", " ", " ", "0", " ", "🙂'", "D", "'s", "a", "̊漢", "'D", "'D", "३", "é", "'M", ">"]} +{"text": "عⅣ,9!!s …ع<
🙂,0 å\tDž👍🏽\r\n\r\nḍ̇ !!३ \n\r\n\r\n𐞁éfi'M", "tokens": 52, "pieces": ["ع", "Ⅳ", ",", "9", "!!", "s", " ", "…ع", "<", "
", "🙂,", "0", " a", "̊", "\tDž", "👍🏽\r\n\r\n", "ḋ", "̣", " ", "!!", "३", " \n\r\n\r\n", "𐞁e", "́fi", "'M"]} +{"text": "!<|endoftext|>३t\"३,😀🏽\t'D𐞁12345678'½", "tokens": 33, "pieces": ["!<|", "endoftext", "|>", "३", "t", "\"", "३", ",<", "META", "_START", ">😀🏽", "\t", "'D", "𐞁", "123", "456", "78", "'", "½"]} +{"text": "Dž<|endoftext|>", "tokens": 9, "pieces": ["Dž", "<|", "endoftext", "|>"]} +{"text": "#$%३>
\r\n\r\n<|endoftext|>字٣٤٥٦fifiå\r
ZEOT\rå㋿‍#$%", "tokens": 49, "pieces": ["#$%", "३", ">", "
\r\n\r\n", "<|", "endoftext", "|>", "字", "٣٤٥", "٦", "fifia", "̊\r", "
ZEOT", "\r", "a", "̊㋿‍#$%"]} +{"text": "𐞁㍿s9​…!ſḍ̇'Re.<|endoftext|>(ꟲs \n'll0 …ḍ̇ 'TDžfi<|fim_prefix|>0EOT​🙂½a0'sA\u000b", "tokens": 74, "pieces": ["𐞁", "㍿s", "9", "​", "…", "!ſḋ", "̣'", "Re", ".<", "META", "_START", "><|", "endoftext", "|>(", "ꟲs", " \n", "'ll", "0", " ", "…ḋ", "̣", " ", " '", "TDžfi", "<|", "fim", "_prefix", "|>", "0", "EOT", "​🙂", "½", "a", "0", "'s", "A", "\u000b"]} +{"text": ">09(!!ſADž- 'Sfi​\u000b'D'VE0!!\t'Se'VE's'D12345678''M", "tokens": 42, "pieces": [">", "09", "(!!", "ſADž", "-", " ", " '", "Sfi", "​", "\u000b", "'D", "'VE", "0", "!!<", "EOT", ">", "\t", "'S", "e", "'VE", "'", "s", "'D", "123", "456", "78", "''", "M"]} +{"text": "fi0m>-'sé \n\r‍9fi,Z\r\n½é9
㋿'re>'lĺéⅣ", "tokens": 49, "pieces": ["fi", "0", "m", ">-<", "EOT", ">'", "sé", " \n\r", "‍<", "EOT", ">", "9", "fi", ",Z", "\r\n", "½", "e", "́<", "META", "_START", ">", "9", "
", "㋿'", "re", ">'", "ll", "́e", "́", "Ⅳ", ""]} +{"text": "😀🏽½ \r-\rEOTét#$%é\r'Tع>é İ'D.㍿<|fim_prefix|>½é½ ‍a\"DžDžAⅣ A<'VE𐞁", "tokens": 63, "pieces": ["😀🏽", "½", " \r", "-\r", "EOTét", "#$%", "é", "\r", "'T", "ع", ">e", "́", " İ", "'D", ".㍿<|", "fim", "_prefix", "|>", "½", "e", "́", "½", " ‍", "a", "\"DžDžA", "Ⅳ", " ", " A", "<<", "EOT", ">'", "VE𐞁"]} +{"text": "ꟲ'M😀🏽🙂­", "tokens": 16, "pieces": ["ꟲ", "'", "M", "😀🏽🙂­"]} +{"text": "㍿Z㍿åfié'ſZ㋿>'VEdeع \n \nm 👍🏽éå३é", "tokens": 43, "pieces": ["㍿Z", "㍿a", "̊fié", "'ſ", "Z", "㋿>'", "VEdeع", " \n \n", "m", " 👍🏽<", "META", "_START", ">e", "́a", "̊", "३", "é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "…''re !m㋿ \n'T9\r\n\r\n<|fim_prefix|>m", "tokens": 22, "pieces": ["…", "''", "re", " ", "!m", "㋿", " \n", "'T", "9", "\r\n\r\n", "<|", "fim", "_prefix", "|>", "m"]} +{"text": "\t9EOT  's9'reİåt\n'D#$%s字>ꟲ", "tokens": 24, "pieces": ["\t", "9", "EOT", " ", " '", "s", "9", "'re", "İa", "̊t", "\n", "'D", "#$%", "s字", ">ꟲ"]} +{"text": "'D(mEOT½\u000bß\r\n\r\n,ḍ̇m's'T́>'ſ'D\r\na字0\t(ß'VE\r\u000b 🙂.عs9(", "tokens": 44, "pieces": ["'D", "(mEOT", "½", "\u000bß", "\r\n\r\n", ",ḋ", "̣m", "'s", "'T", "́>'", "ſ", "'D", "\r\n", "a字", "0", "\t", "(", "ß", "'VE", "\r", "\u000b", " ", "🙂.", "عs", "9", "("]} +{"text": "å're㋿ḍ̇'reZ㋿́'S漢!!
(Aé'S३\r\n\r\n#$% \n're
'VE#$%fi\n,\u000b", "tokens": 50, "pieces": ["a", "̊'", "re", "㋿ḋ", "̣'", "reZ", "㋿́'", "S漢", "!!", "
", "(Aé", "'S", "३", "\r\n\r\n", "#$%", " \n", "'re", "
", "'VE", "#$%", "fi", "\n", ",", "\u000b"]} +{"text": "!!é åéİ'Re㋿((!㋿", "tokens": 17, "pieces": ["!!", "é", " ", " a", "̊éİ", "'Re", "㋿((!㋿"]} +{"text": "'sDž\n字\r\n\r\nm#$%fi
漢'Ret½\u000bß'Tḍ̇9 ½ éEOT're'ſⅣ字३\tm", "tokens": 44, "pieces": ["'s", "Dž", "\n", "字", "\r\n\r\n", "m", "#$%", "fi", "
漢", "'Re", "t", "½", "\u000bß", "'T", "ḋ", "̣", "9", " ", "½", " ", " e", "́EOT", "'re", "'ſ", "Ⅳ", "字", "३", "\tm"]} +{"text": "ع'S\"", "tokens": 7, "pieces": ["ع", "'", "S", "\""]} +{"text": " \nZsⅣ\"sⅣ0é12345678<|fim_prefix|>>", "tokens": 20, "pieces": [" \n", "Zs", "Ⅳ", "\"s", "Ⅳ0", "é", "123", "456", "78", "<|", "fim", "_prefix", "|>>"]} +{"text": " \n
d㍿́12345678ſ'A㋿\" \né#$%\rfi<\r\n\r\n'lle", "tokens": 33, "pieces": [" \n", "
d", "㍿́", "123", "456", "78", "ſ", "'", "A", "㋿\"", " \n", "é", "#$%\r", "fi", "<\r\n\r\n", "'ll", "e"]} +{"text": "'VEmDžd'Re\r\n'Re< ㍿ é ‍
 漢…'TZ t\r'Refi!", "tokens": 41, "pieces": ["'VE", "mDžd", "'Re", "\r\n", "'Re", "<", " ", " ㍿", " ", " e", "́", " ", " ‍<", "EOT", ">", "
", " 漢", "…", "'T", "Z", " ", " t", "\r", "'Re", "fi", "!"]} +{"text": "\t\"३!½#$%\"'Sḍ̇𐞁ꟲ… \nDž́ſéⅣ​👍🏽", "tokens": 56, "pieces": ["\t", "\"", "३", "!", "½", "字", "<", "META", "_START", ">#$%<", "META", "_START", ">\"'", "Sḋ", "̣𐞁ꟲ", "… \n", "Dž", "́ſe", "́", "Ⅳ", "​👍🏽"]} +{"text": "'S(ß-'ll'T!

½>'VE'Td漢ع'Sd漢'VEA'så\r<|endoftext|>ⅣEOT'S 漢 '<|fim_prefix|>\t", "tokens": 60, "pieces": ["'S", "(ß", "-'", "ll", "'T", "!", "
", "
", "½", ">'", "VE", "'T", "d漢ع", "'S", "d漢", "'VE", "A", "'s", "a", "̊\r", "<|", "endoftext", "|><", "META", "_START", ">", "Ⅳ", "EOT", "'S", " ", " 漢", " ", " '<|", "fim", "_prefix", "|>", "\t"]} +{"text": "🙂́Ⅳ
éßZ字,­漢'ſ漢'T<|fim_prefix|>", "tokens": 29, "pieces": ["🙂́", "Ⅳ", "
e", "́ßZ字", ",­", "漢", "'ſ", "漢", "'T", "<|", "fim", "_prefix", "|>"]} +{"text": "!!\tßß", "tokens": 4, "pieces": ["!!", "\tßß"]} +{"text": "Z!ḍ̇'Sꟲ<#$%a#$%a's'edⅣ\teeſmeİ9'Dm", "tokens": 33, "pieces": ["Z", "!ḋ", "̣'", "Sꟲ", "<#$%", "a", "#$%", "a", "'s", "'ed", "Ⅳ", "\teeſmeİ", "9", "'D", "m"]} +{"text": "Dž\u000b12345678\"
tꟲ'Mt\"t½", "tokens": 18, "pieces": ["Dž", "\u000b", "123", "456", "78", "\"", "
tꟲ", "'M", "t", "\"t", "½"]} +{"text": "٣٤٥٦'ſ'M\r0EOT<'re9½<​㍿'ll'lla-s‍<|endoftext|>​tm½a aa½,å\"'D", "tokens": 51, "pieces": ["٣٤٥", "٦", "'ſ", "'M", "\r", "0", "EOT", "<'", "re", "9½", "<​㍿'", "ll", "'ll", "a", "-s", "‍<|", "endoftext", "|>​", "tm", "½", "a", " aa", "½", ",a", "̊\"'", "D"]} +{"text": "123456780Afi", "tokens": 11, "pieces": ["", "123", "456", "780", "Afi"]} +{"text": "İ‍d'll\nEOT'Dd<|endoftext|>'S\u000b", "tokens": 19, "pieces": ["İ", "‍d", "'ll", "\n", "EOT", "'D", "d", "<|", "endoftext", "|>'", "S", "\u000b"]} +{"text": "'M'ReeA12345678!😀🏽, Dž<|endoftext|>'s(\"ſ'\tſ0<|endoftext|>EOT ꟲ#$% \n😀🏽DžDž9عſe<'ReAⅣé", "tokens": 72, "pieces": ["'M", "'Re", "eA", "123", "456", "78", "!😀🏽,", " Dž", "<|", "endoftext", "|>'", "s", "(\"", "ſ", "'", "\tſ", "0", "<|", "endoftext", "|>", "EOT", " ", " ꟲ", "#$%", " \n", "😀🏽", "DžDž", "9", "عſe", "<'", "ReA", "Ⅳ", "e", "́"]} +{"text": "Ⅳ'ḍ̇ 👍🏽0ḍ̇12345678EOT<|fim_prefix|>\r\nⅣ😀🏽'VE\"é!!'S\nß 'T.𐞁é
>ḍ̇\u000b<|fim_prefix|>,́s", "tokens": 80, "pieces": ["Ⅳ", "'<", "EOT", ">ḋ", "̣", " ", "👍🏽", "0", "ḋ", "̣", "123", "456", "78", "EOT", "<|", "fim", "_prefix", "|>\r\n", "Ⅳ", "😀🏽'", "VE", "\"é", "!!'", "S", "\n", "ß", " ", " '", "T", ".𐞁e", "́", "
", ">ḋ", "̣", "\u000b", "<|", "fim", "_prefix", "|>,́", "s"]} +{"text": "!Z\r\n½\u000b\u000bsع\n<.<|fim_prefix|>
'VEsDž㋿d𐞁' \n<|fim_prefix|>İ#$%'MZ३ Ⅳ 'VE\r\n\r\nm9  \r\n", "tokens": 59, "pieces": ["!Z", "\r\n", "½", "\u000b", "\u000bsع", "\n", "<.<|", "fim", "_prefix", "|>", "
", "'VE", "sDž", "㋿d𐞁", "'", " \n", "<|", "fim", "_prefix", "|>", "İ", "#$%'", "MZ", "३", " ", " ", "Ⅳ", " ", " '", "VE", "\r\n\r\n", "m", "9", "  \r\n"]} +{"text": "!åİ㋿('Re\r\n'VEḍ̇t<|endoftext|>\n0İ<9!0(", "tokens": 36, "pieces": ["!a", "̊İ", "㋿<", "META", "_START", ">('", "Re", "\r\n", "'VE", "ḋ", "̣t", "<|", "endoftext", "|>\n", "0", "İ", "<", "9", "!", "0", "("]} +{"text": "fi<''Tde,\t… \n<😀🏽ḍ̇Dž👍🏽 é
\tſعḍ̇Ⅳ🙂t's\"㋿ mm(", "tokens": 60, "pieces": ["fi", "<''", "Tde", ",", "\t… \n", "<😀🏽", "ḋ", "̣Dž", "👍🏽", " e", "́", "
", "\tſعḋ", "̣", "Ⅳ", "🙂t", "'s", "\"㋿", " ", " mm", "("]} +{"text": "'re'Re're ,ß
​㋿'reEOTA!!㋿tDž#$%> \ne9🙂é…9å'red(s\r\n", "tokens": 44, "pieces": ["'re", "'Re", "'re", " ", " ,", "ß", "
", "​㋿'", "reEOTA", "!!㋿", "tDž", "#$%>", " \n", "e", "9", "🙂é", "…", "9", "a", "̊'", "red", "(s", "\r\n"]} +{"text": "fia.9㍿.aꟲ 'llß", "tokens": 17, "pieces": ["fia", ".", "9", "㍿.", "aꟲ", " ", " '", "llß"]} +{"text": "\nDžé <|endoftext|>…字 ­e\r\n…<|endoftext|><'s ꟲ\t \t-!…<|fim_prefix|>𐞁 ſ\r\n\r\n'VEſḍ̇dع", "tokens": 63, "pieces": ["\n", "Džé", " <|", "endoftext", "|>", "…字", " ­", "e", "\r\n", "…", "<|", "endoftext", "|><'", "s", " ꟲ", "\t ", "\t", "-!", "…", "<|", "fim", "_prefix", "|>", "𐞁", " ſ", "\r\n\r\n", "'VE", "ſḋ", "̣dع"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "½,字å<|fim_prefix|>🙂字\u000bé\n'Mß're.ſ½é'D<🙂", "字", "\u000bé", "\n", "'M", "ß", "'re", ".ſ", "½", "é", "'D", "<<", "d漢"]} +{"text": "'S
 ḍ̇m㋿Ae字́ḍ̇㋿'TŹ>mAé
  <|endoftext|>'VE,å \n'ſ", "tokens": 56, "pieces": ["'S", "
", " ḋ", "̣m", "㋿Ae字", "́ḋ", "̣㋿'", "TZ", "́>", "mA", "e", "́", "
 ", " ", "<|", "endoftext", "|>'", "VE", ",a", "̊", " \n", "'ſ"]} +{"text": "9ZDž<字é字<Dž½d‍'T!sⅣEOT \n‍\tEOTåa'ſAå٣٤٥٦İ𐞁", "tokens": 53, "pieces": ["9", "ZDž", "<字é字", "<Dž", "½", "d", "‍'", "T", "!s", "Ⅳ", "EOT", " \n", "‍", "\tEOTa", "̊a", "'ſ", "Aa", "̊", "٣٤٥", "٦", "İ𐞁"]} +{"text": "'ll>३ \n'12345678#$%\r!d‍́,-t😀🏽\r\n\r\n'D>́👍🏽", "tokens": 38, "pieces": ["'ll", ">", "३", " \n", "'", "123", "456", "78", "#$%\r", "!d", "‍́<", "META", "_START", ">,-", "t", "😀🏽\r\n\r\n", "'D", ">́👍🏽"]} +{"text": "12345678'S as­<|fim_prefix|>Dž12345678>es!<|endoftext|>\t", "tokens": 28, "pieces": ["123", "456", "78", "'S", " as", "­<|", "fim", "_prefix", "|>", "Dž", "123", "456", "78", ">es", "!<|", "endoftext", "|>", "\t"]} +{"text": "Z12345678e'S\r\n<|fim_prefix|>é½-", "tokens": 20, "pieces": ["Z", "123", "456", "78", "e", "'S", "\r\n", "<|", "fim", "_prefix", "|>", "e", "́", "½", "-"]} +{"text": "<|fim_prefix|>\r\nm𐞁\u000b漢߅'s'VE'Dßåß ,  'Re'D'MDž.'ll字'Sꟲ…ع", "tokens": 51, "pieces": ["<|", "fim", "_prefix", "|>\r\n", "m𐞁", "\u000b漢ß", "…", "'s", "'VE", "'D", "ß", "a", "̊ß", " ", ",", " ", " <", "META", "_START", ">'", "Re", "'D", "'M", "Dž", ".'", "ll字", "'S", "ꟲ", "…ع"]} +{"text": "'Dḍ̇'D\nfi!'s\r\n\r\nİ́t …,'re\tḍ̇<|endoftext|>", "tokens": 39, "pieces": ["'D", "ḋ", "̣'", "D", "\n", "fi", "!'", "s", "\r\n\r\n", "İ", "́t", " ", "…", ",'", "re", "\t", "ḋ", "̣<|", "endoftext", "|>"]} +{"text": "İ.,'VE'S
\r( !­ßſ'Sꟲs\tꟲ\u000b-'ſع,Aå", "tokens": 35, "pieces": ["İ", ".,'", "VE", "'S", "
\r", "(", " ", " !­", "ßſ", "'S", "ꟲs", "\tꟲ", "\u000b", "-'", "ſع", ",Aa", "̊"]} +{"text": "12345678­٣٤٥٦('Re0ſ<…ع🙂\u000btéⅣa're're", "tokens": 37, "pieces": ["123", "456", "78", "­", "٣٤٥", "٦", "('", "Re", "0", "ſ", "<", "…ع", "🙂", "\u000b", "t", "e", "́", "Ⅳ", "a", "'re", "'re"]} +{"text": "ꟲ字>t́㋿\r\n\r\n字👍🏽ds'Tß<|fim_prefix|>🙂 字", "tokens": 34, "pieces": ["ꟲ字", ">t", "́<", "META", "_START", ">㋿\r\n\r\n", "字", "👍🏽", "ds", "'T", "ß", "<|", "fim", "_prefix", "|>🙂", " 字"]} +{"text": " a  ​'sꟲdés!!\"'VE<|endoftext|>'re👍🏽EOT'sZ're' ́A字é½𐞁's㋿9㍿ſ", "tokens": 60, "pieces": [" a", " ", " ", "​'", "sꟲ", "de", "́s", "!!\"'", "VE", "<|", "endoftext", "|>'", "re", "👍🏽", "EOT", "'s", "Z", "'re", "'", " ́", "A字e", "́", "½", "𐞁", "'s", "㋿", "9", "㍿ſ"]} +{"text": "‍\n'sm,fiع́t,'s!!٣٤٥٦'字😀🏽👍🏽‍'re३!
", "tokens": 44, "pieces": ["‍\n", "'s", "m", ",fiع", "́t", ",'", "s", "!!", "٣٤٥", "٦", "'字", "😀🏽👍🏽‍'", "re", "३", "!", "
"]} +{"text": " …𐞁­'D­Ⅳ0EOT\r🙂字\r\nß'VE३é<|endoftext|>Z👍🏽😀🏽ß٣٤٥٦d𐞁ꟲ㋿<|fim_prefix|>㍿\n", "tokens": 81, "pieces": [" ", "…𐞁", "­'", "D", "­", "Ⅳ0", "EOT", "\r", "🙂字", "\r\n", "ß", "'VE", "३", "e", "́<|", "endoftext", "|>", "Z", "👍🏽😀🏽", "ß", "٣٤٥", "٦", "d𐞁ꟲ", "㋿<|", "fim", "_prefix", "|>㍿<", "META", "_START", ">\n"]} +{"text": "-\r\n\r\nß🙂're<|fim_prefix|>EOT…ésfi<'ll…ꟲ12345678㋿e>", "tokens": 41, "pieces": ["-\r\n\r\n", "ß", "🙂<", "META", "_START", ">'", "re", "<|", "fim", "_prefix", "|>", "EOT", "…e", "́sfi", "<'", "ll", "…ꟲ", "123", "456", "78", "㋿e", ">"]} +{"text": "字.EOTå \n ½ ‍'re'VE'llß
sſ漢'T'M'Då(#$%<|fim_prefix|><­'re\r\n\r\nZ½ß're<|fim_prefix|>'s é", "tokens": 63, "pieces": ["字", ".EOTa", "̊", " \n", " ", " ", "½", " ", "‍'", "re", "'VE", "'ll", "ß", "
sſ漢", "'T", "'M", "'D", "a", "̊(#$%<|", "fim", "_prefix", "|><­'", "re", "\r\n\r\n", "Z", "½", "ß", "'re", "<|", "fim", "_prefix", "|><", "META", "_START", ">'", "s", " é"]} +{"text": "Ⅳ", "tokens": 5, "pieces": ["", "Ⅳ"]} +{"text": " 9ß(😀🏽㋿漢< <|endoftext|>''Mfi𐞁'D\r\n\r\n 12345678\r\n漢å<ع're,\r\nZ", "tokens": 46, "pieces": [" ", "9", "ß", "(😀🏽㋿", "漢", "<", " <|", "endoftext", "|>''", "Mfi𐞁", "'D", "\r\n\r\n", " ", "123", "456", "78", "\r\n", "漢a", "̊<", "ع", "'re", ",\r\n", "Z"]} +{"text": "٣٤٥٦ß!!A\t👍🏽!!e.ꟲ'VE", "tokens": 27, "pieces": ["٣٤٥", "٦", "ß", "!!", "A", "\t", "👍🏽!!", "e", ".ꟲ", "'VE"]} +{"text": "'D\r\n\r\n'sſ𐞁<|fim_prefix|><|endoftext|>12345678<|endoftext|>é́́İ< ſDž'T \nfiⅣſA>Zꟲ㋿ع\r(>ſ' \n \n12345678٣٤٥٦A", "tokens": 84, "pieces": ["'D", "\r\n\r\n", "'s", "ſ𐞁", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "123", "456", "78", "<|", "endoftext", "|>", "é", "́́", "İ", "<", " ", " ſDž", "'T", " \n", "fi", "Ⅳ", "ſA", "><", "EOT", ">Zꟲ", "㋿ع", "\r", "(>", "ſ", "'", " \n \n", "123", "456", "78٣", "٤٥٦", "A"]} +{"text": "'sḍ̇-é३<|fim_prefix|>'T'sſ'Mé<|fim_prefix|>", "tokens": 32, "pieces": ["'s", "ḋ", "̣-", "e", "́", "३", "<|", "fim", "_prefix", "|>'", "T", "'s", "ſ", "'M", "e", "́<|", "fim", "_prefix", "|>"]} +{"text": ">漢d\"\nZ're­'S'D,\n", "tokens": 12, "pieces": [">漢d", "\"\n", "Z", "'re", "­'", "S", "'D", ",\n"]} +{"text": "m 🙂'TAtⅣ\rd'Reſ'VE٣٤٥٦A३㍿'Reİ'Ss å😀🏽́!…\r(\n'M'S'ſ漢Ⅳ 漢…Ⅳ ", "tokens": 70, "pieces": ["m", " ", "🙂'", "TAt", "Ⅳ", "\r", "d", "'Re", "ſ", "'VE", "٣٤٥", "٦", "A", "३", "㍿'", "Reİ", "'S", "s", " a", "̊😀🏽́!", "…\r", "(\n", "'M", "'S", "'ſ", "漢", "Ⅳ", " ", " 漢", "…", "Ⅳ", " "]} +{"text": "🙂́ 12345678<|endoftext|>'M(", "tokens": 16, "pieces": ["🙂́", " ", "123", "456", "78", "<|", "endoftext", "|>'", "M", "("]} +{"text": "<'ſ½'ſ're½'VE\"\n
'S,ßꟲ\n!!ſ'ſ 
!'MßéZ", "tokens": 35, "pieces": ["<'", "ſ", "½", "'ſ", "'re", "½", "'VE", "\"\n", "
", "'S", ",ßꟲ", "\n", "!!", "ſ", "'ſ", " ", "
", "!'", "MßéZ"]} +{"text": "
d́EOT漢!<|fim_prefix|>.½  ३…\"Á'S<|endoftext|>", "tokens": 36, "pieces": ["
d", "́EOT漢", "!<|", "fim", "_prefix", "|>.", "½", "  ", " ", "३", "…", "\"A", "́'", "S", "<|", "endoftext", "|>"]} +{"text": "<|endoftext|>0ع👍🏽'Re字ع're🙂\r\n\r\n𐞁ḍ̇,", "tokens": 33, "pieces": ["<|", "endoftext", "|>", "0", "ع", "👍🏽'", "Re字ع", "'re", "🙂\r\n\r\n", "𐞁ḋ", "̣,"]} +{"text": "'re\n㋿\u000bs३\r\nḍ̇ꟲ's'llå-ḍ̇A \n​㍿'!'Re'rem㋿\r\n \nḍ̇㍿😀🏽½", "tokens": 60, "pieces": ["'re", "\n", "㋿", "\u000bs", "३", "\r\n", "ḋ", "̣ꟲ", "'s", "'ll", "a", "̊-", "ḋ", "̣A", " \n", "​㍿'!'", "Re", "'re", "m", "㋿\r\n", " \n", "ḋ", "̣㍿😀🏽", "½"]} +{"text": "e🙂fi'S<|endoftext|>
met'M's½å🙂㋿👍🏽'M!, ß<|fim_prefix|>\r\n\r\n<|fim_prefix|> m,", "tokens": 55, "pieces": ["e", "🙂fi", "'S", "<|", "endoftext", "|>", "
met", "'M", "'s", "½", "a", "̊🙂㋿👍🏽'", "M", "!,", " ß", "<|", "fim", "_prefix", "|>\r\n\r\n", "<|", "fim", "_prefix", "|>", " m", ","]} +{"text": "<Aét<|fim_prefix|>İ<|fim_prefix|>,
", "tokens": 21, "pieces": ["<Aét", "<|", "fim", "_prefix", "|>", "İ", "<|", "fim", "_prefix", "|>,", "
"]} +{"text": "٣٤٥٦ß aAİa \t\r\n'Dꟲ🙂 ", "tokens": 27, "pieces": ["٣٤٥", "٦", "ß", " aAİa", "", " \t\r\n", "'D", "ꟲ", "🙂", " "]} +{"text": "½#$%​é字12345678ß½fi'llDž­Ⅳ㋿(,Dž \n'S'VEd", "tokens": 32, "pieces": ["½", "#$%​", "e", "́字", "123", "456", "78", "ß", "½", "fi", "'ll", "Dž", "­", "Ⅳ", "㋿(,", "Dž", " \n", "'S", "'VE", "d"]} +{"text": ">'S(#$%'D…Z‍'M㋿\r\n>", "tokens": 18, "pieces": [">'", "S", "(#$%'", "D", "…Z", "‍'", "M", "㋿\r\n", ">"]} +{"text": "aⅣ12345678<|fim_prefix|>>m🙂 !!…'Re< \r\nA​😀🏽#$%-A'Re🙂​'Tt'Re(s٣٤٥٦Z­m", "tokens": 54, "pieces": ["a", "Ⅳ12", "345", "678", "<|", "fim", "_prefix", "|>>", "m", "🙂", " !!", "…", "'Re", "<", " \r\n", "A", "​😀🏽#$%-", "A", "'Re", "🙂​'", "Tt", "'Re", "(s", "٣٤٥", "٦", "Z", "­m"]} +{"text": "å.'M'VE
\tEOT漢漢🙂.12345678ع
EOT!!Zß  dع9ſ.<'S漢\n,👍🏽9🙂\u000b", "tokens": 53, "pieces": ["a", "̊.'", "M", "'VE", "
", "\tEOT漢漢", "🙂.", "123", "456", "78", "ع", "
EOT", "!!", "Zß", "  ", " dع", "9", "ſ", ".<'", "S漢", "\n", ",👍🏽", "9", "🙂", "\u000b"]} +{"text": "ḍ̇A'ſ<", "tokens": 11, "pieces": ["ḋ", "̣A", "'ſ", "<"]} +{"text": "㍿\u000b>e㋿#$%'S\r<|endoftext|>fi \ń(<|fim_prefix|>
.İDžs'M'Ret0
'S'MA12345678", "tokens": 50, "pieces": ["㍿", "\u000b", ">e", "㋿#$%'", "S", "\r", "<|", "endoftext", "|>", "fi", " \n", "́(<|", "fim", "_prefix", "|>", "
", ".İDžs", "'M", "'Re", "t", "0", "
", "'S", "'M", "A", "123", "456", "78"]} +{"text": "å", "tokens": 3, "pieces": ["a", "̊"]} +{"text": " ḍ̇Dž", "tokens": 8, "pieces": [" ", " ḋ", "̣Dž"]} +{"text": "'ſå३ß'MDž'M<|endoftext|>​12345678٣٤٥٦😀🏽 'VE'T'(ع\t(åⅣ", "tokens": 57, "pieces": ["'ſ", "a", "̊", "३", "ß", "'M", "Dž", "'M", "<|", "endoftext", "|>​", "123", "456", "78", "", "٣٤٥", "٦", "😀🏽", " ", " '", "VE", "'T", "'(", "ع", "\t", "(a", "̊", "Ⅳ"]} +{"text": "EOT𐞁d-!!‍'D👍🏽ß'T漢0👍🏽字12345678'Re(…('T字 å(Z½éḍ̇0#$%'S ٣٤٥٦​", "tokens": 75, "pieces": ["EOT𐞁d", "-<", "META", "_START", ">!!‍'", "D", "👍🏽", "ß", "'T", "漢", "0", "👍🏽", "字", "123", "456", "78", "'Re", "(", "…", "('", "T字", " ", "a", "̊(", "Z", "½", "éḋ", "̣", "0", "#$%'", "S", " ", " ", "٣٤٥", "٦", "​"]} +{"text": "0'Re\r\n\r\n-mⅣ½a½́ ꟲ…", "tokens": 15, "pieces": ["0", "'Re", "\r\n\r\n", "-m", "Ⅳ½", "a", "½", "́", " ꟲ", "…"]} +{"text": "9EOT٣٤٥٦'ſe'VE٣٤٥٦字é\"'ll", "tokens": 30, "pieces": ["9", "EOT", "٣٤٥", "٦", "'ſ", "e", "'VE", "٣٤٥", "٦", "字e", "́\"'", "ll"]} +{"text": " EOT'ſfi'S-12345678.ꟲfi🙂🙂 'D½12345678\tEOTdm'll\n'!åå-'D", "tokens": 48, "pieces": [" EOT", "'ſ", "fi", "'S", "-", "123", "456", "78", ".ꟲfi", "🙂<", "EOT", ">🙂", " '", "D", "½12", "345", "678", "\tEOTdm", "'ll", "\n", "'!", "a", "̊a", "̊-'", "D"]} +{"text": "'VE EOT'Mḍ̇'ſ're𐞁İ'ſ\"'reß \n­0­<|endoftext|>漢sDž 'Re­'VEع𐞁.\n३㍿Zdé'M", "tokens": 71, "pieces": ["'VE", " EOT", "'M", "ḋ", "̣<", "EOT", ">'", "ſ", "'re", "𐞁İ", "'ſ", "\"'", "re", "ß", " \n", "­", "0", "­<|", "endoftext", "|>", "漢sDž", " ", "'Re", "­'", "VEع𐞁", ".\n", "३", "㍿Zdé", "'M"]} +{"text": "!!½!!s12345678!!.", "tokens": 8, "pieces": ["!!", "½", "!!", "s", "123", "456", "78", "!!."]} +{"text": "fi㍿t३…Dž12345678é३Z\u000bfi,0­d'ſ‍\r\n\r\nⅣ…fi ", "tokens": 49, "pieces": ["fi", "㍿t", "३", "…Dž", "123", "456", "78", "é", "३", "Z", "\u000bfi", ",", "0", "­<", "EOT", ">d", "'ſ", "‍\r\n\r\n", "", "Ⅳ", "…fi", " "]} +{"text": "dع'Rem\r'M>'S​'ſAdéDž́ſİ9ḍ̇,\nå-字­- 'Ms'M!!٣٤٥٦12345678३Aé́\t", "tokens": 59, "pieces": ["dع", "'Re", "m", "\r", "'M", ">'", "S", "​'", "ſAdéDž", "́ſİ", "9", "ḋ", "̣,\n", "a", "̊-", "字", "­-", " '", "Ms", "'M", "!!", "٣٤٥", "٦12", "345", "678", "३", "Ae", "́́", "\t"]} +{"text": "é🙂\t\n,Asé 🙂'lĺ'Ddß\"字 !!'S'D\u000b'll< \n(DžAå<|endoftext|>0!'re", "tokens": 55, "pieces": ["e", "́🙂", "\t\n", ",Asé", " ", "🙂'", "ll", "́'", "D", "dß", "\"字", " ", "!!'", "S", "'D", "\u000b", "'ll", "<", " \n", "(DžAa", "̊<|", "endoftext", "|>", "0", "!'", "re"]} +{"text": "é", "tokens": 2, "pieces": ["e", "́"]} +{"text": " '​🙂-字'ſ<|endoftext|>'M㋿\"tt㍿d-ſ'Tꟲ'M½٣٤٥٦\r\n\r\n'ſ \n'llt'reZ٣٤٥٦0d", "tokens": 62, "pieces": [" '​🙂-", "字", "'ſ", "<|", "endoftext", "|>'", "M", "㋿\"", "tt", "㍿d", "-ſ", "'T", "ꟲ", "'M", "½٣٤", "٥٦", "\r\n\r\n", "'ſ", " \n", "'ll", "t", "'re", "Z", "٣٤٥", "٦0", "d"]} +{"text": "\"\nt 'ꟲß", "tokens": 11, "pieces": ["\"\n", "t", " ", "'<", "META", "_START", ">ꟲß"]} +{"text": "12345678'll\t​12345678.\r\n\r\nſ!tßfi‍\"'s< \r!!0é½\"('Dḍ̇et'S.", "tokens": 43, "pieces": ["123", "456", "78", "'ll", "\t", "​", "123", "456", "78", ".\r\n\r\n", "ſ", "!tßfi", "‍\"'", "s", "<", " \r", "!!", "0", "e", "́", "½", "\"('", "Dḋ", "̣et", "'S", "."]} +{"text": "-m𐞁å", "tokens": 12, "pieces": ["-m𐞁a", "̊<", "EOT", ">"]} +{"text": "\r\nZ🙂👍🏽­ \u000b", "tokens": 13, "pieces": ["\r\n", "Z", "🙂👍🏽­", " \u000b"]} +{"text": "\t👍🏽e12345678m'Sé>㋿'MEOT's\t㍿\né́'S\"!!0 ßå!d ,A٣٤٥٦ \n‍\n'Re", "tokens": 62, "pieces": ["\t", "👍🏽", "e", "123", "456", "78", "m", "'S", "é", ">㋿'", "MEOT", "'s", "\t", "㍿\n", "é", "́'", "S", "\"!!", "0", " ", " ßa", "̊!", "d", " ", ",A", "٣٤٥", "٦", " \n", "‍\n", "'Re"]} +{"text": "𐞁( 😀🏽\n\r\n\r\n", "tokens": 13, "pieces": ["𐞁", "(", " ", "😀🏽\n\r\n\r\n"]} +{"text": "9tm,­ع-字३. 漢‍  Ⅳ'Re'Re\u000b9Z", "tokens": 25, "pieces": ["9", "tm", ",­", "ع", "-字", "३", ".", " ", " 漢", "‍", " ", " ", "Ⅳ", "'Re", "'Re", "\u000b", "9", "Z"]} +{"text": "‍👍🏽'M­́İḍ̇ 🙂(mDžA㋿\r\n 9'll ><|endoftext|>ſ\n!!🙂ét9🙂>'VE-9 ", "tokens": 61, "pieces": ["‍👍🏽'", "M", "­́", "İḋ", "̣", " ", " 🙂(", "mDžA", "㋿\r\n", " ", " ", "9", "'ll", " ><|", "endoftext", "|>", "ſ", "\n", "!!🙂", "e", "́t", "9", "🙂>'", "VE", "-", "9", "", " "]} +{"text": "-0 ḍ̇", "tokens": 7, "pieces": ["-", "0", " ḋ", "̣"]} +{"text": "é字fifi!!é's👍🏽ſm<|fim_prefix|>(漢s\r\n\r\n
İ३
ݽ's9İaſs'sé'ſ \néİ😀🏽\t", "tokens": 62, "pieces": ["e", "́字fifi", "!!", "é", "'s", "👍🏽", "ſm", "<|", "fim", "_prefix", "|>(", "漢s", "\r\n\r\n", "
İ", "३", "
İ", "½", "'s", "9", "İaſs", "'s", "e", "́'", "ſ", " \n", "e", "́İ", "😀🏽", "\t"]} +{"text": "عs", "tokens": 2, "pieces": ["عs"]} +{"text": "İ's\refi At's­9😀🏽åⅣ\r\n'd‍", "tokens": 27, "pieces": ["İ", "'s", "\r", "efi", " At", "'s", "­", "9", "😀🏽", "a", "̊", "Ⅳ", "\r\n", "'d", "‍"]} +{"text": "<\r漢\u000ba\"½t'T𐞁字0㋿\t३😀🏽'Re'T", "tokens": 30, "pieces": ["<\r", "漢", "\u000ba", "\"", "½", "t", "'T", "𐞁字", "0", "㋿", "\t", "३", "😀🏽'", "Re", "'T"]} +{"text": "👍🏽", "tokens": 6, "pieces": ["👍🏽"]} +{"text": " å👍🏽𐞁\r\n'' 0e!!'D🙂
", "tokens": 26, "pieces": [" a", "̊👍🏽", "𐞁", "\r\n", "''", " ", "0", "e", "!!'", "D", "🙂", "
"]} +{"text": "12345678Z", "tokens": 4, "pieces": ["123", "456", "78", "Z"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "fi…123456780Z0'M", "tokens": 10, "pieces": ["fi", "…", "123", "456", "780", "Z", "0", "'M"]} +{"text": "'ſ 'VE,\t३Ⅳ😀🏽", "tokens": 17, "pieces": ["'ſ", " ", " '", "VE", ",", "\t", "३Ⅳ", "😀🏽"]} +{"text": "字­'漢ꟲ\r\n­‍'S('S\"́sa\u000bßßt#$% <|fim_prefix|>👍🏽 ḍ̇t𐞁ḍ̇ …!ß𐞁㍿", "tokens": 71, "pieces": ["字", "­'", "漢ꟲ", "\r\n", "­‍'", "S", "('", "S", "\"́", "sa", "\u000b", "ßßt", "#$%", " <|", "fim", "_prefix", "|><", "EOT", ">👍🏽", " ḋ", "̣t𐞁ḋ", "̣", " ", "…", "!ß𐞁", "㍿"]} +{"text": "ſ́😀🏽㋿0'T­漢ع‍\t d12345678'VE\tⅣſ🙂㋿dDž't!!", "tokens": 44, "pieces": ["ſ", "́😀🏽㋿", "0", "'T", "­漢", "ع", "‍", "\t", " d", "123", "456", "78", "'VE", "\t", "Ⅳ", "ſ", "🙂㋿", "dDž", "'t", "!!"]} +{"text": "t…عꟲß́
👍🏽\u000b'MAdſ'VE'D", "tokens": 27, "pieces": ["t", "…عꟲß", "́", "
", "👍🏽", "\u000b", "'M", "Adſ", "'VE", "'D"]} +{"text": "İß漢e㋿ ३…<|endoftext|>9Á", "tokens": 25, "pieces": ["İß漢e", "㋿", " ", " ", "३", "…", "<|", "endoftext", "|>", "9", "A", "́"]} +{"text": "ß'ret(ßt'Re👍🏽0,\u000b!!s\rA'​𐞁​字ꟲع\u000b", "tokens": 41, "pieces": ["ß", "'re", "t", "(ßt", "'Re", "👍🏽", "0", ",", "\u000b", "!!", "s", "\r", "A", "'​", "𐞁", "​字ꟲع", "", "\u000b"]} +{"text": "٣٤٥٦A'T\t  mꟲ9 >'T𐞁AaA\r\n>👍🏽'S'Dm
​", "tokens": 52, "pieces": ["٣٤٥", "٦", "A", "'T", "\t ", " mꟲ", "9", "", " >'", "T𐞁AaA", "\r\n", ">👍🏽'", "S", "'D", "m", "
", "​<", "EOT", ">"]} +{"text": "Džḍ̇㍿(<|fim_prefix|>ḍ̇½!!\n('Re<|fim_prefix|>👍🏽 \n…字\n12345678A㋿ḍ̇d éİEOTt.…ſ\u000bå're​9", "tokens": 79, "pieces": ["Džḋ", "̣㍿(<|", "fim", "_prefix", "|>", "ḋ", "̣", "½", "!!\n", "('", "Re", "<|", "fim", "_prefix", "|>👍🏽", " \n", "…字", "\n", "123", "456", "78", "A", "㋿ḋ", "̣d", " éİEOTt", ".", "…ſ", "\u000ba", "̊'", "re", "​", "9"]} +{"text": "\r\n\r\nⅣſ'redé,é \n<|endoftext|>ßſ\rꟲA,\n\r👍🏽(👍🏽😀🏽<|endoftext|>\"-\n𐞁…9", "tokens": 62, "pieces": ["\r\n\r\n", "Ⅳ", "ſ", "'re", "de", "́,", "é", " \n", "<|", "endoftext", "|>", "ßſ", "\r", "ꟲA", ",\n\r", "👍🏽(👍🏽😀🏽<|", "endoftext", "|>\"-\n", "𐞁", "…", "9"]} +{"text": "A0's", "tokens": 4, "pieces": ["A", "0", "'s"]} +{"text": "😀🏽Z\n'Tſfi\r\n'll>.'SßéEOT0 漢…😀🏽́Ⅳ́𐞁AéEOT\u000bt", "tokens": 47, "pieces": ["😀🏽", "Z", "\n", "'T", "ſfi", "\r\n", "'ll", ">.'", "Sße", "́EOT", "0", " 漢", "…", "😀🏽́", "Ⅳ", "́𐞁AéEOT", "\u000bt"]} +{"text": "🙂漢!!İ🙂٣٤٥٦EOTß\r\n\t
#$%å0>-👍🏽\ŕ𐞁𐞁३İ", "tokens": 49, "pieces": ["🙂漢", "!!", "İ", "🙂", "٣٤٥", "٦", "EOTß", "\r\n", "\t", "
", "#$%", "a", "̊", "0", ">-👍🏽\r", "́𐞁𐞁", "३", "İ"]} +{"text": "Z'T're ㋿​½­‍å \n< <'T \nmm½", "tokens": 26, "pieces": ["Z", "'T", "'re", " ", "㋿​", "½", "­‍", "a", "̊", " \n", "<", " ", "<'", "T", " \n", "mm", "", "½"]} +{"text": "'VE​㍿­½<|endoftext|>", "tokens": 19, "pieces": ["'VE", "​㍿­<", "META", "_START", ">", "½", "<|", "endoftext", "|>"]} +{"text": "\r\nå\r ٣٤٥٦!!'ll\nsß're'ſ'S9sZ🙂\r\nß­s", "tokens": 37, "pieces": ["\r\n", "a", "̊\r", " ", " ", "٣٤٥", "٦", "!!'", "ll", "\n", "s", "ß", "'re", "'ſ", "'S", "9", "sZ", "🙂\r\n", "ß", "­s"]} +{"text": "'VE🙂", "tokens": 4, "pieces": ["'VE", "🙂"]} +{"text": "é\u000bDž‍s­!", "tokens": 10, "pieces": ["e", "́", "\u000bDž", "‍s", "­!"]} +{"text": "'SZ#$%Z½m🙂ſ字ع字\råeſ'll😀🏽å", "tokens": 30, "pieces": ["'S", "Z", "#$%", "Z", "½", "m", "🙂ſ字ع字", "\r", "a", "̊eſ", "'ll", "😀🏽", "a", "̊"]} +{"text": "ḍ̇'T<|endoftext|>
9<|fim_prefix|>漢㋿‍0ß🙂 a\u000bA३", "tokens": 42, "pieces": ["ḋ", "̣'", "T", "<|", "endoftext", "|>", "
", "9", "<|", "fim", "_prefix", "|>", "漢", "㋿‍", "0", "ß", "🙂", " a", "\u000bA", "३"]} +{"text": "'ll-'ll0'Re字ZꟲⅣ漢👍🏽\r\n\r\n😀🏽\u000bZ字,dß", "tokens": 31, "pieces": ["'ll", "-'", "ll", "0", "'Re", "字Zꟲ", "Ⅳ", "漢", "👍🏽\r\n\r\n", "😀🏽", "\u000bZ字", ",dß"]} +{"text": "\r\n\r\nſDžZ're \nⅣ9A'ſ\r\n\r\n'VÉé٣٤٥٦𐞁ém!<'Re #$%EOT!!", "tokens": 45, "pieces": ["\r\n\r\n", "ſDžZ", "'re", " \n", "Ⅳ9", "A", "'ſ", "\r\n\r\n", "'VE", "́e", "́", "٣٤٥", "٦", "𐞁e", "́m", "!<'", "Re", " #$%", "EOT", "!!"]} +{"text": "'S<|fim_prefix|>ꟲ😀🏽s d'VE漢<|endoftext|>'llDžſ😀🏽", "tokens": 45, "pieces": ["'S", "<|", "fim", "_prefix", "|>", "ꟲ", "😀🏽", "s", " d", "'VE", "漢", "<|", "endoftext", "|>'", "llDžſ", "😀🏽"]} +{"text": "'Sde㍿漢>🙂Dž!!<👍🏽 \n½ḍ̇'ll­fi12345678\"é<|fim_prefix|>", "tokens": 44, "pieces": ["'S", "de", "㍿漢", ">🙂", "Dž", "!!<👍🏽", " \n", "½", "ḋ", "̣'", "ll", "­fi", "123", "456", "78", "\"é", "<|", "fim", "_prefix", "|>"]} +{"text": "\"\r\n'VE<|fim_prefix|>🙂\"", "tokens": 13, "pieces": ["\"\r\n", "'VE", "<|", "fim", "_prefix", "|>🙂\""]} +{"text": " ,<​ß'T<\r\n\r\nåeß' é<|endoftext|>>'ll-ß's㋿'VEß‍<|fim_prefix|>İt…-12345678'>٣٤٥٦ ß", "tokens": 61, "pieces": [" ,<​", "ß", "'T", "<\r\n\r\n", "a", "̊eß", "'", " e", "́<|", "endoftext", "|>>'", "ll", "-ß", "'s", "㋿'", "VEß", "‍<|", "fim", "_prefix", "|>", "İt", "…", "-", "123", "456", "78", "'>", "٣٤٥", "٦", " ß"]} +{"text": "t​!12345678'll😀🏽\r\n", "tokens": 13, "pieces": ["t", "​!", "123", "456", "78", "'ll", "😀🏽\r\n"]} +{"text": ".é ,<|fim_prefix|>ḍ̇\rEOT'Ret​åé\t'ſt㋿ſⅣ🙂漢e\r\n\r\n'SAa12345678 !­", "tokens": 54, "pieces": [".e", "́", " ", ",<|", "fim", "_prefix", "|>", "ḋ", "̣\r", "EOT", "'Re", "t", "​a", "̊e", "́", "\t", "'ſ", "t", "㋿ſ", "Ⅳ", "🙂漢e", "\r\n\r\n", "'S", "Aa", "123", "456", "78", " ", "!­"]} +{"text": " ́  're12345678dm", "tokens": 9, "pieces": [" ", "́", " ", " ", "'re", "123", "456", "78", "dm"]} +{"text": "ſéſ9́EOTEOT!!\" ßİ!!fia é!…s字e", "tokens": 29, "pieces": ["ſe", "́ſ", "9", "́EOTEOT", "!!\"", " ", " ßİ", "!!", "fia", " é", "!", "…s字e"]} +{"text": "३'s\r'M
㍿,\r\n\r\nعEOT 'll,'re\r\n'M'ſİ!", "tokens": 28, "pieces": ["३", "'s", "\r", "'M", "
", "㍿,\r\n\r\n", "عEOT", " ", "'ll", ",'", "re", "\r\n", "'M", "'ſ", "İ", "!<", "EOT", ">"]} +{"text": "'ſm'S'llſ'Re漢éꟲ9", "tokens": 19, "pieces": ["'ſ", "m", "'S", "'ll", "ſ", "'Re", "漢é", "ꟲ", "9"]} +{"text": "-.ꟲſ'\r'll'VE
å", "tokens": 16, "pieces": ["-.", "ꟲſ", "'\r", "'ll", "'VE", "
a", "̊"]} +{"text": "'S٣٤٥٦é0A🙂é<|endoftext|>s\r\n\r\n", "tokens": 29, "pieces": ["'S", "٣٤٥", "٦", "e", "́", "0", "A", "🙂e", "́<|", "endoftext", "|><", "EOT", ">s", "\r\n\r\n"]} +{"text": "\u000b", "tokens": 1, "pieces": ["\u000b"]} +{"text": ".\"-ßⅣEOT🙂aⅣé
", "tokens": 16, "pieces": [".\"-", "ß", "Ⅳ", "EOT", "🙂a", "Ⅳ", "e", "́", "
"]} +{"text": "½\u000b🙂‍0!a ३9\r\n😀🏽߅.'S", "tokens": 24, "pieces": ["½", "\u000b", "🙂‍", "0", "!a", " ", "३9", "\r\n", "😀🏽", "ß", "…", ".'", "S"]} +{"text": "漢Ⅳḍ̇A((\r\r\n .<|endoftext|>
ſ'Re३٣٤٥٦ꟲm", "tokens": 40, "pieces": ["漢", "Ⅳ", "ḋ", "̣A", "((\r\r\n", " ", ".<|", "endoftext", "|>", "
ſ", "'Re", "३٣٤", "٥٦", "ꟲm"]} +{"text": "!!٣٤٥٦\r\n\n", "tokens": 10, "pieces": ["!!", "٣٤٥", "٦", "\r\n\n"]} +{"text": "٣٤٥٦字İ!!m'T…eé", "tokens": 21, "pieces": ["٣٤٥", "٦", "字İ", "!!", "m", "'T", "…eé", ""]} +{"text": ". 😀🏽d(ÁZ \nA🙂ß'Dİd \n½㍿é‍", "tokens": 30, "pieces": [".", " ", "😀🏽", "d", "(A", "́Z", " \n", "A", "🙂ß", "'D", "İd", " \n", "½", "㍿é", "‍"]} +{"text": "\r\n\r\n​Dž😀🏽s<|endoftext|>½0½½'re…", "tokens": 28, "pieces": ["\r\n\r\n", "​Dž", "😀🏽", "s", "<|", "endoftext", "|>", "½0", "", "½½", "'re", "…"]} +{"text": "Z'Dꟲ‍åéḍ̇mع\r", "tokens": 22, "pieces": ["Z", "'D", "ꟲ", "‍a", "̊éḋ", "̣mع", "\r"]} +{"text": "­>'S'ſ!!étß'll字𐞁m>ꟲ漢\nEOT٣٤٥٦'ll12345678(fi​9å0'D🙂́ Am", "tokens": 56, "pieces": ["­>'", "S", "'ſ", "!!", "e", "́tß", "'ll", "字𐞁m", ">ꟲ漢", "\n", "EOT", "٣٤٥", "٦", "'ll", "123", "456", "78", "(fi", "​", "9", "a", "̊", "0", "'D", "🙂́", " ", " Am"]} +{"text": "s,'ll
…0's9é㍿é३🙂́Džſ0\r\n\r\n'sع'M 
Z!!ḍ̇", "tokens": 45, "pieces": ["s", ",'", "ll", "
", "…", "0", "'s", "9", "e", "́㍿", "e", "́", "३", "🙂́", "Džſ", "0", "\r\n\r\n", "'s", "ع", "'M", " ", "
Z", "!!", "ḋ", "̣"]} +{"text": "é ,İ<|endoftext|>عéé𐞁👍🏽​ſa👍🏽ZZ𐞁\u000ba0 ­9½0\t​\u000bfi''re'Re<|fim_prefix|>​\u000bå‍(<", "tokens": 71, "pieces": ["é", " ", ",İ", "<|", "endoftext", "|>", "عe", "́é𐞁", "👍🏽​", "ſa", "👍🏽", "ZZ𐞁", "\u000ba", "0", " ", "­", "9½0", "\t", "​", "\u000bfi", "''", "re", "'Re", "<|", "fim", "_prefix", "|>​", "\u000ba", "̊‍(<"]} +{"text": "'Tḍ̇½ \n''Re<'re‍½.e12345678'D́é 𐞁fi🙂12345678\t٣٤٥٦fi漢'ſḍ̇12345678'D𐞁 漢é٣٤٥٦", "tokens": 82, "pieces": ["'T", "ḋ", "̣", "½", " \n", "''", "Re", "<'", "re", "‍", "½", ".e", "123", "456", "78", "'D", "́e", "́", " ", " 𐞁fi", "🙂", "123", "456", "78", "\t", "٣٤٥", "٦", "fi漢", "'ſ", "ḋ", "̣<", "META", "_START", ">", "123", "456", "78", "'D", "𐞁", " 漢e", "́", "٣٤٥", "٦"]} +{"text": " s, ­
fi're \nEOT ꟲ३字é漢\"\"!😀🏽ع
s‍méݽ 12345678‍㍿ſ.", "tokens": 56, "pieces": [" ", " s", ",", " ", "­", "
fi", "'re", " \n", "EOT", " ꟲ", "३", "字", "é漢", "\"\"!😀🏽", "ع", "
s", "‍me", "́İ", "½", " ", " ", "123", "456", "78", "‍㍿", "ſ", "."]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " s's…å,ſa\"t½𐞁're\rEOT½'llⅣfi\">\"A½İ-३t", "tokens": 45, "pieces": [" ", " s", "'s", "…a", "̊,", "ſa", "\"t", "½", "𐞁", "'re", "\r", "EOT", "½", "'ll", "Ⅳ", "fi", "\">\"", "A", "½", "İ", "-<", "META", "_START", ">", "३", "t"]} +{"text": "\r\n\r\n<,9Ⅳ३'Re'Re0\n12345678३ḍ̇\u000b\r\n t.\t<|endoftext|>A\r\n\r\n \n\ŕfi'Ss<|fim_prefix|>\r\n\r\n\n!!‍", "tokens": 55, "pieces": ["\r\n\r\n", "<,", "9Ⅳ३", "'Re", "'Re", "0", "\n", "123", "456", "78३", "ḋ", "̣", "\u000b\r\n", " t", ".", "\t", "<|", "endoftext", "|>", "A", "\r\n\r\n \n\r", "́fi", "'S", "s", "<|", "fim", "_prefix", "|>\r\n\r\n\n", "!!‍"]} +{"text": "Ⅳ́é \n​Ⅳع́ ", "tokens": 11, "pieces": ["Ⅳ", "́é", " \n", "​", "Ⅳ", "ع", "́", " "]} +{"text": ">ꟲ\r\n👍🏽٣٤٥٦\"!é​
's#$%½'Re𐞁\r\n\r\né,", "tokens": 41, "pieces": [">ꟲ", "\r\n", "👍🏽", "٣٤٥", "٦", "\"!", "é", "​<", "META", "_START", ">", "
", "'s", "#$%", "½", "'Re", "𐞁", "\r\n\r\n", "é", ","]} +{"text": "İ's👍🏽<|fim_prefix|><|fim_prefix|>'ll#$%'saḍ̇
ſ12345678 'S'VE\n", "tokens": 42, "pieces": ["İ", "'s", "👍🏽<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>'", "ll", "#$%'", "saḋ", "̣", "
ſ", "123", "456", "78", " '", "S", "'VE", "\n"]} +{"text": "३\r\nꟲåéDž
(㋿éfiꟲEOT-a'VE \"'DéⅣ\n'\n
.A!!\" \n👍🏽 🙂…é\r\n\r\n\n<|endoftext|>", "tokens": 66, "pieces": ["३", "\r\n", "ꟲa", "̊éDž", "
", "(㋿", "e", "́fiꟲEOT", "-a", "'VE", " ", "\"'", "Dé", "Ⅳ", "\n", "'\n", "
", ".A", "!!\"", " \n", "👍🏽", " 🙂", "…é", "\r\n\r\n\n", "<|", "endoftext", "|>"]} +{"text": "ß𐞁½,!!­½-😀🏽\r\n\r\n'ſ ‍#$%(,Z,\r\n,tt\u000b'ſ#$%", "tokens": 42, "pieces": ["ß𐞁", "½", ",!!­", "½", "-😀🏽\r\n\r\n", "'ſ", " ", "‍#$%(<", "META", "_START", "><", "EOT", ">,", "Z", ",\r\n", ",tt", "\u000b", "'ſ", "#$%"]} +{"text": "🙂३'ſ\rꟲ", "tokens": 11, "pieces": ["🙂", "३", "'ſ", "\r", "ꟲ"]} +{"text": "ع,'S", "tokens": 3, "pieces": ["ع", ",'", "S"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'VE'VE٣٤٥٦㋿İ'Re‍İ\r\n'VE'ss", "tokens": 25, "pieces": ["'VE", "'VE", "٣٤٥", "٦", "㋿İ", "'Re", "‍İ", "\r\n", "'VE", "'s", "s"]} +{"text": " 'M", "tokens": 2, "pieces": [" '", "M"]} +{"text": "👍🏽 \nع><|fim_prefix|>(㍿", "tokens": 18, "pieces": ["👍🏽", " \n", "ع", "><|", "fim", "_prefix", "|>(㍿"]} +{"text": "½Z", "tokens": 2, "pieces": ["½", "Z"]} +{"text": "'s\r'T\r\nfi\tésEOT \nEOT​Z 12345678ع \né👍🏽…!!'llZ㋿́٣٤٥٦", "tokens": 48, "pieces": ["'s", "\r", "'T", "\r\n", "fi", "\te", "́sEOT", " \n", "EOT", "​Z", " ", "123", "456", "78", "ع", " \n", "e", "́👍🏽", "…", "!!'", "llZ", "㋿́", "٣٤٥", "٦"]} +{"text": "(字'reſd", "tokens": 6, "pieces": ["(字", "'re", "ſd"]} +{"text": "'s…EOT\r\nع \nſås(m#$%㍿'D", "tokens": 23, "pieces": ["'s", "…EOT", "\r\n", "ع", " \n", "ſa", "̊s", "(m", "#$%㍿'", "D"]} +{"text": "
('VEꟲ㋿12345678 \n😀🏽'Rea\u000b \nⅣ\u000b漢'S \u000bß٣٤٥٦…३m'VE<|fim_prefix|>", "tokens": 55, "pieces": ["
", "('", "VEꟲ", "㋿", "123", "456", "78", " \n", "😀🏽'", "Rea", "\u000b \n", "Ⅳ", "\u000b漢", "'S", " ", "\u000bß", "٣٤٥", "٦", "…", "३", "m", "'VE", "<|", "fim", "_prefix", "|>"]} +{"text": "ع\t½0fiß<|endoftext|>-'D٣٤٥٦!", "tokens": 28, "pieces": ["ع", "\t", "½0", "fiß", "<|", "endoftext", "|>-'", "D", "٣٤٥", "٦", "!"]} +{"text": "㍿字tꟲ 'D 'ſéea\n३tDžع字'12345678漢0'S\"", "tokens": 34, "pieces": ["㍿字tꟲ", " ", "'D", " ", "'ſ", "e", "́ea", "\n", "३", "tDžع字", "'", "123", "456", "78", "漢", "0", "'S", "\""]} +{"text": "ſ字'sⅣsⅣZ 12345678're½0'T>👍🏽­t'Séé́३,A㍿\r\n>é EOT \n…'VE'\r\n\r\n‍😀🏽", "tokens": 61, "pieces": ["ſ字", "'s", "Ⅳ", "s", "Ⅳ", "Z", " ", " ", "123", "456", "78", "'re", "½0", "'T", ">👍🏽­", "t", "'S", "e", "́e", "́́", "३", ",A", "㍿\r\n", ">e", "́", " ", " EOT", " \n", "…", "'VE", "'\r\n\r\n", "‍😀🏽"]} +{"text": "३!! \n\r\n \n🙂'D-😀🏽fi\rع12345678'.👍🏽३‍\té👍🏽0ꟲ㋿ſå", "tokens": 58, "pieces": ["३", "!!", " \n\r\n \n", "🙂'", "D", "-😀🏽", "fi", "\r", "ع", "123", "456", "78", "'.<", "META", "_START", ">👍🏽", "३", "‍", "\te", "́👍🏽", "0", "ꟲ", "㋿ſa", "̊"]} +{"text": "漢𐞁é", "tokens": 7, "pieces": ["漢𐞁é"]} +{"text": "ſ'Mfi fi'ré'Reß'Re!", "tokens": 15, "pieces": ["ſ", "'M", "fi", " ", " fi", "'re", "́'", "Reß", "'Re", "!"]} +{"text": "'T\rfi­
\r\n\u000bEOTDžfi \n fis\r½'Re,
٣٤٥٦'T​(ꟲfi ½… ", "tokens": 47, "pieces": ["'T", "\r", "fi", "­", "
\r\n", "\u000bEOTDžfi", " \n", " fis", "\r", "½", "'Re", ",", "
", "٣٤٥", "٦", "'T", "​(", "ꟲfi", " ", "½", "… "]} +{"text": "㍿👍🏽d!\t\t́12345678", "tokens": 17, "pieces": ["㍿👍🏽", "d", "!", "\t", "\t", "́", "123", "456", "78"]} +{"text": "'D😀🏽'M𐞁A<ꟲꟲع'T,.㍿'T\n'll<|fim_prefix|>'D<|fim_prefix|>12345678e's\tꟲ'\u000b😀🏽İ(\"", "tokens": 67, "pieces": ["'D", "😀🏽'", "M𐞁A", "<ꟲꟲع", "'T", ",.㍿<", "META", "_START", ">'", "T", "\n", "'ll", "<|", "fim", "_prefix", "|>'", "D", "<|", "fim", "_prefix", "|>", "123", "456", "78", "e", "'s", "\tꟲ", "'", "\u000b", "😀🏽", "İ", "(\""]} +{"text": "9'D'ſ𐞁\r\n\r\n'M字㋿å👍🏽sm'll\rfi(😀🏽'ſ𐞁#$%", "tokens": 44, "pieces": ["9", "'D", "'ſ", "𐞁", "\r\n\r\n", "'M", "字", "㋿a", "̊👍🏽", "sm", "'ll", "\r", "fi", "(😀🏽'", "ſ𐞁", "#$%"]} +{"text": " ­'ll's!", "tokens": 5, "pieces": [" ­'", "ll", "'s", "!"]} +{"text": "Džİ 'SA​0 …dꟲ'VE9'Re\r\n\r\nEOTꟲḍ̇ mmå-\"‍字ⅣA", "tokens": 47, "pieces": ["Džİ", " ", "'S", "A", "​", "0", " ", "…dꟲ", "'VE", "9", "'Re", "\r\n\r\n", "EOTꟲḋ", "̣", " mma", "̊<", "EOT", ">-\"‍", "字", "Ⅳ", "A"]} +{"text": "-'M.d(", "tokens": 4, "pieces": ["-'", "M", ".d", "("]} +{"text": "\t", "tokens": 1, "pieces": ["\t"]} +{"text": "\n👍🏽'Rem<|fim_prefix|>'M字e'Reée'D!!'Re-", "tokens": 26, "pieces": ["\n", "👍🏽'", "Rem", "<|", "fim", "_prefix", "|>'", "M字e", "'Re", "ée", "'D", "!!'", "Re", "-"]} +{"text": "\t#$%Ⅳ'M\r\n İſ ​'ll\u000b'reDžḍ̇e're(ßZ字㍿ſ<㋿<|fim_prefix|><|endoftext|>㋿ß12345678'", "tokens": 63, "pieces": ["\t", "#$%", "Ⅳ", "'M", "\r\n", " İſ", " ", "​'", "ll", "\u000b", "'re", "Džḋ", "̣e", "'re", "(ßZ字", "㍿ſ", "<㋿<", "META", "_START", "><|", "fim", "_prefix", "|><|", "endoftext", "|>㋿", "ß", "123", "456", "78", "'"]} +{"text": "'ſ'D<|endoftext|>sfi'T<é \r\n\r\ndm\"½Dž́'T\t㋿EOTDž fi", "tokens": 40, "pieces": ["'ſ", "'D", "<|", "endoftext", "|>", "sfi", "'T", "<é", " \r\n\r\n", "dm", "\"", "½", "Dž", "́'", "T", "", "\t", "㋿EOTDž", " fi"]} +{"text": "'ReⅣs'VE🙂'D", "tokens": 10, "pieces": ["'Re", "Ⅳ", "s", "'VE", "🙂'", "D"]} +{"text": " å\u000bع\"><'VEé👍🏽 ", "tokens": 17, "pieces": [" ", " a", "̊", "\u000bع", "\"><'", "VEé", "👍🏽", " "]} +{"text": "t!!'ſ!!!\td½ſ,<|fim_prefix|>'s'D12345678>Z…́Džſ\u000bé𐞁\"'T'Re​'D", "tokens": 49, "pieces": ["t", "!!<", "EOT", ">'", "ſ", "!!!", "\td", "½", "ſ", ",<|", "fim", "_prefix", "|>'", "s", "'D", "123", "456", "78", ">Z", "…", "́Džſ", "\u000bé𐞁", "\"'", "T", "'", "Re", "​'", "D"]} +{"text": "Ⅳ…eé,'ſ'Re\r\nd😀🏽-\nß0Z(👍🏽'S12345678m½\"㍿㍿å٣٤٥٦​<|endoftext|>\te \nZ", "tokens": 67, "pieces": ["Ⅳ", "…ee", "́,'", "ſ", "'Re", "\r\n", "d", "😀🏽-\n", "ß", "0", "Z", "(👍🏽'", "S", "123", "456", "78", "m", "½", "\"㍿㍿", "a", "̊", "٣٤٥", "٦", "​<|", "endoftext", "|>", "\te", " \n", "Z"]} +{"text": "'ll\r\nꟲd", "tokens": 6, "pieces": ["'ll", "\r\n", "ꟲd"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿‍s#$%\t\r\n\r\na<|endoftext|>٣٤٥٦åع\n 0㋿\r\n<​Ⅳ,​9!!'<😀🏽 \n", "tokens": 52, "pieces": ["㍿‍", "s", "#$%", "\t\r\n\r\n", "a", "<|", "endoftext", "|>", "٣٤٥", "٦", "a", "̊ع", "\n", " ", " ", "0", "㋿\r\n", "<​", "Ⅳ", ",​", "9", "!!'<😀🏽", " \n"]} +{"text": "''Re 12345678漢\ntḍ̇0'DEOTEOT'VEꟲ-ꟲ­", "tokens": 30, "pieces": ["''", "Re", " ", "123", "456", "78", "漢", "\n", "tḋ", "̣", "0", "'D", "EOTEOT", "'VE", "ꟲ", "-ꟲ", "­"]} +{"text": "0", "tokens": 4, "pieces": ["", "0"]} +{"text": "<ß're'S🙂'Reſ𐞁 0 ㋿㍿0\r\n𐞁\u000bé>👍🏽 
 \nſfi\t\t́Z'Re'D", "tokens": 53, "pieces": ["<ß", "'re", "'S", "🙂'", "Reſ𐞁", " ", "0", " ㋿㍿", "0", "\r\n", "𐞁", "\u000be", "́>👍🏽", " 
 \n", "ſfi", "\t", "\t", "́Z", "'Re", "'D"]} +{"text": "'Re\t#$%<|endoftext|>é\u000b'ḍ̇,tA‍éAß'T9're 're,㍿'D​𐞁a'D\"fi३", "tokens": 53, "pieces": ["'Re", "\t", "#$%<|", "endoftext", "|>", "é", "\u000b", "'ḋ", "̣,", "tA", "‍éAß", "'T", "9", "'re", " ", " '", "re", ",㍿'", "D", "​𐞁a", "'D", "\"fi", "३"]} +{"text": "'S\r\nZ\rd!!  ​\t ", "tokens": 11, "pieces": ["'S", "\r\n", "Z", "\r", "d", "!!", " ", " ", "​", "\t "]} +{"text": "𐞁'ſt'Re", "tokens": 9, "pieces": ["𐞁", "'ſ", "t", "'Re"]} +{"text": "éd0 \n'ſ \nⅣİ12345678e,Dž漢me字😀🏽", "tokens": 28, "pieces": ["e", "́d", "0", " \n", "'ſ", " \n", "Ⅳ", "İ", "123", "456", "78", "e", ",Dž漢me字", "😀🏽"]} +{"text": " !!<|endoftext|>'VE'T 字'", "tokens": 25, "pieces": [" ", "!!<|", "endoftext", "|>'", "VE", "'", "T", "", " 字", "'<", "META", "_START", ">"]} +{"text": "'s", "tokens": 1, "pieces": ["'s"]} +{"text": "漢!!'VE\t<|fim_prefix|>'Reem㍿\r\n🙂A'Me٣٤٥٦ꟲ\u000bİ👍🏽३ 👍🏽 'T", "tokens": 58, "pieces": ["漢", "!!'", "VE", "\t", "<|", "fim", "_prefix", "|>'", "Reem", "㍿\r\n", "🙂<", "META", "_START", ">A", "'M", "e", "٣٤٥", "٦", "ꟲ", "\u000bİ", "👍🏽", "३", " ", " 👍🏽", " ", "'T"]} +{"text": "'s!!‍'Re'llt-‍​-s12345678 9ḍ̇", "tokens": 29, "pieces": ["'s", "!!‍'", "Re", "'ll", "t", "-‍​-", "s", "123", "456", "78", " ", " ", "9", "ḋ", "̣"]} +{"text": "é'VE!s\"Dž字s𐞁e\r\n\r\nZé's", "tokens": 25, "pieces": ["e", "́'", "VE", "!<", "EOT", "><", "META", "_START", ">s", "\"Dž字s𐞁e", "\r\n\r\n", "Zé", "'s"]} +{"text": " \tⅣ<|fim_prefix|>\n12345678🙂<|fim_prefix|>'Sé'Re<|fim_prefix|> ع\r\nfi𐞁'T'D𐞁́ꟲ…Ad fi's<9s<|endoftext|>😀🏽Ⅳ٣٤٥٦12345678", "tokens": 90, "pieces": [" ", "\t", "Ⅳ", "<|", "fim", "_prefix", "|>\n", "123", "456", "78", "🙂<|", "fim", "_prefix", "|>'", "Sé", "'Re", "<|", "fim", "_prefix", "|>", " ع", "\r\n", "fi𐞁", "'T", "'D", "𐞁", "́ꟲ", "…Ad", " fi", "'s", "<", "9", "s", "<|", "endoftext", "|>😀🏽", "Ⅳ٣٤", "٥٦1", "234", "567", "8"]} +{"text": " 9👍🏽\r㋿>d㍿!\u000b㋿\n'VEdét\u000b>éé👍🏽३ ३😀🏽㍿aAéİ…m‍", "tokens": 61, "pieces": [" ", "9", "👍🏽\r", "㋿>", "d", "㍿!", "\u000b", "㋿\n", "'VE", "de", "́t", "\u000b", ">e", "́é", "👍🏽", "३", " ", "३", "😀🏽㍿", "aAéİ", "…m", "‍"]} +{"text": "s(.<|endoftext|> !ḍ̇'Mꟲ'VEDž", "tokens": 28, "pieces": ["s", "(.<|", "endoftext", "|>", " ", "!<", "META", "_START", ">ḋ", "̣'", "Mꟲ", "'VE", "Dž"]} +{"text": "​'TDž \n…㍿\u000b ſt㍿åé½😀🏽é­9👍🏽<|endoftext|>ḍ̇'reⅣ#$%…ſA", "tokens": 70, "pieces": ["​'", "TDž", " \n", "…", "㍿", "\u000b ", " ſt", "㍿", "a", "̊é", "½", "😀🏽", "é", "­", "9", "👍🏽<|", "endoftext", "|>", "ḋ", "̣'", "re", "Ⅳ", "#$%", "…ſA", ""]} +{"text": "!!!\tß'ſ漢s'Re9<|endoftext|>d'S -'!!​ 字t½\n‍ ḍ̇½३", "tokens": 44, "pieces": ["!!!", "\tß", "'ſ", "漢s", "'Re", "9", "<|", "endoftext", "|>", "d", "'S", " -'!!​", " ", " 字t", "½", "\n", "‍<", "EOT", ">", " ", " ḋ", "̣", "½३"]} +{"text": "'llA -​\u000b'VÉ字'ſß३Ⅳſ٣٤٥٦'😀🏽é<.Dž'VE.0<|fim_prefix|>ع…<|endoftext|>", "tokens": 64, "pieces": ["'ll", "A", " ", "-​", "\u000b", "'VE", "́字", "'ſ", "ß", "३Ⅳ", "ſ", "٣٤٥", "٦", "'😀🏽", "é", "<.", "Dž", "'VE", ".", "0", "<|", "fim", "_prefix", "|>", "ع", "…", "<|", "endoftext", "|>"]} +{"text": "'reꟲd٣٤٥٦EOT9'D\n 字😀🏽㋿字m'VE😀🏽-0 -㍿😀🏽's !!", "tokens": 56, "pieces": ["'re", "ꟲd", "٣٤٥", "٦", "EOT", "9", "'D", "\n", " ", " 字", "😀🏽<", "EOT", ">㋿", "字m", "'VE", "😀🏽-", "0", " -㍿😀🏽'", "s", " ", "!!"]} +{"text": "mt'S.A \n‍३!!<|fim_prefix|>🙂'sZ<|endoftext|>Aḍ̇.Dž㍿12345678s- 'ś12345678ß Ⅳ㍿ꟲåA#$%字å½m", "tokens": 80, "pieces": ["mt", "'S", ".A", " \n", "‍", "३", "!!<|", "fim", "_prefix", "|>🙂'", "sZ", "<|", "endoftext", "|>", "Aḋ", "̣.", "Dž", "㍿", "123", "456", "78", "s", "-", " '", "s", "́", "123", "456", "78", "ß", "", " ", "Ⅳ", "㍿ꟲa", "̊A", "#$%", "字a", "̊", "½", "m"]} +{"text": "👍🏽Ⅳ漢<|endoftext|>EOT́\t㋿\r\nع㍿", "tokens": 29, "pieces": ["👍🏽", "Ⅳ", "漢", "<|", "endoftext", "|>", "EOT", "́", "\t", "㋿\r\n", "ع", "㍿"]} +{"text": " , é12345678Afi​0
t\r\n\r\nİDž😀🏽<|endoftext|>'ll\u000b'S'll  ㋿­EOT­­\r", "tokens": 54, "pieces": [" ", ",", " é", "123", "456", "78", "Afi", "​", "0", "
t", "\r\n\r\n", "İDž", "😀🏽<|", "endoftext", "|><", "META", "_START", ">'", "ll", "\u000b", "'S", "'ll", " ", " <", "EOT", ">", " ", " ㋿­", "EOT", "­­\r"]} +{"text": "\"\r\n\r\n>́ß ", "tokens": 5, "pieces": ["\"\r\n\r\n", ">́", "ß", " "]} +{"text": "字t'SEOTعZ㋿''ll'sEOTfi", "tokens": 17, "pieces": ["字t", "'S", "EOTعZ", "㋿''", "ll", "'s", "EOTfi"]} +{"text": "!!​👍🏽漢aꟲa\r\n🙂́\tⅣḍ̇Ⅳßd🙂Ⅳé'VEEOT…​㍿㍿\r\n\r\n!!…🙂 tfiⅣ🙂'Reé", "tokens": 67, "pieces": ["!!​👍🏽", "漢aꟲa", "\r\n", "🙂́", "\t", "Ⅳ", "ḋ", "̣", "Ⅳ", "ßd", "🙂", "Ⅳ", "e", "́'", "VEEOT", "…", "​㍿㍿\r\n\r\n", "!!", "…", "🙂", " ", " tfi", "Ⅳ", "🙂'", "Reé"]} +{"text": "
…\tfiḍ̇'Re0'é\u000b", "tokens": 18, "pieces": ["
…", "\tfiḋ", "̣'", "Re", "0", "'e", "́", "\u000b"]} +{"text": "Džs㍿-ſ", "tokens": 9, "pieces": ["Džs", "㍿-", "ſ"]} +{"text": "'VEé <|fim_prefix|>é\r'M", "tokens": 16, "pieces": ["'VE", "e", "́", " ", "<|", "fim", "_prefix", "|>", "e", "́\r", "'M"]} +{"text": "ßfi'ſ'ſ𐞁d", "tokens": 14, "pieces": ["ßfi", "'ſ", "'ſ", "𐞁d"]} +{"text": "'ſ t\t fi\tEOT<|fim_prefix|>åafi\n!!m'\r\nع…'ſ ३ع'll…é👍🏽漢t'VEé \u000b­…漢<|endoftext|>", "tokens": 73, "pieces": ["'ſ", " t", "\t ", " <", "EOT", ">fi", "\tEOT", "<|", "fim", "_prefix", "|>", "a", "̊afi", "\n", "!!", "m", "'\r\n", "ع", "…", "'ſ", " ", " ", "३", "ع", "'ll", "…e", "́👍🏽", "漢t", "'VE", "é", " ", "\u000b", "­", "…漢", "<|", "endoftext", "|>"]} +{"text": "😀🏽\r\n12345678 \nⅣ's,\t.9'é­'M😀🏽\rs㍿'M‍å㍿'Msa…ß'ſ9t'Re\tꟲZ 'S're", "tokens": 62, "pieces": ["😀🏽\r\n", "123", "456", "78", " \n", "Ⅳ", "'s", ",", "\t", ".", "9", "'e", "́­'", "M", "😀🏽\r", "s", "㍿'", "M", "‍a", "̊㍿'", "Msa", "…ß", "'ſ", "9", "t", "'Re", "\tꟲZ", " '", "S", "'re"]} +{"text": "EOT", "tokens": 2, "pieces": ["EOT"]} +{"text": "'S'D !9 ­३é, 'ReAßs", "tokens": 18, "pieces": ["'S", "'D", " ", "!", "9", " ", "­", "३", "e", "́,", " ", "'Re", "Aßs"]} +{"text": " 'ſ're#$%0'De'عDž'Dعfi😀🏽Ⅳmå'sd\r\n\r\n'll\r'S\r\n\r\n", "tokens": 39, "pieces": [" ", " '", "ſ", "'re", "#$%<", "EOT", ">", "0", "'D", "e", "'عDž", "'D", "عfi", "😀🏽", "Ⅳ", "ma", "̊'", "sd", "\r\n\r\n", "'ll", "\r", "'S", "\r\n\r\n"]} +{"text": "\n\r\n\r\n<㍿½\"é's​­…㍿ꟲ३(𐞁㋿\r\n\r\n é
字å9\"aå\"", "tokens": 55, "pieces": ["\n\r\n\r\n", "<㍿", "½", "\"e", "́'", "s", "​­", "…", "㍿", "ꟲ", "", "३", "(𐞁", "㋿\r\n\r\n", " ", " e", "́", "
字a", "̊", "9", "\"aa", "̊\""]} +{"text": ".'ſ'ſZt­ 'M-İḍ̇'S12345678#$%0'D㋿İfi\rEOTm'DZßع", "tokens": 42, "pieces": [".'", "ſ", "'ſ", "Zt", "­", " ", " '", "M", "-İḋ", "̣'", "S", "123", "456", "78", "#$%", "0", "'D", "㋿İfi", "\r", "EOTm", "'D", "Zßع"]} +{"text": "9½Dž\r\n\r\nZ!!‍A!\t㋿-åꟲ…fiém 'M'sa­㍿,😀🏽🙂 é\r\n👍🏽's'", "tokens": 60, "pieces": ["9½", "Dž", "\r\n\r\n", "Z", "!!‍", "A", "!", "\t", "㋿-", "a", "̊ꟲ", "…", "fiém", " ", " '", "M", "'s", "a", "­㍿,😀🏽🙂", " e", "́\r\n", "👍🏽'", "s", "'"]} +{"text": "­'Tİ \n'ß\"…'VE'Mé​-<|endoftext|>\r\n\r\n​३'ſ.-a\"", "tokens": 33, "pieces": ["­'", "Tİ", " \n", "'ß", "\"", "…", "'VE", "'M", "e", "́​-<|", "endoftext", "|>\r\n\r\n", "​", "३", "'ſ", ".-", "a", "\""]} +{"text": "\r\n12345678ßm'D
٣٤٥٦ḍ̇​ß,عs9<|fim_prefix|><|fim_prefix|><12345678'S­Ze9", "tokens": 53, "pieces": ["\r\n", "123", "456", "78", "ßm", "'D", "
", "٣٤٥", "٦", "ḋ", "̣​", "ß", ",", "عs", "9", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|><", "123", "456", "78", "'S", "­Ze", "9"]} +{"text": "'T\r\n'S\r\n\r\nꟲ,-  å12345678EOT㋿ \r\n\r\nⅣ<|endoftext|>t#$%👍🏽ßaé \n㋿", "tokens": 48, "pieces": ["'T", "\r\n", "'S", "\r\n\r\n", "ꟲ", ",-", " ", " a", "̊", "123", "456", "78", "EOT", "㋿", " \r\n\r\n", "Ⅳ", "<|", "endoftext", "|>", "t", "#$%👍🏽", "ßaé", " \n", "㋿"]} +{"text": "'s", "tokens": 1, "pieces": ["'s"]} +{"text": "ß字ع#$%e.ee\r\u000bm\n​字\u000bß\r\n", "tokens": 20, "pieces": ["ß字ع", "#$%", "e", ".ee", "\r", "", "\u000bm", "\n", "​字", "\u000bß", "\r\n"]} +{"text": "­
m'lle-A'MA😀🏽!!٣٤٥٦🙂-<|endoftext|>'Re'ſ'S12345678'D\t'll'safi ", "tokens": 52, "pieces": ["­", "
m", "'ll", "e", "-A", "'M", "A", "😀🏽!!", "٣٤٥", "٦", "🙂-<|", "endoftext", "|>'", "Re", "'ſ", "'S", "123", "456", "78", "'D", "\t", "'ll", "'s", "afi", " "]} +{"text": "aDž e'Sİ
d½s'VE …m'S…\nꟲ
ꟲ'Re'T…‍#$%å𐞁 'Re<|endoftext|>'VEEOT'Tꟲeé", "tokens": 66, "pieces": ["aDž", " e", "'S", "İ", "
d", "½", "s", "'VE", " ", "…", "m", "'S", "…\n", "ꟲ", "
ꟲ", "'Re", "'T", "…", "‍#$%", "a", "̊𐞁", " '", "Re", "<|", "endoftext", "|>'", "VEEOT", "'T", "ꟲeé"]} +{"text": "å🙂\t'S\t𐞁ⅣdaEOT\t<|fim_prefix|>Ⅳ½(12345678fi're漢­㍿#$%\n ,>\n<|endoftext|>å(ꟲ're \nḍ̇", "tokens": 67, "pieces": ["a", "̊🙂", "\t", "'S", "\t𐞁", "Ⅳ", "daEOT", "\t", "<|", "fim", "_prefix", "|>", "Ⅳ½", "(", "123", "456", "78", "fi", "'re", "漢", "­㍿#$%\n", " ", " ,>\n", "<|", "endoftext", "|>", "a", "̊(", "ꟲ", "'re", " \n", "ḋ", "̣"]} +{"text": "\r\n", "tokens": 1, "pieces": ["\r\n"]} +{"text": "😀🏽
 .\n👍🏽'‍\n12345678t \n'VE\"", "tokens": 30, "pieces": ["😀🏽", "
", " ", ".\n", "👍🏽'‍\n", "123", "456", "78", "t", " \n", "'VE", "\"<", "META", "_START", ">"]} +{"text": "​👍🏽😀🏽.<|endoftext|>", "tokens": 19, "pieces": ["​👍🏽😀🏽.<|", "endoftext", "|>"]} +{"text": "9'D…é.㍿!dm 'Rem's漢Džd​9s\r \r\n字A'DⅣ㋿𐞁㋿<|endoftext|>'M漢a12345678<", "tokens": 62, "pieces": ["9", "'D", "…é", ".㍿!", "dm", " ", "'Re", "m", "'s", "漢Džd", "​", "9", "s", "\r \r\n", "字A", "'D", "Ⅳ", "㋿𐞁", "㋿<|", "endoftext", "|>'", "M漢a", "123", "456", "78", "<"]} +{"text": "½'m", "tokens": 2, "pieces": ["½", "'m"]} +{"text": "'Re 
<'D
 t<ß \n'VEé',e<|fim_prefix|>३‍<|fim_prefix|>fi\t\r\n🙂9\"<|fim_prefix|>\r\n s…ع", "tokens": 57, "pieces": ["'Re", " ", "
", "<'", "D", "
 ", " t", "<ß", " \n", "'VE", "é", "',", "e", "<|", "fim", "_prefix", "|>", "३", "‍<|", "fim", "_prefix", "|>", "fi", "\t\r\n", "🙂", "9", "\"<|", "fim", "_prefix", "|>\r\n", "", " ", " s", "…ع"]} +{"text": "'s\r\n\t!!s'VEs'Dßå\nd😀🏽EOT \n  \n漢eé​🙂\">Z\t >", "tokens": 44, "pieces": ["'s", "\r\n", "\t", "!!", "s", "'VE", "s", "'D", "ßa", "̊\n", "d", "😀🏽", "EOT", " \n  \n", "漢ee", "́<", "EOT", ">​🙂\">", "Z", "\t ", " >"]} +{"text": "‍sİé́<|fim_prefix|>🙂'ſfi'Mعt٣٤٥٦ ſDž🙂👍🏽‍-mⅣ
́ 'S'漢", "tokens": 61, "pieces": ["‍sİe", "́́<|", "fim", "_prefix", "|>🙂'", "ſfi", "'M", "عt", "", "٣٤٥", "٦", " ſDž", "🙂👍🏽‍-", "m", "Ⅳ", "
", "́", " '", "S", "'漢"]} +{"text": "'MⅣfiḍ̇eſ㍿漢0٣٤٥٦-m's\r\n'Dع ꟲ字字", "tokens": 38, "pieces": ["'M", "Ⅳ", "fiḋ", "̣eſ", "㍿漢", "0٣٤", "٥٦", "-m", "'s", "\r\n", "'D", "ع", " ꟲ字字"]} +{"text": "ḍ̇ḍ̇‍٣٤٥٦e's!", "tokens": 23, "pieces": ["ḋ", "̣ḋ", "̣‍", "٣٤٥", "٦", "e", "'s", "!"]} +{"text": " 're'ſ.‍ꟲ'llå'ſ'ſİDž>­ (At<", "tokens": 35, "pieces": [" '", "re", "'ſ", ".‍", "ꟲ", "'ll", "a", "̊'", "ſ", "'ſ", "İDž", ">­", " ", "(<", "META", "_START", ">At", "<"]} +{"text": " -''D字9", "tokens": 10, "pieces": [" ", "-''", "D", "字", "9"]} +{"text": "(\r 's👍🏽é>\t…'Tعḍ̇'T<ꟲ'reDž½- 'ſ(t㍿\r \n-́", "tokens": 46, "pieces": ["(\r", " ", "'s", "👍🏽", "e", "́>", "\t", "…", "'T", "عḋ", "̣'", "T", "<ꟲ", "'re", "Dž", "½", "-", " ", "'ſ", "(t", "㍿\r", " \n", "-́"]} +{"text": "­åEOT< ßå😀🏽t'ſt३\r\n\r\n'Reع \r
ꟲs<|fim_prefix|> ­", "tokens": 47, "pieces": ["­a", "̊EOT", "<", " ", "ßa", "̊😀🏽", "t", "'ſ", "t", "३", "\r\n\r\n", "'Re", "ع", " \r", "
ꟲs", "<|", "fim", "_prefix", "|>", " ­"]} +{"text": "m\nEOT\u000b'VE,'D३㋿t३<|endoftext|>fi㍿ꟲmEOT漢㍿Z\"\rfi\r\nA\r漢İ👍🏽३,", "tokens": 61, "pieces": ["m", "\n", "EOT", "\u000b", "'VE", ",'", "D", "३", "㋿t", "३", "<|", "endoftext", "|>", "fi", "㍿ꟲmEOT漢", "㍿Z", "\"\r", "fi", "\r\n", "A", "\r", "漢İ", "👍🏽", "३", ","]} +{"text": " 👍🏽\u000b-\r\"\r\nsA Dž'VE", "tokens": 23, "pieces": [" ", " 👍🏽", "\u000b", "-<", "EOT", ">\r", "\"\r\n", "sA", " ", " Dž", "'VE"]} +{"text": "ꟲ--,eع㍿…ḍ̇\nZ \n\"sع \r\nſ'D
Z'M(", "tokens": 30, "pieces": ["ꟲ", "--,", "eع", "㍿", "…ḋ", "̣\n", "Z", " \n", "\"sع", " \r\n", "ſ", "'D", "
Z", "'M", "("]} +{"text": "'M'Re A𐞁ḍ̇,é\r𐞁… Z.\na٣٤٥٦12345678é'T", "tokens": 42, "pieces": ["'M", "'Re", " ", " A𐞁ḋ", "̣,", "é", "\r", "𐞁", "…", " Z", ".\n", "a", "٣٤٥", "٦12", "345", "678", "e", "́'", "T"]} +{"text": "'S'T́", "tokens": 3, "pieces": ["'S", "'T", "́"]} +{"text": "0😀🏽­𐞁sſé 'MⅣEOT\u000b", "tokens": 23, "pieces": ["0", "😀🏽­", "𐞁sſé", " ", " '", "M", "Ⅳ", "EOT", "\u000b"]} +{"text": "Z<fi're9…fi\n!!\r!å9\r\n\r\nZEOTß\"12345678…㍿", "tokens": 31, "pieces": ["Z", "<fi", "'re", "9", "…fi", "\n", "!!\r", "!a", "̊", "9", "\r\n\r\n", "ZEOTß", "\"", "123", "456", "78", "…", "㍿"]} +{"text": "\t\r\n\r\n Ⅳ \u000bꟲt'Sſꟲ'ſ\r\n
 ㋿'M\u000bfi㍿\né0!!㍿", "tokens": 46, "pieces": ["\t\r\n\r\n", " ", "Ⅳ", " ", "\u000bꟲt", "'S", "ſ", "ꟲ", "'ſ", "\r\n", "
", " ㋿'", "M", "\u000bfi", "㍿\n", "e", "́", "0", "!!㍿"]} +{"text": "́A👍🏽 \n'Tع! 're'reİ'Dß,ß\u000b🙂㍿!!㍿
'T½t'M'll<|fim_prefix|>s'!t㋿🙂‍é\r\n\r\n㍿", "tokens": 64, "pieces": ["́A", "👍🏽", " \n", "'T", "ع", "!", " ", "'re", "'re", "İ", "'D", "ß", ",ß", "\u000b", "🙂㍿!!㍿", "
", "'T", "½", "t", "'M", "'ll", "<|", "fim", "_prefix", "|>", "s", "'<", "META", "_START", ">!", "t", "㋿🙂‍", "é", "\r\n\r\n", "㍿"]} +{"text": "#$%ßEOT…½Dž ٣٤٥٦ \n'M,𐞁㋿s ", "tokens": 31, "pieces": ["#$%", "ßEOT", "…", "½", "Dž", " ", "٣٤٥", "٦", " \n", "'M", ",𐞁", "㋿s", " "]} +{"text": "0' \nḍ̇ßꟲ<\"
<|endoftext|>漢́amEOT‍Dž 🙂👍🏽 a", "tokens": 44, "pieces": ["0", "'", " \n", "ḋ", "̣ßꟲ", "<\"", "
", "<|", "endoftext", "|><", "META", "_START", ">漢", "́amEOT", "‍Dž", " ", " 🙂👍🏽", " a"]} +{"text": "½'ſå\t­'Daꟲ👍🏽fi㋿!!EOTé𐞁m👍🏽
 's'Mß're's \n 's\"\u000b'T\r\n", "tokens": 57, "pieces": ["½", "'ſ", "a", "̊", "\t", "­'", "Daꟲ", "👍🏽", "fi", "㋿!!", "EOTe", "́𐞁m", "👍🏽", "
", " ", "'s", "'M", "ß", "'re", "'s", " \n", " ", " '", "s", "\"", "\u000b", "'T", "\r\n"]} +{"text": "­ع́ ٣٤٥٦'Sſ 9٣٤٥٦'T 漢٣٤٥٦-'T!!EOT​<|endoftext|>Ⅳ0\" ꟲ㋿
'ſ<'Mع", "tokens": 72, "pieces": ["­ع", "́", " ", "٣٤٥", "٦", "'S", "ſ", " ", "9٣٤", "٥٦", "'T", " 漢", "٣٤٥", "٦", "-'", "T", "!!", "EOT", "​<|", "endoftext", "|>", "Ⅳ0", "\"", " ꟲ", "㋿", "
", "'ſ", "<'", "Mع"]} +{"text": "'Re", "tokens": 1, "pieces": ["'Re"]} +{"text": "-!,'ll
́  .,.İ'\t#$%s're'VE(­\u000b!​<|endoftext|>'VE'D\r\n\u000bA'SDž", "tokens": 48, "pieces": ["-!,'", "ll", "
", "́", " ", " ", ".,.", "İ", "'", "\t", "#$%", "s", "'", "re", "'VE", "(­", "\u000b", "!<", "EOT", ">​<|", "endoftext", "|>'", "VE", "'D", "\r\n", "\u000bA", "'S", "Dž"]} +{"text": "
\u000b\tß(<|fim_prefix|>-  ㍿😀🏽'll'ſ", "tokens": 27, "pieces": ["
\u000b", "\tß", "(<|", "fim", "_prefix", "|>-", " ", " ㍿😀🏽'", "ll", "'ſ"]} +{"text": "½'३😀🏽((…'Ts>å", "tokens": 17, "pieces": ["½", "'", "३", "😀🏽((", "…", "'T", "s", ">a", "̊"]} +{"text": "e½Z👍🏽ḍ̇12345678<|fim_prefix|>३\"s>\t!!'re'llDž<|endoftext|>ſ́'ll㋿0d😀🏽Dž", "tokens": 63, "pieces": ["e", "½", "Z", "👍🏽", "ḋ", "̣<", "EOT", ">", "123", "456", "78", "<|", "fim", "_prefix", "|>", "३", "\"s", ">", "\t", "!!'", "re", "'ll", "Dž", "<|", "endoftext", "|>", "ſ", "́'", "ll", "㋿", "0", "d", "😀🏽", "Dž"]} +{"text": "ḍ̇字'ſ<|fim_prefix|>字t'S#$%ßa'M 'T\u000b'M­AⅣ'< eé!!fi'reéé !ꟲ!!", "tokens": 53, "pieces": ["ḋ", "̣字", "'ſ", "<|", "fim", "_prefix", "|>", "字t", "'S", "#$%", "ßa", "'M", " ", "'T", "\u000b", "'M", "­A", "Ⅳ", "'<", " ", " ee", "́!!", "fi", "'re", "e", "́é", " ", "!<", "EOT", ">ꟲ", "!!"]} +{"text": "İ㋿🙂\r٣٤٥٦عfi३Ⅳ'Re\u000b'VE‍🙂'D ३\t'ſ½🙂9'll٣٤٥٦ſ­å'S字.eéEOT\n'T !é字", "tokens": 72, "pieces": ["İ", "㋿🙂\r", "٣٤٥", "٦", "عfi", "३Ⅳ", "'Re", "\u000b", "'VE", "‍🙂'", "D", " ", "३", "\t", "'ſ", "½", "🙂", "9", "'ll", "٣٤٥", "٦", "ſ", "­a", "̊'", "S字", ".eéEOT", "\n", "'T", " ", " !", "e", "́字"]} +{"text": "́fi.'S\t'VEDž㍿ \"<ꟲt-𐞁", "tokens": 24, "pieces": ["́fi", ".'", "S", "\t", "'VE", "Dž", "㍿", " ", " \"<", "ꟲt", "-𐞁"]} +{"text": "9A\u000bZAéſſ\r\n\r\nꟲe !!\r\n\r\nİ", "tokens": 28, "pieces": ["9", "A", "\u000bZAe", "́<", "EOT", ">ſſ", "\r\n\r\n", "ꟲe", " ", "!!\r\n\r\n", "İ"]} +{"text": "fi½", "tokens": 3, "pieces": ["fi", "½"]} +{"text": "'s  …㋿'D'Re 'S\r\n's 9's­!İsEOTs12345678İ's'S😀🏽 fié", "tokens": 51, "pieces": ["'s", "  ", "…", "㋿<", "e", "́​,>'", "D", "'Re", "", " ", "'S", "\r\n", "'s", " ", " ", "9", "'s", "­!", "İsEOTs", "123", "456", "78", "İ", "'s", "'S", "😀🏽", " fie", "́"]} +{"text": "!!\r\n\r\n' ß
>…<|endoftext|>'D<|endoftext|>字!!.😀🏽'sß­'Re'VE's\t 字'ſ.(", "tokens": 47, "pieces": ["!!\r\n\r\n", "'", " ß", "
", ">", "…", "<|", "endoftext", "|>'", "D", "<|", "endoftext", "|>", "字", "!!.😀🏽'", "sß", "­'", "Re", "'VE", "'s", "\t", " 字", "'ſ", ".("]} +{"text": " ع(", "tokens": 3, "pieces": [" ", " ع", "("]} +{"text": "字EOT.\r\nfi!ꟲ>.\r", "tokens": 12, "pieces": ["字EOT", ".\r\n", "fi", "!ꟲ", ">.\r"]} +{"text": " <'M0'VE ㍿…👍🏽é>9 漢12345678900EOT\r", "tokens": 35, "pieces": [" ", "<'", "M", "0", "'VE", " ", "㍿", "…", "👍🏽", "é", ">", "9", " ", " 漢", "123", "456", "789", "00", "EOT", "\r"]} +{"text": "'S,'ſ.<|fim_prefix|><|endoftext|>'ſ'sß0A\r9", "tokens": 26, "pieces": ["'S", ",'", "ſ", ".<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "ſ", "'s", "ß", "0", "A", "\r", "9"]} +{"text": "عDžEOT!(👍🏽", "tokens": 12, "pieces": ["عDžEOT", "!(👍🏽"]} +{"text": "<😀🏽'Afi \r\n0Dž-'MDž‍0ſ'll''re\n­ \nع٣٤٥٦👍🏽s
ßfi0 ३12345678​𐞁å\r\n\r\n", "tokens": 79, "pieces": ["<😀🏽'", "Afi", " \r\n", "0", "Dž", "-'", "MDž", "‍<", "META", "_START", ">", "0", "ſ", "'", "ll", "''", "re", "\n", "­", " \n", "ع", "٣٤٥", "٦", "👍🏽", "s", "
ßfi", "0", " ", "३12", "345", "678", "​", "𐞁a", "̊\r\n\r\n"]} +{"text": "'S-a​'reéß𐞁½ 'SZ漢fi ", "tokens": 19, "pieces": ["'S", "-a", "​'", "ree", "́ß𐞁", "½", " '", "SZ漢fi", " "]} +{"text": "\t㋿'VE<|endoftext|>", "tokens": 13, "pieces": ["\t", "㋿'", "VE", "<|", "endoftext", "|>"]} +{"text": "<#$%#$% a9EOT𐞁a \r\n\r\nt<'s  \n-ḍ̇fi٣٤٥٦åaådfi 0…-", "tokens": 55, "pieces": ["<#$%#$%", " a", "9", "EOT𐞁a", " \r\n\r\n", "t", "<'", "s", "  \n", "-ḋ", "̣fi", "٣٤٥", "٦", "a", "̊<", "EOT", ">aa", "̊dfi", " ", " ", "0", "…", "-"]} +{"text": "­😀🏽\"'ع👍🏽12345678's 'll🙂12345678<|fim_prefix|>😀🏽<.'M", "tokens": 45, "pieces": ["­😀🏽\"'<", "EOT", ">ع", "👍🏽", "123", "456", "78", "'s", " ", " '", "ll", "🙂", "123", "456", "78", "<|", "fim", "_prefix", "|>😀🏽<.'", "M"]} +{"text": ". ½.\u000b㍿d𐞁 12345678‍\n'DA", "tokens": 23, "pieces": [".", " ", "½", ".", "\u000b", "㍿d𐞁", " ", "123", "456", "78", "‍\n", "'D", "A"]} +{"text": "½a…😀🏽EOT'll-", "tokens": 13, "pieces": ["½", "a", "…", "😀🏽", "EOT", "'ll", "-"]} +{"text": "\tA‍(😀🏽 \nعꟲ㍿(a!!㍿", "tokens": 29, "pieces": ["\t", "A", "‍(😀🏽", " \n", "عꟲ", "㍿(", "a", "!!㍿"]} +{"text": "́🙂'T\u000b!!a𐞁\te \",!!", "tokens": 16, "pieces": ["́🙂'", "T", "\u000b", "!!", "a𐞁", "\te", " ", "\",!!"]} +{"text": "\r\n\r\n'll👍🏽ts㋿m#$%.tA😀🏽㍿ 'ſعſ12345678‍ſfi>३é's'Ms\t\r\n\r\nDžع", "tokens": 62, "pieces": ["\r\n\r\n", "'ll", "👍🏽", "ts", "㋿m", "#$%.", "tA", "😀🏽㍿", " ", " '", "ſعſ", "123", "456", "78", "‍<", "EOT", ">ſfi", ">", "३", "e", "́'", "s", "'M", "s", "\t\r\n\r\n", "Džع"]} +{"text": "!<|fim_prefix|>㍿", "tokens": 10, "pieces": ["!<|", "fim", "_prefix", "|>㍿"]} +{"text": "-漢é tm
,åß,'S", "tokens": 14, "pieces": ["-漢é", " tm", "
", ",a", "̊ß", ",'", "S"]} +{"text": "\r\n\r\n३㍿!!", "tokens": 7, "pieces": ["\r\n\r\n", "३", "㍿!!"]} +{"text": "'D'smⅣé𐞁ꟲ'12345678EOTå'ſ12345678 \n'S㋿ \néⅣع👍🏽́fi
å é'Dſ-Z'", "sm", "Ⅳ", "é𐞁ꟲ", "'", "123", "456", "78", "EOTa", "̊'", "ſ", "123", "456", "78", " \n", "'S", "㋿", " \n", "é", "Ⅳ", "ع", "👍🏽́", "fi", "
a", "̊", " e", "́'", "Dſ", "-Z", "<|endoftext|>s½,ꟲ\u000b9", "tokens": 30, "pieces": ["etſ", "9", "'re", "㋿\r\n", "½0३", "<|", "endoftext", "|>", "s", "½", ",ꟲ", "\u000b", "9"]} +{"text": "å\r\nsſ字Ⅳ​A­٣٤٥٦𐞁!EOTⅣfi\nßt's­'ll…d", "tokens": 43, "pieces": ["a", "̊\r\n", "sſ字", "Ⅳ", "​A", "­", "٣٤٥", "٦", "𐞁", "!EOT", "Ⅳ", "fi", "\n", "ßt", "'s", "­'", "ll", "…d"]} +{"text": " ḍ̇\r\n>\r\n'll\r\n½Ⅳ㋿,Dž'MDž㋿A'VE\"\t'så…㋿é", "tokens": 41, "pieces": [" ", " ḋ", "̣\r\n", ">\r\n", "'ll", "\r\n", "½Ⅳ", "㋿,", "Dž", "'M", "Dž", "㋿A", "'VE", "\"", "\t", "'s", "a", "̊", "…", "㋿é"]} +{"text": "
‍ésⅣ👍🏽s\r\n\r\n'VE\nḍ̇EOTd.  !!'T", "tokens": 34, "pieces": ["
", "‍e", "́s", "Ⅳ", "👍🏽", "s", "\r\n\r\n", "'VE", "\n", "ḋ", "̣EOTd", ".", "  ", " !!'", "T"]} +{"text": "'Sḍ̇.\r\n\r\n­३ ­!㍿", "tokens": 16, "pieces": ["'S", "ḋ", "̣.\r\n\r\n", "­", "३", " ", "­!㍿"]} +{"text": "'VE‍㍿\naå'S.", "tokens": 14, "pieces": ["'VE", "‍㍿\n", "aa", "̊'", "S", "."]} +{"text": "åDžſ'T😀🏽\r\n\r\n\n'SEOT\r\n\r\nḍ̇İ9 ㋿'Re㋿Dž''re'Ma㋿'T!a,\"\r\n\r\n\na'VE0'Reé", "tokens": 57, "pieces": ["a", "̊Džſ", "'T", "😀🏽\r\n\r\n\n", "'S", "EOT", "\r\n\r\n", "ḋ", "̣İ", "9", " ", "㋿'", "Re", "㋿Dž", "''", "re", "'M", "a", "㋿'", "T", "!a", ",\"\r\n\r\n\n", "a", "'VE", "0", "'Re", "é"]} +{"text": "tt\r\nDžm'a,<🙂\u000b­.‍9EOT<\rع\r
9! \tİع٣٤٥٦👍🏽Z'ſꟲåİ", "tokens": 60, "pieces": ["tt", "\r\n", "Džm", "'<", "EOT", ">a", ",<🙂", "\u000b", "­.‍", "9", "EOT", "<\r", "ع", "\r", "
", "9", "!", " ", "\tİع", "", "٣٤٥", "٦", "👍🏽", "Z", "'ſ", "ꟲa", "̊İ"]} +{"text": "\n\u000bfi<|fim_prefix|>, t'M-a\r\n\r\néⅣ'reAfi \r\n", "tokens": 29, "pieces": ["\n", "\u000bfi", "<|", "fim", "_prefix", "|>,", " ", " t", "'M", "-a", "\r\n\r\n", "é", "Ⅳ", "'re", "Afi", " \r\n"]} +{"text": "'sع!! \nعm👍🏽fiß ́A(<㋿'ſ'D!\t漢's­🙂,", "tokens": 39, "pieces": ["'s", "ع", "!!", " \n", "عm", "👍🏽", "fiß", " ́", "A", "(<㋿'", "ſ", "'D", "!", "\t漢", "'s", "­🙂,"]} +{"text": "́<|fim_prefix|>å12345678>½'S‍ع😀🏽'T‍12345678\nfi", "tokens": 39, "pieces": ["́<|", "fim", "_prefix", "|>", "a", "̊", "123", "456", "78", ">", "½", "'S", "‍ع", "😀🏽'", "T", "‍<", "EOT", ">", "123", "456", "78", "\n", "fi"]} +{"text": "\r\n\r\n DžZfi'ſ<|endoftext|> \n漢09\r'VE😀🏽Z.ß'ſ, \n𐞁\u000b'Re<|endoftext|>éß", "tokens": 56, "pieces": ["\r\n\r\n", "", " DžZfi", "'ſ", "<|", "endoftext", "|>", " \n", "漢", "09", "\r", "'VE", "😀🏽", "Z", ".ß", "'ſ", ",", " \n", "𐞁", "\u000b", "'Re", "<|", "endoftext", "|>", "éß"]} +{"text": "'D㍿'re'ſ's\naé​d>\ré३At​ \n-½ \nfiİfi字㍿'VE'D!Z\r\n\u000b'ſ", "tokens": 52, "pieces": ["'D", "㍿'", "re", "'ſ", "'s", "\n", "ae", "́<", "META", "_START", ">​", "d", ">\r", "e", "́", "३", "At", "​", " \n", "-", "½", " \n", "fiİfi字", "㍿'", "VE", "'D", "!Z", "\r\n", "\u000b", "'ſ"]} +{"text": "🙂'VEſ👍🏽İ're­字( \n", "tokens": 18, "pieces": ["🙂'", "VEſ", "👍🏽", "İ", "'re", "­字", "(", " \n"]} +{"text": "a!عDža​ 😀🏽…åع'M­d‍\"", "tokens": 23, "pieces": ["a", "!عDža", "​", " 😀🏽", "…a", "̊ع", "'M", "­d", "‍\""]} +{"text": "9\r\n\"<> \n…(Ⅳ'M'S
'VE㋿>­٣٤٥٦'M0…字…", "tokens": 38, "pieces": ["9", "\r\n", "\"<<", "EOT", ">>", " \n", "…", "(", "Ⅳ", "'M", "'S", "
", "'VE", "㋿>­", "٣٤٥", "٦", "'M", "0", "…字", "…"]} +{"text": "'Re!­'३ d​-\u000b", "tokens": 11, "pieces": ["'Re", "!­'", "३", " d", "​-", "\u000b"]} +{"text": "½'re'VE…Z9ꟲ", "tokens": 11, "pieces": ["½", "'re", "'VE", "…Z", "9", "ꟲ"]} +{"text": "åⅣd👍🏽Ⅳ㍿dd,३́'Re<|fim_prefix|>\u000b \n>.", "tokens": 37, "pieces": ["a", "̊", "Ⅳ", "d", "👍🏽", "Ⅳ", "㍿dd", ",", "३", "́'", "Re", "<|", "fim", "_prefix", "|>", "\u000b", "", " \n", ">."]} +{"text": "𐞁İ㋿\u000b'Ḿ \nZ‍é't0éZA9३字fi", "tokens": 31, "pieces": ["𐞁İ", "㋿", "\u000b", "'M", "́", " \n", "Z", "‍e", "́'", "t", "0", "e", "́ZA", "9३", "字fi"]} +{"text": "'ll'Sİ(́s­字\r\n\r\nEOT'👍🏽'İ…>é'll 𐞁ݽ字'T'D🙂…sⅣé٣٤٥٦\" e\n", "tokens": 60, "pieces": ["'ll", "'S", "İ", "(́", "s", "­字", "\r\n\r\n", "EOT", "'👍🏽'", "İ", "…", ">é", "'ll", " ", " 𐞁İ", "½", "字", "'T", "'D", "🙂", "…s", "Ⅳ", "e", "́", "٣٤٥", "٦", "\"", " e", "\n"]} +{"text": " 字a(\r\nعⅣfi'M!d😀🏽12345678", "tokens": 20, "pieces": [" ", " 字a", "(\r\n", "ع", "Ⅳ", "fi", "'M", "!d", "😀🏽", "123", "456", "78"]} +{"text": "㍿<ß'VE>d's३'llm…​(ꟲ!'reé漢12345678're\u000b're'M'D", "tokens": 33, "pieces": ["㍿<", "ß", "'VE", ">d", "'s", "३", "'ll", "m", "…", "​(", "ꟲ", "!'", "ree", "́漢", "123", "456", "78", "'re", "\u000b", "'re", "'M", "'D"]} +{"text": "e​ae!<ع", "tokens": 12, "pieces": ["e", "​", "a", "e", "!<", "ع"]} +{"text": "​字e𐞁EOTaſḍ̇#$% \"0\u000bſßm字… \n<|fim_prefix|>åEOT", "tokens": 47, "pieces": ["​字e𐞁EOTaſḋ", "̣#$%", " ", "\"", "0", "\u000bſßm", "字", "… \n", "<|", "fim", "_prefix", "|>", "a", "̊EOT"]} +{"text": "#$%!!Dž½३'VE.!e's Dž#$%Z'D​-🙂३'VEs'S,", "tokens": 35, "pieces": ["#$%!!", "Dž", "½३", "'VE", ".!", "e", "'s", " ", "Dž", "#$%", "Z", "'D", "​-🙂", "३", "'VE", "s", "'S", ","]} +{"text": "-'VE<|fim_prefix|>'<ꟲfi'T٣٤٥٦عſ字字fi\t́字", "tokens": 34, "pieces": ["-'", "VE", "<|", "fim", "_prefix", "|>'<", "ꟲfi", "'T", "٣٤٥", "٦", "عſ字字fi", "\t", "́字"]} +{"text": "<'M's<|endoftext|>d(👍🏽٣٤٥٦字𐞁.t<漢'ſ\n​é<|endoftext|>'re9𐞁", "tokens": 55, "pieces": ["<'", "M", "'s", "<|", "endoftext", "|>", "d", "(👍🏽", "٣٤٥", "٦", "字𐞁", ".t", "<漢", "'ſ", "\n", "​e", "́<|", "endoftext", "|>'", "re", "9", "𐞁"]} +{"text": "\"ع३EOT", "tokens": 6, "pieces": ["\"ع", "३", "EOT"]} +{"text": "'ś\r!.,'M", "tokens": 7, "pieces": ["'s", "́\r", "!.,'", "M"]} +{"text": ">,<|endoftext|>9!!㋿½Ⅳ\r\n\r\nd'T9𐞁İİ\n­㍿​EOT‍\"'M><|endoftext|> \n\r", "tokens": 47, "pieces": [">,<|", "endoftext", "|>", "9", "!!㋿", "½Ⅳ", "\r\n\r\n", "d", "'T", "9", "𐞁İİ", "\n", "­㍿​", "EOT", "‍\"'", "M", "><|", "endoftext", "|>", " \n\r"]} +{"text": "s漢漢 ٣٤٥٦", "tokens": 14, "pieces": ["s漢漢", " ", "٣٤٥", "٦"]} +{"text": "'ſ \n字‍\rḍ̇Dž\"EOT'ſⅣ\"'M'S\u000b!!‍ßm's.fi­'ReⅣ", "tokens": 43, "pieces": ["'ſ", " \n", "字", "‍\r", "ḋ", "̣Dž", "\"EOT", "'ſ", "Ⅳ", "\"<", "EOT", ">'", "M", "'S", "\u000b", "!!‍", "ßm", "'s", ".fi", "­'", "Re", "Ⅳ"]} +{"text": "t#$%#$%''s\r\n\r\nſZ\"<|endoftext|><٣٤٥٦<|fim_prefix|>!!-'Sé😀🏽٣٤٥٦d字#$%0s'Re漢é'Reeḍ̇s㋿ع", "tokens": 77, "pieces": ["t", "#$%#$%''", "s", "\r\n\r\n", "ſZ", "\"<|", "endoftext", "|><", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>!!-'", "Sé", "😀🏽", "٣٤٥", "٦", "d字", "#$%", "0", "s", "'Re", "漢e", "́<", "META", "_START", ">'", "Reeḋ", "̣s", "㋿ع"]} +{"text": "\ta #$%,a\u000bt,Dža'll'Re-efi𐞁ſ٣٤٥٦
,éa㍿12345678👍🏽\r\n'D12345678d㋿12345678åEOT", "tokens": 66, "pieces": ["\ta", " #$%,", "a", "\u000bt", ",Dža", "'ll", "'Re", "-efi𐞁ſ", "٣٤٥", "٦", "
", ",éa", "㍿", "123", "456", "78", "👍🏽\r\n", "'D", "123", "456", "78", "d", "㋿", "123", "456", "78", "a", "̊EOT"]} +{"text": "éḍ̇'T👍🏽 ‍ Ⅳ'ReZ…\"Ⅳİ!12345678ſd, \n 漢s\r\n\r\nDžå<|endoftext|>t(😀🏽 \n'Re!eé'red", "tokens": 69, "pieces": ["éḋ", "̣'", "T", "👍🏽", " ", "‍", " ", " ", "Ⅳ", "'Re", "Z", "…", "\"", "Ⅳ", "İ", "!", "123", "456", "78", "ſd", ",", " \n", " 漢s", "\r\n\r\n", "Dža", "̊<|", "endoftext", "|>", "t", "(😀🏽", " \n", "'Re", "!eé", "'re", "d"]} +{"text": "İ
‍ḍ̇ꟲfi…", "tokens": 17, "pieces": ["İ", "
", "‍ḋ", "̣ꟲfi", "…"]} +{"text": "!!㍿́'ll𐞁㋿\r", "tokens": 15, "pieces": ["!!㍿́'", "ll𐞁", "㋿\r"]} +{"text": " 0<|fim_prefix|>'ll㋿'så­t😀🏽३'Taå\r\nZ㍿12345678𐞁​
ꟲİ ꟲ\r\n", "tokens": 54, "pieces": [" ", " ", "0", "<|", "fim", "_prefix", "|>'", "ll", "㋿'", "sa", "̊­", "t", "😀🏽", "३", "'T", "aa", "̊\r\n", "Z", "㍿", "123", "456", "78", "𐞁", "​", "
ꟲİ", " ꟲ", "\r\n"]} +{"text": "e𐞁㍿é 12345678'sß 12345678meZ㋿字0'Re!'MZ-ſ'T𐞁", "tokens": 41, "pieces": ["e𐞁", "㍿<", "META", "_START", ">é", " ", "123", "456", "78", "'s", "ß", " ", "123", "456", "78", "meZ", "㋿字", "0", "'Re", "!'", "MZ", "-ſ", "'T", "𐞁"]} +{"text": " 0𐞁Z…#$%<|endoftext|>'D …👍🏽12345678á<|fim_prefix|>'re!!'Sfi \n😀🏽's٣٤٥٦'s'T<(", "tokens": 65, "pieces": [" ", "0", "𐞁Z", "…", "#$%<|", "endoftext", "|>'", "D", " ", "…", "👍🏽", "123", "456", "78", "a", "́<|", "fim", "_prefix", "|>'", "re", "!!'", "Sfi", " \n", "😀🏽'", "s", "٣٤٥", "٦", "'s", "'T", "<("]} +{"text": "👍🏽A'reé'reİ\n३字字,åZt३ꟲ<|endoftext|> 'Mꟲ", "tokens": 44, "pieces": ["👍🏽", "A", "'re", "e", "́'", "reİ", "\n", "३", "字", "字", ",a", "̊Zt", "३", "ꟲ", "<|", "endoftext", "|>", " ", "'M", "ꟲ"]} +{"text": "  ‍…\r漢ع're­ع \né'S👍🏽", "tokens": 22, "pieces": [" ", " ", "‍", "…\r", "漢ع", "'re", "­ع", " \n", "é", "'S", "👍🏽"]} +{"text": "\r#$%<|endoftext|>\u000b𐞁'T𐞁#$%e's \u000b…!'T́漢㍿>\r>👍🏽\r
å9,fi'VE‍\r㍿ < 𐞁<|fim_prefix|>漢ſ", "tokens": 82, "pieces": ["\r", "#$%<|", "endoftext", "|>", "\u000b𐞁", "'T", "𐞁", "#$%", "e", "'s", " \u000b", "…", "!'", "T", "́漢", "㍿>\r", ">👍🏽\r", "
a", "̊", "9", ",fi", "'VE", "‍\r", "㍿", " ", "<", " ", " 𐞁", "<|", "fim", "_prefix", "|>", "漢ſ"]} +{"text": "ſ9ع'Reé३'reſ'll\t'ſ㋿Ⅳ'Retfi­'re'VE​", "tokens": 31, "pieces": ["ſ", "9", "ع", "'Re", "é", "३", "'re", "ſ", "'ll", "\t", "'ſ", "㋿", "Ⅳ", "'Re", "tfi", "­'", "re", "'VE", "​"]} +{"text": "EOTd90'llA.'ll\" ", "tokens": 11, "pieces": ["EOTd", "90", "'ll", "A", ".'", "ll", "\"", " "]} +{"text": "dA٣٤٥٦½td<­३å'Té12345678㍿字>\t'D <|fim_prefix|>ꟲ\n'MA'reå 'Dß🙂å 9", "tokens": 62, "pieces": ["dA", "٣٤٥", "٦½", "td", "<­", "३", "a", "̊'", "Té", "123", "456", "78", "㍿字", ">", "\t", "'D", " ", "<|", "fim", "_prefix", "|>", "ꟲ", "\n", "'M", "A", "'re", "a", "̊", " ", "'D", "ß", "🙂a", "̊", " ", "9"]} +{"text": "'İDž\t½(漢", "tokens": 9, "pieces": ["'İDž", "\t", "½", "(漢"]} +{"text": ".ß𐞁‍İ \r\n", "tokens": 17, "pieces": [".", "ß𐞁", "‍", "İ", " \r\n"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'ſ ㋿ée👍🏽9\rEOT'\n<|endoftext|> ꟲ'S", " ꟲ", "'S", " \n​😀🏽 \nſDžſ'VE٣٤٥٦t\t<>", "tokens": 62, "pieces": ["'re", "🙂ḋ", "̣", "9", "​\n", "㋿'", "S", "­", "9", "éd漢", "
eİ", "", " \n", "​😀🏽", " \n", "ſDžſ", "'VE", "٣٤٥", "٦", "<", "META", "_START", ">t", "\t", "<>"]} +{"text": " #$%'D🙂9,…'re", "tokens": 11, "pieces": [" ", "#$%'", "D", "🙂", "9", ",", "…", "'re"]} +{"text": "ſß!<", "tokens": 4, "pieces": ["ſß", "!<"]} +{"text": "ßd'Re", "tokens": 3, "pieces": ["ßd", "'Re"]} +{"text": "fi( 'M\t''ReEOTéa\r\n㍿𐞁½ ३ é😀🏽\u000b>😀🏽'Dع'Dé\"İ ", "tokens": 48, "pieces": ["fi", "(", " ", "'M", "\t", "''", "ReEOTe", "́a", "\r\n", "㍿𐞁", "½", " ", "३", " e", "́😀🏽", "\u000b", ">😀🏽'", "Dع", "'D", "é", "\"İ", " "]} +{"text": "㋿́!!-😀🏽'0ع\t'ſ'D!!#$%<é漢fi<­", "tokens": 47, "pieces": ["㋿́!!-😀🏽'", "0", "ع", "", "\t", "'ſ", "'D", "!!<", "a", "̊a", "̊", " ", " '", "D", "#$%<", "é漢fi", "<­"]} +{"text": ".'Re­>", "tokens": 4, "pieces": [".'", "Re", "­>"]} +{"text": " <|endoftext|>​ \n-", "tokens": 9, "pieces": [" <|", "endoftext", "|>​", " \n", "-"]} +{"text": "'ſ!m漢😀🏽m㋿ſ9( \n𐞁Zꟲ'llé'M‍'s#$%(åḍ̇㍿٣٤٥٦'Re \n\r\nZ", "tokens": 63, "pieces": ["'ſ", "!m漢", "😀🏽", "m", "㋿ſ", "9", "(", " \n", "𐞁Zꟲ", "'ll", "e", "́'", "M", "‍'", "s", "#$%(", "a", "̊ḋ", "̣㍿", "٣٤٥", "٦", "'Re", " \n\r\n", "Z"]} +{"text": "t'ReEOTdA'M\r\n'D'VE\r\n'ſſ𐞁㋿'M'ſt>字", "tokens": 36, "pieces": ["t", "'Re", "EOTdA", "'", "M", "\r\n", "'D", "'VE", "\r\n", "'ſ", "ſ𐞁", "㋿'", "M", "'ſ", "t", ">字"]} +{"text": "'Mſ'T३d​\r\n<|fim_prefix|>'ſḍ̇'VE-Z-'Sİ㋿\r­ⅣEOT㋿\r\n\né😀🏽㍿'D>'M", "tokens": 55, "pieces": ["'M", "ſ", "'T", "३", "d", "​\r\n", "<|", "fim", "_prefix", "|>'", "ſḋ", "̣'", "VE", "-Z", "-'", "Sİ", "㋿\r", "­", "Ⅳ", "EOT", "㋿\r\n\n", "é", "😀🏽㍿'", "D", ">'", "M"]} +{"text": "漢 \r
  
'll½字Ⅳm\r\n\r\n,eémDž<|fim_prefix|>! 0​'S're-!!'D'D
ꟲ\"𐞁ḍ̇\u000b‍", "tokens": 66, "pieces": ["漢", " \r", "
  ", "
", "'ll", "½", "字", "Ⅳ", "m", "\r\n\r\n", ",<", "EOT", ">eémDž", "<|", "fim", "_prefix", "|>!", " ", " ", "0", "​'", "S", "'re", "-!!'", "D", "'", "D", "
ꟲ", "\"𐞁ḋ", "̣", "\u000b", "‍"]} +{"text": "\t12345678ßd\r\nm'Re​", "tokens": 10, "pieces": ["\t", "123", "456", "78", "ßd", "\r\n", "m", "'Re", "​"]} +{"text": ".!!㋿ḍ̇éⅣ'red३'Dåt", "tokens": 26, "pieces": [".!!㋿", "ḋ", "̣e", "́", "Ⅳ", "'re", "d", "", "३", "'D", "a", "̊t"]} +{"text": "-‍\",
", "tokens": 6, "pieces": ["-‍\",", "
"]} +{"text": "'s'T DžEOT12345678'S", "tokens": 11, "pieces": ["'s", "'T", " DžEOT", "123", "456", "78", "'S"]} +{"text": "­㋿\r字\r\n\r\n㍿ (​é\r\n", "tokens": 20, "pieces": ["­㋿<", "META", "_START", ">\r", "字", "\r\n\r\n", "㍿", " ", "(​", "e", "́\r\n"]} +{"text": "字!9", "tokens": 3, "pieces": ["字", "!", "9"]} +{"text": "e\nḍ̇字A-٣٤٥٦\"​
\"'DZⅣ𐞁'S<|endoftext|>> (İ \nm,><漢", "tokens": 58, "pieces": ["e", "\n", "ḋ", "̣字A", "-", "٣٤٥", "٦", "\"<", "EOT", ">​", "
", "\"'", "DZ", "", "Ⅳ", "𐞁", "'S", "<|", "endoftext", "|>>", " ", " (", "İ", " \n", "m", ",<", "META", "_START", ">><", "漢"]} +{"text": "é<‍㍿漢…\u000b​9'M字漢 \r\n㍿ꟲ<'så\rm漢fi‍!!㋿\"٣٤٥٦ſ", "tokens": 38, "pieces": ["é", "Ⅳ", "…", "!!", "m", "(d", "'VE", "漢", ">a", "̊\r", "m漢fi", "‍!!㋿\"", "٣٤٥", "٦", "ſ"]} +{"text": "… \nßⅣ12345678
ḍ̇mعßA​İ-é12345678🙂'M‍ 'M9EOT\r\n'll\r\n\r\n٣٤٥٦ſ ", "tokens": 54, "pieces": ["… \n", "ß", "Ⅳ12", "345", "678", "
ḋ", "̣mعßA", "​İ", "-e", "́", "123", "456", "78", "🙂'", "M", "‍", " ", " '", "M", "9", "EOT", "\r\n", "'ll", "\r\n\r\n", "٣٤٥", "٦", "ſ", " "]} +{"text": "'ll🙂\rꟲ #$%, 'Re½​㋿…'D漢ꟲ'SAA e\t\tḍ̇'reİ👍🏽e😀🏽fi🙂s, 'VEt!", "tokens": 69, "pieces": ["'ll", "🙂\r", "ꟲ", " ", "#$%,", " '", "Re", "½", "​㋿", "…", "'D", "漢ꟲ", "'S", "AA", " ", " e", "\t", "\tḋ", "̣'", "reİ", "👍🏽", "e", "😀🏽<", "EOT", ">fi", "🙂s", ",", " ", " '", "VEt", "!"]} +{"text": "漢ß\t​", "tokens": 5, "pieces": ["漢ß", "\t", "​"]} +{"text": " ​må12345678#$%<|fim_prefix|>½Ⅳ㋿字'D字­!é\rꟲ'T", "tokens": 45, "pieces": ["", " ", "​ma", "̊", "123", "456", "78", "#$%<|", "fim", "_prefix", "|>", "½", "<", "META", "_START", ">", "Ⅳ", "㋿字", "'D", "字", "­!", "e", "́\r", "ꟲ", "'T"]} +{"text": "㍿
'D9𐞁½dß\r\n\r\n\u000bEOTd'sDžså'M<|endoftext|>ZDž 
.é'Re'TZ \n𐞁Ⅳ\t", "tokens": 62, "pieces": ["㍿", "
", "'D", "9", "𐞁", "½", "dß", "\r\n\r\n", "\u000bEOTd", "'", "sDžsa", "̊'", "M", "<|", "endoftext", "|>", "ZDž", " ", "
", ".e", "́'", "Re", "'T", "Z", " \n", "𐞁", "", "Ⅳ", "\t"]} +{"text": "EOTm'll're'reé ­ m字,<|fim_prefix|>'Re'tfi‍'VE🙂0é'M'VEt' ㋿-\n
'ſ'ſ‍9EOT𐞁'S", "tokens": 58, "pieces": ["EOTm", "'ll", "'re", "'re", "é", " ­", " m字", ",<|", "fim", "_prefix", "|>'", "Re", "'t", "fi", "‍'", "VE", "🙂", "0", "é", "'M", "'VE", "t", "'", " ㋿-\n", "
", "'ſ", "'ſ", "‍", "9", "EOT𐞁", "'S"]} +{"text": "Ⅳ#$%ååß 'VE0fi㋿½'T'Re0-.12345678'>'s.🙂>ꟲ…", "tokens": 40, "pieces": ["Ⅳ", "#$%", "a", "̊a", "̊ß", " ", "'VE", "0", "fi", "㋿", "½", "'T", "'Re", "0", "-.", "123", "456", "78", "'>'", "s", ".🙂>", "ꟲ", "…"]} +{"text": "<|endoftext|>fi-㍿'T​", "tokens": 16, "pieces": ["<|", "endoftext", "|>", "fi", "-㍿'", "T", "​"]} +{"text": "İſ٣٤٥٦09字'­#$%\t'VE's'­e\"३\r \n<|fim_prefix|>s㋿Dž'T ", "tokens": 44, "pieces": ["İſ", "٣٤٥", "٦09", "字", "'­#$%", "\t", "'VE", "'s", "'­", "e", "\"", "३", "\r \n", "<|", "fim", "_prefix", "|>", "s", "㋿Dž", "'T", " "]} +{"text": "A​Aſſé-12345678.\u000b- ३½s'Tḍ̇#$%(ع 
ß\nDž'D'MDž 'seß", "tokens": 49, "pieces": ["A", "​A", "ſſé", "-", "123", "456", "78", ".", "\u000b", "-", " ", "३½", "s", "'T", "ḋ", "̣#$%(", "ع", " ", "
ß", "\n", "Dž", "'D", "'M", "Dž", " ", "'s", "eß"]} +{"text": "- \né'Sa'VEſ漢عſ𐞁..EOT<字,'s'ſ<\r\n\r\nt٣٤٥٦ꟲ\r\n\r\n👍🏽é­字9", "tokens": 58, "pieces": ["-", " \n", "e", "́'", "Sa", "'VE", "ſ漢عſ𐞁", "..", "EOT", "<字", ",'", "s", "'ſ", "<\r\n\r\n", "t", "٣٤٥", "٦", "ꟲ", "\r\n\r\n", "👍🏽", "e", "́­", "字", "9", ""]} +{"text": "9\n \n 'D,Dž'ع<<|fim_prefix|><|fim_prefix|>​EOT>İ'!ع字éée字('ll‍👍🏽\t'VEḍ̇字🙂a字 ,", "tokens": 63, "pieces": ["9", "\n \n", " ", "'D", ",Dž", "'ع", "<<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|><", "EOT", ">​", "EOT", ">İ", "'!", "ع字e", "́ée字", "('", "ll", "‍👍🏽", "\t", "'VE", "ḋ", "̣字", "🙂a字", " ", ","]} +{"text": "å<'re<|endoftext|>½ \n<|endoftext|>\r\n\r\n٣٤٥٦fi ​\r0'reZ \nEOTDž. ㍿­\r\n#$%İ, 9 \n\"'M'TA", "tokens": 62, "pieces": ["a", "̊<'", "re", "<|", "endoftext", "|>", "½", " \n", "<|", "endoftext", "|>\r\n\r\n", "٣٤٥", "٦", "fi", " ", " ​\r", "0", "'re", "Z", " \n", "EOTDž", ".", " ", "㍿­\r\n", "#$%", "İ", ",", " ", " ", "9", " \n", "\"'", "M", "'T", "A"]} +{"text": "\u000b<|endoftext|>\u000b㋿'s३ſ‍.'VE
\u000ba㍿Ⅳt​'Re\r ꟲ", "tokens": 47, "pieces": ["\u000b", "<|", "endoftext", "|>", "\u000b", "㋿'", "s", "३", "ſ", "‍.'", "VE", "
", "\u000ba", "㍿", "Ⅳ", "t", "​'", "Re", "\r", " ꟲ", ""]} +{"text": "EOT#$%😀🏽9㋿'T'D
<|endoftext|>'M,", "tokens": 27, "pieces": ["EOT", "#$%😀🏽", "9", "㋿'", "T", "'D", "
", "<|", "endoftext", "|>'", "M", ","]} +{"text": "'ſ ", "tokens": 7, "pieces": ["'ſ", "", " "]} +{"text": "éå'M字
>­​dé\r\n\r\n½\r\n, \u000bmfi.'llſ👍🏽 \n ع​t …é'T12345678fi('t\n½", "tokens": 52, "pieces": ["éa", "̊'", "M字", "
", ">­​", "dé", "\r\n\r\n", "½", "\r\n", ",", " ", "\u000bmfi", ".'", "llſ", "👍🏽", " \n", " ع", "​t", " ", "…é", "'T", "123", "456", "78", "fi", "('", "t", "\n", "½"]} +{"text": "٣٤٥٦å👍🏽t'M'!'ſm'Ś12345678…(ſ\u000b👍🏽漢's'Re'T!", "tokens": 50, "pieces": ["٣٤٥", "٦", "a", "̊👍🏽", "t", "'M", "'!'", "ſm", "'S", "́", "123", "456", "78", "…", "(ſ", "", "\u000b", "👍🏽", "漢", "'s", "'Re", "'T", "!"]} +{"text": "'s­'S\r", "tokens": 5, "pieces": ["'s", "­'", "S", "\r"]} +{"text": "́㋿ 
Ⅳ字! ‍Dž'T\rß(Aß漢字🙂.'M'VEDže'VE'ét٣٤٥٦…(9\r\n\r\n'VE\r\n\r\n
A", "tokens": 60, "pieces": ["́㋿", " ", "
", "Ⅳ", "字", "!", " ", " ‍", "Dž", "'T", "\r", "ß", "(Aß漢字", "🙂.'", "M", "'VE", "Dže", "'VE", "'e", "́t", "٣٤٥", "٦", "…", "(", "9", "\r\n\r\n", "'VE", "\r\n\r\n", "
A"]} +{"text": "DžéaA㋿ .,é
٣٤٥٦ßⅣ12345678", "tokens": 29, "pieces": ["Dže", "́aA", "㋿", " .,", "é", "
", "٣٤٥", "٦", "ß", "Ⅳ12", "345", "678"]} +{"text": " 👍🏽!!'M\t!!>½s३<|fim_prefix|>åA३'D\tZ\r\n\r\n'Re>#$%ḍ̇Ⅳd\r\n\r\n\tdt!!< t \"ع", "tokens": 66, "pieces": [" ", " 👍🏽!!'", "M", "\t", "!!>", "½", "s", "३", "<|", "fim", "_prefix", "|>", "a", "̊A", "३", "'D", "\tZ", "\r\n\r\n", "'Re", ">#$%", "ḋ", "̣", "Ⅳ", "d", "\r\n\r\n", "\t", "dt", "!!<", " t", " ", " \"", "ع"]} +{"text": "㍿३عAſ,>#$%!!'ll12345678'", "tokens": 24, "pieces": ["㍿", "३", "عA", "ſ", ",>#$%!!'", "ll", "123", "456", "78", "'"]} +{"text": "'S- !!\"😀🏽 aåé 'S
", "tokens": 25, "pieces": ["<", "EOT", ">'", "S", "-", " ", "!!\"😀🏽", " aa", "̊é", " ", "'S", "
"]} +{"text": "Dž( \n12345678sé#$%(12345678e😀🏽", "tokens": 20, "pieces": ["Dž", "(", " \n", "123", "456", "78", "sé", "#$%(", "123", "456", "78", "e", "😀🏽"]} +{"text": "'ReéDžé 😀🏽­㍿ 'lle", "tokens": 23, "pieces": ["'Re", "e", "́Dž", "e", "́", " 😀🏽­㍿", " ", "'ll", "e", ""]} +{"text": "३!㋿m\t'D㍿\r\n\r\n.'llDž 'lls'é<", "tokens": 26, "pieces": ["३", "!㋿", "m", "", "\t", "'D", "㍿\r\n\r\n", ".'", "llDž", " ", "'ll", "s", "'é", "<"]} +{"text": " \n'VE'Sd­ ع'Må\r\n\r\n'ree 'ReeZ's‍ ('re🙂 (\r\u000bꟲ", "tokens": 39, "pieces": [" \n", "'VE", "'S", "d", "­", " ع", "'M", "a", "̊\r\n\r\n", "'re", "e", " ", " '", "ReeZ", "'s", "‍", " ", "('", "re", "🙂<", "META", "_START", ">", " ", "(\r", "\u000bꟲ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'sZ", "tokens": 2, "pieces": ["'s", "Z"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n'S!!‍字\r\n\r\ns漢A‍漢0!!‍dém0漢-́Dž \r\n\r\n<漢're‍'M12345678ḍ̇", "tokens": 51, "pieces": ["\r\n", "'S", "!!‍", "字", "\r\n\r\n", "s漢A", "‍漢", "0", "!!‍", "de", "́m", "0", "漢", "-́", "Dž", " \r\n\r\n", "<漢", "'re", "‍'", "M", "123", "456", "78", "ḋ", "̣"]} +{"text": "<|endoftext|>𐞁३'S,ſ́!!e\t­\r\n'Reḍ̇'s'Ss३ ㍿#$%Z'ſa½<(!9", "tokens": 53, "pieces": ["<|", "endoftext", "|>", "𐞁", "३", "'S", ",ſ", "́!!", "e", "\t", "­<", "META", "_START", ">\r\n", "'Re", "ḋ", "̣'", "s", "'S", "s", "३", " ", "㍿#$%", "Z", "'ſ", "a", "½", "<(!", "9"]} +{"text": "méꟲé\n漢's<|fim_prefix|>
9३ éa A'VE३!!漢 \na'VE>éEOT<|endoftext|>ḍ̇ A,字'll#$%ſ", "tokens": 67, "pieces": ["me", "́ꟲe", "́\n", "漢", "'s", "<|", "fim", "_prefix", "|>", "
", "9", "", "३", " éa", " ", " A", "'VE", "३", "!!", "漢", " \n", "a", "'VE", ">éEOT", "<|", "endoftext", "|>", "ḋ", "̣", " A", ",字", "'ll", "#$%", "ſ"]} +{"text": "EOT३('ll㍿e😀🏽A'ſ­é<12345678
漢\u000b ́12345678", "tokens": 36, "pieces": ["EOT", "३", "('", "ll", "㍿e", "😀🏽", "A", "'ſ", "­é", "<", "123", "456", "78", "
漢", "\u000b", " ", "́", "123", "456", "78"]} +{"text": "!👍🏽'S'ſ \r\n'VE‍'!!'ḍ̇>Dž\r\n'ſ٣٤٥٦md é字ع!!<|endoftext|>0!!!'S
𐞁d", "tokens": 64, "pieces": ["!👍🏽'", "S", "'ſ", " \r\n", "'VE", "‍'!!'", "ḋ", "̣>", "Dž", "\r\n", "'ſ", "٣٤٥", "٦", "md", " e", "́字ع", "!!<|", "endoftext", "|>", "0", "!!!'", "S", "
𐞁d"]} +{"text": "é'ſtA'ſ\r\n<|fim_prefix|>\t'ſ३漢\r\n\u000b\nét!'s\r\n\r\n-Dž<|fim_prefix|>ع>9'VEİé㍿", "tokens": 55, "pieces": ["é", "'ſ", "tA", "'ſ", "\r\n", "<|", "fim", "_prefix", "|>", "\t", "'ſ", "३", "漢", "\r\n\u000b\n", "e", "́t", "!'", "s", "\r\n\r\n", "-Dž", "<|", "fim", "_prefix", "|>", "ع", ">", "9", "'VE", "İé", "㍿"]} +{"text": "<12345678👍🏽\r­‍", "tokens": 14, "pieces": ["<", "123", "456", "78", "👍🏽\r", "­‍"]} +{"text": "'s <'s!!ع\n-​\u000b Dž👍🏽Ⅳ's're<‍<|endoftext|>😀🏽\"'M 'll>٣٤٥٦ \tEOTm ", "tokens": 56, "pieces": ["'s", " ", "<'", "s", "!!", "ع", "\n", "-​", "\u000b", " Dž", "👍🏽", "Ⅳ", "'s", "'re", "<‍<|", "endoftext", "|>😀🏽\"'", "M", " ", "'ll", ">", "٣٤٥", "٦", " ", "\tEOTm", " "]} +{"text": "\r\n\"🙂fi é‍٣٤٥٦ḍ̇å<|endoftext|>
!å​ 'Mé'T9!!!३㍿.d12345678'VE'D<(", "tokens": 61, "pieces": ["\r\n", "\"🙂", "fi", " ", " e", "́‍", "٣٤٥", "٦", "ḋ", "̣a", "̊<|", "endoftext", "|>", "
", "!a", "̊​", " ", "'M", "é", "'T", "9", "!!!", "३", "㍿.", "d", "123", "456", "78", "'VE", "'D", "<("]} +{"text": "́ ḍ̇\"㍿'VEsⅣſ(#$%ß  
\rAعt\t𐞁!!👍🏽­sfi!'Reå-fiEOT", "tokens": 57, "pieces": ["́", " ḋ", "̣\"㍿'", "VEs", "Ⅳ", "ſ", "(#$%", "ß", "  
\r", "Aعt", "\t𐞁", "!!👍🏽­", "sfi", "!'", "Rea", "̊-", "fiEOT"]} +{"text": "'T‍३Ⅳ", "tokens": 7, "pieces": ["'T", "‍", "३Ⅳ"]} +{"text": "!!tA'reꟲ½ع­३a!s'll\r\n<<|fim_prefix|>'S're<|endoftext|>👍🏽 A \n,fi\r\n,İ'ſ👍🏽d#$%t-", "tokens": 64, "pieces": ["!!", "tA", "'re", "ꟲ", "½", "ع", "­", "३", "a", "!s", "'ll", "\r\n", "<<|", "fim", "_prefix", "|>'", "S", "'re", "<|", "endoftext", "|>👍🏽", " A", " \n", ",fi", "\r\n", ",İ", "'ſ", "👍🏽", "d", "#$%", "t", "-"]} +{"text": ">'s'll‍ꟲ\u000b​'s#$%'Td👍🏽", "tokens": 21, "pieces": [">'", "s", "'ll", "‍ꟲ", "\u000b", "​'", "s", "#$%'", "Td", "👍🏽"]} +{"text": "\r\n\"​'ſ漢\u000bⅣ-٣٤٥٦t.", "tokens": 24, "pieces": ["\r\n", "\"​'", "ſ漢", "\u000b", "Ⅳ", "-", "٣٤٥", "٦", "t", "."]} +{"text": "9​å\r\n\r\n''VE३d,​́és're'T'ſmḍ̇, \n é
\"mſİ", "tokens": 63, "pieces": ["9", "​a", "̊\r\n\r\n", "''", "VE", "३", "d", ",​́", "e", "́s", "'re", "'T", "'ſ", "mḋ", "̣<", "t", "­👍🏽", " \r\n", "👍🏽!", "ꟲ", "<-<", "META", "_START", ">,", " \n", " é", "
", "\"mſİ"]} +{"text": "‍\r\n㍿Zİ9é.>fi'VEEOT ٣٤٥٦å😀🏽'D  \n#$%‍'re߅\r\n\r\n\u000b\"a 12345678", "tokens": 58, "pieces": ["‍<", "EOT", ">\r\n", "㍿Zİ", "9", "é", ".>", "fi", "'VE", "EOT", " ", "٣٤٥", "٦", "a", "̊😀🏽'", "D", "  \n", "#$%‍'", "reß", "…\r\n\r\n", "\u000b", "\"a", " ", "123", "456", "78"]} +{"text": "A.३-\u000b9字ꟲZ'll㋿(a!𐞁sEOT'll.'å!! ḍ̇", "tokens": 43, "pieces": ["A", ".", "३", "-", "\u000b", "9", "字ꟲZ", "'ll", "㋿(", "a", "!𐞁s", "EOT", "'ll", ".'", "a", "̊!!", " ḋ", "̣"]} +{"text": "!!s'fifi👍🏽m…\u000ba\"Ⅳ're½.9İEOTꟲſe字ꟲ'ſ", "tokens": 41, "pieces": ["!!", "s", "'fifi", "👍🏽", "m", "…", "\u000ba", "\"", "Ⅳ", "'re", "½", ".", "9", "İEOTꟲſe字ꟲ", "'ſ"]} +{"text": "𐞁'Sß​<|fim_prefix|>ſ \n>㋿9ꟲA9 a", "tokens": 30, "pieces": ["𐞁", "'S", "ß", "​<|", "fim", "_prefix", "|>", "ſ", " \n", ">㋿", "9", "ꟲA", "9", " ", " a"]} +{"text": "fi३\u000b're'VEع", "tokens": 9, "pieces": ["fi", "३", "\u000b", "'re", "'VE", "ع"]} +{"text": "👍🏽'séd<­A㋿
'Mm'M½
", "tokens": 25, "pieces": ["👍🏽'", "se", "́d", "<­", "A", "㋿", "
", "'M", "m", "'M", "½", "
"]} +{"text": "'VEs-漢>Z'", "tokens": 8, "pieces": ["'VE", "s", "-漢", ">Z", "'"]} +{"text": "İḍ̇t're½e0😀🏽'\"d", "tokens": 18, "pieces": ["İḋ", "̣t", "'re", "½", "e", "0", "😀🏽'\"", "d"]} +{"text": "efi're🙂\r㋿(字
…", "tokens": 16, "pieces": ["efi", "'re", "🙂\r", "㋿(", "字", "
…"]} +{"text": "9​é", "tokens": 9, "pieces": ["", "9", "​", "é"]} +{"text": "ß😀🏽å's", "tokens": 11, "pieces": ["ß", "😀🏽", "a", "̊'", "s"]} +{"text": "‍Z<|fim_prefix|>३!!\n­", "tokens": 14, "pieces": ["‍Z", "<|", "fim", "_prefix", "|>", "३", "!!\n", "­"]} +{"text": "漢é.…", "tokens": 10, "pieces": ["漢e", "́.", "…", ""]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "9.9 ㋿é(\r\nİ,s
", "tokens": 18, "pieces": ["9", ".", "9", " ", " ㋿<", "EOT", ">é", "(\r\n", "İ", ",s", "
"]} +{"text": "\u000b #$%A \n'VE'ſ🙂́'Tsé'ſt ", "tokens": 28, "pieces": ["\u000b", " ", "#$%", "A", " \n", "'VE", "'ſ", "🙂<", "META", "_START", ">́'", "Tse", "́'", "ſt", " "]} +{"text": "𐞁३ \n٣٤٥٦'ll­‍'ReꟲEOT'D👍🏽'M'll<'ll.d\"'Md'T漢(", "tokens": 48, "pieces": ["𐞁", "३", " \n", "٣٤٥", "٦", "'ll", "­‍'", "ReꟲEOT", "'D", "👍🏽'", "M", "'ll", "<'", "ll", ".d", "\"'", "Md", "'T", "漢", "("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "0fiß.'re\"\u000b 'ſfiß!!'ll\r\n'S​'D㋿\r\n\r\n…", "tokens": 33, "pieces": ["0", "fiß", ".'", "re", "\"", "\u000b", "", " ", " '", "ſfiß", "!!'", "ll", "\r\n", "'S", "​'", "D", "㋿\r\n\r\n", "…"]} +{"text": "ß३'VEع<|endoftext|>Ⅳ👍🏽ß,", "tokens": 23, "pieces": ["ß", "३", "'VE", "ع", "<|", "endoftext", "|>", "Ⅳ", "👍🏽", "ß", ","]} +{"text": "t\r'T9­A'S🙂 #$%!'T\tt½'VE㋿
-m👍🏽\n
 \n", "tokens": 38, "pieces": ["t", "\r", "'T", "9", "­A", "'S", "🙂", " ", "#$%!'", "T", "", "\tt", "½", "'VE", "㋿", "
", "-m", "👍🏽\n", "
 \n"]} +{"text": "0ḍ̇Zdé.\r\n\r\nZ­<|endoftext|>as…'ll'M12345678<|fim_prefix|><\n‍ 'S👍🏽9!e'Re'Re", "tokens": 55, "pieces": ["0", "ḋ", "̣Zdé", ".\r\n\r\n", "Z", "­<|", "endoftext", "|>", "as", "…", "'ll", "'M", "123", "456", "78", "<|", "fim", "_prefix", "|><\n", "‍", " ", " '", "S", "👍🏽", "9", "!e", "'Re", "'Re"]} +{"text": "é漢ḍ̇", "tokens": 8, "pieces": ["é漢ḋ", "̣"]} +{"text": "ß٣٤٥٦'TZt३<|endoftext|>å.漢a\tm'TA A½e‍​ ㋿㋿'M
 \n<'Re \n٣٤٥٦́", "tokens": 67, "pieces": ["ß", "٣٤٥", "٦", "'T", "Zt", "३", "<|", "endoftext", "|>", "a", "̊.", "漢a", "\tm", "'T", "A", " A", "½", "e", "‍​", " ", "㋿㋿'", "M", "
 \n", "<'", "Re", " \n", "٣٤٥", "٦", "́"]} +{"text": " ㍿ 'll A字<|fim_prefix|>'Re <|fim_prefix|>٣٤٥٦'M'Mꟲ\r\nå>Ⅳ.𐞁EOT‍'ll\n\rt!\nm<|fim_prefix|>", "tokens": 72, "pieces": [" ", " ㍿", " ", "'ll", " A字", "<|", "fim", "_prefix", "|>'", "Re", " ", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "'M", "'M", "ꟲ", "\r\n", "a", "̊>", "Ⅳ", ".𐞁", "EOT", "‍'", "ll", "\n\r", "t", "!\n", "m", "<|", "fim", "_prefix", "|>"]} +{"text": "d\nḍ̇fiß\r'M😀🏽'S'T㋿ \r\n\r\n㍿<­漢٣٤٥٦'ReDž\n<|endoftext|>a--!!漢fi…\u000bé\r३­", "tokens": 65, "pieces": ["d", "\n", "ḋ", "̣fiß", "\r", "'M", "😀🏽'", "S", "'T", "㋿", " \r\n\r\n", "㍿<­", "漢", "٣٤٥", "٦", "'Re", "Dž", "\n", "<|", "endoftext", "|>", "a", "--!!", "漢fi", "…", "\u000bé", "\r", "३", "­"]} +{"text": "́a!!", "tokens": 3, "pieces": ["́a", "!!"]} +{"text": "a<|endoftext|>𐞁\nꟲ'S\r\nfiEOT12345678🙂", "tokens": 27, "pieces": ["a", "<|", "endoftext", "|>", "𐞁", "\n", "ꟲ", "'S", "\r\n", "fiEOT", "123", "456", "78", "🙂"]} +{"text": "👍🏽'reé㍿ 🙂0\t㋿\r#$% 
m!!é#$%🙂's㍿s'VEeſ'M#$%'ll漢Z,a( ", "tokens": 52, "pieces": ["👍🏽'", "reé", "㍿", " 🙂", "0", "\t", "㋿\r", "#$%", " ", "
m", "!!", "é", "#$%🙂'", "s", "㍿s", "'VE", "eſ", "'M", "#$%'", "ll漢Z", ",a", "(", " "]} +{"text": "👍🏽…㋿İd ", "tokens": 14, "pieces": ["👍🏽", "…", "㋿İd", " "]} +{"text": " 字 \u000bmEOT​ḍ̇ İ٣٤٥٦👍🏽'ſDža'T\r>'TDžDž!!٣٤٥٦\"'VE", "tokens": 57, "pieces": [" 字", " ", "\u000bmEOT", "​ḋ", "̣", " İ", "٣٤٥", "٦", "👍🏽'", "ſDža", "'T", "\r", ">'", "TDžDž", "!!", "٣٤٥", "٦", "\"<", "EOT", ">'", "VE"]} +{"text": "ḍ̇  \n é<|endoftext|>'ll<|endoftext|>é…\t \né\r\nm.'DéA#$%\u000bⅣ­", "tokens": 45, "pieces": ["ḋ", "̣", "  \n", " e", "́<|", "endoftext", "|>'", "ll", "<|", "endoftext", "|>", "é", "…\t \n", "é", "\r\n", "m", ".'", "Dé", "A", "#$%", "\u000b", "Ⅳ", "­"]} +{"text": "'­ßع<|endoftext|>'llss\"٣٤٥٦é", "tokens": 23, "pieces": ["'­", "ßع", "<|", "endoftext", "|>'", "llss", "\"", "٣٤٥", "٦", "é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "Ⅳ!!éé<|endoftext|>m
Dž'D ḍ̇\"३<|endoftext|>'👍🏽́'́'VEa!\n'll㋿́EOT<|endoftext|>!!‍9ḍ̇", "tokens": 69, "pieces": ["Ⅳ", "!!", "ée", "́<|", "endoftext", "|>", "m", "
Dž", "'D", " ḋ", "̣\"", "३", "<|", "endoftext", "|>'👍🏽́'́'", "VEa", "!\n", "'ll", "㋿́", "EOT", "<|", "endoftext", "|>!!‍", "9", "ḋ", "̣"]} +{"text": " ٣٤٥٦ſ!!ع𐞁-漢å漢३İ‍'M\"ع!A#$%EOTe9\u000b<|endoftext|>‍'VE ٣٤٥٦12345678 ée9>\r\n\r\n\r\n\r\n'S", "tokens": 81, "pieces": [" ", " ", "٣٤٥", "٦", "ſ", "!!", "ع𐞁", "-漢a", "̊漢", "३", "İ", "‍'", "M", "\"ع", "!A", "#$%", "EOTe", "9", "\u000b", "<|", "endoftext", "|>‍'", "VE", " ", "٣٤٥", "٦12", "345", "678", " e", "́e", "9", ">\r\n\r\n\r\n\r\n", "<", "EOT", ">'", "S"]} +{"text": "'D½!é0- \r\n\r\nd😀🏽 9­#$%Z…éꟲ e'Reé>漢'S9", "tokens": 37, "pieces": ["'D", "½", "!e", "́", "0", "-", " \r\n\r\n", "d", "😀🏽", " ", " ", "9", "­#$%", "Z", "…éꟲ", " e", "'Re", "é", ">漢", "'S", "9"]} +{"text": " \r\né t! ' #$%12345678!e12345678! \r", "tokens": 23, "pieces": [" \r\n", "e", "́", " t", "!", " ", "'", " ", " #$%", "123", "456", "78", "!e", "123", "456", "78", "!", " \r"]} +{"text": "EOT!'ſ­\r\nع", "tokens": 8, "pieces": ["EOT", "!'", "ſ", "­\r\n", "ع"]} +{"text": "㍿9#$%(d!!Á
ع٣٤٥٦\"\"eß\u000b \n!!𐞁𐞁٣٤٥٦​👍🏽'll 
", "tokens": 64, "pieces": ["㍿", "9", "#$%(", "d", "!!", "A", "́", "
ع", "", "٣٤٥", "٦", "\"\"", "eß", "\u000b \n", "!!", "𐞁", "𐞁", "٣٤٥", "٦", "​👍🏽'", "ll", " 
"]} +{"text": ", EOT\n<|endoftext|>'ReEOT
ſ'ſ㍿İſ\r\n\r\n㍿½🙂<|endoftext|>'\r\n\r\nⅣꟲm", "tokens": 51, "pieces": [",", " EOT", "\n", "<|", "endoftext", "|>'", "ReEOT", "
ſ", "'ſ", "㍿İſ", "\r\n\r\n", "㍿", "½", "🙂<|", "endoftext", "|>'\r\n\r\n", "Ⅳ", "ꟲ", "m"]} +{"text": "ß \n<|endoftext|>A'llEOT \r åé9é👍🏽㍿㍿'Ḿm\"ꟲḍ̇漢'ſ12345678'T३ \n're'('D\u000b", "tokens": 66, "pieces": ["ß", " \n", "<|", "endoftext", "|>", "A", "'ll", "EOT", " \r", " a", "̊e", "́", "9", "e", "́👍🏽㍿㍿'", "M", "́m", "\"ꟲḋ", "̣漢", "'ſ", "123", "456", "78", "'T", "३", " \n", "'re", "'('", "D", "\u000b"]} +{"text": "\n're😀🏽‍éḍ̇9㋿", "tokens": 20, "pieces": ["\n", "'re", "😀🏽‍", "e", "́ḋ", "̣", "9", "㋿"]} +{"text": "漢\u000b\r\n\r\n
 漢'redİ\r\n", "tokens": 13, "pieces": ["漢", "\u000b\r\n\r\n", "
", " 漢", "'re", "dİ", "\r\n"]} +{"text": "½!e", "tokens": 3, "pieces": ["½", "!e"]} +{"text": " 'VE​t३", "tokens": 7, "pieces": [" ", "'VE", "​t", "३"]} +{"text": "'M,d㋿\r>as>!'re\tſ́<|fim_prefix|>#$%-fi𐞁½㋿́\ré", "tokens": 37, "pieces": ["'M", ",d", "㋿\r", ">as", ">!'", "re", "\tſ", "́<|", "fim", "_prefix", "|>#$%-", "fi𐞁", "½", "㋿́\r", "é"]} +{"text": "e,😀🏽9\r\n#$%'Dt👍🏽s'ReⅣ", "tokens": 26, "pieces": ["e", ",😀🏽", "9", "\r\n", "#$%'", "Dt", "👍🏽", "s", "'Re", "Ⅳ", ""]} +{"text": "…\"\"½#$%😀🏽's!!\"ea< \n", "tokens": 18, "pieces": ["…", "\"\"", "½", "#$%😀🏽'", "s", "!!\"", "ea", "<", " \n"]} +{"text": "'TⅣ're㋿\r\n\r\n🙂'Re<|endoftext|> 'reꟲ ḍ̇m😀🏽字'S'ReåA", "tokens": 47, "pieces": ["'T", "Ⅳ", "'re", "㋿\r\n\r\n", "🙂<", "META", "_START", ">'", "Re", "<|", "endoftext", "|>", " ", "'re", "ꟲ", " ", " ḋ", "̣m", "😀🏽", "字", "'S", "'Re", "a", "̊A"]} +{"text": "'<漢…🙂<|endoftext|> ßé\r\n\r\nḍ̇Dž", "tokens": 28, "pieces": ["'<", "漢", "…", "🙂<", "META", "_START", "><|", "endoftext", "|>", " ße", "́\r\n\r\n", "ḋ", "̣Dž"]} +{"text": "a'fi.EOT( 'T-\r\n\r\n12345678….", "tokens": 17, "pieces": ["a", "'fi", ".EOT", "(", " '", "T", "-\r\n\r\n", "123", "456", "78", "…", "."]} +{"text": "­'Re<㋿", "tokens": 7, "pieces": ["­'", "Re", "<㋿"]} +{"text": " \n>😀🏽 -ع'S३
🙂ß,!'s<-'re'Mmm'TdEOT-­\r\n\r\n", "tokens": 38, "pieces": [" \n", "><", "META", "_START", ">😀🏽", " -", "ع", "'S", "३", "
", "🙂ß", ",!'", "s", "<-'", "re", "'M", "mm", "'T", "dEOT", "-­\r\n\r\n"]} +{"text": "å'Refie <|fim_prefix|>'TEOT a३d'sſ' \n!'reDž(߅(Ⅳeع", "tokens": 41, "pieces": ["a", "̊'", "Refie", " ", "<|", "fim", "_prefix", "|>'", "TEOT", " a", "३", "d", "'s", "ſ", "'", " \n", "!'", "reDž", "(ß", "…", "(", "Ⅳ", "eع"]} +{"text": "ع‍'ſDž12345678😀🏽 \n #$%(s<|endoftext|>'VE", "tokens": 29, "pieces": ["ع", "‍'", "ſDž", "123", "456", "78", "😀🏽", " \n", " ", " #$%(", "s", "<|", "endoftext", "|>'", "VE"]} +{"text": "‍👍🏽", "tokens": 8, "pieces": ["‍👍🏽"]} +{"text": "å 字­👍🏽ésm٣٤٥٦Dže12345678\r\n,9İ'Re
A", "tokens": 37, "pieces": ["a", "̊", " 字", "­👍🏽", "ésm", "٣٤٥", "٦", "Dže", "123", "456", "78", "\r\n", ",", "9", "İ", "'Re", "
A"]} +{"text": "#$%Zd (\r\n,'ſ漢\r\nDž#$%‍A𐞁", "tokens": 24, "pieces": ["#$%", "Zd", " ", "(\r\n", ",'", "ſ漢", "\r\n", "Dž", "#$%‍", "A𐞁"]} +{"text": "(a'D's<|fim_prefix|>(Z åt😀🏽EOTe!\r\na'TEOT…!", "tokens": 32, "pieces": ["(a", "'D", "'s", "<|", "fim", "_prefix", "|>(", "Z", " ", " a", "̊t", "😀🏽", "EOTe", "!\r\n", "a", "'T", "EOT", "…", "!"]} +{"text": "\u000b'De\ns", "tokens": 5, "pieces": ["\u000b", "'D", "e", "\n", "s"]} +{"text": "s", "tokens": 1, "pieces": ["s"]} +{"text": "'Refi EOTſ 𐞁\u000b字!'ReⅣ'ſ", "tokens": 23, "pieces": ["'Re", "fi", " ", " EOTſ", " ", " 𐞁", "\u000b字", "!'", "Re", "Ⅳ", "'ſ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "- 'llAſ İ.\r\n\r\n𐞁#$%½9\u000b㋿\u000b ​m'St‍å0ſ… ", "tokens": 46, "pieces": ["-", " '", "llAſ", " <", "EOT", ">İ", ".\r\n\r\n", "𐞁", "#$%", "½9", "\u000b", "㋿", "\u000b", " ", "​m", "'S", "t", "‍a", "̊", "0", "ſ", "… "]} +{"text": "\r\n'Rete\u000bfi'M\rꟲ9'll㍿ḍ̇ és'9Dž‍'T!​", "tokens": 33, "pieces": ["\r\n", "'Re", "te", "\u000bfi", "'M", "\r", "ꟲ", "9", "'ll", "㍿ḋ", "̣", " ", " és", "'", "9", "Dž", "‍'", "T", "!​"]} +{"text": "‍A'\r\n…\u000b#$%.9'T12345678
\u000b mſ\r\n\r\n", "tokens": 27, "pieces": ["‍A", "'\r\n", "…", "\u000b", "#$%.", "9", "'T", "123", "456", "78", "
\u000b", " m", "ſ", "\r\n\r\n"]} +{"text": "🙂A, 12345678…㋿­३٣٤٥٦12345678A३\r#$%<​­å", "tokens": 46, "pieces": ["🙂A", ",", " ", "123", "456", "78", "…", "㋿­", "३٣٤", "٥٦1", "234", "567", "8", "A", "३", "\r", "#$%<​­", "a", "̊"]} +{"text": "㍿é'M\t'S!!s\n,aİZé12345678‍­('M\"å😀🏽'M<|endoftext|>fiḍ̇d \n𐞁", "tokens": 54, "pieces": ["㍿e", "́'", "M", "\t", "'S", "!!", "s", "\n", ",aİZé", "123", "456", "78", "‍­('", "M", "\"a", "̊😀🏽'", "M", "<|", "endoftext", "|>", "fiḋ", "̣d", " \n", "𐞁"]} +{"text": " '😀🏽ḍ̇<|endoftext|><|endoftext|>Ⅳ\" 'ḍ̇🙂'S'S'll m'VE Ⅳ \n'🙂字'ſ'VE\rEOT\r‍‍!!\r9", "tokens": 69, "pieces": [" ", " '😀🏽", "ḋ", "̣<|", "endoftext", "|><|", "endoftext", "|>", "Ⅳ", "\"", " ", " '", "ḋ", "̣🙂'", "S", "'S", "'ll", " m", "'VE", " ", "Ⅳ", " \n", "'🙂", "字", "'ſ", "'VE", "\r", "EOT", "\r", "‍‍!!\r", "9"]} +{"text": "0<|fim_prefix|>(Dž \n㍿\u000bDž<…​漢A(,.12345678٣٤٥٦\"字​'Re㍿\r\n\r\n‍㍿", "tokens": 52, "pieces": ["0", "<|", "fim", "_prefix", "|>(", "Dž", " \n", "㍿", "\u000bDž", "<", "…", "​漢A", "(,.", "123", "456", "78٣", "٤٥٦", "\"字", "​'", "Re", "㍿\r\n\r\n", "‍㍿"]} +{"text": "漢३DžⅣ\n'ſ३fi \n 'Re'MA<|fim_prefix|>\r\n\r\n\r\n-ꟲ½!!<|endoftext|>½ 's½\r\n३ḍ̇\r\n\r\nſ
𐞁Ⅳa ‍٣٤٥٦", "tokens": 81, "pieces": ["漢", "३", "Dž", "Ⅳ", "\n", "'", "ſ", "३", "fi", " \n", " ", " '", "Re", "'M", "A", "<|", "fim", "_prefix", "|>\r\n\r\n\r\n", "-ꟲ", "½", "!!<|", "endoftext", "|>", "½", " ", " '", "s", "½", "\r\n", "३", "ḋ", "̣\r\n\r\n", "ſ", "
𐞁", "Ⅳ", "a", " ", "‍", "٣٤٥", "٦"]} +{"text": "\t\r\n<am
>('S're\n㋿'Re\u000b‍\"́ \n<ß0
t 're''½>­\"'S'reDž-", "tokens": 44, "pieces": ["\t", "\r\n", "<<", "META", "_START", ">am", "
", ">('", "S", "'re", "\n", "㋿'", "Re", "\u000b", "‍\"́", " \n", "<ß", "0", "
t", " ", "'re", "''", "½", ">­\"'", "S", "'re", "Dž", "-"]} +{"text": "\r\n<𐞁9'M​ß🙂'D", "tokens": 14, "pieces": ["\r\n", "<𐞁", "9", "'M", "​ß", "🙂'", "D"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "…ꟲ 𐞁'Té​Z(é\u000bA\n\u000b!\"!!#$%  ſ\r0're 'T(٣٤٥٦<|fim_prefix|>\r're<|endoftext|>12345678Ⅳ㍿\r 😀🏽", "tokens": 77, "pieces": ["…ꟲ", " 𐞁", "'T", "e", "́​", "Z", "(e", "́", "\u000bA", "\n", "\u000b", "!\"!!#$%", " ", " ſ", "\r", "0", "'re", " ", "'T", "(", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>\r", "'", "re", "<|", "endoftext", "|>", "123", "456", "78Ⅳ", "㍿\r", " 😀🏽"]} +{"text": "#$%eİ\nḍ̇", "tokens": 10, "pieces": ["#$%", "eİ", "\n", "ḋ", "̣"]} +{"text": "
-­😀🏽", "tokens": 9, "pieces": ["
", "-­😀🏽"]} +{"text": "(Ⅳß 'ſ-<|endoftext|>", "tokens": 16, "pieces": ["(", "Ⅳ", "ß", " ", "'ſ", "-<|", "endoftext", "|>"]} +{"text": "mt0#$%sA\nع 'T\r\n\r\nfiå(Džḍ̇İ \n\r", "tokens": 36, "pieces": ["mt", "0", "#$%", "sA", "\n", "ع", " ", "'T", "\r\n\r\n", "fia", "̊(<", "EOT", ">Džḋ", "̣İ", " \n\r"]} +{"text": "İ㋿#$%EOT𐞁३­#$%­-ꟲåḍ̇ⅣⅣḍ̇t's漢漢ꟲs'M'VEs\r\n½ İEOT٣٤٥٦A'll'M12345678", "tokens": 74, "pieces": ["İ", "㋿#$%", "EOT𐞁", "३", "­#$%­-", "ꟲa", "̊ḋ", "̣", "ⅣⅣ", "ḋ", "̣t", "'s", "漢漢ꟲs", "'M", "'VE", "s", "\r\n", "½", " İEOT", "٣٤٥", "٦", "A", "'ll", "'M", "123", "456", "78"]} +{"text": "ſd\r\n\r\ne!!12345678'a'Dſ漢  EOT<|endoftext|>t!! -',\n🙂<|endoftext|> <|endoftext|>", "tokens": 48, "pieces": ["ſd", "\r\n\r\n", "e", "!!", "123", "456", "78", "'a", "'D", "ſ漢", " ", " EOT", "<|", "endoftext", "|>", "t", "!!", " ", " -',\n", "🙂<|", "endoftext", "|>", " ", "<|", "endoftext", "|>"]} +{"text": "!.Z'Dfié", "tokens": 7, "pieces": ["!.", "Z", "'D", "fie", "́"]} +{"text": "​Zé<#$%12345678d
EOT३́\r\n\r\né­!EOTå'T'T३
\r\n12345678
", "tokens": 40, "pieces": ["​Ze", "́<#$%", "123", "456", "78", "d", "
EOT", "३", "́\r\n\r\n", "e", "́­!", "EOTa", "̊'", "T", "'T", "३", "
\r\n", "123", "456", "78", "
"]} +{"text": "ſ🙂'lléع́>😀🏽'VEe#$%😀🏽\tfiDžé'Re‍🙂ß'<Ⅳ9​'M㋿\r\nd'S \t>\n", "tokens": 64, "pieces": ["ſ", "🙂'", "lle", "́ع", "́>😀🏽'", "VEe", "#$%😀🏽", "\tfiDžé", "'Re", "‍🙂", "ß", "'<", "Ⅳ9", "​'", "M", "㋿\r\n", "d", "'S", " ", "", "\t", ">\n"]} +{"text": "İas👍🏽'ſé'llع\tDžé12345678!é!<İ\r\nfi 'VE", "tokens": 42, "pieces": ["İ", "as", "👍🏽'", "ſe", "́'", "ll", "ع", "\tDže", "́", "123", "456", "78", "!e", "́!<", "İ", "\r\n", "fi", " ", "'VE"]} +{"text": "Zdꟲ(EOT'T\n", "tokens": 9, "pieces": ["Zdꟲ", "(EOT", "'T", "\n"]} +{"text": "字'D­'M

'll>İ(½'s३!㍿‍İ's#$%ééaß're.'T \n\"\r\nß­٣٤٥٦ꟲfi😀🏽", "tokens": 57, "pieces": ["字", "'D", "­'", "M", "
", "
", "'ll", ">İ", "(", "½", "'s", "३", "!㍿‍", "İ", "'s", "#$%", "e", "́éaß", "'re", ".'", "T", " \n", "\"\r\n", "ß", "­", "٣٤٥", "٦", "ꟲfi", "😀🏽"]} +{"text": "9('VEß 'T<|fim_prefix|><|endoftext|>🙂\t३>!!å'D㋿(", "tokens": 49, "pieces": ["9", "('", "VEß", "", " ", "'", "T", "<|", "fim", "_prefix", "|><|", "endoftext", "|>🙂", "\t", "३", ">!!", "a", "̊'", "D", "㋿("]} +{"text": "½'D́e'sßEOT >'ReDžt", "tokens": 17, "pieces": ["½", "'", "D", "́e", "'s", "ßEOT", " ", ">'", "ReDžt"]} +{"text": "‍­", "tokens": 3, "pieces": ["‍­"]} +{"text": "<|endoftext|> 'M㋿ß\r\nⅣ㋿​'Re\ne\n 12345678 'Reḍ̇'VE\"ḍ̇", "tokens": 49, "pieces": ["<|", "endoftext", "|>", " '", "M", "㋿ß", "\r\n", "", "Ⅳ", "㋿​'", "Re", "\n", "e", "\n", " ", " ", "123", "456", "78", " ", "'Re", "ḋ", "̣'", "VE", "\"ḋ", "̣"]} +{"text": "é'SZß\r\n\r\n\u000b<|endoftext|>!!  \u000b >(ḍ̇Džds😀🏽'll'", "SZß", "\r\n\r\n", "\u000b", "<|", "endoftext", "|>!!", "  \u000b ", " >(", "ḋ", "̣Džds", "😀🏽'", "ll", "३#$%a'ſ'S­
's字m", "tokens": 24, "pieces": ["३", "\"<|", "endoftext", "|>", "३", "#$%", "a", "'ſ", "'S", "­", "
", "'s", "字m"]} +{"text": "'re٣٤٥٦‍ḍ̇'S'ſ漢m…३ſꟲ‍½  'ſ<|fim_prefix|>́'Re'Reع\té,ḍ̇.ß \n字'ſ", "tokens": 71, "pieces": ["'re", "٣٤٥", "٦", "‍ḋ", "̣'", "S", "'ſ", "漢m", "…", "३", "ſꟲ", "‍", "½", " ", " ", "'ſ", "<|", "fim", "_prefix", "|>́'", "Re", "'Re", "ع", "\te", "́,", "ḋ", "̣.", "ß", " \n", "字", "'ſ", ""]} +{"text": "-d(\u000b\"३fi३ \n🙂\r'T'M", "tokens": 16, "pieces": ["-d", "(", "\u000b", "\"", "३", "fi", "३", " \n", "🙂\r", "'T", "'M"]} +{"text": "\r\n字!\n0٣٤٥٦👍🏽㍿ⅣZ𐞁ß㍿\" é,Z 
३'M", "tokens": 44, "pieces": ["\r\n", "字", "!\n", "0٣٤", "٥٦", "👍🏽㍿", "Ⅳ", "Z𐞁ß", "㍿\"", " é", ",Z", " ", "
", "३", "'M"]} +{"text": ">'d12345678\"…ßⅣ\n12345678Džع𐞁<|endoftext|><|endoftext|>\r'llEOTd½ 'T😀🏽0'​EOT\"٣٤٥٦㋿'ll12345678'll'll,'D", "tokens": 78, "pieces": ["><", "EOT", ">'", "d", "123", "456", "78", "\"", "…ß", "Ⅳ", "\n", "123", "456", "78", "Džع𐞁", "<|", "endoftext", "|><|", "endoftext", "|>\r", "'ll", "EOTd", "½", " ", " '", "T", "😀🏽", "0", "'​", "EOT", "\"", "٣٤٥", "٦", "㋿'", "ll", "123", "456", "78", "'ll", "'ll", ",'", "D"]} +{"text": "­", "tokens": 1, "pieces": ["­"]} +{"text": "Džå\u000b'Re🙂é>'re", "tokens": 13, "pieces": ["Dža", "̊", "\u000b", "'Re", "🙂e", "́>'", "re"]} +{"text": "٣٤٥٦d,\n'll'sعfi \r\n!!é!\t'Sfi😀🏽(­", "tokens": 31, "pieces": ["٣٤٥", "٦", "d", ",\n", "'ll", "'s", "عfi", " \r\n", "!!", "é", "!", "\t", "'S", "fi", "😀🏽(­"]} +{"text": "🙂'sAå \ts漢Z'll-éع", "tokens": 18, "pieces": ["🙂'", "sAa", "̊", " ", "\ts漢Z", "'ll", "-e", "́ع"]} +{"text": "'Re½\r\nm", "tokens": 4, "pieces": ["'Re", "½", "\r\n", "m"]} +{"text": ">'s!! 漢३<|endoftext|>Aعa>!!<|endoftext|>\r!s\r\n‍ Dž \nⅣ's#$% \r\né٣٤٥٦<|endoftext|>'D0Z\t", "tokens": 71, "pieces": [">'", "s", "!!<", "META", "_START", ">", " 漢", "३", "<|", "endoftext", "|>", "Aعa", ">!!<|", "endoftext", "|>\r", "!s", "\r\n", "‍", " ", " Dž", " \n", "Ⅳ", "'s", "#$%", " \r\n", "e", "́", "٣٤٥", "٦", "<|", "endoftext", "|>'", "D", "0", "Z", "\t"]} +{"text": "'s,ßDž'Re'!!!e9'Då\r\nİ \n-㍿d\u000b'T", "tokens": 24, "pieces": ["'s", ",ßDž", "'Re", "'!!!", "e", "9", "'D", "a", "̊\r\n", "İ", " \n", "-㍿", "d", "\u000b", "'T"]} +{"text": "\u000b'T9 😀🏽'Té\n're\t́ß!'M <…
­😀🏽Ⅳa", "tokens": 34, "pieces": ["\u000b", "'T", "9", " ", "😀🏽'", "Té", "\n", "'re", "\t", "́ß", "!'", "M", " ", "<", "…", "
", "­😀🏽", "Ⅳ", "a"]} +{"text": "ḍ̇ꟲ‍m ­ 'Re㋿,'M ㍿-EOT‍́ \nع\n'ḾDž \n\"é'Red½éEOT'S", "tokens": 48, "pieces": ["ḋ", "̣ꟲ", "‍m", " ", " ­", " ", "'Re", "㋿,'", "M", " ", " ㍿-", "EOT", "‍́", " \n", "ع", "\n", "'M", "́Dž", " \n", "\"é", "'Re", "d", "½", "éEOT", "'S"]} +{"text": "'Re \n's­ع㋿字é \nEOT'Re\" 'M😀🏽EOTḍ̇0<|fim_prefix|>漢-!३½Ⅳ𐞁ꟲ's", "tokens": 59, "pieces": ["'Re", " \n", "'s", "­ع", "㋿字e", "́", " \n", "EOT", "'Re", "\"", " ", "'M", "😀🏽", "EOTḋ", "̣", "0", "<|", "fim", "_prefix", "|>", "漢", "-!", "३½Ⅳ", "𐞁ꟲ", "'s"]} +{"text": "fi!fi<‍A‍́😀🏽\r\n\r\n'll\"A漢'Re<|fim_prefix|><|fim_prefix|><‍㍿", "tokens": 44, "pieces": ["fi", "!fi", "<‍", "A", "‍́😀🏽\r\n\r\n", "'ll", "\"A漢", "'Re", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|><‍㍿"]} +{"text": "👍🏽'VE'Reé👍🏽d\" \n-\"𐞁d<|endoftext|>\r\n­EOT(12345678३'reA㋿", "tokens": 57, "pieces": ["👍🏽'", "VE", "'Re", "é", "👍🏽", "d", "\"", " \n", "-\"", "𐞁", "d", "<|", "endoftext", "|>\r\n", "­", "EOT", "(", "123", "456", "78३", "'re", "A", "㋿"]} +{"text": "Aå🙂.A㋿½EOTꟲ12345678İ'M's", "tokens": 29, "pieces": ["Aa", "̊🙂.", "A", "㋿<", "META", "_START", ">", "½", "EOTꟲ", "123", "456", "78", "İ", "'M", "'s"]} +{"text": "-Džt,#$%EOT's0ꟲ
👍🏽mé३ꟲ'S. ع字ß\"''re
å'S \u000bé ", "tokens": 53, "pieces": ["-Džt", ",#$%", "EOT", "'s", "0", "ꟲ", "
", "👍🏽", "mé", "३", "ꟲ", "'S", ".<", "META", "_START", ">", " ", " ع字ß", "\"''", "re", "
a", "̊'", "S", " ", "\u000be", "́", " "]} +{"text": "dDž", "tokens": 12, "pieces": ["", "३", " ", " <", "META", "_START", ">dDž"]} +{"text": "‍'ReDžfiİé'ſ\r'ſꟲA٣٤٥٦👍🏽e½\r\né.'M'", "tokens": 44, "pieces": ["‍'", "ReDžfiİé", "'ſ", "\r", "'ſ", "ꟲA", "٣٤٥", "٦", "👍🏽", "e", "½", "\r\n", "e", "́.'", "M", "'"]} +{"text": "Ⅳ ́0 \n>a

a½'<|fim_prefix|>𐞁ع(t\"<|fim_prefix|>'ſDžt", "tokens": 42, "pieces": ["", "Ⅳ", " ", "́", "0", " \n", ">a", "
", "
a", "½", "'<|", "fim", "_prefix", "|>", "𐞁ع", "(t", "\"<|", "fim", "_prefix", "|>'", "ſDžt"]} +{"text": "\r\n\r\n​३<|endoftext|>­'ll\u000b s㋿a
", "tokens": 23, "pieces": ["\r\n\r\n", "​", "३", "<|", "endoftext", "|>­'", "ll", "\u000b", " s", "㋿a", "
"]} +{"text": "\u000b<|endoftext|>0\n9𐞁fi'Så\u000b‍>㍿é𐞁m İtéé (12345678", "tokens": 47, "pieces": ["\u000b", "<|", "endoftext", "|>", "0", "\n", "9", "𐞁fi", "'S", "a", "̊", "\u000b", "‍>㍿", "e", "́𐞁m", " İ", "téé", " ", "(", "123", "456", "78"]} +{"text": "½'ſfi👍🏽‍\u000b\r\nEOT \r\n\r\nfi'M12345678\u000bdA9 \"٣٤٥٦<<|fim_prefix|> ", "tokens": 52, "pieces": ["", "½", "'ſ", "fi", "👍🏽‍", "\u000b\r\n", "EOT", " \r\n\r\n", "fi", "'M", "123", "456", "78", "\u000bdA", "9", " \"", "٣٤٥", "٦", "<<|", "fim", "_prefix", "|>", " "]} +{"text": "(\n漢EOT😀🏽", "tokens": 10, "pieces": ["(\n", "漢EOT", "😀🏽"]} +{"text": "ſ㍿aZ ß'Re'é…​AعⅣéfi12345678½", "tokens": 27, "pieces": ["ſ", "㍿aZ", " ß", "'Re", "'e", "́", "…", "​Aع", "Ⅳ", "éfi", "123", "456", "78½"]} +{"text": "İ'D㍿ \n字é‍ßd😀🏽ḍ̇a>'ſ漢 >12345678ß<>'ll fi'😀🏽<|fim_prefix|>", "tokens": 53, "pieces": ["İ", "'D", "㍿", " \n", "字e", "́‍", "ßd", "😀🏽", "ḋ", "̣a", ">'", "ſ漢", " >", "123", "456", "78", "ß", "<>'", "ll", " fi", "'😀🏽<|", "fim", "_prefix", "|>"]} +{"text": "\"'M\t\r\ns12345678.éé\"\tß‍\"!!é- 
a#$%ß  \n 'VE", "tokens": 36, "pieces": ["'Re", "½", "👍🏽", "½", "'ſ", "ſ", "!<|", "endoftext", "|>\"!!", "é", "-", " ", "
a", "#$%", "ß", "  \n", " ", " '", "VE"]} +{"text": "㍿!A字Dž", "tokens": 9, "pieces": ["㍿!", "A字Dž"]} +{"text": "٣٤٥٦'Sꟲ\r\n\r\n😀🏽é(!>'re字é㋿e
-́​😀🏽́\r\néDž<|fim_prefix|>𐞁​​'llå\"😀🏽", "tokens": 71, "pieces": ["٣٤٥", "٦", "'S", "ꟲ", "\r\n\r\n", "😀🏽", "e", "́(!>'", "re字e", "́㋿", "e", "
", "-́<", "EOT", ">​😀🏽́\r\n", "éDž", "<|", "fim", "_prefix", "|>", "𐞁", "​​'", "lla", "̊\"😀🏽"]} +{"text": "…

…‍Ⅳ'D'ſ'll.½fi\r\n\r\n-9<|endoftext|> 漢!!", "tokens": 36, "pieces": ["…

", "…", "‍", "Ⅳ", "'D", "'ſ", "'ll", ".", "½", "fi", "\r\n\r\n", "-", "9", "<|", "endoftext", "|>", " ", " 漢", "!!"]} +{"text": "'D́
's​EOT٣٤٥٦İꟲ\r\n\r\nA 9dß'Sm<|endoftext|>😀🏽
's", "tokens": 60, "pieces": ["'D", "́", "
", "'s", "​EOT", "٣٤٥", "٦", "İꟲ", "\r\n\r\n", "A", " ", " ", "9", "d", "", "ß", "'S", "m", "<|", "endoftext", "|>😀🏽", "
", "'s"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ع\n'D(ß㍿‍-'D😀🏽ß½Z'll\u000b'reİ", "tokens": 27, "pieces": ["ع", "\n", "'D", "(ß", "㍿‍-'", "D", "😀🏽", "ß", "½", "Z", "'", "ll", "\u000b", "'re", "İ"]} +{"text": "Ⅳ!!👍🏽Ⅳ🙂ſ ㋿!>½\rß½ ''D<fi'ſ", "tokens": 40, "pieces": ["Ⅳ", "!!👍🏽", "Ⅳ", "🙂ſ", " ", "㋿!>", "½", "\r", "ß", "", "½", " ", " ''", "D", "<fi", "'ſ"]} +{"text": "́EOT'D<‍!'S9字🙂\nå<|fim_prefix|>.a\t\r\né!!0<|endoftext|>Ⅳİ'll!\" 0", "tokens": 44, "pieces": ["́EOT", "'D", "<‍!'", "S", "9", "字", "🙂\n", "a", "̊<|", "fim", "_prefix", "|>.", "a", "\t\r\n", "e", "́!!", "0", "<|", "endoftext", "|>", "Ⅳ", "İ", "'ll", "!\"", " ", "0"]} +{"text": "<9\u000b'ſmⅣ'MßtA😀🏽ꟲ'reع ع'M<,ſİé'ſ-\r\n9A­漢 9👍🏽.३mt", "tokens": 59, "pieces": ["<", "9", "\u000b", "'ſ", "m", "Ⅳ", "'M", "ßtA", "😀🏽", "ꟲ", "'re", "ع", " ع", "'M", "<,", "ſİe", "́'", "ſ", "-\r\n", "9", "A", "­漢", " ", "9", "👍🏽<", "EOT", ">.", "३", "mt"]} +{"text": "0'M", "tokens": 2, "pieces": ["0", "'M"]} +{"text": "åe𐞁
\r,ß'VEa12345678ḍ̇a'Ree.> 
́😀🏽'T㋿Z,'ſ\r\n 𐞁åé\t", "tokens": 62, "pieces": ["a", "̊e𐞁", "
\r", ",ß", "'VE", "a", "123", "456", "78", "ḋ", "̣a", "'Re", "e", ".>", " ", "
", "́😀🏽'", "T", "㋿Z", ",'", "ſ", "\r\n", " ", " 𐞁a", "̊é", "\t"]} +{"text": "'ſ́é \n½\n's𐞁>a​ꟲfi-Z'ſ\r\n㋿ é \n
Dž…́'M…­éⅣ😀🏽字ſ're'Re­d'VE", "tokens": 61, "pieces": ["'ſ", "́é", " \n", "½", "\n", "'s", "𐞁", ">a", "​ꟲfi", "-Z", "'ſ", "\r\n", "㋿", " e", "́", " \n", "
Dž", "…", "́'", "M", "…", "­e", "́", "Ⅳ", "😀🏽", "字ſ", "'re", "'Re", "­d", "'VE"]} +{"text": "\r\n\r\nt 😀🏽ßfi'reé'ſe\u000b漢!ß 'Reſ'Mع<|endoftext|>ſ Afi-s字㋿", "tokens": 48, "pieces": ["\r\n\r\n", "t", " ", "😀🏽", "ßfi", "'re", "e", "́'", "ſe", "\u000b漢", "!ß", " ", "'Re", "ſ", "'M", "ع", "<|", "endoftext", "|>", "ſ", " ", " Afi", "-s字", "㋿"]} +{"text": "\n👍🏽!!9'Re😀🏽", "tokens": 15, "pieces": ["\n", "👍🏽!!", "9", "'Re", "😀🏽"]} +{"text": "<|fim_prefix|>½'½ 'D", "tokens": 15, "pieces": ["<|", "fim", "_prefix", "|>", "½", "'", "½", " ", "'", "D"]} +{"text": "٣٤٥٦Ⅳ", "tokens": 10, "pieces": ["٣٤٥", "٦Ⅳ"]} +{"text": "#$% \nꟲ\"A.😀🏽 9're( ٣٤٥٦ ㍿😀🏽", "tokens": 38, "pieces": ["#$%", " \n", "ꟲ", "\"A", ".😀🏽", " ", " ", "9", "'re", "(", " ", "٣٤٥", "٦", " ", "㍿😀🏽"]} +{"text": "'re'T 😀🏽½EOTſ\r\n\r\n㋿912345678<|endoftext|>'ReeA \n12345678Z,\n'reḍ̇ḍ̇\n", "tokens": 49, "pieces": ["'re", "'T", " ", "😀🏽", "½", "EOTſ", "\r\n\r\n", "㋿", "912", "345", "678", "<|", "endoftext", "|>'", "ReeA", " \n", "123", "456", "78", "Z", ",\n", "'re", "ḋ", "̣ḋ", "̣\n"]} +{"text": "​.ſ", "tokens": 4, "pieces": ["​.", "ſ"]} +{"text": "㍿Ⅳ\téfia\t́,''M> İ३", "tokens": 19, "pieces": ["㍿", "Ⅳ", "\te", "́fia", "\t", "́,''", "M", ">", " ", " İ", "३"]} +{"text": ">👍🏽e㋿\né​ ३('D", "tokens": 23, "pieces": [">👍🏽", "e", "㋿\n", "e", "́​", " ", "३", "('", "D"]} +{"text": "\n'>'Ddé🙂12345678å'Mع0 9 ३㍿!!", "tokens": 29, "pieces": ["\n", "'>'", "Dde", "́🙂", "123", "456", "78", "a", "̊'", "Mع", "0", " ", " ", "9", " ", " ", "३", "㍿!!"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "İåfiå \n­\r\n\r\nt​.ꟲ­-'llꟲ\r\né!!DžA!d🙂\" a'D \né‍ßde(fi \n🙂", "tokens": 54, "pieces": ["İa", "̊fia", "̊", " \n", "­\r\n\r\n", "t", "​.", "ꟲ", "­-'", "llꟲ", "\r\n", "e", "́!!", "DžA", "!d", "🙂\"", " a", "'D", " \n", "e", "́‍", "ßde", "(fi", " \n", "🙂"]} +{"text": "\n­\nꟲ½e\n😀🏽…عİ \n㍿!\"'Re", "tokens": 25, "pieces": ["\n", "­\n", "ꟲ", "½", "e", "\n", "😀🏽", "…عİ", " \n", "㍿!\"'", "Re"]} +{"text": "e><<|fim_prefix|>å'Dع9<'VEé'D…½३! 'S😀🏽's👍🏽", "tokens": 43, "pieces": ["e", "><<|", "fim", "_prefix", "|>", "a", "̊'", "Dع", "9", "<'", "VEe", "́'", "D", "…", "½३", "!", " ", "'S", "😀🏽'", "s", "👍🏽"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": " \u000b'M\nEOT  \n<|fim_prefix|>'SEOTſ'll'VE‍Dž'll字'\r9.'reع ", "tokens": 37, "pieces": [" ", "\u000b", "'M", "\n", "EOT", "  \n", "<|", "fim", "_prefix", "|>'", "SEOTſ", "'ll", "'VE", "‍Dž", "'ll", "字", "'\r", "9", ".'", "reع", " "]} +{"text": "t (12345678😀🏽字½ع'DⅣ👍🏽㍿\"👍🏽<|endoftext|>İfiꟲⅣ(", "tokens": 52, "pieces": ["t", " ", "(", "123", "456", "78", "😀🏽", "字", "½", "ع", "'D", "Ⅳ", "👍🏽㍿\"👍🏽<", "EOT", "><|", "endoftext", "|>", "İfiꟲ", "Ⅳ", "("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\n \n12345678👍🏽İ'llZ!!ſ\u000b\"<㋿Džtİ'ReİAa👍🏽'Så\t<½,. d'T", "tokens": 57, "pieces": ["\n \n", "123", "456", "78", "👍🏽", "İ", "'ll", "Z", "!!", "ſ", "\u000b", "\"<㋿", "Dž", "tİ", "'Re", "İAa", "👍🏽'", "Sa", "̊", "\t", "<", "½", ",.", " d", "'", "T"]} +{"text": "ḍ̇㍿ḍ̇", "tokens": 13, "pieces": ["ḋ", "̣㍿", "ḋ", "̣"]} +{"text": "'VE㍿EOT\nİ", "tokens": 9, "pieces": ["'VE", "㍿EOT", "\n", "İ"]} +{"text": "Ⅳ́ 's \n", "tokens": 6, "pieces": ["Ⅳ", "́", " '", "s", " \n"]} +{"text": "(", "tokens": 10, "pieces": ["a", "̊<|", "endoftext", "|>("]} +{"text": "!‍ \n'Tſ9<𐞁Ⅳ\rt'sfi!!", "tokens": 21, "pieces": ["!‍", " \n", "'T", "ſ", "9", "<𐞁", "Ⅳ", "\r", "t", "'s", "fi", "!!"]} +{"text": "ḍ̇‍A'Mſİع́İ0fi", "tokens": 23, "pieces": ["ḋ", "̣‍", "A", "'M", "ſİع", "́<", "EOT", ">İ", "0", "fi"]} +{"text": "ß👍🏽 ꟲ.'D­字\"😀🏽'll>ع< ſ \r\n\r\n
𐞁😀🏽Z‍٣٤٥٦e", "tokens": 55, "pieces": ["ß", "👍🏽", " ꟲ", ".'", "D", "­字", "\"😀🏽'", "ll", ">", "ع", "<", " ſ", " \r\n\r\n", "
𐞁", "😀🏽", "Z", "‍", "٣٤٥", "٦", "e"]} +{"text": "㋿\"'sꟲ. åéé(😀🏽", "tokens": 20, "pieces": ["㋿\"'", "sꟲ", ".", " a", "̊éé", "(😀🏽"]} +{"text": "!", "tokens": 1, "pieces": ["!"]} +{"text": "<ḍ̇\" 'Re!!#$%​!
😀🏽'T", "tokens": 27, "pieces": ["<ḋ", "̣\"", " ", "'Re", "!!#$%​!", "
", "😀🏽'", "T"]} +{"text": "'VE'T­​A #$%t​ \r\n\r\nꟲm0å'३!\tfit<|fim_prefix|>\r12345678'D.'VE<|fim_prefix|>'re", "tokens": 52, "pieces": ["'VE", "'T", "­​", "A", " ", "#$%", "t", "​", " \r\n\r\n", "ꟲm", "0", "a", "̊'", "३", "!", "\tfit", "<|", "fim", "_prefix", "|>\r", "123", "456", "78", "'D", ".'", "VE", "<|", "fim", "_prefix", "|>'", "re"]} +{"text": "ZZß👍🏽٣٤٥٦d'D漢𐞁\r\n'll'ſ…😀🏽!!\"㍿", "tokens": 44, "pieces": ["ZZ", "ß", "👍🏽", "٣٤٥", "٦", "d", "'D", "漢𐞁", "\r\n", "'ll", "'ſ", "…", "😀🏽!!\"㍿"]} +{"text": "½'Mdꟲ \"9Afi's'D'ſåé
\r\n\n!ḍ̇\"\r\n'T0ḍ̇👍🏽漢 漢12345678👍🏽ع!\"", "tokens": 64, "pieces": ["½", "'M", "dꟲ", " ", "\"", "9", "Afi", "'", "s", "'D", "'ſ", "a", "̊é", "
\r\n\n", "!ḋ", "̣\"\r\n", "'T", "0", "ḋ", "̣👍🏽", "漢", " 漢", "123", "456", "78", "👍🏽", "ع", "!\""]} +{"text": "EOT\r\n\r\n'MZ'D<|endoftext|>ḍ̇'T\"'ſ<|endoftext|>", "tokens": 32, "pieces": ["EOT", "\r\n\r\n", "'", "MZ", "'D", "<|", "endoftext", "|>", "ḋ", "̣'", "T", "\"'", "ſ", "<|", "endoftext", "|>"]} +{"text": "İ\rß>t٣٤٥٦éfi", "tokens": 16, "pieces": ["İ", "\r", "ß", ">t", "٣٤٥", "٦", "e", "́fi"]} +{"text": " ſ!!!\r\n\r\n㍿'TDž३é!'Ś'll9३ḍ̇", "tokens": 33, "pieces": [" ſ", "!!!\r\n\r\n", "㍿'", "TDž", "३", "e", "́!'", "S", "́'", "ll", "9३", "ḋ", "̣"]} +{"text": "\"#$%é🙂fi", "tokens": 8, "pieces": ["\"#$%", "é", "🙂fi"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "(٣٤٥٦३𐞁\r-'D́'漢'ſعEOT.ſt\u000b'…-<|fim_prefix|>😀🏽٣٤٥٦\"12345678ꟲs𐞁\u000b<|fim_prefix|>ḍ̇EOT-,", "tokens": 88, "pieces": ["(", "٣٤٥", "٦३", "𐞁", "\r", "-'", "D", "́'", "漢", "'ſ", "عEOT", ".ſt", "\u000b", "'", "…", "-<|", "fim", "_prefix", "|>😀🏽", "٣٤٥", "٦", "\"", "123", "456", "78", "ꟲs𐞁", "", "\u000b", "<|", "fim", "_prefix", "|>", "ḋ", "̣EOT", "-,"]} +{"text": "-­(Ⅳe>Zع,", "tokens": 13, "pieces": ["-­(", "Ⅳ", "e", ">Z", "ع", ","]} +{"text": "㋿\u000b㍿\r\nİ'\r😀🏽'ſ😀🏽a!!'ſ\t>.漢٣٤٥٦ 's㋿9.ås \n\r\n\r\n0 \n'll\"㋿<|endoftext|>m", "tokens": 72, "pieces": ["㋿", "\u000b", "㍿\r\n", "İ", "'\r", "😀🏽'", "ſ", "😀🏽", "a", "!!'", "ſ", "\t", ">.", "漢", "٣٤٥", "٦", " ", " <", "META", "_START", ">'", "s", "㋿", "9", ".a", "̊s", " \n\r\n\r\n", "0", " \n", "'ll", "\"㋿<|", "endoftext", "|>", "m"]} +{"text": "ſ eⅣt're漢<ḍ̇'Dd'DEOT're​", "tokens": 30, "pieces": ["ſ", " e", "Ⅳ", "t", "'re", "漢", "<ḋ", "̣'", "Dd", "'", "D", "EOT", "'re", "​"]} +{"text": "\u000bEOT00<|endoftext|>‍Z,ḍ̇12345678<|fim_prefix|>sEOTs­am <|fim_prefix|>å👍🏽sA漢", "tokens": 60, "pieces": ["\u000bEOT", "00", "<|", "endoftext", "|>‍", "Z", ",<", "EOT", ">ḋ", "̣", "123", "456", "78", "<|", "fim", "_prefix", "|>", "sEOTs", "­am", " ", " <|", "fim", "_prefix", "|>", "a", "̊👍🏽", "sA漢"]} +{"text": "<|endoftext|>9<ꟲEOTⅣfi!tⅣ ß\r\n\r\n​-३'Re", "tokens": 31, "pieces": ["<|", "endoftext", "|>", "9", "<ꟲEOT", "Ⅳ", "fi", "!t", "Ⅳ", " ", " ß", "\r\n\r\n", "​-", "३", "'Re"]} +{"text": "'D😀🏽 'D0٣٤٥٦\r9m🙂\r \n‍", "tokens": 26, "pieces": ["'D", "😀🏽", " '", "D", "0٣٤", "٥٦", "\r", "9", "m", "🙂\r", " \n", "‍"]} +{"text": "é'M🙂Dž
,<㍿'ll\"\t字aḍ̇字<|fim_prefix|>'ſ!!d🙂<|fim_prefix|>İ'VE .‍'s㍿'VE\t12345678İ\u000bⅣé", "tokens": 76, "pieces": ["é", "'M", "🙂<", "META", "_START", ">Dž", "
", ",<㍿'", "ll", "\"", "\t字aḋ", "̣字", "<|", "fim", "_prefix", "|>'", "ſ", "!!", "d", "🙂<|", "fim", "_prefix", "|>", "İ", "'VE", " ", " .‍'", "s", "㍿'", "VE", "\t", "123", "456", "78", "İ", "\u000b", "Ⅳ", "e", "́"]} +{"text": "0, å漢 ſ", "tokens": 11, "pieces": ["0", ",", " a", "̊漢", " ", " ſ"]} +{"text": "'s<‍\n're‍🙂 \n#$% 'Re'Ms😀🏽s\r'llt…fiḍ̇'ll​ßİ0…\"<ß‍", "tokens": 52, "pieces": ["'s", "<‍\n", "'re", "‍🙂", " \n", "#$%", " ", "'Re", "'M", "s", "😀🏽", "s", "\r", "'ll", "t", "…fiḋ", "̣'", "ll", "​ßİ", "0", "…", "\"<", "ß", "‍"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "9'SDž<|fim_prefix|> 𐞁<|endoftext|>'ll !>㍿\u000b", "tokens": 31, "pieces": ["9", "'S", "Dž", "<|", "fim", "_prefix", "|>", " 𐞁", "<|", "endoftext", "|>'", "ll", " ", " !>㍿", "\u000b"]} +{"text": "㍿EOT'ſ㋿\u000b\r\n\r\n 's \nd \nع🙂0!!ſDž'T>éd३é,Ⅳt<​‍İ12345678\r\n\r\n'S\r", "tokens": 56, "pieces": ["㍿EOT", "'ſ", "㋿", "\u000b\r\n\r\n", " ", "'s", " \n", "d", " \n", "ع", "🙂", "0", "!!", "ſDž", "'T", ">éd", "३", "e", "́,", "Ⅳ", "t", "<​<", "EOT", ">‍", "İ", "123", "456", "78", "\r\n\r\n", "'S", "\r", ""]} +{"text": "㍿-㋿٣٤٥٦d🙂>t 's\r\n\r\nZ.'VEé\r\nDž å👍🏽‍ ", "tokens": 43, "pieces": ["㍿-㋿", "٣٤٥", "٦", "d", "🙂>", "t", " '", "s", "\r\n\r\n", "Z", ".'", "VEé", "\r\n", "Dž", " ", " a", "̊👍🏽‍", " "]} +{"text": "́('D…😀🏽ß\r\n 'll'Re'sⅣ\r\n", "tokens": 19, "pieces": ["́('", "D", "…", "😀🏽", "ß", "\r\n", " '", "ll", "'Re", "'s", "Ⅳ", "\r\n"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "½‍😀🏽'll'Mfiå< 12345678­👍🏽å
ḍ̇\r\ń㍿'Reå​ḍ̇", "tokens": 57, "pieces": ["", "½", "‍😀🏽'", "ll", "'M", "fia", "̊<", " ", "123", "456", "78", "­👍🏽", "a", "̊", "
ḋ", "̣\r\n", "́㍿'", "Rea", "̊​", "ḋ", "̣"]} +{"text": ",ß\r\n'S<ſée‍tt", "tokens": 11, "pieces": [",ß", "\r\n", "'S", "<ſée", "‍tt"]} +{"text": "ſEOT<|fim_prefix|>\u000b0 ꟲ<|fim_prefix|>İ\t <|endoftext|>​<|endoftext|>t,9", "tokens": 43, "pieces": ["ſEOT", "<|", "fim", "_prefix", "|>", "\u000b", "0", " ", " ꟲ", "<|", "fim", "_prefix", "|>", "İ", "\t ", " <|", "endoftext", "|>​<|", "endoftext", "|>", "t", ",", "9"]} +{"text": "\r\n\r\nع३", "tokens": 4, "pieces": ["\r\n\r\n", "ع", "३"]} +{"text": "#$%\r‍­Džḍ̇'T\nḍ̇🙂>!!é'VEé.-𐞁­­A🙂ſ'Sſ\t'ſع字>", "tokens": 60, "pieces": ["#$%\r", "‍­", "Džḋ", "̣'", "T", "\n", "ḋ", "̣🙂>!!", "e", "́'", "VEe", "́.-", "𐞁", "­­", "A", "🙂ſ", "'S", "ſ", "", "\t", "'", "ſع字", ">"]} +{"text": "𐞁\r\n½d \n'Reséé'ſſ\"­'llⅣİ😀🏽<|endoftext|>fi", "tokens": 38, "pieces": ["𐞁", "\r\n", "½", "d", " \n", "'Re", "se", "́é", "'ſ", "ſ", "\"­'", "ll", "Ⅳ", "İ", "😀🏽<|", "endoftext", "|>", "fi"]} +{"text": "å<|fim_prefix|>'T<\n\na👍🏽\rſ#$%'ll\r<|endoftext|>-'T(\u000b'm0‍ß'ReDž", "tokens": 50, "pieces": ["a", "̊<|", "fim", "_prefix", "|>'", "T", "<\n\n", "a", "👍🏽\r", "ſ", "#$%'", "ll", "\r", "<|", "endoftext", "|>-'", "T", "(", "\u000b", "'m", "0", "‍ß", "'", "ReDž"]} +{"text": "ßع👍🏽m-'ſ'('M", "tokens": 15, "pieces": ["ßع", "👍🏽", "m", "-'", "ſ", "'('", "M"]} +{"text": "‍EOTꟲ٣٤٥٦", "tokens": 15, "pieces": ["‍EOTꟲ", "٣٤٥", "٦"]} +{"text": "'M½!!\n​ \nⅣꟲ́ ㋿३Ⅳꟲ<|fim_prefix|>'  stfi<|fim_prefix|>'VEm då\n​", "tokens": 49, "pieces": ["'M", "½", "!!\n", "​", " \n", "Ⅳ", "ꟲ", "́", " ", " ㋿", "३Ⅳ", "ꟲ", "<|", "fim", "_prefix", "|>'", "  ", " stfi", "<|", "fim", "_prefix", "|>'", "VEm", " da", "̊\n", "​"]} +{"text": "👍🏽‍½t", "tokens": 14, "pieces": ["👍🏽‍<", "EOT", ">", "½", "t"]} +{"text": "\r\n,s!!( 'M ꟲ\"‍😀🏽 ​😀🏽EOT'reé", "tokens": 29, "pieces": ["\r\n", ",s", "!!(", " ", "'M", " ꟲ", "\"‍😀🏽", " ", "​😀🏽", "EOT", "'re", "e", "́"]} +{"text": "'M- é \n字Dž'll9…EOTḍ̇eİaⅣ\nfi'llA'S漢.٣٤٥٦½​9.…\n", "tokens": 53, "pieces": ["'M", "-", " é", " \n", "字Dž", "'ll", "9", "…EOTḋ", "̣eİa", "Ⅳ", "\n", "fi", "'ll", "A", "'S", "漢", ".", "٣٤٥", "٦½", "​<", "EOT", ">", "9", ".", "…\n"]} +{"text": "e!🙂'Reſ<ſ́👍🏽㍿'sm👍🏽 0eAm", "tokens": 35, "pieces": ["e", "!🙂'", "Reſ", "<ſ", "́👍🏽㍿'", "sm", "👍🏽", " ", "0", "eAm"]} +{"text": "\n0'<|endoftext|>'ll\u000b­(Ad́", "tokens": 17, "pieces": ["\n", "0", "'<|", "endoftext", "|>'", "ll", "\u000b", "­(", "Ad", "́"]} +{"text": "ſ\t", "tokens": 3, "pieces": ["ſ", "\t"]} +{"text": "‍½m ́​Dž", "tokens": 9, "pieces": ["‍", "½", "m", " ́​", "Dž"]} +{"text": "ḍ̇ fiEOT", "tokens": 13, "pieces": ["ḋ", "̣", " fi", "EOT"]} +{"text": ".a(İ
d'D…ع're字㍿😀🏽…'ll字😀🏽>#$%\r\n\r\nm<|fim_prefix|><|endoftext|>!'Mfi,-.é>­'S9dİ", "tokens": 62, "pieces": [".a", "(İ", "
d", "'D", "…ع", "'re", "字", "㍿😀🏽", "…", "'ll", "字", "😀🏽>#$%\r\n\r\n", "m", "<|", "fim", "_prefix", "|><|", "endoftext", "|>!'", "Mfi", ",-.", "e", "́>­'", "S", "9", "dİ"]} +{"text": "ß🙂ع('s'Da👍🏽å'T'M ́عfiZİ'M", "tokens": 28, "pieces": ["ß", "🙂ع", "('", "s", "'D", "a", "👍🏽", "a", "̊'", "T", "'M", " ", "́عfiZİ", "'M"]} +{"text": "ع'VE'sZ'\tİ​ 😀🏽", "tokens": 19, "pieces": ["ع", "'VE", "'s", "Z", "'", "\t", "İ", "​", " ", "😀🏽"]} +{"text": "\r३'s'VEfiꟲ>", "tokens": 12, "pieces": ["\r", "३", "'s", "'VE", "fiꟲ", ">"]} +{"text": "👍🏽\tEOT漢'Re . tEOT\u000bß​\n\t!Ⅳ👍🏽\r\nع.a𐞁'VEZ
漢字é'ſ", "tokens": 59, "pieces": ["👍🏽", "\tEOT漢", "'Re", " ", " <", "META", "_START", ">", " .", " tEOT", "\u000bß", "​\n", "\t", "!", "Ⅳ", "👍🏽\r\n", "ع", ".a", "𐞁", "'VE", "Z", "
漢字e", "́'", "ſ"]} +{"text": "d<|fim_prefix|>\r\n\r\n漢'Re\"t عéå-0 字''ſ'D́9
  EOT", "tokens": 34, "pieces": ["d", "<|", "fim", "_prefix", "|>\r\n\r\n", "漢", "'Re", "\"t", " عéa", "̊-", "0", " 字", "''", "ſ", "'D", "́", "9", "
 ", " EOT"]} +{"text": " t !", "tokens": 4, "pieces": [" ", " t", " ", " !"]} +{"text": " 👍🏽 Ⅳ\n", "tokens": 10, "pieces": [" ", " 👍🏽", " ", "Ⅳ", "\n"]} +{"text": "<|fim_prefix|>漢s'Me'T'll<|endoftext|><|endoftext|>e!!9㋿,'reⅣ'VEee‍'T>dfiḍ̇İ'lld३", "tokens": 64, "pieces": ["<|", "fim", "_prefix", "|>", "漢s", "'M", "e", "'T", "'ll", "<|", "endoftext", "|><|", "endoftext", "|>", "e", "!!", "9", "㋿,'", "re", "Ⅳ", "'VE", "ee", "‍'", "T", ">dfi", "ḋ", "̣İ", "'", "lld", "३"]} +{"text": "½a\"\t😀🏽ſé٣٤٥٦漢\u000b,\t9Ⅳ'T\r\n", "tokens": 34, "pieces": ["½", "a", "\"", "\t", "😀🏽", "ſe", "́", "٣٤٥", "٦", "漢", "\u000b", ",", "\t", "9", "", "Ⅳ", "'T", "\r\n"]} +{"text": "t! 😀🏽ſⅣ \n ꟲ  m'll漢㍿'漢!!12345678ſd́\t🙂ꟲ\"'Re<|endoftext|>", "tokens": 54, "pieces": ["t", "!", " 😀🏽", "ſ", "Ⅳ", " \n", " ", " ꟲ", "  ", " m", "'ll", "漢", "㍿'", "漢", "!!", "123", "456", "78", "ſd", "́", "\t", "🙂ꟲ", "\"'", "Re", "<|", "endoftext", "|>"]} +{"text": "e\né'T<😀🏽ꟲ.\u000ba \nA'llfi'll\r\n\r\n'll,\rDžeع<|fim_prefix|>👍🏽ét'(t'Re'ſ#$%Z👍🏽​'reZ", "tokens": 66, "pieces": ["e", "\n", "é", "'T", "<😀🏽", "ꟲ", ".", "\u000ba", " \n", "A", "'ll", "fi", "'ll", "\r\n\r\n", "'ll", ",\r", "Džeع", "<|", "fim", "_prefix", "|>👍🏽", "ét", "'(", "t", "'Re", "'", "ſ", "#$%", "Z", "👍🏽​'", "reZ"]} +{"text": "́!!ⅣmEOT", "tokens": 7, "pieces": ["́!!", "Ⅳ", "mEOT"]} +{"text": "!👍🏽 İ\nß>", "tokens": 12, "pieces": ["!👍🏽", " İ", "\n", "ß", ">"]} +{"text": "ſ\r\n\r\n'll(𐞁're㋿'M", "tokens": 15, "pieces": ["ſ", "\r\n\r\n", "'ll", "(𐞁", "'re", "㋿'", "M"]} +{"text": "-(½'Rea!\r\n('Re", "tokens": 7, "pieces": ["-(", "½", "'Re", "a", "!\r\n", "('", "Re"]} +{"text": "!Zع\r\n­'Da漢ZİDž' #$%\r\n", "tokens": 17, "pieces": ["!Zع", "\r\n", "­'", "Da漢ZİDž", "'", " ", "#$%\r\n"]} +{"text": "ſع \n'ſ#$%!é­s\u000b​ß<字m字𐞁éⅣ'T.<<|endoftext|>'ſ́½9­-(🙂'reſ\n12345678🙂", "tokens": 55, "pieces": ["ſع", " \n", "'ſ", "#$%!", "e", "́­", "s", "\u000b", "​ß", "<字m字𐞁é", "Ⅳ", "'T", ".<<|", "endoftext", "|>'", "ſ", "́", "½9", "­-(🙂'", "reſ", "\n", "123", "456", "78", "🙂"]} +{"text": "åEOT \nA", "tokens": 8, "pieces": ["a", "̊EOT", " \n", "A"]} +{"text": "'sd'T  \n漢<>'llas(ḍ̇漢​㍿३字­<<|endoftext|>,…ß㍿㍿", "tokens": 47, "pieces": ["'s", "d", "'T", "  \n", "漢", "<>'", "llas", "(ḋ", "̣漢", "​㍿", "३", "字", "­<<|", "endoftext", "|>,", "…ß", "㍿㍿<", "META", "_START", ">"]} +{"text": "#$%\t½㍿tßEOT\u000b\r\nZ.'llé३>", "tokens": 20, "pieces": ["#$%", "\t", "½", "㍿tßEOT", "\u000b\r\n", "Z", ".'", "llé", "३", ">"]} +{"text": "!!👍🏽👍🏽eİ
­é'll𐞁'll!!<|fim_prefix|>0'Mé \r\n\r\ń㍿<|fim_prefix|>́'ſ'SⅣ-ZAt", "tokens": 61, "pieces": ["!!👍🏽👍🏽", "eİ", "
", "­é", "'ll", "𐞁", "'ll", "!!<|", "fim", "_prefix", "|>", "0", "'M", "e", "́", " \r\n\r\n", "́㍿<|", "fim", "_prefix", "|>́'", "ſ", "'S", "Ⅳ", "-ZAt"]} +{"text": " Dž𐞁 \n字EOT​ḍ̇é𐞁t'Re\u000bm \r\n\r\n'M12345678(<|endoftext|>fiZ.'VE,ꟲ㍿\r\n\r\n🙂'll", "tokens": 61, "pieces": [" ", " Dž𐞁", " \n", "字EOT", "​ḋ", "̣e", "́𐞁t", "'Re", "\u000bm", "", " \r\n\r\n", "'M", "123", "456", "78", "(<|", "endoftext", "|>", "fiZ", ".'", "VE", ",ꟲ", "㍿\r\n\r\n", "🙂'", "ll"]} +{"text": "12345678
عEOTİ,'VE👍🏽Džfi,'VE'👍🏽Ⅳ", "tokens": 35, "pieces": ["", "123", "456", "78", "
عEOTİ", ",'", "VE", "👍🏽", "Džfi", ",'", "VE", "'👍🏽", "Ⅳ"]} +{"text": "< 0s", "tokens": 4, "pieces": ["<", " ", "0", "s"]} +{"text": "ꟲ", "tokens": 3, "pieces": ["ꟲ"]} +{"text": "é'S漢.d'Z!!\ta\u000bA<Ⅳ­😀🏽 é'T\t㋿\"0'ſDž< ", "tokens": 46, "pieces": ["é", "'S", "漢", ".d", "'Z", "!!", "\ta", "\u000bA", "<", "Ⅳ", "­<", "META", "_START", ">😀🏽", " e", "́'", "T", "", "\t", "㋿\"", "0", "'ſ", "Dž", "<", " "]} +{"text": "'re𐞁åm'D \n12345678漢ع<😀🏽\r\n\r\n٣٤٥٦'Seꟲ\r\n\r\n!!<|endoftext|>\t‍'VE<漢٣٤٥٦­ꟲ e'M", "tokens": 69, "pieces": ["'re", "𐞁a", "̊m", "'D", " \n", "123", "456", "78", "漢ع", "<😀🏽\r\n\r\n", "٣٤٥", "٦", "'S", "eꟲ", "\r\n\r\n", "!!<|", "endoftext", "|>", "\t", "‍'", "VE", "<漢", "٣٤٥", "٦", "­ꟲ", " e", "'M"]} +{"text": ",🙂\r\n\r\nعİ ßḍ̇😀🏽12345678 'ſ'", "tokens": 29, "pieces": [",🙂\r\n\r\n", "عİ", " ßḋ", "̣<", "EOT", ">😀🏽", "123", "456", "78", " '", "ſ", "'"]} +{"text": "s\n ­👍🏽90'Re𐞁A>", "tokens": 18, "pieces": ["s", "\n", " ­👍🏽", "90", "'Re", "𐞁A", ">"]} +{"text": "ꟲ(<|endoftext|>", "tokens": 10, "pieces": ["ꟲ", "(<|", "endoftext", "|>"]} +{"text": "‍eEOT12345678'Dꟲ👍🏽ꟲ漢e İ!'Ⅳ­.e٣٤٥٦字\u000b🙂0'Sع'ſ e'Dꟲte", "tokens": 57, "pieces": ["‍eEOT", "123", "456", "78", "'D", "ꟲ", "👍🏽", "ꟲ漢e", " İ", "!'", "Ⅳ", "­.", "e", "٣٤٥", "٦", "字", "\u000b", "🙂", "0", "'S", "ع", "'ſ", " ", " e", "'D", "ꟲte"]} +{"text": "A<|endoftext|>\rEOTé!!é>ḍ̇åé
\nmfi \r\n", "tokens": 33, "pieces": ["A", "<|", "endoftext", "|>\r", "EOTé", "!!", "e", "́>", "ḋ", "̣a", "̊é", "
\n", "mfi", " \r\n"]} +{"text": "'sḍ̇<|fim_prefix|>'٣٤٥٦'re
…ꟲe ٣٤٥٦'VE's\u000b '", "٣٤٥", "٦", "'re", "
", "…ꟲe", " ", "٣٤٥", "٦", "'VE", "'s", "\u000b", " <", "Zİ", "½", "\t", "(㋿", "A", "!ß", "\"", "३", "EOT", "!!🙂\r\n", "'re", "İ", " \n", " ", " !'", "T"]} +{"text": "!! 😀🏽- ​s​A\r\nع‍ß\t字'Re' ſ<'Re\"👍🏽\t>\n<|endoftext|>😀🏽ß\u000b", "tokens": 50, "pieces": ["!!", " ", "😀🏽-", " ​", "s", "​A", "\r\n", "ع", "‍ß", "\t字", "'Re", "'", " ſ", "<'", "Re", "\"👍🏽", "\t", ">\n", "<|", "endoftext", "|>😀🏽", "ß", "\u000b"]} +{"text": "🙂-<|endoftext|>‍𐞁(", "tokens": 17, "pieces": ["🙂-<|", "endoftext", "|>‍", "𐞁", "("]} +{"text": "0\n字\ré!👍🏽'ſ'llEOT0<|endoftext|>.😀🏽\rꟲⅣⅣⅣ㍿'T9🙂٣٤٥٦'VEꟲé­​Ⅳ
", "tokens": 70, "pieces": ["0", "\n", "字", "\r", "e", "́!👍🏽'", "ſ", "'ll", "EOT", "0", "<|", "endoftext", "|>.😀🏽\r", "ꟲ", "ⅣⅣⅣ", "㍿'", "T", "9", "🙂", "٣٤٥", "٦", "'VE", "ꟲé", "­​", "Ⅳ", "
"]} +{"text": "A…fi٣٤٥٦ \r\n\r\n🙂!'s…'Red\r\nZ字#$%🙂12345678ꟲm!!漢'T½𐞁#$%", "tokens": 53, "pieces": ["A", "…fi", "٣٤٥", "٦", " \r\n\r\n", "🙂!'", "s", "…", "'Re", "d", "\r\n", "Z字", "#$%🙂", "123", "456", "78", "ꟲm", "!!", "漢", "'T", "½", "𐞁", "#$%<", "EOT", ">"]} +{"text": "0!a.字9'M'llḍ̇𐞁9漢'ſ>fi\t'T", "tokens": 32, "pieces": ["0", "!a", ".字", "9", "'M", "'", "llḋ", "̣𐞁", "9", "漢", "'ſ", ">fi", "\t", "'T"]} +{"text": "👍🏽Ⅳ\"!İ0,aİ'D字Ⅳ<ꟲ\t½\u000b 😀🏽", "tokens": 38, "pieces": ["👍🏽", "Ⅳ", "\"!", "İ", "0", ",aİ", "'D", "字", "Ⅳ", "<ꟲ", "\t", "½", "<", "EOT", ">", "\u000b", " ", "😀🏽"]} +{"text": "s\"mA12345678å ́🙂'Tع👍🏽'M…(漢ꟲ- 'M\n9", "tokens": 43, "pieces": ["s", "\"mA", "123", "456", "78", "a", "̊", " ́🙂'", "T", "ع", "👍🏽'", "M", "…", "(漢ꟲ", "-", " ", "'M", "\n", "9"]} +{"text": ">ꟲ٣٤٥٦\n漢.ß漢Ⅳ0عå 'ſ'ReEOT'T
​-(12345678, 's🙂stꟲ", "tokens": 51, "pieces": [">ꟲ", "٣٤٥", "٦", "\n", "漢", ".ß漢", "Ⅳ0", "عa", "̊", " ", " '", "ſ", "'Re", "EOT", "'T", "
", "​-(", "123", "456", "78", ",", " ", " '", "s", "🙂stꟲ"]} +{"text": "ß>12345678🙂<|fim_prefix|>\"<|endoftext|>ع <|fim_prefix|>>0'SEOT  <|endoftext|><|fim_prefix|>漢३,é字é‍\r\n're漢e(d(#$%ßå\u000b", "tokens": 76, "pieces": ["ß", ">", "123", "456", "78", "🙂<|", "fim", "_prefix", "|>\"<|", "endoftext", "|>", "ع", " ", " <|", "fim", "_prefix", "|>>", "0", "'S", "EOT", " ", " ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "漢", "३", ",e", "́字e", "́‍\r\n", "'re", "漢e", "(d", "(#$%", "ßa", "̊", "\u000b", ""]} +{"text": "fi İDž\r\n
0.'T'VE'Dḍ̇\r\n\r\nt a­𐞁#$%🙂.'", "T", "'VE", "'D", "ḋ", "̣\r\n\r\n", "t", " a", "­𐞁", "#$%🙂<", "dé", "(𐞁", "½", "s", "\r\n\r\n", "s"]} +{"text": "!!<'ſå \r\n𐞁fiZ 12345678'reİ( \nḍ̇३#$%'S'ſ\r,m­㍿.\n\r\n​\n'VE \n'ſ", "tokens": 52, "pieces": ["!!<'", "ſa", "̊", " \r\n", "𐞁fiZ", " ", "123", "456", "78", "'re", "İ", "(", " \n", "ḋ", "̣", "३", "#$%'", "S", "'ſ", "\r", ",m", "­㍿.\n\r\n", "​\n", "'VE", " \n", "'ſ"]} +{"text": "å\r'VE\u000b\u000b0>m'sa字a0٣٤٥٦.'S0\t", "tokens": 27, "pieces": ["a", "̊\r", "'VE", "\u000b", "\u000b", "0", ">m", "'s", "a字a", "0٣٤", "٥٦", ".'", "S", "0", "\t"]} +{"text": " Z!!'D<|fim_prefix|>­ å('VÉ", "tokens": 23, "pieces": [" Z", "!!'", "D", "<|", "fim", "_prefix", "|>­", " a", "̊('", "VE", "́"]} +{"text": "… 'T'VE👍🏽!‍", "tokens": 16, "pieces": ["… ", " '", "T", "'VE", "👍🏽!‍"]} +{"text": "\r\n\r\n(𐞁ꟲ \u000b'VE㋿👍🏽", "tokens": 22, "pieces": ["\r\n\r\n", "(𐞁ꟲ", " ", "\u000b", "'VE", "㋿👍🏽"]} +{"text": "​s'llé", "tokens": 9, "pieces": ["​s", "'ll", "e", "́<", "META", "_START", ">"]} +{"text": " \u000b 'VE‍m!!!éå. 're'M", "tokens": 20, "pieces": [" \u000b", " '", "VE", "‍<", "EOT", ">m", "!!!", "e", "́a", "̊.", " '", "re", "'M"]} +{"text": "३Ⅳ0\n\r\n\r\n,å", "tokens": 14, "pieces": ["३Ⅳ0", "\n\r\n\r\n", ",a", "̊<", "EOT", ">"]} +{"text": "ms's\r\nd\r\n\r\n12345678'S''Té(A字'VE\tet(
 a
é12345678'D𐞁\r\r\né", "tokens": 40, "pieces": ["ms", "'s", "\r\n", "d", "\r\n\r\n", "123", "456", "78", "'S", "''", "Té", "(A字", "'VE", "\tet", "(", "
", " a", "
e", "́", "123", "456", "78", "'D", "𐞁", "\r\r\n", "e", "́"]} +{"text": " fi👍🏽 'll
< '
'ſ'S \ne👍🏽𐞁‍'sꟲḍ̇", "tokens": 49, "pieces": ["", " ", " fi", "👍🏽", " ", "'ll", "
", "<", " ", "'", "
", "'ſ", "'S", " \n", "e", "👍🏽", "𐞁", "‍'", "sꟲḋ", "̣"]} +{"text": "'re-㍿\nå-㍿\r\"𐞁!!½🙂
Dž𐞁#$%ſꟲ,\n!!😀🏽'll'Re𐞁\r\n…e👍🏽", "tokens": 62, "pieces": ["'re", "-㍿\n", "a", "̊-㍿\r", "\"𐞁", "!!", "½", "🙂", "
Dž𐞁", "#$%", "ſꟲ", ",\n", "!!😀🏽'", "ll", "'Re", "𐞁", "\r\n", "…e", "👍🏽"]} +{"text": "½A><|endoftext|>½'M>Ⅳع\t>字é🙂 \n! d\u000b ,​t㋿漢>\r\n", "tokens": 41, "pieces": ["½", "A", "><|", "endoftext", "|>", "½", "'M", ">", "Ⅳ", "ع", "\t", ">字e", "́🙂", " \n", "!", " d", "\u000b ", " ,​", "t", "㋿漢", ">\r\n"]} +{"text": "'٣٤٥٦ßꟲ漢 'Re​ \n … ​'VE\"0İ12345678 …0'T𐞁", "tokens": 40, "pieces": ["'", "٣٤٥", "٦", "ßꟲ漢", " '", "Re", "​", " \n", " …", " ​'", "VE", "\"", "0", "İ", "123", "456", "78", " ", "…", "0", "'T", "𐞁"]} +{"text": "e漢'S ", "tokens": 5, "pieces": ["e漢", "'S", " "]} +{"text": "ḍ̇ß 
\n", "tokens": 14, "pieces": ["ḋ", "̣ß", " 
\n"]} +{"text": "ſA", "tokens": 4, "pieces": ["ſA"]} +{"text": "\n­Dž٣٤٥٦,.
'D'ſ s'EOT'字e😀🏽0aEOT'M㍿'S\"㋿'VEع", "tokens": 50, "pieces": ["\n", "­Dž", "٣٤٥", "٦", ",.", "
", "'D", "'ſ", " s", "'EOT", "'字e", "😀🏽", "0", "aEOT", "'M", "㍿'", "S", "\"㋿'", "VE", "ع"]} +{"text": "字😀🏽9ḍ̇EOTZ'll\t३३'T
s­İ­12345678'VE漢(\r\n\r\n漢İ're㋿ع'VEİ漢👍🏽Aa", "tokens": 67, "pieces": ["字", "😀🏽", "9", "ḋ", "̣EOTZ", "'ll", "\t", "३३", "'T", "
s", "­İ", "­", "123", "456", "78", "'VE", "漢", "(\r\n\r\n", "漢İ", "'", "re", "㋿ع", "'VE", "İ漢", "👍🏽<", "META", "_START", ">Aa"]} +{"text": "­\"…‍.
d#$%'ſ'S'll'VE.ſ12345678𐞁's<|fim_prefix|> 'ſ12345678d<|endoftext|>\"!!​ å𐞁DžⅣ\r", "tokens": 73, "pieces": ["­\"", "…", "‍.", "
d", "#$%'", "ſ", "'S", "'ll", "'VE", ".ſ", "123", "456", "78", "𐞁", "'s", "<|", "fim", "_prefix", "|>", " ", " '", "ſ", "123", "456", "78", "d", "<|", "endoftext", "|>\"<", "EOT", "><", "META", "_START", ">!!​", " ", " a", "̊𐞁Dž", "Ⅳ", "\r"]} +{"text": "å'SéꟲDžⅣ३㍿m\r\n😀🏽Ⅳ'Reḍ̇㍿ ٣٤٥٦", "tokens": 48, "pieces": ["a", "̊'", "S", "éꟲDž", "Ⅳ३", "㍿m", "\r\n", "😀🏽", "Ⅳ", "'Re", "ḋ", "̣㍿", " ", "٣٤٥", "٦"]} +{"text": "\r\n>ée'S<|endoftext|>ḍ̇A字<|fim_prefix|>fi", "tokens": 29, "pieces": ["\r\n", ">e", "́e", "'S", "<|", "endoftext", "|>", "ḋ", "̣A字", "<|", "fim", "_prefix", "|>", "fi"]} +{"text": "\n'VE…\"ſß's'ſ<|fim_prefix|>\r", "tokens": 21, "pieces": ["\n", "'VE", "…", "\"ſß", "'s", "'ſ", "<|", "fim", "_prefix", "|>\r"]} +{"text": "'Dm​'VEåع12345678!! 👍🏽\n \né", "tokens": 23, "pieces": ["'D", "m", "​'", "VEa", "̊ع", "123", "456", "78", "!!", " ", "👍🏽\n", " \n", "é"]} +{"text": "字-0'VE,­ع#$%'D'T\r\n-fi", "tokens": 20, "pieces": ["字", "-", "0", "'VE", ",­", "ع", "#$%'", "D", "'T", "\r\n", "-fi"]} +{"text": "'ll>🙂漢eḍ̇'M\n\r\n‍ 👍🏽", "tokens": 25, "pieces": ["'ll", ">🙂", "漢eḋ", "̣'", "M", "\n\r\n", "‍", " ", "👍🏽"]} +{"text": "12345678éſ漢!\n9­'ſ>ḍ̇'́ ", "tokens": 24, "pieces": ["123", "456", "78", "e", "́ſ漢", "!\n", "9", "­'", "ſ", ">ḋ", "̣'́", " "]} +{"text": "\t,漢>' dDž>\rDžZꟲ­'ll \n Ⅳ d's0'Re'D​", "tokens": 31, "pieces": ["\t", ",漢", ">'", " dDž", ">\r", "DžZꟲ", "­'", "ll", " \n", " ", "Ⅳ", " ", " d", "'s", "0", "'Re", "'D", "​"]} +{"text": "\r\n\r\n-'S !!-\u000b‍", "tokens": 9, "pieces": ["\r\n\r\n", "-'", "S", " ", " !!-", "\u000b", "‍"]} +{"text": "é​ꟲ'll‍'re…\rع😀🏽tꟲ\t\"a🙂m,\r\n\r\n
३½ \nt,'ſDžع👍🏽𐞁EOTfiß#$%Dž'll'M ", "tokens": 64, "pieces": ["é", "​ꟲ", "'ll", "‍'", "re", "…\r", "ع", "😀🏽", "tꟲ", "\t", "\"a", "🙂m", ",\r\n\r\n", "
", "३½", " \n", "t", ",'", "ſDžع", "👍🏽", "𐞁EOTfiß", "#$%", "Dž", "'ll", "'M", " "]} +{"text": "́㍿'llt'll\"'S 字", "tokens": 12, "pieces": ["́㍿'", "llt", "'ll", "\"'", "S", " 字"]} +{"text": "'M\t é>'D㋿İ", "tokens": 10, "pieces": ["'M", "\t", " é", ">'", "D", "㋿İ"]} +{"text": "\"é>t'D<|endoftext|>", "tokens": 11, "pieces": ["\"é", ">t", "'D", "<|", "endoftext", "|>"]} +{"text": "A😀🏽'ſ#$%'re'll#$%​ ㍿  ", "tokens": 26, "pieces": ["A", "😀🏽'", "ſ", "#$%'", "re", "'", "ll", "#$%​", " ㍿", "  "]} +{"text": "ꟲꟲfi< m漢9EOTs𐞁
'll㋿𐞁.a.Ⅳ… \n<|fim_prefix|>٣٤٥٦🙂½é'Reſ", "tokens": 67, "pieces": ["ꟲꟲfi", "<<", "EOT", ">", " m漢", "9", "EOTs𐞁", "
", "'ll", "㋿𐞁", ".a", ".", "Ⅳ", "… \n", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "🙂", "½", "e", "́<", "META", "_START", ">'", "Reſ"]} +{"text": "'S'D'se'ſ9Ⅳſ🙂㍿Z\r\n­\u000b", "tokens": 21, "pieces": ["'S", "'D", "'s", "e", "'ſ", "9Ⅳ", "ſ", "🙂㍿", "Z", "\r\n", "­", "\u000b"]} +{"text": "!\n,(a…aEOT𐞁'VE0'", "VE", "0", "#$%\"", "tokens": 17, "pieces": ["'ll", "ſ", "\r", "123", "456", "78", "d", "<|", "endoftext", "|>#$%\""]} +{"text": "'", "tokens": 1, "pieces": ["'"]} +{"text": "㋿A漢s'Tع
'Mꟲḍ̇é(- åaſ​Džt\tA'S!漢 -漢 ‍<|endoftext|>😀🏽'D", "tokens": 63, "pieces": ["㋿A漢s", "'T", "ع", "
", "'M", "ꟲḋ", "̣e", "́(-", " a", "̊aſ", "​Džt", "\tA", "'S", "!漢", " ", " -", "漢", " ", " ‍<|", "endoftext", "|>😀🏽'", "D"]} +{"text": "'ll'VE३ß 0字­ 'ReİeEOTⅣ\nZ​", "tokens": 22, "pieces": ["'ll", "'VE", "३", "ß", " ", "0", "字", "­", " ", " '", "ReİeEOT", "Ⅳ", "\n", "Z", "​"]} +{"text": "İAſ #$%㋿'M", "tokens": 13, "pieces": ["İAſ", " ", "#$%㋿'", "M"]} +{"text": "\"", "tokens": 1, "pieces": ["\""]} +{"text": "m#$%>'Mm.ZZ👍🏽㋿'T
㍿e\"㋿٣٤٥٦9\tḍ̇0½", "tokens": 50, "pieces": ["m", "#$%>'", "Mm", ".ZZ", "👍🏽<", "META", "_START", ">㋿'", "T", "
", "㍿e", "\"㋿", "٣٤٥", "٦9", "\tḋ", "̣", "0½"]} +{"text": " \n\t(½EOTßt漢\t㋿EOTß\r\n\r\n㍿'𐞁\u000ba½🙂", "tokens": 31, "pieces": [" \n", "\t", "(", "½", "EOTßt漢", "\t", "㋿EOTß", "\r\n\r\n", "㍿'", "𐞁", "\u000ba", "½", "🙂"]} +{"text": "9漢'SDž𐞁㍿ع'ſ३Ⅳ(é½👍🏽\t\r\n,", "tokens": 32, "pieces": ["9", "漢", "'S", "Dž𐞁", "㍿ع", "'ſ", "३Ⅳ", "(e", "́", "½", "👍🏽", "\t\r\n", ","]} +{"text": "'ſ😀🏽#$%\r\n­<|fim_prefix|><> EOT>​éⅣ'D\r'D\tt­'s 'Re>\"‍EOT#$%\r\nḍ̇fi d\r\n\r\n😀🏽", "tokens": 64, "pieces": ["'ſ", "😀🏽#$%\r\n", "­<|", "fim", "_prefix", "|><>", " ", " EOT", ">​", "e", "́", "Ⅳ", "'", "D", "\r", "'D", "\tt", "­'", "s", " ", " '", "Re", ">\"‍", "EOT", "#$%\r\n", "ḋ", "̣fi", " ", " d", "\r\n\r\n", "😀🏽"]} +{"text": "'Ree'sé(#$%Dž🙂́'ſ‍", "tokens": 17, "pieces": ["'Re", "e", "'s", "é", "(#$%", "Dž", "🙂́'", "ſ", "‍"]} +{"text": "ع𐞁eA\r\n\r\nfi‍\" 'ſ'VE'sſ'Re.Dž𐞁å'll<|endoftext|>,😀🏽㋿👍🏽ḍ̇<|endoftext|>t .㍿<<🙂
", "tokens": 85, "pieces": ["ع𐞁eA", "\r\n\r\n", "fi", "‍\"", " ", "'ſ", "'", "VE", "'s", "ſ", "'Re", ".Dž𐞁a", "̊'", "ll", "<|", "endoftext", "|>,😀🏽㋿👍🏽", "ḋ", "̣<|", "endoftext", "|>", "t", " ", " <", "META", "_START", ">.㍿<<🙂", "
"]} +{"text": "'ll>'s
\r,𐞁<|fim_prefix|>Z afi9字're字 ḍ̇<|endoftext|>字\r\n\r\n'sA0 d\n#$%12345678‍s'VEm字…", "tokens": 66, "pieces": ["'ll", ">'", "s", "", "
\r", ",𐞁", "<|", "fim", "_prefix", "|>", "Z", " ", " afi", "9", "字", "'re", "字", " ḋ", "̣<|", "endoftext", "|>", "字", "\r\n\r\n", "'s", "A", "0", " ", " d", "\n", "#$%", "123", "456", "78", "‍s", "'VE", "m字", "…"]} +{"text": "A\n-漢Dž👍🏽٣٤٥٦'ſ'D३ 漢 é", "tokens": 34, "pieces": ["A", "\n", "-漢Dž", "👍🏽", "٣٤٥", "٦", "'ſ", "'D", "३", " 漢", " e", "́"]} +{"text": "'S#$%'ß(😀🏽,é½EOTꟲ'VE\r'SⅣ12345678<(ßt'D‍ 'M'Ret'VE", "tokens": 40, "pieces": ["'S", "#$%'", "ß", "(😀🏽,", "e", "́", "½", "EOTꟲ", "'VE", "\r", "'S", "Ⅳ12", "345", "678", "<(", "ßt", "'D", "‍", " ", "'M", "'Re", "t", "'VE"]} +{"text": "'re字'M(\u000b're𐞁'Ś9Dž\r\n\r\nDž \n('reꟲḍ̇
\rs \n", "tokens": 37, "pieces": ["'re", "字", "'M", "(", "\u000b", "'re", "𐞁", "'S", "́", "9", "Dž", "\r\n\r\n", "Dž", " \n", "('", "reꟲḋ", "̣", "
\r", "s", " \n"]} +{"text": "​!!İA'M!!'sfimⅣ ꟲ👍🏽t👍🏽…é0é's>å 'ſ<|endoftext|>٣٤٥٦éA Dž(  Dž㍿́ḍ̇😀🏽", "tokens": 85, "pieces": ["​!!", "İA", "'M", "!!'", "sfim", "Ⅳ", " ꟲ", "👍🏽", "t", "👍🏽", "…é", "0", "é", "'s", ">a", "̊", " ", "'ſ", "<|", "endoftext", "|>", "٣٤٥", "٦", "e", "́A", " Dž", "(", " ", " Dž", "㍿́", "ḋ", "̣😀🏽"]} +{"text": "🙂\t😀🏽dA-😀🏽éⅣ́\r\n\r\n👍🏽 ſ́ \n- m\" 'Sd", "tokens": 41, "pieces": ["🙂", "\t", "😀🏽", "dA", "-😀🏽", "e", "́", "Ⅳ", "́\r\n\r\n", "👍🏽", " ſ", "́", " \n", "-", " ", " m", "\"", " ", "'S", "d"]} +{"text": "😀🏽!!éꟲEOT́ꟲ😀🏽'T𐞁½…👍🏽'EOT.ßa𐞁a", "tokens": 47, "pieces": ["😀🏽!!", "éꟲEOT", "́ꟲ", "😀🏽'", "T𐞁", "½", "…", "👍🏽'", "EOT", ".ßa𐞁a"]} +{"text": "!!t.99٣٤٥٦Ⅳع,漢字\u000bİ9\u000b", "tokens": 23, "pieces": ["!!", "t", ".", "99٣", "٤٥٦", "Ⅳ", "ع", ",漢字", "\u000bİ", "9", "\u000b"]} +{"text": ".\"A\r>\u000bte<|endoftext|>#$%<|fim_prefix|>,eeå'Re👍🏽 s'Re 'Reſ\n>12345678#$%漢… \n<|fim_prefix|>\t", "tokens": 61, "pieces": [".\"", "A", "\r", ">", "\u000bte", "<|", "endoftext", "|>#$%<|", "fim", "_prefix", "|>,", "eea", "̊'", "Re", "👍🏽", " s", "'Re", " ", " '", "Reſ", "\n", ">", "123", "456", "78", "#$%", "漢", "… \n", "<|", "fim", "_prefix", "|>", "\t"]} +{"text": "'VE're​​\u000b\n\n'> 'll\u000b𐞁 ß' t㍿m٣٤٥٦‍t0­ İ👍🏽Dž‍t\r", "tokens": 53, "pieces": ["'VE", "'re", "​​", "\u000b\n\n", "'>", " '", "ll", "\u000b𐞁", " ß", "'", " ", " t", "㍿", "m", "٣٤٥", "٦", "‍t", "0", "­", " ", " İ", "👍🏽", "Dž", "‍t", "\r"]} +{"text": " ㍿㍿\nſ‍!'0👍🏽", "tokens": 20, "pieces": [" ", "㍿㍿\n", "ſ", "‍!'", "0", "👍🏽"]} +{"text": "ḍ̇ḍ̇'M字sḍ̇\t#$%t,'VEéß३0'Sé'll‍👍🏽,s", "tokens": 46, "pieces": ["ḋ", "̣ḋ", "̣'", "M字sḋ", "̣", "\t", "#$%", "t", ",'", "VEe", "́ß", "३0", "'S", "e", "́'", "ll", "‍👍🏽,", "s"]} +{"text": "㍿sfi३Z
…😀🏽!!\"漢12345678'sfiZd字Dž𐞁Zå0.d9aß\"<|fim_prefix|>'12345678㋿字Z's're", "tokens": 66, "pieces": ["㍿sfi", "३", "Z", "", "
", "…", "😀🏽!!\"", "漢", "123", "456", "78", "'s", "fiZd字Dž𐞁Za", "̊", "0", ".d", "9", "aß", "\"<|", "fim", "_prefix", "|>'", "123", "456", "78", "㋿字Z", "'s", "'re"]} +{"text": " \"'VE're 9ḍ̇éeⅣ'VE!!㍿\t.", "tokens": 24, "pieces": [" ", " \"'", "VE", "'re", " ", "9", "ḋ", "̣e", "́e", "Ⅳ", "'VE", "!!㍿", "\t", "."]} +{"text": "½å㋿🙂𐞁\r\n12345678's…", "tokens": 20, "pieces": ["½", "a", "̊㋿🙂", "𐞁", "\r\n", "123", "456", "78", "'s", "…"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "
ع…
séİ\r\n\r\n,dſéZ \n 'DEOT'Re", "tokens": 21, "pieces": ["
ع", "…", "
se", "́İ", "\r\n\r\n", ",dſéZ", " \n", " '", "DEOT", "'Re"]} +{"text": "9dZ\rZ'll \t!!(\t​\u000bfi0 sİe­#$%'ſm\u000bm", "tokens": 40, "pieces": ["9", "dZ", "\r", "Z", "A", "'", "ll", " ", "\t", "!!(", "\t", "​", "\u000bfi", "0", " sİe", "­#$%'", "ſm", "\u000bm"]} +{"text": "<0‍ZEOTm'D
'll12345678.Z'llZé<ſfiſ0Ⅳ>", "tokens": 29, "pieces": ["<", "0", "‍ZEOTm", "'D", "
", "'ll", "123", "456", "78", ".Z", "'ll", "Zé", "<ſfiſ", "0Ⅳ", ">"]} +{"text": " å👍🏽12345678٣٤٥٦㍿ ३𐞁s \t\"३'lle'T'‍''VE 漢Ⅳ<|endoftext|> \n.👍🏽漢fí٣٤٥٦ſ12345678'VE", "tokens": 88, "pieces": [" ", " a", "̊👍🏽", "123", "456", "78٣", "٤٥٦", "㍿", " ", "३", "𐞁s", " ", "\t", "\"", "३", "'ll", "e", "'T", "'‍''", "VE", " 漢", "Ⅳ", "<|", "endoftext", "|>", " \n", ".👍🏽", "漢fi", "́", "٣٤٥", "٦", "ſ", "123", "456", "78", "'", "VE"]} +{"text": "Ⅳ \nḍ̇t<|endoftext|>At !,㍿!!٣٤٥٦🙂ⅣEOT#$%A\r\n\r\n", "At", " ", "!,㍿!!", "٣٤٥", "٦", "🙂", "Ⅳ", "EOT", "#$%", "A", "\r\n\r\n", ",½\"\té'reİ\t(漢#$%\r\n\r\nå'!!٣٤٥٦\r'M
Zé٣٤٥٦́İ ́­'३😀🏽å'<|fim_prefix|>", "tokens": 75, "pieces": ["<|", "fim", "_prefix", "|>,", "½", "\"", "\te", "́'", "reİ", "\t", "(漢", "#$%\r\n\r\n", "a", "̊'!!", "٣٤٥", "٦", "\r", "'M", "
Ze", "́", "٣٤٥", "٦", "́", "İ", " ", " ́­'", "३", "😀🏽", "a", "̊'<|", "fim", "_prefix", "|>"]} +{"text": "\r\n…\u000b𐞁é-mſ…\r\n\r\nß<|endoftext|>d0 (字३½s字́'T\r\t'ſ'sⅣ́\tDž\r\n", "tokens": 48, "pieces": ["\r\n", "…", "\u000b𐞁é", "-mſ", "…\r\n\r\n", "ß", "<|", "endoftext", "|>", "d", "0", " (", "字", "३½", "s字", "́'", "T", "\r", "\t", "'ſ", "'s", "Ⅳ", "́", "\tDž", "\r\n"]} +{"text": "#$%!EOTdé'S\u000bß \n३<|endoftext|>ß­s‍'ll0‍", "tokens": 37, "pieces": ["#$%!", "EOTde", "́'", "S", "\u000b", "ß", " \n", "३", "<|", "endoftext", "|>", "ß", "­", "s", "‍'", "ll", "0", "‍"]} +{"text": "İꟲ12345678", "tokens": 7, "pieces": ["İꟲ", "123", "456", "78"]} +{"text": "Dž(!!Zİ<­ 'Ree,
", "tokens": 15, "pieces": ["Dž", "(!!", "Zİ", "<­", " ", " '", "Ree", ",", "
"]} +{"text": "字 🙂s…Dž\r\n\r\n'ſ\tsİ𐞁\r'T\rm'VE-a9,e0!漢eİ'VE's s'Re字#$%‍<|endoftext|>", "tokens": 50, "pieces": ["(s", "!!", "ع", "​d", "'", "ſ", "\tsİ𐞁", "\r", "'T", "\r", "m", "'VE", "-a", "9", ",e", "0", "!漢eİ", "'VE", "'s", " s", "'Re", "字", "#$%‍<|", "endoftext", "|>"]} +{"text": "٣٤٥٦٣٤٥٦EOTé\u000b<|fim_prefix|>𐞁!9漢(字é…😀🏽eİ\"e \t!!A'reḍ̇'saZ\r\ntḍ̇­㍿\r\n\r\n𐞁", "tokens": 80, "pieces": ["٣٤٥", "٦٣٤", "٥٦", "EOTé", "\u000b", "<|", "fim", "_prefix", "|>", "𐞁", "!", "9", "漢", "(字e", "́", "…", "😀🏽", "eİ", "\"e", " ", "\t", "!!", "A", "'re", "ḋ", "̣'", "saZ", "\r\n", "tḋ", "̣­㍿\r\n\r\n", "𐞁"]} +{"text": "éع(A>३#$%!
\n\t'llع\r\n\r\nⅣd'VEs(\n'Re", "tokens": 29, "pieces": ["éع", "(A", ">", "३", "#$%!", "
\n", "\t", "'ll", "ع", "\r\n\r\n", "Ⅳ", "d", "'VE", "s", "(\n", "'Re"]} +{"text": "عAZ٣٤٥٦#$%", "tokens": 18, "pieces": ["عAZ", "٣٤٥", "٦", "#$%"]} +{"text": "\u000bd😀🏽DžA३٣٤٥٦\u000bs#$%\t㍿'S,عé‍m\u000b-s!!­Z'#$%", "tokens": 46, "pieces": ["\u000bd", "😀🏽", "DžA", "३٣٤", "٥٦", "\u000bs", "#$%", "\t", "㍿'", "S", ",عe", "́‍", "m", "\u000b", "-s", "!!­", "Z", "'#$%"]} +{"text": "eé
<|fim_prefix|>'re'D'", "re", "'D", "'T㋿! \n…!\"'s٣٤٥٦٣٤٥٦'re ", "tokens": 40, "pieces": ["Ⅳ", "'VE", "字", "'VE", "'", "T", "㋿!", " \n", "…", "!\"'", "s", "٣٤٥", "٦٣٤", "٥٦", "'re", " "]} +{"text": "-🙂d𐞁🙂 <\r\n㍿t­#$%'T å<|endoftext|>🙂'lls", "tokens": 39, "pieces": ["𐞁", "🙂", " ", "<\r\n", "㍿t", "­#$%'", "T", " a", "̊<|", "endoftext", "|>🙂'", "lls"]} +{"text": "'ſ !㍿,\r\n\r\n<|fim_prefix|>ßefi,㍿'Dİ🙂\t\r\n<|endoftext|>.ß9🙂😀🏽", "tokens": 44, "pieces": ["'ſ", " !㍿,\r\n\r\n", "<|", "fim", "_prefix", "|>", "ßefi", ",㍿'", "Dİ", "🙂", "\t\r\n", "<|", "endoftext", "|>.", "ß", "9", "🙂😀🏽"]} +{"text": "漢漢eعés\rDž'VE Z", "tokens": 14, "pieces": ["漢漢eعés", "\r", "Dž", "'VE", " Z"]} +{"text": "!fit#$%‍Džfi İ🙂🙂 ㍿‍́'ſ<|fim_prefix|>", "tokens": 35, "pieces": ["!fit", "#$%‍", "Džfi", " İ", "🙂🙂", " ", "㍿‍́'", "ſ", "<|", "fim", "_prefix", "|>"]} +{"text": "Z👍🏽s#$%<|fim_prefix|>ḍ̇", "tokens": 22, "pieces": ["Z", "👍🏽", "s", "#$%<|", "fim", "_prefix", "|>", "ḋ", "̣"]} +{"text": "\n'VEZé'll,…😀🏽d<|endoftext|>㋿ß!<|endoftext|>½é\r'lléAt👍🏽.'VE'!!Ⅳ", "tokens": 55, "pieces": ["\n", "'VE", "Ze", "́'", "ll", ",", "…", "😀🏽", "d", "<|", "endoftext", "|>㋿", "ß", "!<|", "endoftext", "|>", "½", "e", "́\r", "'ll", "éAt", "👍🏽.'", "VE", "'!!", "Ⅳ"]} +{"text": "ſ½​­😀🏽½\n'VE('عⅣ…'reé'reee12345678😀🏽- <'ꟲ'ReDž12345678<  'VEt!! é\r漢's", "tokens": 60, "pieces": ["ſ", "½", "​­😀🏽", "½", "\n", "'VE", "('", "ع", "Ⅳ", "…", "'re", "e", "́'", "reee", "123", "456", "78", "😀🏽-", " <'", "ꟲ", "'Re", "Dž", "123", "456", "78", "<", " ", " ", "'VE", "t", "!!", " e", "́\r", "漢", "'s"]} +{"text": "A­ꟲ<'ll
", "tokens": 10, "pieces": ["A", "­ꟲ", "<'", "ll", "
"]} +{"text": "Dž #$%㋿字㍿٣٤٥٦d\t… \n🙂's,-", "tokens": 30, "pieces": ["Dž", " ", "#$%㋿", "字", "㍿", "٣٤٥", "٦", "d", "\t… \n", "🙂'", "s", ",-"]} +{"text": "🙂'VE\r­'ſⅣꟲfié<|fim_prefix|>,(", "tokens": 28, "pieces": ["🙂'", "VE", "\r", "­<", "EOT", ">'", "ſ", "Ⅳ", "ꟲfié", "<|", "fim", "_prefix", "|>,("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "12345678Z<|endoftext|>\".\t> ½\u000b'M 12345678#$%㋿\u000bå漢at\u000bd<|endoftext|>!t👍🏽 > …s\tſ", "tokens": 62, "pieces": ["123", "456", "78", "Z", "<|", "endoftext", "|><", "EOT", ">\".", "\t", ">", " ", "½", "\u000b", "'M", " ", "123", "456", "78", "#$%㋿", "\u000ba", "̊漢at", "\u000bd", "<|", "endoftext", "|>!", "t", "👍🏽", " ", " >", " ", "…s", "\tſ"]} +{"text": "٣٤٥٦ Z漢́ 'VE‍'VEعع🙂३té\t\r\n\r\n>🙂'Sé 'S\r½ ", "tokens": 46, "pieces": ["٣٤٥", "٦", "", " Z漢", "́", " ", "'VE", "‍'", "VEعع", "🙂", "३", "te", "́", "\t\r\n\r\n", "><", "EOT", ">🙂'", "Se", "́", " '", "S", "\r", "½", " "]} +{"text": " <|endoftext|><|endoftext|>..t<|fim_prefix|>'T#$%Dž<'D👍🏽ås'll'fi​'llꟲ'Sd", "tokens": 52, "pieces": [" ", "<|", "endoftext", "|><|", "endoftext", "|>..", "t", "<|", "fim", "_prefix", "|>'", "T", "#$%", "Dž", "<'", "D", "👍🏽", "a", "̊s", "'ll", "'fi", "​'", "llꟲ", "'S", "d"]} +{"text": "(.Ⅳ
Dž9", "tokens": 8, "pieces": ["(.", "Ⅳ", "
Dž", "9"]} +{"text": "… -,-Atfi.عſ'VE'VE㋿㋿'\r> \n'字 ㋿<|fim_prefix|>'ll
'S㋿ 漢9e漢Z", "tokens": 57, "pieces": ["… ", " -,-", "Atfi", ".عſ", "'VE", "'VE", "㋿㋿'\r", ">", " \n", "'字", " ", " ㋿<|", "fim", "_prefix", "|>'", "ll", "
", "'S", "㋿", " 漢", "9", "e漢Z"]} +{"text": "́㋿<|fim_prefix|>🙂\r\n\r\nA
e
\"d12345678fiDž 's'Re <|endoftext|><|fim_prefix|>fi𐞁å㋿ع…🙂­३'(ß🙂\r\n\r\n", "A", "
e", "
", "\"d", "123", "456", "78", "fiDž", " ", "'s", "'Re", " ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "fi𐞁a", "̊㋿", "ع", "…", "🙂­", "३", "'(", "ß", "é>#$% '", "tokens": 47, "pieces": ["!'", "T", "٣٤٥", "٦", "m", "123", "456", "78", "e", "́", " ", "👍🏽", "é", " \n", "…é", ",'", "S", "(­", "İ", "'S", "\n", "<|", "endoftext", "|>", "é", ">#$%", " '"]} +{"text": "!👍🏽
é12345678'll'Re🙂…A#$%!🙂're\rİ>", "tokens": 38, "pieces": ["!👍🏽", "
e", "́", "123", "456", "78", "'ll", "'Re", "🙂", "…A", "#$%!🙂'", "re", "\r", "İ", "><", "EOT", ">"]} +{"text": "DžⅣ \nſ 're…'-tfi'reⅣ \ń.dma", "tokens": 30, "pieces": ["Dž", "Ⅳ", " \n", "ſ", " ", " '", "re", "…", "'-", "t", "fi", "'re", "Ⅳ", " \n", "́.", "dma", ""]} +{"text": "'(👍🏽e#$%'S😀🏽\tß-​Aa㍿\r\n\r\n eå<|fim_prefix|>\u000b!\n\t😀🏽🙂\tt'llZ㍿ ", "tokens": 59, "pieces": ["'(👍🏽", "e", "#$%'", "S", "😀🏽", "\tß", "-​", "Aa", "㍿\r\n\r\n", " ", " <", "META", "_START", ">ea", "̊<|", "fim", "_prefix", "|>", "\u000b", "!\n", "\t", "😀🏽🙂", "\tt", "'ll", "Z", "㍿", " "]} +{"text": "字ꟲ fi​t12345678<'llEOT ­👍🏽\r\n
<|fim_prefix|>a㍿́-<|fim_prefix|>\t <́#$%Z,\nİ \n\r\n", "tokens": 55, "pieces": ["字ꟲ", " fi", "​t", "123", "456", "78", "<'", "llEOT", " ­👍🏽\r\n", "
", "<|", "fim", "_prefix", "|>", "a", "㍿́-<|", "fim", "_prefix", "|>", "\t", " <́#$%", "Z", ",\n", "İ", " \n\r\n"]} +{"text": "㋿", "tokens": 3, "pieces": ["㋿"]} +{"text": "<|fim_prefix|>d字\r\n\r\n-('Reé'VEع 👍🏽٣٤٥٦\r\n\r\nſ𐞁\n٣٤٥٦İ!!>\r​😀🏽9", "tokens": 60, "pieces": ["<|", "fim", "_prefix", "|>", "d字", "\r\n\r\n", "-('", "Ree", "́'", "VEع", " ", "👍🏽", "٣٤٥", "٦", "\r\n\r\n", "ſ𐞁", "\n", "٣٤٥", "٦", "İ", "!!>\r", "​😀🏽", "9"]} +{"text": "
a'Dß,!'S<|endoftext|>\tⅣ'D'S½å😀🏽ع ­ßtع!\u000b!! \n­漢9㋿‍!", "tokens": 52, "pieces": ["
a", "'D", "ß", ",!'", "S", "<|", "endoftext", "|>", "\t", "Ⅳ", "'D", "'S", "½", "a", "̊😀🏽", "ع", " ", "­ß", "tع", "!", "\u000b", "!!", " \n", "­漢", "9", "㋿‍!"]} +{"text": "३ ß😀🏽<|endoftext|>½​́'S\"!!A𐞁漢😀🏽\r٣٤٥٦'VE 😀🏽\u000b​\n
Z\r\n\r\n!!㍿EOT<|fim_prefix|>A‍Z‍ (", "tokens": 82, "pieces": ["३", " ", " ß", "😀🏽<|", "endoftext", "|>", "½", "​́'", "S", "\"!!", "A𐞁漢", "😀🏽\r", "٣٤٥", "٦", "'VE", " ", "😀🏽", "\u000b", "​\n", "
Z", "\r\n\r\n", "!!㍿", "EOT", "<|", "fim", "_prefix", "|>", "A", "‍Z", "‍", " ("]} +{"text": ".a12345678å 'll 's!!𐞁 \n.t'Re३٣٤٥٦ⅣEOTEOT𐞁", "tokens": 40, "pieces": [".a", "123", "456", "78", "a", "̊", " ", " '", "ll", " ", " '", "s", "!!", "𐞁", " \n", ".t", "'Re", "३٣٤", "٥٦Ⅳ", "EOTEOT𐞁"]} +{"text": " \n٣٤٥٦", "tokens": 9, "pieces": [" \n", "٣٤٥", "٦"]} +{"text": ">ſ9!!İ,عß½te\"​é'Té漢m\u000b \t­ #$%9<İ\r's09a\r👍🏽", "tokens": 42, "pieces": [">ſ", "9", "!!", "İ", ",عß", "½", "te", "\"​", "e", "́'", "Te", "́漢m", "\u000b ", "\t", "­", " ", " #$%", "9", "<İ", "\r", "'s", "09", "a", "\r", "👍🏽"]} +{"text": "½​- ३𐞁𐞁éع字's9t'D🙂😀🏽#$%ß字٣٤٥٦s'M", "tokens": 48, "pieces": ["½", "​-<", "EOT", ">", " ", " ", "३", "𐞁𐞁e", "́ع字", "'s", "9", "t", "'D", "🙂😀🏽#$%", "ß字", "٣٤٥", "٦", "s", "'M"]} +{"text": "३é.'Re-٣٤٥٦Dž#$%𐞁​́é<|fim_prefix|>ꟲ\".😀🏽<𐞁t12345678dd\t\t(ḍ̇- ", "tokens": 62, "pieces": ["३", "e", "́.'", "Re", "-", "٣٤٥", "٦", "Dž", "#$%", "𐞁", "​́", "é", "<|", "fim", "_prefix", "|>", "ꟲ", "\".😀🏽<", "𐞁t", "123", "456", "78", "dd", "\t", "\t", "(ḋ", "̣-", " "]} +{"text": "9…३Ⅳ漢EOTe👍🏽𐞁s漢EOTå9fi'll", "tokens": 34, "pieces": ["9", "…", "३Ⅳ", "漢EOTe", "👍🏽", "𐞁s漢EOTa", "̊", "9", "fi", "'ll"]} +{"text": "㋿\r…'VEEOTß12345678å'lléfi​
å'T㍿'s'M \r'TåDžd>漢 d ㍿", "tokens": 58, "pieces": ["㋿\r", "…", "'VE", "EOTß", "123", "456", "78", "a", "̊'", "lléfi", "​<", "EOT", ">", "
a", "̊'", "T", "㍿'", "s", "'M", " \r", "'T", "a", "̊Džd", ">漢", " ", " d", " ", "㍿"]} +{"text": "ꟲ's'ſ…Zt\r\n\r\n<🙂ꟲ\teß'T'Dꟲ'll.𐞁dZdtDžEOT", "tokens": 45, "pieces": ["ꟲ", "'", "s", "'ſ", "…Zt", "\r\n\r\n", "<🙂", "ꟲ", "\teß", "'T", "'D", "ꟲ", "'ll", ".𐞁dZdtDžEOT", ""]} +{"text": "s<|fim_prefix|>​ſ\u000b-d<漢're\"👍🏽(…\r'ſ#$%\"'S<|fim_prefix|>a
'D\n\té-åİ'Ⅳd<|endoftext|>字\r<|fim_prefix|>", "tokens": 76, "pieces": ["s", "<|", "fim", "_prefix", "|>​", "ſ", "\u000b", "-d", "<漢", "'re", "\"👍🏽(", "…\r", "'ſ", "#$%\"'", "S", "<|", "fim", "_prefix", "|>", "a", "
", "'D", "\n", "\té", "-a", "̊İ", "'<", "EOT", ">", "Ⅳ", "d", "<|", "endoftext", "|>", "字", "\r", "<|", "fim", "_prefix", "|>"]} +{"text": "́'re9\r\n\r\n>٣٤٥٦asß>​ع'ſ'VEe\t'S9, Z'Re­𐞁'Re", "tokens": 42, "pieces": ["́'", "re", "9", "\r\n\r\n", ">", "٣٤٥", "٦", "asß", ">​", "ع", "'ſ", "'VE", "e", "\t", "'S", "9", ",", " Z", "'Re", "­<", "META", "_START", ">𐞁", "'Re"]} +{"text": "😀🏽👍🏽\t", "tokens": 12, "pieces": ["😀🏽👍🏽", "\t"]} +{"text": "a㋿'ll!!\r\n\r\né\"! ' -9'reA३ß <#$%'VE\"0!#$%Dž漢'T\r\n", "tokens": 37, "pieces": ["a", "㋿'", "ll", "!!\r\n\r\n", "é", "\"!", " '", " ", "-", "9", "'re", "A", "३", "ß", " ", "<#$%'", "VE", "\"", "0", "!#$%", "Dž漢", "'T", "\r\n"]} +{"text": " \nßDž😀🏽's,m-'ſEOT𐞁𐞁", "tokens": 28, "pieces": [" \n", "ßDž", "😀🏽<", "EOT", ">'", "s", ",m", "-'", "ſEOT𐞁𐞁"]} +{"text": "
\r㍿<|endoftext|>'ll(<|fim_prefix|> fi​eⅣ٣٤٥٦", "tokens": 36, "pieces": ["
\r", "㍿<|", "endoftext", "|>'", "ll", "(<|", "fim", "_prefix", "|>", " fi", "​e", "Ⅳ٣٤", "٥٦"]} +{"text": "İZ \n12345678fi'Re'sſ
字, <|fim_prefix|>'D,12345678\"a ,㋿", "tokens": 34, "pieces": ["İZ", " \n", "123", "456", "78", "fi", "'Re", "'s", "ſ", "
字", ",", " ", "<|", "fim", "_prefix", "|>'", "D", ",", "123", "456", "78", "\"a", " ,㋿"]} +{"text": "EOT漢!fi,İ'se'Re㍿३'VE#$%t0Z\u000bⅣZ", "tokens": 33, "pieces": ["EOT漢", "!fi", ",İ", "'s", "e", "'Re", "㍿", "३", "'VE", "#$%<", "EOT", ">t", "0", "Z", "\u000b", "Ⅳ", "Z"]} +{"text": " \u000b🙂<|fim_prefix|> 
12345678'TtⅣꟲ>fi३…㍿sⅣEOT'VE½étİé", "tokens": 47, "pieces": [" ", "\u000b", "🙂<|", "fim", "_prefix", "|>", " ", "
", "123", "456", "78", "'T", "t", "Ⅳ", "ꟲ", ">fi", "३", "…", "㍿s", "Ⅳ", "EOT", "'VE", "½", "e", "́tİé"]} +{"text": "٣٤٥٦0\r\n12345678<|endoftext|>m'D \n٣٤٥٦'S३ſ\t漢'Re३'M12345678Ⅳ'VE're.字 .<|fim_prefix|>\u000b<|fim_prefix|>m", "tokens": 73, "pieces": ["٣٤٥", "٦0", "\r\n", "123", "456", "78", "<|", "endoftext", "|>", "m", "'D", " \n", "٣٤٥", "٦", "'S", "", "३", "ſ", "\t漢", "'Re", "३", "'M", "123", "456", "78Ⅳ", "'VE", "'re", ".字", " ", ".<|", "fim", "_prefix", "|>", "\u000b", "<|", "fim", "_prefix", "|>", "m"]} +{"text": "㋿ß(12345678\tmع'S", "tokens": 11, "pieces": ["㋿ß", "(", "123", "456", "78", "\tmع", "'S"]} +{"text": "ꟲ​½s\u000b㋿#$%A\"​ Dž!!ḍ̇Z字>fi'S\r\n\rå!e漢.ßa.👍🏽>'", "tokens": 51, "pieces": ["ꟲ", "​", "½", "s", "\u000b", "㋿#$%", "A", "\"​", " Dž", "!!", "ḋ", "̣Z字", ">fi", "'S", "\r\n\r", "a", "̊!", "e漢", ".ßa", ".👍🏽>'"]} +{"text": "‍\t\t'DéA \n <|fim_prefix|>", "tokens": 17, "pieces": ["‍", "\t", "\t", "'D", "e", "́A", " \n", " ", " <|", "fim", "_prefix", "|>"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " Dž#$%'T'T'M漢🙂<|fim_prefix|><'Té'D㋿👍🏽…fi🙂0Dž9👍🏽're😀🏽
éع\téa-.", "tokens": 73, "pieces": [" ", " Dž", "#$%'", "T", "'T", "'M", "漢", "🙂<|", "fim", "_prefix", "|><'", "Té", "'D", "㋿<", "EOT", ">👍🏽", "…fi", "🙂", "0", "Dž", "9", "👍🏽'", "re", "😀🏽", "
e", "́ع", "\téa", "-."]} +{"text": "​\r!,ad字t…𐞁>🙂'VE\r\n\r\nſ Dž, 'ſ'Re \n're're㍿'S0<漢'D,", "tokens": 43, "pieces": ["​\r", "!,", "ad字t", "…𐞁", ">🙂'", "VE", "\r\n\r\n", "ſ", " Dž", ",", " ", " '", "ſ", "'Re", " \n", "'re", "'re", "㍿'", "S", "0", "<漢", "'D", ","]} +{"text": "Z12345678.\"३\u000bİ字fi!!​\r\n\r\n🙂!!'ſ​㍿
'D!!t😀🏽½ſ٣٤٥٦
fi", "tokens": 57, "pieces": ["Z", "123", "456", "78", ".\"", "३", "\u000bİ字fi", "!!​\r\n\r\n", "🙂!!<", "EOT", ">'", "ſ", "​<", "EOT", ">㍿", "
", "'D", "!!", "t", "😀🏽", "½", "ſ", "٣٤٥", "٦", "
fi"]} +{"text": "\r\n\r\ns\t>\"étfiém'VEa're字\"Z\u000b m#$%İ\r<|endoftext|>s🙂ꟲ\te>", "tokens": 39, "pieces": ["\r\n\r\n", "s", "\t", ">\"", "étfie", "́m", "'VE", "a", "'re", "字", "\"Z", "\u000b", " m", "#$%", "İ", "\r", "<|", "endoftext", "|>", "s", "🙂ꟲ", "\te", ">"]} +{"text": "'VE#$%ꟲ'll٣٤٥٦ſꟲ åⅣ.…<|fim_prefix|>٣٤٥٦ea'M!!३Ⅳå'D\r\n​\r\n'S㋿½'ll½ع'MZEOT'", "tokens": 75, "pieces": ["'VE", "#$%", "ꟲ", "'ll", "٣٤٥", "٦", "ſꟲ", " a", "̊", "Ⅳ", ".", "…", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "ea", "'M", "!!", "३Ⅳ", "a", "̊'", "D", "\r\n", "​\r\n", "'S", "㋿", "½", "'ll", "½", "ع", "'M", "ZEOT", "'"]} +{"text": "​Ⅳ EOT<|endoftext|>as!㋿éſd'DDž é,㍿(d​
\r\né a🙂ſ", "tokens": 50, "pieces": ["​", "Ⅳ", " EOT", "<|", "endoftext", "|>", "as", "!㋿", "éſd", "'D", "Dž", " ", " e", "́,㍿(", "d", "​", "
\r\n", "e", "́", " ", " a", "🙂ſ"]} +{"text": "㍿<|endoftext|>fitZ 𐞁'VE'll'D…\t🙂<Dž>A!!'VEEOT'sع字's \"(…😀🏽'T!!'ſ​Z", "tokens": 65, "pieces": ["㍿<|", "endoftext", "|>", "fitZ", " ", " 𐞁", "'VE", "'ll", "'D", "…", "\t", "🙂<", "Dž", ">A", "!!'", "VEEOT", "'s", "ع字", "'s", " ", "\"(", "…", "😀🏽'", "T", "!!'", "ſ", "​Z"]} +{"text": "ḍ̇½'Sa'́9Ⅳfi
 \n\u000bfi e,d…", "tokens": 30, "pieces": ["ḋ", "̣", "½", "'S", "a", "'́", "9", "", "Ⅳ", "fi", "
 \n", "\u000bfi", " ", " e", ",d", "…"]} +{"text": "漢 éå🙂é \n(ſs#$%'s e( ,㋿🙂ꟲ\n㍿
", "tokens": 41, "pieces": ["漢", " éa", "̊🙂", "e", "́", " \n", "(ſs", "#$%'", "s", " ", " e", "(", " ", ",㋿🙂", "ꟲ", "\n", "㍿<", "EOT", ">", "
"]} +{"text": "s'VEß𐞁ſ9 mfi!!㍿<|fim_prefix|>s9  ", "tokens": 43, "pieces": ["s", "'VE", "ß𐞁ſ", "9", " ", " mfi", "!!㍿<|", "fim", "_prefix", "|>", "s", "", "9", "  "]} +{"text": ">-ſEOT'T<<|endoftext|>\r fi漢👍🏽 a\"å㍿ ​EOT\n're-Z \n<|endoftext|>sm٣٤٥٦字", "tokens": 62, "pieces": [">-", "ſEOT", "'T", "<<|", "endoftext", "|>\r", "", " fi漢", "👍🏽", " ", " a", "\"a", "̊㍿", " ", " ​", "EOT", "\n", "'re", "-Z", " \n", "<|", "endoftext", "|>", "sm", "٣٤٥", "٦", "字"]} +{"text": "\"𐞁ſ-٣٤٥٦ßémİ!३٣٤٥٦á>‍\r\n\r\n<|fim_prefix|>#$%ß're
🙂a<|fim_prefix|><|endoftext|>12345678d\",", "tokens": 70, "pieces": ["\"𐞁ſ", "-", "٣٤٥", "٦", "ßémİ", "!", "३٣٤", "٥٦", "a", "́>‍\r\n\r\n", "<|", "fim", "_prefix", "|>#$%", "ß", "'re", "
", "🙂a", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "123", "456", "78", "d", "\","]} +{"text": "😀🏽…,‍('T𐞁#$%
Ⅳع\t'字's're\r\n\r\n'll👍🏽ع .½< 😀🏽é३'VE漢漢'S\nⅣع…A#$%", "tokens": 66, "pieces": ["😀🏽", "…", ",‍('", "T𐞁", "#$%", "
", "Ⅳ", "ع", "\t", "'字", "'s", "'re", "\r\n\r\n", "'ll", "👍🏽", "ع", " ", " .", "½", "<", " ", " 😀🏽", "é", "३", "'VE", "漢漢", "'S", "\n", "Ⅳ", "ع", "…A", "#$%"]} +{"text": "Z \n>३d'S<|endoftext|>12345678👍🏽😀🏽'lĺ(­…㍿\t㋿㋿字fie0'A'll漢s's🙂👍🏽😀🏽'VE<🙂fi", "tokens": 78, "pieces": ["Z", " \n", ">", "३", "d", "'S", "<|", "endoftext", "|>", "123", "456", "78", "👍🏽😀🏽'", "ll", "́(­", "…", "㍿", "\t", "㋿㋿", "字fie", "0", "'A", "'ll", "漢s", "'s", "🙂👍🏽😀🏽'", "VE", "<🙂", "fi"]} +{"text": "é>EOT'VE'VE 'VEⅣ
 'll३0d12345678👍🏽Ⅳ字å字Aḍ̇½<|fim_prefix|>Ⅳ< <|endoftext|>å", "EOT", "'VE", "'VE", " ", "'VE", "Ⅳ", "
", " '", "ll", "३0", "d", "123", "456", "78", "👍🏽", "Ⅳ", "字a", "̊字Aḋ", "̣", "½", "<|", "fim", "_prefix", "|>", "Ⅳ", "<", " ", "<|", "endoftext", "|>", "a", "̊!字<|fim_prefix|>#$%عté12345678😀🏽İ३d­>ꟲ-'sa३½­…é", "tokens": 62, "pieces": ["٣٤٥", "٦", " fi", "!<", "META", "_START", ">字", "<|", "fim", "_prefix", "|>#$%", "عte", "́", "123", "456", "78", "😀🏽", "İ", "३", "d", "­>", "ꟲ", "-'", "sa", "३", "", "½", "­", "…e", "́"]} +{"text": "éEOTe.( ٣٤٥٦<|fim_prefix|>e😀🏽🙂 😀🏽><|endoftext|>é٣٤٥٦Z \n> 'llfié<|fim_prefix|>👍🏽漢ꟲ .'M", "tokens": 84, "pieces": ["éEOT", "e", ".(", " ", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "e", "😀🏽🙂", " 😀🏽><|", "endoftext", "|>", "e", "́", "٣٤٥", "٦", "Z", " \n", "><", "META", "_START", ">", " ", "'ll", "fié", "<|", "fim", "_prefix", "|>👍🏽", "漢ꟲ", " ", " .'", "M"]} +{"text": "ßꟲs𐞁🙂<|fim_prefix|>e\"\r\n\r\n \n", "tokens": 24, "pieces": ["ßꟲs𐞁", "🙂<", "META", "_START", "><|", "fim", "_prefix", "|>", "e", "\"\r\n\r\n", " \n"]} +{"text": "<|fim_prefix|>'T #$%‍'ll<|fim_prefix|>'T \t0字ꟲé😀🏽é\n'🙂( m½é'M\u000b<|fim_prefix|>㋿>", "tokens": 59, "pieces": ["<|", "fim", "_prefix", "|>'", "T", " ", "#$%‍'", "ll", "<|", "fim", "_prefix", "|>'", "T", " ", "\t", "0", "字ꟲé", "😀🏽", "e", "́\n", "'🙂(", " m", "½", "é", "'M", "\u000b", "<|", "fim", "_prefix", "|>㋿>"]} +{"text": "́​ꟲ", "tokens": 5, "pieces": ["́​", "ꟲ"]} +{"text": "'lla३́ꟲ  🙂
", "tokens": 15, "pieces": ["'ll", "a", "३", "́ꟲ", "", " ", " 🙂", "
"]} +{"text": "0٣٤٥٦İ's(12345678\u000b0'M ", "tokens": 68, "pieces": ["0", "", "٣٤٥", "٦", "İ", "'s", "(", "123", "456", "78", "\u000b", "0", "'M", " "]} +{"text": "\u000bſ9́ \n'Re𐞁\r漢<|fim_prefix|>", "tokens": 21, "pieces": ["\u000bſ", "9", "́", " \n", "'Re", "𐞁", "\r", "漢", "<|", "fim", "_prefix", "|>"]} +{"text": "\r\n'D\r\nİ ('M٣٤٥٦ ع", "tokens": 16, "pieces": ["\r\n", "'D", "\r\n", "İ", " ('", "M", "٣٤٥", "٦", " ع"]} +{"text": "Ⅳ\"ꟲ\nß😀🏽9👍🏽é㍿ع㍿Džع!!𐞁 'S9😀🏽9'Ssꟲ'S‍<|endoftext|>\r\n३-
<|fim_prefix|>'EOTḍ̇३a", "tokens": 87, "pieces": ["Ⅳ", "\"ꟲ", "\n", "ß", "😀🏽", "9", "👍🏽", "e", "́㍿", "ع", "㍿Džع", "!!", "𐞁", " ", "'S", "9", "😀🏽", "9", "'S", "sꟲ", "'S", "‍<|", "endoftext", "|>\r\n", "३", "-", "
", "<|", "fim", "_prefix", "|>'", "EOTḋ", "̣", "३", "a"]} +{"text": "🙂́\u000bß!!Ⅳ,\r(漢'sꟲ½字\r\nA,\t'Tå𐞁३!!👍🏽Ⅳ\nt'Se", "tokens": 54, "pieces": ["🙂́<", "META", "_START", ">", "\u000bß", "!!", "Ⅳ", ",\r", "(漢", "'s", "ꟲ", "½", "字", "\r\n", "A", ",", "\t", "'T", "a", "̊𐞁", "३", "!!👍🏽", "Ⅳ", "\n", "t", "'S", "e"]} +{"text": "'D\"se🙂ع12345678e", "tokens": 10, "pieces": ["'D", "\"se", "🙂ع", "123", "456", "78", "e"]} +{"text": "<( \n (", "tokens": 12, "pieces": ["<(<", "META", "_START", ">", " \n", "", " ", "("]} +{"text": "İ'MEOT'ſ\u000b३\r\n\r\nع'M\r\n>\t ", "tokens": 20, "pieces": ["İ", "'M", "EOT", "'ſ", "", "\u000b", "३", "\r\n\r\n", "ع", "'M", "\r\n", ">", "\t "]} +{"text": " -'S\r\né漢0ſ​'s'T'ſ '\u000b'D,㋿'M​td㍿<|endoftext|>", "tokens": 45, "pieces": [" -<", "META", "_START", ">'", "S", "\r\n", "é漢", "0", "ſ", "​<", "EOT", ">'", "s", "'T", "'ſ", " ", " '", "\u000b", "'D", ",㋿'", "M", "​td", "㍿<|", "endoftext", "|>"]} +{"text": "\nꟲ㍿<|endoftext|> 'M'sm½EOT  sꟲ ع🙂!\n!!Ⅳ-<|fim_prefix|>.m'M\r㍿'Rem", "tokens": 54, "pieces": ["\n", "ꟲ", "㍿<|", "endoftext", "|>", " ", " '", "M", "'s", "m", "½", "EOT", " ", " sꟲ", " ع", "🙂!\n", "!!", "Ⅳ", "-<|", "fim", "_prefix", "|>.", "m", "'M", "\r", "㍿'", "Rem"]} +{"text": "ع 'Res", "tokens": 8, "pieces": ["ع", "", " ", "'Re", "s"]} +{"text": "9! \ń'SDž३'Ss'D0(Z🙂ſDž!!ꟲß0…fi", "tokens": 31, "pieces": ["9", "!", " \n", "́'", "SDž", "३", "'S", "s", "'D", "0", "(Z", "🙂ſDž", "!!", "ꟲß", "0", "…fi"]} +{"text": "\"́½ \nⅣAéꟲ👍🏽<A( ́d漢👍🏽'ſ'S12345678\t'D,Dž12345678३", "tokens": 51, "pieces": ["\"́", "½", " \n", "Ⅳ", "Aéꟲ", "👍🏽<", "A", "(", " ", " ́", "d漢", "👍🏽'", "ſ", "'S", "123", "456", "78", "\t", "'D", ",Dž", "123", "456", "78३"]} +{"text": "\r\nå, !!fi ́字½12345678\n're𐞁'Re <|endoftext|>👍🏽tEOT👍🏽 emⅣe", "tokens": 51, "pieces": ["\r\n", "a", "̊,", " ", " !!", "fi", " ", "́字", "½12", "345", "678", "\n", "'re", "𐞁", "'Re", " ", "<|", "endoftext", "|>👍🏽", "tEOT", "👍🏽", " ", " em", "Ⅳ", "e"]} +{"text": "\réße字0's
३ḍ̇0EOT", "tokens": 23, "pieces": ["\r", "e", "́ß", "e字", "0", "'s", "
", "३", "ḋ", "̣", "0", "EOT"]} +{"text": "A'ſ🙂EOT\"ḍ̇Ⅳ㍿́٣٤٥٦", "tokens": 32, "pieces": ["A", "'ſ", "🙂EOT", "\"ḋ", "̣", "Ⅳ", "㍿́", "٣٤٥", "٦"]} +{"text": "३.'M\t", "tokens": 5, "pieces": ["३", ".'", "M", "\t"]} +{"text": "'Re'llİ< \r\n\r\n\nEOT.d'll!'S,", "tokens": 19, "pieces": ["'Re", "'ll", "İ", "<", " \r\n\r\n\n", "EOT", ".d", "'", "ll", "!'", "S", ","]} +{"text": "1234567812345678'D'M\"㍿🙂'S'S<|endoftext|> >\n.'S.é𐞁​🙂 ſ \n<\r\nḍ̇0👍🏽!३,", "tokens": 62, "pieces": ["123", "456", "781", "234", "567", "8", "'D", "'M", "\"㍿🙂'", "S", "'S", "<|", "endoftext", "|>", " >\n", ".'", "S", ".e", "́<", "EOT", ">𐞁", "​🙂", " ſ", " \n", "<\r\n", "ḋ", "̣", "0", "👍🏽!", "३", ","]} +{"text": "'Sß'll
‍ß-é'D.A३!!字're'll(0a'\"ⅣEOT‍🙂\r\n㍿😀🏽A#$%EOT \n!!fíḍ̇", "tokens": 61, "pieces": ["'S", "ß", "'ll", "
", "‍ß", "-e", "́'", "D", ".A", "३", "!!", "字", "'re", "'ll", "(", "0", "a", "'\"", "Ⅳ", "EOT", "‍🙂\r\n", "㍿😀🏽", "A", "#$%", "EOT", " \n", "!!", "fi", "́ḋ", "̣"]} +{"text": "👍🏽Z'ſ \r\n-\r\n!字fi \n,s \n㍿,<|fim_prefix|>12345678🙂漢عemsḍ̇'e\"<|endoftext|>\tⅣ", "tokens": 56, "pieces": ["👍🏽", "Z", "'ſ", " \r\n", "-\r\n", "!字fi", " \n", ",s", " \n", "㍿,<|", "fim", "_prefix", "|>", "123", "456", "78", "🙂漢عemsḋ", "̣'", "e", "\"<|", "endoftext", "|>", "\t", "Ⅳ"]} +{"text": "!!\t\r\n\r\n
😀🏽\r0\u000b😀🏽ع <|endoftext|>0 ٣٤٥٦\u000b​s३<|endoftext|> ́'VEEOT mmå-३字३  漢", "tokens": 73, "pieces": ["!!", "\t\r\n\r\n", "
", "😀🏽\r", "0", "\u000b", "😀🏽", "ع", " <|", "endoftext", "|>", "0", " ", "٣٤٥", "٦", "\u000b", "​s", "३", "<|", "endoftext", "|>", " ", "́'", "VEEOT", " mma", "̊<", "META", "_START", ">-", "३", "字", "", "३", " ", " 漢"]} +{"text": "ع½śdDž!!,Z'llt३́A", "tokens": 16, "pieces": ["ع", "½", "s", "́dDž", "!!,", "Z", "'ll", "t", "३", "́A"]} +{"text": "<|endoftext|>'ll12345678 \n漢're३㍿'S-\"'re…<|endoftext|>'Śḍ̇!!", "tokens": 42, "pieces": ["<|", "endoftext", "|>'", "ll", "123", "456", "78", " \n", "漢", "'re", "३", "㍿'", "S", "-\"'", "re", "…", "<|", "endoftext", "|>'", "S", "́ḋ", "̣!!"]} +{"text": "ḍ̇0!\t​👍🏽9ſ(dⅣt'VE­0'D<漢\r\n\r\nA<|fim_prefix|>EOTfi", "tokens": 44, "pieces": ["ḋ", "̣", "0", "!", "\t", "​👍🏽", "9", "ſ", "(d", "Ⅳ", "t", "'VE", "­", "0", "'D", "<漢", "\r\n\r\n", "A", "<|", "fim", "_prefix", "|>", "EOTfi"]} +{"text": "‍ \"\rsß½ꟲß😀🏽…漢'M㍿<́-t‍漢Zعd<|endoftext|>", "tokens": 43, "pieces": ["‍", " ", " \"\r", "sß", "½", "ꟲß", "😀🏽", "…漢", "'M", "㍿<́-", "t", "‍漢Zعd", "<|", "endoftext", "|>"]} +{"text": " 're!ém<|fim_prefix|>​\r\n<|endoftext|>\rع​#$%'llḍ̇㋿𐞁A!>\rå­12345678ſéꟲ٣٤٥٦Am\r\n\r\nع'ſ'", "tokens": 76, "pieces": [" '", "re", "!e", "́m", "<|", "fim", "_prefix", "|>​\r\n", "<|", "endoftext", "|>\r", "ع", "​#$%'", "llḋ", "̣㋿", "𐞁A", "!>\r", "a", "̊­", "123", "456", "78", "ſe", "́ꟲ", "٣٤٥", "٦", "Am", "\r\n\r\n", "ع", "'ſ", "'"]} +{"text": "'sEOTefi'\r\n\r\ń­ß<|endoftext|>t'reⅣ🙂", "tokens": 23, "pieces": ["'s", "EOTefi", "'\r\n\r\n", "́­", "ß", "<|", "endoftext", "|>", "t", "'re", "Ⅳ", "🙂"]} +{"text": ">(😀🏽'M\téⅣ\u000b'VE!! \n(\tm\t12345678<.t<|fim_prefix|>'re​, ́ꟲ!!​\r\nDž'sḍ̇", "tokens": 56, "pieces": [">(😀🏽'", "M", "\té", "Ⅳ", "\u000b", "'VE", "!!", " \n", "(", "\tm", "\t", "123", "456", "78", "<.", "t", "<|", "fim", "_prefix", "|>'", "re", "​,", " ", " ́", "ꟲ", "!!​<", "EOT", ">\r\n", "Dž", "'s", "ḋ", "̣"]} +{"text": "🙂'ſⅣ'Re漢́㍿'Mmd-t!!🙂-!!'Sİ'ſ३㋿\r𐞁'S'T<|endoftext|>m'VEſ…dع'rem٣٤٥٦", "tokens": 70, "pieces": ["🙂'", "ſ", "Ⅳ", "'Re", "漢", "́㍿<", "META", "_START", ">'", "Mmd", "-t", "!!🙂-!!'", "Sİ", "'ſ", "३", "㋿\r", "𐞁", "'S", "'T", "<|", "endoftext", "|>", "m", "'VE", "ſ", "…dع", "'re", "m", "٣٤٥", "٦"]} +{"text": "-👍🏽­'ll12345678!३㋿ß‍ 'så\r\nⅣ\r\n\r\n
\r\n ㍿३ m \n\t", "tokens": 42, "pieces": ["-👍🏽­'", "ll", "123", "456", "78", "!", "३", "㋿ß", "‍", " '", "sa", "̊\r\n", "Ⅳ", "\r\n\r\n
\r\n", " ", "㍿", "३", " m", " \n\t"]} +{"text": "ḍ̇!m㍿s\r ,𐞁ع fí", "tokens": 23, "pieces": ["ḋ", "̣!", "m", "㍿s", "\r", " ", ",𐞁ع", " ", " fi", "́"]} +{"text": ",İ\r\n\u000b", "tokens": 4, "pieces": [",İ", "\r\n\u000b"]} +{"text": "‍#$%<'S'ReⅣee𐞁'M \r\n9-٣٤٥٦…,'re٣٤٥٦'M'VE 'T٣٤٥٦A-\"-'VEſſ-", "tokens": 70, "pieces": ["‍#$%<'", "S", "'Re", "Ⅳ", "ee𐞁", "'M", " \r\n", "9", "-", "٣٤٥", "٦", "…", ",'", "re", "٣٤٥", "٦", "'M", "'VE", "", " ", " '", "T", "٣٤٥", "٦", "A", "-\"-'", "VEſſ", "-"]} +{"text": "<|endoftext|>३ <‍İ'så३'Mta's'M'ſ're \r\n\r\nꟲ\n😀🏽'Reꟲå\ndꟲ9Z\r\u000b \n\r\nté­İ", "tokens": 64, "pieces": ["<|", "endoftext", "|>", "३", " <‍<", "EOT", ">İ", "'s", "a", "̊", "३", "'M", "ta", "'s", "'M", "'ſ", "'re", " \r\n\r\n", "ꟲ", "\n", "😀🏽'", "Reꟲa", "̊\n", "dꟲ", "9", "Z", "\r\u000b \n\r\n", "té", "­İ"]} +{"text": "<|endoftext|>é‍å \n३!😀🏽 \n'VE\t\r\n\r\n\néßå🙂\r\n\r\nfia", "tokens": 51, "pieces": ["㋿<", "META", "_START", "><|", "endoftext", "|>", "é", "‍", "a", "̊", " \n", "३", "!😀🏽", " \n", "'VE", "\t\r\n\r\n\n", "éßa", "̊🙂\r\n\r\n", "fia"]} +{"text": "字s12345678", "tokens": 5, "pieces": ["字s", "123", "456", "78"]} +{"text": "ḍ̇m㋿\n👍🏽,😀🏽 漢 \n­'T漢\t́'MåEOT're9t­\rß! Ⅳ‍'Re\r", "tokens": 49, "pieces": ["\u000b", "́'", "MİA", "123", "456", "78", "EOT", "<|", "fim", "_prefix", "|>'", "T漢", "\t", "́'", "Ma", "̊EOT", "'re", "", "9", "t", "­\r", "ß", "!", " ", "Ⅳ", "‍'", "Re", "\r"]} +{"text": "漢٣٤٥٦d<|fim_prefix|>'sEOT #$%​<|endoftext|> 'Sfiع!\"ſ漢", "tokens": 42, "pieces": ["漢", "٣٤٥", "٦", "d", "<|", "fim", "_prefix", "|>'", "sEOT", " ", "#$%​<|", "endoftext", "|>", " ", "'S", "fiع", "!\"", "ſ漢"]} +{"text": "'🙂ſ३'VE99
t…🙂字EOT0. \n​عEOTtꟲDžfí'llİ‍\r\né", "tokens": 44, "pieces": ["'🙂", "ſ", "३", "'VE", "99", "
t", "…", "🙂字EOT", "0", ".", " \n", "​عEOTtꟲDžfi", "́'", "llİ", "‍\r\n", "e", "́"]} +{"text": "'S​'ll‍
İ", "tokens": 9, "pieces": ["'S", "​'", "ll", "‍", "
İ"]} +{"text": "å'Så ३é👍🏽'VE><\r\n\r\n😀🏽𐞁'aEOT\r\n\r\nß'll0Dže​Aꟲ", "tokens": 53, "pieces": ["a", "̊'", "Sa", "̊", " ", " ", "३", "é", "👍🏽'", "VE", "><\r\n\r\n", "😀🏽", "𐞁", "'<", "META", "_START", ">aEOT", "\r\n\r\n", "ß", "'ll", "0", "Dže", "​Aꟲ"]} +{"text": "''Mع>🙂ꟲ\r\n 9e\"字
", "tokens": 20, "pieces": ["''", "Mع", ">🙂", "ꟲ", "\r\n", " ", "9", "e", "\"字", "
"]} +{"text": ",
!<|endoftext|>‍㍿…'M‍㍿ \r\n'Mtİḍ̇fi👍🏽.\r\n\r\né>'re\"\r\n", "tokens": 45, "pieces": [",", "
", "!<|", "endoftext", "|>‍㍿", "…", "'M", "‍㍿", " \r\n", "'M", "tİḋ", "̣fi", "👍🏽.\r\n\r\n", "é", ">'", "re", "\"\r\n"]} +{"text": "'M'D‍012345678'll12345678<|fim_prefix|>\u000bA ḍ̇!Ⅳ,'ll‍A>İé \n👍🏽're9''ll漢-.a\n㋿é", "tokens": 65, "pieces": ["'M", "'D", "‍", "012", "345", "678", "'ll", "123", "456", "78", "<|", "fim", "_prefix", "|>", "\u000bA", " ḋ", "̣!", "Ⅳ", ",'", "ll", "‍A", ">İe", "́", " \n", "👍🏽'", "re", "9", "''", "ll", "漢", "-.", "a", "\n", "㋿é"]} +{"text": "'VE-'llḍ̇fi'DⅣ.​,́", "tokens": 18, "pieces": ["'VE", "-'", "llḋ", "̣fi", "'D", "Ⅳ", ".​,́"]} +{"text": "<|endoftext|>tm😀🏽!!'T,'sfi,漢,'ſfiZé<|fim_prefix|>,,!!
\t", "tokens": 42, "pieces": ["<|", "endoftext", "|>", "tm", "😀🏽!!'", "T", ",'", "sfi", ",漢", ",'", "ſfiZe", "́<|", "fim", "_prefix", "|>,,!!", "
\t"]} +{"text": "́'Reꟲ \n'M>. 9 EOT#$%👍🏽t字 \nt​  ٣٤٥٦(<|fim_prefix|>'re\t", "tokens": 45, "pieces": ["́'", "Reꟲ", " \n", "'M", ">.", " ", "9", " EOT", "#$%👍🏽", "t字", " \n", "t", "​", " ", " ", "٣٤٥", "٦", "(<|", "fim", "_prefix", "|>'", "re", "\t"]} +{"text": "\t're…'VEſ字ea🙂\re٣٤٥٦'T字'reé'llꟲ'ſ½ \nⅣa", "tokens": 40, "pieces": ["\t", "'re", "…", "'VE", "ſ字ea", "🙂\r", "e", "٣٤٥", "٦", "'T", "字", "'re", "e", "́'", "llꟲ", "'ſ", "½", " \n", "Ⅳ", "a"]} +{"text": "\tⅣe 0漢\u000b'D, ,漢Ⅳع😀🏽é́
'D \n\t.‍dꟲİ‍#$%A'T", "tokens": 52, "pieces": ["\t", "Ⅳ", "e", " ", "0", "漢", "\u000b", "'", "D", ",", " ", ",漢", "Ⅳ", "ع", "😀🏽", "é", "́", "
", "'D", "", " \n", "\t", ".‍", "dꟲİ", "‍#$%", "A", "'T"]} +{"text": "字!!'s 're'D\r\n\r\n(​👍🏽\r\n\r\n𐞁'll fiDž123456780‍'Tå漢٣٤٥٦", "tokens": 47, "pieces": ["字", "!!'", "s", " ", " '", "re", "'D", "\r\n\r\n", "(​👍🏽\r\n\r\n", "𐞁", "'ll", " fiDž", "123", "456", "780", "‍'", "Ta", "̊漢", "٣٤٥", "٦"]} +{"text": "Ⅳ. ⅣEOT \raſ漢<👍🏽-!e.٣٤٥٦.m", "tokens": 35, "pieces": ["Ⅳ", ".", " ", "Ⅳ", "EOT", " \r", "aſ漢", "<👍🏽-!", "e", ".", "٣٤٥", "٦", ".m"]} +{"text": "-#$%'T…Z<|fim_prefix|>\"9 \n½<|fim_prefix|>'M", "tokens": 25, "pieces": ["-#$%'", "T", "…Z", "<|", "fim", "_prefix", "|>\"", "9", " \n", "½", "<|", "fim", "_prefix", "|>'", "M"]} +{"text": "ꟲ३'T .
<''T'ReAe'S<|endoftext|>​a<|endoftext|>𐞁\r\n >EOT're😀🏽Z'T\"#$%ßZ'll\r\n\r\n㍿", "tokens": 63, "pieces": ["ꟲ", "३", "'T", " ", " .", "
", "<''", "T", "'Re", "Ae", "'S", "<|", "endoftext", "|>​", "a", "<|", "endoftext", "|>", "𐞁", "\r\n", " ", ">EOT", "'re", "😀🏽", "Z", "'T", "\"#$%", "ßZ", "'ll", "\r\n\r\n", "㍿"]} +{"text": ">m­​ \n'Ds\u000bé㋿́", "tokens": 12, "pieces": [">m", "­​", " \n", "'D", "s", "\u000bé", "㋿́"]} +{"text": "<‍\n'MZfiß\tfi!'S\r…'s!!tǻ漢عꟲعå'ſEOT", "tokens": 38, "pieces": ["<‍\n", "'M", "Zfiß", "\tfi", "!'", "S", "\r", "…", "'s", "!!", "ta", "̊́", "漢عꟲعa", "̊'", "ſEOT"]} +{"text": "字😀🏽👍🏽‍ع\u000b'S㍿<|endoftext|>a!'ll", "tokens": 30, "pieces": ["字", "😀🏽👍🏽‍", "ع", "\u000b", "'S", "㍿<|", "endoftext", "|>", "a", "!'", "ll"]} +{"text": "ꟲ<|endoftext|>‍😀🏽​('ſ<|endoftext|>\t ​\u000b'D!!㍿'s\r\n\r\n>\"字 \n‍", "tokens": 48, "pieces": ["ꟲ", "<|", "endoftext", "|>‍😀🏽<", "META", "_START", ">​('", "ſ", "<|", "endoftext", "|>", "\t ", " ​", "\u000b", "'D", "!!㍿'", "s", "\r\n\r\n", ">\"", "字", " \n", "‍"]} +{"text": "m \n漢'S \"İ漢𐞁", "tokens": 17, "pieces": ["m", " \n", "漢", "'", "S", " ", " \"", "İ漢𐞁"]} +{"text": " EOTé٣٤٥٦a#$%عEOT#$%ع\n🙂 aZ​EOT\r\nع字m'Re𐞁Ⅳ'SA'S  \u000b
!EOT'll字", "tokens": 58, "pieces": [" EOTe", "́", "٣٤٥", "٦", "a", "#$%", "عEOT", "#$%", "ع", "\n", "🙂", " ", " aZ", "​EOT", "\r\n", "ع字m", "'Re", "𐞁", "Ⅳ", "'S", "A", "'S", "  \u000b", "
", "!EOT", "'ll", "字"]} +{"text": "́عé३'re!!.'­a !0s…\"åé's😀🏽9", "tokens": 29, "pieces": ["́عe", "́", "३", "'re", "!!.'­", "a", " !", "0", "s", "…", "\"a", "̊e", "́'", "s", "😀🏽", "9"]} +{"text": "ḍ̇é\t>\r.d‍'Re字 \tß!\t \n漢'fi t", "tokens": 29, "pieces": ["ḋ", "̣é", "\t", ">\r", ".d", "‍'", "Re字", " ", "\tß", "!<", "EOT", ">", "\t \n", "漢", "'fi", " t"]} +{"text": "Z'S0A\r\n\r\n\r\n\r\nß,㋿", "tokens": 11, "pieces": ["Z", "'S", "0", "A", "\r\n\r\n\r\n\r\n", "ß", ",㋿"]} +{"text": "½é½😀🏽#$%­😀🏽0 9ſ(\r\r>𐞁'Sß-<|fim_prefix|>👍🏽\n👍🏽İ\"عZ'Z,
", "tokens": 62, "pieces": ["½", "e", "́", "½", "😀🏽#$%­😀🏽", "0", " ", "9", "ſ", "(\r\r", ">𐞁", "'S", "ß", "-<|", "fim", "_prefix", "|>👍🏽\n", "👍🏽", "İ", "\"عZ", "'Z", ",", "
"]} +{"text": "!DžEOT's𐞁t ́\t🙂a٣٤٥٦ḍ̇ꟲAZ .m0 !!İ'Dt\r'll<|endoftext|>Z\nZ#$%'s­㋿‍#$%", "tokens": 71, "pieces": ["!DžEOT", "'s", "𐞁t", " ́", "\t", "🙂a", "٣٤٥", "٦", "ḋ", "̣ꟲAZ", " ", " <", "EOT", ">.", "m", "0", " ", " !!", "İ", "'D", "t", "\r", "'ll", "<|", "endoftext", "|>", "Z", "\n", "Z", "#$%'", "s", "­㋿‍#$%"]} +{"text": "<(d İs 9!!\r\n\r\n<|endoftext|>s…\r\n'ſ'D'ſ", "tokens": 27, "pieces": ["<(", "d", " ", " İs", " ", "9", "!!\r\n\r\n", "<|", "endoftext", "|>", "s", "…\r\n", "'ſ", "'D", "'ſ"]} +{"text": "́½0", "tokens": 3, "pieces": ["́", "½0"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㋿é­", "tokens": 5, "pieces": ["㋿é", "­"]} +{"text": "😀🏽aع'VE\"9㍿#$%😀🏽
eEOT!!'sſ<|fim_prefix|>'Re\u000bZ漢'T>­𐞁३ e'ſ‍", "tokens": 62, "pieces": ["😀🏽<", "EOT", ">aع", "'VE", "\"", "9", "㍿#$%😀🏽", "
eEOT", "!!'", "sſ", "<|", "fim", "_prefix", "|>'", "Re", "\u000bZ漢", "'T", ">­", "𐞁", "३", " e", "'ſ", "‍"]} +{"text": "#$% #$%Zꟲ漢 EOT'ſ'reA­'VE漢😀🏽३😀🏽s9( >\r \n\u000b
m<|fim_prefix|>< \n٣٤٥٦​e\r\n٣٤٥٦
", "tokens": 76, "pieces": ["#$%", " ", "#$%", "Zꟲ漢", " EOT", "'ſ", "'re", "A", "­'", "VE漢", "😀🏽", "३", "😀🏽", "s", "9", "(", " >\r", " \n", "\u000b", "
m", "<|", "fim", "_prefix", "|><", " \n", "٣٤٥", "٦", "​e", "\r\n", "٣٤٥", "٦", "
"]} +{"text": "'s 
\nİ,'VE -a \n\r12345678#$%'S", "tokens": 22, "pieces": ["'s", "", " 
\n", "İ", ",'", "VE", " ", "-a", " \n\r", "123", "456", "78", "#$%'", "S"]} +{"text": "'Mdꟲ字", "tokens": 6, "pieces": ["'M", "dꟲ字"]} +{"text": "…ſ 12345678…㍿  😀🏽m!!ḍ̇", "tokens": 27, "pieces": ["…ſ", " ", "123", "456", "78", "…", "㍿", " ", " ", "😀🏽", "m", "!!", "ḋ", "̣"]} +{"text": "12345678e.fi're#$%­ \n,漢𐞁👍🏽", "tokens": 25, "pieces": ["123", "456", "78", "e", ".fi", "'re", "#$%­", " \n", ",漢𐞁", "👍🏽"]} +{"text": ". 漢sss\u000b\r\n\r\nſs\n😀🏽\t'ret EOT!!‍㍿,\r're\u000b'Re\"Ⅳ'M ", "tokens": 38, "pieces": [".", " 漢sss", "\u000b\r\n\r\n", "ſs", "\n", "😀🏽", "\t", "'re", "t", " EOT", "!!‍㍿,\r", "'re", "\u000b", "'Re", "\"", "Ⅳ", "'M", " "]} +{"text": "!!(!!🙂A㋿㍿0㍿", "tokens": 17, "pieces": ["!!(!!🙂", "A", "㋿㍿", "0", "㍿"]} +{"text": "𐞁Ź\r\n\r\n're'Tſ'D\td'T're9­​e🙂Zع\n'D­\r\n\u000b😀🏽½.ع字😀🏽İEOT½", "tokens": 50, "pieces": ["𐞁Z", "́\r\n\r\n", "'re", "'T", "ſ", "'D", "\td", "'T", "'re", "9", "­<", "EOT", ">​", "e", "🙂Zع", "\n", "'D", "­\r\n", "\u000b", "😀🏽", "½", ".ع字", "😀🏽", "İEOT", "½"]} +{"text": "İ>", "tokens": 2, "pieces": ["İ", ">"]} +{"text": "é𐞁漢", "tokens": 7, "pieces": ["é𐞁漢"]} +{"text": "!! \n½👍🏽😀🏽#$%<|endoftext|><|fim_prefix|>.㋿<|endoftext|>\u000b", "tokens": 40, "pieces": ["!!", " \n", "½", "👍🏽😀🏽#$%<|", "endoftext", "|><|", "fim", "_prefix", "|>.㋿<|", "endoftext", "|>", "\u000b"]} +{"text": "'ll\r\n<|endoftext|>İ­\r\n\r\n's t's'reét\r٣٤٥٦dd(>m<|endoftext|>'D㍿é", "tokens": 46, "pieces": ["'ll", "\r\n", "<|", "endoftext", "|>", "İ", "­\r\n\r\n", "'s", " t", "'s", "'re", "e", "́t", "\r", "٣٤٥", "٦", "dd", "(><", "EOT", ">m", "<|", "endoftext", "|>'", "D", "㍿é"]} +{"text": "𐞁𐞁'T", "tokens": 9, "pieces": ["𐞁𐞁", "'T"]} +{"text": "12345678'T'ſ#$%d​EOTⅣ-…<|fim_prefix|>ꟲ's12345678Džé'llm'VE(Aåİ́", "tokens": 47, "pieces": ["123", "456", "78", "'T", "'ſ", "#$%", "d", "​EOT", "Ⅳ", "-", "…", "<|", "fim", "_prefix", "|>", "ꟲ", "'s", "123", "456", "78", "Džé", "'ll", "m", "'VE", "(Aa", "̊İ", "́"]} +{"text": "👍🏽12345678­  \n0å'S​́ſ0Dž9s12345678é\"ſEOT'llḍ̇-éDžfi're'S", "tokens": 55, "pieces": ["👍🏽", "123", "456", "78", "­", " ", "", " \n", "0", "a", "̊'", "S", "​́", "ſ", "0", "Dž", "9", "s", "123", "456", "78", "e", "́\"", "ſEOT", "'ll", "ḋ", "̣-", "e", "́Džfi", "'re", "'S"]} +{"text": "'llåfi😀🏽e\r\n\t d<|fim_prefix|>", "tokens": 23, "pieces": ["'ll", "a", "̊fi", "😀🏽", "e", "\r\n", "\t", " d", "<|", "fim", "_prefix", "|>"]} +{"text": "'Re\r\n\r\n<|endoftext|>#$%( \n", "tokens": 15, "pieces": ["'Re", "\r\n\r\n", "<|", "endoftext", "|>#$%(", " \n"]} +{"text": ",­s.\t\u000b㋿
!!😀🏽'M<|endoftext|>#$%㋿漢e's🙂\t३e'T'm!!.,EOT
­mꟲZ", "tokens": 62, "pieces": [",<", "META", "_START", ">­", "s", ".", "\t", "\u000b", "㋿", "
", "!!😀🏽'", "M", "<|", "endoftext", "|>#$%㋿", "漢e", "'s", "🙂", "\t", "३", "e", "'T", "'m", "!!.,", "EOT", "
", "­", "mꟲZ"]} +{"text": "​>ꟲꟲ㋿\r\nſ👍🏽'Re'T ½'M(", "tokens": 27, "pieces": ["​>", "ꟲꟲ", "㋿\r\n", "ſ", "👍🏽'", "Re", "'T", " ", "½", "'M", "("]} +{"text": "Aꟲſ
!dfiEOT12345678", "tokens": 18, "pieces": ["Aꟲſ", "
", "!dfiEOT", "123", "456", "78"]} +{"text": "字'S<|endoftext|>9('llꟲ,d-\r\n\r\n12345678té\r\n\r\n٣٤٥٦'S9🙂\"-ꟲA9 \n>'llſ­EOT", "tokens": 51, "pieces": ["字", "'S", "<|", "endoftext", "|>", "9", "('", "llꟲ", ",d", "-\r\n\r\n", "123", "456", "78", "té", "\r\n\r\n", "٣٤٥", "٦", "'S", "9", "🙂\"-", "ꟲA", "9", " \n", ">'", "llſ", "­EOT"]} +{"text": " ३fi㍿(\re'reſ\"'VE0😀🏽!!𐞁12345678<|endoftext|>ع ZZ'Reé­a9🙂́'ſ­s​", "tokens": 53, "pieces": [" ", " ", "३", "fi", "㍿(\r", "e", "'re", "ſ", "\"'", "VE", "0", "😀🏽!!", "𐞁", "123", "456", "78", "<|", "endoftext", "|>", "ع", " ZZ", "'Re", "é", "­a", "9", "🙂́'", "ſ", "­s", "​"]} +{"text": "‍'re!!'Re\rfiDž½🙂'Dḍ̇d'ſ…'s\r\n\r\n\r\n('", "tokens": 31, "pieces": ["‍'", "re", "!!'", "Re", "\r", "fiDž", "½", "🙂'", "Dḋ", "̣d", "'ſ", "…", "'s", "\r\n\r\n\r\n", "('"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "!
\t​İ'M ㋿'Dm>ꟲ
'llEOTꟲ ꟲꟲ\"fi\r\nm!ß(…\u000b…𐞁", "tokens": 49, "pieces": ["!", "
", "\t", "​İ", "'M", " ㋿'", "Dm", ">ꟲ", "
", "'ll", "EOTꟲ", " ꟲꟲ", "\"fi", "\r\n", "m", "!ß", "(", "…\u000b", "…𐞁"]} +{"text": "٣٤٥٦'D \n#$%\t0-aå😀🏽0‍!e㍿🙂#$%fi-…٣٤٥٦é", "tokens": 49, "pieces": ["٣٤٥", "٦", "'D", " \n", "#$%", "\t", "0", "-aa", "̊😀🏽", "0", "‍!", "e", "㍿🙂#$%", "fi", "-", "…", "٣٤٥", "٦", "é"]} +{"text": "'re\r\n​!! \n​漢A\r\n㍿!!\r\nع.😀🏽'ſ<|fim_prefix|>\rZ<\r\n", "tokens": 39, "pieces": ["'re", "\r\n", "​!!", " \n", "​漢A", "\r\n", "㍿!!\r\n", "ع", ".😀🏽'", "ſ", "<|", "fim", "_prefix", "|>\r", "Z", "<\r\n"]} +{"text": "­ꟲ9'M.Z 😀🏽'VE's😀🏽>\r\n\r\nß .
 \ń'sḍ̇'VEm", "tokens": 44, "pieces": ["­ꟲ", "9", "'M", ".Z", " ", "😀🏽'", "VE", "'s", "😀🏽>\r\n\r\n", "ß", " .", "
 \n", "́<", "META", "_START", ">'", "sḋ", "̣<", "EOT", ">'", "VEm"]} +{"text": "Dž \r\n\r\nß㍿'re<|fim_prefix|>👍🏽'Re ​0", "tokens": 27, "pieces": ["Dž", " \r\n\r\n", "ß", "㍿'", "re", "<|", "fim", "_prefix", "|>👍🏽'", "Re", " ​", "0"]} +{"text": "İe'll", "tokens": 3, "pieces": ["İe", "'ll"]} +{"text": "ſ<|endoftext|>
é0'Re#$%daꟲ 'S٣٤٥٦'Re½漢\r\n ٣٤٥٦́ ㍿A'll㋿(", "tokens": 63, "pieces": ["ſ", "<|", "endoftext", "|>", "
e", "́", "0", "'Re", "#$%<", "EOT", ">daꟲ", " ", " '", "S", "٣٤٥", "٦", "'Re", "½", "漢", "\r\n", " ", " ", "٣٤٥", "٦", "́", " ", "㍿A", "'ll", "㋿("]} +{"text": "٣٤٥٦-tꟲ9EOTſ\u000b́Zm0-.<|endoftext|>\r\n ́ <'s<|endoftext|>
", "tokens": 52, "pieces": ["٣٤٥", "٦", "-tꟲ", "9", "EOT", "ſ", "\u000b", "́Zm", "0", "-.<|", "endoftext", "|>\r\n", " ", " ́", " ", " <'", "s", "<|", "endoftext", "|>", "
"]} +{"text": "a'd12345678𐞁", "tokens": 9, "pieces": ["a", "'d", "123", "456", "78", "𐞁"]} +{"text": "ß\r'ſ\u000b٣٤٥٦>Ⅳ㍿Ⅳséåſ𐞁\u000bİ'VE字EOT'Re😀🏽 0😀🏽eaⅣ", "tokens": 73, "pieces": ["ß", "\r", "'ſ", "\u000b", "٣٤٥", "٦", ">", "Ⅳ", "㍿", "Ⅳ", "séa", "̊<", "m", "\"", " \n \n\r\n\r\n", "'Re", "'D", "\u000b𐞁", ".<", "EOT", ">ſ𐞁", "\u000bİ", "'VE", "字EOT", "'Re", "😀🏽", " ", "0", "😀🏽", "ea", "Ⅳ"]} +{"text": "\r\n\r\n<|endoftext|>!!\u000b'Dḍ̇å\r …a<|fim_prefix|>!!#$%👍🏽ꟲⅣ-\tt.\r\n
\"​𐞁\"\r\n\r\n'Z'ſ'S#$%", "tokens": 72, "pieces": ["\r\n\r\n", "<|", "endoftext", "|>!!", "\u000b", "'D", "ḋ", "̣a", "̊\r", " ", "…a", "<|", "fim", "_prefix", "|>!!#$%👍🏽", "ꟲ", "Ⅳ", "-", "\tt", ".<", "EOT", ">\r\n", "
", "\"​", "𐞁", "\"\r\n\r\n", "'Z", "'ſ", "'S", "#$%<", "META", "_START", ">"]} +{"text": "\r'T½\n٣٤٥٦'\"m\"𐞁", "tokens": 26, "pieces": ["\r", "'T", "½", "\n", "٣٤٥", "٦", "'\"", "m", "\"𐞁", ""]} +{"text": "\r\nⅣ\r'Reع㋿​d🙂m<|endoftext|>12345678dḍ̇\"'sſ'Re½t#$%🙂e#$% 'll", "tokens": 50, "pieces": ["\r\n", "Ⅳ", "\r", "'Re", "ع", "㋿​", "d", "🙂m", "<|", "endoftext", "|>", "123", "456", "78", "dḋ", "̣\"'", "sſ", "'", "Re", "½", "t", "#$%🙂", "e", "#$%", " ", "'ll"]} +{"text": "🙂!!\n<|fim_prefix|>​'M0<|endoftext|>…‍fieſaet😀🏽​aZ  'll👍🏽A", "tokens": 51, "pieces": ["🙂!!\n", "<|", "fim", "_prefix", "|>​'", "M", "0", "<|", "endoftext", "|>", "…", "‍fieſaet", "😀🏽​", "aZ", " ", " ", "'ll", "👍🏽", "A"]} +{"text": "#$%>. 😀🏽e\u000b'll👍🏽EOT\"٣٤٥٦eꟲ!!>٣٤٥٦12345678<\"ſa'Re'S\tDž12345678'D", "tokens": 57, "pieces": ["#$%>.", " 😀🏽", "e", "\u000b", "'ll", "👍🏽", "EOT", "\"", "٣٤٥", "٦", "eꟲ", "!!>", "٣٤٥", "٦12", "345", "678", "<\"", "ſa", "'Re", "'S", "\tDž", "123", "456", "78", "'D"]} +{"text": "㍿\nع\r'ſ912345678Ⅳå,'M'Ré", "tokens": 21, "pieces": ["㍿\n", "ع", "\r", "'ſ", "912", "345", "678", "Ⅳ", "a", "̊,'", "M", "'Re", "́"]} +{"text": "t'Dm🙂<\n0\r\n\r\n(d-🙂😀🏽-'SA😀🏽", "tokens": 26, "pieces": ["t", "'D", "m", "🙂<\n", "0", "\r\n\r\n", "(d", "-🙂😀🏽-'", "SA", "😀🏽"]} +{"text": "Ⅳ9små're३ >​\"fi's<|fim_prefix|>!٣٤٥٦d‍İ#$%<|fim_prefix|>", "tokens": 46, "pieces": ["Ⅳ9", "sma", "̊'", "re", "३", " >​\"", "fi", "'s", "<|", "fim", "_prefix", "|>!", "٣٤٥", "٦", "d", "‍İ", "#$%<|", "fim", "_prefix", "|>"]} +{"text": "㋿ Z", "tokens": 5, "pieces": ["㋿", " Z"]} +{"text": "\"'\r\nİ<|fim_prefix|>. \nå.(\u000bfi🙂ḍ̇  ٣٤٥٦İ𐞁åſ👍🏽", "tokens": 53, "pieces": ["\"'\r\n", "İ", "<|", "fim", "_prefix", "|>.", " \n", "a", "̊.(", "\u000bfi", "🙂ḋ", "̣", " ", " ", "٣٤٥", "٦", "İ𐞁a", "̊ſ", "👍🏽"]} +{"text": "é \r𐞁½🙂३", "tokens": 16, "pieces": ["é", " \r", "𐞁", "½", "🙂<", "META", "_START", ">", "३"]} +{"text": "å👍🏽'Dع  ́𐞁㍿'re👍🏽㍿½漢<|fim_prefix|> 😀🏽'ſå e-㍿३", "tokens": 66, "pieces": ["a", "̊👍🏽'", "D", "ع", " ", " ", "́𐞁", "㍿'", "re", "👍🏽㍿", "½", "漢", "<|", "fim", "_prefix", "|>", " ", "😀🏽'", "ſa", "̊", " ", " e", "-㍿", "३"]} +{"text": "漢٣٤٥٦\r<|endoftext|>👍🏽٣٤٥٦½t…'VE'VE", "tokens": 40, "pieces": ["漢", "٣٤٥", "٦", "\r", "<|", "endoftext", "|>👍🏽", "٣٤٥", "٦½", "t", "…", "'VE", "'VE"]} +{"text": " \nİ-<漢\r\"\u000b
'S\r<|fim_prefix|>👍🏽sععt́㍿sꟲ𐞁\t'9\t😀🏽", "tokens": 60, "pieces": [" \n", "İ", "-<", "漢", "\r", "\"", "\u000b", "
", "'S", "\r", "<|", "fim", "_prefix", "|><", "META", "_START", ">👍🏽", "sععt", "́㍿", "s", "ꟲ𐞁", "\t", "'", "9", "\t", "😀🏽"]} +{"text": " 👍🏽 #$%-​<ꟲİ'ſ㍿EOT><,!me👍🏽A\n.d😀🏽Z'VE'VE>m'sZ", "tokens": 51, "pieces": [" ", " 👍🏽", " #$%-​<", "META", "_START", "><", "ꟲİ", "'ſ", "㍿EOT", "><,!", "me", "👍🏽", "A", "\n", ".d", "😀🏽", "Z", "'VE", "'VE", ">m", "'s", "Z"]} +{"text": "ſ 'Re-e\nDž\n", "tokens": 12, "pieces": ["ſ", " ", "'Re", "-e", "\n", "Dž", "\n", ""]} +{"text": "ḍ̇ſésA'é'lls'ſ­'M'll'sa😀🏽s", "tokens": 35, "pieces": ["ḋ", "̣ſe", "́sA", "'e", "́'", "lls", "'ſ", "­'", "M", "'ll", "'s", "a", "😀🏽", "s"]} +{"text": "'​\u000b'Mꟲs字👍🏽t\n𐞁\"İZ'ſ½\r ſ'T🙂㋿😀🏽٣٤٥٦<|fim_prefix|>é½ A​漢'VE9🙂é", "tokens": 78, "pieces": ["'​", "\u000b", "'M", "ꟲs字", "👍🏽", "t", "\n", "𐞁", "\"İZ", "'ſ", "½", "\r", " ſ", "'T", "🙂㋿😀🏽", "٣٤٥", "٦", "<|", "fim", "_prefix", "|><", "META", "_START", ">é", "½", " A", "​<", "META", "_START", ">漢", "'VE", "9", "🙂e", "́"]} +{"text": "
sé'M'VEsDžé.(‍\"'ſ", "tokens": 17, "pieces": ["
sé", "'M", "'VE", "sDžé", ".(‍\"'", "ſ"]} +{"text": "'T12345678-\nd!!mé,9'll 'D\" ḍ̇m'ſ'T…<|endoftext|>12345678A字#$%(fi12345678漢", "tokens": 49, "pieces": ["'T", "123", "456", "78", "-\n", "d", "!!", "mé", ",", "9", "'ll", " '", "D", "\"", " ḋ", "̣m", "'ſ", "'T", "…", "<|", "endoftext", "|>", "123", "456", "78", "A字", "#$%(", "fi", "123", "456", "78", "漢"]} +{"text": "<|endoftext|>Dž", "tokens": 9, "pieces": ["<|", "endoftext", "|>", "Dž"]} +{"text": "‍<|endoftext|>><|fim_prefix|>'ſ \nd's'llmⅣ'ſ'VE٣٤٥٦'VE\u000b'VE#$%dⅣꟲ字‍\r", "tokens": 58, "pieces": ["‍<|", "endoftext", "|>><", "EOT", "><|", "fim", "_prefix", "|>'", "ſ", " \n", "d", "'s", "'ll", "m", "Ⅳ", "'ſ", "'VE", "٣٤٥", "٦", "'VE", "\u000b", "'VE", "#$%", "d", "Ⅳ", "ꟲ字", "‍\r"]} +{"text": "'VEå9fi 'll-🙂!!
‍ééA\t\rfié'D", "tokens": 36, "pieces": ["'VE", "a", "̊", "9", "fi", " ", " '", "ll", "-🙂!!", "
", "‍e", "́e", "́A", "\t\r", "fie", "́<", "EOT", ">'", "D"]} +{"text": "\rDž#$%0\r\n㍿́12345678å😀🏽'ſ👍🏽", "tokens": 31, "pieces": ["\r", "Dž", "#$%", "0", "\r\n", "㍿́", "123", "456", "78", "a", "̊😀🏽'", "ſ", "👍🏽"]} +{"text": " ́<'Re912345678!'re<|fim_prefix|>0\"EOT​t#$%9Z12345678­'ll\"'re🙂😀🏽  0<'ſ", "tokens": 50, "pieces": [" ", " ́<'", "Re", "912", "345", "678", "!'", "re", "<|", "fim", "_prefix", "|><", "META", "_START", ">", "0", "\"EOT", "​t", "#$%", "9", "Z", "123", "456", "78", "­'", "ll", "\"'", "re", "🙂😀🏽", " ", " ", "0", "<'", "ſ"]} +{"text": "\r\n12345678#$%​<|endoftext|>>…EOTⅣ'Séå,́\tA\nßZ 😀🏽٣٤٥٦'M!#$%", "tokens": 51, "pieces": ["\r\n", "123", "456", "78", "#$%​<|", "endoftext", "|>>", "…EOT", "Ⅳ", "'S", "éa", "̊,́", "\tA", "\n", "ßZ", " ", "😀🏽", "٣٤٥", "٦", "'M", "!#$%"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'!e'ſ!漢'M're0\"İ\r\n\r\n\"😀🏽ꟲeåé>s", "tokens": 31, "pieces": ["'!", "e", "'ſ", "!漢", "'M", "'re", "0", "\"İ", "\r\n\r\n", "\"😀🏽", "ꟲea", "̊e", "́>", "s"]} +{"text": "Dž<|endoftext|>!!ſ …t0 å'ſ", "tokens": 24, "pieces": ["Dž", "<|", "endoftext", "|>!!", "ſ", " ", "…t", "0", " a", "̊'", "ſ"]} +{"text": "😀🏽\r's12345678Dž'漢​…(s'0\r\n\r\n…\r٣٤٥٦İ漢<|endoftext|>'Re\u000b ", "tokens": 46, "pieces": ["😀🏽\r", "'s", "123", "456", "78", "Dž", "'漢", "​", "…", "(s", "'", "0", "\r\n\r\n…\r", "٣٤٥", "٦", "İ漢", "<|", "endoftext", "|>'", "Re", "\u000b "]} +{"text": "'D!!m'Sé​ß \"‍Z'T'ssİ‍Dž<\t>Dž", "tokens": 25, "pieces": ["'D", "!!", "m", "'S", "é", "​ß", " ", " \"‍", "Z", "'T", "'s", "sİ", "‍Dž", "<", "\t", ">Dž"]} +{"text": "ꟲ(9 \n
m👍🏽0Ⅳ\r'T 𐞁!\r\n\r\n>­ſ \u000b>\t'd'S🙂Ⅳé<\"", "tokens": 44, "pieces": ["ꟲ", "(", "9", " \n", "
m", "👍🏽", "0Ⅳ", "\r", "'T", " ", " 𐞁", "!\r\n\r\n", ">­", "ſ", " ", "\u000b", ">", "\t", "'d", "'S", "🙂", "Ⅳ", "é", "<\""]} +{"text": "𐞁'S#$%字ſ\u000b", "tokens": 15, "pieces": ["𐞁", "'", "S", "#$%", "字ſ", "\u000b"]} +{"text": ".🙂ꟲ<|endoftext|>s𐞁३12345678!ß(½ét!é  t\r\n३.", "tokens": 39, "pieces": [".🙂", "ꟲ", "<|", "endoftext", "|>", "s𐞁", "३12", "345", "678", "!ß", "(", "½", "e", "́t", "!é", " ", " t", "\r\n", "३", "."]} +{"text": " ́s'VE.'S٣٤٥٦३٣٤٥٦-sſ'Ms…9're😀🏽t́Afi㋿‍'Re<½'M字0-'re!!EOT'Ⅳå", "tokens": 71, "pieces": [" ́", "s", "'VE", ".'", "S", "٣٤٥", "٦३٣", "٤٥٦", "-sſ", "'M", "s", "…", "9", "'re", "😀🏽", "t", "́Afi", "㋿‍'", "Re", "<", "½", "'M", "字", "0", "-'", "re", "!!", "EOT", "'", "Ⅳ", "a", "̊"]} +{"text": "é('D 9ḿ'ſ\r'S", "tokens": 12, "pieces": ["é", "('", "D", " ", "9", "m", "́'", "ſ", "\r", "'S"]} +{"text": " Zſ'\"", "tokens": 4, "pieces": [" Zſ", "'\""]} +{"text": "漢​'Re<|endoftext|>Ⅳfi!!", "tokens": 20, "pieces": ["漢", "​'", "Re", "<|", "endoftext", "|>", "Ⅳ", "fi", "!!"]} +{"text": "'M字㋿éİß!!㋿ḍ̇Asꟲfi'T's漢!(ḍ̇Ⅳ😀🏽0åå9é", "tokens": 55, "pieces": ["'M", "字", "㋿", "éİß", "!!㋿", "ḋ", "̣Asꟲfi", "'T", "'s", "漢", "!(", "ḋ", "̣", "Ⅳ", "😀🏽", "0", "a", "̊a", "̊", "9", "é"]} +{"text": "\te字ꟲ….ſ", "tokens": 10, "pieces": ["\te字ꟲ", "…", ".ſ"]} +{"text": "éd…­ ㋿٣٤٥٦​'t㋿İ.'Ts0𐞁
🙂'lltع !!fi👍🏽!!́ ", "tokens": 53, "pieces": ["e", "́d", "…", "­", " ㋿", "٣٤٥", "٦", "​'", "t", "㋿İ", ".'", "Ts", "0", "𐞁", "
", "🙂'", "lltع", " ", " !!", "fi", "👍🏽!!́", " "]} +{"text": "🙂'Re🙂'll​漢'D(\r\n\r\nİ!👍🏽's'M\n-😀🏽Z\r\r\n<<12345678İa\r\n\r\n", "tokens": 41, "pieces": ["🙂'", "Re", "🙂'", "ll", "​漢", "'D", "(\r\n\r\n", "İ", "!👍🏽'", "s", "'M", "\n", "-😀🏽", "Z", "\r\r\n", "<<", "123", "456", "78", "İa", "\r\n\r\n"]} +{"text": "'ll字é'M\r\n\r\n\n<|endoftext|>'D'🙂Džéİ<'reå'Reé<|endoftext|> 'T\r'D…
s12345678>< sḍ̇𐞁٣٤٥٦'ll<fi\r\n\r\n", "tokens": 77, "pieces": ["'ll", "字é", "'M", "\r\n\r\n\n", "<|", "endoftext", "|>'", "D", "'🙂", "Dže", "́İ", "<'", "rea", "̊'", "Reé", "<|", "endoftext", "|>", " ", "'T", "\r", "'", "D", "…", "
s", "123", "456", "78", "><", " ", " sḋ", "̣𐞁", "٣٤٥", "٦", "'ll", "<fi", "\r\n\r\n"]} +{"text": "
㍿'DA\r\n\r\n'!!>.'ſꟲ\neDž'Re\u000bع-🙂éé,m-\raİ ", "tokens": 40, "pieces": ["
", "㍿'", "DA", "\r\n\r\n", "'!!>.'", "ſꟲ", "\n", "eDž", "'Re", "\u000bع", "-🙂", "e", "́e", "́,", "m", "-\r", "aİ", " "]} +{"text": "é0'Re٣٤٥٦", "tokens": 12, "pieces": ["e", "́", "0", "'Re", "٣٤٥", "٦"]} +{"text": "d👍🏽 '\u000b…字'll٣٤٥٦ ſ'D", "tokens": 29, "pieces": ["d", "👍🏽", " ", " '", "\u000b", "…字", "'ll", "٣٤٥", "٦", "", " ſ", "'D"]} +{"text": "🙂­㋿́s​å​'redfiİ'D
\r\n\r\ne0'T,e…'re漢é!!", "tokens": 34, "pieces": ["🙂­㋿́", "s", "​a", "̊​'", "redfiİ", "'D", "
\r\n\r\n", "e", "0", "'T", ",e", "…", "'re", "漢e", "́!!"]} +{"text": "'0! \nⅣ9å㍿𐞁Z12345678é's…㍿!!('T
", "tokens": 35, "pieces": ["'", "0", "!", " \n", "Ⅳ9", "a", "̊㍿", "𐞁Z", "123", "456", "78", "e", "́'", "s", "…", "㍿!!('", "T", "
"]} +{"text": "m🙂'Ré", "tokens": 6, "pieces": ["m", "🙂'", "Re", "́"]} +{"text": "٣٤٥٦ع́ḍ̇,👍🏽(́\r\nm's­'VE
👍🏽\"\r\nſ\u000b sZ٣٤٥٦.'Te𐞁\r'Da,ḍ̇\u000bꟲ!!", "tokens": 72, "pieces": ["٣٤٥", "٦", "ع", "́ḋ", "̣,👍🏽(́\r\n", "m", "'s", "­'", "VE", "
", "👍🏽\"\r\n", "ſ", "\u000b", " sZ", "٣٤٥", "٦", ".'", "Te𐞁", "\r", "'D", "a", ",ḋ", "̣", "\u000bꟲ", "!!"]} +{"text": "12345678'MsA𐞁e٣٤٥٦🙂 ", "tokens": 27, "pieces": ["123", "456", "78", "'M", "sA𐞁e", "٣٤٥", "٦", "🙂<", "EOT", ">", " "]} +{"text": "fi­A字mⅣ'D𐞁'T👍🏽😀🏽're!#$%٣٤٥٦\nd", "tokens": 41, "pieces": ["fi", "­A字m", "Ⅳ", "'D", "𐞁", "'T", "👍🏽😀🏽'", "re", "!#$%", "٣٤٥", "٦", "\n", "d"]} +{"text": "<'ReA\r\ns's'ret㍿́𐞁.12345678-'S🙂>字\r\n\r\ns㍿\r\ne\r\na12345678👍🏽\r\n", "tokens": 63, "pieces": ["<'", "ReA", "\r\n", "s", "'", "s", "'re", "t", "㍿́", "𐞁", ".", "123", "456", "78", "-'", "S", "🙂>", "字", "\r\n\r\n", "s", "㍿\r\n", "e", "\r\n", "a", "123", "456", "78", "👍🏽\r\n"]} +{"text": "m'ReⅣ​㋿…'VEعßd'S'Re 漢12345678Dž \té \n's.\rfi \n.", "tokens": 39, "pieces": ["m", "'Re", "Ⅳ", "​㋿", "…", "'VE", "عßd", "'S", "'Re", " 漢", "123", "456", "78", "Dž", " ", "\té", " \n", "'s", ".\r", "fi", " \n", ".<", "EOT", ">"]} +{"text": "​>", "tokens": 2, "pieces": ["​>"]} +{"text": "!!t३#$% \n\r\n\r㍿#$%9ß'ḍ̇­mdé
́.", "tokens": 33, "pieces": ["!!", "t", "३", "#$%", " \n\r\n\r", "㍿#$%", "9", "ß", "'ḋ", "̣­", "mde", "́", "
", "́."]} +{"text": "m t 'reḍ̇​<|endoftext|>0<|endoftext|>👍🏽", "tokens": 36, "pieces": ["m", " ", " t", " ", "'re", "ḋ", "̣​<|", "endoftext", "|>", "0", "<|", "endoftext", "|>👍🏽<", "META", "_START", ">"]} +{"text": " s ,İꟲع''s'S字t漢-😀🏽12345678\r<|fim_prefix|>ꟲ's\ré's'Dfi<|endoftext|>ḍ̇'ſDž'll,字", "tokens": 71, "pieces": [" s", " ,", "İꟲع", "'<", "EOT", ">'", "s", "'S", "字t漢", "-<", "EOT", ">😀🏽", "123", "456", "78", "\r", "<|", "fim", "_prefix", "|>", "ꟲ", "'s", "\r", "e", "́'", "s", "'D", "fi", "<|", "endoftext", "|>", "ḋ", "̣'", "ſDž", "'ll", ",字"]} +{"text": "
m<|endoftext|><|endoftext|>㋿<|endoftext|>­\"(½\u000b\r'S\"(a'Sḍ̇é½'ſå😀🏽a­🙂­ \nſ'Re👍🏽,\u000bd 👍🏽", "tokens": 82, "pieces": ["
m", "<|", "endoftext", "|><|", "endoftext", "|>㋿<|", "endoftext", "|>­<", "EOT", ">\"(", "½", "\u000b\r", "'S", "\"(", "a", "'S", "ḋ", "̣é", "½", "'ſ", "a", "̊😀🏽", "a", "­🙂­", " \n", "ſ", "'Re", "👍🏽,", "\u000bd", " ", "👍🏽"]} +{"text": "d‍३", "tokens": 5, "pieces": ["d", "‍", "३"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\u000b,\r٣٤٥٦'ll ‍<-!­.\r", "tokens": 20, "pieces": ["\u000b", ",\r", "٣٤٥", "٦", "'ll", " ", "‍<-!­.\r"]} +{"text": "😀🏽'D​​字", "tokens": 9, "pieces": ["😀🏽'", "D", "​​", "字"]} +{"text": "'S­ \nⅣ>", "tokens": 6, "pieces": ["'S", "­", " \n", "Ⅳ", ">"]} +{"text": "́İ㋿ 字é!'Re😀🏽#$%字 ‍é''VE\tⅣd\rA ", "tokens": 32, "pieces": ["­m", "\r", "\"<('", "Re", " s", ">😀🏽#$%", "字", " ", " ‍", "e", "́''", "VE", "\t", "Ⅳ", "d", "\r", "A", " "]} +{"text": "'VE<|endoftext|>\r\n\r\ntaEOT<\u000b🙂​A-́𐞁-'ſ-👍🏽'́d​ſ​e'M<9\r\n-'s'Déet́9", "tokens": 58, "pieces": ["'VE", "<|", "endoftext", "|>\r\n\r\n", "taEOT", "<", "\u000b", "🙂​", "A", "-́<", "META", "_START", ">𐞁", "-'", "ſ", "-👍🏽'́", "d", "​ſ", "​e", "'M", "<", "9", "\r\n", "-'", "s", "'D", "éet", "́", "9"]} +{"text": "३\t 'VE12345678'S \nſDž\r\néd'M'S…s\n >t<|fim_prefix|>👍🏽<|fim_prefix|>́'VE1234567812345678<|endoftext|>ꟲ३", "tokens": 72, "pieces": ["३", "\t", " ", "'VE", "123", "456", "78", "'S", " \n", "ſDž", "\r\n", "éd", "'M", "'S", "…s", "\n", " ", ">", "t", "<|", "fim", "_prefix", "|>👍🏽<|", "fim", "_prefix", "|>́'", "VE", "123", "456", "781", "234", "567", "8", "<|", "endoftext", "|>", "ꟲ", "३"]} +{"text": "0'M ḍ̇🙂 ", "tokens": 11, "pieces": ["0", "'M", " ḋ", "̣🙂", " "]} +{"text": "-<|fim_prefix|>Dž́㋿-'TⅣ🙂३ \n漢're-e'VE'M(\"𐞁字", "tokens": 39, "pieces": ["-<|", "fim", "_prefix", "|>", "Dž", "́㋿-'", "T", "Ⅳ", "🙂", "३", " \n", "漢", "'re", "-e", "'VE", "'M", "(\"", "𐞁字"]} +{"text": ".㋿㍿", "tokens": 7, "pieces": [".㋿㍿"]} +{"text": ",0'll#$%ꟲ\t-… ३ A ſİ'S'Smfi\n
", "tokens": 30, "pieces": [",", "0", "'ll", "#$%", "ꟲ", "\t", "-", "…", " ", "३", " ", " A", " ſİ", "'S", "'S", "mfi", "\n
"]} +{"text": "ع́​  <字 \n'VÉ#$%\u000b('D å३½'TDžA-EOT ꟲßEOTA'S", "tokens": 40, "pieces": ["ع", "́​", " ", " ", "<字", " \n", "'VE", "́#$%", "\u000b", "('", "D", " a", "̊", "३½", "'T", "DžA", "-EOT", " ꟲßEOTA", "'S"]} +{"text": "ع‍'s'S٣٤٥٦Z字'VEİ<'reéé!!EOT12345678…½<'VE\tåaⅣ…éZ'S\n
A-​\r\n\r\n", "tokens": 57, "pieces": ["ع", "‍'", "s", "'S", "٣٤٥", "٦", "Z字", "'VE", "İ", "<'", "re", "ée", "́!!", "EOT", "123", "456", "78", "…", "½", "<'", "VE", "\ta", "̊a", "Ⅳ", "…e", "́Z", "'S", "\n", "
A", "-​\r\n\r\n"]} +{"text": "­ ½d\u000bedZ're-m字#$%ḍ̇'re
ḍ̇́漢(>(İ👍🏽<|endoftext|>𐞁<|fim_prefix|>㋿'T字'Re\r\nİ0字", "tokens": 67, "pieces": ["­", " ", "½", "d", "\u000bedZ", "'re", "-m字", "#$%", "ḋ", "̣'", "re", "
ḋ", "̣́", "漢", "(>(", "İ", "👍🏽<|", "endoftext", "|>", "𐞁", "<|", "fim", "_prefix", "|>㋿'", "T字", "'Re", "\r\n", "İ", "0", "字"]} +{"text": "0‍é٣٤٥٦<|endoftext|><'Re३ -\r\n٣٤٥٦́\r\te ,''T9㍿👍🏽字​<|fim_prefix|>", "tokens": 63, "pieces": ["0", "‍é", "٣٤٥", "٦", "<|", "endoftext", "|><", "META", "_START", "><'", "Re", "३", " ", " -\r\n", "٣٤٥", "٦", "́\r", "\te", " ", " ,''", "T", "9", "㍿👍🏽", "字", "​<|", "fim", "_prefix", "|>"]} +{"text": "<|endoftext|><|fim_prefix|>'S🙂a!३\"'T字\ré0,123456780 \t­''VE㋿<|endoftext|> é", "tokens": 48, "pieces": ["<|", "endoftext", "|><|", "fim", "_prefix", "|>'", "S", "🙂a", "!", "३", "\"'", "T字", "\r", "e", "́", "0", ",", "123", "456", "780", " ", "\t", "­''", "VE", "㋿<|", "endoftext", "|>", " e", "́"]} +{"text": "!'M\u000b<|endoftext|>å\u000b", "tokens": 14, "pieces": ["!'", "M", "\u000b", "<|", "endoftext", "|>", "a", "̊", "\u000b"]} +{"text": ",'ll\nd漢ſ", "tokens": 8, "pieces": [",'", "ll", "\n", "d漢ſ"]} +{"text": "eEOT'Tå\r'sḍ̇'ß\"EOT\n㋿", "tokens": 25, "pieces": ["eEOT", "'T", "a", "̊\r", "'s", "ḋ", "̣'", "ß", "\"<", "META", "_START", ">EOT", "\n", "㋿"]} +{"text": "'sfißḍ̇Aé\n\u000bAé'Mm<|fim_prefix|>å<|fim_prefix|>٣٤٥٦", "tokens": 46, "pieces": ["'s", "fißḋ", "̣Aé", "\n", "\u000bAe", "́'", "Mm", "<|", "fim", "_prefix", "|>", "a", "̊<|", "fim", "_prefix", "|>", "٣٤٥", "٦"]} +{"text": "'M漢#$%ß..'\r\n0\u000b(́\"\r\nééfiⅣDž \n\u000bAḍ̇As", "tokens": 37, "pieces": ["'M", "漢", "#$%", "ß", "..'\r\n", "0", "\u000b", "(́\"\r\n", "ée", "́fi", "Ⅳ", "Dž", "", " \n", "\u000bAḋ", "̣As"]} +{"text": "é \nſ", "tokens": 5, "pieces": ["e", "́", " \n", "ſ"]} +{"text": "t Z㋿,12345678. EOT'VE
३­ Ⅳ㋿<|endoftext|>ßEOT!EOT'S", "tokens": 42, "pieces": ["t", " Z", "㋿,", "123", "456", "78", ".", " EOT", "'VE", "
", "३", "­", " ", " ", "Ⅳ", "㋿<|", "endoftext", "|>", "ßEOT", "!EOT", "'S"]} +{"text": "٣٤٥٦<|endoftext|>😀🏽३😀🏽'DA'reß 'Re. \n 9éå're'ſ<'reع!! <‍me👍🏽'llm", "tokens": 69, "pieces": ["٣٤٥", "٦", "<|", "endoftext", "|>😀🏽", "३", "😀🏽'", "DA", "'re", "ß", " ", "'Re", ".<", "EOT", ">", " \n", " ", "9", "éa", "̊'", "re", "'ſ", "<'", "reع", "!!", " ", "<‍", "me", "👍🏽'", "llm"]} +{"text": "'M Zſtaé9́ Ⅳ'S٣٤٥٦\r\n\r\n'ſ're'Re…", "tokens": 31, "pieces": ["'M", " Zſtaé", "9", "́", " ", "Ⅳ", "'S", "٣٤٥", "٦", "\r\n\r\n", "'ſ", "'re", "'Re", "…"]} +{"text": "!!漢EOT字<漢", "tokens": 11, "pieces": ["!!", "漢", "EOT字", "<漢"]} +{"text": "\n \n🙂​>‍'s字㍿s
're½EOT'Re'reé. \n", "tokens": 26, "pieces": ["\n \n", "🙂​>‍'", "s字", "㍿s", "
", "'re", "½", "EOT", "'Re", "'re", "e", "́.", " \n"]} +{"text": "\r\n\r\n'D́é-'lld!𐞁\u000bm㋿ع'D", "tokens": 21, "pieces": ["\r\n\r\n", "'D", "́é", "-'", "lld", "!<", "META", "_START", ">𐞁", "\u000bm", "㋿ع", "'D"]} +{"text": "​…३(​9t ", "tokens": 10, "pieces": ["​", "…", "३", "(​", "9", "t", " "]} +{"text": "🙂\r\nd0'VE🙂'M字
㋿!!'!!'D🙂", "tokens": 27, "pieces": ["🙂\r\n", "d", "0", "'VE", "🙂'", "M字", "
", "㋿!!'<", "EOT", ">!!'", "D", "🙂"]} +{"text": "-.m' \r\nt<|fim_prefix|>👍🏽👍🏽a😀🏽½!!'VEDž\r ", "tokens": 39, "pieces": ["-.", "m", "'", " \r\n", "t", "<|", "fim", "_prefix", "|>👍🏽👍🏽", "a", "😀🏽", "½", "!!'", "VEDž", "\r "]} +{"text": " \n<'M#$%\n\r\nİéA🙂 漢å\" ع", "tokens": 25, "pieces": [" \n", "<'", "M", "#$%\n\r\n", "İe", "́A", "🙂", " 漢a", "̊\"", " ع"]} +{"text": "9'Dİ👍🏽㍿🙂m 
e fi<|fim_prefix|>㋿t'M'll \n!Ⅳ'VEİ-'reé½ !fi9\t\n \n<|endoftext|>", "tokens": 60, "pieces": ["9", "'D", "İ", "👍🏽㍿🙂", "m", " ", "
e", " ", " fi", "<|", "fim", "_prefix", "|>㋿", "t", "'M", "'ll", " \n", "!", "Ⅳ", "'VE", "İ", "-'", "reé", "½", " ", " !", "fi", "9", "\t\n \n", "<|", "endoftext", "|>"]} +{"text": "́…́\"Dž ", "tokens": 11, "pieces": ["́", "…", "́\"", "Dž", " "]} +{"text": "\r\n\r\nds,㋿'VE", "tokens": 8, "pieces": ["\r\n\r\n", "ds", ",㋿'", "VE"]} +{"text": "-a<|endoftext|>'s12345678!!s . ", "tokens": 21, "pieces": ["-a", "<|", "endoftext", "|>'", "s", "123", "456", "78", "!!", "s", "", " ", ".", " "]} +{"text": "12345678‍'s(t's 𐞁éعꟲ٣٤٥٦>0\r\nİ12345678!!३\",,fie'rea12345678Džꟲé' .EOT", "tokens": 67, "pieces": ["123", "456", "78", "‍'", "s", "(t", "'s", " 𐞁e", "́ع", "ꟲ", "٣٤٥", "٦", ">", "0", "\r\n", "İ", "123", "456", "78", "!!", "३", "\",<", "META", "_START", ">,", "fie", "'", "rea", "123", "456", "78", "Džꟲé", "'", " .", "EOT"]} +{"text": "a'llEOT-ع𐞁're٣٤٥٦㋿dé<|fim_prefix|>at,é字#$%-0👍🏽😀🏽.½ḍ̇<|fim_prefix|>ß<𐞁
㍿ (", "tokens": 76, "pieces": ["a", "'ll", "EOT", "-ع𐞁", "'re", "٣٤٥", "٦", "㋿dé", "<|", "fim", "_prefix", "|>", "at", ",e", "́字", "#$%-", "0", "👍🏽😀🏽.", "½", "ḋ", "̣<|", "fim", "_prefix", "|>", "ß", "<𐞁", "
", "㍿", " ", "("]} +{"text": "a-ßt𐞁e<|endoftext|>éfi字're
\tß 'VE\r\n\r\n \n㋿'TfiEOTdm𐞁\rß́'M", "tokens": 55, "pieces": ["a", "-ßt𐞁e", "<|", "endoftext", "|>", "e", "́fi字", "'", "re", "
", "\tß", " ", "'VE", "\r\n\r\n \n", "㋿'", "TfiEOTdm𐞁", "\r", "ß", "́'", "M"]} +{"text": "ꟲ!!٣٤٥٦\r\ne's't\n字 EOTmⅣ ß漢0dZ.ßsꟲ \" 'refié", "tokens": 48, "pieces": ["ꟲ", "!!", "٣٤٥", "٦", "\r\n", "e", "'s", "'t", "\n", "字", " EOTm", "Ⅳ", " ß漢", "0", "dZ", ".ßsꟲ", " ", "\"", " ", " '", "re", "fié"]} +{"text": "-😀🏽12345678å.0fiEOT'Tع!㋿ Z>字 'Tꟲ#$%\r\nt'VEع
\u000bßḍ̇'S12345678'S \n\t\u000b٣٤٥٦", "tokens": 69, "pieces": ["-😀🏽", "123", "456", "78", "a", "̊.", "0", "fiEOT", "'T", "ع", "!㋿", " Z", ">字", " ", " '", "Tꟲ", "#$%\r\n", "t", "'VE", "ع", "
", "\u000bßḋ", "̣<", "META", "_START", ">'", "S", "123", "456", "78", "'S", " \n", "\t", "\u000b", "٣٤٥", "٦"]} +{"text": "\nİ\"\n'T.e9'reé'ſZé👍🏽,12345678  👍🏽'll漢 Ⅳ'VEta's'VE \n's'Tعꟲ", "tokens": 52, "pieces": ["\n", "İ", "\"\n", "'T", ".e", "9", "'re", "e", "́'", "ſZe", "́👍🏽,", "123", "456", "78", " ", " ", "👍🏽'", "ll漢", " ", "Ⅳ", "'VE", "ta", "'s", "'VE", " \n", "'s", "'T", "عꟲ"]} +{"text": "'VE'sDž½fi㍿漢ſ<\rDž\r\n<\u000b\"\u000b漢Aḍ̇­ad🙂", "tokens": 37, "pieces": ["'VE", "'s", "Dž", "½", "fi", "㍿漢ſ", "<\r", "Dž", "\r\n", "<", "\u000b", "\"", "\u000b漢Aḋ", "̣­", "ad", "🙂"]} +{"text": "\u000b!\ra漢 EOT.‍!\u000b​​a👍🏽 \r\n'́", "tokens": 28, "pieces": ["\u000b", "!\r", "a漢", " EOT", ".‍!", "\u000b", "​​", "a", "👍🏽", " \r\n", "'́"]} +{"text": " Z!!½-,#$%fi'MDž 漢ꟲ>'reꟲ0t'reéḍ̇12345678(12345678ſ३#$%'ſdAet12345678t३", "tokens": 64, "pieces": [" Z", "!!", "½", "-,#$%", "fi", "'M", "Dž", " ", " 漢", "ꟲ", ">'", "reꟲ", "0", "t", "'re", "éḋ", "̣", "123", "456", "78", "(", "123", "456", "78", "ſ", "३", "#$%'", "ſdA", "et", "123", "456", "78", "t", "३"]} +{"text": "㋿\t\r\n12345678é\u000b!!'ſ'MA,\r\n\r\n\"'ll'EOT٣٤٥٦'MZ", "tokens": 34, "pieces": ["㋿", "\t\r\n", "123", "456", "78", "é", "\u000b", "!!'", "ſ", "'M", "A", ",\r\n\r\n", "\"<", "META", "_START", ">'", "ll", "'EOT", "٣٤٥", "٦", "'M", "Z"]} +{"text": "'Re
‍\t\r\n\r\nİ ㍿d \nmİ#$%ḍ̇字#$%eZ(12345678\r\n\r\n t \n'ReA‍\r\n's#$%ⅣA<|endoftext|>'VE'VE", "tokens": 62, "pieces": ["'Re", "
", "‍", "\t\r\n\r\n", "İ", " ", " ㍿", "d", " \n", "mİ", "#$%", "ḋ", "̣字", "#$%", "eZ", "(", "123", "456", "78", "\r\n\r\n", " ", " t", " \n", "'Re", "A", "‍\r\n", "'s", "#$%", "Ⅳ", "A", "<|", "endoftext", "|>'", "VE", "'VE"]} +{"text": "'lld㍿
'red٣٤٥٦👍🏽aDž'reꟲ<<|fim_prefix|>\r\n'Re'VEa\"<|fim_prefix|>é''🙂EOT'0're'ſ'T", "tokens": 64, "pieces": ["'ll", "d", "㍿", "
", "'re", "d", "٣٤٥", "٦", "👍🏽", "aDž", "'re", "ꟲ", "<<", "EOT", "><|", "fim", "_prefix", "|>\r\n", "'Re", "'VE", "a", "\"<|", "fim", "_prefix", "|>", "é", "''🙂", "EOT", "'", "0", "'re", "'ſ", "'T"]} +{"text": " EOT 漢\"<\"٣٤٥٦ꟲ12345678!\u000b👍🏽'M!A'VE", "tokens": 37, "pieces": [" ", " EOT", " 漢", "\"<\"", "٣٤٥", "٦", "ꟲ", "123", "456", "78", "!", "\u000b", "👍🏽'", "M", "!A", "'VE"]} +{"text": "<|fim_prefix|> 'reſZ‍🙂\r\n\r\nİfi\r\n'S👍🏽ſİ…'S ", "tokens": 38, "pieces": ["<|", "fim", "_prefix", "|>", " ", "'re", "ſZ", "‍🙂\r\n\r\n", "İfi", "\r\n", "'S", "👍🏽", "ſİ", "…", "'", "S", " "]} +{"text": "de३ ", "tokens": 8, "pieces": ["de", "", "३", " "]} +{"text": "\n𐞁㍿Dž ㋿a'll!!ſ字…Ⅳ'sé12345678", "tokens": 35, "pieces": ["\n", "𐞁", "㍿Dž", " ", " ㋿", "a", "'ll", "!!", "ſ字", "…", "Ⅳ", "'s", "e", "́<", "EOT", ">", "123", "456", "78"]} +{"text": "<|fim_prefix|>e😀🏽tZ​\t#$%عå३ ㍿ß😀🏽İ9", "tokens": 41, "pieces": ["<|", "fim", "_prefix", "|>", "e", "😀🏽", "tZ", "​", "\t", "#$%", "عa", "̊", "३", " ", "㍿<", "META", "_START", ">ß", "😀🏽", "İ", "9"]} +{"text": "#$%!!½ ٣٤٥٦'D'VEa
\t ٣٤٥٦('VE! \"​.", "tokens": 37, "pieces": ["#$%!!", "½", " ", " ", "٣٤٥", "٦", "'D", "'VE", "a", "
\t", " ", "٣٤٥", "٦", "('", "VE", "!", " ", "\"​."]} +{"text": "9<'re'M'T😀🏽('ſ'M.'VE\r\n\r\n👍🏽éa\u000bⅣعé<|fim_prefix|>>'D\tZDž  ", "tokens": 49, "pieces": ["9", "<'", "re", "'M", "'T", "😀🏽('", "ſ", "'M", ".'", "VE", "\r\n\r\n", "👍🏽", "e", "́a", "", "\u000b", "Ⅳ", "عe", "́<|", "fim", "_prefix", "|>>'", "D", "\tZDž", "  "]} +{"text": "'‍ſ#$%…३\r\n\r\ne𐞁'T!٣٤٥٦-sé é​'s𐞁d\r\n\r\nſå<|endoftext|>字 \n👍🏽DžⅣ½\",Z👍🏽", "tokens": 78, "pieces": ["'‍", "ſ", "#$%", "…", "३", "\r\n\r\n", "e𐞁", "'T", "!", "٣٤٥", "٦", "-se", "́", " e", "́<", "META", "_START", ">​'", "s𐞁d", "\r\n\r\n", "ſa", "̊<|", "endoftext", "|>", "字", " \n", "👍🏽", "Dž", "Ⅳ½", "\",", "Z", "👍🏽"]} +{"text": "9Z…🙂३ß \n(<|endoftext|>\t9'D٣٤٥٦EOT ꟲ.é
å'll漢İA", "tokens": 51, "pieces": ["9", "Z", "…", "🙂", "३", "ß", " \n", "(<|", "endoftext", "|>", "\t", "", "9", "'D", "٣٤٥", "٦", "EOT", " ", " ꟲ", ".e", "́", "
a", "̊'", "ll漢İA"]} +{"text": "
-\u000b'Tß'M 𐞁😀🏽\r'll३'Ⅳ…9Zm9Ae'Tİ漢<éßå", "tokens": 50, "pieces": ["
", "-<", "EOT", ">", "\u000b", "'T", "ß", "'M", " 𐞁", "😀🏽\r", "'ll", "३", "'", "Ⅳ", "…", "9", "Zm", "9", "Ae", "'T", "İ漢", "<éßa", "̊"]} +{"text": "\r\n'é\r\nع㋿'reEOT a'll \r0'D\u000bİ#$%🙂\r\n>㍿>å\r's😀🏽́ 'S A ", "tokens": 51, "pieces": ["\r\n", "'e", "́\r\n", "ع", "㋿'", "reEOT", " a", "'ll", " \r", "0", "'D", "\u000bİ", "#$%🙂\r\n", ">㍿>", "a", "̊\r", "'s", "😀🏽<", "META", "_START", ">́", " ", "'S", " A", " "]} +{"text": ".A<|fim_prefix|>'ſfiDž­'eée٣٤٥٦'VE<|endoftext|>👍🏽mİ!!
", "tokens": 51, "pieces": [".A", "<|", "fim", "_prefix", "|>'", "ſfiDž", "­'", "e", "ée", "٣٤٥", "٦", "'VE", "<|", "endoftext", "|>👍🏽", "mİ", "!!", "
"]} +{"text": "!漢‍<|fim_prefix|>A'D'ſ 𐞁- .  a", "tokens": 29, "pieces": ["!漢", "‍<|", "fim", "_prefix", "|>", "A", "'D", "'ſ", " ", " 𐞁", "-", " .", " ", " a"]} +{"text": "İé<|fim_prefix|>'re'T½ꟲ!!t Z ​ꟲEOT\r\n(fi's#$%sDž́ ㋿…'T!m\u000b", "tokens": 50, "pieces": ["İe", "́<|", "fim", "_prefix", "|>'", "re", "'T", "½", "ꟲ", "!!", "t", " Z", " ", "​ꟲEOT", "\r\n", "(fi", "'s", "#$%", "sDž", "́", " ", "㋿", "…", "'T", "!m", "\u000b"]} +{"text": "'re漢<|endoftext|>½ (😀🏽a#$%0<|fim_prefix|>å𐞁<|endoftext|>'Tſ\r\n\r\n‍Ⅳ\né<٣٤٥٦('ſ'T 9​漢३é,\u000b12345678İ\u000b", "tokens": 85, "pieces": ["'re", "漢", "<|", "endoftext", "|>", "½", " ", "(😀🏽", "a", "#$%", "0", "<|", "fim", "_prefix", "|>", "a", "̊𐞁", "<|", "endoftext", "|>'", "Tſ", "\r\n\r\n", "‍", "Ⅳ", "\n", "é", "<", "٣٤٥", "٦", "('", "ſ", "'T", " ", " ", "9", "​漢", "३", "é", ",", "\u000b", "123", "456", "78", "İ", "", "\u000b"]} +{"text": "#$%
(㍿Z!!s👍🏽\"EOT<\t𐞁,'T!'DⅣ'Re👍🏽३ḍ̇", "tokens": 46, "pieces": ["#$%", "
", "(㍿", "Z", "!!", "s", "👍🏽\"", "EOT", "<", "\t𐞁", ",'", "T", "!'", "D", "Ⅳ", "'Re", "👍🏽", "३", "ḋ", "̣"]} +{"text": "mm 're𐞁då'ReⅣ🙂  >\r\n'ſ", "tokens": 23, "pieces": ["mm", " ", " '", "re𐞁da", "̊'", "Re", "Ⅳ", "🙂", " ", " ", ">\r\n", "'ſ"]} +{"text": "٣٤٥٦'ll\r\nZ字9½́#$%!t", "tokens": 18, "pieces": ["٣٤٥", "٦", "'ll", "\r\n", "Z字", "9½", "́#$%!", "t"]} +{"text": "fi'Dt.'llḍ̇'så.​\u000b'D㍿ḍ̇'ll'T >Ⅳ \n#$%'D!'ll\r", "tokens": 44, "pieces": ["fi", "'D", "t", ".<", "META", "_START", ">'", "llḋ", "̣'", "sa", "̊.​", "\u000b", "'D", "㍿ḋ", "̣'", "ll", "'T", " ", ">", "Ⅳ", " \n", "#$%'", "D", "!'", "ll", "\r"]} +{"text": "'Reİfi\t'VE12345678字's'Mꟲſé(>'S ٣٤٥٦,😀🏽ſa\r\nmDža…ⅣA'\r\n", "tokens": 53, "pieces": ["'Re", "İfi", "\t", "'VE", "123", "456", "78", "字", "'s", "'M", "ꟲſe", "́(>'", "S", " ", "٣٤٥", "٦", ",😀🏽", "ſa", "\r\n", "mDža", "…", "Ⅳ", "A", "'\r\n"]} +{"text": "!!EOT½​\r\n\r\n\t½'ſ'll's½
\"d½'\r\n\r\n٣٤٥٦t½ ḍ̇(<|fim_prefix|>.ꟲm'D​sm!😀🏽9 ", "tokens": 57, "pieces": ["!!", "EOT", "½", "​\r\n\r\n", "\t", "½", "'ſ", "'ll", "'s", "½", "
", "\"d", "½", "'\r\n\r\n", "٣٤٥", "٦", "t", "½", " ḋ", "̣(<|", "fim", "_prefix", "|>.", "ꟲm", "'D", "​sm", "!😀🏽", "9", " "]} +{"text": "🙂fiع#$%\ŕ", "tokens": 9, "pieces": ["🙂fiع", "#$%\r", "́"]} +{"text": "EOTİ\n#$%٣٤٥٦fi9́'S'VEZa12345678字s0>'re'Re<<|fim_prefix|>́", "tokens": 47, "pieces": ["EOTİ", "\n", "#$%<", "EOT", ">", "٣٤٥", "٦", "fi", "9", "́'", "S", "'VE", "Za", "123", "456", "78", "字s", "0", ">'", "re", "'Re", "<<|", "fim", "_prefix", "|>́"]} +{"text": "…漢'ſeZꟲ'VE­'Sa 0\u000b\r", "tokens": 21, "pieces": ["…漢", "'ſ", "eZꟲ", "'VE", "­'", "Sa", " ", "0", "\u000b\r"]} +{"text": "­ \nm!", "tokens": 4, "pieces": ["­", " \n", "m", "!"]} +{"text": "'DAm½é,'ſ,-e'D😀🏽'S,'T0é<
 ㍿e㍿'D'Re٣٤٥٦#$%\"𐞁'DAa", "tokens": 60, "pieces": ["'D", "Am", "½", "é", ",'", "ſ", ",-", "e", "'D", "😀🏽'", "S", ",'", "T", "0", "e", "́<", "
", " ", "㍿e", "㍿'", "D", "'Re", "٣٤٥", "٦", "#$%\"", "𐞁", "'D", "Aa"]} +{"text": "\r𐞁<|endoftext|>9dd!'T\r\n\r\n'sDž#$%9ß́\u000b<|fim_prefix|>\u000bḍ̇‍ ", "tokens": 42, "pieces": ["\r", "𐞁", "<|", "endoftext", "|>", "9", "dd", "!'", "T", "\r\n\r\n", "'s", "Dž", "#$%", "9", "ß", "́", "\u000b", "<|", "fim", "_prefix", "|>", "\u000bḋ", "̣‍", " "]} +{"text": "-𐞁!!ḍ̇‍. 0ꟲ '字a‍fiEOT\"ſ \ns'DZ're ٣٤٥٦'s", "tokens": 47, "pieces": ["-𐞁", "!!", "ḋ", "̣‍.", " ", "0", "ꟲ", " ", "'字a", "‍fiEOT", "\"ſ", " \n", "s", "'D", "Z", "'re", " ", "٣٤٥", "٦", "'s"]} +{"text": "'VE!!!😀🏽\tEOT'M-t>
\re0\"t\r\n\r\n'SEOT́
Dž字‍(", "tokens": 36, "pieces": ["'VE", "!!!😀🏽", "\tEOT", "'M", "-t", ">", "
\r", "e", "0", "\"t", "\r\n\r\n", "'S", "EOT", "́", "
Dž字", "‍<", "EOT", ">("]} +{"text": "EOT'ſ​'M", "tokens": 8, "pieces": ["EOT", "'ſ", "​'", "M"]} +{"text": "𐞁👍🏽👍🏽Aß0½'MDže!!
EOT'llع<|fim_prefix|>é", "tokens": 41, "pieces": ["𐞁", "👍🏽👍🏽", "Aß", "0½", "'M", "Dže", "!!", "
EOT", "'ll", "ع", "<|", "fim", "_prefix", "|>", "e", "́"]} +{"text": "#$%ع'TEOTs'S#$%ßd­é​ 'Re-#$%́", "tokens": 30, "pieces": ["#$%", "ع", "'T", "EOTs", "'S", "#$%", "ßd", "­e", "́​", " ", " <", "META", "_START", ">'", "Re", "-#$%́<", "EOT", ">"]} +{"text": "ꟲ字​\r'll!!", "tokens": 8, "pieces": ["ꟲ字", "​\r", "'ll", "!!"]} +{"text": "……a㍿m
𐞁𐞁,0fiſ‍d .>👍🏽 -é,\u000b́Dž字👍🏽㍿\n\u000b", "tokens": 60, "pieces": ["…", "…a", "㍿m", "
𐞁𐞁", ",", "0", "fiſ", "‍d", " ", " .>👍🏽", " ", "-", "é", ",", "\u000b", "́Dž字", "👍🏽㍿\n", "\u000b"]} +{"text": "d½Džtaع​s#$%t'reع'VE👍🏽're", "tokens": 26, "pieces": ["d", "½", "Džt", "aع", "​s", "#$%", "t", "'re", "ع", "'VE", "👍🏽'", "re"]} +{"text": "ꟲaſ \n'D-", "tokens": 9, "pieces": ["ꟲaſ", " \n", "'D", "-"]} +{"text": "😀🏽(३'ſZ\r\nſ  t \n<|endoftext|> Ⅳꟲḍ̇İ12345678e<|fim_prefix|>٣٤٥٦\n'S t's İd👍🏽d", "tokens": 75, "pieces": ["😀🏽(", "३", "'ſ", "Z", "\r\n", "ſ", " ", " t", " \n", "<|", "endoftext", "|>", " ", " ", "Ⅳ", "ꟲḋ", "̣İ", "123", "456", "78", "e", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "\n", "'S", " t", "'s", " İd", "👍🏽", "d"]} +{"text": ",…'llḍ̇s\"<\"३a(#$%\"s👍🏽字é'D!!
", "tokens": 33, "pieces": [",", "…", "'ll", "ḋ", "̣s", "\"<<", "META", "_START", ">\"", "३", "a", "(#$%\"", "s", "👍🏽", "字é", "'D", "!!", "
"]} +{"text": "👍🏽 ㍿½'VE#$%­'Re३ ſ'D­t'M३👍🏽A<,'lla<|fim_prefix|>ADž🙂sd", "tokens": 53, "pieces": ["👍🏽", " ", "㍿", "½", "'VE", "#$%­'", "Re", "३", " ſ", "'D", "­t", "'M", "३", "👍🏽", "A", "<,'", "lla", "<|", "fim", "_prefix", "|>", "ADž", "🙂sd"]} +{"text": "A𐞁\r\n\r\n#$%0㋿EOT-عſ'MmⅣd#$%'D12345678d… ", "tokens": 34, "pieces": ["A𐞁", "\r\n\r\n", "#$%", "0", "㋿EOT", "-عſ", "'M", "m", "Ⅳ", "d", "#$%'", "D", "123", "456", "78", "d", "… "]} +{"text": "<|endoftext|>…'㋿<|endoftext|>t> fim's㋿(…e👍🏽<|fim_prefix|>t ,ꟲ", "tokens": 54, "pieces": ["<|", "endoftext", "|>", "…", "'㋿<|", "endoftext", "|>", "t", ">", " ", " fim", "'s", "㋿(", "…e", "👍🏽<|", "fim", "_prefix", "|>", "t", " ,", "ꟲ"]} +{"text": "
字́ſ'M'D<|endoftext|> (́
­,9#$%\"'s'Tſḍ̇>9‍d \n\"३fi!!å'S!td<|fim_prefix|>", "tokens": 60, "pieces": ["
字", "́ſ", "'M", "'D", "<|", "endoftext", "|>", " (́", "
", "­,", "9", "#$%\"'", "s", "'T", "ſḋ", "̣>", "9", "‍d", " \n", "\"", "३", "fi", "!!", "a", "̊'", "S", "!td", "<|", "fim", "_prefix", "|>"]} +{"text": "<|endoftext|>12345678sfi(a d", "tokens": 16, "pieces": ["<|", "endoftext", "|>", "123", "456", "78", "sfi", "(a", " d"]} +{"text": "Džß'll😀🏽EOT٣٤٥٦ſZ0<|fim_prefix|>-", "tokens": 34, "pieces": ["Džß", "'ll", "😀🏽", "EOT", "٣٤٥", "٦", "ſZ", "0", "<|", "fim", "_prefix", "|>-"]} +{"text": "字'VE'T'M'll(\t<'ſſ ‍!!!!d#$%!!", "tokens": 24, "pieces": ["字", "'VE", "'T", "'M", "'ll", "(", "\t", "<'", "ſſ", " ", " ‍!!!!", "d", "#$%!!"]} +{"text": "<|fim_prefix|>!!…́'㍿a­\"㋿\r\n -<|endoftext|>>eḍ̇🙂字'Re->
.s", "tokens": 44, "pieces": ["<|", "fim", "_prefix", "|>!!", "…", "́'㍿", "a", "­\"㋿\r\n", " -<|", "endoftext", "|>>", "eḋ", "̣🙂", "字", "'Re", "->", "
", ".s"]} +{"text": "(9漢'S're\r\ń's'ſ\r\n\r\n
('Dd<​12345678'D٣٤٥٦½#$%t­
ſ'S\r\nꟲ", "tokens": 51, "pieces": ["(", "9", "漢", "'S", "'re", "\r\n", "́'", "s", "'ſ", "\r\n\r\n", "
", "('", "Dd", "<​", "123", "456", "78", "'D", "٣٤٥", "٦½", "#$%", "t", "­", "
ſ", "'S", "\r\n", "ꟲ"]} +{"text": "🙂m(('s.½  ſ.'Tfiåa‍-9\t\"#$%#$%​ꟲ-'Re \n'Tꟲ\n㍿字-‍#$%", "tokens": 50, "pieces": ["🙂m", "(('", "s", ".", "½", " ", " ſ", ".'", "Tfia", "̊a", "‍-", "9", "\t", "\"#$%#$%​", "ꟲ", "-'", "Re", " \n", "'T", "ꟲ", "\n", "㍿字", "-‍#$%"]} +{"text": "(\r<|fim_prefix|>½'re'll​aß", "tokens": 15, "pieces": ["(\r", "<|", "fim", "_prefix", "|>", "½", "'re", "'ll", "​aß"]} +{"text": "\r\n٣٤٥٦d,\r\n\r\n", "tokens": 11, "pieces": ["\r\n", "٣٤٥", "٦", "d", ",\r\n\r\n"]} +{"text": "é\t­ \n0(12345678👍🏽٣٤٥٦ ḍ̇", "tokens": 29, "pieces": ["é", "\t", "­", " \n", "0", "(", "123", "456", "78", "👍🏽", "٣٤٥", "٦", " ḋ", "̣"]} +{"text": "㍿ßém'Tå't🙂<|endoftext|>! 'M३.09Dž'ReZ!!👍🏽\u000bfid(e​漢 
'ſ12345678éA'T", "tokens": 59, "pieces": ["㍿ßém", "'T", "a", "̊'", "t", "🙂<|", "endoftext", "|>!", " ", "'M", "३", ".", "09", "Dž", "'Re", "Z", "!!👍🏽", "\u000bfid", "(e", "​漢", " ", "
", "'ſ", "123", "456", "78", "éA", "'T"]} +{"text": "(½İ<|fim_prefix|>!!a' ", "tokens": 14, "pieces": ["(", "½", "İ", "<|", "fim", "_prefix", "|>!!", "a", "'", " "]} +{"text": "\r\n½ſ㋿ ㋿👍🏽dſ½㋿", "tokens": 24, "pieces": ["\r\n", "½", "ſ", "㋿", " ", "㋿👍🏽", "dſ", "½", "㋿"]} +{"text": "å½EOT's漢EOT‍ Dž' \n9é \n😀🏽é\rع>ḍ̇'VEß'Dmsİḍ̇", "tokens": 51, "pieces": ["a", "̊", "½", "EOT", "'s", "漢EOT", "‍", " Dž", "'", " \n", "9", "e", "́", " \n", "😀🏽", "é", "\r", "ع", ">ḋ", "̣'", "VEß", "'", "Dmsİḋ", "̣"]} +{"text": "Ⅳ12345678ḍ̇字,㍿0عs\u000b0fi'Ré.eßADž​漢ſ", "tokens": 36, "pieces": ["Ⅳ12", "345", "678", "ḋ", "̣字", ",㍿", "0", "عs", "\u000b", "0", "fi", "'Re", "́.", "eßADž", "​漢ſ"]} +{"text": "é'Maa'lls😀🏽
,Dž \nع,", "tokens": 20, "pieces": ["e", "́'", "Maa", "'ll", "s", "😀🏽", "
", ",Dž", " \n", "ع", ","]} +{"text": "'sAé… \ne'T½12345678s", "tokens": 15, "pieces": ["'s", "Ae", "́", "… \n", "e", "'T", "½12", "345", "678", "s"]} +{"text": "12345678<|fim_prefix|>😀🏽t字é'ع㋿👍🏽é\t fid'VE\r\" s\r\nß", "tokens": 45, "pieces": ["123", "456", "78", "<|", "fim", "_prefix", "|>😀🏽", "t字e", "́'", "ع", "㋿👍🏽", "e", "́", "\t", " fid", "'VE", "\r", "\"", " ", " s", "\r\n", "ß"]} +{"text": "ꟲDž,'VE", "tokens": 7, "pieces": ["ꟲDž", ",'", "VE"]} +{"text": "Dž<|endoftext|>-'𐞁'll
𐞁å\u000b<😀🏽#$%\t A\u000bé'S'Ree12345678𐞁‍éA #$%ß'VE", "tokens": 65, "pieces": ["Dž", "<|", "endoftext", "|>-'", "𐞁", "'ll", "
𐞁a", "̊", "\u000b", "<😀🏽#$%", "\t", " A", "\u000be", "́'", "S", "'Re", "e", "123", "456", "78", "𐞁", "‍e", "́A", " <", "EOT", ">#$%", "ß", "'VE"]} +{"text": "\té", "tokens": 2, "pieces": ["\té"]} +{"text": " ḍ̇…㋿ꟲ३,𐞁İ\n-'M字mfi𐞁😀🏽ع👍🏽́!!!!d\"\u000b'VE ३ ( ", "tokens": 58, "pieces": [" ḋ", "̣", "…", "㋿ꟲ", "३", ",𐞁İ", "\n", "-'", "M字mfi𐞁", "😀🏽", "ع", "👍🏽́!!!!", "d", "\"", "\u000b", "'VE", " ", "३", " ", " (", " "]} +{"text": "\r\n0'M\r😀🏽e½
\r\n\r\n\n(.#$% \nß'llعⅣéſⅣs ㍿\r\n\r\n#$%\"EOT'S 👍🏽🙂'VE\n'ſ", "tokens": 63, "pieces": ["\r\n", "0", "'", "M", "\r", "😀🏽", "e", "½", "
\r\n\r\n\n", "(.#$%", " \n", "ß", "'ll", "ع", "Ⅳ", "éſ", "", "Ⅳ", "s", " ", "㍿\r\n\r\n", "#$%\"", "EOT", "'S", " ", "👍🏽🙂'", "VE", "\n", "'ſ"]} +{"text": "𐞁㋿!\n'DA\n\tßé‍Ⅳ>İ.<|fim_prefix|> ", "tokens": 33, "pieces": ["𐞁", "㋿!\n", "'D", "A", "\n", "\tßé", "‍", "Ⅳ", ">İ", ".<|", "fim", "_prefix", "|>", " "]} +{"text": "́㋿ḍ̇!字'ſ EOT​fiⅣ\t漢𐞁'sfi'T'M ‍ \n㍿é'Re-0", "tokens": 49, "pieces": ["́㋿", "ḋ", "̣!", "字", "'ſ", " EOT", "​fi", "Ⅳ", "\t漢𐞁", "'s", "fi", "'T", "'M", " ", " ‍", " \n", "㍿é", "'Re", "-<", "EOT", ">", "0"]} +{"text": "'Re‍sEOT#$%字'ſ,😀🏽d漢 \nA \n!!'M're…!!'T ㍿ İ(a 𐞁s\rß٣٤٥٦\t'VE'Mİa𐞁𐞁", "tokens": 71, "pieces": ["'Re", "‍sEOT", "#$%", "字", "'ſ", ",😀🏽", "d漢", " \n", "A", " \n", "!!'", "M", "'re", "…", "!!'", "T", " ㍿", " İ", "(a", " 𐞁s", "\r", "ß", "٣٤٥", "٦", "\t", "'VE", "'M", "İa𐞁𐞁"]} +{"text": "𐞁#$%dDž漢", "tokens": 14, "pieces": ["𐞁", "#$%<", "EOT", ">dDž漢"]} +{"text": "<|fim_prefix|>ḍ̇'Séſ\n're\r\n\r\n", "tokens": 20, "pieces": ["<|", "fim", "_prefix", "|>", "ḋ", "̣'", "Se", "́ſ", "\n", "'re", "\r\n\r\n"]} +{"text": "ſ٣٤٥٦A'T'll'ſ٣٤٥٦​ İ漢t", "tokens": 35, "pieces": ["ſ", "", "٣٤٥", "٦", "A", "'T", "'ll", "'ſ", "٣٤٥", "٦", "​", " İ漢t"]} +{"text": "­\r'M'VE\u000b0𐞁'D\r\n\r\né😀🏽'…\u000b<Dž!!…
\" d👍🏽ßm'ſ9½-", "tokens": 58, "pieces": ["­\r", "'M", "'VE", "\u000b", "0", "𐞁", "'D", "\r\n\r\n", "e", "́😀🏽'", "…", "\u000b", "<Dž", "!!", "…", "
", "\"<", "e", "́<|", "fim", "_prefix", "|>", " d", "👍🏽", "ßm", "'ſ", "9½", "-"]} +{"text": "9", "tokens": 1, "pieces": ["9"]} +{"text": "(\"字Džé(ꟲ-", "tokens": 14, "pieces": ["(\"", "字Dže", "́(<", "EOT", ">ꟲ", "-"]} +{"text": "\r\n­m½ \r\n \n#$%ꟲ><|endoftext|> 0d.'s漢😀🏽é", "tokens": 32, "pieces": ["\r\n", "­m", "½", " \r\n \n", "#$%", "ꟲ", "><|", "endoftext", "|>", " ", "0", "d", ".'", "s漢", "😀🏽", "e", "́"]} +{"text": "ſEOT\r\n9\r \na", "tokens": 9, "pieces": ["ſEOT", "\r\n", "9", "\r \n", "a"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "👍🏽\n'D", "tokens": 8, "pieces": ["👍🏽\n", "'D"]} +{"text": "'S'S\n
-Z
३.½'ſ'VE½ İⅣ\n\t𐞁\">\n A'VEs㋿\r!!'S🙂!'DⅣ", "tokens": 50, "pieces": ["'S", "'S", "\n", "
", "-Z", "
", "", "३", ".", "½", "'ſ", "'VE", "½", " İ", "Ⅳ", "\n", "\t𐞁", "\">\n", " ", " A", "'VE", "s", "㋿\r", "!!'", "S", "🙂!'", "D", "Ⅳ"]} +{"text": "
\r\n\r\n㍿\r\n\r\n\r\n#$%ꟲ", "tokens": 12, "pieces": ["
\r\n\r\n", "㍿\r\n\r\n\r\n", "#$%", "ꟲ"]} +{"text": "\t'Re#$%ꟲ👍🏽 'ꟲ'llé\té,'ſß", "tokens": 31, "pieces": ["\t", "'Re", "#$%", "ꟲ", "👍🏽<", "EOT", ">", " ", "'ꟲ", "'ll", "e", "́", "\té", ",'", "ſß"]} +{"text": "<
\r-('VEDž\r'S\ré'Ree\r\n😀🏽's\né\r\nm\t", "tokens": 31, "pieces": ["<", "
\r", "-('", "VEDž", "\r", "'S", "\r", "é", "'Re", "e", "\r\n", "😀🏽'", "s", "\n", "e", "́\r\n", "m", "\t"]} +{"text": "<|fim_prefix|>𐞁\r\n'll\t's9fi12345678'12345678…-0\n > \n<|fim_prefix|>-é
😀🏽's'S!!🙂\n㋿‍\n0ḍ̇", "tokens": 67, "pieces": ["<|", "fim", "_prefix", "|>", "𐞁", "\r\n", "'ll", "\t", "'s", "9", "fi", "123", "456", "78", "'", "123", "456", "78", "…", "-", "0", "\n", " ", ">", " \n", "<|", "fim", "_prefix", "|>-", "é", "
", "😀🏽'", "s", "'S", "!!🙂\n", "㋿‍\n", "0", "ḋ", "̣"]} +{"text": "𐞁🙂𐞁ꟲⅣDž½a0,㍿éḍ̇½'s٣٤٥٦٣٤٥٦", "tokens": 54, "pieces": ["𐞁", "🙂𐞁ꟲ", "Ⅳ", "Dž", "½", "a", "0", ",<", "EOT", "><", "EOT", ">㍿", "éḋ", "̣", "½", "'s", "٣٤٥", "٦٣٤", "٥٦"]} +{"text": "''re​> \n\r\n\r\n>字s'M­fi㋿'s\r\n\r\n\r\nſ0.t", "tokens": 23, "pieces": ["''", "re", "​>", " \n\r\n\r\n", ">字s", "'M", "­fi", "㋿'", "s", "\r\n\r\n\r\n", "ſ", "0", ".t"]} +{"text": "9 fiZꟲA👍🏽'M­ع<|endoftext|>🙂#$%­ḍ̇t🙂  字Z ½ꟲ\r'<|endoftext|>́tDž­", "tokens": 61, "pieces": ["9", " fiZꟲA", "👍🏽'", "M", "­ع", "<|", "endoftext", "|>🙂#$%­", "ḋ", "̣t", "🙂", " ", " 字Z", " ", " ", "½", "ꟲ", "\r", "'<|", "endoftext", "|>́", "tDž", "­"]} +{"text": "𐞁'S(㍿!!e'll'Tsfi㍿𐞁EOT're", "tokens": 30, "pieces": ["𐞁", "'S", "(㍿!!", "e", "'", "ll", "'T", "sfi", "㍿𐞁EOT", "'re"]} +{"text": "aⅣEOT字㍿́\"​ع㍿EOT'D\r<|fim_prefix|>'Sé\r𐞁'Dعİ㍿''VEå‍'Ree0 \u000bDž", "tokens": 61, "pieces": ["a", "Ⅳ", "EOT字", "㍿́\"​", "ع", "㍿EOT", "'D", "\r", "<|", "fim", "_prefix", "|><", "META", "_START", ">'", "Sé", "\r", "𐞁", "'D", "عİ", "㍿''", "VE", "a", "̊‍'", "Ree", "0", " ", "\u000bDž"]} +{"text": "å,🙂<|endoftext|>. \n​.\r\n字\r\nİ\r\n\r\n'MA㋿½9", "tokens": 65, "pieces": ["a", "̊<", "e", "'VE", "\u000bꟲ", " ", " '", "llḋ", "̣ḋ", "̣", " ", "३", "'S", " ", "'ſ", "Džꟲ", ",🙂<|", "endoftext", "|>.", " \n", "​.\r\n", "字", "\r\n", "İ", "\r\n\r\n", "'M", "A", "㋿", "½9"]} +{"text": "
<|fim_prefix|>Dž", "tokens": 11, "pieces": ["
", "<|", "fim", "_prefix", "|>", "Dž"]} +{"text": "-\n
(\r\n\r\n\r\n fi'seß\r漢㋿s ­", "tokens": 19, "pieces": ["-\n", "
", "(\r\n\r\n\r\n", " fi", "'s", "eß", "\r", "漢", "㋿s", " ­"]} +{"text": "'VE٣٤٥٦👍🏽's
e😀🏽ß🙂>ß <AA'S!!㍿㋿ '‍㋿é>ḍ̇fi\r\n\r\n'ſ", "tokens": 64, "pieces": ["'VE", "٣٤٥", "٦", "👍🏽'", "s", "
e", "😀🏽", "ß", "🙂>", "ß", " <", "AA", "'S", "!!㍿㋿", " ", " '‍㋿", "é", ">ḋ", "̣fi", "\r\n\r\n", "'ſ"]} +{"text": "'D'ſ…'S9", "tokens": 8, "pieces": ["'D", "'ſ", "…", "'S", "9"]} +{"text": "\rm३ 'S🙂漢㍿.㍿,'EOT!漢 EOTå", "tokens": 28, "pieces": ["\r", "m", "३", " ", "'S", "🙂漢", "㍿.㍿,'", "EOT", "!漢", " EOTa", "̊"]} +{"text": "­İ㍿!!🙂 ​EOT'S0㋿éßå㋿ \n", "tokens": 27, "pieces": ["­İ", "㍿!!🙂", " ", " ​", "EOT", "'S", "0", "㋿e", "́ßa", "̊㋿", " \n"]} +{"text": "s", "tokens": 1, "pieces": ["s"]} +{"text": "'llⅣ''<|endoftext|>a½\r\n\r\n m !d!!aA 'Dß㋿Z<|fim_prefix|>>'red", "tokens": 43, "pieces": ["'ll", "Ⅳ", "''<|", "endoftext", "|>", "a", "½", "\r\n\r\n", " ", " m", " ", "!d", "!!", "aA", " ", "'D", "ß", "㋿Z", "<|", "fim", "_prefix", "|>>'", "red"]} +{"text": "𐞁𐞁t\r\n-́'Te'Dſ…<|fim_prefix|> 9ḍ̇", "tokens": 33, "pieces": ["𐞁𐞁t", "\r\n", "-́'", "Te", "'D", "ſ", "…", "<|", "fim", "_prefix", "|>", " ", "9", "ḋ", "̣"]} +{"text": "'Re", "tokens": 1, "pieces": ["'Re"]} +{"text": " #$%'ſé('ll ‍­'Re'ReDža", "tokens": 18, "pieces": [" ", "#$%'", "ſé", "('", "ll", " ", " ‍­'", "Re", "'Re", "Dža"]} +{"text": "٣٤٥٦\u000b<|fim_prefix|> fi㋿", "tokens": 22, "pieces": ["٣٤٥", "٦", "\u000b", "<|", "fim", "_prefix", "|>", " fi", "㋿"]} +{"text": "ꟲعae\r\n\r\n'ſ
字(-\t!!!!½åßéA!​'Reſ", "tokens": 29, "pieces": ["ꟲعae", "\r\n\r\n", "'ſ", "
字", "(-", "\t", "!!!!", "½", "a", "̊ßéA", "!​'", "Reſ"]} +{"text": "-३ #$%!!👍🏽", "tokens": 13, "pieces": ["-", "३", " ", " #$%!!👍🏽"]} +{"text": "𐞁'\t- 'M.𐞁​", "tokens": 20, "pieces": ["𐞁", "'", "\t", "-", " ", " '", "M", ".𐞁", "​"]} +{"text": " å <|fim_prefix|>EOT字d‍㍿‍-12345678're\"t0👍🏽<|endoftext|>s>\r\n\r\n9Ⅳad'VE're\"­", "tokens": 55, "pieces": [" ", " a", "̊", " ", "<|", "fim", "_prefix", "|>", "EOT字d", "‍㍿‍-", "123", "456", "78", "'re", "\"t", "0", "👍🏽<|", "endoftext", "|>", "s", ">\r\n\r\n", "9Ⅳ", "ad", "'VE", "'re", "\"­"]} +{"text": "ſ", "tokens": 2, "pieces": ["ſ"]} +{"text": "å٣٤٥٦Z😀🏽>'T\r\n\r\n३ß\"½éZ㋿‍ḍ̇9s'D字\n漢ſdt३e", "tokens": 54, "pieces": ["a", "̊", "٣٤٥", "٦", "Z", "😀🏽<", "EOT", ">>'", "T", "\r\n\r\n", "३", "ß", "\"", "½", "éZ", "㋿‍", "ḋ", "̣", "9", "s", "'D", "字", "\n", "漢ſdt", "३", "e"]} +{"text": "㍿-.'s.e ­'VE ­\u000b\r\nå", "tokens": 17, "pieces": ["㍿-.'", "s", ".e", " ", "­'", "VE", " ­", "\u000b\r\n", "a", "̊"]} +{"text": "'T'T", "tokens": 2, "pieces": ["'T", "'T"]} +{"text": "-.('", "tokens": 2, "pieces": ["-.('"]} +{"text": "Z'ſ'VEé'T…ß'ſ\n're ­.>𐞁́å'VE😀🏽 -d\u000bA'T", "tokens": 43, "pieces": ["Z", "'ſ", "'VE", "e", "́'", "T", "…ß", "'ſ", "\n", "'re", " ", "­.>", "𐞁", "́a", "̊'", "VE", "😀🏽", " ", "-d", "\u000bA", "'T"]} +{"text": "ḍ̇३३㋿e'M'Re\r\n
\"d \u000b'D-漢 漢e\r\n\r\n\u000b㍿a㍿", "tokens": 42, "pieces": ["ḋ", "̣<", "META", "_START", ">", "३३", "㋿e", "'M", "'Re", "\r\n", "
", "\"d", " ", "\u000b", "'D", "-漢", " 漢e", "\r\n\r\n", "\u000b", "㍿a", "㍿"]} +{"text": "'ll'VE'VEå👍🏽é9\r\n\r\n<漢", "tokens": 28, "pieces": ["'ll", "'VE", "'", "VEa", "̊👍🏽", "e", "́<", "EOT", ">", "9", "\r\n\r\n", "<漢"]} +{"text": "عⅣ'M'Md٣٤٥٦0!
're ‍<>måm३\"ع-㍿㋿'re🙂0's'Ś​m<|endoftext|>½ \n­m-🙂", "tokens": 61, "pieces": ["ع", "Ⅳ", "'M", "'M", "d", "٣٤٥", "٦0", "!", "
", "'re", " ‍<>", "ma", "̊m", "३", "\"ع", "-㍿㋿'", "re", "🙂", "0", "'s", "'S", "́​", "m", "<|", "endoftext", "|>", "½", " \n", "­m", "-🙂"]} +{"text": "\u000b字'éfi'VEDžⅣ\tDžda\r\n\r\n 'VE½<|fim_prefix|><|endoftext|>🙂-'ll漢Ⅳ", "tokens": 45, "pieces": ["\u000b字", "'e", "́fi", "'VE", "Dž", "Ⅳ", "\tDžda", "\r\n\r\n", " ", " <", "EOT", ">'", "VE", "½", "<|", "fim", "_prefix", "|><|", "endoftext", "|>🙂-'", "ll漢", "Ⅳ"]} +{"text": "İeꟲA", "tokens": 7, "pieces": ["İeꟲA"]} +{"text": "9'Reعd(𐞁\n㋿३a\r\n\r\nⅣ(A \n'S字\"a
fi\u000b're-!", "tokens": 41, "pieces": ["9", "'Re", "عd", "(𐞁", "\n", "㋿", "३", "a", "\r\n\r\n", "Ⅳ", "(A", " \n", "'S", "字", "\"a", "
fi", "\u000b", "'re", "-!<", "META", "_START", ">"]} +{"text": "Dž٣٤٥٦#$%", "tokens": 12, "pieces": ["Dž", "٣٤٥", "٦", "#$%"]} +{"text": "㍿\r\n\r\nd e>…Aſé<|endoftext|>\tḍ̇\r\n\r\n\r", "tokens": 31, "pieces": ["㍿\r\n\r\n", "d", " e", ">", "…Aſe", "́<|", "endoftext", "|>", "\tḋ", "̣\r\n\r\n\r"]} +{"text": "!fi>'re'VEſ­å'Re.EOTſa 
㋿'ſ\r\n'Mß'M'Sem\r\n\r\n12345678ḍ̇ 𐞁!!­é12345678́👍🏽12345678", "tokens": 71, "pieces": ["!fi", ">'", "re", "'VE", "ſ", "­a", "̊'", "Re", ".EOTſa", " ", "
", "㋿'", "ſ", "\r\n", "'M", "ß", "'", "M", "'S", "em", "\r\n\r\n", "123", "456", "78", "ḋ", "̣", " ", " 𐞁", "!!­", "e", "́", "123", "456", "78", "́👍🏽", "123", "456", "78"]} +{"text": "e.́عſe​\r\nt -㋿½\n'VE('re0 0 ㍿'T ßA𐞁!!ḍ̇<|fim_prefix|>字‍½t㋿", "tokens": 65, "pieces": ["e", ".́", "عſe", "​\r\n", "t", " ", " -㋿", "½", "\n", "'VE", "('", "re", "", "0", " ", " ", "0", " ", " ㍿'", "T", " ", " ßA𐞁", "!!", "ḋ", "̣<|", "fim", "_prefix", "|>", "字", "‍", "½", "t", "㋿"]} +{"text": "#$% 's‍'ſém ſ'T0½'M㋿éZ", "tokens": 22, "pieces": ["#$%", " ", " '", "s", "‍'", "ſém", " ſ", "'T", "0½", "'M", "㋿éZ"]} +{"text": " m٣٤٥٦ \n­'D㋿", "tokens": 20, "pieces": [" m", "٣٤٥", "٦", " \n", "­'", "D", "㋿"]} +{"text": "d٣٤٥٦́­́३<\r\n..‍'S-.㍿́
İ<'T", "tokens": 37, "pieces": ["d", "٣٤٥", "٦", "́­́", "३", "<\r\n", "..<", "EOT", ">‍'", "S", "-.㍿́", "
İ", "<'", "T"]} +{"text": "'S\u000bs're>Ⅳt'VE<'Re><|endoftext|>​'Té
<|fim_prefix|>'s\r\n0½.'VE\u000b'M#$%'M​ſZeḍ̇.< !!d're", "tokens": 57, "pieces": ["'S", "\u000bs", "'re", ">", "Ⅳ", "t", "'VE", "<'", "Re", "><|", "endoftext", "|>​'", "Te", "́", "
", "<|", "fim", "_prefix", "|>'", "s", "\r\n", "0½", ".'", "VE", "\u000b", "'M", "#$%'", "M", "​ſZeḋ", "̣.<", " ", "!!", "d", "'re"]} +{"text": "
́عé'll‍ḍ̇!!'ſ're\r'VE…‍'VEſe'VE😀🏽🙂sع", "tokens": 44, "pieces": ["
", "́عé", "'ll", "‍ḋ", "̣!!'", "ſ", "'re", "\r", "'VE", "…", "‍'", "VEſe", "'VE", "😀🏽🙂", "sع"]} +{"text": "!!́ fie 
12345678-\u000bé0're(", "tokens": 21, "pieces": ["!!́", " fie", " ", "
", "123", "456", "78", "-", "\u000bé", "0", "'re", "("]} +{"text": "\r\n.́'Dt<\" ٣٤٥٦'Sd'Md'll😀🏽 ḍ̇İ9\t<|fim_prefix|>ém​A 漢å", "tokens": 53, "pieces": ["\r\n", ".́'", "Dt", "<\"", " ", "٣٤٥", "٦", "'S", "d", "'M", "d", "'ll", "😀🏽", " ", " ḋ", "̣İ", "9", "\t", "<|", "fim", "_prefix", "|>", "e", "́m", "​A", " 漢a", "̊"]} +{"text": "Zfi
åZ👍🏽.'VE're…é123456780ع㋿<|endoftext|>İ'!(9'(㋿
👍🏽'Re ㋿>", "tokens": 63, "pieces": ["Zfi", "
a", "̊Z", "👍🏽.'", "VE", "'re", "", "…e", "́", "123", "456", "780", "ع", "㋿<|", "endoftext", "|>", "İ", "'!(", "9", "'(㋿", "
", "👍🏽'", "Re", " ", "㋿>"]} +{"text": "'M'llEOT'll", "tokens": 9, "pieces": ["'M", "'", "llEOT", "'ll"]} +{"text": "…éd EOTع're'DDž>㍿>İ\r\n\r\n( <­½é-'ſع're​ ", "tokens": 31, "pieces": ["…éd", " ", " EOTع", "'re", "'D", "Dž", ">㍿>", "İ", "\r\n\r\n", "(", " ", "<­", "½", "é", "-'", "ſع", "'re", "​", " "]} +{"text": "ßſßfi'St", "tokens": 8, "pieces": ["ßſßfi", "'S", "t"]} +{"text": "Dž0t'VEm(fi\r", "tokens": 11, "pieces": ["Dž", "0", "t", "'VE", "m", "(fi", "\r"]} +{"text": "<#$%'Re \n \n字\r\n\r\n…́<|fim_prefix|>12345678EOT­9Dž-'' ​m𐞁३'T'T\n", "tokens": 43, "pieces": ["<#$%'", "Re", " \n \n", "字", "\r\n\r\n", "…", "́<|", "fim", "_prefix", "|>", "123", "456", "78", "EOT", "­", "9", "Dž", "-''", " ​", "m𐞁", "३", "'T", "'T", "\n"]} +{"text": "<|fim_prefix|>!(EOTm👍🏽Z\"", "tokens": 22, "pieces": ["<|", "fim", "_prefix", "|>!(", "EOTm", "👍🏽", "Z", "\"<", "EOT", ">"]} +{"text": "\r\n\r\nḍ̇'M३'Dع😀🏽><|endoftext|>!>\r'Re漢'VE", "tokens": 32, "pieces": ["\r\n\r\n", "ḋ", "̣'", "M", "३", "'D", "ع", "😀🏽><|", "endoftext", "|>!>\r", "'Re", "漢", "'VE"]} +{"text": "३'S🙂é,​'reİ㍿ꟲa-<|endoftext|>éZ­-😀🏽\u000b", "tokens": 37, "pieces": ["३", "'S", "🙂e", "́,​'", "reİ", "㍿ꟲa", "-<|", "endoftext", "|>", "éZ", "­-😀🏽", "\u000b"]} +{"text": " 'VEعAßꟲ…ſ'M​­d\ré're!\t!#$%\n😀🏽", "tokens": 37, "pieces": [" ", "'VE", "عAßꟲ", "…ſ", "'M", "​­", "d", "\r", "e", "́'", "re", "!", "\t", "!#$%<", "EOT", ">\n", "😀🏽"]} +{"text": "'ſ…'llåé'Re.…m'reſ(\"e'Ⅳéİ­'D#$%a
'S½½३́\rعA३ḍ̇㍿", "tokens": 63, "pieces": ["'ſ", "…", "'ll", "a", "̊é", "'Re", ".", "…m", "'re", "ſ", "(\"", "e", "'", "Ⅳ", "é", "İ", "­'", "D", "#$%", "a", "
", "'S", "½½३", "́\r", "عA", "३", "ḋ", "̣㍿"]} +{"text": ",'ſ(,'é'Re😀🏽字A㍿㍿(🙂m>'ſ٣٤٥٦å12345678३‍('T㍿‍", "tokens": 63, "pieces": [",'", "ſ", "(,'", "é", "'", "Re", "😀🏽", "字A", "㍿㍿(<", "EOT", ">🙂", "m", ">'", "ſ", "٣٤٥", "٦", "a", "̊", "123", "456", "78३", "‍('", "T", "㍿‍"]} +{"text": "<🙂٣٤٥٦ⅣEOTEOTt'ſ12345678 \n , ́\"\u000b,\r\n\r\n字'VEfi.'s#$%", "tokens": 39, "pieces": ["<🙂", "٣٤٥", "٦Ⅳ", "EOTEOTt", "'ſ", "123", "456", "78", " \n", " ,", " ́\"", "\u000b", ",\r\n\r\n", "字", "'VE", "fi", ".'", "s", "#$%"]} +{"text": ".ß -<|fim_prefix|>99\r\n\n👍🏽ḍ̇㋿t㋿<|fim_prefix|>'refi", "tokens": 40, "pieces": [".ß", " -<|", "fim", "_prefix", "|>", "99", "\r\n\n", "👍🏽", "ḋ", "̣㋿", "t", "㋿<|", "fim", "_prefix", "|>'", "refi"]} +{"text": "­'s12345678.İ‍fit'D(Aé\t'T9…a३İ", "tokens": 28, "pieces": ["­'", "s", "123", "456", "78", ".İ", "‍fit", "'D", "(Ae", "́", "\t", "'T", "9", "…a", "३", "İ"]} +{"text": "​'Sſ漢t e,.'S ½ \nsſé'M㋿'llſ'Dž\t \"🙂tZḍ̇…", "tokens": 50, "pieces": ["​'", "Sſ", "漢t", " e", ",.'", "S", " ", " ", "½", " \n", "sſe", "́'", "M", "㋿'", "llſ", "'Dž", "\t", " ", "\"🙂", "tZḋ", "̣", "…"]} +{"text": ",9\tḍ̇", "tokens": 8, "pieces": [",", "9", "\tḋ", "̣"]} +{"text": "٣٤٥٦عßé\r\n…,ḍ̇\n㍿'T> <|fim_prefix|>ع<|endoftext|>s<", "tokens": 52, "pieces": ["'D", "ḋ", "̣", "\t字", "\n", "s", "'M", "a漢", " ", "㋿'", "re", "\t", ",🙂", "0", ".", "
e", "́<", "EOT", ">>", " ", "<|", "fim", "_prefix", "|>", "ع", "<|", "endoftext", "|>", "s", "<"]} +{"text": "!‍'D9Dž'llḍ̇​A㋿३\r३té'llå'D'Re…­३0漢<|fim_prefix|>9(­½<|fim_prefix|>́ ́t", "tokens": 64, "pieces": ["!‍'", "D", "9", "Dž", "'ll", "ḋ", "̣​", "A", "㋿", "३", "\r", "३", "te", "́'", "lla", "̊'", "D", "'Re", "…", "­", "३0", "漢", "<|", "fim", "_prefix", "|>", "9", "(­", "½", "<|", "fim", "_prefix", "|>́", " ́", "t"]} +{"text": "12345678<|fim_prefix|>\r\nå👍🏽#$%é漢12345678Ⅳ٣٤٥٦t🙂ع𐞁'sİ٣٤٥٦…漢👍🏽ḍ̇EOT're", "tokens": 73, "pieces": ["123", "456", "78", "<|", "fim", "_prefix", "|>\r\n", "a", "̊👍🏽#$%", "é漢", "123", "456", "78Ⅳ", "٣٤٥", "٦", "t", "🙂ع𐞁", "'s", "İ", "٣٤٥", "٦", "…漢", "👍🏽", "ḋ", "̣EOT", "'re"]} +{"text": "ßعs'㍿\"\r\n\r\n字👍🏽0afi'ſ12345678tİ<9 😀🏽\r\n\r\n-", "tokens": 39, "pieces": ["ßعs", "'㍿\"\r\n\r\n", "字", "👍🏽", "0", "afi", "'ſ", "123", "456", "78", "tİ", "<", "9", " ", "😀🏽\r\n\r\n", "-"]} +{"text": "'M字t12345678…#$%­,\n \nDž9字m\r\n", "tokens": 19, "pieces": ["'M", "字t", "123", "456", "78", "…", "#$%­,\n", " \n", "Dž", "9", "字m", "\r\n"]} +{"text": "字's👍🏽\u000b\r\n\r\nع<\u000b३ \nDž>㋿åßfi \n", "tokens": 55, "pieces": ["字", "'s", "👍🏽", "\u000b\r\n\r\n", "ع", "<", "\u000b", "", "३", " \n", "Dž", ">㋿", "a", "̊ßfi", " \n"]} +{"text": "'D'll fi👍🏽EOT‍s👍🏽'VEEOT a😀🏽\r!!
mZ\r\n(­😀🏽ḍ̇EOT\r\n \n", "tokens": 65, "pieces": ["'D", "'ll", "", " fi", "👍🏽", "EOT", "‍s", "👍🏽'", "VEEOT", " ", " <", "EOT", ">a", "😀🏽\r", "!!", "
mZ", "\r\n", "(­<", "META", "_START", ">😀🏽", "ḋ", "̣EOT", "\r\n \n"]} +{"text": "ßꟲ‍­<|fim_prefix|>12345678\r\n\r\n", "tokens": 18, "pieces": ["ßꟲ", "‍­<|", "fim", "_prefix", "|>", "123", "456", "78", "\r\n\r\n"]} +{"text": "́", "tokens": 1, "pieces": ["́"]} +{"text": "…é­<|endoftext|>d字,\nع'M9ḍ̇'Re<'sd", "tokens": 29, "pieces": ["…é", "­<|", "endoftext", "|>", "d字", ",\n", "ع", "'M", "9", "ḋ", "̣'", "Re", "<'", "sd"]} +{"text": "\r\n\r\n", "tokens": 1, "pieces": ["\r\n\r\n"]} +{"text": "🙂 12345678Z𐞁‍'ſZ٣٤٥٦'re😀🏽👍🏽#$%>'ſ90‍12345678m's", "tokens": 53, "pieces": ["🙂", " <", "EOT", ">", "123", "456", "78", "Z𐞁", "‍'", "ſZ", "٣٤٥", "٦", "'re", "😀🏽👍🏽#$%>'", "ſ", "90", "‍", "123", "456", "78", "m", "'s"]} +{"text": "!fi.''D​'S​aé🙂'ſ🙂‍eé'Re½!!\u000b!!'T㍿t", "tokens": 42, "pieces": ["!fi", ".<", "EOT", ">''", "D", "​'", "S", "​ae", "́🙂'", "ſ", "🙂‍", "ee", "́'", "Re", "½", "!!<", "META", "_START", ">", "\u000b", "!!'", "T", "㍿t"]} +{"text": "'S'", "tokens": 6, "pieces": ["'", "S", "'"]} +{"text": "\"ſİ#$%½-½ 𐞁'Re'ſ e-\r\n'M'll
e👍🏽a'ſ", "tokens": 36, "pieces": ["\"ſİ", "#$%", "½", "-", "½", " 𐞁", "'Re", "'ſ", " e", "-\r\n", "'M", "'ll", "
e", "👍🏽", "a", "'ſ"]} +{"text": "👍🏽'VE\"fi-t's'll漢ße'Tåefis\r\r\n're
ع!å㍿å𐞁 \né😀🏽9A(é'T㋿\"-", "tokens": 62, "pieces": ["👍🏽'", "VE", "\"fi", "-t", "'s", "'ll", "漢ße", "'T", "a", "̊efis", "\r\r\n", "'re", "
ع", "!a", "̊㍿", "a", "̊𐞁", " \n", "é", "😀🏽", "9", "A", "(e", "́'", "T", "㋿\"-"]} +{"text": "𐞁​㍿", "tokens": 8, "pieces": ["𐞁", "​㍿"]} +{"text": "ꟲEOT­<|endoftext|>m​ſ \n👍🏽<|fim_prefix|>EOT>#$%t \n'llfi😀🏽Ze,", "tokens": 54, "pieces": ["ꟲEOT", "­<", "META", "_START", "><|", "endoftext", "|>", "m", "​ſ", " \n", "👍🏽<|", "fim", "_prefix", "|>", "EOT", "><", "EOT", ">#$%", "t", " \n", "'ll", "fi", "😀🏽", "Ze", ","]} +{"text": "a\r\n\r\n‍ s'ſ­é \n!!t字0\t,'S", "tokens": 19, "pieces": ["a", "\r\n\r\n", "‍", " s", "'ſ", "­é", " \n", "!!", "t字", "0", "\t", ",'", "S"]} +{"text": "'M!<٣٤٥٦\nDž'ſ-12345678👍🏽'Sß字Z३\r\né.'T#$%-'t𐞁​㍿<é\u000bⅣ​", "tokens": 56, "pieces": ["'M", "!<", "٣٤٥", "٦", "\n", "Dž", "'ſ", "-", "123", "456", "78", "👍🏽'", "Sß字Z", "३", "\r\n", "e", "́.'", "T", "#$%-'", "t𐞁", "​㍿<", "é", "\u000b", "Ⅳ", "​"]} +{"text": "​ 'll\r\n're", "tokens": 9, "pieces": ["​", " ", " '", "ll", "\r\n", "'re"]} +{"text": " \"e'T漢<|endoftext|>Dž(…'VE㋿ſİ\r\n\r\n'reḍ̇'Re.", "tokens": 58, "pieces": [" ", "\"e", "'T", "漢", "<|", "endoftext", "|>", "Dž", "(", "…", "'VE", "㋿ſİ", "\r\n\r\n", "A", "'", "reḋ", "̣'", "Re", "."]} +{"text": "३٣٤٥٦½", "tokens": 11, "pieces": ["३٣٤", "٥٦½"]} +{"text": "ß'Re\"fi ३İ \n<<|endoftext|>12345678🙂'ſ12345678
٣٤٥٦", "tokens": 42, "pieces": ["ß", "'Re", "\"fi", " ", " <", "META", "_START", ">", "३", "İ", " \n", "<<|", "endoftext", "|>", "123", "456", "78", "🙂'", "ſ", "123", "456", "78", "
", "٣٤٥", "٦"]} +{"text": "<|fim_prefix|>'s\t ḍ̇'re🙂 ß漢's\r\n\r\nZ9'Resḍ̇\n'T\"漢\"Ⅳ", "tokens": 46, "pieces": ["<|", "fim", "_prefix", "|>'", "s", "\t ", " ḋ", "̣'", "re", "🙂", " ß漢", "'s", "\r\n\r\n", "Z", "9", "'Re", "sḋ", "̣\n", "'T", "\"漢", "\"", "Ⅳ"]} +{"text": "‍‍㍿\u000b<|fim_prefix|> 𐞁t-​'aA>ḍ̇ 'llß٣٤٥٦da'e㍿ꟲſ", " 𐞁t", "-​'", "aA", ">ḋ", "̣", " ", "'ll", "ß", "٣٤٥", "٦", "da", "'e", "㍿ꟲſ", "<|fim_prefix|>'Refié𐞁½\u000b३ 'Dſḍ̇12345678 
", "tokens": 41, "pieces": ["‍😀🏽><|", "fim", "_prefix", "|>'", "Refié𐞁", "½", "\u000b", "३", " '", "Dſḋ", "̣", "123", "456", "78", " 
"]} +{"text": "\t'T İ​Z😀🏽", "tokens": 11, "pieces": ["\t", "'T", " İ", "​Z", "😀🏽"]} +{"text": "'D!!Ⅳs're ­😀🏽'Re٣٤٥٦́ ㋿٣٤٥٦٣٤٥٦ Dž'M<|endoftext|>
", "tokens": 65, "pieces": ["'D", "!!", "Ⅳ", "​<", "EOT", ">s", "'re", " ", " ­😀🏽'", "Re", "٣٤٥", "٦", "́", " ㋿", "٣٤٥", "٦٣٤", "٥٦", " Dž", "'M", "<|", "endoftext", "|>", "
"]} +{"text": "ßåéḍ̇ꟲfi👍🏽'll\"", "tokens": 31, "pieces": ["ßa", "̊é", "<", "EOT", ">ḋ", "̣ꟲfi", "👍🏽'", "ll", "\""]} +{"text": "\"t'reع漢́٣٤٥٦Ⅳ\t٣٤٥٦<́s'Sع​m\tⅣ.\r'Déé", "tokens": 41, "pieces": ["\"t", "'re", "ع漢", "́", "٣٤٥", "٦Ⅳ", "\t", "٣٤٥", "٦", "<́", "s", "'S", "ع", "​m", "\t", "Ⅳ", ".\r", "'D", "éé"]} +{"text": "('M\r字…-३ \u000b…
é<|fim_prefix|>'M.'Re", "tokens": 26, "pieces": ["('", "M", "\r", "字", "…", "-", "३", " \u000b…", "
é", "<|", "fim", "_prefix", "|>'", "M", ".'", "Re"]} +{"text": "'VE.<|endoftext|>fi字\t​\r\n\r\n字​ꟲ\r Dž́𐞁'D12345678٣٤٥٦'T<\r\n'ſ12345678👍🏽漢
𐞁'll­12345678 ع½😀🏽#$% \r\n\r\n", "tokens": 82, "pieces": ["'VE", ".<|", "endoftext", "|>", "fi字", "\t", "​\r\n\r\n", "字", "​ꟲ", "\r", " ", " Dž", "́𐞁", "'D", "123", "456", "78٣", "٤٥٦", "'T", "<\r\n", "'ſ", "123", "456", "78", "👍🏽", "漢", "
𐞁", "'ll", "­", "123", "456", "78", " ", " ع", "½", "😀🏽#$%", " \r\n\r\n"]} +{"text": "漢!!'Re !!", "tokens": 7, "pieces": ["漢", "!!'", "Re", " ", " !!"]} +{"text": "-EOT", "tokens": 5, "pieces": ["-", "EOT"]} +{"text": "­
ꟲfi㋿٣٤٥٦👍🏽(३'M\"'T('re😀🏽>Afi12345678", "tokens": 50, "pieces": ["­", "
ꟲfi", "㋿", "٣٤٥", "٦", "👍🏽(", "३", "'M", "\"'", "T", "('", "re", "😀🏽>", "Afi", "", "123", "456", "78"]} +{"text": "㍿'s'ſ字\u000b'Re😀🏽,'T́­ 𐞁
!!fi ,漢漢३\t३ß́\rⅣ\r'ſ'ſå\r\r\n\r\n👍🏽", "tokens": 64, "pieces": ["㍿'", "s", "'ſ", "字", "\u000b", "'Re", "😀🏽,'", "T", "́­", " 𐞁", "
", "!!", "fi", " ", ",漢漢", "३", "\t", "३", "ß", "́\r", "Ⅳ", "\r", "'ſ", "'ſ", "a", "̊\r\r\n\r\n", "👍🏽"]} +{"text": "\t12345678🙂İ­Dža­< A", "tokens": 15, "pieces": ["\t", "123", "456", "78", "🙂İ", "­Dža", "­<", " A"]} +{"text": "a<|fim_prefix|>\r\n\r\n\r\n\r\n
­👍🏽 m'så'VE", "tokens": 25, "pieces": ["a", "<|", "fim", "_prefix", "|>\r\n\r\n\r\n\r\n", "
", "­👍🏽", " m", "'s", "a", "̊'", "VE"]} +{"text": " \r\n\r\n😀🏽‍'VE-İ\naA<字<|endoftext|>#$%a🙂é>ع!! ꟲ​dſ½'T", "tokens": 53, "pieces": [" \r\n\r\n", "😀🏽‍'", "VE", "-İ", "\n", "aA", "<字", "<|", "endoftext", "|>#$%", "a", "🙂e", "́>", "ع", "!!<", "EOT", ">", " ꟲ", "​dſ", "½", "'T"]} +{"text": "'D३a漢𐞁🙂12345678<|fim_prefix|>>㍿'VE٣٤٥٦ EOT'T\t🙂e字(", "tokens": 48, "pieces": ["'D", "३", "a漢𐞁", "🙂", "123", "456", "78", "<|", "fim", "_prefix", "|>>㍿<", "META", "_START", ">'", "VE", "٣٤٥", "٦", " EOT", "'T", "\t", "🙂e字", "("]} +{"text": "́  \r\nfi", "tokens": 6, "pieces": ["́", "  \r\n", "fi"]} +{"text": "'Sİt­.'D's𐞁\r\n-'ll'll.İ́\r\n\r\n'\n字éåt \nß", "tokens": 31, "pieces": ["'S", "İt", "­.'", "D", "'s", "𐞁", "\r\n", "-'", "ll", "'ll", ".<", "EOT", ">İ", "́\r\n\r\n", "'\n", "字éa", "̊t", " \n", "ß"]} +{"text": "🙂t\"ꟲ\r\nꟲİ<|endoftext|> ḍ̇é 'T​𐞁漢t ſع12345678fiع", "tokens": 45, "pieces": ["🙂t", "\"ꟲ", "\r\n", "ꟲİ", "<|", "endoftext", "|>", " ḋ", "̣é", " ", "'T", "​𐞁漢t", " ſع", "123", "456", "78", "fiع"]} +{"text": "'Re'D'VEDž.!!İꟲ", "tokens": 12, "pieces": ["'Re", "'D", "'VE", "Dž", ".!!", "İꟲ"]} +{"text": "fiDž­s㍿'
㋿'reſé👍🏽漢𐞁<|fim_prefix|>ḍ̇tDž'sa", "tokens": 52, "pieces": ["fiDž", "­s", "㍿'", "
", "㋿'", "reſé", "👍🏽<", "META", "_START", ">漢𐞁", "<|", "fim", "_prefix", "|>", "ḋ", "̣tDž", "'s", "a"]} +{"text": "éꟲ漢٣٤٥٦ 'M\tfi\u000b're字", "tokens": 23, "pieces": ["éꟲ漢", "٣٤٥", "٦", " ", " '", "M", "\tfi", "\u000b", "'re", "字"]} +{"text": " …ſe'9漢­字Ⅳ😀🏽㋿\u000b\r\n\r\nd'D🙂… d#$%'Reḍ̇e㍿Zt…\u000bİ>e\r­a\r", "tokens": 59, "pieces": [" ", "…ſe", "'", "9", "漢", "­字", "", "Ⅳ", "😀🏽㋿", "\u000b\r\n\r\n", "d", "'D", "🙂", "…", " d", "#$%'", "Reḋ", "̣e", "㍿Zt", "…", "\u000bİ", ">e", "\r", "­a", "\r"]} +{"text": ">,<|endoftext|>İ漢 \r\n\r\ns字!ſ0漢Z\u000bEOT'VE
A👍🏽,", "tokens": 37, "pieces": [">,<|", "endoftext", "|>", "İ漢", " \r\n\r\n", "s字", "!ſ", "0", "漢Z", "\u000bEOT", "'VE", "
A", "👍🏽,"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'TdⅣfi­t\t-­'s<\"'VE'M 'Re'VE漢ſ
'D
‍㍿9#$%ſ'd
\r", "tokens": 52, "pieces": ["'T", "d", "Ⅳ", "fi", "­t", "\t", "-­'", "s", "<<", "META", "_START", ">\"'", "VE", "'M", " ", "'Re", "'VE", "漢ſ", "
", "'D", "
", "‍㍿", "9", "#$%", "ſ", "'d", "
\r"]} +{"text": "eḍ̇ ,'M0tEOT \n'T\"👍🏽t漢
!!㍿\r\nſİ😀🏽<|endoftext|>'reZ's", "tokens": 49, "pieces": ["eḋ", "̣", " ,'", "M", "0", "tEOT", " \n", "'T", "\"👍🏽", "t漢", "
", "!!㍿\r\n", "ſİ", "😀🏽<|", "endoftext", "|>'", "reZ", "'s"]} +{"text": "𐞁'M'ſZ'T'Då\u000bßfi'VE'S\n\n漢'S\r\nſ'SEOT\" Ⅳ", "tokens": 38, "pieces": ["𐞁", "'M", "'ſ", "Z", "'T", "'D", "a", "̊", "\u000bßfi", "'VE", "'", "S", "\n\n", "漢", "'S", "\r\n", "ſ", "'S", "EOT", "\"", " ", "Ⅳ"]} +{"text": "😀🏽  -0'M \n㋿9", "tokens": 15, "pieces": ["😀🏽", " ", " ", "-", "0", "'M", " \n", "㋿", "9"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "½'Da'D-<|fim_prefix|>'re'D​m'å३\r\n\r\n३Afi(, e.'re'Reḍ̇'T", "tokens": 45, "pieces": ["½", "'D", "a", "'D", "-<|", "fim", "_prefix", "|>'", "re", "'D", "​m", "'a", "̊", "३", "\r\n\r\n", "३", "Afi", "(,", " ", " e", ".'", "re", "'Re", "ḋ", "̣<", "META", "_START", ">'", "T"]} +{"text": "\rm‍㍿0('Séع<|endoftext|>­ ", "tokens": 21, "pieces": ["\r", "m", "‍㍿", "0", "('", "Se", "́ع", "<|", "endoftext", "|>­", " "]} +{"text": "'ſaEOT \n\t🙂é漢're​Dž🙂㍿(dm㍿éfiEOT..́…,(0.👍🏽…İ­!㋿", "tokens": 54, "pieces": ["'ſ", "aEOT", " \n", "\t", "🙂e", "́漢", "'re", "​Dž", "🙂㍿(", "dm", "㍿éfiEOT", "..́", "…", ",(", "0", ".👍🏽", "…İ", "­!㋿"]} +{"text": ",'VEéDž", "tokens": 6, "pieces": [",'", "VEe", "́Dž"]} +{"text": "'llÁ㋿12345678 é𐞁Aa9𐞁fi㋿ḍ̇'ſſ'D\nḍ̇ß́>('VE'VE,㋿", "tokens": 57, "pieces": ["'ll", "A", "́㋿", "123", "456", "78", " ", " e", "́𐞁Aa", "9", "𐞁fi", "㋿ḋ", "̣'", "ſſ", "'D", "\n", "ḋ", "̣ß", "́>('", "VE", "'VE", ",㋿"]} +{"text": "'ll-ḍ̇ \r'ſd!! ३,𐞁½ſ éḍ̇sḍ̇é'ſ!\t<|fim_prefix|>\t­
字😀🏽ß<,'re…\u000b.A𐞁'M", "tokens": 77, "pieces": ["'ll", "-ḋ", "̣", " \r", "'ſ", "d", "!!", " ", "३", ",𐞁", "½", "ſ", " e", "́ḋ", "̣sḋ", "̣é", "'ſ", "!", "\t", "<|", "fim", "_prefix", "|>", "\t", "­", "
字", "😀🏽", "ß", "<,'", "re", "…", "\u000b", ".A𐞁", "'M"]} +{"text": "🙂s\n\r­'re EOT!漢🙂漢'reß'D<.'ſ Ⅳ'T9\n's(m're12345678>A<|endoftext|>#$%", "tokens": 52, "pieces": ["🙂s", "\n\r", "­'", "re", " EOT", "!漢", "🙂<", "EOT", ">漢", "'re", "ß", "'D", "<.'", "ſ", " ", "Ⅳ", "'T", "9", "\n", "'s", "(m", "'re", "123", "456", "78", ">A", "<|", "endoftext", "|>#$%"]} +{"text": "fiİ​ſ'VEſ0-0 \n, 9<|fim_prefix|>漢½", "tokens": 30, "pieces": ["fiİ", "​ſ", "'VE", "ſ", "0", "-", "0", " \n", ",", " ", "9", "<|", "fim", "_prefix", "|>", "漢", "½"]} +{"text": "\r\n-\rs३\t́'T#$%\r\ne​½'ll<|fim_prefix|>", "tokens": 27, "pieces": ["\r\n", "-\r", "s", "३", "\t", "́'", "T", "#$%<", "META", "_START", ">\r\n", "e", "​", "½", "'ll", "<|", "fim", "_prefix", "|>"]} +{"text": "漢\"9'ſ're", "tokens": 8, "pieces": ["漢", "\"", "9", "'ſ", "'re"]} +{"text": "(Zs\u000b 
.<'VE\r\n\r\n㍿🙂㍿‍\t(🙂½\r're
 😀🏽🙂!Z \n😀🏽", "tokens": 48, "pieces": ["(Zs", "\u000b ", "
", ".<'", "VE", "\r\n\r\n", "㍿🙂㍿‍", "\t", "(🙂", "½", "\r", "'re", "
 ", " 😀🏽🙂!", "Z", "", " \n", "😀🏽"]} +{"text": "é<|fim_prefix|>d", "tokens": 10, "pieces": ["e", "́<|", "fim", "_prefix", "|>", "d"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "12345678A🙂𐞁👍🏽.<|fim_prefix|>\u000b's\u000b!!!'\n\rEOTdꟲsta'S9fi<Ⅳ\u000b ع\r\n\r\n½ 'S", "tokens": 51, "pieces": ["123", "456", "78", "A", "🙂𐞁", "👍🏽.<|", "fim", "_prefix", "|>", "\u000b", "'s", "\u000b", "!!!'\n\r", "EOTdꟲsta", "'S", "9", "fi", "<", "Ⅳ", "\u000b ", " ع", "\r\n\r\n", "½", " ", "'S"]} +{"text": "(­ꟲ'llZ😀🏽́½㋿< \nå\rßå \n", "tokens": 28, "pieces": ["(­", "ꟲ", "'ll", "Z", "😀🏽́", "½", "㋿<", " \n", "a", "̊\r", "ßa", "̊", " \n"]} +{"text": "!!#$%Ⅳḍ̇漢'llfißeDž'D ‍㍿.(字<|fim_prefix|>ee👍🏽12345678eDžé'T!'VE㋿ m", "tokens": 61, "pieces": ["!!#$%", "Ⅳ", "ḋ", "̣漢", "'ll", "fißeDž", "'D", " ", "‍㍿.(", "字", "<|", "fim", "_prefix", "|>", "ee", "👍🏽", "123", "456", "78", "eDže", "́'", "T", "!'", "VE", "㋿", " ", " m"]} +{"text": "\r'😀🏽'Re  İ'ſ\"🙂 𐞁\r\n\t#$% ­.a!\t  \n\u000bſ", "tokens": 38, "pieces": ["\r", "'😀🏽'", "Re", " ", " İ", "'ſ", "\"🙂", " 𐞁", "\r\n", "\t", "#$%", " ", "­.", "a", "!", "\t  \n", "\u000bſ"]} +{"text": "ßعe're9'sßſ漢'VE'VE½\"👍🏽'ReZ😀🏽'VE12345678ḍ̇👍🏽
㋿Z…\t (\r\n'reⅣ'T'12345678'S
", "tokens": 69, "pieces": ["ßعe", "'re", "9", "'s", "ßſ漢", "'VE", "'VE", "½", "\"👍🏽'", "ReZ", "😀🏽'", "VE", "123", "456", "78", "ḋ", "̣👍🏽", "
", "㋿Z", "…\t", " ", "(\r\n", "'re", "Ⅳ", "'T", "'", "123", "456", "78", "'S", "
"]} +{"text": "ḍ̇
ꟲ'T'VE
'ſ.Dž\r\n'…𐞁 !'D\r\n\r\n'Mt字 ſ​🙂,­s\r\n\r\nꟲ'Mt(㋿½३", "tokens": 60, "pieces": ["ḋ", "̣", "
ꟲ", "'T", "'VE", "
", "'ſ", ".", "Dž", "\r\n", "'", "…𐞁", " ", "!'", "D", "\r\n\r\n", "'M", "t字", " ſ", "​🙂,­", "s", "\r\n\r\n", "ꟲ", "'M", "t", "(㋿", "½३"]} +{"text": "åAⅣ🙂ßfi'", "tokens": 13, "pieces": ["a", "̊A", "Ⅳ", "🙂ßfi", "'"]} +{"text": "'s(\nd👍🏽'll A(\n.", "tokens": 15, "pieces": ["'s", "(\n", "d", "👍🏽'", "ll", " A", "(\n", "."]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "#$%😀🏽𐞁A\r\n\"'re0‍s㍿­㋿'s
0½­fi\r\n>́", "tokens": 42, "pieces": ["#$%😀🏽", "𐞁A", "\r\n", "\"'", "re", "0", "‍s", "㍿­㋿'", "s", "
", "0½", "­fi", "\r\n", ">́"]} +{"text": "'se'M!,‍eع\r\n,9", "tokens": 11, "pieces": ["'s", "e", "'M", "!,‍", "eع", "\r\n", ",", "9"]} +{"text": "<|fim_prefix|>,EOT<…-dfie'reZå 912345678> 'Tfi'D​٣٤٥٦!!🙂EOT\r\n\r\n'ſfi­\r٣٤٥٦‍'re㍿", "tokens": 69, "pieces": ["<|", "fim", "_prefix", "|>,", "EOT", "<", "…", "-dfie", "'re", "Za", "̊", " ", " ", "912", "345", "678", ">", " ", "'T", "fi", "'D", "​", "٣٤٥", "٦", "!!🙂", "EOT", "\r\n\r\n", "'ſ", "fi", "­\r", "٣٤٥", "٦", "‍'", "re", "㍿"]} +{"text": "ع!'Dß", "tokens": 8, "pieces": ["ع", "!'", "Dß", ""]} +{"text": "漢tZ<'ReDž-㍿('Re0'M\u000b-fi½\r\n!!Z", "tokens": 24, "pieces": ["漢tZ", "<'", "ReDž", "-㍿('", "Re", "0", "'M", "\u000b", "-fi", "½", "\r\n", "!!", "Z"]} +{"text": "'s'ſs, -\r'séZDž<|fim_prefix|> \ns३a
👍🏽½!👍🏽's
!!\r㋿tİꟲ \nsåꟲ\u000b", "tokens": 64, "pieces": ["'s", "'ſ", "s", ",", " ", "-\r", "'s", "éZDž", "<|", "fim", "_prefix", "|>", " \n", "s", "३", "a", "
", "👍🏽", "½", "!👍🏽'", "s", "
", "!!\r", "㋿tİꟲ", " \n", "sa", "̊ꟲ", "\u000b"]} +{"text": "t‍!!fi 'VE", "tokens": 9, "pieces": ["t", "‍!!", "fi", " ", "'VE"]} +{"text": "㋿ \"'s​EOT#$%\ne", "tokens": 31, "pieces": ["'re", "عfi", ".­", "eعe", "́", "\tDžA", "'re", "Dž", "'S", "e", "!<", "EOT", ">'", "s", "​EOT", "#$%\n", "e"]} +{"text": "½Dž \n½,e'D٣٤٥٦𐞁\r\n\r\ne
<|endoftext|>\u000bſ 0<('Tfi", "tokens": 66, "pieces": ["½", "Dž", " \n", "½", ",e", "'D", "٣٤٥", "٦", "𐞁", "\r\n\r\n", "e", "
", "<|", "endoftext", "|>", "\u000bſ", " ", "0", "<('", "Tfi"]} +{"text": "​A ​٣٤٥٦́a're'VE‍ ㍿å0'VÉ9're!!'D'S0\"!½'Re", "tokens": 43, "pieces": ["​A", " ", "​", "٣٤٥", "٦", "́a", "'re", "'VE", "‍", " ", " ㍿", "a", "̊", "0", "'VE", "́", "9", "'re", "!!'", "D", "'S", "0", "\"!", "½", "'Re"]} +{"text": "'S½漢́ \r\n'S.,ḍ̇\"'SZ t<|fim_prefix|>😀🏽e'T㋿dé㋿Ⅳ9'ſ字fi!", "tokens": 54, "pieces": ["'S", "½", "漢", "́", " \r\n", "'S", ".,", "ḋ", "̣\"'", "SZ", " ", " t", "<|", "fim", "_prefix", "|>😀🏽", "e", "'T", "㋿dé", "㋿", "Ⅳ9", "'", "ſ字fi", "!"]} +{"text": "🙂fi‍',
é​ \n12345678𐞁…'VEA'Re", "tokens": 27, "pieces": ["🙂fi", "‍',", "
e", "́​", " \n", "123", "456", "78", "𐞁", "…", "'VE", "A", "'Re"]} +{"text": "'Re漢fi\"12345678'ſ'ſ<|endoftext|>㍿𐞁३字9\u000b\" ſ漢m'VEİ<|endoftext|>,'llé 9DžDž٣٤٥٦\n", "tokens": 71, "pieces": ["'Re", "漢fi", "\"", "123", "456", "78", "'ſ", "'ſ", "<|", "endoftext", "|>㍿", "𐞁", "३", "字", "9", "\u000b", "\"", " ſ漢m", "'VE", "İ", "<|", "endoftext", "|>,'", "llé", " ", "9", "DžDž", "٣٤٥", "٦", "\n"]} +{"text": "عꟲ<|endoftext|>é\r\nİDž'Re​ꟲ>-漢字ع́!!ß'ree \n'👍🏽­İß𐞁٣٤٥٦", "tokens": 54, "pieces": ["عꟲ", "<|", "endoftext", "|>", "é", "\r\n", "İDž", "'Re", "​ꟲ", ">-", "漢字ع", "́!!", "ß", "'re", "e", " \n", "'👍🏽­", "İß𐞁", "٣٤٥", "٦"]} +{"text": "…́㍿'Re 'llt ३ \nmZ''re'ſ", "tokens": 25, "pieces": ["…", "́㍿'", "Re", " '", "llt", " ", "३", " \n", "mZ", "'<", "EOT", ">'", "re", "'ſ"]} +{"text": " ㍿Dž <|fim_prefix|>m's٣٤٥٦-éDž ع‍fi\"٣٤٥٦'Re'𐞁𐞁,.\r\n'Re漢", "tokens": 55, "pieces": [" ", "㍿Dž", " <|", "fim", "_prefix", "|>", "m", "'s", "٣٤٥", "٦", "-éDž", " ع", "‍fi", "\"", "٣٤٥", "٦", "'Re", "'𐞁𐞁", ",.\r\n", "'Re", "漢"]} +{"text": "'ReEOT'llꟲDžé0\tfi\"🙂'S'ſ'DZés'll' tḍ̇.", "tokens": 39, "pieces": ["'Re", "EOT", "'ll", "ꟲDžé", "0", "\tfi", "\"🙂<", "EOT", ">'", "S", "'ſ", "'D", "Ze", "́s", "'ll", "'", " tḋ", "̣."]} +{"text": "\u000b<-<|fim_prefix|>#$%ſ'D'ſEOTAḍ̇
12345678!'Re'ſİ'VE're's'll'T0'S'Re\r\n<'re,>٣٤٥٦😀🏽½½\"\t…ع", "tokens": 71, "pieces": ["\u000b", "<-<|", "fim", "_prefix", "|>#$%", "ſ", "'D", "'ſ", "EOTAḋ", "̣", "
", "123", "456", "78", "!'", "Re", "'ſ", "İ", "'VE", "'re", "'s", "'ll", "'T", "0", "'S", "'Re", "\r\n", "<'", "re", ",>", "٣٤٥", "٦", "😀🏽", "½½", "\"", "\t", "…ع"]} +{"text": "9 9İ३ſ\n'ree<|fim_prefix|>'llḍ̇…𐞁9m>ع'D!!\nse‍​å12345678漢'ſ
.Ⅳ", "tokens": 56, "pieces": ["9", " ", "9", "İ", "३", "ſ", "\n", "'re", "e", "<|", "fim", "_prefix", "|>'", "llḋ", "̣", "…𐞁", "9", "m", ">ع", "'D", "!!\n", "se", "‍​", "a", "̊", "123", "456", "78", "漢", "'ſ", "
", ".", "Ⅳ"]} +{"text": "\n99İ​a'Re漢 éſ\r\n\r\nDž", "tokens": 15, "pieces": ["\n", "99", "İ", "​a", "'Re", "漢", " éſ", "\r\n\r\n", "Dž"]} +{"text": "́\"\r\n\r\n'VE.́Z٣٤٥٦😀🏽a<|endoftext|>\"'ſfié  'S\r\n\r\n's12345678'T‍eDžéع", "tokens": 56, "pieces": ["́\"\r\n\r\n", "'VE", ".́", "Z", "٣٤٥", "٦", "😀🏽", "a", "<|", "endoftext", "|>\"'", "ſfié", " ", " <", "META", "_START", ">", " ", " ", "'S", "\r\n\r\n", "'s", "123", "456", "78", "'T", "‍eDže", "́ع"]} +{"text": "٣٤٥٦<|fim_prefix|>''s㍿Ⅳ́  \n३ſ🙂\r\n\r\n\n㍿'s\r\nå", "tokens": 42, "pieces": ["٣٤٥", "٦", "<|", "fim", "_prefix", "|>''", "s", "㍿", "Ⅳ", "́", "  \n", "३", "ſ", "🙂\r\n\r\n\n", "㍿'", "s", "\r\n", "a", "̊"]} +{"text": "!३ \n", "tokens": 4, "pieces": ["!", "३", " \n"]} +{"text": "𐞁'Z㋿㋿Ⅳ‍!'Re  ", "tokens": 20, "pieces": ["𐞁", "'Z", "㋿㋿", "Ⅳ", "‍!'", "Re", "  "]} +{"text": "-
", "tokens": 3, "pieces": ["-", "
"]} +{"text": "‍­ꟲ,<㋿'s🙂", "tokens": 14, "pieces": ["‍­", "ꟲ", ",<㋿'", "s", "🙂"]} +{"text": "\rİⅣ漢0\"​éétḍ̇\"İ\u000b!!'reⅣ 12345678\t㋿0<|endoftext|>12345678…́\r\n\u000bſ\"!éⅣ-", "tokens": 58, "pieces": ["\r", "İ", "Ⅳ", "漢", "0", "\"​", "ée", "́tḋ", "̣\"", "İ", "\u000b", "!!'", "re", "Ⅳ", " ", "123", "456", "78", "\t", "㋿", "0", "<|", "endoftext", "|>", "123", "456", "78", "…", "́\r\n", "\u000bſ", "\"!", "e", "́", "Ⅳ", "-"]} +{"text": "a<|fim_prefix|>", "tokens": 8, "pieces": ["a", "<|", "fim", "_prefix", "|>"]} +{"text": "é>Ź👍🏽­'T‍ …ß<İ'st\r\né­㋿\r\n😀🏽Ⅳfie\u000b", "tokens": 45, "pieces": ["é", ">Z", "́👍🏽­'", "T", "‍", " ", "…ß", "<İ", "'s", "t", "\r\n", "e", "́­㋿\r\n", "😀🏽", "Ⅳ", "fie", "\u000b"]} +{"text": "s\"å!Ⅳİe\r\n\r\n𐞁9\r㋿½\n'M-é ­\u000b#$%", "tokens": 46, "pieces": ["s", "\"a", "̊!", "Ⅳ", "İe", "\r\n\r\n", "𐞁", "9", "\r", "㋿", "½", "\n", "'M", "-é", " ", " <", "s漢", "'S", " ", "'s", "­", " \n", "'M", "Džꟲ", "­>­", "\u000b", "#$%"]} +{"text": "…'re\n\r9Ⅳ \n㋿åå\u000b's'Tꟲ­ꟲ\n'T\r\n\r\n12345678 😀🏽's'll12345678e½", "tokens": 50, "pieces": ["…", "'re", "\n\r", "9Ⅳ", " \n", "㋿a", "̊a", "̊", "\u000b", "'s", "'T", "ꟲ", "­ꟲ", "\n", "'T", "\r\n\r\n", "123", "456", "78", " ", " 😀🏽<", "META", "_START", ">'", "s", "'ll", "123", "456", "78", "e", "½"]} +{"text": "字漢\u000b‍fit'VE\r'S'😀🏽ع…\t", "tokens": 23, "pieces": ["字漢", "\u000b", "‍fit", "'VE", "\r", "'S", "'😀🏽", "ع", "…\t"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": ".0३'MA३ḍ̇\r<|endoftext|>12345678
㍿t", "tokens": 31, "pieces": [".", "0३", "'M", "A", "३", "ḋ", "̣\r", "<|", "endoftext", "|>", "123", "456", "78", "
", "㍿t"]} +{"text": "\r\n\r\n'll㍿>\n字é<|endoftext|>\t\t12345678>-Dž", "tokens": 24, "pieces": ["\r\n\r\n", "'ll", "㍿>\n", "字e", "́<|", "endoftext", "|>", "\t", "\t", "123", "456", "78", ">-", "Dž"]} +{"text": "­ …!ḍ̇m𐞁'Tعſ\r\n\r\nſ́ 0m", "tokens": 33, "pieces": ["­", " ", "…", "!ḋ", "̣m𐞁", "'T", "عſ", "\r\n\r\n", "ſ", "́<", "EOT", ">", " ", "0", "m"]} +{"text": "字\"e9ع'M­tſ👍🏽>ßDž­", "tokens": 20, "pieces": ["字", "\"e", "9", "ع", "'M", "­tſ", "👍🏽>", "ßDž", "­"]} +{"text": "ع…A'D🙂'reEOT#$%! 'M'VE ٣٤٥٦'T‍-👍🏽#$%'Re\n\n 字", "tokens": 48, "pieces": ["ع", "…A", "'D", "🙂'", "reEOT", "#$%!<", "META", "_START", ">", " ", " '", "M", "'VE", " ", "٣٤٥", "٦", "'T", "‍-👍🏽#$%'", "Re", "\n\n", " ", " 字"]} +{"text": "'Tt𐞁ḍ̇ḍ̇३عEOT'Sſ'd.s​t🙂ݽ‍漢t<|fim_prefix|>", "tokens": 44, "pieces": ["'T", "t𐞁ḋ", "̣ḋ", "̣", "३", "عEOT", "'S", "ſ", "'d", ".s", "​t", "🙂İ", "½", "‍漢t", "<|", "fim", "_prefix", "|>"]} +{"text": " 😀🏽é'll!३'VEZs\r\n\r\n\t\n<|endoftext|>\n\r're", "tokens": 24, "pieces": [" 😀🏽", "é", "'ll", "!", "३", "'VE", "Zs", "\r\n\r\n\t\n", "<|", "endoftext", "|>\n\r", "'re"]} +{"text": "mdİa½a'VE-d​#$% ३", "tokens": 15, "pieces": ["mdİa", "½", "a", "'VE", "-d", "​#$%", " ", " ", "३"]} +{"text": "\u000b", "tokens": 1, "pieces": ["\u000b"]} +{"text": "😀🏽'VE㍿.--\u000b're㍿eé'S<|fim_prefix|>🙂𐞁'S-Áİ<|endoftext|>\u000b12345678 Z'Ms\r\n0ſ字", "tokens": 58, "pieces": ["😀🏽'", "VE", "㍿.--", "\u000b", "'re", "㍿eé", "'S", "<|", "fim", "_prefix", "|>🙂", "𐞁", "'S", "-A", "́İ", "<|", "endoftext", "|>", "\u000b", "123", "456", "78", " Z", "'M", "s", "\r\n", "0", "ſ字"]} +{"text": "'D́عe
ꟲt'Re\"­.!!‍‍é
#$%ḍ̇#$%ꟲ>s漢e…mEOTꟲ…", "tokens": 48, "pieces": ["'D", "́عe", "
ꟲt", "'Re", "\"­.!!‍‍", "é", "
", "#$%", "ḋ", "̣#$%", "ꟲ", ">s漢e", "…mEOTꟲ", "…"]} +{"text": "a0­\r\n\r\n (s(字mع t'ſ'T!!", "tokens": 18, "pieces": ["a", "0", "­\r\n\r\n", " ", " (", "s", "(字mع", " t", "'ſ", "'T", "!!"]} +{"text": "‍'sꟲßs!!٣٤٥٦0ꟲ!!́㍿ſ'D\r\n½!!\r\n é\t😀🏽́́'S
…fiⅣ", "tokens": 58, "pieces": ["‍'", "sꟲßs", "!!", "٣٤٥", "٦0", "ꟲ", "!!́㍿", "ſ", "'", "D", "\r\n", "½", "!!\r\n", " ", " e", "́", "\t", "😀🏽́́'", "S", "
", "…fi", "Ⅳ"]} +{"text": "\r\n\r\n<|fim_prefix|>\"‍-''", "tokens": 12, "pieces": ["\r\n\r\n", "<|", "fim", "_prefix", "|>\"‍-''"]} +{"text": " <|fim_prefix|>!!0ſ'Re'ſ<|endoftext|> \n\n ㋿'ſ'Reå'll\r\n9३\"-字.d<Ⅳå", "tokens": 51, "pieces": [" ", "<|", "fim", "_prefix", "|>!!", "0", "ſ", "'Re", "'ſ", "<|", "endoftext", "|>", " \n\n", " ", " ㋿'", "ſ", "'Re", "a", "̊'", "ll", "\r\n", "9३", "\"-", "字", ".d", "<", "Ⅳ", "a", "̊"]} +{"text": "(A <|fim_prefix|>🙂'll,漢 å㋿'VE9'VEDž.𐞁​A\"\r'se", "tokens": 58, "pieces": ["(A", "", " ", "<|", "fim", "_prefix", "|><", "META", "_START", ">🙂<", "EOT", ">'", "ll", ",漢", " ", " a", "̊㋿'", "VE", "9", "'VE", "Dž", ".𐞁", "​<", "META", "_START", ">A", "\"\r", "'s", "e"]} +{"text": "e字'ſ㋿'ſſḍ̇Ⅳt字ſ'Ś…😀🏽\u000b\nİa\n​🙂'M,٣٤٥٦fi \u000b漢", "tokens": 64, "pieces": ["e字", "'ſ", "㋿'", "ſſḋ", "̣", "Ⅳ", "t字", "ſ", "'S", "́", "…", "😀🏽", "\u000b\n", "İa", "\n", "​🙂'", "M", ",", "٣٤٥", "٦", "fi", " ", "\u000b漢"]} +{"text": " '\u000bḍ̇'llß👍🏽­㋿Ⅳſ𐞁漢½漢­३#$%(😀🏽\"½ſⅣDž'VE'VEé a", "tokens": 63, "pieces": [" ", "'", "\u000bḋ", "̣'", "llß", "👍🏽­㋿", "Ⅳ", "ſ", "𐞁漢", "½", "漢", "­", "३", "#$%(😀🏽\"", "½", "ſ", "Ⅳ", "Dž", "'VE", "'VE", "é", " ", " a"]} +{"text": "<|fim_prefix|>!!ß're'reDž#$%12345678\rmå㋿'D\nḍ̇-12345678fi㍿'Re​12345678é\t", "tokens": 50, "pieces": ["<|", "fim", "_prefix", "|>!!", "ß", "'re", "'re", "Dž", "#$%", "123", "456", "78", "\r", "ma", "̊㋿'", "D", "\n", "ḋ", "̣-", "123", "456", "78", "fi", "㍿'", "Re", "​", "123", "456", "78", "é", "\t"]} +{"text": "fi㋿s'ſ́'T'<|fim_prefix|>-İe'll Dž߅'Re ㍿EOT #$%㍿…\u000bZ'VE'S
ſ\r\n\r\n (", "tokens": 60, "pieces": ["fi", "㋿s", "'ſ", "́'", "T", "'<|", "fim", "_prefix", "|>-<", "META", "_START", ">İe", "'ll", " Džß", "…", "'Re", " ", " ㍿", "EOT", " ", " #$%㍿", "…", "\u000bZ", "'VE", "'S", "
ſ", "\r\n\r\n", " ", "("]} +{"text": " 🙂#$%#$%9'VE!!'VE…eDž㋿d\"'SEOT 漢're
'#$%!'Ret0\u000b'S EOT字åعa912345678\r\n\r\n\u000béDž<|fim_prefix|>", "tokens": 27, "pieces": ["e", "́<", "META", "_START", "><", "META", "_START", ">a", "912", "345", "678", "\r\n\r\n", "\u000bé", "Dž", "<|", "fim", "_prefix", "|>"]} +{"text": "👍🏽\r0s३🙂漢ée \n𐞁DžsⅣ(\n'ſ'll½,ſ<|fim_prefix|>İ́>ḍ̇<👍🏽!!'ll'S٣٤٥٦字'VE>", "tokens": 73, "pieces": ["👍🏽\r", "0", "s", "३", "🙂漢ée", " \n", "𐞁Džs", "Ⅳ", "(\n", "'ſ", "'ll", "½", ",ſ", "<|", "fim", "_prefix", "|>", "İ", "́>", "ḋ", "̣<👍🏽!!'", "ll", "'S", "٣٤٥", "٦", "字", "'VE", ">"]} +{"text": "\n!m- \n'll'Tt'VE…Dž👍🏽ع३'ſß\r\nſ å漢漢\u000bA! ́㋿0ع", "tokens": 55, "pieces": ["\n", "!m", "-", " \n", "'", "ll", "'T", "t", "'VE", "…Dž", "👍🏽", "ع", "३", "'ſ", "ß", "\r\n", "ſ", " a", "̊漢漢", "\u000bA", "!", " ", " ́㋿", "0", "ع"]} +{"text": " 'T!!'sDž", "tokens": 7, "pieces": [" '", "T", "!!'", "sDž"]} +{"text": "…\u000b's Ⅳß", "tokens": 11, "pieces": ["…", "\u000b", "'s", "", " ", "Ⅳ", "ß"]} +{"text": "#$%Ⅳ\r\n9 Ⅳ0'll𐞁
,'M'll‍'Dḍ̇Dž㍿٣٤٥٦ \n३,Z🙂é\r\n\r\ns", "tokens": 55, "pieces": ["#$%", "Ⅳ", "\r\n", "9", " ", " <", "META", "_START", ">", "Ⅳ0", "'ll", "𐞁", "
", ",'", "M", "'ll", "‍'", "Dḋ", "̣Dž", "㍿", "٣٤٥", "٦", " \n", "३", ",Z", "🙂é", "\r\n\r\n", "s"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿'ll<|fim_prefix|>s're're(<'S'Re㍿ A​'D", "tokens": 30, "pieces": ["㍿'", "ll", "<|", "fim", "_prefix", "|>", "s", "'re", "'re", "(<'", "S", "'Re", "㍿", " A", "​'", "D"]} +{"text": " 'S('ſ……'D'll🙂'٣٤٥٦­🙂<|endoftext|>​t <|fim_prefix|>> #$%‍ßⅣ're\r
9३'D'Mİ́漢s", "tokens": 64, "pieces": [" '", "S", "('", "ſ", "…", "…", "'D", "'ll", "🙂'", "٣٤٥", "٦", "­🙂<|", "endoftext", "|>​", "t", " ", "<|", "fim", "_prefix", "|>>", " ", "#$%‍", "ß", "Ⅳ", "'re", "\r", "
", "9३", "'D", "'M", "İ", "́漢s"]} +{"text": "👍🏽s​<|endoftext|>'S \nm!!å٣٤٥٦\"fi", "tokens": 36, "pieces": ["👍🏽", "s", "​<|", "endoftext", "|>'", "S", " \n", "m", "!!", "a", "̊", "٣٤٥", "٦", "\"<", "META", "_START", ">fi"]} +{"text": "́'Réd-𐞁\t㍿mdé'́
é३", "tokens": 24, "pieces": ["́'", "Re", "́d", "-𐞁", "\t", "㍿mdé", "'́", "
e", "́", "३"]} +{"text": "ß​'D<'D\"🙂fi 'M…é‍ \n12345678𐞁字㍿ \n9ß😀🏽té0fi३ḍ̇'VE", "tokens": 56, "pieces": ["ß", "​'", "D", "<'", "D", "\"🙂", "fi", " ", " '", "M", "…é", "‍", " \n", "123", "456", "78", "𐞁", "字", "㍿", " \n", "9", "ß", "😀🏽", "te", "́", "0", "fi", "३", "ḋ", "̣'", "VE"]} +{"text": "½-ع12345678\n漢é'T'S\u000b𐞁dåå🙂عİ\r 
'ſḍ̇.'VE
 ", "tokens": 46, "pieces": ["½", "-ع", "123", "456", "78", "\n", "漢e", "́'", "T", "'S", "\u000b𐞁da", "̊a", "̊🙂", "عİ", "\r", " ", "
", "'ſ", "ḋ", "̣.'", "VE", "
 "]} +{"text": ".😀🏽'Sm're'ſ<|fim_prefix|>Aḍ̇.­.0'llⅣ<ß.<|endoftext|>漢𐞁😀🏽'res\u000b३‍", "tokens": 60, "pieces": [".😀🏽'", "Sm", "'re", "'ſ", "<|", "fim", "_prefix", "|>", "Aḋ", "̣.­.", "0", "'ll", "Ⅳ", "<ß", ".<|", "endoftext", "|>", "漢𐞁", "😀🏽'", "res", "\u000b", "३", "‍"]} +{"text": "Z\u000b", "tokens": 2, "pieces": ["Z", "\u000b"]} +{"text": "s­\r\n'Mm…EOTa'VE😀🏽- \n…𐞁<|endoftext|>é'Tm㍿ 'Ret<|fim_prefix|>12345678'T३🙂!é'llEOT㋿s'VE\r\n\r\n​", "tokens": 72, "pieces": ["s", "­\r\n", "'M", "m", "…EOT", "a", "'VE", "😀🏽-", " \n", "…𐞁", "<|", "endoftext", "|>", "e", "́'", "Tm", "㍿", " '", "Ret", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'T", "३", "🙂!", "é", "'ll", "EOT", "㋿s", "'VE", "\r\n\r\n", "​"]} +{"text": "字 0,\u000bZé('VE's\"\r\n\r\nꟲ漢'ReDž\nfiع㍿12345678(㍿", "tokens": 36, "pieces": ["字", " ", "0", ",", "\u000bZe", "́('", "VE", "'s", "\"\r\n\r\n", "ꟲ漢", "'", "ReDž", "\n", "fiع", "㍿", "123", "456", "78", "(㍿"]} +{"text": "'Ms-09ḍ̇DžⅣ d0'<ſß \n", "tokens": 25, "pieces": ["'M", "s", "-", "09", "ḋ", "̣Dž", "Ⅳ", " ", " d", "0", "'<<", "META", "_START", ">ſß", " \n"]} +{"text": "'ſ𐞁\r're\r\n\r\n\n", "tokens": 16, "pieces": ["'ſ", "𐞁", "\r", "'", "re", "\r\n\r\n\n"]} +{"text": "'VE eéA'Sa😀🏽
Ⅳ-'M𐞁,  >'VEtd½\t'Mḍ̇‍\r\n\r\n !ꟲ \t३", "tokens": 53, "pieces": ["'VE", " e", "e", "́A", "'S", "a", "😀🏽", "
", "Ⅳ", "-'", "M𐞁", ",", " ", " ", ">'", "VEtd", "½", "\t", "'M", "ḋ", "̣‍\r\n\r\n", " ", " !", "ꟲ", " ", "\t", "३"]} +{"text": "İ'S0
", "tokens": 5, "pieces": ["İ", "'S", "0", "
"]} +{"text": "<|fim_prefix|><|fim_prefix|>ß 's'Sm😀🏽fi'Tḍ̇\u000bs\"‍'ſ \nmEOT‍", "tokens": 49, "pieces": ["<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", "ß", " ", " '", "s", "'S", "m", "😀🏽", "fi", "'T", "ḋ", "̣", "\u000bs", "\"‍'", "ſ", " \n", "mEOT", "‍"]} +{"text": "'Reḍ̇'s!𐞁d'VEſ👍🏽㍿a('St'ſ㋿𐞁  <|fim_prefix|>\r\n½Dž!!字<|endoftext|>'re字12345678'Reaa aEOT", "tokens": 72, "pieces": ["'Re", "ḋ", "̣'", "s", "!𐞁d", "'VE", "ſ", "👍🏽㍿", "a", "('", "St", "'ſ", "㋿𐞁", " ", " ", "<|", "fim", "_prefix", "|>\r\n", "½", "Dž", "!!", "字", "<|", "endoftext", "|>'", "re字", "123", "456", "78", "'Re", "aa", " ", " aEOT"]} +{"text": "'VE­s\"'M\r\n\r\nZ\r\n\r\n'VE'D9ꟲ", "tokens": 23, "pieces": ["'VE", "­s", "\"'", "M", "\r\n\r\n", "Z", "\r\n\r\n", "'VE", "'D", "9", "ꟲ", ""]} +{"text": ".d🙂", "tokens": 3, "pieces": [".d", "🙂"]} +{"text": "EOTEOTEOTß<|fim_prefix|>㋿ ́fi‍\nİ!! 
½ß", "tokens": 33, "pieces": ["EOTEOTEOTß", "<|", "fim", "_prefix", "|>㋿", " ", " ́", "fi", "‍\n", "İ", "!!", " ", "
", "½", "ß"]} +{"text": "'sdDžsſ ㍿ \n🙂\t!'T\u000bse.‍\n<|fim_prefix|>🙂ſ㍿ ­'D'S'Re٣٤٥٦Ⅳ'S", "tokens": 55, "pieces": ["'s", "dDžsſ", " ", " ㍿", " \n", "🙂", "\t", "!'", "T", "\u000bse", ".‍\n", "<|", "fim", "_prefix", "|>🙂", "ſ", "㍿", " ", " ­'", "D", "'S", "'Re", "٣٤٥", "٦Ⅳ", "'S"]} +{"text": "'ll😀🏽eEOT. \n \r'sعع́\u000b\rⅣé", "tokens": 23, "pieces": ["'ll", "😀🏽", "eEOT", ".", " \n \r", "'s", "عع", "́", "\u000b\r", "Ⅳ", "e", "́"]} +{"text": "<|endoftext|>😀🏽\r\n!İ\t \n'T'#$%m ­😀🏽m9<mꟲå\t \ne\n㍿", "tokens": 50, "pieces": ["<|", "endoftext", "|>😀🏽\r\n", "!İ", "\t \n", "'T", "'#$%", "m", " ", " ­😀🏽", "m", "9", "<<", "EOT", ">mꟲa", "̊", "\t \n", "e", "\n", "㍿"]} +{"text": "'reع\tfi''MEOT㍿9\r\né'T‍eع\r\n\r\n😀🏽<|fim_prefix|>\r\n>​\n\"", "tokens": 43, "pieces": ["'re", "ع", "\tfi", "''", "MEOT", "㍿", "9", "\r\n", "é", "'T", "‍eع", "\r\n\r\n", "😀🏽<|", "fim", "_prefix", "|><", "EOT", ">\r\n", ">​\n", "\""]} +{"text": "'ſ>\t", "tokens": 5, "pieces": ["'ſ", ">", "\t"]} +{"text": "'ſ\n𐞁'Re,-<|endoftext|>'VEſ…<|fim_prefix|>EOT​'Mſ'S0٣٤٥٦12345678👍🏽'llİ12345678!!\tſé'ſ‍漢'ſå\u000b're'T'D\t㍿½", "tokens": 89, "pieces": ["'ſ", "\n", "𐞁", "'Re", ",-<|", "endoftext", "|>'", "VEſ", "…", "<|", "fim", "_prefix", "|>", "EOT", "​'", "Mſ", "'S", "0٣٤", "٥٦1", "234", "567", "8", "👍🏽'", "llİ", "123", "456", "78", "!!", "\tſé", "'ſ", "‍漢", "'ſ", "a", "̊", "\u000b", "'re", "'T", "'D", "\t", "㍿", "½"]} +{"text": "㋿𐞁é'VE­­-å‍'漢,EOT字A३<|fim_prefix|>​🙂🙂'VE''s…t​'re'St㍿ !!", "tokens": 61, "pieces": ["㋿𐞁é", "'VE", "­­-", "a", "̊<", "EOT", ">‍'", "漢", ",EOT字A", "३", "<|", "fim", "_prefix", "|>​🙂🙂'", "VE", "''", "s", "…t", "​'", "re", "'S", "t", "㍿", " ", "!!"]} +{"text": "'ll字", "tokens": 2, "pieces": ["'ll", "字"]} +{"text": "\t㍿s㋿#$%12345678,e#$%'T'Re٣٤٥٦😀🏽👍🏽s
!", "tokens": 41, "pieces": ["\t", "㍿s", "㋿#$%", "123", "456", "78", ",e", "#$%'", "T", "'Re", "٣٤٥", "٦", "😀🏽👍🏽", "s", "
", "!"]} +{"text": "t'ſ‍\t🙂éd\r㋿😀🏽a\r\n\r\n9 'llA'll-(0‍٣٤٥٦­ 字#$%(å'ſ
Ⅳ'SⅣa", "tokens": 68, "pieces": ["t", "'ſ", "‍", "\t", "🙂éd", "\r", "㋿😀🏽", "a", "\r\n\r\n", "9", " ", "'ll", "A", "'ll", "-(", "0", "‍", "٣٤٥", "٦", "­", " 字", "#$%<", "META", "_START", ">(", "a", "̊'", "ſ", "
", "Ⅳ", "'S", "Ⅳ", "a"]} +{"text": "'\r\n\r​ḍ̇­a. EOT'VE­\r\n\r\n'Re​㋿ 👍🏽<|fim_prefix|>>㋿>\r\n\r\n\r\n!Dž'VE 'llſ'ré\t'ſ'sḍ̇,", "tokens": 65, "pieces": ["'\r\n\r", "​ḋ", "̣­", "a", ".", " ", " EOT", "'VE", "­\r\n\r\n", "'Re", "​㋿", " ", " 👍🏽<|", "fim", "_prefix", "|>>㋿>\r\n\r\n\r\n", "!Dž", "'", "VE", " ", "'ll", "ſ", "'re", "́", "\t", "'ſ", "'s", "ḋ", "̣,"]} +{"text": "👍🏽AéEOT\"Z\ré́\u000bAعaZ𐞁<|fim_prefix|>9d३", "tokens": 38, "pieces": ["👍🏽", "Ae", "́EOT", "\"Z", "\r", "é", "́", "\u000bAعaZ𐞁", "<|", "fim", "_prefix", "|>", "9", "d", "३"]} +{"text": "9‍'T字'reⅣt'Re'Dꟲå0'reꟲé's.ع\"'ll字㍿>", "tokens": 36, "pieces": ["9", "‍'", "T字", "'re", "Ⅳ", "t", "'Re", "'D", "ꟲa", "̊", "0", "'re", "ꟲe", "́'", "s", ".ع", "\"'", "ll字", "㍿>"]} +{"text": "#$%㋿é", "tokens": 6, "pieces": ["#$%㋿", "é"]} +{"text": "é🙂‍'re
å'VE३s.…'T漢.!#$%12345678㋿٣٤٥٦m㋿#$%\r", "tokens": 56, "pieces": ["e", "́🙂‍'", "re", "", "
a", "̊'", "VE", "३", "s", ".", "…", "'T", "漢", ".!#$%", "123", "456", "78", "㋿", "٣٤٥", "٦", "m", "㋿<", "EOT", ">#$%\r"]} +{"text": "👍🏽!!", "tokens": 7, "pieces": ["👍🏽!!"]} +{"text": "'sḍ̇Z\r\n\r\n½İ🙂\r\n\r\n'Tß", "tokens": 33, "pieces": ["ße", "0", "Aꟲ", "'D", "𐞁", "\u000b", "
", "Ⅳ", "'S", "<'", "Re", "-s", "<|", "fim", "_prefix", "|>🙂\r\n\r\n", "'T", "ß"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'S>éZſ<|endoftext|>\">३-ع#$%Z!'VE", "tokens": 23, "pieces": ["'S", ">éZſ", "<|", "endoftext", "|>\">", "३", "-ع", "#$%", "Z", "!'", "VE"]} +{"text": "d'ſ<\r🙂<|endoftext|>\nǻ#$%ع́ \n\n½\tå'Re‍🙂ḍ̇'S,m👍🏽EOTfi​㍿\u000b😀🏽<|fim_prefix|>", "tokens": 69, "pieces": ["d", "'ſ", "<\r", "🙂<|", "endoftext", "|>\n", "a", "̊́#$%", "ع", "́", " \n\n", "½", "\ta", "̊'", "Re", "‍🙂", "ḋ", "̣'", "S", ",m", "👍🏽", "EOTfi", "​㍿", "\u000b", "😀🏽<|", "fim", "_prefix", "|>"]} +{"text": "e…é'VE9漢é!!<|fim_prefix|>", "tokens": 19, "pieces": ["e", "…é", "'VE", "9", "漢e", "́!!<|", "fim", "_prefix", "|>"]} +{"text": "ḍ̇
\t३", "tokens": 10, "pieces": ["ḋ", "̣", "
", "\t", "३"]} +{"text": "\t 字é㍿İ<|endoftext|>'VEḍ̇Ⅳ", "tokens": 28, "pieces": ["\t", " 字e", "́㍿<", "EOT", ">İ", "<|", "endoftext", "|>'", "VEḋ", "̣", "Ⅳ"]} +{"text": "Ⅳ​😀🏽İ\"'D🙂'T​9d\råe12345678\r\n\r\n \nع's", "tokens": 38, "pieces": ["Ⅳ", "​😀🏽", "İ", "\"'", "D", "🙂'", "T", "​<", "EOT", ">", "9", "d", "\r", "a", "̊e", "123", "456", "78", "\r\n\r\n \n", "ع", "'s"]} +{"text": "<|endoftext|>👍🏽
́\"Z0عZꟲ'Re🙂…<٣٤٥٦'MDž㍿'D½<|fim_prefix|>(ⅣZd́ ٣٤٥٦㋿ß'Ma'\r\n'ReⅣ㍿\r\n\r\n.", "tokens": 83, "pieces": ["<|", "endoftext", "|>👍🏽", "
", "́\"", "Z", "0", "عZꟲ", "'Re", "🙂", "…", "<", "٣٤٥", "٦", "'M", "Dž", "㍿'", "D", "½", "<|", "fim", "_prefix", "|>(", "Ⅳ", "Zd", "́", " ", "٣٤٥", "٦", "㋿ß", "'M", "a", "'\r\n", "'Re", "Ⅳ", "㍿\r\n\r\n", "."]} +{"text": "​9ſA㋿\r\n\r\n😀🏽Dž字'VE\tİ'reḍ̇#$%<|endoftext|>\"#$%‍\n", "tokens": 45, "pieces": ["​", "9", "ſA", "㋿\r\n\r\n", "😀🏽", "Dž字", "'VE", "\tİ", "'re", "ḋ", "̣#$%<|", "endoftext", "|><", "META", "_START", ">\"#$%‍\n"]} +{"text": "(<|fim_prefix|>", "tokens": 7, "pieces": ["(<|", "fim", "_prefix", "|>"]} +{"text": "ſ'ſ'T­Z. \n
<|endoftext|>!!३9m٣٤٥٦9 \r\n𐞁
ßſ's'३'ſ
'VEḍ̇", "tokens": 59, "pieces": ["ſ", "'ſ", "'T", "­Z", ".", " \n", "
", "<|", "endoftext", "|>!!", "३9", "m", "٣٤٥", "٦9", " \r\n", "𐞁", "
ßſ", "'s", "'", "३", "'ſ", "
", "'VE", "ḋ", "̣"]} +{"text": "'s३'S \n ſ!'D­🙂㋿\rḍ̇e#$% \n,ss '", "tokens": 28, "pieces": ["'s", "३", "'S", " \n", " ſ", "!'", "D", "­🙂㋿\r", "ḋ", "̣e", "#$%", " \n", ",ss", " '"]} +{"text": "'VE\u000b\t\t漢\u000b<|endoftext|>åꟲ😀🏽.'sm's!!'ſ́३", "tokens": 36, "pieces": ["'VE", "\u000b\t", "\t漢", "\u000b", "<|", "endoftext", "|>", "a", "̊ꟲ", "😀🏽.'", "sm", "'s", "!!'", "ſ", "́", "३"]} +{"text": ".\"<\t'S> ſ!!eå👍🏽'M'Re!😀🏽 \n!!…ß", "tokens": 48, "pieces": [".\"<", "\t", "'S", ">", " ſ", "!!", "ea", "̊👍🏽'", "M", "'Re", "!😀🏽", " \n", "!!", "…ß"]} +{"text": "!!'ſ'ſ\r\n->'ś'D0é0㍿<'ſ'VE漢\"é­", "tokens": 34, "pieces": ["!!<", "EOT", ">'", "ſ", "'ſ", "\r\n", "->'", "s", "́'", "D", "0", "e", "́", "0", "㍿<'", "ſ", "'VE", "漢", "\"é", "­"]} +{"text": "12345678ſ'M\n'T㍿d\t", "tokens": 13, "pieces": ["123", "456", "78", "ſ", "'M", "\n", "'T", "㍿d", "\t"]} +{"text": " ſfi🙂0
½0\u000b
\u000b ‍\r\nⅣ
𐞁12345678'VE😀🏽e'S>fi\r", "tokens": 48, "pieces": [" ", " ſfi", "🙂", "0", "
", "½0", "\u000b
\u000b ", " ‍\r\n", "Ⅳ", "
𐞁", "123", "456", "78", "'VE", "😀🏽", "e", "'S", ">fi", "\r"]} +{"text": "'VEå(ḍ̇0#$%
's Aꟲ", "tokens": 22, "pieces": ["'VE", "a", "̊(", "ḋ", "̣", "0", "#$%", "
", "'s", " Aꟲ"]} +{"text": "Dž're
漢ß(\" (é(åfi㍿'VE", "tokens": 27, "pieces": ["Dž", "'re", "
漢", "ß", "(\"", " ", "(e", "́(", "a", "̊fi", "㍿'", "VE"]} +{"text": "<|endoftext|>m​㋿", "tokens": 12, "pieces": ["<|", "endoftext", "|>", "m", "​㋿"]} +{"text": "ḍ̇#$% \nt'e㍿\n \n'", "tokens": 22, "pieces": ["ḋ", "̣#$%", " \n", "t", "'", "e", "㍿\n", " \n", "'"]} +{"text": "­́ع0'''ſsḍ̇", "tokens": 46, "pieces": ["­́<", "META", "_START", ">ع", "0", "'''", "ſsḋ", "̣"]} +{"text": "'T'D'SꟲaDž字字'Mmfim ́ß\"
eعm!…m!٣٤٥٦‍s\r\n\r\n0👍🏽A", "tokens": 67, "pieces": ["'T", "'D", "'S", "ꟲaDž字字", "'M", "aDž", "mfim", " <", "META", "_START", ">́", "ß", "\"", "
eعm", "!", "…m", "!", "٣٤٥", "٦", "‍s", "\r\n\r\n", "0", "👍🏽", "A"]} +{"text": "'Reḍ̇ß𐞁Ⅳ'ſ \n'De­ع0ḍ̇.'\r\nå\u000b ", "tokens": 34, "pieces": ["'Re", "ḋ", "̣ß𐞁", "Ⅳ", "'ſ", " \n", "'D", "e", "­ع", "0", "ḋ", "̣.'\r\n", "a", "̊", "\u000b "]} +{"text": "İ's\nⅣ12345678Džd9\r\n
's<|fim_prefix|>(Dž
'Re'VE'll ß'M‍\n''s漢é३'ſaéⅣḍ̇EOT", "tokens": 62, "pieces": ["İ", "'s", "\n", "Ⅳ12", "345", "678", "Džd", "9", "\r\n", "
", "'s", "<|", "fim", "_prefix", "|>(", "Dž", "
", "'Re", "'VE", "'ll", " ß", "'M", "‍\n", "''", "s漢e", "́", "३", "'ſ", "ae", "́", "Ⅳ", "ḋ", "̣<", "EOT", ">EOT"]} +{"text": "amé'VEſZ𐞁'M🙂.e字", "tokens": 17, "pieces": ["ame", "́'", "VEſZ𐞁", "'M", "🙂.", "e字"]} +{"text": "ꟲ0s.​EOTꟲ\r\n\r\n\r\nع字 0Z 🙂ſ<|fim_prefix|> \nfi👍🏽İ,A\nİt!å‍", "tokens": 52, "pieces": ["ꟲ", "0", "s", ".​", "EOTꟲ", "\r\n\r\n\r\n", "ع字", " ", "0", "Z", " ", "🙂ſ", "<|", "fim", "_prefix", "|>", " \n", "fi", "👍🏽", "İ", ",A", "\n", "İt", "!a", "̊‍"]} +{"text": "'T9A!'D𐞁's
\"٣٤٥٦\t٣٤٥٦'M'T<<'Så३'re­ .EOT­' eꟲm<|endoftext|> ,'Så", "tokens": 78, "pieces": ["a", "̊", " ", " '", "D", "'re", "\r\n\r\n", "漢", "<|", "fim", "_prefix", "|>'", "s", "
", "\"", "٣٤٥", "٦", "\t", "٣٤٥", "٦", "'M", "'T", "<<'", "Sa", "̊", "३", "'re", "­", " ", ".EOT", "­<", "EOT", ">'", " eꟲm", "<|", "endoftext", "|>", " ", ",'", "Sa", "̊"]} +{"text": "\r🙂'Tß字#$%\r\n\r\n\t㍿\r\n\r\n.\u000bİ㋿
12345678'T", "tokens": 27, "pieces": ["\r", "🙂'", "Tß字", "#$%\r\n\r\n", "\t", "㍿\r\n\r\n", ".", "\u000bİ", "㋿", "
", "123", "456", "78", "'T"]} +{"text": "'M
 ß!ſ'Red('ll­\n'S!!Ⅳs#$%.'ss'VE­
'''Ḿa're'M'D", "tokens": 43, "pieces": ["'M", "
", " ß", "!ſ", "'Re", "d", "('", "ll", "­\n", "'S", "!!", "Ⅳ", "s", "#$%.'", "s", "s", "'VE", "­", "
", "'''", "M", "́", "a", "'re", "'M", "'D"]} +{"text": "<|endoftext|><字fiŹ\n12345678㍿EOT#$%Džé👍🏽
\n'S…'D'M'ſ​ع\n<‍", "tokens": 49, "pieces": ["<|", "endoftext", "|><", "字fiZ", "́\n", "123", "456", "78", "㍿EOT", "#$%", "Džé", "👍🏽", "
\n", "'S", "…", "'D", "'M", "'ſ", "​ع", "\n", "<‍"]} +{"text": "<|endoftext|>ß\r\nİ#$%9ß'ſ'T½", "tokens": 19, "pieces": ["<|", "endoftext", "|>", "ß", "\r\n", "İ", "#$%", "9", "ß", "'ſ", "'T", "½"]} +{"text": "9,\tſé٣٤٥٦é'reſ'ſZDžsİDž'S-'s漢0!
're<|endoftext|>𐞁's
é٣٤٥٦㍿!!\r\n 's", "tokens": 71, "pieces": ["9", ",", "\tſé", "٣٤٥", "٦", "é", "'re", "ſ", "'ſ", "ZDžsİDž", "'S", "-'", "s漢", "", "0", "!", "
", "'re", "<|", "endoftext", "|>", "𐞁", "'s", "
é", "٣٤٥", "٦", "㍿!!\r\n", " ", "'s"]} +{"text": "½m½́🙂Ze \nt‍́ 
sꟲå‍'Reé", "tokens": 27, "pieces": ["½", "m", "½", "́🙂", "Ze", " \n", "t", "‍́", " ", "
sꟲa", "̊‍'", "Reé"]} +{"text": "12345678 \n'Re漢(.e字're‍#$%'VE!!aEOT😀🏽­'S🙂\"'VE\re漢\r\n\r\neZ", "tokens": 39, "pieces": ["123", "456", "78", " \n", "'Re", "漢", "(.", "e字", "'re", "‍#$%'", "VE", "!!", "aEOT", "😀🏽­'", "S", "🙂\"'", "VE", "\r", "e漢", "\r\n\r\n", "eZ"]} +{"text": "'ll字,", "tokens": 3, "pieces": ["'ll", "字", ","]} +{"text": "s,'M\rſå", "tokens": 9, "pieces": ["s", ",'", "M", "\r", "ſa", "̊"]} +{"text": "s'\"'ſeع …<'Reꟲ'T­é\r\n\r\n", "tokens": 20, "pieces": ["s", "'\"'", "ſeع", " ", "…", "<'", "Reꟲ", "'T", "­e", "́\r\n\r\n"]} +{"text": "😀🏽#$%\r\n…12345678'Re<\".漢m٣٤٥٦عå'ſe\u000bt12345678å", "tokens": 48, "pieces": ["😀🏽#$%\r\n", "…", "", "123", "456", "78", "'Re", "<\".", "漢m", "٣٤٥", "٦", "ع", "a", "̊'", "ſe", "\u000bt", "123", "456", "78", "a", "̊"]} +{"text": "'VE'lla  \r\n\r\n'llt<", "tokens": 9, "pieces": ["'VE", "'ll", "a", "  \r\n\r\n", "'ll", "t", "<"]} +{"text": "éé­İ'S‍<", "tokens": 9, "pieces": ["e", "́é", "­İ", "'S", "‍<"]} +{"text": "'ll'st漢EOT​Aeİ𐞁0éḍ̇EOT­", "tokens": 27, "pieces": ["'ll", "'s", "t漢EOT", "​Aeİ𐞁", "0", "e", "́ḋ", "̣EOT", "­"]} +{"text": ">\n𐞁sⅣ𐞁#$%t'VE#$%<|endoftext|>-<|endoftext|>ſ­>'D­३d'VE\t👍🏽𐞁tꟲ\n12345678'D#$%ⅣZ'll㍿ 漢a'M", "tokens": 82, "pieces": [">\n", "𐞁s", "Ⅳ", "𐞁", "#$%", "t", "'VE", "#$%<|", "endoftext", "|>-<|", "endoftext", "|>", "ſ", "­>'", "D", "­", "३", "d", "'VE", "\t", "👍🏽", "𐞁tꟲ", "\n", "123", "456", "78", "'D", "#$%", "Ⅳ", "Z", "'ll", "㍿", " 漢a", "'M"]} +{"text": "'Ḿ's \né! ,'T
's \"'D9ع'llDžEOT­A ½漢…", "tokens": 35, "pieces": ["'M", "́'", "s", " \n", "é", "!", " ,'", "T", "
", "'s", " ", "\"'", "D", "9", "ع", "'ll", "DžEOT", "­A", " ", "½", "漢", "…"]} +{"text": "ḍ̇ſe\t \n߅'re're'Sfi'عİ­​​'ll🙂ḍ̇㍿'lle.><|endoftext|>(0,漢ḍ̇½e​\u000b", "tokens": 57, "pieces": ["ḋ", "̣ſe", "\t \n", "ß", "…", "'re", "'re", "'S", "fi", "'عİ", "­​​'", "ll", "🙂ḋ", "̣㍿'", "lle", ".><|", "endoftext", "|>(", "0", ",漢ḋ", "̣", "½", "e", "​", "\u000b"]} +{"text": "३ßZ字0́'Re'T'ſſ12345678३>'remé\r\n­…fit \n!İ….\n'ſ#$% ", "tokens": 42, "pieces": ["३", "ßZ字", "0", "́'", "Re", "'T", "'ſ", "ſ", "123", "456", "78३", ">'", "remé", "\r\n", "­", "…fit", " \n", "!İ", "…", ".\n", "'ſ", "#$%", " "]} +{"text": "'VE字<|endoftext|><|endoftext|>\nAع㋿", "tokens": 22, "pieces": ["'VE", "字", "<|", "endoftext", "|><|", "endoftext", "|>\n", "Aع", "㋿"]} +{"text": "!é \nⅣa!­\"!m\r\n#$%'字𐞁'ع㍿<|endoftext|>'se( ꟲḍ̇Džd漢!. 'S!!d", "tokens": 57, "pieces": ["!e", "́", " \n", "Ⅳ", "a", "!­\"!", "m", "\r\n", "#$%'", "字𐞁", "'ع", "㍿<|", "endoftext", "|>'", "se", "(", " ꟲḋ", "̣Džd漢", "!.", " ", " '", "S", "!!", "d", ""]} +{"text": "!'SDžİⅣ\r\n\r\n\r!㋿-é\ntéé🙂İ​\r\nſ👍🏽!‍​㋿
३e‍ a", "tokens": 53, "pieces": ["!'", "SDžİ", "Ⅳ", "\r\n\r\n\r", "!㋿<", "META", "_START", ">-", "é", "\n", "tée", "́🙂", "İ", "​\r\n", "ſ", "👍🏽!‍​㋿", "
", "३", "e", "‍", " a"]} +{"text": "\u000b'S", "tokens": 2, "pieces": ["\u000b", "'S"]} +{"text": "<\"é ſ ­ꟲ Ⅳ🙂'ſ३ع٣٤٥٦aé𐞁'M
'T<|endoftext|>", "tokens": 45, "pieces": ["<\"", "e", "́", " ſ", " ­", "ꟲ", " ", "Ⅳ", "🙂'", "ſ", "३", "ع", "٣٤٥", "٦", "aé𐞁", "'M", "
", "'T", "<|", "endoftext", "|>"]} +{"text": "!.eعt\r\n\r\n\"<|fim_prefix|>fi EOT\r\n\r\nꟲ
'Reḍ̇ sfi,漢mtm\t", "tokens": 45, "pieces": ["!.<", "META", "_START", ">eعt", "\r\n\r\n", "\"<", "EOT", "><|", "fim", "_prefix", "|>", "fi", " EOT", "\r\n\r\n", "ꟲ", "
", "'Re", "ḋ", "̣", " sfi", ",漢mtm", "\t"]} +{"text": "ßé're912345678t", "tokens": 7, "pieces": ["ßé", "'re", "912", "345", "678", "t"]} +{"text": "İ\"!!", "tokens": 3, "pieces": ["İ", "\"!!"]} +{"text": " .\u000bå🙂t😀🏽<|fim_prefix|>​​,0'(é'ſm'Re", "tokens": 32, "pieces": [" ", ".", "\u000ba", "̊🙂", "t", "😀🏽<|", "fim", "_prefix", "|>​​,", "0", "'(", "e", "́'", "ſm", "'Re"]} +{"text": "Ⅳ'ſİ😀🏽e0𐞁 😀🏽#$%t'S>३9㍿٣٤٥٦'㋿'De'Re'!<|endoftext|>e½​‍ \r\n\r\n'…​ß ' ", "tokens": 72, "pieces": ["Ⅳ", "'ſ", "İ", "😀🏽", "e", "0", "𐞁", " 😀🏽#$%", "t", "'S", ">", "३9", "㍿", "٣٤٥", "٦", "'㋿'", "De", "'Re", "'!<|", "endoftext", "|>", "e", "½", "​‍", " \r\n\r\n", "'<", "EOT", ">", "…", "​ß", " '", " "]} +{"text": "ſ", "tokens": 2, "pieces": ["ſ"]} +{"text": "​३ ('reḍ̇漢ḍ̇m-'𐞁​'D-ꟲ\r\n\r\nEOT字Ⅳ㋿𐞁 \n> \u000b", "tokens": 51, "pieces": ["​", "३", " ", " ('", "reḋ", "̣漢ḋ", "̣m", "-'", "𐞁", "​'", "D", "-ꟲ", "\r\n\r\n", "EOT字", "Ⅳ", "㋿𐞁", " \n", "><", "META", "_START", ">", " \u000b"]} +{"text": "­Ⅳ-ſ! 'sß", "tokens": 13, "pieces": ["­", "Ⅳ", "-", "ſ", "!", " '", "sß"]} +{"text": "漢­­9🙂🙂𐞁😀🏽0 d'TZ", "tokens": 23, "pieces": ["漢", "­­", "9", "🙂🙂", "𐞁", "😀🏽", "0", " ", " d", "'T", "Z"]} +{"text": "İé0<|endoftext|>İ'll🙂́fi ​s-efi<|endoftext|>'ll­́३,", "tokens": 45, "pieces": ["İe", "́", "0", "<|", "endoftext", "|>", "İ", "'ll", "🙂́", "fi", " ", "​s", "-efi", "<|", "endoftext", "|>'", "ll", "­́", "३", ","]} +{"text": "12345678're 漢́é\n0mßḍ̇!!Ze12345678're'VE\tm'M\r\né'll😀🏽Ⅳt½\t👍🏽ḍ̇fi'T<|endoftext|>­㍿Z'Re", "tokens": 64, "pieces": ["\u000b", "́'", "re", "-.‍", "eß", "Ze", "123", "456", "78", "'re", "'VE", "\tm", "'M", "\r\n", "e", "́'", "ll", "😀🏽", "Ⅳ", "t", "½", "\t", "👍🏽", "ḋ", "̣fi", "'T", "<|", "endoftext", "|>­㍿", "Z", "'Re"]} +{"text": "(ع👍🏽ſⅣ'e…t㋿fifi३'T🙂dAém'ss\r\nꟲ㍿9 \n!…Z\" 're٣٤٥٦'re\u000bt!!", "tokens": 65, "pieces": ["(ع", "👍🏽", "ſ", "Ⅳ", "'e", "…t", "㋿fi", "fi", "३", "'T", "🙂dAém", "'s", "s", "\r\n", "ꟲ", "㍿", "9", " \n", "!", "…Z", "\"", " '", "re", "٣٤٥", "٦", "'re", "\u000bt", "!!"]} +{"text": "𐞁漢å <|endoftext|>'VE㋿AZ\u000bm🙂İ\r\n\r\n'sع12345678å🙂\t#$%'-ß'Reع> d…é", "tokens": 53, "pieces": ["𐞁漢a", "̊", " ", "<|", "endoftext", "|>'", "VE", "㋿AZ", "\u000bm", "🙂İ", "\r\n\r\n", "'s", "ع", "123", "456", "78", "a", "̊🙂", "\t", "#$%'-", "ß", "'Re", "ع", ">", " d", "…e", "́"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\n<|endoftext|>'VE½'re𐞁\u000b漢12345678eé<㍿漢Z'M'VE
'M👍🏽s'D ḍ̇-㍿<|fim_prefix|>", "tokens": 61, "pieces": ["\n", "<|", "endoftext", "|>'", "VE", "½", "'re", "𐞁", "\u000b漢", "123", "456", "78", "ee", "́<㍿", "漢Z", "'M", "'VE", "
", "'M", "👍🏽", "s", "'D", " ḋ", "̣-㍿<|", "fim", "_prefix", "|>"]} +{"text": "12345678\r\n\r\nſ<½12345678ḍ̇'ſع<'ReA'D(\n漢‍ #$%​٣٤٥٦🙂­!! ​\r\n\r\n<<|endoftext|>>!!ⅣDžſſ!字", "tokens": 71, "pieces": ["123", "456", "78", "\r\n\r\n", "ſ", "<", "½12", "345", "678", "ḋ", "̣'", "ſع", "<'", "ReA", "'D", "(\n", "漢", "‍", " ", " #$%​", "٣٤٥", "٦", "🙂­!!", " ​\r\n\r\n", "<<|", "endoftext", "|>>!!", "Ⅳ", "Džſſ", "!字"]} +{"text": "
s𐞁tⅣ(٣٤٥٦åZ½​'D ḍ̇<‍ ‍'VEeعfi\r\ń½'S9'T\tſ漢​'re'Ms漢㍿'M", "tokens": 70, "pieces": ["
s𐞁t", "Ⅳ", "(", "٣٤٥", "٦", "a", "̊Z", "½", "​'", "D", " ḋ", "̣<<", "META", "_START", ">‍", " ‍'", "VEeعfi", "\r\n", "́", "½", "'S", "9", "'T", "\tſ漢", "​'", "re", "'M", "s漢", "㍿'", "M"]} +{"text": "fiꟲḍ̇", "tokens": 10, "pieces": ["fiꟲḋ", "̣"]} +{"text": "字😀🏽
İⅣA​é<|fim_prefix|>DžEOT😀🏽!<\u000b㋿\" ", "tokens": 38, "pieces": ["字", "😀🏽", "
İ", "Ⅳ", "A", "​é", "<|", "fim", "_prefix", "|>", "DžEOT", "😀🏽!<", "\u000b", "㋿\"", " "]} +{"text": "0㍿<ḍ̇İ'T!… \r\n \n'M٣٤٥٦İ́漢\rtꟲ…éİ٣٤٥٦é're 'M‍ß.", "tokens": 60, "pieces": ["0", "㍿<", "ḋ", "̣İ", "'T", "!", "… \r\n \n", "'M", "٣٤٥", "٦", "İ", "́漢", "\r", "tꟲ", "…éİ", "٣٤٥", "٦", "é", "'re", " ", " '", "M", "‍ß", "."]} +{"text": "0'llt ", "tokens": 4, "pieces": ["0", "'ll", "t", " "]} +{"text": "🙂'9ß(å\u000ba'Re\r\nḍ̇.\rꟲ㋿ع9'VEsZDž…'M\r\n\r\n\rİA'Mfia.३ḍ̇", "tokens": 60, "pieces": ["🙂<", "META", "_START", ">'<", "META", "_START", ">", "9", "ß", "(a", "̊", "\u000ba", "'Re", "\r\n", "ḋ", "̣.\r", "ꟲ", "㋿ع", "9", "'VE", "sZDž", "…", "'M", "\r\n\r\n\r", "İA", "'M", "fia", ".", "३", "ḋ", "̣"]} +{"text": "   \n'VEt<|fim_prefix|>é㋿Džts,'lĺ'Re\r\n<,", "tokens": 29, "pieces": ["   \n", "'VE", "t", "<|", "fim", "_prefix", "|>", "e", "́㋿", "Džts", ",'", "ll", "́'", "Re", "\r\n", "<,"]} +{"text": "(- 'Té𐞁ǻ…EOTعꟲ'Re\u000b,٣٤٥٦㍿s\u000b", "tokens": 40, "pieces": ["(-", " ", " '", "Te", "́𐞁a", "̊́", "…EOTعꟲ", "'", "Re", "\u000b", ",", "٣٤٥", "٦", "㍿s", "\u000b"]} +{"text": "\"Ⅳ-𐞁-e!'llåİ 漢Zfi㋿", "tokens": 24, "pieces": ["\"", "Ⅳ", "-𐞁", "-e", "!'", "lla", "̊İ", " ", " 漢Zfi", "㋿"]} +{"text": "d<|endoftext|>'TⅣ\r😀🏽ſⅣd‍!㋿", "tokens": 28, "pieces": ["d", "<|", "endoftext", "|>'", "T", "Ⅳ", "\r", "😀🏽", "ſ", "Ⅳ", "d", "‍!㋿"]} +{"text": "👍🏽#$% EOT<|endoftext|>,عſéſ\t\r½'sd…😀🏽é½<\t\n'VEé👍🏽", "tokens": 51, "pieces": ["👍🏽#$%", " ", " EOT", "<|", "endoftext", "|>,", "عſe", "́ſ", "\t\r", "½", "'s", "d", "…", "😀🏽", "é", "½", "<", "\t\n", "'VE", "e", "́👍🏽"]} +{"text": "(åع🙂ع🙂Dž­'S!'s're­㋿'ſfi😀🏽½'D9", "tokens": 34, "pieces": ["(a", "̊ع", "🙂ع", "🙂Dž", "­'", "S", "!'", "s", "'re", "­㋿'", "ſfi", "😀🏽", "½", "'D", "9"]} +{"text": "\t٣٤٥٦'T३\rع…𐞁fi٣٤٥٦🙂'll'll
", "tokens": 37, "pieces": ["\t", "٣٤٥", "٦", "'T", "३", "\r", "ع", "…𐞁fi", "٣٤٥", "٦", "🙂'", "ll", "'ll", "
"]} +{"text": "'­\"́́'ll're🙂👍🏽 字

…0ꟲ'llⅣ\" \n३ ,\u000b'Re \n", "tokens": 43, "pieces": ["'­\"́́'", "ll", "'re", "🙂👍🏽", " 字", "
", "", "
", "…", "0", "ꟲ", "'ll", "Ⅳ", "\"", " \n", "३", " ", ",", "\u000b", "'Re", " \n"]} +{"text": "
Z<|fim_prefix|>!!\n‍é\u000b㋿ \n'Re  0d(d㍿Z'lls12345678 👍🏽'T ​́> >👍🏽 ", "tokens": 67, "pieces": ["
Z", "<|", "fim", "_prefix", "|>!!\n", "‍<", "META", "_START", ">é", "\u000b", "㋿<", "EOT", ">", " \n", "'Re", " ", " ", "0", "d", "(d", "㍿Z", "'ll", "s", "123", "456", "78", " ", "👍🏽'", "T", " ", " ​́>", " >👍🏽", " "]} +{"text": "'S漢​'Sİḍ̇‍\r\nA", "tokens": 20, "pieces": ["'S", "漢", "​<", "META", "_START", ">'", "Sİḋ", "̣‍\r\n", "A"]} +{"text": "㍿ <<|fim_prefix|>å\r\n<|fim_prefix|>㍿'T\r\nſ!", "tokens": 31, "pieces": ["㍿", " ", " <<|", "fim", "_prefix", "|>", "a", "̊\r\n", "<|", "fim", "_prefix", "|>㍿'", "T", "\r\n", "ſ", "!"]} +{"text": ".a
'll'llm<|endoftext|>Z\r\"-'ſ😀🏽>字'sⅣ<|endoftext|>d\tİ­", "tokens": 40, "pieces": [".a", "
", "'ll", "'ll", "m", "<|", "endoftext", "|>", "Z", "\r", "\"-'", "ſ", "😀🏽>", "字", "'s", "Ⅳ", "<|", "endoftext", "|>", "d", "\tİ", "­"]} +{"text": "😀🏽'Re🙂'sعm'VE🙂's('re㋿daa'Dß🙂\u000bİ'll!! åع'så'St12345678\r\n\r\n\neé", "tokens": 51, "pieces": ["😀🏽'", "Re", "🙂'", "sعm", "'VE", "🙂'", "s", "('", "re", "㋿daa", "'D", "ß", "🙂", "\u000bİ", "'ll", "!!", " a", "̊ع", "'s", "a", "̊'", "St", "123", "456", "78", "\r\n\r\n\n", "eé"]} +{"text": "'D're\r\n123456789㍿t‍.​<|endoftext|>!ß㍿\r\nDž<|fim_prefix|>㋿ḍ̇́'M…", "tokens": 52, "pieces": ["'D", "'re", "\r\n", "123", "456", "789", "㍿t", "‍.​<|", "endoftext", "|>!", "ß", "㍿<", "EOT", ">\r\n", "Dž", "<|", "fim", "_prefix", "|>㋿", "ḋ", "̣́'", "M", "…"]} +{"text": "t́e9 \t'VE漢ꟲ12345678 Džİ'ZZ\rع𐞁é.😀🏽İ​\u000b­(0'\rZ…m!!EOT ́😀🏽", "tokens": 58, "pieces": ["t", "́e", "9", " ", "\t", "'VE", "漢ꟲ", "123", "456", "78", " Džİ", "'ZZ", "\r", "ع𐞁e", "́.😀🏽", "İ", "​", "\u000b", "­(", "0", "'\r", "Z", "…m", "!!", "EOT", " ", "́😀🏽"]} +{"text": "EOT٣٤٥٦!!<|fim_prefix|>fifiaß́\n'VE𐞁\n­🙂 \n(t㋿'D\n\r\né", "tokens": 47, "pieces": ["EOT", "٣٤٥", "٦", "!!<|", "fim", "_prefix", "|>", "fifiaß", "́\n", "'VE", "𐞁", "\n", "­🙂", " \n", "(t", "㋿'", "D", "\n\r\n", "e", "́"]} +{"text": "! \n…#$%\r\nſ😀🏽३\"́!…😀🏽́é‍#$%é.ém'll'S", "tokens": 44, "pieces": ["!", " \n", "…", "#$%<", "META", "_START", ">\r\n", "ſ", "😀🏽", "३", "\"́!", "…", "😀🏽́", "é", "‍#$%", "e", "́.", "ém", "'ll", "'S"]} +{"text": "İ", "tokens": 1, "pieces": ["İ"]} +{"text": "'Md३🙂'llİ🙂\"! ‍Dž<½\r\n\r\n'ſ­s漢!! ع 12345678\"½\n👍🏽\r\n३½.𐞁३", "tokens": 56, "pieces": ["'M", "d", "३", "🙂'", "llİ", "🙂<", "META", "_START", ">\"!", " ", "‍Dž", "<", "½", "\r\n\r\n", "'ſ", "­s漢", "!!", " ع", " ", "123", "456", "78", "\"", "½", "\n", "👍🏽\r\n", "३½", ".𐞁", "३"]} +{"text": "\u000b(-<|endoftext|>.३ 👍🏽#$%'VE'res\r\n", "tokens": 24, "pieces": ["\u000b", "(-<|", "endoftext", "|>.", "३", " ", "👍🏽#$%'", "VE", "'re", "s", "\r\n"]} +{"text": "🙂 \n😀🏽<|endoftext|>​('🙂mt'sm\"ع́é­m'S9!㍿𐞁é'ßAm'ſ'S½'ſ
", "tokens": 60, "pieces": ["🙂", " \n", "😀🏽<|", "endoftext", "|>​('🙂", "m", "t", "'s", "m", "\"ع", "́e", "́­", "m", "'S", "9", "!㍿", "𐞁e", "́'", "ßAm", "'ſ", "'S", "½", "'ſ", "
"]} +{"text": "-,漢<|endoftext|>漢\r\n𐞁<|fim_prefix|>𐞁𐞁<|endoftext|>​ع'D㋿'ReⅣ-\"'ſ'Z'll​'Tå  ​Dž", "tokens": 77, "pieces": ["<|", "endoftext", "|>", "漢", "\r\n", "𐞁", "<|", "fim", "_prefix", "|>", "𐞁𐞁", "<|", "endoftext", "|>​", "ع", "'D", "㋿'", "Re", "Ⅳ", "-\"'", "ſ", "'Z", "'ll", "​'", "Ta", "̊", "  ", " ​", "Dž", ""]} +{"text": "ع漢EOT३<|fim_prefix|>ع'llİ  é<|endoftext|>sß!!🙂㋿- ꟲå0ع'Dfi३>٣٤٥٦ 'S\r\n\r\n\rfi", "tokens": 66, "pieces": ["ع漢EOT", "३", "<|", "fim", "_prefix", "|>", "ع", "'ll", "İ", " ", " e", "́<|", "endoftext", "|>", "sß", "!!🙂㋿-", " ꟲa", "̊", "0", "ع", "'D", "fi", "३", ">", "٣٤٥", "٦", " ", "'S", "\r\n\r\n\r", "fi"]} +{"text": "'ſEOT\u000b​t́,漢字,ḍ̇🙂d'T'llİḍ̇‍'VE😀🏽12345678s漢٣٤٥٦漢İ  ", "tokens": 58, "pieces": ["'ſ", "EOT", "\u000b", "​t", "́,", "漢字", ",ḋ", "̣🙂", "d", "'T", "'ll", "İḋ", "̣‍'", "VE", "😀🏽", "123", "456", "78", "s漢", "٣٤٥", "٦", "漢İ", "  "]} +{"text": "Z'Så(ḍ̇'VEe‍", "tokens": 16, "pieces": ["Z", "'S", "a", "̊(", "ḋ", "̣'", "VEe", "‍"]} +{"text": "sßꟲsa\u000baſtع㍿́ع३漢\r\n\tEOTßſé\"­\u000bع'S ́ß", "tokens": 41, "pieces": ["sßꟲsa", "\u000baſtع", "㍿́", "ع", "३", "漢", "\r\n", "\tEOTß", "ſe", "́\"­", "\u000bع", "'S", " ́", "ß"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "Džd३\"İ12345678'll字'llfi\r\ńé३Dž'VE'VE.a<'lldt", "tokens": 34, "pieces": ["Džd", "३", "\"İ", "123", "456", "78", "'ll", "字", "'ll", "fi", "\r\n", "́", "e", "́", "३", "Dž", "'VE", "'VE", ".a", "<'", "lldt"]} +{"text": "'S <|endoftext|>é'S𐞁<|endoftext|>​ſ…­ ''re", "tokens": 32, "pieces": ["'S", " ", " <|", "endoftext", "|>", "e", "́'", "S𐞁", "<|", "endoftext", "|>​", "ſ", "…", "­", " ", "''", "re"]} +{"text": "'s\t'ſZ \"​d👍🏽>!><İſ𐞁'İ<|fim_prefix|>'D- 🙂.'ll'S𐞁sEOT😀🏽dEOT'M Z", "tokens": 61, "pieces": ["'s", "\t", "'ſ", "Z", " ", "\"​", "d", "👍🏽>!><", "İſ𐞁", "'İ", "<|", "fim", "_prefix", "|>'", "D", "-", " ", "🙂.'", "ll", "'S", "𐞁sEOT", "😀🏽", "dEOT", "'M", " Z"]} +{"text": " \u000be,ſs😀🏽字\"ع<|fim_prefix|>'re… \ns\r\r\n ", "tokens": 32, "pieces": [" ", "\u000be", ",ſ", "s", "😀🏽", "字", "\"ع", "<|", "fim", "_prefix", "|>'", "re", "… \n", "s", "\r\r\n "]} +{"text": " \t<<|fim_prefix|> \n>㋿ḍ̇½t🙂'T👍🏽\t<|endoftext|>\r\n\r\n 
🙂", "tokens": 44, "pieces": [" ", "\t", "<<|", "fim", "_prefix", "|>", " \n", ">㋿", "ḋ", "̣", "½", "t", "🙂'", "T", "👍🏽", "\t", "<|", "endoftext", "|>\r\n\r\n", " ", "
", "🙂"]} +{"text": " 0\t's'M<,👍🏽'ſ\t's…👍🏽0", "tokens": 28, "pieces": [" ", " ", "0", "\t", "'s", "'M", "<,👍🏽'", "ſ", "\t", "'s", "…", "👍🏽", "0"]} +{"text": "ḍ̇\nİ12345678३㋿", "tokens": 15, "pieces": ["ḋ", "̣\n", "İ", "123", "456", "78३", "㋿"]} +{"text": "'M 9<|endoftext|>fi('𐞁a<𐞁😀🏽½½a-'VE\tm\r é ½𐞁ⅣDžéd<|fim_prefix|>…\t'sZİع
Ⅳ", "tokens": 73, "pieces": ["'M", " ", "9", "<|", "endoftext", "|>", "fi", "('", "𐞁a", "<𐞁", "😀🏽", "½½", "a", "-'", "VE", "\tm", "\r", " ", " e", "́", " ", "½", "𐞁", "Ⅳ", "Dže", "́d", "<|", "fim", "_prefix", "|>", "…", "\t", "'s", "Z", "İع", "
", "Ⅳ"]} +{"text": "
m \n!!ſⅣ‍ß𐞁's#$%ع-d'́\t!!<|endoftext|>aDžfi㍿😀🏽#$%#$% ḍ̇‍,", "tokens": 58, "pieces": ["
m", " \n", "!!", "ſ", "Ⅳ", "‍ß𐞁", "'s", "#$%", "ع", "-d", "'́", "\t", "!!<|", "endoftext", "|>", "aDžfi", "㍿😀🏽#$%#$%", " ", " ḋ", "̣‍,"]} +{"text": "́a- ३ḍ̇'reꟲ㋿> 'll字ع😀🏽'M\t-'M\r", "tokens": 39, "pieces": ["́a", "-<", "EOT", ">", " ", "३", "ḋ", "̣'", "reꟲ", "㋿>", " '", "ll字ع", "😀🏽'", "M", "\t", "-'", "M", "\r"]} +{"text": "'S
('S🙂Z<'VE tm", "tokens": 12, "pieces": ["'S", "
", "('", "S", "🙂Z", "<'", "VE", " ", " tm"]} +{"text": "-0! \u000b'M0EOT(t'Re e𐞁>dEOT\"​eꟲmm٣٤٥٦!!ḍ̇٣٤٥٦👍🏽12345678", "tokens": 61, "pieces": ["-", "0", "!", " ", "\u000b", "'M", "0", "EOT", "(t", "'Re", " e", "𐞁", ">dEOT", "\"​", "eꟲmm", "٣٤٥", "٦", "!!", "ḋ", "̣", "٣٤٥", "٦", "👍🏽", "123", "456", "78"]} +{"text": "٣٤٥٦́a9><A​\t㍿A!!'<fi>é,0㋿漢Ⅳ'Re‍å字m🙂 'VE", "tokens": 49, "pieces": ["٣٤٥", "٦", "́a", "9", "><", "A", "​", "\t", "㍿A", "!!'<", "fi", ">e", "́,", "0", "㋿漢", "Ⅳ", "'Re", "‍a", "̊字m", "🙂", " ", " '", "VE"]} +{"text": "ع", "tokens": 1, "pieces": ["ع"]} +{"text": " \n0㍿𐞁Dž㋿é字'Re\tḍ̇\r\nꟲ\r 'S'D½", "tokens": 39, "pieces": [" \n", "0", "㍿𐞁Dž", "㋿é字", "'Re", "\tḋ", "̣<", "META", "_START", ">\r\n", "ꟲ", "\r", " ", "'S", "'D", "½", ""]} +{"text": "å'S dfi㍿fi '0Z-'re३
A'ſ<|endoftext|>Ⅳeå…(ſ\r\nm\r\n\r\n<|endoftext|>Dž漢عḍ̇\r\n\r\n-", "tokens": 77, "pieces": ["a", "̊'", "S", "", " ", " dfi", "㍿fi", " ", "'", "0", "Z", "-<", "EOT", ">'", "re", "३", "
A", "'", "ſ", "<|", "endoftext", "|>", "Ⅳ", "ea", "̊", "…", "(ſ", "\r\n", "m", "\r\n\r\n", "<|", "endoftext", "|>", "Dž漢عḋ", "̣\r\n\r\n", "-"]} +{"text": "<'M'M!!(ꟲ½12345678!İ🙂'VEDž३#$%\"㍿'D 'Re>må'D\t
漢<|endoftext|><|endoftext|>'s<|fim_prefix|>'T", "tokens": 64, "pieces": ["<'", "M", "'M", "!!(", "ꟲ", "½12", "345", "678", "!İ", "🙂'", "VEDž", "३", "#$%\"㍿'", "D", " '", "Re", ">ma", "̊'", "D", "\t", "
漢", "<|", "endoftext", "|><|", "endoftext", "|>'", "s", "<|", "fim", "_prefix", "|>'", "T"]} +{"text": "Ⅳ'M.'Tḍ̇­-'ſt㍿㋿é­ḍ̇0'D9#$%<|fim_prefix|>\r\n\r\nEOT'Re 'Re😀🏽,dꟲ𐞁#$%é", "tokens": 66, "pieces": ["Ⅳ", "'", "M", ".'", "Tḋ", "̣­-'", "ſt", "㍿㋿", "é", "­ḋ", "̣", "0", "'D", "9", "#$%<|", "fim", "_prefix", "|>\r\n\r\n", "EOT", "'Re", " '", "Re", "😀🏽,", "dꟲ𐞁", "#$%", "é"]} +{"text": "'Mİ åA\r\n㋿'ſm \n\t>👍🏽!!!ḍ̇İ9𐞁12345678​EOT 字漢fi'ſ9३½ !!\nꟲ", "tokens": 65, "pieces": ["'M", "İ", " ", " a", "̊A", "\r\n", "㋿'", "ſm", " \n", "\t", ">👍🏽!!!", "ḋ", "̣İ", "9", "𐞁", "123", "456", "78", "​EOT", " 字漢fi", "'ſ", "9३½", " ", " !!\n", "ꟲ"]} +{"text": "'Re - 𐞁å'ſꟲeå\r\n's 🙂­ع🙂ع\r\n\r\nꟲ#$%
>ḍ̇é", "tokens": 45, "pieces": ["'Re", " ", "-", " 𐞁a", "̊'", "ſꟲea", "̊\r\n", "'s", " ", "🙂­", "ع", "🙂ع", "\r\n\r\n", "ꟲ", "#$%", "
", ">ḋ", "̣é"]} +{"text": "­(- sådA😀🏽éDž#$%<|endoftext|>字", "tokens": 27, "pieces": ["­(-", " sa", "̊dA", "😀🏽", "éDž", "#$%<|", "endoftext", "|>", "字"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n\r\n-'Sİe \n'MⅣ́👍🏽‍字9🙂漢fi're­a12345678字'M<|endoftext|>\r\n\r\n漢 …,\"Džß", "tokens": 61, "pieces": ["\r\n\r\n", "-'", "Sİe", " \n", "'M", "", "Ⅳ", "́👍🏽‍", "字", "9", "🙂<", "EOT", ">漢fi", "'re", "­a", "123", "456", "78", "字", "'M", "<|", "endoftext", "|>\r\n\r\n", "漢", " ", "…", ",\"", "Džß"]} +{"text": "İs㋿‍fí👍🏽9‍>", "tokens": 20, "pieces": ["İs", "㋿‍", "fi", "́👍🏽", "9", "‍>"]} +{"text": "ḍ̇㋿é'ſ'ſ'D12345678 ß‍!", "tokens": 29, "pieces": ["ḋ", "̣㋿", "e", "́<", "EOT", ">'", "ſ", "'ſ", "'D", "123", "456", "78", " ", " ß", "‍!"]} +{"text": "Z.'ſ (
sd'ſ's\r\n\r\n\r<👍🏽'Re'D>'Re\n\r\n\r\nZ 'Res‍
\t.ع", "tokens": 39, "pieces": ["Z", ".'", "ſ", " (", "
sd", "'ſ", "'s", "\r\n\r\n\r", "<👍🏽'", "Re", "'D", ">'", "Re", "\n\r\n\r\n", "Z", " ", " '", "Res", "‍", "
", "\t", ".ع"]} +{"text": "\r\n\r\n\r\n\r\nAd\t'Sع ", "tokens": 8, "pieces": ["\r\n\r\n\r\n\r\n", "Ad", "\t", "'S", "ع", " "]} +{"text": "'ſꟲ​seaع#$%fi'll0'ſZꟲ<|fim_prefix|>9<ꟲ'D'D", "tokens": 36, "pieces": ["'ſ", "ꟲ", "​seaع", "#$%", "fi", "'ll", "0", "'ſ", "Zꟲ", "<|", "fim", "_prefix", "|>", "9", "<ꟲ", "'D", "'D"]} +{"text": "éß'll'T㋿ \r\n\r\n\n <|fim_prefix|>‍́<|endoftext|>㍿ꟲ\"'字\t ­👍🏽\t'Re'D‍ \nß", "tokens": 59, "pieces": ["éß", "'ll", "'T", "㋿", " \r\n\r\n\n", " ", "<|", "fim", "_prefix", "|>‍́<|", "endoftext", "|>㍿", "ꟲ", "\"'<", "EOT", ">字", "\t", " ­👍🏽", "\t", "'Re", "'", "D", "‍", " \n", "ß"]} +{"text": "0ſ9Aß\rꟲéⅣ\r㋿\u000bß'Reé\t­<|endoftext|>9\r\n 's'­'M", "tokens": 47, "pieces": ["0", "ſ", "9", "Aß", "\r", "ꟲe", "́<", "EOT", "><", "EOT", ">", "Ⅳ", "\r", "㋿", "\u000bß", "'Re", "é", "\t", "­<|", "endoftext", "|>", "9", "\r\n", " ", "'s", "'­'", "M"]} +{"text": "12345678​\r\n漢 \n'Reḍ̇9EOT('\r\n\r\nA'Re.ḍ̇   ſ9'D𐞁字\"e🙂12345678ꟲ'reſe३㋿!!<|fim_prefix|>,å", "tokens": 72, "pieces": ["123", "456", "78", "​\r\n", "漢", " \n", "'Re", "ḋ", "̣", "9", "EOT", "('\r\n\r\n", "A", "'Re", ".ḋ", "̣", "  ", " ſ", "", "9", "'D", "𐞁字", "\"e", "🙂", "123", "456", "78", "ꟲ", "'re", "ſe", "३", "㋿!!<|", "fim", "_prefix", "|>,", "a", "̊"]} +{"text": "३<|endoftext|>!́𐞁३漢#$% 漢>
a9‍", "tokens": 32, "pieces": ["३", "<|", "endoftext", "|>!́", "𐞁", "३", "漢", "#$%", " ", " 漢", ">", "
a", "9", "‍"]} +{"text": "­'Re-…de!!9…'T'D\n\r\néⅣm🙂\ts\ta <'ll", "tokens": 26, "pieces": ["­'", "Re", "-", "…de", "!!", "9", "…", "'T", "'D", "\n\r\n", "é", "Ⅳ", "m", "🙂", "\ts", "\ta", " <'", "ll"]} +{"text": "\u000b́٣٤٥٦<|fim_prefix|>\"", "tokens": 46, "pieces": ["a", "̊<", "META", "_START", ">", "\u000b", "́<", "a", "­㋿<", "½", "!", " t", "\n", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>\""]} +{"text": "ſ'ſ> d'ſ …İ'ſm's#$%. 9 ", "tokens": 25, "pieces": ["ſ", "'ſ", ">", " ", " d", "'ſ", " ", "…İ", "'ſ", "m", "'s", "#$%.", " ", "9", " "]} +{"text": "('M🙂\r½'ſdEOTDžAfi <#$%", "tokens": 50, "pieces": ["('", "M", "🙂\r", "½", "'ſ", "dEOTDžA", "", "fi", " ", "<#$%"]} +{"text": " \nd½!'ReDžİ'VEd٣٤٥٦🙂e \n\n'Tß 'S
dDž", "tokens": 35, "pieces": [" \n", "d", "½", "!'", "ReDžİ", "'VE", "d", "٣٤٥", "٦", "🙂e", " \n\n", "'T", "ß", " ", "'S", "
", "dDž"]} +{"text": ">\u000b㍿#$%'fi\t'll!!é9>,­'VE'​'Re  'll(\u000b'S字👍🏽\n \n're٣٤٥٦ !!\r ع", "tokens": 55, "pieces": [">", "\u000b", "㍿#$%'", "fi", "\t", "'ll", "!!", "é", "9", ">,­'", "VE", "'​'", "Re", "  ", " '", "ll", "(", "\u000b", "'S", "字", "👍🏽\n", " \n", "'re", "٣٤٥", "٦", " ", "!!\r", " ع"]} +{"text": "­३(!", "tokens": 4, "pieces": ["­", "३", "(!"]} +{"text": " t ‍Ⅳa'D'reåd㋿İdZ…½Ⅳ\nعꟲ\td!!", "tokens": 35, "pieces": [" t", " ", "‍", "Ⅳ", "a", "'D", "'re", "a", "̊d", "㋿İdZ", "…", "½", "", "Ⅳ", "\n", "عꟲ", "\td", "!!"]} +{"text": "…a's<'Mع😀🏽'M…9ꟲ'ſ\"A\ta\u000bḍ̇('T-12345678e \n 🙂<|endoftext|>́'D", "tokens": 57, "pieces": ["…a", "'s", "<'", "Mع", "😀🏽'", "M", "…", "9", "ꟲ", "'ſ", "\"A", "\ta", "\u000bḋ", "̣(<", "META", "_START", ">'", "T", "-", "123", "456", "78", "e", " \n", " ", "🙂<|", "endoftext", "|>́'", "D"]} +{"text": "d½t'll㋿\r\n
>字字mع,12345678d'Sé३'VE!!><|endoftext|>EOT́å½12345678'A,A­½\r\n👍🏽é", "tokens": 67, "pieces": ["d", "½", "t", "'ll", "㋿\r\n", "
", ">字字mع", ",", "123", "456", "78", "d", "'S", "é", "३", "'", "VE", "!!><|", "endoftext", "|>", "EOT", "́a", "̊", "½12", "345", "678", "'A", ",A", "­", "½", "\r\n", "👍🏽", "é", ""]} +{"text": "'ſDž…㍿#$%>'re,fi \ńß ( \n漢#$%.", "tokens": 27, "pieces": ["'ſ", "Dž", "…", "㍿#$%>'", "re", ",fi", " \n", "́ß", " ", " (", " \n", "漢", "#$%."]} +{"text": "e ३عm('s漢 \n \né漢m#$%Aé'Sع12345678e\"ſ३'ll😀🏽Z'Re>👍🏽३…­\r\n\r\n😀🏽", "tokens": 63, "pieces": ["e", "", " ", " ", "३", "عm", "('", "s漢", " \n \n", "é漢m", "#$%", "Aé", "'S", "ع", "123", "456", "78", "e", "\"ſ", "३", "'ll", "😀🏽", "Z", "'Re", ">👍🏽", "३", "…", "­\r\n\r\n", "😀🏽"]} +{"text": "\r\n\r\n'ſ\r\n\r\n!👍🏽\t \n㋿#$%.å\r\n're漢\"! ", "tokens": 34, "pieces": ["\r\n\r\n", "'ſ", "\r\n\r\n", "!👍🏽", "\t \n", "㋿#$%.", "a", "̊\r\n", "'re", "漢", "\"!", " "]} +{"text": "​fi漢.\"ع\nعa'lls'ſ", "tokens": 15, "pieces": ["​fi漢", ".\"", "ع", "\n", "عa", "'ll", "s", "'ſ"]} +{"text": "\r'S'D a字fi­<9EOT½…>d,\r😀🏽٣٤٥٦😀🏽'M0ḿ𐞁med's𐞁​", "tokens": 53, "pieces": ["\r", "'S", "'D", " a字fi", "­<", "9", "EOT", "½", "…", ">d", ",\r", "😀🏽", "٣٤٥", "٦", "😀🏽'", "M", "0", "m", "́𐞁med", "'s", "𐞁", "​"]} +{"text": "٣٤٥٦'VE\r\n\r\n \r\n\r\n\r\n\r\ń\r\n𐞁(́'ſ​३!!\u000b Z漢,ع ­'EOT́😀🏽Aſ­Dž <|fim_prefix|> >fi", "tokens": 66, "pieces": ["٣٤٥", "٦", "'VE", "\r\n\r\n \r\n\r\n\r\n\r\n", "́\r\n", "𐞁", "(́'", "ſ", "​", "३", "!!", "\u000b", " Z漢", ",ع", " ", "­'", "EOT", "́<", "META", "_START", ">😀🏽", "Aſ", "­Dž", " <|", "fim", "_prefix", "|>", " ", " >", "fi"]} +{"text": "(å \n‍(-12345678'D٣٤٥٦é<… İ 🙂字,٣٤٥٦EOT'VE,ſ>字'll'll<|endoftext|>fi", "tokens": 61, "pieces": ["(a", "̊", " \n", "‍(-", "123", "456", "78", "'D", "٣٤٥", "٦", "é", "<", "…", " İ", " ", "🙂字", ",", "٣٤٥", "٦", "EOT", "'VE", ",ſ", ">", "字", "'ll", "'ll", "<|", "endoftext", "|>", "fi"]} +{"text": "𐞁👍🏽.㋿('Sfi\nd\r\n<|fim_prefix|>…ꟲ字(​ḍ̇<|fim_prefix|>\u000b𐞁e漢", "tokens": 59, "pieces": ["𐞁", "👍🏽.㋿('", "Sfi", "\n", "d", "\r\n", "<|", "fim", "_prefix", "|><", "EOT", ">", "…ꟲ字", "(​", "ḋ", "̣<|", "fim", "_prefix", "|>", "\u000b𐞁e漢"]} +{"text": "\u000b漢😀🏽\t \r\n\r\n'T!½🙂A9,éfie🙂(\r\n\r\n\r\n!Ⅳ٣٤٥٦>a\u000b'­!!㍿", "tokens": 46, "pieces": ["\u000b漢", "😀🏽", "\t \r\n\r\n", "'T", "!", "½", "🙂A", "9", ",e", "́fie", "🙂(\r\n\r\n\r\n", "!", "Ⅳ٣٤", "٥٦", ">a", "\u000b", "'­!!㍿"]} +{"text": "<|fim_prefix|>🙂, 😀🏽<|endoftext|> !!'llDž字㍿\u000b''M㋿'llⅣ#$%!! 漢 0", "tokens": 50, "pieces": ["<|", "fim", "_prefix", "|>🙂,", " ", " 😀🏽<|", "endoftext", "|>", " ", "!!'", "llDž字", "㍿", "\u000b", "''", "M", "㋿'", "ll", "Ⅳ", "#$%!!", " 漢", " ", "0"]} +{"text": "'T're#$%İ-!!Ź9'VE<|fim_prefix|>٣٤٥٦'ll,As­d'ſ", "tokens": 40, "pieces": ["'T", "'re", "#$%", "İ", "-!!", "Z", "́", "9", "'VE", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "'", "ll", ",As", "­d", "'ſ"]} +{"text": "٣٤٥٦ß٣٤٥٦­d'Dž é >''ſm\t <|fim_prefix|>fi½\t'T \nſ d", "tokens": 49, "pieces": ["٣٤٥", "٦", "ß", "٣٤٥", "٦", "­d", "'Dž", " é", " ", ">''", "ſm", "\t", " ", "<|", "fim", "_prefix", "|>", "fi", "½", "\t", "'T", " \n", "ſ", " d"]} +{"text": "'Dع12345678­ſ㋿<|fim_prefix|>㍿. ‍½ꟲs. ꟲ'M'llß(
<|endoftext|>\ts!
漢", "tokens": 60, "pieces": ["'D", "ع", "123", "456", "78", "­ſ", "㋿<|", "fim", "_prefix", "|>㍿.", " ‍", "½", "ꟲs", ".", " ꟲ", "'M", "'ll", "ß", "(", "
", "<|", "endoftext", "|>", "\t", "<", "EOT", ">s", "!", "
漢"]} +{"text": "\tß ​𐞁<|endoftext|> ㋿\rⅣ>'s\"->,'D
​
A\n Ⅳ!!s\r\n🙂's👍🏽(漢áع", "tokens": 62, "pieces": ["\tß", " ", "​𐞁", "<|", "endoftext", "|>", " ", "㋿<", "EOT", ">\r", "Ⅳ", ">'", "s", "\"->,'", "D", "
", "​", "
A", "\n", " ", "Ⅳ", "!!", "s", "\r\n", "🙂'", "s", "👍🏽(", "漢a", "́ع"]} +{"text": "㋿ e👍🏽s'Refi(👍🏽<|endoftext|>'S½><#$% \r\nA \r­🙂🙂
fi 0ꟲ'D 👍🏽(<|endoftext|>漢", "tokens": 73, "pieces": ["㋿", " ", "e", "👍🏽", "s", "'Re", "fi", "(👍🏽<|", "endoftext", "|>'", "S", "½", "><#$%", " \r\n", "A", " \r", "­🙂🙂", "
fi", " ", "0", "ꟲ", "'D", " ", "👍🏽(<|", "endoftext", "|>", "漢"]} +{"text": "🙂㋿½", "tokens": 6, "pieces": ["🙂㋿", "½"]} +{"text": " 😀🏽㍿ 𐞁'llß\rßé12345678aع'S-\r's\r'#$%!\"eß𐞁", "tokens": 38, "pieces": [" ", " 😀🏽㍿", " 𐞁", "'ll", "ß", "\r", "ße", "́", "123", "456", "78", "aع", "'S", "-\r", "'s", "\r", "'#$%!\"", "eß𐞁"]} +{"text": "‍㍿'Mſ㍿½ ​9s漢ß漢afi
ſå12345678\r\n\r\n­ꟲ…­AA字td漢​eع\"<|fim_prefix|>'ll", "tokens": 66, "pieces": ["‍㍿'", "Mſ", "㍿", "½", " ", " ​", "9", "s漢ß漢afi", "
ſa", "̊", "123", "456", "78", "\r\n\r\n", "­ꟲ", "…", "­AA字td漢", "​eع", "\"<|", "fim", "_prefix", "|>'", "ll"]} +{"text": "12345678'D́ſ\"عfi'llA ㋿'Me㋿'ſZع\r\n\r\n<
\"! ­'D\n é😀🏽ع字sé", "tokens": 52, "pieces": ["123", "456", "78", "'D", "́ſ", "\"عfi", "'ll", "A", " ㋿'", "Me", "㋿'", "ſZع", "\r\n\r\n", "<", "
", "\"!<", "EOT", ">", " ", " ­'", "D", "\n", " é", "😀🏽", "ع字sé"]} +{"text": "å…9㍿ḍ̇A<|endoftext|>­
İꟲmEOT'Mİét'VE'VE🙂\r字ſ", "tokens": 48, "pieces": ["a", "̊", "…", "9", "㍿ḋ", "̣A", "<|", "endoftext", "|>­", "
İꟲmEOT", "'M", "İét", "'VE", "'VE", "🙂\r", "字ſ"]} +{"text": "½fi­9½​m㋿😀🏽\r\nḍ̇#$%字𐞁'ſ#$%字Ae \nZ", "tokens": 40, "pieces": ["½", "fi", "­", "9½", "​m", "㋿😀🏽\r\n", "ḋ", "̣#$%", "字𐞁", "'ſ", "#$%", "字Ae", " \n", "Z"]} +{"text": "🙂sé<|fim_prefix|>'ll 'M漢!!!!Dž㍿٣٤٥٦ Zḍ̇<|endoftext|>'ſ <|endoftext|>A,İ#$%", "tokens": 58, "pieces": ["🙂sé", "<|", "fim", "_prefix", "|>'", "ll", " '", "M漢", "!!!!", "Dž", "㍿", "٣٤٥", "٦", " Zḋ", "̣<|", "endoftext", "|>'", "ſ", " <|", "endoftext", "|>", "A", ",İ", "#$%"]} +{"text": "('Re'ſ'Re", "tokens": 9, "pieces": ["('", "Re", "'ſ", "'", "Re"]} +{"text": " å\r\n\r\n<ßfifié<|fim_prefix|>'reA'DžⅣ́!!#$%'<|fim_prefix|>ع㍿", "tokens": 46, "pieces": [" a", "̊\r\n\r\n", "<ßfifie", "́<|", "fim", "_prefix", "|>'", "reA", "'Dž", "Ⅳ", "́!!#$%'<|", "fim", "_prefix", "|><", "META", "_START", ">ع", "㍿"]} +{"text": "\r\n\r\n­
३\n字'ſ'T𐞁\u000b's\rsd<|endoftext|>fifiDž!'s ", "tokens": 38, "pieces": ["\r\n\r\n", "­", "
", "३", "\n", "字", "'ſ", "'T", "𐞁", "\u000b", "'s", "\r", "sd", "<|", "endoftext", "|>", "fifiDž", "!'", "s", " "]} +{"text": "\u000b ,é\r\n\r\n\r 'S
😀🏽-'re𐞁\r\n\r\n\r\n\r\n", "tokens": 24, "pieces": ["\u000b ", " ,", "e", "́\r\n\r\n\r", " ", " '", "S", "
", "😀🏽-'", "re𐞁", "\r\n\r\n\r\n\r\n"]} +{"text": "‍-ß
te👍🏽…d-́\r\n<|endoftext|>🙂!!\n<|endoftext|>fi\r\n\r\n​Ⅳ\n", "tokens": 46, "pieces": ["‍-", "ß", "
", "te", "👍🏽", "…d", "-́\r\n", "<|", "endoftext", "|>🙂!!\n", "<|", "endoftext", "|>", "fi", "\r\n\r\n", "​", "Ⅳ", "\n"]} +{"text": "\ntſ\t'll३e'ſ'T'',🙂0Dž'TsDž'llḍ̇'D३½👍🏽‍-", "tokens": 46, "pieces": ["\n", "tſ", "\t", "'ll", "३", "e", "'ſ", "'T", "'',🙂", "0", "Dž", "'T", "sDž", "'ll", "ḋ", "̣'", "D", "३", "", "½", "👍🏽‍-"]} +{"text": "m \t<㋿#$%́12345678🙂'S漢 \n's‍㍿'VE​ßs12345678t'SZ0,,9'ſ<|endoftext|>!12345678'ſ३🙂<|fim_prefix|><>A", "tokens": 71, "pieces": ["m", " ", "\t", "<㋿#$%́", "123", "456", "78", "🙂'", "S漢", " \n", "'s", "‍㍿'", "VE", "​ßs", "123", "456", "78", "t", "'S", "Z", "0", ",,", "9", "'ſ", "<|", "endoftext", "|>!", "123", "456", "78", "'ſ", "३", "🙂<|", "fim", "_prefix", "|><>", "A"]} +{"text": "aḍ̇ꟲ!!é(\n'Re ㋿!!>!'\u000bé\r\n\r\n'ſ \nA,é\néfi9 👍🏽,<👍🏽ꟲd'Tḍ̇", "tokens": 64, "pieces": ["aḋ", "̣ꟲ", "!!", "é", "(\n", "'Re", " ㋿!!>!<", "META", "_START", ">'", "\u000bé", "\r\n\r\n", "'ſ", " \n", "A", ",é", "\n", "éfi", "9", " ", " 👍🏽,<👍🏽", "ꟲ", "d", "'T", "ḋ", "̣"]} +{"text": "'s漢0'D 𐞁.>漢….\"\r½'M'sé<|endoftext|>>12345678ع12345678𐞁", "tokens": 40, "pieces": ["'s", "漢", "0", "'D", " 𐞁", ".>", "漢", "…", ".\"\r", "½", "'M", "'s", "é", "<|", "endoftext", "|>>", "123", "456", "78", "ع", "123", "456", "78", "𐞁"]} +{"text": "!!'S\"'ſfi. '", "tokens": 14, "pieces": ["!!'", "S", "\"<", "META", "_START", ">'", "ſfi", ".", " ", "'"]} +{"text": "<ع's", "tokens": 3, "pieces": ["<ع", "'s"]} +{"text": "㍿漢te
'D m \n३<|endoftext|>ⅣDž,.३ \nß字ſß㍿''>9ع.", "tokens": 49, "pieces": ["㍿漢t", "e", "
", "'D", " m", " \n", "३", "<|", "endoftext", "|>", "Ⅳ", "Dž", ",.", "३", " \n", "ß字ſß", "㍿''>", "9", "ع", "."]} +{"text": ">🙂😀🏽'Rees're( '12345678‍9​\"'D👍🏽
", "tokens": 44, "pieces": [">🙂😀🏽'", "Rees", "'", "re", "(", " '", "123", "456", "78", "‍", "9", "​\"'", "D", "👍🏽", "
"]} +{"text": " ३'Re\r\n\r9‍
ſ\r\nع\r\n\r\nå'T'Z𐞁'VE ­‍.0!é👍🏽fiEOT\nt🙂EOT><­ꟲ ", "tokens": 68, "pieces": [" ", " ", "३", "'Re", "\r\n\r", "9", "‍", "
", "ſ", "\r\n", "ع", "\r\n\r\n", "a", "̊'", "T", "'Z𐞁", "'VE", " ", " ­‍.", "0", "!", "é", "👍🏽", "fiEOT", "\n", "t", "🙂EOT", "><­", "ꟲ", " "]} +{"text": "Z'VEEOT<|endoftext|>㋿'s😀🏽9#$%😀🏽d 'Re((Džꟲ>< ع'\u000bé'VE㍿#$%'ReEOT​'T!! ३-'M", "tokens": 64, "pieces": ["Z", "'VE", "EOT", "<|", "endoftext", "|>㋿'", "s", "😀🏽", "9", "#$%😀🏽", "d", " ", "'Re", "((", "Džꟲ", "><", " ع", "'", "\u000be", "́'", "VE", "㍿#$%'", "ReEOT", "​'", "T", "!!", " ", "३", "-'", "M"]} +{"text": "mꟲ\"aé,ḍ̇'VE'Ma‍d👍🏽a…(字\rd'VE9é'Re<|endoftext|> mꟲ're‍½Džꟲ‍,'s 漢", "tokens": 65, "pieces": ["mꟲ", "\"ae", "́,", "ḋ", "̣'", "VE", "'M", "a", "‍d", "👍🏽", "a", "…", "(字", "\r", "d", "'VE", "9", "é", "'Re", "<|", "endoftext", "|>", " mꟲ", "'re", "‍", "½", "Džꟲ", "‍,'", "s", " 漢"]} +{"text": "t<|fim_prefix|>9㍿½EOTå.Z'T‍漢Ⅳ'ſe𐞁", "tokens": 38, "pieces": ["t", "<|", "fim", "_prefix", "|>", "9", "㍿", "½", "EOTa", "̊<", "META", "_START", ">.", "Z", "'T", "‍漢", "Ⅳ", "'ſ", "e𐞁"]} +{"text": "\r\n\r\n0'S,!!aEOTéå're\r\n\r\n", "tokens": 16, "pieces": ["\r\n\r\n", "0", "'S", ",!!", "aEOTe", "́a", "̊'", "re", "\r\n\r\n"]} +{"text": "٣٤٥٦ \néåfiع🙂𐞁>'Reḍ̇'M'lleA'llA", "tokens": 41, "pieces": ["٣٤٥", "٦", " \n", "éa", "̊fi", "ع", "🙂𐞁", ">'", "Reḋ", "̣'", "M", "'ll", "eA", "'ll", "A"]} +{"text": " \n­\u000b­! ḍ̇​!Dž\"𐞁!!12345678­\u000bs٣٤٥٦é'så㍿\r\ńZ…", "tokens": 49, "pieces": [" \n", "­", "\u000b", "­!", " ḋ", "̣​!", "Dž", "\"𐞁", "!!", "123", "456", "78", "­", "\u000bs", "٣٤٥", "٦", "e", "́'", "sa", "̊㍿\r\n", "́Z", "…"]} +{"text": "'s<|fim_prefix|>İ!漢<|fim_prefix|>(ꟲ'reé å𐞁", "tokens": 33, "pieces": ["'s", "<|", "fim", "_prefix", "|>", "İ", "!漢", "<|", "fim", "_prefix", "|>(", "ꟲ", "'re", "e", "́", " ", " a", "̊𐞁"]} +{"text": " 😀🏽<|endoftext|>‍👍🏽\".>\r\n𐞁Dž\n", "tokens": 32, "pieces": [" ", " 😀🏽<|", "endoftext", "|>‍👍🏽\".>\r\n", "𐞁Dž", "\n"]} +{"text": "\r\nḍ̇\r\n\r\nA'D\n….­", "tokens": 15, "pieces": ["\r\n", "ḋ", "̣\r\n\r\n", "A", "'D", "\n", "…", ".­"]} +{"text": "'D ㍿'refi!!<|fim_prefix|>\n👍🏽…𐞁…'re0<|endoftext|>>𐞁'VE‍́ \n'll", "tokens": 52, "pieces": ["'D", " ", " ㍿'", "refi", "!!<|", "fim", "_prefix", "|>\n", "👍🏽", "…𐞁", "…", "'re", "0", "<|", "endoftext", "|>>", "𐞁", "'VE", "‍́", " \n", "'ll"]} +{"text": "\rå", "tokens": 4, "pieces": ["\r", "a", "̊"]} +{"text": "㍿İ'ſ#$%́\r\n\r\n\r\n\r\n\u000b字Z
.a'llEOT 9 <'D\"!é'D'Sm'S‍Z'Dꟲİa#$%<|fim_prefix|> \n", "tokens": 58, "pieces": ["㍿İ", "'ſ", "#$%́\r\n\r\n\r\n\r\n", "\u000b字Z", "
", ".a", "'ll", "EOT", " ", " ", "9", " ", "<'", "D", "\"!", "e", "́'", "D", "'S", "m", "'", "S", "‍Z", "'D", "ꟲİa", "#$%<|", "fim", "_prefix", "|>", " \n"]} +{"text": "'re>'ſ<|fim_prefix|>́
é ", "tokens": 17, "pieces": ["'re", ">'", "ſ", "<|", "fim", "_prefix", "|>́", "
e", "́", " "]} +{"text": "\r\n\r\n'Td(İ'T>dꟲ\r\nm", "tokens": 12, "pieces": ["\r\n\r\n", "'T", "d", "(İ", "'T", ">dꟲ", "\r\n", "m"]} +{"text": ".'Re're'S'Ss'ſAⅣé٣٤٥٦#$%
!-㍿'D 'VEa's㋿'De\"
  ", "tokens": 52, "pieces": [".'", "Re", "'re", "'S", "'S", "s", "'ſ", "A", "Ⅳ", "é", "٣٤٥", "٦", "#$%", "
", "!-㍿'", "D", " ", "'VE", "a", "'s", "㋿'", "D", "e", "\"", "
  "]} +{"text": "३A#$%0m", "tokens": 12, "pieces": ["३", "A", "#$%", "0", "m"]} +{"text": ",\u000b-'ll \n'𐞁Aée'll9fi㋿Dž​\rs字", "tokens": 33, "pieces": [",", "\u000b", "-'", "ll", " \n", "'<", "EOT", ">𐞁Aée", "'ll", "9", "fi", "㋿", "Dž", "​\r", "s字"]} +{"text": "é<|fim_prefix|>\r\n\r\n㍿ ‍é", "tokens": 19, "pieces": ["e", "́<", "META", "_START", "><|", "fim", "_prefix", "|>\r\n\r\n", "㍿", " ", "‍é"]} +{"text": "<|endoftext|>", "tokens": 7, "pieces": ["<|", "endoftext", "|>"]} +{"text": "ſ عfi\r‍-EOT🙂‍'S
 ­'ꟲ#$%\re\"ꟲ#$%<|endoftext|>👍🏽\"09
Ⅳ <|endoftext|>>🙂é", "tokens": 76, "pieces": ["ſ", " عfi", "\r", "‍-", "EOT", "🙂‍'", "S", "
 ", " ­'", "ꟲ", "#$%\r", "e", "\"<", "META", "_START", ">ꟲ", "#$%<|", "endoftext", "|><", "META", "_START", ">👍🏽\"", "09", "
", "Ⅳ", " ", "<|", "endoftext", "|>><", "EOT", ">🙂", "e", "́"]} +{"text": "㋿fi\tDž \n'Tḍ̇éa🙂 \n!!ß😀🏽\r<'reŹ0​\u000b­efi#$%'Sa'S٣٤٥٦
d", "tokens": 55, "pieces": ["㋿fi", "\tDž", " \n", "'T", "ḋ", "̣e", "́a", "🙂", " \n", "!!", "ß", "😀🏽\r", "<'", "reZ", "́", "0", "​", "\u000b", "­efi", "#$%'", "Sa", "'S", "٣٤٥", "٦", "
d"]} +{"text": "0a'såEOTEOTḍ̇'Tß's'VE'D,<|fim_prefix|>…'Dž'S<\r…fim👍🏽ḍ̇३", "tokens": 54, "pieces": ["0", "a", "'s", "a", "̊EOTEOTḋ", "̣'", "Tß", "'s", "'VE", "'D", ",<|", "fim", "_prefix", "|>", "…", "'Dž", "'S", "<\r", "…fim", "👍🏽", "ḋ", "̣", "३"]} +{"text": "㍿!!A'D!'ſ\n٣٤٥٦‍é", "tokens": 23, "pieces": ["㍿!!", "A", "'D", "!'", "ſ", "\n", "٣٤٥", "٦", "‍e", "́"]} +{"text": "m😀🏽½'T𐞁​ß🙂 \n,ß­ \n(́'ſDž­३'EOT>'ſ\rع'M㋿tfiZ'S🙂 ", "tokens": 54, "pieces": ["m", "😀🏽", "½", "'T", "𐞁", "​<", "EOT", ">ß", "🙂", " \n", ",ß", "­", " \n", "(́'", "ſDž", "­", "३", "'EOT", ">'", "ſ", "\r", "ع", "'M", "㋿tfiZ", "'S", "🙂", " "]} +{"text": "åé ½ é\t#$%\r\n'D're'D('st å<|fim_prefix|>t'Td
d \n,㍿'s9'Mḍ̇", "tokens": 51, "pieces": ["a", "̊é", " ", "½", " ", " e", "́", "\t", "#$%\r\n", "'D", "'re", "'D", "('", "st", " a", "̊<|", "fim", "_prefix", "|>", "t", "'T", "d", "
d", " \n", ",㍿'", "s", "", "9", "'M", "ḋ", "̣"]} +{"text": "<ꟲ<|endoftext|>å
'll'ſé字d<|endoftext|>\neeḍ̇\r\n\r\n\r\t\t \"\"td \n're", "tokens": 44, "pieces": ["<ꟲ", "<|", "endoftext", "|>", "a", "̊", "
", "'ll", "'ſ", "é字d", "<|", "endoftext", "|>\n", "eeḋ", "̣\r\n\r\n\r", "\t\t", " ", "\"\"", "td", " \n", "'re"]} +{"text": ",‍e😀🏽
('M٣٤٥٦㋿fi㋿", "fi", ",fi👍🏽…e'Reع 'ſ'Re Ⅳ'M'S ", "tokens": 40, "pieces": [".漢", "
d", "㋿🙂<", "META", "_START", ">,", "fi", "👍🏽", "…e", "'Re", "ع", " ", "'ſ", "'Re", " ", " ", "Ⅳ", "'M", "'S", " "]} +{"text": "EOT d\r ſ‍
d३ \n㋿12345678Zé", "tokens": 23, "pieces": ["EOT", " d", "\r", " ſ", "‍", "
d", "३", " \n", "㋿", "123", "456", "78", "Ze", "́"]} +{"text": "-AEOT!!fi \n🙂<漢,½३é字'ś 字'ſtsd漢12345678", "tokens": 34, "pieces": ["-AEOT", "!!", "fi", " \n", "🙂<", "漢", ",", "½३", "e", "́字", "'s", "́", " 字", "'ſ", "tsd漢", "123", "456", "78"]} +{"text": "​(fi#$%­\n\r<|endoftext|><\r\n'VE漢 𐞁٣٤٥٦mDž'ſ́'TEOT<|fim_prefix|>ſ", "tokens": 56, "pieces": ["​(", "fi", "#$%­\n\r", "<|", "endoftext", "|><\r\n", "'VE", "漢", " 𐞁", "٣٤٥", "٦", "mDž", "'ſ", "́'", "TEOT", "<|", "fim", "_prefix", "|>", "ſ"]} +{"text": "9字٣٤٥٦t٣٤٥٦ ­​\r\n३", "tokens": 25, "pieces": ["9", "字", "٣٤٥", "٦", "t", "٣٤٥", "٦", " ", "­​\r\n", "३"]} +{"text": "ع'M!½<|fim_prefix|>EOTeꟲ \nſ!<👍🏽½३a!", "tokens": 34, "pieces": ["ع", "'M", "!", "½", "<|", "fim", "_prefix", "|><", "EOT", ">EOTeꟲ", " \n", "ſ", "!<👍🏽", "½३", "a", "!"]} +{"text": "'re!!İ٣٤٥٦İ( ,\tZ#$%
İ😀🏽…İ.'Re 'ReEOT#$%'T#$%字<|fim_prefix|>m­İ", "tokens": 55, "pieces": ["'re", "!!", "İ", "٣٤٥", "٦", "İ", "(", " ", ",", "\tZ", "#$%", "
İ", "😀🏽", "…İ", ".'", "Re", " ", "'Re", "EOT", "#$%'", "T", "#$%", "字", "<|", "fim", "_prefix", "|><", "EOT", ">m", "­İ"]} +{"text": "s\u000b‍'re 'ſfi0ma.​9́\t.\n字'Ré'M'll'Sİ,
EOT…👍🏽\t'VE​s,", "tokens": 60, "pieces": ["٣٤٥", "٦", "ß", "<|", "fim", "_prefix", "|>", " '", "ſfi", "0", "ma", ".​", "9", "́", "\t", ".\n", "字", "'Re", "́'", "M", "'", "ll", "'S", "İ", ",", "
EOT", "…", "👍🏽", "\t", "'VE", "​s", ","]} +{"text": "'S\"Ⅳ.\ta३ A(㋿'VEß!!é'VE😀🏽-'VE٣٤٥٦🙂 e'VE'\t0ß👍🏽字#$% \r\n", "tokens": 61, "pieces": ["'S", "\"", "Ⅳ", ".", "\ta", "३", " A", "(㋿'", "VEß", "!!", "é", "'VE", "😀🏽-'", "VE", "٣٤٥", "٦", "🙂", " e", "'VE", "'", "\t", "0", "ß", "👍🏽", "字", "#$%", " \r\n"]} +{"text": "12345678ſ​ #$%ꟲ !'VEm", "tokens": 16, "pieces": ["123", "456", "78", "ſ", "​", " ", " #$%", "ꟲ", " !'", "VEm"]} +{"text": "!  \n字' 'ſ\r\nå", "tokens": 66, "pieces": ["!", "  \n", "字", "'<", "META", "_START", ">A", "", " ", " '", "ſ", "\r\n", "a", "̊"]} +{"text": "'ſ\tꟲ𐞁m9\"…\u000b­‍éع'reⅣ,‍٣٤٥٦('sDžZ'm👍🏽👍🏽", "tokens": 64, "pieces": ["'ſ", "\tꟲ𐞁m", "9", "\"", "…", "\u000b", "­<", "META", "_START", ">‍", "éع", "'", "re", "Ⅳ", ",<", "EOT", ">‍", "٣٤٥", "٦", "('", "sDžZ", "'m", "👍🏽👍🏽"]} +{"text": "'sA 'T'sfi9­'Re ḍ̇e'M­\r\" EOT\u000btDž", "tokens": 37, "pieces": ["'s", "A", " <", "EOT", ">'", "T", "'s", "fi", "9", "­'", "Re", " ", " ḋ", "̣e", "'M", "­\r", "\"", " EOT", "\u000btDž", ""]} +{"text": "'S'VE😀🏽'D,½㍿a <<|endoftext|>'reß㍿👍🏽!'re
é…-
½ع𐞁\"fi,
́é>'D-😀🏽0,㋿\t", "tokens": 75, "pieces": ["'S", "'VE", "😀🏽'", "D", ",", "½", "㍿a", " ", " <<|", "endoftext", "|>'", "reß", "㍿👍🏽!'", "re", "
e", "́", "…", "-", "
", "½", "ع𐞁", "\"fi", ",", "
", "́e", "́>'", "D", "-😀🏽", "0", ",㋿", "\t"]} +{"text": "'S‍𐞁٣٤٥٦३Džaé t<|fim_prefix|>A\"-'Rea\r\né\r\n\u000b \u000b9'S'S>", "tokens": 49, "pieces": ["'S", "‍𐞁", "٣٤٥", "٦३", "Džae", "́", " t", "<|", "fim", "_prefix", "|>", "A", "\"-<", "EOT", ">'", "Rea", "\r\n", "é", "\r\n", "\u000b ", "\u000b", "9", "'S", "'S", ">"]} +{"text": "e'VE'Då'Sḍ̇👍🏽", "tokens": 27, "pieces": ["e", "'VE", "'D", "a", "̊'", "Sḋ", "̣👍🏽<", "EOT", "><", "EOT", ">"]} +{"text": "㍿\r\n'll٣٤٥٦at㋿>Dž😀🏽\n ", "tokens": 27, "pieces": ["㍿\r\n", "'ll", "٣٤٥", "٦", "at", "㋿>", "Dž", "😀🏽\n", " "]} +{"text": "'śé0३😀🏽 \n#$%,#$%<|endoftext|>-İ
!३", "tokens": 30, "pieces": ["'s", "́e", "́", "0३", "😀🏽", " \n", "#$%,#$%<|", "endoftext", "|>-", "İ", "
", "!", "३"]} +{"text": "ſéⅣ\r\n🙂Dž㍿", "tokens": 14, "pieces": ["ſe", "́", "Ⅳ", "\r\n", "🙂Dž", "㍿"]} +{"text": ".́\", fi\r\n\r\n३9 é!!㋿aé👍🏽!!'s३éꟲ\r!!'ſḍ̇ß🙂", "tokens": 45, "pieces": [".́\",", " fi", "\r\n\r\n", "३9", " ", " é", "!!㋿", "ae", "́👍🏽!!'", "s", "३", "éꟲ", "\r", "!!'", "ſḋ", "̣ß", "🙂"]} +{"text": "m'VEs's'll<A👍🏽㋿'ſ\r\r\n\r\n.12345678\tع\r\n\r\n字🙂ḍ̇½'reſ\"!!!!́Ⅳfiꟲ<|fim_prefix|>\r‍s", "tokens": 63, "pieces": ["m", "'VE", "s", "'s", "'ll", "<A", "👍🏽㋿'", "ſ", "\r\r\n\r\n", ".", "123", "456", "78", "\tع", "\r\n\r\n", "字", "🙂ḋ", "̣", "½", "'re", "ſ", "\"!!!!́", "Ⅳ", "fiꟲ", "<|", "fim", "_prefix", "|>\r", "‍s"]} +{"text": " \n\tDžḍ̇EOTe ㋿'Re𐞁\r'VEaDž<|endoftext|>9#$%ꟲå字're
!\r\n😀🏽'reAaḍ̇ Dž 'M漢\r\n'ſ're", "tokens": 81, "pieces": [" \n", "\tDžḋ", "̣EOTe", " ", " <", "META", "_START", ">㋿'", "Re𐞁", "\r", "'VE", "aDž", "<|", "endoftext", "|>", "9", "#$%", "ꟲa", "̊字", "'re", "
", "!\r\n", "😀🏽'", "reAaḋ", "̣", " ", " Dž", " ", "'M", "漢", "\r\n", "'ſ", "'re"]} +{"text": "漢'Ré'reEOT​m", "tokens": 10, "pieces": ["漢", "'Re", "́'", "reEOT", "​m"]} +{"text": "‍é'Dd……!e३>.<Ⅳ३३", "tokens": 27, "pieces": ["‍", "é", "'D", "d", "…", "…", "!<", "EOT", ">e", "३", ">.<", "Ⅳ३३"]} +{"text": "'TDž're0're\r\n'Re
Ⅳḍ̇ ㋿‍ß-'TdEOTfi å!(fiḍ̇ \n", "tokens": 43, "pieces": ["'T", "Dž", "'re", "0", "'re", "\r\n", "'Re", "
", "Ⅳ", "ḋ", "̣", " ", "㋿‍", "ß", "-'", "TdEOTfi", " ", " a", "̊!(", "fiḋ", "̣", " \n"]} +{"text": "'s½,­३d‍ꟲ\t漢㋿𐞁.t… \r\n\r\n9fi字­٣٤٥٦👍🏽fi👍🏽('s12345678\r\n\r\nEOT \n'ع#$%'ll½", "tokens": 69, "pieces": ["'s", "½", ",­", "३", "d", "‍ꟲ", "\t漢", "㋿𐞁", ".t", "… \r\n\r\n", "9", "fi字", "­", "٣٤٥", "٦", "👍🏽", "fi", "👍🏽('", "s", "123", "456", "78", "\r\n\r\n", "EOT", " \n", "'ع", "#$%'", "ll", "½"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\nå,a字字'Re'S😀🏽12345678 ſfiḍ̇'Re#$%㋿'D", "tokens": 39, "pieces": ["\r\n", "a", "̊<", "META", "_START", ">,", "a字字", "'Re", "'S", "😀🏽", "123", "456", "78", " ſfiḋ", "̣'", "Re", "#$%㋿'", "D"]} +{"text": "\r\néDž👍🏽漢", "tokens": 12, "pieces": ["\r\n", "éDž", "👍🏽", "漢"]} +{"text": "A\r\n\r\n\r\n㍿عtZⅣ​d'retſ!<|fim_prefix|> 'VE,", "tokens": 36, "pieces": ["A", "\r\n\r\n\r\n", "㍿عtZ", "Ⅳ", "​<", "EOT", ">d", "'re", "tſ", "!<", "EOT", "><", "META", "_START", "><|", "fim", "_prefix", "|>", " '", "VE", ","]} +{"text": " \n", "tokens": 1, "pieces": [" \n"]} +{"text": "t-\t", "tokens": 3, "pieces": ["t", "-", "\t"]} +{"text": "'M12345678é<|endoftext|>'Sḍ̇㍿é!ſḍ̇a'ſfi're#$%́", "tokens": 41, "pieces": ["'M", "123", "456", "78", "é", "<|", "endoftext", "|>'", "Sḋ", "̣㍿", "e", "́!", "ſḋ", "̣a", "'ſ", "fi", "'re", "#$%́"]} +{"text": "'T ع(㋿\"­İ", "tokens": 13, "pieces": ["'T", " ع", "(㋿<", "META", "_START", ">\"­", "İ"]} +{"text": "字'lĺ­!㍿'ſ<|fim_prefix|>½a0\t<|endoftext|>३😀🏽\r\n\r\n.m​'ſfiåEOT'‍'M\r\n\r\n字\r\n\r\n㋿s're", "tokens": 70, "pieces": ["字", "'ll", "́­!㍿'", "ſ", "<|", "fim", "_prefix", "|>", "½", "a", "0", "", "\t", "<|", "endoftext", "|>", "३", "😀🏽\r\n\r\n", ".m", "​'", "ſfia", "̊EOT", "'‍'", "M", "\r\n\r\n", "字", "\r\n\r\n", "㋿s", "'", "re"]} +{"text": ",…<|fim_prefix|>­ 'D­\r\ne\u000b'SⅣ<|fim_prefix|>.\u000b😀🏽s'VEa\t𐞁'Rea Ⅳ", "tokens": 50, "pieces": [",", "…", "<|", "fim", "_prefix", "|>­", " ", "'D", "­\r\n", "e", "\u000b", "'S", "Ⅳ", "<|", "fim", "_prefix", "|>.", "\u000b", "😀🏽", "s", "'VE", "a", "", "\t𐞁", "'Re", "a", " ", "Ⅳ"]} +{"text": "­'s#$%\u000bİ", "tokens": 7, "pieces": ["­'", "s", "#$%", "\u000bİ"]} +{"text": "㋿'M​", "tokens": 6, "pieces": ["㋿'", "M", "​"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "字12345678㋿ß\u000b\u000b \n½t字'S9½\r🙂å३0½< ſ\r'sſ'sع\r-d‍ 'refi's", "tokens": 57, "pieces": ["字", "123", "456", "78", "㋿ß", "\u000b\u000b \n", "½", "t字", "'S", "9½", "\r", "🙂a", "̊", "३0½", "<", " ", " ſ", "\r", "'s", "ſ", "'s", "ع", "\r", "-d", "‍", " ", "'re", "fi", "'s"]} +{"text": "0‍٣٤٥٦
0­éß漢d漢
漢's-ꟲås\r字'll-.İsA ​́", "tokens": 46, "pieces": ["0", "‍", "٣٤٥", "٦", "
", "0", "­e", "́ß漢d漢", "
漢", "'s", "-ꟲa", "̊s", "\r", "字", "'ll", "-.", "İsA", " ​́"]} +{"text": ". é३'re.>३'Mḍ̇'re👍🏽<|fim_prefix|>'T́漢12345678'VE\r\n\r\n👍🏽­'ſDž", "tokens": 53, "pieces": [".", " é", "३", "'re", ".>", "३", "'M", "ḋ", "̣'", "re", "👍🏽<|", "fim", "_prefix", "|>'", "T", "́漢", "123", "456", "78", "'VE", "\r\n\r\n", "👍🏽­'", "ſDž"]} +{"text": "\"ⅣEOT'VE
'VE!!…\u000bd'S>\"", "tokens": 18, "pieces": ["\"", "Ⅳ", "EOT", "'VE", "
", "'VE", "!!", "…", "\u000bd", "'S", ">\""]} +{"text": "­\nŹ\r\n", "tokens": 5, "pieces": ["­\n", "Z", "́\r\n"]} +{"text": ">'T㍿fi#$%­字Ⅳ0'M'İmⅣEOT-👍🏽 A!EOTⅣ\n0\r\n'VEe\u000b ع\u000b漢🙂­", "tokens": 59, "pieces": [">'", "T", "㍿", "fi", "#$%­", "字", "Ⅳ0", "'M", "'<", "EOT", ">İm", "Ⅳ", "EOT", "-👍🏽", " ", " A", "!EOT", "Ⅳ", "\n", "0", "\r\n", "'VE", "e", "\u000b ", " ع", "\u000b漢", "🙂­"]} +{"text": "tA \"s's\n's ­<\n𐞁,AEOT㍿A\r\n\r\n½漢\rEOT'EOTs.", "tokens": 36, "pieces": ["tA", " ", "\"s", "'s", "\n", "'s", " ", "­<\n", "𐞁", ",AEOT", "㍿A", "\r\n\r\n", "½", "漢", "\r", "EOT", "'EOTs", "."]} +{"text": "Džſ😀🏽😀🏽<>​<‍é 'ſEOT<|endoftext|>‍­'Sfi
", "tokens": 46, "pieces": ["Džſ", "😀🏽😀🏽<>​<‍", "é", " ", " '", "ſEOT", "<|", "endoftext", "|>‍­'", "Sfi", "", "
"]} +{"text": " Zåİ12345678字", "tokens": 10, "pieces": [" Za", "̊İ", "123", "456", "78", "字"]} +{"text": "#$%٣٤٥٦İé'Se‍fi!!Dž<'ſ ½", "tokens": 32, "pieces": ["#$%", "٣٤٥", "٦", "İé", "'S", "e", "‍fi", "!!", "Dž", "<'", "ſ", " <", "META", "_START", ">", "½", ""]} +{"text": "👍🏽s", "tokens": 7, "pieces": ["👍🏽", "s"]} +{"text": "#$%\r\n\r\ns٣٤٥٦<|fim_prefix|>३\u000bm9'Re\ta½<|fim_prefix|>\"\nZ​!!\n#$%ß\rm's\"'T", "tokens": 49, "pieces": ["#$%\r\n\r\n", "s", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "३", "\u000bm", "9", "'Re", "\ta", "", "½", "<|", "fim", "_prefix", "|>\"\n", "Z", "​!!\n", "#$%", "ß", "\r", "m", "'s", "\"'", "T"]} +{"text": "'re'T'S\ń\r\n​fi<ع!字12345678\n9'M‍😀🏽́'re\u000b漢😀🏽́'", "re", "\u000b漢", "
'rea \néEOT😀🏽", "tokens": 37, "pieces": ["e字", "<字", "'re", "٣٤٥", "٦ⅣⅣ", "<|", "fim", "_prefix", "|>", "
", "'re", "a", " \n", "éEOT", "😀🏽"]} +{"text": "٣٤٥٦'ſ<|endoftext|>0EOT👍🏽0  ­…å\r\"'T\r\nḍ̇éa-'re<|endoftext|>'S\r\n\r\n", "tokens": 59, "pieces": ["٣٤٥", "٦", "'ſ", "<|", "endoftext", "|>", "0", "EOT", "👍🏽", "0", " ", " ", "­", "…a", "̊\r", "\"'", "T", "\r\n", "ḋ", "̣e", "́a", "-'", "re", "<|", "endoftext", "|>'", "S", "\r\n\r\n"]} +{"text": "Dž​fi're'll", "tokens": 7, "pieces": ["Dž", "​fi", "'re", "'ll"]} +{"text": "\r\nå…,漢
fiß\" \n\r-𐞁'VE>\tm㋿'s
>- \n㍿", "tokens": 38, "pieces": ["\r\n", "a", "̊", "…", ",漢", "
fiß", "\"", " \n\r", "-𐞁", "'VE", ">", "\tm", "㋿'", "s", "
", ">-", " \n", "㍿"]} +{"text": "ma'0½३'D'rea'ſ\u000b😀🏽漢'Ḿ'D!…𐞁 \nee٣٤٥٦ǻ>", "tokens": 46, "pieces": ["ma", "'", "0½३", "'D", "'re", "a", "'ſ", "\u000b", "😀🏽", "漢", "'M", "́'", "D", "!", "…𐞁", " \n", "ee", "٣٤٥", "٦", "a", "̊́>"]} +{"text": "​ع<|endoftext|>㍿ \nZ\u000b\tß'Re\r\n\r\n३EOT", "tokens": 23, "pieces": ["​ع", "<|", "endoftext", "|>㍿", " \n", "Z", "\u000b", "\tß", "'Re", "\r\n\r\n", "३", "EOT"]} +{"text": "'s\tḍ̇ t
 ㋿-٣٤٥٦ße🙂😀🏽ß12345678😀🏽m<|endoftext|>٣٤٥٦字ſ(\r\néfi0é'T", "tokens": 73, "pieces": ["'s", "\tḋ", "̣", " t", "
 ", " ㋿<", "META", "_START", ">-", "٣٤٥", "٦", "ße", "🙂😀🏽", "ß", "123", "456", "78", "😀🏽", "m", "<|", "endoftext", "|>", "٣٤٥", "٦", "字ſ", "(\r\n", "éfi", "0", "e", "́'", "T"]} +{"text": "A'VE<|fim_prefix|>", "tokens": 11, "pieces": ["A", "'VE", "<|", "fim", "_prefix", "|>"]} +{"text": "fi‍Aa'S-9㍿!­EOT.'M>\r\né", "tokens": 25, "pieces": ["fi", "‍", "Aa", "'S", "-", "9", "㍿!­", "EOT", ".'", "M", ">\r\n", "é"]} +{"text": "ſ'VEḍ̇३å漢Ⅳ0m🙂👍🏽ꟲ㍿", "tokens": 34, "pieces": ["ſ", "'VE", "ḋ", "̣", "३", "a", "̊漢", "Ⅳ0", "m", "🙂👍🏽", "ꟲ", "㍿"]} +{"text": "'ſ'll<|endoftext|>< tm३ Z-<|endoftext|>'St.m", "tokens": 27, "pieces": ["'ſ", "'ll", "<|", "endoftext", "|><", " ", " tm", "३", " Z", "-<|", "endoftext", "|>'", "St", ".m"]} +{"text": "\"-Z<|endoftext|>'rea\n  Zéé'Taé0'Méå Dž<|fim_prefix|>ⅣDž9's", "tokens": 43, "pieces": ["\"-", "Z", "<|", "endoftext", "|>'", "rea", "\n", " ", " Zée", "́'", "Tae", "́", "0", "'M", "e", "́a", "̊", " Dž", "<|", "fim", "_prefix", "|>", "Ⅳ", "Dž", "9", "'s"]} +{"text": "Ⅳ😀🏽\r\ndé½\r\n\r\né'D's(", "tokens": 16, "pieces": ["Ⅳ", "😀🏽\r\n", "de", "́", "½", "\r\n\r\n", "é", "'D", "'s", "("]} +{"text": " (🙂're漢İⅣ \nß'Tİ'T'M", "tokens": 65, "pieces": [" ", "(🙂'", "re漢İ", "Ⅳ", "", " \n", "ß", "'T", "İ", "'T", "'M"]} +{"text": "\r\n\r\n!!\"\t ́\tſEOT\t漢𐞁\"ع's㋿(­३'MDž\t\nع\té>d'Sİ\"\n", "tokens": 43, "pieces": ["\r\n\r\n", "!!\"", "\t", " ", "́", "\tſEOT", "\t漢𐞁", "\"<", "EOT", ">ع", "'s", "㋿(­", "३", "'M", "Dž", "\t\n", "ع", "\te", "́>", "d", "'S", "İ", "\"\n"]} +{"text": "\r\n\r\nİ字'T㍿å \n‍.fi\r!!!t!<,Z​́Z", "tokens": 25, "pieces": ["\r\n\r\n", "İ字", "'T", "㍿a", "̊", " \n", "‍.", "fi", "\r", "!!!", "t", "!<,", "Z", "​́", "Z"]} +{"text": "s 'D٣٤٥٦Z\" 's<|endoftext|>\t#$%\n \n'llé'MZm𐞁'D'T", "tokens": 49, "pieces": ["s", " ", " '", "D", "٣٤٥", "٦", "Z", "\"", " ", " '", "s", "<|", "endoftext", "|>", "\t", "#$%\n", " \n", "'ll", "e", "́<", "META", "_START", ">'", "MZm𐞁", "'D", "'", "T", ""]} +{"text": "🙂<|fim_prefix|>🙂'Re12345678\n>Dž#$%'s12345678!é字عa👍🏽A字a漢\u000b0\t#$%", "tokens": 64, "pieces": ["🙂<", "EOT", "><", "me", "́!!'", "M", "<|", "endoftext", "|><|", "fim", "_prefix", "|>🙂'", "Re", "123", "456", "78", "\n", ">Dž", "#$%'", "s", "123", "456", "78", "!e", "́字عa", "👍🏽", "A字a漢", "\u000b", "0", "\t", "#$%"]} +{"text": "‍½'D'D‍sDž㋿-'s\"́é'D d\u000bعEOT​㋿", "tokens": 29, "pieces": ["‍", "½", "'D", "'D", "‍sDž", "㋿-'", "s", "\"́", "é", "'D", " d", "\u000bعEOT", "​㋿"]} +{"text": "å'Tt<|fim_prefix|>字𐞁𐞁\u000b'TⅣ12345678é字", "tokens": 34, "pieces": ["a", "̊'", "T", "t", "<|", "fim", "_prefix", "|>", "字𐞁𐞁", "\u000b", "'T", "Ⅳ12", "345", "678", "e", "́字"]} +{"text": ">😀🏽ꟲ", "tokens": 9, "pieces": [">😀🏽", "ꟲ"]} +{"text": "ß🙂 !İ'ſ<|endoftext|>'re-Zß9ḍ̇\n\téع٣٤٥٦(éEOT>㍿é#$%-m12345678ſDžİ", "tokens": 57, "pieces": ["ß", "🙂", " ", "!İ", "'ſ", "<|", "endoftext", "|>'", "re", "-Zß", "9", "ḋ", "̣\n", "\te", "́ع", "٣٤٥", "٦", "(éEOT", ">㍿", "é", "#$%-", "m", "123", "456", "78", "ſDžİ"]} +{"text": "9'́<|fim_prefix|>🙂\"!​ꟲꟲ\r\n\r\n'reße0Z'D漢", "tokens": 29, "pieces": ["9", "'́<|", "fim", "_prefix", "|>🙂\"!​", "ꟲꟲ", "\r\n\r\n", "'re", "ße", "0", "Z", "'D", "漢"]} +{"text": "!0're,'s'D.'VE#$%👍🏽😀🏽İ​🙂'Sé 'MEOT<٣٤٥٦'s\u000bfi0<|endoftext|>عé", "tokens": 55, "pieces": ["!", "0", "'re", ",'", "s", "'D", ".'", "VE", "#$%👍🏽😀🏽", "İ", "​🙂'", "Sé", " ", " '", "MEOT", "<", "٣٤٥", "٦", "'s", "\u000bfi", "0", "<|", "endoftext", "|>", "عé"]} +{"text": "\u000be㋿👍🏽'D٣٤٥٦'reع'M'll
12345678Ⅳa漢㍿>Ⅳ.-'D'VEs ३­\n", "tokens": 53, "pieces": ["\u000be", "㋿👍🏽'", "D", "٣٤٥", "٦", "'re", "ع", "'M", "'ll", "
", "123", "456", "78Ⅳ", "a漢", "㍿>", "Ⅳ", ".-'", "D", "'VE", "s", " ", " ", "३", "­\n"]} +{"text": "é>'12345678㋿½٣٤٥٦३'Re\"t !!‍ ́", "tokens": 32, "pieces": ["é", ">'", "123", "456", "78", "㋿", "½٣٤", "٥٦३", "'Re", "\"t", " ", "!!‍", " ", "́<", "EOT", ">"]} +{"text": "Ź's𐞁e字A9­ ßA漢tå\rd'MDž<", "tokens": 29, "pieces": ["Z", "́'", "s𐞁e字A", "9", "­", " ßA漢ta", "̊\r", "d", "'M", "Dž", "<"]} +{"text": "\r\n\u000bEOTa'sEOT½0​½'VE-\r\t(👍🏽\u000bꟲ\"İ३\néİ <|endoftext|>'M0", "tokens": 49, "pieces": ["\r\n", "\u000bEOTa", "'s", "EOT", "½0", "​", "½", "'VE", "-\r", "\t", "(👍🏽<", "META", "_START", ">", "\u000bꟲ", "\"İ", "३", "\n", "éİ", " ", "<|", "endoftext", "|>'", "M", "0"]} +{"text": "㋿'T ae🙂ſ0٣٤٥٦👍🏽m,‍\ré'VE Ⅳ'Så­>\n'Re३'Re<​.㍿'Tß'T👍🏽", "tokens": 63, "pieces": ["㋿'", "T", " ae", "🙂ſ", "0٣٤", "٥٦", "👍🏽", "m", ",‍\r", "e", "́'", "VE", " ", "Ⅳ", "'S", "a", "̊­>\n", "'Re", "३", "'Re", "<​.㍿'", "Tß", "'T", "👍🏽"]} +{"text": "<|fim_prefix|>", "tokens": 7, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "a!!EOT'", "tokens": 5, "pieces": ["a", "!!", "EOT", "'"]} +{"text": "ſ🙂å漢ſéfi ", "tokens": 20, "pieces": ["ſ", "🙂a", "̊漢ſe", "́<", "META", "_START", ">fi", " "]} +{"text": "́́
0Dž‍㍿sع'Re'sm'Re
<|fim_prefix|>३å३字👍🏽\n", "tokens": 42, "pieces": ["́́", "
", "0", "Dž", "‍㍿", "sع", "'Re", "'s", "m", "'Re", "
", "<|", "fim", "_prefix", "|>", "३", "a", "̊", "३", "字", "👍🏽\n"]} +{"text": "t\né\t'å \n​,'M\"\u000b½m㋿ \n<‍ß12345678d Z'VE'ſA٣٤٥٦'M#$%éDž", "tokens": 51, "pieces": ["t", "\n", "e", "́", "\t", "'a", "̊", " \n", "​,'", "M", "\"", "\u000b", "½", "m", "㋿", " \n", "<‍", "ß", "123", "456", "78", "d", " Z", "'VE", "'ſ", "A", "٣٤٥", "٦", "'M", "#$%", "éDž"]} +{"text": "'T<'S!!\r\nZDž.㍿ås'em\n'T 12345678'T9🙂 \n's㍿\"'D9's\r.ſ12345678👍🏽", "tokens": 49, "pieces": ["'T", "<'", "S", "!!\r\n", "ZDž", ".㍿", "a", "̊s", "'em", "\n", "'T", " ", "123", "456", "78", "'T", "9", "🙂", " \n", "'s", "㍿\"'", "D", "9", "'s", "\r", ".ſ", "123", "456", "78", "👍🏽"]} +{"text": "me<|fim_prefix|>#$%é​", "tokens": 13, "pieces": ["me", "<|", "fim", "_prefix", "|>#$%", "e", "́​"]} +{"text": "\tst…\reßé 'ſé'S\tḍ̇", "tokens": 21, "pieces": ["\tst", "…\r", "eßé", " ", " '", "ſe", "́'", "S", "\tḋ", "̣"]} +{"text": "İaé<|endoftext|>#$%<'ſs\n", "tokens": 17, "pieces": ["İaé", "<|", "endoftext", "|>#$%<'", "ſs", "\n"]} +{"text": "é\n eZ's­👍🏽 ​'T.mm🙂d<|endoftext|> 's\"!é­ ‍\".", "tokens": 39, "pieces": ["e", "́\n", " eZ", "'s", "­👍🏽", " ", " ​'", "T", ".mm", "🙂d", "<|", "endoftext", "|>", " '", "s", "\"!", "é", "­", " ", " ‍\"."]} +{"text": "🙂​​\u000bⅣ-<|fim_prefix|>'VE½½'ſ…å'll\"😀🏽 \n३Ata<½ \n<", "tokens": 43, "pieces": ["🙂​​", "\u000b", "Ⅳ", "-<|", "fim", "_prefix", "|>'", "VE", "½½", "'ſ", "…a", "̊'", "ll", "\"😀🏽", " \n", "३", "Ata", "<", "½", " \n", "<"]} +{"text": " e'VE👍🏽­EOT<|fim_prefix|>漢ßå​.'re<|fim_prefix|>​😀🏽‍½ع𐞁,é 9­㍿
.…\n ٣٤٥٦aA<|fim_prefix|>s ", "tokens": 85, "pieces": [" ", " e", "'VE", "👍🏽­", "EOT", "<|", "fim", "_prefix", "|>", "漢ßa", "̊​.'", "re", "<|", "fim", "_prefix", "|>​😀🏽‍", "½", "ع𐞁", ",é", " ", "9", "­㍿", "
", ".", "…\n", " ", "٣٤٥", "٦", "aA", "<|", "fim", "_prefix", "|>", "s", " "]} +{"text": "é३!!-a12345678!!", "tokens": 10, "pieces": ["é", "३", "!!-", "a", "123", "456", "78", "!!"]} +{"text": " \n're.", "tokens": 7, "pieces": [" \n", "'", "re", "."]} +{"text": "'T'DA३𐞁'T", "tokens": 11, "pieces": ["'T", "'D", "A", "३", "𐞁", "'T"]} +{"text": " <|endoftext|>", "tokens": 8, "pieces": [" ", "<|", "endoftext", "|>"]} +{"text": "👍🏽…­漢\r\nⅣ åZAZ'T<|fim_prefix|>\n\nsZm\" EOTZ\r\n\r\n
'll 
Z👍🏽漢㋿٣٤٥٦😀🏽EOTꟲ0're½‍", "tokens": 79, "pieces": ["👍🏽", "…", "­漢", "\r\n", "Ⅳ", " a", "̊ZAZ", "'T", "<|", "fim", "_prefix", "|>\n\n", "sZm", "\"", " ", " EOTZ", "\r\n\r\n", "
", "'ll", " ", "
Z", "👍🏽", "漢", "㋿", "٣٤٥", "٦", "😀🏽", "EOTꟲ", "0", "'re", "½", "‍"]} +{"text": "🙂as\nå\r.å​Dž(\t​!!Ⅳd‍>㋿#$%​\u000b字'll<|endoftext|>fi!,", "tokens": 52, "pieces": ["🙂<", "META", "_START", ">as", "\n", "a", "̊\r", ".a", "̊​", "Dž", "(", "\t", "​!!", "Ⅳ", "d", "‍<", "EOT", ">>㋿#$%​", "\u000b字", "'ll", "<|", "endoftext", "|>", "fi", "!,"]} +{"text": "\"٣٤٥٦å٣٤٥٦'s éZ!!👍🏽( 12345678İ́ <|endoftext|>\r\n\r\n< …\r\n\r\n'T'٣٤٥٦㋿😀🏽<|endoftext|>٣٤٥٦te'MmꟲA", "tokens": 89, "pieces": ["\"", "٣٤٥", "٦", "a", "̊", "٣٤٥", "٦", "'s", " ", " éZ", "!!👍🏽(", " ", "123", "456", "78", "İ", "́", " <|", "endoftext", "|>\r\n\r\n", "<", " …\r\n\r\n", "'T", "'", "٣٤٥", "٦", "㋿😀🏽<|", "endoftext", "|>", "٣٤٥", "٦", "te", "'M", "mꟲA"]} +{"text": ">\u000b😀🏽漢́s-'M'sEOT 漢9#$%ß👍🏽,", "tokens": 31, "pieces": [">", "\u000b", "😀🏽", "漢", "́s", "-'", "M", "'s", "EOT", " ", " 漢", "9", "#$%", "ß", "👍🏽,"]} +{"text": "́> DžDžEOTfi\r\n🙂a'Re >EOTAEOT.Z0<|fim_prefix|><|endoftext|>.\"é ss'D0‍ å…", "tokens": 53, "pieces": ["́>", " ", " DžDžEOTfi", "\r\n", "🙂a", "'Re", " ", ">EOTAEOT", ".Z", "0", "<|", "fim", "_prefix", "|><|", "endoftext", "|>.\"", "e", "́", " ss", "'D", "0", "‍", " ", " a", "̊", "…"]} +{"text": "İ'T'sd!å½ḍ̇㋿ع're👍🏽é漢ß\n'VE​Ⅳ\t㍿'VE!,٣٤٥٦'ſ", "tokens": 67, "pieces": ["İ", "'T", "'s", "d", "!<", "m漢ß", "<|", "endoftext", "|>", "a", "̊", "½", "ḋ", "̣㋿", "ع", "'re", "👍🏽", "é漢ß", "\n", "'VE", "​", "Ⅳ", "\t", "㍿'", "VE", "!,", "٣٤٥", "٦", "'ſ"]} +{"text": "EOT𐞁afi  \taDžfí 字‍ß \n", "tokens": 22, "pieces": ["EOT𐞁afi", "  ", "\taDžfi", "́", " 字", "‍ß", " \n"]} +{"text": "'VE\n'ſ'Deéḍ̇\"𐞁Z'ſ ,12345678 \n
Zꟲ 9😀🏽.…!!!eDž<-, ", "tokens": 51, "pieces": ["'VE", "\n", "'ſ", "'D", "eéḋ", "̣\"", "𐞁Z", "'ſ", " ,", "123", "456", "78", " \n", "
Zꟲ", " ", "9", "😀🏽.", "…", "!!!", "eDž", "<-,", " "]} +{"text": "EOTe'T'ſ𐞁<|endoftext|>\n0ZZع(\u000b>½m ", "tokens": 30, "pieces": ["EOTe", "'T", "'ſ", "𐞁", "<|", "endoftext", "|>\n", "0", "Z", "Zع", "(", "\u000b", ">", "½", "m", " "]} +{"text": "sZ‍<|fim_prefix|>'Re!'Té \r🙂ea'D'D<㍿é́ e'D'T'Tſ(EOT😀🏽\r\n\r\n", "tokens": 46, "pieces": ["sZ", "‍<|", "fim", "_prefix", "|>'", "Re", "!'", "Té", " \r", "🙂ea", "'D", "'D", "<㍿", "e", "́́", " <", "EOT", ">e", "'D", "'T", "'T", "ſ", "(EOT", "😀🏽\r\n\r\n"]} +{"text": "'T<İ'S
​('sA३'lled(", "tokens": 44, "pieces": ["ſe", "-", "٣٤٥", "٦", "'s", "A", "३", "'ll", "ed", "("]} +{"text": "''Re'D\t​eſ́.<12345678'S'VE\"ḍ̇'Re​'Reḍ̇s<fiDž''Te🙂ꟲſ'D(​'VEé9́", "tokens": 59, "pieces": ["''", "Re", "'D", "\t", "​eſ", "́.<", "123", "456", "78", "'S", "'VE", "\"ḋ", "̣'", "Re", "​'", "Reḋ", "̣s", "<fiDž", "''", "Te", "🙂ꟲſ", "'", "D", "(​'", "VEé", "9", "́"]} +{"text": ",\t\"\re>Aé!!\r\n\r\n'ś#$%'reEOT9!'S'llEOT\u000b", "tokens": 25, "pieces": [",", "\t", "\"\r", "e", ">Aé", "!!\r\n\r\n", "'s", "́#$%'", "reEOT", "9", "!'", "S", "'ll", "EOT", "\u000b"]} +{"text": "#$%İa'Dß \n
­
㋿…'Re ", "tokens": 19, "pieces": ["#$%", "İa", "'D", "ß", " \n", "
", "­", "
", "㋿", "…", "'Re", " "]} +{"text": "'s 'DEOTé0'll👍🏽mⅣ\u000b'Re'D\n're", "tokens": 25, "pieces": ["'s", " ", "'D", "EOTé", "0", "'ll", "👍🏽", "m", "Ⅳ", "\u000b", "'Re", "'", "D", "\n", "'re"]} +{"text": "a'Ds'VEⅣ're 'VE 9​'Re", "tokens": 21, "pieces": ["a", "'D", "s", "'VE", "", "Ⅳ", "'re", " ", "'VE", " ", " ", "9", "​'", "Re"]} +{"text": "'s­👍🏽…fi\u000b漢,\r\n'reع!!-fis", "tokens": 23, "pieces": ["'s", "­👍🏽", "…fi", "\u000b漢", ",\r\n", "'re", "ع", "!!-", "fis"]} +{"text": ".m  \u000b㋿d'ſ́'D𐞁\r\n'Mså​9EOT㋿\"9.
عa", "tokens": 40, "pieces": [".m", "  ", "\u000b", "㋿d", "'ſ", "́'", "D𐞁", "\r\n", "'M", "sa", "̊​", "9", "EOT", "㋿\"", "9", ".", "
عa"]} +{"text": " \nßA'M 9'S ­'Re're're漢漢's<|fim_prefix|>éé\rDž…ß#$%👍🏽­
'D\r\t#$%Z,'VE३#$%,'re-", "tokens": 65, "pieces": [" \n", "ßA", "'M", " ", " ", "9", "'S", " ", "­'", "Re", "'re", "'re", "漢漢", "'s", "<|", "fim", "_prefix", "|>", "éé", "\r", "Dž", "…ß", "#$%👍🏽­", "
", "'D", "\r", "\t", "#$%", "Z", ",'", "VE", "३", "#$%,'", "re", "-"]} +{"text": "㍿ß½m\u000b<㋿'ſ 'A ſ漢d\råtEOT
‍>", "tokens": 38, "pieces": ["㍿ß", "½", "m", "\u000b", "<㋿'", "ſ", " '", "A", " ſ", "漢d", "\r", "a", "̊tEOT", "
", "‍>"]} +{"text": "\"é३🙂'Reḍ̇!A're", "tokens": 17, "pieces": ["\"é", "३", "🙂'", "Reḋ", "̣!", "A", "'re"]} +{"text": "३A\"m<|fim_prefix|><12345678t'M<|endoftext|>'s㋿٣٤٥٦…\r\n\r\n12345678ś", "tokens": 45, "pieces": ["३", "A", "\"m", "<|", "fim", "_prefix", "|><", "123", "456", "78", "t", "'M", "<|", "endoftext", "|>'", "s", "㋿", "٣٤٥", "٦", "…\r\n\r\n", "123", "456", "78", "s", "́"]} +{"text": "'ll🙂'٣٤٥٦éEOTfi漢EOT<|fim_prefix|>\r\n\r\n \n'VEDž\n<|fim_prefix|>", "tokens": 42, "pieces": ["'ll", "🙂'", "٣٤٥", "٦", "e", "́EOTfi漢EOT", "<|", "fim", "_prefix", "|>\r\n\r\n", " \n", "'VE", "Dž", "\n", "<|", "fim", "_prefix", "|>"]} +{"text": "'re३Dž's!! \n𐞁'll'SEOT9!!㍿éDž\r0é", "tokens": 28, "pieces": ["'re", "३", "Dž", "'s", "!!", " \n", "𐞁", "'ll", "'S", "EOT", "9", "!!㍿", "e", "́Dž", "\r", "0", "é"]} +{"text": "'D-
'12345678́<|endoftext|>#$%٣٤٥٦㍿<|endoftext|>'VE\"åDž,é
\r…'𐞁​́Dž
字 Dž", "tokens": 64, "pieces": ["'D", "-", "
", "'", "123", "456", "78", "́<|", "endoftext", "|>#$%", "٣٤٥", "٦", "㍿<|", "endoftext", "|>'", "VE", "\"a", "̊Dž", ",e", "́", "
\r", "…", "'𐞁", "​́", "Dž", "
字", " Dž"]} +{"text": "​‍\r\n\r\n\"EOT\n\r\n🙂 . ㋿ع́éعع>İsعİ́ 'ſ…́é字", "tokens": 39, "pieces": ["​‍\r\n\r\n", "\"EOT", "\n", "\r\n", "🙂", " ", ".", " ", "㋿ع", "́éعع", ">İsعİ", "́", " ", "'ſ", "…", "́e", "́字"]} +{"text": "\n\n#$% \n\r\tع \nⅣmعs३٣٤٥٦
<\r\n.ḍ̇👍🏽", "tokens": 39, "pieces": ["\n\n", "#$%", " \n\r", "\tع", " \n", "Ⅳ", "mعs", "३٣٤", "٥٦", "
", "<\r\n", ".ḋ", "̣👍🏽"]} +{"text": "('VE㋿fiḍ̇ ", "tokens": 17, "pieces": ["('", "VE", "㋿<", "META", "_START", ">fiḋ", "̣", " "]} +{"text": "m't", "tokens": 2, "pieces": ["m", "'t"]} +{"text": "9\r#$% 'Re𐞁\rt#$%'SEOTİ𐞁é'0sḍ̇!!'ſtꟲ9a<|fim_prefix|>\"㋿'Re㋿, ٣٤٥٦👍🏽-e👍🏽Ⅳ㋿​", "tokens": 86, "pieces": ["9", "\r", "#$%", " ", "'Re", "𐞁", "\r", "t", "#$%'", "SEOTİ𐞁e", "́'", "0", "sḋ", "̣!!'", "ſtꟲ", "9", "a", "<|", "fim", "_prefix", "|>\"㋿'", "Re", "㋿,", " ", "٣٤٥", "٦", "👍🏽-", "e", "👍🏽", "Ⅳ", "㋿​"]} +{"text": "𐞁​\n12345678.𐞁0d𐞁㍿३e½A३'re", "tokens": 32, "pieces": ["𐞁", "​\n", "123", "456", "78", ".𐞁", "0", "d𐞁", "㍿", "३", "e", "½", "A", "३", "'re"]} +{"text": "ßé'Re ́#$%'Sm
0'12345678's0\r\ńⅣEOT㍿ \naſİꟲ漢(", "tokens": 38, "pieces": ["ßé", "'Re", " ", " ́#$%'", "Sm", "
", "0", "'", "123", "456", "78", "'s", "0", "\r\n", "́", "Ⅳ", "EOT", "㍿", " \n", "aſİꟲ漢", "("]} +{"text": "t\rſ#$%'Mꟲ'ſ​!!EOT0", "tokens": 18, "pieces": ["t", "\r", "ſ", "#$%'", "Mꟲ", "'ſ", "​!!", "EOT", "0"]} +{"text": "‍afi\n0漢", "tokens": 9, "pieces": ["‍afi", "\n", "0", "漢"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'Smع'll\r\n'VE\r\n\r\n😀🏽\t\nſ漢㍿
'Re.ع\r\n\r\n", "😀🏽", "\t\n", "ſ漢", "㍿", "
", "'Re", ".ع", "½ḍ̇Dž​(ADž'T9𐞁.'Ts㋿<|fim_prefix|>‍Dž½>ſ,Z'Ms.٣٤٥٦d>'D're'ſ😀🏽9<|endoftext|>a", "tokens": 76, "pieces": ["", "½", "ḋ", "̣Dž", "​(", "ADž", "'T", "9", "𐞁", ".'", "Ts", "㋿<|", "fim", "_prefix", "|>‍", "Dž", "½", ">ſ", ",Z", "'M", "s", ".", "٣٤٥", "٦", "d", ">'", "D", "'re", "'ſ", "😀🏽", "9", "<|", "endoftext", "|>", "a"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " İ­İſDž\u000b", "tokens": 9, "pieces": [" İ", "­İſDž", "\u000b"]} +{"text": "३\r\n\r\n㋿A٣٤٥٦eعs!!'s㋿½Ⅳ'Té\r\n\r\n\"'D!<|fim_prefix|>عſ\r\n\r\n\r\ń́\r,\n \n<३", "tokens": 62, "pieces": ["३", "\r\n\r\n", "㋿A", "٣٤٥", "٦", "eعs", "!!'", "s", "㋿", "½Ⅳ", "'T", "e", "́\r\n\r\n", "\"'", "D", "!<|", "fim", "_prefix", "|>", "عſ", "\r\n\r\n\r\n", "́́\r", ",\n", " \n", "<", "३", ""]} +{"text": "Z>sé\t\r\nſß<|endoftext|>…!😀🏽0#$%😀🏽é <|endoftext|> \n99'VE
å٣٤٥٦-fi漢Ⅳ‍'D're㋿'Re,", "tokens": 79, "pieces": ["Z", ">sé", "\t\r\n", "ſß", "<|", "endoftext", "|>", "…", "!😀🏽", "0", "#$%😀🏽", "é", " <|", "endoftext", "|>", " \n", "99", "'VE", "
a", "̊<", "EOT", ">", "٣٤٥", "٦", "-fi漢", "Ⅳ", "‍'", "D", "'re", "㋿'", "Re", ","]} +{"text": "é 👍🏽عdꟲ'T'Re", "tokens": 15, "pieces": ["é", " ", "👍🏽", "عdꟲ", "'T", "'Re"]} +{"text": "'Da\u000b#$% \nZ're <|endoftext|>…​Z 0.\u000b㋿e", "tokens": 26, "pieces": ["'D", "a", "\u000b", "#$%", " \n", "Z", "'re", " <|", "endoftext", "|>", "…", "​Z", " ", "0", ".", "\u000b", "㋿e"]} +{"text": "\u000b३𐞁'M…­\n㋿", "tokens": 15, "pieces": ["\u000b", "३", "𐞁", "'M", "…", "­\n", "㋿"]} +{"text": "'s\n'SZ㍿", "tokens": 7, "pieces": ["'s", "\n", "'S", "Z", "㍿"]} +{"text": "👍🏽!!‍\"ß\"ḍ̇\r\n\r\n३fi", "tokens": 22, "pieces": ["👍🏽!!‍\"", "ß", "\"ḋ", "̣\r\n\r\n", "३", "fi"]} +{"text": "éé'll \nA !-", "tokens": 12, "pieces": ["e", "́e", "́'", "ll", " \n", "A", " ", " !-"]} +{"text": "'Re 'ſtḍ̇… 9漢ß字!#$%𐞁…a12345678㋿9", "tokens": 40, "pieces": ["'Re", " ", " '", "ſtḋ", "̣", "… ", " ", "9", "漢ß字", "!#$%", "𐞁", "…a", "", "123", "456", "78", "㋿", "9"]} +{"text": "#$%字 Dž\r", "tokens": 7, "pieces": ["#$%", "字", " Dž", "\r"]} +{"text": "\" \n㋿<|fim_prefix|>EOT\r\n\r\n ", "tokens": 16, "pieces": ["\"", " \n", "㋿<|", "fim", "_prefix", "|>", "EOT", "\r\n\r\n "]} +{"text": "\r\n\r\n㋿👍🏽!'VEß12345678'll\r'ſ!!ß'DⅣ", "tokens": 30, "pieces": ["\r\n\r\n", "㋿👍🏽!'", "VEß", "123", "456", "78", "'ll", "\r", "'ſ", "!!<", "META", "_START", ">ß", "'D", "Ⅳ"]} +{"text": "EOT😀🏽👍🏽<|fim_prefix|>12345678t'D0!ꟲ0fi…12345678👍🏽é㋿12345678.", "tokens": 56, "pieces": ["EOT", "😀🏽👍🏽<|", "fim", "_prefix", "|>", "123", "456", "78", "t", "'D", "0", "!ꟲ", "", "0", "fi", "…", "123", "456", "78", "👍🏽", "e", "́㋿", "123", "456", "78", "."]} +{"text": "<|endoftext|> <|endoftext|>0 'reع'Ma½", "tokens": 21, "pieces": ["<|", "endoftext", "|>", " ", " <|", "endoftext", "|>", "0", " ", "'re", "ع", "'M", "a", "½"]} +{"text": "\r's‍(. .\n👍🏽0A<|endoftext|>'D\ré­\r\n\r\ń​'D'T", "tokens": 37, "pieces": ["\r", "'s", "‍(.", " ", ".\n", "👍🏽", "0", "A", "<|", "endoftext", "|>'", "D", "\r", "e", "́­\r\n\r\n", "́​'", "D", "'T"]} +{"text": "'ReDž-…🙂漢Z漢
 'll.\r३é​'VE\u000b\"'reé'S 'VE漢0ß👍🏽😀🏽ع\t‍😀🏽ꟲ", "tokens": 71, "pieces": ["'Re", "Dž", "-", "…", "🙂漢Z漢", "
", " ", "'", "ll", ".\r", "३", "é", "​'", "VE", "\u000b", "\"'", "ree", "́'", "S", " '", "VE漢", "0", "ß", "👍🏽😀🏽", "ع", "\t", "‍<", "EOT", ">😀🏽", "ꟲ"]} +{"text": "0<>'Tİ.ḍ̇åDž\r\n\r\nعA", "tokens": 20, "pieces": ["0", "<>'", "Tİ", ".ḋ", "̣a", "̊Dž", "\r\n\r\n", "عA"]} +{"text": "s9ḍ̇字𐞁'll'३A é Z‍½s>'Mİa'Sé#$%té'<|endoftext|>", "tokens": 44, "pieces": ["s", "9", "ḋ", "̣字𐞁", "'ll", "'", "३", "A", " é", " Z", "‍", "½", "s", ">'", "Mİa", "'S", "e", "́#$%", "té", "'<|", "endoftext", "|>"]} +{"text": "𐞁å
t'llZéDž…e'TعEOT#$%#$%\u000b​. EOTa> ́'(ß<|endoftext|>㋿'re😀🏽-.", "tokens": 64, "pieces": ["𐞁a", "̊", "
t", "'ll", "Ze", "́Dž", "…e", "'T", "عEOT", "#$%#$%", "\u000b", "​.", " EOTa", ">", " ", " ́'(", "ß", "<|", "endoftext", "|>㋿'", "re", "😀🏽<", "EOT", ">-."]} +{"text": "Dž٣٤٥٦EOT\r٣٤٥٦ 字å𐞁👍🏽ſ0​ '㋿<|fim_prefix|>𐞁#$%字'llA\t#$%å \né<|fim_prefix|>㍿#$%s'VEع're", "tokens": 91, "pieces": ["Dž", "٣٤٥", "٦", "EOT", "\r", "٣٤٥", "٦", " 字a", "̊𐞁", "👍🏽", "ſ", "0", "​", " ", " '㋿<|", "fim", "_prefix", "|>", "𐞁", "#$%", "字", "'ll", "A", "\t", "#$%", "a", "̊", " \n", "e", "́<|", "fim", "_prefix", "|>㍿#$%", "s", "'VE", "ع", "'re", ""]} +{"text": "…\n½ >'D\r\n\r\n\t­-é\t9​'ll٣٤٥٦mſ漢-Dž٣٤٥٦İEOT'½\u000b\r\n٣٤٥٦漢.ḍ̇ß½", "tokens": 69, "pieces": ["…\n", "½", " >'", "D", "\r\n\r\n", "\t", "­-", "e", "́", "\t", "9", "​'", "ll", "٣٤٥", "٦", "mſ漢", "-Dž", "٣٤٥", "٦", "İEOT", "'", "½", "\u000b\r\n", "٣٤٥", "٦", "漢", ".ḋ", "̣ß", "½"]} +{"text": " \n'Re㋿👍🏽éZ½aſ́-ſ'll'Re🙂(<|fim_prefix|>'\rfi'VEDžİ㋿d!é>", "tokens": 47, "pieces": [" \n", "'Re", "㋿👍🏽", "éZ", "½", "aſ", "́-", "ſ", "'ll", "'Re", "🙂(<|", "fim", "_prefix", "|>'\r", "fi", "'VE", "Džİ", "㋿d", "!é", ">"]} +{"text": " \n''VEḍ̇'T", "tokens": 14, "pieces": [" \n", "''", "VE", "ḋ", "̣'", "T"]} +{"text": "!!#$%\n0é٣٤٥٦å.(<
12345678'ſa\n漢Aḍ̇e t('ſå'll\"Z'VE \"\té#$%mt", "tokens": 63, "pieces": ["!!#$%<", "EOT", ">\n", "0", "e", "́", "٣٤٥", "٦", "a", "̊.(<", "
", "123", "456", "78", "'ſ", "a", "\n", "漢Aḋ", "̣e", " t", "('", "ſa", "̊'", "ll", "\"Z", "'VE", " \"", "\te", "́#$%", "mt"]} +{"text": "'e߅!å''MⅣ㋿'VE", "tokens": 17, "pieces": ["'eß", "…", "!a", "̊''", "M", "Ⅳ", "㋿'", "VE"]} +{"text": "🙂fi<|fim_prefix|>👍🏽é'D'Re\r'ſ", "tokens": 26, "pieces": ["🙂fi", "<|", "fim", "_prefix", "|>👍🏽", "e", "́'", "D", "'Re", "\r", "'ſ"]} +{"text": " ́\r\ń\n😀🏽'ſfi'Re0éß", "tokens": 19, "pieces": [" ", "́\r\n", "́\n", "😀🏽'", "ſfi", "'Re", "0", "éß"]} +{"text": "㍿'re'SA字", "tokens": 9, "pieces": ["㍿'", "re", "'S", "A字"]} +{"text": "é漢\"m#$%!!\r\n\r\n३👍🏽EOT\t漢 9 'D\"٣٤٥٦\r\n\r\n\r\nꟲ#$%́'M- \nعDž \n😀🏽9!", "tokens": 67, "pieces": ["é漢", "\"m", "#$%!!<", "EOT", ">\r\n\r\n", "३", "👍🏽", "EOT", "\t", "漢", " ", " ", "9", " '", "D", "\"", "٣٤٥", "٦", "\r\n\r\n\r\n", "ꟲ", "#$%́'", "M", "-", " \n", "عDž", " \n", "😀🏽", "9", "!<", "EOT", ">"]} +{"text": "d-\te d👍🏽,½‍åå٣٤٥٦", "tokens": 29, "pieces": ["d", "-", "\te", " d", "👍🏽,", "½", "‍a", "̊a", "̊", "٣٤٥", "٦"]} +{"text": "\r\nEOTś٣٤٥٦'D0'S👍🏽字", "tokens": 25, "pieces": ["\r\n", "EOT", "s", "́", "٣٤٥", "٦", "'D", "0", "'S", "👍🏽", "字"]} +{"text": ",é½", "tokens": 3, "pieces": [",e", "́", "½"]} +{"text": "eEOT👍🏽'ſ <|endoftext|>", "tokens": 19, "pieces": ["eEOT", "👍🏽'", "ſ", " ", " <|", "endoftext", "|>"]} +{"text": "ḍ̇\tß𐞁\r'VE\r\n12345678'llſ\u000bé३🙂.­'VEꟲ<|fim_prefix|>\u000b㍿9Z㍿\r\n\r\n0ſ >́'Mfi,'S漢\u000b'<|fim_prefix|>", "tokens": 61, "pieces": ["-", "9", "'ll", "‍!", "漢tDž", "<|", "endoftext", "|>'", "VEꟲ", "<|", "fim", "_prefix", "|>", "\u000b", "㍿", "9", "Z", "㍿\r\n\r\n", "0", "ſ", " ", ">́'", "Mfi", ",'", "S漢", "\u000b", "'<|", "fim", "_prefix", "|>"]} +{"text": "'D!'ll㍿aZ>­EOT­", "EOT", " \t,'ll<'SⅣ漢'!!(a½#$%é㋿eEOT😀🏽t#$%9 \t٣٤٥٦<|endoftext|>'Re a", "tokens": 71, "pieces": ["EOTs", "'VE", "#$%<|", "fim", "_prefix", "|>", " ", "\t", ",'", "ll", "<'", "S", "Ⅳ", "漢", "'!!(", "a", "½", "#$%", "e", "́㋿", "eEOT", "😀🏽", "t", "#$%", "9", " ", "\t", "٣٤٥", "٦", "<|", "endoftext", "|>'", "Re", " a"]} +{"text": "\"09\r\n'T", "tokens": 4, "pieces": ["\"", "09", "\r\n", "'T"]} +{"text": "́😀🏽EOT\r<|fim_prefix|>Z'T٣٤٥٦é㍿ß
 ½\"''VE
12345678عtZ<|endoftext|>", "tokens": 59, "pieces": ["́😀🏽", "EOT", "\r", "<|", "fim", "_prefix", "|>", "Z", "'T", "٣٤٥", "٦", "e", "́㍿", "ß", "
", " ", "½", "\"''", "VE", "
", "123", "456", "78", "عtZ", "<|", "endoftext", "|>"]} +{"text": "d!< \n㋿t😀🏽12345678㋿…e<(‍>ع'Tſ", "tokens": 33, "pieces": ["d", "!<", " \n", "㋿t", "😀🏽", "123", "456", "78", "㋿", "…e", "<(‍>", "ع", "'T", "ſ"]} +{"text": "🙂", "tokens": 2, "pieces": ["🙂"]} +{"text": "dfi​'DⅣéZ- \nſ", "tokens": 15, "pieces": ["dfi", "​'", "D", "Ⅳ", "e", "́Z", "-", " \n", "ſ"]} +{"text": "'Tm👍🏽0🙂12345678́!!½s \n٣٤٥٦\re#$%㋿漢EOT", "tokens": 38, "pieces": ["'T", "m", "👍🏽", "0", "🙂", "123", "456", "78", "́!!", "½", "s", " \n", "٣٤٥", "٦", "\r", "e", "#$%㋿", "漢EOT"]} +{"text": "㍿mméZ-ع,DžDž'S½\r\n\r\n'T㋿\"tm<ꟲ'Re\r\n\r\nEOT<|endoftext|>​", "tokens": 41, "pieces": ["㍿mméZ", "-ع", ",<", "EOT", ">DžDž", "'S", "½", "\r\n\r\n", "'T", "㋿\"", "tm", "<ꟲ", "'Re", "\r\n\r\n", "EOT", "<|", "endoftext", "|>​"]} +{"text": "d\t𐞁m\tém३½👍🏽 \"\tⅣ-'ſ'M", "tokens": 28, "pieces": ["d", "\t𐞁m", "\te", "́m", "३½", "👍🏽", " ", " \"", "\t", "Ⅳ", "-'", "ſ", "'M"]} +{"text": "  m३!!'Re ́e!!ßḍ̇ß İ\n'Té<|endoftext|>9 ", "tokens": 34, "pieces": [" ", " m", "३", "!!'", "Re", " ", "́<", "META", "_START", ">e", "!!", "ßḋ", "̣ß", " İ", "\n", "'T", "é", "<|", "endoftext", "|>", "9", " "]} +{"text": "ḍ̇s\r\r\n\r\n<\t\n🙂>0ß'Re\r\n\r\nAEOTdAt9'T.😀🏽!fi'llé<|endoftext|>'T‍ꟲ
", "tokens": 61, "pieces": ["ḋ", "̣s", "\r\r\n\r\n", "<", "\t\n", "🙂>", "0", "ß", "'Re", "\r\n\r\n", "AEOTdAt", "9", "'T", ".😀🏽<", "META", "_START", ">!", "fi", "'ll", "e", "́<|", "endoftext", "|><", "EOT", ">'", "T", "‍ꟲ", "
"]} +{"text": "\n\r\n#$% ́\u000bm.\t<|endoftext|> ꟲa A­㍿́㋿(́'ll'😀🏽t​", "tokens": 50, "pieces": ["\n\r\n", "#$%", " ́", "\u000b", "m", ".", "\t", "<|", "endoftext", "|>", " ꟲa", " ", " A", "­㍿́㋿(́'", "ll", "'<", "META", "_START", ">😀🏽", "t", "​"]} +{"text": "ſ\"½é>‍#$%🙂're ½ \n<👍🏽am漢­.12345678 \n\t'ſ", "tokens": 37, "pieces": ["ſ", "\"", "½", "é", ">‍#$%🙂'", "re", " ", "½", " \n", "<👍🏽", "am漢", "­.", "123", "456", "78", " \n", "\t", "'ſ"]} +{"text": "㍿'VEå'VE \nEOT <|fim_prefix|>\nåaå'S\n\r‍", "tokens": 32, "pieces": ["㍿'", "VEa", "̊'", "VE", " \n", "EOT", " ", " <|", "fim", "_prefix", "|>\n", "a", "̊aa", "̊'", "S", "\n\r", "‍"]} +{"text": "'Re'll#$%!-'s<­d'Re​a0😀🏽 'S漢\t", "tokens": 27, "pieces": ["'Re", "'ll", "#$%!-<", "META", "_START", ">'", "s", "<­", "d", "'Re", "​a", "0", "😀🏽", " '", "S漢", "\t"]} +{"text": "fiDž😀🏽m''Re!! t>a'å\t(́ݽꟲ-DžAt㍿İ's😀🏽9‍m\ta \"", "tokens": 49, "pieces": ["fiDž", "😀🏽", "m", "''", "Re", "!!", " t", ">a", "'a", "̊", "\t", "(́", "İ", "½", "ꟲ", "-DžAt", "㍿İ", "'s", "😀🏽", "9", "‍m", "\ta", " ", "\""]} +{"text": "­'D're", "tokens": 4, "pieces": ["­'", "D", "'re"]} +{"text": "mḍ̇9ḍ̇fi'M", "tokens": 22, "pieces": ["m", "ḋ", "̣", "9", "ḋ", "̣fi", "'M", ""]} +{"text": "́ \n㍿\r\n\r\n👍🏽👍🏽!!Aa㋿½Dže­🙂\r\n\r\n'T !!dZt👍🏽\u000b>ع'S\"mt", "tokens": 58, "pieces": ["́", " \n", "㍿\r\n\r\n", "👍🏽👍🏽!!", "Aa", "㋿", "½", "Dže", "­🙂\r\n\r\n", "'T", " ", "!!", "dZt", "👍🏽", "\u000b", ">", "ع", "'S", "\"mt"]} +{"text": "👍🏽#$%\t Z<|endoftext|>'Reḍ̇,0😀🏽! \n!́½\r\n\r\n\tſ'​ꟲ\"\u000bß‍EOT\"​(ß👍🏽", "tokens": 61, "pieces": ["👍🏽#$%", "\t ", " Z", "<|", "endoftext", "|>'", "Reḋ", "̣,", "0", "😀🏽!", " \n", "!́", "½", "\r\n\r\n", "\tſ", "'​", "ꟲ", "\"", "\u000bß", "‍EOT", "\"​(", "ß", "👍🏽"]} +{"text": "
'Sé‍Ⅳ<|fim_prefix|>漢9\r\n\r\n​ßs#$%- …​漢1234567812345678-Ⅳ#$%,,㍿é'saß(字'ſ\u000b", "tokens": 63, "pieces": ["
", "'", "Sé", "‍", "Ⅳ", "<|", "fim", "_prefix", "|>", "漢", "9", "\r\n\r\n", "​ßs", "#$%-", " ", "…", "​漢", "123", "456", "781", "234", "567", "8", "-", "Ⅳ", "#$%,,㍿<", "META", "_START", ">e", "́'", "saß", "(字", "'ſ", "\u000b"]} +{"text": "(d\r\n\r\n,'Re\u000bꟲ'VE <|endoftext|>
ḍ̇'ſ 'Re\r\n\r\nİ字👍🏽'M漢 ٣٤٥٦'漢‍ſ​ \n<|fim_prefix|>", "tokens": 73, "pieces": ["(d", "\r\n\r\n", ",'", "Re", "\u000bꟲ", "'VE", " ", "<|", "endoftext", "|>", "
ḋ", "̣'", "ſ", " ", " <", "META", "_START", ">'", "Re", "\r\n\r\n", "İ字", "👍🏽'", "M漢", " ", " ", "٣٤٥", "٦", "'漢", "‍ſ", "​", " \n", "<|", "fim", "_prefix", "|>"]} +{"text": " ३", "tokens": 3, "pieces": [" ", "३"]} +{"text": "​'S<|fim_prefix|>s🙂0-'ſع­t\"㋿ \r\n", "tokens": 25, "pieces": ["​'", "S", "<|", "fim", "_prefix", "|>", "s", "🙂", "0", "-'", "ſع", "­t", "\"㋿", " \r\n"]} +{"text": "ⅣEOT ", "tokens": 8, "pieces": ["Ⅳ", "EOT", "", " "]} +{"text": "EOT", "tokens": 2, "pieces": ["EOT"]} +{"text": "a'ſ12345678🙂漢​ſḍ̇Z-‍m\n'ſDž", "tokens": 30, "pieces": ["a", "'ſ", "123", "456", "78", "🙂漢", "​ſḋ", "̣Z", "-‍", "m", "\n", "'ſ", "Dž"]} +{"text": "EOT字'VE", "tokens": 9, "pieces": ["EOT", "字", "'VE"]} +{"text": "­'re\"\r\n𐞁ع.a'DEOT\"‍\u000b'Re", "tokens": 23, "pieces": ["­'", "re", "\"\r", "\n", "𐞁ع", ".a", "'D", "EOT", "\"‍", "\u000b", "'Re"]} +{"text": "…,e'ſd''VEA٣٤٥٦​DždEOT漢'ſ'll\rſ㍿", "tokens": 43, "pieces": ["…", ",e", "'ſ", "d", "''", "VE", "A", "٣٤٥", "٦", "​", "DždEOT漢", "'ſ", "'ll", "\r", "ſ", "㍿"]} +{"text": "㋿,", "tokens": 4, "pieces": ["㋿,"]} +{"text": "ḍ̇㍿٣٤٥٦<|fim_prefix|>t \"İ''Tfí‍å‍'retd'İé\"d", "tokens": 49, "pieces": ["ḋ", "̣㍿", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "t", " \"", "İ", "''", "Tfi", "́‍", "a", "̊‍'", "ret", "d", "'İe", "́\"", "d"]} +{"text": "'s🙂'D 
عé 's'VE.İ'İ..12345678A\r㋿é ½'ſ'VE-ꟲ #$%<|endoftext|>", "tokens": 73, "pieces": ["'s", "🙂'", "D", " ", "
", "عé", " ", " '", "s", "'VE", ".İ", "'İ", "..", "123", "456", "78", "A", "\r", "㋿e", "́", " ", "½", "'ſ", "'VE", "-ꟲ", " ", " #$%<|", "endoftext", "|>"]} +{"text": "mA🙂​㍿漢'll( eAß12345678\t#$%३t's 'T!!!!><😀🏽漢Dž're", "tokens": 45, "pieces": ["mA", "🙂​㍿", "漢", "'", "ll", "(", " eAß", "123", "456", "78", "\t", "#$%", "३", "t", "'s", " ", "'T", "!!!!><😀🏽", "漢Dž", "'re"]} +{"text": "('S'å's\n‍Ⅳ½\r's'VE 字", "tokens": 19, "pieces": ["('", "S", "'a", "̊'", "s", "\n", "‍", "Ⅳ½", "\r", "'s", "'VE", " ", " 字"]} +{"text": "fi'D㋿", "tokens": 6, "pieces": ["fi", "'D", "㋿"]} +{"text": "'re", "tokens": 1, "pieces": ["'re"]} +{"text": "'D9…٣٤٥٦!!t
🙂字🙂漢-e'Re 'M​\u000b'ſ", "tokens": 32, "pieces": ["'D", "9", "…", "٣٤٥", "٦", "!!", "t", "
", "🙂字", "🙂漢", "-e", "'Re", " ", "'M", "​", "\u000b", "'ſ"]} +{"text": "\nḍ̇fiEOT12345678㍿👍🏽㋿Džع\u000b३<'re𐞁'VE'Mßḍ̇\r>𐞁 90'T½ع𐞁 ‍é<|endoftext|>'reDž३12345678.", "tokens": 81, "pieces": ["\n", "ḋ", "̣fiEOT", "123", "456", "78", "㍿👍🏽㋿", "Džع", "\u000b", "३", "<'", "re𐞁", "'VE", "'M", "ßḋ", "̣\r", ">𐞁", " ", "90", "'T", "½", "ع𐞁", " ", "‍é", "<|", "endoftext", "|>'", "reDž", "३12", "345", "678", "."]} +{"text": " e'Re​é9㍿s​'Mİ,Ⅳ12345678ع'sAs Ⅳ\t.12345678", "tokens": 36, "pieces": [" e", "'Re", "​é", "9", "㍿<", "EOT", ">s", "​'", "Mİ", ",", "Ⅳ12", "345", "678", "ع", "'s", "As", " ", "Ⅳ", "\t", ".", "123", "456", "78"]} +{"text": "'EOT\"…𐞁9é🙂're9,>a,'VEmꟲ-ß٣٤٥٦", "tokens": 35, "pieces": ["'EOT", "\"", "…𐞁", "9", "é", "🙂'", "re", "9", ",>", "a", ",'", "VEmꟲ", "-ß", "٣٤٥", "٦"]} +{"text": "ꟲ­ !!­'S's😀🏽…­Z#$%'re sDž'll12345678'Sعs٣٤٥٦́३t­ ", "tokens": 47, "pieces": ["ꟲ", "­", " ", "!!­'", "S", "'s", "😀🏽", "…", "­Z", "#$%'", "re", " sDž", "'ll", "123", "456", "78", "'S", "عs", "٣٤٥", "٦", "́", "३", "t", "­", " "]} +{"text": "s\"é<|endoftext|>👍🏽!漢>ḍ̇\u000b\tⅣ\u000b12345678𐞁\r\n字\u000bm!㍿('VE'ſ0 \n㍿​'M\r\n\r\n'M12345678fi㋿tⅣ", "tokens": 72, "pieces": ["s", "\"e", "́<|", "endoftext", "|>👍🏽!", "漢", ">ḋ", "̣", "\u000b", "\t", "Ⅳ", "\u000b", "123", "456", "78", "𐞁", "\r\n", "字", "\u000bm", "!㍿('", "VE", "'ſ", "0", " \n", "㍿​'", "M", "\r\n\r\n", "'M", "123", "456", "78", "fi", "㋿t", "Ⅳ"]} +{"text": "12345678\n(ß㋿Á12345678'S½<|endoftext|>", "tokens": 26, "pieces": ["123", "456", "78", "\n", "(ß", "㋿A", "́", "123", "456", "78", "'S", "½", "<|", "endoftext", "|>"]} +{"text": "‍㍿", "tokens": 5, "pieces": ["‍㍿"]} +{"text": "é#$%9s\r\nſ'VE<|endoftext|>'ll👍🏽sfi'Re'reå", "tokens": 33, "pieces": ["e", "́#$%", "9", "s", "\r\n", "ſ", "'VE", "<|", "endoftext", "|>'", "ll", "👍🏽", "sfi", "'Re", "'re", "a", "̊"]} +{"text": "Ⅳꟲåå(å<|fim_prefix|>!sİ!\t \n'ſ'ſ३ éDž
́é㍿Dž\r\ńs٣٤٥٦<|endoftext|><|fim_prefix|>", "tokens": 73, "pieces": ["Ⅳ", "ꟲa", "̊a", "̊(", "a", "̊<|", "fim", "_prefix", "|>!", "sİ", "!", "\t \n", "'ſ", "'ſ", "३", " e", "́Dž", "
", "́é", "㍿Dž", "\r\n", "́s", "٣٤٥", "٦", "<|", "endoftext", "|><|", "fim", "_prefix", "|>"]} +{"text": "𐞁\n½٣٤٥٦A㋿0\"å!fi字́'ll ", "tokens": 31, "pieces": ["𐞁", "\n", "½٣٤", "٥٦", "A", "㋿", "0", "\"a", "̊!", "fi字", "́'", "ll", " "]} +{"text": "ꟲⅣ-ꟲaé漢­!'Sḍ̇EOT!İas", "tokens": 26, "pieces": ["ꟲ", "Ⅳ", "-ꟲaé漢", "­!'", "Sḋ", "̣EOT", "!İas"]} +{"text": "Dž'T<|fim_prefix|>字㋿́漢'Dm12345678 ", "tokens": 23, "pieces": ["Dž", "'T", "<|", "fim", "_prefix", "|>", "字", "㋿́", "漢", "'D", "m", "123", "456", "78", " "]} +{"text": ">", "tokens": 1, "pieces": [">"]} +{"text": "'re>'m#$%…!!𐞁\t\n9ſ㍿EOTİ'漢", "tokens": 25, "pieces": ["'re", ">'", "m", "#$%", "…", "!!", "𐞁", "\t\n", "9", "ſ", "㍿EOTİ", "'漢"]} +{"text": "s#$% fi<|fim_prefix|>\u000b 漢A😀🏽ßfi'M'T'T'MZßAꟲ>\n0 !#$%٣٤٥٦\r\nDž'VE٣٤٥٦é'll'ret", "tokens": 71, "pieces": ["s", "#$%", " ", " fi", "<|", "fim", "_prefix", "|>", "\u000b", " 漢A", "😀🏽", "ßfi", "'M", "'T", "'T", "'M", "ZßAꟲ", ">\n", "0", " !#$%", "٣٤٥", "٦", "\r\n", "Dž", "'VE", "٣٤٥", "٦", "é", "'ll", "'re", "t"]} +{"text": "0…'T's\u000bZfi", "tokens": 9, "pieces": ["0", "…", "'T", "'s", "\u000bZfi"]} +{"text": "e漢dé\r\n\r\n'M", "tokens": 11, "pieces": ["e漢", "de", "́\r\n\r\n", "'M"]} +{"text": "😀🏽", "tokens": 5, "pieces": ["😀🏽"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "-é'S<Ⅳ३ ع\tſ!! \n", "tokens": 15, "pieces": ["-e", "́'", "S", "<", "Ⅳ३", " ع", "\tſ", "!!", " \n"]} +{"text": "<", "tokens": 1, "pieces": ["<"]} +{"text": "Z㋿0!!½'ll('Tt'remDžEOT9>'S \u000bDž. \u000bⅣfia३'re漢'M'Re", "tokens": 39, "pieces": ["Z", "㋿", "0", "!!", "½", "'ll", "('", "Tt", "'re", "mDžEOT", "9", ">'", "S", " ", "\u000bDž", ".", " ", "\u000b", "Ⅳ", "fia", "३", "'re", "漢", "'M", "'Re"]} +{"text": "😀🏽!‍>'D \na‍ A", "tokens": 16, "pieces": ["😀🏽!‍>'", "D", " \n", "a", "‍", " A"]} +{"text": "Z<|fim_prefix|>𐞁'llé", "tokens": 18, "pieces": ["Z", "<|", "fim", "_prefix", "|>", "𐞁", "'", "lle", "́"]} +{"text": "AEOT<Dž's👍🏽'VE'VEİ\r\n\r\nEOT! \n३9t", "tokens": 28, "pieces": ["AEOT", "<Dž", "'s", "👍🏽'", "VE", "'VE", "İ", "\r\n\r\n", "EOT", "!", " \n", "३9", "t"]} +{"text": "\r\n½\ne", "tokens": 4, "pieces": ["\r\n", "½", "\n", "e"]} +{"text": "'DEOTeꟲs\rEOT㋿­'ſ12345678'", "tokens": 27, "pieces": ["'", "DEOTeꟲs", "\r", "EOT", "㋿­<", "EOT", ">'", "ſ", "123", "456", "78", "'"]} +{"text": "­ſ٣٤٥٦å12345678", "tokens": 17, "pieces": ["­ſ", "٣٤٥", "٦", "a", "̊", "123", "456", "78"]} +{"text": "12345678'VEA½sDž字", "tokens": 12, "pieces": ["123", "456", "78", "'VE", "A", "½", "sDž字"]} +{"text": "३'M
(…ZꟲⅣ'VE字½𐞁 ‍", "tokens": 28, "pieces": ["३", "'M", "
", "(", "…Zꟲ", "Ⅳ", "'VE", "字", "½", "𐞁", " ", "‍"]} +{"text": "३'(é𐞁३!½'Sfié!#$%'ſ㍿ß '(ſ'Refi,!𐞁'llt'ſ 🙂 ß😀🏽𐞁és#$%'", "tokens": 60, "pieces": ["३", "'(", "é𐞁", "३", "!", "½", "'S", "fie", "́!#$%'", "ſ", "㍿ß", " ", "'(", "ſ", "'Re", "fi", ",!", "𐞁", "'ll", "t", "'ſ", " ", "🙂", " ß", "😀🏽", "𐞁és", "#$%'"]} +{"text": "👍🏽𐞁é\r\n\r\n!!'ReſßعZ'T!!İ字ſEOT́-é३ 漢'll字漢ſmⅣ,👍🏽m(", "tokens": 54, "pieces": ["👍🏽", "𐞁é", "\r\n\r\n", "!!'", "ReſßعZ", "'T", "!!", "İ字ſEOT", "́-", "é", "३", " 漢", "'ll", "字漢ſm", "Ⅳ", ",👍🏽", "m", "("]} +{"text": " \t<|fim_prefix|>𐞁'll'Tdꟲḍ̇🙂'T\r\né́e​Ⅳ", "tokens": 36, "pieces": [" ", "\t", "<|", "fim", "_prefix", "|>", "𐞁", "'ll", "'T", "dꟲḋ", "̣🙂'", "T", "\r\n", "e", "́́", "e", "​", "Ⅳ"]} +{"text": "ſZß漢!!ꟲ!!㋿㋿ Ⅳ", "tokens": 24, "pieces": ["ſZ", "ß漢", "!!", "ꟲ", "!!㋿㋿", " ", "Ⅳ"]} +{"text": "!ḍ̇​dsåm\r\n\r\n><|endoftext|>,'Re<|fim_prefix|>'llfis\r\n\r\n½", "tokens": 37, "pieces": ["!ḋ", "̣​", "dsa", "̊m", "\r\n\r\n", "><|", "endoftext", "|>,<", "EOT", ">'", "Re", "<|", "fim", "_prefix", "|>'", "llfis", "\r\n\r\n", "½"]} +{"text": "tEOT'D('Re'M><|endoftext|>\u000b‍,0İ's're'T\tA", "tokens": 26, "pieces": ["tEOT", "'D", "('", "Re", "'M", "><|", "endoftext", "|>", "\u000b", "‍,", "0", "İ", "'s", "'re", "'T", "\tA"]} +{"text": "㋿'M'll-ḍ̇é
9Ⅳfi!!><|endoftext|>é字 #$%​0'Re!!İ'll9<字<…ſå", "tokens": 55, "pieces": ["㋿'", "M", "'ll", "-ḋ", "̣e", "́", "
", "", "9Ⅳ", "fi", "!!><|", "endoftext", "|>", "e", "́字", " ", "#$%​", "0", "'Re", "!!", "İ", "'ll", "9", "<字", "<", "…ſa", "̊"]} +{"text": "Z'\u000b٣٤٥٦ſ
\u000b''D'D‍m½\nZ……<'T >ém'MaA\u000b\r\n \n å\u000bå½\r#$%9", "tokens": 58, "pieces": ["Z", "'", "\u000b", "٣٤٥", "٦", "ſ", "
", "\u000b", "''", "D", "'D", "‍m", "½", "\n", "Z", "…", "…", "<'", "T", " ", " >", "ém", "'M", "aA", "\u000b\r\n \n", " ", " a", "̊", "\u000ba", "̊<", "META", "_START", ">", "½", "\r", "#$%", "9"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "😀🏽İDž٣٤٥٦\r're''S\r😀🏽é.0s🙂'ع½ \n's\"(😀🏽\u000b'ſ!!😀🏽İZ漢ſ عé-", "tokens": 66, "pieces": ["😀🏽", "İDž", "٣٤٥", "٦", "\r", "'", "re", "''", "S", "\r", "😀🏽", "e", "́.", "0", "s", "🙂'", "ع", "½", " \n", "'s", "\"(😀🏽", "\u000b", "'ſ", "!!😀🏽", "İZ漢ſ", " عé", "-"]} +{"text": "\r\nEOT
å(٣٤٥٦\t#$%'lld ٣٤٥٦9#$%㋿- ,
 \n\rtḍ̇>​Z", "tokens": 60, "pieces": ["\r\n", "EOT", "
a", "̊(", "٣٤٥", "٦", "\t", "#$%'", "lld", " ", "٣٤٥", "٦9", "#$%㋿-", " ", " ,", "
 \n\r", "tḋ", "̣>​", "Z", ""]} +{"text": "9​ ſ𐞁Ⅳ'ſꟲDž😀🏽عßZ㋿afi…'re'lla\t …字́३Ⅳ'st", "tokens": 49, "pieces": ["9", "​", " ", " ſ𐞁", "Ⅳ", "'ſ", "ꟲDž", "😀🏽", "عßZ", "㋿afi", "…", "'re", "'ll", "a", "\t ", "…字", "́", "३Ⅳ", "'s", "t"]} +{"text": "👍🏽!!\n\nⅣAm,٣٤٥٦,A,Zß'ſ 12345678\n9ḍ̇Ⅳß \n🙂é 👍🏽<|fim_prefix|><|endoftext|>", "tokens": 67, "pieces": ["👍🏽!!\n\n", "Ⅳ", "Am", ",", "٣٤٥", "٦", ",A", ",Zß", "'ſ", " ", "123", "456", "78", "\n", "9", "ḋ", "̣", "Ⅳ", "ß", " \n", "🙂e", "́", " ", " 👍🏽<|", "fim", "_prefix", "|><|", "endoftext", "|>"]} +{"text": ".३'S'S\"a'D­٣٤٥٦
'llAA𐞁0字9", "tokens": 30, "pieces": [".", "३", "'S", "'S", "\"a", "'D", "­", "٣٤٥", "٦", "
", "'ll", "AA𐞁", "0", "字", "9"]} +{"text": "\r\n'!!é'll<|endoftext|>\t'MⅣt漢‍‍𐞁A'ſe>​ EOT'Reſ'-𐞁Z's", "tokens": 50, "pieces": ["\r\n", "'!!", "é", "'ll", "<|", "endoftext", "|>", "\t", "'M", "Ⅳ", "t漢", "‍‍", "𐞁A", "'ſ", "e", ">​", " EOT", "'Re", "ſ", "'-", "𐞁Z", "'s"]} +{"text": "\t<|endoftext|>👍🏽ß'Re漢a9>éå'D12345678İDž…9\t㍿At", "tokens": 43, "pieces": ["\t", "<|", "endoftext", "|>👍🏽", "ß", "'Re", "漢a", "9", ">e", "́a", "̊'", "D", "123", "456", "78", "İDž", "…", "9", "\t", "㍿At"]} +{"text": "漢Dž!é'M‍'VEꟲ\t㋿'ſ…'så're㋿İ9fi!!<|endoftext|>", "tokens": 44, "pieces": ["漢Dž", "!é", "'M", "‍'", "VEꟲ", "\t", "㋿'", "ſ", "…", "'s", "a", "̊'", "re", "㋿İ", "9", "fi", "!!<|", "endoftext", "|>"]} +{"text": "́
fi'Rem\u000b!å'Reſ㋿ …", "tokens": 25, "pieces": ["́", "
fi", "'Re", "m", "\u000b", "!a", "̊'", "Re", "ſ", "㋿", " …"]} +{"text": "<|endoftext|>a(,'DDž\r\n 'T'Så‍ a ㍿\r<|endoftext|>", "tokens": 37, "pieces": ["<|", "endoftext", "|>", "a", "(,'", "DDž", "\r\n", " '", "T", "'S", "a", "̊‍", " a", " ", " ㍿\r", "<|", "endoftext", "|>"]} +{"text": "‍<<|fim_prefix|>9", "tokens": 10, "pieces": ["‍<<|", "fim", "_prefix", "|>", "9"]} +{"text": "​‍ع٣٤٥٦ḍ̇👍🏽!३ Dž​
\n(字 ٣٤٥٦t,\r\n'M'T's'Dfi𐞁-ḍ̇dß𐞁", "tokens": 70, "pieces": ["​‍", "ع", "٣٤٥", "٦", "ḋ", "̣👍🏽!", "३", " Dž", "​", "
\n", "(字", " ", "٣٤٥", "٦", "t", ",\r\n", "'M", "'T", "'s", "'D", "fi𐞁", "-", "ḋ", "̣dß𐞁"]} +{"text": "́'reⅣß'll👍🏽\r\n\r\n​👍🏽'\r\n\r\nß㍿", "tokens": 29, "pieces": ["́<", "EOT", ">'", "re", "Ⅳ", "ß", "'ll", "👍🏽\r\n\r\n", "​👍🏽'\r\n\r\n", "ß", "㍿"]} +{"text": "å… 'M,eع'Ḿ\r‍🙂\" å‍İع𐞁m'TtEOT'S  'D", "tokens": 38, "pieces": ["a", "̊", "…", " ", "'M", ",eع", "'M", "́\r", "‍🙂\"", " a", "̊‍", "İع𐞁m", "'T", "tEOT", "'S", " ", " ", "'D"]} +{"text": "'ReA…,'s\n ", "tokens": 9, "pieces": ["'Re", "A", "…", ",'", "s", "\n "]} +{"text": "aé é're👍🏽a\r\n\r\n字ḍ̇‍12345678('Re's0‍>éé,", "tokens": 41, "pieces": ["aé", " ", " e", "́'", "re", "👍🏽", "a", "\r\n\r\n", "字ḋ", "̣‍", "123", "456", "78", "('", "Re", "'s", "0", "‍>", "e", "́e", "́,"]} +{"text": "0́…½ße‍a𐞁٣٤٥٦<|endoftext|>
", "tokens": 30, "pieces": ["0", "́", "…", "½", "ße", "‍a𐞁", "٣٤٥", "٦", "<|", "endoftext", "|>", "
"]} +{"text": "fimİ<|endoftext|>'M<|fim_prefix|>३ \n
.字Aİt漢é'Re12345678", "tokens": 39, "pieces": ["fimİ", "<|", "endoftext", "|>'", "M", "<|", "fim", "_prefix", "|>", "३", " \n", "
", ".字Aİt漢e", "́'", "Re", "123", "456", "78"]} +{"text": "٣٤٥٦ſ A\u000bİſ're<|fim_prefix|>ع'll…\r\n\r\n!!'Dİ9'Mعåd३ 0‍‍\r\n\r\nd'D­", "tokens": 57, "pieces": ["٣٤٥", "٦", "ſ", " ", " A", "\u000bİſ", "'re", "<|", "fim", "_prefix", "|>", "ع", "'ll", "…\r\n\r\n", "!!'", "Dİ", "9", "'M", "عa", "̊d", "३", " ", "0", "‍‍\r\n\r\n", "d", "'D", "­"]} +{"text": "ꟲe0\r😀🏽 \nḍ̇'S
<|endoftext|>EOT\u000b!!Aİ\n'Reſ", "tokens": 39, "pieces": ["ꟲe", "0", "\r", "😀🏽", " \n", "ḋ", "̣'", "S", "
", "<|", "endoftext", "|>", "EOT", "\u000b", "!!", "Aİ", "\n", "'Re", "ſ"]} +{"text": "('T \n𐞁 \nå👍🏽aꟲ!!­'ſ\r\n\r\na'VE‍'Se㍿\"<|fim_prefix|>>İ'S,", "tokens": 47, "pieces": ["('", "T", " \n", "𐞁", " \n", "a", "̊👍🏽", "aꟲ", "!!­'", "ſ", "\r\n\r\n", "a", "'VE", "‍'", "Se", "㍿\"<|", "fim", "_prefix", "|>>", "İ", "'S", ","]} +{"text": "'T'T👍🏽…<|fim_prefix|>'VE\té㋿", "tokens": 23, "pieces": ["'T", "'T", "👍🏽", "…", "<|", "fim", "_prefix", "|>'", "VE", "\té", "㋿"]} +{"text": "­'re…'T,a…३½ \n-\r \na'VE-ꟲ", "tokens": 23, "pieces": ["­'", "re", "…", "'T", ",a", "…", "३½", " \n", "-\r", " \n", "a", "'VE", "-ꟲ"]} +{"text": "İ!!é \nⅣ\n…\r\n\r\n
m ", "tokens": 19, "pieces": ["İ", "!!", "e", "́", " \n", "", "Ⅳ", "\n…\r\n\r\n", "
m", " "]} +{"text": "ع😀🏽'ſ㋿.t åfiZ", "tokens": 20, "pieces": ["ع", "😀🏽'", "ſ", "㋿.", "t", " a", "̊fiZ"]} +{"text": "'s\r\n!td'M'D!​́ \u000b \n'll😀🏽'Re!!>́9́", "9", "EOT'sع\r\n\r\ndꟲé<㍿efi 🙂'T…'Re", "tokens": 46, "pieces": [".-", "İe", "́#$%", "e", "́-!!", "e", " 漢", "­", " ", "\u000b", ">EOT", "'s", "ع", "\r\n\r\n", "dꟲe", "́<㍿", "efi", " ", "🙂'", "T", "…", "'", "Re"]} +{"text": "\"\u000b're!!\nsſ👍🏽e
…ſ<|endoftext|>'D́s!!é9३\"㍿,
é😀🏽Ⅳ\t<\net-漢fiDžt", "tokens": 63, "pieces": ["\"", "\u000b", "'re", "!!\n", "sſ", "👍🏽", "e", "
", "…ſ", "<|", "endoftext", "|>'", "D", "́s", "!!", "e", "́", "9३", "\"㍿,", "
e", "́😀🏽", "Ⅳ", "\t", "<\n", "et", "-漢fiDžt"]} +{"text": "Ⅳ३㋿'ll'S𐞁 fi\u000b<|fim_prefix|>漢'ſ#$% \nm t‍'VE<|endoftext|> ३ 
", "tokens": 52, "pieces": ["Ⅳ३", "㋿'", "ll", "'S", "𐞁", " fi", "\u000b", "<|", "fim", "_prefix", "|>", "漢", "'ſ", "#$%", " \n", "m", " t", "‍'", "VE", "<|", "endoftext", "|>", " ", "३", " 
"]} +{"text": "< ꟲ…'s#$%\u000b🙂İs\r\n\r\n㋿dßm>\"​\r\n\r\nⅣ'D𐞁-​'Śtß\r­!漢d…३👍🏽", "tokens": 62, "pieces": ["<", " ", " ꟲ", "…", "'s", "#$%", "\u000b", "🙂İs", "\r\n\r\n", "㋿", "dßm", ">\"​\r\n\r\n", "Ⅳ", "'D", "𐞁", "-​'", "S", "́tß", "\r", "­<", "EOT", ">!", "漢d", "…", "३", "👍🏽"]} +{"text": "٣٤٥٦㋿0am<|fim_prefix|>åé
#$%Ⅳ9eⅣ12345678 \n…mßعe㍿ع'", "tokens": 54, "pieces": ["٣٤٥", "٦", "㋿", "0", "am", "<|", "fim", "_prefix", "|>", "a", "̊é", "", "
", "#$%", "Ⅳ9", "e", "Ⅳ12", "345", "678", " \n", "…mßعe", "㍿ع", "'"]} +{"text": "<½'Re'm12345678\nm \n漢Džſ", "tokens": 16, "pieces": ["<", "½", "'Re", "'m", "123", "456", "78", "\n", "m", " \n", "漢Džſ"]} +{"text": "漢<|endoftext|> 'S'sꟲZꟲas<|endoftext|>A㍿́fiſ0'M​912345678ſ", "tokens": 45, "pieces": ["漢", "<|", "endoftext", "|>", " ", "'S", "'s", "ꟲZꟲas", "<|", "endoftext", "|>", "A", "㍿́", "fiſ", "0", "'M", "​", "912", "345", "678", "ſ"]} +{"text": " 'D३éDžİ'Sd
EOT. t", "tokens": 17, "pieces": [" '", "D", "३", "e", "́Džİ", "'S", "d", "
EOT", ".", " t"]} +{"text": "'ll😀🏽'DfiZ'VE  -ꟲ
\rꟲe…
", "tokens": 30, "pieces": ["'ll", "😀🏽'", "DfiZ", "'VE", " ", " ", "-ꟲ", "
\r", "ꟲe", "…
"]} +{"text": "३!\u000bds'll'M\r\n", "tokens": 11, "pieces": ["३", "!", "\u000bds", "'ll", "'M", "\r\n"]} +{"text": "𐞁'll½ḍ̇", "tokens": 11, "pieces": ["𐞁", "'ll", "½", "ḋ", "̣"]} +{"text": "!!ꟲß ß'D(12345678👍🏽\td \r\n\r\nİAſ½dd㋿😀🏽 é0éå<|endoftext|>é३", "tokens": 59, "pieces": ["!!", "ꟲß", " ", " ß", "'D", "(", "123", "456", "78", "👍🏽", "\td", " ", " <", "META", "_START", ">\r\n\r\n", "İAſ", "½", "dd", "㋿😀🏽", " e", "́", "0", "e", "́a", "̊<|", "endoftext", "|>", "é", "३"]} +{"text": "''llma㍿'s<|fim_prefix|>ꟲDž\n!​(-#$%…Z👍🏽eé tꟲ.\"\r…Dž😀🏽'll ", "tokens": 56, "pieces": ["''", "llma", "㍿'", "s", "<|", "fim", "_prefix", "|>", "ꟲDž", "\n", "!​(-#$%", "…Z", "👍🏽", "ee", "́", " tꟲ", ".\"\r", "…Dž", "😀🏽'", "ll", " "]} +{"text": "\r\nt'Re३'re!!\"㍿éß'M\r\n\r\n'Rea½㍿'M<|endoftext|>'ſ'D'Mß🙂Zd字'Re9e \nmé'!!'Re's", "tokens": 51, "pieces": ["\r\n", "t", "'Re", "३", "'re", "!!\"㍿", "éß", "'M", "\r\n\r\n", "'Re", "a", "½", "㍿'", "M", "<|", "endoftext", "|>'", "ſ", "'D", "'M", "ß", "🙂Zd字", "'Re", "9", "e", " \n", "me", "́'!!'", "Re", "'s"]} +{"text": "ß>'Refi12345678ßꟲé🙂\"'S٣٤٥٦!३!𐞁\u000b 👍🏽", "tokens": 41, "pieces": ["ß", ">'", "Refi", "123", "456", "78", "ßꟲe", "́🙂\"'", "S", "٣٤٥", "٦", "!", "३", "!𐞁", "\u000b ", " 👍🏽"]} +{"text": "'SZ­t. (𐞁 \n'Reé
EOT\r'll…\r\n㍿ſ!!!İ ſ,\r\nfi\r㍿㍿\"<|fim_prefix|>ع", "tokens": 54, "pieces": ["'S", "Z", "­t", ".", " ", "(𐞁", " \n", "'Re", "é", "
EOT", "\r", "'ll", "…\r\n", "㍿ſ", "!!!", "İ", " ", " ſ", ",\r\n", "fi", "\r", "㍿㍿\"<|", "fim", "_prefix", "|>", "ع", ""]} +{"text": "#$%>é!!", "tokens": 6, "pieces": ["#$%>", "e", "́!!"]} +{"text": "İḍ̇́9m'T12345678'S12345678s½३́é\r\n\r\n👍🏽'ſ㋿ꟲd 'Reḍ̇‍́😀🏽'Dß're EOT! \r\n\r\n", "tokens": 68, "pieces": ["İḋ", "̣́", "9", "m", "'T", "123", "456", "78", "'S", "123", "456", "78", "s", "½३", "́é", "\r\n\r\n", "👍🏽'", "ſ", "㋿ꟲd", " ", "'Re", "ḋ", "̣‍́<", "META", "_START", ">😀🏽'", "Dß", "'re", " EOT", "!", " \r\n\r\n"]} +{"text": ">漢ß🙂're‍'VE!!<|fim_prefix|>ḍ̇", "tokens": 25, "pieces": [">漢ß", "🙂'", "re", "‍'", "VE", "!!<|", "fim", "_prefix", "|>", "ḋ", "̣"]} +{"text": "'lld'Re!𐞁>!漢Dž12345678\t漢٣٤٥٦㍿", "tokens": 42, "pieces": ["'ll", "d", "'Re", "!𐞁", ">!", "漢", "Dž", "123", "456", "78", "\t漢", "٣٤٥", "٦", "㍿"]} +{"text": "é<|fim_prefix|>Dž٣٤٥٦½👍🏽½'Re9'll𐞁d12345678́ſa ß>\",'TaA \n😀🏽ß!<\r\n\r\nå \n漢<|endoftext|>'llA", "tokens": 78, "pieces": ["e", "́<|", "fim", "_prefix", "|>", "Dž", "٣٤٥", "٦½", "👍🏽", "½", "'Re", "9", "'ll", "𐞁d", "123", "456", "78", "́ſa", " ", " ß", ">\",'", "TaA", " \n", "😀🏽", "ß", "!<\r\n\r\n", "a", "̊", " \n", "漢", "<|", "endoftext", "|>'", "llA"]} +{"text": "­\t Z<|endoftext|>EOTé(amd'M㋿ 字a \nḍ̇ é'VEſAḍ̇'ſß'ſ' <'s'D<", "tokens": 55, "pieces": ["­", "\t", " Z", "<|", "endoftext", "|>", "EOTe", "́(", "amd", "'M", "㋿", " 字a", " \n", "ḋ", "̣", " é", "'VE", "ſAḋ", "̣'", "ſß", "'ſ", "'", " ", "<'", "s", "'D", "<"]} +{"text": "!!m𐞁", "tokens": 6, "pieces": ["!!", "m𐞁"]} +{"text": "Z३\u000b\r'am! \n,<|endoftext|>㋿🙂EOT㋿字#$%!…٣٤٥٦字<|fim_prefix|>!\rſ'S'D½​½ſ>\u000b
\r\n\r\n0", "tokens": 63, "pieces": ["Z", "३", "\u000b\r", "'am", "!", " \n", ",<|", "endoftext", "|>㋿🙂", "EOT", "㋿字", "#$%!", "…", "٣٤٥", "٦", "字", "<|", "fim", "_prefix", "|>!\r", "ſ", "'S", "'D", "½", "​", "½", "ſ", ">", "\u000b
\r\n\r\n", "0"]} +{"text": "\r🙂'D!!'lla-­🙂\u000b.'S ​dfi\rꟲİ㋿-\nⅣ\n é 0", "tokens": 36, "pieces": ["\r", "🙂'", "D", "!!'", "lla", "-­🙂", "\u000b", ".'", "S", " ", "​dfi", "\r", "ꟲİ", "㋿-\n", "Ⅳ", "\n", " é", " ", "0"]} +{"text": "\n'll\r'ſ\n're0👍🏽d", "tokens": 19, "pieces": ["\n", "'ll", "\r", "'ſ", "\n", "'re", "0", "👍🏽", "d"]} +{"text": "!!A

😀🏽\r\n३'M'Reꟲ㍿éDž, A<|endoftext|>", "tokens": 37, "pieces": ["!!", "A", "
", "
", "😀🏽\r\n", "३", "'M", "'Re", "ꟲ", "㍿éDž", ",", " ", " A", "<|", "endoftext", "|>"]} +{"text": "𐞁0-'ſ'VE<|fim_prefix|>a", "tokens": 18, "pieces": ["𐞁", "0", "-'", "ſ", "'VE", "<|", "fim", "_prefix", "|>", "a"]} +{"text": "éA😀🏽
EOT㍿ 😀🏽'VE! ½å'lléⅣ, <\u000b#$%漢.\tts\"'T<|fim_prefix|>\r,ßé'SédADž", "tokens": 67, "pieces": ["e", "́A", "😀🏽", "
EOT", "㍿", " ", "😀🏽'", "VE", "!", " ", " ", "½", "a", "̊'", "lle", "́", "Ⅳ", ",", " ", "<", "\u000b", "#$%", "漢", ".", "\tts", "\"'", "T", "<|", "fim", "_prefix", "|>\r", ",ße", "́'", "Se", "́dADž"]} +{"text": "'T३​漢'll0٣٤٥٦ꟲ ", "tokens": 20, "pieces": ["'T", "३", "​漢", "'ll", "0٣٤", "٥٦", "ꟲ", " "]} +{"text": "㍿m,㍿t\t
<|endoftext|>0tt123456780t'ſß're\u000be㍿٣٤٥٦'llⅣ'ſ½dåꟲ३!!9ḍ̇🙂", "tokens": 67, "pieces": ["㍿m", ",㍿", "t", "\t", "
", "<|", "endoftext", "|>", "0", "tt", "123", "456", "780", "t", "'ſ", "ß", "'re", "\u000be", "㍿", "٣٤٥", "٦", "'ll", "Ⅳ", "'ſ", "½", "da", "̊ꟲ", "३", "!!", "9", "ḋ", "̣🙂"]} +{"text": "!!t,'reDž12345678,s \r\n\r\n<|endoftext|>'Sḍ̇A'S,🙂é'VE12345678éß
\r\n\r\n\r\n\r\n'S\r\nDž'T<|fim_prefix|>é३ſⅣ!!३
İ🙂", "tokens": 71, "pieces": ["!!", "t", ",'", "reDž", "123", "456", "78", ",s", " \r\n\r\n", "<|", "endoftext", "|>'", "Sḋ", "̣A", "'S", ",🙂", "e", "́'", "VE", "123", "456", "78", "e", "́ß", "
\r\n\r\n\r\n\r\n", "'S", "\r\n", "Dž", "'T", "<|", "fim", "_prefix", "|>", "é", "३", "ſ", "Ⅳ", "!!", "३", "
İ", "🙂"]} +{"text": "'M0\r\n\r\n\r0ßEOT'reA
", "tokens": 13, "pieces": ["'M", "0", "\r\n\r\n\r", "0", "ßEOT", "'re", "A", "
"]} +{"text": "å‍A's9'ſ🙂漢's
\u000b\t
a\"👍🏽عع-0Džfi\u000bt", "tokens": 41, "pieces": ["a", "̊‍", "A", "'s", "9", "'ſ", "🙂漢", "'s", "
\u000b\t", "
a", "\"👍🏽", "عع", "-", "0", "Džfi", "\u000bt"]} +{"text": "'set", "tokens": 2, "pieces": ["'s", "et"]} +{"text": "…𐞁fiꟲ", "tokens": 11, "pieces": ["…𐞁fiꟲ"]} +{"text": "'re漢\r\n\u000b'Re㍿ee!!fisḍ̇​३\n<|endoftext|>😀🏽,t'ſ", "tokens": 40, "pieces": ["'re", "漢", "\r\n", "\u000b", "'Re", "㍿ee", "!!", "fisḋ", "̣​", "३", "\n", "<|", "endoftext", "|>😀🏽,", "t", "'ſ"]} +{"text": "🙂 \n𐞁'T­'re(İ'reDž-'reDžs½\r\n>🙂😀🏽ḍ̇'re\r漢́", "tokens": 42, "pieces": ["🙂", " \n", "𐞁", "'T", "­'", "re", "(İ", "'re", "Dž", "-'", "reDžs", "½", "\r\n", ">🙂😀🏽", "ḋ", "̣'", "re", "\r", "漢", "́"]} +{"text": "'lĺ -d㋿\n!!'Re\r\n.é'D­å𐞁\u000bꟲ\r\n\r\n'S!!<|fim_prefix|>­e٣٤٥٦", "tokens": 49, "pieces": ["'ll", "́", " ", " -", "d", "㋿\n", "!!'", "Re", "\r\n", ".e", "́'", "D", "­a", "̊𐞁", "\u000bꟲ", "\r\n\r\n", "'S", "!!<|", "fim", "_prefix", "|>­", "e", "٣٤٥", "٦"]} +{"text": "m'T­>d\u000bA'sⅣ fia🙂\u000b👍🏽fi👍🏽ſtع", "tokens": 36, "pieces": ["m", "'T", "­>", "d", "\u000bA", "'s", "Ⅳ", " fia", "🙂", "\u000b", "👍🏽", "fi", "👍🏽", "ſtع"]} +{"text": "'!\u000b㋿.9३t\u000ba.'Te 'D٣٤٥٦Ⅳ​ İ\"m 'Re'Re𐞁 <|endoftext|> ㋿\rꟲ😀🏽", "tokens": 64, "pieces": ["'!", "\u000b", "㋿.", "9३", "t", "\u000ba", ".'", "Te", " ", "'D", "٣٤٥", "٦Ⅳ", "​", " İ", "\"m", " ", " '", "Re", "'Re", "𐞁", " ", "<|", "endoftext", "|>", " ㋿\r", "ꟲ", "😀🏽"]} +{"text": "\n'Re㍿\nm‍'½ḍ̇Dž \n…  ३.é", "tokens": 31, "pieces": ["\n", "'Re", "㍿\n", "m", "‍'", "½", "ḋ", "̣Dž", " \n", "…  ", " ", "३", ".e", "́"]} +{"text": "ḍ̇㍿'S🙂>\"<|endoftext|>A'३é!! <|fim_prefix|>'३\u000b'T\r\" e🙂", "tokens": 52, "pieces": ["ḋ", "̣㍿'", "S", "🙂>\"<|", "endoftext", "|>", "A", "'", "३", "e", "́!!<", "EOT", ">", " ", "<|", "fim", "_prefix", "|>'", "३", "\u000b", "'T", "\r", "\"", " ", " e", "🙂"]} +{"text": "३𐞁m㋿AⅣ", "tokens": 14, "pieces": ["३", "𐞁m", "㋿A", "Ⅳ"]} +{"text": "fi ", "tokens": 3, "pieces": ["fi", " "]} +{"text": " s're​EOT'reꟲ½'sfié́", "tokens": 20, "pieces": [" ", " s", "'re", "​EOT", "'re", "ꟲ", "½", "'s", "fie", "́́"]} +{"text": " \t­'D🙂'ſ漢d", "tokens": 42, "pieces": [" ", "\t", "­'", "D", "🙂'", "ſ", "<", "EOT", ">漢d"]} +{"text": "字ꟲDž́🙂<|endoftext|>tA's12345678­<|fim_prefix|>'ſ́Z👍🏽mAś's'D'VE#$%'VE\téß㍿ß'M", "tokens": 68, "pieces": ["字ꟲDž", "́🙂<|", "endoftext", "|>", "tA", "'s", "123", "456", "78", "­<|", "fim", "_prefix", "|>'", "ſ", "́Z", "👍🏽", "mAs", "́'", "s", "'", "D", "'VE", "#$%'", "VE", "\téß", "㍿ß", "'M"]} +{"text": "0
३'ll🙂'll", "tokens": 10, "pieces": ["0", "
", "३", "'ll", "🙂'", "ll"]} +{"text": "\r!!'T!!ꟲe٣٤٥٦fi<|fim_prefix|>👍🏽-#$%́𐞁\r\n#$%㍿'sé㍿", "tokens": 52, "pieces": ["\r", "!!'", "T", "!!", "ꟲe", "٣٤٥", "٦", "fi", "<|", "fim", "_prefix", "|>👍🏽-#$%́", "𐞁", "\r\n", "#$%㍿'", "se", "́㍿"]} +{"text": "'D", "tokens": 4, "pieces": ["'", "D"]} +{"text": "ꟲ.­Dž's\r\n 'VE \n٣٤٥٦é!!9fi-𐞁\r\u000b12345678\r\n.EOT12345678İ'ſaſA-㍿ \n0", "tokens": 67, "pieces": ["ꟲ", ".­", "Dž", "'s", "\r\n", " '", "VE", " \n", "٣٤٥", "٦", "é", "!!", "9", "fi", "-𐞁", "\r", "\u000b", "123", "456", "78", "\r\n", ".EOT", "", "123", "456", "78", "İ", "'ſ", "a", "ſA", "-㍿", " \n", "0"]} +{"text": "😀🏽s.​㍿,…å-e'VE9
fi  ", "tokens": 28, "pieces": ["😀🏽", "s", ".​㍿,", "…a", "̊-", "e", "'VE", "9", "
fi", "  "]} +{"text": "\u000b12345678<|endoftext|>", "tokens": 11, "pieces": ["\u000b", "123", "456", "78", "<|", "endoftext", "|>"]} +{"text": "ḍ̇Dž's.\n,㍿𐞁㍿s9'Ta \n<|endoftext|>́👍🏽e'Re'lls \n'll.d
.…m", "tokens": 52, "pieces": ["ḋ", "̣Dž", "'s", ".\n", ",㍿", "𐞁", "㍿s", "9", "'T", "a", " \n", "<|", "endoftext", "|>́👍🏽", "e", "'Re", "'ll", "s", " \n", "'ll", ".d", "
", ".", "…m"]} +{"text": "(㋿#$%'M\r 'DmEOT're'reeå­\u000b,å字aİ👍🏽‍‍fiaḍ̇'re'VE \n,'T!!'så\r", "tokens": 59, "pieces": ["(㋿#$%'", "M", "\r", " ", "'D", "mEOT", "'re", "'re", "ea", "̊­", "\u000b", ",a", "̊字", "aİ", "👍🏽‍‍", "fiaḋ", "̣'", "re", "'VE", " \n", ",'", "T", "!!'", "sa", "̊\r"]} +{"text": ".mİéå😀🏽\r½𐞁ßİ३…<عé'Re\r\r\n\r\n ß​åddع0", "tokens": 39, "pieces": [".mİéa", "̊😀🏽\r", "½", "𐞁ßİ", "३", "…", "<عé", "'Re", "\r\r\n\r\n", " ", " ß", "​a", "̊ddع", "0"]} +{"text": "#$%\nſd३!!३ḍ̇㋿12345678'll, \n'ſ!! <|fim_prefix|>'ſé㍿,0…'ll३'TDž!٣٤٥٦Z😀🏽00AZ", "tokens": 73, "pieces": ["#$%<", "META", "_START", ">\n", "ſd", "३", "!!", "३", "ḋ", "̣㋿", "123", "456", "78", "'ll", ",", " \n", "'ſ", "!!", " <|", "fim", "_prefix", "|>'", "ſé", "㍿,", "0", "…", "'ll", "३", "'T", "Dž", "!", "٣٤٥", "٦", "Z", "😀🏽", "00", "AZ"]} +{"text": "\t'DZ(><,!!عdDžm\nat-👍🏽\tß😀🏽३­́½'Sfimع", "tokens": 49, "pieces": ["\t", "'D", "Z", "(><,!!", "عdDžm", "\n", "at", "-👍🏽", "\tß", "😀🏽", "३", "­́", "½", "'S", "fi", "漢", "mع"]} +{"text": "ß٣٤٥٦ꟲ fi
a", "tokens": 17, "pieces": ["ß", "٣٤٥", "٦", "ꟲ", " fi", "
a"]} +{"text": " 'Ḿ​t é㋿Ⅳ𐞁 ſ漢d‍fim<|fim_prefix|>", "tokens": 35, "pieces": [" ", "'M", "́​", "t", " e", "́㋿", "Ⅳ", "𐞁", " ſ漢d", "‍fim", "<|", "fim", "_prefix", "|>"]} +{"text": "'Re('S\n", "tokens": 4, "pieces": ["'Re", "('", "S", "\n"]} +{"text": "Dž👍🏽Ⅳ\u000bfi'VE㍿at 字eḍ̇\"字…\t字 é'Reſ", "tokens": 41, "pieces": ["Dž", "👍🏽", "Ⅳ", "\u000bfi", "'VE", "㍿at", "", " ", " 字eḋ", "̣\"", "字", "…", "\t字", " é", "'Re", "ſ"]} +{"text": "'D<|fim_prefix|>d𐞁\r'll'Re٣٤٥٦३㍿'ll('ſ́\u000bs…'S\r\n\r\n👍🏽'D'<'VE'SZ09'D'S㋿<|endoftext|>", "tokens": 67, "pieces": ["'D", "<|", "fim", "_prefix", "|>", "d𐞁", "\r", "'ll", "'Re", "٣٤٥", "٦३", "㍿'", "ll", "('", "ſ", "́", "\u000bs", "…", "'S", "\r\n\r\n", "👍🏽'", "D", "'<'", "VE", "'S", "Z", "09", "'D", "'S", "㋿<|", "endoftext", "|>"]} +{"text": "ḍ̇0‍🙂​​'Re0㍿>", "tokens": 18, "pieces": ["ḋ", "̣", "0", "‍🙂​​'", "Re", "0", "㍿>"]} +{"text": "é'M\n\r'M'DA\"", "tokens": 11, "pieces": ["e", "́'", "M", "\n\r", "'M", "'D", "A", "\""]} +{"text": "e㍿' EOTéåDž fiß's'll🙂\u000b'Tİ\r\nİ!", "tokens": 30, "pieces": ["e", "㍿'", " EOTéa", "̊Dž", " fiß", "'", "s", "'ll", "🙂", "\u000b", "'T", "İ", "\r\n", "İ", "!"]} +{"text": "\r0'VE
'VEZعs \t'VE㋿İ ḍ̇́-té😀🏽A'ſ\n\r \n0é'S​", "tokens": 50, "pieces": ["\r", "0", "'VE", "", "
", "'VE", "Zعs", " ", "\t", "'VE", "㋿İ", " ḋ", "̣́-", "te", "́😀🏽", "A", "'ſ", "\n\r \n", "0", "é", "'S", "​"]} +{"text": " ſt \r\nś'M'D", "tokens": 11, "pieces": [" ſt", " \r\n", "s", "́'", "M", "'D"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "9'M漢漢३!!é½''Dß字'll''ſ\r㍿Ⅳ", "tokens": 25, "pieces": ["9", "'M", "漢漢", "३", "!!", "é", "½", "''", "Dß字", "'ll", "''", "ſ", "\r", "㍿", "Ⅳ"]} +{"text": "́", "tokens": 1, "pieces": ["́"]} +{"text": "->\r\n\r\n­'Re​ꟲ<|endoftext|>‍m…‍㋿0
eDž'S字\"<|endoftext|>Ⅳé㍿‍ \nع'M' >́'VEfi", "tokens": 60, "pieces": ["->\r\n\r\n", "­'", "Re", "​ꟲ", "<|", "endoftext", "|>‍", "m", "…", "‍㋿", "0", "
eDž", "'S", "字", "\"<|", "endoftext", "|>", "Ⅳ", "é", "㍿‍", " \n", "ع", "'M", "'", " ", ">́'", "VEfi"]} +{"text": "0㍿'é're.!!ß0", "tokens": 13, "pieces": ["0", "㍿'", "e", "́'", "re", ".!!", "ß", "0"]} +{"text": "'ſ'T🙂 'M\u000b½a'ſ‍字!!m(!!😀🏽ع", "tokens": 38, "pieces": ["'ſ", "'T", "🙂", " ", " '", "M", "", "\u000b", "½", "a", "'ſ", "‍<", "EOT", "><", "EOT", ">字", "!!", "m", "(!!😀🏽", "ع"]} +{"text": "\r\ń12345678Ⅳ😀🏽'Re😀🏽<|fim_prefix|>漢", "tokens": 31, "pieces": ["\r\n", "́", "123", "456", "78", "", "Ⅳ", "😀🏽'", "Re", "😀🏽<|", "fim", "_prefix", "|>", "漢"]} +{"text": "\u000b'Re-<|endoftext|>𐞁-👍🏽'D\u000b!!🙂\u000b\r", "tokens": 32, "pieces": ["\u000b", "'Re", "-<|", "endoftext", "|>", "𐞁", "-👍🏽'", "D", "\u000b", "!!🙂", "\u000b\r"]} +{"text": "'s\t#$%EOT0漢. .(t‍s>-'re\rİ12345678EOT-'", "re", "\r", "İ", "123", "456", "78", "EOT", "'ll👍🏽>,", "tokens": 21, "pieces": [",t", "㍿‍\r\n\r\n", "m", "…", "'", "ll", "👍🏽>,"]} +{"text": "'ll>३\r\r\n\r\ń", "tokens": 7, "pieces": ["'ll", ">", "३", "\r\r\n\r\n", "́"]} +{"text": "​", "tokens": 1, "pieces": ["​"]} +{"text": "'D३😀🏽<|fim_prefix|><'T\r\n\r\nDž!\"tDž\r\n漢. (a<|fim_prefix|>\r\n\r\n-㋿\n.\né<|fim_prefix|>DžⅣ字ad!!٣٤٥٦", "tokens": 66, "pieces": ["'D", "३", "😀🏽<|", "fim", "_prefix", "|><'", "T", "\r\n\r\n", "Dž", "!\"", "tDž", "\r\n", "漢", ".", " ", "(a", "<|", "fim", "_prefix", "|>\r\n\r\n", "-㋿\n", ".\n", "é", "<|", "fim", "_prefix", "|>", "Dž", "Ⅳ", "字ad", "!!", "٣٤٥", "٦"]} +{"text": "\r", "tokens": 1, "pieces": ["\r"]} +{"text": "!!", "tokens": 1, "pieces": ["!!"]} +{"text": ".𐞁​㍿\"e३'M \ń \nfifi\t 😀🏽mt <|fim_prefix|>", "tokens": 37, "pieces": [".𐞁", "​㍿\"", "e", "३", "'M", " \n", "́<", "META", "_START", ">", " \n", "fifi", "\t ", " 😀🏽", "mt", " <|", "fim", "_prefix", "|>"]} +{"text": "Z'MamA 'ſd½㋿㋿a\"d字 Ⅳ<|endoftext|>́İ'ſ'VE'ſ\n12345678㍿ 👍🏽𐞁İ", "tokens": 63, "pieces": ["Z", "'M", "amA", " ", "'ſ", "d", "½", "㋿㋿<", "META", "_START", ">a", "\"d字", " ", " ", "Ⅳ", "<|", "endoftext", "|>́", "İ", "'ſ", "'VE", "'ſ", "\n", "123", "456", "78", "㍿", " ", "👍🏽", "𐞁İ"]} +{"text": "٣٤٥٦!Zſꟲ-…'ll ", "tokens": 20, "pieces": ["٣٤٥", "٦", "!Zſꟲ", "-", "…", "'ll", " "]} +{"text": "'ll‍ İdİ\naEOTd!\r\n\r\n\r\n\r\n\u000b'M'VEꟲ㍿<|endoftext|>0 \n<|endoftext|>'ſ­㋿'ſ", "tokens": 49, "pieces": ["'ll", "‍", " ", " İdİ", "\n", "aEOTd", "!\r\n\r\n\r\n\r\n", "\u000b", "'M", "'VE", "ꟲ", "㍿<|", "endoftext", "|>", "0", " \n", "<|", "endoftext", "|>'", "ſ", "­㋿'", "ſ"]} +{"text": "fi\r\n\r\n…", "tokens": 5, "pieces": ["fi", "\r\n\r\n…"]} +{"text": "ſ>'s'Sa9ع٣٤٥٦\r\n\r\n-Ⅳ\nß", "tokens": 22, "pieces": ["ſ", ">'", "s", "'S", "a", "9", "ع", "٣٤٥", "٦", "\r\n\r\n", "-", "Ⅳ", "\n", "ß"]} +{"text": "٣٤٥٦<|fim_prefix|>0\u000b​(-३é字'S!!!!字fi'ſ<|endoftext|>0 's\u000bZ㍿'Red\r>👍🏽s\n👍🏽'll'll
e", "tokens": 75, "pieces": ["٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "0", "\u000b", "​(-", "३", "e", "́字", "'S", "!!!!", "字fi", "'ſ", "<|", "endoftext", "|>", "0", " ", "'s", "\u000bZ", "㍿'", "Red", "\r", ">👍🏽", "s", "\n", "👍🏽'", "ll", "'ll", "
e"]} +{"text": "­t漢 (\nfi >t\rs!!'T're!!½'refi ­­!!\nḍ̇😀🏽#$%­EOT'T>", "tokens": 40, "pieces": ["­t漢", " (\n", "fi", " >", "t", "\r", "s", "!!'", "T", "'re", "!!", "½", "'re", "fi", " ", "­­!!\n", "ḋ", "̣😀🏽#$%­", "EOT", "'T", ">"]} +{"text": "'S<|fim_prefix|>.㋿\"👍🏽½​ḍ̇\u000b>9", "tokens": 32, "pieces": ["'S", "<|", "fim", "_prefix", "|>.㋿\"👍🏽", "½", "​ḋ", "̣<", "EOT", ">", "\u000b", ">", "9"]} +{"text": "٣٤٥٦😀🏽㍿-0!!'VE''S!!Z(-0'!!́३
", "tokens": 37, "pieces": ["٣٤٥", "٦", "😀🏽㍿-", "0", "!!'", "VE", "''", "S", "!!<", "EOT", ">Z", "(-", "0", "'!!́", "३", "
"]} +{"text": "'ſ㍿!İ're t\r\n‍½ſfi\u000b'Dé!漢­½‍'ſ\"ḍ̇éA<'Ss,", "tokens": 52, "pieces": ["'ſ", "㍿!", "İ", "'re", " t", "\r\n", "‍<", "EOT", ">", "½", "ſfi", "\u000b", "'D", "é", "!漢", "­", "½", "‍'", "ſ", "\"ḋ", "̣éA", "<'", "Ss", ","]} +{"text": ".ḍ̇åİ\"!!ḍ̇ꟲ漢  ­A\nſ>३12345678#$%a३\u000b'ſ9,", "tokens": 49, "pieces": [".ḋ", "̣a", "̊İ", "\"!!", "ḋ", "̣ꟲ漢", " ", " ­", "A", "\n", "ſ", ">", "३12", "345", "678", "#$%", "a", "३", "\u000b", "'ſ", "9", ","]} +{"text": "㋿12345678>٣٤٥٦<|fim_prefix|>,'Re<|fim_prefix|>>́😀🏽ADž9'VE \n're", "tokens": 46, "pieces": ["㋿", "123", "456", "78", ">", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>,'", "Re", "<|", "fim", "_prefix", "|>>́😀🏽", "ADž", "9", "'VE", " \n", "'re"]} +{"text": "Džſ­< \n \rme12345678dعſ­#$%-ḿ…\" (Dž 's‍fi!٣٤٥٦<|endoftext|>\r\n'll
… \n're", "tokens": 62, "pieces": ["Džſ", "­<", " \n \r", "me", "123", "456", "78", "dعſ", "­#$%-", "m", "́", "…", "\"", " ", "(Dž", " ", " '", "s", "‍fi", "!", "٣٤٥", "٦", "<|", "endoftext", "|>\r\n", "'ll", "
… \n", "'re"]} +{"text": "İſ,㍿!a0'M‍'ll漢'VE<|endoftext|>Dž३…!Dž ḍ̇'Dé'll🙂ع३aåß \n,½'Re\r\n\r\n漢'll'ß", "tokens": 65, "pieces": ["İſ", ",㍿!", "a", "0", "'M", "‍'", "ll漢", "'", "VE", "<|", "endoftext", "|>", "Dž", "३", "…", "!Dž", " ḋ", "̣'", "Dé", "'ll", "🙂ع", "३", "aa", "̊ß", " \n", ",", "½", "'Re", "\r\n\r\n", "漢", "'ll", "'ß"]} +{"text": "٣٤٥٦A㋿EOT​\r\n\r\n\"‍'Sa'VE٣٤٥٦Z!ꟲ🙂 'ſ'll", "tokens": 43, "pieces": ["٣٤٥", "٦", "A", "㋿EOT", "​\r\n\r\n", "\"‍'", "Sa", "'VE", "٣٤٥", "٦", "Z", "!ꟲ", "🙂", " '", "ſ", "'ll"]} +{"text": "\t(s(", "tokens": 3, "pieces": ["\t", "(s", "("]} +{"text": "ADž𐞁a-😀🏽é\r 漢
​'\u000b𐞁EOTḍ̇'VE…>㍿\t
字🙂!'s­A#$%'VE漢0ß'M're㍿", "tokens": 71, "pieces": ["ADž𐞁a", "-😀🏽", "e", "́\r", " 漢", "
", "​'", "\u000b", "𐞁EOTḋ", "̣'", "VE", "…", ">㍿", "\t", "
字", "🙂!'", "s", "­A", "#$%'", "VE漢", "0", "ß", "'M", "'re", "㍿"]} +{"text": "-!🙂'll٣٤٥٦ '.㍿Dž!!eſe", "tokens": 26, "pieces": ["-!🙂'", "ll", "٣٤٥", "٦", " ", "'.㍿", "Dž", "!!", "eſe"]} +{"text": " 😀🏽9İ\t㋿…ß m\t", "tokens": 15, "pieces": [" 😀🏽", "9", "İ", "\t", "㋿", "…ß", " m", "\t"]} +{"text": "…㋿🙂tm🙂字>EOT 漢'T­\r\r\n\r\n ٣٤٥٦éé", "tokens": 31, "pieces": ["…", "㋿🙂", "tm", "🙂字", ">EOT", " 漢", "'T", "­\r\r\n\r\n", " ", "٣٤٥", "٦", "éé"]} +{"text": "'ſ\r漢
<|endoftext|>😀🏽\rå😀🏽<|endoftext|>a‍<|fim_prefix|> ع\u000bå", "tokens": 55, "pieces": ["'ſ", "\r", "漢", "
", "<|", "endoftext", "|>😀🏽\r", "a", "̊😀🏽<", "EOT", "><|", "endoftext", "|>", "a", "‍<|", "fim", "_prefix", "|>", " ع", "\u000ba", "̊"]} +{"text": "ß'T'🙂EOTZdEOT🙂 ½!!'D<|endoftext|>fi0\" \n㋿٣٤٥٦-é\"'VE 'S", "tokens": 48, "pieces": ["ß", "'T", "'🙂", "EOTZdEOT", "🙂", " ", " ", "½", "!!'", "D", "<|", "endoftext", "|>", "fi", "0", "\"", " \n", "㋿", "٣٤٥", "٦", "-e", "́\"'", "VE", " ", "'S"]} +{"text": "İ\n's12345678ß<|fim_prefix|>\r\n\r\ns漢 -٣٤٥٦#$% 'Re0!!!㋿\tDž", "tokens": 50, "pieces": ["İ", "\n", "'s", "123", "456", "78", "ß", "<|", "fim", "_prefix", "|>\r\n\r\n", "s漢", " -", "٣٤٥", "٦", "#$%", " ", " '", "Re", "0", "!!!㋿", "\t", "Dž", ""]} +{"text": "\"́ع'T\rḍ̇é㍿👍🏽ḍ̇\r\n\r\n​🙂's#$%,½'T<|endoftext|>ßDž ३'M​å<|endoftext|>''T", "tokens": 66, "pieces": ["\"́", "ع", "'T", "\r", "ḋ", "̣e", "́㍿👍🏽", "ḋ", "̣<", "META", "_START", ">\r\n\r\n", "​🙂'", "s", "#$%,", "½", "'T", "<|", "endoftext", "|>", "ßDž", " ", "३", "'M", "​a", "̊<|", "endoftext", "|>''", "T"]} +{"text": "12345678ééİ Dž<|endoftext|>!!𐞁३㋿👍🏽漢‍<|fim_prefix|><|endoftext|>s", "tokens": 56, "pieces": ["123", "456", "78", "e", "́e", "́İ", " ", " Dž", "<|", "endoftext", "|>!!", "𐞁", "", "३", "㋿👍🏽", "漢", "‍<|", "fim", "_prefix", "|><|", "endoftext", "|>", "s"]} +{"text": "㍿🙂A<­> éع123456780\r\n𐞁12345678'VE'S漢<|fim_prefix|>éDžİ'D're…Zeé>٣٤٥٦ \t'VE>fiéß", "tokens": 69, "pieces": ["㍿🙂", "A", "<­>", " e", "́ع", "123", "456", "780", "\r\n", "𐞁", "123", "456", "78", "'VE", "'S", "漢", "<|", "fim", "_prefix", "|><", "EOT", ">e", "́Džİ", "'D", "'re", "…Zee", "́>", "٣٤٥", "٦", " ", "\t", "'VE", ">fie", "́ß"]} +{"text": "👍🏽a'VE9ſعDž ٣٤٥٦­Dž\r\n३fiß‍‍'De<|fim_prefix|>ſع漢<|fim_prefix|>\"­A", "tokens": 69, "pieces": ["👍🏽", "a", "'VE", "9", "ſ", "عDž", " ", " ", "٣٤٥", "٦", "­Dž", "\r\n", "३", "fiß", "‍‍'", "De", "<|", "fim", "_prefix", "|>", "ſع漢", "<|", "fim", "_prefix", "|>\"<", "META", "_START", ">­", "A"]} +{"text": "å,\r\n\r\nع½㋿\r\né३-'M<|fim_prefix|>'M­ 'Téßİ'st३'s\"tA\r\n\r\nd\u000b'sḍ̇ſ٣٤٥٦\r\n", "tokens": 59, "pieces": ["a", "̊,\r\n\r\n", "ع", "½", "㋿\r\n", "e", "́", "३", "-'", "M", "<|", "fim", "_prefix", "|>'", "M", "­", " '", "Téßİ", "'s", "t", "३", "'s", "\"tA", "\r\n\r\n", "d", "\u000b", "'s", "ḋ", "̣ſ", "٣٤٥", "٦", "\r\n"]} +{"text": "ß 'M#$%'s ३㍿ sdDž'll'sEOT
\r\n\r\n<|endoftext|> ‍\u000bEOT\u000b\n㍿Dž éEOT'S㋿ \né", "tokens": 56, "pieces": ["ß", " '", "M", "#$%'", "s", " ", "३", "㍿", " sdDž", "'ll", "'s", "EOT", "
\r\n\r\n", "<|", "endoftext", "|>", " ", "‍", "\u000bEOT", "\u000b\n", "㍿Dž", " <", "EOT", ">éEOT", "'S", "㋿", " \n", "é"]} +{"text": "٣٤٥٦åſ ‍és'ſ'lĺ12345678'T#$%٣٤٥٦!ع ́­", "tokens": 46, "pieces": ["٣٤٥", "٦", "a", "̊ſ", " ", "‍és", "'ſ", "'ll", "́", "123", "456", "78", "'T", "#$%<", "META", "_START", ">", "٣٤٥", "٦", "!ع", " ", " ́­"]} +{"text": ">EOT'reꟲß\nſ'S's'llm\u000b👍🏽'㋿'lĺ\t㋿Ⅳ>漢👍🏽 A", "tokens": 46, "pieces": [">EOT", "'re", "ꟲß", "\n", "ſ", "'S", "'s", "'ll", "m", "\u000b", "👍🏽'㋿'", "ll", "́", "\t", "㋿", "Ⅳ", ">漢", "👍🏽", " ", " A"]} +{"text": "'re٣٤٥٦EOTEOT<३😀🏽're-Ⅳ'D­ß\té३​漢٣٤٥٦-'S>ḍ̇ſ0sſ9-d0İ", "tokens": 61, "pieces": ["'re", "٣٤٥", "٦", "EOTEOT", "<", "३", "😀🏽'", "re", "-", "Ⅳ", "'D", "­ß", "\te", "́", "३", "​漢", "٣٤٥", "٦", "-'", "S", ">ḋ", "̣ſ", "0", "sſ", "9", "-d", "0", "İ"]} +{"text": "12345678\tſ", "tokens": 6, "pieces": ["123", "456", "78", "\tſ"]} +{"text": " EOT\"Ⅳ😀🏽'ſ", "tokens": 14, "pieces": [" ", " EOT", "\"", "Ⅳ", "😀🏽'", "ſ"]} +{"text": "½ßs\u000b\"😀🏽'llDžA .9İ'Re#$%12345678é", "tokens": 27, "pieces": ["½", "ßs", "\u000b", "\"😀🏽'", "llDžA", " ", ".", "9", "İ", "'Re", "#$%", "123", "456", "78", "é"]} +{"text": " 👍🏽\u000b३ !!​!!漢<|fim_prefix|>\n
fi a½\t12345678'Re' \u000b12345678", "tokens": 63, "pieces": ["", " ", " 👍🏽", "\u000b", "", "३", " ", " !!​!!", "漢", "<|", "fim", "_prefix", "|>\n", "
fi", " a", "½", "\t", "123", "456", "78", "'Re", "'", " ", "\u000b", "123", "456", "78", "<", "EOT", ">"]} +{"text": "'re😀🏽EOTß\ns'll\n#$%0<́é>ع<Ⅳ'ع", "tokens": 48, "pieces": ["'re", "😀🏽", "EOTß", "\n", "s", "'ll", "\n", "#$%", "0", "<<", "EOT", ">́", "e", "́>", "ع", "<", "Ⅳ", "'ع"]} +{"text": "(字as'ſ!ꟲ
9m👍🏽\"ⅣZḿ\"'Tİá12345678Dž'll😀🏽0\nDž!㋿ḍ̇Aa'll'M٣٤٥٦ꟲ 
12345678", "tokens": 66, "pieces": ["Dž", "123", "456", "78", "'re", "'ll", "㋿d", "9", "\"👍🏽<", "META", "_START", ">Dž", "'ll", "😀🏽", "0", "\n", "Dž", "!㋿", "ḋ", "̣Aa", "'ll", "'M", "٣٤٥", "٦", "ꟲ", " ", "
", "123", "456", "78"]} +{"text": "aa‍!Z㍿👍🏽!!>٣٤٥٦ ꟲDžA! ½ꟲéé.Dždafi𐞁Dž­ \n…A-👍🏽, 😀🏽𐞁👍🏽", "tokens": 84, "pieces": ["aa", "‍!", "Z", "㍿👍🏽!!>", "٣٤٥", "٦", " ꟲDžA", "!", " ", "½", "ꟲéé", ".Dždafi", "𐞁Dž", "­", " \n", "…A", "-👍🏽,", " ", " 😀🏽", "𐞁", "👍🏽"]} +{"text": "<|endoftext|>½३㋿#$%a9,.åDžDž'S ß漢12345678….'TDž", "tokens": 40, "pieces": ["<|", "endoftext", "|>", "½३", "㋿#$%", "a", "9", ",.", "a", "̊DžDž", "'S", " ", " ß漢", "123", "456", "78", "…", ".'", "TDž"]} +{"text": "dⅣfifi'MⅣ३‍'s\n­'ll (!!'s é\"-", "tokens": 29, "pieces": ["d", "Ⅳ", "fifi", "'M", "Ⅳ३", "‍'", "s", "\n", "­'", "ll", " ", "(!!'", "s", " ", " e", "́\"-"]} +{"text": "㋿🙂 ḍ̇#$%\r\n ٣٤٥٦½ꟲ🙂'ſß\u000b\rعa‍<|endoftext|> A \na­Z'll\r\n'T\r\n\r\nsⅣ'ſſEOT #$%​", "tokens": 77, "pieces": ["㋿🙂", " ḋ", "̣#$%\r\n", " ", " ", "٣٤٥", "٦½", "ꟲ", "🙂'", "ſß", "\u000b\r", "عa", "‍<|", "endoftext", "|>", " <", "EOT", ">A", " \n", "a", "­Z", "'ll", "\r\n", "'T", "\r\n\r\n", "s", "Ⅳ", "'ſ", "ſEOT", " <", "EOT", ">#$%​"]} +{"text": "漢ع Ⅳ🙂​'ll<|endoftext|> 'ḍ̇​İ…३\r\n٣٤٥٦", "tokens": 40, "pieces": ["漢ع", " ", "Ⅳ", "🙂​'", "ll", "<|", "endoftext", "|>", " ", " '", "ḋ", "̣​", "İ", "…", "३", "\r\n", "٣٤٥", "٦"]} +{"text": "s​🙂9ſ0\r's👍🏽'VE 9(A", "tokens": 23, "pieces": ["s", "​🙂", "9", "ſ", "0", "\r", "'s", "👍🏽'", "VE", " ", "9", "(A"]} +{"text": ".'S'VE字漢'lĺaDž!!éEOT\u000b< A㋿aådع'SEOT\u000b​­<'Re", "tokens": 40, "pieces": [".'", "S", "'VE", "字漢", "'ll", "́aDž", "!!", "e", "́EOT", "\u000b", "<", " A", "㋿aa", "̊dع", "'", "SEOT", "\u000b", "​­<'", "Re"]} +{"text": "fi!! ", "tokens": 4, "pieces": ["fi", "!!", " "]} +{"text": "\"‍'ſ'Ma9'VE>\r\n'llfi", "tokens": 20, "pieces": ["\"‍'", "ſ", "'M", "a", "9", "'", "VE", ">\r\n", "'ll", "fi"]} +{"text": "mß'll'M 9عd9e'retEOT\r\n\r\n३'ſ9
㋿", "tokens": 26, "pieces": ["mß", "'ll", "'M", " ", "9", "عd", "9", "e", "'re", "tEOT", "\r\n\r\n", "३", "'ſ", "9", "
", "㋿"]} +{"text": "🙂0!!!‍½'a½½>s字‍dfi-'\r'ſ'reⅣZß'Re…", "tokens": 36, "pieces": ["🙂", "0", "!!!‍", "½", "'a", "½½", ">s字", "‍dfi", "-'\r", "'ſ", "'", "re", "Ⅳ", "Zß", "'Re", "…"]} +{"text": "!!eⅣ!<|fim_prefix|>", "tokens": 11, "pieces": ["!!", "e", "Ⅳ", "!<|", "fim", "_prefix", "|>"]} +{"text": "a'ſ ", "tokens": 5, "pieces": ["a", "'ſ", " "]} +{"text": "<|endoftext|>½és're‍", "tokens": 28, "pieces": ["<|", "endoftext", "|>", "½", "és", "'re", "‍"]} +{"text": "'M's​\r­ 're'S'sdfi're'VE(​Ⅳ", "tokens": 19, "pieces": ["'M", "'s", "​\r", "­", " ", "'re", "'S", "'s", "dfi", "'re", "'VE", "(​", "Ⅳ"]} +{"text": "
३EOT éfi'll\r!…İ", "tokens": 16, "pieces": ["
", "३", "EOT", " ", " éfi", "'ll", "\r", "!", "…İ"]} +{"text": "'Ts,m", "tokens": 3, "pieces": ["'T", "s", ",m"]} +{"text": "𐞁Z12345678é\"\u000b'ſ ́٣٤٥٦ꟲZ", "tokens": 36, "pieces": ["𐞁Z", "123", "456", "78", "e", "́\"<", "EOT", ">", "\u000b", "'ſ", " ", " <", "EOT", ">́", "٣٤٥", "٦", "ꟲZ"]} +{"text": "'M're Ⅳ𐞁­\"", "tokens": 11, "pieces": ["'M", "'re", " ", "Ⅳ", "𐞁", "­\""]} +{"text": "
㋿\r\n \nés…", "tokens": 10, "pieces": ["
", "㋿\r\n", " \n", "és", "…"]} +{"text": "<(٣٤٥٦.m0字İ're٣٤٥٦m\r\n\r\n\r\n 00Z'M \n'ſZ­Z's㋿ \n🙂 \n\nDž<|fim_prefix|>mİ's<|endoftext|>", "tokens": 62, "pieces": ["<(", "٣٤٥", "٦", ".m", "0", "字İ", "'re", "٣٤٥", "٦", "m", "\r\n\r\n\r\n", " ", "00", "Z", "'M", " \n", "'ſ", "Z", "­Z", "'s", "㋿", " \n", "🙂", " \n\n", "Dž", "<|", "fim", "_prefix", "|>", "mİ", "'s", "<|", "endoftext", "|>"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'ll३\u000b<|endoftext|>🙂\"!…Z字'T㋿s 'T字!👍🏽…🙂…㍿,fi\"ḍ̇!ع\nå ​Ⅳ ½\"t", "tokens": 71, "pieces": ["'ll", "३", "\u000b", "<|", "endoftext", "|>🙂\"!", "…Z字", "'T", "㋿s", "", " ", "'T", "字", "!👍🏽", "…", "🙂", "…", "㍿,", "fi", "\"ḋ", "̣!", "ع", "\n", "a", "̊", " ​", "Ⅳ", " ", "½", "\"t"]} +{"text": " ꟲ \u000b12345678 \r
👍🏽𐞁", "tokens": 23, "pieces": [" ꟲ", " ", "\u000b", "123", "456", "78", " \r", "
", "👍🏽", "𐞁"]} +{"text": "½9🙂ß", "tokens": 5, "pieces": ["½9", "🙂ß"]} +{"text": "'Re9́<\n漢t\"½#$%\u000b\u000b字٣٤٥٦('M'M#$%'‍ \"'S字​<|fim_prefix|> 'VE\u000b👍🏽'TZ​d", "tokens": 54, "pieces": ["'Re", "9", "́<\n", "漢t", "\"", "½", "#$%", "\u000b", "\u000b字", "٣٤٥", "٦", "('", "M", "'M", "#$%'‍", " ", "\"'", "S字", "​<|", "fim", "_prefix", "|>", " '", "VE", "\u000b", "👍🏽'", "TZ", "​d"]} +{"text": "!\r\n​  0ſ'T'reḍ̇​ ㋿#$%9EOT​", "tokens": 26, "pieces": ["!\r\n", "​", " ", " ", "0", "ſ", "'T", "'re", "ḋ", "̣​", " ", " ㋿#$%", "9", "EOT", "​"]} +{"text": "'VE'S.\r\n\r\n٣٤٥٦<|fim_prefix|>\r'VEA\nAZ\u000b.", "tokens": 34, "pieces": ["'VE", "'S", ".\r\n\r\n", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>\r", "'VE", "A", "\n", "AZ", "\u000b", "."]} +{"text": "'VE!.12345678éA'fiea \nḍ̇\t'Re0'm#$%𐞁!!‍EOT<|endoftext|> A,३Dž", "tokens": 53, "pieces": ["'VE", "!.", "123", "456", "78", "e", "́A", "'<", "EOT", ">fiea", " \n", "ḋ", "̣", "\t", "'Re", "0", "'m", "#$%", "𐞁", "!!‍", "EOT", "<|", "endoftext", "|>", " A", ",", "३", "Dž"]} +{"text": ".9aA!!åé字#$%Ⅳé 
EOT​mm(…‍'字字EOT \n👍🏽\r\n'.'VE🙂", "tokens": 49, "pieces": [".", "9", "aA", "!!", "a", "̊é字", "#$%", "Ⅳ", "é", " ", "
EOT", "​mm", "(", "…", "‍'", "字字EOT", " \n", "👍🏽\r\n", "'.'", "VE", "🙂"]} +{"text": "\r\n👍🏽.sesEOT㋿\"'De!'VE9", "tokens": 21, "pieces": ["\r\n", "👍🏽.", "sesEOT", "㋿\"'", "De", "!'", "VE", "9"]} +{"text": "\t!!…'Re<|endoftext|> å12345678Ⅳ漢<'re!!mEOT\tEOTß <|endoftext|>EOTm-𐞁A ḍ̇漢ⅣZ!é'ſ😀🏽EOT", "tokens": 77, "pieces": ["\t", "!!", "…", "'Re", "<|", "endoftext", "|>", " a", "̊", "123", "456", "78Ⅳ", "漢", "<'", "re", "!!", "mEOT", "\tEOT", "ß", " ", "<|", "endoftext", "|>", "EOTm", "-𐞁A", " ḋ", "̣漢", "Ⅳ", "Z", "!e", "́'", "ſ", "😀🏽", "EOT"]} +{"text": "\u000b\rm­s'T𐞁Ⅳ're…३é>Ⅳ‍'s'ſ(\r\nfi<字tſEOT½éd́-'reDž\"<|endoftext|>m㋿'M", "tokens": 60, "pieces": ["\u000b\r", "m", "­s", "'T", "𐞁", "Ⅳ", "'re", "…", "३", "e", "́>", "Ⅳ", "‍'", "s", "'ſ", "(\r\n", "fi", "<字tſEOT", "½", "e", "́d", "́-'", "reDž", "\"<|", "endoftext", "|>", "m", "㋿'", "M"]} +{"text": "Ⅳ#$%'Ree\u000b½<|endoftext|>ß​½A½s 'sEOT𐞁\r\u000b½ḍ̇'٣٤٥٦", "tokens": 47, "pieces": ["Ⅳ", "#$%'", "Ree", "\u000b", "½", "<|", "endoftext", "|>", "ß", "​", "½", "A", "½", "s", " ", "'s", "EOT𐞁", "\r", "\u000b", "½", "ḋ", "̣'", "٣٤٥", "٦"]} +{"text": "ſé٣٤٥٦0‍t­Ⅳ½ꟲ'llⅣe\"½ ", "tokens": 30, "pieces": ["ſe", "́", "٣٤٥", "٦0", "‍t", "­", "Ⅳ½", "ꟲ", "'ll", "Ⅳ", "e", "\"", "½", " "]} +{"text": "
>'T½émå9e>😀🏽ém İéEOT \n#$%Ⅳ'Rea<|endoftext|>å12345678𐞁ZEOT 字t'T", "tokens": 57, "pieces": ["
", ">'", "T", "½", "e", "́ma", "̊", "9", "e", ">😀🏽", "e", "́m", " ", " İe", "́EOT", " \n", "#$%", "Ⅳ", "'Re", "a", "<|", "endoftext", "|>", "a", "̊", "123", "456", "78", "𐞁ZEOT", " 字t", "'T"]} +{"text": "'D.é㋿!\r12345678٣٤٥٦\"fi\ts'M३'Dع㋿\r\u000bⅣm\n-𐞁㍿", "tokens": 62, "pieces": ["字ß", "'Re", "\n", "𐞁a", "̊ḋ", "̣漢a", "̊", "\t", "'S", "!!'", "ſ", "9", "<|", "fim", "_prefix", "|>", "\ts", "'M", "३", "'D", "ع", "㋿<", "EOT", ">\r", "\u000b", "Ⅳ", "m", "\n", "-𐞁", "㍿"]} +{"text": "\n're'reDžfi
Dž  \r\n", "tokens": 13, "pieces": ["\n", "'re", "'re", "Džfi", "
Dž", "  \r\n"]} +{"text": "!EOT.'ſ \n'T're'VE'Dſ  字 ", "tokens": 24, "pieces": ["!EOT", ".'", "ſ", " \n", "'T", "'", "re", "'VE", "'D", "ſ", " ", " 字", " <", "EOT", ">"]} +{"text": "'sm!  t\"٣٤٥٦漢Zé'll 'll!٣٤٥٦ -00㋿́\u000b \n m ́'Ddع .Dž𐞁", "tokens": 56, "pieces": ["'s", "m", "!", "  ", " t", "\"", "٣٤٥", "٦", "漢Ze", "́'", "ll", " '", "ll", "!", "٣٤٥", "٦", " ", " -", "00", "㋿́", "\u000b \n", " m", " ", "́'", "Ddع", " .", "Dž𐞁"]} +{"text": "'S 'ſ \nt
​fi\nſ㍿9'VE12345678Z㍿<|endoftext|>Ⅳ👍🏽's<\n\r\n\r\n", "tokens": 53, "pieces": ["'S", " ", " '", "ſ", " \n", "t", "
", "​", "fi", "\n", "ſ", "㍿", "9", "'VE", "123", "456", "78", "Z", "㍿<|", "endoftext", "|>", "Ⅳ", "👍🏽'", "s", "<<", "META", "_START", ">\n\r\n\r\n"]} +{"text": "漢👍🏽.Džſ0!!\r\n'T9å åaåEOTİꟲ㍿‍㋿\t\tḍ̇", "tokens": 52, "pieces": ["漢", "👍🏽.", "Džſ", "0", "!!\r\n", "'T", "9", "a", "̊", " ", "a", "̊aa", "̊EOTİꟲ", "㍿‍㋿", "\t", "\tḋ", "̣"]} +{"text": "Z\r\n\r\n\r'ſ'MZ\r,12345678㍿㍿", "tokens": 19, "pieces": ["Z", "\r\n\r\n\r", "'ſ", "'M", "Z", "\r", ",", "123", "456", "78", "㍿㍿"]} +{"text": " \rd'ſa😀🏽're३<㋿\r\n\r\nꟲZ<🙂's <|fim_prefix|>mꟲ👍🏽字", "tokens": 48, "pieces": [" \r", "d", "'ſ", "a", "😀🏽'", "re", "३", "<㋿\r\n\r\n", "ꟲZ", "<🙂'", "s", " ", " <|", "fim", "_prefix", "|>", "mꟲ", "👍🏽", "字"]} +{"text": "ßİm(\r\n'lléA're\rs'DDž!fi३عdA'D-
​e㍿!!‍", "tokens": 36, "pieces": ["ßİm", "(\r\n", "'ll", "e", "́A", "'re", "\r", "s", "'D", "Dž", "!fi", "३", "عdA", "'D", "-", "
", "​e", "㍿!!‍"]} +{"text": "㍿ꟲ0dßt's漢 \n🙂😀🏽­ꟲ's👍🏽'Dtm<|endoftext|>é٣٤٥٦#$%!字éع'll👍🏽A", "tokens": 73, "pieces": ["㍿ꟲ", "0", "dßt", "'s", "漢", " \n", "🙂😀🏽­", "ꟲ", "'s", "👍🏽'", "Dtm", "<|", "endoftext", "|>", "e", "́", "٣٤٥", "٦", "#$%!<", "EOT", ">字éع", "'", "ll", "👍🏽", "A"]} +{"text": "\u000b<|endoftext|>A‍ß0#$%ⅣtZ>
're Dž-ع!e\tß'ſ é عß,< 
ḍ̇", "tokens": 54, "pieces": ["\u000b", "<|", "endoftext", "|>", "A", "‍ß", "0", "#$%", "Ⅳ", "tZ", ">", "
", "'re", " Dž", "-ع", "!e", "\tß", "'ſ", " e", "́", " ", " ع", "ß", ",<", " ", "
ḋ", "̣"]} +{"text": "éḍ̇\r\n\t0!'T\ts字㋿Dž\u000bع'reEOT'Re!.Z'ſ", "tokens": 30, "pieces": ["e", "́ḋ", "̣\r\n", "\t", "0", "!'", "T", "\ts字", "㋿Dž", "\u000bع", "'re", "EOT", "'Re", "!.", "Z", "'ſ"]} +{"text": "ḍ̇ée!㍿\"'D👍🏽\"fidA'M'VE'!!🙂½\r\n\r\n>'s's
Zḍ̇'ll9漢m٣٤٥٦Z#$%(­", "tokens": 66, "pieces": ["ḋ", "̣é", "e", "!㍿\"'", "D", "👍🏽\"", "fidA", "'M", "'VE", "'!!🙂", "½", "\r\n\r\n", ">'", "s", "'s", "
Zḋ", "̣'", "ll", "9", "漢m", "٣٤٥", "٦", "Z", "#$%(­"]} +{"text": "0'lléa😀🏽 \r\n\r\n字… e'Tt> e 'Té​12345678EOTZ㋿
.½0A㍿字𐞁́漢m'sa", "tokens": 54, "pieces": ["0", "'ll", "éa", "😀🏽", " \r\n\r\n", "字", "…", " e", "'T", "t", ">", " e", " ", " '", "Te", "́​", "123", "456", "78", "EOTZ", "㋿", "
", ".", "½0", "A", "㍿字𐞁", "́漢m", "'s", "a"]} +{"text": "<|fim_prefix|>漢 fi𐞁ع#$%.😀🏽#$%#$%12345678½ꟲ'Re!å'Reع😀🏽\n'sfi\ńA½\t.> 'M's🙂", "tokens": 65, "pieces": ["<|", "fim", "_prefix", "|>", "漢", " fi𐞁ع", "#$%.😀🏽#$%#$%", "123", "456", "78½", "ꟲ", "'Re", "!a", "̊'", "Reع", "😀🏽\n", "'s", "fi", "\n", "́A", "½", "\t", ".>", " ", "'M", "'s", "🙂"]} +{"text": "½'s
  12345678 \nſEOT\r㍿Zḍ̇!EOTſ'T\r\nß 𐞁", "tokens": 40, "pieces": ["½", "'s", "
 ", " ", "123", "456", "78", " \n", "ſEOT", "\r", "㍿Zḋ", "̣!", "EOTſ", "'T", "\r\n", "ß", " 𐞁"]} +{"text": "'ſ<|fim_prefix|>'Rea12345678'll­٣٤٥٦Z0३9'Res", "tokens": 36, "pieces": ["'ſ", "<|", "fim", "_prefix", "|>'", "Rea", "123", "456", "78", "'ll", "­", "٣٤٥", "٦", "Z", "0३9", "'Re", "s", ""]} +{"text": "३12345678e-Zİe Ⅳ'(\r\n\r\n‍\r\n\r\né \n漢㍿\nA㍿å<
 \nEOT­Ⅳ,\"9<|endoftext|>\t9\u000b\r", "tokens": 56, "pieces": ["३12", "345", "678", "e", "-Zİe", " ", " ", "Ⅳ", "'(\r\n\r\n", "‍\r\n\r\n", "é", " \n", "漢", "㍿\n", "A", "㍿a", "̊<", "
 \n", "EOT", "­", "Ⅳ", ",\"", "9", "<|", "endoftext", "|>", "\t", "9", "\u000b\r"]} +{"text": "é'VE''Ta 
<.­👍🏽\nİ\u000b#$%d٣٤٥٦'ſ'll '漢'Re‍½😀🏽'VE٣٤٥٦", "tokens": 58, "pieces": ["e", "́'", "VE", "''", "Ta", " ", "
", "<.­👍🏽\n", "İ", "\u000b", "#$%", "d", "٣٤٥", "٦", "'ſ", "'ll", " '", "漢", "'Re", "‍", "½", "😀🏽'", "VE", "٣٤٥", "٦"]} +{"text": "'Mfi \r\n12345678EOT  ٣٤٥٦éZİſ …,👍🏽!", "tokens": 36, "pieces": ["'M", "fi", " \r\n", "123", "456", "78", "EOT", "  ", " ", "٣٤٥", "٦", "éZİſ", " ", "…", ",👍🏽!"]} +{"text": "'🙂😀🏽a\"- 0 '𐞁'VEßé'D३👍🏽İ'D👍🏽ßA'VEt ㋿ \tİé", "tokens": 61, "pieces": ["'🙂😀🏽", "a", "\"-", " ", " ", "0", " ", " <", "EOT", ">'", "𐞁", "'VE", "ße", "́'", "D", "३", "👍🏽", "İ", "'D", "👍🏽", "ßA", "'VE", "t", " ", "㋿", " ", "\tİé"]} +{"text": "👍🏽aßa\" .å", "tokens": 14, "pieces": ["👍🏽", "aßa", "\"", " .", "a", "̊"]} +{"text": " \nꟲ½\te\u000b३ع'Re漢'T9Ⅳ'>३-<|fim_prefix|><­-'D'll'D'sİ<|endoftext|>12345678EOT
", "tokens": 52, "pieces": [" \n", "ꟲ", "½", "\te", "\u000b", "३", "ع", "'Re", "漢", "'T", "9Ⅳ", "'>", "३", "-<", "EOT", "><|", "fim", "_prefix", "|><­-'", "D", "'ll", "'D", "'s", "İ", "<|", "endoftext", "|>", "123", "456", "78", "EOT", "
"]} +{"text": "İåeEOT\n‍👍🏽‍😀🏽 \r\n\r\n12345678", "tokens": 27, "pieces": ["İa", "̊eEOT", "\n", "‍👍🏽‍😀🏽", " \r\n\r\n", "123", "456", "78"]} +{"text": "#$%'ſ\r\n 90😀🏽<|fim_prefix|>'ſ \n \n­㋿då", "tokens": 31, "pieces": ["#$%'", "ſ", "\r\n", " ", " ", "90", "😀🏽<|", "fim", "_prefix", "|>'", "ſ", " \n \n", "­㋿", "da", "̊"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'s'<|endoftext|>EOTعdİ\"å\ta", "tokens": 17, "pieces": ["'s", "'<|", "endoftext", "|>", "EOTعdİ", "\"a", "̊", "\ta"]} +{"text": "'A🙂 \n's\r12345678're", "tokens": 12, "pieces": ["'A", "🙂", " \n", "'s", "\r", "123", "456", "78", "'re"]} +{"text": "ꟲ<'M>", "tokens": 6, "pieces": ["ꟲ", "<'", "M", ">"]} +{"text": "👍🏽fißå<İ㍿#$%'Re٣٤٥٦0́٣٤٥٦Z<|endoftext|>‍<|endoftext|>ß👍🏽ßfi,", "tokens": 70, "pieces": ["👍🏽<", "META", "_START", ">fißa", "̊<", "İ", "㍿#$%'", "Re", "٣٤٥", "٦0", "́", "٣٤٥", "٦", "Z", "<|", "endoftext", "|>‍<|", "endoftext", "|>", "ß", "👍🏽", "ßfi", ","]} +{"text": "​<|fim_prefix|>\r'T㋿'VE字.​EOT,'T३", "tokens": 24, "pieces": ["​<|", "fim", "_prefix", "|>\r", "'T", "㋿'", "VE字", ".​", "EOT", ",'", "T", "३"]} +{"text": "\u000b", "tokens": 5, "pieces": ["", "\u000b"]} +{"text": "12345678㋿Džé½>\"!!e'Re…'M٣٤٥٦EOTİ😀🏽", "tokens": 37, "pieces": ["123", "456", "78", "㋿Džé", "½", ">\"!!", "e", "'Re", "…", "'M", "٣٤٥", "٦", "EOTİ", "😀🏽"]} +{"text": "\"å\u000bⅣ.e é३", "tokens": 11, "pieces": ["\"a", "̊", "\u000b", "Ⅳ", ".e", " ", " é", "३"]} +{"text": "👍🏽ZDž<|endoftext|> ", "tokens": 17, "pieces": ["👍🏽", "ZDž", "<|", "endoftext", "|>", " "]} +{"text": "​\r\nḍ̇́å'T Z漢0

'Re\r\nå٣٤٥٦\t½㋿ꟲ'T㋿ ㍿😀🏽 'VEß'refi>\nḍ̇m's12345678-ß\r\n", "tokens": 76, "pieces": ["​\r\n", "ḋ", "̣́", "a", "̊'", "T", " Z漢", "0", "
", "
", "'Re", "\r\n", "a", "̊", "٣٤٥", "٦", "\t", "½", "㋿ꟲ", "'T", "㋿", " ", "㍿😀🏽", " '", "VEß", "'re", "fi", ">\n", "ḋ", "̣m", "'s", "123", "456", "78", "-ß", "\r\n"]} +{"text": "½ 12345678('ſ😀🏽<|endoftext|>𐞁😀🏽\r\n\r\n(!'Sſ<|fim_prefix|>Z's👍🏽 👍🏽,", "tokens": 62, "pieces": ["½", " ", " ", "123", "456", "78", "('", "ſ", "😀🏽<|", "endoftext", "|>", "𐞁", "😀🏽\r\n\r\n", "(!'", "Sſ", "<|", "fim", "_prefix", "|>", "Z", "'s", "👍🏽", " ", "👍🏽,"]} +{"text": "'llⅣ३>½12345678ḍ̇́é>a>'D(\r\n\r\n,12345678'M", "tokens": 29, "pieces": ["'ll", "Ⅳ३", ">", "½12", "345", "678", "ḋ", "̣́", "e", "́>", "a", ">'", "D", "(\r\n\r\n", ",", "123", "456", "78", "'M"]} +{"text": "İ字㋿㋿Z'Sع
ḍ̇EOT.'D
㋿e 🙂12345678\u000b,ßå'Reé🙂, \ne'ſ३'s>‍\"ꟲ's'", "tokens": 65, "pieces": ["İ字", "㋿㋿", "Z", "'S", "ع", "
ḋ", "̣EOT", ".'", "D", "
", "㋿e", " 🙂", "123", "456", "78", "\u000b", ",<", "META", "_START", ">ßa", "̊'", "Ree", "́🙂,", " \n", "e", "'ſ", "३", "'s", ">‍\"", "ꟲ", "'s", "'"]} +{"text": "m👍🏽", "tokens": 7, "pieces": ["m", "👍🏽"]} +{"text": "字ⅣDž​ -0漢12345678", "tokens": 14, "pieces": ["字", "Ⅳ", "Dž", "​", " ", " -", "0", "漢", "123", "456", "78"]} +{"text": "'DEOT' \n 0字Z३", "tokens": 11, "pieces": ["'D", "EOT", "'", " \n", " ", "0", "字Z", "३"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "字Ⅳm​d‍½'re👍🏽,\nDžs㍿𐞁Z12345678\r\n", "tokens": 44, "pieces": ["字", "Ⅳ", "m", "​d", "‍", "½", "'re", "👍🏽,\n", "Džs", "㍿𐞁Z", "123", "456", "78", "\r\n"]} +{"text": "'ſ<|endoftext|>t'Tdaع👍🏽字\rZ'Mꟲ\r 漢\t0fi<|endoftext|>'MⅣe',३­At­​Ⅳé½><|fim_prefix|>", "tokens": 68, "pieces": ["'ſ", "<|", "endoftext", "|>", "t", "'T", "daع", "👍🏽", "字", "\r", "Z", "'M", "ꟲ", "\r", " ", " 漢", "\t", "0", "fi", "<|", "endoftext", "|>'", "M", "Ⅳ", "e", "',", "३", "­At", "­​", "Ⅳ", "e", "́", "½", "><|", "fim", "_prefix", "|>"]} +{"text": "İ'Ré Ⅳ ſ'reZsm\ń 'Tḍ̇'reEOT ​  \n𐞁\r'Sa'Té٣٤٥٦­́٣٤٥٦DžⅣA", "tokens": 63, "pieces": ["İ", "'Re", "́", " ", "Ⅳ", " ſ", "'re", "Zsm", "\n", "́", " '", "Tḋ", "̣'", "reEOT", " ", "​", "  \n", "𐞁", "\r", "'", "Sa", "'T", "é", "٣٤٥", "٦", "­́", "٣٤٥", "٦", "Dž", "Ⅳ", "A"]} +{"text": "…字\u000btå>'T字'VEḍ̇½'", "T字", "'VE", "ḋ", "̣", "½", "İ'sé\r\n\r\n­ßs\r\n\r\nfi<9're\n'MéⅣ<|fim_prefix|>#$%.\"'Te३ \r\n\r\n> e٣٤٥٦\u000bed", "tokens": 60, "pieces": ["Z", "́'", "s", "123", "456", "78", "Z", "İ", "'s", "e", "́\r\n\r\n", "­ßs", "\r\n\r\n", "fi", "<", "9", "'re", "\n", "'M", "e", "́", "Ⅳ", "<|", "fim", "_prefix", "|>#$%.\"'", "Te", "३", " \r\n\r\n", ">", " e", "٣٤٥", "٦", "\u000bed"]} +{"text": "'s'll ३Džad'll-  ㍿'T'ſ\"Ⅳ 字漢 ३ !!İa\r\n\r\n३0s", "tokens": 40, "pieces": ["'s", "'ll", " ", " ", "३", "Džad", "'ll", "-", " ", " ", "㍿'", "T", "'ſ", "\"", "Ⅳ", " ", " 字漢", " ", "३", " ", "!!", "İa", "\r\n\r\n", "३0", "s"]} +{"text": " ‍㍿'VEEOT-\t🙂́12345678\r\n\r\n'Re#$%😀🏽​🙂-\r\n\r\n😀🏽're#$%ع,'VEEOT㍿👍🏽é𐞁m'D 👍🏽٣٤٥٦\"(ſ", "tokens": 80, "pieces": [" ", "‍㍿'", "VEEOT", "-", "\t", "🙂́", "123", "456", "78", "\r\n\r\n", "'Re", "#$%😀🏽​🙂-\r\n\r\n", "😀🏽'", "re", "#$%", "ع", ",'", "VEEOT", "㍿👍🏽", "é𐞁m", "'D", " ", "👍🏽", "٣٤٥", "٦", "\"(", "ſ"]} +{"text": ".a\r\n
'Sꟲ㋿fiDž\n9fi 9🙂'Re👍🏽 d……३", "tokens": 38, "pieces": [".a", "\r\n", "
", "'S", "ꟲ", "㋿fiDž", "\n", "9", "fi", " ", "9", "🙂'", "Re", "👍🏽", " d", "…", "…", "३"]} +{"text": "d👍🏽fi!m漢漢!!‍EOT\r\n<|fim_prefix|>'Re­'ll㍿ 🙂\r\n\r\n,'Dḍ̇e­ ­étع́'llß<|endoftext|>\r -", "tokens": 71, "pieces": ["d", "👍🏽", "fi", "!m漢", "漢", "!!‍", "EOT", "\r\n", "<|", "fim", "_prefix", "|>'", "Re", "­'", "ll", "㍿", " ", "🙂\r\n\r\n", ",'", "Dḋ", "̣e", "­", " ", "­e", "́tع", "́'", "llß", "<|", "endoftext", "|>\r", " ", "-"]} +{"text": "<|endoftext|>'S!eⅣ\r\naéEOT'ſ\r\n\r\n­<|endoftext|>sß 👍🏽ⅣéEOT​", "tokens": 46, "pieces": ["<|", "endoftext", "|>'", "S", "!e", "Ⅳ", "\r\n", "aéEOT", "'ſ", "\r\n\r\n", "­<|", "endoftext", "|>", "sß", " ", "👍🏽", "Ⅳ", "éEOT", "​"]} +{"text": "\ŕ<|fim_prefix|>'D'M 
ḍ̇s \n😀🏽#$%Aéİ ㋿\n ", "tokens": 38, "pieces": ["\r", "́<|", "fim", "_prefix", "|>'", "D", "'M", " ", "
ḋ", "̣s", " \n", "😀🏽#$%", "Aéİ", " ", "㋿\n", " "]} +{"text": "ꟲ\"Zḍ̇ååfiꟲ", "tokens": 21, "pieces": ["ꟲ", "\"Zḋ", "̣a", "̊a", "̊fiꟲ"]} +{"text": "\r\n\r\nm'll're'VE\u000bEOTd\t🙂're٣٤٥٦ꟲ漢\t字#$%ḍ̇\"de", "tokens": 42, "pieces": ["\r\n\r\n", "m", "'ll", "'re", "'VE", "\u000b", "EOTd", "\t", "🙂'", "re", "٣٤٥", "٦", "ꟲ漢", "\t字", "#$%", "ḋ", "̣\"", "de"]} +{"text": ".漢\t'㋿12345678ßZEOT'VE <\r \n…㍿(fis㍿'VEDž's-ſ12345678٣٤٥٦", "tokens": 51, "pieces": [".漢", "\t", "'㋿", "123", "456", "78", "ßZEOT", "'VE", " ", "<\r", " \n", "…", "㍿(", "fis", "㍿'", "VEDž", "'s", "-ſ", "123", "456", "78٣", "٤٥٦"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'TEOT\"-ſ㋿ \n㋿", "tokens": 13, "pieces": ["'T", "EOT", "\"-", "ſ", "㋿", " \n", "㋿"]} +{"text": ".'llåm0'T­‍'VE're m½\ŕ\t", "tokens": 19, "pieces": [".'", "lla", "̊m", "0", "'T", "­‍'", "VE", "'re", " ", " m", "½", "\r", "́", "\t"]} +{"text": "'Re😀🏽३'Mt", "tokens": 14, "pieces": ["'Re", "😀🏽", "३", "'M", "t", ""]} +{"text": "ⅣZå\r!('llDž\r\n
ßſ're 𐞁'sꟲ", "tokens": 28, "pieces": ["Ⅳ", "Za", "̊\r", "!('", "llDž", "\r\n", "
ßſ", "'re", " 𐞁", "'s", "ꟲ"]} +{"text": "👍🏽d́.\r\n३ \n\" \t😀🏽e Z'ḍ̇EOT👍🏽>", "tokens": 38, "pieces": ["👍🏽", "d", "́.\r\n", "३", " \n", "\"", " ", "\t", "😀🏽", "e", " ", " Z", "'ḋ", "̣EOT", "👍🏽>"]} +{"text": "9m㋿12345678's'S'Ts'Re'Re\t<|fim_prefix|>.\"\r\n", "tokens": 23, "pieces": ["9", "m", "㋿", "123", "456", "78", "'s", "'S", "'T", "s", "'Re", "'Re", "\t", "<|", "fim", "_prefix", "|>.\"\r\n"]} +{"text": "EOT!字,'ſ字\r😀🏽éİ#$%٣٤٥٦३m🙂\".é'S🙂ḍ̇ſ \n \n\" 9EOTe", "tokens": 59, "pieces": ["EOT", "!字", ",'", "ſ字", "\r", "😀🏽", "e", "́<", "EOT", ">İ", "#$%", "٣٤٥", "٦३", "m", "🙂\"<", "META", "_START", ">.", "é", "'S", "🙂ḋ", "̣ſ", " \n \n", "\"", " ", "9", "EOTe"]} +{"text": " \na\r\n\r\n9'll<'VE'Re漢
­\r\nⅣ\u000b㋿字🙂(>A", "tokens": 31, "pieces": [" \n", "a", "\r\n\r\n", "9", "'", "ll", "<'", "VE", "'Re", "漢", "
", "­\r\n", "Ⅳ", "\u000b", "㋿字", "🙂(>", "A"]} +{"text": "㋿ſ Ⅳ ><|fim_prefix|>
Z🙂
d\t", "tokens": 28, "pieces": ["㋿", "ſ", " ", "Ⅳ", " ", "><|", "fim", "_prefix", "|>", "
Z", "🙂", "
d", "\t"]} +{"text": "
((½Ⅳ 𐞁‍#$%(é'T<#$%éEOTZ12345678's\n‍\u000b.ſe<|endoftext|>㋿​́>㋿<|fim_prefix|>
字Dž٣٤٥٦", "tokens": 76, "pieces": ["
", "((", "½Ⅳ", " 𐞁", "‍#$%(", "é", "'T", "<#$%", "éEOTZ", "123", "456", "78", "'s", "\n", "‍", "\u000b", ".ſe", "<|", "endoftext", "|>㋿​́>㋿<|", "fim", "_prefix", "|>", "
字", "Dž", "٣٤٥", "٦"]} +{"text": "­漢EOT'D३é \r\n\r\n‍'s㍿ \n'll\n12345678ßEOTtd İع‍e#$% İ'VEßA", "tokens": 42, "pieces": ["­漢EOT", "'D", "३", "é", " \r\n\r\n", "‍'", "s", "㍿", " \n", "'ll", "\n", "123", "456", "78", "ßEOTtd", " İع", "‍e", "#$%", " ", " İ", "'VE", "ßA"]} +{"text": "fi'S,\r\n\"​<😀🏽", "tokens": 12, "pieces": ["fi", "'S", ",\r\n", "\"​<😀🏽"]} +{"text": "ſ­ EOT'ſ
'Re👍🏽😀🏽Dž\u000bad🙂s", "tokens": 33, "pieces": ["ſ", "­", " EOT", "'ſ", "", "
", "'Re", "👍🏽😀🏽", "Dž", "\u000bad", "🙂s"]} +{"text": "<|endoftext|>İ\n…-👍🏽㍿m's ­Džt <|fim_prefix|>12345678'T<|fim_prefix|>dḍ̇a", "tokens": 54, "pieces": ["<|", "endoftext", "|>", "İ", "\n", "…", "-👍🏽㍿", "m", "'s", " ", "­Džt", " ", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'T", "<|", "fim", "_prefix", "|>", "dḋ", "̣a"]} +{"text": "('D\r'ſ", "tokens": 6, "pieces": ["('", "D", "\r", "'ſ"]} +{"text": "!!aée\"漢 漢 .㍿ ", "tokens": 19, "pieces": ["!!", "ae", "́<", "EOT", ">e", "\"漢", " ", " 漢", " .㍿", " "]} +{"text": "å½­12345678\t,ع,as'VE\r ٣٤٥٦ḍ̇३ \n㍿٣٤٥٦ ſå½𐞁ßḍ̇\n12345678ǻ", "tokens": 70, "pieces": ["a", "̊", "½", "­", "123", "456", "78", "\t", ",ع", ",as", "'VE", "\r", " ", " ", "٣٤٥", "٦", "ḋ", "̣", "३", " \n", "㍿", "٣٤٥", "٦", " ſa", "̊", "½", "𐞁ßḋ", "̣\n", "123", "456", "78", "a", "̊́"]} +{"text": "é'Tm
.́字
're​ \né'll", "tokens": 19, "pieces": ["e", "́'", "Tm", "
", ".́", "字", "
", "'re", "​", " \n", "e", "́'", "ll"]} +{"text": "㋿9EOT½,'re'S!'ſ<|endoftext|>🙂३İ're'T\r\n…ꟲ#$%'Sḍ̇😀🏽12345678'ſḍ̇\r's𐞁!!\n
漢#$%", "tokens": 70, "pieces": ["㋿", "9", "EOT", "½", ",'", "re", "'S", "!'", "ſ", "<|", "endoftext", "|>🙂", "३", "İ", "'re", "'T", "\r\n", "…ꟲ", "#$%'", "Sḋ", "̣😀🏽", "123", "456", "78", "'ſ", "ḋ", "̣\r", "'s", "𐞁", "!!\n", "
漢", "#$%"]} +{"text": ".fi ꟲt12345678\n㍿ḍ̇\r\n\r\ń\n", "㍿ḋ", "̣\r\n\r\n", "́<", "td", "\r\n", "́(", "t", "!"]} +{"text": "İ<|endoftext|>EOTA\"e>  \"㍿", "tokens": 20, "pieces": ["İ", "<|", "endoftext", "|>", "EOTA", "\"e", ">", " ", " \"㍿"]} +{"text": "ſ<|endoftext|>9A-'re>½>́ ", "tokens": 19, "pieces": ["ſ", "<|", "endoftext", "|>", "9", "A", "-'", "re", ">", "½", ">́", " "]} +{"text": "A \n(Ⅳ'VEſ(ZZå\r\n\r\nå\" s㍿ḍ̇㍿9e.!!́\n", "tokens": 39, "pieces": ["A", " \n", "(", "Ⅳ", "'VE", "ſ", "(ZZa", "̊\r\n\r\n", "a", "̊\"", " ", " s", "㍿ḋ", "̣㍿", "9", "e", ".!!́\n"]} +{"text": "ꟲDž !<\r\n're -", "tokens": 11, "pieces": ["ꟲDž", " !<\r\n", "'re", " ", " -"]} +{"text": "ß9<|fim_prefix|>'sⅣ'sḍ̇éſ\t#$%A'Mé­9\u000bés> ​漢's😀🏽'll ½!!\r\n'M'T‍\n", "tokens": 61, "pieces": ["ß", "9", "<|", "fim", "_prefix", "|>'", "s", "Ⅳ", "'s", "ḋ", "̣e", "́ſ", "\t", "#$%", "A", "'M", "é", "­", "9", "\u000be", "́s", ">", " ", "​<", "EOT", ">漢", "'s", "😀🏽'", "ll", " ", "½", "!!\r\n", "'M", "'T", "‍\n"]} +{"text": "́,字́\"d's'S\r\n ½9'D ḍ̇½🙂ꟲ'́漢!!🙂", "tokens": 32, "pieces": ["́,", "字", "́\"", "d", "'s", "'S", "\r\n", " ", "½9", "'D", " ", " ḋ", "̣", "½", "🙂ꟲ", "'́", "漢", "!!🙂"]} +{"text": "#$%ZꟲAam!9Ⅳß½́12345678!!", "tokens": 20, "pieces": ["#$%", "ZꟲAam", "!", "9Ⅳ", "ß", "½", "́", "123", "456", "78", "!!"]} +{"text": "
\r\n\r\n㍿<0\r\nß'9<|endoftext|>ſ", "tokens": 21, "pieces": ["
\r\n\r\n", "㍿<", "0", "\r\n", "ß", "'", "9", "<|", "endoftext", "|>", "ſ"]} +{"text": "
ع…-å'S9👍🏽EOT́\"'ḍ̇.'é­!́EOT<|fim_prefix|>字åå ", "tokens": 52, "pieces": ["
ع", "…", "-a", "̊'", "S", "9", "👍🏽", "EOT", "́\"'", "ḋ", "̣.'", "é", "­<", "META", "_START", ">!́", "EOT", "<|", "fim", "_prefix", "|>", "字a", "̊a", "̊", " "]} +{"text": "<|fim_prefix|>𐞁'ſ9ḍ̇
字ꟲع éEOT𐞁 #$%!!'ſ​३mEOT\r\nZ(​'s㍿#$%!!\"", "tokens": 64, "pieces": ["<|", "fim", "_prefix", "|>", "𐞁", "'ſ", "9", "ḋ", "̣", "
字ꟲع", " éEOT𐞁", " ", "#$%!!<", "EOT", ">'", "ſ", "​", "३", "mEOT", "\r\n", "Z", "(​'", "s", "㍿#$%!!\""]} +{"text": "Ⅳ½​m\r\nA'T…\u000bdDžaİ <|fim_prefix|>‍'llⅣ‍\"\u000b'S-", "tokens": 44, "pieces": ["Ⅳ½", "​m", "\r\n", "A", "'T", "…", "\u000bdDžaİ", " ", "<|", "fim", "_prefix", "|>‍'", "ll", "", "Ⅳ", "‍\"", "\u000b", "'S", "-"]} +{"text": "ſe… EOT", "tokens": 8, "pieces": ["ſe", "…", " EOT"]} +{"text": "३('VE漢\r\n\r\n 👍🏽-́a‍#$%<\u000b>😀🏽𐞁ſꟲ'Re
ꟲte'llİ", "tokens": 46, "pieces": ["३", "('", "VE漢", "\r\n\r\n", " ", " 👍🏽-́", "a", "‍#$%<", "\u000b", ">😀🏽", "𐞁ſꟲ", "'Re", "
ꟲte", "'ll", "İ"]} +{"text": "½‍\r­'reåⅣİé㍿,'s'S", "tokens": 19, "pieces": ["½", "‍\r", "­'", "rea", "̊", "Ⅳ", "İé", "㍿,'", "s", "'S"]} +{"text": "((
Dž'… 0d\r>aé're'll<|endoftext|>'D'llع<|fim_prefix|>'>عé're9\t \nAa", "tokens": 50, "pieces": ["((", "
Dž", "'", "…", " ", "0", "d", "\r", ">aé", "'re", "'ll", "<|", "endoftext", "|>'", "D", "'ll", "ع", "<|", "fim", "_prefix", "|>'>", "عe", "́'", "re", "", "9", "\t \n", "Aa"]} +{"text": "ع'll😀🏽're'Re㋿'M字 ", "tokens": 21, "pieces": ["ع", "'ll", "😀🏽'", "re", "'Re", "㋿'", "M", "字", " "]} +{"text": "fi㋿(d< 0\"Z漢ß‍😀🏽漢''ſé<|fim_prefix|>\r\n!9\"#$%…­EOT👍🏽\t👍🏽'TDž-", "tokens": 63, "pieces": ["fi", "㋿(", "d", "<", " ", "0", "\"Z漢ß", "‍😀🏽", "漢", "''", "ſé", "<|", "fim", "_prefix", "|>\r\n", "!", "9", "\"#$%", "…", "­EOT", "👍🏽", "\t", "👍🏽'", "TDž", "-"]} +{"text": "ſm🙂…Ⅳ𐞁½́st!'re", "tokens": 18, "pieces": ["ſm", "🙂", "…", "Ⅳ", "𐞁", "½", "́st", "!'", "re"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n\r\n😀🏽ع́字🙂ß३'Re\"𐞁\"'ll \n㋿12345678🙂\n's\n #$%!!ꟲ'ſ#$%!!ſ\r\n👍🏽's're'T", "tokens": 60, "pieces": ["\r\n\r\n", "😀🏽", "ع", "́字", "🙂ß", "३", "'Re", "\"𐞁", "\"'", "ll", " \n", "㋿", "123", "456", "78", "🙂\n", "'s", "\n", " ", " #$%!!", "ꟲ", "'ſ", "#$%!!", "ſ", "\r\n", "👍🏽'", "s", "'re", "'T"]} +{"text": "㍿㍿<👍🏽\"🙂 déEOTåḍ̇\r\n\r\n's\t<|fim_prefix|>><'S漢㋿å're \n½'D-", "tokens": 62, "pieces": ["㍿㍿<👍🏽\"🙂", " ", " déEOTa", "̊<", "EOT", ">ḋ", "̣\r\n\r\n", "'s", "\t", "<|", "fim", "_prefix", "|>><'", "S漢", "㋿", "a", "̊'", "re", " \n", "½", "'D", "-"]} +{"text": "'VE.́ …\rⅣİ'Re\"é٣٤٥٦'VE\u000b字٣٤٥٦ḍ̇👍🏽0!!A३é12345678mat\r\n'S👍🏽å漢‍'sDžA'S", "tokens": 83, "pieces": ["'VE", ".́", " …\r", "Ⅳ", "İ", "'Re", "\"é", "٣٤٥", "٦", "'VE", "\u000b字", "٣٤٥", "٦", "ḋ", "̣👍🏽", "0", "!!", "A", "३", "e", "́", "123", "456", "78", "mat", "\r\n", "'S", "👍🏽", "a", "̊漢", "‍'", "sDžA", "'S"]} +{"text": "!!字é<|fim_prefix|>'D\r(Ⅳ'Dꟲ,🙂…EOT३'D( \n \n­ \n'lléas
'Ree\r\n- ", "tokens": 45, "pieces": ["!!", "字é", "<|", "fim", "_prefix", "|>'", "D", "\r", "(", "Ⅳ", "'D", "ꟲ", ",🙂", "…EOT", "३", "'D", "(", " \n \n", "­", " \n", "'ll", "e", "́as", "
", "'Re", "e", "\r\n", "-", " "]} +{"text": "…", "tokens": 2, "pieces": ["…"]} +{"text": "'re#$%Dž'😀🏽㋿ع \nå0­
'lla字''s'S😀🏽'ſ \u000bZ-é \"'ll­👍🏽<|fim_prefix|>ꟲ!!\n", "tokens": 67, "pieces": ["'re", "#$%", "Dž", "'😀🏽㋿", "ع", " \n", "a", "̊", "0", "­", "
", "'ll", "a字", "''", "s", "'S", "😀🏽'", "ſ", " ", "\u000bZ", "-e", "́", " ", "\"'", "ll", "­👍🏽<|", "fim", "_prefix", "|>", "ꟲ", "!!\n"]} +{"text": "'D…\u000b🙂a🙂 EOT!!٣٤٥٦12345678\r'Té३\r\n\r\n9'VE", "tokens": 32, "pieces": ["'D", "…", "\u000b", "🙂a", "🙂", " EOT", "!!", "٣٤٥", "٦12", "345", "678", "\r", "'T", "é", "३", "\r\n\r\n", "9", "'VE"]} +{"text": "'Så\u000be𐞁'ſ㍿'s㋿İ'S㍿d\t㋿'S'ſ're's😀🏽३#$%İⅣfi fifidꟲ'", "ſ", "㍿'", "s", "㋿İ", "'S", "㍿d", "\t", "㋿'", "S", "'ſ", "'re", "'s", "😀🏽", "३", "#$%", "İ", "Ⅳ", "fi", " fifidꟲ", "éå㍿'s\té\r٣٤٥٦å­å'Re\r\n🙂٣٤٥٦12345678­Ⅳ", "tokens": 53, "pieces": ["t", "-a", "'D", ">éa", "̊㍿'", "s", "\te", "́\r", "٣٤٥", "٦", "a", "̊­", "a", "̊'", "Re", "\r\n", "🙂", "٣٤٥", "٦12", "345", "678", "­", "Ⅳ"]} +{"text": "½'Re'ſ9>d \n<'llé0dß12345678'D­ſ#$%'s", "tokens": 27, "pieces": ["½", "'", "Re", "'ſ", "9", ">d", " \n", "<'", "llé", "0", "dß", "123", "456", "78", "'D", "­ſ", "#$%'", "s"]} +{"text": "'VE", "tokens": 6, "pieces": ["'VE", ""]} +{"text": "'re's३'Re12345678eA'VE\nA,'s#$%\n𐞁\"Dž", "tokens": 31, "pieces": ["'re", "'s", "३", "'Re", "123", "456", "78", "eA", "'VE", "\n", "A", ",'", "s", "#$%\n", "𐞁", "\"Dž"]} +{"text": "\n'VE🙂½<|endoftext|><\n'M!'Mꟲ­𐞁ꟲ're\u000bfis! 9'Da\r\na👍🏽😀🏽're( 12345678'", "tokens": 64, "pieces": ["\n", "'VE", "🙂", "½", "<|", "endoftext", "|><\n", "'M", "!'", "Mꟲ", "­<", "EOT", ">𐞁ꟲ", "'re", "\u000bfis", "!", " ", "9", "'D", "a", "\r\n", "a", "👍🏽😀🏽'", "re", "(", " ", " ", "123", "456", "78", "'"]} +{"text": "fi \t👍🏽12345678're​ꟲ👍🏽EOT٣٤٥٦'ll٣٤٥٦ å('.", "tokens": 47, "pieces": ["fi", " ", "\t", "👍🏽", "123", "456", "78", "'re", "​ꟲ", "👍🏽", "EOT", "٣٤٥", "٦", "'ll", "٣٤٥", "٦", " a", "̊('."]} +{"text": "\"\rd㋿\t \t 'S\tså'VE🙂", "tokens": 17, "pieces": ["\"\r", "d", "㋿", "\t \t", " ", "'S", "\tsa", "̊'", "VE", "🙂"]} +{"text": "m३\r\n 𐞁👍🏽\r's,s😀🏽́­ \r\n\r\n🙂𐞁", "tokens": 32, "pieces": ["m", "३", "\r\n", " 𐞁", "👍🏽\r", "'s", ",s", "😀🏽́­", " \r\n\r\n", "🙂𐞁"]} +{"text": "0ſ\r!EOT​é'Ḿ\r'M'S​et'reſ\u000b", "tokens": 24, "pieces": ["", "0", "ſ", "\r", "!EOT", "​é", "'M", "́\r", "'M", "'S", "​et", "'re", "ſ", "\u000b"]} +{"text": "#$%,d
EOT漢\n​Ⅳꟲe'M", "tokens": 18, "pieces": ["#$%,", "d", "
EOT漢", "\n", "​", "Ⅳ", "ꟲe", "'M"]} +{"text": "><|endoftext|>!ḍ̇'Ré.漢>'M", "tokens": 24, "pieces": ["><|", "endoftext", "|>!", "ḋ", "̣'", "Re", "́.", "漢", ">'", "M", ""]} +{"text": "Zéſ--Dž's\r\n\r\nꟲ́'VE字漢're!!s\r\n'-<", "EOT", ">'", "s", "\r\n\r\n", "ꟲ", "́'", "VE字漢", "'re", "!!", "s", "\r\n", "'-<", "EOT", "🙂", " ", " ", "#$%"]} +{"text": " 👍🏽
!!\t!!'sꟲ́<ꟲ🙂é'VEA\"Z𐞁'-İ<|endoftext|>", "
", "!!", "\t", "!!'", "sꟲ", "́<", "ꟲ", "🙂é", "'VE", "A", "\"Z𐞁", "'-", "İ", "<|", "endoftext", "|><", "s", "9", "​"]} +{"text": ",tع'D \nḍ̇㍿'Re\r\n 𐞁'VEd👍🏽​'ll'ReA𐞁 \n🙂", "tokens": 42, "pieces": [",tع", "'D", " \n", "ḋ", "̣㍿'", "Re", "\r\n", " 𐞁", "'VE", "d", "👍🏽​'", "ll", "'Re", "A𐞁", " \n", "🙂"]} +{"text": "́'VEDž\nⅣé𐞁‍'M", "tokens": 17, "pieces": ["́'", "VEDž", "\n", "Ⅳ", "é𐞁", "‍'", "M"]} +{"text": "é \nEOT
<'‍'Ttå!", "tokens": 16, "pieces": ["e", "́", " \n", "EOT", "
", "<'‍'", "Tta", "̊!"]} +{"text": "!!字\r\n\r\n'ſ'D Z㋿'T( Ⅳ0eé㋿a're \n", "tokens": 28, "pieces": ["!!", "字", "\r\n\r\n", "'ſ", "'D", " Z", "㋿'", "T", "(", " ", " ", "Ⅳ0", "eé", "㋿a", "'re", " \n"]} +{"text": "aſ'D👍🏽­㍿aḍ̇\u000b​! \n'S<|endoftext|>s're9fi>", "tokens": 38, "pieces": ["aſ", "'D", "👍🏽­㍿", "aḋ", "̣", "\u000b", "​!", " \n", "'S", "<|", "endoftext", "|>", "s", "'re", "9", "fi", ">"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "Ⅳ३A٣٤٥٦0éſ\u000b\r𐞁 Džé½ḍ̇sd,'Sa\r\n\r\n३é", "tokens": 42, "pieces": ["Ⅳ३", "A", "٣٤٥", "٦0", "éſ", "\u000b\r", "𐞁", " Džé", "½", "ḋ", "̣sd", ",'", "Sa", "\r\n\r\n", "३", "e", "́"]} +{"text": "t d​ß'ſfi-!😀🏽a३'sḍ̇12345678Z'll­'Re#$%
 ,s'Så'Té'ſ(\r\n\r\ntſm'\r\n", "tokens": 58, "pieces": ["t", " ", " d", "​ß", "'ſ", "fi", "-!😀🏽", "a", "३", "'s", "ḋ", "̣", "123", "456", "78", "Z", "'ll", "­'", "Re", "#$%", "
 ", " ,", "s", "'S", "a", "̊'", "Te", "́'", "ſ", "(\r\n\r\n", "tſm", "'\r\n"]} +{"text": "​漢<'DtEOT­ḍ̇<𐞁<字́ 'M're0", "tokens": 28, "pieces": ["​", "漢", "<'", "DtEOT", "­ḋ", "̣<", "𐞁", "<字", "́", " '", "M", "'re", "0"]} +{"text": ">dé,!!-\rEOT- \nt Ⅳ \n\tḍ̇Z字", "tokens": 31, "pieces": [">", "de", "́,!!-\r", "EOT", "-", " \n", "t", " ", " ", "Ⅳ", " \n", "", "\tḋ", "̣Z字"]} +{"text": "\r\n
'Sſ٣٤٥٦!!…DžZ३'Tع#$%e", "tokens": 31, "pieces": ["\r\n", "
", "'S", "ſ", "٣٤٥", "٦", "!!", "…", "DžZ", "३", "'T", "ع", "#$%", "e"]} +{"text": "fi🙂'👍🏽'M \n'Ta🙂t \u000b >Afi", "🙂'👍🏽'", "M", " \n", "'T", "a", "🙂t", " \u000b", " ", ">A", "#$%å'ſ𐞁<|fim_prefix|>ß!!🙂….ꟲ\"", "tokens": 54, "pieces": [".", "123", "456", "78", "'𐞁", "३", "ع", "'M", "(,", "½", "́\n", "३", ">#$%", "a", "̊'", "ſ𐞁", "<|", "fim", "_prefix", "|>", "ß", "!!🙂", "…", ".ꟲ", "\""]} +{"text": "<|fim_prefix|>> \n𐞁e", "tokens": 13, "pieces": ["<|", "fim", "_prefix", "|>>", " \n", "𐞁e"]} +{"text": "'Dt-ḍ̇३'ſ'dZ.m\r\n‍\r\n\r\n,Dž\r''M'VE", "tokens": 28, "pieces": ["'D", "t", "-ḋ", "̣", "३", "'ſ", "'d", "Z", ".m", "\r\n", "‍\r\n\r\n", ",Dž", "\r", "''", "M", "'VE"]} +{"text": "'VE'TA<\r\n'S'VEfi漢-#$%\u000b½sééd.½ 'S.'ll>🙂A's('T​漢'T", "tokens": 42, "pieces": ["'VE", "'T", "A", "<\r\n", "'S", "'VE", "fi漢", "-#$%", "\u000b", "½", "se", "́e", "́d", ".", "½", " ", "'S", ".'", "ll", ">🙂", "A", "'s", "('", "T", "​漢", "'T"]} +{"text": "'S\"𐞁\r\n\r\n字‍,
\"🙂‍.aa!9'Dta'M​\t\nZ\t …m­İ𐞁,Za㋿'re", "tokens": 53, "pieces": ["'S", "\"", "𐞁", "\r\n\r\n", "字", "‍,", "
", "\"🙂‍.", "aa", "!", "9", "'D", "ta", "'M", "​", "\t\n", "Z", "\t ", "…m", "­İ", "𐞁", ",Za", "㋿'", "re"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㋿å㍿12345678(-<|endoftext|>>!!\r\n'ſ'VEſ
'sZ<|endoftext|>😀🏽9ßİ‍…EOTⅣع's(ſ <|fim_prefix|>", "tokens": 77, "pieces": ["㋿a", "̊㍿", "123", "456", "78", "(-<|", "endoftext", "|>><", "EOT", ">!!\r\n", "'ſ", "'", "VEſ", "
", "'s", "Z", "<|", "endoftext", "|>😀🏽", "9", "ßİ", "‍", "…EOT", "Ⅳ", "ع", "'s", "(ſ", " <|", "fim", "_prefix", "|>"]} +{"text": " '‍\r\n9-'ree'rea\r\n\r\n३𐞁(́Dž…'Ree😀🏽\"🙂'D0,fi", "tokens": 43, "pieces": [" ", " '‍\r\n", "9", "-'", "ree", "'re", "a", "\r\n\r\n", "३", "𐞁", "(́", "Dž", "", "…", "'Re", "e", "😀🏽\"🙂'", "D", "0", ",fi"]} +{"text": "🙂0漢(Dž'VE<|fim_prefix|> \"ḍ̇­Z👍🏽٣٤٥٦漢(å字EOT\"12345678's\n'D", "tokens": 57, "pieces": ["🙂", "0", "漢", "(Dž", "'VE", "<|", "fim", "_prefix", "|>", " ", "\"ḋ", "̣­", "Z", "👍🏽", "٣٤٥", "٦", "漢", "(a", "̊字", "EOT", "\"", "123", "456", "78", "'s", "\n", "'D"]} +{"text": "<|fim_prefix|><|endoftext|>'Re…'T#$% d<|endoftext|><|fim_prefix|>12345678", "tokens": 36, "pieces": ["<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "Re", "…", "'T", "#$%", " d", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "123", "456", "78"]} +{"text": " ㋿𐞁İ\nmḍ̇'re<|fim_prefix|>'Sḍ̇mfi9<٣٤٥٦'Dß're \n ḍ̇㍿\r‍'👍🏽<|endoftext|>🙂ſ'S", "tokens": 78, "pieces": [" ", "㋿𐞁İ", "\n", "mḋ", "̣'", "re", "<|", "fim", "_prefix", "|>'", "Sḋ", "̣mfi", "9", "<", "٣٤٥", "٦", "'D", "ß", "'re", " \n", " ḋ", "̣㍿\r", "‍'👍🏽<|", "endoftext", "|>🙂", "ſ", "'S"]} +{"text": "'D're½½As9>'M,😀🏽.'Re ́🙂'Ms's३­\r\n­9'", "re", "½½", "As", "9", ">'", "M", ",😀🏽.'", "Re", " ", "́🙂'", "Ms", "'s", "३", "­\r\n", "­", "9", "é́t𐞁Ⅳ#$%'VE'sfi,ſḍ̇'M-,EOT \nA's", "tokens": 55, "pieces": ["İ", "‍​", "
", "㋿", " 漢", "‍\r", "३", "!!́<", "EOT", ">é", "́t𐞁", "Ⅳ", "#$%'", "VE", "'s", "fi", ",ſḋ", "̣'", "M", "-,", "EOT", " \n", "A", "'s"]} +{"text": "'re…0!!٣٤٥٦ßdZ‍'sZ", "tokens": 21, "pieces": ["'re", "…", "0", "!!", "٣٤٥", "٦", "ßdZ", "‍'", "sZ"]} +{"text": "fiA \nEOT\t½'ll,s A9", "tokens": 14, "pieces": ["fiA", " \n", "EOT", "\t", "½", "'ll", ",s", " A", "9"]} +{"text": "𐞁!!漢٣٤٥٦ ㋿'S\r\n\r\nſ", "tokens": 24, "pieces": ["𐞁", "!!", "漢", "٣٤٥", "٦", " ", "㋿'", "S", "\r\n\r\n", "ſ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "!\r\n\r\n.\u000b­ſ.åe", "tokens": 11, "pieces": ["!\r\n\r\n", ".", "\u000b", "­ſ", ".a", "̊e"]} +{"text": " \t9'Mſ9!!\"\"9'llꟲ\u000bİ㋿!!fiZع'ſⅣⅣſ", "tokens": 33, "pieces": [" ", "\t", "9", "'M", "ſ", "9", "!!\"\"", "9", "'ll", "ꟲ", "\u000bİ", "㋿!!", "fiZع", "'ſ", "ⅣⅣ", "ſ"]} +{"text": "EOT३ 12345678'T'Reḍ̇ \n'Dm'sd😀🏽\t㍿,t 'Tع\rå", "tokens": 47, "pieces": ["EOT", "३", "", " ", "123", "456", "78", "'", "T", "'Re", "ḋ", "̣", " \n", "'D", "m", "'s", "d", "😀🏽", "\t", "㍿,", "t", "", " ", "'T", "ع", "\r", "a", "̊"]} +{"text": "\u000bAé
e\r'T <|endoftext|>t!!e'D٣٤٥٦𐞁👍🏽Ⅳ\rſ'lltDž('", "tokens": 52, "pieces": ["\u000bAe", "́", "
e", "\r", "'T", " ", " <|", "endoftext", "|>", "t", "!!<", "EOT", ">e", "'D", "٣٤٥", "٦", "𐞁", "👍🏽", "Ⅳ", "\r", "ſ", "'ll", "tDž", "('"]} +{"text": "\nḍ̇㋿㍿😀🏽漢a", "tokens": 20, "pieces": ["\n", "ḋ", "̣㋿㍿😀🏽", "漢a"]} +{"text": "s ㍿'reAt!!𐞁½d㋿EOT…d!!㍿㍿e
éİ.​‍ßZ
३0Ⅳ, é'S'VE漢३0's", "tokens": 62, "pieces": ["s", " ", "㍿'", "reAt", "!!", "𐞁", "½", "d", "㋿EOT", "…d", "!!㍿㍿", "e", "
éİ", ".​‍", "ßZ", "
", "३0Ⅳ", ",", " é", "'S", "'VE", "漢", "३0", "'s"]} +{"text": " <㍿ع😀🏽Z𐞁ſ
mm​dß'Re漢\r\n\r\n \n\"…\"", "tokens": 32, "pieces": [" <㍿", "ع", "😀🏽", "Z𐞁ſ", "
mm", "​dß", "'Re", "漢", "\r\n\r\n \n", "\"", "…", "\""]} +{"text": " \ném'S½'s
字­'ſ\r\n\r\n‍ſ٣٤٥٦\r\n\r\n
'VE​éDž㍿\u000bååع\r\n'se\rعt''ll", "tokens": 59, "pieces": [" \n", "e", "́m", "'S", "½", "'s", "
字", "­<", "EOT", ">'", "ſ", "\r\n\r\n", "‍ſ", "٣٤٥", "٦", "\r\n\r\n", "
", "'VE", "​e", "́Dž", "㍿", "\u000ba", "̊a", "̊ع", "\r\n", "'s", "e", "\r", "عt", "''", "ll"]} +{"text": "字mſ🙂'", "tokens": 7, "pieces": ["字mſ", "🙂'"]} +{"text": "🙂 ", "tokens": 3, "pieces": ["🙂", " "]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'s<|endoftext|>㋿\"👍🏽\rEOT́!e", "tokens": 24, "pieces": ["'s", "<|", "endoftext", "|>㋿\"👍🏽\r", "EOT", "́!", "e"]} +{"text": ".㋿at字<'Sm'M9字e
\u000bع!👍🏽ß㍿..-d ḍ̇🙂.'S\r'ſ\r\n\r\nA'ſ12345678'VE
é'ſ\n", "tokens": 63, "pieces": [".㋿", "at字", "<'", "Sm", "'M", "9", "字e", "
", "\u000bع", "!👍🏽", "ß", "㍿..-", "d", " ", " ḋ", "̣🙂.'", "S", "\r", "'ſ", "\r\n\r\n", "A", "'ſ", "123", "456", "78", "'VE", "
e", "́'", "ſ", "\n"]} +{"text": "ḍ̇e'ſ👍🏽<|fim_prefix|><|endoftext|>s!­ é😀🏽ſ½…'Mꟲ\u000b -㋿ 'S\r#$%fi字-<\u000b\r(-\reé㋿", "tokens": 78, "pieces": ["ḋ", "̣e", "'ſ", "👍🏽<|", "fim", "_prefix", "|><|", "endoftext", "|>", "s", "!­", " ", "e", "́😀🏽", "ſ", "½", "…", "'M", "ꟲ", "\u000b", " -㋿", " ", " '", "S", "\r", "#$%", "fi字", "-<", "\u000b\r", "(-\r", "eé", "㋿"]} +{"text": "fid <|endoftext|>!­ 'D३Z🙂<|endoftext|>#$%字'ſé", "tokens": 37, "pieces": ["fid", " ", " <|", "endoftext", "|><", "META", "_START", ">!­", " ", "'D", "३", "Z", "🙂<|", "endoftext", "|>#$%", "字", "'ſ", "e", "́"]} +{"text": "'ll !!​<|fim_prefix|>ع‍'ll‍‍''S.EOT#$%'
\u000b👍🏽Dž'\r\n\r\nİ́", "tokens": 46, "pieces": ["'ll", " ", " !!​<", "EOT", "><|", "fim", "_prefix", "|>", "ع", "‍'", "ll", "‍‍''", "S", ".EOT", "#$%'", "
", "\u000b", "👍🏽", "Dž", "'\r\n\r\n", "İ", "́"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "<|fim_prefix|>a­fi\"'S'ſ३<0'Sé>'VE㋿a.'Re'T\r\nſDža🙂0\"ſ", "tokens": 44, "pieces": ["<|", "fim", "_prefix", "|>", "a", "­fi", "\"'", "S", "'ſ", "३", "<", "0", "'S", "e", "́>'", "VE", "㋿a", ".'", "Re", "'T", "\r\n", "ſDža", "🙂", "0", "\"ſ"]} +{"text": "㍿12345678'S!!#$%fi'ß\t㋿åe.­😀🏽'VE\r\n\r\nDž‍漢٣٤٥٦İ㍿ >9'Re\u000b9!\u000b0 \nß‍'Re'ſſ", "tokens": 70, "pieces": ["㍿", "123", "456", "78", "'S", "!!#$%", "fi", "'ß", "\t", "㋿a", "̊e", ".­😀🏽'", "VE", "\r\n\r\n", "Dž", "‍漢", "٣٤٥", "٦", "İ", "㍿", " ", " >", "9", "'Re", "\u000b", "9", "!", "\u000b", "0", " \n", "ß", "‍'", "Re", "'ſ", "ſ"]} +{"text": "tm'll\t12345678<,३A🙂12345678Dž'll𐞁İå𐞁\u000b\u000b
", "tokens": 36, "pieces": ["tm", "'ll", "\t", "123", "456", "78", "<,", "३", "A", "🙂", "123", "456", "78", "Dž", "'ll", "𐞁İa", "̊𐞁", "\u000b\u000b
"]} +{"text": "'Reßå>12345678å­m😀🏽", "tokens": 19, "pieces": ["'Re", "ßa", "̊>", "123", "456", "78", "a", "̊­", "m", "😀🏽"]} +{"text": "​ ſå㍿ꟲ<|fim_prefix|>-EOT<|fim_prefix|>'T,(㍿ 
ßعé'Mḍ̇.", "tokens": 50, "pieces": ["​", " ſa", "̊㍿", "ꟲ", "<|", "fim", "_prefix", "|>-", "EOT", "<|", "fim", "_prefix", "|>'", "T", ",(㍿", " ", "", "
ßعé", "'M", "ḋ", "̣."]} +{"text": " ㋿㋿\u000bꟲ", "tokens": 11, "pieces": [" ", "㋿㋿", "\u000bꟲ"]} +{"text": " ꟲ३'ſ'ſ\n'll!!'M'S1234567812345678🙂\r\nZ\r", "tokens": 29, "pieces": [" ꟲ", "३", "'ſ", "'ſ", "\n", "'ll", "!!'", "M", "'S", "123", "456", "781", "234", "567", "8", "🙂\r\n", "Z", "\r"]} +{"text": "fi​'D\"Džmſ 
>३ꟲ\r\n'ſa", "tokens": 25, "pieces": ["fi", "​'", "D", "\"Džmſ", " ", "
", ">", "३", "ꟲ", "\r\n", "'ſ", "a"]} +{"text": "é>…‍́漢́ꟲ09…12345678d- <|endoftext|> \n\r'' EOTå- ,'sEOT#$% \n<|endoftext|>9!!…😀🏽", "tokens": 71, "pieces": ["é", ">", "…", "‍́", "漢", "́ꟲ", "09", "…", "123", "456", "78", "d", "-", " ", "<|", "endoftext", "|>", " \n\r", "''", " ", " EOTa", "̊-<", "META", "_START", ">", " ", ",'", "sEOT", "#$%", " \n", "<|", "endoftext", "|>", "9", "!!", "…", "😀🏽"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "å'lls<|fim_prefix|>'M'-12345678 \n'D é", "tokens": 23, "pieces": ["a", "̊'", "lls", "<|", "fim", "_prefix", "|>'", "M", "'-", "123", "456", "78", " \n", "'D", " ", " e", "́"]} +{"text": "३漢t!!!!es\u000b\u000b\r\n​'ſ<|fim_prefix|>'TZ\"å!!!!", "es", "\u000b\u000b\r\n", "​'", "ſ", "<|", "fim", "_prefix", "|>'", "TZ", "\"a", "̊<", "eꟲ", " ", "9", "\r", "ſ", "(字", "‍​", "字", " '", "M", " \n", "👍🏽", "0"]} +{"text": "🙂 99 ㍿'D🙂-", "tokens": 16, "pieces": ["🙂", " ", "", "99", " ㍿'", "D", "🙂-"]} +{"text": "'T \n'T‍mZ'llmA!\r\n\r\n(ds\t", "tokens": 15, "pieces": ["'T", " \n", "'T", "‍mZ", "'ll", "mA", "!\r\n\r\n", "(ds", "\t"]} +{"text": "­
Dž\"ßİ#$%'M'sꟲ(\r\n\r\n<ꟲ'S'M😀🏽d३….​½ꟲ٣٤٥٦", "tokens": 50, "pieces": ["­", "
Dž", "\"", "ßİ", "#$%'", "M", "'s", "ꟲ", "(\r\n\r\n", "<ꟲ", "'S", "'M", "😀🏽", "d", "३", "…", ".​", "½", "ꟲ", "٣٤٥", "٦"]} +{"text": "'Da‍㍿,'Reß12345678'DeZ-'re'D's,-12345678'VE'll३(", "tokens": 30, "pieces": ["'D", "a", "‍㍿,'", "Reß", "123", "456", "78", "'D", "eZ", "-'", "re", "'D", "'s", ",-", "123", "456", "78", "'VE", "'ll", "३", "("]} +{"text": "​ \n'VEſ", "tokens": 6, "pieces": ["​", " \n", "'VE", "ſ"]} +{"text": "-ß'Re३å .", "tokens": 9, "pieces": ["-ß", "'Re", "३", "a", "̊", " ."]} +{"text": "åḍ̇!!<|fim_prefix|>\r's'S‍'sſ٣٤٥٦'D12345678.#$%-!fi fi< 'é‍😀🏽३å> ", "tokens": 64, "pieces": ["a", "̊ḋ", "̣!!<|", "fim", "_prefix", "|>\r", "'s", "'S", "‍'", "sſ", "٣٤٥", "٦", "'D", "123", "456", "78", ".#$%-!", "fi", " fi", "<", " ", " '", "é", "‍😀🏽", "३", "a", "̊>", " "]} +{"text": "ß😀🏽ſ'llt'T''T'M0́!a'S,㍿٣٤٥٦>
…s>́'re\n's", "tokens": 43, "pieces": ["ß", "😀🏽", "ſ", "'ll", "t", "'T", "''", "T", "'M", "0", "́!", "a", "'S", ",㍿", "٣٤٥", "٦", ">", "
", "…s", ">́'", "re", "\n", "'s"]} +{"text": "
\r\n
İ0'Red#$% 12345678åe \nع'VE'Re d'T漢> 'll'Tİ́㍿३𐞁ß12345678 ㋿", "tokens": 57, "pieces": ["
\r\n", "
İ", "0", "'Re", "d", "#$%<", "EOT", ">", " ", "123", "456", "78", "a", "̊e", " \n", "ع", "'VE", "'Re", " d", "'T", "漢", ">", " '", "ll", "'T", "İ", "́㍿", "३", "𐞁ß", "123", "456", "78", " ", " ㋿"]} +{"text": "\u000bعع<|endoftext|>㍿ ​<|endoftext|>EOT'ſtd㍿'s‍'M\r(#$%9\n<|fim_prefix|>! \n'ſ9٣٤٥٦'T\tⅣ", "tokens": 68, "pieces": ["\u000bعع", "<|", "endoftext", "|>㍿", " ", " ​<|", "endoftext", "|>", "EOT", "'ſ", "td", "㍿'", "s", "‍'", "M", "\r", "(#$%", "9", "\n", "<|", "fim", "_prefix", "|>!", " \n", "'ſ", "9٣٤", "٥٦", "'T", "\t", "Ⅳ"]} +{"text": "eEOT''ress'VEm>'S- 'M字\r\n\r\n'Sḍ̇İ0're'Re
́dEOT", "''", "ress", "'VE", "m", ">'", "S", "-", " '", "M字", "\r\n\r\n", "'S", "ḋ", "̣İ", "0", "'re", "'Re", "
", "́d", "EOT‍ß\r
\n字<\u000bꟲḍ̇Dž'D́ésⅣ!\u000b<…(<|endoftext|>0's'Re 're'Sſa", "tokens": 67, "pieces": ["'re", "0", "A", ".㋿<", "EOT", ">EOT", "‍ß", "\r
\n", "字", "<", "\u000bꟲḋ", "̣Dž", "'D", "́e", "́s", "Ⅳ", "!", "\u000b", "<", "…", "(<|", "endoftext", "|>", "0", "'", "s", "'Re", " <", "META", "_START", ">'", "re", "'S", "ſa"]} +{"text": "\" é", "tokens": 4, "pieces": ["\"", " ", " e", "́"]} +{"text": "😀🏽's…½\t३#$%ḍ̇'T\r'Reé-0İ㍿<|fim_prefix|>Dž \n ", "tokens": 62, "pieces": ["ꟲع", " ßAe", "(\r\n\r\n", "eß", "㍿", " ", " ḋ", "̣'", "sꟲ", "<|", "endoftext", "|>", "ḋ", "̣'", "T", "\r", "'Re", "e", "́-", "0", "İ", "㍿<|", "fim", "_prefix", "|>", "Dž", " \n "]} +{"text": "عe 'D३'s'reéa#$%9字\r!'s", "tokens": 18, "pieces": ["عe", " ", "'D", "३", "'s", "'re", "e", "́a", "#$%", "9", "字", "\r", "!'", "s"]} +{"text": "'ſⅣⅣ12345678>Dž​٣٤٥٦३", "tokens": 25, "pieces": ["'ſ", "ⅣⅣ1", "234", "567", "8", ">Dž", "​", "٣٤٥", "٦३"]} +{"text": "\rſ'ſ.ſ", "tokens": 9, "pieces": ["\r", "ſ", "'ſ", ".ſ"]} +{"text": "​(ꟲ'M🙂'VE\re'VE\n12345678İ's!!>٣٤٥٦Z\r\n\r\n\tA३\r\nZ㍿\r,­٣٤٥٦-\r !!'VE<|endoftext|><|endoftext|>", "tokens": 72, "pieces": ["​(", "ꟲ", "'M", "🙂'", "VE", "\r", "e", "'VE", "\n", "123", "456", "78", "İ", "'s", "!!>", "٣٤٥", "٦", "Z", "\r\n\r\n", "\tA", "३", "\r\n", "Z", "㍿\r", ",­", "٣٤٥", "٦", "-\r", " ", "!!'", "VE", "<|", "endoftext", "|><|", "endoftext", "|>"]} +{"text": "ⅣⅣZé'T \tm…'T<|fim_prefix|>ḍ̇- \n㋿Z'sé漢३​m \nd", "tokens": 42, "pieces": ["ⅣⅣ", "Ze", "́'", "T", " ", "\tm", "…", "'T", "<|", "fim", "_prefix", "|>", "ḋ", "̣-", " \n", "㋿Z", "'s", "e", "́漢", "३", "​m", " \n", "d"]} +{"text": "!!ꟲ'Ts, #$%,…🙂>​aİ\r\n\r\n
ée

é‍,a'VE​aß\r\n\t9t<|fim_prefix|>\rſ٣٤٥٦<|fim_prefix|>Džae", "tokens": 74, "pieces": ["ſ", "\n", "😀🏽'", "re", "<|", "fim", "_prefix", "|>🙂>​", "aİ", "\r\n\r\n", "
ée", "
", "
e", "́‍,", "a", "'VE", "​aß", "\r\n", "\t", "9", "t", "<|", "fim", "_prefix", "|>\r", "ſ", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "Džae"]} +{"text": "<|endoftext|>㍿­ås½ ع\r\n(ع𐞁👍🏽-'ll!'llm㋿Z", "tokens": 40, "pieces": ["<|", "endoftext", "|>㍿­", "a", "̊s", "½", " ", " ع", "\r\n", "(ع𐞁", "👍🏽-'", "ll", "!'", "llm", "㋿Z"]} +{"text": "<|fim_prefix|>İd㍿ß ­🙂!>,\r\n\r\n𐞁٣٤٥٦\" ,ꟲ…tå a½é<|endoftext|>,٣٤٥٦é>", "tokens": 67, "pieces": ["<|", "fim", "_prefix", "|>", "İd", "㍿ß", " ­🙂!>,\r\n\r\n", "𐞁", "٣٤٥", "٦", "\"", " ,", "ꟲ", "…ta", "̊", " a", "½", "e", "́<|", "endoftext", "|>,", "٣٤٥", "٦", "e", "́>"]} +{"text": "> ́!\t're'T㍿ſ㍿EOT\r\n\r\n­'ReEOTⅣ'M", "tokens": 29, "pieces": [">", " ", "́!<", "META", "_START", ">", "\t", "'re", "'T", "㍿ſ", "㍿EOT", "\r\n\r\n", "­'", "ReEOT", "Ⅳ", "'M"]} +{"text": "m-½𐞁Zꟲ漢㋿'D<|fim_prefix|>.<|fim_prefix|>'re'T0''lle\ńmsd12345678!!'Reſ(Ⅳ", "tokens": 52, "pieces": ["m", "-", "½", "𐞁Zꟲ漢", "㋿'", "D", "<|", "fim", "_prefix", "|>.<|", "fim", "_prefix", "|>'", "re", "'T", "0", "''", "lle", "\n", "́msd", "123", "456", "78", "!!'", "Reſ", "(", "Ⅳ"]} +{"text": "ꟲ👍🏽٣٤٥٦ Dž‍9'M㋿fi<|endoftext|>'ſs\té(㋿", "tokens": 53, "pieces": ["ꟲ", "👍🏽<", "META", "_START", ">", "٣٤٥", "٦", " Dž", "‍<", "META", "_START", ">", "9", "'M", "㋿fi", "<|", "endoftext", "|>'", "ſs", "\té", "(㋿"]} +{"text": "𐞁'VE\r\n🙂‍字字(", "tokens": 14, "pieces": ["𐞁", "'VE", "\r\n", "🙂‍", "字字", "("]} +{"text": "́­'Re>Áå.'VE½\n'Tt9åⅣ\"fi🙂A", "tokens": 33, "pieces": ["́­'", "Re", "><", "EOT", ">A", "́a", "̊.'", "VE", "½", "\n", "'T", "t", "9", "a", "̊", "Ⅳ", "\"fi", "🙂A"]} +{"text": "<ſå½EOTeDžé ", "tokens": 14, "pieces": ["<ſa", "̊", "½", "EOTeDžé", " "]} +{"text": "éİ(\r㍿ \né३Z!!9\u000bⅣ\u000b<|fim_prefix|>'ſ ḍ̇漢-\tm'Té", "tokens": 43, "pieces": ["éİ", "(\r", "㍿", " \n", "e", "́", "३", "Z", "!!", "9", "", "\u000b", "Ⅳ", "\u000b", "<|", "fim", "_prefix", "|>'", "ſ", " ḋ", "̣漢", "-", "\tm", "'T", "é"]} +{"text": "'ll \"\u000b 🙂३<‍🙂12345678Dž <㋿e漢'ſⅣ0ع<|endoftext|>a㍿", "tokens": 44, "pieces": ["'ll", " ", " \"", "\u000b ", " 🙂", "३", "<‍🙂", "123", "456", "78", "Dž", " ", " <㋿", "e漢", "'ſ", "Ⅳ0", "ع", "<|", "endoftext", "|>", "a", "㍿"]} +{"text": "!!'ſa'Re\rA!A9m", "tokens": 14, "pieces": ["!!'", "ſa", "'Re", "\r", "A", "!A", "9", "m"]} +{"text": "ع12345678", "tokens": 4, "pieces": ["ع", "123", "456", "78"]} +{"text": "​字ع're\n ㍿'VEé'llfi", "tokens": 15, "pieces": ["​字ع", "'re", "\n", " ㍿'", "VEé", "'ll", "fi"]} +{"text": " \n.عİ12345678d'fia\n​३ḍ̇m!>'re#$%(​👍🏽𐞁
'S­'llſſ\n👍🏽 \nDž'Re 'sfi", "tokens": 71, "pieces": [" \n", ".عİ", "123", "456", "78", "d", "'fia", "\n", "​", "३", "ḋ", "̣m", "!><", "META", "_START", ">", "३", "'", "re", "#$%(​👍🏽", "𐞁", "
", "'S", "­'", "llſſ", "\n", "👍🏽", " \n", "Dž", "'Re", " ", "'s", "fi"]} +{"text": "d", "tokens": 1, "pieces": ["d"]} +{"text": "\t㋿", "tokens": 4, "pieces": ["\t", "㋿"]} +{"text": ">  ㋿'llZ\t<|endoftext|><|endoftext|>'VE­ḍ̇!!'s­-\r'", "tokens": 37, "pieces": [">", " ", " ", "㋿'", "llZ", "\t", "<|", "endoftext", "|><|", "endoftext", "|>'", "VE", "­ḋ", "̣!!'", "s", "­-\r", "'"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ꟲDž㍿𐞁d<|endoftext|><|fim_prefix|>< (عA字‍‍'T👍🏽's's#$%!!<Ⅳß ٣٤٥٦>'Dع ", "tokens": 67, "pieces": ["ꟲDž", "㍿𐞁d", "<|", "endoftext", "|><|", "fim", "_prefix", "|><", " ", " (", "عA字", "‍‍'", "T", "👍🏽'", "s", "'s", "#$%!!<", "Ⅳ", "ß", " ", "٣٤٥", "٦", ">'", "Dع", " "]} +{"text": "m \n's9İé(!>३㋿­<|fim_prefix|> !漢𐞁ḍ̇s#$%åß12345678'D", "tokens": 52, "pieces": ["m", " \n", "'s", "9", "İé", "(!>", "३", "㋿­<|", "fim", "_prefix", "|>", " ", " <", "META", "_START", ">!", "漢𐞁", "ḋ", "̣s", "#$%", "a", "̊ß", "123", "456", "78", "'D"]} +{"text": "!é🙂\r\n३s𐞁'llEOT>
㋿>,́<𐞁عḍ̇\"عté0sam", "tokens": 40, "pieces": ["!é", "🙂\r\n", "३", "s𐞁", "'ll", "EOT", ">", "
", "㋿>,́<", "𐞁عḋ", "̣\"", "عte", "́", "0", "sam"]} +{"text": "fiZ'ſ!\nß\u000bsea0'ſ
Ⅳ🙂👍🏽 \n‍
('M t('DA!éⅣ‍ m\té9d😀🏽'ſ", "tokens": 63, "pieces": ["fiZ", "'ſ", "!\n", "ß", "\u000bsea", "0", "'ſ", "
", "Ⅳ", "🙂👍🏽", " \n", "‍", "
", "('", "M", " t", "('", "DA", "!e", "́", "Ⅳ", "‍", " m", "\té", "9", "d", "😀🏽'", "ſ"]} +{"text": "12345678\n'…><|fim_prefix|>A(d'ſ!'ll<|fim_prefix|>'s", "tokens": 30, "pieces": ["123", "456", "78", "\n", "'", "…", "><|", "fim", "_prefix", "|>", "A", "(d", "'ſ", "!'", "ll", "<|", "fim", "_prefix", "|>'", "s"]} +{"text": "\r\n Z>'ſåſ\"ſ½­-(< ع\r\n\r\n
", "tokens": 23, "pieces": ["\r\n", " Z", ">'", "ſa", "̊ſ", "\"ſ", "½", "­-(<", " ع", "\r\n\r\n
"]} +{"text": "!😀🏽🙂EOT'VEſ", "tokens": 14, "pieces": ["!😀🏽🙂", "EOT", "'VE", "ſ"]} +{"text": "é#$%m\r\n\r\nß𐞁>\t(‍A\r\n\r\n'ſḍ̇३'S🙂İ(#$%", "tokens": 39, "pieces": ["e", "́#$%", "m", "\r\n\r\n", "ß", "𐞁", ">", "\t", "(‍", "A", "\r\n\r\n", "'ſ", "ḋ", "̣", "३", "'S", "🙂İ", "(#$%"]} +{"text": ".'Sḍ̇'S…", "tokens": 15, "pieces": [".'", "Sḋ", "̣'", "S", "", "…"]} +{"text": "ع …A🙂'T. \n!'Re🙂é12345678Z\r\n", "tokens": 27, "pieces": ["ع", "", " ", "…A", "🙂'", "T", ".", " \n", "!'", "Re", "🙂e", "́", "123", "456", "78", "Z", "\r\n"]} +{"text": "\t३‍A🙂Ⅳ㋿İḍ̇
<\r\n\r\n½<|fim_prefix|>ꟲ'DعADž'D'VE<|endoftext|>'ſ", "tokens": 53, "pieces": ["\t", "३", "‍A", "🙂", "Ⅳ", "㋿İḋ", "̣", "
", "<\r\n\r\n", "½", "<|", "fim", "_prefix", "|>", "ꟲ", "'D", "عADž", "'D", "'VE", "<|", "endoftext", "|>'", "ſ"]} +{"text": "maſ\r\n\r\n'TDž'D Dž\rⅣ'T12345678å\t
Dž漢(!ع(!! ½EOT\r\nḍ̇!! <|fim_prefix|>", "tokens": 57, "pieces": ["maſ", "\r\n\r\n", "'T", "Dž", "'D", " ", " Dž", "\r", "Ⅳ", "'T", "123", "456", "78", "a", "̊", "\t", "
Dž漢", "(!", "ع", "(!!", " ", " ", "½", "EOT", "\r\n", "ḋ", "̣!!<", "EOT", ">", " ", "<|", "fim", "_prefix", "|>"]} +{"text": "​<|fim_prefix|>\t'lls漢Ⅳ >'T٣٤٥٦! Z#$%\"'re>
ع'M'ſEOT's👍🏽'\r\n\r\n‍​s\n­<|endoftext|>​!!ḍ̇0,", "tokens": 73, "pieces": ["​<|", "fim", "_prefix", "|>", "\t", "'ll", "s漢", "Ⅳ", " ", ">'", "T", "٣٤٥", "٦", "!", " Z", "#$%\"'", "re", ">", "
ع", "'M", "'ſ", "EOT", "'s", "👍🏽'\r\n\r\n", "‍​", "s", "\n", "­<|", "endoftext", "|>​!!", "ḋ", "̣", "0", ","]} +{"text": "½!t३\n .d㍿'Res​12345678 \n\r\n.漢…ع…ß'Re漢'VE<|endoftext|>٣٤٥٦ß
.\u000b'Séd
漢", "tokens": 70, "pieces": ["½", "!t", "३", "\n", " ", " .", "d", "㍿'", "Res", "​", "123", "456", "78", " \n\r\n", ".", "漢", "…ع", "…ß", "'Re", "漢", "'VE", "<|", "endoftext", "|>", "٣٤٥", "٦", "ß", "", "
", ".", "\u000b", "'S", "éd", "
漢"]} +{"text": "…s👍🏽e३'VE''reZ  \n", "tokens": 18, "pieces": ["٣٤٥", "٦", "<|", "endoftext", "|>", "Z", "  \n"]} +{"text": "m>'S12345678\r\nḍ̇m‍ḍ̇漢d9<|fim_prefix|>ds, ( EOT!'D9é, a
#$%𐞁d́🙂", "tokens": 60, "pieces": ["m", ">'", "S", "123", "456", "78", "\r\n", "ḋ", "̣m", "‍ḋ", "̣漢d", "9", "<|", "fim", "_prefix", "|>", "ds", ",", " (", " ", " EOT", "!<", "META", "_START", ">'", "D", "9", "e", "́,", " a", "
", "#$%", "𐞁d", "́🙂"]} +{"text": "(", "tokens": 1, "pieces": ["("]} +{"text": ">​-<|fim_prefix|>㍿'Tt(< 'D🙂 Z\r\n\r\n
(s'VE.#$%Dž​'T㋿\t​
Dž'-", "tokens": 51, "pieces": [">​-<|", "fim", "_prefix", "|>㍿'", "Tt", "(<", " ", "'D", "🙂", " Z", "\r\n\r\n", "
", "(s", "'VE", ".#$%", "Dž", "​'", "T", "㋿", "\t", "​", "
Dž", "'-"]} +{"text": "'Ḿ9", "tokens": 3, "pieces": ["'M", "́", "9"]} +{"text": "'s'S>", "tokens": 3, "pieces": ["'s", "'S", ">"]} +{"text": "‍㋿Ⅳ", "tokens": 7, "pieces": ["‍㋿", "Ⅳ"]} +{"text": "٣٤٥٦\r\n\r\n­å ­0​½½9字e \u000b", "tokens": 26, "pieces": ["٣٤٥", "٦", "\r\n\r\n", "­a", "̊", " ­", "0", "​", "½½9", "字e", " \u000b"]} +{"text": "ßⅣ> m'VEfi ḍ̇", "tokens": 16, "pieces": ["ß", "Ⅳ", ">", " ", " m", "'VE", "fi", " ḋ", "̣"]} +{"text": "'M<|fim_prefix|>ع\t<ꟲ'S><|endoftext|>‍'ſ\u000b́
EOTé­
 \n9", "tokens": 42, "pieces": ["'M", "<|", "fim", "_prefix", "|>", "ع", "\t", "<", "ꟲ", "'S", "><|", "endoftext", "|>‍'", "ſ", "\u000b", "́", "
EOTé", "­", "
 \n", "9"]} +{"text": "\rß\rZa\tm 👍🏽0 🙂漢EOT'll're''A\u000b< \naa-'llå\r\nḍ̇t­<12345678'Mſ", "tokens": 51, "pieces": ["\r", "ß", "\r", "Za", "\tm", " ", "👍🏽", "0", " ", "🙂漢", "EOT", "'ll", "'re", "''", "A", "\u000b", "<", " \n", "aa", "-'", "lla", "̊\r\n", "ḋ", "̣t", "­<", "123", "456", "78", "'M", "ſ"]} +{"text": " ßd٣٤٥٦EOT\r\n\r\n9-🙂'ſEOT\r\u000b<|fim_prefix|>'🙂 !!字mZꟲ<|endoftext|>!!٣٤٥٦'ſ!'ſ!!ſ", "tokens": 66, "pieces": [" ßd", "٣٤٥", "٦", "EOT", "\r\n\r\n", "9", "-🙂'", "ſEOT", "\r", "\u000b", "<|", "fim", "_prefix", "|>'🙂", " !!", "字mZꟲ", "<|", "endoftext", "|>!!", "٣٤٥", "٦", "'ſ", "!'", "ſ", "!!", "ſ"]} +{"text": "0're<…,12345678­Džå'Tḍ̇'Re👍🏽\nDž漢𐞁", "tokens": 43, "pieces": ["0", "'re", "<", "…", ",", "123", "456", "78", "­", "Dža", "̊'", "Tḋ", "̣'", "Re", "👍🏽\n", "Dž漢𐞁"]} +{"text": "-'VE…<|fim_prefix|>'́́'0e‍s,''ll'ſ\t…s0👍🏽é", "tokens": 40, "pieces": ["-'", "VE", "…", "<|", "fim", "_prefix", "|>'́́'", "0", "e", "‍s", ",''", "ll", "'ſ", "\t", "…s", "0", "👍🏽", "e", "́"]} +{"text": "ḍ̇  ३'Tḍ̇漢 ſ's…'D…  <\rعİétḍ̇å🙂0é\u000bع<|endoftext|>'T\u000b३", "tokens": 65, "pieces": ["ḋ", "̣", " <", "EOT", ">", " ", "३", "'T", "ḋ", "̣漢", " ſ", "'s", "…", "'D", "… ", " ", "<\r", "عİétḋ", "̣a", "̊🙂", "0", "e", "́", "\u000bع", "<|", "endoftext", "|>'", "T", "\u000b", "३", ""]} +{"text": "٣٤٥٦(é👍🏽e9 \n(😀🏽0\r'Sſ're漢é\r\né👍🏽,0 ſds's㍿9ḍ̇'re(🙂ſa ", "tokens": 66, "pieces": ["٣٤٥", "٦", "(é", "👍🏽", "e", "9", " \n", "(😀🏽", "0", "\r", "'S", "ſ", "'re", "漢é", "\r\n", "é", "👍🏽,", "0", " ſds", "'s", "㍿", "9", "ḋ", "̣'", "re", "(🙂", "ſa", " "]} +{"text": "ßİ0​'ſ#$%​\"㋿́'Sꟲ<ع-٣٤٥٦", "tokens": 31, "pieces": ["ßİ", "0", "​'", "ſ", "#$%​\"㋿́'", "Sꟲ", "<ع", "-", "٣٤٥", "٦"]} +{"text": "😀🏽<|fim_prefix|>s!'M!!😀🏽\u000bea", "tokens": 23, "pieces": ["😀🏽<|", "fim", "_prefix", "|>", "s", "!'", "M", "!!😀🏽", "\u000bea"]} +{"text": ".½\r\n𐞁ſ㋿½👍🏽Dž''ll㋿'S's'ſ½0 mfi Z,s'Redİ", "tokens": 44, "pieces": [".", "½", "\r\n", "𐞁ſ", "㋿", "½", "👍🏽", "Dž", "''", "ll", "㋿'", "S", "'s", "'ſ", "½0", " mfi", " Z", ",s", "'Re", "dİ"]} +{"text": "㍿", "tokens": 3, "pieces": ["㍿"]} +{"text": "é12345678fi 𐞁'ſ<|fim_prefix|>३!!İ\t0Z!d.<|fim_prefix|>🙂#$%>…!", "tokens": 49, "pieces": ["é", "123", "456", "78", "fi", " ", " 𐞁", "'ſ", "<|", "fim", "_prefix", "|><", "META", "_START", ">", "३", "!!", "İ", "\t", "0", "Z", "!d", ".<|", "fim", "_prefix", "|>🙂#$%>", "…", "!"]} +{"text": "#$%㍿'Mm
,!!0'S'ſ'VE'D", "tokens": 23, "pieces": ["#$%㍿'", "M", "m", "
", ",!!", "0", "'S", "'ſ", "'VE", "'D"]} +{"text": "<|endoftext|>'S字m'VEt́9#$%''ret\r'll­", "tokens": 22, "pieces": ["<|", "endoftext", "|>'", "S字m", "'VE", "t", "́", "9", "#$%''", "ret", "\r", "'ll", "­"]} +{"text": "mⅣé", "tokens": 4, "pieces": ["m", "Ⅳ", "é"]} +{"text": "DžEOT ('Re'll½🙂㋿ⅣÁ(👍🏽>́", "tokens": 34, "pieces": ["DžEOT", " ", " (<", "EOT", ">'", "Re", "'ll", "½", "🙂㋿", "Ⅳ", "A", "́(👍🏽><", "META", "_START", ">́"]} +{"text": "३ A(­'D s Dž🙂Dž \n㍿ 'ſ \t🙂A ३#$%Dž0'T'0\u000b", "tokens": 45, "pieces": [" ", "😀🏽", "A", "<|", "endoftext", "|>", " \n", "㍿", " ", "'ſ", " ", "\t", "🙂A", " ", "३", "#$%", "Dž", "", "0", "'T", "'", "0", "\u000b"]} +{"text": "A<|endoftext|>fiß9ꟲ<|endoftext|>e ½ >㋿İ😀🏽' ㍿😀🏽́\u000b३\n\u000b9d'M'Tße", "tokens": 60, "pieces": ["A", "<|", "endoftext", "|>", "fiß", "9", "ꟲ", "<|", "endoftext", "|>", "e", " ", "½", " ", ">㋿", "İ", "😀🏽'", " ", "㍿😀🏽́", "\u000b", "३", "\n", "\u000b", "9", "d", "'M", "'T", "ße"]} +{"text": "…½é\r\ne\t9 #$%'ll(s漢'ſ<|endoftext|>A<|fim_prefix|>'S<|endoftext|>'D>EOT㋿٣٤٥٦🙂<½'M½té>'T", "tokens": 67, "pieces": ["…", "½", "e", "́\r\n", "e", "\t", "9", " ", " #$%'", "ll", "(s漢", "'ſ", "<|", "endoftext", "|>", "A", "<|", "fim", "_prefix", "|>'", "S", "<|", "endoftext", "|>'", "D", ">EOT", "㋿", "٣٤٥", "٦", "🙂<", "½", "'M", "½", "te", "́>'", "T"]} +{"text": " Zma<‍és'VE'sm'S \n👍🏽ꟲ9😀🏽", "tokens": 31, "pieces": [" Zma", "<‍", "és", "'VE", "'s", "m", "'S", " \n", "👍🏽", "ꟲ", "9", "😀🏽"]} +{"text": "ß'VE<|fim_prefix|>d\t!!ḍ̇å\ńſ\"́é​ 'D'VE㋿!'sé😀🏽EOT 'lld😀🏽́'s", "tokens": 61, "pieces": ["ß", "'VE", "<|", "fim", "_prefix", "|>", "d", "\t", "!!", "ḋ", "̣a", "̊\n", "́ſ", "\"́", "e", "́​", " ", "'D", "'VE", "㋿!'", "se", "́😀🏽", "EOT", " ", "'ll", "d", "😀🏽́'", "s", ""]} +{"text": "é<|endoftext|>EOT́<|fim_prefix|> \n( fiß-㋿'ſ<|fim_prefix|>,'s \n#$%-
\r", "tokens": 46, "pieces": ["e", "́<|", "endoftext", "|>", "EOT", "́<|", "fim", "_prefix", "|>", " \n", "(", " fiß", "-㋿'", "ſ", "<|", "fim", "_prefix", "|>,'", "s", " \n", "#$%-", "
\r"]} +{"text": "
é٣٤٥٦'D \n'll\u000bå'MEOT 'll'ſⅣ'\rfifid😀🏽­…字😀🏽३'ll'Dd,ḍ̇", "tokens": 62, "pieces": ["
e", "́", "٣٤٥", "٦", "'D", " \n", "'ll", "\u000ba", "̊'", "MEOT", " ", " '", "ll", "'ſ", "Ⅳ", "'\r", "fifid", "😀🏽­", "…字", "😀🏽", "३", "'ll", "'D", "d", ",ḋ", "̣"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿½<ſ'll \n(½'ll字'Re!!!!<…ꟲ½a\tA'VE'lltⅣDžm'Re ", "tokens": 37, "pieces": ["㍿", "½", "<ſ", "'ll", " \n", "(", "½", "'ll", "字", "'Re", "!!!!<", "…ꟲ", "½", "a", "\tA", "'VE", "'ll", "t", "Ⅳ", "Džm", "'Re", " "]} +{"text": "'s…\u000b\r-'s'll's'ſ​\r\n\r\n'Sm\u000b½ !\ré9😀🏽s ſ", "tokens": 34, "pieces": ["'s", "…\u000b\r", "-'", "s", "'ll", "'s", "'ſ", "​\r\n\r\n", "'S", "m", "\u000b", "½", " !\r", "é", "9", "😀🏽<", "EOT", ">s", " ſ"]} +{"text": "😀🏽ſ'ſİ
字ém'Re \n​<́३ 'T", "tokens": 29, "pieces": ["😀🏽", "ſ", "'ſ", "İ", "
字", "e", "́m", "'Re", " \n", "​<́", "३", " ", "'T"]} +{"text": "'VE<#$%'re<|fim_prefix|>!fié٣٤٥٦'s>fis\"<|fim_prefix|> ३'ll\r\n12345678𐞁>,", "tokens": 51, "pieces": ["'VE", "<#$%'", "re", "<|", "fim", "_prefix", "|>!", "fie", "́", "٣٤٥", "٦", "'s", ">fis", "\"<|", "fim", "_prefix", "|>", " ", "३", "'ll", "\r\n", "123", "456", "78", "𐞁", ">,"]} +{"text": " 
a éꟲ👍🏽
ꟲ'ſ 'VEtZ🙂½漢", "tokens": 38, "pieces": [" ", "
a", " e", "́ꟲ", "👍🏽", "
ꟲ", "'ſ", " '", "VEt", "<", "EOT", ">Z", "🙂", "½", "漢"]} +{"text": "Ⅳİ''ll\t…å😀🏽 \n\u000b!عDž", "tokens": 22, "pieces": ["Ⅳ", "İ", "''", "ll", "\t", "…a", "̊😀🏽", " \n", "\u000b", "!عDž"]} +{"text": "\r\nfi 👍🏽!!,😀🏽İ字EOT\r\n\r\nⅣⅣ.ḍ̇…🙂Ⅳ
(('Ss t́Ⅳ \nZ\t 's\r\n\r\n!!", "tokens": 54, "pieces": ["\r\n", "fi", " ", "👍🏽!!,😀🏽", "İ字EOT", "\r\n\r\n", "ⅣⅣ", ".ḋ", "̣", "…", "🙂", "Ⅳ", "
", "(('", "Ss", " ", " t", "́", "Ⅳ", " \n", "Z", "\t ", " '", "s", "\r\n\r\n", "!!"]} +{"text": "é", "tokens": 1, "pieces": ["é"]} +{"text": "­-👍🏽A😀🏽s\t.12345678½ \r\nḍ̇''Re​‍.
sḍ̇a", "tokens": 43, "pieces": ["­-👍🏽", "A", "😀🏽", "s", "\t", ".", "123", "456", "78½", " \r\n", "ḋ", "̣''", "Re", "​‍.", "
sḋ", "̣a"]} +{"text": "A‍👍🏽\t​ås字 ع'Ds", "tokens": 20, "pieces": ["A", "‍👍🏽", "\t", "​a", "̊s字", " ع", "'D", "s"]} +{"text": "9", "tokens": 1, "pieces": ["9"]} +{"text": "Dž", "tokens": 2, "pieces": ["Dž"]} +{"text": "́,İt'VE!!å'ſⅣ🙂'ſ's9\r\nſ𐞁ḍ̇ <|fim_prefix|>Ź٣٤٥٦'Dİfiꟲß,s0a 字", "tokens": 68, "pieces": ["́,", "İt", "'VE", "!!", "a", "̊'", "ſ", "Ⅳ", "🙂'", "ſ", "'s", "9", "\r\n", "ſ", "𐞁ḋ", "̣", " ", "<|", "fim", "_prefix", "|>", "Z", "́", "٣٤٥", "٦", "'D", "İfiꟲß", ",s", "0", "a", " ", " 字"]} +{"text": "'S<|fim_prefix|><<|fim_prefix|>…-\r\n\"é\rZ'llع(fi", "tokens": 31, "pieces": ["'S", "<|", "fim", "_prefix", "|><<|", "fim", "_prefix", "|><", "META", "_START", ">", "…", "-\r\n", "\"e", "́\r", "Z", "'ll", "ع", "(fi"]} +{"text": "s'½ \nééſ'll👍🏽'ſ😀🏽fi\r\n9é‍'VE\r… 𐞁\r'Dİſ\"é 字İé'M'DA🙂,ſ", "tokens": 66, "pieces": ["s", "'", "½", " \n", "ééſ", "'ll", "👍🏽'", "ſ", "😀🏽", "fi", "\r\n", "9", "é", "‍'", "VE", "\r", "…", "", " 𐞁", "\r", "'D", "İſ", "\"é", " ", " 字İe", "́'", "M", "'D", "A", "🙂,", "ſ"]} +{"text": "'re fi e'D('D", "tokens": 9, "pieces": ["'re", " ", " fi", " e", "'D", "('", "D"]} +{"text": "9're字­'D字'🙂12345678å>🙂's३ſ \n", "tokens": 26, "pieces": ["9", "'re", "字", "­'", "D字", "'🙂", "123", "456", "78", "a", "̊>🙂'", "s", "३", "ſ", " \n"]} +{"text": "12345678😀🏽!字\r\n'Tḍ̇٣٤٥٦­½'T'VEⅣ\"", "tokens": 37, "pieces": ["123", "456", "78", "😀🏽!", "字", "\r\n", "'T", "ḋ", "̣", "٣٤٥", "٦", "­", "½", "'T", "'VE", "", "Ⅳ", "\""]} +{"text": ".👍🏽\r\n\r\nd­‍<|endoftext|>😀🏽 Ⅳ<|endoftext|>12345678m\r\n\r\n'S漢ꟲ", "tokens": 45, "pieces": [".👍🏽\r\n\r\n", "d", "­‍<|", "endoftext", "|>😀🏽", " ", "Ⅳ", "<|", "endoftext", "|>", "123", "456", "78", "m", "\r\n\r\n", "'S", "漢ꟲ"]} +{"text": ".>#$%Z12345678…㍿", "tokens": 13, "pieces": [".>#$%", "Z", "123", "456", "78", "…", "㍿"]} +{"text": "#$% \n'Re \n<́(ꟲ㍿​fiéfi\t​Ⅳ'S ́ßemd\t0\n", "tokens": 35, "pieces": ["#$%", " \n", "'Re", " \n", "<́(<", "EOT", ">ꟲ", "㍿​", "fiéfi", "\t", "​", "Ⅳ", "'S", " ", "́ßemd", "\t", "0", "\n"]} +{"text": "(…'lleá#$%EOT!! ​9ß­'res\"ⅣⅣ-'M३e#$%'s12345678'll<|fim_prefix|>…Džfi‍åZ<|endoftext|>", "tokens": 64, "pieces": ["(", "…", "'ll", "ea", "́#$%", "EOT", "!!", " ", "​", "9", "ß", "­'", "res", "\"", "ⅣⅣ", "-'", "M", "३", "e", "#$%'", "s", "123", "456", "78", "'ll", "<|", "fim", "_prefix", "|>", "…Džfi", "‍a", "̊Z", "<|", "endoftext", "|>"]} +{"text": "
‍३'re0\r\nDž­🙂a👍🏽­㍿-ḍ̇Dž", "tokens": 37, "pieces": ["
", "‍", "३", "'re", "0", "\r\n", "Dž", "­🙂", "a", "👍🏽­<", "EOT", ">㍿-", "ḋ", "̣Dž"]} +{"text": "\té'ſ \nḍ̇aꟲ''VE'VEé 漢…' A'Re's.", "tokens": 37, "pieces": ["\té", "'ſ", " \n", "ḋ", "̣aꟲ", "''", "VE", "'VE", "e", "́", " ", " 漢", "…", "'", " A", "'Re", "'s", ".<", "META", "_START", ">"]} +{"text": "12345678'stⅣع.… ,EOT字İ漢㋿ !!'M🙂㋿
", "tokens": 35, "pieces": ["123", "456", "78", "'s", "t", "Ⅳ", "ع", ".", "…", " ", ",EOT字İ漢", "㋿", " ", "!!<", "META", "_START", ">'", "M", "🙂㋿", "
"]} +{"text": " \n.Aİ,½👍🏽're\t'VE𐞁عéDžſ(#$%'reſ ㋿😀🏽ع!å𐞁‍s👍🏽å­Z'", "tokens": 72, "pieces": [" \n", ".A", "İ", ",", "½", "👍🏽'", "re", "\t", "'VE", "𐞁عéDžſ", "(#$%'", "reſ", " ", " ㋿😀🏽", "ع", "!a", "̊𐞁", "‍s", "👍🏽", "a", "̊­", "Z", "'"]} +{"text": "ḍ̇ꟲms12345678e'S9٣٤٥٦字ḍ̇\"ع㍿İ'M'llſ\r\n 🙂a<|endoftext|>fi漢!३<|fim_prefix|> \u000b٣٤٥٦>", "tokens": 79, "pieces": ["ḋ", "̣ꟲms", "123", "456", "78", "e", "'S", "9", "", "٣٤٥", "٦", "字ḋ", "̣\"", "ع", "㍿İ", "'M", "'ll", "ſ", "\r\n", " ", " 🙂", "a", "<|", "endoftext", "|>", "fi漢", "!", "३", "<|", "fim", "_prefix", "|>", " ", "\u000b", "٣٤٥", "٦", ">"]} +{"text": "ßåع0ع12345678½㋿ Džꟲtſ-sꟲ#$%'ſ'ſt''D sDž\"a‍Ⅳ字12345678 \n", "tokens": 55, "pieces": ["ßa", "̊<", "EOT", ">ع", "0", "ع", "123", "456", "78½", "㋿", " Džꟲtſ", "-sꟲ", "#$%'", "ſ", "'ſ", "t", "''", "D", " ", " sDž", "\"a", "‍", "Ⅳ", "字", "123", "456", "78", " \n"]} +{"text": "m<|endoftext|>Ⅳ\r\n\r\n!\r\n\r\n'漢  'S'Ses'D're𐞁'VE.aDžm(", "tokens": 37, "pieces": ["m", "<|", "endoftext", "|>", "Ⅳ", "\r\n\r\n", "!\r\n\r\n", "'漢", " ", " ", "'S", "'S", "es", "'D", "'re", "𐞁", "'VE", ".a", "Džm", "("]} +{"text": "\téa\t0t漢ꟲEOT\u000bEOT́e­-\u000b​", "tokens": 25, "pieces": ["\téa", "\t", "0", "t漢ꟲEOT", "\u000bEOT", "́e", "­<", "EOT", ">-", "\u000b", "​"]} +{"text": "<|endoftext|>½EOTfiḍ̇​½́
\"ḍ̇#$%Dž​Aꟲ\"<", "tokens": 39, "pieces": ["<|", "endoftext", "|>", "½", "EOTfiḋ", "̣​", "½", "́", "
", "\"ḋ", "̣#$%", "Dž", "​Aꟲ", "\"<"]} +{"text": "٣٤٥٦0(😀🏽'lld0\r.<İ'Re'Re#$%<|endoftext|>#$%<|endoftext|>ß­'Re>sa", "tokens": 50, "pieces": ["٣٤٥", "٦0", "(😀🏽'", "lld", "0", "\r", ".<", "EOT", "><", "İ", "'Re", "'Re", "#$%<|", "endoftext", "|>#$%<|", "endoftext", "|>", "ß", "­'", "Re", ">sa"]} +{"text": "😀🏽12345678㋿ḍ̇.𐞁\r\n<|endoftext|>,ß -😀🏽é.'S9\r\n\r\n !!漢", "tokens": 45, "pieces": ["😀🏽", "123", "456", "78", "㋿ḋ", "̣.", "𐞁", "\r\n", "<|", "endoftext", "|>,", "ß", " -😀🏽", "é", ".'", "S", "9", "\r\n\r\n", " ", "!!", "漢"]} +{"text": "👍🏽\u000b ſ>‍!!字٣٤٥٦\"ḍ̇'३ß9é३\"e\nß0
!Ad  A9\n", "tokens": 55, "pieces": ["👍🏽", "\u000b ", " ſ", ">‍!!", "字", "٣٤٥", "٦", "\"ḋ", "̣'", "३", "ß", "9", "e", "́", "३", "\"e", "\n", "ß", "0", "
", "!Ad", " ", " A", "9", "\n"]} +{"text": "t12345678 'ſfiḍ̇ſfiⅣ㍿>", "tokens": 24, "pieces": ["t", "123", "456", "78", " '", "ſfiḋ", "̣ſfi", "Ⅳ", "㍿>"]} +{"text": "('Mꟲ字'Dß-<|endoftext|>٣٤٥٦\u000bd\r\nſ😀🏽\n…aémA…㍿½👍🏽३s
 ", "tokens": 80, "pieces": [" ", " <'", "sZ", " ", "‍", "½", "e", "'ſ", "e", "
e", "9", " \n", "<|", "fim", "_prefix", "|>'", "Dß", "-<|", "endoftext", "|>", "٣٤٥", "٦", "\u000bd", "\r\n", "ſ", "😀🏽\n", "…ae", "́mA", "…", "㍿", "½", "👍🏽", "३", "s", "
 "]} +{"text": " ſ>!'Re㋿'M's'T,İ!ع9-(­å㍿ع-'ll𐞁'Re​d😀🏽ꟲ !!😀🏽", "tokens": 54, "pieces": [" ſ", ">!'", "Re", "㋿'", "M", "'s", "'T", ",İ", "!ع", "9", "-(­", "a", "̊㍿", "ع", "-'", "ll", "𐞁", "'Re", "​d", "😀🏽", "ꟲ", " ", "!!😀🏽"]} +{"text": "३- \n٣٤٥٦é ſ!!'ll'ſ \n>'å­'ll½'M𐞁́Dž​<|endoftext|>!fißſa12345678<|endoftext|>́9'", "a", "̊­'", "ll", "½", "'M", "𐞁", "́Dž", "​<|", "endoftext", "|>!", "fißſa", "123", "456", "78", "<|", "endoftext", "|>́", "9", "字<|endoftext|>(0​\rA 0İ㍿'D㋿ḍ̇㋿\"Z", "tokens": 51, "pieces": [" \n", "9", "👍🏽", "é", "'re", "字", "<|", "endoftext", "|>(", "0", "​\r", "A", " ", "0", "İ", "㍿'", "D", "㋿ḋ", "̣<", "EOT", ">㋿\"", "Z"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'😀🏽ع!𐞁 s\"😀🏽字٣٤٥٦٣٤٥٦İ,٣٤٥٦ꟲ#$%,\r\n\r\n字a", "tokens": 55, "pieces": ["'😀🏽", "ع", "!𐞁", " s", "\"😀🏽", "字", "٣٤٥", "٦٣٤", "٥٦", "İ", ",", "٣٤٥", "٦", "ꟲ", "#$%,\r\n\r\n", "字a"]} +{"text": "\rDž'ſ're12345678‍'Re'T\"\r'M
ع\u000b\tt", "tokens": 26, "pieces": ["Z漢", "-'", "VEDž", " ", " 🙂,", "9", "字", "9", "…", "\"\r", "'M", "
ع", "\u000b", "\tt"]} +{"text": "<ḍ̇漢ꟲ'T<('Reꟲع\r\n\r\nDžⅣ'll㍿!!Dž \n'T'Dعfi<|fim_prefix|>漢é­m'ſ㋿ \n㋿0\u000b'refi", "tokens": 71, "pieces": ["<ḋ", "̣漢", "ꟲ", "'T", "<('", "Reꟲع", "\r\n\r\n", "Dž", "Ⅳ", "'ll", "㍿!!", "Dž", " \n", "'T", "'D", "عfi", "<|", "fim", "_prefix", "|>", "漢é", "­m", "'ſ", "㋿", " \n", "㋿", "0", "\u000b", "'", "refi"]} +{"text": "0'Tt漢漢 <|endoftext|><|fim_prefix|><|fim_prefix|>'T\"m'S½\u000bt'ſḍ̇>ßſ'Tt>s éZ 12345678'Ma­'Re😀🏽", "tokens": 66, "pieces": ["0", "'T", "t漢漢", " ", "<|", "endoftext", "|><|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>'", "T", "\"m", "'S", "½", "\u000bt", "'ſ", "ḋ", "̣>", "ßſ", "'T", "t", ">s", " e", "́Z", " ", "123", "456", "78", "'M", "a", "­'", "Re", "😀🏽"]} +{"text": "'VE'll\r\n\r\n's'M're­'ss'D३ß㍿", "tokens": 17, "pieces": ["'VE", "'ll", "\r\n\r\n", "'s", "'M", "'re", "­'", "ss", "'D", "३", "ß", "㍿"]} +{"text": "​㍿㍿ 字३३Dž!12345678s'VEt漢­EOT", "tokens": 28, "pieces": ["​㍿㍿", " ", " 字", "३३", "Dž", "!", "123", "456", "78", "s", "'VE", "t漢", "­EOT"]} +{"text": "'S‍s😀🏽'Re ㋿émfi0", "tokens": 22, "pieces": ["'S", "‍s", "😀🏽'", "Re", " ", " ㋿", "e", "́mfi", "0"]} +{"text": "#$%dé'ſ́Z'MEOT.'VE>", "tokens": 15, "pieces": ["#$%", "de", "́'", "ſ", "́Z", "'M", "EOT", ".'", "VE", ">"]} +{"text": "t>Dž'Re aEOT­\r\n\r\n字9EOT-!(\u000b'Mm-", "tokens": 23, "pieces": ["t", ">Dž", "'Re", " aEOT", "­\r\n\r\n", "字", "9", "EOT", "-!(", "\u000b", "'M", "m", "-"]} +{"text": "字㍿ !('T \r9'll😀🏽'D EOT‍<|endoftext|>(Dž-'D​\r\nEOT", "tokens": 46, "pieces": ["字", "㍿", " ", "!('", "T", " \r", "9", "'ll", "😀🏽'", "D", "", " EOT", "‍<|", "endoftext", "|><", "EOT", ">(", "Dž", "-'", "D", "​\r\n", "EOT"]} +{"text": ">漢㋿d t😀🏽‍Dž \n'TsdEOT\"<|endoftext|>ع🙂\r👍🏽(ß9's,", "tokens": 44, "pieces": [">漢", "㋿d", " t", "😀🏽‍", "Dž", " \n", "'T", "sdEOT", "\"<|", "endoftext", "|>", "ع", "🙂\r", "👍🏽(", "ß", "9", "'s", ","]} +{"text": "㋿\u000b‍EOTåt字\r\né\r #$%ꟲ>㋿é'S 'm.𐞁", "tokens": 37, "pieces": ["㋿", "\u000b", "‍EOTa", "̊t字", "\r\n", "é", "\r", " ", " #$%", "ꟲ", ">㋿", "e", "́'", "S", " '", "m", ".𐞁"]} +{"text": "\r\n 'ſḍ̇​'Re,", "tokens": 16, "pieces": ["\r\n", " <", "META", "_START", ">'", "ſḋ", "̣​'", "Re", ","]} +{"text": "'ſ!
\"'S>'sſ'Re.
 ½<|fim_prefix|> \nd'<'ſé'll", "tokens": 38, "pieces": ["'ſ", "!", "
", "\"'", "S", ">'", "s", "ſ", "'Re", ".", "
", " ", "½", "<|", "fim", "_prefix", "|>", " \n", "d", "'<'", "ſe", "́'", "ll"]} +{"text": "(Ź\n<|fim_prefix|>́'VEA<|fim_prefix|>", "tokens": 22, "pieces": ["(Z", "́\n", "<|", "fim", "_prefix", "|>́'", "VEA", "<|", "fim", "_prefix", "|>"]} +{"text": "😀🏽'VEfiḍ̇\r\n\r\nßéſ#$%Am!'re½ 'Re­३\nⅣ (>'reعe\r漢", "tokens": 46, "pieces": ["😀🏽'", "VEfiḋ", "̣\r\n\r\n", "ßéſ", "#$%", "Am", "!'", "re", "½", " ", "'Re", "­", "३", "\n", "Ⅳ", " (>'", "reعe", "\r", "漢"]} +{"text": "'s㍿m'VE>ßß'ſ (Ⅳ‍'Dßß", "'", "ſ", " ", "(", "Ⅳ", "‍'", "D", "字", "tokens": 41, "pieces": ["३", "𐞁", "-", "\t", "…", "㍿Dž字", "!!👍🏽", "ꟲ", "'VE", "é", "#$%\r\n", "<'", "s", "<|", "endoftext", "|>", "字"]} +{"text": "㋿9e,🙂å\r\n­'M>.eꟲ.'st'ſ३0Z'M.", "tokens": 37, "pieces": ["㋿", "9", "e", ",🙂", "a", "̊<", "EOT", ">\r\n", "­'", "M", ">.", "eꟲ", ".'", "st", "'ſ", "३", "", "0", "Z", "'M", "."]} +{"text": " \n'Dž.'ssé
9́(#$%'M'S㍿Ⅳ漢éé's \n­\r\u000b'M😀🏽", "tokens": 49, "pieces": [" \n", "'Dž", ".'", "ssé", "
", "9", "́(#$%'", "M", "'S", "㍿", "Ⅳ", "漢ée", "́'", "s", " \n", "­\r", "", "\u000b", "'", "M", "😀🏽"]} +{"text": "ḍ̇é漢", "tokens": 8, "pieces": ["ḋ", "̣é漢"]} +{"text": "Z12345678\u000b('D,é>'<|endoftext|>'\r\n漢'sm<|endoftext|>Zt", "tokens": 32, "pieces": ["Z", "123", "456", "78", "\u000b", "('", "D", ",é", ">'<|", "endoftext", "|>'\r\n", "漢", "'s", "m", "<|", "endoftext", "|>", "Zt"]} +{"text": "!!'re'VE'll're ع'D \u000b9!!", "tokens": 14, "pieces": ["!!'", "re", "'VE", "'ll", "'re", " ع", "'D", " ", "\u000b", "9", "!!"]} +{"text": "
å å‍>👍🏽\r\n<|endoftext|>ݽe­ꟲDž-ḍ̇!!ꟲfi<|fim_prefix|>\t'VEs <é\r\n\r\n'll٣٤٥٦", "tokens": 70, "pieces": ["
a", "̊", " ", " a", "̊‍>👍🏽\r\n", "<|", "endoftext", "|>", "İ", "½", "e", "­ꟲDž", "-ḋ", "̣!!", "ꟲfi", "<|", "fim", "_prefix", "|>", "\t", "'VE", "s", " <", "é", "\r\n\r\n", "'ll", "٣٤٥", "٦"]} +{"text": "\r\n'VE>dd́字'Re'lle's\né  t३ 漢'ſ!!#$%漢३\n\"
'👍🏽😀🏽EOT", "tokens": 49, "pieces": ["\r\n", "'VE", ">dd", "́字", "'Re", "'ll", "e", "'s", "\n", "é", "  ", " t", "३", " 漢", "'ſ", "!!#$%", "漢", "३", "\n", "\"", "
", "'👍🏽😀🏽", "EOT"]} +{"text": "etd12345678 ", "tokens": 9, "pieces": ["etd", "123", "456", "78", "", " "]} +{"text": "\n½'reİİ\"\n 'M>'ſ's\u000b'VE㋿d 🙂a\nEOTé​", "tokens": 40, "pieces": ["\n", "½", "'re", "İİ", "\"\n", " '", "M", ">'", "ſ", "'s", "\u000b", "'VE", "㋿d", "", " ", "🙂a", "\n", "EOTe", "́<", "META", "_START", ">​"]} +{"text": ".\r\n\r\n٣٤٥٦", "tokens": 9, "pieces": [".\r\n\r\n", "٣٤٥", "٦"]} +{"text": "å字( å .s- m३9( ́!​<'Tm", "tokens": 42, "pieces": ["a", "̊字", "(", " a", "̊", " ", " .", "s", "-", " m", "३9", "(<", "EOT", ">", " ́!​<<", "META", "_START", "><", "EOT", ">'", "Tm"]} +{"text": "#$%ع\"\n", "tokens": 4, "pieces": ["#$%", "ع", "\"\n"]} +{"text": " \nſꟲ-👍🏽(mm#$%!!(\t
'D\" 
'Ré#$%>", "tokens": 42, "pieces": [" \n", "ſ", "ꟲ", "-👍🏽(", "mm", "#$%!!(", "\t", "
", "'D", "\"", " ", " <", "EOT", ">", "
", "'Re", "́#$%>"]} +{"text": "!! \nßåå", "tokens": 9, "pieces": ["!!", " \n", "ßa", "̊a", "̊"]} +{"text": "ꟲ'VE­́Ⅳte!-'M#$%", "tokens": 15, "pieces": ["ꟲ", "'VE", "­́", "Ⅳ", "te", "!-'", "M", "#$%"]} +{"text": "​㋿0\r'M漢é́Dždåe s \nå😀🏽'Resd \n३>'ll\u000b 'refi'lla'VE½\n", "tokens": 47, "pieces": ["​㋿", "0", "\r", "'M", "漢é", "́Džda", "̊e", " s", " \n", "a", "̊😀🏽'", "Resd", " \n", "३", ">'", "ll", "\u000b ", " '", "refi", "'ll", "a", "'VE", "½", "\n"]} +{"text": "e'Re9٣٤٥٦​٣٤٥٦'T", "tokens": 21, "pieces": ["e", "'Re", "9٣٤", "٥٦", "​", "٣٤٥", "٦", "'T"]} +{"text": "🙂m", "tokens": 3, "pieces": ["🙂m"]} +{"text": " 
́'ll字İ", "tokens": 8, "pieces": [" ", "
", "́'", "ll字İ"]} +{"text": "å'Mꟲ'S\t٣٤٥٦३…e­ ­å३é<|endoftext|>  \r\t.", "tokens": 45, "pieces": ["a", "̊'", "Mꟲ", "'S", "\t", "٣٤٥", "٦३", "…e", "­", " ", "­a", "̊", "३", "é", "<|", "endoftext", "|>", "  \r", "\t", "."]} +{"text": "٣٤٥٦t𐞁 \n\r\n!éDž>'T'S.'s0'ſ ", "tokens": 29, "pieces": ["٣٤٥", "٦", "t𐞁", " \n\r\n", "!éDž", ">'", "T", "'S", ".'", "s", "0", "'ſ", " "]} +{"text": "٣٤٥٦'reſ \n\r\n<Ⅳ \r३0
s½'VE \n#$%٣٤٥٦\u000b𐞁", "tokens": 52, "pieces": ["٣٤٥", "٦", "'re", "ſ", " \n\r\n", "<", "Ⅳ", " ", "\r", "३0", "
s", "½", "'VE", " \n", "#$%", "٣٤٥", "٦", "\u000b", "𐞁"]} +{"text": "' ḍ̇…d'é<|endoftext|>EOT \u000b㋿'Re'll­ \n😀🏽½𐞁<|endoftext|>#$%é9'VE9's(İ", "tokens": 59, "pieces": ["'", " ḋ", "̣", "…d", "'é", "<|", "endoftext", "|>", "EOT", " ", "\u000b", "㋿'", "Re", "'ll", "­", " \n", "😀🏽", "½", "𐞁", "<|", "endoftext", "|>#$%", "é", "9", "'", "VE", "9", "'s", "(İ"]} +{"text": "​'漢00
…㍿!😀🏽‍-'𐞁 😀🏽\r\n\r\n👍🏽A漢字<<𐞁m\r\n", "tokens": 53, "pieces": ["​'", "漢", "00", "
", "", "…", "㍿!😀🏽‍-'", "𐞁", " ", " 😀🏽\r\n\r\n", "👍🏽", "A漢字", "<<", "𐞁m", "\r\n"]} +{"text": "fi‍(", "tokens": 5, "pieces": ["fi", "‍("]} +{"text": "­!ꟲ字é㍿m  ㍿éAⅣEOTfi 'ſ𐞁a \n‍ſ é\r\n\r\n\"å'S", "tokens": 55, "pieces": ["­!", "ꟲ字e", "́㍿", "m", " ", " ", "㍿éA", "Ⅳ", "EOTfi", " ", "'ſ", "𐞁a", " \n", "‍<", "EOT", ">ſ", " é", "\r\n\r\n", "\"a", "̊'", "S"]} +{"text": "Z \n!!ß(<|endoftext|>EOT३ ḍ̇>åEOTß!字ꟲ½字.㍿!!é 'VE\t", "tokens": 48, "pieces": ["Z", " \n", "!!", "ß", "(<|", "endoftext", "|>", "EOT", "३", " ", "ḋ", "̣>", "a", "̊EOTß", "!字ꟲ", "½", "字", ".㍿!!", "é", " '", "VE", "\t"]} +{"text": "'s\t#$%‍‍  \n Dž ٣٤٥٦ 0.fiꟲ<\u000b'D…'re12345678ꟲ…㍿ع'T<|fim_prefix|>\tå­'T'D>!!'ſm", "tokens": 69, "pieces": ["'s", "\t", "#$%‍‍", "  \n", " Dž", " ", "٣٤٥", "٦", " ", "0", ".fiꟲ", "<", "\u000b", "'D", "…", "'re", "123", "456", "78", "ꟲ", "…", "㍿ع", "'T", "<|", "fim", "_prefix", "|>", "\ta", "̊­'", "T", "'D", ">!!'", "ſm"]} +{"text": " 'VEع'reeⅣ \n\r\n٣٤٥٦३ ‍ EOT-. 😀🏽'sa", "tokens": 37, "pieces": [" ", "'VE", "ع", "'re", "e", "Ⅳ", " \n", "\r\n", "٣٤٥", "٦३", " ‍", " ", " EOT", "-.", " ", "😀🏽'", "sa"]} +{"text": "㍿EOTſ<12345678𐞁\tßa .​t漢<|fim_prefix|><ع😀🏽\u000b\n
 字Z0Dž<|fim_prefix|>", "tokens": 54, "pieces": ["㍿EOTſ", "<", "123", "456", "78", "𐞁", "\tßa", " ", " .​", "t漢", "<|", "fim", "_prefix", "|><", "ع", "😀🏽", "\u000b\n", "
", " 字Z", "0", "Dž", "<|", "fim", "_prefix", "|>"]} +{"text": "\"'T'D'Re ́'sİ<|fim_prefix|>\r३\r\nꟲaßſ½́½'ll", "tokens": 35, "pieces": ["\"'", "T", "'D", "'Re", " ", " <", "META", "_START", ">́'", "sİ", "<|", "fim", "_prefix", "|>\r", "३", "\r\n", "ꟲaßſ", "½", "́", "½", "'ll"]} +{"text": "<'s\nİ३#$%İd ‍ſ字㋿㍿ſ-字> 😀🏽́ß\na", "tokens": 35, "pieces": ["<'", "s", "\n", "İ", "३", "#$%", "İd", " ‍", "ſ字", "㋿㍿", "ſ", "-字", ">", " ", " 😀🏽́", "ß", "\n", "a"]} +{"text": "字 ", "tokens": 2, "pieces": ["字", " "]} +{"text": "-'lltİ12345678ꟲd½İ0ßḍ̇t-< 👍🏽", "tokens": 33, "pieces": ["-'", "lltİ", "123", "456", "78", "ꟲd", "½", "İ", "0", "ß", "ḋ", "̣t", "-<", " ", " 👍🏽"]} +{"text": " ㋿\r<|fim_prefix|> 0é!!.㋿'ſḍ̇é'll 'T#$%\t\r\n\r\n12345678fi\"\"Džé'Reع<", "tokens": 51, "pieces": [" ", "㋿\r", "<|", "fim", "_prefix", "|>", " ", "0", "é", "!!.㋿'", "ſḋ", "̣e", "́'", "ll", " ", " '", "T", "#$%", "\t\r\n\r\n", "123", "456", "78", "fi", "\"\"", "Dže", "́'", "Reع", "<"]} +{"text": "'llå,\r­ \u000b㍿\r\n\r\n\r'👍🏽\r\n\r\né", "tokens": 24, "pieces": ["'ll", "a", "̊,\r", "­", " ", "\u000b", "㍿\r\n\r\n\r", "'👍🏽\r\n\r\n", "e", "́"]} +{"text": "'ſ", "tokens": 3, "pieces": ["'ſ"]} +{"text": " 're'éé㍿ḍ̇DžZ(m😀🏽
عé\rⅣ'M㋿ḍ̇İé½\u000b", "tokens": 44, "pieces": [" '", "re", "'e", "́é", "㍿ḋ", "̣DžZ", "(m", "😀🏽", "
عe", "́\r", "Ⅳ", "'M", "㋿ḋ", "̣İe", "́", "½", "\u000b"]} +{"text": "'ſⅣt𐞁­\n👍🏽🙂a🙂'M<|fim_prefix|>字Dž𐞁ع́㋿", "Ⅳ", "t𐞁", "­\n", "👍🏽🙂", "a", "🙂'", "M", "<|", "fim", "_prefix", "|>", "字Dž𐞁ع", "́㋿<", "e", "́漢", "…", ",e", "(𐞁"]} +{"text": "​.'ś‍­.'ll9m٣٤٥٦ßⅣ'ſḍ̇'Re\tmß>𐞁#$%Ⅳ(\t !ꟲAZ 0d0", "tokens": 56, "pieces": ["​.'", "s", "́‍­.'", "ll", "9", "m", "٣٤٥", "٦", "ß", "Ⅳ", "'ſ", "ḋ", "̣'", "Re", "\tmß", ">𐞁", "#$%", "Ⅳ", "(", "\t ", " !", "ꟲAZ", " ", "0", "d", "0"]} +{"text": "́!'DZ Dž<|fim_prefix|> \n'Re", "tokens": 16, "pieces": ["́!'", "DZ", " Dž", "<|", "fim", "_prefix", "|>", " \n", "'Re"]} +{"text": "Ⅳ\r🙂'M9-́ \n!!İ­\t!字'D<|fim_prefix|>ꟲe'VE \n's😀🏽mعꟲet 👍🏽'D字🙂'Re", "tokens": 61, "pieces": ["Ⅳ", "\r", "🙂'", "M", "9", "-́", " \n", "!!", "İ", "­", "\t", "!字", "'D", "<|", "fim", "_prefix", "|>", "ꟲe", "'VE", " \n", "'s", "😀🏽", "mعꟲet", " ", "👍🏽'", "D字", "🙂<", "EOT", ">'", "Re"]} +{"text": " ,> Z٣٤٥٦(‍'s…\r'Re>…é\u000bİꟲ'S'Re!\ndḍ̇'DéZ字 ſ'Re0ع'S \u000bİ", "tokens": 56, "pieces": [" ", ",>", " ", " Z", "٣٤٥", "٦", "(‍'", "s", "…\r", "'Re", ">", "…é", "\u000bİꟲ", "'", "S", "'Re", "!\n", "dḋ", "̣'", "DéZ字", " ſ", "'Re", "0", "ع", "'S", " ", "\u000bİ"]} +{"text": "'MZ‍!!'S३ß<|fim_prefix|>(<|fim_prefix|>e'red٣٤٥٦'VE \n'VEe0👍🏽 ", "tokens": 51, "pieces": ["'M", "Z", "‍!!'", "S", "३", "ß", "<|", "fim", "_prefix", "|>(<|", "fim", "_prefix", "|>", "e", "'re", "d", "٣٤٥", "٦", "'VE", " \n", "'VE", "e", "0", "👍🏽", " "]} +{"text": "#$%‍ å𐞁é㍿ <EOTd'ſ12345678\r!漢­'Tfi\n.'Sm#$%", "tokens": 49, "pieces": ["#$%‍", " a", "̊𐞁é", "㍿", " ", "<<", "META", "_START", ">EOTd", "'ſ", "123", "456", "78", "\r", "!漢", "<", "META", "_START", ">­'", "Tfi", "\n", ".'", "Sm", "#$%"]} +{"text": ",漢½m'T३👍🏽Dž\r\n
́-s\r\n\r\n<|fim_prefix|>a३éİ­🙂عfi­ſ'Mfi٣٤٥٦\n", "tokens": 56, "pieces": [",漢", "½", "m", "'T", "३", "👍🏽", "Dž", "\r\n", "
", "́-", "s", "\r\n\r\n", "<|", "fim", "_prefix", "|>", "a", "३", "éİ", "­🙂", "عfi", "­ſ", "'M", "fi", "٣٤٥", "٦", "\n"]} +{"text": "…#$%", "tokens": 4, "pieces": ["…", "#$%"]} +{"text": "<|fim_prefix|>", "tokens": 7, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "👍🏽fi \nes'T é0å😀🏽ع\t12345678İd́A­ḍ̇<|fim_prefix|>\rAß! ", "tokens": 55, "pieces": ["👍🏽", "fi", " \n", "es", "'T", " e", "́", "0", "a", "̊😀🏽", "ع", "\t", "", "123", "456", "78", "İd", "́A", "­ḋ", "̣<|", "fim", "_prefix", "|>\r", "Aß", "!", " "]} +{"text": " ­<", "tokens": 3, "pieces": [" ", "­<"]} +{"text": "\"e🙂>ع-é12345678Džḍ̇.t👍🏽!३", "tokens": 29, "pieces": ["\"e", "🙂>", "ع", "-e", "́", "123", "456", "78", "Džḋ", "̣.", "t", "👍🏽!", "३"]} +{"text": "३\r\nå!\nZ㋿ſ'Re,𐞁'll(\n'D!'re'ſ 字m", "tokens": 33, "pieces": ["३", "\r\n", "a", "̊!\n", "Z", "㋿ſ", "'Re", ",𐞁", "'ll", "(\n", "'D", "!<", "META", "_START", ">'", "re", "'ſ", " ", " 字m"]} +{"text": "<|fim_prefix|>字𐞁'VEİs \nعⅣ'ReéfiⅣDž\u000bfi", "tokens": 35, "pieces": ["<|", "fim", "_prefix", "|>", "字", "𐞁", "'VE", "İs", " \n", "ع", "Ⅳ", "'Re", "e", "́fi", "Ⅳ", "Dž", "\u000bfi"]} +{"text": "m-<\r\n\r\n'ſ\u000b\"", "tokens": 9, "pieces": ["m", "-<\r\n\r\n", "'ſ", "\u000b", "\""]} +{"text": "'ſ're㍿m'S'Reꟲ\r
🙂.12345678sfi,'<|fim_prefix|>9ꟲ\u000b漢\nt>字\r\nsſ'Red'TmZEOT\t'T", "tokens": 63, "pieces": ["'ſ", "'re", "㍿m", "'S", "'Re", "ꟲ", "\r", "
", "🙂.", "123", "456", "78", "sfi", ",'<|", "fim", "_prefix", "|>", "9", "ꟲ", "\u000b漢", "\n", "t", ">字", "\r\n", "sſ", "'Re", "d", "'T", "m", "ZEOT", "\t", "'T"]} +{"text": ", ", "tokens": 2, "pieces": [",", " "]} +{"text": "EOT🙂!EOT're,ß'Ms\r\n9\r\n\r\nZ#$%\"'sEOT'ſ \n'll😀🏽㍿'M- ½'ſ!a'D'll0\n", "tokens": 50, "pieces": ["EOT", "🙂!", "EOT", "'re", ",ß", "'M", "s", "\r\n", "9", "\r\n\r\n", "Z", "#$%\"'", "sEOT", "'ſ", " \n", "'ll", "😀🏽㍿'", "M", "-", " ", " ", "½", "'ſ", "!a", "'D", "'ll", "0", "\n"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " \n Ⅳ0e 🙂", "tokens": 11, "pieces": [" \n", " ", "Ⅳ", "", "0", "e", " 🙂"]} +{"text": "d<|endoftext|>'sİ!'VEß,́Dž!tⅣ'M<ꟲ字 \n\u000b!!㍿٣٤٥٦ !!<\rDž'T''VE<|endoftext|>(漢㋿", "tokens": 65, "pieces": ["d", "<|", "endoftext", "|>'", "sİ", "!'", "VEß", ",́", "Dž", "!t", "Ⅳ", "'M", "<ꟲ字", " \n", "\u000b", "!!㍿", "٣٤٥", "٦", " ", "!!<\r", "Dž", "'T", "''", "VE", "<|", "endoftext", "|>(", "漢", "㋿"]} +{"text": "#$%\r\n​aé  !३'é\r­\rZ字!!é0٣٤٥٦'Re'M ", "tokens": 35, "pieces": ["#$%\r\n", "​aé", "", "  ", " !", "३", "'e", "́\r", "­\r", "Z字", "!!", "e", "́", "0٣٤", "٥٦", "'Re", "'M", " "]} +{"text": "e!EOTع­<|endoftext|>0t½½ع́'ſ", "tokens": 25, "pieces": ["e", "!EOTع", "­<|", "endoftext", "|>", "0", "t", "½½", "ع", "́'", "ſ"]} +{"text": "éA㍿ḍ̇e'VEDž\r 12345678½é're'Re0\t\r\n\r\ń½!!\r\n\r\n👍🏽'Dfi'S<|endoftext|>s\r\n-ꟲa<|fim_prefix|>'ſ 'res", "tokens": 73, "pieces": ["e", "́A", "㍿ḋ", "̣e", "'VE", "Dž", "\r", " ", "123", "456", "78½", "é", "'re", "'Re", "0", "\t\r\n\r\n", "́", "½", "!!\r\n\r\n", "👍🏽'", "Dfi", "'S", "<|", "endoftext", "|>", "s", "\r\n", "-ꟲa", "<|", "fim", "_prefix", "|>'", "ſ", "", " ", "'re", "s"]} +{"text": "\"‍٣٤٥٦é0!!ꟲA", "tokens": 20, "pieces": ["\"‍", "٣٤٥", "٦", "e", "́", "0", "!!", "ꟲA"]} +{"text": "és", "tokens": 3, "pieces": ["e", "́s"]} +{"text": "Ⅳع­𐞁#$%\t😀🏽fi ꟲfi'ſ!!t'D'MZꟲ", "tokens": 46, "pieces": ["Ⅳ", "ع", "­<", "META", "_START", ">𐞁", "#$%", "\t", "😀🏽", "fi", " ꟲfi", "'ſ", "!!", "t", "'", "D", "'M", "Zꟲ"]} +{"text": "🙂DžA9fi'D𐞁EOT.0‍<漢,'M>३㍿‍漢'fi>'S", "tokens": 47, "pieces": ["🙂DžA", "9", "<", "META", "_START", ">fi", "'D", "𐞁EOT", ".", "0", "‍<", "漢", ",'", "M", ">", "३", "㍿‍", "漢", "'fi", ">'", "S"]} +{"text": "​ <|endoftext|>字(​ 'ſEOT'Mß‍㍿<|endoftext|>字ſ 'sꟲ12345678Ⅳ'Dꟲ३'reⅣs'll'll\n ㋿A a
㍿३ \n", "tokens": 72, "pieces": ["​", " ", " <|", "endoftext", "|>", "字", "(​", " '", "ſEOT", "'M", "ß", "‍㍿<|", "endoftext", "|>", "字ſ", " ", "'s", "ꟲ", "123", "456", "78Ⅳ", "'D", "ꟲ", "३", "'re", "Ⅳ", "s", "'ll", "'ll", "\n", " ㋿", "A", " a", "
", "㍿", "३", " \n"]} +{"text": "𐞁's㍿dſ'ſ 👍🏽\t#$%Ⅳſ<
🙂\"m‍🙂ſ", "tokens": 41, "pieces": ["𐞁", "'s", "㍿dſ", "'ſ", " ", "👍🏽", "\t", "#$%", "Ⅳ", "ſ", "<", "
", "🙂\"", "m", "‍🙂", "ſ"]} +{"text": "\r", "tokens": 1, "pieces": ["\r"]} +{"text": "m\"'T<  👍🏽'T ꟲ字㋿é! #$%\u000b!!٣٤٥٦漢'M", "tokens": 43, "pieces": ["m", "\"'", "T", "<<", "META", "_START", ">", " ", " ", "👍🏽'", "T", " ꟲ字", "㋿e", "́!", " ", " #$%", "\u000b", "!!", "٣٤٥", "٦", "漢", "'M"]} +{"text": "<'så漢'M'ſ🙂٣٤٥٦\nfi😀🏽d #$%(", "tokens": 32, "pieces": ["<'", "sa", "̊漢", "'M", "'ſ", "🙂", "٣٤٥", "٦", "\n", "fi", "😀🏽", "d", " ", " #$%("]} +{"text": "d'S
 ", "tokens": 5, "pieces": ["d", "'S", "
 "]} +{"text": " EOT́😀🏽­d\tA123456780.'VE😀🏽éåé9<|endoftext|>­ß'S'EOTDž'llßİ", "tokens": 50, "pieces": [" EOT", "́😀🏽­", "d", "\tA", "123", "456", "780", ".'", "VE", "😀🏽", "é", "a", "̊e", "́", "9", "<|", "endoftext", "|>­", "ß", "'S", "'EOTDž", "'ll", "ßİ"]} +{"text": "👍🏽İ<|fim_prefix|>", "tokens": 14, "pieces": ["👍🏽", "İ", "<|", "fim", "_prefix", "|>"]} +{"text": "<|endoftext|>\r\n-d\r\n\r\n<|endoftext|> \n'D>½…\t🙂", "tokens": 25, "pieces": ["<|", "endoftext", "|>\r\n", "-d", "\r\n\r\n", "<|", "endoftext", "|>", " \n", "'D", ">", "½", "…", "\t", "🙂"]} +{"text": "ع'reå", "tokens": 5, "pieces": ["ع", "'re", "a", "̊"]} +{"text": "EOT'll're åt\r\nZ 9<|fim_prefix|>'M…>­㍿!!fiعEOT\u000b'Re", "tokens": 36, "pieces": ["EOT", "'ll", "'re", " a", "̊t", "\r\n", "Z", " ", "9", "<|", "fim", "_prefix", "|>'", "M", "…", ">­㍿!!", "fiعEOT", "\u000b", "'Re"]} +{"text": "㋿ß漢…Z㍿\"ß<#$%
\r\n\tZEOTZ9́>AZ'M­'Re \r é字!!é", "tokens": 46, "pieces": ["㋿ß漢", "…Z", "㍿\"", "ß", "<#$%", "
\r\n", "\tZEOTZ", "9", "́><", "EOT", ">AZ", "'M", "­'", "Re", " \r", " é字", "!!", "é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿'ſ'VE'VE'Re'Mḍ̇ ­​\u000b(ḍ̇​'Re㍿㋿ \u000ba'reé\r㍿\n,12345678 𐞁", "tokens": 64, "pieces": ["㍿'", "ſ", "'VE", "'VE", "'Re", "'", "Mḋ", "̣", " ", "­​", "\u000b", "(<", "EOT", ">ḋ", "̣​'", "Re", "㍿㋿", " ", "\u000ba", "'re", "e", "́\r", "㍿\n", ",", "123", "456", "78", " ", " 𐞁"]} +{"text": "​
字>㍿😀🏽\"'D'reéDž 're‍é'Re\"'T٣٤٥٦12345678٣٤٥٦Z㍿😀🏽\"'", "D", "'re", "éDž", " ", " '", "re", "‍e", "́'", "Re", "\"'", "T", "٣٤٥", "٦12", "345", "678", "٣٤٥", "٦", "Z", "'ſ s漢́\t>'Re<­", "tokens": 24, "pieces": ["ß", "#$%🙂", "३", "a", "'VE", "\r\n", ">'", "ſ", " s漢", "́", "\t", ">'", "Re", "<­"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "é\rDž12345678 'll\tḍ̇'T\"!'llعs'll'll ㋿", "tokens": 28, "pieces": ["é", "\r", "Dž", "123", "456", "78", " ", "'ll", "\tḋ", "̣'", "T", "\"!'", "llعs", "'ll", "'ll", " ", "㋿"]} +{"text": ">A#$%åEOT's字ß‍ \t9'ſ>'M 'S'Res<|endoftext|><|endoftext|>İ㍿é9Džé's漢'", "tokens": 59, "pieces": [">A", "#$%", "a", "̊EOT", "'s", "字ß", "‍", " ", "\t", "9", "'ſ", ">'", "M", " ", " <", "META", "_START", ">'", "S", "'Re", "s", "<|", "endoftext", "|><|", "endoftext", "|>", "İ", "㍿é", "9", "Dže", "́'", "s漢", "'"]} +{"text": "Ⅳ!!0\u000bé\r\n\r\nA!!Džꟲ'9…9'D㍿\n0Z12345678'M#$%
\r\nfi'reß字", "tokens": 62, "pieces": ["Ⅳ", "!!", "0", "\u000be", "́\r\n\r\n", "A", "!!", "Džꟲ", "'", "9", "…", "9", "'D", "㍿\n", "", "0", "Z", "123", "456", "78", "'M", "#$%", "
\r\n", "fi", "'re", "ß", "字"]} +{"text": "عſ३es9​㍿\r\n'D\r\n\r\n", "tokens": 24, "pieces": ["عſ", "३", "es", "9", "​㍿\r\n", "'D", "\r\n\r\n"]} +{"text": "éDž.­ …>…\r\n­", "tokens": 14, "pieces": ["e", "́Dž", ".­", " ", "…", ">", "…\r\n", "­"]} +{"text": "​㋿'T'D㋿㍿>字 ㍿", "tokens": 19, "pieces": ["​㋿'", "T", "'D", "㋿㍿>", "字", " ", "㍿"]} +{"text": "\r!!'re's‍!m(字­
\téع́d😀🏽EOTaDž\"e'VE\rå0\nꟲA漢>\n9… 'Déعع", "tokens": 56, "pieces": ["\r", "!!'", "re", "'s", "‍!", "m", "(字", "­", "
", "\téع", "́d", "😀🏽", "EOTaDž", "\"e", "'VE", "\r", "a", "̊", "0", "\n", "ꟲA漢", ">\n", "9", "…", " ", "'D", "e", "́عع"]} +{"text": "A\r​a\t🙂 'Dع​t𐞁'llfiİ", "tokens": 21, "pieces": ["A", "\r", "​a", "\t", "🙂", " ", "'D", "ع", "​t𐞁", "'ll", "fiİ"]} +{"text": "EOT😀🏽<|endoftext|>‍\r\n\"Džꟲ'\rd.<9\r\r\n\r\n!\n…9 ٣٤٥٦", "tokens": 46, "pieces": ["EOT", "😀🏽<|", "endoftext", "|>‍\r\n", "\"Dž", "ꟲ", "'\r", "d", ".<", "9", "\r\r\n\r\n", "!\n", "…", "9", " ", "٣٤٥", "٦"]} +{"text": "'re 👍🏽字\r!!<|endoftext|>sⅣ'T( Z\r\n😀🏽٣٤٥٦0🙂👍🏽'M<|endoftext|>\r漢ḍ̇\r\n\r\n İ<9Ⅳ'Ds", "tokens": 74, "pieces": ["'re", " ", "👍🏽", "字", "\r", "!!<|", "endoftext", "|>", "s", "Ⅳ", "'T", "(", " Z", "\r\n", "😀🏽", "٣٤٥", "٦0", "🙂👍🏽'", "M", "<|", "endoftext", "|>\r", "漢ḋ", "̣\r\n\r\n", " İ", "<", "9Ⅳ", "'D", "s"]} +{"text": "\u000b\tſ漢é", "tokens": 8, "pieces": ["\u000b", "\tſ漢e", "́"]} +{"text": "Dž#$% t'Re‍ ꟲ", "tokens": 11, "pieces": ["Dž", "#$%", " t", "'Re", "‍", " ꟲ"]} +{"text": "Zꟲ½e#$%e", "tokens": 9, "pieces": ["Zꟲ", "½", "e", "#$%", "e"]} +{"text": " …‍​'s'ſ🙂'll <|fim_prefix|>ع\t!é٣٤٥٦ḍ̇e…漢Dž", "tokens": 69, "pieces": [" ", "…", "‍​'", "s", "'ſ", "🙂'", "ll", " <|", "fim", "_prefix", "|>", "ع", "", "\t", "!e", "́", "٣٤٥", "٦", "ḋ", "̣e", "…漢Dž"]} +{"text": "'T<|fim_prefix|>12345678'\r\n
á İ\" İ 's", "tokens": 23, "pieces": ["'T", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'\r\n", "
a", "́", " İ", "\"", " ", " İ", " ", "'s"]} +{"text": "'s\t9EOT'ſ<😀🏽​>0漢sſ ꟲ'VEǻ‍", "tokens": 33, "pieces": ["'s", "\t", "9", "EOT", "'ſ", "<😀🏽​>", "0", "漢sſ", " ꟲ", "'VE", "a", "̊́‍"]} +{"text": ">'M", "tokens": 5, "pieces": ["><", "META", "_START", ">'", "M"]} +{"text": ".!<|endoftext|> ३#$%", "tokens": 13, "pieces": [".!<|", "endoftext", "|>", " ", "३", "#$%"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'T​å३t𐞁å½­​0a", "tokens": 20, "pieces": ["'T", "​a", "̊", "३", "t𐞁a", "̊", "½", "­​", "0", "a"]} +{"text": "‍", "tokens": 2, "pieces": ["‍"]} +{"text": "\u000bétA\r\nå \n​0'T EOT9\r\n\r\n \nt\"'Sa", "tokens": 23, "pieces": ["\u000be", "́tA", "\r\n", "a", "̊", " \n", "​", "0", "'T", " EOT", "9", "\r\n\r\n \n", "t", "\"'", "Sa"]} +{"text": "(-‍
'S­Z漢Z🙂12345678 'VEs'Re漢👍🏽​>  ​́s😀🏽Ⅳ'D
Z'Z", "tokens": 59, "pieces": ["(-‍", "
", "'", "S", "­<", "META", "_START", ">Z漢Z", "🙂<", "EOT", ">", "123", "456", "78", " ", "'VE", "s", "'Re", "漢", "👍🏽​>", " ", " ​́", "s", "😀🏽", "Ⅳ", "'D", "
Z", "'Z"]} +{"text": "…fi'll's #$%㋿std#$%.Ⅳ😀🏽👍🏽'T㍿​ß'M", "tokens": 47, "pieces": ["…fi", "'ll", "'s", " ", "#$%㋿", "s", "td", "#$%.", "Ⅳ", "😀🏽👍🏽'", "T", "㍿​", "ß", "'M"]} +{"text": "㍿é0\r\n\r\n㋿d '…12345678漢㋿Z ſ 'Re ( 'reDž\n\r\n\r\n", "tokens": 42, "pieces": ["㍿", "e", "́", "0", "\r\n\r\n", "㋿d", " ", "'", "…", "123", "456", "78", "漢", "㋿Z", " ſ", " ", " '", "Re", " ", " (", " ", " '", "reDž", "\n\r\n\r\n"]} +{"text": "'llDž12345678aé<'T-'S ३eEOTſ\r'ſ\u000bꟲEOT‍ .DžⅣZ!!-'", "S", " ", "३", "eEOTſ", "\r", "'ſ", "\u000bꟲEOT", "‍", " .", "Dž", "Ⅳ", "Z", "!!<", "EOTtع", "'D", " e", "́,", "a", "̊́", "A", "३"]} +{"text": "#$%‍½İ'T㋿‍漢e ", "tokens": 16, "pieces": ["#$%‍", "½", "İ", "'T", "㋿‍", "漢e", " "]} +{"text": "'VE!Z३fi३9Dž 'T
'T>‍'Re'Re ,.㋿'s#$%'T\r\n㍿9½<|endoftext|>'lle \n漢 字​", "tokens": 54, "pieces": ["'VE", "!Z", "३", "fi", "३9", "Dž", " ", "'T", "
", "'T", ">‍'", "Re", "'Re", " ", ",.㋿'", "s", "#$%'", "T", "\r\n", "㍿", "9½", "<|", "endoftext", "|>'", "lle", " \n", "漢", " 字", "​"]} +{"text": "é👍🏽ꟲ𐞁 漢Z'ſſ'll!!<|fim_prefix|>\r\nſ­ \n­\r\n\r\n'll٣٤٥٦\tfi12345678ḍ̇é \n\"12345678İt'Sḍ̇🙂-", "tokens": 75, "pieces": ["é", "👍🏽", "ꟲ𐞁", " 漢Z", "'ſ", "ſ", "'ll", "!!<|", "fim", "_prefix", "|>\r\n", "ſ", "­", " \n", "­\r\n\r\n", "'ll", "٣٤٥", "٦", "\tfi", "123", "456", "78", "ḋ", "̣é", " \n", "\"", "123", "456", "78", "İt", "'S", "ḋ", "̣🙂-"]} +{"text": "-㍿́#$%ꟲ-㍿\r\n\r\n\r\n'D 
‍ꟲ ꟲعé#$%
​🙂ßع👍🏽'ſ'D​<|endoftext|>!'ll\r\n\r\nDž½ßd", "tokens": 65, "pieces": ["-㍿́#$%", "ꟲ", "-㍿\r\n\r\n\r\n", "'D", " ", "
", "‍ꟲ", " ꟲعe", "́#$%", "
", "​🙂", "ßع", "👍🏽'", "ſ", "'D", "​<|", "endoftext", "|>!'", "ll", "\r\n\r\n", "Dž", "½", "ßd"]} +{"text": "'sſ 's
…ع-٣٤٥٦>e! ٣٤٥٦
㍿<|endoftext|>
<fi😀🏽'VE>İé\n<ꟲ­ßſ🙂12345678'DZ ㋿ Ⅳ", "tokens": 72, "pieces": ["\n\n", ">-", "٣٤٥", "٦", ">e", "!", " ", "٣٤٥", "٦", "
", "㍿<|", "endoftext", "|>", "
", "<fi", "😀🏽'", "VE", ">İe", "́\n", "<ꟲ", "­ßſ", "🙂", "123", "456", "78", "'D", "Z", " ", "㋿", " ", "Ⅳ"]} +{"text": "\td#$%#$%>\r\n,t\t0<|fim_prefix|>\nع३½\r\n", "tokens": 27, "pieces": ["\td", "#$%#$%><", "EOT", ">\r\n", ",t", "\t", "0", "<|", "fim", "_prefix", "|>\n", "ع", "", "३½", "\r\n"]} +{"text": "́-­'T'lld0.e'M,'Se\r\n\r\n\r\n!!㍿'reDž字", "tokens": 25, "pieces": ["́-­'", "T", "'ll", "d", "0", ".e", "'M", ",'", "Se", "\r\n", "\r\n\r\n", "!!㍿'", "reDž字"]} +{"text": ".'sEOT\r\n\r\nⅣå!s", "tokens": 12, "pieces": [".'", "sEOT", "\r\n\r\n", "Ⅳ", "a", "̊!", "s"]} +{"text": "३İ'll३漢", "tokens": 8, "pieces": ["३", "İ", "'ll", "३", "漢"]} +{"text": "Z \n‍A'VE\r\nméå\t㋿\té​‍\"é'ſt㋿!", "tokens": 34, "pieces": ["Z", " \n", "‍A", "'VE", "\r\n", "me", "́a", "̊", "\t", "㋿", "\te", "́​‍\"", "e", "́'", "ſt", "㋿!"]} +{"text": "fi'ſ\r\n\r\n!!. \"ع", "tokens": 10, "pieces": ["fi", "'ſ", "\r\n\r\n", "!!.", " ", " \"", "ع"]} +{"text": "𐞁ꟲ're\n<\"İém­'re", "tokens": 17, "pieces": ["𐞁ꟲ", "'re", "\n", "<\"", "İe", "́m", "­'", "re"]} +{"text": "\r\nå0​👍🏽A're'TEOT", "tokens": 18, "pieces": ["\r\n", "a", "̊", "0", "​👍🏽", "A", "'re", "'T", "EOT"]} +{"text": "\n>'Sfí㋿ed
'Re<|fim_prefix|>aZ<|fim_prefix|>漢
", "tokens": 33, "pieces": ["\n", ">'", "Sfi", "́㋿", "ed", "
", "'Re", "<|", "fim", "_prefix", "|>", "aZ", "<|", "fim", "_prefix", "|>", "漢", "
"]} +{"text": "123456789
ꟲ­<|fim_prefix|>㋿٣٤٥٦­'ſ'S!ß'Smḍ̇ḍ̇", "tokens": 46, "pieces": ["123", "456", "789", "
ꟲ", "­<|", "fim", "_prefix", "|>㋿", "٣٤٥", "٦", "­'", "ſ", "'S", "!ß", "'S", "mḋ", "̣ḋ", "̣"]} +{"text": "३", "tokens": 2, "pieces": ["३"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿​٣٤٥٦12345678'Dع🙂9ſ😀🏽ß㋿İ'Ret३Zع9İ'Reſ‍字‍!३​'s<|fim_prefix|><|fim_prefix|>𐞁12345678​\n'llm", "tokens": 82, "pieces": ["㍿​", "٣٤٥", "٦12", "345", "678", "'D", "ع", "🙂", "9", "ſ", "😀🏽", "ß", "㋿İ", "'Re", "t", "३", "Zع", "9", "İ", "'Re", "ſ", "‍字", "‍!", "३", "​'", "s", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", "𐞁", "123", "456", "78", "​\n", "'ll", "m"]} +{"text": "字as'ś'sA漢12345678å", "tokens": 16, "pieces": ["字as", "'s", "́'", "sA漢", "123", "456", "78", "a", "̊"]} +{"text": "<|endoftext|>s012345678<|endoftext|>0>\"\r\n\r\n㋿İ𐞁.漢,'Ta'Dd\n \n㋿'S½漢\r\n \n漢", "tokens": 49, "pieces": ["<|", "endoftext", "|>", "s", "012", "345", "678", "<|", "endoftext", "|>", "0", ">\"\r\n\r\n", "㋿İ𐞁", ".漢", ",'", "Ta", "'D", "d", "\n \n", "㋿'", "S", "½", "漢", "\r\n \n", "漢"]} +{"text": " ㋿३ßda'T!!", "tokens": 10, "pieces": [" ㋿", "३", "ßda", "'T", "!!"]} +{"text": "<Dž 𐞁㋿\u000bع‍", "tokens": 15, "pieces": ["<Dž", " 𐞁", "㋿", "\u000bع", "‍"]} +{"text": "🙂e٣٤٥٦'llZ<|fim_prefix|>'T", "tokens": 21, "pieces": ["🙂e", "٣٤٥", "٦", "'ll", "Z", "<|", "fim", "_prefix", "|>'", "T"]} +{"text": "m,'\"e字㋿字\u000b'll\u000b0å\r'M9ds𐞁A\r\n 're12345678ḍ̇éaEOT\r½dZ\r\n\r\n \n>e,9é", "tokens": 52, "pieces": ["m", ",'\"", "e字", "㋿字", "\u000b", "'ll", "\u000b", "0", "a", "̊\r", "'M", "9", "ds𐞁A", "\r\n", " ", " '", "re", "123", "456", "78", "ḋ", "̣éaEOT", "\r", "½", "dZ", "\r\n\r\n \n", ">e", ",", "9", "e", "́"]} +{"text": "12345678'M😀🏽.٣٤٥٦'T'Re३e(<㍿åd('T'll\r\n\r\n'VE㋿🙂m​,ḍ̇ \n#$%ß \n‍<|endoftext|>ß'T½'
Dž#$%", "tokens": 78, "pieces": ["123", "456", "78", "'M", "😀🏽.", "٣٤٥", "٦", "'T", "'Re", "३", "e", "(<㍿", "a", "̊d", "('", "T", "'", "ll", "\r\n\r\n", "'VE", "㋿🙂", "m", "​,", "ḋ", "̣", " \n", "#$%", "ß", " \n", "‍<|", "endoftext", "|>", "ß", "'T", "½", "'", "
Dž", "#$%"]} +{"text": "a\r\n\r\nEOT \nḍ̇>12345678's<|fim_prefix|>'ſ.EOT12345678'Rema'D𐞁'T<|endoftext|>fi.", "tokens": 47, "pieces": ["a", "\r\n\r\n", "EOT", " \n", "ḋ", "̣>", "123", "456", "78", "'s", "<|", "fim", "_prefix", "|>'", "ſ", ".EOT", "123", "456", "78", "'Re", "ma", "'D", "𐞁", "'T", "<|", "endoftext", "|>", "fi", "."]} +{"text": "!!!!fi漢İAe.ḍ̇t漢㍿​!!'reda'M.'ſ‍é> 'VÉ", "tokens": 47, "pieces": ["!!<", "EOT", ">!!", "fi漢İ", "Ae", ".ḋ", "̣t漢", "㍿​!!'", "reda", "'M", ".'", "ſ", "‍e", "́>", " ", " '", "VE", "́"]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "­>\"ådſ  ", "tokens": 10, "pieces": ["­>\"", "a", "̊dſ", "  "]} +{"text": "'D'D 'Re12345678'Re'T ZEOT ع㍿\t's㍿­ſ, ", "tokens": 27, "pieces": ["'D", "'D", " ", " '", "Re", "123", "456", "78", "'Re", "'T", " ZEOT", " ع", "㍿", "\t", "'s", "㍿­", "ſ", ",", " "]} +{"text": "'Re'S,-👍🏽 'T
é #$%", "tokens": 19, "pieces": ["'Re", "'S", ",-👍🏽", " ", " '", "T", "
e", "́", " ", "#$%"]} +{"text": "😀🏽\u000b'llA\r\n\r\n\r\n<|endoftext|>(㍿.'s👍🏽½dDž\"", "tokens": 33, "pieces": ["😀🏽", "\u000b", "'ll", "A", "\r\n\r\n\r\n", "<|", "endoftext", "|>(㍿.'", "s", "👍🏽", "½", "dDž", "\""]} +{"text": "9'Re0İ0'M\n<|endoftext|>'T😀🏽٣٤٥٦0", "tokens": 29, "pieces": ["9", "'Re", "0", "İ", "0", "'M", "\n", "<|", "endoftext", "|>'", "T", "😀🏽", "٣٤٥", "٦0"]} +{"text": "<|endoftext|>s㋿\r\n\r\n👍🏽éad.'sſ'S", "tokens": 32, "pieces": ["<|", "endoftext", "|><", "EOT", ">s", "㋿\r\n\r\n", "👍🏽", "e", "́ad", ".'", "sſ", "'", "S"]} +{"text": "a \u000b'ſꟲfi<'VE'Re\r\n", "tokens": 18, "pieces": ["a", " ", "\u000b", "'ſ", "ꟲfi", "<'", "VE", "'Re", "\r\n"]} +{"text": "\r\n\r\n­Dž's'M'S<́'VE'Re\t '३'🙂é👍🏽\u000b<|endoftext|>Ⅳ३‍㋿'re", "tokens": 46, "pieces": ["\r\n\r\n", "­Dž", "'s", "'M", "'S", "<́'", "VE", "'Re", "\t", " ", "'", "३", "'🙂", "é", "👍🏽", "\u000b", "<|", "endoftext", "|>", "Ⅳ३", "‍㋿'", "re"]} +{"text": "é漢a t'Z\tع> #$%\tİ…'ReDž…٣٤٥٦'D!!é\nⅣmAt sZ😀🏽'Mfi'll", "tokens": 59, "pieces": ["e", "́漢a", " ", " t", "'Z", "\tع", ">", " ", " #$%", "\tİ", "…", "'Re", "Dž", "…", "٣٤٥", "٦", "'D", "!!", "e", "́\n", "Ⅳ", "mAt", " sZ", "😀🏽'", "Mfi", "'ll"]} +{"text": "٣٤٥٦ß'Re…!'re\tZmé字9fi!'TEOT -'ſ's(漢a", "tokens": 33, "pieces": ["٣٤٥", "٦", "ß", "'Re", "…", "!'", "re", "\tZme", "́字", "9", "fi", "!'", "TEOT", " -'", "ſ", "'s", "(漢a"]} +{"text": "<|endoftext|>m½!!‍0 <|fim_prefix|>-\r12345678'T'Zs12345678. 'sZ‍<|endoftext|>d'VEd漢'S'VEⅣ
\u000bé漢'Re \n", "tokens": 69, "pieces": ["<|", "endoftext", "|>", "m", "½", "!!‍", "0", " ", "<|", "fim", "_prefix", "|>-\r", "", "123", "456", "78", "'T", "'Zs", "123", "456", "78", ".", " ", "'s", "Z", "‍<|", "endoftext", "|>", "d", "'VE", "d漢", "'S", "'VE", "Ⅳ", "
", "\u000be", "́漢", "'Re", " \n"]} +{"text": "ſ‍A(ſ३\t'D \tⅣ's\r\"EOT.٣٤٥٦ \nİ \"", "tokens": 33, "pieces": ["ſ", "‍A", "(ſ", "३", "\t", "'D", " ", "\t", "Ⅳ", "'s", "\r", "\"EOT", ".", "٣٤٥", "٦", " \n", "İ", " \""]} +{"text": " (!'Re'Re<|fim_prefix|>\rع\u000bDž're ß‍'T99d!!åⅣ,‍'D!!㍿
a'll\nŹ'D ", "tokens": 57, "pieces": [" ", "(!'", "Re", "'", "Re", "<|", "fim", "_prefix", "|>\r", "ع", "\u000bDž", "'re", " ß", "‍'", "T", "99", "d", "!!", "a", "̊", "Ⅳ", ",‍'", "D", "!!㍿", "
a", "'ll", "\n", "Z", "́'", "D", " "]} +{"text": "'D're㋿0!", "tokens": 7, "pieces": ["'D", "'re", "㋿", "0", "!"]} +{"text": "Ⅳ'ſ‍<|fim_prefix|>9t'Re\rḍ̇🙂'D\r\n\r\nſ'Re字! ع'D\"… e'll\n!İ !<|fim_prefix|> tA\"fi", "tokens": 60, "pieces": ["Ⅳ", "'ſ", "‍<|", "fim", "_prefix", "|>", "9", "t", "'Re", "\r", "ḋ", "̣🙂'", "D", "\r\n\r\n", "ſ", "'Re", "字", "!", " ع", "'D", "\"", "…", " e", "'ll", "\n", "!İ", " !<|", "fim", "_prefix", "|>", " tA", "\"fi"]} +{"text": "🙂\u000bİ12345678\r\n12345678", "tokens": 11, "pieces": ["🙂", "\u000bİ", "123", "456", "78", "\r\n", "123", "456", "78"]} +{"text": "åe\u000b0m <\t'T㍿…'VE𐞁e9<|fim_prefix|>'sſ!!\n\r\n㋿Ⅳ'D <0é", "tokens": 53, "pieces": ["a", "̊e", "\u000b", "0", "m", " <", "\t", "'T", "㍿", "…", "'VE", "𐞁e", "9", "<|", "fim", "_prefix", "|>'", "sſ", "!!<", "EOT", ">\n\r\n", "㋿", "Ⅳ", "'D", " ", "<", "0", "e", "́"]} +{"text": "'ll'Ddm>㋿'İ're'S-'S>𐞁 …A", "tokens": 26, "pieces": ["'ll", "'D", "dm", ">㋿'", "İ", "'re", "'S", "-'", "S", ">", "𐞁", " ", "…A"]} +{"text": "12345678'M\r,A 字'sm\r\n\r\n m>𐞁 \n عſe٣٤٥٦", "tokens": 37, "pieces": ["123", "456", "78", "'M", "\r", ",A", " ", " 字", "'s", "m", "\r\n\r\n", " m", ">𐞁", " \n", " عſe", "٣٤٥", "٦"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿ 'DZ. \n!\r\n\r\n\n-👍🏽'ſ\"'D\"㍿ \n'D,'ſ", "tokens": 34, "pieces": ["㍿", " ", "'", "DZ", ".", " \n", "!\r\n\r\n\n", "-👍🏽'", "ſ", "\"'", "D", "\"㍿", " \n", "'D", ",'", "ſ"]} +{"text": "ZaعeéEOT🙂(!!'T٣٤٥٦ع 0éⅣ\n'ſ<|endoftext|>'re👍🏽\r\n\r\nå.< 字", "tokens": 61, "pieces": ["Zaعee", "́EOT", "🙂(!!<", "META", "_START", ">'", "T", "٣٤٥", "٦", "ع", " ", "0", "e", "́", "Ⅳ", "\n", "'ſ", "<", "EOT", "><|", "endoftext", "|>'", "re", "👍🏽\r\n\r\n", "a", "̊.<", " ", " 字"]} +{"text": "İm12345678\r''12345678'T-t ß12345678
<|fim_prefix|>‍", "tokens": 31, "pieces": ["İm", "123", "456", "78", "\r", "''", "123", "456", "78", "'T", "-t", " ß", "123", "456", "78", "
", "<|", "fim", "_prefix", "|>‍"]} +{"text": "ḍ̇ea\" 字ع\r
👍🏽
👍🏽😀🏽åḍ̇३A\u000b<|endoftext|>'M­'TZ'ſ👍🏽(‍㋿३0", "tokens": 74, "pieces": ["ḋ", "̣ea", "\"", " 字ع", "\r", "
", "👍🏽", "
", "👍🏽😀🏽", "a", "̊ḋ", "̣", "३", "A", "\u000b", "<|", "endoftext", "|>'", "M", "­'", "TZ", "'ſ", "👍🏽(‍㋿", "३0"]} +{"text": "'D\n½½㍿\r\nꟲ́12345678 a\t#$%٣٤٥٦\n!d0'ſ12345678٣٤٥٦㍿ßé😀🏽.'", "tokens": 59, "pieces": ["'D", "\n", "½½", "㍿<", "META", "_START", ">\r\n", "ꟲ", "́", "123", "456", "78", " a", "\t", "#$%", "٣٤٥", "٦", "\n", "!d", "0", "'ſ", "123", "456", "78٣", "٤٥٦", "㍿ßé", "😀🏽.'"]} +{"text": "Dž>Aa㍿́<(ḍ̇İ#$%\r' \n e!!!!9<'T", "tokens": 31, "pieces": ["Dž", ">Aa", "㍿́<(", "ḋ", "̣İ", "#$%\r", "'<", "META", "_START", ">", " \n", " e", "!!!!", "9", "<'", "T"]} +{"text": "#$%!字\"👍🏽㋿a'Re\t​A​İ'M\rſ'T\u000bİ字<'D!!
", "tokens": 34, "pieces": ["#$%!", "字", "\"👍🏽㋿", "a", "'Re", "\t", "​A", "​İ", "'M", "\r", "ſ", "'T", "\u000bİ字", "<'", "D", "!!", "
"]} +{"text": "ꟲ're 😀🏽\"'", "tokens": 10, "pieces": ["ꟲ", "'re", " ", " 😀🏽\"'"]} +{"text": " …'s \nßEOTé", "tokens": 10, "pieces": [" ", "…", "'s", " \n", "ßEOTe", "́"]} +{"text": "́(", "tokens": 2, "pieces": ["́("]} +{"text": "\r\n\r\n
å<|fim_prefix|>EOT'Ss \r\nt…㋿'ll!!'VE're 0Zt👍🏽ع'D", "tokens": 43, "pieces": ["\r\n\r\n", "
a", "̊<|", "fim", "_prefix", "|>", "EOT", "'S", "s", " \r\n", "t", "…", "㋿'", "ll", "!!'", "VE", "'re", " ", "0", "Zt", "👍🏽", "ع", "'D"]} +{"text": "'Dß字9\r\n\r\n३…d'VEfi👍🏽Z…ſ(\tDže३", "tokens": 35, "pieces": ["'D", "ß字", "9", "\r\n\r\n", "३", "…d", "'VE", "fi", "👍🏽", "Z", "…ſ", "(", "\tDže", "", "३"]} +{"text": "\"ḍ̇ع\"!,!! ''T 12345678漢́\"漢 <|endoftext|>🙂'ſ's㋿👍🏽>‍", "tokens": 63, "pieces": ["\"ḋ", "̣ع", "\"!,!!", " ", "''", "T", " ", "123", "456", "78", "漢", "́\"", "漢", " <|", "endoftext", "|>🙂'", "ſ", "'s", "㋿👍🏽>‍"]} +{"text": "'\r\nع'Dé'<|endoftext|>'Re12345678\r\n\r\nfi#$%t\r\n\r\n字EOT<<|fim_prefix|> \n-'VE'Re​ \n", "tokens": 41, "pieces": ["'\r\n", "ع", "'D", "é", "'<|", "endoftext", "|>'", "Re", "123", "456", "78", "\r\n\r\n", "fi", "#$%", "t", "\r\n\r\n", "字EOT", "<<|", "fim", "_prefix", "|>", " \n", "-'", "VE", "'Re", "​", " \n", ""]} +{"text": "'ReEOT𐞁ḍ̇<|fim_prefix|>e½İ\u000b½å", "tokens": 27, "pieces": ["'Re", "EOT𐞁ḋ", "̣<|", "fim", "_prefix", "|>", "e", "½", "İ", "\u000b", "½", "a", "̊"]} +{"text": "ſe​", "tokens": 4, "pieces": ["ſe", "​"]} +{"text": "'Re😀🏽'\r😀🏽12345678\tß\r\n\r\n <|fim_prefix|>́ſ<|endoftext|>#$%('s\r\n\r\n,!!ꟲ𐞁👍🏽­́ß ", "tokens": 61, "pieces": ["'Re", "😀🏽'\r", "😀🏽", "123", "456", "78", "\tß", "\r\n\r\n", " ", "<|", "fim", "_prefix", "|>́", "ſ", "<|", "endoftext", "|>#$%('", "s", "\r\n\r\n", ",!!", "ꟲ𐞁", "👍🏽­́", "ß", " "]} +{"text": "'D👍🏽́'s'", "s", "#$%m", "tokens": 55, "pieces": ["\u000b", "#$%<", "d字", "́'", "ſß", "m"]} +{"text": "å 's \"<|fim_prefix|>12345678'll. \n - \u000bé½ß'Dſ\"!!ꟲm😀🏽'Rea12345678Ⅳ", "tokens": 48, "pieces": ["a", "̊", " ", " '", "s", " ", "\"<|", "fim", "_prefix", "|>", "123", "456", "78", "'ll", ".", " \n", " -", " ", "\u000bé", "½", "ß", "'D", "ſ", "\"!!", "ꟲm", "😀🏽'", "Rea", "123", "456", "78Ⅳ"]} +{"text": "'ḍ̇ ३'ReA३\r\n
s( .​9
-\n½…½­😀🏽'D(fi", "tokens": 42, "pieces": ["'ḋ", "̣", " ", " ", "३", "'Re", "A", "३", "\r\n", "
s", "(", " ", " .​", "9", "
", "-\n", "½", "…", "½", "­😀🏽'", "D", "(fi"]} +{"text": "'Mḍ̇s(sfiZ٣٤٥٦d'D‍漢\r\n\r\n!12345678३EOT'S0́㋿漢 ½", "tokens": 44, "pieces": ["'M", "ḋ", "̣s", "(sfiZ", "٣٤٥", "٦", "d", "'D", "‍漢", "\r\n\r\n", "!", "123", "456", "78३", "EOT", "'S", "0", "́㋿", "漢", " ", "½"]} +{"text": "漢! \r\ń😀🏽'😀🏽 'SZ\"t,ع'ſ𐞁<|endoftext|>Z#$%\"", "tokens": 42, "pieces": ["漢", "!", " \r\n", "́😀🏽'😀🏽", " <", "META", "_START", ">'", "SZ", "\"t", ",ع", "'ſ", "𐞁", "<|", "endoftext", "|>", "Z", "#$%\""]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ſ
s👍🏽㍿­å<|fim_prefix|>ßa", "tokens": 30, "pieces": ["ſ", "
s", "👍🏽㍿­", "a", "̊<|", "fim", "_prefix", "|>", "ß", "a"]} +{"text": "Ⅳſİ'S'll'D'llé-ß'S ​́­m", "tokens": 18, "pieces": ["Ⅳ", "ſİ", "'S", "'ll", "'D", "'ll", "e", "́-", "ß", "'S", " ​́­", "m"]} +{"text": "0EOT-३s𐞁 \nḍ̇'M\r\n㋿fi!𐞁a'D''se<|fim_prefix|>\"İ<|fim_prefix|>A­ḍ̇", "tokens": 57, "pieces": ["0", "EOT", "-", "३", "s𐞁", " \n", "ḋ", "̣'", "M", "\r\n", "㋿fi", "!𐞁a", "'D", "''", "se", "<|", "fim", "_prefix", "|>\"", "İ", "<|", "fim", "_prefix", "|>", "A", "­ḋ", "̣"]} +{"text": "­\neḍ̇fi\"s.'S漢  'VE😀🏽é\r\n\r\n!!s Z-ع(mé\r\ń's. t३!", "tokens": 44, "pieces": ["­\n", "eḋ", "̣fi", "\"s", ".'", "S漢", " ", " ", "'VE", "😀🏽", "é", "\r\n\r\n", "!!", "s", " ", " Z", "-ع", "(mé", "\r\n", "́'", "s", ".", " t", "३", "!"]} +{"text": "漢#$%9fiⅣ½㍿s😀🏽fi\r", "tokens": 22, "pieces": ["漢", "#$%", "9", "fi", "Ⅳ½", "㍿s", "😀🏽", "fi", "\r"]} +{"text": "<|endoftext|>\r\n\r\nꟲ\u000b'VE\n…Ⅳ𐞁 \n👍🏽9'll12345678​", "tokens": 41, "pieces": ["<|", "endoftext", "|>\r\n\r\n", "ꟲ", "\u000b", "'VE", "\n", "", "…", "Ⅳ", "𐞁", " \n", "👍🏽", "9", "'ll", "123", "456", "78", "​"]} +{"text": "'sDž'Re", "tokens": 7, "pieces": ["'s", "Dž", "'", "Re"]} +{"text": "🙂 ­'M d'ſ 😀🏽㍿eå<0'll\"'Dm", "tokens": 29, "pieces": ["🙂", " ", " ­'", "M", " d", "'ſ", " 😀🏽㍿", "e", "a", "̊<", "0", "'ll", "\"'", "Dm"]} +{"text": "\n…'Re'll👍🏽maé٣٤٥٦åİ'VE㋿🙂٣٤٥٦<|endoftext|>ée'Re\t\r\n!'Mfi
字🙂", "tokens": 74, "pieces": ["e", "́'", "ll", "'ll", "eİ", " <|", "fim", "_prefix", "|>", "…", "'Re", "'ll", "👍🏽", "mae", "́", "٣٤٥", "٦", "a", "̊İ", "'VE", "㋿🙂", "٣٤٥", "٦", "<|", "endoftext", "|>", "e", "́e", "'Re", "\t\r\n", "!'", "Mfi", "
字", "🙂"]} +{"text": "Džå0‍.'M\u000b'Se‍漢!!'T\u000bⅣ́", "tokens": 28, "pieces": ["Dža", "̊", "0", "‍.'", "M", "\u000b", "'S", "e", "‍漢", "!!'", "T", "\u000b", "Ⅳ", "́"]} +{"text": "Z\"", "tokens": 2, "pieces": ["Z", "\""]} +{"text": "aſ'ſ(e>'T Z12345678
<'Reé👍🏽s!!٣٤٥٦'D\r\ns👍🏽dDž\"ß'ع\rꟲa", "tokens": 62, "pieces": ["aſ", "'ſ", "(e", ">'", "T", " Z", "123", "456", "78", "
", "<'", "Reé", "👍🏽", "s", "!!", "٣٤٥", "٦", "'D", "\r\n", "s", "👍🏽", "d", "Dž", "\"ß", "'ع", "\r", "ꟲa"]} +{"text": ".t(ta'M‍😀🏽'ré🙂'Sé12345678 é", "tokens": 28, "pieces": [".t", "(ta", "'", "M", "‍😀🏽'", "re", "́🙂'", "Se", "́", "123", "456", "78", " é"]} +{"text": "'M å​'ſ漢é!", "tokens": 13, "pieces": ["'M", " a", "̊​'", "ſ漢é", "!"]} +{"text": "e…d\r\n\r\n½ 'S'M‍\"\tİe- \n!!'re३(\n字 ​fi 's ,\n9ꟲ­", "tokens": 39, "pieces": ["e", "…d", "\r\n\r\n", "½", " ", "'S", "'M", "‍\"", "\tİe", "-", " \n", "!!'", "re", "३", "(<", "META", "_START", ">\n", "字", " ", "​fi", " ", "'s", " ,\n", "9", "ꟲ", "­"]} +{"text": "#$%\r \n½ a'S­'D", "tokens": 11, "pieces": ["#$%\r", " \n", "½", " a", "'S", "­'", "D"]} +{"text": "🙂aß㍿<|fim_prefix|>'VE 𐞁#$%ſ'll#$%\r\nA<|endoftext|>", "tokens": 37, "pieces": ["🙂aß", "㍿<|", "fim", "_prefix", "|>'", "VE", " ", " 𐞁", "#$%", "ſ", "'ll", "#$%\r\n", "A", "<|", "endoftext", "|>"]} +{"text": "𐞁 å!ae(\r\ns\t字'M'll'TA😀🏽٣٤٥٦ع\t́😀🏽éZİ\u000b", "tokens": 44, "pieces": ["𐞁", " a", "̊!", "ae", "(\r\n", "s", "\t字", "'M", "'ll", "'T", "A", "😀🏽", "٣٤٥", "٦", "ع", "\t", "́😀🏽", "e", "́Zİ", "\u000b"]} +{"text": "0m٣٤٥٦Ⅳ'T", "tokens": 13, "pieces": ["0", "m", "٣٤٥", "٦Ⅳ", "'T"]} +{"text": "å'M", "tokens": 5, "pieces": ["a", "̊'", "M"]} +{"text": "å'Ret'Retꟲ𐞁
m🙂🙂\n𐞁e", "tokens": 27, "pieces": ["a", "̊'", "Ret", "'Re", "tꟲ𐞁", "
m", "🙂🙂\n", "𐞁e"]} +{"text": "'re're>åſ㍿​\u000bé'M0'M\r", "tokens": 19, "pieces": ["'re", "'re", ">a", "̊ſ", "㍿​", "\u000be", "́'", "M", "0", "'M", "\r"]} +{"text": "e \u000b'S.a", "tokens": 5, "pieces": ["e", " ", "\u000b", "'S", ".a"]} +{"text": " <  \n漢d're​eİe\"\t㋿
ع́\t(", "tokens": 23, "pieces": [" ", "<", "  \n", "漢d", "'re", "​eİe", "\"", "\t", "㋿", "
ع", "́", "\t", "("]} +{"text": "!!'s'S'Tt", "tokens": 9, "pieces": ["!!'", "s", "'S", "'T", "t"]} +{"text": ",'ſḍ̇<|fim_prefix|>d é字'ſ㋿㋿e", "tokens": 37, "pieces": [",'", "ſḋ", "̣<|", "fim", "_prefix", "|>", "d", " ", " é", "字", "'ſ", "㋿<", "META", "_START", ">㋿", "e"]} +{"text": "Dž 'D𐞁
👍🏽#$%𐞁!٣٤٥٦Z!!t", "tokens": 37, "pieces": ["Dž", " '", "D𐞁", "
", "👍🏽#$%", "𐞁", "!<", "EOT", ">", "٣٤٥", "٦", "Z", "!!", "t"]} +{"text": "\t,Ⅳ", "tokens": 4, "pieces": ["\t", ",", "Ⅳ"]} +{"text": " ​'re \nعt𐞁'Dta \nåé'D漢👍🏽så㍿́DžDžt Z'S< ", "tokens": 49, "pieces": [" ", " ​'", "re", " \n", "عt𐞁", "'D", "ta", " \n", "a", "̊e", "́'", "D漢", "👍🏽", "sa", "̊㍿́", "DžDžt", " ", " Z", "'S", "<", " "]} +{"text": "<|fim_prefix|>…'s.𐞁عfi字 >\r\n're", "tokens": 25, "pieces": ["<|", "fim", "_prefix", "|>", "…", "'s", ".𐞁عfi字", "", " ", " >\r\n", "'re"]} +{"text": "'VE9👍🏽", "tokens": 9, "pieces": ["'VE", "9", "👍🏽"]} +{"text": "Ⅳfi(\"'D\r\nİ>A漢😀🏽'S .mEOTZ
……!!9字ḍ̇𐞁<ꟲ\r\n<", "tokens": 56, "pieces": ["Ⅳ", "fi", "(\"'", "D", "\r\n", "İ", ">A漢", "😀🏽'", "S", " ", ".mEOTZ", "
", "", "…", "…", "!!", "9", "字ḋ", "̣𐞁", "<ꟲ", "\r\n", "<"]} +{"text": "'M#$%٣٤٥٦(#$%.m㍿㋿9#$%'T㋿\r\n\r\n!!Aḍ̇", "tokens": 37, "pieces": ["'M", "#$%", "٣٤٥", "٦", "(#$%.", "m", "㍿㋿", "9", "#$%'", "T", "㋿\r\n\r\n", "!!", "Aḋ", "̣"]} +{"text": "'s<|fim_prefix|>'ReZ😀🏽㋿漢s𐞁Ⅳḍ̇t İ ́e👍🏽…ḍ̇fi \n 漢ß'VE", "tokens": 60, "pieces": ["'s", "<|", "fim", "_prefix", "|>'", "ReZ", "😀🏽㋿", "漢s𐞁", "Ⅳ", "ḋ", "̣t", " ", " İ", " ́", "e", "👍🏽", "…ḋ", "̣fi", " \n", " 漢ß", "'VE"]} +{"text": "'re'Dİ字9​.🙂​㋿\t-
İé12345678‍
<|endoftext|>㍿12345678👍🏽\r㍿Zfi'VE> é", "tokens": 62, "pieces": ["'re", "'D", "İ字", "9", "​.🙂​㋿", "\t", "-", "
İé", "123", "456", "78", "‍", "
", "<|", "endoftext", "|>㍿", "123", "456", "78", "👍🏽\r", "㍿Zfi", "'VE", ">", " e", "́<", "EOT", ">"]} +{"text": "0 \n­  …'re漢­'T\r٣٤٥٦9s9d. 're\n'sⅣ\r\n'T<|fim_prefix|>", "tokens": 47, "pieces": ["0", " \n", "­", "  ", "…", "'re", "漢", "­'", "T", "\r", "٣٤٥", "٦9", "s", "", "9", "d", ".", " '", "re", "\n", "'s", "Ⅳ", "\r\n", "'T", "<|", "fim", "_prefix", "|>"]} +{"text": "🙂㋿'re字'll,s\r½ ḍ̇𐞁漢'D<|fim_prefix|>٣٤٥٦٣٤٥٦'re'Re,👍🏽!!­\r9", "tokens": 64, "pieces": ["🙂㋿'", "re字", "'ll", ",<", "EOT", ">s", "\r", "½", " ḋ", "̣𐞁漢", "'D", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦٣٤", "٥٦", "'re", "'Re", ",👍🏽!!­\r", "9"]} +{"text": "'VE", "tokens": 2, "pieces": ["'VE"]} +{"text": "
ß 's'Re#$% \r\n㍿́m 'M12345678!!
🙂<|fim_prefix|>ß>é'M'D!!DžݽEOT", "tokens": 36, "pieces": ["!ß", "🙂", "٣٤٥", "٦", "!", "
", "'T", "'M", "ß", ">e", "́'", "M", "'D", "!!", "Džİ", "", "½", "EOT"]} +{"text": "mZas !Ⅳ>
'S㍿0\r­ꟲ,#$%漢'T'll
-½'ſ'D!!Zté\"漢9字字's
½", "tokens": 49, "pieces": ["mZas", " ", "!", "Ⅳ", ">", "
", "'S", "㍿", "0", "\r", "­ꟲ", ",#$%", "漢", "'T", "'ll", "
", "-", "½", "'ſ", "'D", "!!", "Zté", "\"漢", "9", "字字", "'s", "
", "½"]} +{"text": "<'Re­ 'T㋿'VE0 \nEOT\n mééİ.\r\n\r\né<|endoftext|>å<|endoftext|>🙂!12345678\"İ'S-Dž", "tokens": 52, "pieces": ["<'", "Re", "­", " ", " '", "T", "㋿'", "VE", "0", " \n", "EOT", "\n", " mééİ", ".\r\n\r\n", "é", "<|", "endoftext", "|>", "a", "̊<|", "endoftext", "|>🙂!", "123", "456", "78", "\"İ", "'S", "-Dž"]} +{"text": "­'Tå'D're\r\n\r\n-é,12345678.<,٣٤٥٦ſ٣٤٥٦\t́\"'VE½m\u000bꟲ", "tokens": 51, "pieces": ["­'", "T", "a", "̊'", "D", "'re", "\r\n\r\n", "-e", "́,", "123", "456", "78", ".<,", "٣٤٥", "٦", "ſ", "", "٣٤٥", "٦", "\t", "́\"'", "VE", "½", "m", "\u000bꟲ"]} +{"text": "'M🙂12345678e\r\n9,12345678İ'reعA", "tokens": 18, "pieces": ["'M", "🙂", "123", "456", "78", "e", "\r\n", "9", ",", "123", "456", "78", "İ", "'re", "عA"]} +{"text": " \"-m'३㍿字🙂 \tꟲ‍ḍ̇éḍ̇#$%­å,'s", "tokens": 45, "pieces": [" ", " \"-", "m", "'<", "EOT", ">", "३", "㍿字", "🙂", " ", "\tꟲ", "‍", "ḋ", "̣éḋ", "̣#$%­", "a", "̊,'", "s"]} +{"text": "㍿\rEOTḍ̇'S ßİ​d,½٣٤٥٦Ⅳ' ٣٤٥٦漢'Dé 'S𐞁ḍ̇mfi\u000b\rm'Re ½<|fim_prefix|><|endoftext|>> \n'", "tokens": 87, "pieces": ["㍿\r", "EOTḋ", "̣'", "S", " ßİ", "​d", ",<", "EOT", ">", "½٣٤", "٥٦Ⅳ", "'", " ", " ", "٣٤٥", "٦", "漢", "'D", "é", " ", "'S", "𐞁ḋ", "̣mfi", "\u000b\r", "m", "'Re", " ", " ", "½", "<|", "fim", "_prefix", "|><|", "endoftext", "|>>", " \n", "'"]} +{"text": "\r\n\r\né< (ḍ̇ ", "tokens": 12, "pieces": ["\r\n\r\n", "e", "́<", " ", "(ḋ", "̣", " "]} +{"text": "t,!㍿😀🏽éé", "tokens": 22, "pieces": ["t", "Ⅳ", "İ𐞁", "'T", "A", ".🙂'", "ll", "Ⅳ", "́>", "éé"]} +{"text": "é́٣٤٥٦Džmſ\r𐞁<|fim_prefix|>­s\r\n9ß \n å'Re…", "tokens": 40, "pieces": ["é", "́", "٣٤٥", "٦", "Džmſ", "\r", "𐞁", "<|", "fim", "_prefix", "|>­", "s", "\r\n", "9", "ß", " \n", " a", "̊'", "Re", "…"]} +{"text": "!!­d(ſe<|endoftext|>Ⅳſ'D\r\n\r\n#$%dd漢㋿(ع(漢EOT.'s½字", "tokens": 39, "pieces": ["!!­", "d", "(ſe", "<|", "endoftext", "|>", "Ⅳ", "ſ", "'D", "\r\n\r\n", "#$%", "dd漢", "㋿(", "ع", "(漢EOT", ".'", "s", "½", "字"]} +{"text": "ꟲ'Re fi漢!!å>'D", "tokens": 19, "pieces": ["ꟲ", "'Re", " ", " fi漢", "!!", "a", "̊>'", "D"]} +{"text": "12345678­ꟲ!! 漢́'reéZ's", "tokens": 22, "pieces": ["123", "456", "78", "­ꟲ", "!!", " 漢", "́'", "reéZ", "'s", ""]} +{"text": "́éé.㋿İé३​><…漢­🙂'VE\t \n\n!Ⅳ", "tokens": 30, "pieces": ["́e", "́e", "́.㋿", "İe", "́", "३", "​><", "…漢", "­🙂'", "VE", "\t \n\n", "!", "Ⅳ"]} +{"text": "eⅣ😀🏽'D're३'s's9-0!", "tokens": 19, "pieces": ["e", "Ⅳ", "😀🏽'", "D", "'re", "३", "'s", "'s", "9", "-", "0", "!"]} +{"text": "0 \n(½å<|fim_prefix|> m عd'llé!Z字EOTå😀🏽㍿é.İ A \nfi
.'ßfi​Z'VEfi \n", "tokens": 60, "pieces": ["0", " \n", "(", "½", "a", "̊<|", "fim", "_prefix", "|>", " m", " عd", "'ll", "e", "́!", "Z字EOTa", "̊😀🏽㍿", "e", "́.", "İ", " A", " \n", "fi", "
", ".'", "ßfi", "​Z", "'VE", "fi", " \n"]} +{"text": "<|endoftext|> \n…Aİ‍ ,", "tokens": 17, "pieces": ["<|", "endoftext", "|>", " \n", "…Aİ", "‍", " ", " ,"]} +{"text": "a'S!", "tokens": 3, "pieces": ["a", "'S", "!"]} +{"text": "👍🏽\r\n\r\n<|fim_prefix|>字'M.ḍ̇ꟲ", "tokens": 25, "pieces": ["👍🏽\r\n\r\n", "<|", "fim", "_prefix", "|>", "字", "'M", ".ḋ", "̣ꟲ"]} +{"text": " ½字\nm\r\n\r\n'T<>>mſaå㋿ 'T!𐞁ع<|endoftext|>\r\n\r\na're,123456789'S \n-", "tokens": 47, "pieces": ["", " ", "½", "字", "\n", "m", "\r\n\r\n", "'T", "<>>", "mſaa", "̊㋿", " ", "'T", "!𐞁ع", "<|", "endoftext", "|>\r\n\r\n", "a", "'re", ",", "123", "456", "789", "'S", " \n", "-"]} +{"text": "'Re<|endoftext|>Dž\n٣٤٥٦ḍ̇ta!fi…a٣٤٥٦ -Ⅳ😀🏽é\n#$%漢👍🏽   😀🏽'T'Mſ'ſé", "tokens": 81, "pieces": ["'Re", "<|", "endoftext", "|>", "Dž", "\n", "٣٤٥", "٦", "ḋ", "̣ta", "!fi", "…a", "٣٤٥", "٦", " ", "-", "Ⅳ", "😀🏽", "é", "\n", "#$%", "漢", "👍🏽", "  ", " ", "😀🏽'", "T", "'", "Mſ", "'ſ", "é"]} +{"text": "́㍿'Re9d#$%#$%ꟲAꟲ<|fim_prefix|>\r\n\n<|endoftext|>EOTZ㍿m", "tokens": 42, "pieces": ["́㍿'", "Re", "9", "d", "#$%#$%", "ꟲAꟲ", "<|", "fim", "_prefix", "|>\r\n\n", "<|", "endoftext", "|>", "EOTZ", "㍿m"]} +{"text": "'Tt㍿😀🏽efi'T12345678sİ 'S​(​'S'T", "tokens": 27, "pieces": ["'T", "t", "㍿😀🏽", "efi", "'T", "123", "456", "78", "sİ", " '", "S", "​(​'", "S", "'T"]} +{"text": "9'<ⅣⅣ 0'll d'ſDž", "tokens": 20, "pieces": ["9", "'<", "ⅣⅣ", " ", " ", "0", "'", "ll", " d", "'ſ", "Dž"]} +{"text": "\r\n.d<|fim_prefix|>字­,ꟲꟲ!EOT's", "tokens": 22, "pieces": ["\r\n", ".d", "<|", "fim", "_prefix", "|>", "字", "­,", "ꟲꟲ", "!EOT", "'s"]} +{"text": "ſ.'ſ<|endoftext|>👍🏽EOT👍🏽A Ⅳ👍🏽9'ſ å…\"m'M>\u000b.<'VEm", "tokens": 61, "pieces": ["ſ", ".'", "ſ", "<|", "endoftext", "|>👍🏽", "EOT", "👍🏽", "A", " ", "Ⅳ", "👍🏽", "9", "'ſ", " ", "<", "EOT", ">a", "̊", "…", "\"m", "'M", ">", "\u000b", ".<'", "VEm"]} +{"text": " 'll12345678'D\u000bſ\t'Re𐞁…㍿e'ſ­EOT𐞁.A'ReEOT\n", "tokens": 43, "pieces": [" ", " '", "ll", "123", "456", "78", "'", "D", "\u000bſ", "\t", "'Re", "𐞁", "…", "㍿e", "'ſ", "­EOT𐞁", ".A", "'Re", "EOT", "\n"]} +{"text": "s३漢㍿\r\n\r\nꟲع
DžEOT\"ſ३ḍ̇mDž", "tokens": 35, "pieces": ["s", "३", "漢", "㍿\r\n\r\n", "ꟲع", "
DžEOT", "\"ſ", "३", "ḋ", "̣m", "Dž"]} +{"text": " ſ…'MåDžḍ̇\t'Re,<|endoftext|>ⅣA字 \n İfi.\tétſsꟲ<|endoftext|>㍿>ꟲ", "tokens": 59, "pieces": [" ", " ſ", "…", "'M", "a", "̊Džḋ", "̣", "\t", "'Re", ",<|", "endoftext", "|>", "Ⅳ", "A字", " \n", " ", " İfi", ".", "\te", "́tſsꟲ", "<|", "endoftext", "|>㍿>", "ꟲ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "<|endoftext|> ", "tokens": 8, "pieces": ["<|", "endoftext", "|>", " "]} +{"text": "12345678're'½'ſ😀🏽漢‍(ſ m𐞁漢->Ⅳ😀🏽'ſ(字\u000bm9Džعḍ̇漢ḍ̇ \n½ß ", "tokens": 63, "pieces": ["123", "456", "78", "'re", "'", "½", "'ſ", "😀🏽", "漢", "‍(", "ſ", " m𐞁漢", "->", "Ⅳ", "😀🏽'", "ſ", "(字", "\u000bm", "9", "Džعḋ", "̣漢ḋ", "̣", " \n", "½", "ß", " "]} +{"text": "'Re -e𐞁EOT9'VE<|endoftext|>\r'D㋿e!\r\n😀🏽‍́ḍ̇…ع字é<|fim_prefix|>'ſ", "tokens": 61, "pieces": ["'Re", " ", "-e", "<", "META", "_START", ">𐞁EOT", "9", "'VE", "<|", "endoftext", "|>\r", "'D", "㋿e", "!\r\n", "😀🏽‍́", "ḋ", "̣", "…ع字é", "<|", "fim", "_prefix", "|>'", "ſ"]} +{"text": "<|endoftext|>>a#$%dZEOT m\naع", "tokens": 17, "pieces": ["<|", "endoftext", "|>>", "a", "#$%", "dZEOT", " m", "\n", "aع"]} +{"text": "🙂‍DžDž㍿字\u000b's字'é're", "tokens": 19, "pieces": ["🙂‍", "DžDž", "㍿字", "\u000b", "'s", "字", "'e", "́'", "re"]} +{"text": "ع,<|fim_prefix|>'e \n>ḍ̇\u000bd'D12345678're a#$%…Dž𐞁\nå㍿\u000bⅣ!!\"½\n\"​#$%​Ⅳ½́́t", "tokens": 60, "pieces": ["ع", ",<|", "fim", "_prefix", "|>'", "e", " \n", ">ḋ", "̣", "\u000bd", "'D", "123", "456", "78", "'re", " ", " a", "#$%", "…Dž𐞁", "\n", "a", "̊㍿", "\u000b", "Ⅳ", "!!\"", "½", "\n", "\"​#$%​", "Ⅳ½", "́́", "t"]} +{"text": "-.'Sfi🙂9İ­#$%Ⅳ9\"m\r'll३s\r\n\r\nfiZe३'ll", "tokens": 37, "pieces": ["-<", "META", "_START", ">.'", "Sfi", "🙂", "9", "İ", "­#$%", "Ⅳ9", "\"m", "\r", "'", "ll", "३", "s", "\r\n\r\n", "fiZe", "३", "'ll"]} +{"text": "​\rA(٣٤٥٦d👍🏽<ع'DعAé'Re\r\n\r\n \t'ſt#$%Z\"Dž‍.'Sa\r 𐞁 \n\r\n", "tokens": 55, "pieces": ["​\r", "A", "(", "٣٤٥", "٦", "d", "👍🏽<", "ع", "'D", "عAe", "́'", "Re", "\r\n\r\n", " ", "\t", "'ſ", "t", "#$%", "Z", "\"Dž", "‍.'", "Sa", "\r", " 𐞁", " \n\r\n"]} +{"text": "ꟲ,'M", "tokens": 5, "pieces": ["ꟲ", ",'", "M"]} +{"text": "㋿ fi,ع'll12345678'reé\r\n,\ns'", "tokens": 19, "pieces": ["㋿", " fi", ",ع", "'ll", "123", "456", "78", "'re", "e", "́\r\n", ",\n", "s", "'"]} +{"text": "'re-Ⅳ\"ḍ̇ع!!s", "tokens": 13, "pieces": ["'re", "-", "Ⅳ", "\"ḋ", "̣ع", "!!", "s"]} +{"text": "'lléꟲ\n12345678ḍ̇Ⅳ0㍿'VE'D12345678e㍿ß㋿'sß-½s\u000b\t'll…#$%\r Z'TtEOT 𐞁,ꟲd", "tokens": 67, "pieces": ["'ll", "e", "́ꟲ", "\n", "123", "456", "78", "ḋ", "̣", "Ⅳ0", "㍿'", "VE", "'D", "123", "456", "78", "e", "㍿ß", "㋿'", "sß", "-", "½", "s", "\u000b", "\t", "'ll", "…", "#$%\r", " <", "EOT", ">Z", "'T", "tEOT", " 𐞁", ",ꟲd"]} +{"text": "\u000b
åZ‍ \r\n\r\n\r\n\r\nEOT \n‍tع…'Re\r\n!ⅣſⅣ!!😀🏽9<(\r\n>'​fi\u000b🙂", "tokens": 45, "pieces": ["\u000b", "
a", "̊Z", "‍", " \r\n\r\n\r\n\r\n", "EOT", " \n", "‍tع", "…", "'Re", "\r\n", "!", "Ⅳ", "ſ", "Ⅳ", "!!😀🏽", "9", "<(\r\n", ">'​", "fi", "\u000b", "🙂"]} +{"text": " …३\r'M
!漢- 
12345678t<|endoftext|>
'D,\na'Red ٣٤٥٦漢\"é㋿ s‍>\r\n\r\ns9DžZ½😀🏽", "tokens": 66, "pieces": [" ", "…", "३", "\r", "'M", "
", "!漢", "-", " ", "
", "123", "456", "78", "t", "<|", "endoftext", "|>", "
", "'D", ",\n", "a", "'Re", "d", " ", "٣٤٥", "٦", "漢", "\"e", "́㋿", " s", "‍>\r\n\r\n", "s", "9", "DžZ", "½", "😀🏽"]} +{"text": "'ll𐞁\r…'ll!Ⅳ-<|endoftext|>ß \r\n\r\n\r\n\r\nét㋿\n'Tt12345678", "tokens": 35, "pieces": ["'ll", "𐞁", "\r", "…", "'ll", "!", "Ⅳ", "-<|", "endoftext", "|>", "ß", " \r\n\r\n\r\n\r\n", "e", "́t", "㋿\n", "'T", "t", "123", "456", "78"]} +{"text": "'s\u000b-'s‍EOT#$%ſ\u000bßEOT\u000b(åعfi", "tokens": 23, "pieces": ["'s", "\u000b", "-'", "s", "‍EOT", "#$%", "ſ", "\u000bßEOT", "\u000b", "(a", "̊عfi"]} +{"text": "\t< '漢e漢12345678'Re'D'D912345678\u000b éad ḍ̇İmm!!12345678३㋿\r\n", "tokens": 41, "pieces": ["\t", "<", " ", " '", "漢e漢", "123", "456", "78", "'Re", "'D", "'D", "912", "345", "678", "\u000b ", " e", "́ad", " ḋ", "̣İmm", "!!", "123", "456", "78३", "㋿\r\n"]} +{"text": "Z09٣٤٥٦0é'T're 'T", "tokens": 16, "pieces": ["Z", "09٣", "٤٥٦", "0", "é", "'T", "'re", " ", "'T"]} +{"text": "㍿٣٤٥٦\nt㍿字\r\n\r\n漢٣٤٥٦\n<'re‍>\t㍿㍿Ⅳ\r\n!'VE.9e'.ع", "tokens": 51, "pieces": ["㍿", "٣٤٥", "٦", "\n", "t", "㍿字", "\r\n\r\n", "漢", "٣٤٥", "٦", "\n", "<'", "re", "‍>", "\t", "㍿㍿", "Ⅳ", "\r\n", "!'", "VE", ".", "9", "e", "'.", "ع"]} +{"text": "!! ‍é½३A'VEſ<|fim_prefix|>ع​é😀🏽'T\r<|fim_prefix|> d!!a'll٣٤٥٦½A<字s😀🏽#$%d½㋿e#$%", "tokens": 77, "pieces": ["!!", " ", "‍é", "½३", "A", "'VE", "ſ", "<|", "fim", "_prefix", "|>", "ع", "​e", "́😀🏽'", "T", "\r", "<|", "fim", "_prefix", "|>", " d", "!!", "a", "'", "ll", "٣٤٥", "٦½", "A", "<字s", "😀🏽#$%", "d", "½", "㋿e", "#$%"]} +{"text": "ع​A(t12345678>'S'T😀🏽a'Re٣٤٥٦\r\nd", "tokens": 34, "pieces": ["ع", "​A", "(<", "META", "_START", ">t", "123", "456", "78", ">'", "S", "'T", "😀🏽", "a", "'Re", "٣٤٥", "٦", "\r\n", "d"]} +{"text": "!!\"
#$%\"'lléß
ß 12345678->t👍🏽\r\n\r\n're漢 ‍<|fim_prefix|>d0 \n 0's㋿e#$%0😀🏽", "tokens": 65, "pieces": ["!!\"", "
", "#$%\"'", "lléß", "
ß", " ", "123", "456", "78", "->", "t", "👍🏽\r\n\r\n", "'re", "漢", " ", "‍<|", "fim", "_prefix", "|>", "d", "0", " \n", " ", "0", "'s", "㋿e", "#$%<", "META", "_START", ">", "0", "😀🏽"]} +{"text": "'Da<३éſd'Ret'Re𐞁字字'DEOT\r\n\r\nEOT\"½'S­ -\r.tå\r\n\r\n​​ع㋿😀🏽", "tokens": 46, "pieces": ["'D", "a", "<", "३", "e", "́ſd", "'Re", "t", "'Re", "𐞁字字", "'D", "EOT", "\r\n\r\n", "EOT", "\"", "½", "'S", "­", " ", "-\r", ".ta", "̊\r\n\r\n", "​​", "ع", "㋿😀🏽"]} +{"text": "\r\n'llA", "tokens": 4, "pieces": ["\r\n", "'ll", "A"]} +{"text": ">🙂\"'Rea", "tokens": 6, "pieces": [">🙂\"'", "Rea"]} +{"text": "'D'ſ'٣٤٥٦ <\u000b漢㋿0'llA.\t'D🙂'T٣٤٥٦ ٣٤٥٦-<|endoftext|>!! <|fim_prefix|><|endoftext|>字Džé㍿ع\r\nd\u000b're㍿​ḍ̇", "tokens": 95, "pieces": ["'D", "'ſ", "'", "٣٤٥", "٦", " ", " <", "\u000b漢", "㋿", "0", "'ll", "A", ".", "\t", "'D", "🙂'", "T", "٣٤٥", "٦", " ", " ", "٣٤٥", "٦", "-<|", "endoftext", "|>!!", " ", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "字Dže", "́㍿", "ع", "\r\n", "d", "\u000b", "'re", "㍿​", "ḋ", "̣"]} +{"text": "' 字'D٣٤٥٦dſZ \u000b­ 字é'reéé
ḍ̇fi😀🏽İḍ̇é‍३md𐞁<|endoftext|>­👍🏽​'VE'‍\r", "tokens": 80, "pieces": ["'", " 字", "'D", "٣٤٥", "٦", "dſZ", " ", "\u000b", "­", " 字é", "'re", "e", "́é", "
", "ḋ", "̣fi", "😀🏽", "İḋ", "̣é", "‍", "३", "md𐞁", "<|", "endoftext", "|>­👍🏽​'", "VE", "'‍\r"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "😀🏽 ㍿'ſ😀🏽👍🏽ḍ̇'Dİ'S字\r9😀🏽漢字fi<|endoftext|>s(\t­\r(", "tokens": 58, "pieces": ["😀🏽", " ", "㍿'", "ſ", "😀🏽👍🏽", "ḋ", "̣'", "Dİ", "'S", "字", "\r", "9", "😀🏽", "漢字fi", "<|", "endoftext", "|>", "s", "(", "\t", "­\r", "("]} +{"text": "åm½s\"'Re'D字\r\n''ſ字ع0s12345678😀🏽A\u000b<|endoftext|>\"
EOT're", "tokens": 41, "pieces": ["a", "̊m", "½", "s", "\"'", "Re", "'D", "字", "\r\n", "''", "ſ字ع", "0", "s", "123", "456", "78", "😀🏽", "A", "\u000b", "<|", "endoftext", "|>\"", "
EOT", "'re"]} +{"text": " \ń👍🏽1234567812345678", "tokens": 14, "pieces": [" \n", "́👍🏽", "123", "456", "781", "234", "567", "8"]} +{"text": "\r\"\"é\r\n\r\n#$%😀🏽-𐞁…\r\n de…t­s>EOT!!", "tokens": 29, "pieces": ["\r", "\"\"", "e", "́\r\n\r\n", "#$%😀🏽-", "𐞁", "…\r\n", " ", " de", "…t", "­s", ">EOT", "!!"]} +{"text": "́ḍ̇", "tokens": 6, "pieces": ["́ḋ", "̣"]} +{"text": "''字9ꟲ 'VE\u000b😀🏽\n'ſ!0Zß٣٤٥٦m 😀🏽<<|fim_prefix|>字å 'Re\r\n३'T'D<", "字a", "̊", " ", " '", "Re", "\r\n", "३", "'T", "'D", "<<", "d", "👍🏽", "İ", "…", "("]} +{"text": "ع<", "tokens": 2, "pieces": ["ع", "<"]} +{"text": "
\t\u000bḍ̇'\"", "tokens": 10, "pieces": ["
\t", "\u000bḋ", "̣'\""]} +{"text": "\"'M३ds😀🏽'T​\r\n.\t d🙂٣٤٥٦
३'Ss\r\n\r\nfi'T\r\n\r\n\u000b", "tokens": 39, "pieces": ["\"'", "M", "३", "ds", "😀🏽'", "T", "​\r\n", ".", "\t ", " d", "🙂", "٣٤٥", "٦", "
", "३", "'S", "s", "\r\n\r\n", "fi", "'T", "\r\n\r\n\u000b"]} +{"text": "0‍'SZ(12345678㋿🙂'D12345678
\"ⅣEOT'T\r\n\r\n字<|endoftext|>#$%<|endoftext|>'D\u000b<|fim_prefix|>​ A'Refi­Dž0'<ſfi-,🙂", "tokens": 73, "pieces": ["0", "‍'", "SZ", "(", "123", "456", "78", "㋿🙂'", "D", "123", "456", "78", "
", "\"", "Ⅳ", "EOT", "'T", "\r\n\r\n", "字", "<|", "endoftext", "|>#$%<|", "endoftext", "|>'", "D", "\u000b", "<|", "fim", "_prefix", "|>​", " A", "'Re", "fi", "­Dž", "0", "'<", "ſfi", "-,🙂"]} +{"text": "< \"'Re<|fim_prefix|>", "tokens": 16, "pieces": ["<", " ", "\"'", "Re", "<|", "fim", "_prefix", "|><", "META", "_START", ">"]} +{"text": "½\u000b'ſADž.\r­><|fim_prefix|>!!½s ß🙂", "tokens": 26, "pieces": ["½", "\u000b", "'ſ", "ADž", ".\r", "­><|", "fim", "_prefix", "|>!!", "½", "s", " ß", "🙂"]} +{"text": "0.३é!90🙂d9'VE​字 t.'S३😀🏽0ß字<\rA٣٤٥٦🙂字m<<|fim_prefix|>👍🏽", "tokens": 63, "pieces": ["0", ".", "३", "e", "́!", "90", "🙂d", "9", "'VE", "​字", " t", ".'", "S", "३", "😀🏽", "0", "ß字", "<\r", "A", "٣٤٥", "٦", "🙂字m", "<<|", "fim", "_prefix", "|>👍🏽"]} +{"text": "ḍ̇漢'Dé​#$%㋿'re''re<>\r\né😀🏽12345678", "tokens": 32, "pieces": ["ḋ", "̣漢", "'D", "e", "́​#$%㋿'", "re", "''", "re", "<>\r\n", "e", "́😀🏽", "123", "456", "78"]} +{"text": "'ll३-té½e'…\n​ß\u000bⅣ😀🏽Ⅳ'M9!३'", "tokens": 29, "pieces": ["'ll", "३", "-te", "́", "½", "e", "'", "…\n", "​ß", "\u000b", "Ⅳ", "😀🏽", "Ⅳ", "'M", "9", "!", "३", "'"]} +{"text": "'llع😀🏽 -'s Džع99( 'Re9até9٣٤٥٦㍿漢->\"EOTéEOT½!!ßa12345678​
'VE'Re", "tokens": 59, "pieces": ["'ll", "ع", "😀🏽", " -'", "s", " Džع", "99", "(", " '", "Re", "9", "ate", "́", "9٣٤", "٥٦", "㍿漢", "->\"", "EOTéEOT", "½", "!!", "ßa", "123", "456", "78", "​", "
", "'VE", "'Re"]} +{"text": "ع'VE字", "tokens": 4, "pieces": ["ع", "'VE", "字"]} +{"text": "٣٤٥٦𐞁12345678'…​'VE𐞁(ß", "tokens": 30, "pieces": ["٣٤٥", "٦", "𐞁", "123", "456", "78", "'<", "META", "_START", ">", "…", "​'", "VE𐞁", "(ß"]} +{"text": "12345678字(İ 字😀🏽३\u000bm𐞁<|fim_prefix|>👍🏽 \"fi'T'Msꟲ́½​\"漢fi\u000b<|endoftext|>㍿fi", "tokens": 68, "pieces": ["123", "456", "78", "字", "(İ", " ", " 字", "😀🏽", "३", "\u000bm𐞁", "<|", "fim", "_prefix", "|>👍🏽", " ", " \"", "fi", "'T", "'M", "sꟲ", "́", "½", "​\"<", "EOT", ">漢fi", "\u000b", "<|", "endoftext", "|>㍿", "fi"]} +{"text": "㍿12345678­ꟲ'VE'DꟲDž'Sḍ̇ḍ̇.'T'\r\n\r\n'll\rDž \n'llZåſ<|endoftext|>𐞁.", "tokens": 56, "pieces": ["㍿", "123", "456", "78", "­ꟲ", "'VE", "'D", "ꟲDž", "'S", "ḋ", "̣ḋ", "̣.'", "T", "'\r\n\r\n", "'ll", "\r", "Dž", " \n", "'ll", "Za", "̊ſ", "<|", "endoftext", "|>", "𐞁", "."]} +{"text": "<ꟲ३é<|fim_prefix|>0(​😀🏽", "tokens": 23, "pieces": ["<ꟲ", "३", "e", "́<|", "fim", "_prefix", "|>", "0", "(​😀🏽"]} +{"text": " 's\"'ll\r'M… 'D'\t­ ‍A㋿\r\n\r\n \n.", "tokens": 29, "pieces": [" '", "s", "\"'", "ll", "\r", "'M", "… ", " '", "D", "'", "\t", "­", " ", "‍A", "㋿\r\n\r\n", " \n", "."]} +{"text": " 'ReEOTİ́é­s𐞁'Dع٣٤٥٦'M0½㋿\n'VEa", "tokens": 32, "pieces": [" '", "ReEOTİ", "́é", "­s𐞁", "'D", "ع", "٣٤٥", "٦", "'M", "0½", "㋿\n", "'VE", "a"]} +{"text": "­\"A", "tokens": 4, "pieces": ["­\"", "A"]} +{"text": "!!\r\na㋿'reéDž\ré!!!!t字tḍ̇!!'DDž .\r\u000be漢,", "tokens": 43, "pieces": ["!!\r\n", "a", "㋿'", "reéDž", "\r", "é", "!!!!", "t字tḋ", "̣!!<", "m", "'", "DDž", " ", ".\r", "\u000be漢", ","]} +{"text": " 12345678!((\rm‍e ½ſ'VE 'S३\t‍.9<", "tokens": 28, "pieces": [" ", "123", "456", "78", "!((\r", "m", "‍e", " ", "½", "ſ", "'VE", " ", " '", "S", "३", "\t", "‍.", "9", "<"]} +{"text": "!<|endoftext|>12345678\ns½0<\"٣٤٥٦​
A㍿ \n
İ字é​", "tokens": 38, "pieces": ["!<|", "endoftext", "|>", "123", "456", "78", "\n", "s", "½0", "<\"", "٣٤٥", "٦", "​", "
A", "㍿", " \n", "
İ字é", "​"]} +{"text": "🙂ꟲ\u000bt\n
​㋿㍿.\r\"​12345678㋿字e\"'ſ漢'T\t!ſ​
're<…​", "tokens": 56, "pieces": ["🙂ꟲ", "", "\u000bt", "\n", "
", "​㋿㍿.\r", "\"​", "123", "456", "78", "㋿字e", "\"'", "ſ漢", "'T", "\t", "!ſ", "​", "
", "'re", "<", "…", "​"]} +{"text": "ßEOT½ 0½عZ '‍‍'VEé\u000b", "tokens": 20, "pieces": ["ßEOT", "½", " ", "0½", "عZ", " ", "'‍‍'", "VEe", "́", "\u000b"]} +{"text": "åİa\t字9㋿9…<|fim_prefix|>, ‍'re…\t'12345678-'D<|fim_prefix|>漢 \n!'M 'D\t字…åe", "tokens": 57, "pieces": ["a", "̊İa", "\t字", "9", "㋿", "9", "…", "<|", "fim", "_prefix", "|>,", " ", "‍'", "re", "…", "\t", "'", "123", "456", "78", "-'", "D", "<|", "fim", "_prefix", "|>", "漢", " \n", "!'", "M", " ", "'D", "\t字", "…a", "̊e"]} +{"text": "ḍ̇\nḍ̇​٣٤٥٦", "tokens": 20, "pieces": ["ḋ", "̣\n", "ḋ", "̣​", "٣٤٥", "٦"]} +{"text": "fi ,ß\"İZ09\u000b \né\"'Reİꟲ\"ß'T́!!('ree-'S'reḍ̇Z
'VE", "tokens": 39, "pieces": ["fi", " ", " ,", "ß", "\"İZ", "09", "\u000b \n", "e", "́\"'", "Reİꟲ", "\"ß", "'T", "́!!('", "ree", "-'", "S", "'re", "ḋ", "̣Z", "
", "'VE"]} +{"text": "漢ꟲ🙂İ👍🏽\r\n\r\n字' A😀🏽(ꟲ𐞁<|endoftext|>AZéa👍🏽٣٤٥٦\r", "tokens": 63, "pieces": ["漢ꟲ", "🙂İ", "👍🏽\r\n\r\n", "字", "'", " A", "😀🏽(", "ꟲ𐞁", "<|", "endoftext", "|>", "AZéa", "👍🏽", "٣٤٥", "٦", "\r"]} +{"text": "12345678​
\n‍­9½­'re<|endoftext|>İ👍🏽fi'ſé½'ſ​\u000b㋿0
字é<|endoftext|>d३0A㍿­٣٤٥٦12345678>EOT-🙂", "tokens": 83, "pieces": ["123", "456", "78", "​", "
\n", "‍­", "9½", "­'", "re", "<|", "endoftext", "|>", "İ", "👍🏽", "fi", "'ſ", "e", "́", "½", "'ſ", "​", "\u000b", "㋿", "0", "
字é", "<|", "endoftext", "|>", "d", "३0", "A", "㍿­", "٣٤٥", "٦12", "345", "678", ">EOT", "-🙂"]} +{"text": "عt\"é<|fim_prefix|>Dž!!é㋿'VEé½\n-٣٤٥٦", "tokens": 35, "pieces": ["عt", "\"e", "́<|", "fim", "_prefix", "|>", "Dž", "!!", "e", "́㋿'", "VEe", "́", "½", "\n", "-", "٣٤٥", "٦"]} +{"text": "İ㋿'T'VE'SⅣ0EOT٣٤٥٦,.٣٤٥٦A½", "tokens": 34, "pieces": ["İ", "㋿'", "T", "'VE", "'S", "Ⅳ0", "EOT", "٣٤٥", "٦", ",.", "٣٤٥", "٦", "A", "½"]} +{"text": "ße'D३Z‍", "tokens": 7, "pieces": ["ße", "'D", "३", "Z", "‍"]} +{"text": "é<|fim_prefix|>9dåſ!!Z ", "tokens": 18, "pieces": ["e", "́<|", "fim", "_prefix", "|>", "9", "da", "̊ſ", "!!", "Z", " "]} +{"text": " 'SEOTDž\r\n\u000b
‍ع😀🏽\r\nع'sḍ̇å
", "tokens": 33, "pieces": [" '", "SEOTDž", "\r\n", "\u000b", "
", "‍ع", "😀🏽\r\n", "ع", "'", "sḋ", "̣a", "̊", "
"]} +{"text": "𐞁㋿ḍ̇\n ½ſḍ̇m漢", "tokens": 28, "pieces": ["𐞁", "㋿ḋ", "̣\n", " ", "½", "ſḋ", "̣m漢", ""]} +{"text": "'T३!!9'T'MⅣ\r\n\r\n🙂½.ꟲ-Z…'s0漢عع\r'Re(ع'll123456780㋿(<|endoftext|>\r\n३é", "tokens": 47, "pieces": ["'T", "३", "!!", "9", "'T", "'M", "Ⅳ", "\r\n\r\n", "🙂", "½", ".ꟲ", "-Z", "…", "'s", "0", "漢عع", "\r", "'Re", "(ع", "'ll", "123", "456", "780", "㋿(<|", "endoftext", "|>\r\n", "३", "é"]} +{"text": "(
\"
>", "tokens": 7, "pieces": ["(", "
", "\"", "
", ">"]} +{"text": "漢́\r\n\r\n\t́字字a 'Dé", "tokens": 16, "pieces": ["漢", "́\r\n\r\n", "\t", "́", "字字a", " ", " '", "Dé"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " Dž 'refi\r'\u000b<|endoftext|>Z\t0", "tokens": 20, "pieces": [" Dž", " '", "refi", "\r", "'", "\u000b", "<|", "endoftext", "|>", "Z", "\t", "0"]} +{"text": "'Re㋿İ-éEOTAſa‍字\"㍿…İع​'VE'VE㋿Aḍ̇ꟲ 's.👍🏽>\r\n\r\n३ \n\r\n…12345678ad \n㍿", "tokens": 70, "pieces": ["'", "Re", "㋿İ", "-e", "́EOTAſa", "‍字", "\"㍿", "…İع", "​'", "VE", "'VE", "㋿Aḋ", "̣ꟲ", " '", "s", ".👍🏽>\r\n\r\n", "३", " \n\r\n", "…", "123", "456", "78", "ad", " \n", "㍿"]} +{"text": "\",\r\n\r\nḍ̇\r\n\r\n<|endoftext|>(tİ<|fim_prefix|>'re\r\n\r\n🙂'llea\u000bAe", "tokens": 39, "pieces": ["\",\r\n\r\n", "ḋ", "̣\r\n\r\n", "<|", "endoftext", "|>(", "tİ", "<|", "fim", "_prefix", "|>'", "re", "\r\n\r\n", "🙂'", "llea", "\u000b", "Ae"]} +{"text": "sZ0's😀🏽
३", "tokens": 13, "pieces": ["sZ", "0", "'s", "😀🏽", "
", "३"]} +{"text": "\rs字 \n漢'D​m𐞁e0,Z0½", "tokens": 18, "pieces": ["\r", "s字", " \n", "漢", "'D", "​m𐞁e", "0", ",Z", "0½"]} +{"text": "12345678 \r\n'Re>ſ\ne A\t­ \n…mAfi#$%éå'", "tokens": 30, "pieces": ["123", "456", "78", " \r\n", "'Re", ">ſ", "\n", "e", " A", "\t", "­", " \n", "…mAfi", "#$%", "éa", "̊'"]} +{"text": "<|fim_prefix|> \na\r\n\r\n'.'𐞁🙂३'re.0\r\n\r\n👍🏽,<|endoftext|>< éDžé'T𐞁㍿\r'll.𐞁👍🏽\u000bİ‍", "tokens": 67, "pieces": ["<|", "fim", "_prefix", "|>", " \n", "a", "\r\n\r\n", "'.'", "𐞁", "🙂", "३", "'re", ".", "0", "\r\n\r\n", "👍🏽,<|", "endoftext", "|><", " e", "́Džé", "'T", "𐞁", "㍿\r", "'ll", ".𐞁", "👍🏽", "\u000bİ", "‍"]} +{"text": "‍'ſ
 'ſ​!!.EOT عİ#$%\r\n\r\n漢's'Sfi", "tokens": 27, "pieces": ["‍'", "ſ", "
 ", " '", "ſ", "​!!.", "EOT", " ", " عİ", "#$%\r\n\r\n", "漢", "'s", "'S", "fi"]} +{"text": "! \n#$%ع'llmꟲDž'VE\t\r\n\r\n-½­'VE𐞁EOT're's🙂", "tokens": 30, "pieces": ["!", " \n", "#$%", "ع", "'ll", "mꟲDž", "'VE", "\t\r\n\r\n", "-", "½", "­'", "VE𐞁EOT", "'re", "'s", "🙂"]} +{"text": "fi\nⅣ'Re㍿!!­", "tokens": 11, "pieces": ["fi", "\n", "Ⅳ", "'Re", "㍿!!­"]} +{"text": "'M!!a0'VE\r\n\r\nß漢(Ⅳm ع>(!", "tokens": 18, "pieces": ["'M", "!!", "a", "0", "'VE", "\r\n\r\n", "ß漢", "(", "Ⅳ", "m", " ", " ع", ">(!"]} +{"text": "'VE#$%efi're.'re'ſ'T'VEḍ̇㋿'S㋿<|endoftext|>٣٤٥٦👍🏽", "tokens": 50, "pieces": ["'VE", "#$%", "efi", "'re", ".'", "re", "'ſ", "'T", "'VE", "ḋ", "̣㋿'", "S", "㋿<|", "endoftext", "|>", "٣٤٥", "٦", "👍🏽"]} +{"text": "<|endoftext|>'ſ 'T0'll𐞁<|endoftext|>'S<|endoftext|>'s👍🏽🙂👍🏽9d🙂ßعZ, 's<|endoftext|>😀🏽'T's\u000bŹ \n<<<\ré𐞁", "tokens": 86, "pieces": ["<|", "endoftext", "|>'", "ſ", " ", " '", "T", "0", "'ll", "𐞁", "<|", "endoftext", "|>'", "S", "<|", "endoftext", "|>'", "s", "👍🏽🙂👍🏽", "9", "d", "🙂ßعZ", ",", " ", " '", "s", "<|", "endoftext", "|>😀🏽'", "T", "'s", "\u000bZ", "́", " \n", "<<<\r", "e", "́𐞁"]} +{"text": "fi'ſ.#$%\r\n\r\n'Md‍åsfié !!å\r12345678! <|fim_prefix|>é\"漢字漢\né½> \r㍿", "tokens": 55, "pieces": ["fi", "'ſ", ".#$%\r\n\r\n", "'M", "d", "‍a", "̊sfié", " ", "!!", "a", "̊\r", "123", "456", "78", "!", " ", "<|", "fim", "_prefix", "|>", "é", "\"漢字漢", "\n", "e", "́", "½", ">", " \r", "㍿"]} +{"text": "e'ReⅣm 'T#$%👍🏽\u000b,", "tokens": 17, "pieces": ["e", "'Re", "Ⅳ", "m", " ", "'T", "#$%👍🏽", "\u000b", ","]} +{"text": "'ſع9ꟲfi", "tokens": 10, "pieces": ["'ſ", "ع", "9", "ꟲfi"]} +{"text": "(å字téعⅣé,", "tokens": 12, "pieces": ["(a", "̊字téع", "Ⅳ", "e", "́,"]} +{"text": "<́'Re9'㍿ ½\"<|fim_prefix|>\r\n.<|endoftext|>  're\r\n\r\nms٣٤٥٦é😀🏽'VE漢ßdd́>sḍ̇é'Sm", "tokens": 61, "pieces": ["<́'", "Re", "9", "'㍿", " ", "½", "\"<|", "fim", "_prefix", "|>\r\n", ".<|", "endoftext", "|>", " ", " ", "'re", "\r\n\r\n", "ms", "٣٤٥", "٦", "é", "😀🏽'", "VE漢ßdd", "́>", "sḋ", "̣é", "'S", "m"]} +{"text": ">‍\n'ſ- Z字'ſſ12345678åꟲعs'llDž\r\n\r\nfiⅣß'Mé👍🏽Dž漢㋿👍🏽'S\u000b'Så", "tokens": 65, "pieces": [">‍\n", "'ſ", "-", " Z字", "'ſ", "ſ", "123", "456", "78", "a", "̊ꟲعs", "'ll", "Dž", "\r\n\r\n", "fi", "Ⅳ", "ß", "'M", "e", "́👍🏽", "Dž漢", "㋿👍🏽'", "S", "\u000b", "'S", "a", "̊"]} +{"text": "漢ꟲ…\r\n\r\nés'Sİ12345678ſaع'MEOTⅣ
'Tꟲ", "tokens": 29, "pieces": ["漢ꟲ", "…\r\n\r\n", "és", "'S", "İ", "123", "456", "78", "ſaع", "'M", "EOT", "Ⅳ", "
", "'T", "ꟲ"]} +{"text": "Ⅳ 'VE>ḍ̇\n'S\n🙂<|endoftext|>.Dž''MZ㍿'ll\r\n\r\né‍12345678٣٤٥٦😀🏽'llé'Mt㋿'ll\n\tEOTß", "tokens": 66, "pieces": ["Ⅳ", " ", " '", "VE", ">ḋ", "̣\n", "'S", "\n", "🙂<|", "endoftext", "|>.", "Dž", "''", "MZ", "㍿'", "ll", "\r\n\r\n", "é", "‍", "123", "456", "78٣", "٤٥٦", "😀🏽'", "lle", "́'", "Mt", "㋿'", "ll", "\n", "\tEOTß"]} +{"text": "ḍ̇fi", "tokens": 7, "pieces": ["ḋ", "̣fi"]} +{"text": "åDžßDžⅣ.'Ré-\"12345678fifi!!<|endoftext|>", "tokens": 32, "pieces": ["a", "̊DžßDž", "Ⅳ", ".'", "Re", "́-\"", "123", "456", "78", "fifi", "!!<|", "endoftext", "|>"]} +{"text": "½<|fim_prefix|>​12345678a", "tokens": 13, "pieces": ["½", "<|", "fim", "_prefix", "|>​", "123", "456", "78", "a"]} +{"text": "e EOTß👍🏽<'VE𐞁(<|fim_prefix|>#$%<ḍ̇'Re're#$%#$%å'ſſ́\rå字 \n>'M're\t", "tokens": 64, "pieces": ["e", " ", " EOTß", "👍🏽<'", "VE𐞁", "<", "EOT", ">(<|", "fim", "_prefix", "|>#$%<", "ḋ", "̣'", "Re", "'re", "#$%#$%", "a", "̊'", "ſſ", "́\r", "a", "̊字", " \n", ">'", "M", "'re", "\t"]} +{"text": "
ḍ̇'s\r\n0 \n㍿'T!'T,'ll🙂<|endoftext|>Dž!'s🙂's'll'VE's'\u000b", "tokens": 44, "pieces": ["
ḋ", "̣'", "s", "\r\n", "0", " \n", "㍿'", "T", "!'", "T", ",'", "ll", "🙂<|", "endoftext", "|>", "Dž", "!'", "s", "🙂'", "s", "'ll", "'VE", "'s", "'", "\u000b"]} +{"text": "\u000b\t👍🏽'll#$%m<٣٤٥٦‍\t 'D😀🏽<\n​­EOT३.㋿!㋿Z", "tokens": 48, "pieces": ["\u000b", "\t", "👍🏽'", "ll", "#$%", "m", "<", "٣٤٥", "٦", "‍", "\t ", " '", "D", "😀🏽<\n", "​­", "EOT", "३", ".㋿!㋿", "Z"]} +{"text": "(9'll", "tokens": 3, "pieces": ["(", "9", "'ll"]} +{"text": "🙂
…ſd#$%\u000b'VEé'Re\u000ba(<|endoftext|>mÁ🙂字​'S \r\n'llß'llⅣ", "tokens": 44, "pieces": ["🙂", "
", "…ſd", "#$%", "\u000b", "'VE", "e", "́'", "Re", "\u000ba", "(<|", "endoftext", "|>", "mA", "́🙂", "字", "​'", "S", " \r\n", "'ll", "ß", "'ll", "Ⅳ"]} +{"text": " EOT'Re\u000b0字", "tokens": 6, "pieces": [" EOT", "'Re", "\u000b", "0", "字"]} +{"text": "Z. 
9عt 'Rea🙂字m!<㍿ꟲḍ̇fi'D ḍ̇9're('ſß\r\n\r\nss", "tokens": 51, "pieces": ["Z", ".<", "EOT", ">", " ", "
", "9", "عt", " '", "Rea", "🙂字m", "!<㍿", "ꟲḋ", "̣fi", "'D", " ḋ", "̣", "9", "'re", "('", "ſß", "\r\n\r\n", "ss", ""]} +{"text": "ⅣEOT<|endoftext|>fi'll<|fim_prefix|>e \tZ 'T\"ꟲ字,EOT漢Dž\"\r\n'ſ\r\nſ><\r\n\r\n'S­👍🏽\råé#$%", "tokens": 64, "pieces": ["Ⅳ", "EOT", "<|", "endoftext", "|>", "fi", "'ll", "<|", "fim", "_prefix", "|>", "e", " ", "\tZ", " ", "'T", "\"ꟲ字", ",EOT漢", "Dž", "\"\r\n", "'ſ", "\r\n", "ſ", "><\r\n\r\n", "'S", "­👍🏽\r", "a", "̊é", "#$%"]} +{"text": "0'ſ㋿'T >\n㍿'re字!!'reDž\t👍🏽 dd-​‍0İ ", "tokens": 37, "pieces": ["0", "'ſ", "㋿'", "T", " ", ">\n", "㍿'", "re字", "!!'", "reDž", "\t", "👍🏽", " dd", "-​‍", "0", "İ", " "]} +{"text": "\r-'>12345678\"!!漢㍿٣٤٥٦­AmAع", "tokens": 28, "pieces": ["\r", "-'>", "123", "456", "78", "\"!!", "漢", "㍿", "٣٤٥", "٦", "­AmAع"]} +{"text": "\u000b٣٤٥٦\n'D😀🏽Z'S\"'ſ٣٤٥٦​9\r\n\r\n(#$%!!🙂ع<>İt३'ḍ̇  ́Ⅳ㍿㍿", "tokens": 65, "pieces": ["\u000b", "٣٤٥", "٦", "\n", "'D", "😀🏽", "Z", "'S", "\"'", "ſ", "٣٤٥", "٦", "​", "9", "\r\n\r\n", "(#$%<", "EOT", ">!!🙂", "ع", "<>", "İt", "३", "'ḋ", "̣", " ", " ", "́", "Ⅳ", "㍿㍿"]} +{"text": "'s\r\n", "tokens": 5, "pieces": ["'", "s", "\r\n"]} +{"text": "… 'ſ​>Dž,'s<'Re<|fim_prefix|>fißß<'S'VE<|fim_prefix|>t🙂Ⅳ\u000b\r'VE<|fim_prefix|>e,a<|endoftext|>字AİDž", "tokens": 75, "pieces": ["… ", " '", "ſ", "​>", "Dž", ",'", "s", "<'", "Re", "<|", "fim", "_prefix", "|>", "fißß", "<'", "S", "'VE", "<|", "fim", "_prefix", "|>", "t", "🙂<", "EOT", ">", "Ⅳ", "\u000b\r", "'VE", "<|", "fim", "_prefix", "|>", "e", ",a", "<|", "endoftext", "|>", "字AİDž"]} +{"text": "漢's'T𐞁\u000b\t12345678'D9ſḍ̇9'VE'S㋿Z👍🏽d­漢éd", "tokens": 46, "pieces": ["漢", "'s", "'T", "𐞁", "\u000b", "\t", "123", "456", "78", "'", "D", "9", "ſḋ", "̣", "9", "'VE", "'S", "㋿Z", "👍🏽", "d", "­漢e", "́d"]} +{"text": "é٣٤٥٦­#$%ḍ̇½ḍ̇ ㍿ḍ̇\t­½٣٤٥٦'Re𐞁EOTع<|fim_prefix|><|fim_prefix|>!Ⅳfié<|fim_prefix|>\r…0#$%​'M \nt're", "tokens": 95, "pieces": ["é", "٣٤٥", "٦", "­#$%", "ḋ", "̣", "½", "ḋ", "̣", " ", " ㍿", "ḋ", "̣", "\t", "­", "½٣٤", "٥٦", "'Re", "𐞁EOTع", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>!", "Ⅳ", "fie", "́<|", "fim", "_prefix", "|>\r", "…", "0", "#$%​'", "M", " \n", "t", "'re"]} +{"text": "\n­Ⅳ\r\n\r\n 'M,
\r\n\r\n,é e'Sfi👍🏽", "tokens": 25, "pieces": ["\n", "­", "Ⅳ", "\r\n\r\n", " ", "'M", ",", "
", "\r\n\r\n", ",é", " e", "'S", "fi", "👍🏽"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\u000b(́'res ꟲm\r\néé\r\n\r\nḍ̇!.'S字
\r\nfia'ſDž'T's­'VEm३d字'll", "tokens": 42, "pieces": ["👍🏽", "123", "456", "78३", "­​<", "META", "_START", ">.'", "S字", "
\r\n", "fia", "'ſ", "Dž", "'T", "'s", "­'", "VEm", "३", "d字", "'ll"]} +{"text": "<\r\n\r\nå<'Tfi9#$%Dž\r\n😀🏽🙂‍😀🏽\t,'T.'s\u000b字e#$% \u000b t9٣٤٥٦\t'ſ \n", "tokens": 63, "pieces": ["<<", "META", "_START", ">\r\n\r\n", "a", "̊<'", "Tfi", "9", "#$%<", "EOT", ">Dž", "\r\n", "😀🏽🙂‍😀🏽", "\t", ",'", "T", ".'", "s", "\u000b字e", "#$%", " \u000b ", " t", "9٣٤", "٥٦", "\t", "'ſ", " \n"]} +{"text": ">\ń٣٤٥٦-a‍<|endoftext|>\u000bé'T12345678Dž0eß\r\nfi\r\n\r\n ß́́ ⅣDž<|fim_prefix|>t​…åZ​a<|fim_prefix|>", "tokens": 74, "pieces": [">\n", "́", "٣٤٥", "٦", "-a", "‍<|", "endoftext", "|>", "\u000be", "́'", "T", "123", "456", "78", "Dž", "0", "eß", "\r\n", "fi", "\r\n\r\n", " ß", "́<", "META", "_START", ">́", " ", "Ⅳ", "Dž", "<|", "fim", "_prefix", "|>", "t", "​", "…a", "̊Z", "​a", "<|", "fim", "_prefix", "|>"]} +{"text": "­..😀🏽're漢\t\t\r\n😀🏽\u000bß'VE𐞁é漢<|endoftext|>​Ⅳ'Ś…<|fim_prefix|>(Ⅳ\tm३", "tokens": 57, "pieces": ["­..😀🏽'", "re漢", "\t\t\r\n", "😀🏽", "\u000bß", "'VE", "𐞁é漢", "<|", "endoftext", "|>​", "Ⅳ", "'", "S", "́", "…", "<|", "fim", "_prefix", "|>(", "Ⅳ", "\tm", "३"]} +{"text": "😀🏽漢😀🏽 's-\"\r'ſDž0#$%!!👍🏽! Dž㋿'<|endoftext|>'VEع m\r字Ⅳ٣٤٥٦>Dž½<​<|fim_prefix|>", "tokens": 84, "pieces": ["😀🏽", "漢", "😀🏽", " ", " '", "s", "-\"\r", "'ſ", "Dž", "0", "#$%!!👍🏽!<", "META", "_START", ">", " Dž", "㋿'<|", "endoftext", "|>'", "VEع", " ", " m", "\r", "字", "Ⅳ٣٤", "٥٦", ">Dž", "½", "<​<", "EOT", "><|", "fim", "_prefix", "|><", "EOT", ">"]} +{"text": "‍#$%éİ
३>0\"'VEé", "tokens": 16, "pieces": ["‍#$%", "éİ", "
", "३", ">", "0", "\"'", "VEe", "́"]} +{"text": "‍,'res'VE‍DžⅣ's٣٤٥٦\r\n\r\n'Re 🙂'M\r\n\r\n\rdſ \nZ½é", "tokens": 36, "pieces": ["‍,'", "res", "'VE", "‍Dž", "Ⅳ", "'s", "٣٤٥", "٦", "\r\n\r\n", "'Re", " ", " 🙂'", "M", "\r\n\r\n\r", "dſ", " \n", "Z", "½", "é"]} +{"text": "'re!ع\rmDž😀🏽#$%…é
\nİ\"é'VEß(sd'VE <漢­ꟲ\r\n\r\nså'M", "tokens": 44, "pieces": ["'re", "!ع", "\r", "mDž", "😀🏽#$%", "…é", "
\n", "İ", "\"e", "́'", "VEß", "(sd", "'VE", " ", "<漢", "­ꟲ", "\r\n\r\n", "sa", "̊'", "M"]} +{"text": "'D, \r\n\r\nعꟲ'ſ", "tokens": 10, "pieces": ["'D", ",", " \r\n\r\n", "عꟲ", "'ſ"]} +{"text": "漢<|fim_prefix|>aéع \na'VE­‍ſDž字­t'\te𐞁0́'VEsİ
ß字'll!İ(-ꟲ", "tokens": 56, "pieces": ["漢", "<|", "fim", "_prefix", "|>", "ae", "́ع", " \n", "a", "'VE", "­‍", "ſDž", "字", "­t", "'", "\te𐞁", "0", "́'", "VEsİ", "
ß字", "'ll", "!İ", "(-<", "EOT", ">ꟲ"]} +{"text": "ßİꟲ😀🏽'\r\ne<|fim_prefix|>Dž٣٤٥٦t(ſ'ReA", "tokens": 36, "pieces": ["ßİꟲ", "😀🏽'\r\n", "e", "<|", "fim", "_prefix", "|>", "Dž", "٣٤٥", "٦", "t", "(ſ", "'Re", "A"]} +{"text": "'ll字,aß'Re½ A", "tokens": 8, "pieces": ["'ll", "字", ",aß", "'Re", "½", " A"]} +{"text": "Ⅳ's漢İ
(𐞁0ée\r\n\r\nd½\rḍ̇9!9
​,٣٤٥٦\r\nع'ſ३'Re", "tokens": 49, "pieces": ["Ⅳ", "'s", "漢İ", "
", "(𐞁", "0", "e", "́e", "\r\n\r\n", "d", "½", "\r", "ḋ", "̣", "9", "!", "9", "
", "​,", "٣٤٥", "٦", "\r\n", "ع", "'ſ", "३", "'Re"]} +{"text": "e'M 9 (…'Re!Z<|fim_prefix|>'D", "tokens": 22, "pieces": ["e", "'M", " ", "9", " ", "(", "…", "'Re", "!", "Z", "<|", "fim", "_prefix", "|>'", "D"]} +{"text": "Z \r😀🏽‍'VE字A😀🏽-\t#$%'>́Dž\r-\r\n're漢'ſ", "tokens": 36, "pieces": ["Z", " \r", "😀🏽‍'", "VE字A", "😀🏽-", "\t", "#$%'>́", "Dž", "\r", "-\r\n", "'re", "漢", "'ſ"]} +{"text": "é
٣٤٥٦ 'M🙂fim 'S<", "tokens": 22, "pieces": ["é", "
", "٣٤٥", "٦", " ", " '", "M", "🙂fim", " '", "S", "<"]} +{"text": "å
🙂ⅣDž\n👍🏽s's'res \n's'll!fi😀🏽​a're", "tokens": 36, "pieces": ["a", "̊", "
", "🙂", "Ⅳ", "Dž", "\n", "👍🏽", "s", "'s", "'re", "s", " \n", "'s", "'ll", "!fi", "😀🏽​", "a", "'re"]} +{"text": "9EOT‍ 're\"<|endoftext|>३t'T\n.İ\n'T́s🙂<'D\ta", "tokens": 32, "pieces": ["9", "EOT", "‍", " ", "'re", "\"<|", "endoftext", "|>", "३", "t", "'T", "\n", ".İ", "\n", "'T", "́s", "🙂<'", "D", "\ta"]} +{"text": "'ſ'llⅣ", "tokens": 9, "pieces": ["'ſ", "'ll", "Ⅳ", ""]} +{"text": "🙂ß𐞁'll<|endoftext|> \n<|endoftext|>A<|fim_prefix|>'M字㍿'Re's!!!fiEOTⅣ", "tokens": 51, "pieces": ["🙂ß𐞁", "'ll", "<|", "endoftext", "|>", " \n", "<|", "endoftext", "|>", "A", "<|", "fim", "_prefix", "|>'", "M字", "㍿'", "Re", "'s", "!<", "META", "_START", ">!!", "fiEOT", "Ⅳ"]} +{"text": "\u000bé‍'re
é́漢  t<|endoftext|>'Re12345678\"­ ́\u000b字e\r'll", "tokens": 50, "pieces": ["\u000be", "́‍'", "re", "
é", "́漢", " ", " t", "<|", "endoftext", "|>'", "Re", "123", "456", "78", "\"­", " ́<", "META", "_START", "><", "EOT", ">㋿<", "META", "_START", ">", "\u000b字e", "\r", "'ll"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "fi'Dİ'reéåع\"Ⅳ(téİ", "tokens": 20, "pieces": ["fi", "'D", "İ", "'re", "e", "́<", "EOT", ">a", "̊ع", "\"", "Ⅳ", "(téİ"]} +{"text": "9å'reⅣ're­\u000b'Téd , 字😀🏽'Såع٣٤٥٦ḍ̇\r,½ꟲ漢‍ꟲ…Ⅳ's", "tokens": 72, "pieces": ["9", "a", "̊'", "re", "Ⅳ", "'re", "­", "\u000b", "'T", "e", "́d", " ", ",", " 字", "😀🏽'", "Sa", "̊ع", "", "٣٤٥", "٦", "ḋ", "̣\r", ",", "½", "ꟲ漢", "‍ꟲ", "…", "Ⅳ", "'s"]} +{"text": "㋿‍!!​'re-EOT\r\n\r\n-​,'ll​éd\n!ß­ m'Rea'ſ \r'Red…0", "tokens": 44, "pieces": ["㋿‍!!​'", "re", "-EOT", "\r\n\r\n", "-​,'", "ll", "​e", "́d", "\n", "!", "ß", "­", " m", "'Re", "a", "'ſ", " \r", "'Re", "d", "…", "0"]} +{"text": "'M'ſ'D½\t…𐞁….\n12345678İ
.'M‍㍿<|fim_prefix|>", "tokens": 39, "pieces": ["'M", "'ſ", "'D", "½", "\t", "…𐞁", "…", ".\n", "123", "456", "78", "İ", "
", ".'", "M", "‍㍿<|", "fim", "_prefix", "|>"]} +{"text": "ß\ra
9😀🏽'ſ>…😀🏽Ⅳ­'T<㋿İꟲ‍'ſ 0'll.t", "tokens": 44, "pieces": ["ß", "\r", "a", "
", "9", "😀🏽'", "ſ", ">", "…", "😀🏽", "Ⅳ", "­'", "T", "<㋿", "İꟲ", "‍'", "ſ", " ", "0", "'ll", ".t"]} +{"text": "'sſ'Re\"\n123456780'ſa ", "tokens": 13, "pieces": ["'s", "ſ", "'Re", "\"\n", "123", "456", "780", "'ſ", "a", " "]} +{"text": "\r\n 漢t ḍ̇😀🏽sꟲ'Dt\rfi\" 漢td ", "tokens": 32, "pieces": ["\r\n", " 漢t", " ḋ", "̣😀🏽", "sꟲ", "'D", "t", "\r", "fi", "\"", " ", " 漢td", " "]} +{"text": "ع㋿Z'M𐞁 \n…'ll'VE,e.Džs", "tokens": 21, "pieces": ["ع", "㋿Z", "'M", "𐞁", " \n", "…", "'ll", "'VE", ",e", ".Džs"]} +{"text": " 'DéA<\r\n\r\n>Dž!d字#$%'D!½e‍­s!<\réAعfi'Re", "tokens": 32, "pieces": [" '", "DéA", "<\r\n\r\n", ">Dž", "!d字", "#$%'", "D", "!", "½", "e", "‍­", "s", "!<\r", "e", "́Aعfi", "'Re"]} +{"text": "(aſİ'ſ
字A\"३e9!\r ½
's​\r\n\r\n'VE", "tokens": 29, "pieces": ["(aſİ", "'ſ", "
字A", "\"", "३", "e", "9", "!\r", " ", " ", "½", "
", "'s", "​\r\n\r\n", "'VE"]} +{"text": "A \nꟲ'T\"fis
's𐞁🙂å\r\na\r\n\r\n", "tokens": 26, "pieces": ["A", " \n", "ꟲ", "'T", "\"fis", "
", "'s", "𐞁", "🙂a", "̊\r\n", "a", "\r\n\r\n"]} +{"text": "m 'VE-'.👍🏽EOTé\r\nétå", "tokens": 21, "pieces": ["m", " ", "'VE", "-'.👍🏽", "EOTé", "\r\n", "e", "́ta", "̊"]} +{"text": "'S'D-ſ‍#$%'Dß́'S s!'MZé>é", "tokens": 21, "pieces": ["'S", "'D", "-ſ", "‍#$%'", "Dß", "́'", "S", " ", " s", "!'", "MZé", ">é"]} +{"text": "'ll,'VEs'VE're​a​\"-Zع'Séå…ꟲ字", "tokens": 25, "pieces": ["'ll", ",'", "VEs", "'VE", "'re", "​a", "​\"-", "Zع", "'S", "e", "́a", "̊", "…ꟲ字"]} +{"text": "\n \n s mfi !\t'Dt!!ꟲ,dİ12345678㍿…'Reع<|endoftext|>​漢ع😀🏽'ſ  m", "tokens": 49, "pieces": ["\n \n", " ", " s", " ", " mfi", " !", "\t", "'D", "t", "!!", "ꟲ", ",dİ", "123", "456", "78", "㍿", "…", "'Re", "ع", "<|", "endoftext", "|>​", "漢ع", "😀🏽'", "ſ", " ", " m"]} +{"text": "㍿ßßm'DDž ٣٤٥٦'Re
 𐞁(½\n \n٣٤٥٦<|endoftext|>å'Mß́0\"…,'Re'M(fi<|fim_prefix|>EOT㋿Ⅳ٣٤٥٦", "tokens": 86, "pieces": ["㍿ßßm", "'D", "Dž", " ", " ", "٣٤٥", "٦", "'Re", "
", " 𐞁", "(", "½", "\n \n", "٣٤٥", "٦", "<|", "endoftext", "|>", "a", "̊'", "Mß", "́", "0", "\"", "…", ",'", "Re", "'M", "(", "fi", "<|", "fim", "_prefix", "|>", "EOT", "㋿", "Ⅳ٣٤", "٥٦"]} +{"text": "(ꟲé <|fim_prefix|> \n<|fim_prefix|> 🙂😀🏽fi0a>\n٣٤٥٦éꟲ😀🏽́é", " \n", "<|", "fim", "_prefix", "|>", " ", " 🙂😀🏽", "fi", "0", "a", ">\n", "٣٤٥", "٦", "éꟲ", "😀🏽́", "é", "½12345678…😀🏽Ⅳ<|endoftext|>\"३ع㍿té
d​m½", "tokens": 53, "pieces": ["…", "'ll", "éع", " \n", "<'", "re", "Ⅳ", "<ß", "<|", "fim", "_prefix", "|>", "½12", "345", "678", "…", "😀🏽", "Ⅳ", "<|", "endoftext", "|>\"", "३", "ع", "㍿te", "́", "
d", "​m", "½"]} +{"text": "٣٤٥٦Ⅳ ​Ⅳ'reſ́'Re٣٤٥٦éé<<|endoftext|> <|fim_prefix|>عåⅣ😀🏽 ḍ̇'D'D
٣٤٥٦㍿'s'VEs'Re👍🏽0'VEå\r\n>字", "tokens": 98, "pieces": ["٣٤٥", "٦Ⅳ", " ", " ​", "Ⅳ", "'re", "ſ", "́'", "Re", "٣٤٥", "٦", "ée", "́<<|", "endoftext", "|>", " ", " <|", "fim", "_prefix", "|>", "عa", "̊", "Ⅳ", "😀🏽", " ", " ḋ", "̣'", "D", "'D", "
", "٣٤٥", "٦", "㍿'", "s", "'VE", "s", "'Re", "👍🏽", "0", "'VE", "a", "̊\r\n", ">字"]} +{"text": "#$%!!㋿Z½٣٤٥٦ḍ̇\tAtعd㋿漢'Téع9A'é\né­#$%Dž…", "tokens": 53, "pieces": ["#$%!!㋿", "Z", "½٣٤", "٥٦", "ḋ", "̣", "\tAtعd", "㋿<", "META", "_START", ">漢", "'T", "e", "́ع", "9", "A", "'é", "\n", "é", "­#$%", "Dž", "…"]} +{"text": "m
ḍ̇ع're're漢 \n\r\nAⅣ\n'res<|fim_prefix|>‍عé12345678", "tokens": 37, "pieces": ["m", "
ḋ", "̣ع", "'re", "'re", "漢", " \n\r\n", "A", "Ⅳ", "\n", "'re", "s", "<|", "fim", "_prefix", "|>‍", "عe", "́", "123", "456", "78"]} +{"text": "é'SDž'sعß'T.", "tokens": 9, "pieces": ["é", "'S", "Dž", "'s", "عß", "'T", "."]} +{"text": "\" --Dž-‍mA\r\n\n\" \n're", "tokens": 17, "pieces": ["\"", " ", "--", "Dž", "-‍", "mA", "\r\n\n", "\"", " \n", "'re"]} +{"text": "\u000bmé0‍­'S㍿­३𐞁Dž‍", "tokens": 23, "pieces": ["\u000bmé", "0", "‍­'", "S", "㍿­", "३", "𐞁Dž", "‍"]} +{"text": "!!字㍿ Ⅳ #$% !!漢<|fim_prefix|>😀🏽,
\"ꟲEOT  Z\"'Tعå٣٤٥٦0>😀🏽-tfi\t㋿😀🏽", "tokens": 76, "pieces": ["!!", "字", "㍿", " ", "Ⅳ", " ", " #$%", " ", "!!", "漢", "<|", "fim", "_prefix", "|>😀🏽,", "
", "\"ꟲEOT", " ", " Z", "\"'", "Tعa", "̊", "٣٤٥", "٦0", "><", "EOT", ">😀🏽-", "tfi", "\t", "㋿😀🏽"]} +{"text": "EOT'RedⅣZ🙂('Re'D\rDž'VE\t\t.'T㋿\u000b​㍿​😀🏽é0٣٤٥٦\r\nZ-'re🙂fi漢½'re.ع", "tokens": 60, "pieces": ["EOT", "'Re", "d", "Ⅳ", "Z", "🙂('", "Re", "'D", "\r", "Dž", "'VE", "\t", "\t", ".'", "T", "㋿", "\u000b", "​㍿​😀🏽", "e", "́", "0٣٤", "٥٦", "\r\n", "Z", "-'", "re", "🙂fi漢", "½", "'re", ".ع"]} +{"text": "#$%½sé", "tokens": 5, "pieces": ["#$%", "½", "sé"]} +{"text": "\u000bat<İ…ß🙂عꟲ字're३,'T٣٤٥٦'Re…A\r\n ́t>\r\n's​
éDž9'Dع‍", "tokens": 57, "pieces": ["\u000bat", "<İ", "", "…ß", "🙂عꟲ字", "'re", "३", ",'", "T", "٣٤٥", "٦", "'Re", "…A", "\r\n", " ", " ́", "t", ">\r\n", "'s", "​<", "EOT", ">", "
éDž", "9", "'D", "ع", "‍"]} +{"text": "a漢\u000bİ​\n\r's!'Re'D👍🏽İ\r\n'VEİ'St ", "tokens": 26, "pieces": ["a漢", "\u000bİ", "​\n\r", "'s", "!'", "Re", "'D", "👍🏽", "İ", "\r\n", "'VE", "İ", "'S", "t", " "]} +{"text": "𐞁d'Sع㋿éݽ'S\u000bſ'T\"9ſ123456780\r\nß'smA,½ ", "tokens": 34, "pieces": ["𐞁d", "'S", "ع", "㋿éİ", "½", "'S", "\u000bſ", "'T", "\"", "9", "ſ", "123", "456", "780", "\r\n", "ß", "'s", "mA", ",", "½", " "]} +{"text": "́m#$%'T ('VE\r,Ⅳ😀🏽!!½ ́\t- 0'D३å‍>eſd", "tokens": 43, "pieces": ["́m", "#$%'", "T", " ", "('", "VE", "\r", ",", "Ⅳ", "😀🏽!!", "½", " ", " ́", "\t", "-", " ", "0", "'D", "३", "a", "̊‍>", "eſd"]} +{"text": " \n ", "tokens": 2, "pieces": [" \n "]} +{"text": "\t \n<|endoftext|>\r\n9,\r\n12345678EOT", "tokens": 15, "pieces": ["\t \n", "<|", "endoftext", "|>\r\n", "9", ",\r\n", "123", "456", "78", "EOT"]} +{"text": "'Dſ'S9३İ ,'T🙂\u000b're\t'e'Re<fi​🙂<|fim_prefix|>'D,‍<|fim_prefix|> -12345678字12345678é ꟲ", "tokens": 60, "pieces": ["'D", "ſ", "'S", "9", "", "३", "İ", " ,'", "T", "🙂", "\u000b", "'re", "\t", "'e", "'Re", "<fi", "​🙂<|", "fim", "_prefix", "|>'", "D", ",‍<|", "fim", "_prefix", "|>", " ", "-", "123", "456", "78", "字", "123", "456", "78", "e", "́", " ꟲ"]} +{"text": "(㋿\u000b0\r\n\r\n'M", "tokens": 8, "pieces": ["(㋿", "\u000b", "0", "\r\n\r\n", "'M"]} +{"text": "\t
'ſⅣ㍿ 字'D >", "tokens": 20, "pieces": ["\t", "
", "'ſ", "Ⅳ", "㍿", " ", "字", "'D", " ", ">"]} +{"text": "m\r㍿( Dž\n३-­\"\t's", "tokens": 20, "pieces": ["m", "\r", "㍿(", " Dž", "\n", "३", "-­\"<", "EOT", ">", "\t", "'s"]} +{"text": "३!İ🙂(,字d0​漢a", "tokens": 15, "pieces": ["३", "!İ", "🙂(,", "字d", "0", "​漢a"]} +{"text": "!!e \n<#$% a >ع \n​३\u000b'VE­\r", "tokens": 19, "pieces": ["!!", "e", " \n", "<#$%", " a", " >", "ع", " \n", "​", "३", "\u000b", "'VE", "­\r"]} +{"text": ",'Dž", "tokens": 3, "pieces": [",'", "Dž"]} +{"text": "A'ſ \n", "tokens": 6, "pieces": ["A", "'ſ", " \n"]} +{"text": "ſEOT0‍-éZع\r", "tokens": 12, "pieces": ["ſEOT", "0", "‍-", "éZع", "\r"]} +{"text": "'re EOTŹ'३عś\r.ع
… ß👍🏽", "tokens": 27, "pieces": ["'re", " EOTZ", "́'", "३", "عs", "́\r", ".ع", "
…", " ß", "👍🏽"]} +{"text": "'DEOTA'Re\r\n\u000b \n😀🏽ae,ſDžꟲ.㍿'llİ३𐞁A'M'VE…é'll9<|endoftext|><|endoftext|>'", "tokens": 59, "pieces": ["'D", "EOTA", "'Re", "\r\n\u000b \n", "😀🏽", "ae", ",ſDžꟲ", ".㍿'", "llİ", "३", "𐞁A", "'M", "'VE", "…é", "'ll", "9", "<|", "endoftext", "|><|", "endoftext", "|>'"]} +{"text": "Ⅳ('s\"!!'S#$%'llé'll\u000b ㋿!a's<|fim_prefix|>('re <|fim_prefix|>\r\né
½'Re\t½½'s!́'ll३ \n ", "tokens": 60, "pieces": ["Ⅳ", "('", "s", "\"!!'", "S", "#$%'", "ll", "e", "́'", "ll", "\u000b", " ㋿!", "a", "'s", "<|", "fim", "_prefix", "|>('", "re", " ", "<|", "fim", "_prefix", "|>\r\n", "e", "́", "
", "½", "'Re", "\t", "½½", "'s", "!́'", "ll", "३", " \n "]} +{"text": "e12345678", "tokens": 4, "pieces": ["e", "123", "456", "78"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "", "tokens": 0, "pieces": []} +{"text": "fi0‍ḍ̇<|endoftext|>ḍ̇>9 '<|fim_prefix|>\nß́'VE\r\n​(𐞁𐞁", "tokens": 50, "pieces": ["fi", "0", "‍ḋ", "̣<", "META", "_START", "><|", "endoftext", "|>", "ḋ", "̣>", "9", " ", " '<|", "fim", "_prefix", "|>\n", "ß", "́'", "VE", "\r\n", "​(", "𐞁𐞁"]} +{"text": "­-'D9👍🏽👍🏽é\"👍🏽e's<|fim_prefix|>12345678İt åé३'D >('VE\n", "tokens": 55, "pieces": ["­-<", "EOT", ">'", "D", "9", "👍🏽👍🏽", "e", "́\"👍🏽", "e", "'s", "<|", "fim", "_prefix", "|>", "123", "456", "78", "İt", " a", "̊e", "́", "३", "'D", " ", ">('", "VE", "\n"]} +{"text": "'lls\r\n\r\n \"'TA'Re٣٤٥٦sé 😀🏽", "tokens": 24, "pieces": ["'ll", "s", "\r\n\r\n", " \"'", "TA", "'Re", "٣٤٥", "٦", "se", "́", " ", "😀🏽"]} +{"text": "s\n'VE\rEOT\n", "tokens": 8, "pieces": ["s", "\n", "'VE", "\r", "EOT", "\n"]} +{"text": "fiⅣ
>", "tokens": 7, "pieces": ["fi", "Ⅳ", "
", ">"]} +{"text": "½!!عDžⅣⅣİ😀🏽'T \néß'Dm'Tعß!, ع>", "tokens": 32, "pieces": ["½", "!!", "عDž", "ⅣⅣ", "İ", "😀🏽'", "T", " \n", "éß", "'D", "m", "'T", "عß", "!,", " ع", ">"]} +{"text": " \r\n\r\nⅣ'Rea", "tokens": 5, "pieces": [" \r\n\r\n", "Ⅳ", "'Re", "a"]} +{"text": "#$%å's🙂㍿'Re\t😀🏽Z ­e‍'s漢're>!-'sß٣٤٥٦'s<|fim_prefix|>㍿0!! \ń<|endoftext|>👍🏽9m­A", "tokens": 84, "pieces": ["#$%", "a", "̊'", "s", "🙂㍿'", "Re", "\t", "😀🏽", "Z", "", " ", "­e", "‍'", "s漢", "'re", ">!-'", "sß", "٣٤٥", "٦", "'s", "<|", "fim", "_prefix", "|>㍿", "0", "!!", " \n", "́<|", "endoftext", "|>👍🏽", "9", "m", "­<", "META", "_START", ">A"]} +{"text": "A>㋿Ⅳ#$%­­'VEs\rİ'Md  \r\n…<'s.ſ٣٤٥٦EOT\"9​< t\u000b", "tokens": 50, "pieces": ["A", ">㋿", "Ⅳ", "#$%­­'", "VEs", "\r", "İ", "'M", "d", "  \r\n", "…", "<'", "s", ".", "ſ", "٣٤٥", "٦", "EOT", "\"", "9", "​<", " t", "\u000b"]} +{"text": "Aé-'\n", "tokens": 9, "pieces": ["Aé", "-'\n"]} +{"text": "<|fim_prefix|>a'Re<𐞁漢
½", "tokens": 22, "pieces": ["<|", "fim", "_prefix", "|>", "a", "'Re", "<<", "META", "_START", ">𐞁漢", "
", "½"]} +{"text": "<|endoftext|>٣٤٥٦Džéś9A 🙂s!<,㍿12345678\n'M'VE<|endoftext|>😀🏽", "tokens": 54, "pieces": ["<|", "endoftext", "|>", "٣٤٥", "٦", "Džés", "́", "9", "A", " ", "🙂s", "!<,㍿", "123", "456", "78", "\n", "'", "M", "'VE", "<|", "endoftext", "|>😀🏽"]} +{"text": "é🙂漢𐞁Ⅳfi'VEd.­ \n…\u000b㋿ \n𐞁㋿d\"EOT--'ſZ\tm", "tokens": 54, "pieces": ["e", "́🙂", "漢", "𐞁", "Ⅳ", "fi", "'VE", "d", ".­", " \n", "…", "\u000b", "㋿", " \n", "𐞁", "㋿d", "\"EOT", "--'", "ſZ", "\tm"]} +{"text": "'M'ret👍🏽ꟲ\u000bt12345678'M -٣٤٥٦!'re", "tokens": 30, "pieces": ["'M", "'re", "t", "👍🏽", "ꟲ", "\u000bt", "123", "456", "78", "'M", " ", " -", "٣٤٥", "٦", "!'", "re"]} +{"text": "\r\n\r\nİ'VE٣٤٥٦­éZ\t \n", "tokens": 20, "pieces": ["\r\n\r\n", "İ", "'VE", "٣٤٥", "٦", "­e", "́<", "EOT", ">Z", "\t \n"]} +{"text": "<|endoftext|>'s🙂𐞁​‍d", "tokens": 18, "pieces": ["<|", "endoftext", "|>'", "s", "🙂𐞁", "​‍", "d"]} +{"text": "'sßꟲ ,12345678㍿🙂're<|fim_prefix|>'MAſ'll㋿½e!,'ll‍ꟲⅣ>", "tokens": 50, "pieces": ["'", "sßꟲ", " ,", "123", "456", "78", "㍿🙂'", "re", "<|", "fim", "_prefix", "|>'", "MAſ", "'ll", "㋿", "½", "e", "!,'", "ll", "‍ꟲ", "Ⅳ", ">"]} +{"text": "<|endoftext|>\"'sfi", "tokens": 11, "pieces": ["<|", "endoftext", "|>\"'", "sfi"]} +{"text": "(m12345678㍿३'Re(İ'D( #$%sZ'llé0🙂‍­9", "tokens": 28, "pieces": ["(m", "123", "456", "78", "㍿", "३", "'Re", "(İ", "'D", "(", " ", "#$%", "sZ", "'ll", "é", "0", "🙂‍­", "9"]} +{"text": "ſ(字\t​!ſA٣٤٥٦😀🏽  ३'TfiA…'s㋿<'VE", "tokens": 44, "pieces": ["ſ", "(字", "\t", "​!", "ſA", "٣٤٥", "٦", "😀🏽", " ", " ", "३", "'T", "fiA", "…", "'s", "㋿<'", "VE"]} +{"text": "' d字😀🏽­Z<|fim_prefix|>Ⅳ
fi", "tokens": 24, "pieces": ["'", " ", " d字", "😀🏽­", "Z", "<|", "fim", "_prefix", "|>", "Ⅳ", "
fi"]} +{"text": "😀🏽㍿", "tokens": 8, "pieces": ["😀🏽㍿"]} +{"text": "\t\n<|endoftext|>m,\r\n\r\n\u000b🙂\"<|endoftext|>\na", "tokens": 23, "pieces": ["\t\n", "<|", "endoftext", "|><", "EOT", ">m", ",\r\n\r\n", "\u000b", "🙂\"<|", "endoftext", "|>\n", "a"]} +{"text": "s,\"…!!'res‍Dž ḍ̇漢ⅣDž12345678'D'M😀🏽👍🏽 \n'D 
", "tokens": 44, "pieces": ["s", ",\"", "…", "!!'", "res", "‍Dž", " ", " ḋ", "̣漢", "Ⅳ", "Dž", "123", "456", "78", "'D", "'M", "😀🏽👍🏽", " \n", "'D", " 
"]} +{"text": "ḍ̇㋿ع'M''M m漢\r 'D0s'TZEOTfi­", "tokens": 30, "pieces": ["ḋ", "̣㋿", "ع", "'M", "''", "M", " m漢", "\r", " ", "'D", "0", "s", "'T", "ZEOTfi", "­"]} +{"text": "🙂åe㋿'D….ḍ̇́\n.DžⅣḍ̇😀🏽'VE'S漢😀🏽A!!d(a𐞁 \nZ<|fim_prefix|>٣٤٥٦EOT", "tokens": 77, "pieces": ["🙂<", "EOT", ">a", "̊e", "㋿'", "D", "…", ".ḋ", "̣́\n", ".Dž", "Ⅳ", "ḋ", "̣😀🏽'", "VE", "'S", "漢", "😀🏽", "A", "!!", "d", "(a𐞁", " \n", "Z", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "EOT"]} +{"text": "‍ aé<|endoftext|>A٣٤٥٦ſ\u000bꟲ\nt'll㋿ß\"!!\nséå- ع'M漢\r\n\r\n 'DEOT㍿12345678½A", "tokens": 60, "pieces": ["‍", " aé", "<|", "endoftext", "|>", "A", "٣٤٥", "٦", "ſ", "\u000bꟲ", "\n", "t", "'ll", "㋿ß", "\"!!\n", "se", "́a", "̊-", " ع", "'M", "漢", "\r\n\r\n", " '", "DEOT", "㍿", "123", "456", "78½", "A"]} +{"text": "dét\u000b \n#$%\t#$%'S12345678\r\n𐞁\u000b'T‍ 'ree㍿㍿\"İA. \nd\r\n\r\nééaß\n", "tokens": 47, "pieces": ["dét", "\u000b \n", "#$%", "\t", "#$%'", "S", "123", "456", "78", "\r\n", "𐞁", "\u000b", "'T", "‍", " ", " '", "ree", "㍿㍿\"", "İA", ".", " \n", "d", "\r\n\r\n", "ée", "́<", "META", "_START", ">aß", "\n"]} +{"text": "12345678İ\u000b'VE\u000b<|fim_prefix|>99,👍🏽A'D😀🏽½ 👍🏽'll३-á'Reß'll", "tokens": 48, "pieces": ["123", "456", "78", "İ", "\u000b", "'VE", "\u000b", "<|", "fim", "_prefix", "|>", "99", ",👍🏽", "A", "'D", "😀🏽", "½", " ", " 👍🏽'", "ll", "३", "-a", "́'", "Reß", "'ll"]} +{"text": " ㋿\"", "tokens": 6, "pieces": [" ", " ㋿\""]} +{"text": "!!é's.m>'M🙂İ'<|fim_prefix|>\r\n( \u000b's ́,٣٤٥٦.'T'll
½m<'VEd'ſ", "tokens": 46, "pieces": ["!!", "e", "́'", "s", ".m", ">'", "M", "🙂İ", "'<|", "fim", "_prefix", "|>\r\n", "(", " ", "\u000b", "'s", " ", "́,", "٣٤٥", "٦", ".'", "T", "'ll", "
", "½", "m", "<'", "VEd", "'ſ"]} +{"text": "-\n😀🏽(dd३!!'ſå'T\r\n\u000bſ'VE'VEa…\u000b​ 're", "tokens": 34, "pieces": ["-\n", "😀🏽(", "dd", "३", "!!'", "ſa", "̊'", "T", "\r\n", "\u000bſ", "'VE", "'VE", "a", "…", "\u000b", "​", " '", "re"]} +{"text": "\r\n\r\n<|endoftext|>12345678­‍\r\nع'sé'T>Ⅳ
ꟲ!́<|endoftext|>", "tokens": 38, "pieces": ["\r\n\r\n", "<|", "endoftext", "|>", "123", "456", "78", "­‍\r\n", "ع", "'s", "e", "́'", "T", ">", "Ⅳ", "
ꟲ", "!́<|", "endoftext", "|>"]} +{"text": "'D'VEt\r㋿\r", "tokens": 12, "pieces": ["'D", "'VE", "t", "\r", "㋿\r"]} +{"text": "'Re", "tokens": 11, "pieces": ["'Re", "a", "̊<", "EOT", ">"]} +{"text": "t'll٣٤٥٦­\ra A ḍ̇🙂-𐞁'reⅣDžåé㍿\r\n\r\n'Re!\n👍🏽\r\n\r\n\r'S👍🏽å३‍\u000b㍿
­", "tokens": 74, "pieces": ["t", "'ll", "٣٤٥", "٦", "­\r", "a", " A", " ḋ", "̣🙂-", "𐞁", "'re", "Ⅳ", "Dža", "̊e", "́㍿\r\n\r\n", "'Re", "!\n", "👍🏽\r\n\r\n\r", "'S", "👍🏽", "a", "̊", "३", "‍", "\u000b", "㍿", "
", "­"]} +{"text": "!!Ⅳ🙂Dž\r\n\r\n're \ńd漢½'S🙂må\"'S٣٤٥٦!🙂ß\t‍é", "tokens": 40, "pieces": ["!!", "Ⅳ", "🙂Dž", "\r\n\r\n", "'re", " \n", "́d漢", "½", "'S", "🙂ma", "̊\"'", "S", "٣٤٥", "٦", "!🙂", "ß", "\t", "‍e", "́"]} +{"text": ".­é 12345678'Re½👍🏽ꟲ字,!ſ'ſDž
👍🏽🙂३\n ", "tokens": 42, "pieces": [".­", "e", "́", " ", "123", "456", "78", "'Re", "½", "👍🏽", "ꟲ字", ",!", "ſ", "'ſ", "Dž", "
", "👍🏽🙂", "३", "\n "]} +{"text": "eعé٣٤٥٦🙂 >­é ꟲ🙂\r\n\r\n\r", "tokens": 26, "pieces": ["eعé", "٣٤٥", "٦", "🙂", " ", " >­", "e", "́", " ", " ꟲ", "🙂\r\n\r\n\r"]} +{"text": "åd<|endoftext|>'́\r\n\r\n's㋿😀🏽ḍ̇😀🏽'Re🙂å0'D‍#$%'VEḍ̇🙂 (🙂sd#$%éa\"\u000b­ſİ\r\n\r\n", "tokens": 69, "pieces": ["a", "̊d", "<|", "endoftext", "|>'́\r\n\r\n", "'s", "㋿😀🏽", "ḋ", "̣😀🏽'", "Re", "🙂a", "̊", "0", "'D", "‍#$%'", "VEḋ", "̣🙂", " ", "(🙂", "sd", "#$%", "éa", "\"", "\u000b", "­ſİ", "\r\n\r\n"]} +{"text": "\rDžZmZ'Dt!'s!㍿", "tokens": 14, "pieces": ["\r", "DžZmZ", "'D", "t", "!'", "s", "!㍿"]} +{"text": "ݽ#$%ß👍🏽'll­'ſ\n字'Dé㍿a'३'D 😀🏽\t <|endoftext|>Dž३fiꟲEOTtdé…>'ſ", "tokens": 59, "pieces": ["İ", "½", "#$%", "ß", "👍🏽'", "ll", "­'", "ſ", "\n", "字", "'D", "e", "́㍿", "a", "'", "३", "'D", " 😀🏽", "\t ", " <|", "endoftext", "|>", "Dž", "३", "fiꟲEOTtdé", "…", ">'", "ſ"]} +{"text": "EOT ́<|fim_prefix|>a'llßع \n!,😀🏽é😀🏽-İ'sß#$%㋿é", "tokens": 39, "pieces": ["EOT", " ", "́<|", "fim", "_prefix", "|>", "a", "'ll", "ßع", " \n", "!,😀🏽", "é", "😀🏽-", "İ", "'s", "ß", "#$%㋿", "e", "́"]} +{"text": "½<😀🏽'Re", "tokens": 9, "pieces": ["½", "<😀🏽'", "Re"]} +{"text": "३'Re<|endoftext|>😀🏽!३ḍ̇!!.😀🏽\u000b'M>㍿", "tokens": 35, "pieces": ["३", "'Re", "<|", "endoftext", "|>😀🏽!", "३", "ḋ", "̣!!.😀🏽", "\u000b", "'M", ">㍿"]} +{"text": "((", "tokens": 1, "pieces": ["(("]} +{"text": "'sée 'S‍\t.,Džm½….‍㍿Ⅳ 𐞁'll٣٤٥٦عa🙂
-ꟲ\r\n\r\n३'ſ<'T's!'-EOT
", "tokens": 63, "pieces": ["'s", "e", "́e", " '", "S", "‍", "\t", ".,", "Džm", "½", "…", ".‍㍿", "Ⅳ", " 𐞁", "'ll", "٣٤٥", "٦", "عa", "🙂", "
", "-ꟲ", "\r\n\r\n", "३", "'ſ", "<'", "T", "'s", "!'-", "EOT", "
"]} +{"text": "\t字…'s字em sae\" 'ſ12345678é'reİ'ss㋿<|fim_prefix|>#$% e漢'VE.fi>…é½ꟲ…", "tokens": 60, "pieces": ["Ⅳ", "'Re", "\u000b\n", "'ſ", "\tt", "", " sae", "\"", " ", " '", "ſ", "123", "456", "78", "é", "'re", "İ", "'s", "s", "㋿<|", "fim", "_prefix", "|>#$%", " e漢", "'VE", ".fi", ">", "…e", "́", "½", "ꟲ", "…"]} +{"text": "ḍ̇Dž<|endoftext|>'MEOTA ſ'lls٣٤٥٦<|endoftext|>Zꟲ,a,\t\r\n\r\n\r\n", "tokens": 46, "pieces": ["ḋ", "̣Dž", "<|", "endoftext", "|>'", "MEOTA", " ", " ſ", "'ll", "s", "٣٤٥", "٦", "<|", "endoftext", "|>", "Zꟲ", ",a", ",", "\t\r\n\r\n\r\n"]} +{"text": "EOT…㋿'T,t9'Me\u000b( A!!('ſⅣfi🙂're!0(ḍ̇'T'D㍿
's'll", "tokens": 47, "pieces": ["EOT", "…", "㋿'", "T", ",t", "9", "'M", "e", "\u000b", "(", " A", "!!('", "ſ", "Ⅳ", "fi", "🙂'", "re", "!", "0", "(ḋ", "̣'", "T", "'D", "㍿", "
", "'s", "'ll"]} +{"text": "m<<|fim_prefix|>!!å🙂!!<|fim_prefix|>fi😀🏽éꟲſ'Re३ \n0😀🏽\ńå½
\r\n\r\n!!a'VE-ع>", "tokens": 62, "pieces": ["m", "<<|", "fim", "_prefix", "|>!!", "a", "̊🙂!!<|", "fim", "_prefix", "|>", "fi", "😀🏽", "e", "́ꟲſ", "'Re", "३", " \n", "0", "😀🏽\n", "́a", "̊", "½", "
\r\n\r\n", "!!", "a", "'VE", "-ع", ">"]} +{"text": "<|endoftext|>ß \nfi'Re9­'T\t 'Må\"㋿", "tokens": 25, "pieces": ["<|", "endoftext", "|>", "ß", " \n", "fi", "'Re", "9", "­'", "T", "\t", " '", "Ma", "̊\"㋿"]} +{"text": "Ⅳ(d­\r>éⅣ", "tokens": 9, "pieces": ["Ⅳ", "(d", "­\r", ">é", "Ⅳ"]} +{"text": "ßİZ,Aé\"t 'VE‍'>'T're½!㍿\tm.<|endoftext|>ḍ̇,'VEd0!Zm!!e漢😀🏽t­å.­", "tokens": 58, "pieces": ["ßİZ", ",Aé", "\"t", " '", "VE", "‍'>'", "T", "'re", "½", "!㍿", "\tm", ".<|", "endoftext", "|>", "ḋ", "̣,'", "VEd", "0", "!Zm", "!!", "e漢", "😀🏽", "t", "­a", "̊.­"]} +{"text": "'se'reeDžZ- e Ⅳ½Ⅳ👍🏽🙂tat𐞁ḍ̇'VE9<ꟲſİ𐞁(‍étꟲ", "tokens": 57, "pieces": ["'s", "e", "'re", "eDžZ", "-", " e", " ", "Ⅳ½Ⅳ", "👍🏽🙂", "tat𐞁ḋ", "̣'", "VE", "9", "<ꟲſİ𐞁", "(‍", "e", "́tꟲ"]} +{"text": "ꟲ'ſ<|fim_prefix|>'M½ß\r\n\r\né½å​٣٤٥٦\u000b…'T​Ⅳsع🙂0", "tokens": 43, "pieces": ["ꟲ", "'ſ", "<|", "fim", "_prefix", "|>'", "M", "½", "ß", "\r\n\r\n", "é", "½", "a", "̊​", "٣٤٥", "٦", "\u000b", "…", "'T", "​", "Ⅳ", "sع", "🙂", "0"]} +{"text": "é're ३.‍dßع½𐞁< <\r're'T's(ḍ̇𐞁 \r\n\r\n\n\r'S12345678'\r\n\r\né\r\n", "tokens": 43, "pieces": ["é", "'re", " ", "३", ".‍", "dßع", "½", "𐞁", "<", " ", "<\r", "'re", "'T", "'s", "(ḋ", "̣𐞁", " \r\n\r\n\n\r", "'S", "123", "456", "78", "'\r\n\r\n", "é", "\r\n"]} +{"text": "a
. 漢- -<|fim_prefix|>ع字٣٤٥٦fiꟲ'ſ漢'T<㋿'T#$%a#$%dⅣ#$%😀🏽", "tokens": 59, "pieces": ["a", "
", ".", " 漢", "-", " ", " -<|", "fim", "_prefix", "|>", "ع字", "٣٤٥", "٦", "fiꟲ", "'ſ", "漢", "'T", "<㋿'", "T", "#$%", "a", "#$%", "d", "Ⅳ", "#$%😀🏽"]} +{"text": "㋿d's'Dž'sDž<|endoftext|>'ll'VE,(­åſ𐞁ßDž\u000bfi>-‍t å", "tokens": 50, "pieces": ["㋿d", "'s", "'Dž", "'s", "Dž", "<|", "endoftext", "|>'", "ll", "'VE", ",(­", "a", "̊ſ𐞁ßDž", "\u000bfi", ">-‍", "t", " ", " a", "̊<", "META", "_START", ">"]} +{"text": "'re#$%½a'S\t-< \n12345678'VE\r\n\r\n
…,\"t  ", "tokens": 26, "pieces": ["'re", "#$%", "½", "a", "'S", "\t", "-<", " \n", "123", "456", "78", "'", "VE", "\r\n\r\n", "
", "…", ",\"", "t", "  "]} +{"text": "३(漢'ſ­㍿ >", "tokens": 14, "pieces": ["३", "(漢", "'ſ", "­㍿", " ", ">"]} +{"text": "
'T'ſ٣٤٥٦\"\u000bꟲ<|fim_prefix|>'M'llſ\r\n'D\r\n\r\n'ſ٣٤٥٦", "tokens": 44, "pieces": ["
", "'T", "'ſ", "٣٤٥", "٦", "\"", "\u000bꟲ", "<|", "fim", "_prefix", "|>'", "M", "'ll", "ſ", "\r\n", "'D", "\r\n\r\n", "'ſ", "٣٤٥", "٦"]} +{"text": "İ​é'S\r𐞁\r\n\r\n<…​!!9'Re", "tokens": 23, "pieces": ["İ", "​", "é", "'S", "\r", "𐞁", "\r\n\r\n", "<", "…", "​!!", "9", "'Re"]} +{"text": "\r\n\r\n0fi0-\u000b", "tokens": 9, "pieces": ["", "0", "fi", "0", "-", "\u000b"]} +{"text": "'ſꟲZ<|endoftext|>\u000b", "tokens": 15, "pieces": ["'ſ", "ꟲZ", "<|", "endoftext", "|>", "\u000b"]} +{"text": "dt!!Z \n'T'D0\rEOT'Re<|fim_prefix|>aß\u000baⅣ !字İ'DAſ's\"ḍ̇'S漢Ae३a\r\n\r\nå'Re", "tokens": 58, "pieces": ["dt", "!!", "Z", " \n", "'T", "'D", "0", "\r", "EOT", "'Re", "<|", "fim", "_prefix", "|>", "aß", "\u000ba", "Ⅳ", " !", "字İ", "'D", "Aſ", "'s", "\"ḋ", "̣'", "S漢Ae", "३", "a", "\r\n\r\n", "a", "̊'", "Re"]} +{"text": "'S 'T‍\t12345678m(\"字", "tokens": 12, "pieces": ["'S", " ", "'T", "‍", "\t", "123", "456", "78", "m", "(\"", "字"]} +{"text": "\r\n'ſ<漢'VE>,'Re\u000btfié漢
ḍ̇
'Mḍ̇ß's#$%\rع \nA!!", "tokens": 49, "pieces": ["\r\n", "'ſ", "<漢", "'VE", ">,'", "Re", "\u000btfie", "́漢", "
ḋ", "̣", "
", "'M", "ḋ", "̣<", "META", "_START", ">ß", "'s", "#$%\r", "ع", " \n", "A", "!!"]} +{"text": "!字㋿字ß<|fim_prefix|><🙂ḍ̇\r漢 ſfi…\r\n\r\né​EOT'Mḍ̇m
ḿDžé­½<|endoftext|>字9", "tokens": 68, "pieces": ["!", "字", "㋿字ß", "<|", "fim", "_prefix", "|><🙂", "ḋ", "̣\r", "漢", " ſfi", "…\r\n\r\n", "é", "​EOT", "'M", "ḋ", "̣m", "
m", "́Dže", "́­", "½", "<|", "endoftext", "|>", "字", "", "9"]} +{"text": "'M👍🏽字'ḍ̇d'D\r\n\r\nfi \n​åİZ'S<|fim_prefix|>'ſ'ſ\r\n\r\neA㍿漢're漢d<|fim_prefix|>9\n#$%😀🏽İ", "tokens": 72, "pieces": ["'M", "👍🏽", "字", "'ḋ", "̣d", "'D", "\r\n\r\n", "fi", " \n", "​a", "̊İ", "Z", "'S", "<|", "fim", "_prefix", "|>'", "ſ", "'ſ", "\r\n\r\n", "eA", "㍿漢", "'re", "漢d", "<|", "fim", "_prefix", "|>", "9", "\n", "#$%😀🏽", "İ"]} +{"text": "'reع\r\n\nA'T😀🏽", "tokens": 37, "pieces": ["'", "reع", "\r\n", "\n", "A", "'T", "😀🏽"]} +{"text": " !!12345678\t'MZ'T.\t'VE𐞁\r!!٣٤٥٦!! ", "tokens": 32, "pieces": [" ", "!!", "123", "456", "78", "\t", "'M", "Z", "'", "T", ".", "\t", "'VE", "𐞁", "\r", "!!", "٣٤٥", "٦", "!!", " "]} +{"text": "!!ꟲ漢 𐞁½👍🏽9'!!漢,'re漢 ​é\r字!'VEefi\n<|fim_prefix|>d\r\nꟲ", "tokens": 51, "pieces": ["!!", "ꟲ漢", " 𐞁", "½", "👍🏽", "9", "'!!", "漢", ",'", "re漢", " ", "​e", "́\r", "字", "!'", "VEefi", "\n", "<|", "fim", "_prefix", "|>", "d", "\r\n", "ꟲ"]} +{"text": " \n\u000b", "tokens": 2, "pieces": [" \n\u000b"]} +{"text": "<'Reé<|endoftext|>.'ReEOT \n", "tokens": 21, "pieces": ["<<", "EOT", ">'", "Reé", "<|", "endoftext", "|><", "META", "_START", ">.'", "ReEOT", " \n"]} +{"text": "m\n­ع é́'S'Mt!!''ll\r\n\r\n<'ſ,'", "tokens": 21, "pieces": ["m", "\n", "­ع", " ", " é", "́'", "S", "'M", "t", "!!''", "ll", "\r\n\r\n", "<'", "ſ", ",'"]} +{"text": "t 'M
fi.EOTꟲḍ̇\t​\u000be'SA'Re\u000b,", "tokens": 27, "pieces": ["t", " '", "M", "
fi", ".EOTꟲḋ", "̣", "\t", "​", "\u000be", "'S", "A", "'Re", "\u000b", ","]} +{"text": "ßm'D .🙂\u000b<|endoftext|>A'VE'T३9<|fim_prefix|>ḍ̇ḍ̇‍s<|fim_prefix|><|fim_prefix|> #$%12345678å<|endoftext|>\r0tع(字\r", "tokens": 82, "pieces": ["ßm", "'D", " ", ".🙂", "\u000b", "<|", "endoftext", "|>", "A", "'VE", "'T", "३9", "<|", "fim", "_prefix", "|>", "ḋ", "̣ḋ", "̣‍", "s", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", " ", "#$%", "123", "456", "78", "a", "̊<|", "endoftext", "|><", "EOT", ">\r", "0", "tع", "(字", "\r"]} +{"text": "0'VE'll 'VE ,'sEOT12345678字<字㍿㋿!'Mꟲ!!Dž\r.åm'S0\tⅣDž#$%s½'ll🙂㍿'ll…㋿", "tokens": 61, "pieces": ["0", "'VE", "'ll", " ", "'VE", " ", " ,'", "sEOT", "123", "456", "78", "字", "<字", "㍿㋿!'", "Mꟲ", "!!", "Dž", "\r", ".a", "̊m", "'S", "0", "\t", "Ⅳ", "Dž", "#$%", "s", "½", "'ll", "🙂㍿'", "ll", "…", "㋿"]} +{"text": "<-EOT​ \n㍿漢'VEdꟲ123456789", "tokens": 23, "pieces": ["<-", "EOT", "​", " \n", "㍿", "漢", "'VE", "dꟲ", "123", "456", "789"]} +{"text": "\r .٣٤٥٦漢㋿'ll👍🏽t🙂'D\t㋿,'VE..'ReA'ſ!!<|fim_prefix|> \u000b <(", "tokens": 57, "pieces": ["\r", " ", ".", "٣٤٥", "٦", "漢", "㋿'", "ll", "👍🏽", "t", "🙂'", "D", "\t", "㋿,'", "VE", "..'", "ReA", "'ſ", "!!<|", "fim", "_prefix", "|>", " \u000b", " <("]} +{"text": "…då!('T👍🏽,字漢㋿'ll12345678½\u000båå,👍🏽\"", "tokens": 42, "pieces": ["…da", "̊!('", "T", "👍🏽,", "字漢", "㋿'", "ll", "123", "456", "78½", "\u000ba", "̊a", "̊,👍🏽\""]} +{"text": "ꟲéꟲ<|endoftext|>😀🏽s<|fim_prefix|>'Daéعt­'12345678 ३d​\t३'M#$%#$%\tDž09​👍🏽字<|fim_prefix|>漢fiꟲ", "tokens": 75, "pieces": ["ꟲéꟲ", "<|", "endoftext", "|>😀🏽", "s", "<|", "fim", "_prefix", "|>'", "Daéعt", "­'", "123", "456", "78", " ", "३", "d", "​", "\t", "३", "'M", "#$%#$%", "\tDž", "09", "​👍🏽", "字", "<|", "fim", "_prefix", "|>", "漢fiꟲ"]} +{"text": "éfiꟲß'ſ\r\n\r\n'VE\u000b\r ­'llḍ̇𐞁12345678ßꟲ½12345678İA漢'S12345678fi9🙂'Mt!!s", "tokens": 64, "pieces": ["e", "́fiꟲ", "ß", "'ſ", "\r\n\r\n", "'VE", "\u000b\r", " ­'", "llḋ", "̣𐞁", "123", "456", "78", "ßꟲ", "½12", "345", "678", "İA漢", "'S", "123", "456", "78", "fi", "9", "🙂'", "Mt", "!!", "s"]} +{"text": "#$%\"Džé\r\nEOTⅣ\u000b \n.éaEOTḍ̇㍿ꟲd𐞁㍿Džſ‍s\"İ㋿\"fi<|fim_prefix|>'D\r\n\r\n", "tokens": 63, "pieces": ["#$%\"", "Dž", "é", "\r\n", "EOT", "Ⅳ", "\u000b \n", ".éaEOTḋ", "̣㍿", "ꟲd𐞁", "㍿Džſ", "‍s", "\"İ", "㋿\"", "fi", "<|", "fim", "_prefix", "|>'", "D", "\r\n\r\n"]} +{"text": "\tt'S'SEOT​Z.eꟲ'll'D½İ३字ḿ'!!tå é'll'Dİ(#$%­𐞁'll٣٤٥٦​t'Re\t …", "tokens": 56, "pieces": ["\tt", "'S", "'S", "EOT", "​Z", ".eꟲ", "'ll", "'D", "½", "İ", "३", "字m", "́'!!", "ta", "̊", " ", " e", "́'", "ll", "'D", "İ", "(#$%­", "𐞁", "'ll", "٣٤٥", "٦", "​t", "'Re", "\t …"]} +{"text": "'VE>'ll٣٤٥٦", "tokens": 12, "pieces": ["'VE", ">'", "ll", "٣٤٥", "٦"]} +{"text": "­漢عe𐞁>ts e'!!'Red漢're!d!!ع\rsfi👍🏽", "tokens": 39, "pieces": ["­漢", "عe𐞁", ">", "ts", " ", " e", "'!!'", "Red漢", "'re", "!d", "!!", "ع", "\r", "sfi", "👍🏽"]} +{"text": "㋿३\r\n<|fim_prefix|>Ⅳ'T'Mḍ̇ZZ\r\t \n\",'S'VE", "tokens": 34, "pieces": ["㋿", "३", "\r\n", "<", "META", "_START", "><|", "fim", "_prefix", "|>", "Ⅳ", "'T", "'M", "ḋ", "̣ZZ", "\r\t \n", "\",'", "S", "'VE"]} +{"text": "…'ſ\r\n\r\n\r\rⅣ ​'S>'T‍<|fim_prefix|>9ß\r'ſ''VE.!!'VEm'Dé…\r\n㍿ſꟲé\n\raEOT", "tokens": 64, "pieces": ["…", "'ſ", "\r\n\r\n\r\r", "", "Ⅳ", " ", "​'", "S", ">'", "T", "‍<|", "fim", "_prefix", "|><", "EOT", ">", "9", "ß", "\r", "'ſ", "''", "VE", ".!!'", "VEm", "'D", "e", "́", "…\r\n", "㍿ſꟲé", "\n\r", "aEOT"]} +{"text": "#$%\r\n字'St‍'VE", "tokens": 9, "pieces": ["#$%\r\n", "字", "'S", "t", "‍'", "VE"]} +{"text": "Z​m'sß٣٤٥٦'ſ㋿ \n'́", "tokens": 22, "pieces": ["Z", "​m", "'s", "ß", "٣٤٥", "٦", "'ſ", "㋿", " \n", "'́"]} +{"text": "ſع\u000bⅣ㋿!! -'re३fi<fi", "tokens": 21, "pieces": ["ſع", "\u000b", "Ⅳ", "㋿!!", " ", " -'", "re", "३", "fi", "<fi"]} +{"text": "😀🏽\"ع-́", "tokens": 9, "pieces": ["😀🏽\"", "ع", "-́"]} +{"text": "ꟲ'sⅣ 0́́'T!!㋿#$%३At<", "tokens": 24, "pieces": ["ꟲ", "'s", "Ⅳ", " ", "0", "́́'", "T", "!!㋿#$%", "३", "At", "<"]} +{"text": "> 'S\r\n\r\n'll''VEꟲꟲ\r\n­AZ'Re\r\n漢9<½\r\n<‍İ\r\n­‍\n12345678'M.<|fim_prefix|>\t
tt", "tokens": 50, "pieces": [">", " '", "S", "\r\n\r\n", "'ll", "''", "VEꟲꟲ", "\r\n", "­AZ", "'Re", "\r\n", "漢", "9", "<", "½", "\r\n", "<‍", "İ", "\r\n", "­‍\n", "123", "456", "78", "'M", ".<|", "fim", "_prefix", "|>", "\t", "
tt"]} +{"text": "𐞁", "tokens": 4, "pieces": ["𐞁"]} +{"text": "\t​", "tokens": 2, "pieces": ["\t", "​"]} +{"text": "\u000b́\r\naع'S\r\n'S\u000bḍ̇'S'Re㋿ 'S'ſ,👍🏽'll12345678s 😀🏽\t\u000bm 'ḍ̇#$%", "tokens": 58, "pieces": ["\u000b", "́\r\n", "aع", "'S", "\r\n", "'S", "\u000bḋ", "̣'", "S", "'Re", "㋿", " ", "'S", "'ſ", ",👍🏽'", "ll", "123", "456", "78", "s", " ", " 😀🏽", "\t", "\u000bm", " ", "'<", "EOT", ">ḋ", "̣#$%"]} +{"text": "\"é're<|endoftext|>", "tokens": 10, "pieces": ["\"é", "'re", "<|", "endoftext", "|>"]} +{"text": "👍🏽ꟲ字Ⅳ𐞁9e!!㋿'Mfi㍿'M9!", "tokens": 33, "pieces": ["👍🏽", "ꟲ字", "Ⅳ", "𐞁", "9", "e", "!!㋿'", "Mfi", "㍿'", "M", "9", "!"]} +{"text": "İé́'S'S'VE‍\"'D\r\n\r\n👍🏽 daAḍ̇ \nA'DDž㍿", "tokens": 40, "pieces": ["İé", "́'", "S", "'S", "'VE", "‍\"'", "D", "\r\n\r\n", "👍🏽", " ", " daAḋ", "̣", " \n", "A", "'", "DDž", "㍿"]} +{"text": "Dž\r\n\r\n.é A<|fim_prefix|>es.ع漢'ſ<|endoftext|>ß<|fim_prefix|>ß<|endoftext|><́>(🙂ع<|fim_prefix|>'ll字", "tokens": 63, "pieces": ["Dž", "\r\n\r\n", ".é", " A", "<|", "fim", "_prefix", "|>", "es", ".ع", "漢", "'ſ", "<|", "endoftext", "|>", "ß", "<|", "fim", "_prefix", "|>", "ß", "<|", "endoftext", "|><́>(🙂", "ع", "<|", "fim", "_prefix", "|>'", "ll字"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'Re字\n#$%!
\r\n\r\n<|fim_prefix|>㍿ꟲA9'…
<|fim_prefix|>\r\n\r\n३EOT 
 \né'D0'M‍", "tokens": 50, "pieces": ["'Re", "字", "\n", "#$%!", "
\r\n\r\n", "<|", "fim", "_prefix", "|>㍿", "ꟲA", "9", "'", "…", "
", "<|", "fim", "_prefix", "|>\r\n\r\n", "३", "EOT", " 
 \n", "é", "'D", "0", "'M", "‍"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ſ#$%'SA!! \n\"‍!!ḍ̇\r\n­å(,#$%😀🏽-…e \r(!! 9ß m\tꟲ('Re0é🙂㍿-", "tokens": 62, "pieces": ["ſ", "#$%'", "SA", "!!", " \n", "\"‍!!", "ḋ", "̣\r\n", "­a", "̊(,#$%😀🏽<", "EOT", ">-", "…e", " \r", "(!!", " ", "9", "ß", " m", "\tꟲ", "('", "Re", "0", "é", "🙂㍿-"]} +{"text": "ꟲé>\"'ll'S( ḍ̇\r<|endoftext|> -<🙂", "tokens": 54, "pieces": ["ꟲé", ">\"'", "ll", "'S", "(", " ", " ḋ", "̣\r", "<|", "endoftext", "|>", " ", "-<", "ta", "…", "‍\r\n", "́a", "'S", "a", "̊", "Ⅳ", "a", "!!\r", "🙂<|", "endoftext", "|><🙂"]} +{"text": "\"\nsDžte\u000b\u000bé㋿½
", "tokens": 15, "pieces": ["\"\n", "sDžte", "\u000b", "\u000be", "́㋿", "½", "
"]} +{"text": "<|endoftext|>>em\u000b\ŕßé(Ⅳ㋿'M'MA éḍ̇!fi!\r\n\r\n'T's.🙂'ReZ'Re…#$%", "tokens": 48, "pieces": ["<|", "endoftext", "|>>", "em", "\u000b\r", "́ßé", "(", "Ⅳ", "㋿'", "M", "'M", "A", " éḋ", "̣!", "fi", "!\r\n\r\n", "'T", "'s", ".🙂'", "ReZ", "'Re", "…", "#$%"]} +{"text": "Z字9#$%!!efim123456780<|endoftext|>", "tokens": 23, "pieces": ["Z字", "9", "#$%!!", "efi", "m", "123", "456", "780", "<|", "endoftext", "|>"]} +{"text": "👍🏽'Re㍿ſß߅'re'reꟲ\t\"字ſ\"12345678", "tokens": 24, "pieces": ["\r", "<|", "endoftext", "|>", "…", "'re", "'re", "ꟲ", "\t", "\"字ſ", "\"", "123", "456", "78"]} +{"text": "'T'll'VE (‍ſ३½👍🏽𐞁'VE😀🏽Ⅳ'\n👍🏽㍿​'D'lla#$% ", "tokens": 49, "pieces": ["'T", "'ll", "'VE", " (‍", "ſ", "३½", "👍🏽", "𐞁", "'VE", "😀🏽", "Ⅳ", "'\n", "👍🏽㍿​'", "D", "'ll", "a", "#$%", " "]} +{"text": "d…-\r\n\r\n'S!漢's…!m­­'T12345678'M'Reع
\u000b'S<|fim_prefix|>ع\r", "tokens": 37, "pieces": ["d", "…", "-\r\n\r\n", "'S", "!漢", "'s", "…", "!m", "­­'", "T", "123", "456", "78", "'M", "'Re", "ع", "
", "\u000b", "'S", "<|", "fim", "_prefix", "|>", "ع", "\r"]} +{"text": "'D👍🏽𐞁<|fim_prefix|>Ⅳḍ̇'M字<|endoftext|>३", "tokens": 39, "pieces": ["'D", "👍🏽", "𐞁", "<|", "fim", "_prefix", "|>", "Ⅳ", "ḋ", "̣'", "M字", "<|", "endoftext", "|>", "३"]} +{"text": "9㋿e'TEOTafi😀🏽!!<字漢'M<'VE‍åꟲ fi\u000b😀🏽é​'M#$%", "tokens": 52, "pieces": ["9", "㋿e", "'T", "EOTafi", "😀🏽<", "EOT", ">!!<", "字漢", "'M", "<'", "VE", "‍a", "̊ꟲ", " ", " fi", "\u000b", "😀🏽", "e", "́​'", "M", "#$%"]} +{"text": "\"ḍ̇!ع٣٤٥٦EOT ,s>­'VE-İ'llⅣ", "tokens": 34, "pieces": ["\"ḋ", "̣!", "ع", "٣٤٥", "٦", "EOT", " ", " ,", "s", ">­'", "VE", "-<", "META", "_START", ">İ", "'ll", "Ⅳ"]} +{"text": "(\r\n\r\n'MfiⅣ!!ſ\rfim‍9A<|fim_prefix|>'VEA ", "tokens": 30, "pieces": ["(\r\n\r\n", "'M", "fi", "Ⅳ", "!!", "ſ", "\r", "fim", "‍", "9", "A", "<|", "fim", "_prefix", "|>'", "VEA", " "]} +{"text": "e­-\n12345678字", "tokens": 7, "pieces": ["e", "­-\n", "123", "456", "78", "字"]} +{"text": "Aé \nß9'TEOT𐞁#$%>\"912345678
12345678\n\"'s", "tokens": 27, "pieces": ["Aé", " \n", "ß", "9", "'T", "EOT𐞁", "#$%>\"", "912", "345", "678", "
", "123", "456", "78", "\n", "\"'", "s"]} +{"text": "'VE's \n\r\nİ ३ \nm<|endoftext|>㋿'llZ'll­İ\"9e-ꟲ'Re", "tokens": 38, "pieces": ["'VE", "'", "s", " \n\r\n", "İ", " ", "३", " \n", "m", "<|", "endoftext", "|>㋿'", "llZ", "'ll", "­İ", "\"", "9", "e", "-ꟲ", "'Re"]} +{"text": "漢\r\n\r\n\r\n\r\n<|endoftext|>EOT
<|endoftext|>\n٣٤٥٦字́'ſ'S< …\u000b\t'a \n'Re 'T𐞁'Re", "tokens": 57, "pieces": ["漢", "\r\n\r\n\r\n\r\n", "<|", "endoftext", "|>", "EOT", "", "
", "<|", "endoftext", "|>\n", "٣٤٥", "٦", "字", "́'", "ſ", "'S", "<", " …\u000b", "\t", "'a", "", " \n", "'Re", " ", "'T", "𐞁", "'Re"]} +{"text": "́…'D\"s>'lla", "tokens": 7, "pieces": ["́", "…", "'D", "\"s", ">'", "lla"]} +{"text": "s\nét-'M \n'De'ſfiḍ̇'ſ(İa½ EOT🙂ed㋿😀🏽12345678a\té­​\rZe…'", "tokens": 59, "pieces": ["s", "\n", "ét", "-'", "M", " \n", "'D", "e", "'ſ", "fiḋ", "̣'", "ſ", "(İa", "½", " EOT", "🙂ed", "㋿😀🏽", "123", "456", "78", "a", "\té", "­​\r", "Ze", "…", "'"]} +{"text": "½字
字<|endoftext|>-dZ'Re 'ſ'VE‍字#$%0EOT'İ <|endoftext|>>(…A漢‍‍٣٤٥٦.!!.a're", "tokens": 60, "pieces": ["½", "字", "
字", "<|", "endoftext", "|>-", "dZ", "'Re", " ", "'ſ", "'VE", "‍字", "#$%", "0", "EOT", "'İ", " ", " <|", "endoftext", "|>>(", "…A漢", "‍‍", "٣٤٥", "٦", ".!!.", "a", "'re"]} +{"text": "Z\r\n\r\né<|fim_prefix|>Z字㍿
å'M'Tſm½'Re\rⅣ \n's,9DžéDž字EOT漢ḍ̇ a12345678're ‍0\u000b漢", "tokens": 62, "pieces": ["Z", "\r\n\r\n", "e", "́<|", "fim", "_prefix", "|>", "Z字", "㍿", "
a", "̊'", "M", "'T", "ſm", "½", "'Re", "\r", "Ⅳ", " \n", "'s", ",", "9", "DžéDž字EOT漢ḋ", "̣", " a", "123", "456", "78", "'re", " ‍", "0", "\u000b漢"]} +{"text": "‍ 'ꟲ'T'Re<#$%'S<٣٤٥٦sDž\t \né\n's\r\nm0👍🏽😀🏽🙂 \r\n\r\nⅣ12345678ßt", "tokens": 54, "pieces": ["‍", " ", "'ꟲ", "'T", "'Re", "<#$%'", "S", "<", "٣٤٥", "٦", "sDž", "\t \n", "é", "\n", "'s", "\r\n", "m", "0", "👍🏽😀🏽🙂", " \r\n\r\n", "Ⅳ12", "345", "678", "ßt"]} +{"text": "aae 're", "tokens": 4, "pieces": ["aae", " '", "re"]} +{"text": "s'ſع \r\n", "tokens": 6, "pieces": ["s", "'ſ", "ع", " \r\n"]} +{"text": "३\n🙂.a‍  012345678🙂 .… 'VE
½fiſ😀🏽", "tokens": 35, "pieces": ["३", "\n", "🙂.", "a", "‍", " ", " ", "012", "345", "678", "🙂", " ", " .", "…", " ", "'VE", "
", "½", "fiſ", "😀🏽"]} +{"text": "<|fim_prefix|>́.…😀🏽Ⅳ漢👍🏽éḍ̇字-EOTZ'ſḍ̇'#$%s😀🏽३!(ع  ́𐞁👍🏽(ZZ½-", "tokens": 77, "pieces": ["<|", "fim", "_prefix", "|>́.<", "EOT", ">", "…", "😀🏽", "Ⅳ", "漢", "👍🏽", "éḋ", "̣字", "-EOTZ", "'ſ", "ḋ", "̣'#$%", "s", "😀🏽", "३", "!(", "ع", " ", " ", "́𐞁", "👍🏽(", "ZZ", "½", "-"]} +{"text": ">­ ‍Džḍ̇  \t٣٤٥٦,\t12345678'S\u000b\tEOT'S\"‍ \nt'll Dž", "tokens": 42, "pieces": [">­", " ", " ‍", "Džḋ", "̣", "  ", "\t", "٣٤٥", "٦", ",", "\t", "123", "456", "78", "'S", "\u000b", "\tEOT", "'S", "\"‍", " \n", "t", "'ll", " Dž"]} +{"text": "å mع'S're're!!#$%'VEé\n‍", "tokens": 18, "pieces": ["a", "̊", " mع", "'S", "'re", "'re", "!!#$%'", "VEe", "́\n", "‍"]} +{"text": "٣٤٥٦३ ꟲe.…Zd​Dž\n㍿ ß́٣٤٥٦'re ㋿'㋿ ś\nḍ̇", "tokens": 58, "pieces": ["٣٤٥", "٦३", " ꟲe", ".", "…Zd", "​Dž", "\n", "㍿", " ß", "́", "٣٤٥", "٦", "'re", " ", "㋿'㋿", " <", "META", "_START", ">s", "́\n", "ḋ", "̣"]} +{"text": " \n\t'MZ", "tokens": 7, "pieces": ["", " \n", "\t", "'M", "Z"]} +{"text": "ß٣٤٥٦㍿ ‍'s><㍿ع's漢
 İꟲ­\t!漢‍'Re \nß'VE", "tokens": 45, "pieces": ["ß", "٣٤٥", "٦", "㍿", " ", " ‍'", "s", "><㍿", "ع", "'s", "漢", "
 ", " İꟲ", "­", "\t", "!漢", "‍'", "Re", " \n", "ß", "'VE"]} +{"text": "İ३", "tokens": 3, "pieces": ["İ", "३"]} +{"text": " \ń字aİ!'re \r\n\r\né-\rmm,\u000b'S\r\n\u000b​!½'D\"'ſꟲs३\n0​ع ꟲEOTé🙂", "tokens": 48, "pieces": [" \n", "́字aİ", "!'", "re", " \r\n\r\n", "é", "-\r", "mm", ",", "\u000b", "'S", "\r\n", "\u000b", "​!", "½", "'D", "\"'", "ſꟲs", "३", "\n", "0", "​ع", " ꟲEOTe", "́🙂"]} +{"text": "३'D ' åZ12345678ع'S\"३ḍ̇🙂>\tḍ̇Za\"å \n're-é‍𐞁", "tokens": 51, "pieces": ["३", "'D", " ", "'", " a", "̊Z", "", "123", "456", "78", "ع", "'S", "\"", "३", "ḋ", "̣🙂>", "\tḋ", "̣Za", "\"a", "̊", " \n", "'re", "-é", "‍𐞁"]} +{"text": "😀🏽İⅣ½…ꟲm ꟲ", "tokens": 22, "pieces": ["😀🏽", "İ", "Ⅳ½", "…ꟲm", " ", "ꟲ"]} +{"text": "🙂0EOTḍ̇漢(,\réꟲ٣٤٥٦éA<<|fim_prefix|> \n\r\n\r\n!!", "tokens": 40, "pieces": ["🙂", "0", "EOTḋ", "̣漢", "(,\r", "éꟲ", "٣٤٥", "٦", "éA", "<<|", "fim", "_prefix", "|>", " \n\r\n\r\n", "!!"]} +{"text": "…é(٣٤٥٦é<|endoftext|>‍>fiſ\r\n0
('D<'D!!!!s're!!🙂\r\n'ſ'VE½३\r\n\r\né!'VE", "tokens": 55, "pieces": ["…é", "(", "٣٤٥", "٦", "é", "<|", "endoftext", "|>‍>", "fiſ", "\r\n", "0", "
", "('", "D", "<'", "D", "!!!!", "s", "'re", "!!🙂\r\n", "'ſ", "'VE", "½३", "\r\n\r\n", "e", "́!'", "VE"]} +{"text": "d!㍿<|endoftext|>‍12345678d­​漢😀🏽é<|fim_prefix|>  ſs\"ḍ̇‍", "tokens": 49, "pieces": ["d", "!㍿<|", "endoftext", "|>‍", "123", "456", "78", "d", "­​", "漢", "😀🏽", "e", "́<|", "fim", "_prefix", "|>", "  ", " ſs", "\"ḋ", "̣‍"]} +{"text": "㋿​<|endoftext|>½'ReꟲEOTs", "tokens": 19, "pieces": ["㋿​<|", "endoftext", "|>", "½", "'Re", "ꟲEOTs"]} +{"text": "㍿.ZDž𐞁'VE…½", "tokens": 16, "pieces": ["㍿.", "ZDž𐞁", "'VE", "…", "½"]} +{"text": "'SAd­EOT😀🏽ع!a/b,", "tokens": 17, "pieces": ["'S", "Ad", "­EOT", "😀🏽", "ع", "!a", "/b", ","]} +{"text": "İ'Sé㍿…0ſ'M", "tokens": 12, "pieces": ["İ", "'S", "é", "㍿", "…", "0", "ſ", "'M"]} +{"text": "㍿İ ­🙂Ⅳ…!!('s/\r\n½٣٤٥٦𐞁字", "tokens": 30, "pieces": ["㍿İ", " ", "­🙂", "Ⅳ", "…", "!!('", "s", "/\r\n", "½٣٤", "٥٦", "𐞁字"]} +{"text": "#$%EOT!mİſ!!aB́😀🏽/🙂>aB\r字𐞁aa", "tokens": 31, "pieces": ["#$%", "EOT", "!mİſ", "!!", "aB", "́😀🏽/🙂>", "aB", "\r", "字𐞁aa"]} +{"text": "HTTPServer's,", "tokens": 4, "pieces": ["HTTPServer", "'s", ","]} +{"text": "\u000b9-ßß'T३㍿\n/𐞁camelCaset'A/ \n
😀🏽é'Re 0Z😀🏽漢字Z'M0\r\n'T\u000b‍<😀🏽", "tokens": 61, "pieces": ["\u000b", "9", "-ßß", "'T", "३", "㍿\n", "/𐞁camelCaset", "'A", "/", " \n", "
", "😀🏽", "e", "́'", "Re", " ", "0", "Z", "😀🏽", "漢字Z", "'M", "0", "\r\n", "'T", "\u000b", "‍<😀🏽"]} +{"text": "\r\n\r\n \n ३Ze", "tokens": 6, "pieces": ["\r\n\r\n \n", " ", "३", "Ze"]} +{"text": "eAABC<\r#$%\r \n EOTꟲ0aBᵃ字< ㍿
<|endoftext|>👍🏽ḍ̇!!fiaİ\r\n\r\nEOTᵃaBABC\r\n\r\n#$%'retBßHTTPServer", "tokens": 69, "pieces": ["eAABC", "<\r", "#$%\r", " \n", " EOTꟲ", "0", "aBᵃ字", "<", " ", "㍿", "
", "<|", "endoftext", "|>👍🏽", "ḋ", "̣!!", "fiaİ", "\r\n\r\n", "EOTᵃaBABC", "\r\n\r\n", "#$%'", "retBßHTTPServer"]} +{"text": "ꟲ٣٤٥٦t", "tokens": 12, "pieces": ["ꟲ", "٣٤٥", "٦", "t"]} +{"text": "́åm'll12345678👍🏽ḍ̇12345678́fi字HTTPServer mDžungla\r\naBABC12345678٣٤٥٦/-'Sé字́<|endoftext|>Z#$%<|fim_prefix|>", "tokens": 73, "pieces": ["́a", "̊m", "'ll", "123", "456", "78", "👍🏽", "ḋ", "̣", "123", "456", "78", "́fi字HTTPServer", " mDžungla", "\r\n", "aBABC", "123", "456", "78٣", "٤٥٦", "/-'", "Se", "́字", "́<|", "endoftext", "|>", "Z", "#$%<|", "fim", "_prefix", "|>"]} +{"text": "'re\t \n!! \nå…!!ḍ̇<|fim_prefix|>\r\n\r\naB'Mds\n/>\t …ع<ſꟲiOS𐞁camelCase/\r\n٣٤٥٦>,'  camelCaseDžunglaé😀🏽", "tokens": 71, "pieces": ["'re", "\t \n", "!!", " \n", "a", "̊", "…", "!!", "ḋ", "̣<|", "fim", "_prefix", "|>\r\n\r\n", "aB", "'M", "ds", "\n", "/>", "\t ", "…ع", "<ſꟲiOS𐞁camelCase", "/\r\n", "٣٤٥", "٦", ">,'", " ", " camelCaseDžunglaé", "😀🏽"]} +{"text": "camelCase३字iOS/!!-Aſß-!!'ResDžunglaDž​́<\t", "tokens": 28, "pieces": ["camelCase", "३", "字iOS", "/!!-", "Aſß", "-!!'", "ResDžunglaDž", "​́<", "\t"]} +{"text": "fiiOS…'re३HTTPServer\n­'é", "tokens": 15, "pieces": ["fiiOS", "…", "'re", "३", "HTTPServer", "\n", "­'", "e", "́"]} +{"text": "'re\n \nſ!!'re \n 
Džungla'ſ-'Tå maB", "tokens": 28, "pieces": ["'re", "\n \n", "ſ", "!!'", "re", "", " \n", " ", "
Džungla", "'ſ", "-'", "Ta", "̊", " ", " maB"]} +{"text": "aB\r/.㍿字'SDžungla漢(''Re\u000b/a/bDžEOTe<|fim_prefix|>a/b12345678<|fim_prefix|>\r\na/b㍿<\r\n\r\n\n\n/\t#$%'T'<|endoftext|>é㍿\r\n ", "tokens": 75, "pieces": ["aB", "\r", "/.㍿", "字", "'S", "Džungla漢", "(''", "Re", "\u000b", "/a", "/b", "DžEOTe", "<|", "fim", "_prefix", "|>", "a", "/b", "123", "456", "78", "<|", "fim", "_prefix", "|>\r\n", "a", "/b", "㍿<\r\n\r\n\n\n", "/", "\t", "#$%'", "T", "'<|", "endoftext", "|>", "é", "㍿\r\n", " "]} +{"text": "ḍ̇a٣٤٥٦\ta/b,,iOS/\r\n😀🏽Z0/\r\nİDžé,<|endoftext|>e'D\" \n 
३ꟲ", "tokens": 74, "pieces": ["ḋ", "̣a", "٣٤٥", "٦", "\ta", "/b", ",,", "iOS", "/\r\n", "😀🏽", "Z", "0", "/\r\n", "İDže", "́,<", "a", "/b", "‍\n", "\t", "㍿!!", "sAb", "
", "!!", "t", "<<|", "endoftext", "|><|", "endoftext", "|>", "e", "'D", "\"", " \n", " ", "
", "३", "ꟲ"]} +{"text": "㍿A<.camelCaseİ\r\n\r\n éZ/\r\n\r\n\r\nع'll½\u000ba/b\naDžungla<|endoftext|>'reAb漢\n/(HTTPServer  
<|fim_prefix|>'re, ", "tokens": 55, "pieces": ["㍿A", "<.", "camelCaseİ", "\r\n\r\n", " éZ", "/\r\n\r\n\r\n", "ع", "'ll", "½", "\u000ba", "/b", "\n", "aDžungla", "<|", "endoftext", "|>'", "reAb漢", "\n", "/(", "HTTPServer", "  ", "
", "<|", "fim", "_prefix", "|>'", "re", ",", " "]} +{"text": "<|endoftext|>Ⅳ \n å字\td're漢😀🏽ᵃ#$%ᵃ\r\nع<9
", "tokens": 37, "pieces": ["<|", "endoftext", "|>", "Ⅳ", " \n", " a", "̊字", "\td", "'re", "漢", "😀🏽", "ᵃ", "#$%", "ᵃ", "\r\n", "ع", "<", "9", "
"]} +{"text": "Džunglaꟲ\n<|endoftext|>ß㍿Džungla\"0, 👍🏽ᵃ12345678\na/bḍ̇\u000bdİع'T/\r\ń🙂👍🏽dA Z🙂㋿m!", "tokens": 77, "pieces": ["Džunglaꟲ", "\n", "<|", "endoftext", "|>", "ß", "㍿Džungla", "\"", "0", ",<", "EOT", ">", " ", "👍🏽", "ᵃ", "123", "456", "78", "\n", "a", "/bḋ", "̣", "\u000bdİع", "'T", "/\r\n", "́🙂👍🏽", "dA", " Z", "🙂㋿", "m", "!"]} +{"text": "ḍ̇", "tokens": 5, "pieces": ["ḋ", "̣"]} +{"text": "👍🏽>\u000b>😀🏽é
Dž,'M漢'D\n\"ꟲ Z\r<|fim_prefix|>'ſⅣ Džunglaḍ̇½", "tokens": 55, "pieces": ["👍🏽>", "\u000b", ">😀🏽", "e", "́", "
Dž", ",'", "M漢", "'D", "\n", "\"ꟲ", " ", " Z", "\r", "<|", "fim", "_prefix", "|>'", "ſ", "Ⅳ", " Džunglaḋ", "̣", "½"]} +{"text": "camelCased-m٣٤٥٦AHTTPServer", "tokens": 20, "pieces": ["camelCased", "-m", "٣٤٥", "٦", "A", "HTTPServer"]} +{"text": "㋿漢'M'VEiOS<|endoftext|>", "tokens": 16, "pieces": ["㋿漢", "'M", "'VE", "iOS", "<|", "endoftext", "|>"]} +{"text": "/,HTTPServeŕ ḍ̇iOS٣٤٥٦'DHTTPServer'M\r\n're'reꟲ👍🏽'reaBB<|endoftext|>\u000be​ſ'måꟲſſd👍🏽m½", "tokens": 69, "pieces": ["/,", "HTTPServer", "́", " ḋ", "̣iOS", "٣٤٥", "٦", "'D", "HTTPServer", "'M", "\r\n", "'re", "'re", "ꟲ", "👍🏽'", "reaBB", "<|", "endoftext", "|>", "\u000be", "​ſ", "'m", "a", "̊ꟲſſd", "👍🏽", "m", "½"]} +{"text": "Abḍ̇\r\n", "tokens": 7, "pieces": ["Abḋ", "̣\r\n"]} +{"text": "ḍ̇\r\nİ​m#$% ३­HTTPServeré'Re", "tokens": 19, "pieces": ["ḋ", "̣\r\n", "İ", "​m", "#$%", " ", "३", "­HTTPServeré", "'Re"]} +{"text": "​३0.‍ZⅣ'Td'sßſs㋿ſ㋿ABC🙂𐞁\"", "tokens": 39, "pieces": ["​", "३0", ".‍", "Z", "Ⅳ", "'T", "d", "'s", "ßſs", "㋿ſ", "㋿ABC", "🙂𐞁", "\"<", "EOT", ">"]} +{"text": "s12345678!'ll \n 'ReⅣs B(㍿  'Tß'DABCAb>HTTPServer\r#$%\r\nDžunglaEOT\r\n\r\n12345678字", "tokens": 45, "pieces": ["s", "123", "456", "78", "!'", "ll", " \n", " '", "Re", "Ⅳ", "s", " B", "(㍿", " ", " ", "'T", "ß", "'D", "ABCAb", ">HTTPServer", "\r", "#$%\r\n", "DžunglaEOT", "\r\n\r\n", "123", "456", "78", "字"]} +{"text": "A'D's /ḍ̇ABCEOTſḍ̇s", "tokens": 21, "pieces": ["A", "'D", "'s", " /", "ḋ", "̣ABCEOTſḋ", "̣s"]} +{"text": "<|endoftext|>Ⅳm're0a/b
méé's\n/Dž'Dd\n漢漢<|endoftext|> é \naB🙂३'ReB‍m'ſ㍿.camelCase", "tokens": 66, "pieces": ["<|", "endoftext", "|>", "Ⅳ", "m", "'re", "0", "a", "/b", "
m", "éé", "'s", "\n", "/Dž", "'D", "d", "\n", "漢漢", "<|", "endoftext", "|>", " e", "́", " \n", "aB", "🙂", "३", "'Re", "B", "‍m", "'ſ", "㍿.", "camelCase"]} +{"text": "éAb½dḍ̇d'T\rꟲᵃ>HTTPServera/b­'VEé३<…\n/<|endoftext|>A<|fim_prefix|>", "tokens": 52, "pieces": ["éAb", "½", "dḋ", "̣<", "EOT", ">d", "'T", "\r", "ꟲᵃ", ">HTTPServera", "/b", "­'", "VEé", "३", "<", "…\n", "/<|", "endoftext", "|>", "A", "<|", "fim", "_prefix", "|>"]} +{"text": "㍿\ré ſéſ́ \r\n\r\n!ع(fi-ع
㋿camelCase#$%HTTPServerİ é…字 \n \"", "tokens": 42, "pieces": ["㍿\r", "e", "́", " ſe", "́ſ", "́", " \r\n\r\n", "!ع", "(fi", "-ع", "
", "㋿camelCase", "#$%", "HTTPServerİ", " é", "", "…字", " \n", " \""]} +{"text": "A Bé ꟲBAb,de🙂
", "tokens": 17, "pieces": ["A", " Be", "́", " ", " ꟲBAb", ",de", "🙂", "
"]} +{"text": "\u000b<𐞁\"ᵃå,'T㍿㋿ABC\n/<|endoftext|>​'T👍🏽ꟲꟲİ'VE'\r\n\r\n,
a/b", "tokens": 57, "pieces": ["\u000b", "<𐞁", "\"ᵃa", "̊,'", "T", "㍿㋿", "ABC", "\n", "/<|", "endoftext", "|><", "META", "_START", ">​'", "T", "👍🏽", "ꟲꟲİ", "'VE", "'\r\n\r\n", ",", "
a", "/b"]} +{"text": "
'll/㍿ع<|fim_prefix|>/\r\niOS\n/…\r\n\r\n éå­ḍ̇'SDžungla𐞁9", "tokens": 48, "pieces": ["
", "'ll", "/㍿", "ع", "<|", "fim", "_prefix", "|><", "META", "_START", ">/\r\n", "iOS", "\n", "/", "…\r\n\r\n", " e", "́a", "̊­", "ḋ", "̣'", "SDžungla𐞁", "9"]} +{"text": "a'll9Z ", "tokens": 5, "pieces": ["a", "'ll", "9", "Z", " "]} +{"text": "Dž'M字fi.'reİ'ſ  #$%", "tokens": 16, "pieces": ["Dž", "'M", "字fi", ".'", "reİ", "'ſ", "  ", " #$%"]} +{"text": "/ß'T½㍿<\rcamelCase<|endoftext|>'re", "tokens": 19, "pieces": ["/ß", "'T", "½", "㍿<\r", "camelCase", "<|", "endoftext", "|>'", "re"]} +{"text": "ß\u000ba/b<|fim_prefix|>½'T…!İ(𐞁漢‍\ta👍🏽'M\r<|endoftext|>t", "tokens": 44, "pieces": ["ß", "\u000ba", "/b", "<|", "fim", "_prefix", "|>", "½", "'T", "…", "!İ", "(𐞁漢", "‍", "\ta", "👍🏽'", "M", "\r", "<|", "endoftext", "|>", "t"]} +{"text": "👍🏽\n/!!iOS12345678", "tokens": 16, "pieces": ["👍🏽\n", "/!!", "iOS", "123", "456", "78", ""]} +{"text": "'VE 'll🙂ḍ̇\rs", "tokens": 18, "pieces": ["'VE", " ", " '", "ll", "🙂ḋ", "̣\r", "s", ""]} +{"text": "camelCase½Z…HTTPServer-'llſB's<|fim_prefix|>\n/\r\n \n fifi'.٣٤٥٦at'ſ", "tokens": 40, "pieces": ["camelCase", "½", "Z", "…HTTPServer", "-'", "llſB", "'s", "<|", "fim", "_prefix", "|>\n", "/\r\n", " \n", " fifi", "'.", "٣٤٥", "٦", "at", "'ſ"]} +{"text": "e'S#$%👍🏽iOSⅣ😀🏽/\r\n­'DžunglaEOT#$%'DiOS/#$%sB\r'res", "tokens": 41, "pieces": ["e", "'S", "#$%👍🏽", "iOS", "Ⅳ", "😀🏽/\r\n", "­'", "DžunglaEOT", "#$%'", "DiOS", "/#$%", "sB", "\r", "'re", "s"]} +{"text": " 9㋿a/b㍿'Re३字\n/#$%B
<|fim_prefix|>…a", "/b", "㍿'", "Re", "३", "字", "\n", "/#$%", "B", "
", "<|", "fim", "_prefix", "|>", "…", "mع㍿EOT\n<|fim_prefix|> \nEOT\r\nDžungla\"\r\nⅣ😀🏽A'Re.<'ll<|endoftext|> eDžungla👍🏽漢12345678B\n'Re e​ \n ", "tokens": 69, "pieces": ["mع", "㍿EOT", "\n", "<|", "fim", "_prefix", "|>", " \n", "EOT", "\r\n", "Džungla", "\"\r\n", "Ⅳ", "😀🏽", "A", "'Re", ".<'", "ll", "<|", "endoftext", "|>", " eDžungla", "👍🏽", "漢", "123", "456", "78", "B", "\n", "'Re", " e", "​", " \n "]} +{"text": "́٣٤٥٦ \n ‍\u000béaBiOS", "tokens": 17, "pieces": ["́", "٣٤٥", "٦", " \n", " ‍", "\u000béaBiOS"]} +{"text": "t🙂.#$%> camelCase AHTTPServerté…!Ⅳ𐞁12345678Zsm!‍ ⅣB\n\n/", "tokens": 41, "pieces": ["t", "🙂.#$%>", " ", " camelCase", " ", " AHTTPServerte", "́", "…", "!", "Ⅳ", "𐞁", "123", "456", "78", "Zsm", "!‍", " ", " ", "Ⅳ", "B", "\n\n", "/"]} +{"text": "́e-👍🏽Ab'Re'ſ'Re.𐞁­𐞁ß's//\r\n٣٤٥٦(🙂0/ \n \r‍
", "tokens": 52, "pieces": ["́e", "-👍🏽<", "EOT", ">Ab", "'Re", "'ſ", "'Re", ".𐞁", "­𐞁ß", "'s", "//\r\n", "٣٤٥", "٦", "(🙂", "0", "/", " \n \r", "‍", "
"]} +{"text": "d<'D", "tokens": 3, "pieces": ["d", "<'", "D"]} +{"text": "fi12345678ABC/\r\nᵃAb#$%İ12345678BiOS\n/
/½́
A <|fim_prefix|>́", "tokens": 37, "pieces": ["fi", "123", "456", "78", "ABC", "/\r\n", "ᵃAb", "#$%", "İ", "123", "456", "78", "BiOS", "\n", "/", "
", "/", "½", "́", "
A", " <|", "fim", "_prefix", "|>́"]} +{"text": "Ⅳ́0é'́\r\nDž🙂​\r\n\r\n<|fim_prefix|>\n👍🏽​", "tokens": 36, "pieces": ["Ⅳ", "́<", "EOT", ">", "0", "é", "'́\r\n", "Dž", "🙂<", "EOT", ">​\r\n\r\n", "<|", "fim", "_prefix", "|>\n", "👍🏽​"]} +{"text": " \n é/\n\r9'M
 \n ㍿ddḍ̇é\u000bع\r\n\r\n", "tokens": 24, "pieces": [" \n", " é", "/\n\r", "9", "'M", "
 \n", " ㍿", "ddḋ", "̣e", "́", "\u000bع", "\r\n\r\n"]} +{"text": "'VE<|fim_prefix|>½t>\u000bİd­é㍿\n/漢å", "tokens": 31, "pieces": ["'VE", "<|", "fim", "_prefix", "|>", "½", "t", ">", "\u000bİd", "­e", "́㍿\n", "/漢a", "̊"]} +{"text": "\r\n\r\nᵃcamelCaseZ12345678'D'Tm字㋿å\r\n\r\n'Re!!'ll\r\n12345678३🙂\nABCA漢-​㍿\n#$%HTTPServer!!'Tß​<|fim_prefix|>漢å!!/\r\n", "tokens": 71, "pieces": ["\r\n\r\n", "ᵃcamelCaseZ", "123", "456", "78", "'D", "'T", "m字", "㋿a", "̊\r\n\r\n", "'Re", "!!'", "ll", "\r\n", "123", "456", "78३", "🙂\n", "ABCA漢", "-​㍿\n", "#$%", "HTTPServer", "!!'", "Tß", "​<|", "fim", "_prefix", "|>", "漢a", "̊!!/\r\n"]} +{"text": "'s\n/", "tokens": 3, "pieces": ["'s", "\n", "/"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'re漢", "tokens": 3, "pieces": ["'re", "漢"]} +{"text": "ſ'a9Džungla", "tokens": 8, "pieces": ["ſ", "'a", "9", "Džungla"]} +{"text": "\n/\r\nm/'ſ​d'M٣٤٥٦ABC d٣٤٥٦ع\r!\r!ᵃ𐞁 (Dž'\r\n\r\nſ> ​é ", "tokens": 54, "pieces": ["\n", "/\r\n", "m", "/'", "ſ", "​", "d", "'M", "٣٤٥", "٦", "ABC", " d", "٣٤٥", "٦", "ع", "\r", "!\r", "!ᵃ𐞁", " ", "(Dž", "'\r\n\r\n", "ſ", ">", " ", "​é", " "]} +{"text": ",'reZꟲᵃ/", "tokens": 10, "pieces": [",'", "reZꟲᵃ", "/"]} +{"text": "\rⅣ'ſDžHTTPServer'Reå12345678㋿9ſ㋿ABC🙂BåaDžungla‍!camelCase12345678<|endoftext|>/' ㋿\"'Ts", "tokens": 59, "pieces": ["\r", "Ⅳ", "'ſ", "DžHTTPServer", "'Re", "a", "̊", "123", "456", "78", "㋿", "9", "ſ", "㋿ABC", "🙂Ba", "̊aDžungla", "‍!", "camelCase", "123", "456", "78", "<|", "endoftext", "|>/'", " ", "㋿\"'", "Ts"]} +{"text": "\n'a/b‍<|fim_prefix|>½字EOT३", "tokens": 18, "pieces": ["\n", "'a", "/b", "‍<|", "fim", "_prefix", "|>", "½", "字EOT", "३"]} +{"text": " \n ع\r\n\r\nå😀🏽#$%  'M‍İ ㍿\r\n​'S­/\r\ncamelCase'S>…#$%a/b\n/Bé", "tokens": 46, "pieces": [" \n", " ع", "\r\n\r\n", "a", "̊😀🏽#$%", " ", " ", "'M", "‍İ", " ", " ㍿\r\n", "​'", "S", "­/\r\n", "camelCase", "'S", ">", "…", "#$%", "a", "/b", "\n", "/Bé"]} +{"text": " (#$%\"é.camelCase(-t \r\n\r\n>㍿\n/\r\n \n 's㋿a/b\t<|fim_prefix|>٣٤٥٦ᵃt\t > ", "tokens": 50, "pieces": [" ", " (#$%\"", "e", "́.", "camelCase", "(-", "t", " \r\n\r\n", ">㍿\n", "/\r\n", " \n", " '", "s", "㋿a", "/b", "\t", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "ᵃt", "\t", " >", " "]} +{"text": "Z㋿EOT㋿'sdZⅣ're(dꟲ㍿'½عß", "tokens": 26, "pieces": ["Z", "㋿EOT", "㋿'", "sdZ", "Ⅳ", "'re", "(dꟲ", "㍿'", "½", "عß"]} +{"text": "ABC/\r\na👍🏽ᵃcamelCase½mß́camelCase👍🏽t \n<|fim_prefix|>a 12345678/漢Za/b́m\r>aB
m ́'M'ſ
 >…\u000b'll", "tokens": 69, "pieces": ["ABC", "/\r\n", "a", "👍🏽", "ᵃcamelCase", "½", "mß", "́camelCase", "👍🏽", "t", " \n", "<|", "fim", "_prefix", "|>", "a", " ", "123", "456", "78", "/漢Za", "/b", "́m", "\r", ">aB", "
m", " ́'", "M", "'ſ", "
", " ", ">", "…", "\u000b", "'ll"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n३㋿é㋿", "tokens": 11, "pieces": ["\r\n", "३", "㋿e", "́㋿"]} +{"text": " \r\n\r\n!ḍ̇ABC'VEå𐞁<|fim_prefix|>'s٣٤٥٦ꟲé<|endoftext|>a/btHTTPServeŕ'VEḍ̇!!\"t\t \n ,camelCase'VE>camelCase\n👍🏽", "tokens": 83, "pieces": [" \r\n\r\n", "!ḋ", "̣ABC", "'VE", "a", "̊𐞁", "<|", "fim", "_prefix", "|>'", "s", "٣٤٥", "٦", "ꟲ", "e", "́<|", "endoftext", "|>", "a", "/btHTTPServer", "́'", "VEḋ", "̣!!\"<", "EOT", ">t", "\t \n", " ,", "camelCase", "'VE", ">camelCase", "\n", "👍🏽"]} +{"text": ".㋿!a \n !aBḍ̇‍👍🏽12345678Dž", "tokens": 31, "pieces": [".<", "EOT", ">㋿!", "a", " \n", " !", "aBḋ", "̣‍👍🏽", "123", "456", "78", "Dž"]} +{"text": "عZ \n\nİ\u000b. …'s/İ,👍🏽", "tokens": 19, "pieces": ["عZ", " \n\n", "İ", "\u000b", ".", " ", "…", "'s", "/İ", ",👍🏽"]} +{"text": "s​", "tokens": 2, "pieces": ["s", "​"]} +{"text": "३EOT٣٤٥٦d\t're0iOSABCAbſaB\r\n\r\nİ\t!<|endoftext|>㍿ ع'llꟲaB३\"aB​.ᵃ9 t", "tokens": 59, "pieces": ["३", "EOT", "٣٤٥", "٦", "d", "\t", "'re", "0", "iOSABCAbſaB", "\r\n\r\n", "İ", "\t", "!<|", "endoftext", "|>㍿<", "EOT", ">", " ع", "'ll", "ꟲaB", "३", "\"aB", "​.", "ᵃ", "9", " t"]} +{"text": "…‍ß\t字​…aB<|endoftext|> 'ſعAb /😀🏽", "tokens": 28, "pieces": ["漢", "<|", "endoftext", "|><|", "endoftext", "|>", " ", "'ſ", "عAb", " ", "/😀🏽"]} +{"text": "iOS\r\n\r\n 'Džungla<|endoftext|>'D
EOTſ", "tokens": 22, "pieces": ["iOS", "\r\n\r\n", " ", "'Džungla", "<|", "endoftext", "|>'", "D", "
EOTſ"]} +{"text": "!😀🏽字'M‍Dž0 12345678fi½…a", "tokens": 23, "pieces": ["!😀🏽", "字", "'M", "‍Dž", "0", " ", "123", "456", "78", "fi", "½", "…a"]} +{"text": "'S'Mm'\"#$%'re\t-👍🏽㍿\r\n\r\n.Z'ſEOT AbAb<|fim_prefix|>/. \n 字iOS‍ḍ̇字9", "tokens": 52, "pieces": ["'S", "'M", "m", "'\"#$%'", "re", "\t", "-<", "META", "_START", ">👍🏽㍿\r\n\r\n", ".Z", "'ſ", "EOT", " ", " AbAb", "<|", "fim", "_prefix", "|>/.", " \n", " 字iOS", "‍ḋ", "̣字", "9"]} +{"text": " 's\n/㋿'Re👍🏽/\r\n' fi'S/DžunglacamelCase \n's(́m<|fim_prefix|>ᵃ👍🏽\r,'Re'M­𐞁\r\n\r\n\"字­🙂e<|endoftext|>漢Z12345678", "tokens": 78, "pieces": [" ", "'s", "\n", "/㋿'", "Re", "👍🏽/\r\n", "'", " fi", "'S", "/DžunglacamelCase", " \n", "'s", "(́", "m", "<|", "fim", "_prefix", "|>", "ᵃ", "👍🏽\r", ",'", "Re", "'M", "­𐞁", "\r\n\r\n", "\"字", "­🙂", "e", "<|", "endoftext", "|>", "漢Z", "123", "456", "78"]} +{"text": "camelCase<ꟲ㍿\"ع😀🏽😀🏽", "tokens": 24, "pieces": ["camelCase", "<ꟲ", "㍿\"", "ع", "😀🏽😀🏽"]} +{"text": "é'D‍<|fim_prefix|>ḍ̇å½ſ​…ḍ̇'S'VE\r\n\r\n'll字a0ᵃ0Dž३0 \n ß \u000b𐞁ABCDž", "tokens": 60, "pieces": ["é", "'D", "‍<|", "fim", "_prefix", "|>", "ḋ", "̣a", "̊", "½", "ſ", "​", "…ḋ", "̣'", "S", "'VE", "\r\n\r\n", "'ll", "字a", "0", "ᵃ", "0", "Dž", "३0", " \n", " ß", " ", "\u000b𐞁ABCDž"]} +{"text": "!/\r\n‍s­Dža/b😀🏽", "tokens": 15, "pieces": ["!/\r\n", "‍s", "­Dža", "/b", "😀🏽"]} +{"text": "é's‍…٣٤٥٦t<|fim_prefix|>'VE٣٤٥٦,a/bDžungla\nAb/\r\n/\r\n\tacamelCase9éABC", "tokens": 50, "pieces": ["e", "́'", "s", "‍", "…", "٣٤٥", "٦", "t", "<|", "fim", "_prefix", "|>'", "VE", "٣٤٥", "٦", ",a", "/bDžungla", "\n", "Ab", "/\r\n", "/\r\n", "\tacamelCase", "9", "e", "́ABC"]} +{"text": " ⅣéZİ­'ſå\r\n\r\nAbAb🙂٣٤٥٦'ſᵃꟲ😀🏽ᵃ/ 12345678/\r\nABC𐞁a𐞁ع字12345678😀🏽ع", "tokens": 77, "pieces": [" ", "Ⅳ", "éZİ", "­'", "ſa", "̊\r\n\r\n", "AbAb", "🙂", "٣٤٥", "٦", "'ſ", "ᵃꟲ", "😀🏽", "ᵃ", "/<", "META", "_START", ">", " ", " ", "123", "456", "78", "/\r\n", "ABC𐞁a𐞁ع字", "123", "456", "78", "😀🏽", "ع", ""]} +{"text": ">'t", "tokens": 2, "pieces": [">'", "t"]} +{"text": "ꟲꟲe!aB­\r漢,é​éHTTPServerDžungla ‍ #$%< #$%ᵃ👍🏽/\r\nHTTPServer🙂 -9­DžZ𐞁", "tokens": 58, "pieces": ["ꟲꟲe", "!aB", "­\r", "漢", ",e", "́​", "e", "́HTTPServerDžungla", " ‍", " #$%<", " ", " #$%", "ᵃ", "👍🏽/\r\n", "HTTPServer", "🙂", " ", "-", "9", "­DžZ𐞁"]} +{"text": "!'s'>'Re9>३🙂Džunglaß字字aB\r<|endoftext|>ADžDž'll<|endoftext|>ᵃaB!<|fim_prefix|>ḍ̇́å \n٣٤٥٦", "tokens": 72, "pieces": ["!'", "s", "'>'", "Re", "9", ">", "३", "🙂Džunglaß字字aB", "\r", "<|", "endoftext", "|>", "ADžDž", "'ll", "<|", "endoftext", "|>", "ᵃaB", "!<|", "fim", "_prefix", "|>", "ḋ", "̣́", "a", "̊", " \n", "٣٤٥", "٦"]} +{"text": "aⅣ<|endoftext|>åß㋿/ \n  \rß-\r\n! \nع'reſDžungla <|fim_prefix|>İ,‍字", "tokens": 49, "pieces": ["a", "Ⅳ", "<|", "endoftext", "|>", "a", "̊ß", "㋿/<", "EOT", ">", " \n  \r", "ß", "-\r\n", "!", " \n", "ع", "'re", "ſDžungla", " ", "<|", "fim", "_prefix", "|>", "İ", ",‍", "字"]} +{"text": "'ſ", "tokens": 3, "pieces": ["'ſ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ABCß/\n/>'aB㋿e0\nع'\n𐞁 \n😀🏽.é𐞁/\r\n é­ İ'll<|endoftext|>!", "tokens": 47, "pieces": ["ABCß", "/\n", "/>'", "aB", "㋿e", "0", "\n", "ع", "'\n", "𐞁", " \n", "😀🏽.", "é𐞁", "/\r\n", " e", "́­", " İ", "'ll", "<|", "endoftext", "|>!"]} +{"text": ".és٣٤٥٦İfi'D\n", "tokens": 15, "pieces": [".és", "٣٤٥", "٦", "İfi", "'D", "\n"]} +{"text": "/\r\n'a/b'ſZDž \n𐞁ꟲ'T<,,<|endoftext|>㍿'DABCꟲ👍🏽ᵃ\u000b३ \nA", "tokens": 55, "pieces": ["/\r\n", "'", "a", "/b", "'ſ", "ZDž", " \n", "𐞁ꟲ", "'T", "<,,<|", "endoftext", "|>㍿'", "DABCꟲ", "👍🏽", "ᵃ", "\u000b", "३", " \n", "A"]} +{"text": "😀🏽ꟲtAb/\r\n/\r\n३٣٤٥٦Ⅳ​>/EOTcamelCaseA٣٤٥٦", "tokens": 40, "pieces": ["😀🏽", "ꟲtAb", "/\r\n", "/\r\n", "३٣٤", "٥٦Ⅳ", "​>/", "EOTcamelCaseA", "٣٤٥", "٦"]} +{"text": "å \n å 字A㋿\ra/bß'Re!­\n/!㋿", "tokens": 28, "pieces": ["a", "̊", " \n", " a", "̊", " 字A", "㋿\r", "a", "/bß", "'Re", "!­\n", "/!㋿"]} +{"text": "a/bé9㍿عå'M👍🏽 iOS/ᵃ", "tokens": 27, "pieces": ["a", "/be", "́", "9", "㍿عa", "̊'", "M", "👍🏽", " iOS", "/<", "META", "_START", ">ᵃ"]} +{"text": "ß/\r\nt(‍'S!'TiOS0a/bİDž㋿ \n , å'EOT<|endoftext|>a/bm'\r½", "tokens": 57, "pieces": ["ß", "/\r\n", "t", "(‍'", "S", "!'", "TiOS", "0", "a", "/bİDž", "㋿", " \n", " ,", " ", "a", "̊'", "EOT", "<|", "endoftext", "|>", "a", "/bm", "'\r", "½"]} +{"text": "'ſ'sß'SsABC😀🏽\t'T'll\r\n३'٣٤٥٦ḍ̇\r\n'T漢/İ12345678aBEOT", "tokens": 45, "pieces": ["'ſ", "'s", "ß", "'S", "sABC", "😀🏽", "\t", "'T", "'ll", "\r\n", "३", "'", "٣٤٥", "٦", "ḋ", "̣\r\n", "'T", "漢", "/İ", "123", "456", "78", "aBEOT"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " 字0a 'VE\t'VE 𐞁'ſ\u000b𐞁​", "tokens": 23, "pieces": [" 字", "0", "a", " ", "'VE", "\t", "'VE", " 𐞁", "'ſ", "\u000b𐞁", "​"]} +{"text": "\r\n<|endoftext|>", "tokens": 8, "pieces": ["\r\n", "<|", "endoftext", "|>"]} +{"text": "fiEOT🙂.,aB\n/\n/ 🙂, <|fim_prefix|>'Mdſ camelCaseEOT🙂…👍🏽\u000b٣٤٥٦a!!EOT'Re👍🏽fiå'll<|endoftext|>‍m", "tokens": 76, "pieces": ["fiEOT", "🙂.,", "aB", "\n", "/\n", "/", " ", "🙂,", " <|", "fim", "_prefix", "|>'", "Mdſ", " camelCaseEOT", "🙂", "…", "👍🏽", "\u000b", "٣٤٥", "٦", "a", "!!", "EOT", "'Re", "👍🏽", "fia", "̊'", "ll", "<|", "endoftext", "|>‍", "m"]} +{"text": "​12345678m!!éİéiOSABC'll >Ⅳmİ<|fim_prefix|>\r!Ab /camelCase're'M12345678½'M \n /\r\nm-'ſ HTTPServer🙂", "tokens": 53, "pieces": ["​", "123", "456", "78", "m", "!!", "éİéiOSABC", "'ll", " ", ">", "Ⅳ", "mİ", "<|", "fim", "_prefix", "|>\r", "!Ab", " ", " /", "camelCase", "'re", "'", "M", "123", "456", "78½", "'M", " \n", " /\r\n", "m", "-'", "ſ", " HTTPServer", "🙂"]} +{"text": "ABC're\r\n\r\n'TEOT'VEa/b(​字\r\n\t", "tokens": 15, "pieces": ["ABC", "'re", "\r\n\r\n", "'T", "EOT", "'VE", "a", "/b", "(​", "字", "\r\n\t"]} +{"text": "㍿字EOTİfi'ſ(", "tokens": 13, "pieces": ["㍿字EOTİfi", "'ſ", "("]} +{"text": "( Džunglá​㋿İ'VE'!", "tokens": 20, "pieces": ["(<", "META", "_START", ">", " ", " Džungla", "́​㋿", "İ", "'VE", "'!"]} +{"text": " \nfi‍字é‍\"HTTPServer­-é'fi \n/\n're\u000b㋿d'Ree", "tokens": 34, "pieces": [" <", "EOT", ">", " \n", "fi", "‍字é", "‍\"", "HTTPServer", "­-", "e", "́'", "fi", " \n", "/\n", "'re", "\u000b", "㋿d", "'Re", "e"]} +{"text": "/At'T㋿/\r\n
0HTTPServerEOT½ \tß‍𐞁\" ABC‍…'M'D/\r\nAb,​a/b'Mع-𐞁\t", "tokens": 52, "pieces": ["/At", "'T", "㋿/\r\n", "
", "0", "HTTPServerEOT", "", "½", " ", "\tß", "‍𐞁", "\"", " ", " ABC", "‍", "…", "'M", "'D", "/\r\n", "Ab", ",​", "a", "/b", "'M", "ع", "-𐞁", "\t"]} +{"text": "'ſAé\r\n\r\na'S👍🏽<|fim_prefix|>‍漢​'ll", "tokens": 33, "pieces": ["'ſ", "Aé", "\r\n\r\n", "a", "'S", "👍🏽<|", "fim", "_prefix", "|>‍<", "EOT", ">漢", "​'", "ll"]} +{"text": " ㍿>'VE\t'Té'VEZعDžsA\t>…<Džungla>", "tokens": 28, "pieces": [" ㍿>'", "VE", "\t", "'T", "é", "'VE", "ZعDžsA", "\t", ">", "…", "<Džungla", ">"]} +{"text": "HTTPServer('ſHTTPServer३camelCaseİBDž\rå>tA'M#$%/å٣٤٥٦9ꟲé ta/\r\n<|fim_prefix|>s.ꟲ\r\n/", "tokens": 63, "pieces": ["HTTPServer", "('", "ſHTTPServer", "३", "camelCaseİBDž", "\r", "a", "̊>", "tA", "'M", "#$%/", "a", "̊", "٣٤٥", "٦9", "ꟲe", "́", " ta", "/\r\n", "<|", "fim", "_prefix", "|>", "s", ".ꟲ", "\r\n", "/"]} +{"text": "a\n/Ab", "tokens": 48, "pieces": ["a", "\n", "/Ab", ""]} +{"text": "
/(em'T's(漢'Re  \n e 're\n/Z'ſå\r\n\r\n#$%#$%٣٤٥٦'Re
camelCase­'VE🙂", "tokens": 49, "pieces": ["
", "/(", "em", "'T", "'s", "(漢", "'Re", "  \n", " e", " ", "'re", "\n", "/Z", "'ſ", "a", "̊\r\n\r\n", "#$%#$%", "٣٤٥", "٦", "'Re", "
camelCase", "­'", "VE", "🙂"]} +{"text": "<|fim_prefix|>0 <|fim_prefix|>Ⅳ́ZcamelCaseⅣ‍'Re-<|endoftext|> ſ0a/bmᵃ", "tokens": 44, "pieces": ["<|", "fim", "_prefix", "|>", "0", " <|", "fim", "_prefix", "|>", "Ⅳ", "́ZcamelCase", "Ⅳ", "‍'", "Re", "-<|", "endoftext", "|>", " ", " ſ", "0", "a", "/bmᵃ"]} +{"text": "字éİ#$%ſZꟲᵃa iOS,camelCase👍🏽a​a0a0\u000bEOT😀🏽é", "tokens": 42, "pieces": ["字éİ", "#$%", "ſZꟲᵃa", " ", " iOS", ",camelCase", "👍🏽", "a", "​a", "0", "a", "0", "\u000bEOT", "😀🏽", "e", "́"]} +{"text": "\u000ba/bAdeᵃᵃ(漢½
fi٣٤٥٦a/b", "tokens": 57, "pieces": ["\u000ba", "/bAdeᵃᵃ", "(漢", "", "½", "
fi", "٣٤٥", "٦", "a", "/b"]} +{"text": "İ'VEİdDžungla​!!Abd\r'll'S'D'VE\n/", "tokens": 21, "pieces": ["İ", "'VE", "İdDžungla", "​!!", "Abd", "\r", "'ll", "'S", "'D", "'VE", "\n", "/"]} +{"text": "Ab½#$%'ll­(.𐞁३\r'll­😀🏽-d'D३½ع漢­'😀🏽
ꟲ./\r\n \naB\n\n0'Re \nd<|endoftext|>", "tokens": 63, "pieces": ["Ab", "½", "#$%'", "ll", "­(.<", "EOT", ">𐞁", "३", "\r", "'ll", "­😀🏽-", "d", "'D", "३½", "ع漢", "­'😀🏽", "
ꟲ", "./\r\n", " \n", "aB", "\n\n", "0", "'Re", " \n", "d", "<|", "endoftext", "|>"]} +{"text": "fi😀🏽\r\n\r\n<ع #$%\u000b \n 
", "tokens": 20, "pieces": ["fi", "😀🏽\r\n\r\n", "<", "ع", " ", "#$%", "\u000b \n 
"]} +{"text": "#$%'Me.Džunglaa'T٣٤٥٦a…fi\r\n\r\na'Tḍ̇字éAbAbå'Té're're-'\r\n\r\nscamelCasefi", "tokens": 51, "pieces": ["#$%'", "Me", ".Džunglaa", "'T", "٣٤٥", "٦", "a", "…fi", "\r\n\r\n", "a", "'T", "ḋ", "̣字e", "́AbAba", "̊'", "Té", "'re", "'re", "-'\r\n\r\n", "scamelCasefi"]} +{"text": "#$% \nḍ̇ZsⅣ<|endoftext|>Ⅳ㍿😀🏽", "tokens": 29, "pieces": ["#$%", " \n", "ḋ", "̣Zs", "Ⅳ", "<|", "endoftext", "|>", "Ⅳ", "㍿😀🏽"]} +{"text": "a/b­😀🏽'TAع㍿>'re٣٤٥٦a/b\"𐞁a/b0 😀🏽ß'Re<|fim_prefix|>", "tokens": 49, "pieces": ["a", "/b", "­😀🏽'", "TAع", "㍿>'", "re", "٣٤٥", "٦", "a", "/b", "\"𐞁a", "/b", "0", " 😀🏽", "ß", "'Re", "<|", "fim", "_prefix", "|>"]} +{"text": "sB,'D", "tokens": 4, "pieces": ["sB", ",'", "D"]} +{"text": "dt漢m#$%B𐞁fiḍ̇٣٤٥٦\n'sꟲDžunglaa👍🏽'VEſ字A㍿ſſB\u000be>\r\n'DİſtiOS ᵃcamelCase>iOS", "tokens": 77, "pieces": ["dt漢m", "#$%<", "EOT", ">B𐞁fiḋ", "̣", "٣٤٥", "٦", "\n", "'s", "ꟲDžunglaa", "👍🏽'", "VEſ字A", "㍿ſſB", "\u000be", ">\r\n", "'D", "İſtiOS", " ", " ᵃcamelCase", ">iOS"]} +{"text": "'T-A😀🏽ſ'TEOT \n \r\n\r\n\n/\nİꟲ漢>fimaBſ३", "tokens": 76, "pieces": ["'T", "-A", "😀🏽", "ſ", "'T", "EOT", "", " \n \r\n\r\n\n", "/\n", "İꟲ漢", ">fimaB", "", "ſ", "३"]} +{"text": "d\r\nꟲⅣ", "tokens": 7, "pieces": ["d", "\r\n", "ꟲ", "Ⅳ"]} +{"text": "'MiOS(\r\n\r\na/b'Mḍ̇éé\nİ‍fi\r\n\r\nßḍ̇ſⅣ'\"-İ(\u000b'VE(", "tokens": 63, "pieces": ["'M", "iOS", "(\r\n\r\n", "a", "/b", "'M", "ḋ", "̣<", "e", "'T", "AbcamelCase", "<|", "endoftext", "|>", "e", "́é", "\n", "İ", "‍fi", "\r\n\r\n", "ßḋ", "̣<", "EOT", ">ſ", "Ⅳ", "'\"-", "İ", "(", "\u000b", "'VE", "("]} +{"text": ",…\r‍Dž​'reé0\rDžungla'Re'Dfi/\r\n३­t'Rett'T'TDž字
٣٤٥٦>'VEa/b'Re\u000béé½Džİ'M0", "tokens": 57, "pieces": [",", "…\r", "‍Dž", "​'", "reé", "0", "\r", "Džungla", "'Re", "'D", "fi", "/\r\n", "३", "­t", "'Re", "tt", "'T", "'T", "Dž字", "
", "٣٤٥", "٦", ">'", "VEa", "/b", "'Re", "\u000béé", "½", "Džİ", "'M", "0"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "t#$%\t㋿EOT", "tokens": 9, "pieces": ["t", "#$%", "\t", "㋿EOT"]} +{"text": "90're \n(#$%-Dž .ᵃ
'll'ſſ­'DDžungla.漢åaBed👍🏽 \n ع!!0\"/字", "tokens": 52, "pieces": ["90", "'re", "", " \n", "(#$%-", "Dž", " ", ".ᵃ", "
", "'ll", "'ſ", "ſ", "­'", "DDžungla", ".漢a", "̊aBed", "👍🏽", " \n", " ع", "!!", "0", "\"/", "字"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "0>
A😀🏽'ſ٣٤٥٦!/\r\n!\r\n\r\n >é'VEⅣB'Ss😀🏽'sAb…-ᵃḍ̇iOS…/\r\ncamelCase\r\n#$%", "tokens": 64, "pieces": ["0", ">", "
A", "😀🏽'", "ſ", "٣٤٥", "٦", "!/\r\n", "!\r\n\r\n", " ", " >", "é", "'VE", "Ⅳ", "B", "'S", "s", "😀🏽'", "sAb", "…", "-ᵃḋ", "̣iOS", "…", "/\r\n", "camelCase", "\r\n", "#$%"]} +{"text": "Džfi0é\n/½'T#$%Ⅳ>…é\t‍d𐞁 <|endoftext|>!ß#$%tİİa‍/ع>…'S'D🙂", "tokens": 59, "pieces": ["Džfi", "0", "e", "́\n", "/", "½", "'T", "#$%", "Ⅳ", ">", "…", "e", "́", "\t", "‍d𐞁", " ", "<|", "endoftext", "|>!", "ß", "#$%", "tİİa", "‍/", "ع", ">", "…", "'S", "'D", "🙂"]} +{"text": "'M\r\n\u000b\ta/b\n//\r\n\r\nع㍿,Abᵃ\r", "tokens": 17, "pieces": ["'M", "\r\n", "\u000b", "\ta", "/b", "\n", "//\r\n\r\n", "ع", "㍿,", "Abᵃ", "\r"]} +{"text": "‍  ſABCå\"\naé👍🏽 a", "̊\"\n", "ae", "́👍🏽", " ", " ꟲ😀🏽camelCase,åDž\n/sé👍🏽Ab\rع'llcamelCase ,-iOS\rDžungla३…ꟲ㍿'re/aB", "tokens": 68, "pieces": ["9", "漢", "<|", "endoftext", "|>", "e", "́<", "EOT", ">ꟲ", "😀🏽", "camelCase", ",a", "̊Dž", "\n", "/se", "́👍🏽", "Ab", "\r", "ع", "'ll", "camelCase", " ", " ,-", "iOS", "\r", "Džungla", "३", "…ꟲ", "㍿'", "re", "/aB"]} +{"text": "0<|fim_prefix|>👍🏽'VE'Re!'re,Džunglaſ\r 'ſ>'Reſt
㋿Z!!'S", "tokens": 45, "pieces": ["0", "<|", "fim", "_prefix", "|>👍🏽'", "VE", "'Re", "!'", "re", ",Džunglaſ", "\r", " ", " '", "ſ", ">'", "Reſt", "
", "㋿Z", "!!'", "S"]} +{"text": "<\" \n 👍🏽HTTPServer'llꟲeém😀🏽👍🏽.<ᵃⅣ\u000b>𐞁s('M\u000bcamelCaseå字éDžd're a/b-㍿'D''T", "tokens": 70, "pieces": ["<\"<", "META", "_START", ">", " \n", " 👍🏽", "HTTPServer", "'ll", "ꟲee", "́m", "😀🏽👍🏽.<", "ᵃ", "Ⅳ", "\u000b", ">𐞁s", "('", "M", "\u000bcamelCasea", "̊字e", "́Džd", "'re", " ", " a", "/b", "-㍿'", "D", "''", "T"]} +{"text": "𐞁🙂'T<|endoftext|>\nDžungla \n'ReHTTPServer\raHTTPServerB ABC< <|endoftext|>aBⅣ​ écamelCase \n \r\n\r\n9A漢Dž½.t<
fi
\n", "tokens": 66, "pieces": ["𐞁", "🙂'", "T", "<|", "endoftext", "|>\n", "Džungla", " \n", "'Re", "HTTPServer", "\r", "aHTTPServerB", " ", " ABC", "<", " ", " <|", "endoftext", "|>", "aB", "Ⅳ", "​", " e", "́camelCase", " \n \r\n\r\n", "9", "A漢Dž", "½", ".t", "<", "
fi", "
\n"]} +{"text": "t​ée ", "tokens": 4, "pieces": ["t", "​ée", " "]} +{"text": "iOS \n­< !! \n<㋿  \rſcamelCaseⅣꟲ३camelCase- عiOS'S  ㋿é", "tokens": 38, "pieces": ["iOS", " \n", "­<", " ", "!!", " \n", "<㋿", "  \r", "ſcamelCase", "Ⅳ", "ꟲ", "३", "camelCase", "-", " عiOS", "'S", " ", " ", "㋿e", "́"]} +{"text": "<|fim_prefix|>ḍ̇camelCase'se'TeHTTPServerEOT \r👍🏽d😀🏽'VEDž३", "tokens": 53, "pieces": ["<|", "fim", "_prefix", "|>", "ḋ", "̣camelCase", "'s", "e", "'T", "e", "HTTPServerEOT", "", " \r", "👍🏽", "d", "😀🏽'", "VE", "Dž", "३"]} +{"text": "HTTPServer'reeé\r\n \nß", "tokens": 8, "pieces": ["HTTPServer", "'re", "ee", "́\r\n", " \n", "ß"]} +{"text": ">'VE0ᵃ>", "tokens": 11, "pieces": [">'", "VE", "", "0", "ᵃ", ">"]} +{"text": "/\r\né'Dß<|fim_prefix|>🙂'T00\r\n\r\n\"ع'TcamelCase-\u000b12345678ꟲa/b", "tokens": 37, "pieces": ["/\r\n", "e", "́'", "Dß", "<|", "fim", "_prefix", "|>🙂'", "T", "00", "\r\n\r\n", "\"ع", "'", "TcamelCase", "-", "\u000b", "123", "456", "78", "ꟲa", "/b"]} +{"text": "Džungla12345678>HTTPServer/\r\n​'re'VE.㋿12345678​ B'll- Zm​'é'SعⅣ'DⅣ/\r\nعB́B'D'Re\u000b٣٤٥٦", "tokens": 57, "pieces": ["Džungla", "123", "456", "78", ">HTTPServer", "/\r\n", "​'", "re", "'VE", ".㋿", "123", "456", "78", "​", " B", "'ll", "-", " Zm", "​'", "e", "́'", "Sع", "Ⅳ", "'D", "Ⅳ", "/\r\n", "عB", "́B", "'D", "'Re", "\u000b", "٣٤٥", "٦"]} +{"text": "\t𐞁🙂Dž\nDž
#$%́ ‍ \n \r\n\r\n'MAbmdABC-", "tokens": 35, "pieces": ["\t𐞁", "🙂Dž", "\n", "Dž", "
", "#$%́<", "META", "_START", ">", " ", " ‍", " \n \r\n\r\n", "'M", "Ab", "mdABC", "-"]} +{"text": "m½'D\n😀🏽\"🙂‍'Re 'D…éÁ३ꟲEOT\n/\r'M\u000b\n<́", "tokens": 42, "pieces": ["m", "½", "'D", "\n", "😀🏽\"🙂‍'", "Re", " ", "'D", "…e", "́A", "́", "३", "ꟲEOT", "\n", "/\r", "'M", "\u000b", "\n", "<́"]} +{"text": "<éBABC…👍🏽 'Ds'S0tHTTPServerABC​\r\n'll' 0<|endoftext|>", "tokens": 35, "pieces": ["<éBABC", "…", "👍🏽", " ", " '", "Ds", "'S", "0", "tHTTPServerABC", "​\r\n", "'ll", "'", " ", " ", "0", "<|", "endoftext", "|>"]} +{"text": "ßm\t\r\n\r\n
iOSDžungla \n a/b", "tokens": 13, "pieces": ["ßm", "\t\r\n\r\n", "
iOSDžungla", " \n", " a", "/b"]} +{"text": "‍\r\n\r\naBع'M㍿t\n 'SZ/\r\n/\r\n½é \n're. (
字", "tokens": 27, "pieces": ["‍\r\n\r\n", "aBع", "'M", "㍿t", "\n", " ", " '", "SZ", "/\r\n", "/\r\n", "½", "é", " \n", "'re", ".", " ", "(", "
字"]} +{"text": "漢ſå\r\n\r\n<|fim_prefix|>\r\n😀🏽EOT<|fim_prefix|>½ iOS \n", "tokens": 33, "pieces": ["漢ſa", "̊\r\n\r\n", "<|", "fim", "_prefix", "|>\r\n", "😀🏽", "EOT", "<|", "fim", "_prefix", "|>", "½", " ", " iOS", " \n"]} +{"text": "'sᵃ㋿å'DABC'Teå<-'D½½9å", "tokens": 31, "pieces": ["'s", "ᵃ", "㋿a", "̊'", "DABC", "'T", "ea", "̊<-'", "D", "½", "", "½9", "a", "̊"]} +{"text": "İ<|fim_prefix|>éeaBa/b\n/iOS'D \n EOT,Ab'll!🙂d\r\n\r\n\r\n\r\nᵃcamelCase \n ABC٣٤٥٦", "tokens": 43, "pieces": ["İ", "<|", "fim", "_prefix", "|>", "e", "́eaBa", "/b", "\n", "/iOS", "'D", " \n", " EOT", ",Ab", "'ll", "!🙂", "d", "\r\n\r\n\r\n\r\n", "ᵃcamelCase", " \n", " ABC", "٣٤٥", "٦"]} +{"text": "ⅣaBİ<|endoftext|>'M…#$%'VEDž ", "tokens": 21, "pieces": ["Ⅳ", "aBİ", "<|", "endoftext", "|>'", "M", "…", "#$%'", "VEDž", " "]} +{"text": "a/b३t0/\r\n \n\r\n\r\n­\tḍ̇ᵃ'Re><|fim_prefix|>'THTTPServer字­camelCase \n'M ३'T'T", "tokens": 42, "pieces": ["a", "/b", "३", "t", "0", "/\r\n", " \n\r\n\r\n", "­", "\tḋ", "̣ᵃ", "'Re", "><|", "fim", "_prefix", "|>'", "THTTPServer字", "­camelCase", " \n", "'M", " ", " ", "३", "'T", "'T"]} +{"text": "​12345678😀🏽<|endoftext|>Dž \n ", "tokens": 30, "pieces": ["iOS", "🙂", " ", " B", "/\r\n", "\t", "㋿>", "123", "456", "78", "😀🏽<|", "endoftext", "|>", "Dž", " \n "]} +{"text": "ᵃ-\u000b12345678ſiOSa/b9<|endoftext|>'D'll'Dž'll'sſ#$%éſ<|endoftext|>9(EOTfi é#$%ée­­𐞁'llZ ㋿", "tokens": 67, "pieces": ["ᵃ", "-", "\u000b", "123", "456", "78", "ſiOSa", "/b", "9", "<|", "endoftext", "|>'", "D", "'ll", "'Dž", "'ll", "'s", "ſ", "#$%", "e", "́ſ", "<|", "endoftext", "|>", "9", "(EOTfi", " é", "#$%", "e", "́e", "­­", "𐞁", "'ll", "Z", " ", " ㋿"]} +{"text": "a/b(Dž३<|fim_prefix|>…😀🏽9éᵃ", "tokens": 26, "pieces": ["a", "/b", "(Dž", "३", "<|", "fim", "_prefix", "|>", "…", "😀🏽", "9", "éᵃ"]} +{"text": "-𐞁/ḍ̇!İ'VEDžunglaéfi'Td
(́(ع/", "tokens": 31, "pieces": ["-𐞁", "/ḋ", "̣!", "İ", "'VE", "Džunglaéfi", "'T", "d", "
", "(́(", "ع", "/"]} +{"text": "12345678-\n/aB­/!<|endoftext|>", "tokens": 15, "pieces": ["123", "456", "78", "-\n", "/aB", "­/!<|", "endoftext", "|>"]} +{"text": "å㋿d…é'/  \n İ\"é\u000b漢-😀🏽s <|endoftext|>½é­ABC,
 ḍ̇\n HTTPServerHTTPServer", "tokens": 58, "pieces": ["a", "̊㋿", "d", "", "…e", "́'/", "  \n", " İ", "\"e", "́", "\u000b漢", "-😀🏽", "s", " ", "<|", "endoftext", "|>", "½", "e", "́­", "ABC", ",", "
", " ḋ", "̣\n", " ", " HTTPServerHTTPServer"]} +{"text": "İßs
ꟲa/b", "tokens": 10, "pieces": ["İßs", "
ꟲa", "/b"]} +{"text": "'ll'sé<|endoftext|>३å'M>!", "tokens": 19, "pieces": ["'ll", "'s", "é", "<|", "endoftext", "|>", "३", "a", "̊'", "M", ">!"]} +{"text": "sZAb  \n0ſB(\u000bd𐞁\r\n\r\n>'SDž /३عⅣd'M(aBé,…\n<|endoftext|>'S'll0Ab👍🏽
", "tokens": 56, "pieces": ["sZAb", "  \n", "0", "ſB", "(", "\u000bd𐞁", "\r\n\r\n", ">'", "SDž", " ", "/", "३", "ع", "Ⅳ", "d", "'M", "(aBé", ",", "…\n", "<|", "endoftext", "|>'", "S", "'ll", "0", "Ab", "👍🏽", "
"]} +{"text": "‍e'śع\"\r 12345678ß-́,9#$%\n/ſ \n -d‍字'SaB\na/bcamelCaseİ\n/", "tokens": 43, "pieces": ["‍e", "'s", "́ع", "\"\r", " ", " ", "123", "456", "78", "ß", "-́,", "9", "#$%\n", "/ſ", " \n", " -", "d", "‍字", "'S", "aB", "\n", "a", "/bcamelCase", "İ", "\n", "/"]} +{"text": "Dž \n 𐞁🙂<|endoftext|>'s9<|endoftext|>\u000b字'T́(\"9\r\n\r\n'siOS字<|fim_prefix|><|endoftext|>'Re'M\r\nd<|fim_prefix|>'reé🙂", "tokens": 64, "pieces": ["Dž", " \n", " 𐞁", "🙂<|", "endoftext", "|>'", "s", "9", "<|", "endoftext", "|>", "\u000b字", "'T", "́(\"", "9", "\r\n\r\n", "'s", "iOS字", "<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "Re", "'M", "\r\n", "d", "<|", "fim", "_prefix", "|>'", "ree", "́🙂"]} +{"text": " EOT\u000bfidZ#$%HTTPServer/iOSᵃ㍿́\r\"ݽ…\nm.Ab'M'S٣٤٥٦Džungla(éꟲ'll", "tokens": 50, "pieces": [" EOT", "\u000bfidZ", "#$%", "HTTPServer", "/iOSᵃ", "㍿́\r", "\"İ", "½", "…\n", "m", ".Ab", "'M", "'S", "٣٤٥", "٦", "Džungla", "(e", "́ꟲ", "'ll"]} +{"text": "ᵃ\r\n.#$%!!٣٤٥٦\ndAB0\n/́ꟲ !!𐞁B(ſABC字HTTPServer<|fim_prefix|>\r \nABCé\n😀🏽HTTPServera/b\t'12345678", "tokens": 72, "pieces": ["ᵃ", "\r\n", ".#$%!!", "٣٤٥", "٦", "\n", "dAB", "0", "\n", "/́", "ꟲ", " !!<", "META", "_START", ">𐞁B", "(ſABC字HTTPServer", "<|", "fim", "_prefix", "|>\r", " \n", "ABCe", "́\n", "😀🏽", "HTTPServera", "/b", "\t", "'", "123", "456", "78"]} +{"text": "( \n㋿td🙂å'så'M\n/camelCase½\n,ABCå\n0/", "tokens": 31, "pieces": ["(", " \n", "㋿td", "🙂a", "̊'", "sa", "̊'", "M", "\n", "/camelCase", "½", "\n", ",ABCa", "̊\n", "0", "/"]} +{"text": "\n/>-HTTPServer\nAb'VE'S#$%<|endoftext|>camelCaseß'VEعDžungla>ꟲéß\r'(", "tokens": 50, "pieces": ["\n", "/>-", "HTTPServer", "\n", "Ab", "'VE", "'S", "#$%<|", "endoftext", "|>", "camelCaseß", "'VE", "ع", "Džungla", ">ꟲe", "́ß", "\r", "'(<", "META", "_START", ">"]} +{"text": "ꟲa/b", "tokens": 5, "pieces": ["ꟲa", "/b"]} +{"text": "09HTTPServer's㋿iOS#$%ᵃ,ⅣaB'siOS", "tokens": 20, "pieces": ["09", "HTTPServer", "'s", "㋿iOS", "#$%", "ᵃ", ",", "Ⅳ", "aB", "'s", "iOS"]} +{"text": "é", "tokens": 1, "pieces": ["é"]} +{"text": "Džungla㋿'s", "tokens": 9, "pieces": ["Džungla", "㋿'", "s"]} +{"text": "字…‍HTTPServerA‍' \n 'D\n\r\n a/b…\n😀🏽12345678Bfi'Re🙂s\"㋿!😀🏽\r\n­", "tokens": 58, "pieces": ["字", "", "…", "‍HTTPServer", "A", "‍'", " \n", " '", "D", "\n\r\n", " a", "/b", "…\n", "😀🏽", "123", "456", "78", "Bfi", "'Re", "🙂s", "\"㋿!😀🏽\r\n", "­"]} +{"text": " #$%éḍ̇\t३́afi, Ab\r\n\r\na٣٤٥٦\",­'re!!ABC\r\né!\r٣٤٥٦'re\r\r\n\r\n­ᵃ'VE\té<|fim_prefix|>", "tokens": 64, "pieces": [" ", "#$%", "éḋ", "̣", "\t", "३", "́afi", ",", " Ab", "\r\n\r\n", "a", "٣٤٥", "٦", "\",­'", "re", "!!", "ABC", "\r\n", "é", "!\r", "٣٤٥", "٦", "'re", "\r\r\n\r\n", "­ᵃ", "'VE", "\te", "́<|", "fim", "_prefix", "|>"]} +{"text": "İAé,", "tokens": 5, "pieces": ["İAé", ","]} +{"text": "‍ <|fim_prefix|>,'D.'ſ/\r.a0m\n!'reAb\"👍🏽/Ab!!Ab३​\tABC", "tokens": 42, "pieces": ["‍", " ", " <|", "fim", "_prefix", "|>,'", "D", ".'", "ſ", "/\r", ".a", "0", "m", "\n", "!'", "reAb", "\"👍🏽<", "META", "_START", ">/", "Ab", "!!", "Ab", "३", "​", "\tABC"]} +{"text": "ⅣcamelCase<|endoftext|>\" \r\n<|fim_prefix|>'rea/bᵃtfi'reᵃB👍🏽٣٤٥٦ABCe'DBᵃ\rm\"\r\r\n d عDžungla\r\n0'Re𐞁 \n!ſ", "tokens": 76, "pieces": ["Ⅳ", "camelCase", "<|", "endoftext", "|>\"", " \r\n", "<|", "fim", "_prefix", "|>'", "rea", "/bᵃtfi", "'re", "ᵃB", "👍🏽", "٣٤٥", "٦", "ABCe", "'D", "Bᵃ", "\r", "m", "\"\r\r\n", " d", " عDžungla", "\r\n", "0", "'Re", "𐞁", " \n", "!ſ"]} +{"text": "İ漢\"aZ½‍<|fim_prefix|>d12345678B're", "tokens": 21, "pieces": ["İ漢", "\"aZ", "½", "‍<|", "fim", "_prefix", "|>", "d", "123", "456", "78", "B", "'re"]} +{"text": " 0字…½\n'ſ/\r\né½t12345678漢ḍ̇.漢漢>é-/\r\n'ḍ̇\"Ⅳ'D👍🏽\n#$%<|endoftext|>", "tokens": 60, "pieces": [" ", "0", "字", "…", "½", "\n", "'ſ", "/\r\n", "e", "́", "½", "t", "123", "456", "78", "漢ḋ", "̣.", "漢漢", ">é", "-/\r\n", "'ḋ", "̣\"", "Ⅳ", "'D", "👍🏽\n", "#$%<|", "endoftext", "|>"]} +{"text": "
å'VE‍'D \n
ß'VE \n👍🏽٣٤٥٦​𐞁'TAb𐞁…tHTTPServer👍🏽", "tokens": 54, "pieces": ["
a", "̊'", "VE", "‍'", "D", " \n", "
ß", "'VE", " \n", "👍🏽", "٣٤٥", "٦", "​𐞁", "'T", "Ab𐞁", "…tHTTPServer", "👍🏽"]} +{"text": "'ſDžes\n'ſHTTPServer'S㋿e
  \n HTTPServer…\n
iOSZ㍿ß\n\u000b/👍🏽'reع漢 \n ḍ̇-'VE½", "tokens": 57, "pieces": ["'ſ", "Džes", "\n", "'ſ", "HTTPServer", "'S", "㋿e", "
  \n", " HTTPServer", "…\n", "
iOSZ", "㍿ß", "\n", "\u000b", "/👍🏽'", "reع漢", " \n", " ḋ", "̣-'", "VE", "½"]} +{"text": "'T½\"😀🏽'TBd/'re\r\n\r\nſ é٣٤٥٦'Re…<
", "tokens": 32, "pieces": ["'T", "½", "\"😀🏽'", "TBd", "/'", "re", "\r\n\r\n", "ſ", " ", " é", "٣٤٥", "٦", "'Re", "…", "<", "
"]} +{"text": "-\u000ba0Z9é '½.", "tokens": 11, "pieces": ["-", "\u000ba", "0", "Z", "9", "e", "́", " '", "½", "."]} +{"text": "३'S\nİa/b​A. s \n ㋿/å.e!'Mḍ̇'DiOS", "tokens": 34, "pieces": ["३", "'S", "\n", "İa", "/b", "​A", ".", " s", " \n", " ㋿/", "a", "̊.", "e", "!'", "Mḋ", "̣'", "DiOS"]} +{"text": "…/Džunglaİ
é\te \né \n eB <\rAᵃ'ßBa\n/\n/½\t३sAb𐞁camelCase're/ 🙂", "tokens": 54, "pieces": ["…", "/<", "EOT", ">Džunglaİ", "
e", "́", "\te", " \n", "é", " \n", " eB", "", " ", " <\r", "Aᵃ", "'ßBa", "\n", "/\n", "/", "½", "\t", "३", "sAb𐞁camelCase", "'re", "/", " 🙂"]} +{"text": "­'ReZ
字Z", "tokens": 8, "pieces": ["­'", "ReZ", "
字Z"]} +{"text": "ABC\n/漢'll\n\r\n\r\n🙂́ᵃAꟲ'll'Réd>字-#$%😀🏽e½/\r\naBa/b'Re'", "ll", "'Re", "́d", ">字", "-#$%😀🏽", "e", "½", "/\r\n", "aBa", "/b", "'Re", "d9a\r\u000bꟲ'Re
漢éAb're<|endoftext|>\n0
EOTAd٣٤٥٦😀🏽", "tokens": 75, "pieces": ["a", "'D", "a", "̊‍", " ", "'VE", "a", "/b", "'ll", "漢", " ", "\"", "…", "㋿ꟲ", "\n", "Dž", "d", "9", "a", "\r", "\u000bꟲ", "'Re", "
漢éAb", "'re", "<|", "endoftext", "|>\n", "0", "
EOTAd", "٣٤٥", "٦", "😀🏽"]} +{"text": "12345678İ \nꟲEOTaعZ İ'ſ‍ \n12345678Džunglaé㍿eDž<|fim_prefix|>‍‍٣٤٥٦s\téⅣ12345678Z字aB(㋿fiAbé", "tokens": 75, "pieces": ["123", "456", "78", "İ", " \n", "ꟲEOTaعZ", " İ", "'ſ", "‍", " \n", "123", "456", "78", "Džunglaé", "㍿eDž", "<|", "fim", "_prefix", "|>‍‍", "٣٤٥", "٦", "s", "\te", "́", "Ⅳ12", "345", "678", "Z字aB", "(㋿", "fiAbe", "́"]} +{"text": "DžABC\r\n<|fim_prefix|>🙂‍Ab㍿٣٤٥٦Dž <|fim_prefix|>B#$%HTTPServer", "tokens": 41, "pieces": ["DžABC", "\r\n", "<|", "fim", "_prefix", "|>🙂‍", "Ab", "㍿", "٣٤٥", "٦", "Dž", " ", " <|", "fim", "_prefix", "|>", "B", "#$%", "HTTPServer"]} +{"text": "'sDžunglaHTTPServer  \n \n/漢'VE👍🏽ⅣAb0éⅣaå३12345678🙂'M,\u000b\r…٣٤٥٦Džungla३'Re(ſ'D", "tokens": 69, "pieces": ["'s", "DžunglaHTTPServer", "  \n \n", "/漢", "'VE", "👍🏽", "Ⅳ", "Ab", "0", "é", "Ⅳ", "aa", "̊", "३12", "345", "678", "🙂'", "M", ",", "\u000b\r", "…", "٣٤٥", "٦", "Džungla", "३", "'", "Re", "(ſ", "'D"]} +{"text": "'Mİ‍'sA.-½ ꟲ'VEḍ̇éᵃ \nAb12345678", "tokens": 30, "pieces": ["'M", "İ", "‍'", "sA", ".-", "½", " ꟲ", "'VE", "ḋ", "̣éᵃ", " \n", "Ab", "123", "456", "78"]} +{"text": "A A𐞁å㋿aß#$%a/\r\n,
s½>😀🏽(½
漢EOTABCع\u000b'S'-İ
m👍🏽<|endoftext|>\r\naᵃ", "tokens": 66, "pieces": ["A", " A𐞁a", "̊㋿", "aß", "#$%", "a", "/\r\n", ",", "
s", "½", ">😀🏽(", "½", "
漢EOTABCع", "\u000b", "'S", "'-", "İ", "
m", "👍🏽<|", "endoftext", "|>\r\n", "aᵃ"]} +{"text": "#$%'VE(İ \n Ab­ 0a'Re - 'S'Re­/\r\nB", "tokens": 22, "pieces": ["#$%'", "VE", "(İ", " \n", " Ab", "­", " ", " ", "0", "a", "'Re", " ", "-", " ", " '", "S", "'Re", "­/\r\n", "B"]} +{"text": "!!٣٤٥٦'Re \n👍🏽ꟲDžunglaa/b'ſßcamelCase(\r.'DaZs", "tokens": 41, "pieces": ["!!", "٣٤٥", "٦", "'Re", " \n", "👍🏽", "ꟲDžunglaa", "/b", "'ſ", "ßcamelCase", "(\r", ".'", "DaZs"]} +{"text": "sⅣ\n/ꟲa🙂½'M\n", "tokens": 20, "pieces": ["s", "", "Ⅳ", "\n", "/ꟲa", "🙂", "½", "'M", "\n"]} +{"text": "­​字Dž\t'Tß\r\niOS's'MAbİ >é'll/\r\nAbiOSa.​DžB😀🏽́", "tokens": 37, "pieces": ["­​", "字Dž", "\t", "'T", "ß", "\r\n", "iOS", "'s", "'M", "Abİ", " ", ">e", "́'", "ll", "/\r\n", "AbiOSa", ".​", "DžB", "😀🏽́"]} +{"text": "B\t/\r\n𐞁a/b<|fim_prefix|>Dž#$%\u000b\n/mcamelCase", "tokens": 28, "pieces": ["B", "\t", "/\r\n", "𐞁a", "/b", "<|", "fim", "_prefix", "|>", "Dž", "#$%", "\u000b\n", "/mcamelCase"]} +{"text": "!#$%'VEع<|fim_prefix|>HTTPServer\n<­\r\n\r\n‍字😀🏽é­,B<|fim_prefix|>३ \ne👍🏽\r\n\r\nd>", "tokens": 66, "pieces": ["!#$%'", "VEع", "<|", "fim", "_prefix", "|>", "HTTPServer", "\n", "<­\r\n\r\n", "‍字", "😀🏽", "e", "́<", "sꟲdß", "<𐞁", "
", ">­,", "B", "<|", "fim", "_prefix", "|>", "३", " \n", "e", "👍🏽\r\n\r\n", "d", ">"]} +{"text": "½tİ\u000b
", "tokens": 6, "pieces": ["½", "tİ", "\u000b
"]} +{"text": "'VE!!\n/ع😀🏽", "tokens": 10, "pieces": ["'VE", "!!\n", "/ع", "😀🏽"]} +{"text": "
 \n ABC\tfi\"'VEAb㍿/s🙂ABCa/b‍ 'M0t\r\n'sEOT", "tokens": 32, "pieces": ["
 \n", " ABC", "\tfi", "\"'", "VEAb", "㍿/", "s", "🙂ABCa", "/b", "‍", " ", "'M", "0", "t", "\r\n", "'s", "EOT"]} +{"text": "\r\n\r\n#$%!!​½A'M 'TiOS", "tokens": 13, "pieces": ["\r\n\r\n", "#$%!!​", "½", "A", "'M", " ", " '", "TiOS"]} +{"text": "9éDžunglaEOTtda/bİ12345678<|fim_prefix|>#$%ß𐞁m\r\n/\r\nع9camelCases\u000b'SDžunglaḍ̇éꟲEOT<|fim_prefix|>aſ,/", "tokens": 70, "pieces": ["9", "e", "́DžunglaEOTt", "da", "/bİ", "123", "456", "78", "<|", "fim", "_prefix", "|>#$%", "ß𐞁m", "\r\n", "/\r\n", "ع", "9", "camelCases", "\u000b", "'S", "Džunglaḋ", "̣e", "́ꟲEOT", "<|", "fim", "_prefix", "|>", "aſ", ",/"]} +{"text": "EOT0 0Ⅳꟲ#$%字
Z'ſ'Ta/bḍ̇'llß", "tokens": 30, "pieces": ["EOT", "0", " ", "0Ⅳ", "ꟲ", "#$%", "字", "
Z", "'ſ", "'T", "a", "/bḋ", "̣'", "llß"]} +{"text": " \nDžunglam12345678\n/Ⅳ12345678EOT'lliOS'ſß", "tokens": 23, "pieces": [" \n", "Džunglam", "123", "456", "78", "\n", "/", "Ⅳ12", "345", "678", "EOT", "'ll", "iOS", "'ſ", "ß"]} +{"text": " \nEOTDž‍字\u000b👍🏽😀🏽字🙂", "tokens": 23, "pieces": [" \n", "EOTDž", "‍字", "\u000b", "👍🏽😀🏽", "字", "🙂"]} +{"text": "ꟲé12345678aB \n 12345678EOT \n<३'VE00…Ⅳ字iOSé'Reꟲ're's'siOSt 0aeعé's", "tokens": 56, "pieces": ["ꟲé", "123", "456", "78", "aB", " \n", " ", "123", "456", "78", "EOT", " \n", "<", "३", "'VE", "", "00", "…", "Ⅳ", "字iOSe", "́'", "Reꟲ", "'re", "'s", "'s", "iOSt", " ", "0", "aeعé", "'s", ""]} +{"text": "'ll'Sé/a/b(d eꟲ/\r\nAå​'reAb'M#$%'sss\rḍ̇ ㋿/\r\nm", "tokens": 37, "pieces": ["'ll", "'S", "é", "/a", "/b", "(d", " eꟲ", "/\r\n", "Aa", "̊​'", "reAb", "'M", "#$%'", "sss", "\r", "ḋ", "̣", " ", " ㋿/\r\n", "m"]} +{"text": "'ll \n/­", "tokens": 8, "pieces": ["'ll", " \n", "/­<", "EOT", ">"]} +{"text": "\"­t'll\"ééfi0Z's\n '㍿aB\"'re #$%…'ll\n/m漢३\u000b'D𐞁.ᵃ!!\t㍿Džungla😀🏽𐞁", "tokens": 64, "pieces": ["\"­<", "META", "_START", ">t", "'ll", "\"ééfi", "0", "Z", "'s", "\n", " ", " '㍿", "aB", "\"'", "re", " #$%", "…", "'ll", "\n", "/m漢", "३", "\u000b", "'D", "𐞁", ".ᵃ", "!!", "\t", "㍿Džungla", "😀🏽", "𐞁"]} +{"text": "sßAb'M\n\"\r\n\r\n<漢\tAb🙂DžA\t㍿\n/ ", "tokens": 24, "pieces": ["sßAb", "'M", "\n", "\"\r\n\r\n", "<漢", "\tAb", "🙂DžA", "\t", "㍿\n", "/", " "]} +{"text": "'s \n 👍🏽㍿ḍ̇🙂", "tokens": 17, "pieces": ["'s", " \n", " 👍🏽㍿", "ḋ", "̣🙂"]} +{"text": "(½ Z9s\r#$%12345678m'S
­\nDžunglafi𐞁ᵃmHTTPServer 'Re㍿㍿ \n B!!!'reZعİ<|endoftext|>", "tokens": 57, "pieces": ["(", "½", " Z", "9", "s", "\r", "#$%", "123", "456", "78", "m", "'S", "
", "­\n", "Džunglafi𐞁ᵃmHTTPServer", " '", "Re", "㍿㍿", " \n", " B", "!!!'", "reZعİ", "<|", "endoftext", "|>"]} +{"text": "\réA a/ba­…٣٤٥٦\n12345678漢", "tokens": 25, "pieces": ["\r", "éA", " a", "/ba", "­", "…", "٣٤٥", "٦", "\n", "123", "456", "78", "漢"]} +{"text": "\r\n\r\n'T#$%< \n 'VE ", "tokens": 9, "pieces": ["\r\n\r\n", "'T", "#$%<", " \n", " '", "VE", " "]} +{"text": "\n/字iOSع½\r\n\r\n
camelCase#$%٣٤٥٦å<|fim_prefix|>­\r'SfiⅣ", "tokens": 41, "pieces": ["\n", "/字iOSع", "½", "\r\n\r\n", "
", "camelCase", "#$%", "٣٤٥", "٦", "a", "̊<|", "fim", "_prefix", "|>­\r", "'S", "fi", "Ⅳ"]} +{"text": "'s漢𐞁İiOScamelCasedİe #$%d'MABCs \n 'reİa/b-ع'reHTTPServer \n!३camelCase<|fim_prefix|>😀🏽é.", "tokens": 54, "pieces": ["'s", "漢𐞁İiOScamelCasedİe", " ", " #$%", "d", "'M", "ABCs", " \n", " '", "reİa", "/b", "-", "ع", "'re", "HTTPServer", " \n", "!", "३", "camelCase", "<|", "fim", "_prefix", "|>😀🏽", "é", "."]} +{"text": "12345678Džungla'D ½a \n …<12345678ᵃ/aBDžungla\u000b㍿camelCase🙂mع\u000b<|fim_prefix|>'Tm字iOS<|endoftext|><|endoftext|>iOSſABC\tꟲ😀🏽\n/Ab\t", "tokens": 83, "pieces": ["123", "456", "78", "Džungla", "'D", " ", "½", "a", " \n", " ", "…", "<", "123", "456", "78", "ᵃ", "/aBDžungla", "\u000b", "㍿camelCase", "🙂mع", "\u000b", "<|", "fim", "_prefix", "|>'", "Tm字iOS", "<|", "endoftext", "|><|", "endoftext", "|>", "iOSſABC", "\tꟲ", "😀🏽\n", "/Ab", "\t"]} +{"text": "漢'ſ", "tokens": 5, "pieces": ["漢", "'ſ"]} +{"text": "… camelCasefiİ<३#$%㍿ḍ̇dB😀🏽", "tokens": 27, "pieces": ["… ", " camelCasefiİ", "<", "३", "#$%㍿", "ḋ", "̣dB", "😀🏽"]} +{"text": "\u000b\n/!\r\n\r\niOS'ſ12345678/\r\n ㍿'M<|fim_prefix|>Ab\r😀🏽ſ'll­", "tokens": 37, "pieces": ["\u000b\n", "/!\r\n\r\n", "iOS", "'ſ", "123", "456", "78", "/\r\n", " ㍿'", "M", "<|", "fim", "_prefix", "|>", "Ab", "\r", "😀🏽", "ſ", "'ll", "­"]} +{"text": "'Taſ​'ll0३'S'ſ'Sſ!!😀🏽,!!\u000bHTTPServer/​mé
9𐞁ß𐞁\r\n\r\n's'ᵃ🙂s", "tokens": 53, "pieces": ["'T", "aſ", "​'", "ll", "0३", "'S", "'ſ", "'S", "ſ", "!!😀🏽,!!", "\u000bHTTPServer", "/​", "mé", "
", "9", "𐞁ß𐞁", "\r\n\r\n", "'s", "'ᵃ", "🙂s"]} +{"text": "/\r\n👍🏽­​İİß're\n/'llZmB\n/́ſ́aBİع😀🏽're …'ſ ' .㋿camelCaseHTTPServerfit'sm", "tokens": 58, "pieces": ["/\r\n", "👍🏽­​", "İİß", "'re", "\n", "/'", "llZmB", "\n", "/́", "ſ", "́aBİع", "😀🏽'", "re", " ", "…", "'ſ", " ", " '", " ", ".㋿", "camelCaseHTTPServerfit", "'s", "m"]} +{"text": "ꟲ!!m'D,.,ع🙂!!iOS#$%\r\n…\"\rAb'SaBABC👍🏽>\u000b👍🏽t#$%Dž½", "tokens": 44, "pieces": ["ꟲ", "!!", "m", "'D", ",.,", "ع", "🙂!!", "iOS", "#$%\r\n", "…", "\"\r", "Ab", "'S", "aBABC", "👍🏽>", "\u000b", "👍🏽", "t", "#$%", "Dž", "½"]} +{"text": "Ab0\r\nſ🙂é,B-,fi'll'VE'BEOT
12345678\n/HTTPServerEOT👍🏽ع \n ⅣcamelCase🙂字🙂ßEOT\u000beiOS…🙂\"٣٤٥٦", "tokens": 68, "pieces": ["Ab", "0", "\r\n", "ſ", "🙂e", "́,", "B", "-,", "fi", "'ll", "'VE", "'BEOT", "
", "123", "456", "78", "\n", "/HTTPServerEOT", "👍🏽", "ع", " \n", " ", "Ⅳ", "camelCase", "🙂字", "🙂ßEOT", "\u000beiOS", "…", "🙂\"", "٣٤٥", "٦"]} +{"text": "EOT'reİ😀🏽 ㋿aB
HTTPServer <३'D", "tokens": 24, "pieces": ["EOT", "'re", "İ", "😀🏽", " ㋿", "aB", "
HTTPServer", " ", "<", "३", "'D"]} +{"text": "Ⅳ\r\u000b漢
<|endoftext|> \n 'll½­å Džungla åⅣᵃiOSḍ̇́ſ!‍aB/\r\nAbع ́aBe\u000b'ſ㋿
", "tokens": 71, "pieces": ["Ⅳ", "\r", "\u000b漢", "
", "<|", "endoftext", "|><", "META", "_START", ">", " \n", " '", "ll", "½", "­a", "̊", " ", " Džungla", " a", "̊", "Ⅳ", "ᵃiOSḋ", "̣́", "ſ", "!‍", "aB", "/\r\n", "Abع", " ́", "aBe", "\u000b", "'ſ", "㋿", "
"]} +{"text": "\nZ!!é#$%٣٤٥٦ ㋿\t​­/
ABC㋿", "tokens": 28, "pieces": ["\n", "Z", "!!", "é", "#$%", "٣٤٥", "٦", " ", "㋿", "\t", "​­/", "
ABC", "㋿"]} +{"text": " \n Zm\r\n'sé漢!!#$%a/bEOTꟲ<|endoftext|>'sß👍🏽😀🏽\n/dcamelCase'T'D9s>\niOS\n/ꟲ'VE's/\rᵃDž…iOS", "tokens": 70, "pieces": [" \n", " Zm", "\r\n", "'s", "e", "́漢", "!!#$%<", "META", "_START", ">a", "/bEOTꟲ", "<|", "endoftext", "|>'", "sß", "👍🏽😀🏽\n", "/dcamelCase", "'T", "'D", "9", "s", ">\n", "iOS", "\n", "/ꟲ", "'VE", "'s", "/\r", "ᵃDž", "…iOS"]} +{"text": "\rsAعsEOT
 \tſ \n \n  \n a/b\t'VEe>'VEDžungla.e'ReZ…'<|fim_prefix|>0BⅣ\"\ta", "tokens": 51, "pieces": ["\r", "sAعsEOT", "
 ", "\t", "ſ", " \n \n  \n", " a", "/b", "\t", "'VE", "e", ">'", "VEDžungla", ".e", "'Re", "Z", "…", "'<|", "fim", "_prefix", "|>", "0", "B", "Ⅳ", "\"", "\ta"]} +{"text": "Ⅳ", "tokens": 6, "pieces": ["Ⅳ", ""]} +{"text": "ſiOSéEOT  \nsß\r\n\r\n字½!!㋿(é😀🏽
>", "tokens": 27, "pieces": ["ſiOSéEOT", "  \n", "sß", "\r\n\r\n", "字", "½", "!!㋿(", "é", "😀🏽", "
", ">"]} +{"text": "​½ꟲ㋿.dⅣ'M३1234567812345678iOS‍…'Refi'llİ", "tokens": 31, "pieces": ["​", "½", "ꟲ", "㋿.", "d", "Ⅳ", "'M", "३12", "345", "678", "123", "456", "78", "iOS", "‍", "…", "'Re", "fi", "'ll", "İ"]} +{"text": "㋿é
'Mſ#$%s#$%
👍🏽 😀🏽३İ‍'VEع-HTTPServer字­", "tokens": 40, "pieces": ["㋿é", "
", "'M", "ſ", "#$%", "s", "#$%", "
", "👍🏽", " ", " 😀🏽", "३", "İ", "‍'", "VEع", "-HTTPServer字", "­"]} +{"text": "३fi9­HTTPServer", "tokens": 8, "pieces": ["३", "fi", "9", "­HTTPServer"]} +{"text": " Abꟲ\"ABC", "tokens": 7, "pieces": [" Abꟲ", "\"ABC"]} +{"text": "字́d/Ⅳḍ̇'a/bZ,ᵃ
'St​'ll‍३s𐞁İ", "tokens": 36, "pieces": ["字", "́d", "/", "Ⅳ", "ḋ", "̣'", "a", "/bZ", ",ᵃ", "
", "'S", "t", "​'", "ll", "‍", "३", "s𐞁İ"]} +{"text": "漢ꟲéİſfiZ\u000b㍿.\rDž३iOS>\n
EOT㍿ \n'T", "tokens": 34, "pieces": ["漢ꟲe", "́İſfiZ", "\u000b", "㍿.\r", "Dž", "३", "iOS", ">\n", "
EOT", "㍿", " \n", "'T"]} +{"text": ">😀🏽's🙂>'re\r\ń/\r\n<|endoftext|>'DB\r\n9…'sA\n/(eß'D/a…- \n m\r\nem
ABC😀🏽'", "s", "🙂>'", "re", "\r\n", "́/\r\n", "<|", "endoftext", "|>'", "DB", "\r\n", "9", "…", "'s", "A", "\n", "/(", "eß", "'D", "/", "a", "…", "-", " \n", " ", " m", "\r\n", "em", "
ABC", "12345678'T\n'VE𐞁𐞁\u000b<ß<|fim_prefix|>diOSaBABC…Ⅳa\r\n​", "tokens": 51, "pieces": [" ", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'T", "\n", "'VE", "𐞁𐞁", "\u000b", "<ß", "<|", "fim", "_prefix", "|>", "d", "iOSaBABC", "…", "Ⅳ", "a", "\r\n", "​"]} +{"text": "å😀🏽عcamelCase /\r\n ḍ̇ \n Ab漢", "tokens": 28, "pieces": ["a", "̊😀🏽", "ع", "camelCase", " ", " /\r\n", " ", " ḋ", "̣", " \n", " Ab漢"]} +{"text": "ḍ̇\u000b­Z/Džungla٣٤٥٦!\n … >EOT<|fim_prefix|>
'ReAbéعABC\n/'VE½A\u000be(👍🏽", "tokens": 57, "pieces": ["ḋ", "̣", "\u000b", "­Z", "/Džungla", "٣٤٥", "٦", "!\n", " …", " ", ">EOT", "<|", "fim", "_prefix", "|>", "
", "'Re", "AbéعABC", "\n", "/'", "VE", "½", "A", "\u000be", "(👍🏽"]} +{"text": "'S>m!B>#$%!!0ſAbABCİ👍🏽é'D ​ B \n\u000bDžungla\n/ aBB \n/ \r'M", "tokens": 41, "pieces": ["'S", ">m", "!B", ">#$%!!", "0", "ſAbABCİ", "👍🏽", "é", "'D", " ", "​", " B", " \n", "\u000bDžungla", "\n", "/", " ", " aBB", " \n", "/", " \r", "'M"]} +{"text": "/😀🏽-İ-åع'ſ\r\n dcamelCase😀🏽's㋿(㍿㍿'et…a<|endoftext|>/\r\n!", "tokens": 56, "pieces": ["/😀🏽-", "İ", "-a", "̊ع", "'ſ", "\r\n", " dcamelCase", "😀🏽'", "s", "㋿(㍿㍿'<", "META", "_START", ">et", "…a", "<|", "endoftext", "|>/\r\n", "!<", "META", "_START", ">"]} +{"text": "Zع'fi\r\n\r\n\r\nABCe0\rZ३ficamelCase'DDžungla'MsEOTd\r0ſꟲ'ReDžungla9…Ab½DžunglaAbDžungla½", "tokens": 54, "pieces": ["Zع", "'fi", "\r\n\r\n\r\n", "ABCe", "0", "\r", "Z", "३", "ficamelCase", "'D", "Džungla", "'M", "sEOTd", "\r", "0", "ſꟲ", "'Re", "Džungla", "9", "…Ab", "½", "DžunglaAbDžungla", "½"]} +{"text": "ABCaBß 'VEe'sDžunglaᵃ", "tokens": 16, "pieces": ["ABCaBß", " ", "'VE", "e", "'s", "Džunglaᵃ"]} +{"text": "Z 🙂é!!漢‍\r\n漢👍🏽B", "tokens": 19, "pieces": ["Z", " ", " 🙂", "é", "!!", "漢", "‍\r\n", "漢", "👍🏽", "B"]} +{"text": "EOTDžungla (iOS'😀🏽 \n /\r\n,\rt\r\n\r\n🙂B'Ⅳ0EOTa/b\r\n\r\n!‍", "tokens": 41, "pieces": ["EOTDžungla", " ", " (", "iOS", "'😀🏽", " \n", " /\r\n", ",\r", "t", "\r\n\r\n", "🙂B", "'", "Ⅳ0", "EOTa", "/b", "\r\n\r\n", "!‍<", "EOT", ">"]} +{"text": "\u000bEOT.s
ſ👍🏽👍🏽İع. \n >🙂㍿,s'T𐞁iOSḍ̇‍BDžungla😀🏽Džungla>İ🙂DžunglacamelCase'D'ſ!d㋿", "tokens": 83, "pieces": ["\u000bEOT", ".s", "
ſ", "👍🏽👍🏽", "İع", ".", " \n", " >🙂㍿,", "s", "'", "T𐞁iOSḋ", "̣‍", "BDžungla", "😀🏽", "Džungla", ">İ", "🙂DžunglacamelCase", "'D", "'", "ſ", "!d", "㋿"]} +{"text": "''S9ABC㋿,\n/9\r'Ś(ABCⅣ́", "tokens": 21, "pieces": ["''", "S", "", "9", "ABC", "㋿,\n", "/", "9", "\r", "'S", "́(", "ABC", "Ⅳ", "́"]} +{"text": "漢㋿​'VE​ßḍ̇<㋿iOS…İ㍿ fiA  ́㋿ \n Ab'VE", "tokens": 41, "pieces": ["漢", "㋿​'", "VE", "​ßḋ", "̣<㋿", "iOS", "…İ", "㍿", " fiA", " ", " ", "́㋿", " \n", " Ab", "'VE"]} +{"text": "d字\r\n\r\n𐞁camelCaseHTTPServer'ſ́9aåå<|fim_prefix|>< .éḍ̇dⅣßfi", "tokens": 43, "pieces": ["d字", "\r\n\r\n", "𐞁camelCaseHTTPServer", "'ſ", "́", "9", "aa", "̊a", "̊<|", "fim", "_prefix", "|><", " ", " .", "éḋ", "̣d", "Ⅳ", "ßfi"]} +{"text": "'VEⅣåst \n fi\"( \nİ٣٤٥٦0camelCase٣٤٥٦'D'TA/\r\nBdt字३", "tokens": 43, "pieces": ["'VE", "Ⅳ", "a", "̊st", " \n", " fi", "\"(", " \n", "İ", "٣٤٥", "٦0", "camelCase", "٣٤٥", "٦", "'D", "'T", "A", "/\r\n", "Bdt字", "३"]} +{"text": "字\rcamelCase\r\n\r\néåt#$%fi\n/,!HTTPServer#$% 'ss", "tokens": 28, "pieces": ["字", "\r", "camelCase", "\r\n\r\n", "e", "́a", "̊t", "#$%", "fi", "\n", "/<", "EOT", ">,!", "HTTPServer", "#$%", " ", " '", "ss"]} +{"text": "\t𐞁'll….\rⅣ9\u000bᵃ٣٤٥٦𐞁iOS\n\t…​'M ३\t'S👍🏽 d'", "tokens": 51, "pieces": ["\t𐞁", "'ll", "…", ".\r", "Ⅳ9", "\u000bᵃ", "٣٤٥", "٦", "𐞁iOS", "\n", "\t", "…", "​'", "M", " ", "३", "\t", "'S", "👍🏽", " d", "'"]} +{"text": "fi漢👍🏽", "tokens": 10, "pieces": ["fi漢", "👍🏽"]} +{"text": "字<|endoftext|>\r\n>𐞁\t\r\nß AbZ \n 'Sd\r­/\r\n​ \n ", "tokens": 30, "pieces": ["字", "<|", "endoftext", "|>\r\n", ">𐞁", "\t\r\n", "ß", " AbZ", "", " \n", " '", "Sd", "\r", "­/\r\n", "​", " \n "]} +{"text": "​Džungla", "tokens": 5, "pieces": ["​Džungla"]} +{"text": "\r\n  éfi㍿Z३Dž", "tokens": 14, "pieces": ["\r\n", " ", " éfi", "㍿Z", "३", "Dž"]} +{"text": "9\u000b'VE'll", "tokens": 5, "pieces": ["9", "\u000b", "'VE", "'ll"]} +{"text": " 🙂!​\n'Mm\n/Bé'D>s\r\n字'M㋿\nEOT's/\r\n fi('Ta/bmd", "tokens": 31, "pieces": [" 🙂!​\n", "'M", "m", "\n", "/Be", "́'", "D", ">s", "\r\n", "字", "'M", "㋿\n", "EOT", "'s", "/\r\n", " fi", "('", "Ta", "/bmd"]} +{"text": "/\r\n‍å>B's字'VE'𐞁iOS-ꟲå३Džunglae‍ḍ̇​fi>.", "tokens": 47, "pieces": ["/\r\n", "‍a", "̊>", "B", "'s", "字", "'VE", "'𐞁iOS", "-ꟲa", "̊", "३", "Džunglae", "‍ḋ", "̣​", "fi", ">."]} +{"text": " \nABCiOS३İ\nABCDžs", "tokens": 11, "pieces": [" \n", "ABCiOS", "३", "İ", "\n", "ABCDžs"]} +{"text": "e字,\r\n\r\ncamelCase'llcamelCaseåHTTPServereعtEOT'ReABCé>", "tokens": 23, "pieces": ["e字", ",\r\n\r\n", "camelCase", "'ll", "camelCasea", "̊HTTPServereعtEOT", "'Re", "ABCe", "́>"]} +{"text": "009<|endoftext|>a'll\" 'reḍ̇Džungla /\r\n émİꟲ'M", "tokens": 37, "pieces": ["009", "<|", "endoftext", "|>", "a", "'ll", "\"<", "META", "_START", ">", " ", " '", "reḋ", "̣Džungla", " ", "/\r\n", " ", " e", "́mİꟲ", "'M"]} +{"text": "\n漢'VE \n ", "tokens": 7, "pieces": ["\n", "漢", "'VE", " \n "]} +{"text": "\n/\t\nEOTꟲ0㋿édéB㋿,", "tokens": 24, "pieces": ["\n", "/<", "META", "_START", ">", "\t\n", "EOTꟲ", "0", "㋿e", "́déB", "㋿,"]} +{"text": "ᵃ漢 \n­\n/aBédéaB \n#$%fiZ\r\n!!HTTPServer'ſ'llcamelCase", "tokens": 30, "pieces": ["ᵃ漢", " \n", "­\n", "/aBédéaB", " \n", "#$%", "fiZ", "\r\n", "!!", "HTTPServer", "'ſ", "'ll", "camelCase"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'llİᵃ9😀🏽s‍'ſⅣ㋿<|endoftext|>ᵃ're🙂𐞁'ſ…\u000beعßB're", "tokens": 50, "pieces": ["'ll", "İᵃ", "9", "😀🏽", "s", "‍'", "ſ", "Ⅳ", "㋿<|", "endoftext", "|>", "ᵃ", "'re", "🙂𐞁", "'ſ", "…", "\u000beعßB", "'re"]} +{"text": "'Re!camelCase字👍🏽(å-!\"٣٤٥٦\u000b/½/​ع<ᵃ \n <|endoftext|>'Se 12345678'll🙂<|fim_prefix|>éDžunglaé'MHTTPServer!!‍!", "tokens": 73, "pieces": ["'Re", "!camelCase字", "👍🏽(", "a", "̊-!\"", "٣٤٥", "٦", "\u000b", "/", "½", "/​", "ع", "<ᵃ", " \n", " <|", "endoftext", "|>'", "Se", " ", "123", "456", "78", "'ll", "🙂<|", "fim", "_prefix", "|>", "e", "́Džunglae", "́'", "MHTTPServer", "!!‍!"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'VE㋿", "tokens": 5, "pieces": ["'VE", "㋿"]} +{"text": "
iOS''Refi👍🏽\nfi👍🏽\r\n\r\n!! \n ٣٤٥٦'ll٣٤٥٦t!!…😀🏽ᵃmİⅣ", "tokens": 59, "pieces": ["
iOS", "''", "Refi", "👍🏽\n", "fi", "👍🏽\r\n\r\n", "!!", " \n", " ", "٣٤٥", "٦", "'ll", "٣٤٥", "٦", "t", "!!", "…", "😀🏽", "ᵃmİ", "Ⅳ"]} +{"text": "…'Me<|fim_prefix|>\r\n\r\n 𐞁camelCase३\u000biOS12345678d('D", "tokens": 28, "pieces": ["…", "'M", "e", "<|", "fim", "_prefix", "|>\r\n\r\n", " 𐞁camelCase", "३", "\u000biOS", "123", "456", "78", "d", "('", "D"]} +{"text": "Džungla\rEOT Džungla-…aB\r\n ", "tokens": 20, "pieces": ["Džungla", "\r", "EOT", " ", " Džungla", "-", "…aB", "\r\n "]} +{"text": "İå'Sm-!fi'S\u000bſBiOS​ \n >camelCase\"912345678\r\n\r\n\tZ", "tokens": 27, "pieces": ["İa", "̊'", "Sm", "-!", "fi", "'S", "\u000bſBiOS", "​", " \n", " >", "camelCase", "\"", "912", "345", "678", "\r\n\r\n", "\tZ"]} +{"text": "‍9‍­/\r\n-sé,…!!d(\r\n \nſ#$%漢é#$%!㋿ᵃ½ABC\r\t\r\nZed\n/𐞁 ​ ", "tokens": 49, "pieces": ["‍", "9", "‍­/\r\n", "-sé", ",", "…", "!!", "d", "(\r\n", " \n", "ſ", "#$%", "漢é", "#$%!㋿", "ᵃ", "½", "ABC", "\r\t\r\n", "Zed", "\n", "/𐞁", "", " ​", " "]} +{"text": "HTTPServer'DßAb漢m'll9Dž!0漢'ſḍ̇\u000b-ſa/bfiB\r\n\r\n‍'TDž (Ⅳ/\r\n", "tokens": 49, "pieces": ["HTTPServer", "'D", "ßAb漢m", "'ll", "9", "Dž", "!", "0", "漢", "'ſ", "ḋ", "̣", "\u000b", "-ſa", "/bfiB", "\r\n\r\n", "‍'", "TDž", " ", " (", "Ⅳ", "/\r\n"]} +{"text": "!å('T ½a/b​0\r\n.DžunglaHTTPServer'ſ㍿fi㋿!字!​漢😀🏽\tᵃZ!Z字 ", "tokens": 51, "pieces": ["!a", "̊('", "T", " ", "½", "a", "/b", "​", "0", "\r\n", ".DžunglaHTTPServer", "'ſ", "㍿fi", "㋿!", "字", "!​", "漢", "😀🏽", "\tᵃZ", "!Z字", " "]} +{"text": "ع EOT!!0ſ'VEA", "tokens": 11, "pieces": ["ع", " EOT", "!!", "0", "ſ", "'VE", "A"]} +{"text": "\r<|fim_prefix|>'s \n DžåAb\t­\n३B'sAbs\n😀🏽…", "tokens": 33, "pieces": ["\r", "<|", "fim", "_prefix", "|>'", "s", " \n", " Dža", "̊Ab", "\t", "­\n", "३", "B", "'s", "Abs", "\n", "😀🏽", "…"]} +{"text": "m👍🏽e\r\n\r\n9‍a/bⅣ<🙂́(ع(<|fim_prefix|>­- \n \"m<|endoftext|>>㋿\t
­-", " \n", " \"", "m", "<|", "endoftext", "|>>㋿", "\t", "
", "㍿ⅣHTTPServer \nABC", "tokens": 29, "pieces": [" \n", "!!", "İ", "\r\n\r\n", "٣٤٥", "٦", " ", "<|", "fim", "_prefix", "|>㍿", "Ⅳ", "HTTPServer", " \n", "ABC"]} +{"text": "'ll,🙂camelCase…­😀🏽-m'D'D ⅣB'Sᵃ😀🏽'HTTPServer", "tokens": 34, "pieces": ["'ll", ",🙂", "camelCase", "…", "­😀🏽-", "m", "'D", "'D", " ", "Ⅳ", "B", "'S", "ᵃ", "😀🏽'", "HTTPServer"]} +{"text": "é.Ⅳé
Ⅳ> \n12345678a/b👍🏽Dž\r\n\r\n\r\nBs𐞁<'T>…9,9字 'reᵃᵃ!>­ſ\"Džungla漢", "tokens": 64, "pieces": ["e", "́.", "Ⅳ", "e", "́", "
", "Ⅳ", ">", " \n", "123", "456", "78", "a", "/b", "👍🏽", "Dž", "\r\n\r\n\r\n", "Bs𐞁", "<'", "T", ">", "…", "9", ",", "9", "字", " '", "reᵃᵃ", "!>­", "ſ", "\"Džungla漢"]} +{"text": "aEOT'ſ'ſ s \n iOS \n ᵃ'refiABC३\n'D", "tokens": 24, "pieces": ["aEOT", "'ſ", "'ſ", " s", " \n", " iOS", " \n", " ᵃ", "'re", "fiABC", "३", "\n", "'D"]} +{"text": "tiOS'SA\r\n漢9‍ \t12345678'M٣٤٥٦>𐞁३٣٤٥٦\n/!!d漢'VEHTTPServer'reAABC", "tokens": 62, "pieces": ["tiOS", "'S", "A", "\r\n", "漢", "9", "‍", " ", "\t", "123", "456", "78", "'M", "٣٤٥", "٦", ">𐞁", "३٣٤", "٥٦", "\n", "/!!", "d漢", "'VE", "HTTPServer", "'re", "AABC"]} +{"text": "'S'T‍12345678字ABC", "tokens": 9, "pieces": ["'S", "'T", "‍", "123", "456", "78", "字ABC"]} +{"text": "㍿​'DB!!<Dž😀🏽'D漢\n//tAb  ᵃ'ReHTTPServerfi\r\n字saBa字iOS'll#$%Z½㋿", "tokens": 50, "pieces": ["㍿​'", "DB", "!!<", "Dž", "😀🏽'", "D漢", "\n", "/<", "META", "_START", ">/", "tAb", " ", " ᵃ", "'Re", "HTTPServerfi", "\r\n", "字saBa字iOS", "'ll", "#$%", "Z", "½", "㋿"]} +{"text": "s,\niOS!#$%A\r…😀🏽,a/bZ's­mZḍ̇३'VE\r\n/'ll\u000b.HTTPServereعİm㋿'S'M'Re", "tokens": 50, "pieces": ["s", ",\n", "iOS", "!#$%", "A", "\r", "…", "😀🏽,", "a", "/bZ", "'s", "­mZḋ", "̣", "३", "'VE", "\r\n", "/'", "ll", "\u000b", ".HTTPServereعİm", "㋿'", "S", "'M", "'Re"]} +{"text": "ßt 👍🏽́ع'VE's٣٤٥٦!!12345678<|endoftext|>\r/\r\n½ ḍ̇", "tokens": 40, "pieces": ["ßt", " 👍🏽́", "ع", "'VE", "'s", "٣٤٥", "٦", "!!", "123", "456", "78", "<|", "endoftext", "|>\r", "/\r\n", "½", " ḋ", "̣"]} +{"text": "\n/12345678camelCase/٣٤٥٦EOT'ſ's\n/<\ra/biOSİ!!😀🏽a\r\n\r\n\r\n!!åééa㍿a/b", "tokens": 51, "pieces": ["\n", "/", "123", "456", "78", "camelCase", "/", "٣٤٥", "٦", "EOT", "'ſ", "'s", "\n", "/<\r", "a", "/biOSİ", "!!😀🏽", "a", "\r\n\r\n\r\n", "!!", "a", "̊e", "́e", "́a", "㍿a", "/b"]} +{"text": "👍🏽½Ⅳ
'M \n字 'T ,½́/é'ſ'", "tokens": 27, "pieces": ["👍🏽", "½Ⅳ", "
", "'M", " \n", "字", " ", "'T", " ", " ,", "½", "́/", "e", "́'", "ſ", "'"]} +{"text": "\rZ'sEOTHTTPServerfi0½AbHTTPServera/b<|endoftext|>…'Re\n<|fim_prefix|>-", "tokens": 38, "pieces": ["\r", "Z", "'s", "EOTHTTPServer", "fi", "0½", "AbHTTPServera", "/b", "<|", "endoftext", "|>", "…", "'Re", "\n", "<|", "fim", "_prefix", "|>-"]} +{"text": "s'D'rét漢é<|endoftext|>…HTTPServerHTTPServerABC9!!'re½'D😀🏽,ß漢Z", "tokens": 39, "pieces": ["s", "'D", "'re", "́t漢e", "́<|", "endoftext", "|>", "…HTTPServerHTTPServerABC", "9", "!!'", "re", "½", "'D", "😀🏽,", "ß漢Z"]} +{"text": "😀🏽'VEt३ \t9'Reé́!,𐞁́<|endoftext|><|fim_prefix|>㋿9>­́'ReZ're/\r\n 👍🏽'ſ
é'T㍿\r\n\r\n", "tokens": 71, "pieces": ["😀🏽'", "VEt", "३", " ", "\t", "9", "'Re", "e", "́́!,", "𐞁", "́<|", "endoftext", "|><|", "fim", "_prefix", "|>㋿", "9", ">­́'", "ReZ", "'re", "/\r\n", " ", "👍🏽'", "ſ", "
e", "́'", "T", "㍿\r\n\r\n"]} +{"text": "
<|endoftext|>", "tokens": 9, "pieces": ["
", "<|", "endoftext", "|>"]} +{"text": "s.👍🏽é'M漢ꟲ'ß \n #$%0\r\n‍(", "tokens": 29, "pieces": ["s", ".👍🏽", "é", "'M", "漢ꟲ", "'ß", " \n", " #$%", "0", "\r\n", "‍("]} +{"text": "å漢EOT👍🏽🙂12345678\n/#$% \n'S Džungla'D🙂Ⅳ12345678fi🙂EOT'", "tokens": 44, "pieces": ["a", "̊漢EOT", "👍🏽🙂", "123", "456", "78", "\n", "/#$%", " \n", "'S", " Džungla", "'D", "🙂", "Ⅳ12", "345", "678", "fi", "🙂EOT", "'"]} +{"text": "३ſ'VEع\"'VE𐞁­'s٣٤٥٦'SZ/!!'ſꟲ<'M 
/\r\nHTTPServerfi", "tokens": 44, "pieces": ["३", "ſ", "'VE", "ع", "\"'", "VE𐞁", "­'", "s", "٣٤٥", "٦", "'S", "Z", "/!!'", "ſꟲ", "<'", "M", " ", "
", "/\r\n", "HTTPServerfi"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\u000b're'SA'Re𐞁9\r\n\n/,m\r\n\r\ncamelCase𐞁३é\r\n\r\nع­​…\t'SA
aBſ…ḍ̇Z/㍿Z‍a/b", "tokens": 61, "pieces": ["\u000b", "'re", "'S", "A", "'Re", "𐞁", "9", "\r\n\n", "/,", "m", "\r\n\r\n", "camelCase𐞁", "३", "e", "́\r\n\r\n", "ع", "­​", "…", "\t", "'S", "A", "
", "aBſ", "…ḋ", "̣Z", "/㍿", "Z", "‍a", "/b"]} +{"text": "-‍.<|endoftext|>́Ⅳꟲa/b字'S#$%m'T
½", "tokens": 27, "pieces": ["-‍.<|", "endoftext", "|>́", "Ⅳ", "ꟲa", "/b字", "'S", "#$%", "m", "'T", "
", "½"]} +{"text": "'T,,e0/ \n' B'S'SDžunglaABCZ\u000b.\r\n9\ts\r\n\r\n'M㋿㋿12345678mſ!!
aB́>-
/\r\n👍🏽", "tokens": 54, "pieces": ["'T", ",,", "e", "0", "/", " \n", "'<", "EOT", ">", " ", " B", "'S", "'S", "DžunglaABCZ", "\u000b", ".\r\n", "9", "\ts", "\r\n\r\n", "'M", "㋿㋿", "123", "456", "78", "mſ", "!!", "
aB", "́>-", "
", "/\r\n", "👍🏽"]} +{"text": "ſe­'VE½'re\"́㍿'.0>d,٣٤٥٦Ab'\r\n ZZs're漢Ⅳ­A \n\u000bZ0aB​/\r\nᵃEOT", "tokens": 51, "pieces": ["ſe", "­'", "VE", "½", "'re", "\"́㍿'.", "0", ">d", ",", "٣٤٥", "٦", "Ab", "'\r\n", " ", " ZZs", "'re", "漢", "Ⅳ", "­A", " \n", "\u000bZ", "0", "aB", "​/\r\n", "ᵃEOT"]} +{"text": "…,😀🏽\n//\r\nHTTPServer!㋿Z'ſ-", "tokens": 21, "pieces": ["…", ",😀🏽\n", "//\r\n", "HTTPServer", "!㋿", "Z", "'ſ", "-"]} +{"text": "Dž<|endoftext|>…HTTPServer-ꟲ'VE 字 ßaBDžع9\u000bEOT9'.ꟲ\r\nssaB  Dž", "tokens": 44, "pieces": ["Dž", "<|", "endoftext", "|>", "…HTTPServer", "-ꟲ", "'VE", " 字", " ßaBDžع", "9", "\u000bEOT", "9", "'.", "ꟲ", "\r\n", "ssaB", " ", " Dž"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "/\r\nİe‍#$%'ScamelCase'llEOT/'Dé
字字fi\tA\"'MABC/ع'reꟲ­ꟲ­aEOT𐞁字Džꟲßſ", "tokens": 58, "pieces": ["/\r\n", "İe", "‍#$%'", "ScamelCase", "'ll", "EOT", "/'", "De", "́", "
", "字字fi", "\tA", "\"'", "MABC", "/ع", "'re", "ꟲ", "­ꟲ", "­aEOT𐞁字Džꟲßſ"]} +{"text": "'VEᵃİ're.'s, !'ſ\r\r\n\r\n \n 9'T-\r\nm​camelCase/\r\n \n \r\n㋿٣٤٥٦字'ſ'VE>'s \n", "tokens": 52, "pieces": ["'VE", "ᵃİ", "'re", ".'", "s", ",", " ", "!'", "ſ", "\r\r\n\r\n \n", " ", "9", "'T", "-\r\n", "m", "​camelCase", "/\r\n", " \n \r\n", "㋿", "٣٤٥", "٦", "字", "'ſ", "'VE", ">'", "s", " \n", ""]} +{"text": "'M\rå<|fim_prefix|> \n9ḍ̇e/ᵃᵃcamelCase\t‍'D<|endoftext|>漢(", "tokens": 51, "pieces": ["'M", "\r", "a", "̊<", "EOT", "><|", "fim", "_prefix", "|>", " \n", "9", "ḋ", "̣e", "/ᵃᵃcamelCase", "\t", "‍'", "D", "<|", "endoftext", "|>", "漢", "("]} +{"text": "d-Ab \nḍ̇­aB漢' …㋿>", "tokens": 22, "pieces": ["d", "-Ab", " \n", "ḋ", "̣­", "aB漢", "'", " ", "…", "㋿>"]} +{"text": "𐞁camelCase<|fim_prefix|>é'll ", "tokens": 16, "pieces": ["𐞁camelCase", "<|", "fim", "_prefix", "|>", "é", "'ll", " "]} +{"text": "<👍🏽camelCasecamelCase", "tokens": 11, "pieces": ["<👍🏽", "camelCasecamelCase"]} +{"text": "EOT-.́½ꟲ!!漢ABCd/\r\n!Džungla'lléABC'Reſſ'Mfim\"'Z'saBEOT'll‍Z", "tokens": 44, "pieces": ["EOT", "-.́", "½", "ꟲ", "!!", "漢ABCd", "/\r\n", "!Džungla", "'ll", "éABC", "'Re", "ſſ", "'M", "fi", "m", "\"'", "Z", "'s", "aBEOT", "'ll", "‍Z"]} +{"text": "å 'lla/b\n/\r\n㍿\r'  \n Dž<|fim_prefix|>BAb字😀🏽Džungla\n/d-'Då\r'sᵃcamelCaseDž,\n'MfiZ\rA'ſå", "tokens": 70, "pieces": ["a", "̊", " '", "lla", "/b", "\n", "/\r\n", "㍿<", "META", "_START", ">\r", "'", "  \n", " Dž", "<|", "fim", "_prefix", "|>", "BAb字", "😀🏽", "Džungla", "\n", "/d", "-'", "Da", "̊\r", "'s", "ᵃcamelCaseDž", ",\n", "'M", "fiZ", "\r", "A", "'ſ", "a", "̊"]} +{"text": "👍🏽's", "tokens": 8, "pieces": ["👍🏽'", "s"]} +{"text": ".'Re😀🏽Dž", "tokens": 9, "pieces": [".'", "Re", "😀🏽", "Dž"]} +{"text": "camelCaseᵃé🙂a'll mᵃABC<|fim_prefix|>", "tokens": 22, "pieces": ["camelCaseᵃé", "🙂a", "'ll", " mᵃABC", "<|", "fim", "_prefix", "|>"]} +{"text": "عع\r\naBs\r\n\r\n'Re", "tokens": 10, "pieces": ["عع", "\r\n", "aBs", "\r\n\r\n", "'Re"]} +{"text": "
漢a ß\r\n\r\n9㋿Ⅳ'Re'sABC‍\tABC<'Re㍿㋿漢,'re\"", "tokens": 34, "pieces": ["
漢a", " ß", "\r\n\r\n", "9", "㋿", "Ⅳ", "'Re", "'s", "ABC", "‍", "\tABC", "<'", "Re", "㍿㋿", "漢", ",'", "re", "\""]} +{"text": "'ſ३fié- ́!B'T𐞁\n/!<👍🏽0'T0DžcamelCase\"!'​ \n'S .\r<|fim_prefix|>åmABC😀🏽 ", "tokens": 67, "pieces": ["'ſ", "३", "fie", "́-", " ", "́!<", "META", "_START", ">B", "'T", "𐞁", "\n", "/!<👍🏽", "0", "'T", "0", "DžcamelCase", "\"!'​", " \n", "'S", " ", ".\r", "<|", "fim", "_prefix", "|>", "a", "̊mABC", "😀🏽", " "]} +{"text": "B㍿İ/s𐞁
", "tokens": 12, "pieces": ["B", "㍿İ", "/s𐞁", "
"]} +{"text": "…<|fim_prefix|>HTTPServer'Re३' \n 'sABC<|fim_prefix|>s \n ',", "tokens": 32, "pieces": ["…", "<|", "fim", "_prefix", "|>", "HTTPServer", "'Re", "३", "'", " \n", " '", "sABC", "<|", "fim", "_prefix", "|>", "s", " \n", " '<", "META", "_START", ">,"]} +{"text": "!!é\u000ba/bß'llfi字漢(Abé\u000b\r…‍d٣٤٥٦ \r'S.é'llEOT.AbcamelCase-\u000b'Re", "tokens": 47, "pieces": ["!!", "e", "́", "\u000ba", "/bß", "'ll", "fi字漢", "(Abe", "́", "\u000b\r", "…", "‍d", "٣٤٥", "٦", " \r", "'S", ".e", "́'", "llEOT", ".AbcamelCase", "-", "\u000b", "'Re"]} +{"text": "㋿.Dž
 /\r\n\t३/\r\niOS‍𐞁㍿́\r\n\r\n٣٤٥٦'0😀🏽<|fim_prefix|>㋿ß's㍿'ll ", "tokens": 66, "pieces": ["㋿.", "Dž", "
", " ", "/\r\n", "\t", "३", "/\r\n", "iOS", "‍𐞁", "㍿́\r\n\r\n", "٣٤٥", "٦", "'", "0", "😀🏽<|", "fim", "_prefix", "|>㋿", "ß", "'", "s", "㍿'", "ll", " "]} +{"text": ">,iOSHTTPServer\"(­ᵃt😀🏽", "tokens": 15, "pieces": [">,", "iOSHTTPServer", "\"(­", "ᵃt", "😀🏽"]} +{"text": "'ll", "tokens": 1, "pieces": ["'ll"]} +{"text": " \n…'MaBt'T'M字'lld\tEOT!! ‍Ⅳ\r\n\r\nEOT😀🏽", "tokens": 32, "pieces": [" \n", "…", "'M", "aBt", "'T", "'M", "字", "'ll", "d", "\tEOT", "!!", " ", " ‍", "Ⅳ", "\r\n\r\n", "EOT", "😀🏽<", "META", "_START", ">"]} +{"text": "iOS½,a/b',EOT\"㋿
'sEOT", "tokens": 16, "pieces": ["iOS", "½", ",a", "/b", "',", "EOT", "\"㋿", "
", "'s", "EOT"]} +{"text": "ABC<|fim_prefix|>#$% <|endoftext|>İéſAb'VEZ㍿0㋿㍿t9'VE😀🏽a/b🙂Džungla0'ReDžungla漢#$%'T🙂😀🏽\t", "tokens": 74, "pieces": ["ABC", "<|", "fim", "_prefix", "|>#$%", " ", " <|", "endoftext", "|>", "İéſAb", "'VE", "Z", "㍿", "0", "㋿㍿<", "META", "_START", ">t", "9", "'VE", "😀🏽", "a", "/b", "🙂Džungla", "0", "'Re", "Džungla漢", "#$%'", "T", "🙂😀🏽", "\t"]} +{"text": "㋿<|endoftext|>0ḿs\"-<|endoftext|>mᵃ #$%\"\taB're㍿s/'Re漢👍🏽Ⅳ'Re३. Źᵃ", "tokens": 57, "pieces": ["㋿<|", "endoftext", "|>", "0", "m", "́s", "\"-<|", "endoftext", "|>", "mᵃ", " ", "#$%\"", "\taB", "'re", "㍿s", "/'", "Re漢", "👍🏽", "Ⅳ", "'Re", "३", ".", " Z", "́ᵃ"]} +{"text": "ḍ̇\t\n/'re!!  å😀🏽åmt!\né<|fim_prefix|>ع-\"''T\t'T́/\r\n!/#$%😀🏽-㋿㋿ d", "tokens": 62, "pieces": ["ḋ", "̣", "\t\n", "/'", "re", "!!", " ", " a", "̊😀🏽", "a", "̊mt", "!\n", "e", "́<|", "fim", "_prefix", "|>", "ع", "-\"''", "T", "\t", "'T", "́/\r\n", "!/#$%😀🏽-㋿㋿", " d", ""]} +{"text": " ḍ̇…fiBe\n/\r\nAbⅣ\n/'ll", "tokens": 18, "pieces": [" ḋ", "̣", "…fiBe", "\n", "/\r\n", "Ab", "Ⅳ", "\n", "/'", "ll"]} +{"text": "HTTPServer\n/\n-éaB #$%9½.>'T \n\r\n\r\n", "tokens": 20, "pieces": ["HTTPServer", "\n", "/\n", "-e", "́aB", " #$%", "9½", ".>'", "T", " \n", "\r\n\r\n"]} +{"text": "ع-३٣٤٥٦\r\n\r\n012345678t३Ⅳ,漢ſiOS𐞁'Tİå'S12345678EOT'D𐞁ma/béⅣ'D😀🏽İ0漢/\r\n a/b
å", "tokens": 72, "pieces": ["ع", "-", "३٣٤", "٥٦", "\r\n\r\n", "012", "345", "678", "t", "३Ⅳ", ",漢ſiOS𐞁", "'T", "İa", "̊'", "S", "123", "456", "78", "EOT", "'D", "𐞁ma", "/be", "́", "Ⅳ", "'D", "😀🏽", "İ", "0", "漢", "/\r\n", " ", " a", "/b", "
a", "̊"]} +{"text": "Džİ ᵃ‍'MEOT'll!>…'sAba/bع \n0٣٤٥٦­/ \nß'sA ‍.\r,‍", "tokens": 47, "pieces": ["Džİ", " ᵃ", "‍'", "MEOT", "'ll", "!>", "…", "'s", "Aba", "/bع", " \n", "0٣٤", "٥٦", "­/", " \n", "ß", "'s", "A", " ", "‍.\r", ",‍"]} +{"text": " e'M'Stſ'ſ ­Džungla​<|endoftext|>ꟲHTTPServer'D/\r\n​<|fim_prefix|>🙂ꟲ(.>/\r\n'Re'll\té𐞁字 \nABC0're", "tokens": 63, "pieces": [" e", "'M", "'S", "tſ", "'ſ", " ", "­Džungla", "​<|", "endoftext", "|>", "ꟲHTTPServer", "'D", "/\r\n", "​<|", "fim", "_prefix", "|>🙂", "ꟲ", "(.>/\r\n", "'", "Re", "'ll", "\té𐞁字", " \n", "ABC", "0", "'re"]} +{"text": "'ll(
> \n t🙂\n/>\r\n,éꟲ aHTTPServer>éAb'VE…", "tokens": 27, "pieces": ["'ll", "(", "
", ">", " \n", " t", "🙂\n", "/>\r\n", ",e", "́ꟲ", " aHTTPServer", ">éAb", "'VE", "…"]} +{"text": "!!½𐞁<­ᵃ\"ع ", "tokens": 14, "pieces": ["!!", "½", "𐞁", "<­", "ᵃ", "\"ع", " "]} +{"text": "\"(­­‍‍漢#$%'D\r\n\r\n \n 'llfi0eİ!aB\rå", "tokens": 28, "pieces": ["\"(­­‍‍", "漢", "#$%'", "D", "\r\n\r\n \n", " '", "llfi", "0", "eİ", "!aB", "\r", "a", "̊"]} +{"text": "'M'T漢ABĆ㍿a/bHTTPServer½a/bém#$%A㍿ABC>
Ⅳ३'ſ𐞁'>aBABCB㍿ \n tABC sa/bm'reᵃ12345678", "tokens": 64, "pieces": ["'M", "'T", "漢ABC", "́㍿", "a", "/bHTTPServer", "½", "a", "/be", "́m", "#$%", "A", "㍿ABC", ">", "
", "Ⅳ३", "'ſ", "𐞁", "'>", "aBABCB", "㍿", " \n", " tABC", " sa", "/bm", "'re", "ᵃ", "123", "456", "78"]} +{"text": "ع 
é́ꟲ-'M'S\"\t12345678>sİå", "tokens": 22, "pieces": ["ع", " ", "
é", "́ꟲ", "-'", "M", "'S", "\"", "\t", "123", "456", "78", ">sİa", "̊"]} +{"text": " \n /\r\n­'Dß👍🏽t\n/Ⅳ#$% B9​B\"Ⅳ'Re½A<|endoftext|>字́", "tokens": 41, "pieces": [" \n", " /\r\n", "­'", "Dß", "👍🏽", "t", "\n", "/", "Ⅳ", "#$%", " B", "9", "​B", "\"", "Ⅳ", "'Re", "½", "A", "<|", "endoftext", "|>", "字", "́"]} +{"text": "e/\r\n \n ३sᵃsDžunglaDžß \n  'Re!ḍ̇́'ſ'D'sABCABC<|endoftext|>ABC0\r\n\r\n\r\n…!'VE👍🏽𐞁'Re'ſé,", "tokens": 71, "pieces": ["e", "/\r\n", " \n", " ", "३", "sᵃsDžunglaDžß", " \n", " ", " ", "'Re", "!ḋ", "̣́'", "ſ", "'D", "'s", "ABCABC", "<|", "endoftext", "|>", "ABC", "0", "\r\n\r\n\r\n", "…", "!'", "VE", "👍🏽", "𐞁", "'", "Re", "'ſ", "e", "́,"]} +{"text": "\n/B'VE३'a/b\rİ\r­字  Džunglaſ
ꟲé/\r\nHTTPServer>\">", "tokens": 33, "pieces": ["\n", "/B", "'VE", "३", "'a", "/b", "\r", "İ", "\r", "­字", " ", " Džunglaſ", "
ꟲe", "́/\r\n", "HTTPServer", ">\">"]} +{"text": "'MHTTPServer\r\n\r\n㍿téᵃt㍿ع!se0\r\nᵃ/\r\nße'Mé😀🏽‍\t́Dž😀🏽,#$%", "tokens": 50, "pieces": ["'M", "HTTPServer", "\r\n\r\n", "㍿téᵃt", "㍿ع", "!", "se", "0", "\r\n", "ᵃ", "/\r\n", "ße", "'M", "é", "😀🏽‍", "\t", "́Dž", "😀🏽,#$%"]} +{"text": " EOT'll'ſ́ⅣDžungla
 \n(.a/b\n/字s", "tokens": 27, "pieces": [" EOT", "'ll", "'ſ", "́<", "EOT", ">", "Ⅳ", "Džungla", "
 \n", "(.", "a", "/b", "\n", "/字s"]} +{"text": "‍'re\r\n\r\n字'lla-'s<12345678'D'MiOSé­ \n", "tokens": 24, "pieces": ["‍'", "re", "\r\n\r\n", "字", "'ll", "a", "-'", "s", "<", "123", "456", "78", "'D", "'M", "iOSé", "­", " \n"]} +{"text": "A'VE'M/ 👍🏽Džungla12345678'ſ9m \n \n\n/daDžungla'T'S<|fim_prefix|>ſaB<|endoftext|>m's𐞁‍Z🙂é'S'reBfi", "tokens": 81, "pieces": ["A", "'VE", "'", "M", "/", " ", "👍🏽", "Džungla", "123", "456", "78", "'ſ", "", "9", "m", " \n \n\n", "/daDžungla", "'T", "'S", "<|", "fim", "_prefix", "|>", "ſ", "aB", "<|", "endoftext", "|>", "m", "'s", "𐞁", "‍Z", "🙂é", "'S", "'re", "Bfi"]} +{"text": "……''re\r\n٣٤٥٦-B<|fim_prefix|>😀🏽(-­'ll(
iOS<|fim_prefix|>'s👍🏽
\r\ne\r\n\r\n​a/b 𐞁,", "tokens": 64, "pieces": ["…", "…", "''", "re", "\r\n", "٣٤٥", "٦", "-B", "<|", "fim", "_prefix", "|>😀🏽(-­'", "ll", "(", "
iOS", "<|", "fim", "_prefix", "|>'", "s", "👍🏽", "
\r\n", "e", "\r\n\r\n", "​a", "/b", " 𐞁", ","]} +{"text": "­", "tokens": 1, "pieces": ["­"]} +{"text": "ß\r\n🙂m'T9camelCase'Re\u000bAb'M ", "tokens": 14, "pieces": ["ß", "\r\n", "🙂m", "'T", "9", "camelCase", "'Re", "\u000bAb", "'M", " "]} +{"text": "åⅣ<|endoftext|>", "tokens": 12, "pieces": ["a", "̊", "Ⅳ", "<|", "endoftext", "|>"]} +{"text": " \n camelCaseß\rḍ̇.٣٤٥٦ſ३३EOT", "tokens": 27, "pieces": [" \n", " camelCaseß", "\r", "ḋ", "̣.", "٣٤٥", "٦", "ſ", "३३", "EOT"]} +{"text": "٣٤٥٦aB/\r\n½½'VEDž/\r\nm字Am👍🏽9dᵃ\n,\"camelCase'M漢'sß,A'Tع𐞁Z's fiꟲå'D", "tokens": 69, "pieces": ["٣٤٥", "٦", "aB", "/\r\n", "½½", "'VE", "Dž", "/\r\n", "m字Am", "👍🏽", "9", "dᵃ", "\n", ",\"", "camelCase", "'M", "漢", "'s", "ß", ",A", "'T", "ع𐞁Z", "'s", " ", " fiꟲa", "̊'", "D"]} +{"text": "/\r\nfi,0", "tokens": 9, "pieces": ["/\r\n", "fi", ",", "0"]} +{"text": "ᵃ-İe­🙂s'Tå're\r\n\r\nA\r\n\r\n👍🏽ḍ̇>", "tokens": 32, "pieces": ["ᵃ", "-İe", "­🙂", "s", "'T", "a", "̊'", "re", "\r\n\r\n", "A", "\r\n\r\n", "👍🏽", "ḋ", "̣>"]} +{"text": " \n B ٣٤٥٦👍🏽👍🏽,'Dᵃ<|endoftext|><|fim_prefix|>iOS'll/fi'D'VEDžungla \n ꟲsDžZ👍🏽𐞁", "tokens": 72, "pieces": [" \n", " B", " ", "٣٤٥", "٦", "👍🏽👍🏽,'", "Dᵃ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "iOS", "'ll", "/fi", "'D", "'VE", "Džungla", " \n", " ꟲsDžZ", "👍🏽", "𐞁"]} +{"text": "d½\n/iOS .B-\n'́字iOS<|fim_prefix|>😀🏽camelCase‍å३½ꟲ'Sfi a/b,e'reAb/\r\ńDžcamelCase'S", "tokens": 57, "pieces": ["d", "½", "\n", "/iOS", " ", " <", "META", "_START", ">.", "B", "-\n", "'́", "字iOS", "<|", "fim", "_prefix", "|>😀🏽", "camelCase", "‍a", "̊", "३½", "ꟲ", "'S", "fi", " a", "/b", ",e", "'re", "Ab", "/\r\n", "́DžcamelCase", "'S"]} +{"text": "'M''T(\u000bé\r\n dⅣ're‍'a(𐞁㍿३a/b\n(m\r\n\r\n/\r\n\r.‍>/\r\n \n /\r\n'ReDžZ", "tokens": 47, "pieces": ["'M", "''", "T", "(", "\u000bé", "\r\n", " d", "Ⅳ", "'re", "‍'", "a", "(", "𐞁", "㍿", "३", "a", "/b", "\n", "(m", "\r\n\r\n", "/\r\n\r", ".‍>/\r\n", " \n", " /\r\n", "'Re", "DžZ"]} +{"text": "́ \r\n", "tokens": 2, "pieces": ["́", " \r\n"]} +{"text": "#$%\n\n.ع́'Rea٣٤٥٦\r\né-/\r\né ", "tokens": 23, "pieces": ["#$%\n\n", ".ع", "́'", "Rea", "٣٤٥", "٦", "\r\n", "e", "́-/\r\n", "é", " "]} +{"text": "<|fim_prefix|>", "tokens": 7, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "ABCDžunglaaB9camelCase
å>s👍🏽 'TADžungla­\r.'T👍🏽,", "tokens": 46, "pieces": ["ABCDžunglaaB", "9", "camelCase", "
a", "̊>", "s", "👍🏽", " '", "TADžungla", "­\r", ".'", "T", "👍🏽,"]} +{"text": " \t'D#$%Ab/…", "tokens": 9, "pieces": [" ", "\t", "'D", "#$%", "Ab", "/", "…"]} +{"text": "'DHTTPServermḍ̇<Džungla \"åA漢𐞁㋿Džungla>ſEOT٣٤٥٦ⅣA'sAsfiḍ̇sZ\ts0", "tokens": 73, "pieces": ["'D", "HTTPServermḋ", "̣<", "Džungla", " ", " \"", "a", "̊A漢𐞁", "㋿Džungla", ">ſEOT", "٣٤٥", "٦Ⅳ", "A", "'s", "As", "fiḋ", "̣sZ", "\ts", "", "0"]} +{"text": "mİ漢 🙂㍿\n/e'ſ …
camelCaseaBḍ̇Ⅳ", "tokens": 30, "pieces": ["mİ漢", " ", "🙂㍿\n", "/e", "'ſ", " …", "
camelCaseaBḋ", "̣", "Ⅳ"]} +{"text": "𐞁(👍🏽漢Dž… 's>ꟲåfi३'T
d!!>\n/t\tß½ⅣAås\na/b'M३!漢a/b", "tokens": 67, "pieces": ["𐞁", "(👍🏽", "漢Dž", "…", "", " ", "'s", ">ꟲa", "̊<", "META", "_START", ">fi", "३", "'T", "
d", "!!>\n", "/t", "\tß", "½Ⅳ", "Aa", "̊s", "\n", "a", "/b", "'M", "३", "!漢a", "/b"]} +{"text": "12345678 \n ḍ̇'S'", "S", "'ḍ̇\r\n'M0字!!(🙂'reEOTAba/bḍ̇m\n/ ٣٤٥٦\n/-a/b<|fim_prefix|>ß👍🏽åDžungla\r\n\r\n#$%a/b<", "tokens": 74, "pieces": ["<|", "fim", "_prefix", "|>'", "ḋ", "̣\r\n", "'M", "0", "字", "!!(🙂'", "reEOTAba", "/bḋ", "̣m", "\n", "/", " ", "٣٤٥", "٦", "\n", "/-", "a", "/b", "<|", "fim", "_prefix", "|>", "ß", "👍🏽", "a", "̊Džungla", "\r\n\r\n", "#$%", "a", "/b", "<"]} +{"text": "𐞁<|endoftext|>'½\r\n\r\nḍ̇ ABC字!!ᵃ😀🏽12345678\ré!aB\nAb'T🙂\t\"Dž\t字Džع३\nDž<|endoftext|>t🙂camelCase㋿ḍ̇", "tokens": 77, "pieces": ["𐞁", "<|", "endoftext", "|>'", "½", "\r\n\r\n", "ḋ", "̣", " ABC字", "!!", "ᵃ", "😀🏽", "123", "456", "78", "\r", "é", "!aB", "\n", "Ab", "'T", "🙂", "\t", "\"Dž", "\t字Džع", "३", "\n", "Dž", "<|", "endoftext", "|>", "t", "🙂camelCase", "㋿ḋ", "̣"]} +{"text": "'VE'Re٣٤٥٦. \n 'THTTPServerA'll३ m🙂#$%𐞁fiiOS
漢\nEOT㍿éiOSa/b0a'
 \n 
", "tokens": 61, "pieces": ["'VE", "'Re", "٣٤٥", "٦", ".", " \n", " '", "THTTPServerA", "'ll", "३", " ", " m", "🙂#$%", "𐞁fiiOS", "", "
漢", "\n", "EOT", "㍿éiOSa", "/b", "0", "a", "'", "
 \n 
"]} +{"text": "३Z 9/<|endoftext|>", "tokens": 16, "pieces": ["", "३", "Z", " ", "9", "/<|", "endoftext", "|>"]} +{"text": "'T.½AbİABCḍ̇३iOS'ᵃ'M", "tokens": 19, "pieces": ["'T", ".", "½", "AbİABCḋ", "̣", "३", "iOS", "'ᵃ", "'M"]} +{"text": "tⅣ'ReiOSḍ̇'M<|endoftext|>,㍿A#$%0\tZ.'‍½a/b9… /\r\n HTTPServerḍ̇/\r\n're''VEHTTPServer \n漢 Ⅳ", "tokens": 64, "pieces": ["t", "Ⅳ", "'Re", "iOSḋ", "̣'", "M", "<|", "endoftext", "|>,㍿", "A", "#$%", "0", "", "\tZ", ".'‍", "½", "a", "/b", "9", "… ", " /\r\n", " HTTPServerḋ", "̣/\r\n", "'re", "''", "VEHTTPServer", " \n", "漢", " ", "Ⅳ"]} +{"text": "'D\tiOSt'\u000b\reſa 👍🏽 \n .ᵃ'TeAḍ̇a/b!!", "tokens": 37, "pieces": ["'D", "\tiOSt", "'", "\u000b\r", "eſa", " ", " 👍🏽", " \n", " <", "META", "_START", ">.", "ᵃ", "'T", "eAḋ", "̣a", "/b", "!!"]} +{"text": "३'s<ḍ̇ ſ0åß𐞁 🙂camelCase12345678", "tokens": 32, "pieces": ["३", "'", "s", "<ḋ", "̣", " ", " ſ", "0", "a", "̊ß𐞁", " ", "🙂camelCase", "123", "456", "78"]} +{"text": "عEOT<|endoftext|>'Re\r­½a'ReDžungla0ḍ̇aå  'S‍́ Dž🙂", "tokens": 41, "pieces": ["عEOT", "<|", "endoftext", "|>'", "Re", "\r", "­", "½", "a", "'Re", "Džungla", "0", "ḋ", "̣aa", "̊", " ", " ", "'S", "‍́", " ", " Dž", "🙂"]} +{"text": "‍㍿ᵃé\"<|endoftext|><|fim_prefix|>\ttᵃ​'Ta<|endoftext|>'½ \té字ᵃEOTABC<|fim_prefix|>'s", "tokens": 59, "pieces": ["‍㍿", "ᵃe", "́\"<|", "endoftext", "|><|", "fim", "_prefix", "|>", "\ttᵃ", "​'", "Ta", "<|", "endoftext", "|>'", "½", " ", "\té字ᵃEOTABC", "<|", "fim", "_prefix", "|>'", "s"]} +{"text": "'Bd\"ßsa/b .BAb…>é‍es\n/ \nAb9ḍ̇\r\nBd㍿٣٤٥٦EOT/\r\nABC", "tokens": 48, "pieces": ["'Bd", "\"ßsa", "/b", " ", " .", "BAb", "…", ">e", "́‍", "es", "\n", "/", " \n", "Ab", "9", "ḋ", "̣\r\n", "Bd", "㍿", "٣٤٥", "٦", "EOT", "/\r\n", "ABC"]} +{"text": "
fiåꟲ  ꟲ12345678ᵃ…-‍B'Dm>å#$%­/İ ß'ſ'llA.!! /", "tokens": 53, "pieces": ["", "
fia", "̊ꟲ", " ", " ꟲ", "123", "456", "78", "ᵃ", "…", "-‍", "B", "'D", "m", ">a", "̊#$%­/", "İ", " ß", "'ſ", "'ll", "A", ".!!", " ", " /"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "عꟲ​/\r\n s🙂'VE👍🏽٣٤٥٦0Džungla㋿/\r\nAb㋿'TDž/0A३٣٤٥٦漢ß9ꟲ", "tokens": 64, "pieces": ["عꟲ", "​/\r\n", " s", "🙂'", "VE", "👍🏽", "٣٤٥", "٦0", "Džungla", "㋿/\r\n", "Ab", "㋿'", "TDž", "/", "0", "A", "३٣٤", "٥٦", "漢ß", "9", "ꟲ"]} +{"text": "'ſ/!!漢/\r\nm👍🏽sfi9ſ\r\n👍🏽's\t!!३camelCaseſfi\rꟲ😀🏽>ś㍿éAb \n ,\t㋿", "tokens": 63, "pieces": ["'ſ", "/!!", "漢", "/\r\n", "m", "👍🏽", "sfi", "9", "ſ", "\r\n", "👍🏽'", "s", "\t", "!!", "३", "camelCaseſfi", "\r", "ꟲ", "😀🏽>", "s", "́㍿", "éAb", " \n", " ,", "\t", "㋿"]} +{"text": "camelCase'llⅣe >12345678\r\n \n ㍿㍿'VEcamelCase​ABC'res𐞁漢ḍ̇ ", "tokens": 40, "pieces": ["camelCase", "'ll", "Ⅳ", "e", " ", ">", "123", "456", "78", "\r\n \n", " ㍿㍿'", "VEcamelCase", "​ABC", "'re", "s𐞁漢ḋ", "̣", " "]} +{"text": "eDžungla‍<|fim_prefix|><", "ta", "̊e", "́Džß", "३", "́t", "-a", "\t", "…ḋ", "̣𐞁", "9", "/\r\n", "ḋ", "̣👍🏽", "\u000b", "…İ", "
"]} +{"text": "'ll dm/\r\nDž́𐞁<|fim_prefix|>\n/😀🏽́Dž字\n字\r(Z", "tokens": 32, "pieces": ["'ll", " dm", "/\r\n", "Dž", "́𐞁", "<|", "fim", "_prefix", "|>\n", "/😀🏽́", "Dž字", "\n", "字", "\r", "(Z"]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "٣٤٥٦'T​🙂s​!
'DžHTTPServerZḍ̇\u000b'ſ!'M \u000ba/b​👍🏽-​éDžunglaB \n,/ᵃ३", "tokens": 60, "pieces": ["٣٤٥", "٦", "'T", "​🙂", "s", "​!", "
", "'DžHTTPServerZḋ", "̣", "\u000b", "'ſ", "!'", "M", " ", "\u000ba", "/b", "​👍🏽-​", "éDžunglaB", " \n", ",/", "ᵃ", "३"]} +{"text": "🙂é\r(
'T>12345678­ 'll Á\reaBḍ̇ \n 'VEdé(Z<‍Ⅳꟲ/😀🏽😀🏽'VE\u000bDž\r\tå𐞁", "tokens": 68, "pieces": ["🙂é", "\r", "(", "
", "'T", ">", "123", "456", "78", "­", " '", "ll", " ", " A", "́\r", "eaBḋ", "̣", " \n", " '", "VEdé", "(Z", "<‍", "Ⅳ", "ꟲ", "/😀🏽😀🏽'", "VE", "\u000bDž", "\r", "\ta", "̊𐞁"]} +{"text": "\u000b\n/'Mſ'ſ'd", "tokens": 10, "pieces": ["\u000b\n", "/'", "Mſ", "'ſ", "'d"]} +{"text": "<|fim_prefix|>Dž><|fim_prefix|> \n/\r\naiOŚİ\r\n\r'M/\rå'ſ'Tع\n ᵃعABC'Re", "tokens": 42, "pieces": ["<|", "fim", "_prefix", "|>", "Dž", "><|", "fim", "_prefix", "|>", " \n", "/\r\n", "aiOS", "́İ", "\r\n\r", "'M", "/\r", "a", "̊'", "ſ", "'T", "ع", "\n", " ᵃعABC", "'Re"]} +{"text": "HTTPServerB३EOTⅣtm12345678'll!!", "tokens": 18, "pieces": ["HTTPServerB", "", "३", "EOT", "Ⅳ", "tm", "123", "456", "78", "'ll", "!!"]} +{"text": "EOTcamelCase
३å🙂m-\u000b🙂9Ab٣٤٥٦\u000bDžungla㍿!!iOS\n/<|endoftext|>'TaB٣٤٥٦/\r\n\r\n'sa/b…fi<|fim_prefix|>\n‍EOT", "tokens": 75, "pieces": ["EOTcamelCase", "
", "३", "a", "̊🙂", "m", "-", "\u000b", "🙂", "9", "Ab", "٣٤٥", "٦", "\u000bDžungla", "㍿!!", "iOS", "\n", "/<|", "endoftext", "|>'", "TaB", "٣٤٥", "٦", "/\r\n\r\n", "'s", "a", "/b", "…fi", "<|", "fim", "_prefix", "|>\n", "‍EOT"]} +{"text": "DžunglaABC𐞁\r /\r\n𐞁 \n t👍🏽Ab'sdméé9'M!!é…ꟲ\t-/\r\n'éé", "tokens": 44, "pieces": ["DžunglaABC𐞁", "\r", " /\r\n", "𐞁", " \n", " t", "👍🏽", "Ab", "'s", "dméé", "9", "'M", "!!", "é", "…ꟲ", "\t", "-/\r\n", "'ée", "́"]} +{"text": "ß,ᵃs\r\n
B<9'll \n 🙂é9Z
>iOS<|endoftext|>\r漢ß<|endoftext|>👍🏽٣٤٥٦㋿", "tokens": 63, "pieces": ["ß", ",ᵃs", "\r\n", "
B", "<", "9", "'ll", " \n", " <", "EOT", ">🙂", "e", "́", "9", "Z", "
", ">iOS", "<|", "endoftext", "|>\r", "漢ß", "<|", "endoftext", "|>👍🏽", "٣٤٥", "٦", "㋿"]} +{"text": "camelCase'Re,t'rem'ſꟲ🙂\r\nſ 👍🏽fi\r/iOSZ(", "tokens": 38, "pieces": ["camelCase", "'Re", ",t", "'re", "m", "'ſ", "ꟲ", "🙂<", "META", "_START", ">\r\n", "ſ", " ", "👍🏽", "fi", "\r", "/iOSZ", "("]} +{"text": "'re'VE'/\r\n#$%Ⅳ\r\n'DZfiعEOT'Re
t 字😀🏽<|endoftext|>", "tokens": 35, "pieces": ["'re", "'VE", "'/\r\n", "#$%", "Ⅳ", "\r\n", "'D", "ZfiعEOT", "'Re", "
t", " 字", "😀🏽<|", "endoftext", "|>"]} +{"text": "éHTTPServerᵃ👍🏽\r\n\r\n<Ⅳ're\"0\u000b9👍🏽​ع! ­'T\r\n\r\n ३'VE<|fim_prefix|>​\n/‍\n0\u000b٣٤٥٦Džungla", "tokens": 71, "pieces": ["e", "́HTTPServerᵃ", "👍🏽\r\n\r\n", "<", "Ⅳ", "'re", "\"", "0", "\u000b", "9", "👍🏽​", "ع", "!", " ", "­'", "T", "\r\n\r\n", " ", "३", "'VE", "<|", "fim", "_prefix", "|>​\n", "/‍\n", "0", "\u000b", "٣٤٥", "٦", "Džungla"]} +{"text": "' ..!0eſB \n ㋿'ſ's'T'漢\r\nZ\n/Abaé dAbEOT<|endoftext|>́👍🏽", "tokens": 48, "pieces": ["'", " ", "..!", "0", "eſB", " \n", " ㋿'", "ſ", "'s", "'T", "'漢", "\r\n", "Z", "\n", "/Abae", "́", " ", " dAbEOT", "<|", "endoftext", "|>́👍🏽"]} +{"text": "'M/\r\nZHTTPServer <ᵃ \n e\n/fi.d9fi㋿'M字camelCasemd/👍🏽Ⅳ㍿字é😀🏽\r\r\t'T'll", "tokens": 54, "pieces": ["'M", "/\r\n", "ZHTTPServer", " ", "<ᵃ", " \n", " e", "\n", "/fi", ".d", "9", "fi", "㋿'", "M字camelCasemd", "/👍🏽", "Ⅳ", "㍿字e", "́😀🏽\r\r", "\t", "'T", "'ll"]} +{"text": "🙂ᵃ9ḍ̇‍AbⅣ३߅ſ\"", "tokens": 48, "pieces": ["🙂", "ᵃ", "9", "ḋ", "̣‍", "Ab", "Ⅳ३", "ß", "…ſ", "\""]} +{"text": "­𐞁EOTDžHTTPServerEOTعa\r\n\r\n", "tokens": 16, "pieces": ["­𐞁EOTDžHTTPServerEOTعa", "\r\n\r\n"]} +{"text": "éع!", "tokens": 3, "pieces": ["éع", "!"]} +{"text": "åaDžunglaacamelCase \n'D㋿/\r\n'sfí\n(\r!!EOTfi#$%iOS's'👍🏽(eEOT漢camelCase'S<\"㋿\r\n\u000b'T ", "tokens": 61, "pieces": ["a", "̊aDžungla", "acamelCase", " \n", "'D", "㋿/\r\n", "'s", "fi", "́\n", "(\r", "!!", "EOTfi", "#$%", "iOS", "'s", "'👍🏽(", "eEOT漢camelCase", "'S", "<\"㋿\r\n", "\u000b", "'T", " "]} +{"text": "\n\r\na/bAb\r
字­ſEOT漢0A", "tokens": 19, "pieces": ["\n\r\n", "a", "/bAb", "\r", "
字", "­ſEOT漢", "0", "A"]} +{"text": "ꟲs​ſ\t ́'ReaBaDž/'ReéⅣAbficamelCase𐞁aå\r\n\r\n<|endoftext|>İ\t \n m\u000b", "tokens": 46, "pieces": ["ꟲs", "​ſ", "\t", " ́'", "ReaBaDž", "/'", "Ree", "́", "Ⅳ", "AbficamelCase𐞁aa", "̊\r\n\r\n", "<|", "endoftext", "|>", "İ", "\t \n", " m", "\u000b"]} +{"text": "㍿(iOS\n/ᵃ㋿ #$%aİétaB 👍🏽A!!('ll#$%!!HTTPServerᵃ́ HTTPServer‍'re", "tokens": 48, "pieces": ["㍿(", "iOS", "\n", "/ᵃ", "㋿", " ", "#$%", "aİétaB", " ", " 👍🏽", "A", "!!('", "ll", "#$%!!", "HTTPServerᵃ", "́", " HTTPServer", "‍'", "re"]} +{"text": "३Ⅳ!!camelCase'reaAbå​\r👍🏽 😀🏽́ḍ̇ⅣⅣ", "tokens": 37, "pieces": ["३Ⅳ", "!!", "camelCase", "'re", "aAba", "̊​\r", "👍🏽", " ", "😀🏽́", "ḋ", "̣", "ⅣⅣ"]} +{"text": "İ9mA㍿ \u000b\r\n\r\n<|endoftext|>Z'M'Reİ \n mZ🙂m12345678Ⅳ😀🏽٣٤٥٦\"12345678EOT(EOTB0\n𐞁😀🏽t'llå", "tokens": 75, "pieces": ["İ", "9", "mA", "㍿", " \u000b\r\n\r\n", "<|", "endoftext", "|>", "Z", "'M", "'Re", "İ", " \n", " mZ", "🙂m", "123", "456", "78Ⅳ", "😀🏽<", "EOT", ">", "٣٤٥", "٦", "\"", "123", "456", "78", "EOT", "(EOTB", "0", "\n", "𐞁", "😀🏽", "t", "'ll", "a", "̊"]} +{"text": "'re'res!'s'll'D'Reé漢''ſ 'Re \n aB\r", "tokens": 20, "pieces": ["'re", "'re", "s", "!'", "s", "'ll", "'D", "'Re", "é漢", "''", "ſ", " '", "Re", " \n", " aB", "\r"]} +{"text": "> \ń­\r\n\r\n12345678/…HTTPServerᵃm\n 'T…,ſ\n/ꟲ/\r\n \nå㋿.DžunglaHTTPServerᵃ\t", "tokens": 57, "pieces": [">", " \n", "́­\r\n\r\n", "123", "456", "78", "/", "…HTTPServerᵃm", "\n", " ", " '", "T", "…", ",ſ", "\n", "/ꟲ", "/\r\n", " \n", "a", "̊㋿.", "DžunglaHTTPServerᵃ", "\t", ""]} +{"text": ">å,camelCasesA字Z<|fim_prefix|>s🙂fi're'iOS\n/𐞁­m'VE'D'ſ🙂½", "tokens": 42, "pieces": [">a", "̊,", "camelCasesA字Z", "<|", "fim", "_prefix", "|>", "s", "🙂fi", "'re", "'iOS", "\n", "/𐞁", "­m", "'VE", "'D", "'ſ", "🙂", "½"]} +{"text": "\r\n\r\n\"!! fi'S🙂0İ0\u000b👍🏽('S\"EOT", "tokens": 23, "pieces": ["\r\n\r\n", "\"!!", " fi", "'S", "🙂", "0", "İ", "0", "\u000b", "👍🏽('", "S", "\"EOT"]} +{"text": "ꟲ🙂/\r\nå912345678㍿३ABCſع\n'Tİ'T/\r\n/", "tokens": 30, "pieces": ["ꟲ", "🙂/\r\n", "a", "̊", "912", "345", "678", "㍿", "३", "ABCſع", "\n", "'T", "İ", "'T", "/\r\n", "/<", "EOT", ">"]} +{"text": "(tB(<|fim_prefix|>'D\"'re㍿\rDž\r\n'reB.'ſ'M", "tokens": 25, "pieces": ["(tB", "(<|", "fim", "_prefix", "|>'", "D", "\"'", "re", "㍿\r", "Dž", "\r\n", "'re", "B", ".'", "ſ", "'M"]} +{"text": "'漢iOS\n/ \néd漢,é\r\"m\rſDžunglaaBéß(İ漢!!ſع\r, 'SDžungla camelCase", "tokens": 46, "pieces": ["'漢iOS", "\n", "/", " \n", "éd漢", ",é", "\r", "\"m", "\r", "ſDžunglaaBéß", "(İ漢", "!!", "ſع", "\r", ",", " ", " '", "SDžungla", " camelCase"]} +{"text": "t(12345678dEOT👍🏽", "tokens": 14, "pieces": ["t", "(", "123", "456", "78", "dEOT", "👍🏽"]} +{"text": "ꟲ字🙂s", "tokens": 7, "pieces": ["ꟲ字", "🙂s"]} +{"text": "Abå\r\n\r\nDžungla<|endoftext|>A're'DⅣéaB!aBZiOS,\rDžungla#$%A0e\n ́Ⅳ\r\n\n/🙂.㋿iOS", "tokens": 57, "pieces": ["Aba", "̊\r\n\r\n", "Džungla", "<|", "endoftext", "|>", "A", "'re", "'D", "Ⅳ", "e", "́aB", "!aBZiOS", ",\r", "Džungla", "#$%", "A", "0", "e", "\n", " ", "́", "Ⅳ", "\r\n\n", "/🙂.㋿", "iOS"]} +{"text": "'T½漢>éḍ̇'re#$%a/bdB'S", "tokens": 22, "pieces": ["'T", "½", "漢", ">e", "́ḋ", "̣'", "re", "#$%<", "EOT", ">a", "/bdB", "'S"]} +{"text": "0'\r\n\r\n", "tokens": 2, "pieces": ["0", "'\r\n\r\n"]} +{"text": "'T٣٤٥٦12345678/ ,㍿\"'TBⅣ㍿", "tokens": 25, "pieces": ["'T", "٣٤٥", "٦12", "345", "678", "/", " ", " ,㍿\"'", "TB", "Ⅳ", "㍿"]} +{"text": "m!<|endoftext|>.é🙂> 'ſ \n !!å", "tokens": 21, "pieces": ["m", "!<|", "endoftext", "|>.", "e", "́🙂>", " '", "ſ", " \n", " !!", "a", "̊"]} +{"text": "t-\r\n\r\nDž‍", "tokens": 7, "pieces": ["t", "-\r\n\r\n", "Dž", "‍"]} +{"text": "<|endoftext|>\r\n\r\nſ 'D(​Džungla‍EOTİ 9B. \n camelCase!!t…>…'ll'Mḍ̇­EOT.½\t\u000b", "tokens": 55, "pieces": ["<|", "endoftext", "|>\r\n\r\n", "ſ", " ", "'D", "(​", "Džungla", "‍EOT", "İ", " ", " ", "9", "B", ".", " \n", " camelCase", "!!", "t", "…", ">", "…", "'ll", "'M", "ḋ", "̣­", "EOT", ".", "½", "\t\u000b"]} +{"text": "iOS٣٤٥٦\tt/\r\nᵃ<|fim_prefix|>.Ⅳ👍🏽🙂12345678\u000bmDžB/\r\n", "tokens": 43, "pieces": ["iOS", "٣٤٥", "٦", "\tt", "/\r\n", "ᵃ", "<|", "fim", "_prefix", "|>.", "Ⅳ", "👍🏽🙂", "123", "456", "78", "\u000bmDžB", "/\r\n"]} +{"text": ".\r\nA<́9d's 'll😀🏽é\r\n\r\n aB\rZcamelCase'sm>0<|endoftext|>s'reABC'Reḍ̇‍\r\n\r\n字Džungla\t\u000ba/bAb", "tokens": 56, "pieces": [".\r\n", "A", "<́", "9", "d", "'s", " ", "'ll", "😀🏽", "é", "\r\n\r\n", " aB", "\r", "ZcamelCase", "'s", "m", ">", "0", "<|", "endoftext", "|>", "s", "'re", "ABC", "'Re", "ḋ", "̣‍\r\n\r\n", "字Džungla", "\t", "\u000ba", "/bAb"]} +{"text": " !漢fiEOT㍿a/bm'ree\r\n\r\n!!<|endoftext|>\nع…", "tokens": 27, "pieces": [" !", "漢fiEOT", "㍿a", "/bm", "'re", "e", "\r\n\r\n", "!!<|", "endoftext", "|>\n", "ع", "…"]} +{"text": "s​\n/å,漢\u000b", "tokens": 10, "pieces": ["s", "​\n", "/a", "̊,", "漢", "\u000b"]} +{"text": "字٣٤٥٦'ſßſ/\r\na/bꟲ!!'T<|endoftext|>🙂å\r\n\r\naå", "tokens": 40, "pieces": ["字", "٣٤٥", "٦", "'ſ", "ßſ", "/\r\n", "a", "/bꟲ", "!!'", "T", "<|", "endoftext", "|>🙂", "a", "̊\r\n\r\n", "aa", "̊"]} +{"text": "ZaB >٣٤٥٦<|endoftext|>'llB\rZ(<|endoftext|>HTTPServerEOT>0😀🏽'D'T(HTTPServer,ß \n a/bⅣ\u000baB B\r\n/'llße", "tokens": 63, "pieces": ["ZaB", " ", " >", "٣٤٥", "٦", "<|", "endoftext", "|>'", "llB", "\r", "Z", "(<|", "endoftext", "|>", "HTTPServerEOT", ">", "0", "😀🏽'", "D", "'T", "(HTTPServer", ",ß", " \n", " a", "/b", "Ⅳ", "\u000baB", " B", "\r\n", "/'", "llße"]} +{"text": "s'", "tokens": 2, "pieces": ["s", "'"]} +{"text": "<|endoftext|>e\n­'llé!12345678ſ'S're-𐞁٣٤٥٦\r\n\n ('sᵃ", "tokens": 40, "pieces": ["<|", "endoftext", "|>", "e", "\n", "­'", "lle", "́!", "123", "456", "78", "ſ", "'S", "'re", "-𐞁", "٣٤٥", "٦", "\r\n\n", " ('", "sᵃ"]} +{"text": "DžunglaDžungla(- a/b👍🏽\n .👍🏽0ع\n!!'ReéiOS ­ \naB ", "tokens": 42, "pieces": ["DžunglaDžungla", "(-", " ", " a", "/b", "👍🏽\n", " ", ".👍🏽", "0", "ع", "\n", "!!'", "Ree", "́iOS", " ", " ­", " \n", "aB", " "]} +{"text": "å#$%a ḍ̇å'VE'S‍<漢‍<😀🏽\r\n'S\t. \tiOSAb \ns's٣٤٥٦'ll/\r\nع.ḍ̇'Re 9a/bAbABC'S", "tokens": 68, "pieces": ["a", "̊#$%", "a", " ḋ", "̣a", "̊'", "VE", "'S", "‍<", "漢", "‍<😀🏽\r\n", "'S", "\t", ".", " ", "\tiOSAb", " \n", "s", "'s", "٣٤٥", "٦", "'ll", "/\r\n", "ع", ".ḋ", "̣'", "Re", " ", "9", "a", "/bAbABC", "'S"]} +{"text": "t!'Refis\t'ſ#$%\r\na Dž㍿0\r\n\r\nİ12345678ABCß字㋿HTTPServer𐞁Dž!!(", "tokens": 42, "pieces": ["t", "!'", "Refis", "\t", "'ſ", "#$%\r\n", "a", " ", " Dž", "㍿", "0", "\r\n\r\n", "İ", "123", "456", "78", "ABCß字", "㋿HTTPServer𐞁Dž", "!!("]} +{"text": "٣٤٥٦३", "tokens": 10, "pieces": ["٣٤٥", "٦३"]} +{"text": "ß Ⅳ\n/'VE\r\nZ-#$%㋿🙂 \n ㍿iOS \n 'T㋿ع'ReAb!́é", "tokens": 38, "pieces": ["ß", " ", " ", "Ⅳ", "\n", "/'", "VE", "\r\n", "Z", "-#$%㋿🙂", " \n", " ", " ㍿", "iOS", " \n", " '", "T", "㋿ع", "'Re", "Ab", "!́", "e", "́"]} +{"text": "'ſ-½\n \niOSé  >0/\r\n /'sa'Re'\t\"!! B­عDž \n9\té \n", "tokens": 34, "pieces": ["'ſ", "-", "½", "\n \n", "iOSe", "́", " ", " ", ">", "0", "/\r\n", " ", " /'", "sa", "'Re", "'", "\t", "\"!!", " B", "­عDž", " \n", "9", "\té", " \n"]} +{"text": "0'TåAb", "tokens": 6, "pieces": ["0", "'T", "a", "̊Ab"]} +{"text": "漢é\r­Ab'ReᵃiOS!!ᵃſ<|fim_prefix|>𐞁 éß
\r", "­Ab", "'Re", "ᵃiOS", "!!", "ᵃſ", "<|", "fim", "_prefix", "|>", "𐞁", " ", " e", "́ß", "
", ".\r0'VE㍿'T­<|endoftext|>/!!", "tokens": 45, "pieces": [" '", "S", "!!", " ", " '", "M", ",𐞁ß", "
", "'VE", "AiOS", "‍", "0", "𐞁", ">.\r", "0", "'VE", "㍿'", "T", "­<|", "endoftext", "|>/!!"]} +{"text": "HTTPServer!!a/bDž😀🏽DžⅣ<\tét\n/ \n <|fim_prefix|>AbaBé٣٤٥٦iOS'M𐞁३9! \n /
s", "tokens": 55, "pieces": ["HTTPServer", "!!", "a", "/bDž", "😀🏽", "Dž", "Ⅳ", "<", "\tét", "\n", "/", " \n", " <|", "fim", "_prefix", "|>", "AbaBé", "٣٤٥", "٦", "iOS", "'M", "𐞁", "३9", "!", " \n", " /", "
s"]} +{"text": "HTTPServerA>'reḍ̇ \"'S\r\n\r\n漢aBA< \n B/​­iOS­漢-a/baB​㍿㋿'ſ'S'VEß", "tokens": 52, "pieces": ["HTTPServerA", ">'", "reḋ", "̣", " ", "\"'", "S", "\r\n\r\n", "漢aBA", "<", " \n", " <", "META", "_START", ">B", "/​­", "iOS", "­漢", "-a", "/baB", "​㍿㋿'", "ſ", "'S", "'VE", "ß"]} +{"text": "‍9ꟲ \ńᵃfiİ㍿ \nfi'VEå​Aᵃ0漢camelCase'reſAbEOT", "tokens": 42, "pieces": ["‍", "9", "ꟲ", " \n", "́ᵃfiİ", "㍿", " \n", "fi", "'VE", "a", "̊​", "Aᵃ", "0", "漢camelCase", "'re", "ſAbEOT"]} +{"text": "a/b'Tꟲ/\r\nABC0字", "tokens": 14, "pieces": ["a", "/b", "'T", "ꟲ", "/\r\n", "ABC", "", "0", "字"]} +{"text": "'VE'VEᵃ𐞁'ſcamelCase‍<|endoftext|>så½İ", "tokens": 33, "pieces": ["'VE", "'VE", "ᵃ𐞁", "'ſ", "camelCase", "‍<|", "endoftext", "|>", "s", "a", "̊", "½", "İ"]} +{"text": "Dž'ſ!!ABC'lladé\r\n\r\n㍿Ab're", "tokens": 16, "pieces": ["Dž", "'ſ", "!!", "ABC", "'ll", "ade", "́\r\n\r\n", "㍿Ab", "'re"]} +{"text": "ḍ̇", "tokens": 5, "pieces": ["ḋ", "̣"]} +{"text": "HTTPServers\nꟲ🙂>a㍿", "tokens": 16, "pieces": ["HTTPServers", "\n", "ꟲ", "🙂>", "a", "㍿"]} +{"text": "😀🏽Ab/\r\n!ß㍿ 'ſ \n/.", " '", "ſ", " \n", "/.<", "m", " HTTPServer", "😀🏽", " "]} +{"text": "'ſ'llaBᵃ\n/", "tokens": 11, "pieces": ["'ſ", "'ll", "aBᵃ", "\n", "/"]} +{"text": "½ \n ​-0字å \n.'ſ٣٤٥٦Dž9 \n'ſ'Rea😀🏽 \t…t漢#$%Ab🙂ḍ̇", "tokens": 52, "pieces": ["½", " \n", " ​-", "0", "字a", "̊", " \n", ".'", "ſ", "٣٤٥", "٦", "Dž", "9", " \n", "'ſ", "'Re", "a", "😀🏽", " \t", "…t漢", "#$%", "Ab", "🙂ḋ", "̣"]} +{"text": "EOTⅣ9", "tokens": 8, "pieces": ["EOT", "Ⅳ9"]} +{"text": "EOT,\r<|endoftext|>😀🏽३!Bİ‍ᵃiOSعAb ᵃ'ſ/<|endoftext|> ḍ̇İ'ſᵃ'M/ꟲ😀🏽ع\r'red!<|endoftext|>㍿aB", "tokens": 85, "pieces": ["EOT", ",\r", "<|", "endoftext", "|>😀🏽", "३", "!Bİ", "‍ᵃiOSعAb", " ᵃ", "'ſ", "/<|", "endoftext", "|>", " ḋ", "̣İ", "'ſ", "ᵃ", "'M", "/ꟲ", "😀🏽", "ع", "\r", "'re", "d", "!<|", "endoftext", "|><", "EOT", ">㍿", "aB"]} +{"text": " éſd,<|endoftext|>å12345678Džt!!\nعa😀🏽EOT­\"'s½…'Ⅳ­'D \n ,'ſ\n-fi0", "tokens": 55, "pieces": [" éſd", ",<|", "endoftext", "|>", "a", "̊", "123", "456", "78", "Džt", "!!\n", "عa", "😀🏽", "EOT", "­\"'", "s", "½", "…", "'", "Ⅳ", "­'", "D", " \n", " ,'", "ſ", "\n", "-fi", "0"]} +{"text": "<|fim_prefix|>", "tokens": 7, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "/mEOT​ ABC", "tokens": 9, "pieces": ["/m", "EOT", "​", " ", " ABC"]} +{"text": "\r\n\r\nfi ½a/b㋿\" \u000bZt's漢३é", "tokens": 23, "pieces": ["\r\n\r\n", "fi", " ", " ", "½", "a", "/b", "㋿\"", " ", "\u000bZt", "'s", "漢", "३", "e", "́"]} +{"text": "t…٣٤٥٦Z'Reİ", "tokens": 14, "pieces": ["t", "…", "٣٤٥", "٦", "Z", "'Re", "İ"]} +{"text": " 'Re㍿😀🏽\" \n㋿iOS(12345678İA½,‍🙂‍'S's. …a/b-'sİ­<|fim_prefix|>İA", "tokens": 55, "pieces": [" ", " '", "Re", "㍿😀🏽\"", " \n", "㋿iOS", "(", "123", "456", "78", "İA", "½", ",‍🙂‍'", "S", "'s", ".", " ", "…a", "/b", "-'", "sİ", "­<|", "fim", "_prefix", "|>", "İA"]} +{"text": "a\r\n\r\nA​<|endoftext|>'s's'T३'M.sAABCDž👍🏽Džungla\r\n𐞁Džunglaع\u000bEOT​'ll\r
HTTPServer🙂½", "tokens": 58, "pieces": ["a", "\r\n\r\n", "A", "​<|", "endoftext", "|>'", "s", "'s", "'T", "३", "'M", ".sAABCDž", "👍🏽", "Džungla", "\r\n", "𐞁Džunglaع", "\u000bEOT", "​'", "ll", "\r", "
HTTPServer", "🙂", "½"]} +{"text": "Ⅳ٣٤٥٦'VE0#$%🙂9-#$%é!EOT𐞁ḍ̇e fiAb", "tokens": 39, "pieces": ["Ⅳ٣٤", "٥٦", "'VE", "0", "#$%🙂", "9", "-#$%", "é", "!EOT𐞁ḋ", "̣e", " fiAb"]} +{"text": "👍🏽字/\r\nBDž \n#$%\"eßß½\"", "tokens": 19, "pieces": ["👍🏽", "字", "/\r\n", "BDž", " \n", "#$%\"", "eßß", "½", "\""]} +{"text": "'T-'Re𐞁", "tokens": 7, "pieces": ["'T", "-'", "Re𐞁"]} +{"text": "!!\r\n'Ret \n\r\ns#$%Ⅳ", "tokens": 14, "pieces": ["!!\r\n", "'Re", "t", " \n\r\n", "s", "#$%", "Ⅳ", ""]} +{"text": "!!ſ𐞁👍🏽ABC\nḍ̇a/b<", "tokens": 23, "pieces": ["!!", "ſ𐞁", "👍🏽", "ABC", "\n", "ḋ", "̣a", "/b", "<"]} +{"text": "å DžunglaEOT!!ḍ̇-Ab>.'sdaB'reḍ̇\"d\r\n\r\n #$%A\n/", "tokens": 38, "pieces": ["a", "̊", " DžunglaEOT", "!!", "ḋ", "̣-", "Ab", ">.'", "sdaB", "'re", "ḋ", "̣\"", "d", "\r\n\r\n", " #$%", "A", "\n", "/"]} +{"text": "'M'ſABC'​👍🏽 ", "tokens": 14, "pieces": ["'M", "'ſ", "ABC", "'​👍🏽", " "]} +{"text": "\n'D'ſ/ \n'M­dḍ̇>㋿<|fim_prefix|>ᵃ漢\" \n\r\n0​ſſ​-", "tokens": 42, "pieces": ["\n", "'D", "'ſ", "/", " \n", "'M", "­dḋ", "̣>㋿<|", "fim", "_prefix", "|>", "ᵃ漢", "\"", " \n\r\n", "0", "​ſſ", "​-"]} +{"text": "Z'ſ'عᵃfi", "tokens": 11, "pieces": ["Z", "'ſ", "'عᵃfi"]} +{"text": "ſ!'ſ/\r\n \n𐞁eaB👍🏽́ da", "tokens": 22, "pieces": ["ſ", "!'", "ſ", "/\r\n", " \n", "𐞁eaB", "👍🏽́", " da"]} +{"text": "A漢 12345678ḍ̇HTTPServer㍿\n/'Re\r\n\r\n‍B9ßaB'S!!👍🏽'M!!'llAꟲ­👍🏽Dž9m½ع.字m\u000bA", "tokens": 66, "pieces": ["A漢", " ", "123", "456", "78", "ḋ", "̣HTTPServer", "㍿\n", "/'", "Re", "\r\n\r\n", "‍B", "9", "ßaB", "'S", "!!👍🏽'", "M", "!!'", "llAꟲ", "­👍🏽", "Dž", "9", "m", "½", "ع", ".字m", "\u000bA"]} +{"text": "camelCasee \n ٣٤٥٦\u000b ABC'T", "tokens": 17, "pieces": ["camelCasee", " \n", " ", "٣٤٥", "٦", "\u000b", " ABC", "'T"]} +{"text": "'s'Må\u000b ́Ab'T(‍", "tokens": 13, "pieces": ["'s", "'M", "a", "̊", "\u000b", " ́", "Ab", "'T", "(‍"]} +{"text": " \n\n/#$%ḍ̇'Re'Ta/b㋿>ḍ̇ \u000bع'll'ſ
 \t!३éA \n camelCasea/bå'ſB0B…< \n ,,a", "tokens": 67, "pieces": [" \n\n", "/#$%", "ḋ", "̣'", "Re", "'", "Ta", "/b", "㋿>", "ḋ", "̣", " ", "\u000bع", "'ll", "'ſ", "
 ", "\t", "!", "३", "éA", " \n", " camelCasea", "/ba", "̊'", "ſB", "0", "B", "…", "<", " \n", " ,,", "a"]} +{"text": " \n …<|fim_prefix|>\"𐞁!㍿'ss", "tokens": 21, "pieces": [" \n", " ", "…", "<|", "fim", "_prefix", "|>\"", "𐞁", "!㍿'", "ss"]} +{"text": "ᵃ😀🏽 é,…fi", "tokens": 16, "pieces": ["ᵃ", "😀🏽", " e", "́,", "…fi"]} +{"text": "é're-\r\n", "tokens": 3, "pieces": ["é", "'re", "-\r\n"]} +{"text": "12345678e½'M\rABC's㍿<\"𐞁éſ㋿👍🏽Ab iOS'T\rⅣ('ll'llABCABCBiOS­>Ⅳ", "tokens": 47, "pieces": ["123", "456", "78", "e", "½", "'M", "\r", "ABC", "'s", "㍿<\"", "𐞁éſ", "㋿👍🏽", "Ab", " iOS", "'T", "\r", "Ⅳ", "('", "ll", "'ll", "ABCABCBiOS", "­>", "Ⅳ"]} +{"text": "\r<٣٤٥٦\"İ́", "tokens": 13, "pieces": ["\r", "<", "٣٤٥", "٦", "\"İ", "́"]} +{"text": "<|fim_prefix|>.aBA𐞁é‍mAbABCåé'reꟲaaBåDž字‍9!é'llA!\rDžᵃ'Ret", "tokens": 61, "pieces": ["<|", "fim", "_prefix", "|>.", "aBA𐞁e", "́‍", "mAbABCa", "̊<", "META", "_START", ">e", "́'", "reꟲaaBa", "̊Dž字", "‍", "9", "!e", "́'", "llA", "!\r", "Džᵃ", "'Re", "t"]} +{"text": "fi0\n/🙂camelCase<ᵃ𐞁", "tokens": 23, "pieces": ["fi", "0", "\n", "/<", "META", "_START", ">🙂", "camelCase", "<ᵃ𐞁", ""]} +{"text": "​ß\r\nå漢å\n/>< \n 'T३EOT'٣٤٥٦ABC漢t9s !!३", "tokens": 42, "pieces": ["​ß", "\r\n", "a", "̊漢a", "̊<", "META", "_START", ">\n", "/><", " \n", " '", "T", "३", "EOT", "'", "٣٤٥", "٦", "ABC漢t", "9", "s", " ", "!!", "३"]} +{"text": "(d㍿ e.'s‍fiåİ\nB/\r\n \n­ (ß", "tokens": 27, "pieces": ["(d", "㍿", " e", ".'", "s", "‍fi", "a", "̊İ", "\n", "B", "/\r\n", " \n", "­", " ", "(ß"]} +{"text": "'VE\r\n!!🙂\r\nⅣB​Ab12345678#$%'TdDžſABC", "tokens": 23, "pieces": ["'VE", "\r\n", "!!🙂\r\n", "Ⅳ", "B", "​Ab", "123", "456", "78", "#$%'", "TdDžſABC"]} +{"text": "𐞁're'sDžungladᵃ9\r­t-…/𐞁ḍ̇<\n(𐞁aB字, \nå\u000b<|fim_prefix|><|endoftext|>'​‍ꟲ", "tokens": 66, "pieces": ["𐞁", "'re", "'s", "Džungladᵃ", "9", "\r", "­t", "-", "…", "/𐞁ḋ", "̣<\n", "(<", "META", "_START", ">𐞁aB字", ",", " \n", "a", "̊", "\u000b", "<|", "fim", "_prefix", "|><|", "endoftext", "|>'​‍", "ꟲ"]} +{"text": "İ'VE", "tokens": 6, "pieces": ["İ", "'", "VE"]} +{"text": "Ab …ꟲ٣٤٥٦­HTTPServer…<|fim_prefix|>ſ", "tokens": 29, "pieces": ["Ab", " ", "…ꟲ", "٣٤٥", "٦", "­HTTPServer", "…", "<|", "fim", "_prefix", "|>", "ſ"]} +{"text": ">å\u000bAABC,HTTPServerǻ-(½EOTZ'D🙂Džungla字ᵃİ👍🏽\t\n३🙂👍🏽m 'll'S\u000bſ/\r\ncamelCase
", "tokens": 64, "pieces": [">a", "̊", "\u000bAABC", ",HTTPServera", "̊́-(", "½", "EOTZ", "'D", "🙂Džungla字ᵃİ", "👍🏽", "\t\n", "३", "🙂👍🏽", "m", " ", "'ll", "'S", "\u000bſ", "/\r\n", "camelCase", "
"]} +{"text": " 'T\"'T<|endoftext|>9Dž/\r\nABC😀🏽éDž字>!!!HTTPServer ㍿\r\n\r\nHTTPServer٣٤٥٦'Re12345678Džungla½", "tokens": 59, "pieces": [" '", "T", "\"'", "T", "<|", "endoftext", "|>", "9", "Dž", "/\r\n", "ABC", "😀🏽", "e", "́Dž字", ">!!!", "HTTPServer", " ", " ㍿\r\n\r\n", "HTTPServer", "٣٤٥", "٦", "'Re", "123", "456", "78", "Džungla", "½"]} +{"text": "<|endoftext|>عAbⅣ\r\n\r\nméßⅣ\n/ABC𐞁ßſcamelCase'M𐞁(ficamelCaseᵃiOSB\r\n\r\nſ½é'Saa", "tokens": 54, "pieces": ["<|", "endoftext", "|>", "عAb", "Ⅳ", "\r\n\r\n", "méß", "Ⅳ", "\n", "/ABC𐞁ßſ", "camelCase", "'M", "𐞁", "(ficamelCaseᵃiOSB", "\r\n\r\n", "ſ", "½", "é", "'S", "aa"]} +{"text": "a\r🙂'ree字<|endoftext|>aB­./漢", "tokens": 22, "pieces": ["a", "\r", "🙂'", "ree字", "<|", "endoftext", "|><", "EOT", ">aB", "­./", "漢"]} +{"text": "A٣٤٥٦'DZ'VE㋿mB/\r\n, \n👍🏽Z㍿ßfi'T३\n/DžunglaDž('M…\n/9…\r", "tokens": 56, "pieces": ["A", "٣٤٥", "٦", "'D", "Z", "'VE", "㋿mB", "/\r\n", ",", " \n", "👍🏽", "Z", "㍿ßfi", "'T", "३", "\n", "/DžunglaDž", "('", "M", "…\n", "/", "9", "…\r"]} +{"text": "'Se½fi٣٤٥٦३İ😀🏽 ,-́fia\n!<|fim_prefix|>fiAb!/Z.écamelCaseB½'D㋿HTTPServera'VEå漢", "tokens": 59, "pieces": ["'S", "e", "½", "fi", "٣٤٥", "٦३", "İ", "😀🏽", " ,-́", "fia", "\n", "!<|", "fim", "_prefix", "|>", "fiAb", "!/", "Z", ".écamelCaseB", "½", "'D", "㋿HTTPServera", "'VE", "a", "̊漢"]} +{"text": "e👍🏽'­
\rsḍ̇ \r\n\r\n/\r\n/DžAb<|endoftext|>\r\nعⅣ
iOSfi\r\n👍🏽d३ Dž\r9tEOT½e#$%a/bDžungla", "tokens": 67, "pieces": ["e", "👍🏽'­", "
\r", "sḋ", "̣", " \r\n\r\n", "/\r\n", "/DžAb", "<|", "endoftext", "|>\r\n", "ع", "Ⅳ", "
iOSfi", "\r\n", "👍🏽", "d", "३", " Dž", "\r", "9", "tEOT", "½", "e", "#$%", "a", "/bDžungla"]} +{"text": "́​'Re'D", "tokens": 5, "pieces": ["́​'", "Re", "'D"]} +{"text": " \n …d㍿
<ꟲé.ع're're", "tokens": 20, "pieces": [" \n", " ", "…d", "㍿", "
", "<ꟲe", "́.", "ع", "'re", "'re"]} +{"text": "å'DABC's'D𐞁<|endoftext|>́😀🏽,​\r\n\r\n<|endoftext|>ḍ̇e​­👍🏽aBéعaB'Reꟲ㍿", "tokens": 69, "pieces": ["a", "̊'", "DABC", "'s", "'D", "𐞁", "<|", "endoftext", "|>́😀🏽,​\r\n\r\n", "<|", "endoftext", "|>", "ḋ", "̣e", "​­<", "META", "_START", ">👍🏽", "aBéعaB", "'Re", "ꟲ", "㍿"]} +{"text": "9 ३́/\r\n\t", "tokens": 7, "pieces": ["9", " ", "३", "́/\r\n", "\t"]} +{"text": "'re'字!é'ſaB漢٣٤٥٦'Re٣٤٥٦é", "tokens": 31, "pieces": ["'re", "'字", "!e", "́'", "ſaB漢", "٣٤٥", "٦", "'Re", "٣٤٥", "٦", "é"]} +{"text": "́a/b\u000b\r\n\r\n!!é 12345678'll\t,Ab'Sſſm-.\n/a/bAbⅣ<|fim_prefix|>/\r\n(>mcamelCase#$%a/b\t'HTTPServer>漢\t", "tokens": 54, "pieces": ["́a", "/b", "\u000b\r\n\r\n", "!!", "e", "́", " ", "123", "456", "78", "'ll", "\t", ",Ab", "'S", "ſſm", "-.\n", "/a", "/bAb", "Ⅳ", "<|", "fim", "_prefix", "|>/\r\n", "(>", "mcamelCase", "#$%", "a", "/b", "\t", "'HTTPServer", ">漢", "\t"]} +{"text": "عdABs'dcamelCase/\"DžunglaaBt𐞁>'ſ12345678aBZ.#$%éꟲ½\u000bs<", "tokens": 50, "pieces": ["عdABs", "'d", "camelCase", "/\"", "DžunglaaBt𐞁", ">'", "ſ", "123", "456", "78", "aBZ", ".<", "META", "_START", ">#$%", "e", "́ꟲ", "½", "\u000bs", "<<", "EOT", ">"]} +{"text": "9t<|fim_prefix|>́\n/-12345678dꟲ<|endoftext|>ſa'M½٣٤٥٦ \n \n३d>d", "tokens": 44, "pieces": ["9", "t", "<|", "fim", "_prefix", "|>́\n", "/-", "123", "456", "78", "dꟲ", "<|", "endoftext", "|>", "ſa", "'M", "½٣٤", "٥٦", " \n \n", "३", "d", ">d"]} +{"text": "\ta/bİ٣٤٥٦<|fim_prefix|>é漢>'Sſ<\t'D\u000bAb<|endoftext|>tḍ̇'MéDž \nḍ̇漢ᵃDžunglá", "tokens": 64, "pieces": ["\ta", "/bİ", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "é漢", ">'", "Sſ", "<", "\t", "'D", "\u000bAb", "<|", "endoftext", "|>", "tḋ", "̣'", "MéDž", " \n", "ḋ", "̣漢ᵃDžungla", "́"]} +{"text": "ſ'T/!aBaEOT३aB'Re/\"a/bſ", "tokens": 19, "pieces": ["ſ", "'T", "/!", "aBaEOT", "३", "aB", "'Re", "/\"", "a", "/bſ"]} +{"text": "'T.!Z
t漢ᵃ\r\n\r\n9👍🏽/\r\na/bع\réſZZ'Re,", "tokens": 30, "pieces": ["'T", ".!", "Z", "
t漢ᵃ", "\r\n\r\n", "9", "👍🏽/\r\n", "a", "/bع", "\r", "éſZZ", "'Re", ","]} +{"text": "ABCEOT>/\r\n'D're", "tokens": 9, "pieces": ["ABC", "EOT", ">/\r\n", "'D", "'re"]} +{"text": "dſ9HTTPServerEOTꟲ/", "tokens": 14, "pieces": ["dſ", "9", "HTTPServerEOTꟲ", "/"]} +{"text": "00ådDžungla'👍🏽…Dž're<\u000bZiOS𐞁'ſ'VEé9s'DDž A<|fim_prefix|>'T𐞁\n/ß s'M('ſ", "tokens": 66, "pieces": ["00", "a", "̊dDžungla", "'👍🏽", "…Dž", "'re", "<", "\u000bZiOS𐞁", "'ſ", "'VE", "é", "9", "s", "'", "DDž", " ", " A", "<|", "fim", "_prefix", "|>'", "T𐞁", "\n", "/ß", " s", "'M", "('", "ſ"]} +{"text": " ḍ̇", "tokens": 6, "pieces": [" ḋ", "̣"]} +{"text": "'s㍿ 0aB\tHTTPServerEOTfi🙂 dé́ 'D!!\r\n\r\nt'M\r\nåDž.\n/HTTPServer", "tokens": 37, "pieces": ["'s", "㍿", " ", "0", "aB", "\tHTTPServerEOTfi", "🙂", " ", " dé", "́", " ", " '", "D", "!!\r\n\r\n", "t", "'M", "\r\n", "a", "̊Dž", ".\n", "/HTTPServer"]} +{"text": ">'S\n/A'StEOT㍿३(å👍🏽\r/\r\n'll'M'/\r\n\r\n\r\n\r\n(.'D​  😀🏽ᵃ🙂!!Aſ \nB ", "tokens": 62, "pieces": [">'", "S", "\n", "/A", "'S", "tEOT", "㍿", "३", "(a", "̊👍🏽\r", "/\r\n", "'ll", "'M", "'/\r\n\r\n\r\n\r\n", "(.'", "D", "​", " ", " <", "EOT", ">", " ", "😀🏽", "ᵃ", "🙂<", "META", "_START", ">!!", "Aſ", " \n", "B", " "]} +{"text": "­é", "tokens": 2, "pieces": ["­é"]} +{"text": "<|fim_prefix|>\t𐞁字fi
", "tokens": 17, "pieces": ["<|", "fim", "_prefix", "|>", "\t𐞁字fi", "
"]} +{"text": ",ᵃ字aBa/bZABC字'漢­👍🏽字0,字 <|fim_prefix|>eABC㍿aB
", "tokens": 42, "pieces": [",ᵃ字aBa", "/bZABC字", "'漢", "­👍🏽", "字", "0", ",字", " ", "<|", "fim", "_prefix", "|>", "eABC", "㍿aB", "
"]} +{"text": "㋿३'S \n'siOS\n/٣٤٥٦camelCase'ſ­Ab'SAaBsé", "tokens": 32, "pieces": ["㋿", "३", "'S", " \n", "'s", "iOS", "\n", "/", "٣٤٥", "٦", "camelCase", "'ſ", "­Ab", "'S", "AaBsé"]} +{"text": "漢!>'scamelCasé", "tokens": 8, "pieces": ["漢", "!>'", "scamelCase", "́"]} +{"text": "-9Džunglaåm/e'M-Ab\u000b \nAᵃ\u000b", "tokens": 40, "pieces": ["-", "9", "Džunglaa", "̊m", "/e", "'M", "-Ab", "", "\u000b \n", "Aᵃ", "\u000b"]} +{"text": " (camelCase -#$%'Ta99!é#$%\n/'Dé㋿Džungla
\r\n\r\nᵃ<|fim_prefix|>aB㍿-  \r\n/​\n\r\n\r\neABCa/b
", "tokens": 58, "pieces": [" (", "camelCase", " ", "-#$%'", "Ta", "99", "!e", "́#$%\n", "/'", "De", "́㋿", "Džungla", "
\r\n\r\n", "ᵃ", "<|", "fim", "_prefix", "|>", "aB", "㍿-", "  \r\n", "/<", "META", "_START", ">​\n\r\n\r\n", "eABCa", "/b", "
"]} +{"text": "​é😀🏽ḍ̇Dž.\r\n\r\n's½\u000b'T́", "tokens": 20, "pieces": ["​é", "😀🏽", "ḋ", "̣Dž", ".\r\n\r\n", "'s", "½", "\u000b", "'T", "́"]} +{"text": ">å(
<㋿'s٣٤٥٦-(a/b#$%İ'll#$%'T'S\"", "tokens": 32, "pieces": [">a", "̊(", "
", "<㋿'", "s", "٣٤٥", "٦", "-(", "a", "/b", "#$%", "İ", "'ll", "#$%'", "T", "'S", "\""]} +{"text": "\r\nḍ̇\r\n12345678 \n (Ⅳd<|endoftext|>ABC'll…éDžZ'll‍­'T\u000b0­ \n ٣٤٥٦(éAEOTeå!́𐞁ḍ̇…m<|endoftext|>", "tokens": 79, "pieces": ["\r\n", "ḋ", "̣\r\n", "123", "456", "78", " \n", " (", "Ⅳ", "d", "<|", "endoftext", "|>", "ABC", "'ll", "…éDžZ", "'ll", "‍­'", "T", "\u000b", "0", "­", " \n", " ", "٣٤٥", "٦", "(éAEOTea", "̊!́", "𐞁ḋ", "̣", "…m", "<|", "endoftext", "|>"]} +{"text": "é 'ſ㋿Z'VE<|fim_prefix|>\n/12345678'Re
s<㋿eᵃ\t漢\r\n\r\n漢ᵃ\r\n12345678ᵃ­åsZⅣ", "tokens": 59, "pieces": ["e", "́", " ", " '", "ſ", "㋿Z", "'VE", "<|", "fim", "_prefix", "|>\n", "/", "123", "456", "78", "'Re", "
s", "<㋿", "eᵃ", "\t漢", "\r\n\r\n", "漢ᵃ", "\r\n", "123", "456", "78", "ᵃ", "­a", "̊sZ", "Ⅳ"]} +{"text": "𐞁𐞁½'s", "tokens": 10, "pieces": ["𐞁𐞁", "½", "'s"]} +{"text": "/\r\n ", "tokens": 2, "pieces": ["/\r\n", " "]} +{"text": "漢>0're", "tokens": 5, "pieces": ["漢", ">", "0", "'re"]} +{"text": "'SEOT\n/ᵃßEOTa/bZ㍿ \nع,عHTTPServer‍  ‍m'DA'Reعع", "tokens": 36, "pieces": ["'S", "EOT", "\n", "/ᵃßEOTa", "/bZ", "㍿", " \n", "ع", ",عHTTPServer", "‍", " ", " ", "‍m", "'D", "A", "'Re", "عع"]} +{"text": "!ßⅣḍ̇ꟲ'ſ㍿", "tokens": 18, "pieces": ["!ß", "Ⅳ", "ḋ", "̣ꟲ", "'ſ", "㍿"]} +{"text": "ꟲ12345678\rſ<|fim_prefix|>٣٤٥٦​Džunglaé𐞁Dž#$%012345678\r\n\r\n٣٤٥٦!Džungla😀🏽!é\n/½12345678'D ㍿\r\nd\r\n\r\nİ
Ab-'😀🏽fi", "tokens": 94, "pieces": ["ꟲ", "123", "456", "78", "\r", "ſ", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "​Džunglaé𐞁", "Dž", "#$%", "012", "345", "678", "\r\n\r\n", "٣٤٥", "٦", "!Džungla", "😀🏽!", "é", "\n", "/", "½12", "345", "678", "'D", " ", "㍿\r\n", "d", "\r\n\r\n", "İ", "", "
Ab", "-'😀🏽", "fi"]} +{"text": "𐞁'Re\n/Ab𐞁ABC३ß㍿ \n३ABC'ſ字字m're 'ReDžungla s👍🏽ḍ̇DžunglasDž9<|endoftext|>\t\n/३\r\n(𐞁ḍ̇'M'VEABC", "tokens": 83, "pieces": ["𐞁", "'Re", "\n", "/Ab𐞁ABC", "३", "ß", "㍿", " \n", "३", "ABC", "'ſ", "字字m", "'re", " ", "'Re", "Džungla", " s", "👍🏽", "ḋ", "̣DžunglasDž", "9", "<|", "endoftext", "|>", "\t\n", "/", "३", "\r\n", "(𐞁ḋ", "̣'", "M", "'VE", "ABC"]} +{"text": "iOS(ABCfiDžunglae🙂EOT­Džungla㍿#$%12345678camelCase'ſ#$%#$%'MaB٣٤٥٦'Reé'llſſ", "tokens": 53, "pieces": ["iOS", "(ABCfiDžunglae", "🙂EOT", "­Džungla", "㍿#$%", "123", "456", "78", "camelCase", "'ſ", "#$%#$%'", "MaB", "٣٤٥", "٦", "'Re", "é", "'ll", "ſſ"]} +{"text": "٣٤٥٦㋿a\r\naa('llcamelCaseDžunglaDž\n/a ZiOS🙂३ABCa/b> \n0'ſ\n/-#$%𐞁'll\r\n\u000b'D>㍿\r\n\r\n", "tokens": 65, "pieces": ["٣٤٥", "٦", "㋿a", "\r\n", "aa", "('", "llcamelCaseDžunglaDž", "\n", "/a", " ", " <", "META", "_START", ">ZiOS", "🙂", "३", "ABCa", "/b", ">", " \n", "0", "'ſ", "\n", "/-#$%", "𐞁", "'ll", "\r\n", "\u000b", "'D", ">㍿\r\n\r\n"]} +{"text": "😀🏽Džꟲ0Dž'rea👍🏽 \nꟲ'D́A­Z 9As,#$% DžHTTPServer.\"Ⅳ", "tokens": 48, "pieces": ["😀🏽", "Džꟲ", "0", "Dž", "'re", "a", "👍🏽", " \n", "ꟲ", "'D", "́A", "­Z", " ", " ", "9", "As", ",#$%", " DžHTTPServer", ".\"", "Ⅳ"]} +{"text": "३ſ<|endoftext|>ⅣBB! \n\t'Dm…🙂-/\r\nⅣaB 're👍🏽camelCase😀🏽ḍ̇aBéßEOT👍🏽😀🏽½#$%'sع'Då", "tokens": 79, "pieces": ["३", "ſ", "<|", "endoftext", "|>", "Ⅳ", "BB", "!", " \n", "\t", "'D", "m", "…", "🙂-/\r\n", "Ⅳ", "aB", " '", "re", "👍🏽", "camelCase", "😀🏽", "ḋ", "̣aBe", "́ßEOT", "👍🏽😀🏽<", "EOT", ">", "½", "#$%'", "sع", "'D", "a", "̊"]} +{"text": "9< \n!m12345678a\r\n٣٤٥٦iOS \n
عé'ſ­,ꟲcamelCaseA३.\"'Re!\u000b,ᵃ٣٤٥٦'ſ'M\t", "tokens": 64, "pieces": ["9", "<", " \n", "!m", "123", "456", "78", "a", "\r\n", "٣٤٥", "٦", "iOS", "", " \n", "
عe", "́'", "ſ", "­,", "ꟲcamelCaseA", "३", ".\"'", "Re", "!", "\u000b", ",ᵃ", "٣٤٥", "٦", "'ſ", "'M", "\t"]} +{"text": "m\n…½\r३ Džungla!/ ع,<> 'S<|fim_prefix|>aB'Ma/bsaBİ.iOS", "tokens": 37, "pieces": ["m", "\n", "…", "½", "\r", "३", " Džungla", "!/", " ع", ",<>", " '", "S", "<|", "fim", "_prefix", "|>", "aB", "'M", "a", "/bsaBİ", ".iOS"]} +{"text": ">>å'sZ(", "tokens": 8, "pieces": [">>", "a", "̊'", "sZ", "("]} +{"text": "0㋿🙂!!- \naBa/b \n/,
9ꟲ-a<|fim_prefix|>Dž 9fiAiOS…'VE'D<|fim_prefix|>'T12345678 \n \n é12345678t\r ", "tokens": 62, "pieces": ["0", "㋿🙂!!-", " \n", "aBa", "/b", " \n", "/,", "
", "9", "ꟲ", "-a", "<|", "fim", "_prefix", "|>", "Dž", " ", "9", "fiAiOS", "…", "'VE", "'D", "<|", "fim", "_prefix", "|>'", "T", "123", "456", "78", " \n \n", " é", "123", "456", "78", "t", "\r "]} +{"text": "'S \n عd'VE३(0-㍿", "tokens": 14, "pieces": ["'S", " \n", " عd", "'VE", "३", "(", "0", "-㍿"]} +{"text": "('Re's👍🏽12345678'VE‍EOTع.12345678!!漢 \n'ſA漢,a/b\r\n\r\n 
…'S\"A\r\n…ᵃsEOTA'", "tokens": 62, "pieces": ["('", "Re", "'s", "👍🏽", "123", "456", "78", "'VE", "‍EOTع", ".", "123", "456", "78", "!!", "漢", " \n", "'ſ", "A漢", ",a", "/b", "\r\n\r\n", " ", "
", "", "…", "'S", "\"A", "\r\n", "…ᵃsEOTA", "'"]} +{"text": "­ e­'s\r\n\r\n", "tokens": 7, "pieces": ["­", " ", " e", "­'", "s", "\r\n\r\n"]} +{"text": "!\r\n\r\nåßDžunglaEOT漢HTTPServer漢camelCase<|endoftext|>\t \n ABC\r\n🙂𐞁\u000b<|fim_prefix|>\"'T'Re'sta 'DaBBé(‍\t👍🏽", "tokens": 70, "pieces": ["!\r\n\r\n", "a", "̊ßDžunglaEOT漢HTTPServer漢camelCase", "<|", "endoftext", "|>", "\t \n", " ABC", "\r\n", "🙂𐞁", "\u000b", "<|", "fim", "_prefix", "|>\"'", "T", "'Re", "'s", "ta", " ", "'", "DaBB", "e", "́(‍", "\t", "👍🏽"]} +{"text": " \n \r\n\r\n", "tokens": 2, "pieces": [" \n \r\n\r\n"]} +{"text": "٣٤٥٦aB́B\n/'ſ'sa\n/👍🏽9é🙂e!!𐞁३HTTPServer,A", "tokens": 48, "pieces": ["٣٤٥", "٦", "aB", "́B", "\n", "/'", "ſ", "'s", "a", "\n", "/👍🏽", "9", "e", "́🙂", "e", "!!", "𐞁", "३", "HTTPServer", ",A"]} +{"text": "ع<|endoftext|>­iOS漢", "tokens": 12, "pieces": ["ع", "<|", "endoftext", "|>­", "iOS漢"]} +{"text": "'VE \n 'S
s\n/'T㍿.aB9camelCase12345678A", "tokens": 25, "pieces": ["'VE", " \n", " ", "'S", "
s", "\n", "/'", "T", "㍿.", "aB", "9", "camelCase", "123", "456", "78", "A"]} +{"text": "ᵃİ,㍿e9!!㍿\r!!", "tokens": 19, "pieces": ["ᵃİ", ",㍿", "e", "9", "!!㍿\r", "!!"]} +{"text": "e<|endoftext|>t12345678ꟲé㍿fi'MA\na/b 'ABC(́å", "tokens": 36, "pieces": ["e", "<|", "endoftext", "|>", "t", "123", "456", "78", "ꟲe", "́㍿", "fi", "'M", "A", "\n", "a", "/b", " ", "'ABC", "(́", "a", "̊"]} +{"text": "عHTTPServer\n/", "tokens": 5, "pieces": ["عHTTPServer", "\n", "/"]} +{"text": "ⅣiOS㋿'sABC'M<|endoftext|>.iOS >sßa/bAb\rḍ̇३!!­٣٤٥٦'DZ<𐞁/\r\n(#$%,½<'re𐞁", "tokens": 63, "pieces": ["Ⅳ", "iOS", "㋿'", "sABC", "'M", "<|", "endoftext", "|>.", "iOS", " >", "sßa", "/bAb", "\r", "ḋ", "̣", "३", "!!­", "٣٤٥", "٦", "'D", "Z", "<𐞁", "/\r\n", "(#$%,", "½", "<'", "re𐞁"]} +{"text": "-字ḍ̇s-aABCDžaB12345678d<|endoftext|>HTTPServerAḍ̇\r\nZ0\nB\"ᵃeé\"", "tokens": 46, "pieces": ["-字ḋ", "̣s", "-aABCDžaB", "123", "456", "78", "d", "<|", "endoftext", "|>", "HTTPServerAḋ", "̣\r\n", "Z", "0", "\n", "B", "\"ᵃeé", "\""]} +{"text": "édḍ̇ABC\u000bḍ̇ᵃ9<|fim_prefix|>\r㍿'reHTTPServerét\"İå🙂d ßaB㋿‍٣٤٥٦.aB'D
\r\n\r\n", "tokens": 67, "pieces": ["e", "́dḋ", "̣ABC", "\u000bḋ", "̣ᵃ", "9", "<|", "fim", "_prefix", "|>\r", "㍿'", "reHTTPServerét", "\"İa", "̊🙂", "d", " ", " ßaB", "㋿‍", "٣٤٥", "٦", ".aB", "'D", "
\r\n\r\n"]} +{"text": "🙂㋿Zꟲ½,‍", "tokens": 13, "pieces": ["🙂㋿", "Zꟲ", "½", ",‍"]} +{"text": " \n
३'ſꟲ", "tokens": 11, "pieces": [" \n", "
", "३", "'ſ", "ꟲ"]} +{"text": " 12345678Z\tİ㍿\n/'VE#$%…!.'Džunglas\r\n\r\n-s'Re‍'MHTTPServerfi<0عZ३㋿EOTA𐞁", "tokens": 61, "pieces": [" ", "123", "456", "78", "Z", "\tİ", "㍿\n", "/'", "VE", "#$%", "…", "!.'", "Džunglas", "<", "META", "_START", ">\r\n\r\n", "-s", "'", "Re", "‍'", "MHTTPServerfi", "<", "0", "عZ", "३", "㋿EOTA𐞁"]} +{"text": "'reaBſ<|endoftext|>😀🏽0tiOSAſHTTPServer", "tokens": 30, "pieces": ["'re", "aBſ", "<|", "endoftext", "|>😀🏽", "0", "tiOS", "AſHTTPServer"]} +{"text": "a/३camelCaseAZ'VE", "tokens": 11, "pieces": ["a", "/", "३", "camelCaseAZ", "'VE"]} +{"text": "🙂<|endoftext|>EOT", "tokens": 11, "pieces": ["🙂<|", "endoftext", "|>", "EOT"]} +{"text": "a're𐞁'llEOTAd​Dž'T<|fim_prefix|>/\r\n … /\r\n'Abꟲ३e#$%'Rea/b-", "tokens": 47, "pieces": ["a", "'re", "𐞁", "'ll", "EOTAd", "​Dž", "'T", "<|", "fim", "_prefix", "|>/\r\n", " ", "…", "", " ", "/\r\n", "'Abꟲ", "३", "e", "#$%'", "Rea", "/b", "-"]} +{"text": "İ\t🙂𐞁#$%dm‍m,Ab漢\r\n\t👍🏽 \n 'sm", "tokens": 29, "pieces": ["İ", "\t", "🙂𐞁", "#$%", "dm", "‍m", ",Ab漢", "\r\n", "\t", "👍🏽", " \n", " '", "sm"]} +{"text": "\" \n 're\t-'D\nd \n 'SᵃZ\r\n\r\nع'ſ \n😀🏽'Då\n/ 'T''D", "tokens": 37, "pieces": ["\"", " \n", " '", "re", "\t", "-'", "D", "\n", "d", " \n", " '", "SᵃZ", "\r\n\r\n", "ع", "'ſ", " \n", "😀🏽'", "Da", "̊\n", "/", " ", "'T", "''", "D"]} +{"text": "!!'ſ…\n#$%…\tİ𐞁½( ⅣHTTPServer12345678iOSZ \r\n\r\nt!!Ⅳ
9\n're", "tokens": 40, "pieces": ["!!'", "ſ", "…\n", "#$%", "…", "\tİ𐞁", "½", "(", " ", "Ⅳ", "HTTPServer", "123", "456", "78", "iOSZ", " \r\n\r\n", "t", "!!", "Ⅳ", "
", "9", "\n", "'re"]} +{"text": "\tcamelCase'S/\r\nABC'ReiOS㋿'Res \r\n12345678…\r \n  -iOS漢🙂٣٤٥٦é…㍿ \r\n\r\nA's!!-camelCase\rABC<|fim_prefix|>>/ḍ̇!!", "tokens": 69, "pieces": ["\tcamelCase", "'S", "/\r\n", "ABC", "'Re", "iOS", "㋿'", "Res", " \r\n", "123", "456", "78", "…\r \n", " ", " ", "-iOS漢", "🙂", "٣٤٥", "٦", "é", "…", "㍿", " \r\n\r\n", "A", "'s", "!!-", "camelCase", "\r", "ABC", "<|", "fim", "_prefix", "|>>/", "ḋ", "̣!!"]} +{"text": "'T å'VE \n \n 'T\nꟲſ ßé'M‍\rAb😀🏽/\r\nA३/\r\nع !!é㍿", "tokens": 44, "pieces": ["'T", " a", "̊'", "VE", " \n \n", " '", "T", "\n", "ꟲſ", " ", " ßé", "'M", "‍\r", "Ab", "😀🏽/\r\n", "A", "३", "/\r\n", "ع", " ", "!!", "e", "́㍿"]} +{"text": "s㍿>٣٤٥٦12345678\r\n\r\n ​\r\n's字s字😀🏽's𐞁Džungla㋿ABC'ſDžungla🙂t'VE\n/eDž \n'SeiOS<字'SAbع", "tokens": 73, "pieces": ["s", "㍿>", "٣٤٥", "٦12", "345", "678", "\r\n\r\n", " ​\r\n", "'s", "字s字", "😀🏽'", "s𐞁Džungla", "㋿ABC", "'ſ", "Džungla", "🙂t", "'VE", "\n", "/eDž", " \n", "'S", "eiOS", "<<", "META", "_START", ">字", "'S", "Abع"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "‍! \n're​.", "tokens": 7, "pieces": ["‍!", " \n", "'re", "​."]} +{"text": "㍿'TiOSiOS!'re'S >ſ ꟲ'T٣٤٥٦'Reſ漢'ReᵃiOS/\r\n-\r\n‍ع٣٤٥٦́0😀🏽ABC \n/\r\n\r\n", "tokens": 60, "pieces": ["㍿'", "TiOSiOS", "!'", "re", "'S", " ", ">ſ", " ꟲ", "'T", "٣٤٥", "٦", "'Re", "ſ漢", "'Re", "ᵃiOS", "/\r\n", "-\r\n", "‍ع", "٣٤٥", "٦", "́", "0", "😀🏽", "ABC", " \n", "/\r\n\r\n"]} +{"text": "\"!\"½字½'reZéfiDžungla", "tokens": 14, "pieces": ["\"!\"", "½", "字", "½", "'re", "ZéfiDžungla"]} +{"text": "afi\t#$%<|endoftext|>Ab/\r\n EOT \n ꟲ'll\t👍🏽…aBaB'Ret㍿#$%ḍ̇", "tokens": 46, "pieces": ["afi", "\t", "#$%<|", "endoftext", "|>", "Ab", "/\r\n", " EOT", " \n", " ꟲ", "'ll", "\t", "👍🏽", "…aBaB", "'Re", "t", "㍿#$%", "ḋ", "̣"]} +{"text": "é-\r\n\r\n\u000b字('re'D \n ٣٤٥٦३\r\n\r\n!عB!iOSé-edABC\n/ \ncamelCase.漢mcamelCasedé", "tokens": 47, "pieces": ["é", "-\r\n\r\n", "\u000b字", "('", "re", "'D", " \n", " ", "٣٤٥", "٦", "", "३", "\r\n\r\n", "!عB", "!iOSé", "-edABC", "\n", "/", " \n", "camelCase", ".漢mcamelCasede", "́"]} +{"text": "́sDžunglaat \n 漢DžHTTPServer㋿ \n ABCꟲm", "tokens": 24, "pieces": ["́sDžunglaat", " \n", " 漢DžHTTPServer", "㋿", " \n", " ABCꟲm"]} +{"text": "½'M!!!", "tokens": 3, "pieces": ["½", "'M", "!!!"]} +{"text": "👍🏽 'll<|endoftext|>'D 𐞁0Dž­ſßZiOS👍🏽\"३å.ḍ̇­ ­-\r \ns'M", "tokens": 59, "pieces": ["👍🏽", " ", "'ll", "<|", "endoftext", "|>'", "D", " <", "EOT", ">𐞁", "0", "Dž", "­ſßZiOS", "👍🏽\"", "३", "a", "̊.", "ḋ", "̣­", " ", " ­-\r", " \n", "s", "'M"]} +{"text": "<|fim_prefix|>>字'M9a/béiOS12345678B ㋿å'll'Mt!!ḍ̇ḍ̇́<|endoftext|>!", "tokens": 50, "pieces": ["<|", "fim", "_prefix", "|>>", "字", "'M", "9", "a", "/be", "́iOS", "123", "456", "78", "B", " ", " ㋿", "a", "̊'", "ll", "'M", "t", "!!", "ḋ", "̣ḋ", "̣́<|", "endoftext", "|>!"]} +{"text": "ع\u000b/ \n #$%", "tokens": 10, "pieces": ["ع", "\u000b", "/", " \n", " <", "META", "_START", ">#$%"]} +{"text": "'réécamelCase/\r\n\"/\r\na/b'reABCfi<ع#$%aB\t", "tokens": 21, "pieces": ["'re", "́écamelCase", "/\r\n", "\"/\r\n", "a", "/b", "'re", "ABCfi", "<ع", "#$%", "aB", "\t"]} +{"text": "٣٤٥٦<|fim_prefix|>٣٤٥٦iOSḍ̇Dž'T\n/…Džungla…é", "tokens": 45, "pieces": ["٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "iOSḋ", "̣Dž", "'T", "\n", "/", "…Džungla", "…é"]} +{"text": "ABCcamelCase'DcamelCase 😀🏽字12345678ꟲ🙂 漢", "tokens": 23, "pieces": ["ABCcamelCase", "'D", "camelCase", " ", " 😀🏽", "字", "123", "456", "78", "ꟲ", "🙂", " 漢"]} +{"text": "ſ漢/iOS٣٤٥٦>\r㋿Z12345678'MDžungla\t'TDžungla\"'VEⅣEOTZ\n/½ta/b\r\n a'D… <|endoftext|>é'St\u000b", "tokens": 69, "pieces": ["ſ漢", "/iOS", "٣٤٥", "٦", ">\r", "㋿Z", "123", "456", "78", "'M", "Džungla", "\t", "'T", "Džungla", "\"'", "VE", "Ⅳ", "EOTZ", "\n", "/", "½", "ta", "/b", "\r\n", " a", "'D", "…", " ", "<|", "endoftext", "|>", "e", "́'", "S", "t", "\u000b"]} +{"text": "\"-", "tokens": 1, "pieces": ["\"-"]} +{"text": "/", "tokens": 1, "pieces": ["/"]} +{"text": "e's 'll'ſZ\u000bDž\u000b<|endoftext|>'M \n 's/\r\nås'VEm,\r\n\r\nåḍ̇/😀🏽<|endoftext|>,s", "tokens": 58, "pieces": ["e", "'s", " ", "'ll", "'ſ", "Z", "\u000bDž", "\u000b", "<|", "endoftext", "|>'", "M", " \n", " '", "s", "/\r\n", "a", "̊s", "'VE", "m", ",\r\n\r\n", "a", "̊<", "EOT", ">ḋ", "̣/😀🏽<|", "endoftext", "|>,", "s"]} +{"text": "😀🏽", "tokens": 5, "pieces": ["😀🏽"]} +{"text": "/Ⅳḍ̇(ABC \r.­½​iOS…\r\n\r\n\"\n <12345678ß.HTTPServerZ ᵃ-'S\ta#$%", "tokens": 43, "pieces": ["/", "Ⅳ", "ḋ", "̣(", "ABC", " \r", ".­", "½", "​iOS", "…\r\n\r\n", "\"\n", " ", " <", "123", "456", "78", "ß", ".HTTPServerZ", " ᵃ", "-<", "META", "_START", ">'", "S", "\ta", "#$%"]} +{"text": "aⅣİAb", "tokens": 5, "pieces": ["a", "Ⅳ", "İAb"]} +{"text": " (t'M​ꟲ \n(́ǻEOT \n e'sEOT \n å<|fim_prefix|>camelCase ㍿\t\r\n\r\n\r\n", "tokens": 41, "pieces": [" ", "(t", "'M", "​ꟲ", " \n", "(́", "a", "̊́", "EOT", " \n", " e", "'s", "EOT", " \n", " a", "̊<|", "fim", "_prefix", "|>", "camelCase", " ", "㍿", "\t\r\n\r\n\r\n"]} +{"text": "ⅣEOTt\t/\r\n­'VE𐞁\t<३-,'s\n/㋿'VE'ſ\n<|fim_prefix|>9३'Dع'M 'S
Ⅳ0aB३ABC\r\n​s🙂", "tokens": 69, "pieces": ["Ⅳ", "EOTt", "\t", "/\r\n", "­'", "VE𐞁", "\t", "<", "३", "-,'", "s", "\n", "/㋿'", "VE", "'ſ", "\n", "<|", "fim", "_prefix", "|>", "9३", "'D", "ع", "'M", " ", "'", "S", "", "
", "Ⅳ0", "aB", "३", "ABC", "\r\n", "​s", "🙂"]} +{"text": "å­𐞁<|fim_prefix|>'M٣٤٥٦ddsdcamelCase,iOSꟲé…\n/㋿écamelCase'Re'T'", "tokens": 48, "pieces": ["a", "̊­", "𐞁", "<|", "fim", "_prefix", "|>'", "M", "٣٤٥", "٦", "ddsdcamelCase", ",iOSꟲe", "́", "…\n", "/㋿", "écamelCase", "'Re", "'T", "'"]} +{"text": "‍ſ'reé\u000bḍ̇e🙂…ſ're….㋿é'T𐞁‍#$%٣٤٥٦ 
-é", "tokens": 55, "pieces": ["‍ſ", "'re", "e", "́", "\u000bḋ", "̣e", "🙂", "…ſ", "'re", "…", ".㋿", "é", "'T", "𐞁", "‍#$%", "٣٤٥", "٦", " ", "
", "-é"]} +{"text": "<|fim_prefix|>漢-\r\nZ㍿!!'ſDžungla\rDžungla'Mt…Aḍ̇​iOS
é'T𐞁(DžéeABCé", "tokens": 56, "pieces": ["<|", "fim", "_prefix", "|>", "漢", "-\r\n", "Z", "㍿!!'", "ſDžungla", "\r", "Džungla", "'M", "t", "…Aḋ", "̣​", "iOS", "
é", "'T", "𐞁", "(Dže", "́eABCé"]} +{"text": ".漢", "tokens": 3, "pieces": [".漢"]} +{"text": "ᵃm", "tokens": 4, "pieces": ["ᵃm"]} +{"text": "́😀🏽DžHTTPServer.\n/İ< /\r\nḍ̇/\r\n're🙂'T'ſⅣ字a/b\u000b", "tokens": 36, "pieces": ["́😀🏽", "DžHTTPServer", ".\n", "/İ", "<", " /\r\n", "ḋ", "̣/\r\n", "'re", "🙂'", "T", "'ſ", "Ⅳ", "字a", "/b", "\u000b"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿½٣٤٥٦a/bå'Reå'T", "tokens": 24, "pieces": ["㍿", "½٣٤", "٥٦", "a", "/ba", "̊'", "Rea", "̊'", "T"]} +{"text": "ꟲ,
 \nmß漢\u000b\na/bB́a/b/\r\n\r\nİfiİ", "tokens": 24, "pieces": ["ꟲ", ",", "
 \n", "mß漢", "\u000b\n", "a", "/bB", "́a", "/b", "/\r\n\r\n", "İfiİ"]} +{"text": " \n ३'ſ \n", "tokens": 8, "pieces": [" \n", " ", "३", "'ſ", " \n"]} +{"text": "9\n/DžunglaHTTPServer/漢ᵃ'TiOSHTTPServer߅'S", "tokens": 27, "pieces": ["", "9", "\n", "/DžunglaHTTPServer", "/漢ᵃ", "'T", "iOSHTTPServerß", "…", "'S"]} +{"text": "fi's", "tokens": 3, "pieces": ["fi", "'s"]} +{"text": "\r'Re ́/'MaBB😀🏽#$%<|endoftext|>\"d🙂aB", "tokens": 26, "pieces": ["\r", "'Re", " ", "́/'", "MaBB", "😀🏽#$%<|", "endoftext", "|>\"", "d", "🙂aB"]} +{"text": "\tDžungla🙂漢字½ \n/\"'Mß\r\n\r\n\r\n/\r\n㋿å\"é.", "tokens": 28, "pieces": ["\tDžungla", "🙂漢字", "½", " \n", "/\"'", "Mß", "\r\n\r\n\r\n", "/\r\n", "㋿a", "̊\"", "e", "́."]} +{"text": "iOS㍿a", "tokens": 5, "pieces": ["iOS", "㍿a"]} +{"text": "ᵃB\"HTTPServerᵃ, \r\n\r\né<|endoftext|>İ­EOT\r\nåḍ̇'DⅣ,'Re…٣٤٥٦\"a/b(.'re-'D
'Tåd 😀🏽\r'", "tokens": 72, "pieces": ["ᵃB", "\"HTTPServerᵃ", ",", " \r\n\r\n", "e", "́<|", "endoftext", "|>", "İ", "­EOT", "\r\n", "a", "̊ḋ", "̣'", "D", "Ⅳ", ",'", "Re", "…", "٣٤٥", "٦", "\"a", "/b", "(.'", "re", "-'", "D", "
", "'T", "a", "̊d", " ", "😀🏽\r", "'"]} +{"text": "'re!!EOT'llAbs'VEå🙂…<',12345678a/b ​㍿\n/
 HTTPServer(\n'M​", "tokens": 36, "pieces": ["'re", "!!", "EOT", "'ll", "Abs", "'VE", "a", "̊🙂", "…", "<',", "123", "456", "78", "a", "/b", " ​㍿\n", "/", "
", " HTTPServer", "(\n", "'M", "​"]} +{"text": "eEOTfi' \n\" 'M", "tokens": 10, "pieces": ["eEOTfi", "'", " \n", "\"", " ", "'M"]} +{"text": "‍'SAbt㋿ !<㍿HTTPServer 'T/éåA/\r\n㍿३/\r\nع>!!<|fim_prefix|>\rt'TaB.9", "tokens": 55, "pieces": ["‍'", "SAbt", "㋿", " <", "META", "_START", ">!<㍿", "HTTPServer", "", " ", "'T", "/e", "́a", "̊A", "/\r\n", "㍿", "३", "/\r\n", "ع", ">!!<|", "fim", "_prefix", "|>\r", "t", "'T", "aB", ".", "9"]} +{"text": "ḍ̇३9HTTPServer ,camelCaseåḍ̇İ\"'ſDžungla12345678 \n <|endoftext|><|fim_prefix|>éſ,Ⅳ𐞁åHTTPServerꟲ½fi'Reع #$%å ", "tokens": 77, "pieces": ["ḋ", "̣", "३9", "HTTPServer", " ", ",camelCasea", "̊ḋ", "̣İ", "\"'", "ſDžungla", "123", "456", "78", " \n", " <|", "endoftext", "|><|", "fim", "_prefix", "|>", "e", "́ſ", ",", "Ⅳ", "𐞁a", "̊HTTPServerꟲ", "½", "fi", "'Re", "ع", " ", " #$%", "a", "̊", " "]} +{"text": "< \n㍿'M😀🏽<㍿'Re\n/ḍ̇(\r\n\r\n<|endoftext|>ſ \n,'ſ", "tokens": 43, "pieces": ["<", " \n", "㍿<", "EOT", ">'", "M", "😀🏽<㍿'", "Re", "\n", "/ḋ", "̣(\r\n\r\n", "<|", "endoftext", "|>", "ſ", " \n", ",'", "ſ"]} +{"text": "ᵃ/ \né'll­iOSḍ̇HTTPServera½字漢👍🏽iOSé's<|fim_prefix|>iOS(…éHTTPServer‍!!0,ſa/b/\r\n
", "tokens": 60, "pieces": ["ᵃ", "/", " \n", "é", "'ll", "­iOS", "ḋ", "̣HTTPServera", "½", "字漢", "👍🏽", "iOSé", "'s", "<|", "fim", "_prefix", "|>", "iOS", "(", "…éHTTPServer", "‍!!", "0", ",ſa", "/b", "/\r\n", "
"]} +{"text": "­Ab㍿'ll字㋿>'re\t'reZ'漢'T\r\n\r\nfi'S…\n/'VE<|fim_prefix|>9s'Sé٣٤٥٦", "tokens": 49, "pieces": ["­Ab", "㍿'", "ll字", "㋿>'", "re", "\t", "'re", "Z", "'漢", "'T", "\r\n\r\n", "fi", "'S", "…\n", "/'", "VE", "<|", "fim", "_prefix", "|>", "9", "s", "'S", "e", "́", "٣٤٥", "٦"]} +{"text": "é'ſZ -ß,\r\n\r\ncamelCase/\r\n\" !!𐞁\r\n\u000b\r\nZ \n<|endoftext|>#$%٣٤٥٦​𐞁 \n", "tokens": 47, "pieces": ["é", "'ſ", "Z", " ", "-ß", ",\r\n\r\n", "camelCase", "/\r\n", "\"", " ", " !!", "𐞁", "\r\n\u000b\r\n", "Z", " \n", "<|", "endoftext", "|>#$%", "٣٤٥", "٦", "​𐞁", " \n"]} +{"text": "👍🏽iOS'SABC", "tokens": 9, "pieces": ["👍🏽", "iOS", "'S", "ABC"]} +{"text": "<漢ḍ̇/\r\n㋿fi字\u000b-", "tokens": 17, "pieces": ["<漢ḋ", "̣/\r\n", "㋿fi字", "\u000b", "-"]} +{"text": " \nåḍ̇ſ\"", "tokens": 16, "pieces": [" \n", "/,'", "S", "!<", "EOT", ">ḋ", "̣ſ", "\""]} +{"text": "d\u000b字ABC٣٤٥٦👍🏽", "tokens": 18, "pieces": ["d", "\u000b字ABC", "٣٤٥", "٦", "👍🏽"]} +{"text": "㋿\t", "tokens": 4, "pieces": ["㋿", "\t"]} +{"text": "ᵃ", "tokens": 6, "pieces": ["ᵃ"]} +{"text": "字'll\r\n /d \nDž Džungla.12345678​ !!Dž'VEḍ̇/\r\n­'ſAåfi'VE", "tokens": 52, "pieces": ["字", "'ll", "\r\n", " /", "d", " \n", "Dž", " Džungla", ".", "123", "456", "78", "​", " <", "EOT", ">!!", "Dž", "'", "VEḋ", "̣/\r\n", "­'", "ſAa", "̊fi", "'VE", ""]} +{"text": "'Må‍३'VE<|fim_prefix|>
Z'é<\r\n\r\n'S!fié‍", "tokens": 30, "pieces": ["'M", "a", "̊‍", "३", "'VE", "<|", "fim", "_prefix", "|>", "
Z", "'é", "<\r\n\r\n", "'S", "!fié", "‍"]} +{"text": "å /\r\n/!Ⅳ \n !٣٤٥٦", "tokens": 19, "pieces": ["a", "̊", " /\r\n", "/!", "Ⅳ", " \n", " !", "٣٤٥", "٦"]} +{"text": "'T 漢 \n 'Dḍ̇mDž<|fim_prefix|> ​
…ſ<३  \u000b>Z'D 👍🏽ea­iOS0漢Džungla 'ſⅣZ.", "tokens": 63, "pieces": ["'T", " ", " 漢", " \n", " '", "Dḋ", "̣mDž", "<|", "fim", "_prefix", "|>", " ", " ​", "
", "…ſ", "<", "३", "  ", "\u000b", ">Z", "'D", " ", "👍🏽", "ea", "­iOS", "0", "漢Džungla", " '", "ſ", "Ⅳ", "Z", "."]} +{"text": "HTTPServerß>camelCase12345678e İDžungla'Re\t", "tokens": 20, "pieces": ["HTTPServerß", ">", "camelCase", "123", "456", "78", "e", " İDžungla", "'Re", "\t"]} +{"text": ",\r\n\r\n<|endoftext|>å ABC字'S \n'VE㍿Dž(HTTPServer \n'S🙂é​'S'ſeaé /\r\n \r\n'll字", "tokens": 52, "pieces": [",\r\n\r\n", "<|", "endoftext", "|>", "a", "̊", " ", " ABC字", "'", "S", " \n", "'VE", "㍿Dž", "(HTTPServer", " \n", "'S", "🙂é", "​'", "S", "'ſ", "eaé", " ", " /\r\n", " ", " <", "META", "_START", ">\r\n", "'ll", "字"]} +{"text": "㍿iOSEOT<|fim_prefix|>a/b'T/\r\nABC'll'll'VEt'Sſet'SAb0-Džunglaſ'ſ'S\n字字>\r\ndé🙂å㋿👍🏽𐞁0Dž", "tokens": 71, "pieces": ["㍿iOSEOT", "<|", "fim", "_prefix", "|>", "a", "/b", "'T", "/\r\n", "ABC", "'", "ll", "'ll", "'VE", "t", "'S", "ſet", "'S", "Ab", "0", "-Džunglaſ", "'ſ", "'S", "\n", "字字", ">\r\n", "dé", "🙂a", "̊㋿👍🏽", "𐞁", "0", "Dž"]} +{"text": "'M½'ſ'll
\u000b iOSDž!!\t/\r\n字Z㍿ḍ̇<|fim_prefix|>ع㍿!", "tokens": 37, "pieces": ["'M", "½", "'ſ", "'ll", "
\u000b", " iOSDž", "!!", "\t", "/\r\n", "字Z", "㍿ḋ", "̣<|", "fim", "_prefix", "|>", "ع", "㍿!"]} +{"text": ">😀🏽ᵃ<|fim_prefix|>ꟲABC/\r\n \n ½\r\n> \n\r🙂's३ \n 漢ꟲ\n/'ſ(\niOS\r\n\t, \n\r\n", "tokens": 53, "pieces": [">😀🏽", "ᵃ", "<|", "fim", "_prefix", "|>", "ꟲABC", "/\r\n", " \n", " ", " ", "½", "\r\n", ">", " \n\r", "🙂'", "s", "३", " \n", " 漢ꟲ", "\n", "/'", "ſ", "(\n", "iOS", "\r\n", "\t", ",", " \n\r\n"]} +{"text": "㍿'S\n/-㋿
fi‍e'ſ'se‍é", "tokens": 29, "pieces": ["㍿'", "S", "\n", "/-㋿", "
fi", "‍e", "'ſ", "'", "se", "‍e", "́"]} +{"text": "🙂Džungla \n éABCABC \n/\r\n漢'T㋿👍🏽Ab😀🏽३\r\n'll\"'ſBaB<㋿㋿å\nd-iOS#$%漢aBHTTPServer", "tokens": 61, "pieces": ["🙂Džungla", " \n", " éABCABC", " \n", "/\r\n", "漢", "'T", "㋿👍🏽", "Ab", "😀🏽", "३", "\r\n", "'ll", "\"'", "ſBaB", "<㋿㋿", "a", "̊\n", "d", "-iOS", "#$%", "漢aBHTTPServer"]} +{"text": "/\r\n \n'VE/ \ne!\n/½'ſ/\r\n \n fi٣٤٥٦12345678
9-\ncamelCase\n/'sA\n!", "tokens": 21, "pieces": ["e", "'M", "", "123", "456", "78", "
", "9", "-\n", "camelCase", "\n", "/'", "sA", "\n", "!"]} +{"text": "ꟲEOT👍🏽字>dABC\t字's<'ſfi'S­㋿'Mß'D \r", "tokens": 33, "pieces": ["ꟲEOT", "👍🏽", "字", ">dABC", "\t字", "'s", "<'", "ſfi", "'S", "­㋿'", "Mß", "'D", " \r"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " 'S'Re(", "tokens": 4, "pieces": [" ", "'S", "'Re", "("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "​ \n \n'VE Ⅳ(a/bEOT<|endoftext|>!!#$%𐞁🙂३\r字/\r\n\"ß🙂'sḍ̇́", "tokens": 56, "pieces": ["​", " \n \n", "'VE", " ", " ", "Ⅳ", "(a", "/bEOT", "<|", "endoftext", "|>!!<", "META", "_START", ">#$%", "𐞁", "🙂", "३", "\r", "字", "/\r\n", "\"ß", "🙂'", "sḋ", "̣́"]} +{"text": "㋿å's", "tokens": 8, "pieces": ["㋿a", "̊'", "s"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "'ś३'re\r\r\n\r\n㋿.ꟲé㍿'M \n ſ<|endoftext|>å9Z0Z㍿>ABCAbⅣ'SaB'll'reB", "tokens": 52, "pieces": ["'s", "́", "३", "'re", "\r\r\n\r\n", "㋿.", "ꟲé", "㍿'", "M", " \n", " ſ", "<|", "endoftext", "|>", "a", "̊", "9", "Z", "0", "Z", "㍿>", "ABCAb", "Ⅳ", "'S", "aB", "'ll", "'re", "B"]} +{"text": "ABCⅣ-ᵃ \n 'THTTPServeréHTTPServer٣٤٥٦ \n<字'ſfi m𐞁é'llå1234567812345678", "tokens": 49, "pieces": ["ABC", "Ⅳ", "-ᵃ", " \n", " '", "THTTPServeréHTTPServer", "٣٤٥", "٦", " \n", "<字", "'ſ", "fi", " ", " m𐞁e", "́'", "lla", "̊", "123", "456", "781", "234", "567", "8"]} +{"text": "#$%mİع㍿a/bfi𐞁fi३\r\n\r\n>'llHTTPServer'ſé<|fim_prefix|>…ſeZt0!! 漢\u000b\r\n½ \n.İ㋿fi漢 ", "tokens": 71, "pieces": ["#$%", "m", "İع", "㍿a", "/bfi𐞁fi", "३", "\r\n\r\n", ">'", "llHTTPServer", "'ſ", "e", "́<|", "fim", "_prefix", "|>", "…ſeZt", "0", "!!", " ", " 漢", "\u000b\r\n", "½", "", " \n", ".İ", "㋿fi漢", " "]} +{"text": " (.", "tokens": 2, "pieces": [" ", "(."]} +{"text": "㋿'ll…३😀🏽ABCEOT½'Re,'M'VE'reZAb 'ſ'VE('ll.ꟲ\t/'Dé-é\"!", "tokens": 49, "pieces": ["㋿'", "ll", "…", "३", "😀🏽", "ABCEOT", "½", "'Re", ",'", "M", "'VE", "'re", "ZAb", " ", "'ſ", "'VE", "('", "ll", ".ꟲ", "\t", "/'", "De", "́<", "EOT", ">-", "é", "\"!"]} +{"text": "'VEDžunglaA'ſ'S'Meé/\r\n.<|endoftext|>Ⅳ>As٣٤٥٦12345678½  'ſå12345678' sſ", "tokens": 60, "pieces": ["'VE", "DžunglaA", "'ſ", "'S", "'M", "ee", "́/\r\n", ".<", "META", "_START", "><|", "endoftext", "|>", "Ⅳ", ">As", "٣٤٥", "٦12", "345", "678", "½", " ", " ", "'ſ", "a", "̊", "123", "456", "78", "'", " ", " sſ"]} +{"text": "'s.'ll🙂ſ/!aB", "tokens": 11, "pieces": ["'s", ".'", "ll", "🙂ſ", "/!", "aB"]} +{"text": ">(𐞁EOT>aa ! !A漢9å \n 'Dᵃå \n ", "tokens": 37, "pieces": [">(", "𐞁EOT", ">aa", " <", "META", "_START", ">!", " !", "A漢", "9", "a", "̊", " \n", " '", "Dᵃa", "̊", " \n", " <", "META", "_START", ">"]} +{"text": "aiOS \n ‍‍DžacamelCaseecamelCase
camelCaseḍ̇㋿​a/b\rt \n  camelCase‍t<|fim_prefix|>", "tokens": 47, "pieces": ["aiOS", " \n", " ‍‍", "DžacamelCaseecamelCase", "
camelCaseḋ", "̣㋿​", "a", "/b", "\r", "t", " \n", " ", " camelCase", "‍t", "<|", "fim", "_prefix", "|>"]} +{"text": "́'sfiع<|fim_prefix|>(!!٣٤٥٦Ⅳ>ḍ̇Z", "tokens": 35, "pieces": ["́'", "sfiع", "<|", "fim", "_prefix", "|>(!!", "٣٤٥", "٦Ⅳ", ">ḋ", "̣Z"]} +{"text": "\r\n\r\néHTTPServer👍🏽\r\n\r\nacamelCase‍­,(0\rHTTPServer👍🏽\n/å0,ß \n Dž🙂s \n ㍿", "tokens": 48, "pieces": ["\r\n\r\n", "e", "́HTTPServer", "👍🏽\r\n\r\n", "acamelCase", "‍­,(", "0", "\r", "HTTPServer", "👍🏽\n", "/a", "̊", "0", ",ß", " \n", " Dž", "🙂s", " \n", " ㍿"]} +{"text": "­
m…'T٣٤٥٦Ⅳ!'S!", "tokens": 20, "pieces": ["­", "
m", "…", "'T", "٣٤٥", "٦Ⅳ", "!'", "S", "!"]} +{"text": "<|fim_prefix|>.\u000baB \n/Abß٣٤٥٦a/b
  ㍿å\tع>ß\r\niOS'll㍿EOT", "tokens": 46, "pieces": ["<|", "fim", "_prefix", "|>.", "\u000baB", " \n", "/Abß", "٣٤٥", "٦", "a", "/b", "
 ", " ", "㍿a", "̊", "\tع", ">ß", "\r\n", "iOS", "'ll", "㍿EOT"]} +{"text": "9.Z<|fim_prefix|>😀🏽ᵃt'D", "tokens": 23, "pieces": ["9", ".Z", "<|", "fim", "_prefix", "|>😀🏽", "ᵃt", "'D", ""]} +{"text": "'D३(字'S9½/\r\n
\ré'T", "tokens": 14, "pieces": ["'D", "३", "(字", "'S", "9½", "/\r\n", "
\r", "é", "'T"]} +{"text": "'VE-fiAb𐞁,😀🏽 \nHTTPServerİᵃ/(B", "tokens": 25, "pieces": ["'VE", "-fiAb𐞁", ",😀🏽", " \n", "HTTPServerİᵃ", "/(", "B"]} +{"text": "'Reé½('ABCa'D/\r\t👍🏽 \nİfi<​\">ßEOTꟲ9ßsd🙂/'re\r\n\r\n<12345678", "tokens": 46, "pieces": ["'Re", "e", "́", "½", "('", "ABCa", "'D", "/\r", "\t", "👍🏽", " \n", "İfi", "<​\">", "ßEOTꟲ", "9", "ßsd", "🙂/'", "re", "\r\n\r\n", "<", "123", "456", "78", ""]} +{"text": "é\r\r\n\r0'ſ'#$%<|fim_prefix|>a", "tokens": 19, "pieces": ["e", "́\r\r\n\r", "0", "'ſ", "'#$%<|", "fim", "_prefix", "|>", "a"]} +{"text": "/\r\n\r\nDžungla'M'ReAb\n/'ll㋿
३٣٤٥٦ 'SaBZé\tEOT12345678ع\nEOTſ#$%éiOS-tsa/bDž(㍿!!0", "tokens": 62, "pieces": ["/\r\n\r\n", "Džungla", "'M", "'Re", "Ab", "\n", "/'", "ll", "㋿", "
", "३٣٤", "٥٦", " ", "'S", "aBZe", "́", "\tEOT", "", "123", "456", "78", "ع", "\n", "EOTſ", "#$%", "e", "́iOS", "-tsa", "/bDž", "(㍿!!", "0"]} +{"text": "fi𐞁é‍t.", "tokens": 11, "pieces": ["fi𐞁é", "‍t", "."]} +{"text": "-…#$%.'<Džungla'StEOTEOT9'T½(\u000b!!\n/e#$%iOS \n", "tokens": 26, "pieces": ["-", "…", "#$%.'<", "Džungla", "'S", "tEOTEOT", "9", "'T", "½", "(", "\u000b", "!!\n", "/e", "#$%", "iOS", " \n"]} +{"text": "\r\n!!HTTPServer\t\"­sd\"½'VE'D're<|fim_prefix|>!!-ع", "tokens": 27, "pieces": ["\r\n", "!!", "HTTPServer", "\t", "\"­", "sd", "\"", "½", "'VE", "'D", "'re", "<|", "fim", "_prefix", "|>!!-", "ع"]} +{"text": "😀🏽​­'ſ's. \r\n\r\néßm 're", "tokens": 20, "pieces": ["😀🏽​­'", "ſ", "'s", ".", " \r\n\r\n", "e", "́ßm", " ", "'re"]} +{"text": "Z0‍e", "tokens": 12, "pieces": ["Z", "", "0", "‍", "e"]} +{"text": "\n𐞁ꟲ'sa/bA >a/be12345678 \nA\r<|endoftext|>a\tDž.…d㋿", "tokens": 45, "pieces": ["\n", "𐞁ꟲ", "'s", "a", "/bA", " >", "a", "/be", "123", "456", "78", " \n", "A", "\r", "<|", "endoftext", "|>", "a", "\tDž", ".", "…d", "㋿"]} +{"text": "🙂 \n iOS\"camelCaseA", "tokens": 9, "pieces": ["🙂", " \n", " iOS", "\"camelCaseA"]} +{"text": "camelCaseDž'Re \nAb字 a \n!!éZé!
 \n 'A,/😀🏽!! 's٣٤٥٦ſ/ ٣٤٥٦", "tokens": 59, "pieces": ["camelCase", "Dž", "'Re", " \n", "Ab字", " a", " \n", "!!", "e", "́Zé", "!", "
 \n", " '", "A", ",/😀🏽!!", " '", "s", "٣٤٥", "٦", "ſ", "/", " ", " ", "٣٤٥", "٦"]} +{"text": "/İ>‍漢 \nfi​ /\r\nß(‍👍🏽'DⅣعᵃ \n aDžungla  \n s", "tokens": 40, "pieces": ["/İ", ">‍", "漢", " \n", "fi", "​", " ", "/\r\n", "ß", "(‍👍🏽'", "D", "Ⅳ", "عᵃ", " \n", " aDžungla", "  \n", " s"]} +{"text": " /\r\nعHTTPServer३./\r\n9३'ſ 'VEAb<|fim_prefix|>Aé '३\r\n३㋿\nB9\r\na \n​\" ,عfiEOT-​", "tokens": 57, "pieces": [" ", "/\r\n", "عHTTPServer", "३", "./\r\n", "9३", "'ſ", " ", "'VE", "Ab", "<|", "fim", "_prefix", "|>", "Ae", "́", " ", " '", "३", "\r\n", "३", "㋿\n", "B", "9", "\r\n", "a", " \n", "​\"", " ", " ,", "عfiEOT", "-​"]} +{"text": "👍🏽\r\n٣٤٥٦'S<|endoftext|>fiABCᵃ𐞁,sa/b\u000b\r\n.Džéꟲᵃ🙂", "tokens": 53, "pieces": ["👍🏽\r\n", "٣٤٥", "٦", "'S", "<|", "endoftext", "|>", "fiABCᵃ𐞁", ",sa", "/b", "\u000b", "\r\n", ".Dže", "́ꟲᵃ", "🙂"]} +{"text": "ABC'ReaB \nfi !.'ll  -ꟲ\r\n\r\n
é\u000b//dZ👍🏽ḍ̇'Mßs ​t🙂 \na/b'SB-‍ ", "tokens": 54, "pieces": ["ABC", "'Re", "aB", " \n", "fi", " !.'", "ll", " ", " -", "ꟲ", "\r\n\r\n", "
é", "\u000b", "//", "dZ", "👍🏽", "ḋ", "̣'", "Mßs", " ", "​t", "🙂", " \n", "a", "/b", "'", "SB", "-‍", " "]} +{"text": "a/béZ\rDžunglaᵃ😀🏽 \n\u000b're\rABC", "tokens": 22, "pieces": ["a", "/be", "́Z", "\r", "Džunglaᵃ", "😀🏽", " \n", "\u000b", "'re", "\r", "ABC"]} +{"text": "\u000bé0㍿fiB'S\t­'ll'camelCase \n'M", "tokens": 19, "pieces": ["\u000bé", "0", "㍿fiB", "'S", "\t", "­'", "ll", "'camelCase", " \n", "'M"]} +{"text": "\n-", "tokens": 2, "pieces": ["\n", "-"]} +{"text": "字é😀🏽ABC'ſ \n \n…'T‍Dž‍a/b३m/!'T'T🙂HTTPServer½㋿", " \n", "…", "'T", "‍Dž", "‍a", "/b", "३", "m", "/!'", "T", "'T", "🙂HTTPServer", "½", "㋿<", "AbꟲiOS", "/\r\n", "e", "'re", " ", "㍿HTTPServers", "३", "/\r\n\r\n"]} +{"text": "<|endoftext|> 漢'T३/\r\nع('St字\r\n\r\nع㍿ꟲ٣٤٥٦é#$%漢'VEcamelCase", "tokens": 44, "pieces": ["<|", "endoftext", "|>", " 漢", "'T", "३", "/\r\n", "ع", "('", "St字", "\r\n\r\n", "ع", "㍿ꟲ", "٣٤٥", "٦", "e", "́#$%", "漢", "'VE", "camelCase"]} +{"text": " \n'Mfi字 ㍿'a/ba/ḍ̇camelCasem\r \n!!'ſ!eå\r\n('ſå\t\r\n\r\n'D👍🏽\r\n\r\n\r\n\r\nAb", "tokens": 53, "pieces": [" \n", "'M", "fi字", " ", "㍿'", "a", "/ba", "/ḋ", "̣camelCasem", "\r \n", "!!'", "ſ", "!ea", "̊\r\n", "('", "ſa", "̊", "\t\r\n\r\n", "'D", "👍🏽\r\n\r\n\r\n\r\n", "Ab"]} +{"text": "\r\n\r\nİZ🙂'İİa/b", "tokens": 12, "pieces": ["\r\n\r\n", "İZ", "🙂'", "İİ", "a", "/b"]} +{"text": "ß9.½'Re字//\r\n🙂a/baBعfi½camelCase/\r\n", "tokens": 23, "pieces": ["ß", "9", ".", "½", "'Re", "字", "//\r\n", "🙂a", "/baBعfi", "½", "camelCase", "/\r\n"]} +{"text": "\u000bZİ", "tokens": 5, "pieces": ["\u000b", "Zİ"]} +{"text": ">HTTPServerDžungla-HTTPServer aBAbm!!", "tokens": 15, "pieces": [">HTTPServerDžungla", "-HTTPServer", " aBAbm", "!!"]} +{"text": "iOS\"Bḍ̇ttᵃᵃ12345678ḍ̇9İ३🙂İ‍/.ſ'VE'ſ'SiOS'T'ꟲ'‍mᵃ. s", "tokens": 60, "pieces": ["iOS", "\"Bḋ", "̣ttᵃᵃ", "123", "456", "78", "ḋ", "̣", "9", "İ", "३", "🙂İ", "‍/.", "ſ", "'VE", "'ſ", "'S", "iOS", "'T", "'ꟲ", "'‍", "mᵃ", ".", " s"]} +{"text": "字<|fim_prefix|>", "tokens": 8, "pieces": ["字", "<|", "fim", "_prefix", "|>"]} +{"text": "🙂 're İ'S­ſ\r\r\n\r\nś'ſ‍0 \nſ'Re ½(aB're'Reé-ꟲ'ReDžungla‍EOT… (", "tokens": 48, "pieces": ["🙂", " '", "re", " İ", "'S", "­ſ", "\r\r\n\r\n", "s", "́'", "ſ", "‍", "0", " \n", "ſ", "'Re", " ", "½", "(aB", "'re", "'Re", "e", "́-", "ꟲ", "'Re", "Džungla", "‍EOT", "…", " ", "("]} +{"text": "BcamelCase'HTTPServer", "tokens": 9, "pieces": ["BcamelCase", "'<", "META", "_START", ">HTTPServer"]} +{"text": "'ll'EOT\r٣٤٥٦ꟲa/b'VE\r\n<|endoftext|>'ſḍ̇<|fim_prefix|>'S''llAb .'re٣٤٥٦ḍ̇", "tokens": 67, "pieces": ["'ll", "'EOT", "\r", "٣٤٥", "٦", "ꟲa", "/b", "'VE", "\r\n", "<|", "endoftext", "|>'", "ſḋ", "̣<|", "fim", "_prefix", "|><", "META", "_START", ">'", "S", "''", "llAb", " ", ".'", "re", "٣٤٥", "٦", "ḋ", "̣"]} +{"text": "#$% \n ३'reꟲ", "tokens": 10, "pieces": ["#$%", " \n", " ", "३", "'re", "ꟲ"]} +{"text": " \n's'Re\n/漢३​\t \n…", "tokens": 13, "pieces": [" \n", "'s", "'Re", "\n", "/漢", "३", "​", "\t \n…"]} +{"text": "Ⅳ㋿漢-İꟲᵃ\"aB
字#$%<|fim_prefix|>\r\n\r\n sİ/\r\n'll>'DEOT<­aDžunglaB,٣٤٥٦aAb(/å\t", "tokens": 61, "pieces": ["Ⅳ", "㋿漢", "-İꟲᵃ", "\"aB", "
字", "#$%<|", "fim", "_prefix", "|>\r\n\r\n", " sİ", "/\r\n", "'ll", ">'", "DEOT", "<­", "aDžunglaB", ",", "٣٤٥", "٦", "aAb", "(/", "a", "̊", "\t"]} +{"text": "EOT㍿\ré\"ſ9<|endoftext|>Džungla's'S'll٣٤٥٦fi<|fim_prefix|>Ⅳ\n­ 0camelCaseé12345678e ' \naB<|fim_prefix|> AåDžungla0㋿‍", "tokens": 83, "pieces": ["EOT", "㍿\r", "e", "́\"", "ſ", "9", "<|", "endoftext", "|>", "Džungla", "'s", "'S", "'ll", "٣٤٥", "٦", "fi", "<|", "fim", "_prefix", "|>", "Ⅳ", "\n", "­", " ", "0", "camelCaseé", "123", "456", "78", "e", " ", " '", " \n", "aB", "<|", "fim", "_prefix", "|>", " Aa", "̊Džungla", "0", "㋿‍"]} +{"text": "👍🏽㍿\r\n\r\n\u000bé12345678㋿camelCaseⅣa/bZ<\r\n\r\n漢fi", "tokens": 63, "pieces": ["éDž", "\"", "123", "456", "78", "/\r\n", " ", "'M", "<|", "fim", "_prefix", "|>㋿", "camelCase", "Ⅳ", "a", "/bZ", "<\r\n\r\n", "漢fi", ""]} +{"text": "㋿字'VE!!Ⅳ'T­#$%㍿漢camelCase", "tokens": 17, "pieces": ["'re", "\r\n", " 𐞁t", "/\r\n", ">㍿", "漢camelCase"]} +{"text": " \n 0a<​EOT٣٤٥٦‍( \n'reiOS'Reꟲ(字'MAb\n/'re­,#$%tEOTé😀🏽 ㍿a/b", "tokens": 54, "pieces": [" \n", " ", " ", "0", "a", "<​", "EOT", "٣٤٥", "٦", "‍(", " \n", "'re", "iOS", "'Re", "ꟲ", "(字", "'M", "Ab", "\n", "/'", "re", "­,#$%", "tEOTe", "́😀🏽", " ", "㍿a", "/b"]} +{"text": "HTTPServerſ'T0", "tokens": 6, "pieces": ["HTTPServerſ", "'T", "0"]} +{"text": "'M😀🏽٣٤٥٦<́é'D 'ſ㋿ ½a/bᵃ'Re…Ⅳ-ZſaB,​‍'\r\naBEOT‍ᵃ0,'ſ", "tokens": 61, "pieces": ["'M", "😀🏽", "٣٤٥", "٦", "<́", "e", "́'", "D", " '", "ſ", "㋿", " ", " ", "½", "a", "/bᵃ", "'Re", "…", "Ⅳ", "-ZſaB", ",​‍'\r\n", "aBEOT", "‍ᵃ", "0", ",'", "ſ"]} +{"text": "B\n'Re🙂 ABC-B", "tokens": 8, "pieces": ["B", "\n", "'Re", "🙂", " ", " ABC", "-B"]} +{"text": "-camelCaseeⅣé.३字\"deع'll😀🏽漢३'aB'Da/b\"camelCase(عſm漢
", "tokens": 41, "pieces": ["-camelCasee", "Ⅳ", "é", ".", "३", "字", "\"deع", "'ll", "😀🏽", "漢", "३", "'aB", "'D", "a", "/b", "\"camelCase", "(عſm漢", "
"]} +{"text": "eAbé३'T12345678 \n", "tokens": 11, "pieces": ["eAbe", "́", "३", "'T", "123", "456", "78", " \n"]} +{"text": "Z12345678é\n/0 m İ​EOT>!\r\n\n/'ll‍-'s", "tokens": 24, "pieces": ["Z", "123", "456", "78", "é", "\n", "/", "0", " ", " m", " İ", "​EOT", ">!\r\n\n", "/'", "ll", "‍-'", "s"]} +{"text": "'å㋿/\r\nAḍ̇dⅣ
d👍🏽!\tå<​३'İ ­,'re", "tokens": 46, "pieces": ["'a", "̊㋿/\r\n", "Aḋ", "̣d", "Ⅳ", "
d", "👍🏽!", "\ta", "̊<<", "META", "_START", "><", "EOT", ">​", "३", "'İ", " ", " ­,'", "re"]} +{"text": "Ⅳ Džungla,'VEꟲ㍿", "tokens": 22, "pieces": ["Ⅳ", " Džungla", ",'", "VEꟲ", "㍿"]} +{"text": "'ſ३㍿\u000b'Re​'reꟲᵃ😀🏽0٣٤٥٦\r\nmm\nAb/\r\n'T½'Re<|fim_prefix|>عéعcamelCase.'ſ \r\n\r\n12345678AAb‍<|endoftext|>\nfi", "tokens": 81, "pieces": ["'", "ſ", "३", "㍿", "\u000b", "'Re", "​'", "reꟲᵃ", "😀🏽", "0٣٤", "٥٦", "\r\n", "mm", "\n", "Ab", "/\r\n", "'T", "½", "'Re", "<|", "fim", "_prefix", "|>", "عe", "́عcamelCase", ".<", "EOT", ">'", "ſ", " \r\n\r\n", "123", "456", "78", "AAb", "‍<|", "endoftext", "|>\n", "fi"]} +{"text": "\raBcamelCasetḍ̇aficamelCaseع'ſßfiع­٣٤٥٦t/ 'Re
'VEaB'll'ſ \n EOT'aBdd \n Aعa", "tokens": 58, "pieces": ["\r", "aBcamelCasetḋ", "̣aficamelCaseع", "'ſ", "ßfiع", "­", "٣٤٥", "٦", "t", "/", " ", "'Re", "
", "'VE", "aB", "'ll", "'ſ", " \n", " EOT", "'aBdd", " \n", " Aعa"]} +{"text": "'s9\n/éſ🙂\n/漢عHTTPServer漢㍿iOSع<ß'ZdiOS.\r\n\r\n're ḍ̇ḍ̇\tß\ns<|endoftext|>\n/'VEſ", "tokens": 57, "pieces": ["'s", "9", "\n", "/e", "́ſ", "🙂\n", "/漢عHTTPServer漢", "㍿iOSع", "<ß", "'ZdiOS", ".\r\n\r\n", "'re", " ḋ", "̣ḋ", "̣", "\tß", "\n", "s", "<|", "endoftext", "|>\n", "/'", "VEſ"]} +{"text": "'sm'S'reع
\n\r­ᵃm٣٤٥٦३'ll>漢'T'llHTTPServeråſ!! 漢𐞁Ab's", "tokens": 47, "pieces": ["'s", "m", "'S", "'re", "ع", "
\n\r", "­ᵃm", "٣٤٥", "٦३", "'ll", ">漢", "'T", "'ll", "HTTPServera", "̊ſ", "!!", " 漢𐞁Ab", "'s"]} +{"text": "9a/b​½12345678 \n 123456780'M<|endoftext|>㋿'TDžungla\t🙂a/bDž \n\neDž<|endoftext|>ḍ̇!HTTPServers\r\nDžunglaع9
ꟲ/ \nDžungla'Mé", "tokens": 80, "pieces": ["9", "a", "/b", "​", "½12", "345", "678", " \n", " ", "123", "456", "780", "'M", "<|", "endoftext", "|>㋿'", "TDžungla", "\t", "🙂a", "/bDž", " \n\n", "e", "Dž", "<|", "endoftext", "|>", "ḋ", "̣!", "HTTPServers", "\r\n", "Džunglaع", "9", "
ꟲ", "/", " \n", "Džungla", "'M", "é"]} +{"text": "ع㍿́å'Ⅳ👍🏽aBms\r🙂Z​!<'VE9t㋿<", "tokens": 37, "pieces": ["ع", "㍿́<", "EOT", ">a", "̊'", "Ⅳ", "👍🏽", "aBms", "\r", "🙂Z", "​!<'", "VE", "9", "t", "㋿<"]} +{"text": "m/d <|endoftext|>'M<|endoftext|>", "tokens": 16, "pieces": ["m", "/d", " <|", "endoftext", "|>'", "M", "<|", "endoftext", "|>"]} +{"text": "'iOS\tDžungla!\r\n\"-!'VE\r\n\r\nEOT \n/\r\n\r\n>tع99漢\t'ſDžᵃ'aé's…'٣٤٥٦\r\n\r\n-fi0", "tokens": 51, "pieces": ["'iOS", "\tDžungla", "!\r\n", "\"-!'", "VE", "\r\n\r\n", "EOT", " \n", "/\r\n\r\n", ">tع", "99", "漢", "\t", "'ſ", "Džᵃ", "'ae", "́'", "s", "…", "'", "٣٤٥", "٦", "\r\n\r\n", "-fi", "0"]} +{"text": "m 'Re12345678a/bAma0", "tokens": 12, "pieces": ["m", " ", "'Re", "123", "456", "78", "a", "/bAma", "0"]} +{"text": " \nå<㋿aaB\n/ABCiOSaB-😀🏽'T\r<\n/fifi-३åḍ̇<|endoftext|>\r\nd 
'Re٣٤٥٦'re", "tokens": 70, "pieces": [" \n", "a", "̊<㋿<", "META", "_START", ">aaB", "\n", "/ABCiOSaB", "-😀🏽'", "T", "\r", "<\n", "/fifi", "-", "३", "a", "̊ḋ", "̣<|", "endoftext", "|>\r\n", "d", "", " ", "
", "'Re", "٣٤٥", "٦", "'re"]} +{"text": " \n ", "tokens": 2, "pieces": [" \n "]} +{"text": " \n ३a/b\r\nZiOSé字!!Dž", "tokens": 16, "pieces": [" \n", " ", "३", "a", "/b", "\r\n", "ZiOSe", "́字", "!!", "Dž"]} +{"text": "'M\"AſⅣİⅣ'ss12345678ſ'TaBaB0", "tokens": 23, "pieces": ["'M", "\"Aſ", "Ⅳ", "İ", "Ⅳ", "'s", "s", "123", "456", "78", "ſ", "'T", "aBaB", "0"]} +{"text": "mع'll\nDž😀🏽 \n ..<|endoftext|>\t٣٤٥٦ \r\n\r\niOS's
!'TEOTEOTꟲ'> 'M😀🏽 /'Re​", "tokens": 64, "pieces": ["mع", "'", "ll", "\n", "Dž", "😀🏽", " \n", " <", "META", "_START", ">..<|", "endoftext", "|>", "\t", "٣٤٥", "٦", " ", " <", "META", "_START", ">\r\n\r\n", "iOS", "'s", "
", "!'", "TEOTEOTꟲ", "'>", " ", "'M", "😀🏽", " ", " /'", "Re", "​"]} +{"text": "漢.٣٤٥٦a/baBDž>३ß<|fim_prefix|>漢Džungla- ><|fim_prefix|>9", "३", "ß", "<|", "fim", "_prefix", "|>", "漢Džungla", "-", " ", " ><|", "fim", "_prefix", "|>", "9", "fiå' \n fi/\r\nDž siOSḍ̇'VEé½'ll'ſ \n é'T\n \n'🙂ßé字'T字HTTPServerᵃ", "tokens": 69, "pieces": ["!!'", "Tع", "\n", "/'", "s", "👍🏽\r\n", "<|", "endoftext", "|>", "fia", "̊'", " \n", " fi", "/\r\n", "Dž", " siOSḋ", "̣'", "VEe", "́", "½", "'ll", "'ſ", " \n", " e", "́'", "T", "\n \n", "'🙂", "ßé字", "'T", "字HTTPServerᵃ"]} +{"text": "'VEaB\n<​a/bHTTPServer!e\n/fiſ́🙂d", "tokens": 26, "pieces": ["'VE", "aB", "\n", "<​", "a", "/bHTTPServer", "!e", "\n", "/fiſ", "́🙂", "d"]} +{"text": "
AbHTTPServerAbA", "tokens": 8, "pieces": ["
AbHTTPServerAbA"]} +{"text": "‍٣٤٥٦‍<|fim_prefix|>漢'S.ſᵃ!Bt,B½
''VEB   'T0å\nAbcamelCaseꟲſ \n", "tokens": 55, "pieces": ["‍", "٣٤٥", "٦", "‍<|", "fim", "_prefix", "|>", "漢", "'S", ".ſᵃ", "!Bt", ",B", "½", "
", "''", "VEB", "  ", " ", "'T", "0", "a", "̊\n", "AbcamelCaseꟲſ", " \n"]} +{"text": "½fiB\n!!#$%㋿ssEOT/‍Dž", "tokens": 19, "pieces": ["½", "fiB", "\n", "!!#$%㋿", "ssEOT", "/‍", "Dž"]} +{"text": "aB (iOS\r\n㋿,\n<İ 👍🏽㋿Džع\r\ns㍿ع'ſ'reİßⅣ­", "tokens": 41, "pieces": ["aB", " (", "iOS", "\r\n", "㋿,\n", "<İ", " 👍🏽㋿", "Dž", "ع", "\r\n", "s", "㍿ع", "'ſ", "'re", "İß", "Ⅳ", "­"]} +{"text": "'\"éaaB٣٤٥٦fi-…>\u000b½/\r\n'S\r\n\r\n/\r\n!!Ⅳ'TmiOSA 'VE ꟲ\"ع", "tokens": 45, "pieces": ["'\"", "e", "́aaB", "٣٤٥", "٦", "fi", "-", "…", ">", "\u000b", "½", "/\r\n", "'S", "\r\n\r\n", "/\r\n", "!!", "Ⅳ", "'T", "miOSA", " ", "'VE", " ", " ꟲ", "\"ع"]} +{"text": "​.ⅣHTTPServer fi…­\nḍ̇'M<|endoftext|>٣٤٥٦'ll'M//㍿㋿ḍ̇Ⅳa\n/  \rß٣٤٥٦B!!𐞁0fi​😀🏽३", "tokens": 82, "pieces": ["​.", "Ⅳ", "HTTPServer", " fi", "…", "­\n", "ḋ", "̣'", "M", "<|", "endoftext", "|>", "٣٤٥", "٦", "'ll", "'M", "//㍿㋿", "ḋ", "̣", "Ⅳ", "a", "\n", "/", "  \r", "ß", "٣٤٥", "٦", "B", "!!", "𐞁", "0", "fi", "​😀🏽", "३"]} +{"text": " e<|endoftext|>>🙂\r's'llAb åDžungla0", "tokens": 24, "pieces": [" ", " e", "<|", "endoftext", "|>>🙂\r", "'s", "'ll", "Ab", " ", " a", "̊Džungla", "0"]} +{"text": "'ſ‍👍🏽\r\n\r\nmHTTPServer\r\nAb", "tokens": 17, "pieces": ["'ſ", "‍👍🏽\r\n\r\n", "mHTTPServer", "\r\n", "Ab"]} +{"text": " ‍\"d'D
'M å𐞁's.ḍ̇/\r\n३'½/ꟲB­fi/'Re'Re-\u000b/\r\n", "tokens": 43, "pieces": [" ", "‍\"", "d", "'D", "
", "'M", " a", "̊𐞁", "'s", ".ḋ", "̣/\r\n", "३", "'", "½", "/ꟲB", "­fi", "/'", "Re", "'Re", "-", "\u000b", "/\r\n"]} +{"text": "EOT​é ́\r \n Dž,🙂\néꟲ iOS'Re", "tokens": 24, "pieces": ["EOT", "​e", "́", " ", " ́\r", " \n", " Dž", ",🙂\n", "éꟲ", " iOS", "'Re"]} +{"text": "İA'٣٤٥٦字\taBḍ̇𐞁‍\u000b'M\n/\r\n'½(#$%ß", "tokens": 36, "pieces": ["İA", "'", "٣٤٥", "٦", "字", "\taBḋ", "̣𐞁", "‍", "\u000b", "'M", "\n", "/\r\n", "'", "½", "(#$%", "ß"]} +{"text": " \n B#$%́\r\n a/b­ \nm\u000b fi", "tokens": 16, "pieces": [" \n", " B", "#$%́\r\n", " a", "/b", "­", " \n", "m", "\u000b", " fi"]} +{"text": "Dž👍🏽
're'Rea/bBᵃ­Dž​m𐞁-0--aABC😀🏽mta/bfi \n ḍ̇ ½३éd ", "tokens": 59, "pieces": ["Dž", "👍🏽", "
", "'re", "'Re", "a", "/bBᵃ", "­Dž", "​m𐞁", "-", "0", "--", "aABC", "😀🏽", "mta", "/bfi", " \n", " ḋ", "̣", " ", "½", "", "३", "e", "́d", " "]} +{"text": "㍿s9iOSſ- B!s🙂e/0'\rع́éa/bEOT 字Džungla😀🏽'S", "tokens": 41, "pieces": ["㍿s", "9", "iOSſ", "-", " ", " B", "!s", "🙂e", "/", "0", "'\r", "ع", "́e", "́a", "/bEOT", " 字Džungla", "😀🏽'", "S"]} +{"text": "'iOS0 \r!!<|fim_prefix|> 12345678\r<|fim_prefix|>éEOT!!'M/\r\n'Dfi\r\n\r\nHTTPServer\n're", "tokens": 41, "pieces": ["'iOS", "0", " \r", "!!<|", "fim", "_prefix", "|>", " ", "123", "456", "78", "\r", "<|", "fim", "_prefix", "|>", "e", "́EOT", "!!'", "M", "/\r\n", "'D", "fi", "\r\n\r\n", "HTTPServer", "\n", "'re"]} +{"text": "😀🏽>aB>İ/'S12345678 <|endoftext|>
a/bAa\t'VE<|fim_prefix|>'VE字fi३́'llé9ß<|endoftext|>Ⅳ<|endoftext|>\r\n\r\nß🙂'MEOT'M", "tokens": 73, "pieces": ["😀🏽>", "aB", ">İ", "/'", "S", "123", "456", "78", " <|", "endoftext", "|>", "
a", "/bAa", "\t", "'VE", "<|", "fim", "_prefix", "|>'", "VE字fi", "३", "́'", "llé", "9", "ß", "<|", "endoftext", "|>", "Ⅳ", "<|", "endoftext", "|>\r\n\r\n", "ß", "🙂'", "MEOT", "'M"]} +{"text": "'smDž'VEİé", "tokens": 9, "pieces": ["'s", "mDž", "'VE", "İe", "́"]} +{"text": ">-aDžAZ>漢\u000bع👍🏽'Mꟲ'Mté", "tokens": 26, "pieces": [">-", "aDžAZ", ">漢", "\u000bع", "👍🏽'", "Mꟲ", "'M", "te", "́"]} +{"text": "㋿३éABC(!'T½\"👍🏽9/\r\n'VE's #$%\ra/b
camelCase's​​ iOS🙂‍", "tokens": 46, "pieces": ["㋿", "३", "e", "́ABC", "(!'", "T", "½", "\"👍🏽", "9", "/\r\n", "'VE", "'s", " ", "#$%\r", "a", "/b", "
camelCase", "'s", "​​", " iOS", "🙂‍"]} +{"text": "ⅣAb ​'s٣٤٥٦ \u000b½½ꟲ/\r\n>", "tokens": 24, "pieces": ["Ⅳ", "Ab", " ", "​'", "s", "٣٤٥", "٦", " ", "\u000b", "½½", "ꟲ", "/\r\n", ">"]} +{"text": "🙂BABC㍿Ab'llZAZ\t…>9字​", "tokens": 20, "pieces": ["🙂BABC", "㍿Ab", "'ll", "ZAZ", "\t", "…", ">", "9", "字", "​"]} +{"text": " \n\r'Dſ'S", "tokens": 6, "pieces": [" \n\r", "'D", "ſ", "'S"]} +{"text": "a/b/\r\n🙂٣٤٥٦!.‍<|fim_prefix|>>  \nſDž\n/camelCaseå𐞁 ", "tokens": 45, "pieces": ["a", "/b", "/\r\n", "🙂<", "EOT", ">", "٣٤٥", "٦", "!.‍<|", "fim", "_prefix", "|>>", "  \n", "ſDž", "\n", "/camelCasea", "̊𐞁", " "]} +{"text": "9Džungla\u000b\u000b12345678İB'VE-'M漢'VE\r​😀🏽é'M٣٤٥٦d<|endoftext|>'S><|fim_prefix|>9 \n AbfiEOTDž
! \n 'M'T", "tokens": 73, "pieces": ["9", "Džungla", "\u000b", "\u000b", "123", "456", "78", "İB", "'VE", "-'", "M漢", "'VE", "\r", "​😀🏽", "é", "'M", "٣٤٥", "٦", "d", "<|", "endoftext", "|>'", "S", "><|", "fim", "_prefix", "|>", "9", " \n", " AbfiEOTDž", "
", "!", " \n", " '", "M", "'T", ""]} +{"text": "å­٣٤٥٦å.\" ᵃa/b\"\n/½a/b😀🏽 \u000b \n's<|endoftext|>'s \n\u000b 9½‍­İ'㋿m
字🙂ts", "tokens": 67, "pieces": ["a", "̊­", "٣٤٥", "٦", "a", "̊.\"", " ᵃa", "/b", "\"\n", "/", "½", "a", "/b", "😀🏽", " \u000b \n", "'s", "<|", "endoftext", "|>'", "s", " \n", "\u000b", " ", "9½", "‍­", "İ", "'㋿", "m", "", "
字", "🙂ts"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": ", ('M(\n/'re \n ٣٤٥٦\r\n<​字👍🏽\n \n 😀🏽a/bᵃ>'ſ<|fim_prefix|>漢́", "tokens": 55, "pieces": [",", " ", " ('", "M", "(\n", "/'", "re", " \n", " ", "٣٤٥", "٦", "\r\n", "<​", "字", "👍🏽\n", "", " \n", " 😀🏽", "a", "/bᵃ", ">'", "ſ", "<|", "fim", "_prefix", "|>", "漢", "́"]} +{"text": "<|endoftext|>'Sع½<३-Džungla'ScamelCase<|endoftext|>­\r<|fim_prefix|><|fim_prefix|>Ⅳ0-ABCHTTPServer('ll‍½-\n…\r\n\r\n字,\r\n\r\n<<|fim_prefix|>́0", "tokens": 76, "pieces": ["<|", "endoftext", "|>'", "S", "ع", "½", "<", "३", "-Džungla", "'S", "camelCase", "<|", "endoftext", "|>­\r", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", "Ⅳ0", "-ABCHTTPServer", "('", "ll", "‍", "½", "-\n", "…\r\n\r\n", "字", ",\r\n\r\n", "<<|", "fim", "_prefix", "|>́", "0"]} +{"text": "<|endoftext|>…👍🏽\"'Re字Za/b ㍿ #$%<|fim_prefix|>ß👍🏽fi're.ſⅣ\n/३EOT,漢aع\r åſ‍👍🏽camelCase漢d\n!", "tokens": 82, "pieces": ["<|", "endoftext", "|>", "…", "👍🏽\"'", "Re字Za", "/b", " ", "㍿", " ", "#$%<|", "fim", "_prefix", "|>", "ß", "👍🏽", "fi", "'re", ".ſ", "Ⅳ", "\n", "/", "३", "EOT", ",漢aع", "\r", " a", "̊ſ", "‍👍🏽", "camelCase漢d", "\n", "!"]} +{"text": " <|endoftext|>\r\n'T12345678a/bß'Re\"½\u000b're/́dᵃ'S­😀🏽 漢'M\ncamelCase
ᵃ­ \n ٣٤٥٦ \n", "tokens": 56, "pieces": [" ", " <|", "endoftext", "|>\r\n", "'T", "123", "456", "78", "a", "/bß", "'Re", "\"", "½", "\u000b", "'re", "/́", "dᵃ", "'S", "­😀🏽", " 漢", "'M", "\n", "camelCase", "
ᵃ", "­", " \n", " ", "٣٤٥", "٦", " \n"]} +{"text": "Abå.!\n/Ab/\">\n/\r\nDžungla !HTTPServerABC­!!<|fim_prefix|>é", "tokens": 34, "pieces": ["Aba", "̊.!\n", "/Ab", "/\">\n", "/\r\n", "Džungla", " ", "!HTTPServerABC", "­!!<|", "fim", "_prefix", "|>", "é"]} +{"text": "/\r\n'T' 12345678İDžunglaDž'ſBᵃ😀🏽!'TBBß\r\n\r\n#$%EOT<́漢字 ", "tokens": 41, "pieces": ["/\r\n", "'T", "'", " ", "123", "456", "78", "İDžunglaDž", "'ſ", "Bᵃ", "😀🏽!'", "TBBß", "\r\n\r\n", "#$%", "EOT", "<́", "漢字", " "]} +{"text": "-…12345678㋿'", "tokens": 10, "pieces": ["-", "…", "123", "456", "78", "㋿'"]} +{"text": "Džᵃ\".EOTABC'll'rea/b12345678\n<|fim_prefix|>'ll㍿Džungla字​-e…ᵃعiOSعḍ̇­'ll٣٤٥٦'s😀🏽Ab…", "tokens": 69, "pieces": ["Džᵃ", "\".", "EOTABC", "'ll", "'re", "a", "/b", "123", "456", "78", "\n", "<|", "fim", "_prefix", "|>'", "ll", "㍿Džungla字", "​-", "e", "…ᵃعiOSعḋ", "̣­'", "ll", "٣٤٥", "٦", "'s", "😀🏽", "Ab", "…"]} +{"text": " 9a‍ İ 👍🏽camelCaseİ'!ꟲᵃ-,tt'VEås\"'Re 'reᵃ…'s'S>", "tokens": 50, "pieces": [" ", " ", "9", "a", "‍", " İ", " ", "👍🏽", "camelCaseİ", "'!", "ꟲᵃ", "-,", "tt", "'VE", "a", "̊s", "\"<", "META", "_START", ">'", "Re", " ", " '", "reᵃ", "…", "'s", "'S", ">"]} +{"text": "Džunglaſ😀🏽>ᵃ'TcamelCaseⅣ \nⅣ", "tokens": 23, "pieces": ["Džunglaſ", "😀🏽>", "ᵃ", "'T", "camelCase", "Ⅳ", " \n", "Ⅳ"]} +{"text": "åt​ſⅣiOS  \r\niOSEOTiOS'Re-\t'ſ<|endoftext|>\r(㍿ \n iOS'Re\n'ſ漢'T ᵃ \n ", "tokens": 51, "pieces": ["a", "̊t", "​ſ", "Ⅳ", "iOS", "  \r\n", "iOSEOTiOS", "'Re", "-", "\t", "'ſ", "<|", "endoftext", "|>\r", "(㍿", " \n", " iOS", "'Re", "\n", "'ſ", "漢", "'T", " ᵃ", " \n "]} +{"text": "́३३Ab<🙂", "tokens": 12, "pieces": ["́", "३३", "Ab", "<🙂"]} +{"text": "'s'lls😀🏽/ \n 0å\u000b", "tokens": 16, "pieces": ["'s", "'ll", "s", "😀🏽/", " \n", " ", "0", "a", "̊", "\u000b"]} +{"text": " \u000bs<|endoftext|>'ll>́\nt,'M,\u000b​\r", "tokens": 21, "pieces": [" ", "\u000bs", "<|", "endoftext", "|>'", "ll", ">́\n", "t", ",'", "M", ",", "\u000b", "​\r"]} +{"text": " ́!!Z 'DHTTPServerm㍿9'D½/", "tokens": 17, "pieces": [" ", " ́!!", "Z", " ", "'D", "HTTPServerm", "㍿", "9", "'D", "½", "/"]} +{"text": "a/bᵃ½.'D/\r\n\n", "tokens": 10, "pieces": ["a", "/bᵃ", "½", ".'", "D", "/\r\n\n"]} +{"text": "'ſ½\"ḍ̇漢/.-t\t'DBe\"\u000bß \n ३ßᵃ\tſ9ꟲ́Ⅳ­(ß's\u000b𐞁́
#$%ᵃ字B", "tokens": 58, "pieces": ["'ſ", "½", "\"ḋ", "̣漢", "/.-", "t", "\t", "'D", "Be", "\"", "\u000bß", " \n", " ", "३", "ßᵃ", "\tſ", "9", "ꟲ", "́", "Ⅳ", "­(", "ß", "'s", "\u000b𐞁", "́", "
", "#$%", "ᵃ字B"]} +{"text": "'Aß 'T…'Refit('llaB \naB́ \n /३\r
,\r\n\r\n're​camelCase", "tokens": 31, "pieces": ["'Aß", " ", "'T", "…", "'Re", "fit", "('", "llaB", " \n", "aB", "́", " \n", " /", "३", "\r", "
", ",\r\n\r\n", "'re", "​camelCase"]} +{"text": "!!\n/漢İ३\"('VEfiꟲd㋿9iOS,'Ree>", "tokens": 25, "pieces": ["!!\n", "/漢İ", "३", "\"('", "VEfiꟲd", "㋿", "9", "iOS", ",'", "Ree", ">"]} +{"text": "\r\nZ's'D\rſ'VE<|fim_prefix|>'D \n >", "tokens": 19, "pieces": ["\r\n", "Z", "'s", "'D", "\r", "ſ", "'VE", "<|", "fim", "_prefix", "|>'", "D", " \n", " >"]} +{"text": "'re \n a'reİe's𐞁𐞁", "tokens": 21, "pieces": ["'re", " \n", " a", "'re", "İ", "e", "'s", "𐞁𐞁", ""]} +{"text": "'ſ mfié0
éᵃé\n/­12345678t‍mEOT#$%ꟲ\r३३३aaBZ-‍👍🏽ßB", "tokens": 58, "pieces": ["'ſ", " mfié", "0", "
éᵃe", "́\n", "/­", "123", "456", "78", "t", "‍mEOT", "#$%", "ꟲ", "\r", "३३३", "aaBZ", "-<", "META", "_START", ">‍👍🏽", "ßB"]} +{"text": "'ſ'Msß's🙂!/㍿'sEOTåfi'saBsaḍ̇AİꟲåDžungla0(‍HTTPServer", "tokens": 52, "pieces": ["'ſ", "'", "Msß", "'s", "🙂!/㍿'", "sEOTa", "̊fi", "'s", "aBsaḋ", "̣Aİꟲa", "̊Džungla", "0", "(‍", "HTTPServer"]} +{"text": " \n !́ſAbå>-\rḍ̇👍🏽BHTTPServercamelCaseḍ̇'T,\tt-字", "tokens": 40, "pieces": [" \n", " !́", "ſAb", "a", "̊>-\r", "ḋ", "̣👍🏽", "BHTTPServercamelCaseḋ", "̣'", "T", ",", "\tt", "-字"]} +{"text": "EOTعiOŚDžungla㋿👍🏽字🙂½t0'ſßⅣB!!
<|endoftext|>Dž👍🏽🙂Džḍ̇ .", "tokens": 62, "pieces": ["EOTعiOS", "́Džungla", "㋿👍🏽", "字", "🙂", "½", "t", "0", "'", "ſß", "Ⅳ", "B", "!!", "
", "<|", "endoftext", "|>", "Dž", "👍🏽🙂", "Džḋ", "̣", " ", " ."]} +{"text": "A<|fim_prefix|>B'ree㍿字ABC/\r\ncamelCase0a/\r\n漢\ta٣٤٥٦३", "tokens": 36, "pieces": ["A", "<|", "fim", "_prefix", "|>", "B", "'re", "e", "㍿字ABC", "/\r\n", "camelCase", "0", "a", "/\r\n", "漢", "\ta", "٣٤٥", "٦३"]} +{"text": " aHTTPServer\r\n\n,ZaB'll ㋿\r㍿'VE A's'VE12345678're😀🏽'S( \n ABC‍ſ\t", "tokens": 52, "pieces": [" aHTTPServer", "\r\n\n", ",ZaB", "'ll", "", " ", " ㋿\r", "㍿'", "VE", " <", "META", "_START", ">A", "'s", "'VE", "123", "456", "78", "'re", "😀🏽'", "S", "(", " \n", " ABC", "‍ſ", "\t"]} +{"text": "-a/b/\r\n- \n !é!\n/­>.é​\" \n \n -漢'S\r'Re'D!Džungla.'M३", "tokens": 33, "pieces": ["-a", "/b", "/\r\n", "-", " \n", " !", "e", "́!\n", "/­>.", "e", "́​\"", " \n \n", " -", "漢", "'S", "\r", "'Re", "'D", "!Džungla", ".'", "M", "३"]} +{"text": "\r\n🙂-/İDžunglaꟲ​ \n're🙂\u000bA'M'éé३\naé0𐞁fi12345678#$%HTTPServer\n/'re½\"ß", "tokens": 54, "pieces": ["\r\n", "🙂<", "EOT", ">-/", "İDžunglaꟲ", "​", " \n", "'re", "🙂", "\u000bA", "'M", "'e", "́e", "́", "३", "\n", "aé", "0", "𐞁fi", "123", "456", "78", "#$%", "HTTPServer", "\n", "/'", "re", "½", "\"ß"]} +{"text": "e-,𐞁t'ſⅣ​a/b>'ll'ſ<|fim_prefix|>!e12345678\n٣٤٥٦\tⅣt字ABC'reⅣ'reée'\r\n'
EOTHTTPServer'iOS<|fim_prefix|><|endoftext|>B", "tokens": 76, "pieces": ["e", "-,", "𐞁t", "'ſ", "Ⅳ", "​a", "/b", ">'", "ll", "'ſ", "<|", "fim", "_prefix", "|>!", "e", "123", "456", "78", "\n", "٣٤٥", "٦", "\t", "Ⅳ", "t字ABC", "'re", "Ⅳ", "'re", "ée", "'\r\n", "'", "
EOTHTTPServer", "'iOS", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "B"]} +{"text": "-\n/ABC \n 👍🏽́́ \n \naB/\r\n👍🏽'Dſ'ſ३camelCase0٣٤٥٦‍siOSDž#$%'😀🏽<\rꟲ<|endoftext|>'VE \n'ſa'T0'Re", "tokens": 80, "pieces": ["-\n", "/ABC", " \n", " ", " 👍🏽́́", " \n \n", "aB", "/\r\n", "👍🏽'", "Dſ", "'ſ", "", "३", "camelCase", "0٣٤", "٥٦", "‍siOSDž", "#$%'😀🏽<\r", "ꟲ", "<|", "endoftext", "|>'", "VE", " \n", "'ſ", "a", "'T", "0", "'Re"]} +{"text": "'reeé٣٤٥٦>٣٤٥٦ABĆ'ſ--'VEt0<|fim_prefix|>\r\n\r\n​
\tDž<­fi字''s", "tokens": 50, "pieces": ["'re", "eé", "٣٤٥", "٦", ">", "٣٤٥", "٦", "ABC", "́'", "ſ", "--'", "VEt", "0", "<|", "fim", "_prefix", "|>\r\n\r\n", "​", "
", "\tDž", "<­", "fi字", "''", "s"]} +{"text": "Džungla'llABC㍿ 9. 𐞁ᵃmſ𐞁Z", "tokens": 32, "pieces": ["Džungla", "'", "llABC", "㍿", " ", "9", ".", " 𐞁ᵃmſ𐞁Z"]} +{"text": "Dž12345678\u000b\"/ß\r\n𐞁'reé½Ab\n/0're'ſ>tAbß­a/a/b/AbHTTPServer\u000b'S", "tokens": 41, "pieces": ["Dž", "", "123", "456", "78", "\u000b", "\"/", "ß", "\r\n", "𐞁", "'re", "é", "½", "Ab", "\n", "/", "0", "'re", "'ſ", ">tAbß", "­a", "/a", "/b", "/AbHTTPServer", "\u000b", "'S"]} +{"text": "Ⅳ \n#$%İA<|fim_prefix|>sDžt!​'re\nfi", "tokens": 26, "pieces": ["Ⅳ", " \n", "#$%", "İA", "<|", "fim", "_prefix", "|>", "sDžt", "!​'", "re", "\n", "fi"]} +{"text": "''Tfi㋿!!", "tokens": 27, "pieces": ["''", "Tfi", "㋿!!"]} +{"text": "9 \n𐞁 \u000b  é-ḍ̇ \n aB'S'lla'T\u000b👍🏽🙂ABC漢åİ<|fim_prefix|>.漢'Ms", "tokens": 51, "pieces": ["9", " \n", "𐞁", " \u000b  ", " é", "-ḋ", "̣", " \n", " aB", "'S", "'ll", "a", "'T", "\u000b", "👍🏽🙂", "ABC漢a", "̊İ", "<|", "fim", "_prefix", "|>.", "漢", "'M", "s"]} +{"text": "camelCase'M‍ݽ‍fi<|endoftext|> ㋿>aBs㍿\r\r\nع", "tokens": 30, "pieces": ["camelCase", "'M", "‍İ", "½", "‍fi", "<|", "endoftext", "|>", " ", "㋿>", "aBs", "㍿\r\r\n", "ع"]} +{"text": "ḍ̇a/b#$%👍🏽
ᵃteåA½émⅣ", "tokens": 56, "pieces": ["", "ḋ", "̣a", "/b", "#$%👍🏽", "
ᵃtea", "̊A", "½", "ém", "Ⅳ"]} +{"text": "'ſ漢​\r\n!d", "tokens": 9, "pieces": ["'ſ", "漢", "​\r\n", "!d"]} +{"text": "\t…👍🏽ḍ̇!!", "tokens": 22, "pieces": ["\t", "", "…", "👍🏽", "ḋ", "̣<", "META", "_START", ">!!"]} +{"text": "\rDž \n s👍🏽's字-mع🙂👍🏽/‍!å12345678🙂\r\n\r\n…/'Reḍ̇ <|endoftext|>👍🏽<|fim_prefix|><|fim_prefix|>​,Bå", "tokens": 84, "pieces": ["\r", "Dž", " \n", " s", "👍🏽'", "s字", "-mع", "🙂👍🏽/‍!", "a", "̊", "123", "456", "78", "🙂\r\n\r\n", "…", "/'", "Reḋ", "̣", " ", " <|", "endoftext", "|>👍🏽<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>​,", "Ba", "̊"]} +{"text": "ſ<|endoftext|>a'S😀🏽'EOTe(😀🏽İEOT9éDž", "tokens": 33, "pieces": ["ſ", "<|", "endoftext", "|>", "a", "'S", "😀🏽'", "EOTe", "(😀🏽", "İEOT", "9", "éDž"]} +{"text": "\r\n\r\n\tḍ̇aB\r\u000b‍\u000b  ­a/b''S'Me12345678३ABC​…< te're\n/ ", "tokens": 39, "pieces": ["\r\n\r\n", "\tḋ", "̣aB", "\r", "\u000b", "‍", "\u000b ", " ", "­a", "/b", "''", "S", "'M", "e", "123", "456", "78३", "ABC", "​", "…", "<", " ", " te", "'re", "\n", "/", " "]} +{"text": "'re!!/ \n", "tokens": 4, "pieces": ["'re", "!!/", " \n"]} +{"text": "‍𐞁漢'ſ😀🏽\r\n\r\nḍ̇sß Džungla'M\n!/\r\n", "tokens": 33, "pieces": ["‍𐞁漢", "'ſ", "😀🏽\r\n\r\n", "ḋ", "̣sß", " Džungla", "'M", "\n", "!/\r\n"]} +{"text": "٣٤٥٦'D're'\t \n 'fi'TⅣ३EOT!#$%", "tokens": 25, "pieces": ["٣٤٥", "٦", "'D", "'re", "'", "\t \n", " '", "fi", "'T", "Ⅳ३", "EOT", "!#$%"]} +{"text": "‍e'T ३‍/\r\n", "tokens": 11, "pieces": ["‍e", "'T", " ", " ", "३", "‍/\r\n"]} +{"text": "🙂'㋿.Z'­>é𐞁​", "tokens": 18, "pieces": ["🙂'㋿.", "Z", "'­>", "e", "́𐞁", "​"]} +{"text": "iOS㍿", "tokens": 4, "pieces": ["iOS", "㍿"]} +{"text": "EOT‍Ⅳé'M0
عiOSé漢dcamelCase0", "tokens": 20, "pieces": ["EOT", "‍", "Ⅳ", "é", "'M", "0", "
عiOSé漢dcamelCase", "0"]} +{"text": "/🙂\t<|endoftext|><|endoftext|>‍Ab#$%​
ABC-ḍ̇𐞁HTTPServer
́\ré'Re\u000b½'ll#$%‍'VE ½!!<|fim_prefix|>!!é're<|endoftext|>漢'll", "tokens": 78, "pieces": ["/🙂", "\t", "<|", "endoftext", "|><|", "endoftext", "|>‍", "Ab", "#$%​", "
ABC", "-ḋ", "̣𐞁HTTPServer", "
", "́\r", "é", "'Re", "\u000b", "½", "'ll", "#$%‍'", "VE", " ", "½", "!!<|", "fim", "_prefix", "|>!!", "e", "́'", "re", "<|", "endoftext", "|>", "漢", "'ll"]} +{"text": "'VE ḍ̇é‍ḍ̇ /\r\n-\rDž字,ḍ̇'S٣٤٥٦\t<|endoftext|>­.d'RemHTTPServer \n ½ßfiåع 'Re<|fim_prefix|>ss'Sd𐞁ABC", "tokens": 82, "pieces": ["'VE", " ", " ḋ", "̣e", "́‍", "ḋ", "̣", " /\r\n", "-\r", "Dž字", ",ḋ", "̣'", "S", "٣٤٥", "٦", "\t", "<|", "endoftext", "|>­.", "d", "'Re", "mHTTPServer", " \n", " ", "½", "ßfia", "̊ع", " ", "'Re", "<|", "fim", "_prefix", "|>", "ss", "'S", "d𐞁ABC"]} +{"text": "😀🏽s'Re's'D>'reta/bꟲHTTPServerm,!​ \n", "tokens": 21, "pieces": ["😀🏽", "s", "'Re", "'s", "'D", ">'", "reta", "/bꟲHTTPServerm", ",!​", " \n"]} +{"text": "camelCase<|fim_prefix|><|endoftext|>'s٣٤٥٦EOT//HTTPServer'ReBDž ABCßİⅣDž", "tokens": 40, "pieces": ["camelCase", "<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "s", "٣٤٥", "٦", "EOT", "//", "HTTPServer", "'Re", "BDž", " ABCßİ", "Ⅳ", "Dž"]} +{"text": "éع‍e\r\ne٣٤٥٦'re३ßaéꟲ漢'D.‍Džſ𐞁­-12345678👍🏽ꟲḍ̇(Ⅳ/\r\n,/fi ß½
", "tokens": 69, "pieces": ["éع", "‍e", "\r\n", "e", "٣٤٥", "٦", "'re", "३", "ßaéꟲ漢", "'D", ".‍", "Džſ𐞁", "­-", "123", "456", "78", "👍🏽", "ꟲḋ", "̣(", "Ⅳ", "/\r\n", ",/", "fi", " ß", "½", "
"]} +{"text": ",\r\n\r\nDžAb­\r\n\u000b12345678漢 \n .㍿́Džéᵃ's👍🏽,dé<|endoftext|>/
ßḍ̇<|fim_prefix|><|endoftext|>\nm'ſ \n ", "tokens": 73, "pieces": [",\r\n\r\n", "DžAb", "­\r\n", "\u000b", "123", "456", "78", "漢", " \n", " .㍿́", "Dže", "́<", "META", "_START", ">ᵃ", "'s", "👍🏽,", "de", "́<|", "endoftext", "|>/", "
ßḋ", "̣<|", "fim", "_prefix", "|><|", "endoftext", "|>\n", "m", "'ſ", " \n "]} +{"text": "\n/漢-İ9ꟲ're'Re<|endoftext|>#$%-㍿<|endoftext|>ABCfi.'llABCDžungla/\r\n'ſ'VEꟲ,漢,漢½Dž\n𐞁m… <|endoftext|>ꟲ", "tokens": 81, "pieces": ["\n", "/漢", "-İ", "9", "ꟲ", "'re", "'Re", "<|", "endoftext", "|>#$%-㍿<|", "endoftext", "|>", "ABCfi", ".'", "llABCDžungla", "/\r\n", "'ſ", "'VE", "ꟲ", ",", "漢", ",漢", "½", "Dž", "\n", "𐞁m", "…", " ", "<|", "endoftext", "|>", "ꟲ"]} +{"text": "🙂\"'ſßcamelCase
-'s!\"", "tokens": 16, "pieces": ["🙂\"'", "ſßcamelCase", "", "
", "-'", "s", "!\""]} +{"text": "İ́Ⅳ​iOS're'٣٤٥٦㍿ \u000b\r\n\r\n‍'VE ", "tokens": 37, "pieces": ["Ⅳ", "'D", "a字", "<|", "endoftext", "|>", "Ⅳ", "​iOS", "'re", "'", "٣٤٥", "٦", "㍿", " \u000b\r\n\r\n", "‍'", "VE", " "]} +{"text": "ꟲa/b<|fim_prefix|>B\r字0<'s/\r\n!!𐞁\u000b\n/", "tokens": 27, "pieces": ["ꟲa", "/b", "<|", "fim", "_prefix", "|>", "B", "\r", "字", "0", "<'", "s", "/\r\n", "!!", "𐞁", "\u000b\n", "/"]} +{"text": "Ⅳ́\u000bEOTe'T s‍ \n'VE𐞁<|fim_prefix|>٣٤٥٦ABC0 \t/\r\n!/\r\n‍/ ㍿'s'sDž !!😀🏽", "tokens": 69, "pieces": ["a", "̊'", "ſ", "!", "Ⅳ३", ">", "\u000bEOTe", "'T", " s", "‍", " \n", "'VE", "𐞁", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "ABC", "0", " ", "\t", "/\r\n", "!/\r\n", "‍/", " ", "㍿'", "s", "'s", "Dž", " ", "!!😀🏽"]} +{"text": "B'T0fié \né<|fim_prefix|>/‍iOSDž
", "tokens": 27, "pieces": ["B", "'T", "0", "fie", "́", " \n", "e", "́<|", "fim", "_prefix", "|>/‍", "iOSDž", "
"]} +{"text": "9字'VEe", "tokens": 9, "pieces": ["9", "字", "'VE", "e", ""]} +{"text": "\ra/b/'S'sfiİ'D's<|endoftext|>0​Ź३B ('M漢㍿ABC𐞁'Mع
'DAſ'>/å㋿ᵃ", "tokens": 57, "pieces": ["\r", "a", "/b", "/'", "S", "'s", "fiİ", "'D", "'s", "<|", "endoftext", "|>", "0", "​Z", "́", "३", "B", " ('", "M漢", "㍿ABC𐞁", "'M", "ع", "
", "'D", "Aſ", "'>/", "a", "̊㋿", "ᵃ"]} +{"text": "'s'Td9 🙂åꟲéé३Ⅳ३12345678'Re\u000b\r\n \nZ\n/!!'llḍ̇camelCase", "tokens": 42, "pieces": ["'s", "'T", "d", "9", " ", " 🙂", "a", "̊ꟲe", "́e", "́", "३Ⅳ३", "123", "456", "78", "'Re", "\u000b\r\n \n", "Z", "\n", "/!!'", "llḋ", "̣camelCase"]} +{"text": "fiZDžungla­<'re, !aB(A\r\n\t ٣٤٥٦😀🏽‍", "tokens": 38, "pieces": ["fiZDžungla", "­<'", "re", ",", " !", "aB", "(A", "\r\n", "\t", "", " ", "٣٤٥", "٦", "😀🏽‍"]} +{"text": "(a/bmßa/baB'aBé<|fim_prefix|>Džungla / (.'sDž(>iOS…𐞁½ABC.\r\nfia/b!!iOS‍9 \r\n\r\nḍ̇ ", "tokens": 61, "pieces": ["(a", "/bmßa", "/b", "aB", "'aBe", "́<|", "fim", "_prefix", "|>", "Džungla", " ", " /", " ", " (.'", "sDž", "(>", "iOS", "…𐞁", "½", "ABC", ".\r\n", "fia", "/b", "!!", "iOS", "‍", "9", " \r\n\r\n", "ḋ", "̣", " "]} +{"text": "​𐞁\n/ \n's\r\n\r\nmḍ̇'s\tAİß\"Ab#$%'ll👍🏽's㋿㋿\n!! d/عfi\n\"'re\rḍ̇
0", "tokens": 72, "pieces": ["​𐞁", "\n", "/<", "EOT", ">", " \n", "'s", "\r\n\r\n", "mḋ", "̣'", "s", "\t", "Aİß", "\"Ab", "#$%'", "ll", "👍🏽'", "s", "㋿㋿\n", "!!", " d", "/عfi", "\n", "\"'", "re", "\r", "ḋ", "̣", "
", "0"]} +{"text": "ſſEOT漢३t٣٤٥٦Ab-ſ́<|fim_prefix|>'ſiOS\"'ll\n\r\n\r\nſعaB'DiOS㍿🙂", "tokens": 53, "pieces": ["ſſEOT漢", "३", "t", "٣٤٥", "٦", "Ab", "-ſ", "́<|", "fim", "_prefix", "|>'", "ſiOS", "\"'", "ll", "\n\r\n\r\n", "ſعaB", "'D", "iOS", "㍿🙂"]} +{"text": "३!!a!Džunglaꟲ\r\nDžungla'…e½é漢ſé'ſEOTéİ'VE>Abꟲ \n\r\n\r\ńAiOS㋿👍🏽t'ſ", "tokens": 64, "pieces": ["३", "!!", "a", "!Džunglaꟲ", "\r\n", "Džungla", "'", "…e", "½", "é", "漢ſé", "'ſ", "EOTéİ", "'VE", ">Abꟲ", " \n\r\n\r\n", "́AiOS", "㋿👍🏽", "t", "'ſ"]} +{"text": "'re🙂\"\naB!!EOT\"åaBiOSḍ̇<-字ع/\r\nDž​é'ſꟲع\nꟲEOTBt ", "tokens": 44, "pieces": ["'re", "🙂\"\n", "aB", "!!", "EOT", "\"a", "̊aBiOSḋ", "̣<-", "字ع", "/\r\n", "Dž", "​é", "'ſ", "ꟲع", "\n", "ꟲEOTBt", " "]} +{"text": "​é 字३é9é \n é\n/-'D", "tokens": 22, "pieces": ["​é", " <", "META", "_START", ">", " ", " 字", "३", "e", "́", "9", "e", "́", " \n", " e", "́\n", "/-'", "D"]} +{"text": "\r\n٣٤٥٦>

\rEOTme Z<|endoftext|>/ḍ̇㍿字 \n'T", "tokens": 37, "pieces": ["\r\n", "٣٤٥", "٦", ">", "

\r", "EOTme", " Z", "<|", "endoftext", "|>/", "ḋ", "̣㍿", "字", " \n", "'T"]} +{"text": " İ\r\n\r\nHTTPServerAb\n/-/'ll㍿ß😀🏽EOT ́㋿'M㋿㋿m<|endoftext|>sé½\r\n…", "tokens": 49, "pieces": [" ", " İ", "\r\n\r\n", "HTTPServerAb", "\n", "/-/'", "ll", "㍿ß", "😀🏽", "EOT", " ", " ́㋿'", "M", "㋿㋿", "m", "<|", "endoftext", "|>", "sé", "½", "\r\n…"]} +{"text": "'M!Dž㋿\n\n/İ ‍iOS'MᵃDžungla!!tß\na\u000b\n\nß -'VE\n/fiaBé9å<|fim_prefix|>12345678'ſé", "tokens": 60, "pieces": ["'M", "!Dž", "㋿\n\n", "/İ", " ", "‍iOS", "'M", "ᵃDžungla", "!!", "tß", "\n", "a", "\u000b\n\n", "ß", " ", " -'", "VE", "\n", "/fiaBé", "9", "a", "̊<|", "fim", "_prefix", "|>", "123", "456", "78", "'ſ", "e", "́"]} +{"text": ",ᵃcamelCase", "tokens": 6, "pieces": [",ᵃcamelCase"]} +{"text": "Ab
-ꟲ'D'\u000b>­#$%t ㍿iOS'M iOS\r\n\"🙂iOS👍🏽-ßa/b0 𐞁#$%\t", "tokens": 52, "pieces": ["Ab", "
", "-ꟲ", "'D", "'", "\u000b", "><", "META", "_START", ">­#$%", "t", " ㍿", "iOS", "'M", " iOS", "\r\n", "\"🙂", "iOS", "👍🏽-", "ßa", "/b", "0", " ", "𐞁", "#$%", "\t"]} +{"text": "ᵃ\r\n­३ZHTTPServerABC\n/<|endoftext|>", "tokens": 19, "pieces": ["ᵃ", "\r\n", "­", "३", "ZHTTPServerABC", "\n", "/<|", "endoftext", "|>"]} +{"text": "d३>'Re-\u000b", "tokens": 7, "pieces": ["d", "३", ">'", "Re", "-", "\u000b"]} +{"text": "9 ㋿", "tokens": 5, "pieces": ["9", " ", "㋿"]} +{"text": "\u000bB\t㋿/s", "tokens": 8, "pieces": ["\u000bB", "\t", "㋿/", "s"]} +{"text": "ꟲ😀🏽Dž/ABCaBAßiOS'Rea", "tokens": 20, "pieces": ["ꟲ", "😀🏽", "Dž", "/ABCaBAßiOS", "'Re", "a"]} +{"text": "éḍ̇<٣٤٥٦HTTPServer,, ", "tokens": 19, "pieces": ["éḋ", "̣<", "٣٤٥", "٦", "HTTPServer", ",,", " "]} +{"text": "'T
İ'VEⅣß", "tokens": 9, "pieces": ["'T", "
İ", "'VE", "Ⅳ", "ß"]} +{"text": "12345678DžAb\u000be(.\tABCiOSß​ꟲå漢'll­<|fim_prefix|>­é㋿ \n ", "tokens": 38, "pieces": ["123", "456", "78", "DžAb", "\u000be", "(.", "\tABCiOSß", "​ꟲa", "̊漢", "'ll", "­<|", "fim", "_prefix", "|>­", "é", "㋿", " \n "]} +{"text": "ꟲ ع\n<|endoftext|> \n 'S'VE#$%'T👍🏽/\r\né", "tokens": 35, "pieces": ["ꟲ", " ", " ع", "\n", "<|", "endoftext", "|>", " \n", " '", "S", "'VE", "#$%'", "T", "👍🏽/\r\n", "<", "EOT", ">e", "́"]} +{"text": "\n/Z \n ", "tokens": 4, "pieces": ["\n", "/Z", " \n "]} +{"text": "!\n/'Reå😀🏽\r\n\n/Ⅳ'Me字fi'VEſ漢 \n!a/bDž­🙂.𐞁-", "tokens": 41, "pieces": ["!\n", "/'", "Rea", "̊😀🏽\r\n\n", "/", "Ⅳ", "'M", "e字fi", "'VE", "ſ漢", " \n", "!a", "/bDž", "­🙂.", "𐞁", "-"]} +{"text": "\rEOT​Z<|endoftext|>́#$%/", "tokens": 15, "pieces": ["\r", "EOT", "​Z", "<|", "endoftext", "|>́#$%/"]} +{"text": "-字‍\r‍!­'re३ß½a/b'٣٤٥٦0iOSA🙂ḍ̇ꟲ㍿camelCase/🙂ḍ̇‍'ll\tſḍ̇>'re३-a /\r\n", "tokens": 78, "pieces": ["-字", "‍\r", "‍!­'", "re", "३", "ß", "½", "a", "/b", "'", "٣٤٥", "٦0", "iOSA", "🙂ḋ", "̣ꟲ", "㍿camelCase", "/🙂", "ḋ", "̣‍'", "ll", "\tſḋ", "̣>'", "re", "३", "-<", "META", "_START", ">a", " ", "/\r\n"]} +{"text": "😀🏽㋿­.éHTTPServer/\r\n9eABC!iOSİ'S12345678é'S-
  \n 's/​/-", "tokens": 36, "pieces": ["😀🏽㋿­.", "éHTTPServer", "/\r\n", "9", "eABC", "!iOSİ", "'S", "123", "456", "78", "é", "'S", "-", "
  \n", " '", "s", "/​/-"]} +{"text": "éAb\r\n/>BEOT!!sDžå0're\n'S\r\n\r\n\n/😀🏽 \n <|endoftext|>\u000b<'T'Re'TaBß \n<|endoftext|>/ \n㍿\u000b", "tokens": 58, "pieces": ["e", "́Ab", "\r\n", "/>", "BEOT", "!!", "sDža", "̊", "0", "'re", "\n", "'S", "\r\n\r\n\n", "/😀🏽", " \n", " <|", "endoftext", "|>", "\u000b", "<'", "T", "'Re", "'T", "aBß", " \n", "<|", "endoftext", "|>/", " \n", "㍿", "\u000b"]} +{"text": " 0,é'sé\r\n\r\nABC's\r\n\rⅣte!A ㋿", "tokens": 23, "pieces": [" ", " ", "0", ",e", "́'", "se", "́\r\n\r\n", "ABC", "'s", "\r\n\r", "Ⅳ", "te", "!A", " ", "㋿"]} +{"text": "ma/b字Džungla😀🏽B/\r\n🙂 td𐞁 \"#$%ABC'ssB\"'T\r\n'M", "tokens": 33, "pieces": ["ma", "/b字Džungla", "😀🏽", "B", "/\r\n", "🙂", " td𐞁", " \"#$%", "ABC", "'s", "sB", "\"'", "T", "\r\n", "'M"]} +{"text": "٣٤٥٦0㍿ \n", "tokens": 13, "pieces": ["٣٤٥", "٦0", "㍿", " \n"]} +{"text": "'T\"å \n aBDžZ ́İ
'Re\r\n'S' 𐞁½\nå 漢s\t0\r\n<‍'ll'll٣٤٥٦㍿tHTTPServerfi", "tokens": 62, "pieces": ["'T", "\"a", "̊", " \n", " aBDžZ", " ́", "İ", "
", "'Re", "\r\n", "'S", "'", " ", " 𐞁", "½", "\n", "a", "̊", " 漢s", "\t", "0", "\r\n", "<<", "EOT", ">‍'", "ll", "'ll", "٣٤٥", "٦", "㍿tHTTPServerfi"]} +{"text": "İå㋿", "tokens": 7, "pieces": ["İa", "̊㋿"]} +{"text": "'VE're'ſſEOT\r\n\r\n", "tokens": 14, "pieces": ["'VE", "'re", "'", "ſſEOT", "\r\n\r\n"]} +{"text": "Bs ع>ſ\r\nd字漢", "tokens": 11, "pieces": ["Bs", " ع", ">ſ", "\r\n", "d字漢"]} +{"text": "Abḍ̇½mfi'retDž\r\ńss\r\n🙂", "tokens": 20, "pieces": ["Abḋ", "̣", "½", "mfi", "'re", "tDž", "\r\n", "́ss", "\r\n", "🙂"]} +{"text": "𐞁 \n ㋿😀🏽 \n aB0 \n EOTEOTꟲ áſſ٣٤٥٦​́! 'M\u000b ㍿", "tokens": 51, "pieces": ["𐞁", " \n", " ㋿😀🏽", " \n", " aB", "0", " \n", " EOTEOTꟲ", " a", "́ſſ", "٣٤٥", "٦", "​́!", " ", " '", "M", "\u000b", " ", "㍿"]} +{"text": "㍿('T\r<|endoftext|>,é'T'Std<…", "tokens": 20, "pieces": ["㍿('", "T", "\r", "<|", "endoftext", "|>,", "é", "'T", "'S", "td", "<", "…"]} +{"text": "(.'re're٣٤٥٦ \n 'sEOTⅣ'll㍿aBⅣ!'Re\n\t .", "tokens": 32, "pieces": ["(.'", "re", "'re", "٣٤٥", "٦", " \n", " '", "sEOT", "Ⅳ", "'ll", "㍿aB", "Ⅳ", "!'", "Re", "\n", "\t", " ."]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ZaB/\r\nZmm​' 's'M!́'s'ſ\r\n👍🏽İAbfiع👍🏽Z'ReéHTTPServer( Z字ᵃ9½!!aaBAb\n/'re", "tokens": 58, "pieces": ["ZaB", "/\r\n", "Zmm", "​'", " ", " '", "s", "'M", "!́'", "s", "'ſ", "\r\n", "👍🏽", "İAbfiع", "👍🏽", "Z", "'Re", "éHTTPServer", "(", " ", " Z字ᵃ", "9½", "!!", "aaBAb", "\n", "/'", "re"]} +{"text": ">ß'S12345678>👍🏽Abᵃ!fi👍🏽<|endoftext|>\r12345678­a/b🙂\"sHTTPServer\u000b#$%/ ß\t", "tokens": 52, "pieces": [">ß", "'S", "123", "456", "78", ">👍🏽", "Abᵃ", "!fi", "👍🏽<|", "endoftext", "|>\r", "123", "456", "78", "­a", "/b", "🙂\"", "sHTTPServer", "\u000b", "#$%/", " ß", "\t"]} +{"text": "t'!!Džungladß12345678", "tokens": 11, "pieces": ["t", "'!!", "Džungladß", "123", "456", "78"]} +{"text": " 'Re👍🏽 \n 'Dḍ̇ 'ſ'll \r\n\r\n\"🙂́'!!!", "tokens": 29, "pieces": [" ", " '", "Re", "👍🏽", " \n", " '", "Dḋ", "̣", " ", "'ſ", "'ll", " \r\n\r\n", "\"🙂́'!!!"]} +{"text": "fi<|endoftext|>ſ'så\r\n<|endoftext|>𐞁𐞁m½e𐞁d́½Ab/<|fim_prefix|>'Td'MaB0​(㍿'ſ'D'S'Re'Re­ſ́", "tokens": 76, "pieces": ["fi", "<|", "endoftext", "|>", "ſ", "'s", "a", "̊\r\n", "<|", "endoftext", "|>", "𐞁𐞁m", "½", "e𐞁", "d", "́", "½", "Ab", "/<|", "fim", "_prefix", "|>'", "Td", "'M", "aB", "0", "​(㍿'", "ſ", "'D", "'S", "'Re", "'Re", "­ſ", "́"]} +{"text": "aB\r\n\r\n㍿İ\r\n\r\n‍EOT \n d Z\r\n\r\n… \n \n's'Re0\tİ\n/Ab'll字ß\n/'D", "tokens": 34, "pieces": ["aB", "\r\n\r\n", "㍿İ", "\r\n\r\n", "‍EOT", " \n", " d", " Z", "\r\n\r\n… \n \n", "'s", "'Re", "0", "\tİ", "\n", "/Ab", "'ll", "字ß", "\n", "/'", "D"]} +{"text": "'D!12345678𐞁\r\n\r\na'DA'VE😀🏽'S(­
🙂mß\"٣٤٥٦ Džungla\u000b\"", "tokens": 47, "pieces": ["'D", "!", "123", "456", "78", "𐞁", "\r\n\r\n", "a", "'D", "A", "'VE", "😀🏽'", "S", "(­", "
", "🙂mß", "\"", "٣٤٥", "٦", " Džungla", "\u000b", "\""]} +{"text": "ea/b're👍🏽عåtaB<|endoftext|>As½㋿-Ab🙂,Zfi\n\r\n\r\n‍㍿'re<|fim_prefix|>­0\n/#$%İ're", "As", "½", "㋿-", "Ab", "🙂,", "Zfi", "\n\r\n\r\n", "‍㍿'", "re", "<|", "fim", "_prefix", "|>­", "0", "\n", "/#$%", "İ", "'re", ",e're're", "tokens": 19, "pieces": [" ", " Abe", "́ᵃꟲ", "<|", "fim", "_prefix", "|>,", "e", "'re", "'re"]} +{"text": " 'llA<|endoftext|>Ab'll'sDž 'M#$%'D9#$%mᵃ/‍ſ'Re", "tokens": 37, "pieces": [" ", "'ll", "A", "<|", "endoftext", "|>", "Ab", "'ll", "'s", "Dž", "", " ", "'M", "#$%'", "D", "9", "#$%", "mᵃ", "/‍", "ſ", "'Re"]} +{"text": "ꟲ\n9\r\n\r\né!'DiOS👍🏽//s\r'reᵃḍ̇'Ree(aAb'DᵃB\r 𐞁🙂!!", "tokens": 47, "pieces": ["ꟲ", "\n", "9", "\r\n\r\n", "é", "!'", "DiOS", "👍🏽//", "s", "\r", "'re", "ᵃḋ", "̣'", "Ree", "(aAb", "'D", "ᵃB", "\r", " 𐞁", "🙂!!"]} +{"text": "‍AbſcamelCase𐞁0\r\n\r\n12345678ḍ̇e!ḍ̇Bᵃ'Rea/b", "tokens": 35, "pieces": ["‍AbſcamelCase𐞁", "0", "\r\n\r\n", "123", "456", "78", "ḋ", "̣e", "!ḋ", "̣Bᵃ", "'Re", "a", "/b"]} +{"text": "'re…㋿​", "tokens": 7, "pieces": ["'re", "…", "㋿​"]} +{"text": " EOT're's'D'VEİ
s\t\u000b.'llDž,😀🏽", "tokens": 23, "pieces": [" EOT", "'re", "'s", "'D", "'VE", "İ", "
s", "\t", "\u000b", ".'", "llDž", ",😀🏽"]} +{"text": "👍🏽ḍ̇'D", "tokens": 13, "pieces": ["👍🏽", "ḋ", "̣'", "D"]} +{"text": "/Aa/b 'M🙂é字😀🏽'ſ\t漢 字­😀🏽m३'D\u000bm​ 字'M\"ꟲ'M<|endoftext|> 'D<|endoftext|>å", "tokens": 69, "pieces": ["/Aa", "/b", " ", "'M", "🙂e", "́字", "😀🏽'", "ſ", "\t漢", " ", " 字", "­😀🏽", "m", "३", "'", "D", "\u000bm", "​", " 字", "'M", "\"ꟲ", "'M", "<|", "endoftext", "|>", " '", "D", "<|", "endoftext", "|>", "a", "̊"]} +{"text": "é½\r\n\r\n\r\n<|fim_prefix|>́Dž३'TcamelCase", "tokens": 19, "pieces": ["e", "́", "½", "\r\n\r\n\r\n", "<|", "fim", "_prefix", "|>́", "Dž", "३", "'T", "camelCase"]} +{"text": "d٣٤٥٦'ſ\u000b​>३
'M", "tokens": 23, "pieces": ["d", "٣٤٥", "٦", "'ſ", "\u000b", "​>", "३", "
", "'M", ""]} +{"text": "\n'reع字BſéficamelCase👍🏽'ABC,9aB \n𐞁
\n/fi're", "tokens": 43, "pieces": ["\n", "'re", "ع字Bſe", "́ficamelCase", "👍🏽<", "META", "_START", ">'", "ABC", ",", "9", "aB", " \n", "𐞁", "
\n", "/fi", "'re"]} +{"text": "/\r\n's<|endoftext|>'s½…fia/b!!t'ſ9'\r\n é'SaBm\u000bZ'0é
<|fim_prefix|>d\rA㋿", "tokens": 55, "pieces": ["/\r\n", "'s", "<|", "endoftext", "|>'", "s", "½", "…fia", "/b", "!!", "t", "'", "ſ", "9", "'\r\n", " ", " e", "́'", "SaBm", "\u000bZ", "'", "0", "é", "
", "<|", "fim", "_prefix", "|>", "d", "\r", "A", "㋿"]} +{"text": "ḍ̇漢fiḍ̇0--ꟲ\r​12345678ßİ å<|fim_prefix|>DžunglaHTTPServer'Re\u000bAb👍🏽('e'T'T\n!A㍿㍿­", "tokens": 67, "pieces": ["ḋ", "̣漢fiḋ", "̣", "0", "--", "ꟲ", "\r", "​", "123", "456", "78", "ßİ", " a", "̊<|", "fim", "_prefix", "|>", "DžunglaHTTPServer", "'Re", "\u000bAb", "👍🏽('", "e", "'T", "'T", "\n", "!A", "㍿㍿­"]} +{"text": "a/b- 𐞁\n/ t.fi e", "tokens": 16, "pieces": ["a", "/b", "-", " 𐞁", "\n", "/", " t", ".fi", " e"]} +{"text": "ᵃ!!'Re--. Z", "tokens": 10, "pieces": ["ᵃ", "!!'", "Re", "--.", " Z"]} +{"text": "İZDžungla㋿(å㍿iOS🙂aß३", "tokens": 23, "pieces": ["İZDžungla", "㋿(", "a", "̊㍿", "iOS", "🙂aß", "३"]} +{"text": "३å'M<|fim_prefix|>EOT'S
!!👍🏽é! ", "tokens": 29, "pieces": ["३", "a", "̊'", "M", "<|", "fim", "_prefix", "|>", "EOT", "'S", "
", "!!👍🏽", "é", "!", " "]} +{"text": "­‍\"\n/EOT'll'D'llⅣ", "tokens": 11, "pieces": ["­‍\"\n", "/EOT", "'ll", "'D", "'ll", "Ⅳ"]} +{"text": "'\n‍'ll\n/t👍🏽 \n HTTPServer٣٤٥٦'T字🙂9\n", "tokens": 38, "pieces": ["'\n", "‍'", "ll", "\n", "/t", "👍🏽<", "EOT", ">", " \n", " <", "META", "_START", ">HTTPServer", "٣٤٥", "٦", "'T", "字", "🙂", "9", "\n"]} +{"text": "a/bİ\tİta/b'DcamelCase'll'ſ \n \r\n\r\n", "tokens": 16, "pieces": ["a", "/bİ", "\tİta", "/b", "'D", "camelCase", "'ll", "'ſ", " \n \r\n\r\n"]} +{"text": "\"́/ᵃ㋿​ >٣٤٥٦", "tokens": 23, "pieces": ["\"́/", "ᵃ", "㋿<", "EOT", ">​", " >", "٣٤٥", "٦"]} +{"text": "'VEABC's(e#$%<|endoftext|>𐞁'Da", "tokens": 20, "pieces": ["'VE", "ABC", "'s", "(e", "#$%<|", "endoftext", "|>", "𐞁", "'D", "a"]} +{"text": "t\r!!'sEOT'llع'T🙂😀🏽字​, ‍\r\n\r\nḍ̇\r\n\r\nB/'re<'ſm", "tokens": 47, "pieces": ["t", "\r", "!!'", "s", "EOT", "'ll", "ع", "'", "T", "🙂😀🏽", "字", "​,", " ", "‍<", "EOT", ">\r\n\r\n", "ḋ", "̣\r\n\r\n", "B", "/'", "re", "<'", "ſm"]} +{"text": "ع٣٤٥٦\r漢Ab<|fim_prefix|><|fim_prefix|>><|endoftext|> \n 
👍🏽d½㍿<'sZ'M'll./\r\n're é", "tokens": 61, "pieces": ["ع", "٣٤٥", "٦", "\r", "漢Ab", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>><|", "endoftext", "|>", " \n", " ", "
", "👍🏽", "d", "", "½", "㍿<'", "sZ", "'M", "'ll", "./\r\n", "'re", " ", " é"]} +{"text": "Dž'MAb", "tokens": 4, "pieces": ["Dž", "'M", "Ab"]} +{"text": "<\rꟲ٣٤٥٦a/bå'refiDžunglaᵃ0camelCaseḍ̇''ll'SⅣ'sⅣé<|endoftext|>/\r\n", "tokens": 54, "pieces": ["<\r", "ꟲ", "٣٤٥", "٦", "a", "/ba", "̊'", "refiDžunglaᵃ", "0", "camelCaseḋ", "̣''", "ll", "'S", "Ⅳ", "'s", "Ⅳ", "é", "<|", "endoftext", "|>/\r\n"]} +{"text": "/\r\n  -\u000b(<|fim_prefix|>emHTTPServer#$%ḿ½​#$%", "tokens": 23, "pieces": ["/\r\n", " ", " ", "-", "\u000b", "(<|", "fim", "_prefix", "|>", "emHTTPServer", "#$%", "m", "́", "½", "​#$%"]} +{"text": "é/0/\r\n!!a'll \nꟲ(Ab/٣٤٥٦'reå🙂<|fim_prefix|>'Re EOT'M\r\n\r\n/\r\n­ABC#$%ßHTTPServer'Deſ", "tokens": 54, "pieces": ["e", "́/", "0", "/\r\n", "!!", "a", "'ll", " \n", "ꟲ", "(Ab", "/", "٣٤٥", "٦", "'re", "a", "̊🙂<|", "fim", "_prefix", "|>'", "Re", " EOT", "'M", "\r\n\r\n", "/\r\n", "­ABC", "#$%", "ßHTTPServer", "'D", "eſ"]} +{"text": "'reABC'reHTTPServerHTTPServerfi
'M 'Tİ\"
'M…A'S\ré٣٤٥٦\r\n\r\nfi­\r0Zᵃ́å!!'ſcamelCase.camelCase#$%>'Så", "tokens": 63, "pieces": ["'re", "ABC", "'re", "HTTPServerHTTPServerfi", "
", "'M", " ", "'T", "İ", "\"", "
", "'M", "…A", "'S", "\r", "e", "́", "٣٤٥", "٦", "\r\n\r\n", "fi", "­\r", "0", "Zᵃ", "́a", "̊!!'", "ſcamelCase", ".camelCase", "#$%>'", "Sa", "̊"]} +{"text": "'t \n 'ſع'D½ع漢12345678­عå😀🏽😀🏽'", "tokens": 40, "pieces": ["'t", " \n", " '", "ſع", "'D", "½", "ع漢", "123", "456", "78", "­ع", "a", "̊😀🏽😀🏽'"]} +{"text": ".٣٤٥٦DžABC <😀🏽‍,B9\"Ae, \n ", "tokens": 31, "pieces": [".", "٣٤٥", "٦", "DžABC", " ", " <😀🏽‍,", "B", "9", "\"Ae", ",", " \n "]} +{"text": "\r\n\rße\u000bⅣ字-\r\n\r\n'Mḍ̇#$%12345678.३',ᵃ👍🏽då'ᵃ<-\r\n\r\nꟲ-'ſ's👍🏽(", "tokens": 56, "pieces": ["\r\n\r", "ße", "\u000b", "Ⅳ", "字", "-\r\n\r\n", "'M", "ḋ", "̣#$%", "123", "456", "78", ".", "३", "',", "ᵃ", "👍🏽", "da", "̊'", "ᵃ", "<-\r\n\r\n", "ꟲ", "-'", "ſ", "'s", "👍🏽("]} +{"text": "Båß\r!!ꟲ  ᵃ😀🏽😀🏽\"0- 'M", "tokens": 28, "pieces": ["Ba", "̊ß", "\r", "!!", "ꟲ", " ", " ᵃ", "😀🏽😀🏽\"", "0", "-", " ", "'M"]} +{"text": "'ReABCå'DAa İ12345678/\r\n9", "tokens": 16, "pieces": ["'Re", "ABCa", "̊'", "DAa", " İ", "123", "456", "78", "/\r\n", "9"]} +{"text": "'VEİßꟲ((​😀🏽<|fim_prefix|>AZ'Dß<|fim_prefix|>'ſ,'re#$%", "tokens": 42, "pieces": ["'VE", "İßꟲ", "((​😀🏽<|", "fim", "_prefix", "|>", "AZ", "'D", "ß", "<|", "fim", "_prefix", "|>'", "ſ", ",'", "re", "#$%"]} +{"text": ",\n \ndſ'D😀🏽#$%/\r\nss/e>字fia/b/Džungla\n/ 🙂", "tokens": 32, "pieces": [",\n", " \n", "dſ", "'D", "😀🏽#$%/\r\n", "ss", "/e", ">字fia", "/b", "/Džungla", "\n", "/", " ", "🙂"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r​!!'resA…İ٣٤٥٦Ⅳꟲ''s½\tZ\"9३३\t\"İ\r\n\r\n'll(\n", "tokens": 39, "pieces": ["\r", "​!!'", "resA", "…İ", "٣٤٥", "٦Ⅳ", "ꟲ", "''", "s", "½", "\tZ", "\"", "9३३", "\t", "\"İ", "\r\n\r\n", "'ll", "(\n"]} +{"text": "så  \n ſⅣ/\r\n\u000b\"🙂're ㋿ \n - \n ZA Ab
d,🙂🙂'Re'Ma/b字<|fim_prefix|>'VEⅣ😀🏽\r\n\r\n\r\n<|fim_prefix|>İ\u000b<12345678, st", "tokens": 62, "pieces": ["'𐞁", "Z", "A", " ", " Ab", "
d", ",🙂🙂'", "Re", "'M", "a", "/b字", "<|", "fim", "_prefix", "|>'", "VE", "Ⅳ", "😀🏽\r\n\r\n\r\n", "<|", "fim", "_prefix", "|>", "İ", "\u000b", "<", "123", "456", "78", ",", " ", " st"]} +{"text": "camelCasedåABCDž𐞁'VÉ'Mع😀🏽", "tokens": 24, "pieces": ["camelCaseda", "̊ABCDž𐞁", "'VE", "́'", "Mع", "😀🏽"]} +{"text": "Z\r\n😀🏽ABC'S .'Sddḍ̇'Res½'SiOS\nABC<|endoftext|>a/b12345678-İ \r\n\r\n👍🏽 ꟲ'T\"EOTé", "tokens": 58, "pieces": ["Z", "\r\n", "😀🏽", "ABC", "'S", " ", " .'", "Sddḋ", "̣'", "Res", "", "½", "'S", "iOS", "\n", "ABC", "<|", "endoftext", "|>", "a", "/b", "123", "456", "78", "-İ", " \r\n\r\n", "👍🏽", " ꟲ", "'T", "\"EOTé"]} +{"text": "\tİB'ſ😀🏽<|endoftext|>İHTTPServer\r\nt٣٤٥٦'VE's㍿
🙂\r\n​'Re\n", "İHTTPServer", "\r\n", "t", "٣٤٥", "٦", "'VE", "'s", "㍿", "
", "🙂\r\n", "​'", "Re", "\n", "३字Džungla'S㋿'Re/…㍿tABCå'S<|fim_prefix|>३,́.aB'VE \n‍e​ 𐞁", "tokens": 58, "pieces": ["", "३", "字Džungla", "'S", "㋿'", "Re", "/", "…", "㍿tABC", "a", "̊'", "S", "<|", "fim", "_prefix", "|>", "३", ",́.", "aB", "'VE", " \n", "‍e", "​", " 𐞁"]} +{"text": " \tZaBcamelCaset'ſ,ſ(mßå/12345678㍿½
Z'red", "tokens": 34, "pieces": [" ", "\t", "ZaBcamelCaset", "'ſ", ",ſ", "(mßa", "̊/", "123", "456", "78", "㍿", "½", "
Z", "'re", "d"]} +{"text": "'🙂camelCase<|fim_prefix|>", "tokens": 12, "pieces": ["'🙂", "camelCase", "<|", "fim", "_prefix", "|>"]} +{"text": "camelCaseḍ̇-ſ/\r\nåaB'ſ\u000b㍿㋿camelCase
'‍'.Z/\r\n-iOS#$%\t'S're㍿ḍ̇ fi‍\"", "tokens": 67, "pieces": ["camelCaseḋ", "̣-", "ſ", "/\r\n", "a", "̊<", "META", "_START", ">aB", "'ſ", "\u000b", "㍿㋿", "camelCase", "
", "'‍'.", "Z", "/\r\n", "-<", "META", "_START", ">iOS", "#$%", "\t", "'S", "'re", "㍿ḋ", "̣", " ", " fi", "‍\""]} +{"text": "aB's''  A'/\r\nA'.", "tokens": 13, "pieces": ["aB", "'s", "''", " ", " A", "'/\r\n", "A", "'."]} +{"text": "'ſ  'llå‍ſABC'S😀🏽12345678-㋿Bm /\r\n…AbB㋿\r\n\r\n​é 'Ś‍\rDžungla字å", "tokens": 57, "pieces": ["'ſ", " ", " ", "'ll", "a", "̊‍", "ſABC", "'S", "😀🏽", "123", "456", "78", "-㋿", "Bm", " ", " /\r\n", "…AbB", "㋿\r\n\r\n", "​e", "́", " ", "'S", "́‍\r", "Džungla字a", "̊"]} +{"text": "'S\r\n​‍ 'så<|endoftext|>٣٤٥٦å!!\t/\r\n
A!ḍ̇scamelCase(éꟲé½/\r\nꟲß<|endoftext|>å🙂\n/#$%ß", "tokens": 73, "pieces": ["'S", "\r\n", "​‍", " ", " '", "sa", "̊<|", "endoftext", "|>", "٣٤٥", "٦", "a", "̊!!", "\t", "/\r\n", "
A", "!ḋ", "̣scamelCase", "(éꟲé", "½", "/\r\n", "ꟲß", "<|", "endoftext", "|>", "a", "̊🙂\n", "/#$%", "ß"]} +{"text": " \u000b/\r\n'll‍ſsß,३eéaB'ré'Sm㋿ /\r\n\n/Z😀🏽ſ\"iOS'㋿ᵃcamelCase𐞁 \n Aa/bEOTB!", "tokens": 63, "pieces": [" ", "\u000b", "/\r\n", "'ll", "‍ſsß", ",", "३", "ee", "́aB", "'re", "́'", "Sm", "㋿", " ", "/\r\n\n", "/Z", "😀🏽", "ſ", "\"iOS", "'㋿", "ᵃcamelCase𐞁", " \n", " Aa", "/bEOTB", "!"]} +{"text": "\n\taB,s9d Ⅳ\nع\n🙂…ABC…é/\r\nm㍿tt'ſ9é­#$%a's<|fim_prefix|>åABCعꟲḍ̇𐞁", "tokens": 61, "pieces": ["\n", "\taB", ",s", "9", "d", " ", "Ⅳ", "\n", "ع", "\n", "🙂", "…ABC", "…é", "/\r\n", "m", "㍿tt", "'ſ", "9", "e", "́­#$%", "a", "'s", "<|", "fim", "_prefix", "|>", "a", "̊ABCعꟲḋ", "̣𐞁"]} +{"text": "(iOS\r\n fi/'T字", "tokens": 15, "pieces": ["(iOS", "\r\n", "", " fi", "/'", "T字"]} +{"text": "𐞁'D'ſ🙂/­", "tokens": 19, "pieces": ["𐞁", "'", "D", "'ſ", "🙂<", "EOT", ">/­"]} +{"text": " åEOT camelCase­A\r\nm\r\nd", "tokens": 16, "pieces": [" ", " a", "̊EOT", " ", " camelCase", "­A", "\r\n", "m", "\r\n", "d"]} +{"text": "12345678'T'S,", "tokens": 6, "pieces": ["123", "456", "78", "'T", "'S", ","]} +{"text": "camelCase‍!'Re<|endoftext|>0Be('", "Re", "<|", "endoftext", "|>", "0", "Be", "(<", "m"]} +{"text": "İ'T \n/ABCm٣٤٥٦🙂ꟲ", "tokens": 19, "pieces": ["İ", "'T", " \n", "/ABCm", "٣٤٥", "٦", "🙂ꟲ"]} +{"text": "!0DžcamelCase٣٤٥٦a/bm 字a/b
ſ\r\n.'Ⅳ\r\n\r\n \na/b!İ\nA's'll<|fim_prefix|><|fim_prefix|>'Re'sꟲ<ꟲABCꟲEOT'D", "tokens": 73, "pieces": ["!", "0", "DžcamelCase", "٣٤٥", "٦", "a", "/bm", " ", " 字a", "/b", "", "
ſ", "\r\n", ".'", "Ⅳ", "\r\n\r\n \n", "a", "/b", "!İ", "\n", "A", "'s", "'ll", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>'", "Re", "'s", "ꟲ", "<ꟲABCꟲEOT", "'D"]} +{"text": " \n🙂!!", "tokens": 4, "pieces": [" \n", "🙂!!"]} +{"text": "m'VE 'll
.'Re<'ll m", "tokens": 17, "pieces": ["m", "'VE", " ", " '", "ll", "
", ".'", "Re", "<'", "ll", "", " ", " m"]} +{"text": "9d#$%Ⅳ!\t12345678𐞁\n/\r😀🏽12345678ع\ta,a/b'D!!漢#$%\"ſ", "tokens": 38, "pieces": ["9", "d", "#$%", "Ⅳ", "!", "\t", "123", "456", "78", "𐞁", "\n", "/\r", "😀🏽", "123", "456", "78", "ع", "\ta", ",a", "/b", "'D", "!!", "漢", "#$%\"", "ſ"]} +{"text": "e'M! \n\r\nſḍ̇'DDž́ABCABC A'Ret'Sa's<ꟲݽs", "tokens": 34, "pieces": ["e", "'M", "!", " \n\r\n", "ſḋ", "̣'", "DDž", "́ABCABC", " A", "'Re", "t", "'S", "a", "'s", "<ꟲİ", "½", "s"]} +{"text": "<|endoftext|> /-🙂<", "tokens": 12, "pieces": ["<|", "endoftext", "|>", " ", "/-🙂<"]} +{"text": "s\u000b", "tokens": 2, "pieces": ["s", "\u000b"]} +{"text": " \n 'D'reİ𐞁a/b <|fim_prefix|>½́dm<|endoftext|>字\n/é0fi字\u000bHTTPServer漢'T ­🙂/9å ᵃ\u000b", "tokens": 61, "pieces": [" \n", " '", "D", "'re", "İ𐞁a", "/b", " ", "<|", "fim", "_prefix", "|>", "½", "́dm", "<|", "endoftext", "|>", "字", "\n", "/é", "0", "fi字", "\u000bHTTPServer漢", "'T", " ", " <", "EOT", ">­🙂/", "9", "a", "̊", " ᵃ", "\u000b"]} +{"text": "‍­", "tokens": 3, "pieces": ["‍­"]} +{"text": "iOS٣٤٥٦\r\n\r\n\r,\r \u000b's٣٤٥٦ßß­ \nABC漢😀🏽🙂'VE字0३a'sḍ̇\u000bHTTPServera३字\r\n\r\n<|endoftext|>
😀🏽٣٤٥٦'D㍿", "tokens": 85, "pieces": ["iOS", "٣٤٥", "٦", "\r\n\r\n\r", ",\r", " ", "\u000b", "'s", "٣٤٥", "٦", "ßß", "­", " \n", "ABC漢", "😀🏽🙂'", "VE字", "0३", "a", "'s", "ḋ", "̣", "\u000bHTTPServera", "३", "字", "\r\n\r\n", "<|", "endoftext", "|>", "
", "😀🏽", "٣٤٥", "٦", "'D", "㍿"]} +{"text": "\n/<|fim_prefix|>EOT 'M'D٣٤٥٦EOT.A'DⅣ\t!!< \n/d<0,👍🏽<|endoftext|>t-BaBZABC", "tokens": 56, "pieces": ["\n", "/<|", "fim", "_prefix", "|>", "EOT", " '", "M", "'D", "٣٤٥", "٦", "EOT", ".A", "'D", "Ⅳ", "\t", "!!<", " \n", "/d", "<", "0", ",👍🏽<|", "endoftext", "|>", "t", "-BaBZABC"]} +{"text": "EOTdsa/b🙂\r \n\r\n\r\n0'M", "tokens": 11, "pieces": ["EOTdsa", "/b", "🙂\r", " \n\r\n\r\n", "0", "'M"]} +{"text": "é​𐞁/\r\n-,fiA٣٤٥٦<|endoftext|>m​é\r>-d'Re'ſ>aBDž'\nABC", "tokens": 44, "pieces": ["é", "​𐞁", "/\r\n", "-,", "fiA", "٣٤٥", "٦", "<|", "endoftext", "|>", "m", "​e", "́\r", ">-", "d", "'Re", "'ſ", ">aBDž", "'\n", "ABC"]} +{"text": "ſ>'sḿ'rea👍🏽\"iOS12345678", "tokens": 18, "pieces": ["ſ", ">'", "sm", "́'", "rea", "👍🏽\"", "iOS", "123", "456", "78"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "AcamelCase", "tokens": 4, "pieces": ["AcamelCase"]} +{"text": "Zs😀🏽\r\nd-09 !\r0EOT'McamelCase\u000b­३\r\n'Re \n <|fim_prefix|>👍🏽é", "tokens": 43, "pieces": ["Zs", "😀🏽\r\n", "d", "-", "09", "", " ", " !\r", "0", "EOT", "'M", "camelCase", "\u000b", "­", "३", "\r\n", "'Re", " \n", " <|", "fim", "_prefix", "|>👍🏽", "é"]} +{"text": "Z9Afi٣٤٥٦\r…å३", "tokens": 22, "pieces": ["Z", "9", "Afi", "٣٤٥", "٦", "\r", "…a", "̊", "३"]} +{"text": "#$% 12345678'll ३'s<|fim_prefix|><|endoftext|>𐞁
'RefiHTTPServerſ \n Bİ​ ‍åſa'T12345678eå>'re
\u000b\n/ /\r\n", "tokens": 67, "pieces": ["#$%", " ", "123", "456", "78", "'ll", " ", "३", "'s", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "𐞁", "
", "'Re", "fiHTTPServerſ", " \n", " Bİ", "​", " ", " ‍", "a", "̊ſa", "'T", "123", "456", "78", "ea", "̊>'", "re", "
\u000b\n", "/", " ", " /\r\n"]} +{"text": "㍿ 'VE😀🏽½/>'ſ字m'T字Dž㋿ḍ̇'ſ!'ſ½İ ḍ̇‍fi🙂tss\t👍🏽\"'!!", "tokens": 65, "pieces": ["㍿", " ", "'VE", "😀🏽", "½", "/>'", "ſ字m", "'T", "字", "Dž", "㋿ḋ", "̣'", "ſ", "!'", "ſ", "½", "İ", " ḋ", "̣‍", "fi", "🙂tss", "\t", "👍🏽\"'!!"]} +{"text": "/a/b \"𐞁Bd''VE0Ⅳᵃᵃ'Mé'Sعm#$%'VE's👍🏽…m\n!ᵃ<|fim_prefix|>aB½/\r\n 'T,
", "tokens": 63, "pieces": ["/a", "/b", " \"", "𐞁Bd", "''", "VE", "0Ⅳ", "ᵃᵃ", "'M", "e", "́'", "Sعm", "#$%'", "VE", "'s", "👍🏽", "…m", "\n", "!ᵃ", "<|", "fim", "_prefix", "|><", "EOT", ">aB", "½", "/\r\n", " '", "T", ",", "
"]} +{"text": "<|fim_prefix|>aa/b<😀🏽'M½/\r\n­\r'ſABC​é‍!!­\r\nHTTPServeŕ.Z漢s/ \n ", "tokens": 44, "pieces": ["<|", "fim", "_prefix", "|>", "aa", "/b", "<😀🏽'", "M", "½", "/\r\n", "­\r", "'ſ", "ABC", "​e", "́‍!!­\r\n", "HTTPServer", "́.", "Z漢s", "/", " \n "]} +{"text": "0­٣٤٥٦Džungla𐞁'ſAeABC're'VEع.(t'rea/b'ſ(…", "tokens": 40, "pieces": ["0", "­", "٣٤٥", "٦", "Džungla𐞁", "'ſ", "AeABC", "'re", "'VE", "ع", ".(", "t", "'re", "a", "/b", "'ſ", "(", "…"]} +{"text": "m iOS'TA12345678\u000b
'Re\r\n\r\ncamelCaseé٣٤٥٦DžDž0,́!!👍🏽å'D ½s
9s㋿ABCDž'ſAb(­", "tokens": 73, "pieces": ["m", " iOS", "'T", "A", "123", "456", "78", "\u000b", "
", "'Re", "\r\n\r\n", "camelCaseé", "٣٤٥", "٦", "DžDž", "0", ",́!!👍🏽", "a", "̊<", "EOT", "><", "META", "_START", ">'", "D", " ", " ", "½", "s", "
", "9", "s", "㋿", "ABCDž", "'ſ", "Ab", "(­"]} +{"text": "é'­
#$%é\n/٣٤٥٦a dDžungla", "tokens": 26, "pieces": ["e", "́'­", "
", "#$%", "é", "\n", "/", "٣٤٥", "٦", "a", " ", " dDžungla"]} +{"text": "‍#$%👍🏽 \nİ/ſ㋿<\r\n\r\n\r\nİ,éaBADžungla\r\n\r\nAßABCaB\r\n('S'McamelCase…㋿😀🏽\t(9", "tokens": 57, "pieces": ["‍#$%👍🏽", " \n", "İ", "/ſ", "㋿<\r\n\r\n\r\n", "İ", ",éaBADžungla", "\r\n\r\n", "AßABCaB", "\r\n", "('", "S", "'M", "camelCase", "…", "㋿😀🏽", "\t", "(", "9"]} +{"text": "\ré\r\nA /\r'TEOT!0\"'VEꟲḍ̇㍿​‍㋿fi'T😀🏽.HTTPServer", "tokens": 45, "pieces": ["\r", "é", "\r\n", "A", " /\r", "'T", "EOT", "!", "0", "\"'", "VEꟲḋ", "̣㍿​‍㋿", "fi", "'T", "😀🏽.", "HTTPServer"]} +{"text": "'Re", "tokens": 1, "pieces": ["'Re"]} +{"text": "DžunglafiABC漢 ''M
\r\n\r\nfia/bᵃé\"'Re…٣٤٥٦/\r\nDž​'s!!m,B", "tokens": 44, "pieces": ["DžunglafiABC漢", " ", "''", "M", "
\r\n\r\n", "fia", "/bᵃé", "\"'", "Re", "…", "٣٤٥", "٦", "/\r\n", "Dž", "​'", "s", "!!", "m", ",B"]} +{"text": "漢éßtiOS½\n/t/A<|endoftext|>​é.\t \n Dž㋿.>'S㋿
> A\tåaB>'Dfi9", "tokens": 51, "pieces": ["漢e", "́ßtiOS", "½", "\n", "/t", "/A", "<|", "endoftext", "|>​", "é", ".", "\t \n", " Dž", "㋿.>'", "S", "㋿", "
", ">", " A", "\ta", "̊aB", ">'", "Dfi", "9"]} +{"text": "iOS're(A<|endoftext|>sEOT  \u000b😀🏽.'Reſ \n­㍿#$%Ⅳİꟲ!<|endoftext|>½ß漢HTTPServer!!<|endoftext|>12345678é\u000b/<|endoftext|>", "tokens": 76, "pieces": ["iOS", "'re", "(A", "<|", "endoftext", "|>", "s", "EOT", "  ", "\u000b", "😀🏽.'", "Reſ", " \n", "­㍿#$%", "Ⅳ", "İꟲ", "!<|", "endoftext", "|>", "½", "ß漢HTTPServer", "!!<|", "endoftext", "|>", "123", "456", "78", "é", "\u000b", "/<|", "endoftext", "|>"]} +{"text": "ABCDžungla३…\n", "tokens": 10, "pieces": ["ABCDžungla", "३", "…\n"]} +{"text": "t\u000b'll!camelCase…EOT,m'S/\r\nA \n ABC'TeⅣ/​漢s'Mİ-👍🏽‍'TcamelCase", "tokens": 48, "pieces": ["t", "\u000b", "'ll", "!camelCase", "…EOT", ",m", "'S", "/\r\n", "A", " \n", " ABC", "'T", "e", "Ⅳ", "/​<", "META", "_START", "><", "META", "_START", ">漢s", "'M", "İ", "-👍🏽‍'", "TcamelCase"]} +{"text": "…㍿Ⅳ,Džİ \ncamelCase'!!0́½ß \né\r< \n/", "tokens": 28, "pieces": ["…", "㍿", "Ⅳ", ",Džİ", " \n", "camelCase", "'!!", "0", "́", "½", "ß", " \n", "e", "́\r", "<", " \n", "/"]} +{"text": "12345678ſ!ABC0aBsdEOTHTTPServerA Ab-漢½/må", "tokens": 27, "pieces": ["123", "456", "78", "ſ", "!ABC", "0", "aBsdEOTHTTPServerA", " Ab", "-漢", "½", "/ma", "̊"]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "aB", "tokens": 2, "pieces": ["aB"]} +{"text": "漢 \n \"
A½ḍ̇<|fim_prefix|><|endoftext|>Džع9", "tokens": 34, "pieces": ["漢", "", " \n", " \"", "
A", "½", "ḋ", "̣<|", "fim", "_prefix", "|><|", "endoftext", "|>", "Džع", "9"]} +{"text": "ꟲ", "tokens": 3, "pieces": ["ꟲ"]} +{"text": "ḍ̇", "tokens": 5, "pieces": ["ḋ", "̣"]} +{"text": "e३'s'Re,é,at🙂/\r\n½iOS'VEfi-🙂aB\r\nſDžungla \n㋿((", "tokens": 38, "pieces": ["e", "३", "'s", "'Re", ",é", ",at", "🙂/\r\n", "½", "iOS", "'VE", "fi", "-🙂", "aB", "\r\n", "ſDžungla", " \n", "㋿<", "EOT", ">(("]} +{"text": "ع​/\r\n \n İ", "tokens": 5, "pieces": ["ع", "​/\r\n", " \n", " İ"]} +{"text": " 𐞁😀🏽​", "tokens": 12, "pieces": [" ", " 𐞁", "😀🏽​"]} +{"text": " ‍‍#$%\tétaB''M", "tokens": 13, "pieces": [" ", " ‍‍#$%", "\tétaB", "''", "M"]} +{"text": "9㍿es", "tokens": 5, "pieces": ["9", "㍿es"]} +{"text": "\u000b9'Re­\t''ſ'llḍ̇m\u000bDžunglaḍ̇ \n a", "tokens": 28, "pieces": ["\u000b", "9", "'Re", "­", "\t", "''", "ſ", "'ll", "ḋ", "̣m", "\u000bDžunglaḋ", "̣", " \n", " ", " a"]} +{"text": "camelCase9Dž12345678camelCaseAb'sAbABCcamelCase'/\r\nDžungla!!>Dž­e…A", "tokens": 32, "pieces": ["camelCase", "9", "Dž", "123", "456", "78", "camelCaseAb", "'s", "AbABCcamelCase", "'/\r\n", "Džungla", "!!>", "Dž", "­e", "…A"]} +{"text": "İ'VE,", "tokens": 4, "pieces": ["İ", "'VE", ","]} +{"text": "aB🙂…ABCع'ſ'DHTTPServer😀🏽as#$%\r'VE'a/b٣٤٥٦३d <|endoftext|>㍿#$%'Re's 😀🏽", "tokens": 56, "pieces": ["aB", "🙂", "…ABCع", "'ſ", "'D", "HTTPServer", "😀🏽", "as", "#$%\r", "'VE", "'a", "/b", "٣٤٥", "٦३", "d", " <|", "endoftext", "|>㍿#$%'", "Re", "'s", " ", " 😀🏽"]} +{"text": "aB ­ ㍿ét,#$%㍿s's\r\nḍ̇ß\r\n!9ᵃ", "tokens": 29, "pieces": ["aB", " ­", " ", "㍿ét", ",#$%㍿", "s", "'s", "\r\n", "ḋ", "̣ß", "\r\n", "!", "9", "ᵃ"]} +{"text": "\n/åéHTTPServer", "tokens": 7, "pieces": ["\n", "/a", "̊éHTTPServer"]} +{"text": "\ré字\n", "tokens": 4, "pieces": ["\r", "é字", "\n"]} +{"text": "EOTİ(́HTTPServer'Re> a
(\r\nfi 
/ iOS'D", "tokens": 23, "pieces": ["EOTİ", "(́", "HTTPServer", "'Re", ">", " a", "
", "(\r\n", "fi", " ", "
", "/", " iOS", "'D"]} +{"text": "EOT- \r\n\r\n\nt'MB​ſ'Reaé-
㍿Džungla. 0㍿é", "tokens": 33, "pieces": ["EOT", "-", " \r\n\r\n\n", "t", "'M", "B", "​ſ", "'Re", "ae", "́-", "
", "㍿Džungla", ".", " ", " ", "0", "㍿é"]} +{"text": "a12345678<|endoftext|>!!édDž\rAb \nB#$% 𐞁🙂𐞁#$%'ll0'ſ'D🙂'Re123456780👍🏽're \n", "tokens": 56, "pieces": ["a", "123", "456", "78", "<|", "endoftext", "|>!!", "édDž", "\r", "Ab", " \n", "B", "#$%", " 𐞁", "🙂𐞁", "#$%'", "ll", "0", "'ſ", "'D", "🙂'", "Re", "123", "456", "780", "👍🏽'", "re", " \n"]} +{"text": "\r>ABC٣٤٥٦\"٣٤٥٦½ 'sⅣEOT0­­ \n a(/\r\n🙂漢.- DžunglaB
 \n عß👍🏽ع/'s", "tokens": 65, "pieces": ["\r", "><", "META", "_START", ">ABC", "٣٤٥", "٦", "\"", "٣٤٥", "٦½", " ", "'s", "Ⅳ", "EOT", "0", "­­", " \n", " a", "(/\r\n", "🙂漢", ".-", " DžunglaB", "
 \n", " عß", "👍🏽", "ع", "/'", "s"]} +{"text": " 👍🏽
‍'ll.Džungla'ReABC12345678𐞁­Z\n/‍​'Re𐞁'M,", "tokens": 44, "pieces": [" ", " 👍🏽", "
", "‍'", "ll", ".Džungla", "'Re", "ABC", "123", "456", "78", "𐞁", "­Z", "\n", "/‍​<", "META", "_START", ">'", "Re𐞁", "'M", ","]} +{"text": "éⅣ0<|fim_prefix|>'fitdAtfi'reİ🙂a/bfiſ㋿\r\n\r\n😀🏽ß
字 ß<'M", "tokens": 49, "pieces": ["é", "Ⅳ0", "<|", "fim", "_prefix", "|>'", "fitdAtfi", "'re", "İ", "🙂a", "/bfiſ", "㋿\r\n\r\n", "😀🏽", "ß", "
字", " <", "META", "_START", ">ß", "<'", "M"]} +{"text": "字 \ndſ, ­(½\n<|fim_prefix|>ABC\"\r 'VE३'re'TiOS/½\"'ll३,'(>", "tokens": 42, "pieces": ["字", " \n", "dſ", ",", " ", "­(", "½", "\n", "<|", "fim", "_prefix", "|>", "ABC", "\"\r", " '", "VE", "३", "'re", "'T", "iOS", "/", "½", "\"'", "ll", "३", ",'(>"]} +{"text": "\naZᵃ12345678'D9< \n aBᵃⅣiOSḍ̇", "tokens": 26, "pieces": ["\n", "aZᵃ", "123", "456", "78", "'D", "9", "<", " \n", " aBᵃ", "Ⅳ", "iOSḋ", "̣"]} +{"text": "Ab \n ́‍/\reß/\r\ne 'Re/½\r'TⅣEOT", "tokens": 22, "pieces": ["Ab", " \n", " ́‍/\r", "eß", "/\r\n", "e", " '", "Re", "/", "½", "\r", "'T", "Ⅳ", "EOT"]} +{"text": "ſ/ HTTPServerꟲ", "tokens": 9, "pieces": ["ſ", "/", " HTTPServerꟲ"]} +{"text": "ABCB'Re\n(ᵃ㋿\r\n 🙂/\r\n.!!'s'ſ́éiOSZ\n0", "tokens": 28, "pieces": ["ABCB", "'Re", "\n", "(ᵃ", "㋿\r\n", " ", " 🙂/\r\n", ".!!'", "s", "'ſ", "́éiOSZ", "\n", "0"]} diff --git a/litellm-rust/crates/token-counter/tests/fixtures/generate.py b/litellm-rust/crates/token-counter/tests/fixtures/generate.py new file mode 100644 index 00000000000..1bfbdf00218 --- /dev/null +++ b/litellm-rust/crates/token-counter/tests/fixtures/generate.py @@ -0,0 +1,422 @@ +"""Pin tiktoken reference counts for the Rust parity tests of one encoding. + +Run from the repository root with the project environment, once per encoding: + + uv run --no-sync python litellm-rust/crates/token-counter/tests/fixtures/generate.py cl100k_base + uv run --no-sync python litellm-rust/crates/token-counter/tests/fixtures/generate.py o200k_base + +`/texts.jsonl` holds `{"text", "tokens", "pieces"}` lines: `tokens` +counted with `tiktoken.get_encoding(name).encode(text, disallowed_special=())`, +the same call `litellm.token_counter` makes, and `pieces` the installed +encoding's split pattern applied with the `regex` module tiktoken itself uses, +so a scanner that splits differently fails even where BPE would count the same. +`/requests.jsonl` holds `{"body", "input_tokens"}` lines, `body` being +the exact request bytes as a JSON string, counted with the proxy's admission +counter (`_count_input_tokens(body, model)`) for a model Python counts with that +encoding. Every message in the 50k-token body is shorter than the Python chunk +size so the chunked Python count equals the exact whole-text tiktoken count the +Rust counter produces. +""" + +import itertools +import json +import random +import sys +from collections.abc import Iterator +from pathlib import Path +from typing import Final + +import regex +import tiktoken + +from litellm.constants import TIKTOKEN_ENCODE_CHUNK_SIZE_CHARS +from litellm.litellm_core_utils.token_counter import openai_tokenizer_encoding +from litellm.proxy.spend_tracking.budget_reservation import _count_input_tokens + +HERE: Final = Path(__file__).resolve().parent +MODELS: Final = {"cl100k_base": "gpt-4", "o200k_base": "gpt-4o"} +ENCODING_NAME: Final = sys.argv[1] +MODEL: Final = MODELS[ENCODING_NAME] +ENCODING: Final = tiktoken.get_encoding(ENCODING_NAME) +assert openai_tokenizer_encoding(MODEL).name == ENCODING_NAME +SPLIT_PATTERN: Final = regex.compile(ENCODING._pat_str) # pyright: ignore[reportPrivateUsage] # tiktoken has no public accessor +OUT: Final = HERE / ENCODING_NAME.removesuffix("_base") + +# Mirrors ALPHABET in src/byte_level.rs, plus the pieces the tiktoken patterns treat differently. +ALPHABET: Final = ( + "a", + "Z", + "e", + "s", + "t", + "d", + "m", + "'", + "'s", + "'re", + "'ll", + "'S", + "0", + "9", + " ", + " ", + "\t", + "\n", + "\r\n", + "\x0b", + ".", + ",", + "!", + "-", + "(", + '"', + "\xa0", + "\x85", + "\u2028", + "\u3000", + "\u200b", + "\u200d", + "é", + "e\u0301", + "ß", + "漢", + "字", + "ع", + "३", + "½", + "Ⅳ", + "🙂", + "👍🏽", + "A", + "fi", + "㍿", + "㋿", + "ꟲ", + "𐞁", + "a\u030a", + "\u1e0b\u0323", + "<", + ">", + "EOT", + "", + "", + "'D", + "'M", + "'T", + "'VE", + "'Re", + "'ſ", + "ſ", + "12345678", + "٣٤٥٦", + "<|endoftext|>", + "<|fim_prefix|>", + "\r", + "\r\n\r\n", + " \n", + "!!", + "#$%", + "\u00ad", + "\u0301", + "\U0001f600\U0001f3fd", + "İ", + "Dž", +) + +# The pieces the o200k case-shaped letter branch and slash-absorbing symbol branch split differently. +CASE_ALPHABET: Final = ALPHABET + ( + "B", + "Ab", + "aB", + "ABC", + "ᵃ", + "camelCase", + "HTTPServer", + "iOS", + "Džungla", + "/", + "\n/", + "/\r\n", + " \n ", + "a/b", +) + +CORPUS: Final = ( + "", + "Hello, how are you today?", + "I'm sure they're right, we'll see. WE'LL SEE, I'M SURE THEY'RE RIGHT, IT'S HERS AND IT'D BE 'D", + "don't Don'T DON'T won'T i've I'VE i'Ve you'RE 'S 'T 'M 'D 'LL 'VE 'RE 'ſ 'x", + "1234567890 123 12 1 0000000 ٣٤٥٦٧٨ ३४५६ 1,234,567.89 2026-09-11T18:00:00Z", + "$abc %def &ghi @jkl _mno #pqr ~stu ^vwx |yz \\a /b :c ;d ?e !f (g )h [i ]j {k }l n =o +p *q", + "foo bar baz \t qux\t\tquux \n\nline\r\nline\r\n\r\n \n\t\r\n x ", + "trailing spaces ", + "trailing tabs\t\t", + "trailing newline\n", + "\n\n\n", + "\r\n\r\n\r\n", + " ", + "😀😃😄 👍🏽 🇺🇸 👨‍👩‍👧‍👦 ✈️ ❤️‍🔥 ٭ ※ ⌘ ⏎", + "漢字かな交じり文、東京都千代田区。日本語のテキストです。中文测试。한국어 텍스트", + "مرحبا بالعالم، هذا نص عربي مع أرقام ١٢٣٤٥٦٧ و علامات ترقيم!", + "Zürich, façade, naïve, Ærøskøbing, Ελληνικά, Русский текст, עברית, हिन्दी, ไทย", + "e\u0301 a\u030a \u1e0b\u0323 \u0301\u0301 combining\u0308 marks\u0301!", + "ΣΊΣΥΦΟΣ Džungla İstanbul file flow Abc ㍿ ㋿ ꟲ 𐞁", + "<|endoftext|> <|fim_prefix|>code<|fim_middle|>more<|fim_suffix|> <|endofprompt|> <|im_start|>", + " [INST] [/INST] <>", + "def f(x):\n return {'a': x ** 2, \"b\": [1, 2, 3]} # comment\n\nprint(f(10))\n", + '{"model":"gpt-4","messages":[{"role":"user","content":"hi\\n"}],"temperature":0.7}', + "https://example.com/path?query=1&other=two#fragment user@example.com 192.168.0.1", + "a" * 3000, + " " * 3000, + "." * 3000, + "ab" * 1500, + "\n" * 3000, + "0" * 3000, + "!" * 3000, + "😀" * 1000, + "漢" * 1000, + "\u00a0abc\u00a0! \u2028x \u3000y \u200bz \u200d\u200d q", + "x\u0085y \x0b\x0c z", + "\x00\x01\x02 \x7f \ufffd", + "tab\tseparated\tvalues\n1\t2\t3\n", + "MiXeD cAsE wOrDs AND ACRONYMS like NASA, HTTP/2, gRPC, iOS, macOS", + "snake_case_identifier camelCaseIdentifier PascalCaseIdentifier SCREAMING_SNAKE_CASE kebab-case", + "x'sy x'ty x'rey x'vey x'my x'lly x'dy x'S x'T x'RE x'VE x'M x'LL x'D x'sS x'llL", + "IT'SOK it'Dbe x'Sy x'Ty x'My x'Dy x'LLy x'VEy x'REy x'Ly x'Vy x'Ry 'Sx'Tx'Mx'LLx'VEx'REx'Dx", + "'s't're've'm'll'd 'S'T'RE'VE'M'LL'D ''s '''s", + "9'9 9's a'9 '9 ' 's' ' 's", + "١٢٣٤ ½⅓¼ ⅣⅤ 𝟘𝟙𝟚𝟛𝟜𝟝𝟞𝟟𝟠𝟡 ①②③", + "camelCase PascalCase ABCdef ABCdeF ABC aB Ab ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyzABC", + "日本ABC ABC日本 日本語abc abc日本語 漢字Kanji kanji漢字 KANJI漢字kanji مرحباABC ABCمرحبا abcمرحبا", + "\u0301ABC \u0301abc \u0301\u0301A A\u0301\u0301 E\u0301A aE\u0301 !!\u0301a \u00a0\u0301A x\u0308Y X\u0308y", + "ᵃbc ᵃBC Aᵃbc Aᵃ ᵃ' ᵃ's Džungla aDžB ADžB ADžb DžDž Ljx İi ΣΊΣΥΦΟΣσ ΣσΣ", + "don'tx ABC's abc'S abc'ſ ABC'ſx IT'SOK it'Dbe 'sabc x's 's 'Sx'Tx x’s X'LLx X'Ll", + "!ABC !AbC !!abc #camelCase (ABCdef) \u00a0ABC\u00a0abc\u00a0Abc \tABC\tabc", + "!!/\n/x a/b !!\n/x /x // path/to/file.rs http://x.y/z?a=b/c \\/\\/ //\r\n//\n", + "x \n x \r\n \r\n y x \n a b \n\n c x\t\ty x\t\t end \n \n", + "12345 6 1abc abc1 ABC123abc 123ABC ١٢٣٤٥abc", +) + +WORDS: Final = ( + "the", + "quick", + "brown", + "fox", + "jumps", + "over", + "lazy", + "dog", + "while", + "counting", + "tokens", + "for", + "budget", + "reservation", + "before", + "admission", + "on", + "the", + "gateway", + "and", + "every", + "request", + "body", + "is", + "scanned", + "exactly", + "once", + "with", + "a", + "hand", + "written", + "piece", + "scanner", + "that", + "mirrors", + "tiktoken's", + "regex", + "boundaries", + "It's", + "faster", + "because", + "there's", + "no", + "backtracking", + "engine", + "involved", + "so", + "we'll", + "keep", + "it", + "that", + "way", + "Zürich", + "café", + "naïve", + "東京", + "مرحبا", + "🙂", + "42", + "1999", + "3.14159", + "$1,234.56", + "100%", + "user@example.com", + "https://example.com/a/b?c=d", + "C++", + "F#", + "node.js", + "v1.2.3", + "(parens)", + "[brackets]", + "{braces}", + "", + '"quotes"', + "'single'", + "don't", + "WON'T", + "I'M", + "They'RE", +) + + +def random_text(rng: random.Random, alphabet: tuple[str, ...]) -> str: + return "".join(rng.choice(alphabet) for _ in range(rng.randrange(0, 40))) + + +def paragraph(rng: random.Random, words: int) -> str: + return " ".join(rng.choice(WORDS) for _ in range(words)) + + +def short_paragraphs(rng: random.Random) -> Iterator[str]: + while True: + content = paragraph(rng, rng.randrange(60, 140)) + if len(content) < TIKTOKEN_ENCODE_CHUNK_SIZE_CHARS: + yield content + + +def chat_body(rng: random.Random, target_tokens: int) -> dict[str, object]: + candidates: Final = tuple(itertools.islice(short_paragraphs(rng), 2000)) + running: Final = tuple(itertools.accumulate(len(ENCODING.encode(content)) + 3 for content in candidates)) + turns: Final = next(index for index, total in enumerate(running) if total >= target_tokens) + 1 + contents: Final = candidates[: turns + (turns % 2)] + return { + "model": MODEL, + "messages": [ + {"role": "system", "content": "You are a helpful assistant. Answer precisely and cite sources."}, + *( + {"role": "user" if index % 2 == 0 else "assistant", "content": content} + for index, content in enumerate(contents) + ), + {"role": "user", "content": "Summarise the conversation so far in three sentences."}, + ], + } + + +TOOLS: Final = [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get weather", + "parameters": { + "type": "object", + "properties": { + "location": {"type": "string", "description": "City name"}, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + "days": {"type": "integer"}, + "tags": {"type": "array", "items": {"type": "string"}}, + "opts": { + "type": "object", + "properties": {"verbose": {"type": "boolean"}, "level": {"type": "integer", "enum": [1, 2]}}, + "required": ["verbose"], + }, + "anything": {}, + }, + "required": ["location"], + }, + }, + }, + {"type": "function", "function": {"name": "noop"}}, +] + +SMALL_REQUESTS: Final = ( + {"model": MODEL, "messages": [{"role": "user", "content": "Hello, how are you today?"}]}, + { + "model": MODEL, + "messages": [ + {"role": "system", "content": "You are a terse assistant."}, + { + "role": "user", + "name": "alice", + "content": [ + {"type": "text", "text": "Summarise this paragraph about ships and harbours."}, + "plain string item", + ], + }, + {"role": "assistant", "content": [{"type": "text", "text": "Sure."}]}, + ], + }, + { + "model": MODEL, + "messages": [{"role": "user", "content": "weather?"}], + "tools": TOOLS, + "tool_choice": {"type": "function", "function": {"name": "get_weather"}}, + }, + { + "model": MODEL, + "messages": [{"role": "system", "content": "sys"}, {"role": "user", "content": "weather?"}], + "tools": TOOLS, + "tool_choice": "none", + }, + {"model": MODEL, "prompt": "Write a haiku about ships."}, + {"model": MODEL, "prompt": ["first prompt", "second prompt"]}, + { + "model": MODEL, + "input": [ + { + "role": "user", + "content": [ + {"type": "input_text", "text": 'Summarise caf\u00e9 menus, na\u00efve \u2014 ok? "quoted"\n'} + ], + }, + {"role": "assistant", "content": "Sure."}, + ], + "instructions": "be terse", + }, + {"model": MODEL, "input": [[101, 2023, 5], [7]], "encoding_format": "float"}, + { + "model": MODEL, + "query": "best harbour", + "documents": [ + "doc one", + {"text": "doc two", "title": "T", "n": 3, "ok": True, "none": None, "tags": ["a", "b"]}, + ], + }, +) + + +def main() -> None: + rng: Final = random.Random(2026) + case_rng: Final = random.Random(200_000) + texts: Final = ( + tuple(CORPUS) + + tuple(random_text(rng, ALPHABET) for _ in range(3000)) + + tuple(random_text(case_rng, CASE_ALPHABET) for _ in range(1000)) + ) + OUT.mkdir(exist_ok=True) + with (OUT / "texts.jsonl").open("w", encoding="utf-8") as handle: + for text in texts: + tokens = len(ENCODING.encode(text, disallowed_special=())) + pieces = SPLIT_PATTERN.findall(text) + handle.write(json.dumps({"text": text, "tokens": tokens, "pieces": pieces}, ensure_ascii=False) + "\n") + bodies: Final = tuple(SMALL_REQUESTS) + (chat_body(rng, 50_000),) + with (OUT / "requests.jsonl").open("w", encoding="utf-8") as handle: + for body in bodies: + input_tokens = _count_input_tokens(dict(body), MODEL) + assert input_tokens is not None + handle.write(json.dumps({"body": json.dumps(body), "input_tokens": input_tokens}) + "\n") + + +if __name__ == "__main__": + main() diff --git a/litellm-rust/crates/token-counter/tests/fixtures/o200k/requests.jsonl b/litellm-rust/crates/token-counter/tests/fixtures/o200k/requests.jsonl new file mode 100644 index 00000000000..f50b85c3922 --- /dev/null +++ b/litellm-rust/crates/token-counter/tests/fixtures/o200k/requests.jsonl @@ -0,0 +1,10 @@ +{"body": "{\"model\": \"gpt-4o\", \"messages\": [{\"role\": \"user\", \"content\": \"Hello, how are you today?\"}]}", "input_tokens": 14} +{"body": "{\"model\": \"gpt-4o\", \"messages\": [{\"role\": \"system\", \"content\": \"You are a terse assistant.\"}, {\"role\": \"user\", \"name\": \"alice\", \"content\": [{\"type\": \"text\", \"text\": \"Summarise this paragraph about ships and harbours.\"}, \"plain string item\"]}, {\"role\": \"assistant\", \"content\": [{\"type\": \"text\", \"text\": \"Sure.\"}]}]}", "input_tokens": 39} +{"body": "{\"model\": \"gpt-4o\", \"messages\": [{\"role\": \"user\", \"content\": \"weather?\"}], \"tools\": [{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Get weather\", \"parameters\": {\"type\": \"object\", \"properties\": {\"location\": {\"type\": \"string\", \"description\": \"City name\"}, \"unit\": {\"type\": \"string\", \"enum\": [\"celsius\", \"fahrenheit\"]}, \"days\": {\"type\": \"integer\"}, \"tags\": {\"type\": \"array\", \"items\": {\"type\": \"string\"}}, \"opts\": {\"type\": \"object\", \"properties\": {\"verbose\": {\"type\": \"boolean\"}, \"level\": {\"type\": \"integer\", \"enum\": [1, 2]}}, \"required\": [\"verbose\"]}, \"anything\": {}}, \"required\": [\"location\"]}}}, {\"type\": \"function\", \"function\": {\"name\": \"noop\"}}], \"tool_choice\": {\"type\": \"function\", \"function\": {\"name\": \"get_weather\"}}}", "input_tokens": 104} +{"body": "{\"model\": \"gpt-4o\", \"messages\": [{\"role\": \"system\", \"content\": \"sys\"}, {\"role\": \"user\", \"content\": \"weather?\"}], \"tools\": [{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Get weather\", \"parameters\": {\"type\": \"object\", \"properties\": {\"location\": {\"type\": \"string\", \"description\": \"City name\"}, \"unit\": {\"type\": \"string\", \"enum\": [\"celsius\", \"fahrenheit\"]}, \"days\": {\"type\": \"integer\"}, \"tags\": {\"type\": \"array\", \"items\": {\"type\": \"string\"}}, \"opts\": {\"type\": \"object\", \"properties\": {\"verbose\": {\"type\": \"boolean\"}, \"level\": {\"type\": \"integer\", \"enum\": [1, 2]}}, \"required\": [\"verbose\"]}, \"anything\": {}}, \"required\": [\"location\"]}}}, {\"type\": \"function\", \"function\": {\"name\": \"noop\"}}], \"tool_choice\": \"none\"}", "input_tokens": 97} +{"body": "{\"model\": \"gpt-4o\", \"prompt\": \"Write a haiku about ships.\"}", "input_tokens": 7} +{"body": "{\"model\": \"gpt-4o\", \"prompt\": [\"first prompt\", \"second prompt\"]}", "input_tokens": 4} +{"body": "{\"model\": \"gpt-4o\", \"input\": [{\"role\": \"user\", \"content\": [{\"type\": \"input_text\", \"text\": \"Summarise caf\\u00e9 menus, na\\u00efve \\u2014 ok? \\\"quoted\\\"\\n\"}]}, {\"role\": \"assistant\", \"content\": \"Sure.\"}], \"instructions\": \"be terse\"}", "input_tokens": 60} +{"body": "{\"model\": \"gpt-4o\", \"input\": [[101, 2023, 5], [7]], \"encoding_format\": \"float\"}", "input_tokens": 5} +{"body": "{\"model\": \"gpt-4o\", \"query\": \"best harbour\", \"documents\": [\"doc one\", {\"text\": \"doc two\", \"title\": \"T\", \"n\": 3, \"ok\": true, \"none\": null, \"tags\": [\"a\", \"b\"]}]}", "input_tokens": 43} +{"body": "{\"model\": \"gpt-4o\", \"messages\": [{\"role\": \"system\", \"content\": \"You are a helpful assistant. Answer precisely and cite sources.\"}, {\"role\": \"user\", \"content\": \"\\ud83d\\ude42 a every WON'T They'RE involved counting caf\\u00e9 backtracking boundaries Z\\u00fcrich WON'T 100% \\\"quotes\\\" caf\\u00e9 tiktoken's budget They'RE request request regex v1.2.3 hand hand fox tiktoken's on 3.14159 mirrors that don't WON'T admission before budget the admission WON'T no 1999 100% no admission budget mirrors way caf\\u00e9 dog a quick mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 $1,234.56 hand gateway regex on jumps because {braces} 'single' the the the {braces} piece hand scanned reservation [brackets] we'll 3.14159 request don't \\\"quotes\\\" na\\u00efve C++ caf\\u00e9 caf\\u00e9 for https://example.com/a/b?c=d we'll no over the node.js for while over way\"}, {\"role\": \"assistant\", \"content\": \"it tiktoken's node.js it scanned v1.2.3 boundaries (parens) and while reservation lazy the that https://example.com/a/b?c=d quick don't budget engine boundaries F# budget every we'll before jumps scanned jumps there's counting the don't jumps a once admission 100% 3.14159 brown brown that because jumps They'RE 100% caf\\u00e9 WON'T They'RE boundaries \\u6771\\u4eac exactly scanner caf\\u00e9 fox (parens) body dog a 1999 boundaries 'single' {braces} reservation mirrors while {braces} {braces} so admission the F# https://example.com/a/b?c=d brown a caf\\u00e9 on admission that on counting that we'll over lazy https://example.com/a/b?c=d way over engine reservation gateway body I'M for \\\"quotes\\\" 42 and F# there's 'single' quick \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T scanner written every (parens) so\"}, {\"role\": \"user\", \"content\": \"I'M over faster counting https://example.com/a/b?c=d way while it over way mirrors boundaries keep body hand na\\u00efve once because that node.js for every the 'single' every caf\\u00e9 caf\\u00e9 piece there's v1.2.3 v1.2.3 tiktoken's brown so counting there's for we'll boundaries 'single' no https://example.com/a/b?c=d node.js \\u6771\\u4eac written with budget (parens) because \\\"quotes\\\" \\u6771\\u4eac while \\u6771\\u4eac counting scanned 1999 \\u6771\\u4eac hand keep so that gateway caf\\u00e9 for scanned it request keep counting user@example.com admission request \\\"quotes\\\" v1.2.3 caf\\u00e9 over that F# we'll regex a faster with that They'RE piece because tiktoken's engine hand 100% WON'T reservation (parens) caf\\u00e9 on Z\\u00fcrich dog \\u6771\\u4eac that They'RE 'single' the 'single' na\\u00efve involved the {braces} there's scanned no engine dog backtracking there's\"}, {\"role\": \"assistant\", \"content\": \"every node.js C++ for 3.14159 \\u6771\\u4eac 'single' gateway once the user@example.com a 100% every every tokens written and every brown na\\u00efve the body the because quick brown engine fox 3.14159 (parens) F# while with a 100% C++ with the written WON'T on request Z\\u00fcrich hand on https://example.com/a/b?c=d don't way way 42 that we'll \\ud83d\\ude42 [brackets] admission tiktoken's request hand caf\\u00e9 every every (parens) They'RE that jumps the regex 100% \\\"quotes\\\" is budget\"}, {\"role\": \"user\", \"content\": \"engine with I'M backtracking while F# quick 'single' mirrors \\\"quotes\\\" 'single' regex caf\\u00e9 It's Z\\u00fcrich exactly a counting tiktoken's we'll once \\u0645\\u0631\\u062d\\u0628\\u0627 42 v1.2.3 scanner WON'T \\u6771\\u4eac there's node.js It's budget that budget {braces} because [brackets] It's a $1,234.56 that v1.2.3 42 gateway backtracking it C++ scanned keep \\ud83d\\ude42 I'M hand a counting scanner reservation \\u6771\\u4eac written 3.14159 there's once fox I'M C++ lazy the \\u0645\\u0631\\u062d\\u0628\\u0627 before don't we'll gateway exactly \\ud83d\\ude42 Z\\u00fcrich 3.14159 with WON'T there's node.js \\ud83d\\ude42 quick written faster counting 1999 backtracking 'single' counting we'll engine counting don't \\\"quotes\\\" engine \\\"quotes\\\" way I'M budget [brackets] backtracking \\\"quotes\\\" 3.14159 don't written written \\u0645\\u0631\\u062d\\u0628\\u0627 body I'M 'single' while on and reservation \\ud83d\\ude42 regex and no budget regex (parens) we'll They'RE na\\u00efve v1.2.3\"}, {\"role\": \"assistant\", \"content\": \"hand lazy dog budget lazy involved because scanned piece quick that involved 'single' dog quick na\\u00efve budget scanner exactly $1,234.56 fox quick I'M hand (parens) node.js while jumps boundaries \\ud83d\\ude42 They'RE way request engine 'single' fox backtracking regex \\u0645\\u0631\\u062d\\u0628\\u0627 because and over jumps 3.14159 WON'T counting for and mirrors admission 'single' caf\\u00e9 https://example.com/a/b?c=d that \\ud83d\\ude42 faster \\u0645\\u0631\\u062d\\u0628\\u0627 admission jumps quick for $1,234.56 exactly exactly $1,234.56 scanned keep 3.14159 backtracking piece tiktoken's is is hand reservation before regex budget 3.14159 tiktoken's I'M it no budget user@example.com budget written budget over that https://example.com/a/b?c=d no written so quick tokens C++\"}, {\"role\": \"user\", \"content\": \"once body involved every brown every it that hand engine with scanned reservation \\u6771\\u4eac backtracking and {braces} over [brackets] brown over for is \\ud83d\\ude42 100% before quick no counting no v1.2.3 counting \\\"quotes\\\" mirrors 3.14159 1999 there's way \\\"quotes\\\" piece on na\\u00efve They'RE no every exactly node.js 42 because over there's 1999 C++ we'll mirrors F# it scanner jumps It's scanner mirrors mirrors It's tokens backtracking body brown $1,234.56 keep admission 100% the exactly we'll caf\\u00e9 jumps over 3.14159 we'll so 42 \\u6771\\u4eac I'M WON'T lazy brown It's a na\\u00efve \\\"quotes\\\" https://example.com/a/b?c=d (parens) \\u6771\\u4eac tiktoken's so dog body 'single' scanned piece body 3.14159 scanned body the so \\\"quotes\\\" counting\"}, {\"role\": \"assistant\", \"content\": \"boundaries request the that 100% gateway the there's hand the They'RE caf\\u00e9 fox C++ scanner written \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE scanner WON'T keep before for it so counting that (parens) {braces} They'RE scanner every {braces} no node.js keep I'M jumps backtracking gateway https://example.com/a/b?c=d don't tiktoken's a is 1999 don't F# v1.2.3 involved hand scanner They'RE backtracking exactly and for exactly for $1,234.56 counting v1.2.3 request gateway mirrors that no \\ud83d\\ude42 every dog once They'RE 100% reservation It's a with and 100% because so It's admission there's gateway 42 on over gateway is every 3.14159 boundaries no for admission quick lazy lazy fox node.js because while $1,234.56 quick I'M lazy involved WON'T {braces} na\\u00efve {braces} there's with caf\\u00e9 https://example.com/a/b?c=d \\\"quotes\\\" [brackets] lazy lazy it 100% caf\\u00e9 way lazy tokens C++ that it \\u6771\\u4eac lazy\"}, {\"role\": \"user\", \"content\": \"Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors budget reservation 'single' it scanner F# is exactly They'RE \\ud83d\\ude42 I'M boundaries F# {braces} https://example.com/a/b?c=d over backtracking node.js is \\ud83d\\ude42 \\ud83d\\ude42 $1,234.56 so and counting It's Z\\u00fcrich once node.js once v1.2.3 on that we'll 'single' mirrors I'M \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking They'RE tokens hand counting 1999 user@example.com Z\\u00fcrich They'RE tokens reservation exactly for https://example.com/a/b?c=d mirrors and boundaries regex the brown while keep \\\"quotes\\\" we'll the involved 42 {braces} scanner reservation Z\\u00fcrich no we'll [brackets] caf\\u00e9 written that hand user@example.com $1,234.56 42 while budget every 3.14159 a exactly way body scanned admission C++ (parens) tiktoken's body for WON'T hand no dog 1999 (parens) on don't 1999 we'll I'M v1.2.3 that WON'T fox scanned 1999\"}, {\"role\": \"assistant\", \"content\": \"gateway $1,234.56 $1,234.56 \\\"quotes\\\" scanner there's scanner $1,234.56 100% it budget \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) admission admission with I'M admission 'single' hand jumps scanned It's \\\"quotes\\\" is F# before engine counting dog because admission tiktoken's backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 that 42 written it body quick Z\\u00fcrich I'M F# I'M that brown It's request caf\\u00e9 is boundaries jumps don't written written faster for admission Z\\u00fcrich engine reservation and Z\\u00fcrich scanned written keep scanned is keep jumps the so F# 'single' user@example.com keep because They'RE over because C++ written faster 42 mirrors na\\u00efve 42 tiktoken's [brackets] na\\u00efve hand on no written so $1,234.56 there's 100% 100% $1,234.56 is \\ud83d\\ude42 scanner dog lazy 100% https://example.com/a/b?c=d tiktoken's They'RE [brackets] user@example.com while hand \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d we'll 42 WON'T it a mirrors v1.2.3\"}, {\"role\": \"user\", \"content\": \"that involved dog C++ way that regex with https://example.com/a/b?c=d because 1999 jumps and and caf\\u00e9 F# backtracking that 3.14159 the quick v1.2.3 engine piece request brown (parens) because tokens na\\u00efve we'll and \\ud83d\\ude42 user@example.com that na\\u00efve \\u6771\\u4eac na\\u00efve tokens jumps 3.14159 tiktoken's na\\u00efve brown It's jumps before and a the user@example.com Z\\u00fcrich WON'T mirrors It's fox It's involved don't \\u6771\\u4eac 'single' written that written fox and the 3.14159 Z\\u00fcrich way on once na\\u00efve \\u0645\\u0631\\u062d\\u0628\\u0627 $1,234.56 request fox so fox 'single' admission admission node.js mirrors backtracking it the\"}, {\"role\": \"assistant\", \"content\": \"engine so $1,234.56 don't caf\\u00e9 lazy I'M while 'single' budget scanned \\ud83d\\ude42 Z\\u00fcrich piece tokens exactly budget boundaries admission written tokens every \\u0645\\u0631\\u062d\\u0628\\u0627 before with backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 engine regex faster brown \\ud83d\\ude42 before we'll reservation na\\u00efve regex there's faster counting na\\u00efve reservation quick Z\\u00fcrich tiktoken's that gateway we'll C++ They'RE scanned for engine budget 100% na\\u00efve $1,234.56 written admission no we'll WON'T scanned body dog node.js while\"}, {\"role\": \"user\", \"content\": \"once exactly scanner boundaries scanned every there's mirrors $1,234.56 there's so dog dog They'RE lazy fox because 'single' so gateway Z\\u00fcrich faster v1.2.3 quick I'M \\u6771\\u4eac exactly written tokens node.js na\\u00efve brown [brackets] mirrors while on \\u6771\\u4eac $1,234.56 once hand over exactly quick backtracking over exactly {braces} it don't regex 3.14159 way the 100% $1,234.56 is I'M admission admission [brackets] scanned while boundaries piece counting that node.js reservation \\u6771\\u4eac body jumps while node.js I'M Z\\u00fcrich with fox C++ reservation F# \\\"quotes\\\" They'RE {braces} (parens) caf\\u00e9 1999 there's {braces} every WON'T no dog WON'T 1999 admission [brackets] that body 'single' gateway while mirrors with with scanner hand no over scanned hand na\\u00efve It's the while with once \\\"quotes\\\" boundaries \\\"quotes\\\" It's tokens and\"}, {\"role\": \"assistant\", \"content\": \"jumps 100% before body there's a C++ $1,234.56 on exactly exactly every hand lazy dog don't every that so F# Z\\u00fcrich jumps caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 written scanned tokens admission WON'T backtracking scanned 1999 engine brown exactly counting node.js 'single' 1999 counting that keep keep $1,234.56 Z\\u00fcrich They'RE before scanned They'RE \\u6771\\u4eac It's over tiktoken's once hand request tiktoken's while gateway node.js no \\ud83d\\ude42 it https://example.com/a/b?c=d It's 100% written there's hand {braces} there's admission $1,234.56 F# [brackets]\"}, {\"role\": \"user\", \"content\": \"hand mirrors exactly 42 They'RE [brackets] before \\u0645\\u0631\\u062d\\u0628\\u0627 that while written caf\\u00e9 body because fox tokens no WON'T dog faster and fox keep don't caf\\u00e9 'single' node.js (parens) reservation C++ I'M fox F# \\u6771\\u4eac mirrors over 1999 written keep na\\u00efve gateway a 3.14159 keep once body \\\"quotes\\\" F# we'll $1,234.56 https://example.com/a/b?c=d regex fox backtracking is admission request involved so over engine caf\\u00e9 way backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac They'RE It's a mirrors \\\"quotes\\\" jumps They'RE way way the engine C++ request because backtracking with with written the scanned boundaries https://example.com/a/b?c=d (parens) no for gateway dog \\ud83d\\ude42 boundaries node.js\"}, {\"role\": \"assistant\", \"content\": \"and tiktoken's They'RE caf\\u00e9 exactly the quick gateway caf\\u00e9 budget hand node.js the node.js we'll faster fox 'single' involved scanner na\\u00efve reservation https://example.com/a/b?c=d {braces} way we'll backtracking keep while and no with dog exactly the WON'T dog \\u0645\\u0631\\u062d\\u0628\\u0627 regex [brackets] written 1999 keep v1.2.3 I'M mirrors admission \\u6771\\u4eac before we'll \\\"quotes\\\" (parens) 100% over mirrors 42 a request F# WON'T a counting \\ud83d\\ude42 scanner no we'll fox gateway boundaries 1999 WON'T I'M Z\\u00fcrich faster there's caf\\u00e9 They'RE that the with for faster quick that body reservation 42 don't I'M I'M scanned faster hand for\"}, {\"role\": \"user\", \"content\": \"and the admission caf\\u00e9 admission once F# written reservation budget written https://example.com/a/b?c=d scanner a that Z\\u00fcrich once the Z\\u00fcrich faster C++ no I'M counting is hand caf\\u00e9 before engine on C++ 'single' \\\"quotes\\\" I'M Z\\u00fcrich quick that boundaries node.js lazy 3.14159 while na\\u00efve Z\\u00fcrich it reservation request that before F# boundaries once written the WON'T https://example.com/a/b?c=d the written piece mirrors 100% mirrors https://example.com/a/b?c=d over before regex scanner \\u0645\\u0631\\u062d\\u0628\\u0627 before node.js WON'T \\u0645\\u0631\\u062d\\u0628\\u0627 way jumps for scanned that involved for every (parens) lazy C++ boundaries backtracking it user@example.com no that C++ faster engine I'M piece with faster for that brown 3.14159 user@example.com it while so on involved that 42 once quick written we'll a budget\"}, {\"role\": \"assistant\", \"content\": \"admission \\ud83d\\ude42 100% tiktoken's on jumps we'll hand body is tiktoken's and 100% \\ud83d\\ude42 the it 3.14159 $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 scanner over 100% scanned \\ud83d\\ude42 so way engine that scanner 100% so so tokens boundaries node.js we'll jumps user@example.com that tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 once don't way is body so user@example.com on request is lazy and every engine tokens no lazy admission jumps reservation v1.2.3 scanned \\u0645\\u0631\\u062d\\u0628\\u0627 the \\ud83d\\ude42 node.js so \\\"quotes\\\" every it \\u6771\\u4eac F# regex don't faster every $1,234.56 on way engine regex every keep $1,234.56 no that engine written don't involved {braces} \\u6771\\u4eac v1.2.3 while request 42 (parens)\"}, {\"role\": \"user\", \"content\": \"keep fox we'll backtracking once it engine with lazy (parens) faster caf\\u00e9 F# with It's (parens) user@example.com for over counting we'll it https://example.com/a/b?c=d fox the (parens) way over because https://example.com/a/b?c=d before written fox involved written [brackets] before it tokens \\\"quotes\\\" there's once [brackets] 'single' tiktoken's C++ v1.2.3 quick 100% user@example.com dog \\u6771\\u4eac I'M 'single' keep a before node.js F# exactly request the Z\\u00fcrich exactly tokens so faster admission lazy and every counting $1,234.56 brown\"}, {\"role\": \"assistant\", \"content\": \"'single' every budget 42 It's quick way dog {braces} because don't \\u0645\\u0631\\u062d\\u0628\\u0627 way brown involved counting body don't (parens) body tiktoken's scanned \\u0645\\u0631\\u062d\\u0628\\u0627 dog it (parens) the for once once engine written there's that node.js every \\u0645\\u0631\\u062d\\u0628\\u0627 1999 $1,234.56 caf\\u00e9 written body dog gateway for It's WON'T 42 'single' 3.14159 that Z\\u00fcrich jumps C++ while WON'T admission so involved It's \\u0645\\u0631\\u062d\\u0628\\u0627 node.js on with It's it They'RE lazy is engine way scanner $1,234.56 and lazy boundaries 'single' before dog that faster 3.14159 tokens It's tokens we'll once and reservation over They'RE\"}, {\"role\": \"user\", \"content\": \"I'M node.js C++ na\\u00efve once engine with \\ud83d\\ude42 involved 100% brown scanned and is scanner scanned 'single' counting don't fox way way C++ before every faster piece hand They'RE so brown piece because while on a WON'T 1999 backtracking budget \\u6771\\u4eac piece and engine every we'll dog user@example.com na\\u00efve https://example.com/a/b?c=d brown Z\\u00fcrich it way and so hand on \\\"quotes\\\" (parens) before we'll\"}, {\"role\": \"assistant\", \"content\": \"They'RE there's node.js mirrors there's the there's reservation engine is boundaries fox a body fox 42 user@example.com (parens) 100% on (parens) written request https://example.com/a/b?c=d {braces} It's Z\\u00fcrich every counting $1,234.56 scanned once user@example.com C++ so 100% exactly exactly {braces} body jumps on no involved 100% and admission every scanned reservation tokens exactly keep there's Z\\u00fcrich v1.2.3 keep 42 the 42 the with boundaries request involved there's faster keep on https://example.com/a/b?c=d user@example.com \\u0645\\u0631\\u062d\\u0628\\u0627 on (parens) na\\u00efve (parens) WON'T \\u6771\\u4eac with engine na\\u00efve body 1999 the engine quick counting boundaries there's counting {braces} for 100% involved there's involved so WON'T lazy a over way keep They'RE tokens keep piece na\\u00efve node.js exactly reservation body every faster backtracking\"}, {\"role\": \"user\", \"content\": \"and scanned C++ jumps They'RE scanned boundaries C++ that hand budget tokens \\\"quotes\\\" scanned I'M 100% 'single' counting because (parens) regex (parens) na\\u00efve F# (parens) I'M and with so once once for regex piece dog quick They'RE while keep exactly before 1999 is 100% keep caf\\u00e9 before 42 hand that reservation 3.14159 because fox that regex body gateway don't once gateway and engine with once don't I'M piece way 3.14159 admission the don't it body piece \\\"quotes\\\" 42 mirrors on body don't 100% that quick backtracking {braces} is we'll and body we'll budget every every v1.2.3 exactly way I'M \\\"quotes\\\" involved gateway scanned once \\u0645\\u0631\\u062d\\u0628\\u0627 keep jumps \\u6771\\u4eac backtracking dog engine \\u6771\\u4eac tiktoken's that 42 user@example.com don't scanner (parens) I'M\"}, {\"role\": \"assistant\", \"content\": \"the piece 1999 over v1.2.3 C++ It's https://example.com/a/b?c=d because request that fox \\ud83d\\ude42 way \\u6771\\u4eac tokens tokens brown (parens) way v1.2.3 that na\\u00efve mirrors \\\"quotes\\\" admission dog 100% 100% regex backtracking reservation 1999 user@example.com \\\"quotes\\\" 3.14159 I'M budget mirrors 1999 lazy admission a on that user@example.com They'RE They'RE \\\"quotes\\\" user@example.com node.js 100% so na\\u00efve and It's \\\"quotes\\\" with the piece boundaries we'll 42 42 while https://example.com/a/b?c=d user@example.com budget faster na\\u00efve and 1999 a admission fox and 'single' for They'RE boundaries scanned\"}, {\"role\": \"user\", \"content\": \"quick gateway They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 fox the (parens) mirrors because while we'll we'll every mirrors counting Z\\u00fcrich 42 on gateway WON'T written backtracking no user@example.com caf\\u00e9 \\u6771\\u4eac that boundaries while so fox backtracking involved They'RE exactly 1999 {braces} [brackets] exactly regex mirrors \\u6771\\u4eac 1999 over reservation way involved \\u0645\\u0631\\u062d\\u0628\\u0627 42 3.14159 while quick so a (parens) hand there's F# a They'RE body engine F# v1.2.3 faster hand over dog backtracking while brown that before I'M 1999 way before piece counting request that budget boundaries v1.2.3 exactly that They'RE faster involved \\\"quotes\\\" scanner C++ It's 100% $1,234.56 written\"}, {\"role\": \"assistant\", \"content\": \"https://example.com/a/b?c=d way \\\"quotes\\\" fox fox C++ is admission It's \\ud83d\\ude42 tiktoken's WON'T tokens so caf\\u00e9 way admission a scanned $1,234.56 3.14159 that F# don't fox counting every engine scanner scanned don't 'single' tiktoken's https://example.com/a/b?c=d I'M keep there's piece \\u6771\\u4eac is while way hand over fox WON'T \\u6771\\u4eac scanner v1.2.3 \\u6771\\u4eac don't written \\\"quotes\\\" keep the body and brown that counting before budget request 3.14159 boundaries {braces}\"}, {\"role\": \"user\", \"content\": \"exactly It's 3.14159 hand body They'RE tokens don't v1.2.3 backtracking on quick (parens) 100% body quick so scanner so \\ud83d\\ude42 backtracking is once exactly regex https://example.com/a/b?c=d na\\u00efve before there's every C++ [brackets] tokens They'RE with the 1999 piece a reservation request caf\\u00e9 it 'single' while \\\"quotes\\\" boundaries because lazy \\u0645\\u0631\\u062d\\u0628\\u0627 for Z\\u00fcrich It's (parens) a 100% there's dog caf\\u00e9 \\ud83d\\ude42 (parens) because quick with node.js https://example.com/a/b?c=d engine piece \\ud83d\\ude42 counting brown way It's user@example.com user@example.com \\u0645\\u0631\\u062d\\u0628\\u0627 scanned reservation 'single' don't request admission\"}, {\"role\": \"assistant\", \"content\": \"reservation it so before no scanned backtracking a 1999 every exactly jumps 100% over It's 1999 we'll boundaries don't scanned once and before \\u6771\\u4eac faster backtracking regex backtracking every 'single' admission caf\\u00e9 brown 3.14159 \\ud83d\\ude42 there's \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T 42 $1,234.56 is before reservation it admission backtracking before over $1,234.56 {braces} mirrors It's {braces} before admission quick hand lazy scanned \\ud83d\\ude42 exactly faster while reservation C++ They'RE it exactly gateway it body for quick so quick 3.14159 1999 for user@example.com involved {braces} https://example.com/a/b?c=d quick while I'M lazy brown backtracking keep\"}, {\"role\": \"user\", \"content\": \"quick keep v1.2.3 request budget a and regex keep body gateway so It's before \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 I'M 3.14159 and written counting {braces} v1.2.3 for brown engine user@example.com mirrors mirrors over \\u6771\\u4eac (parens) regex jumps keep written and Z\\u00fcrich a no dog and jumps piece C++ the counting with quick on the user@example.com request is jumps boundaries quick 'single' 100% 'single' over 100% the over there's na\\u00efve 1999 engine quick https://example.com/a/b?c=d 3.14159 it jumps way no hand \\ud83d\\ude42 [brackets] {braces} admission F# scanner 3.14159 F# it 3.14159 boundaries budget keep don't that quick 'single' reservation C++ the gateway and dog exactly body so https://example.com/a/b?c=d \\\"quotes\\\" 100%\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 budget \\\"quotes\\\" quick once body I'M that way no once a user@example.com boundaries admission gateway engine caf\\u00e9 $1,234.56 They'RE They'RE reservation user@example.com They'RE They'RE budget for \\u0645\\u0631\\u062d\\u0628\\u0627 dog is WON'T faster involved lazy so while faster backtracking 1999 user@example.com we'll engine reservation v1.2.3 it on tokens counting \\ud83d\\ude42 every every over written it every tiktoken's no before (parens) body the request quick every admission the \\ud83d\\ude42 body don't admission regex there's \\u6771\\u4eac \\\"quotes\\\" dog don't no Z\\u00fcrich $1,234.56 and scanned scanned https://example.com/a/b?c=d request while way written engine reservation there's scanned the request that jumps it caf\\u00e9 and v1.2.3 scanned na\\u00efve involved brown no user@example.com tokens because counting there's \\\"quotes\\\" on node.js quick exactly every that written\"}, {\"role\": \"user\", \"content\": \"counting {braces} \\ud83d\\ude42 C++ admission boundaries C++ it 'single' that mirrors jumps the while that don't They'RE while 42 Z\\u00fcrich 1999 scanner so quick v1.2.3 there's don't keep Z\\u00fcrich no keep faster reservation request is (parens) body we'll 100% don't na\\u00efve (parens) for backtracking fox boundaries backtracking v1.2.3 $1,234.56 we'll 42 tokens faster counting once keep budget fox v1.2.3 'single' mirrors no engine exactly fox 42 way the C++ quick tiktoken's counting 1999 It's we'll counting because hand \\ud83d\\ude42 user@example.com is request quick 42 it keep don't They'RE WON'T once and fox v1.2.3 and with na\\u00efve way lazy user@example.com written is 'single' brown scanned request is caf\\u00e9 we'll node.js a so while we'll WON'T way admission They'RE fox C++ once node.js WON'T that v1.2.3 every\"}, {\"role\": \"assistant\", \"content\": \"C++ don't no boundaries scanner https://example.com/a/b?c=d node.js backtracking hand [brackets] keep F# counting caf\\u00e9 scanner v1.2.3 https://example.com/a/b?c=d counting no https://example.com/a/b?c=d the backtracking it I'M I'M F# involved WON'T with exactly (parens) brown body v1.2.3 body https://example.com/a/b?c=d keep it It's 1999 F# \\u6771\\u4eac so \\ud83d\\ude42 scanner for user@example.com v1.2.3 I'M don't fox written request engine once regex counting tokens the admission (parens) admission reservation faster C++ 3.14159 tokens hand \\\"quotes\\\" counting caf\\u00e9 [brackets] It's don't Z\\u00fcrich 3.14159 the tiktoken's I'M v1.2.3 admission C++ They'RE hand tokens dog no $1,234.56 engine while boundaries so caf\\u00e9 100% \\u0645\\u0631\\u062d\\u0628\\u0627 F# scanned jumps faster while na\\u00efve lazy\"}, {\"role\": \"user\", \"content\": \"WON'T hand keep brown \\u0645\\u0631\\u062d\\u0628\\u0627 because counting faster with that involved is because because 42 Z\\u00fcrich that engine with the the brown {braces} tiktoken's [brackets] I'M It's over (parens) C++ It's over faster \\u6771\\u4eac user@example.com user@example.com 100% on it on with mirrors 3.14159 42 budget It's WON'T node.js keep tiktoken's \\u6771\\u4eac https://example.com/a/b?c=d 1999 involved body scanned hand involved engine faster every is and that tokens every 42 on dog admission (parens) body for caf\\u00e9 once before a \\u6771\\u4eac before with don't tokens It's because \\\"quotes\\\" node.js na\\u00efve it engine is backtracking [brackets] regex brown They'RE so is is involved engine so scanner gateway scanned They'RE keep na\\u00efve body reservation 42 scanned WON'T faster there's backtracking it caf\\u00e9 v1.2.3 brown regex {braces} lazy node.js C++ don't written WON'T with user@example.com piece over we'll v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich jumps (parens)\"}, {\"role\": \"assistant\", \"content\": \"na\\u00efve scanned every reservation no we'll faster before it {braces} admission 1999 scanned budget written there's WON'T $1,234.56 faster brown reservation (parens) 'single' every (parens) $1,234.56 It's is counting way jumps no scanned I'M the node.js that na\\u00efve over faster [brackets] 1999 there's keep user@example.com 100% user@example.com before once reservation brown don't C++ dog it over backtracking every \\\"quotes\\\" quick a budget \\ud83d\\ude42 budget because v1.2.3 dog 100% once v1.2.3 and and every don't 3.14159 once once quick gateway before for 3.14159 v1.2.3 \\u6771\\u4eac tokens that on because it budget a involved scanned scanner lazy 'single' caf\\u00e9 They'RE engine request while Z\\u00fcrich engine WON'T\"}, {\"role\": \"user\", \"content\": \"100% over written reservation https://example.com/a/b?c=d WON'T scanner F# before https://example.com/a/b?c=d we'll that while counting is admission 42 caf\\u00e9 C++ body tiktoken's body 3.14159 the I'M with node.js It's \\u6771\\u4eac [brackets] don't https://example.com/a/b?c=d C++ before on [brackets] 'single' piece brown before 1999 {braces} written {braces} way reservation request F# so https://example.com/a/b?c=d 100% $1,234.56 piece piece keep https://example.com/a/b?c=d way budget 100% boundaries node.js lazy because F# tokens (parens) and C++ backtracking tiktoken's piece a 3.14159 WON'T scanned backtracking node.js because regex backtracking so keep 1999 the is because WON'T node.js we'll lazy quick with once and once \\\"quotes\\\" we'll keep It's the on It's so is faster\"}, {\"role\": \"assistant\", \"content\": \"hand every for written mirrors backtracking $1,234.56 It's WON'T because faster Z\\u00fcrich don't don't {braces} jumps 'single' regex gateway user@example.com once there's a on over \\\"quotes\\\" 100% node.js regex a body that F# 100% once on 3.14159 while backtracking v1.2.3 hand \\u6771\\u4eac tokens admission \\\"quotes\\\" 100% admission and It's Z\\u00fcrich that so while and $1,234.56 caf\\u00e9 every F# involved fox backtracking the \\u0645\\u0631\\u062d\\u0628\\u0627 and https://example.com/a/b?c=d exactly node.js before mirrors 'single' exactly tokens that scanned body https://example.com/a/b?c=d na\\u00efve once it the the faster v1.2.3 scanned fox that jumps v1.2.3 1999 gateway F# caf\\u00e9 'single' fox brown [brackets] 100% over user@example.com on na\\u00efve https://example.com/a/b?c=d node.js They'RE and scanner 'single' involved the because hand scanner fox user@example.com\"}, {\"role\": \"user\", \"content\": \"1999 $1,234.56 that 'single' no counting (parens) because Z\\u00fcrich on (parens) a fox user@example.com admission over is there's \\\"quotes\\\" {braces} exactly piece that with I'M F# over for counting and \\ud83d\\ude42 scanner over every \\ud83d\\ude42 100% admission before \\u0645\\u0631\\u062d\\u0628\\u0627 reservation WON'T brown scanner on faster 100% caf\\u00e9 piece [brackets] counting scanner It's written {braces} C++ that is boundaries exactly \\u0645\\u0631\\u062d\\u0628\\u0627 once 'single' quick F# hand 'single' user@example.com reservation jumps it F# reservation request that 'single' (parens) the way WON'T (parens) a backtracking backtracking that \\\"quotes\\\" C++ I'M C++ C++ gateway with body body\"}, {\"role\": \"assistant\", \"content\": \"budget that 42 It's body backtracking caf\\u00e9 1999 because hand hand F# piece is (parens) and Z\\u00fcrich jumps user@example.com don't I'M They'RE regex body reservation once (parens) with F# piece is every node.js no piece because {braces} with 42 Z\\u00fcrich tiktoken's keep I'M is scanner no body 'single' 1999 that request so \\u0645\\u0631\\u062d\\u0628\\u0627 3.14159 \\u6771\\u4eac and https://example.com/a/b?c=d engine tiktoken's on while $1,234.56 over budget 1999 1999 backtracking is \\ud83d\\ude42 the tiktoken's involved over don't Z\\u00fcrich I'M so gateway written dog before written v1.2.3 backtracking gateway keep dog [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9\"}, {\"role\": \"user\", \"content\": \"exactly involved user@example.com fox keep boundaries and 42 reservation {braces} mirrors the It's budget C++ tiktoken's budget mirrors faster [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 over scanner way quick while \\\"quotes\\\" exactly 'single' F# no faster [brackets] backtracking scanner WON'T WON'T tokens quick that the node.js it regex user@example.com involved while boundaries regex while hand mirrors way na\\u00efve a hand request that before v1.2.3 1999 3.14159 caf\\u00e9 because on over scanner so body tokens quick because counting exactly hand user@example.com that It's\"}, {\"role\": \"assistant\", \"content\": \"so quick dog (parens) Z\\u00fcrich backtracking for the \\ud83d\\ude42 because regex involved tokens dog we'll 'single' we'll Z\\u00fcrich once 100% regex we'll [brackets] 42 while with 1999 Z\\u00fcrich fox admission while written the C++ over every mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 on it https://example.com/a/b?c=d scanned gateway no 1999 dog https://example.com/a/b?c=d we'll 1999 before the every budget on 1999 with piece 1999 faster user@example.com I'M body keep while the once scanned I'M\"}, {\"role\": \"user\", \"content\": \"tiktoken's quick budget on budget the \\u0645\\u0631\\u062d\\u0628\\u0627 for Z\\u00fcrich I'M don't WON'T we'll user@example.com It's jumps \\u6771\\u4eac we'll involved gateway it that jumps for for counting that $1,234.56 so while 'single' WON'T quick the brown we'll once mirrors [brackets] \\u6771\\u4eac \\\"quotes\\\" boundaries way for because {braces} it user@example.com faster $1,234.56 no a 'single' body backtracking before the 3.14159 scanner {braces} backtracking hand because so body [brackets] because that involved scanned WON'T there's \\u0645\\u0631\\u062d\\u0628\\u0627 it hand we'll dog a before the and faster \\ud83d\\ude42 hand 100% on body [brackets] 42 the node.js it counting and jumps we'll a fox 3.14159 once (parens) \\ud83d\\ude42 hand every don't way jumps body over involved \\u6771\\u4eac is budget way scanned $1,234.56 because request gateway involved the that dog the reservation while \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"assistant\", \"content\": \"counting written brown on once it They'RE involved we'll every hand \\ud83d\\ude42 it They'RE dog for reservation on every no tokens regex piece on (parens) tiktoken's before gateway it every the while that we'll gateway a the a v1.2.3 the over lazy node.js gateway budget WON'T \\ud83d\\ude42 is exactly jumps over lazy piece backtracking written with 3.14159 jumps involved lazy scanner keep keep [brackets] boundaries that no that because quick F# v1.2.3 reservation involved\"}, {\"role\": \"user\", \"content\": \"caf\\u00e9 engine request na\\u00efve jumps involved lazy regex scanned {braces} 3.14159 engine mirrors \\\"quotes\\\" Z\\u00fcrich on there's (parens) They'RE regex for over regex 1999 tiktoken's scanner that once admission F# mirrors 42 the body WON'T the WON'T dog 100% 42 is lazy \\ud83d\\ude42 that request once 100% na\\u00efve dog tokens budget jumps \\ud83d\\ude42 $1,234.56 user@example.com $1,234.56 don't admission that brown (parens) it node.js tokens na\\u00efve we'll it v1.2.3 once so is written admission faster way we'll request jumps v1.2.3 fox the lazy gateway $1,234.56 42 counting Z\\u00fcrich the 'single' exactly once way I'M \\u6771\\u4eac They'RE before caf\\u00e9 boundaries over counting piece brown faster while so counting na\\u00efve hand \\ud83d\\ude42 quick the every Z\\u00fcrich {braces} scanned \\\"quotes\\\" with regex keep while once engine before fox\"}, {\"role\": \"assistant\", \"content\": \"mirrors {braces} because \\u6771\\u4eac mirrors scanned exactly budget way mirrors https://example.com/a/b?c=d boundaries involved 100% 3.14159 exactly engine brown They'RE is keep that hand lazy exactly tokens It's keep every \\ud83d\\ude42 the F# written {braces} user@example.com the counting $1,234.56 keep regex tiktoken's and mirrors fox the gateway \\ud83d\\ude42 keep They'RE written I'M there's we'll na\\u00efve WON'T is brown C++ involved brown and piece \\ud83d\\ude42 reservation [brackets] reservation I'M \\\"quotes\\\" while exactly way https://example.com/a/b?c=d don't \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac over keep scanner \\u6771\\u4eac scanned over tiktoken's boundaries mirrors once for quick faster a \\u6771\\u4eac lazy (parens) na\\u00efve gateway \\u6771\\u4eac \\u6771\\u4eac brown They'RE there's on keep once dog that 1999 [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 while it It's no\"}, {\"role\": \"user\", \"content\": \"that body over Z\\u00fcrich before that dog keep and \\\"quotes\\\" and regex tiktoken's way piece is scanner is quick C++ node.js written dog regex the It's dog https://example.com/a/b?c=d scanned engine we'll counting it 'single' counting a boundaries is \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 brown budget fox the Z\\u00fcrich jumps {braces} that boundaries piece because the 'single' v1.2.3 {braces} once that mirrors \\ud83d\\ude42 backtracking no no mirrors on on scanned F# They'RE regex scanned the every for faster v1.2.3 \\ud83d\\ude42 I'M we'll exactly that is once for because it caf\\u00e9 It's caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 keep budget admission\"}, {\"role\": \"assistant\", \"content\": \"and node.js is so admission body They'RE https://example.com/a/b?c=d the 1999 we'll It's keep mirrors request so piece gateway engine way hand exactly is C++ 100% exactly for body piece with jumps WON'T the no keep WON'T so F# don't $1,234.56 {braces} and \\ud83d\\ude42 request reservation written tiktoken's that written way written the tokens 3.14159 WON'T admission \\u0645\\u0631\\u062d\\u0628\\u0627 C++ budget gateway every before brown WON'T piece 1999 [brackets] a dog [brackets] while scanner \\ud83d\\ude42 we'll v1.2.3 scanner quick dog a while admission hand so there's every 42 node.js that engine faster na\\u00efve is \\\"quotes\\\"\"}, {\"role\": \"user\", \"content\": \"scanner hand boundaries admission it \\u6771\\u4eac WON'T na\\u00efve user@example.com Z\\u00fcrich F# F# They'RE a is caf\\u00e9 \\\"quotes\\\" before on request $1,234.56 boundaries (parens) don't brown the every 1999 (parens) [brackets] so 1999 boundaries 100% the hand 42 tokens every over that the the F# that quick They'RE before because tokens caf\\u00e9 we'll exactly boundaries because brown keep no 'single' there's brown v1.2.3 user@example.com 42 1999 \\u6771\\u4eac {braces} for mirrors on tiktoken's WON'T \\\"quotes\\\" na\\u00efve They'RE {braces} v1.2.3 brown written involved tokens jumps boundaries budget jumps boundaries way 42 body written na\\u00efve written is request every admission counting before backtracking with They'RE fox\"}, {\"role\": \"assistant\", \"content\": \"\\ud83d\\ude42 we'll backtracking involved counting request node.js caf\\u00e9 lazy $1,234.56 so I'M while is every it 42 exactly node.js that They'RE there's I'M I'M a and counting admission no 'single' admission v1.2.3 request boundaries 'single' tiktoken's budget gateway 100% caf\\u00e9 jumps hand no the \\u6771\\u4eac I'M we'll fox 1999 piece and body {braces} na\\u00efve hand Z\\u00fcrich v1.2.3 over quick for 100% while backtracking scanned scanner scanner request for hand fox the that hand scanned It's while (parens) written\"}, {\"role\": \"user\", \"content\": \"body there's scanned na\\u00efve quick It's a request lazy hand 1999 https://example.com/a/b?c=d jumps and caf\\u00e9 every It's before we'll on body no lazy there's 'single' mirrors we'll tokens over (parens) while \\u0645\\u0631\\u062d\\u0628\\u0627 C++ v1.2.3 keep gateway it because request mirrors for there's $1,234.56 keep over gateway https://example.com/a/b?c=d brown before They'RE 42 [brackets] tokens once once while scanner scanned and I'M Z\\u00fcrich it involved na\\u00efve I'M hand exactly before node.js 'single' because no lazy 3.14159 It's scanner F# 'single' while exactly It's (parens) \\ud83d\\ude42 counting while for tokens backtracking fox (parens) v1.2.3 once quick because is there's reservation because It's node.js that regex for don't\"}, {\"role\": \"assistant\", \"content\": \"with with so written budget is jumps admission before that I'M \\\"quotes\\\" because 3.14159 on WON'T \\\"quotes\\\" caf\\u00e9 is for scanned engine so engine on admission no body before once node.js node.js [brackets] They'RE Z\\u00fcrich written the \\ud83d\\ude42 reservation once so it a budget the request \\u6771\\u4eac https://example.com/a/b?c=d request \\u0645\\u0631\\u062d\\u0628\\u0627 a the 3.14159 with mirrors because body [brackets] exactly \\ud83d\\ude42 the a 1999 every lazy 'single' the request keep once dog user@example.com the scanned \\\"quotes\\\" is regex scanner every hand keep faster request quick WON'T while 3.14159 It's 3.14159 mirrors reservation written the a It's don't reservation C++ It's 42 v1.2.3 involved reservation lazy keep \\\"quotes\\\" the body \\ud83d\\ude42 while 3.14159 node.js dog no user@example.com budget [brackets] hand over\"}, {\"role\": \"user\", \"content\": \"before 1999 \\ud83d\\ude42 on that don't Z\\u00fcrich keep a way for request \\u6771\\u4eac every \\\"quotes\\\" https://example.com/a/b?c=d 3.14159 once for every hand that WON'T 'single' request written jumps regex F# once quick \\u6771\\u4eac dog na\\u00efve the node.js way [brackets] gateway (parens) piece exactly \\ud83d\\ude42 v1.2.3 jumps They'RE written fox the scanner exactly WON'T na\\u00efve so body faster brown the that is with exactly Z\\u00fcrich [brackets] Z\\u00fcrich budget it lazy and the engine budget 3.14159 counting \\ud83d\\ude42 https://example.com/a/b?c=d it WON'T $1,234.56 {braces} a because we'll exactly it with request the that don't scanner jumps involved and tiktoken's the quick so \\ud83d\\ude42 [brackets] regex engine involved is\"}, {\"role\": \"assistant\", \"content\": \"They'RE body because engine 100% WON'T {braces} the tiktoken's before node.js it every na\\u00efve lazy the don't caf\\u00e9 admission 1999 faster dog 3.14159 {braces} once tokens we'll before the admission WON'T the it and It's because engine with user@example.com it \\u6771\\u4eac C++ over is tokens piece exactly it piece backtracking gateway node.js no the 3.14159 exactly regex fox \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors a 1999 v1.2.3 request 100% scanned reservation node.js I'M the 100% so quick before piece brown caf\\u00e9 gateway 42 involved It's involved admission Z\\u00fcrich and {braces} It's quick \\ud83d\\ude42 body fox counting\"}, {\"role\": \"user\", \"content\": \"I'M admission regex that involved engine it \\ud83d\\ude42 counting every dog hand before scanned the 100% we'll exactly Z\\u00fcrich \\u6771\\u4eac v1.2.3 engine na\\u00efve node.js body don't lazy tiktoken's gateway don't node.js request F# scanner there's 100% scanner fox the a 42 quick na\\u00efve admission while 1999 way regex there's \\\"quotes\\\" scanned is so before tiktoken's involved [brackets] quick 1999 \\u6771\\u4eac 1999 for lazy v1.2.3 regex lazy body F# hand boundaries once mirrors so brown \\\"quotes\\\" dog $1,234.56 tokens that once request hand tokens lazy lazy [brackets] quick don't mirrors quick don't lazy It's body boundaries mirrors F# [brackets] there's while 100% body\"}, {\"role\": \"assistant\", \"content\": \"budget before over They'RE I'M request is that while written before $1,234.56 {braces} F# that admission and with for [brackets] fox \\ud83d\\ude42 42 body \\\"quotes\\\" the F# reservation dog 3.14159 WON'T It's tiktoken's exactly regex 3.14159 there's involved backtracking don't 1999 with faster dog body before tiktoken's with and over 42 written there's $1,234.56 hand 42 \\\"quotes\\\" lazy gateway backtracking 100% because \\\"quotes\\\" hand caf\\u00e9 faster request on over \\\"quotes\\\" once scanned over 100% engine mirrors \\u6771\\u4eac gateway user@example.com fox every before [brackets] body admission WON'T engine no dog for node.js gateway node.js na\\u00efve piece the way scanner a the 'single' gateway user@example.com They'RE is don't \\u6771\\u4eac before scanned\"}, {\"role\": \"user\", \"content\": \"It's way that \\u0645\\u0631\\u062d\\u0628\\u0627 written tiktoken's over faster It's with WON'T it because faster because with backtracking Z\\u00fcrich the while because scanned mirrors that scanned They'RE fox exactly 100% It's the tiktoken's hand so user@example.com request \\u6771\\u4eac faster v1.2.3 quick on once over dog piece counting node.js They'RE the because (parens) hand F# \\u6771\\u4eac with hand dog over budget written Z\\u00fcrich jumps there's budget C++ backtracking v1.2.3 'single' don't fox They'RE for \\ud83d\\ude42 admission involved v1.2.3 3.14159 the while 42 quick regex 100% on node.js tokens scanner written every with They'RE \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"100% because user@example.com caf\\u00e9 way 1999 Z\\u00fcrich over written 'single' 3.14159 $1,234.56 tokens I'M C++ [brackets] na\\u00efve $1,234.56 we'll $1,234.56 node.js we'll it (parens) {braces} v1.2.3 with written admission 100% exactly {braces} reservation there's {braces} fox with engine na\\u00efve for 1999 brown hand while dog backtracking backtracking na\\u00efve user@example.com tokens \\u0645\\u0631\\u062d\\u0628\\u0627 body brown don't C++ piece over scanner (parens) 100% no F# keep there's v1.2.3 [brackets] we'll don't hand dog every engine 'single' v1.2.3\"}, {\"role\": \"user\", \"content\": \"before and every $1,234.56 caf\\u00e9 the mirrors reservation https://example.com/a/b?c=d 3.14159 the quick gateway exactly it $1,234.56 1999 v1.2.3 $1,234.56 it on node.js backtracking (parens) there's \\ud83d\\ude42 no quick regex \\ud83d\\ude42 before once \\u0645\\u0631\\u062d\\u0628\\u0627 involved gateway caf\\u00e9 admission $1,234.56 caf\\u00e9 backtracking scanner na\\u00efve a engine [brackets] a way written is caf\\u00e9 1999 gateway 100% boundaries [brackets] backtracking there's tokens while 42 na\\u00efve counting it C++ v1.2.3 100% v1.2.3 scanned \\u0645\\u0631\\u062d\\u0628\\u0627 'single' \\u6771\\u4eac WON'T there's keep It's so WON'T\"}, {\"role\": \"assistant\", \"content\": \"keep 42 because quick and because faster is 'single' scanned fox regex {braces} C++ I'M it 100% written tiktoken's engine so scanned reservation \\ud83d\\ude42 3.14159 \\\"quotes\\\" before the that exactly for with there's 100% over with WON'T It's quick Z\\u00fcrich v1.2.3 'single' the tiktoken's (parens) while we'll written [brackets] reservation the on engine way na\\u00efve caf\\u00e9 exactly $1,234.56 https://example.com/a/b?c=d before written 'single' (parens) piece They'RE Z\\u00fcrich written counting the every that written na\\u00efve hand admission over piece Z\\u00fcrich we'll no quick https://example.com/a/b?c=d keep we'll while 'single' request quick on we'll tiktoken's on regex Z\\u00fcrich jumps written that a because mirrors\"}, {\"role\": \"user\", \"content\": \"\\\"quotes\\\" reservation na\\u00efve there's 100% dog reservation [brackets] 3.14159 F# node.js \\u0645\\u0631\\u062d\\u0628\\u0627 the fox boundaries C++ na\\u00efve before It's so written brown C++ lazy 3.14159 100% [brackets] tokens lazy tiktoken's {braces} with and and fox I'M for the exactly gateway keep backtracking that fox Z\\u00fcrich quick lazy every \\u0645\\u0631\\u062d\\u0628\\u0627 hand Z\\u00fcrich \\u6771\\u4eac caf\\u00e9 C++ there's keep the over v1.2.3 mirrors user@example.com It's They'RE tiktoken's 1999 because piece jumps it dog a budget boundaries brown exactly piece lazy 100% so scanned https://example.com/a/b?c=d there's that don't the engine fox tiktoken's 42 $1,234.56 \\u6771\\u4eac keep the dog admission boundaries piece It's mirrors fox user@example.com boundaries I'M faster reservation It's for no fox \\u0645\\u0631\\u062d\\u0628\\u0627 and because boundaries budget involved written They'RE v1.2.3 admission while brown F# so 'single' body that\"}, {\"role\": \"assistant\", \"content\": \"body tokens exactly over dog Z\\u00fcrich WON'T exactly F# no engine and \\ud83d\\ude42 scanner way {braces} scanner jumps tokens piece [brackets] hand that na\\u00efve \\\"quotes\\\" dog for node.js the v1.2.3 that They'RE once gateway so {braces} with keep that regex no [brackets] admission on caf\\u00e9 fox scanner tokens don't before quick They'RE budget dog C++ once jumps user@example.com a node.js (parens) because WON'T \\ud83d\\ude42 while request 3.14159 with because there's regex don't Z\\u00fcrich written user@example.com with while request keep They'RE mirrors 1999 over involved \\u6771\\u4eac $1,234.56 C++ counting involved that tokens way for \\ud83d\\ude42 regex jumps we'll because na\\u00efve lazy with I'M don't before regex reservation fox that so\"}, {\"role\": \"user\", \"content\": \"involved because dog 1999 exactly 1999 keep boundaries node.js tiktoken's once Z\\u00fcrich [brackets] way na\\u00efve v1.2.3 faster C++ scanner mirrors \\u6771\\u4eac \\u6771\\u4eac [brackets] hand \\ud83d\\ude42 node.js tiktoken's don't counting exactly tiktoken's keep WON'T while with 42 brown it way Z\\u00fcrich over because keep before way {braces} the before way boundaries that \\u6771\\u4eac 1999 request mirrors over It's gateway with scanner It's tiktoken's Z\\u00fcrich gateway faster we'll Z\\u00fcrich admission\"}, {\"role\": \"assistant\", \"content\": \"once They'RE node.js we'll \\u0645\\u0631\\u062d\\u0628\\u0627 don't Z\\u00fcrich admission {braces} It's \\ud83d\\ude42 \\u6771\\u4eac gateway brown is so scanner [brackets] {braces} there's tiktoken's a \\u0645\\u0631\\u062d\\u0628\\u0627 node.js and [brackets] don't counting \\\"quotes\\\" They'RE every mirrors reservation body on piece on regex node.js a 'single' \\u6771\\u4eac tiktoken's \\\"quotes\\\" for scanned there's faster gateway They'RE mirrors jumps https://example.com/a/b?c=d faster tokens every WON'T WON'T that request budget It's counting we'll scanner faster tokens that [brackets] tiktoken's there's F# lazy the reservation on the because with lazy user@example.com $1,234.56 jumps gateway C++ [brackets]\"}, {\"role\": \"user\", \"content\": \"over on regex lazy piece don't (parens) don't so https://example.com/a/b?c=d WON'T v1.2.3 brown node.js 100% \\u0645\\u0631\\u062d\\u0628\\u0627 over admission {braces} https://example.com/a/b?c=d scanned hand It's tiktoken's C++ quick don't They'RE {braces} WON'T scanner there's admission the body $1,234.56 dog \\u6771\\u4eac Z\\u00fcrich involved (parens) for so hand 3.14159 so scanned \\u0645\\u0631\\u062d\\u0628\\u0627 42 100% \\u6771\\u4eac counting tiktoken's 1999 fox keep reservation caf\\u00e9 there's tokens quick F# no is v1.2.3 body \\ud83d\\ude42 brown dog way boundaries scanned node.js quick over admission user@example.com boundaries faster regex on 42 admission na\\u00efve engine lazy WON'T that user@example.com 100% I'M with because faster exactly way for faster C++ because scanner written the \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"there's for I'M on reservation once once mirrors the body body 3.14159 fox dog \\\"quotes\\\" written we'll \\ud83d\\ude42 regex with user@example.com fox 1999 Z\\u00fcrich na\\u00efve is hand hand before on for over piece exactly that the exactly 100% caf\\u00e9 that brown counting way admission 100% 1999 v1.2.3 don't \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} mirrors it once piece Z\\u00fcrich gateway caf\\u00e9 and the budget that because hand scanner tokens request regex for exactly over piece F# we'll [brackets] I'M while and fox scanner \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} on fox quick mirrors It's scanner the scanner no it because node.js 'single' regex written caf\\u00e9 v1.2.3 https://example.com/a/b?c=d {braces} the don't jumps It's hand 'single' scanned 3.14159 involved every fox over we'll keep there's mirrors written on I'M WON'T F# that F# that 3.14159 1999 keep 'single'\"}, {\"role\": \"user\", \"content\": \"exactly involved caf\\u00e9 that 1999 over a there's it F# exactly \\ud83d\\ude42 tokens we'll don't regex fox because that while involved backtracking involved reservation there's over They'RE admission 1999 regex counting keep $1,234.56 for engine it 'single' regex the that body v1.2.3 WON'T once It's 42 tiktoken's written no 'single' piece fox and before it tiktoken's a caf\\u00e9 v1.2.3 user@example.com quick user@example.com scanned admission the scanner on is over reservation request \\u6771\\u4eac we'll 3.14159 with before \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking 3.14159 counting reservation don't way with WON'T involved before It's admission keep C++ written with counting 'single' brown the 100% They'RE 42 before while every and \\\"quotes\\\" for They'RE there's 42 tokens v1.2.3 https://example.com/a/b?c=d before involved there's body once\"}, {\"role\": \"assistant\", \"content\": \"Z\\u00fcrich backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 the 3.14159 involved \\\"quotes\\\" 42 node.js once scanner mirrors {braces} na\\u00efve I'M brown the admission over They'RE piece it faster so request don't over way na\\u00efve (parens) once mirrors 100% don't scanner Z\\u00fcrich hand budget fox there's there's boundaries Z\\u00fcrich don't \\u6771\\u4eac scanner 42 v1.2.3 \\\"quotes\\\" na\\u00efve 3.14159 100% every we'll 3.14159 quick 1999 backtracking {braces} don't lazy dog lazy a dog the tokens with body that quick brown tokens lazy \\u6771\\u4eac piece piece every dog every 1999 \\\"quotes\\\" mirrors faster fox the 3.14159 way piece boundaries user@example.com there's caf\\u00e9 faster on F# https://example.com/a/b?c=d we'll admission \\u0645\\u0631\\u062d\\u0628\\u0627 that written piece scanned counting and admission we'll Z\\u00fcrich dog 3.14159 node.js every engine Z\\u00fcrich body involved backtracking mirrors 100% 100%\"}, {\"role\": \"user\", \"content\": \"(parens) that on 3.14159 \\u6771\\u4eac over is 42 \\\"quotes\\\" 'single' body faster exactly mirrors F# way (parens) no exactly mirrors that \\ud83d\\ude42 $1,234.56 faster brown Z\\u00fcrich a brown tiktoken's \\ud83d\\ude42 tokens regex while They'RE I'M budget node.js https://example.com/a/b?c=d 42 tiktoken's https://example.com/a/b?c=d over over every 'single' 100% tiktoken's scanner gateway for don't tiktoken's a hand v1.2.3 {braces} fox that quick every that fox faster exactly 'single' (parens) v1.2.3 request dog over no is jumps for Z\\u00fcrich WON'T so body once scanner 3.14159 every that \\u6771\\u4eac so\"}, {\"role\": \"assistant\", \"content\": \"no while faster the because there's \\\"quotes\\\" piece tiktoken's a reservation Z\\u00fcrich because caf\\u00e9 gateway tokens gateway over Z\\u00fcrich They'RE that 3.14159 quick They'RE na\\u00efve mirrors faster a 100% reservation I'M on F# lazy 1999 and budget and $1,234.56 exactly budget written fox 100% body and there's don't once body the on while budget \\ud83d\\ude42 100% \\\"quotes\\\" request \\u0645\\u0631\\u062d\\u0628\\u0627 keep lazy 1999 before piece that before involved \\ud83d\\ude42 written don't gateway the once user@example.com (parens) that backtracking budget request involved the v1.2.3 42 hand tokens mirrors I'M written \\u6771\\u4eac \\\"quotes\\\" body [brackets] written 'single' brown so request 3.14159 node.js budget node.js dog (parens) scanner {braces} a WON'T v1.2.3 because there's that engine tiktoken's every no over before on mirrors mirrors don't \\u0645\\u0631\\u062d\\u0628\\u0627 budget for quick 42 quick over lazy over v1.2.3 gateway \\u0645\\u0631\\u062d\\u0628\\u0627 and\"}, {\"role\": \"user\", \"content\": \"that caf\\u00e9 user@example.com hand over we'll no that Z\\u00fcrich brown 42 so fox counting quick every regex https://example.com/a/b?c=d brown once body 'single' reservation v1.2.3 $1,234.56 I'M that on \\ud83d\\ude42 F# scanner faster over F# we'll dog because mirrors \\ud83d\\ude42 we'll scanned regex budget on and I'M admission is written It's fox na\\u00efve with WON'T involved 'single' user@example.com WON'T the [brackets] exactly 42 'single' {braces} involved user@example.com They'RE written before keep tokens dog It's over on \\ud83d\\ude42 a that 1999 reservation I'M for that fox it boundaries no hand for $1,234.56 They'RE jumps hand WON'T https://example.com/a/b?c=d faster over [brackets] 3.14159 Z\\u00fcrich\"}, {\"role\": \"assistant\", \"content\": \"100% the quick scanner is with node.js [brackets] 100% keep scanned because brown every that $1,234.56 \\\"quotes\\\" because 3.14159 it 42 tiktoken's because once 1999 user@example.com node.js the admission it quick scanner involved It's $1,234.56 user@example.com 1999 3.14159 exactly https://example.com/a/b?c=d na\\u00efve dog counting exactly quick v1.2.3 piece involved once v1.2.3 exactly 42 scanner gateway (parens) request \\u6771\\u4eac fox keep 42 node.js while keep with the gateway with F# over no involved caf\\u00e9 [brackets] request They'RE scanner on with v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 every the 42 It's every jumps a scanned involved the brown scanned quick the tokens I'M I'M over no over faster v1.2.3 caf\\u00e9 \\\"quotes\\\" body request that fox fox before node.js brown dog a dog body engine faster Z\\u00fcrich https://example.com/a/b?c=d before I'M we'll hand WON'T admission and counting while on quick and\"}, {\"role\": \"user\", \"content\": \"na\\u00efve \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 mirrors mirrors while tiktoken's tiktoken's hand Z\\u00fcrich there's backtracking brown brown \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors is (parens) it that tokens v1.2.3 way admission that caf\\u00e9 so exactly engine with scanner budget WON'T \\u6771\\u4eac 'single' hand na\\u00efve there's 'single' \\u6771\\u4eac piece na\\u00efve $1,234.56 is before faster admission request \\ud83d\\ude42 \\u6771\\u4eac counting backtracking written \\\"quotes\\\" there's involved lazy quick faster that request scanned v1.2.3 $1,234.56 every dog with for regex once with [brackets] v1.2.3 so caf\\u00e9 It's written it Z\\u00fcrich lazy caf\\u00e9 scanner counting there's no caf\\u00e9 every F# 100% the C++ every faster we'll on tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 engine $1,234.56 faster faster 1999 that exactly mirrors for admission budget 'single' 3.14159 no engine\"}, {\"role\": \"assistant\", \"content\": \"https://example.com/a/b?c=d there's gateway mirrors body mirrors written for brown node.js for brown node.js 'single' scanner na\\u00efve request is every 100% user@example.com dog 3.14159 no we'll that Z\\u00fcrich don't involved with https://example.com/a/b?c=d fox dog F# reservation C++ a v1.2.3 v1.2.3 backtracking WON'T F# no piece C++ {braces} a mirrors way $1,234.56 that WON'T body They'RE no we'll jumps F# once caf\\u00e9 reservation WON'T that piece 1999 They'RE counting piece They'RE request tiktoken's a engine na\\u00efve scanned body lazy faster It's over It's that tokens Z\\u00fcrich C++ scanner with \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"user\", \"content\": \"scanner while tiktoken's fox on boundaries tiktoken's tokens before tiktoken's it body gateway and over (parens) the (parens) 'single' on C++ faster budget every request once on exactly every that it Z\\u00fcrich quick regex and user@example.com jumps it user@example.com quick 100% it exactly every 'single' mirrors body jumps counting caf\\u00e9 tokens admission that piece scanner {braces} tokens is \\ud83d\\ude42 request admission C++ is node.js body for body They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking WON'T\"}, {\"role\": \"assistant\", \"content\": \"dog C++ (parens) It's with 'single' we'll [brackets] request is there's 3.14159 scanned 100% for request on admission $1,234.56 over there's scanned engine scanner They'RE that lazy \\u6771\\u4eac before budget user@example.com once C++ $1,234.56 that scanner exactly tokens is engine \\ud83d\\ude42 dog user@example.com na\\u00efve They'RE we'll \\ud83d\\ude42 that 'single' jumps the Z\\u00fcrich on jumps involved hand involved tiktoken's It's C++ \\ud83d\\ude42 user@example.com that before user@example.com $1,234.56 don't once scanner and a user@example.com user@example.com exactly and user@example.com $1,234.56 admission backtracking budget written 100% we'll 1999 scanned mirrors every \\u0645\\u0631\\u062d\\u0628\\u0627 a brown dog mirrors reservation 1999 regex hand budget \\u6771\\u4eac user@example.com it before 'single' hand over with {braces} no the a keep quick tiktoken's 3.14159 caf\\u00e9 every Z\\u00fcrich F# \\u6771\\u4eac the I'M Z\\u00fcrich na\\u00efve no way gateway na\\u00efve https://example.com/a/b?c=d request {braces} backtracking\"}, {\"role\": \"user\", \"content\": \"\\\"quotes\\\" \\\"quotes\\\" scanned is lazy user@example.com v1.2.3 regex a It's tokens piece They'RE engine over Z\\u00fcrich 3.14159 {braces} 'single' (parens) a boundaries node.js request jumps body before for hand dog boundaries budget on the boundaries we'll every because {braces} Z\\u00fcrich way https://example.com/a/b?c=d 1999 body Z\\u00fcrich there's budget 1999 before so v1.2.3 WON'T user@example.com F# it written lazy with once dog faster faster once we'll \\ud83d\\ude42 exactly user@example.com faster 42 \\u0645\\u0631\\u062d\\u0628\\u0627 a admission over \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T lazy\"}, {\"role\": \"assistant\", \"content\": \"na\\u00efve the (parens) for the brown https://example.com/a/b?c=d and 42 Z\\u00fcrich on {braces} \\ud83d\\ude42 regex over regex na\\u00efve F# over C++ before with the for don't 'single' counting \\u6771\\u4eac tiktoken's before mirrors fox gateway faster so regex \\ud83d\\ude42 keep $1,234.56 reservation so \\ud83d\\ude42 42 https://example.com/a/b?c=d tokens user@example.com for the for exactly \\\"quotes\\\" it written for F# dog I'M engine faster \\u6771\\u4eac gateway dog na\\u00efve is {braces} tokens over request backtracking 1999 body request written 'single' They'RE tokens F# tokens brown C++ boundaries admission every budget we'll no budget it keep don't https://example.com/a/b?c=d every exactly caf\\u00e9 'single' there's C++ 'single' dog v1.2.3 backtracking They'RE engine (parens) 'single' gateway every it $1,234.56 1999 lazy once reservation over fox that {braces} 1999 budget node.js written C++ way [brackets] before \\ud83d\\ude42 gateway on tiktoken's and (parens) the written \\u6771\\u4eac\"}, {\"role\": \"user\", \"content\": \"user@example.com scanned 42 user@example.com piece backtracking every engine that \\u0645\\u0631\\u062d\\u0628\\u0627 request $1,234.56 lazy Z\\u00fcrich mirrors fox no that so It's {braces} It's piece mirrors v1.2.3 regex quick counting for They'RE F# 3.14159 once brown no that no \\u6771\\u4eac so 100% scanner 1999 counting fox there's once engine caf\\u00e9 brown [brackets] piece hand mirrors budget lazy over the counting fox counting is for I'M faster a with node.js we'll \\u6771\\u4eac it 1999 user@example.com that It's mirrors written lazy lazy the before faster piece na\\u00efve fox piece 3.14159 fox na\\u00efve dog regex [brackets] [brackets] \\u6771\\u4eac over body user@example.com 42 node.js while and https://example.com/a/b?c=d na\\u00efve piece faster brown (parens) \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 \\\"quotes\\\"\"}, {\"role\": \"assistant\", \"content\": \"faster written 100% $1,234.56 engine 'single' gateway faster the boundaries exactly on C++ so so mirrors once backtracking caf\\u00e9 quick 1999 while we'll WON'T a Z\\u00fcrich boundaries involved piece scanner {braces} written 1999 and 3.14159 there's \\ud83d\\ude42 that the because quick scanner 3.14159 that it way counting na\\u00efve reservation for before $1,234.56 over quick engine scanner no the we'll scanned backtracking that no don't before 1999 \\u6771\\u4eac we'll before 3.14159 $1,234.56 budget counting tokens jumps while before tokens node.js tokens gateway scanner that \\ud83d\\ude42 3.14159 'single' WON'T so way (parens) backtracking \\\"quotes\\\" the v1.2.3 {braces} body body C++ it 'single'\"}, {\"role\": \"user\", \"content\": \"because request that that dog counting 100% mirrors I'M fox there's the piece with budget regex while before tokens with over there's \\u0645\\u0631\\u062d\\u0628\\u0627 reservation body on reservation tiktoken's and \\u0645\\u0631\\u062d\\u0628\\u0627 while so quick that budget quick brown and user@example.com \\u6771\\u4eac 100% dog keep there's It's WON'T caf\\u00e9 quick there's $1,234.56 way user@example.com budget for v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 quick way boundaries is reservation 42 no that \\ud83d\\ude42 the hand \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) na\\u00efve $1,234.56 3.14159 quick 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors na\\u00efve hand once every F# the \\ud83d\\ude42 once piece I'M 42 with involved the {braces} [brackets] a 'single' caf\\u00e9 that once before once 'single' counting {braces} every reservation budget jumps WON'T\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 {braces} it They'RE scanned for \\u6771\\u4eac hand mirrors written 42 with node.js quick admission I'M counting \\u0645\\u0631\\u062d\\u0628\\u0627 brown \\u6771\\u4eac 100% over \\u6771\\u4eac that \\ud83d\\ude42 \\u6771\\u4eac It's \\ud83d\\ude42 scanner budget so dog that regex keep hand piece lazy no is user@example.com counting tiktoken's request admission we'll mirrors hand gateway boundaries no https://example.com/a/b?c=d on engine reservation no that {braces} mirrors brown Z\\u00fcrich admission piece user@example.com scanned and with 100% the tiktoken's fox WON'T written\"}, {\"role\": \"user\", \"content\": \"Z\\u00fcrich is over 'single' C++ It's gateway [brackets] no over way quick admission no faster written node.js 42 v1.2.3 involved {braces} 3.14159 because [brackets] reservation body before the exactly every [brackets] tokens so fox piece boundaries boundaries v1.2.3 caf\\u00e9 I'M F# 1999 backtracking user@example.com with with don't body a C++ brown don't It's caf\\u00e9 regex They'RE way https://example.com/a/b?c=d scanned 'single' boundaries that reservation \\ud83d\\ude42 brown fox node.js backtracking gateway na\\u00efve (parens)\"}, {\"role\": \"assistant\", \"content\": \"piece WON'T body it $1,234.56 that 3.14159 1999 it reservation \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's boundaries 'single' It's reservation gateway C++ body budget mirrors is the backtracking and \\u0645\\u0631\\u062d\\u0628\\u0627 gateway a 'single' is and while 42 hand it They'RE caf\\u00e9 backtracking counting gateway exactly F# because na\\u00efve scanned that C++ is 100% involved admission piece 3.14159 quick it the lazy involved WON'T a before It's tokens They'RE it mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 reservation it \\ud83d\\ude42 na\\u00efve so is a and that request while that written body keep reservation the on every They'RE WON'T budget scanned reservation $1,234.56 fox is there's $1,234.56 brown involved admission\"}, {\"role\": \"user\", \"content\": \"way F# don't brown v1.2.3 the regex na\\u00efve involved brown budget because dog that WON'T and v1.2.3 for it counting because because because brown \\u6771\\u4eac backtracking (parens) I'M that dog reservation keep that scanner request WON'T They'RE backtracking admission every 42 because 100% [brackets] https://example.com/a/b?c=d admission way the so don't I'M \\ud83d\\ude42 3.14159 that C++ 3.14159 https://example.com/a/b?c=d na\\u00efve \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com node.js way I'M over https://example.com/a/b?c=d engine once mirrors Z\\u00fcrich F# fox on \\u0645\\u0631\\u062d\\u0628\\u0627 na\\u00efve mirrors piece with the \\ud83d\\ude42 WON'T counting involved don't written boundaries gateway keep 1999 boundaries so Z\\u00fcrich way piece every user@example.com exactly They'RE piece boundaries written brown I'M C++ gateway keep the tiktoken's reservation counting Z\\u00fcrich a involved\"}, {\"role\": \"assistant\", \"content\": \"way is a for scanned dog exactly don't body na\\u00efve [brackets] fox na\\u00efve {braces} that body for scanner reservation user@example.com quick fox hand faster reservation mirrors (parens) counting no we'll a exactly reservation user@example.com tiktoken's hand tokens don't C++ body na\\u00efve hand with backtracking for engine for They'RE admission {braces} fox boundaries keep quick it with once body the body the budget \\\"quotes\\\" caf\\u00e9 a admission $1,234.56 we'll way v1.2.3 3.14159 reservation fox WON'T 3.14159 once that written Z\\u00fcrich for na\\u00efve once piece tiktoken's over engine \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 F# 3.14159 lazy \\ud83d\\ude42 user@example.com na\\u00efve \\u6771\\u4eac don't because 'single' They'RE \\ud83d\\ude42 \\u6771\\u4eac body because lazy a for exactly request tiktoken's gateway no C++ before with every {braces} keep v1.2.3 faster mirrors (parens) (parens) and quick scanned I'M budget engine with\"}, {\"role\": \"user\", \"content\": \"regex with exactly \\u0645\\u0631\\u062d\\u0628\\u0627 dog budget that 42 {braces} tokens https://example.com/a/b?c=d tiktoken's that I'M we'll 100% engine counting F# no regex node.js no boundaries way tiktoken's because \\\"quotes\\\" exactly 3.14159 and [brackets] mirrors written keep that F# backtracking scanned They'RE it gateway over {braces} node.js 100% budget scanner budget \\\"quotes\\\" we'll v1.2.3 we'll caf\\u00e9 \\ud83d\\ude42 tokens don't once because mirrors backtracking na\\u00efve every faster once budget caf\\u00e9 it I'M caf\\u00e9 a backtracking regex fox while engine faster scanned I'M tiktoken's don't \\\"quotes\\\" 3.14159 scanner It's na\\u00efve request boundaries hand lazy no $1,234.56 every for jumps written 1999 mirrors {braces} WON'T lazy engine and\"}, {\"role\": \"assistant\", \"content\": \"and reservation caf\\u00e9 it (parens) before piece backtracking and WON'T brown counting that 1999 $1,234.56 tokens user@example.com user@example.com it faster It's the body and request brown caf\\u00e9 keep scanner \\u0645\\u0631\\u062d\\u0628\\u0627 for tokens caf\\u00e9 'single' request boundaries 'single' request They'RE user@example.com tokens quick there's keep exactly a scanned is regex it hand quick hand {braces} so 'single' gateway and\"}, {\"role\": \"user\", \"content\": \"na\\u00efve is request that It's the way caf\\u00e9 the 42 there's quick user@example.com the over that $1,234.56 with user@example.com quick They'RE budget the it F# gateway It's fox we'll involved tiktoken's na\\u00efve over while https://example.com/a/b?c=d that na\\u00efve tiktoken's 42 scanned we'll na\\u00efve and tiktoken's (parens) piece 'single' exactly Z\\u00fcrich budget WON'T (parens) It's engine we'll body involved and keep\"}, {\"role\": \"assistant\", \"content\": \"way v1.2.3 it so once before \\\"quotes\\\" WON'T WON'T written and over F# the with no boundaries v1.2.3 body that dog exactly Z\\u00fcrich backtracking there's on \\u6771\\u4eac mirrors request on They'RE hand regex engine reservation it scanned \\\"quotes\\\" hand boundaries for reservation before brown WON'T once hand \\u0645\\u0631\\u062d\\u0628\\u0627 body 42 \\u0645\\u0631\\u062d\\u0628\\u0627 once every admission body $1,234.56 before $1,234.56 WON'T {braces} na\\u00efve mirrors before the that 3.14159 way brown It's 1999 gateway reservation F# for user@example.com exactly jumps https://example.com/a/b?c=d is and node.js 100% boundaries v1.2.3 'single' we'll no gateway They'RE with F# [brackets] C++ mirrors for tokens written every\"}, {\"role\": \"user\", \"content\": \"WON'T na\\u00efve 'single' It's 3.14159 \\ud83d\\ude42 don't 'single' \\\"quotes\\\" involved C++ caf\\u00e9 with mirrors 100% It's that no user@example.com WON'T before is gateway once node.js gateway while on quick \\ud83d\\ude42 boundaries \\u6771\\u4eac budget 1999 admission They'RE mirrors engine I'M is \\u0645\\u0631\\u062d\\u0628\\u0627 I'M budget for 'single' on \\u6771\\u4eac with WON'T and because brown regex hand reservation scanner 100% 100% user@example.com there's it\"}, {\"role\": \"assistant\", \"content\": \"there's exactly [brackets] faster budget while $1,234.56 (parens) engine caf\\u00e9 They'RE no 100% for because $1,234.56 regex Z\\u00fcrich https://example.com/a/b?c=d tokens jumps because counting brown user@example.com request lazy \\u0645\\u0631\\u062d\\u0628\\u0627 request scanned WON'T $1,234.56 exactly brown every mirrors with 'single' engine 42 \\ud83d\\ude42 engine reservation hand lazy 'single' counting hand (parens) gateway that counting user@example.com 42 fox request \\\"quotes\\\" hand lazy 100% 42 I'M It's WON'T https://example.com/a/b?c=d engine while\"}, {\"role\": \"user\", \"content\": \"C++ \\\"quotes\\\" we'll gateway \\ud83d\\ude42 counting written piece na\\u00efve \\u6771\\u4eac WON'T over caf\\u00e9 backtracking engine https://example.com/a/b?c=d the v1.2.3 that so quick counting na\\u00efve \\u6771\\u4eac don't WON'T It's we'll that the 100% request I'M brown na\\u00efve tokens engine budget before the scanned mirrors dog engine {braces} {braces} 42 don't WON'T It's piece \\\"quotes\\\" that no quick 3.14159 F# once fox don't for scanner \\u6771\\u4eac it brown backtracking tokens $1,234.56 reservation written They'RE the \\\"quotes\\\" caf\\u00e9 backtracking C++ reservation\"}, {\"role\": \"assistant\", \"content\": \"regex (parens) Z\\u00fcrich engine https://example.com/a/b?c=d written tiktoken's C++ dog It's lazy dog exactly with it that admission way no scanner a 3.14159 on They'RE It's involved caf\\u00e9 brown on dog that https://example.com/a/b?c=d don't \\u6771\\u4eac jumps \\ud83d\\ude42 and \\u6771\\u4eac so dog so don't (parens) scanned engine budget the exactly F# tokens over \\u0645\\u0631\\u062d\\u0628\\u0627 jumps backtracking 3.14159 budget 100% boundaries gateway backtracking don't scanner dog with [brackets] gateway 3.14159 caf\\u00e9 faster budget na\\u00efve caf\\u00e9 way for every reservation backtracking request They'RE It's \\u6771\\u4eac written\"}, {\"role\": \"user\", \"content\": \"42 1999 so boundaries $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} budget backtracking [brackets] we'll admission piece node.js admission Z\\u00fcrich fox a involved request is it node.js node.js so \\u6771\\u4eac [brackets] no \\u6771\\u4eac brown node.js quick na\\u00efve scanner C++ user@example.com admission 'single' involved 1999 user@example.com I'M v1.2.3 It's 100% there's gateway F# node.js gateway scanned that caf\\u00e9 3.14159 Z\\u00fcrich no admission {braces} tiktoken's over It's I'M 1999 [brackets] body written scanned that scanner on the on with that scanned hand\"}, {\"role\": \"assistant\", \"content\": \"{braces} regex [brackets] piece \\ud83d\\ude42 mirrors there's tiktoken's It's lazy mirrors every v1.2.3 I'M that boundaries 'single' node.js on C++ user@example.com budget user@example.com no user@example.com every engine while a I'M over [brackets] WON'T piece over there's lazy I'M that $1,234.56 budget https://example.com/a/b?c=d it a request exactly 42 keep They'RE caf\\u00e9 tokens \\u0645\\u0631\\u062d\\u0628\\u0627 keep scanner while dog F# mirrors backtracking we'll gateway scanner so we'll request there's budget it lazy scanner for exactly faster we'll reservation scanner no WON'T request (parens) tiktoken's exactly before so while hand involved with $1,234.56 mirrors 3.14159 on the [brackets] boundaries 100% with request way engine keep the https://example.com/a/b?c=d \\u6771\\u4eac the WON'T that so\"}, {\"role\": \"user\", \"content\": \"body hand backtracking (parens) counting na\\u00efve They'RE \\ud83d\\ude42 WON'T boundaries v1.2.3 budget na\\u00efve admission once the that because tokens dog tiktoken's scanner scanner (parens) I'M \\\"quotes\\\" budget 100% there's once WON'T keep {braces} the before fox (parens) the counting because lazy 1999 \\u0645\\u0631\\u062d\\u0628\\u0627 admission 1999 and there's while admission 3.14159 a counting C++ with exactly don't don't lazy \\ud83d\\ude42 42 \\\"quotes\\\" backtracking C++ there's keep user@example.com request is node.js piece before brown budget 42 100% because admission and with tiktoken's no the 100%\"}, {\"role\": \"assistant\", \"content\": \"with 1999 Z\\u00fcrich that is with mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 the brown $1,234.56 way scanner https://example.com/a/b?c=d lazy written scanner quick v1.2.3 \\\"quotes\\\" regex \\\"quotes\\\" They'RE 3.14159 \\ud83d\\ude42 [brackets] Z\\u00fcrich the counting for so $1,234.56 It's WON'T the quick hand that mirrors the 42 caf\\u00e9 exactly node.js we'll backtracking while I'M and fox is before written it 1999 no for involved \\u0645\\u0631\\u062d\\u0628\\u0627 the \\\"quotes\\\" piece the way involved that no 100% before there's 100% we'll is [brackets] C++ the we'll the node.js reservation is while [brackets] engine scanner we'll no C++ don't dog v1.2.3 that so tiktoken's before \\ud83d\\ude42 mirrors 'single' it once caf\\u00e9 there's backtracking gateway body over quick way and that once \\u6771\\u4eac 'single' we'll (parens) and before for quick They'RE 1999 regex there's \\ud83d\\ude42\"}, {\"role\": \"user\", \"content\": \"engine They'RE https://example.com/a/b?c=d we'll quick tokens written WON'T don't regex backtracking It's it lazy 1999 brown budget on [brackets] node.js {braces} They'RE involved a mirrors exactly tokens with $1,234.56 F# 42 backtracking involved engine It's and {braces} \\u6771\\u4eac Z\\u00fcrich it it [brackets] gateway keep na\\u00efve request scanned don't regex we'll C++ counting tokens so request while scanned tokens scanner for v1.2.3 with tokens \\ud83d\\ude42 They'RE na\\u00efve 'single' keep node.js v1.2.3 'single' it way gateway It's keep reservation backtracking body fox with that [brackets] the and 1999 request no engine counting is [brackets] lazy on request quick involved \\ud83d\\ude42 brown v1.2.3 involved fox exactly hand {braces} we'll every regex\"}, {\"role\": \"assistant\", \"content\": \"(parens) faster tiktoken's with the every mirrors boundaries that caf\\u00e9 backtracking involved once 42 \\u6771\\u4eac Z\\u00fcrich {braces} boundaries that [brackets] 1999 it admission before a jumps and 3.14159 way scanner don't {braces} budget for a 3.14159 mirrors boundaries while caf\\u00e9 counting \\u6771\\u4eac a written tiktoken's and we'll https://example.com/a/b?c=d jumps while so [brackets] F# 3.14159 mirrors that {braces} request request I'M keep the over hand engine keep node.js dog na\\u00efve [brackets] every body $1,234.56 \\\"quotes\\\" $1,234.56 we'll \\u0645\\u0631\\u062d\\u0628\\u0627 that 42 budget scanned it They'RE no with the {braces} we'll tiktoken's a \\ud83d\\ude42 tiktoken's https://example.com/a/b?c=d It's exactly It's for over that 100% a \\\"quotes\\\" caf\\u00e9 no 3.14159 hand gateway quick budget over user@example.com I'M\"}, {\"role\": \"user\", \"content\": \"mirrors engine so Z\\u00fcrich and reservation caf\\u00e9 1999 na\\u00efve exactly involved caf\\u00e9 and 1999 over tiktoken's user@example.com over once that It's jumps involved C++ and is node.js brown gateway 100% 100% keep hand backtracking 3.14159 involved hand 'single' once because budget They'RE brown 3.14159 every faster request it regex while \\ud83d\\ude42 1999 exactly way we'll 'single' request scanner boundaries on (parens) we'll\"}, {\"role\": \"assistant\", \"content\": \"once 1999 fox that Z\\u00fcrich \\u6771\\u4eac that we'll 42 'single' dog \\u6771\\u4eac mirrors (parens) because na\\u00efve every and piece C++ 3.14159 https://example.com/a/b?c=d 42 once so gateway is scanned no user@example.com node.js caf\\u00e9 faster node.js tiktoken's It's F# brown it admission body don't don't before scanner before F# 3.14159 na\\u00efve backtracking user@example.com F# that (parens) (parens) there's exactly a reservation it before backtracking It's piece tiktoken's lazy written is {braces} 1999 scanned keep body because mirrors no written that piece boundaries request once scanner \\\"quotes\\\" with engine hand v1.2.3 we'll v1.2.3 the we'll $1,234.56 the quick and tokens backtracking is quick 'single' 42 once reservation admission fox scanner a scanner engine 1999 quick [brackets] \\u6771\\u4eac budget for quick engine keep user@example.com 100% C++ [brackets] hand on Z\\u00fcrich hand Z\\u00fcrich regex 1999\"}, {\"role\": \"user\", \"content\": \"\\u0645\\u0631\\u062d\\u0628\\u0627 scanner They'RE budget C++ exactly {braces} is \\ud83d\\ude42 because way body brown counting with regex it \\\"quotes\\\" we'll body node.js over a node.js that while it v1.2.3 3.14159 quick request that dog node.js They'RE 1999 that gateway 3.14159 gateway there's backtracking lazy C++ admission node.js 42 request They'RE we'll gateway so once 1999 brown na\\u00efve 3.14159 budget faster \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" backtracking that 100% because a fox once involved on involved v1.2.3 WON'T mirrors 'single' a \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" is They'RE that involved reservation They'RE (parens) v1.2.3 Z\\u00fcrich WON'T Z\\u00fcrich fox that don't boundaries engine regex don't exactly na\\u00efve don't caf\\u00e9 request budget [brackets] F# tokens exactly It's request it\"}, {\"role\": \"assistant\", \"content\": \"scanned (parens) reservation {braces} engine tokens v1.2.3 node.js scanner {braces} It's with admission the 100% F# \\u0645\\u0631\\u062d\\u0628\\u0627 body na\\u00efve with 'single' boundaries regex (parens) because mirrors there's keep once lazy it brown involved boundaries request engine before on It's \\ud83d\\ude42 na\\u00efve fox WON'T we'll backtracking jumps exactly scanner it admission mirrors don't that \\u6771\\u4eac faster body mirrors reservation (parens) \\\"quotes\\\" regex mirrors keep exactly They'RE 3.14159 there's a \\u0645\\u0631\\u062d\\u0628\\u0627 and with admission written C++ https://example.com/a/b?c=d the piece jumps that is lazy tokens before 100% boundaries tokens dog tiktoken's it $1,234.56 I'M 100% once (parens) I'M way brown keep\"}, {\"role\": \"user\", \"content\": \"no no hand F# jumps so faster every don't so tiktoken's 'single' I'M [brackets] \\\"quotes\\\" v1.2.3 over admission exactly scanner node.js 100% quick gateway \\ud83d\\ude42 faster budget over faster over and engine request faster fox 'single' backtracking body brown no na\\u00efve is it \\\"quotes\\\" piece with \\ud83d\\ude42 for tokens I'M scanned while admission way faster na\\u00efve mirrors tokens boundaries \\u6771\\u4eac jumps tiktoken's it there's faster\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 no It's piece request \\u6771\\u4eac Z\\u00fcrich don't \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE 3.14159 quick Z\\u00fcrich once tiktoken's with engine jumps over I'M because caf\\u00e9 faster tiktoken's that 'single' we'll dog 42 \\u6771\\u4eac way once counting it C++ on that the that backtracking we'll scanned before 100% na\\u00efve request dog every request 1999 request quick over 1999 gateway https://example.com/a/b?c=d because 3.14159 that reservation mirrors that a over I'M with regex Z\\u00fcrich mirrors piece regex caf\\u00e9 engine Z\\u00fcrich the written v1.2.3 \"}, {\"role\": \"user\", \"content\": \"hand is It's scanned regex 1999 way on It's we'll tokens with reservation while way mirrors regex no over 42 is user@example.com scanner scanned keep budget {braces} brown budget while is involved [brackets] body jumps F# dog no 3.14159 exactly over 100% (parens) It's jumps so we'll \\u6771\\u4eac faster admission exactly admission that for tiktoken's 100% once engine and every a before budget regex They'RE counting\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 involved backtracking regex na\\u00efve way tokens we'll 100% gateway the gateway the body lazy I'M budget is regex body (parens) C++ that body no the jumps [brackets] exactly 1999 budget request brown \\ud83d\\ude42 jumps keep scanned user@example.com for exactly regex dog node.js scanner that node.js 42 engine fox \\ud83d\\ude42 tiktoken's WON'T mirrors written faster while 42 100% \\u6771\\u4eac user@example.com dog engine lazy [brackets] exactly quick node.js written written 100% every \\ud83d\\ude42 body 3.14159 hand because once gateway faster before with It's a 1999 42 boundaries jumps that {braces} 100% caf\\u00e9 a They'RE 3.14159 [brackets] reservation 1999 faster \\ud83d\\ude42 hand 1999 [brackets] scanner\"}, {\"role\": \"user\", \"content\": \"https://example.com/a/b?c=d dog for C++ v1.2.3 because every 100% \\u6771\\u4eac and brown is request piece boundaries on before because that body WON'T [brackets] that keep tokens lazy WON'T keep on for WON'T body quick {braces} gateway body written tokens quick the mirrors is is with I'M 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 don't \\u0645\\u0631\\u062d\\u0628\\u0627 piece involved user@example.com written tiktoken's fox written and no 1999 don't on because counting \\\"quotes\\\" They'RE 42 a gateway \\u6771\\u4eac scanner admission user@example.com 3.14159 there's on over 'single' written that a {braces} that because gateway the\"}, {\"role\": \"assistant\", \"content\": \"WON'T backtracking \\ud83d\\ude42 node.js 3.14159 keep it \\u0645\\u0631\\u062d\\u0628\\u0627 there's is so user@example.com on there's 3.14159 mirrors involved keep \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's {braces} no engine brown brown there's dog way regex that 'single' for and 3.14159 node.js every there's a every They'RE WON'T exactly scanner tiktoken's They'RE regex the dog counting over the for request scanner with and mirrors dog the counting tokens gateway https://example.com/a/b?c=d for https://example.com/a/b?c=d backtracking\"}, {\"role\": \"user\", \"content\": \"don't because that fox Z\\u00fcrich with Z\\u00fcrich way piece (parens) every (parens) user@example.com {braces} keep so a that boundaries 100% keep C++ node.js 1999 piece $1,234.56 \\ud83d\\ude42 backtracking {braces} that involved gateway boundaries no 42 hand 3.14159 \\ud83d\\ude42 na\\u00efve na\\u00efve that for Z\\u00fcrich $1,234.56 engine lazy fox scanned before once hand no https://example.com/a/b?c=d admission WON'T admission It's no a because \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac exactly written \\u0645\\u0631\\u062d\\u0628\\u0627 faster that is user@example.com v1.2.3 'single' boundaries gateway faster backtracking no engine fox $1,234.56 lazy the $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 that WON'T before I'M scanned\"}, {\"role\": \"assistant\", \"content\": \"it is admission admission brown the They'RE we'll we'll user@example.com [brackets] way tokens $1,234.56 on caf\\u00e9 I'M gateway 3.14159 written 'single' on involved mirrors node.js that fox 'single' quick once \\u6771\\u4eac gateway $1,234.56 no exactly body Z\\u00fcrich that 'single' engine and Z\\u00fcrich don't no so tiktoken's I'M that over hand dog we'll (parens) 'single' \\ud83d\\ude42 while 100% is node.js no quick over while on budget that \\\"quotes\\\" budget It's don't regex on C++ 1999 [brackets] scanned 1999 that there's $1,234.56 way\"}, {\"role\": \"user\", \"content\": \"budget while tiktoken's backtracking gateway and hand no while \\ud83d\\ude42 request node.js we'll and don't while na\\u00efve quick WON'T don't [brackets] quick there's fox that that I'M keep \\ud83d\\ude42 tiktoken's v1.2.3 the body the \\u0645\\u0631\\u062d\\u0628\\u0627 reservation written F# jumps admission boundaries user@example.com while I'M brown lazy It's regex fox user@example.com {braces} is on 'single' written request there's engine exactly tiktoken's WON'T caf\\u00e9 the tokens on and brown exactly scanner with involved regex on v1.2.3 jumps that written F# on exactly before over once body \\u6771\\u4eac \\ud83d\\ude42 lazy that \\ud83d\\ude42\"}, {\"role\": \"assistant\", \"content\": \"don't every there's scanned while They'RE a tokens is scanned mirrors so is request counting once way (parens) {braces} counting 100% \\\"quotes\\\" before boundaries brown tiktoken's that way engine quick tiktoken's over involved 100% before for the C++ It's that node.js [brackets] because https://example.com/a/b?c=d engine 100% budget we'll with tokens 42 budget it dog 1999 user@example.com 42 C++ keep the for a 42 keep no caf\\u00e9 node.js 'single' the for don't It's C++ admission a involved 3.14159 'single' don't They'RE WON'T quick so backtracking because once body \\\"quotes\\\" WON'T user@example.com caf\\u00e9 v1.2.3 engine 'single' scanner hand I'M 'single' $1,234.56 [brackets] {braces} counting it don't there's no that\"}, {\"role\": \"user\", \"content\": \"so jumps dog hand no 100% while request tiktoken's there's that so regex $1,234.56 don't 'single' boundaries \\\"quotes\\\" that 100% the boundaries keep backtracking (parens) no \\u6771\\u4eac fox F# mirrors 3.14159 we'll there's \\\"quotes\\\" scanner backtracking $1,234.56 100% request the WON'T boundaries caf\\u00e9 tiktoken's node.js backtracking counting user@example.com C++ exactly is reservation 100% we'll quick regex \\ud83d\\ude42 'single' I'M request written https://example.com/a/b?c=d boundaries They'RE there's It's piece F# v1.2.3 boundaries [brackets] a 100% that scanner v1.2.3 42 because https://example.com/a/b?c=d [brackets] that I'M once body faster because with engine 'single' scanned on engine the 1999 \\u6771\\u4eac quick written no piece involved boundaries [brackets] counting while I'M Z\\u00fcrich scanner for the written $1,234.56 (parens) that boundaries the because\"}, {\"role\": \"assistant\", \"content\": \"(parens) https://example.com/a/b?c=d 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich Z\\u00fcrich \\u6771\\u4eac involved don't reservation there's we'll \\\"quotes\\\" reservation They'RE tokens because lazy there's a exactly It's jumps every $1,234.56 v1.2.3 exactly 42 a quick https://example.com/a/b?c=d caf\\u00e9 tiktoken's piece reservation every C++ caf\\u00e9 v1.2.3 scanned backtracking na\\u00efve 1999 piece engine while that I'M involved on fox 42 https://example.com/a/b?c=d (parens) tiktoken's for dog once the involved a dog once there's https://example.com/a/b?c=d tokens na\\u00efve (parens) once 1999 over na\\u00efve caf\\u00e9 hand They'RE every tiktoken's involved we'll node.js\"}, {\"role\": \"user\", \"content\": \"that request while jumps faster exactly written 3.14159 scanner that \\\"quotes\\\" with exactly dog that keep with piece piece boundaries no [brackets] backtracking \\u0645\\u0631\\u062d\\u0628\\u0627 F# hand $1,234.56 engine written (parens) and It's involved node.js Z\\u00fcrich boundaries with backtracking na\\u00efve and body counting node.js keep Z\\u00fcrich that while \\u6771\\u4eac because there's \\u6771\\u4eac boundaries no They'RE counting jumps v1.2.3 with don't a \\u0645\\u0631\\u062d\\u0628\\u0627 fox user@example.com no that it F# once it so every 100% a caf\\u00e9 the \\\"quotes\\\" [brackets] that gateway while we'll scanned caf\\u00e9 keep involved \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"the while I'M gateway WON'T tiktoken's 'single' over 'single' request faster (parens) faster faster \\u0645\\u0631\\u062d\\u0628\\u0627 hand tiktoken's keep while v1.2.3 $1,234.56 that written body WON'T way \\u0645\\u0631\\u062d\\u0628\\u0627 brown a F# scanned \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE keep lazy caf\\u00e9 100% 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 {braces} we'll exactly the C++ body Z\\u00fcrich Z\\u00fcrich way with reservation mirrors C++ backtracking involved WON'T exactly with involved WON'T way a counting the admission 1999 hand because once They'RE the backtracking 1999 no a 42 brown\"}, {\"role\": \"user\", \"content\": \"node.js so jumps is piece a way on because the that backtracking request the reservation {braces} {braces} the https://example.com/a/b?c=d it tokens reservation the lazy They'RE reservation a 1999 I'M boundaries no with no C++ is way 3.14159 WON'T hand 'single' counting Z\\u00fcrich on F# with 1999 admission every (parens) we'll request mirrors that the 'single' it faster dog \\\"quotes\\\" jumps that faster WON'T faster every 100% involved a there's \\ud83d\\ude42 way way 'single' na\\u00efve no on it (parens) WON'T user@example.com na\\u00efve boundaries mirrors [brackets] it regex regex 42 engine engine and faster \\\"quotes\\\" gateway \\u0645\\u0631\\u062d\\u0628\\u0627 over request $1,234.56 They'RE user@example.com that node.js I'M https://example.com/a/b?c=d involved on (parens) it a that 3.14159 way over no\"}, {\"role\": \"assistant\", \"content\": \"boundaries so budget involved exactly written node.js [brackets] with jumps caf\\u00e9 node.js (parens) [brackets] 100% tiktoken's 42 it that piece (parens) engine user@example.com involved no gateway \\\"quotes\\\" exactly It's budget exactly once It's a involved (parens) budget we'll fox node.js tokens Z\\u00fcrich no body \\\"quotes\\\" na\\u00efve while it is I'M the hand mirrors WON'T the caf\\u00e9 \\u6771\\u4eac mirrors https://example.com/a/b?c=d counting before way because \\\"quotes\\\" a admission lazy dog keep the budget quick mirrors there's [brackets] hand Z\\u00fcrich https://example.com/a/b?c=d mirrors user@example.com 42 regex request scanned \\\"quotes\\\" engine because the It's Z\\u00fcrich F# 42 boundaries there's keep engine \\ud83d\\ude42 I'M They'RE because over $1,234.56 admission for fox because exactly because piece regex on \\u6771\\u4eac 'single' backtracking a\"}, {\"role\": \"user\", \"content\": \"They'RE backtracking \\u6771\\u4eac boundaries mirrors that 100% {braces} 42 gateway boundaries tokens so on counting gateway tiktoken's tiktoken's tokens tiktoken's there's while tokens It's C++ mirrors the tokens budget hand we'll over over [brackets] jumps the \\ud83d\\ude42 way and that boundaries 100% counting reservation there's [brackets] 3.14159 They'RE keep the regex It's while budget request Z\\u00fcrich it https://example.com/a/b?c=d C++ admission quick engine keep quick \\u6771\\u4eac $1,234.56 engine budget hand \\u6771\\u4eac body 100% don't scanned $1,234.56 dog the \\u6771\\u4eac piece over 1999 $1,234.56 $1,234.56 the dog we'll gateway is dog there's fox Z\\u00fcrich over They'RE so a engine 'single' regex and lazy na\\u00efve tokens tokens \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 \\ud83d\\ude42 (parens)\"}, {\"role\": \"assistant\", \"content\": \"\\\"quotes\\\" {braces} so with brown (parens) that the because keep boundaries before because fox exactly every because F# so mirrors way C++ request gateway there's that on request tokens it \\\"quotes\\\" regex written dog scanner budget while it that $1,234.56 scanned every no we'll is because while reservation They'RE don't dog scanner dog WON'T fox no scanned \\ud83d\\ude42 budget no dog request it scanned is hand over no 3.14159 so regex backtracking exactly backtracking the 42 100% faster quick scanner before 'single' reservation engine it we'll caf\\u00e9 every quick for no faster 100% $1,234.56 on Z\\u00fcrich written keep\"}, {\"role\": \"user\", \"content\": \"on tokens don't F# counting keep exactly every \\ud83d\\ude42 keep Z\\u00fcrich for fox counting no caf\\u00e9 reservation na\\u00efve mirrors They'RE It's F# the that piece \\\"quotes\\\" \\u6771\\u4eac \\u6771\\u4eac engine no $1,234.56 \\\"quotes\\\" because dog 100% na\\u00efve user@example.com tokens [brackets] no with that boundaries I'M Z\\u00fcrich before (parens) while and scanned gateway user@example.com faster request boundaries \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d written C++ I'M while quick [brackets] before 1999 and They'RE C++ v1.2.3 for no {braces} request C++ They'RE over WON'T on caf\\u00e9 {braces} 100% the body quick keep C++ \\ud83d\\ude42 tokens scanned admission keep hand a node.js WON'T keep 100% $1,234.56 hand $1,234.56 (parens) 1999 because the \\ud83d\\ude42 is 42 body Z\\u00fcrich way [brackets] it once that dog C++ jumps fox tiktoken's piece \\\"quotes\\\" v1.2.3 exactly for 42 \\ud83d\\ude42 'single' na\\u00efve admission \\u6771\\u4eac They'RE fox mirrors regex\"}, {\"role\": \"assistant\", \"content\": \"They'RE scanner tiktoken's quick (parens) https://example.com/a/b?c=d budget quick for engine (parens) hand tiktoken's piece written https://example.com/a/b?c=d counting counting over no once engine I'M because fox exactly regex no once keep and written regex faster 100% the admission counting a scanner C++ quick involved gateway brown piece Z\\u00fcrich regex written is faster \\u6771\\u4eac body brown no tokens scanned every request it I'M \\ud83d\\ude42 for (parens) Z\\u00fcrich written 3.14159 'single' I'M no na\\u00efve quick regex gateway it counting (parens) the over so hand gateway is body way dog v1.2.3 faster so request while before counting body F# faster [brackets] engine involved a $1,234.56 dog that that budget mirrors user@example.com the brown that no C++ it every na\\u00efve exactly that body Z\\u00fcrich \\u6771\\u4eac with don't so It's \\u0645\\u0631\\u062d\\u0628\\u0627 that gateway a the\"}, {\"role\": \"user\", \"content\": \"dog is once budget \\ud83d\\ude42 quick regex fox (parens) [brackets] gateway \\ud83d\\ude42 no regex because lazy request with that written we'll It's that request user@example.com is \\\"quotes\\\" \\\"quotes\\\" faster the don't quick node.js \\u0645\\u0631\\u062d\\u0628\\u0627 faster na\\u00efve They'RE the 100% written that the 100% backtracking \\ud83d\\ude42 written the fox scanner na\\u00efve it piece It's no so It's before na\\u00efve gateway there's because gateway reservation \\ud83d\\ude42 quick involved \\u0645\\u0631\\u062d\\u0628\\u0627 with WON'T request F# scanned exactly F# reservation scanned engine jumps v1.2.3 we'll 'single' it quick WON'T jumps na\\u00efve tokens brown node.js backtracking reservation 100% v1.2.3 once exactly\"}, {\"role\": \"assistant\", \"content\": \"fox user@example.com it (parens) we'll 100% involved dog and brown 'single' exactly mirrors 100% WON'T na\\u00efve 1999 F# faster 'single' tokens tiktoken's admission \\\"quotes\\\" the WON'T request that on we'll I'M piece there's body Z\\u00fcrich boundaries because jumps the They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 boundaries \\u6771\\u4eac the https://example.com/a/b?c=d caf\\u00e9 piece don't brown exactly node.js the budget every so admission that and 100% fox 'single' {braces} 100% $1,234.56 C++ counting WON'T brown don't https://example.com/a/b?c=d boundaries I'M 100% [brackets] lazy quick It's brown \\u6771\\u4eac for is \\u6771\\u4eac for is and request\"}, {\"role\": \"user\", \"content\": \"{braces} counting I'M request fox involved boundaries with WON'T WON'T $1,234.56 is faster don't request https://example.com/a/b?c=d while It's every body the a brown engine 42 for we'll request (parens) involved is scanned backtracking jumps C++ gateway keep scanner for counting 42 reservation hand \\\"quotes\\\" hand C++ every reservation caf\\u00e9 that don't with $1,234.56 there's WON'T {braces} that 3.14159 F# before over exactly caf\\u00e9 budget \\u6771\\u4eac the before and body gateway hand admission It's C++ that F# It's faster no that we'll written (parens) fox mirrors we'll user@example.com no https://example.com/a/b?c=d node.js that brown request\"}, {\"role\": \"assistant\", \"content\": \"I'M admission counting Z\\u00fcrich user@example.com faster a \\ud83d\\ude42 mirrors once lazy is the involved \\ud83d\\ude42 \\u6771\\u4eac every scanned every because budget hand and it jumps there's while there's [brackets] na\\u00efve on \\\"quotes\\\" \\\"quotes\\\" a \\ud83d\\ude42 hand \\u6771\\u4eac we'll and It's with the {braces} piece WON'T 'single' for \\u6771\\u4eac no backtracking lazy fox a [brackets] fox 42 \\ud83d\\ude42 (parens) 1999 https://example.com/a/b?c=d 1999 lazy and 1999 reservation way (parens) 100% it na\\u00efve 100%\"}, {\"role\": \"user\", \"content\": \"regex on scanner once $1,234.56 1999 (parens) gateway budget dog F# before user@example.com once brown jumps 100% $1,234.56 that budget Z\\u00fcrich the reservation because a on \\\"quotes\\\" backtracking hand dog because backtracking a is reservation F# F# brown gateway for the we'll the node.js piece I'M keep tokens once body WON'T there's tokens piece 'single' They'RE node.js so 100% body over a is user@example.com no \\\"quotes\\\" boundaries F# They'RE the C++ backtracking and \\ud83d\\ude42 over WON'T\"}, {\"role\": \"assistant\", \"content\": \"is 'single' scanned \\u6771\\u4eac 3.14159 mirrors {braces} there's scanner every reservation mirrors reservation backtracking involved that tiktoken's {braces} node.js we'll F# jumps v1.2.3 v1.2.3 exactly keep \\u0645\\u0631\\u062d\\u0628\\u0627 1999 scanner with tokens once counting so for https://example.com/a/b?c=d piece the while request Z\\u00fcrich it [brackets] dog involved the involved the admission and (parens) engine F# \\u0645\\u0631\\u062d\\u0628\\u0627 gateway quick \\ud83d\\ude42 v1.2.3 tiktoken's https://example.com/a/b?c=d faster \\ud83d\\ude42 that 42 (parens) jumps caf\\u00e9 so hand budget admission [brackets] gateway brown I'M quick admission C++ C++ caf\\u00e9 \\u6771\\u4eac 'single' user@example.com hand that don't C++ once\"}, {\"role\": \"user\", \"content\": \"there's na\\u00efve reservation na\\u00efve before gateway 100% 1999 request that jumps tiktoken's for admission on quick on scanned boundaries jumps \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors we'll dog because a budget \\u6771\\u4eac it na\\u00efve 1999 that the caf\\u00e9 involved counting jumps tokens with \\u0645\\u0631\\u062d\\u0628\\u0627 involved before the over admission I'M dog v1.2.3 engine tiktoken's we'll scanner every tiktoken's They'RE once It's exactly request jumps node.js WON'T 3.14159 over that exactly engine \\u0645\\u0631\\u062d\\u0628\\u0627 once budget reservation because it 42 keep 'single' boundaries so \\u0645\\u0631\\u062d\\u0628\\u0627 node.js scanner budget 3.14159 engine node.js C++ I'M budget [brackets] a caf\\u00e9 user@example.com jumps tokens every 1999 It's\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 scanner (parens) fox gateway mirrors 42 They'RE reservation we'll fox written brown \\\"quotes\\\" mirrors so 42 hand because budget 100% I'M counting request quick 42 caf\\u00e9 C++ C++ 3.14159 request https://example.com/a/b?c=d jumps jumps scanned brown keep brown and \\ud83d\\ude42 once tokens admission fox F# Z\\u00fcrich user@example.com the gateway tiktoken's https://example.com/a/b?c=d \\\"quotes\\\" tokens \\ud83d\\ude42 \\ud83d\\ude42 [brackets] while node.js [brackets] 1999 on \\\"quotes\\\" caf\\u00e9 1999 jumps 3.14159 request over body admission tiktoken's \\ud83d\\ude42 regex that the once scanner because They'RE regex 'single' admission 42\"}, {\"role\": \"user\", \"content\": \"gateway \\u0645\\u0631\\u062d\\u0628\\u0627 keep quick https://example.com/a/b?c=d F# involved hand It's written request {braces} over mirrors there's gateway every so it caf\\u00e9 involved {braces} quick that gateway over 42 fox lazy involved every keep don't (parens) It's request every for \\u6771\\u4eac admission piece on 3.14159 backtracking written quick gateway and jumps F# (parens) 100% the budget quick keep it 100% \\ud83d\\ude42 for (parens) is \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"assistant\", \"content\": \"way boundaries F# (parens) that It's \\\"quotes\\\" scanned \\ud83d\\ude42 \\u6771\\u4eac that brown regex is over na\\u00efve 3.14159 because Z\\u00fcrich 1999 (parens) \\u0645\\u0631\\u062d\\u0628\\u0627 C++ over way so that user@example.com and user@example.com tiktoken's scanned \\u6771\\u4eac [brackets] engine lazy 1999 involved it 100% a so \\u0645\\u0631\\u062d\\u0628\\u0627 on every body na\\u00efve faster regex engine fox that tokens caf\\u00e9 na\\u00efve 3.14159 and hand engine with\"}, {\"role\": \"user\", \"content\": \"keep there's that every [brackets] $1,234.56 \\\"quotes\\\" on body https://example.com/a/b?c=d involved https://example.com/a/b?c=d tokens scanner 1999 https://example.com/a/b?c=d https://example.com/a/b?c=d node.js 1999 backtracking user@example.com that [brackets] don't body lazy WON'T 42 written hand we'll WON'T no \\u0645\\u0631\\u062d\\u0628\\u0627 $1,234.56 that $1,234.56 way piece https://example.com/a/b?c=d admission admission C++ boundaries we'll reservation scanner the C++ \\u0645\\u0631\\u062d\\u0628\\u0627 hand engine na\\u00efve I'M backtracking while regex \\ud83d\\ude42 budget quick v1.2.3 dog v1.2.3 quick [brackets] fox for node.js [brackets] admission \\ud83d\\ude42 They'RE that on boundaries every that regex 42 a {braces} way every scanned backtracking boundaries budget we'll \\\"quotes\\\" it no caf\\u00e9 scanner jumps backtracking while involved and it 1999 with way {braces} \\\"quotes\\\" 'single'\"}, {\"role\": \"assistant\", \"content\": \"42 is \\\"quotes\\\" budget It's They'RE dog fox exactly fox written regex backtracking written with \\\"quotes\\\" written WON'T that backtracking for a every a engine lazy 42 the once budget on because dog we'll and node.js it $1,234.56 mirrors (parens) It's 42 na\\u00efve it scanned https://example.com/a/b?c=d v1.2.3 C++ 'single' boundaries caf\\u00e9 1999 before once 3.14159 is [brackets] don't tiktoken's scanner fox [brackets] \\u6771\\u4eac over once [brackets] They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 we'll with 42 counting \\u0645\\u0631\\u062d\\u0628\\u0627 C++ 3.14159 it the quick\"}, {\"role\": \"user\", \"content\": \" (parens) with before so is that node.js C++ we'll on admission the \\u6771\\u4eac faster the caf\\u00e9 jumps Z\\u00fcrich (parens) fox \\u6771\\u4eac {braces} every {braces} fox mirrors They'RE They'RE fox counting C++ keep \\\"quotes\\\" 1999 once 100% C++ don't once way dog They'RE once involved exactly \\ud83d\\ude42 we'll piece request reservation quick the don't (parens) scanner I'M \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 Z\\u00fcrich once admission no regex tiktoken's exactly 1999 over engine regex there's involved 42 admission no 'single' over there's (parens) caf\\u00e9 because request keep \\u6771\\u4eac keep \\\"quotes\\\" budget piece piece faster $1,234.56 quick WON'T it piece hand\"}, {\"role\": \"assistant\", \"content\": \"involved scanner WON'T {braces} {braces} It's F# piece it \\\"quotes\\\" I'M because body before quick counting piece na\\u00efve reservation hand WON'T hand the request 1999 involved because {braces} so don't F# and [brackets] faster na\\u00efve way for brown brown hand it Z\\u00fcrich 3.14159 the user@example.com with 1999 while every scanner lazy (parens) tokens [brackets] for tokens body exactly I'M \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 over caf\\u00e9 exactly counting a tiktoken's while no https://example.com/a/b?c=d \\u0645\\u0631\\u062d\\u0628\\u0627 lazy lazy (parens) backtracking that involved keep and a tiktoken's is body They'RE it it 100% scanned jumps jumps (parens) na\\u00efve so for v1.2.3 F# 'single' way no exactly 'single' tiktoken's there's dog caf\\u00e9 engine $1,234.56 is tokens\"}, {\"role\": \"user\", \"content\": \"a C++ 100% faster the node.js gateway every It's budget reservation there's scanner before Z\\u00fcrich \\ud83d\\ude42 written written tokens faster tokens $1,234.56 counting quick Z\\u00fcrich for lazy tokens node.js before before backtracking WON'T It's jumps before Z\\u00fcrich $1,234.56 with boundaries mirrors the request once node.js faster that jumps backtracking 1999 boundaries node.js fox admission tiktoken's mirrors way the we'll involved backtracking request 3.14159 the jumps na\\u00efve na\\u00efve \\\"quotes\\\" mirrors 3.14159 na\\u00efve scanner F# because node.js 3.14159 that caf\\u00e9 100% and and (parens) \\\"quotes\\\" WON'T $1,234.56 \\u6771\\u4eac quick $1,234.56 reservation while the na\\u00efve exactly once written backtracking written the na\\u00efve\"}, {\"role\": \"assistant\", \"content\": \"tokens \\u0645\\u0631\\u062d\\u0628\\u0627 scanned [brackets] every admission [brackets] scanned 'single' exactly scanner tokens that [brackets] tokens scanned the before before They'RE every because \\\"quotes\\\" C++ scanner with with brown exactly is for user@example.com while caf\\u00e9 boundaries it WON'T scanner that admission reservation 100% is admission before gateway reservation caf\\u00e9 100% request tiktoken's 3.14159 before backtracking while so https://example.com/a/b?c=d 42 faster and and faster It's we'll mirrors 42 tiktoken's regex node.js there's scanned\"}, {\"role\": \"user\", \"content\": \"written v1.2.3 exactly reservation piece every F# reservation node.js node.js \\u6771\\u4eac that scanner https://example.com/a/b?c=d 42 {braces} it and keep It's no regex jumps with user@example.com body over It's \\\"quotes\\\" and 'single' counting I'M that \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) engine the way $1,234.56 scanned They'RE for caf\\u00e9 \\\"quotes\\\" lazy don't backtracking tokens \\\"quotes\\\" quick {braces} \\\"quotes\\\" involved involved engine scanner because hand tiktoken's v1.2.3 F# regex $1,234.56 engine {braces} gateway counting that (parens) that the before admission that every backtracking \\\"quotes\\\" [brackets] way gateway engine lazy fox 42 don't [brackets] once na\\u00efve \\\"quotes\\\" They'RE They'RE is jumps every 100% written backtracking body caf\\u00e9 that backtracking v1.2.3 (parens) F# keep exactly piece once v1.2.3 F# They'RE backtracking 100% the fox the Z\\u00fcrich involved F#\"}, {\"role\": \"assistant\", \"content\": \"{braces} 42 engine over na\\u00efve hand involved They'RE request It's Z\\u00fcrich before {braces} 42 dog every node.js is over https://example.com/a/b?c=d na\\u00efve engine fox is [brackets] is on over \\ud83d\\ude42 (parens) \\ud83d\\ude42 regex They'RE They'RE \\ud83d\\ude42 (parens) C++ node.js 100% jumps once that scanner on mirrors regex that faster tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's while F# exactly 100% 3.14159 engine \\u6771\\u4eac scanner {braces} na\\u00efve I'M the over brown written don't tokens user@example.com is F# once faster on It's a there's C++ 1999 we'll [brackets] over the dog there's request dog C++ the brown regex way na\\u00efve written before Z\\u00fcrich engine caf\\u00e9 WON'T 1999 with engine and scanned $1,234.56 keep I'M every user@example.com 'single' a and 1999 don't I'M written while and that\"}, {\"role\": \"user\", \"content\": \"budget {braces} hand so and (parens) fox for that no way is counting tokens WON'T tokens tokens lazy tokens piece on that request dog mirrors I'M mirrors scanned regex on na\\u00efve C++ once user@example.com before lazy user@example.com \\\"quotes\\\" I'M once over involved that request written backtracking \\\"quotes\\\" is no brown there's {braces} because \\ud83d\\ude42 keep reservation written 100% caf\\u00e9 v1.2.3 tiktoken's body F# \\u6771\\u4eac counting quick every faster It's while \\u6771\\u4eac every backtracking lazy They'RE tokens fox {braces} tokens \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" 1999 piece involved [brackets] tokens They'RE the a don't hand caf\\u00e9 quick before 'single' so \\u6771\\u4eac that [brackets] quick \\\"quotes\\\" counting gateway it that is It's scanner the 'single' the \\u0645\\u0631\\u062d\\u0628\\u0627 keep 100% it fox way na\\u00efve exactly mirrors caf\\u00e9 mirrors there's because piece every a\"}, {\"role\": \"assistant\", \"content\": \"counting no tiktoken's hand dog {braces} They'RE $1,234.56 involved \\\"quotes\\\" on \\u0645\\u0631\\u062d\\u0628\\u0627 the because tiktoken's 3.14159 that because counting a 3.14159 so node.js counting backtracking https://example.com/a/b?c=d C++ quick involved (parens) request (parens) F# no gateway exactly every backtracking hand on user@example.com It's written scanned before reservation {braces} there's [brackets] $1,234.56 I'M engine involved brown reservation Z\\u00fcrich request F# brown a way I'M written exactly faster \\u6771\\u4eac Z\\u00fcrich mirrors regex 42 tiktoken's caf\\u00e9 42 100% F# way 1999 no node.js way involved while and {braces} that 1999 They'RE node.js admission I'M the is there's 'single' scanner na\\u00efve scanned tiktoken's \\u6771\\u4eac before backtracking 'single'\"}, {\"role\": \"user\", \"content\": \"na\\u00efve tiktoken's boundaries admission 42 100% {braces} 'single' (parens) and and way I'M 3.14159 budget jumps with gateway the fox engine because on scanner \\\"quotes\\\" 1999 user@example.com that written 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 quick [brackets] WON'T 42 \\u6771\\u4eac \\\"quotes\\\" jumps so budget scanned 'single' boundaries and F# quick every https://example.com/a/b?c=d regex tiktoken's lazy involved we'll \\ud83d\\ude42 it admission v1.2.3 reservation counting scanner that \\u0645\\u0631\\u062d\\u0628\\u0627 over there's counting because 100% keep F# lazy $1,234.56 so that is once request is \\ud83d\\ude42 before once na\\u00efve keep\"}, {\"role\": \"assistant\", \"content\": \"with Z\\u00fcrich a 3.14159 faster faster faster jumps 'single' before over dog It's $1,234.56 node.js 'single' na\\u00efve body way we'll It's once regex the counting 3.14159 [brackets] gateway C++ on piece jumps body is node.js lazy user@example.com don't with v1.2.3 exactly user@example.com [brackets] once It's it Z\\u00fcrich [brackets] {braces} faster \\u0645\\u0631\\u062d\\u0628\\u0627 1999 that C++ involved on quick C++ scanner reservation reservation regex node.js faster because jumps It's mirrors that F# brown the request with every reservation backtracking (parens) v1.2.3 I'M {braces} once tokens reservation 42 a 3.14159 F# we'll (parens) way caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 1999 F# piece 1999 quick for scanned\"}, {\"role\": \"user\", \"content\": \"scanned piece F# reservation because regex regex counting written na\\u00efve budget 42 the before keep keep way It's It's exactly node.js because 3.14159 tokens keep with \\u0645\\u0631\\u062d\\u0628\\u0627 body exactly that reservation v1.2.3 42 [brackets] keep scanner dog piece it F# [brackets] regex that tokens exactly body node.js that once faster hand with lazy exactly boundaries jumps that scanned that written\"}, {\"role\": \"assistant\", \"content\": \"brown They'RE gateway the Z\\u00fcrich gateway user@example.com over quick and \\\"quotes\\\" boundaries regex \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 the tokens 'single' is every tokens 42 reservation (parens) request I'M backtracking regex tokens boundaries on is hand scanner boundaries caf\\u00e9 engine and jumps It's the over user@example.com user@example.com I'M on lazy while 1999 WON'T brown that reservation It's v1.2.3 WON'T exactly gateway with quick F# F# the I'M gateway I'M don't way boundaries \\\"quotes\\\" reservation Z\\u00fcrich and 3.14159 on faster na\\u00efve 42 v1.2.3 v1.2.3 3.14159 jumps hand we'll once (parens) C++ the faster keep They'RE node.js fox scanned 42 jumps faster way gateway hand 1999 it that request keep \\u6771\\u4eac $1,234.56 tokens engine that we'll F# we'll written \\\"quotes\\\" a is no keep tiktoken's node.js node.js on counting user@example.com \\\"quotes\\\" {braces} 1999 user@example.com hand fox I'M node.js is\"}, {\"role\": \"user\", \"content\": \"'single' \\u6771\\u4eac and scanned v1.2.3 dog \\u6771\\u4eac with budget $1,234.56 a exactly and scanned body way na\\u00efve on brown mirrors a 3.14159 dog fox jumps {braces} (parens) jumps the way over and we'll involved faster user@example.com fox body there's gateway body brown jumps there's for 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking It's mirrors Z\\u00fcrich 'single' tokens 'single' for user@example.com 100% faster there's keep C++ and don't gateway budget {braces} faster hand don't is involved \\u6771\\u4eac regex regex reservation scanner\"}, {\"role\": \"assistant\", \"content\": \"F# body is regex jumps WON'T reservation Z\\u00fcrich tokens piece that and \\u0645\\u0631\\u062d\\u0628\\u0627 [brackets] for 'single' we'll engine F# there's \\\"quotes\\\" so counting \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d [brackets] piece [brackets] involved exactly fox {braces} \\ud83d\\ude42 3.14159 way F# 3.14159 scanned while keep every request \\u6771\\u4eac way C++ na\\u00efve engine engine user@example.com because admission I'M gateway v1.2.3 v1.2.3 It's it Z\\u00fcrich https://example.com/a/b?c=d 3.14159 Z\\u00fcrich hand [brackets] [brackets] dog that 'single' exactly \\u6771\\u4eac regex brown reservation involved and WON'T F# na\\u00efve before It's a tiktoken's I'M we'll with written budget engine caf\\u00e9 tokens 100% scanned It's brown every a for\"}, {\"role\": \"user\", \"content\": \"involved because caf\\u00e9 so there's I'M 42 no na\\u00efve na\\u00efve exactly tiktoken's once scanner tiktoken's counting exactly dog the a and faster WON'T so They'RE 'single' faster for written C++ 3.14159 tiktoken's $1,234.56 C++ way exactly https://example.com/a/b?c=d every way node.js budget 42 before before gateway it mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 way faster budget (parens) budget every that don't over so v1.2.3 request we'll because with \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 don't 'single' reservation fox tokens there's request it They'RE [brackets] it node.js gateway 'single' node.js body once way quick 3.14159\"}, {\"role\": \"assistant\", \"content\": \"3.14159 while reservation body before on once (parens) request because https://example.com/a/b?c=d involved 42 user@example.com and no request They'RE dog F# na\\u00efve They'RE exactly [brackets] regex engine mirrors scanner engine It's there's counting \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich way 3.14159 over counting dog gateway no we'll for \\\"quotes\\\" no lazy 42 involved counting involved written counting 1999 a $1,234.56 tokens regex quick and no once \\u6771\\u4eac every so brown is \\u0645\\u0631\\u062d\\u0628\\u0627 fox https://example.com/a/b?c=d caf\\u00e9 brown \\u6771\\u4eac 42 exactly a scanner over \\ud83d\\ude42 we'll budget dog {braces} for before faster\"}, {\"role\": \"user\", \"content\": \" faster the over keep while \\u0645\\u0631\\u062d\\u0628\\u0627 tiktoken's once the because involved scanner regex that \\u6771\\u4eac there's brown WON'T dog \\u0645\\u0631\\u062d\\u0628\\u0627 budget lazy v1.2.3 v1.2.3 caf\\u00e9 counting so don't the scanner I'M caf\\u00e9 1999 don't once engine lazy (parens) request keep user@example.com the keep fox the Z\\u00fcrich https://example.com/a/b?c=d 3.14159 caf\\u00e9 (parens) It's piece I'M written \\\"quotes\\\" 'single' It's that {braces} reservation piece\"}, {\"role\": \"assistant\", \"content\": \"fox quick C++ brown before no lazy faster the that the every 100% na\\u00efve that 100% {braces} scanner we'll involved na\\u00efve before that {braces} faster mirrors because 'single' it faster v1.2.3 involved is so with request piece piece while that engine admission tokens gateway scanned before F# every $1,234.56 hand 1999 request is over v1.2.3 over quick regex v1.2.3 v1.2.3 boundaries Z\\u00fcrich and\"}, {\"role\": \"user\", \"content\": \"node.js WON'T faster \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 dog tokens and the no because They'RE and WON'T WON'T caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 42 na\\u00efve written keep don't the WON'T is admission $1,234.56 and every the I'M is C++ budget node.js tokens (parens) node.js caf\\u00e9 'single' with engine https://example.com/a/b?c=d v1.2.3 written node.js {braces} tiktoken's boundaries tiktoken's lazy scanned user@example.com quick body regex backtracking involved jumps faster quick They'RE request na\\u00efve body on written \\ud83d\\ude42 regex a It's a a no \\\"quotes\\\" \\ud83d\\ude42 while involved the jumps and \\ud83d\\ude42 node.js https://example.com/a/b?c=d boundaries na\\u00efve node.js brown hand 42 na\\u00efve for before C++ while https://example.com/a/b?c=d 42 because that for reservation \\\"quotes\\\" gateway budget is a so It's Z\\u00fcrich budget regex once scanned piece regex budget before over na\\u00efve regex\"}, {\"role\": \"assistant\", \"content\": \"WON'T on scanned user@example.com They'RE exactly faster the for a boundaries faster faster $1,234.56 I'M with don't {braces} jumps \\u0645\\u0631\\u062d\\u0628\\u0627 jumps so is keep while Z\\u00fcrich 42 so tiktoken's [brackets] written {braces} request engine no faster node.js counting \\u6771\\u4eac lazy counting faster is mirrors it there's jumps tiktoken's piece [brackets] \\u6771\\u4eac so brown admission brown Z\\u00fcrich way mirrors reservation user@example.com v1.2.3 $1,234.56 budget because the \"}, {\"role\": \"user\", \"content\": \"every jumps 'single' engine so dog scanned F# there's \\ud83d\\ude42 https://example.com/a/b?c=d user@example.com exactly backtracking node.js admission is na\\u00efve v1.2.3 tokens the fox involved gateway that admission don't reservation hand engine C++ so for we'll over written dog https://example.com/a/b?c=d while way dog and written with brown F# budget backtracking so tiktoken's \\u6771\\u4eac exactly written involved They'RE because reservation a caf\\u00e9 over boundaries regex is 'single' admission keep admission I'M WON'T admission counting scanned dog {braces} admission admission every F# 1999 scanner They'RE and the na\\u00efve 42 there's with involved 42 exactly 100% it [brackets] way They'RE node.js \\ud83d\\ude42 keep we'll I'M so I'M tiktoken's reservation WON'T caf\\u00e9 boundaries C++\"}, {\"role\": \"assistant\", \"content\": \"node.js the It's so C++ scanned so 1999 the budget request F# fox body body keep is [brackets] 42 over scanner there's \\u0645\\u0631\\u062d\\u0628\\u0627 node.js \\u0645\\u0631\\u062d\\u0628\\u0627 don't involved F# hand na\\u00efve scanned with involved engine jumps WON'T {braces} mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 is counting C++ fox it (parens) 3.14159 regex engine {braces} it keep 'single' the \\u0645\\u0631\\u062d\\u0628\\u0627 gateway engine WON'T user@example.com keep [brackets] once before budget the reservation regex hand keep user@example.com It's written reservation no fox \\\"quotes\\\" on \\u6771\\u4eac (parens) Z\\u00fcrich backtracking \\ud83d\\ude42 [brackets] mirrors gateway scanner boundaries v1.2.3 brown admission 1999 because way tokens regex admission so \\u0645\\u0631\\u062d\\u0628\\u0627 with counting fox tiktoken's na\\u00efve tokens counting gateway\"}, {\"role\": \"user\", \"content\": \"v1.2.3 dog counting 42 1999 with (parens) 42 scanner involved the it 'single' jumps over and user@example.com once I'M I'M engine before mirrors a faster brown quick Z\\u00fcrich v1.2.3 \\u6771\\u4eac body it mirrors na\\u00efve fox budget 'single' with mirrors WON'T user@example.com faster so Z\\u00fcrich scanned It's that written reservation is that 42 v1.2.3 over 1999 on 1999 that over na\\u00efve brown there's the 100% \\u6771\\u4eac request \\\"quotes\\\" $1,234.56 budget regex piece na\\u00efve 3.14159 the scanner \\u6771\\u4eac scanner scanner It's v1.2.3 on tokens engine 1999 42\"}, {\"role\": \"assistant\", \"content\": \"faster $1,234.56 piece user@example.com for \\u0645\\u0631\\u062d\\u0628\\u0627 while so on Z\\u00fcrich backtracking no It's \\\"quotes\\\" hand node.js before engine it tokens jumps reservation dog jumps [brackets] boundaries scanner for (parens) {braces} and request before scanned because before the for admission mirrors every piece admission keep and gateway we'll brown 1999 is I'M They'RE over https://example.com/a/b?c=d 3.14159 while hand once exactly way counting caf\\u00e9 3.14159 hand na\\u00efve every scanned \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors because They'RE backtracking \\ud83d\\ude42 \\\"quotes\\\" over admission piece on don't scanned exactly reservation and (parens)\"}, {\"role\": \"user\", \"content\": \"every quick that that while na\\u00efve fox {braces} while brown that v1.2.3 dog a C++ F# over for regex while every jumps it is 1999 caf\\u00e9 it and there's C++ we'll over counting boundaries brown 42 I'M involved and reservation boundaries so before I'M with scanned every is involved https://example.com/a/b?c=d lazy regex piece It's a while v1.2.3 user@example.com the 'single' I'M a is is \\\"quotes\\\" user@example.com because node.js faster F# Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 that na\\u00efve because brown involved the so involved WON'T body backtracking quick engine backtracking gateway that a user@example.com body [brackets] tokens every boundaries while \\ud83d\\ude42 is on boundaries scanner node.js Z\\u00fcrich is there's for regex it 100% piece tiktoken's\"}, {\"role\": \"assistant\", \"content\": \"reservation hand before F# counting $1,234.56 caf\\u00e9 faster \\u6771\\u4eac [brackets] tokens Z\\u00fcrich tiktoken's on written {braces} It's keep \\\"quotes\\\" while tokens na\\u00efve the no because keep https://example.com/a/b?c=d because counting request \\\"quotes\\\" it a because hand involved reservation node.js and backtracking 42 on piece I'M gateway 'single' mirrors the Z\\u00fcrich quick lazy brown {braces} admission Z\\u00fcrich C++ before https://example.com/a/b?c=d It's engine backtracking don't \\u6771\\u4eac piece dog no keep Z\\u00fcrich piece and 42 regex C++ caf\\u00e9 with we'll 3.14159 'single' mirrors counting Z\\u00fcrich it dog scanner a node.js piece {braces} we'll we'll because They'RE user@example.com fox caf\\u00e9 1999 42 every counting tiktoken's It's before backtracking scanner 42 the \\u6771\\u4eac admission lazy F# gateway mirrors It's C++ tiktoken's is dog no written https://example.com/a/b?c=d \\u6771\\u4eac exactly \\\"quotes\\\" body exactly\"}, {\"role\": \"user\", \"content\": \"that https://example.com/a/b?c=d the over boundaries scanner don't [brackets] quick every because because I'M because so once dog mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" quick on 'single' (parens) 42 and that before while while engine while over tiktoken's counting exactly {braces} tiktoken's piece gateway user@example.com It's that admission scanner It's request \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com that 42 for so \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 user@example.com for F# exactly that 100% backtracking boundaries it fox tiktoken's piece counting once engine admission piece Z\\u00fcrich a budget keep and {braces} is reservation scanner with https://example.com/a/b?c=d gateway no user@example.com na\\u00efve so user@example.com written brown\"}, {\"role\": \"assistant\", \"content\": \"piece They'RE body v1.2.3 \\ud83d\\ude42 C++ na\\u00efve It's we'll backtracking on because the counting a involved counting that there's request WON'T jumps that node.js Z\\u00fcrich quick https://example.com/a/b?c=d They'RE Z\\u00fcrich 3.14159 engine budget \\ud83d\\ude42 \\\"quotes\\\" scanner don't {braces} F# request don't $1,234.56 \\\"quotes\\\" tokens over while admission every with quick I'M \\\"quotes\\\" {braces} fox [brackets] gateway dog keep Z\\u00fcrich every [brackets] every mirrors C++ \\ud83d\\ude42 They'RE is [brackets] 3.14159 scanner hand and request 3.14159 with scanner no the regex 1999 They'RE every dog there's the admission that tiktoken's 'single' engine\"}, {\"role\": \"user\", \"content\": \"boundaries It's way dog I'M body lazy https://example.com/a/b?c=d budget faster for I'M regex every body piece with \\ud83d\\ude42 regex before for quick 'single' exactly with over brown there's jumps the don't {braces} and [brackets] it jumps on while don't no exactly every tokens and the 1999 \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 no [brackets] keep \\ud83d\\ude42 with Z\\u00fcrich gateway while backtracking the [brackets] every is \\u6771\\u4eac while faster for while budget [brackets] \\ud83d\\ude42 don't once quick so hand 'single' It's on dog exactly with once engine 'single' no written involved over tokens way over [brackets] with so regex caf\\u00e9 3.14159 \\ud83d\\ude42 a node.js every because faster reservation budget v1.2.3 with lazy I'M engine because and for before before engine mirrors \\\"quotes\\\" that node.js node.js counting a admission counting on WON'T no piece I'M once boundaries\"}, {\"role\": \"assistant\", \"content\": \"piece counting quick \\u0645\\u0631\\u062d\\u0628\\u0627 brown the there's mirrors regex faster involved tokens backtracking F# backtracking \\\"quotes\\\" \\ud83d\\ude42 with \\u6771\\u4eac body with gateway counting lazy exactly request [brackets] 'single' because while keep is I'M jumps body {braces} mirrors 1999 $1,234.56 faster it so {braces} v1.2.3 keep 42 admission admission It's that involved https://example.com/a/b?c=d na\\u00efve \\u6771\\u4eac \\u6771\\u4eac 42 fox node.js the (parens) (parens) [brackets] on faster hand node.js engine before \\u0645\\u0631\\u062d\\u0628\\u0627 jumps body we'll engine regex Z\\u00fcrich body admission 3.14159 They'RE Z\\u00fcrich scanner before scanner reservation \\u6771\\u4eac admission involved don't lazy $1,234.56 $1,234.56 that lazy scanned F# for request https://example.com/a/b?c=d there's before scanner reservation They'RE on \\u0645\\u0631\\u062d\\u0628\\u0627 quick we'll 100% lazy while budget 'single' once quick brown admission node.js gateway 1999 v1.2.3 piece because https://example.com/a/b?c=d reservation $1,234.56 for na\\u00efve \\ud83d\\ude42 exactly gateway that exactly on node.js and\"}, {\"role\": \"user\", \"content\": \"hand counting request while reservation there's body regex tokens don't regex every no a engine it a 1999 F# reservation with on mirrors for before Z\\u00fcrich v1.2.3 on the \\u0645\\u0631\\u062d\\u0628\\u0627 quick involved budget every node.js the hand it with node.js 1999 quick faster backtracking quick 42 while counting mirrors over so reservation involved [brackets] no written and tokens fox tiktoken's It's written \\ud83d\\ude42 over admission with over don't \\u6771\\u4eac way counting \\ud83d\\ude42 $1,234.56 https://example.com/a/b?c=d body hand over request tiktoken's that budget https://example.com/a/b?c=d exactly scanned gateway regex the while It's there's written for lazy F# counting body caf\\u00e9 jumps that counting jumps\"}, {\"role\": \"assistant\", \"content\": \"faster Z\\u00fcrich boundaries faster don't counting before dog engine I'M Z\\u00fcrich 3.14159 C++ \\u0645\\u0631\\u062d\\u0628\\u0627 for scanned involved engine backtracking brown \\u6771\\u4eac mirrors is (parens) backtracking tokens admission that Z\\u00fcrich don't C++ tiktoken's user@example.com that mirrors lazy scanner (parens) piece the body every \\\"quotes\\\" the request tokens body C++ (parens) a don't They'RE exactly the 3.14159 is budget user@example.com it \\\"quotes\\\" on \\ud83d\\ude42 backtracking admission written \\ud83d\\ude42 C++ quick fox 1999 I'M for it don't 1999 because way there's body reservation 100% WON'T dog 1999 tokens once a over once so $1,234.56 jumps They'RE scanner 3.14159 regex request 'single' exactly $1,234.56 exactly\"}, {\"role\": \"user\", \"content\": \"we'll 1999 lazy over 42 user@example.com over scanned gateway lazy piece so lazy exactly 'single' caf\\u00e9 node.js there's and once v1.2.3 while faster gateway every way 1999 gateway while 1999 1999 on caf\\u00e9 so written backtracking \\\"quotes\\\" backtracking scanner https://example.com/a/b?c=d body admission dog piece way brown fox 3.14159 with WON'T \\u6771\\u4eac \\\"quotes\\\" and exactly lazy na\\u00efve 1999 (parens) and involved keep that 3.14159 F# keep lazy no na\\u00efve request https://example.com/a/b?c=d jumps 'single' fox 'single' [brackets] tiktoken's counting once there's na\\u00efve admission faster F# we'll \\\"quotes\\\" https://example.com/a/b?c=d \\u0645\\u0631\\u062d\\u0628\\u0627 exactly \\u6771\\u4eac hand over https://example.com/a/b?c=d F# fox scanned 1999\"}, {\"role\": \"assistant\", \"content\": \"1999 no there's is 'single' C++ na\\u00efve that over admission \\\"quotes\\\" F# scanner gateway piece 100% while for no {braces} with fox scanned counting \\\"quotes\\\" there's gateway user@example.com keep body involved 3.14159 Z\\u00fcrich Z\\u00fcrich \\\"quotes\\\" [brackets] quick admission request a is 42 scanned caf\\u00e9 \\\"quotes\\\" I'M piece C++ mirrors 'single' every $1,234.56 3.14159 regex quick dog 42 WON'T brown no {braces} piece body quick before Z\\u00fcrich for for request {braces} engine while \\u6771\\u4eac Z\\u00fcrich brown gateway counting backtracking before don't $1,234.56 the request so {braces} \\u6771\\u4eac [brackets] because a node.js a 3.14159 boundaries every quick we'll jumps na\\u00efve that on we'll scanner tokens \\u6771\\u4eac https://example.com/a/b?c=d tokens 'single' keep\"}, {\"role\": \"user\", \"content\": \"fox admission exactly dog \\u6771\\u4eac backtracking {braces} node.js \\\"quotes\\\" way gateway on 1999 engine scanner keep quick way written boundaries because $1,234.56 don't there's way so over tokens F# that we'll v1.2.3 scanned F# for \\u0645\\u0631\\u062d\\u0628\\u0627 involved there's while {braces} \\u0645\\u0631\\u062d\\u0628\\u0627 admission admission [brackets] They'RE \\ud83d\\ude42 regex body quick that v1.2.3 dog quick user@example.com over before admission hand is budget brown admission we'll body node.js na\\u00efve boundaries hand (parens) there's we'll v1.2.3 100% caf\\u00e9 for [brackets] 100% dog \\\"quotes\\\" piece boundaries I'M admission backtracking gateway engine no C++ there's we'll tiktoken's regex \\u0645\\u0631\\u062d\\u0628\\u0627 way WON'T {braces} once counting 1999 it before quick request WON'T we'll I'M admission lazy tokens involved 42 gateway lazy faster written that reservation exactly \\u6771\\u4eac 100% piece 100% scanner admission\"}, {\"role\": \"assistant\", \"content\": \"exactly because body and we'll so backtracking over fox F# I'M that jumps don't don't lazy scanned I'M lazy way piece It's They'RE scanner gateway 1999 so fox no because gateway 1999 that boundaries engine They'RE on admission scanner over Z\\u00fcrich that Z\\u00fcrich written v1.2.3 WON'T [brackets] tiktoken's scanner 3.14159 don't \\u6771\\u4eac tiktoken's user@example.com [brackets] before \\\"quotes\\\" no C++ node.js scanned 42\"}, {\"role\": \"user\", \"content\": \"once so 3.14159 because 100% with $1,234.56 faster no jumps once written gateway written tokens [brackets] before brown a counting \\\"quotes\\\" tiktoken's They'RE the quick (parens) exactly request jumps before involved mirrors dog budget \\ud83d\\ude42 every scanned \\ud83d\\ude42 fox https://example.com/a/b?c=d gateway body node.js lazy na\\u00efve v1.2.3 exactly way [brackets] gateway scanned WON'T [brackets] https://example.com/a/b?c=d piece \\u6771\\u4eac C++ 3.14159 fox v1.2.3 tokens mirrors caf\\u00e9 engine brown Z\\u00fcrich They'RE hand quick $1,234.56 WON'T tokens lazy written na\\u00efve 'single' dog lazy it that way \\ud83d\\ude42\"}, {\"role\": \"assistant\", \"content\": \"dog a They'RE It's so They'RE counting user@example.com jumps F# request no node.js there's na\\u00efve mirrors \\u6771\\u4eac tiktoken's scanner backtracking hand boundaries 1999 brown written mirrors written faster that budget because hand with 3.14159 and over WON'T every that way regex boundaries C++ \\u0645\\u0631\\u062d\\u0628\\u0627 piece Z\\u00fcrich because it They'RE involved https://example.com/a/b?c=d body 42 a fox that every and request {braces} with node.js F# [brackets] $1,234.56 on no budget way on backtracking node.js so 100%\"}, {\"role\": \"user\", \"content\": \"They'RE scanned dog body on that don't 1999 \\ud83d\\ude42 scanner 1999 that exactly C++ They'RE every caf\\u00e9 brown so \\ud83d\\ude42 caf\\u00e9 mirrors and while request WON'T budget scanner na\\u00efve Z\\u00fcrich the fox 3.14159 100% 'single' before \\ud83d\\ude42 Z\\u00fcrich scanner before that 1999 v1.2.3 body every boundaries that and v1.2.3 1999 100% because user@example.com dog engine keep the we'll I'M scanned na\\u00efve \\\"quotes\\\" there's that lazy https://example.com/a/b?c=d {braces} Z\\u00fcrich exactly \\\"quotes\\\" budget reservation involved that mirrors boundaries na\\u00efve that backtracking \\ud83d\\ude42 once \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 WON'T WON'T I'M node.js involved don't {braces} a and lazy piece backtracking for hand regex\"}, {\"role\": \"assistant\", \"content\": \"WON'T admission admission F# because don't [brackets] tiktoken's before the so na\\u00efve don't hand no na\\u00efve request on mirrors dog \\u0645\\u0631\\u062d\\u0628\\u0627 and before while body $1,234.56 v1.2.3 mirrors piece reservation so backtracking user@example.com every 100% F# regex 1999 brown [brackets] gateway 42 and on with 42 regex (parens) {braces} WON'T v1.2.3 {braces} a there's every I'M the 'single' 42 F# involved \\ud83d\\ude42 piece keep scanner the before I'M reservation written 3.14159 before there's piece once It's \\\"quotes\\\" dog written brown (parens) because while fox quick exactly user@example.com $1,234.56 boundaries Z\\u00fcrich is admission WON'T scanner we'll $1,234.56 hand [brackets] written {braces} {braces} {braces} the\"}, {\"role\": \"user\", \"content\": \"counting once C++ it the and involved keep it so there's I'M C++ 1999 no keep piece It's no user@example.com 'single' I'M WON'T node.js there's 100% quick 100% dog \\ud83d\\ude42 faster Z\\u00fcrich is before there's on don't F# {braces} keep that it na\\u00efve faster counting faster is over hand hand before involved tokens v1.2.3 {braces} boundaries backtracking so on before regex hand backtracking for while keep so (parens) tokens is with exactly backtracking (parens) admission $1,234.56 once no 42 C++ brown reservation Z\\u00fcrich \\\"quotes\\\"\"}, {\"role\": \"assistant\", \"content\": \"so {braces} It's involved body $1,234.56 lazy scanner caf\\u00e9 na\\u00efve v1.2.3 mirrors 3.14159 na\\u00efve the fox It's before we'll over no involved counting C++ brown quick hand \\u6771\\u4eac every exactly 42 boundaries $1,234.56 dog WON'T quick scanned na\\u00efve dog na\\u00efve boundaries quick backtracking body [brackets] tokens piece 100% exactly F# on admission the WON'T budget scanner \\\"quotes\\\" and WON'T we'll boundaries on budget scanned with \\ud83d\\ude42 $1,234.56 before we'll tiktoken's https://example.com/a/b?c=d https://example.com/a/b?c=d over no the there's It's don't over way keep [brackets] with body They'RE the once jumps keep admission They'RE tokens tokens jumps no 1999 fox 3.14159 100% reservation (parens) lazy [brackets] because https://example.com/a/b?c=d \\u0645\\u0631\\u062d\\u0628\\u0627 reservation 3.14159 keep while They'RE I'M brown quick They'RE admission exactly F# regex gateway and v1.2.3 while (parens) because with \\\"quotes\\\"\"}, {\"role\": \"user\", \"content\": \"scanner while node.js scanner 42 hand tokens tiktoken's They'RE engine is 'single' 'single' piece and v1.2.3 while it faster don't brown don't body na\\u00efve (parens) \\ud83d\\ude42 node.js the \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d caf\\u00e9 that $1,234.56 mirrors admission \\ud83d\\ude42 scanner that \\\"quotes\\\" C++ (parens) WON'T Z\\u00fcrich They'RE It's \\ud83d\\ude42 caf\\u00e9 that with WON'T it hand that over \\\"quotes\\\" 'single' na\\u00efve 1999 piece keep {braces} quick the no regex \\u0645\\u0631\\u062d\\u0628\\u0627 \\ud83d\\ude42 piece na\\u00efve the \\u0645\\u0631\\u062d\\u0628\\u0627 'single' no that a body with\"}, {\"role\": \"assistant\", \"content\": \"that C++ over caf\\u00e9 tiktoken's scanned \\u0645\\u0631\\u062d\\u0628\\u0627 fox \\\"quotes\\\" for node.js once keep C++ counting once body so for jumps every before reservation user@example.com budget [brackets] boundaries 42 1999 written jumps $1,234.56 that the https://example.com/a/b?c=d faster Z\\u00fcrich request and once involved [brackets] so over over na\\u00efve caf\\u00e9 for for \\\"quotes\\\" while that scanned tokens scanner no no It's F# request request reservation Z\\u00fcrich jumps the na\\u00efve (parens) the backtracking jumps na\\u00efve scanner caf\\u00e9 keep I'M \\u6771\\u4eac quick scanner body hand scanner I'M way https://example.com/a/b?c=d it caf\\u00e9 mirrors (parens) F# keep {braces} brown WON'T regex engine user@example.com reservation user@example.com brown don't engine brown hand They'RE written https://example.com/a/b?c=d way (parens) dog\"}, {\"role\": \"user\", \"content\": \"\\u6771\\u4eac F# scanner that regex It's lazy na\\u00efve 100% involved keep the admission na\\u00efve user@example.com a for Z\\u00fcrich {braces} exactly lazy counting while tokens with hand written a is counting I'M [brackets] [brackets] 100% tokens body \\\"quotes\\\" \\\"quotes\\\" tiktoken's v1.2.3 https://example.com/a/b?c=d counting $1,234.56 $1,234.56 don't keep that before scanned lazy before gateway na\\u00efve scanner fox $1,234.56 tiktoken's faster so I'M involved over (parens) admission lazy dog caf\\u00e9 don't jumps 100% node.js request because caf\\u00e9 https://example.com/a/b?c=d 42 before 1999 [brackets] {braces} once [brackets] caf\\u00e9 scanner quick there's\"}, {\"role\": \"assistant\", \"content\": \"scanner caf\\u00e9 (parens) keep so {braces} on F# it because so over brown It's exactly I'M I'M F# faster $1,234.56 tokens exactly $1,234.56 caf\\u00e9 the na\\u00efve engine user@example.com \\\"quotes\\\" written C++ so regex don't backtracking engine the caf\\u00e9 C++ They'RE fox lazy written WON'T boundaries keep It's caf\\u00e9 while {braces} C++ keep jumps no node.js 3.14159 there's faster no 'single' so It's so a once scanner \\u0645\\u0631\\u062d\\u0628\\u0627 engine every once before backtracking request because F# scanned jumps I'M exactly we'll we'll and once [brackets] there's exactly 100%\"}, {\"role\": \"user\", \"content\": \"C++ tiktoken's $1,234.56 body scanner dog I'M regex that https://example.com/a/b?c=d mirrors every before user@example.com \\\"quotes\\\" with involved scanner budget I'M reservation hand that for 42 fox 42 tokens scanned They'RE They'RE \\\"quotes\\\" keep reservation piece Z\\u00fcrich the 3.14159 counting counting boundaries F# every while body boundaries 42 don't hand faster the reservation no \\u0645\\u0631\\u062d\\u0628\\u0627 that \\u6771\\u4eac while involved lazy involved caf\\u00e9 that on \\u6771\\u4eac dog 'single' on over once keep Z\\u00fcrich lazy https://example.com/a/b?c=d backtracking scanned 100% 1999 C++ body \\u0645\\u0631\\u062d\\u0628\\u0627 scanner https://example.com/a/b?c=d {braces} don't 3.14159 node.js regex fox faster I'M we'll lazy because backtracking \\ud83d\\ude42 fox F# \\\"quotes\\\" mirrors tokens It's and the for a WON'T regex reservation 42 dog They'RE request on node.js 'single' tokens node.js that admission 1999 on WON'T budget \\ud83d\\ude42 admission exactly 1999 that we'll \\u0645\\u0631\\u062d\\u0628\\u0627 Z\\u00fcrich\"}, {\"role\": \"assistant\", \"content\": \"it 1999 engine \\\"quotes\\\" v1.2.3 every body boundaries over fox before I'M faster that fox \\u0645\\u0631\\u062d\\u0628\\u0627 involved node.js keep They'RE fox the with WON'T jumps hand caf\\u00e9 (parens) exactly user@example.com engine Z\\u00fcrich for request that involved na\\u00efve boundaries faster before no way 'single' piece don't tokens \\ud83d\\ude42 regex $1,234.56 \\ud83d\\ude42 keep [brackets] over 100% once that exactly node.js (parens) Z\\u00fcrich v1.2.3 scanned the admission there's F# and is tiktoken's on \\u0645\\u0631\\u062d\\u0628\\u0627 no involved I'M C++ 42 there's while before v1.2.3 piece over brown we'll 1999 lazy that user@example.com budget C++ on quick engine 1999 {braces} faster and that node.js budget body\"}, {\"role\": \"user\", \"content\": \"user@example.com no tiktoken's involved jumps exactly \\ud83d\\ude42 They'RE \\ud83d\\ude42 once because faster node.js I'M engine hand lazy we'll [brackets] fox budget quick every the tokens C++ a quick They'RE {braces} faster caf\\u00e9 for mirrors caf\\u00e9 it user@example.com node.js written on that engine regex https://example.com/a/b?c=d there's tokens don't na\\u00efve counting \\ud83d\\ude42 user@example.com mirrors engine $1,234.56 3.14159 backtracking hand request brown keep tiktoken's \\u6771\\u4eac a there's it scanner keep engine because fox scanned no I'M no scanned faster 42 [brackets] hand \\u0645\\u0631\\u062d\\u0628\\u0627 that budget scanned that tokens admission C++ and [brackets] that 'single'\"}, {\"role\": \"assistant\", \"content\": \"node.js regex while and WON'T body 100% gateway there's https://example.com/a/b?c=d WON'T keep brown 100% budget with budget \\u0645\\u0631\\u062d\\u0628\\u0627 and so don't while don't exactly backtracking faster tokens gateway node.js lazy over written counting 100% node.js tiktoken's brown dog v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 we'll C++ the with don't we'll tokens 100% no no written scanned counting involved hand user@example.com \\\"quotes\\\" lazy \\u0645\\u0631\\u062d\\u0628\\u0627 body counting for [brackets] 'single' backtracking piece It's regex don't admission once boundaries over lazy F# \\u6771\\u4eac tokens it \\u6771\\u4eac user@example.com fox that while tiktoken's Z\\u00fcrich and (parens) every [brackets] keep counting 100% 42 3.14159 counting it {braces} piece 'single' involved https://example.com/a/b?c=d involved quick \\ud83d\\ude42 tokens written that request tiktoken's lazy admission 42 I'M tokens \\\"quotes\\\" Z\\u00fcrich engine reservation tiktoken's once so a scanned {braces} scanner dog before piece written keep https://example.com/a/b?c=d while It's {braces} Z\\u00fcrich it\"}, {\"role\": \"user\", \"content\": \"Z\\u00fcrich 3.14159 They'RE They'RE there's budget we'll gateway a brown [brackets] jumps once fox there's F# boundaries node.js for once 42 F# dog once so jumps 42 and 100% while it user@example.com exactly brown so [brackets] Z\\u00fcrich 'single' request with tiktoken's Z\\u00fcrich every reservation hand boundaries keep a \\ud83d\\ude42 it once admission piece and we'll budget dog na\\u00efve 3.14159 backtracking so gateway 42 Z\\u00fcrich faster (parens) node.js 100% lazy jumps fox scanner engine 'single' counting request \\u6771\\u4eac a regex brown that 1999 https://example.com/a/b?c=d user@example.com They'RE regex I'M \\ud83d\\ude42 3.14159 \\u0645\\u0631\\u062d\\u0628\\u0627 scanner budget tokens [brackets] WON'T that quick budget v1.2.3 way way user@example.com request the [brackets] $1,234.56 gateway They'RE quick scanned $1,234.56 \\ud83d\\ude42 na\\u00efve don't admission \\u0645\\u0631\\u062d\\u0628\\u0627 budget \\u0645\\u0631\\u062d\\u0628\\u0627\"}, {\"role\": \"assistant\", \"content\": \"WON'T admission piece \\u6771\\u4eac gateway no \\ud83d\\ude42 WON'T tokens 'single' piece admission 3.14159 It's while I'M Z\\u00fcrich the keep boundaries Z\\u00fcrich WON'T It's [brackets] brown tokens They'RE exactly boundaries WON'T admission piece (parens) so tiktoken's backtracking is \\u0645\\u0631\\u062d\\u0628\\u0627 before mirrors 3.14159 dog backtracking written caf\\u00e9 hand scanned way that \\ud83d\\ude42 tiktoken's quick way They'RE {braces} counting na\\u00efve once reservation fox admission $1,234.56 hand and fox we'll hand because quick the is \\u6771\\u4eac {braces} \\\"quotes\\\" exactly and dog so scanner involved \\u6771\\u4eac [brackets]\"}, {\"role\": \"user\", \"content\": \"a \\u0645\\u0631\\u062d\\u0628\\u0627 I'M we'll before backtracking [brackets] scanner don't tokens admission tokens https://example.com/a/b?c=d C++ $1,234.56 is \\ud83d\\ude42 exactly dog piece no tiktoken's admission 'single' the quick that every before before dog It's engine so tiktoken's the boundaries https://example.com/a/b?c=d 'single' 1999 written body {braces} there's scanned budget admission don't for we'll \\\"quotes\\\" body quick regex quick there's we'll it https://example.com/a/b?c=d user@example.com involved na\\u00efve 3.14159 on we'll the over na\\u00efve na\\u00efve \\u6771\\u4eac gateway I'M v1.2.3 it way the scanner jumps budget C++ C++ \\u6771\\u4eac it with quick F# F# with piece budget involved request before backtracking the tiktoken's scanner $1,234.56 backtracking body admission over They'RE on engine request node.js while that They'RE backtracking request v1.2.3 42 3.14159 and 100% scanned a caf\\u00e9 every 100% exactly request tokens is gateway before a exactly caf\\u00e9 we'll node.js because \\ud83d\\ude42 admission\"}, {\"role\": \"assistant\", \"content\": \"F# \\u0645\\u0631\\u062d\\u0628\\u0627 the we'll user@example.com mirrors F# \\u6771\\u4eac faster over with C++ no because piece scanned because 42 piece (parens) scanned admission 100% Z\\u00fcrich It's 42 for 1999 the regex the dog faster budget (parens) brown engine It's on (parens) so https://example.com/a/b?c=d piece over v1.2.3 scanned don't It's piece regex admission https://example.com/a/b?c=d every don't dog F# no regex hand is dog piece no counting v1.2.3 don't https://example.com/a/b?c=d the 3.14159 because once https://example.com/a/b?c=d \\u6771\\u4eac\"}, {\"role\": \"user\", \"content\": \"we'll body tiktoken's F# lazy is is request we'll because dog \\u6771\\u4eac boundaries lazy caf\\u00e9 exactly quick jumps They'RE 3.14159 brown \\u6771\\u4eac 3.14159 'single' on keep faster 'single' because I'M no 'single' tiktoken's exactly F# v1.2.3 tokens hand because faster that over and gateway don't 42 \\u0645\\u0631\\u062d\\u0628\\u0627 lazy before that once for keep body lazy Z\\u00fcrich before faster piece gateway caf\\u00e9 exactly involved scanner the \\ud83d\\ude42 node.js WON'T way before C++ written 1999 It's with that is every na\\u00efve 1999 keep \\u6771\\u4eac quick request regex budget It's 1999 before a hand lazy engine WON'T don't while brown admission \\u0645\\u0631\\u062d\\u0628\\u0627 1999 caf\\u00e9 na\\u00efve WON'T boundaries counting reservation \\\"quotes\\\" there's the once over counting request\"}, {\"role\": \"assistant\", \"content\": \"gateway reservation once F# C++ written 42 on piece https://example.com/a/b?c=d [brackets] that \\u0645\\u0631\\u062d\\u0628\\u0627 we'll reservation They'RE node.js the fox piece https://example.com/a/b?c=d user@example.com every F# $1,234.56 \\\"quotes\\\" admission body [brackets] 'single' that and because regex we'll \\u6771\\u4eac 42 keep na\\u00efve F# user@example.com keep admission we'll user@example.com 1999 {braces} 100% because no $1,234.56 counting the 42 there's boundaries F# no admission 42 so so v1.2.3 for mirrors while the regex piece budget 42 node.js {braces} na\\u00efve regex with jumps that counting a involved $1,234.56 that admission tiktoken's na\\u00efve involved before $1,234.56 budget 1999 because v1.2.3 that backtracking \\u6771\\u4eac mirrors [brackets] node.js hand {braces} don't brown counting Z\\u00fcrich scanned over \\ud83d\\ude42 with I'M lazy tokens\"}, {\"role\": \"user\", \"content\": \"quick keep it body so every C++ gateway It's tokens 'single' 100% we'll quick regex so scanner v1.2.3 the involved It's while {braces} $1,234.56 that written that C++ scanned dog 100% 100% boundaries (parens) gateway exactly scanner regex gateway \\u6771\\u4eac mirrors brown 1999 regex v1.2.3 $1,234.56 caf\\u00e9 scanned brown there's caf\\u00e9 1999 \\\"quotes\\\" no F# for that 100% backtracking exactly admission the quick (parens) 'single' [brackets] {braces} that written boundaries reservation https://example.com/a/b?c=d request on quick 1999 hand so brown regex while exactly piece hand piece reservation tiktoken's scanner \\\"quotes\\\" caf\\u00e9 way because na\\u00efve (parens) tiktoken's 3.14159 admission so don't on involved na\\u00efve C++ dog reservation It's boundaries admission it (parens) gateway\"}, {\"role\": \"assistant\", \"content\": \"keep there's with scanner that budget no on node.js budget scanner dog before exactly quick 1999 WON'T reservation node.js it F# $1,234.56 caf\\u00e9 1999 scanned engine and \\ud83d\\ude42 [brackets] with scanned and admission \\u6771\\u4eac \\ud83d\\ude42 faster {braces} WON'T on I'M F# scanned while dog no exactly written C++ because tiktoken's request reservation written don't 100% gateway budget \\u6771\\u4eac that jumps counting \\u6771\\u4eac every engine scanned They'RE \\ud83d\\ude42 for engine F# and {braces} exactly exactly \\\"quotes\\\" the hand before\"}, {\"role\": \"user\", \"content\": \"jumps $1,234.56 we'll we'll {braces} user@example.com way hand fox way It's brown 3.14159 https://example.com/a/b?c=d regex jumps don't once dog written quick lazy tiktoken's WON'T way it budget the that quick It's WON'T They'RE brown keep because so keep \\u6771\\u4eac don't written because way with is I'M https://example.com/a/b?c=d 3.14159 scanner counting v1.2.3 gateway the it 42 while WON'T tokens caf\\u00e9 jumps once regex budget v1.2.3 C++ backtracking brown exactly involved every regex boundaries the mirrors F# F# $1,234.56 [brackets] WON'T \\u6771\\u4eac $1,234.56 jumps every that It's \\u0645\\u0631\\u062d\\u0628\\u0627 the keep while keep hand regex once jumps v1.2.3 caf\\u00e9 don't brown (parens) 'single'\"}, {\"role\": \"assistant\", \"content\": \"quick because faster $1,234.56 reservation F# don't the boundaries v1.2.3 on user@example.com reservation WON'T body 1999 brown \\u6771\\u4eac jumps \\\"quotes\\\" quick WON'T that (parens) the keep $1,234.56 C++ written {braces} so tokens on {braces} dog https://example.com/a/b?c=d \\ud83d\\ude42 3.14159 written backtracking gateway the way 'single' is gateway that jumps it fox piece backtracking tokens that faster backtracking there's a request request [brackets] and tokens $1,234.56 for user@example.com \\u6771\\u4eac so 'single' 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 WON'T don't 100% [brackets] for lazy the node.js hand because It's faster\"}, {\"role\": \"user\", \"content\": \"is faster admission budget that 1999 counting quick engine no while exactly while the is written v1.2.3 faster v1.2.3 mirrors once They'RE [brackets] tiktoken's 3.14159 quick with because fox while so [brackets] a dog while involved keep 1999 that involved \\ud83d\\ude42 keep is v1.2.3 100% once C++ node.js piece backtracking I'M 3.14159 that (parens) They'RE admission budget fox lazy scanned boundaries on on once that keep https://example.com/a/b?c=d lazy exactly because brown with exactly written \\u6771\\u4eac\"}, {\"role\": \"assistant\", \"content\": \"that 1999 node.js C++ \\u0645\\u0631\\u062d\\u0628\\u0627 no tiktoken's boundaries request so engine before 42 user@example.com a faster piece involved regex body for before lazy hand faster 'single' and hand tiktoken's lazy that node.js jumps budget fox way [brackets] v1.2.3 [brackets] 'single' brown piece It's counting mirrors every 3.14159 \\\"quotes\\\" https://example.com/a/b?c=d na\\u00efve we'll F# 100% once budget because scanned tokens Z\\u00fcrich engine scanned\"}, {\"role\": \"user\", \"content\": \"'single' It's over lazy piece 100% gateway caf\\u00e9 caf\\u00e9 \\\"quotes\\\" a way the request written the budget reservation counting $1,234.56 with Z\\u00fcrich counting I'M keep tokens once we'll the while https://example.com/a/b?c=d a on the tokens involved caf\\u00e9 gateway backtracking before fox na\\u00efve with on 3.14159 written lazy {braces} \\\"quotes\\\" there's don't quick 'single' that for They'RE faster because I'M is scanner \\u6771\\u4eac counting the every Z\\u00fcrich body $1,234.56 quick keep backtracking scanner 'single' lazy and 42 counting written so tokens engine 'single' because scanned They'RE v1.2.3 tokens \\u6771\\u4eac exactly 3.14159 on jumps jumps reservation tokens \\u0645\\u0631\\u062d\\u0628\\u0627 budget They'RE na\\u00efve na\\u00efve request\"}, {\"role\": \"assistant\", \"content\": \"lazy is reservation budget it jumps quick {braces} body 'single' is no jumps \\u6771\\u4eac Z\\u00fcrich no admission 100% and WON'T admission 'single' piece is we'll v1.2.3 is the we'll admission engine request body that body WON'T user@example.com 42 and piece I'M backtracking {braces} WON'T boundaries [brackets] engine request so backtracking budget {braces} caf\\u00e9 3.14159 {braces} It's hand once $1,234.56 They'RE budget It's quick brown regex 42 (parens) and 1999 regex Z\\u00fcrich Z\\u00fcrich Z\\u00fcrich exactly https://example.com/a/b?c=d way tokens no tiktoken's counting that hand backtracking jumps way over WON'T na\\u00efve {braces} piece it that 100% (parens) exactly and is over the WON'T user@example.com 1999 request written counting brown while on because 42 tokens no because (parens) Z\\u00fcrich lazy reservation admission once counting 42 exactly They'RE is dog\"}, {\"role\": \"user\", \"content\": \"that They'RE the admission F# (parens) https://example.com/a/b?c=d tiktoken's 3.14159 scanned reservation 'single' boundaries written exactly faster F# na\\u00efve while and [brackets] It's brown \\ud83d\\ude42 caf\\u00e9 jumps with \\\"quotes\\\" backtracking on reservation C++ written over regex [brackets] counting [brackets] lazy scanned $1,234.56 faster 'single' {braces} (parens) They'RE \\\"quotes\\\" on request F# while the dog faster https://example.com/a/b?c=d brown tokens 100% because WON'T way counting don't scanner a scanner 'single' dog because hand admission tiktoken's {braces} hand caf\\u00e9 that a \\ud83d\\ude42 fox \\ud83d\\ude42 \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors $1,234.56 once tiktoken's node.js \\u6771\\u4eac exactly \\\"quotes\\\" Z\\u00fcrich WON'T (parens) the hand It's once every no They'RE F# C++ lazy every don't 1999 user@example.com piece C++ admission 1999 because I'M budget\"}, {\"role\": \"assistant\", \"content\": \"dog over caf\\u00e9 the $1,234.56 before is that 'single' piece It's jumps exactly na\\u00efve every budget hand tokens $1,234.56 \\ud83d\\ude42 we'll (parens) hand for (parens) exactly way na\\u00efve body gateway brown keep It's before every lazy and while quick budget 100% (parens) no 1999 the 3.14159 regex that fox user@example.com 1999 it quick $1,234.56 there's no that exactly written there's \\u0645\\u0631\\u062d\\u0628\\u0627 request mirrors on involved involved C++ no [brackets] exactly na\\u00efve https://example.com/a/b?c=d 3.14159 gateway the jumps once scanned with that admission budget They'RE 'single' caf\\u00e9 mirrors WON'T \\u0645\\u0631\\u062d\\u0628\\u0627 reservation request It's the body admission reservation scanned we'll node.js regex counting before with \\ud83d\\ude42 backtracking [brackets] https://example.com/a/b?c=d 3.14159 the once It's request a written WON'T mirrors 1999\"}, {\"role\": \"user\", \"content\": \"1999 WON'T so user@example.com jumps fox \\ud83d\\ude42 {braces} there's budget before written quick mirrors {braces} for with 42 and gateway counting we'll request it so a for engine {braces} WON'T Z\\u00fcrich user@example.com because https://example.com/a/b?c=d the node.js is fox I'M 3.14159 tiktoken's faster is once 42 involved no 'single' I'M caf\\u00e9 WON'T I'M They'RE [brackets] $1,234.56 we'll I'M so 42 It's every (parens) we'll once \\ud83d\\ude42 once mirrors jumps budget body {braces} \\ud83d\\ude42 F# fox {braces} exactly \\ud83d\\ude42 WON'T that on 100% budget engine reservation C++ fox F# 42 no (parens) counting mirrors v1.2.3 for fox [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 while way 1999 boundaries piece 100% involved every \\ud83d\\ude42 'single' brown once v1.2.3 before 'single' (parens) v1.2.3 Z\\u00fcrich we'll involved 100% {braces}\"}, {\"role\": \"assistant\", \"content\": \"engine backtracking budget on na\\u00efve mirrors dog reservation don't jumps budget gateway counting dog tokens brown for way Z\\u00fcrich dog piece 'single' that quick it there's quick budget 42 It's that dog They'RE F# written dog piece \\u0645\\u0631\\u062d\\u0628\\u0627 F# regex every no hand body once (parens) every lazy 100% over quick once mirrors don't and v1.2.3 They'RE faster brown 1999 a \\ud83d\\ude42 42 gateway https://example.com/a/b?c=d Z\\u00fcrich for over na\\u00efve it lazy admission admission brown hand \\\"quotes\\\" gateway reservation 1999 keep and {braces} keep F# request written tiktoken's every budget every regex there's exactly I'M \\\"quotes\\\" \\u0645\\u0631\\u062d\\u0628\\u0627 there's backtracking and don't boundaries F# caf\\u00e9 3.14159 (parens) 3.14159 user@example.com https://example.com/a/b?c=d tiktoken's keep every\"}, {\"role\": \"user\", \"content\": \"gateway 'single' v1.2.3 request that hand tiktoken's over v1.2.3 'single' so \\ud83d\\ude42 1999 once once na\\u00efve admission na\\u00efve piece that \\\"quotes\\\" 3.14159 way exactly engine F# the \\u0645\\u0631\\u062d\\u0628\\u0627 3.14159 user@example.com on https://example.com/a/b?c=d It's https://example.com/a/b?c=d F# there's written with {braces} admission {braces} a on user@example.com dog dog [brackets] that caf\\u00e9 over for admission v1.2.3 100% node.js boundaries scanner 42 piece because is (parens) (parens) body no is {braces} that user@example.com \\u0645\\u0631\\u062d\\u0628\\u0627 hand \\\"quotes\\\" user@example.com on that \\\"quotes\\\" backtracking \\\"quotes\\\" on no hand 3.14159 brown faster 100% (parens)\"}, {\"role\": \"assistant\", \"content\": \"there's user@example.com They'RE It's lazy user@example.com before before https://example.com/a/b?c=d request WON'T WON'T fox WON'T and faster for body involved 3.14159 keep written admission on gateway written I'M with quick v1.2.3 C++ there's jumps quick admission scanned jumps node.js before while 1999 over node.js with reservation every once node.js brown don't there's mirrors a [brackets] so faster before \\ud83d\\ude42 tiktoken's admission F# don't written na\\u00efve admission is fox we'll way dog tokens scanned the \\\"quotes\\\" hand no 'single' \\ud83d\\ude42 over we'll that regex so every for (parens) is with the \\\"quotes\\\" exactly piece regex exactly brown for for hand tokens C++ user@example.com tiktoken's engine I'M the It's faster regex is dog 'single' C++ user@example.com \\ud83d\\ude42 Z\\u00fcrich {braces} node.js for \\u6771\\u4eac brown regex the with tiktoken's 42 $1,234.56 (parens)\"}, {\"role\": \"user\", \"content\": \"request it reservation that https://example.com/a/b?c=d so engine user@example.com jumps that and the 3.14159 because 3.14159 brown tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 scanner \\u0645\\u0631\\u062d\\u0628\\u0627 fox counting there's dog lazy node.js tiktoken's I'M we'll 42 it tokens faster faster before \\u0645\\u0631\\u062d\\u0628\\u0627 reservation so with the regex it once backtracking for It's hand regex way there's 100% exactly dog no we'll 42 brown lazy scanned budget na\\u00efve WON'T because and we'll 42 way tiktoken's WON'T gateway we'll I'M dog https://example.com/a/b?c=d scanned and that mirrors \\ud83d\\ude42 C++ They'RE with counting on keep na\\u00efve 3.14159 once request keep admission \\\"quotes\\\" C++ \\ud83d\\ude42 once quick boundaries user@example.com tokens budget budget scanner caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 exactly node.js hand\"}, {\"role\": \"assistant\", \"content\": \"fox for user@example.com there's jumps quick fox I'M boundaries mirrors before involved over node.js admission that is lazy It's 1999 1999 fox involved user@example.com \\\"quotes\\\" boundaries 100% \\u0645\\u0631\\u062d\\u0628\\u0627 that \\u6771\\u4eac backtracking 100% there's jumps because 42 tiktoken's caf\\u00e9 that v1.2.3 tokens brown \\ud83d\\ude42 backtracking user@example.com F# 100% 3.14159 no boundaries It's quick https://example.com/a/b?c=d Z\\u00fcrich tokens 42 100% over WON'T They'RE for lazy body involved every on on user@example.com \\u6771\\u4eac (parens) quick once faster quick over backtracking don't once They'RE engine caf\\u00e9 tokens fox because there's every dog we'll \\u6771\\u4eac exactly 3.14159 \\ud83d\\ude42 boundaries no every the F# 100% faster faster once boundaries a I'M {braces} 1999 over \\\"quotes\\\" mirrors that fox 3.14159 lazy na\\u00efve admission that reservation no for written because scanned caf\\u00e9 body\"}, {\"role\": \"user\", \"content\": \"and involved body jumps \\\"quotes\\\" {braces} it we'll regex it na\\u00efve counting the reservation keep admission once before while admission v1.2.3 I'M reservation 1999 scanned tokens counting don't boundaries lazy exactly we'll $1,234.56 and \\u0645\\u0631\\u062d\\u0628\\u0627 \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 engine I'M 42 F# keep every (parens) budget 100% written tiktoken's (parens) don't na\\u00efve I'M body because the is (parens) admission brown hand piece 42 every request don't $1,234.56 once there's 1999 with we'll that backtracking F# with fox mirrors [brackets] for WON'T with jumps request fox scanned a request while no that faster way I'M \\u6771\\u4eac boundaries that brown dog so (parens) written\"}, {\"role\": \"assistant\", \"content\": \" 42 exactly and \\\"quotes\\\" mirrors admission F# faster F# a on faster with quick brown reservation piece (parens) fox \\u0645\\u0631\\u062d\\u0628\\u0627 budget C++ v1.2.3 so over node.js Z\\u00fcrich WON'T tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 F# 1999 42 the involved once 100% is quick (parens) before jumps dog don't over because $1,234.56 reservation 3.14159 that WON'T caf\\u00e9 backtracking request the \\u0645\\u0631\\u062d\\u0628\\u0627 piece every involved 100% budget there's boundaries counting once written we'll na\\u00efve 42 \\ud83d\\ude42 I'M 3.14159 regex scanner reservation tiktoken's on I'M \\u6771\\u4eac boundaries on brown there's that 'single' the exactly scanned no admission request we'll tiktoken's 'single' because \\\"quotes\\\" \\u6771\\u4eac no tiktoken's caf\\u00e9 jumps it tiktoken's lazy faster exactly body tiktoken's fox budget\"}, {\"role\": \"user\", \"content\": \"so fox a backtracking Z\\u00fcrich admission budget fox scanned 42 request don't because reservation way admission gateway v1.2.3 the so \\u6771\\u4eac involved 'single' with a 100% 'single' v1.2.3 user@example.com dog backtracking body C++ piece the so mirrors written once $1,234.56 while fox 42 [brackets] F# we'll Z\\u00fcrich piece budget while for we'll the we'll written exactly $1,234.56 mirrors counting \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors on 'single' na\\u00efve is node.js while is a on\"}, {\"role\": \"assistant\", \"content\": \"on involved 42 'single' They'RE piece while [brackets] dog that and there's before once jumps counting that so is \\u6771\\u4eac [brackets] over and exactly with involved it once on 1999 42 3.14159 is keep \\u6771\\u4eac jumps that there's They'RE piece lazy over with lazy request \\ud83d\\ude42 written node.js admission backtracking way the over backtracking node.js and while $1,234.56 [brackets] while 42 \\u0645\\u0631\\u062d\\u0628\\u0627 I'M 'single' while user@example.com WON'T \\\"quotes\\\" before admission the faster budget mirrors counting we'll once jumps involved WON'T 3.14159 v1.2.3 [brackets] keep over for 'single' \\u6771\\u4eac regex written na\\u00efve way (parens) user@example.com request there's every involved I'M for scanned It's WON'T https://example.com/a/b?c=d C++ caf\\u00e9 once user@example.com [brackets] a 1999 it tiktoken's 'single' a over mirrors jumps request there's dog reservation mirrors tokens jumps tiktoken's is user@example.com tokens\"}, {\"role\": \"user\", \"content\": \"backtracking once over user@example.com body scanner piece faster \\ud83d\\ude42 over for C++ exactly https://example.com/a/b?c=d no engine 'single' piece 42 over hand for for backtracking quick way 1999 regex scanned every reservation no before over \\\"quotes\\\" so $1,234.56 [brackets] tokens na\\u00efve request a budget fox budget user@example.com tokens brown I'M \\\"quotes\\\" hand I'M \\u0645\\u0631\\u062d\\u0628\\u0627 \\u6771\\u4eac engine once engine backtracking while there's no [brackets] jumps the for [brackets] there's \\u0645\\u0631\\u062d\\u0628\\u0627 dog 42 admission so na\\u00efve F# there's \\u6771\\u4eac reservation F# keep piece keep user@example.com caf\\u00e9 the that It's $1,234.56 v1.2.3 It's na\\u00efve engine lazy They'RE counting faster 'single' piece scanned $1,234.56 user@example.com boundaries [brackets] 'single' \\u6771\\u4eac brown so scanned 'single' scanner boundaries 'single' admission admission \\\"quotes\\\" caf\\u00e9 piece\"}, {\"role\": \"assistant\", \"content\": \"Z\\u00fcrich counting while the the no 100% \\u0645\\u0631\\u062d\\u0628\\u0627 body 3.14159 before the while 1999 once that on caf\\u00e9 1999 on it tiktoken's piece scanned 'single' Z\\u00fcrich and [brackets] regex with regex \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com counting admission way keep scanner scanner \\u0645\\u0631\\u062d\\u0628\\u0627 every so [brackets] tokens $1,234.56 every budget boundaries because reservation and keep 1999 \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d tiktoken's https://example.com/a/b?c=d scanner admission caf\\u00e9 $1,234.56 F# faster so engine so it regex 'single' before hand with that admission fox keep Z\\u00fcrich no scanned I'M tiktoken's the it admission a jumps brown that there's it that (parens) 100% (parens) gateway for\"}, {\"role\": \"user\", \"content\": \"so 100% once it there's (parens) that with 1999 lazy written written keep na\\u00efve brown caf\\u00e9 regex the tiktoken's that a while brown It's dog and It's 'single' $1,234.56 42 \\u6771\\u4eac I'M v1.2.3 once na\\u00efve I'M {braces} node.js scanned It's hand piece no caf\\u00e9 backtracking admission with we'll is on node.js fox on tokens brown brown regex scanner admission $1,234.56 lazy regex piece regex node.js (parens) once on while quick dog gateway body that admission for written it scanned I'M it mirrors because the v1.2.3 every 3.14159 boundaries request na\\u00efve It's because faster hand dog tokens v1.2.3 hand that tiktoken's keep caf\\u00e9 keep tokens we'll it jumps no \\ud83d\\ude42 node.js exactly na\\u00efve na\\u00efve every involved tiktoken's \\u6771\\u4eac counting dog over user@example.com https://example.com/a/b?c=d 1999 3.14159 node.js is hand F# written a\"}, {\"role\": \"assistant\", \"content\": \"we'll fox there's 3.14159 C++ so budget because we'll Z\\u00fcrich the on before we'll that Z\\u00fcrich fox keep scanned dog once that \\\"quotes\\\" while scanned brown engine for fox while there's (parens) 42 every we'll caf\\u00e9 https://example.com/a/b?c=d budget exactly gateway F# https://example.com/a/b?c=d I'M dog way before brown with it 3.14159 don't no engine the piece keep user@example.com Z\\u00fcrich It's once written and They'RE the admission scanner with written that fox engine body on scanner 1999 the admission WON'T quick user@example.com I'M admission tiktoken's https://example.com/a/b?c=d while https://example.com/a/b?c=d every with involved the admission gateway there's 'single' gateway a scanner and request that is engine counting dog user@example.com that hand counting lazy admission boundaries engine while way WON'T brown \\ud83d\\ude42 F# I'M F# scanner \\u6771\\u4eac 3.14159 a backtracking scanner I'M brown\"}, {\"role\": \"user\", \"content\": \"I'M lazy It's over on faster fox tiktoken's while that It's for that reservation https://example.com/a/b?c=d keep node.js the \\u6771\\u4eac the admission body 1999 exactly and way dog (parens) node.js \\u6771\\u4eac on They'RE quick that faster F# boundaries tokens boundaries reservation reservation backtracking tokens is brown 100% {braces} quick engine we'll node.js 100% scanned over F# hand na\\u00efve that request jumps quick because regex\"}, {\"role\": \"assistant\", \"content\": \"with 100% $1,234.56 boundaries we'll so no \\\"quotes\\\" $1,234.56 Z\\u00fcrich we'll brown on exactly 1999 WON'T caf\\u00e9 every dog we'll regex before tokens before [brackets] lazy over faster once Z\\u00fcrich request lazy They'RE tokens https://example.com/a/b?c=d [brackets] don't It's with keep hand hand (parens) I'M budget {braces} faster it way is there's every dog gateway \\\"quotes\\\" it Z\\u00fcrich way the over quick tokens C++ hand mirrors It's It's (parens) F# 100% boundaries caf\\u00e9 F# we'll scanner admission\"}, {\"role\": \"user\", \"content\": \"it node.js caf\\u00e9 'single' quick C++ $1,234.56 3.14159 hand 42 $1,234.56 I'M every exactly 100% lazy the \\u6771\\u4eac tiktoken's WON'T jumps with \\u0645\\u0631\\u062d\\u0628\\u0627 I'M counting written piece and keep it regex on F# keep 3.14159 'single' dog and boundaries tokens \\\"quotes\\\" engine with hand that 3.14159 brown WON'T don't scanner keep I'M WON'T fox F# exactly 'single' piece engine and so on request and It's reservation written that boundaries is caf\\u00e9 \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 on every counting a written so jumps with scanner with quick don't jumps way \\u6771\\u4eac keep mirrors regex (parens) we'll it scanned node.js $1,234.56 reservation They'RE caf\\u00e9 node.js for while scanner it faster [brackets] gateway it scanner for mirrors na\\u00efve admission dog on\"}, {\"role\": \"assistant\", \"content\": \"that v1.2.3 written node.js I'M keep user@example.com tiktoken's tiktoken's way backtracking They'RE quick while scanned way we'll user@example.com way 1999 way caf\\u00e9 https://example.com/a/b?c=d hand faster is involved 1999 (parens) quick is (parens) on brown scanner with dog don't admission involved exactly F# They'RE is with exactly way before counting \\\"quotes\\\" backtracking once gateway faster 3.14159 [brackets] way [brackets] a keep \\ud83d\\ude42 hand (parens) counting boundaries budget I'M mirrors v1.2.3 F# once way 1999 WON'T reservation [brackets] for it scanned because 100% reservation 100% tiktoken's They'RE engine a it that so mirrors written over mirrors node.js F# \\ud83d\\ude42 brown dog it v1.2.3 user@example.com the 1999 reservation counting [brackets] budget way mirrors C++ F# and so there's so They'RE I'M scanner request request WON'T and written budget Z\\u00fcrich faster written lazy C++ brown user@example.com \\ud83d\\ude42 lazy reservation caf\\u00e9 node.js caf\\u00e9 brown\"}, {\"role\": \"user\", \"content\": \"budget budget and \\u6771\\u4eac tiktoken's tiktoken's don't that boundaries that 1999 backtracking with (parens) tiktoken's there's fox counting because exactly the every [brackets] engine Z\\u00fcrich admission way scanned boundaries \\ud83d\\ude42 user@example.com counting na\\u00efve F# fox 100% so faster 1999 because so quick engine way fox that engine lazy quick 1999 way $1,234.56 quick while body that 'single' Z\\u00fcrich They'RE na\\u00efve keep 3.14159 https://example.com/a/b?c=d the involved I'M jumps\"}, {\"role\": \"assistant\", \"content\": \"every jumps Z\\u00fcrich piece 'single' F# we'll 'single' over exactly written C++ 3.14159 with scanner exactly while user@example.com a body engine there's engine way 100% request [brackets] exactly way jumps Z\\u00fcrich budget once gateway before I'M \\\"quotes\\\" https://example.com/a/b?c=d user@example.com written admission exactly I'M {braces} \\ud83d\\ude42 They'RE with WON'T on backtracking node.js a with a lazy request na\\u00efve that scanner that once scanned quick 'single'\"}, {\"role\": \"user\", \"content\": \"v1.2.3 budget brown every [brackets] that na\\u00efve there's the and (parens) is brown They'RE F# there's every there's because the so because \\u0645\\u0631\\u062d\\u0628\\u0627 and over 3.14159 brown piece F# involved (parens) so regex while Z\\u00fcrich (parens) counting user@example.com while on $1,234.56 it v1.2.3 {braces} engine v1.2.3 while 1999 keep regex while https://example.com/a/b?c=d once engine WON'T involved \\u6771\\u4eac I'M it with for on hand \\ud83d\\ude42 v1.2.3 on a na\\u00efve regex there's na\\u00efve before v1.2.3 dog (parens) for brown 3.14159 v1.2.3 [brackets] exactly exactly no user@example.com on mirrors It's with that {braces} faster\"}, {\"role\": \"assistant\", \"content\": \"for once admission backtracking engine because once for I'M brown fox the quick quick backtracking there's lazy engine mirrors every don't caf\\u00e9 before mirrors engine 3.14159 on user@example.com with don't keep it node.js before hand regex reservation while scanner na\\u00efve body body no with F# admission Z\\u00fcrich F# WON'T no I'M \\ud83d\\ude42 way scanned $1,234.56 with a no it caf\\u00e9 lazy the dog https://example.com/a/b?c=d involved caf\\u00e9 jumps we'll scanner caf\\u00e9 42 na\\u00efve is we'll\"}, {\"role\": \"user\", \"content\": \"'single' scanner 100% we'll it a {braces} while it v1.2.3 while mirrors there's exactly caf\\u00e9 It's gateway (parens) fox every dog (parens) a written while while before hand involved admission written counting It's mirrors C++ 1999 scanned fox don't written don't reservation exactly mirrors it https://example.com/a/b?c=d there's no scanned that I'M involved $1,234.56 lazy fox scanned written I'M 100% is we'll every don't F# brown every It's keep while scanned jumps tokens 'single' hand Z\\u00fcrich request admission It's 'single' quick backtracking user@example.com request tokens budget exactly gateway jumps {braces} 'single' once tokens a Z\\u00fcrich before admission admission way jumps fox piece hand They'RE we'll exactly reservation on it written mirrors every once $1,234.56 scanner scanner Z\\u00fcrich jumps is over (parens) written piece Z\\u00fcrich $1,234.56 before request 'single' is 'single' keep\"}, {\"role\": \"assistant\", \"content\": \"scanned $1,234.56 gateway over because It's because tokens (parens) C++ admission is hand C++ brown v1.2.3 user@example.com brown \\\"quotes\\\" once backtracking lazy [brackets] with scanner dog lazy 'single' budget 1999 we'll reservation exactly on exactly the counting the [brackets] engine They'RE scanner caf\\u00e9 (parens) involved na\\u00efve on dog (parens) before don't while fox F# request once the lazy \\ud83d\\ude42 gateway user@example.com once that for 42 involved body user@example.com on regex the for lazy backtracking node.js body over involved v1.2.3 \\u6771\\u4eac regex mirrors there's involved regex faster that before request 3.14159 3.14159 that body 1999 faster for 42 'single' user@example.com request user@example.com piece scanned 1999 keep 3.14159 every fox F# scanned before jumps that admission counting They'RE that tokens I'M\"}, {\"role\": \"user\", \"content\": \"exactly quick and https://example.com/a/b?c=d a brown is dog hand regex It's tiktoken's the every with on scanner every keep scanned backtracking because budget keep fox there's so with so that on exactly counting once it exactly (parens) 3.14159 and that while fox admission 42 reservation It's the na\\u00efve 42 regex we'll piece hand so written regex faster fox backtracking that They'RE while 1999 \\u6771\\u4eac \\\"quotes\\\" request \\u6771\\u4eac They'RE and 100% the written for over way 1999 tokens https://example.com/a/b?c=d brown I'M brown brown keep while with $1,234.56 hand user@example.com {braces} [brackets] that faster na\\u00efve scanner on it caf\\u00e9 counting They'RE \\\"quotes\\\" {braces} counting there's body don't jumps budget fox reservation faster \\u0645\\u0631\\u062d\\u0628\\u0627 faster we'll with gateway [brackets] over counting It's there's \\u6771\\u4eac tokens that backtracking scanner that it so\"}, {\"role\": \"assistant\", \"content\": \"boundaries body fox counting that F# on there's caf\\u00e9 scanned jumps 42 dog hand jumps lazy \\\"quotes\\\" for WON'T is 42 on while 1999 scanner over there's engine 100% \\u0645\\u0631\\u062d\\u0628\\u0627 \\\"quotes\\\" jumps before don't [brackets] is counting involved It's I'M boundaries the (parens) with I'M engine 100% for exactly 'single' piece C++ na\\u00efve {braces} we'll every a no a \\\"quotes\\\" user@example.com body the C++ written https://example.com/a/b?c=d counting I'M a v1.2.3 dog 'single' Z\\u00fcrich keep scanner written before over lazy scanned the with keep over 'single' lazy 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 (parens) dog scanned the the dog no $1,234.56 with a so because quick Z\\u00fcrich fox v1.2.3 once faster \\ud83d\\ude42 faster boundaries it 1999 mirrors way a 3.14159 so it written \\u0645\\u0631\\u062d\\u0628\\u0627 https://example.com/a/b?c=d because the before involved Z\\u00fcrich budget while v1.2.3 scanned boundaries regex we'll backtracking\"}, {\"role\": \"user\", \"content\": \"involved $1,234.56 while jumps \\ud83d\\ude42 \\\"quotes\\\" piece hand written fox jumps v1.2.3 hand piece a jumps [brackets] brown involved Z\\u00fcrich engine that faster 'single' way faster budget F# 1999 so that tiktoken's involved regex WON'T budget mirrors mirrors F# before gateway I'M They'RE https://example.com/a/b?c=d lazy counting gateway regex boundaries \\\"quotes\\\" \\\"quotes\\\" 100% {braces} keep written 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 boundaries before written na\\u00efve 3.14159 that v1.2.3 no before request brown dog for $1,234.56 we'll node.js written mirrors scanned WON'T because we'll we'll way F# piece They'RE and while\"}, {\"role\": \"assistant\", \"content\": \"and that scanner brown They'RE we'll Z\\u00fcrich C++ the it 1999 hand F# admission \\u6771\\u4eac is mirrors so once the it admission counting is boundaries gateway that tiktoken's gateway lazy \\u6771\\u4eac na\\u00efve the we'll fox we'll is 42 that hand 3.14159 while reservation piece there's 1999 (parens) so involved we'll because involved written it caf\\u00e9 before once because https://example.com/a/b?c=d hand fox $1,234.56 admission is hand $1,234.56 boundaries for backtracking WON'T 1999 gateway the They'RE [brackets] budget na\\u00efve because that 42 They'RE C++ is don't \\ud83d\\ude42 C++ I'M $1,234.56 once 3.14159 1999 Z\\u00fcrich boundaries reservation quick exactly gateway $1,234.56 {braces}\"}, {\"role\": \"user\", \"content\": \"42 written (parens) backtracking jumps 1999 a \\\"quotes\\\" involved scanned admission They'RE gateway so backtracking 42 100% keep written It's WON'T user@example.com scanner that we'll WON'T Z\\u00fcrich don't na\\u00efve \\ud83d\\ude42 lazy mirrors (parens) before dog 'single' backtracking tokens faster 100% the body with 42 while {braces} while lazy brown I'M way exactly involved $1,234.56 'single' tokens engine counting tiktoken's $1,234.56 it user@example.com so reservation a 3.14159 dog that brown body fox mirrors quick before don't lazy gateway 100% \\u6771\\u4eac tokens the 42 dog 100% the gateway lazy because request there's $1,234.56 It's and regex 100% It's 'single' {braces} that keep involved caf\\u00e9 with {braces} They'RE faster the the $1,234.56 I'M backtracking I'M reservation 42 1999 before lazy request keep mirrors [brackets] 3.14159 42 {braces} $1,234.56 body 42 It's request 3.14159 {braces}\"}, {\"role\": \"assistant\", \"content\": \"and dog body user@example.com fox quick faster scanned the it faster $1,234.56 $1,234.56 with user@example.com over gateway fox tokens piece quick Z\\u00fcrich and counting engine no dog dog reservation on it it 100% every regex {braces} node.js mirrors before the Z\\u00fcrich v1.2.3 mirrors \\ud83d\\ude42 scanned WON'T brown there's \\u6771\\u4eac that mirrors for I'M it mirrors it node.js is \\u6771\\u4eac body \\\"quotes\\\" dog WON'T tiktoken's regex the It's the It's keep 3.14159 quick regex budget WON'T don't and counting over piece scanned C++ scanned is no \\u0645\\u0631\\u062d\\u0628\\u0627 1999 we'll it https://example.com/a/b?c=d keep https://example.com/a/b?c=d user@example.com admission backtracking once user@example.com piece request that (parens) budget written faster we'll budget scanned the user@example.com regex for WON'T piece counting\"}, {\"role\": \"user\", \"content\": \"{braces} 100% for It's the 1999 \\\"quotes\\\" scanned the 'single' quick caf\\u00e9 \\ud83d\\ude42 once once there's C++ It's we'll I'M engine no It's it na\\u00efve involved fox we'll written jumps 3.14159 dog exactly there's It's gateway tiktoken's every It's C++ WON'T we'll na\\u00efve reservation user@example.com 3.14159 every that with backtracking hand every v1.2.3 before so tiktoken's brown It's 3.14159 v1.2.3 caf\\u00e9 because \\\"quotes\\\" faster user@example.com It's before It's (parens) for I'M request written written dog \\u6771\\u4eac counting faster boundaries because tiktoken's on and gateway fox C++ admission It's {braces} v1.2.3 [brackets] 3.14159 exactly that budget that C++ (parens) so 1999 regex involved budget because scanner {braces} dog because I'M reservation exactly [brackets] reservation don't\"}, {\"role\": \"assistant\", \"content\": \"scanner They'RE \\\"quotes\\\" gateway 100% 3.14159 $1,234.56 so request 1999 C++ na\\u00efve quick faster we'll WON'T dog before engine 'single' tiktoken's node.js \\u0645\\u0631\\u062d\\u0628\\u0627 scanner it and na\\u00efve \\ud83d\\ude42 regex so once there's (parens) before 3.14159 dog [brackets] hand 100% backtracking over don't is once jumps caf\\u00e9 https://example.com/a/b?c=d so lazy that 3.14159 reservation on admission \\ud83d\\ude42 (parens) exactly budget v1.2.3 backtracking with brown request It's it with tokens (parens) scanner 1999 backtracking a fox faster v1.2.3 $1,234.56 I'M node.js the reservation piece [brackets] so \\u6771\\u4eac lazy counting the $1,234.56 I'M backtracking over 3.14159 fox fox keep the backtracking with It's [brackets] 3.14159 (parens) that They'RE so [brackets] piece [brackets] {braces} counting written boundaries mirrors budget (parens) gateway \\ud83d\\ude42 exactly hand engine WON'T 3.14159 faster WON'T fox request\"}, {\"role\": \"user\", \"content\": \"while quick \\u0645\\u0631\\u062d\\u0628\\u0627 so every 42 \\ud83d\\ude42 boundaries counting gateway that \\u0645\\u0631\\u062d\\u0628\\u0627 'single' no 100% and regex because before request and 'single' the once F# na\\u00efve $1,234.56 quick $1,234.56 reservation don't node.js scanner 42 node.js 1999 $1,234.56 https://example.com/a/b?c=d jumps lazy 'single' WON'T They'RE WON'T brown {braces} that scanner because Z\\u00fcrich written dog 3.14159 budget \\\"quotes\\\" engine 3.14159 3.14159 for before exactly tiktoken's jumps user@example.com is (parens) scanner lazy piece request brown Z\\u00fcrich node.js admission It's\"}, {\"role\": \"assistant\", \"content\": \"$1,234.56 C++ node.js because it 42 dog boundaries mirrors piece don't C++ 3.14159 https://example.com/a/b?c=d scanner there's is keep \\ud83d\\ude42 once before because backtracking over $1,234.56 [brackets] quick we'll gateway because we'll exactly once every gateway with over because 3.14159 keep every [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 the because 42 1999 no 100% backtracking 1999 \\u6771\\u4eac faster (parens) faster 42 the F# budget dog 1999 regex before\"}, {\"role\": \"user\", \"content\": \"no it \\ud83d\\ude42 mirrors https://example.com/a/b?c=d reservation C++ on budget \\u6771\\u4eac the for {braces} [brackets] and tiktoken's written 42 so body \\\"quotes\\\" involved C++ budget is written WON'T jumps It's {braces} counting reservation tokens brown piece hand before no It's gateway gateway [brackets] piece mirrors there's \\ud83d\\ude42 while a caf\\u00e9 \\u6771\\u4eac caf\\u00e9 counting counting it 3.14159 don't caf\\u00e9 so written for tiktoken's boundaries tokens 1999 fox keep quick that boundaries that no scanner Z\\u00fcrich budget it written that backtracking involved quick mirrors backtracking brown jumps the lazy They'RE and 42 backtracking 100% brown tokens gateway boundaries caf\\u00e9 reservation way https://example.com/a/b?c=d $1,234.56 100% regex \\u0645\\u0631\\u062d\\u0628\\u0627 lazy we'll \\u0645\\u0631\\u062d\\u0628\\u0627 involved jumps \\u0645\\u0631\\u062d\\u0628\\u0627 while node.js Z\\u00fcrich Z\\u00fcrich counting 100% (parens) that 100% on {braces} counting once user@example.com faster gateway \\u0645\\u0631\\u062d\\u0628\\u0627 that with\"}, {\"role\": \"assistant\", \"content\": \"faster on that boundaries on They'RE F# https://example.com/a/b?c=d that node.js quick for 100% the regex v1.2.3 don't with 42 C++ gateway is before keep is I'M the (parens) They'RE It's \\\"quotes\\\" so mirrors every backtracking tiktoken's v1.2.3 the 3.14159 They'RE scanner over no so that and WON'T that every the don't \\\"quotes\\\" 42 3.14159 node.js budget every 'single' caf\\u00e9 node.js before counting engine\"}, {\"role\": \"user\", \"content\": \"3.14159 piece tiktoken's with request regex on body [brackets] Z\\u00fcrich every gateway the that tiktoken's https://example.com/a/b?c=d no so budget node.js 1999 Z\\u00fcrich we'll They'RE keep and because piece every \\ud83d\\ude42 reservation faster written while request 3.14159 because {braces} fox WON'T written \\\"quotes\\\" (parens) over jumps mirrors body that jumps hand keep 3.14159 a the the it because so we'll gateway tiktoken's regex the scanned 1999 Z\\u00fcrich and don't and\"}, {\"role\": \"assistant\", \"content\": \"quick v1.2.3 100% node.js the na\\u00efve na\\u00efve involved once hand that \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors that request I'M tokens C++ don't na\\u00efve it mirrors there's with tokens https://example.com/a/b?c=d a there's so I'M mirrors $1,234.56 hand mirrors $1,234.56 admission C++ \\u0645\\u0631\\u062d\\u0628\\u0627 scanned user@example.com piece fox admission node.js https://example.com/a/b?c=d caf\\u00e9 It's \\ud83d\\ude42 scanned before https://example.com/a/b?c=d (parens) way https://example.com/a/b?c=d because F# I'M is the It's is reservation exactly user@example.com It's https://example.com/a/b?c=d lazy while the \\ud83d\\ude42 42 because for tiktoken's\"}, {\"role\": \"user\", \"content\": \"fox over {braces} is because caf\\u00e9 hand {braces} backtracking scanner F# engine engine user@example.com It's 42 quick jumps C++ \\\"quotes\\\" Z\\u00fcrich involved C++ na\\u00efve don't we'll admission exactly I'M a {braces} regex $1,234.56 exactly for [brackets] piece F# tokens \\\"quotes\\\" na\\u00efve dog lazy na\\u00efve there's on that with request \\u0645\\u0631\\u062d\\u0628\\u0627 quick a the gateway \\\"quotes\\\" and user@example.com 1999 (parens) over dog involved on engine over so hand over gateway Z\\u00fcrich counting keep that and $1,234.56 tokens scanned [brackets] reservation Z\\u00fcrich it node.js\"}, {\"role\": \"assistant\", \"content\": \"while on because fox \\u6771\\u4eac mirrors quick lazy I'M F# while a the because {braces} \\\"quotes\\\" once mirrors 100% a that on before body over request tokens we'll jumps that for 1999 mirrors hand lazy we'll over dog \\\"quotes\\\" because over way scanned that brown gateway over brown Z\\u00fcrich don't \\ud83d\\ude42 involved mirrors that once v1.2.3 \\ud83d\\ude42 v1.2.3 counting (parens) quick 'single' $1,234.56 scanner https://example.com/a/b?c=d no It's boundaries tokens hand caf\\u00e9 request body scanned 42 a counting with engine there's v1.2.3 fox gateway hand so counting scanned Z\\u00fcrich written https://example.com/a/b?c=d that {braces} v1.2.3 involved there's hand it there's is the exactly \\u0645\\u0631\\u062d\\u0628\\u0627 that scanner\"}, {\"role\": \"user\", \"content\": \"\\u0645\\u0631\\u062d\\u0628\\u0627 piece way scanned we'll hand written jumps fox https://example.com/a/b?c=d lazy $1,234.56 Z\\u00fcrich na\\u00efve WON'T scanner brown \\\"quotes\\\" mirrors It's caf\\u00e9 I'M \\\"quotes\\\" while admission fox engine brown tokens hand over while no scanner 3.14159 reservation budget \\\"quotes\\\" 100% hand while the It's faster tokens hand admission no and F# because \\u6771\\u4eac It's 'single' body that tiktoken's reservation tiktoken's is They'RE every (parens) Z\\u00fcrich scanner request the that F# caf\\u00e9 it They'RE way because\"}, {\"role\": \"assistant\", \"content\": \"brown \\\"quotes\\\" and don't regex [brackets] before on a that is every don't mirrors (parens) over lazy with F# on mirrors boundaries body jumps dog the It's \\u0645\\u0631\\u062d\\u0628\\u0627 so way before jumps WON'T request lazy v1.2.3 caf\\u00e9 it https://example.com/a/b?c=d $1,234.56 tiktoken's and {braces} admission \\u0645\\u0631\\u062d\\u0628\\u0627 fox on jumps jumps faster node.js user@example.com user@example.com so https://example.com/a/b?c=d counting It's \\\"quotes\\\" admission scanner on that while I'M piece written node.js way dog dog 100% before with \\u6771\\u4eac node.js hand mirrors involved 1999 a mirrors\"}, {\"role\": \"user\", \"content\": \"hand tiktoken's [brackets] 42 before 'single' F# F# for 'single' with body way every gateway {braces} brown keep scanner I'M the tiktoken's https://example.com/a/b?c=d for dog 100% tokens body counting over v1.2.3 https://example.com/a/b?c=d no 1999 They'RE {braces} tokens exactly scanner a for \\u0645\\u0631\\u062d\\u0628\\u0627 node.js way $1,234.56 gateway backtracking the brown a tiktoken's dog na\\u00efve I'M tokens for node.js boundaries a It's it C++ that tokens way C++\"}, {\"role\": \"assistant\", \"content\": \"a a counting don't WON'T no tokens quick we'll engine that written \\u0645\\u0631\\u062d\\u0628\\u0627 so 'single' every I'M Z\\u00fcrich there's v1.2.3 lazy there's caf\\u00e9 that scanned keep is fox 1999 100% tokens (parens) jumps 1999 v1.2.3 3.14159 WON'T {braces} for dog [brackets] the involved brown fox [brackets] (parens) They'RE (parens) gateway 1999 before a is scanned keep piece don't I'M exactly na\\u00efve we'll don't that 42 every gateway (parens) the na\\u00efve with Z\\u00fcrich backtracking lazy 100% is node.js faster while quick 42 engine tokens with \\u0645\\u0631\\u062d\\u0628\\u0627 C++ scanner\"}, {\"role\": \"user\", \"content\": \"that hand every exactly \\u6771\\u4eac $1,234.56 tokens {braces} body we'll engine quick once \\u6771\\u4eac on F# v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 I'M over over scanner tiktoken's tokens na\\u00efve that piece tiktoken's once and scanned scanned request with it jumps counting {braces} scanner 100% WON'T scanned exactly is piece 'single' request \\u6771\\u4eac regex for quick written 'single' every F# we'll way na\\u00efve and we'll tokens v1.2.3 regex piece v1.2.3 scanned is with counting with dog gateway admission 100% lazy because that \\ud83d\\ude42 3.14159 F# way hand exactly $1,234.56 while user@example.com It's jumps faster reservation It's don't it mirrors scanned $1,234.56 it while keep $1,234.56 fox $1,234.56 backtracking involved admission 3.14159 mirrors while with on over mirrors C++ for there's fox budget a keep\"}, {\"role\": \"assistant\", \"content\": \"100% scanner user@example.com backtracking every na\\u00efve quick hand jumps body tokens for v1.2.3 exactly \\u0645\\u0631\\u062d\\u0628\\u0627 that \\\"quotes\\\" \\ud83d\\ude42 gateway no caf\\u00e9 that written https://example.com/a/b?c=d the engine piece They'RE budget https://example.com/a/b?c=d involved the https://example.com/a/b?c=d the for involved once exactly every written exactly while 3.14159 user@example.com backtracking request fox is engine way budget counting 1999 piece so 3.14159 tokens F# involved brown a \\u6771\\u4eac scanned keep v1.2.3 for tiktoken's 1999 They'RE and on backtracking counting boundaries na\\u00efve regex\"}, {\"role\": \"user\", \"content\": \"fox is regex https://example.com/a/b?c=d is node.js mirrors user@example.com 100% involved request piece with \\u6771\\u4eac admission the on quick way \\u6771\\u4eac backtracking keep tiktoken's because F# lazy 100% WON'T tiktoken's Z\\u00fcrich involved quick reservation [brackets] written don't {braces} na\\u00efve caf\\u00e9 a $1,234.56 {braces} regex hand 100% 100% so reservation is there's engine \\u0645\\u0631\\u062d\\u0628\\u0627 is 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 backtracking v1.2.3 we'll so quick is before\"}, {\"role\": \"assistant\", \"content\": \"[brackets] request lazy C++ 100% scanned F# a because budget They'RE 1999 that is so \\u0645\\u0631\\u062d\\u0628\\u0627 I'M hand involved 3.14159 admission node.js faster once admission tiktoken's jumps is hand quick tokens mirrors lazy It's 42 fox gateway because request once $1,234.56 mirrors there's so backtracking a over engine we'll \\ud83d\\ude42 body engine fox {braces} {braces} counting hand we'll (parens) that 'single' \\\"quotes\\\" lazy budget na\\u00efve lazy WON'T body v1.2.3 the (parens) \"}, {\"role\": \"user\", \"content\": \"for before quick v1.2.3 scanner tokens faster is hand gateway lazy backtracking C++ budget \\u6771\\u4eac brown (parens) They'RE \\u6771\\u4eac It's \\ud83d\\ude42 3.14159 tokens na\\u00efve backtracking (parens) don't WON'T request WON'T that \\u6771\\u4eac piece v1.2.3 mirrors engine the over mirrors na\\u00efve gateway Z\\u00fcrich C++ written 3.14159 dog involved written \\\"quotes\\\" keep \\\"quotes\\\" They'RE fox written 'single' for backtracking gateway written 1999 the a for F# we'll F# brown faster once engine https://example.com/a/b?c=d it quick piece {braces} \\u0645\\u0631\\u062d\\u0628\\u0627 mirrors [brackets] boundaries $1,234.56 {braces} request keep piece involved body quick once for for They'RE that involved scanned on for 100% mirrors involved budget 3.14159 They'RE faster with \\u6771\\u4eac backtracking reservation the 1999 for \\u0645\\u0631\\u062d\\u0628\\u0627 \\u0645\\u0631\\u062d\\u0628\\u0627 42 \\ud83d\\ude42 keep no caf\\u00e9 engine\"}, {\"role\": \"assistant\", \"content\": \"that tokens dog exactly piece Z\\u00fcrich quick the request 100% every it user@example.com that https://example.com/a/b?c=d way the reservation 1999 F# while a that They'RE backtracking It's Z\\u00fcrich admission budget C++ They'RE we'll keep counting \\ud83d\\ude42 \\\"quotes\\\" that \\ud83d\\ude42 lazy the engine They'RE \\u6771\\u4eac quick involved while no jumps the https://example.com/a/b?c=d is that 3.14159 100% 42 the no while tiktoken's over while written I'M backtracking node.js tokens F# $1,234.56 caf\\u00e9 it while keep (parens) boundaries admission so over no that fox involved F# brown na\\u00efve $1,234.56 reservation lazy body mirrors every $1,234.56 v1.2.3 that scanned user@example.com before quick body with [brackets] brown tiktoken's v1.2.3 fox budget \\\"quotes\\\" 42 every tokens budget don't counting counting no tiktoken's request C++ It's I'M and \\\"quotes\\\" tiktoken's every (parens) user@example.com the the we'll no I'M boundaries F# over on over tokens\"}, {\"role\": \"user\", \"content\": \"tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 It's counting piece $1,234.56 42 once counting there's faster [brackets] 100% piece counting quick \\ud83d\\ude42 caf\\u00e9 exactly written node.js is way on lazy counting it 42 scanned lazy It's v1.2.3 dog dog and because tiktoken's lazy \\ud83d\\ude42 don't involved and once 3.14159 1999 reservation boundaries caf\\u00e9 written written reservation we'll regex {braces} it the https://example.com/a/b?c=d lazy a once body na\\u00efve 1999 tiktoken's caf\\u00e9 caf\\u00e9 for body WON'T \\ud83d\\ude42 no because regex lazy 3.14159 with dog request scanned and involved involved scanned keep that fox https://example.com/a/b?c=d on we'll we'll budget boundaries exactly exactly admission lazy backtracking that F# while fox 1999 boundaries request \\u6771\\u4eac mirrors brown brown 100% Z\\u00fcrich {braces} They'RE 3.14159 v1.2.3 there's v1.2.3 once 1999 faster and WON'T there's we'll on written C++ v1.2.3\"}, {\"role\": \"assistant\", \"content\": \"user@example.com \\u6771\\u4eac because before so keep written for WON'T the They'RE is hand \\ud83d\\ude42 tiktoken's on 42 F# we'll written body user@example.com \\u6771\\u4eac 1999 admission with reservation 42 and that before 3.14159 piece request exactly 1999 for node.js gateway 100% no hand hand it and admission body scanner https://example.com/a/b?c=d involved user@example.com dog boundaries on while 42 body written don't once keep that node.js on reservation \\u0645\\u0631\\u062d\\u0628\\u0627 lazy \\\"quotes\\\" They'RE that with regex while Z\\u00fcrich na\\u00efve that lazy caf\\u00e9 don't jumps before {braces} don't for https://example.com/a/b?c=d keep user@example.com once lazy because counting brown written that 'single' F# \\u6771\\u4eac regex tiktoken's admission I'M a that It's exactly while with so hand don't we'll on\"}, {\"role\": \"user\", \"content\": \"{braces} (parens) fox on over tiktoken's budget lazy for \\\"quotes\\\" 100% so because 100% boundaries Z\\u00fcrich gateway boundaries don't jumps faster don't F# the \\\"quotes\\\" {braces} tokens backtracking They'RE regex every tokens the \\\"quotes\\\" once user@example.com {braces} every \\\"quotes\\\" counting budget It's gateway there's exactly regex is node.js request for F# piece so don't quick 'single' is user@example.com once I'M because user@example.com before user@example.com it v1.2.3 piece a lazy scanned 100% lazy counting dog and we'll we'll for tiktoken's exactly no fox once 1999 \\u6771\\u4eac we'll every the because regex $1,234.56 100% scanner so \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 v1.2.3 on while that na\\u00efve body brown way 3.14159 scanner scanner on They'RE don't over because the 42 involved node.js that a They'RE caf\\u00e9 for \\ud83d\\ude42 user@example.com user@example.com I'M so faster\"}, {\"role\": \"assistant\", \"content\": \"a don't It's gateway fox for \\u6771\\u4eac exactly that exactly I'M piece https://example.com/a/b?c=d \\ud83d\\ude42 faster so once jumps reservation piece boundaries v1.2.3 scanned a while that user@example.com [brackets] keep no there's brown the 1999 backtracking They'RE brown node.js v1.2.3 tiktoken's with is v1.2.3 that hand so I'M and engine boundaries C++ user@example.com tokens keep na\\u00efve {braces} gateway because backtracking [brackets] gateway don't we'll for we'll 100% boundaries counting 3.14159 is the F# it fox node.js for is tiktoken's no backtracking engine and 'single' and involved before fox tokens It's https://example.com/a/b?c=d mirrors written boundaries (parens) reservation the tiktoken's (parens) fox hand body (parens) (parens) it is It's and way because {braces} dog tiktoken's the counting while \\ud83d\\ude42 42 scanned over the\"}, {\"role\": \"user\", \"content\": \"it brown brown tokens backtracking jumps over boundaries with quick counting faster gateway {braces} \\u6771\\u4eac 42 mirrors \\ud83d\\ude42 engine hand \\u6771\\u4eac reservation v1.2.3 quick no no that involved over the scanner faster every mirrors for admission I'M we'll and scanned node.js tokens brown budget 42 written gateway lazy 3.14159 engine it we'll it 3.14159 $1,234.56 the that it (parens) over \\u0645\\u0631\\u062d\\u0628\\u0627 node.js for that {braces} \\\"quotes\\\" way before that no quick tokens and scanner the WON'T boundaries dog C++ boundaries while F# once backtracking 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 a while counting node.js Z\\u00fcrich engine fox faster it F# 1999 gateway written It's 'single' na\\u00efve there's 3.14159 na\\u00efve Z\\u00fcrich brown 3.14159 \\u6771\\u4eac user@example.com WON'T exactly Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 node.js 100% gateway regex while user@example.com and no quick lazy with\"}, {\"role\": \"assistant\", \"content\": \"1999 100% dog so for mirrors tiktoken's backtracking on that fox I'M every for involved involved \\\"quotes\\\" 100% so while 42 and it for F# node.js that gateway backtracking $1,234.56 fox boundaries on budget backtracking mirrors body Z\\u00fcrich for on 42 hand 1999 counting way body brown with scanned 42 [brackets] a jumps I'M written there's no They'RE faster backtracking I'M the don't once admission scanner 'single' \\ud83d\\ude42 way on\"}, {\"role\": \"user\", \"content\": \"every regex \\ud83d\\ude42 3.14159 we'll fox it tiktoken's Z\\u00fcrich \\\"quotes\\\" user@example.com na\\u00efve quick \\u0645\\u0631\\u062d\\u0628\\u0627 fox fox it C++ fox \\u0645\\u0631\\u062d\\u0628\\u0627 body gateway 3.14159 100% fox there's keep They'RE so reservation it lazy with lazy gateway request keep scanned hand every fox budget for quick $1,234.56 piece piece faster hand we'll 42 user@example.com written 1999 \\u6771\\u4eac every na\\u00efve request It's involved caf\\u00e9 [brackets] faster scanned keep 42 It's \\u6771\\u4eac every way is 3.14159 written mirrors brown WON'T once WON'T exactly the They'RE there's scanner because [brackets] gateway v1.2.3 fox tiktoken's 'single' Z\\u00fcrich {braces} jumps caf\\u00e9 I'M https://example.com/a/b?c=d C++ boundaries \\\"quotes\\\" with written we'll quick a with the every way involved scanned so scanner request scanner $1,234.56 exactly the caf\\u00e9 so there's is lazy that jumps user@example.com so with don't faster 1999 request admission They'RE while boundaries reservation\"}, {\"role\": \"assistant\", \"content\": \"jumps no that dog \\u0645\\u0631\\u062d\\u0628\\u0627 the lazy request user@example.com dog with lazy Z\\u00fcrich and tokens \\u6771\\u4eac boundaries engine keep boundaries 'single' so C++ \\ud83d\\ude42 gateway \\ud83d\\ude42 mirrors F# on {braces} once the before body user@example.com boundaries na\\u00efve and keep counting \\\"quotes\\\" while hand request body tiktoken's request we'll so I'M \\\"quotes\\\" on {braces} https://example.com/a/b?c=d They'RE is regex don't dog They'RE (parens) once caf\\u00e9 involved written faster keep before scanner while They'RE boundaries admission once gateway reservation every caf\\u00e9 dog tokens regex 'single' 3.14159 there's counting that tokens dog https://example.com/a/b?c=d piece F# before engine scanner exactly \\\"quotes\\\" don't while\"}, {\"role\": \"user\", \"content\": \"WON'T node.js v1.2.3 involved mirrors on brown keep every tiktoken's F# 100% 'single' They'RE \\u0645\\u0631\\u062d\\u0628\\u0627 na\\u00efve don't 3.14159 \\\"quotes\\\" there's once I'M $1,234.56 dog tokens that the Z\\u00fcrich \\\"quotes\\\" exactly C++ 3.14159 (parens) v1.2.3 no It's no keep user@example.com 42 Z\\u00fcrich a over written tiktoken's keep tiktoken's written with no WON'T $1,234.56 scanner keep keep that fox regex admission every [brackets] WON'T \\u6771\\u4eac 42 engine (parens) C++ we'll tokens a it while [brackets] counting involved the \\u0645\\u0631\\u062d\\u0628\\u0627 lazy 'single' WON'T 3.14159 1999 $1,234.56 regex that 3.14159 because caf\\u00e9 mirrors \\\"quotes\\\" counting 3.14159 with exactly admission every They'RE because 100% body piece na\\u00efve regex counting I'M that admission lazy \\ud83d\\ude42 scanner quick fox we'll brown admission C++\"}, {\"role\": \"assistant\", \"content\": \"\\ud83d\\ude42 the that \\u6771\\u4eac exactly while engine counting \\ud83d\\ude42 42 admission admission once caf\\u00e9 for tokens v1.2.3 jumps piece piece scanner caf\\u00e9 is \\\"quotes\\\" boundaries so jumps hand so while exactly involved written lazy (parens) quick F# with 1999 C++ jumps Z\\u00fcrich faster because quick the scanned C++ because because F# for no there's keep caf\\u00e9 jumps while is tokens quick lazy C++ boundaries 'single' exactly once tiktoken's node.js Z\\u00fcrich https://example.com/a/b?c=d before exactly {braces} boundaries written no caf\\u00e9 100% dog regex WON'T counting dog jumps F# for tokens no budget scanner that {braces} user@example.com C++ scanned \\u6771\\u4eac \\u0645\\u0631\\u062d\\u0628\\u0627 for every way and 'single' mirrors\"}, {\"role\": \"user\", \"content\": \"gateway because tokens jumps \\u6771\\u4eac exactly 3.14159 I'M 'single' It's WON'T body 42 and way is tiktoken's dog that admission on (parens) \\u6771\\u4eac C++ node.js over boundaries piece piece backtracking node.js dog hand on there's C++ 'single' so mirrors (parens) that (parens) the 1999 scanner hand node.js keep (parens) {braces} every [brackets] user@example.com that They'RE over tokens is with every so is quick request piece so reservation 42 brown on \\u6771\\u4eac fox They'RE $1,234.56 They'RE na\\u00efve boundaries jumps 42 quick exactly \\u0645\\u0631\\u062d\\u0628\\u0627 the 3.14159 over na\\u00efve is reservation mirrors it tokens no \\u6771\\u4eac on \\u6771\\u4eac mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 dog for engine before before node.js $1,234.56 100% every no 100% a budget backtracking jumps there's faster there's boundaries tokens every tiktoken's way F# $1,234.56 that every exactly piece\"}, {\"role\": \"assistant\", \"content\": \"over 3.14159 body 100% admission is F# tiktoken's user@example.com once with \\u0645\\u0631\\u062d\\u0628\\u0627 no with quick \\\"quotes\\\" the and $1,234.56 with faster exactly is involved keep exactly before 'single' there's body regex for involved counting https://example.com/a/b?c=d every over user@example.com C++ there's there's They'RE I'M scanned https://example.com/a/b?c=d scanned hand \\ud83d\\ude42 every counting don't fox 'single' WON'T na\\u00efve 100% gateway 'single' I'M no once involved node.js 1999 every admission caf\\u00e9 regex with regex written mirrors $1,234.56 tokens while backtracking we'll on it dog na\\u00efve that I'M every [brackets] a tokens I'M engine \\u0645\\u0631\\u062d\\u0628\\u0627 F# body don't request admission mirrors WON'T keep Z\\u00fcrich engine $1,234.56 request $1,234.56\"}, {\"role\": \"user\", \"content\": \"It's way body mirrors involved written 100% WON'T [brackets] no boundaries piece brown and while over before is there's we'll jumps scanned so \\ud83d\\ude42 backtracking keep don't dog and 42 that \\u0645\\u0631\\u062d\\u0628\\u0627 over and before and \\u6771\\u4eac on for (parens) https://example.com/a/b?c=d the request user@example.com I'M \\u0645\\u0631\\u062d\\u0628\\u0627 [brackets] 3.14159 so Z\\u00fcrich \\u0645\\u0631\\u062d\\u0628\\u0627 involved the hand once there's caf\\u00e9 jumps is body gateway v1.2.3 budget They'RE that it I'M 'single' gateway scanner that brown brown They'RE {braces} caf\\u00e9 hand admission budget scanner $1,234.56 mirrors hand 1999 scanned 3.14159 a It's\"}, {\"role\": \"assistant\", \"content\": \"regex Z\\u00fcrich https://example.com/a/b?c=d lazy engine keep regex admission with before body \\\"quotes\\\" boundaries once tiktoken's gateway 'single' and the way I'M $1,234.56 backtracking the engine dog Z\\u00fcrich 3.14159 Z\\u00fcrich user@example.com fox that keep while over once {braces} \\ud83d\\ude42 admission tokens regex tokens budget no I'M that on caf\\u00e9 faster I'M dog 'single' dog we'll with quick no because tiktoken's that counting tiktoken's we'll involved node.js involved with C++ body (parens) dog over before v1.2.3 100% na\\u00efve while request F# every jumps gateway over a that for it body exactly Z\\u00fcrich \\u6771\\u4eac a mirrors caf\\u00e9 I'M that Z\\u00fcrich brown that https://example.com/a/b?c=d no 'single' boundaries \\u0645\\u0631\\u062d\\u0628\\u0627 we'll They'RE exactly request once lazy It's 3.14159 before \\\"quotes\\\" \\u6771\\u4eac scanned scanned It's jumps C++ They'RE C++ and\"}, {\"role\": \"user\", \"content\": \"the fox They'RE budget faster with brown 3.14159 because backtracking \\ud83d\\ude42 that because reservation the so {braces} is WON'T request 3.14159 100% over a \\u0645\\u0631\\u062d\\u0628\\u0627 brown while admission because there's and once tokens [brackets] is no with a over there's over written scanned don't exactly I'M gateway is boundaries it tokens faster exactly tokens 42 lazy node.js https://example.com/a/b?c=d is engine with\"}, {\"role\": \"assistant\", \"content\": \"3.14159 I'M gateway scanner tokens C++ node.js 1999 engine for jumps user@example.com It's request (parens) so [brackets] \\ud83d\\ude42 that it admission 3.14159 budget WON'T caf\\u00e9 that involved budget v1.2.3 [brackets] node.js They'RE mirrors \\u0645\\u0631\\u062d\\u0628\\u0627 request with v1.2.3 WON'T we'll admission brown don't 1999 \\u6771\\u4eac \\u6771\\u4eac F# for the scanner is 'single' is it piece tokens body body Z\\u00fcrich tiktoken's we'll involved \\u6771\\u4eac on no tokens on {braces} It's mirrors I'M [brackets] \\u0645\\u0631\\u062d\\u0628\\u0627 lazy Z\\u00fcrich a while caf\\u00e9 involved request $1,234.56 tiktoken's once lazy\"}, {\"role\": \"user\", \"content\": \"3.14159 lazy piece fox on lazy v1.2.3 dog while piece 'single' involved written 100% jumps hand scanner v1.2.3 don't 'single' scanned with piece don't They'RE piece (parens) for with F# scanned don't no backtracking written 100% 100% lazy backtracking involved user@example.com scanner way we'll 42 'single' reservation 'single' lazy the written don't 'single' every and before reservation so it Z\\u00fcrich before a node.js before F# involved jumps counting involved that scanned regex so that user@example.com it with \\u0645\\u0631\\u062d\\u0628\\u0627 involved admission caf\\u00e9 the reservation [brackets] over mirrors that hand {braces} \\u0645\\u0631\\u062d\\u0628\\u0627 don't that we'll [brackets] hand faster I'M regex boundaries piece keep reservation WON'T hand backtracking counting piece $1,234.56 \\\"quotes\\\" Z\\u00fcrich tokens jumps it body before don't scanner {braces} body It's 42 caf\\u00e9 $1,234.56 scanner tokens\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 engine involved faster involved keep with $1,234.56 while piece F# Z\\u00fcrich 100% tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 is budget that tiktoken's admission scanned node.js 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 that on on body dog with regex admission 3.14159 piece na\\u00efve so 'single' mirrors boundaries boundaries keep on no while \\u0645\\u0631\\u062d\\u0628\\u0627 reservation the node.js F# 42 piece reservation It's tiktoken's reservation $1,234.56 hand tiktoken's that I'M it \\ud83d\\ude42 (parens) counting Z\\u00fcrich is [brackets] https://example.com/a/b?c=d reservation boundaries lazy (parens) that with over Z\\u00fcrich don't counting\"}, {\"role\": \"user\", \"content\": \"body dog backtracking that we'll don't \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com keep quick before scanned caf\\u00e9 It's body quick na\\u00efve 100% before quick jumps \\\"quotes\\\" once regex 42 I'M \\ud83d\\ude42 and keep a faster brown caf\\u00e9 regex \\\"quotes\\\" we'll engine exactly so 42 written They'RE every caf\\u00e9 the the once na\\u00efve piece no hand that boundaries scanned before there's there's reservation https://example.com/a/b?c=d admission I'M v1.2.3 dog They'RE dog scanned engine It's user@example.com no reservation \\u0645\\u0631\\u062d\\u0628\\u0627 regex scanned Z\\u00fcrich is it a and \\\"quotes\\\" involved exactly exactly way hand jumps \\ud83d\\ude42 reservation quick tokens mirrors over user@example.com boundaries 3.14159 I'M node.js piece F# https://example.com/a/b?c=d before tiktoken's I'M piece brown is na\\u00efve and {braces} WON'T It's don't it is {braces} on so so \\u6771\\u4eac it it user@example.com brown \\\"quotes\\\" I'M 100% scanned C++ body tokens \\u6771\\u4eac [brackets]\"}, {\"role\": \"assistant\", \"content\": \"mirrors F# $1,234.56 for dog (parens) keep that [brackets] (parens) 3.14159 v1.2.3 WON'T is don't the a It's https://example.com/a/b?c=d faster 'single' it user@example.com is dog dog counting so $1,234.56 v1.2.3 boundaries every on regex backtracking before quick brown quick C++ quick v1.2.3 scanner written \\ud83d\\ude42 user@example.com before is 42 3.14159 on engine dog dog once request https://example.com/a/b?c=d lazy on caf\\u00e9 faster don't before is exactly written brown C++ piece a 3.14159 no WON'T faster it https://example.com/a/b?c=d and boundaries \\ud83d\\ude42 reservation that engine we'll fox while the dog v1.2.3 because boundaries scanned It's \\u6771\\u4eac \\\"quotes\\\" scanned that so is tiktoken's scanned {braces} gateway exactly regex quick over is fox It's we'll {braces} faster\"}, {\"role\": \"user\", \"content\": \"$1,234.56 scanned node.js budget with na\\u00efve \\\"quotes\\\" hand because admission before F# [brackets] C++ v1.2.3 backtracking with 'single' hand on the for backtracking (parens) lazy is 1999 faster is [brackets] over v1.2.3 don't there's C++ na\\u00efve quick scanned lazy engine I'M scanned gateway They'RE gateway before there's scanned is v1.2.3 reservation once gateway (parens) It's so 'single' regex it It's brown counting 3.14159 lazy tokens engine backtracking with \\\"quotes\\\" there's [brackets] caf\\u00e9 reservation for scanned there's lazy mirrors WON'T (parens) so user@example.com It's It's {braces} {braces} 3.14159 42 that budget boundaries once no faster F# it WON'T admission because it that involved and every way C++ on jumps user@example.com request for engine keep engine regex It's node.js\"}, {\"role\": \"assistant\", \"content\": \"v1.2.3 brown piece counting backtracking no for budget C++ boundaries I'M tiktoken's tiktoken's {braces} jumps dog \\u6771\\u4eac while brown it with {braces} that counting tokens na\\u00efve because 100% na\\u00efve that I'M $1,234.56 with backtracking with keep reservation brown F# with v1.2.3 exactly jumps that exactly {braces} 1999 tokens $1,234.56 backtracking tokens once exactly \\u6771\\u4eac It's jumps user@example.com boundaries 'single' backtracking scanner scanned node.js 42 mirrors no backtracking caf\\u00e9 3.14159 {braces} It's WON'T dog I'M way that\"}, {\"role\": \"user\", \"content\": \"dog mirrors budget Z\\u00fcrich don't while request is a 1999 that jumps 3.14159 body caf\\u00e9 don't mirrors \\ud83d\\ude42 because Z\\u00fcrich way na\\u00efve tokens dog \\u0645\\u0631\\u062d\\u0628\\u0627 with reservation 100% with there's reservation request with 1999 scanned is faster that keep $1,234.56 I'M budget request because is counting the that and 3.14159 user@example.com scanner request WON'T [brackets] lazy \\ud83d\\ude42 because {braces} with on scanned the there's Z\\u00fcrich mirrors while jumps 'single' \\u6771\\u4eac keep so https://example.com/a/b?c=d mirrors don't $1,234.56 no there's engine (parens) mirrors regex scanned quick \\u6771\\u4eac They'RE WON'T (parens) $1,234.56 node.js \\ud83d\\ude42 don't counting They'RE (parens) because WON'T over request hand It's before a body quick brown caf\\u00e9 caf\\u00e9 $1,234.56 there's brown involved I'M we'll 42 lazy (parens) \\u0645\\u0631\\u062d\\u0628\\u0627 the involved quick over engine caf\\u00e9 v1.2.3 [brackets] reservation a F# counting we'll involved tiktoken's brown\"}, {\"role\": \"assistant\", \"content\": \"I'M backtracking no faster exactly 'single' the before on boundaries because with \\u0645\\u0631\\u062d\\u0628\\u0627 because that the every the involved na\\u00efve It's a jumps it it fox written WON'T and we'll reservation piece tiktoken's fox $1,234.56 v1.2.3 counting hand tokens budget jumps counting v1.2.3 budget budget that and \\u0645\\u0631\\u062d\\u0628\\u0627 dog every 100% that dog brown \\ud83d\\ude42 admission tiktoken's I'M is user@example.com admission while gateway because v1.2.3 for we'll and {braces} request reservation fox exactly no tokens jumps body because before faster once {braces} tokens hand https://example.com/a/b?c=d once jumps 3.14159 budget so C++ exactly lazy once tiktoken's 100% because on 1999 1999 tokens every \\ud83d\\ude42 body\"}, {\"role\": \"user\", \"content\": \"v1.2.3 \\u0645\\u0631\\u062d\\u0628\\u0627 They'RE we'll way fox scanner jumps node.js the user@example.com C++ regex scanned lazy exactly while v1.2.3 hand scanner piece there's involved 42 \\u6771\\u4eac 3.14159 C++ fox the hand reservation before 3.14159 gateway body the jumps na\\u00efve with regex the \\\"quotes\\\" the \\\"quotes\\\" scanned so dog no lazy \\\"quotes\\\" It's scanned \\ud83d\\ude42 mirrors regex body fox faster because engine that dog tiktoken's gateway 42 tiktoken's on body keep \\\"quotes\\\" They'RE is and written $1,234.56 every engine request caf\\u00e9 hand Z\\u00fcrich v1.2.3 mirrors the way piece the 1999 (parens) 42 $1,234.56 engine regex https://example.com/a/b?c=d F# 42 body 100% is caf\\u00e9 Z\\u00fcrich user@example.com with [brackets] mirrors keep over exactly C++ reservation it budget Z\\u00fcrich caf\\u00e9 fox way budget caf\\u00e9 budget boundaries (parens) fox 100% boundaries mirrors\"}, {\"role\": \"assistant\", \"content\": \"[brackets] over because a every boundaries keep tiktoken's the hand 42 backtracking written jumps hand before there's a with don't body [brackets] https://example.com/a/b?c=d request request reservation and 42 admission scanned node.js engine piece 100% reservation I'M https://example.com/a/b?c=d admission no involved F# fox [brackets] regex so They'RE 1999 budget admission no I'M node.js exactly scanner so tiktoken's node.js written \\\"quotes\\\" every brown body regex na\\u00efve that backtracking for 1999 brown brown while v1.2.3 $1,234.56 written hand node.js every scanner Z\\u00fcrich way [brackets] no is It's {braces} \\u6771\\u4eac regex dog user@example.com request for $1,234.56 quick node.js $1,234.56 the engine on admission that written https://example.com/a/b?c=d boundaries a It's once body budget exactly every the before 'single' F# because {braces} once fox no because involved once quick that 42 no reservation there's there's\"}, {\"role\": \"user\", \"content\": \"They'RE 3.14159 42 scanned mirrors jumps reservation we'll caf\\u00e9 counting mirrors {braces} quick a fox written no scanner body https://example.com/a/b?c=d 3.14159 \\\"quotes\\\" scanner mirrors no on with no budget WON'T \\ud83d\\ude42 over brown piece jumps It's counting hand https://example.com/a/b?c=d lazy involved 'single' (parens) with scanned \\u6771\\u4eac engine no request {braces} jumps with {braces} admission for It's keep \\u6771\\u4eac don't gateway because admission don't there's https://example.com/a/b?c=d for v1.2.3 Z\\u00fcrich [brackets] piece because no the engine mirrors 3.14159 42 1999 no scanner the don't mirrors node.js\"}, {\"role\": \"assistant\", \"content\": \"before while [brackets] involved involved https://example.com/a/b?c=d It's hand on keep way we'll quick every exactly They'RE mirrors way before the reservation 1999 body 'single' and quick WON'T C++ {braces} over caf\\u00e9 and brown 42 counting hand user@example.com that so (parens) hand hand hand a hand engine every that counting tiktoken's admission WON'T counting dog on https://example.com/a/b?c=d tiktoken's $1,234.56 https://example.com/a/b?c=d scanner gateway while lazy C++ na\\u00efve 3.14159 don't {braces} faster over body because is gateway backtracking\"}, {\"role\": \"user\", \"content\": \"we'll that once tokens before lazy Z\\u00fcrich no Z\\u00fcrich the scanned \\\"quotes\\\" way is we'll jumps [brackets] involved \\ud83d\\ude42 v1.2.3 while WON'T written F# there's (parens) admission way I'M node.js \\ud83d\\ude42 written budget (parens) {braces} 1999 mirrors so engine \\u0645\\u0631\\u062d\\u0628\\u0627 because 'single' WON'T {braces} body reservation caf\\u00e9 so mirrors once over fox on regex quick dog exactly WON'T every a I'M regex \\u0645\\u0631\\u062d\\u0628\\u0627 don't na\\u00efve way backtracking admission over budget (parens) exactly 3.14159 we'll jumps there's v1.2.3 for counting reservation budget exactly counting while Z\\u00fcrich is $1,234.56 don't mirrors 3.14159 we'll user@example.com {braces} so so [brackets] tokens the scanned user@example.com na\\u00efve that involved don't the there's fox 42 there's budget before scanned It's boundaries tokens piece It's \\u0645\\u0631\\u062d\\u0628\\u0627 for counting written boundaries {braces} jumps $1,234.56 over lazy F# written on\"}, {\"role\": \"assistant\", \"content\": \"on F# mirrors hand \\ud83d\\ude42 exactly Z\\u00fcrich every admission with hand $1,234.56 scanned hand \\u0645\\u0631\\u062d\\u0628\\u0627 node.js that counting piece is there's don't fox node.js for so admission quick faster every that that jumps because tiktoken's over caf\\u00e9 F# once [brackets] F# lazy I'M \\\"quotes\\\" caf\\u00e9 over a no that before v1.2.3 piece {braces} because quick \\u6771\\u4eac once tokens while https://example.com/a/b?c=d na\\u00efve every lazy WON'T don't F#\"}, {\"role\": \"user\", \"content\": \"a a 'single' \\u6771\\u4eac (parens) counting F# with quick the (parens) tiktoken's \\u0645\\u0631\\u062d\\u0628\\u0627 I'M quick because counting backtracking https://example.com/a/b?c=d written 'single' \\u0645\\u0631\\u062d\\u0628\\u0627 caf\\u00e9 caf\\u00e9 dog counting It's and keep caf\\u00e9 tiktoken's budget budget piece piece over piece tiktoken's scanned before over request brown that involved quick reservation the that scanner tiktoken's hand 42 'single' keep exactly it 100% on engine engine budget (parens) the regex written node.js \\u0645\\u0631\\u062d\\u0628\\u0627 user@example.com (parens) faster \\ud83d\\ude42 jumps 100% WON'T https://example.com/a/b?c=d a reservation there's so Z\\u00fcrich https://example.com/a/b?c=d every with It's Z\\u00fcrich tiktoken's\"}, {\"role\": \"assistant\", \"content\": \"don't a a faster 42 that 'single' backtracking before that over engine it hand don't it jumps budget 'single' $1,234.56 piece there's the with \\ud83d\\ude42 42 boundaries It's no that reservation backtracking dog https://example.com/a/b?c=d tokens don't piece on body once the once jumps while \\u6771\\u4eac fox $1,234.56 \\u0645\\u0631\\u062d\\u0628\\u0627 3.14159 we'll every hand v1.2.3 They'RE that a user@example.com https://example.com/a/b?c=d for budget quick hand 100% They'RE caf\\u00e9 while budget is request way They'RE tiktoken's once piece with They'RE gateway scanner hand engine C++ piece a exactly [brackets] we'll before for 42 \\u6771\\u4eac and reservation dog request \\u6771\\u4eac once Z\\u00fcrich v1.2.3 3.14159 \\u6771\\u4eac faster lazy regex user@example.com gateway tokens involved fox They'RE before that (parens) and C++ mirrors node.js once user@example.com over before 'single' there's 100% that faster the don't written scanner admission dog user@example.com\"}, {\"role\": \"user\", \"content\": \"na\\u00efve budget keep request C++ \\ud83d\\ude42 \\\"quotes\\\" involved that Z\\u00fcrich that https://example.com/a/b?c=d with before we'll user@example.com backtracking before while 'single' budget 100% fox They'RE keep {braces} mirrors quick Z\\u00fcrich before caf\\u00e9 way jumps na\\u00efve that hand [brackets] for and It's na\\u00efve Z\\u00fcrich it faster 'single' boundaries with scanned fox faster I'M engine na\\u00efve written node.js na\\u00efve a It's 42 $1,234.56 backtracking body scanner 1999 3.14159 quick scanned scanned tokens 'single' (parens) every 42 quick engine that tokens because admission node.js request backtracking https://example.com/a/b?c=d They'RE counting for we'll body is piece C++ \\ud83d\\ude42 we'll na\\u00efve 3.14159 3.14159 the \\\"quotes\\\" v1.2.3 that the [brackets] $1,234.56 scanner \\ud83d\\ude42 user@example.com v1.2.3 'single' v1.2.3 lazy don't request \\u6771\\u4eac scanned dog They'RE admission [brackets] there's piece It's involved so F#\"}, {\"role\": \"assistant\", \"content\": \"body once mirrors dog there's because that is quick reservation fox because with C++ every tokens node.js 100% 3.14159 no WON'T 1999 counting a no 100% every don't the boundaries engine 42 WON'T budget lazy 42 caf\\u00e9 budget scanner keep hand tiktoken's WON'T because the body regex because there's node.js written involved lazy counting exactly so with once 'single' dog because it before tokens once we'll 42 keep WON'T dog quick na\\u00efve it keep 42 quick\"}, {\"role\": \"user\", \"content\": \"node.js the backtracking scanner while \\u0645\\u0631\\u062d\\u0628\\u0627 while we'll Z\\u00fcrich It's C++ piece \\u0645\\u0631\\u062d\\u0628\\u0627 na\\u00efve counting over {braces} over admission it because and there's a mirrors tokens while keep body exactly WON'T written F# we'll I'M don't for request before with mirrors scanner boundaries (parens) counting that hand so on piece (parens) that before gateway boundaries request 'single' boundaries mirrors written a exactly C++ quick engine involved [brackets] for jumps gateway tiktoken's written piece so 100% engine v1.2.3 3.14159 It's 'single' \\u6771\\u4eac 1999 na\\u00efve is the\"}, {\"role\": \"assistant\", \"content\": \"so mirrors over the lazy and 3.14159 for brown dog engine \\\"quotes\\\" dog engine C++ It's F# user@example.com reservation piece that for caf\\u00e9 body gateway involved is body involved dog tokens involved reservation user@example.com (parens) [brackets] v1.2.3 caf\\u00e9 https://example.com/a/b?c=d quick written hand because lazy 42 42 3.14159 gateway keep I'M lazy counting It's caf\\u00e9 regex a scanned there's there's\"}, {\"role\": \"user\", \"content\": \"node.js lazy faster gateway budget we'll once \\u0645\\u0631\\u062d\\u0628\\u0627 admission tiktoken's that na\\u00efve boundaries budget user@example.com that Z\\u00fcrich gateway user@example.com way it 3.14159 {braces} for before that They'RE brown so for written counting body that 'single' na\\u00efve lazy there's body tokens it engine 3.14159 so mirrors written 3.14159 dog on counting request They'RE we'll admission na\\u00efve reservation jumps $1,234.56 a for every tiktoken's before caf\\u00e9\"}, {\"role\": \"assistant\", \"content\": \"piece the quick 1999 reservation way na\\u00efve F# every v1.2.3 (parens) Z\\u00fcrich counting once na\\u00efve Z\\u00fcrich admission \\u6771\\u4eac exactly [brackets] there's we'll budget we'll 1999 budget lazy \\\"quotes\\\" {braces} counting tokens boundaries keep tiktoken's dog piece mirrors keep exactly over caf\\u00e9 admission C++ quick F# lazy \\ud83d\\ude42 before keep jumps It's the involved so admission the C++ node.js tokens mirrors WON'T reservation once no\"}, {\"role\": \"user\", \"content\": \"Summarise the conversation so far in three sentences.\"}]}", "input_tokens": 50422} diff --git a/litellm-rust/crates/token-counter/tests/fixtures/o200k/texts.jsonl b/litellm-rust/crates/token-counter/tests/fixtures/o200k/texts.jsonl new file mode 100644 index 00000000000..9d78131a456 --- /dev/null +++ b/litellm-rust/crates/token-counter/tests/fixtures/o200k/texts.jsonl @@ -0,0 +1,4053 @@ +{"text": "", "tokens": 0, "pieces": []} +{"text": "Hello, how are you today?", "tokens": 7, "pieces": ["Hello", ",", " how", " are", " you", " today", "?"]} +{"text": "I'm sure they're right, we'll see. WE'LL SEE, I'M SURE THEY'RE RIGHT, IT'S HERS AND IT'D BE 'D", "tokens": 32, "pieces": ["I'm", " sure", " they're", " right", ",", " we'll", " see", ".", " WE'LL", " SEE", ",", " I'M", " SURE", " THEY'RE", " RIGHT", ",", " IT'S", " HERS", " AND", " IT'D", " BE", " '", "D"]} +{"text": "don't Don'T DON'T won'T i've I'VE i'Ve you'RE 'S 'T 'M 'D 'LL 'VE 'RE 'ſ 'x", "tokens": 33, "pieces": ["don't", " Don'T", " DON'T", " won'T", " i've", " I'VE", " i'Ve", " you'RE", " '", "S", " '", "T", " '", "M", " '", "D", " '", "LL", " '", "VE", " '", "RE", " '", "ſ", " '", "x"]} +{"text": "1234567890 123 12 1 0000000 ٣٤٥٦٧٨ ३४५६ 1,234,567.89 2026-09-11T18:00:00Z", "tokens": 48, "pieces": ["123", "456", "789", "0", " ", "123", " ", "12", " ", "1", " ", "000", "000", "0", " ", "٣٤٥", "٦٧٨", " ", "३४५", "६", " ", "1", ",", "234", ",", "567", ".", "89", " ", "202", "6", "-", "09", "-", "11", "T", "18", ":", "00", ":", "00", "Z"]} +{"text": "$abc %def &ghi @jkl _mno #pqr ~stu ^vwx |yz \\a /b :c ;d ?e !f (g )h [i ]j {k }l n =o +p *q", "tokens": 56, "pieces": ["$abc", " %", "def", " &", "ghi", " @", "jkl", " _", "mno", " #", "pqr", " ~", "stu", " ^", "vwx", " |", "yz", " \\", "a", " /", "b", " :", "c", " ;", "d", " ?", "e", " !", "f", " (", "g", " )", "h", " [", "i", " ]", "j", " {", "k", " }", "l", " <", "m", " >", "n", " =", "o", " +", "p", " *", "q"]} +{"text": "foo bar baz \t qux\t\tquux \n\nline\r\nline\r\n\r\n \n\t\r\n x ", "tokens": 22, "pieces": ["foo", " ", " bar", " ", " baz", " \t", " qux", "\t", "\tquux", " \n\n", "line", "\r\n", "line", "\r\n\r\n \n\t\r\n", " ", " x", " "]} +{"text": "trailing spaces ", "tokens": 4, "pieces": ["trailing", " spaces", " "]} +{"text": "trailing tabs\t\t", "tokens": 4, "pieces": ["trailing", " tabs", "\t\t"]} +{"text": "trailing newline\n", "tokens": 4, "pieces": ["trailing", " newline", "\n"]} +{"text": "\n\n\n", "tokens": 1, "pieces": ["\n\n\n"]} +{"text": "\r\n\r\n\r\n", "tokens": 1, "pieces": ["\r\n\r\n\r\n"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "😀😃😄 👍🏽 🇺🇸 👨‍👩‍👧‍👦 ✈️ ❤️‍🔥 ٭ ※ ⌘ ⏎", "tokens": 38, "pieces": ["😀😃😄", " 👍🏽", " 🇺🇸", " 👨‍👩‍👧‍👦", " ✈️", " ❤️‍🔥", " ٭", " ※", " ⌘", " ⏎"]} +{"text": "漢字かな交じり文、東京都千代田区。日本語のテキストです。中文测试。한국어 텍스트", "tokens": 30, "pieces": ["漢字かな交じり文", "、東京都千代田区", "。日本語のテキストです", "。中文测试", "。한국어", " 텍스트"]} +{"text": "مرحبا بالعالم، هذا نص عربي مع أرقام ١٢٣٤٥٦٧ و علامات ترقيم!", "tokens": 24, "pieces": ["مرحبا", " بالعالم", "،", " هذا", " نص", " عربي", " مع", " أرقام", " ", "١٢٣", "٤٥٦", "٧", " و", " علامات", " ترقيم", "!"]} +{"text": "Zürich, façade, naïve, Ærøskøbing, Ελληνικά, Русский текст, עברית, हिन्दी, ไทย", "tokens": 29, "pieces": ["Zürich", ",", " façade", ",", " naïve", ",", " Ærøskøbing", ",", " Ελληνικά", ",", " Русский", " текст", ",", " עברית", ",", " हिन्दी", ",", " ไทย"]} +{"text": "é å ḍ̇ ́́ combining̈ markś!", "tokens": 16, "pieces": ["é", " å", " ḍ̇", " ́́", " combining̈", " markś", "!"]} +{"text": "ΣΊΣΥΦΟΣ Džungla İstanbul file flow Abc ㍿ ㋿ ꟲ 𐞁", "tokens": 40, "pieces": ["ΣΊΣΥΦΟΣ", " Džungla", " İstanbul", " file", " flow", " Abc", " ㍿", " ㋿", " ꟲ", " 𐞁"]} +{"text": "<|endoftext|> <|fim_prefix|>code<|fim_middle|>more<|fim_suffix|> <|endofprompt|> <|im_start|>", "tokens": 40, "pieces": ["<|", "endoftext", "|>", " <|", "fim", "_prefix", "|>", "code", "<|", "fim", "_middle", "|>", "more", "<|", "fim", "_suffix", "|>", " <|", "endofprompt", "|>", " <|", "im", "_start", "|>"]} +{"text": " [INST] [/INST] <>", "tokens": 22, "pieces": ["", " <", "META", "_START", ">", " <", "s", ">", " ", " [", "INST", "]", " [/", "INST", "]", " <<", "SYS", ">>"]} +{"text": "def f(x):\n return {'a': x ** 2, \"b\": [1, 2, 3]} # comment\n\nprint(f(10))\n", "tokens": 35, "pieces": ["def", " f", "(x", "):\n", " ", " return", " {'", "a", "':", " x", " **", " ", "2", ",", " \"", "b", "\":", " [", "1", ",", " ", "2", ",", " ", "3", "]}", " ", " #", " comment", "\n\n", "print", "(f", "(", "10", "))\n"]} +{"text": "{\"model\":\"gpt-4\",\"messages\":[{\"role\":\"user\",\"content\":\"hi\\n\"}],\"temperature\":0.7}", "tokens": 27, "pieces": ["{\"", "model", "\":\"", "gpt", "-", "4", "\",\"", "messages", "\":[{\"", "role", "\":\"", "user", "\",\"", "content", "\":\"", "hi", "\\n", "\"}],\"", "temperature", "\":", "0", ".", "7", "}"]} +{"text": "https://example.com/path?query=1&other=two#fragment user@example.com 192.168.0.1", "tokens": 26, "pieces": ["https", "://", "example", ".com", "/path", "?query", "=", "1", "&other", "=two", "#fragment", " user", "@example", ".com", " ", "192", ".", "168", ".", "0", ".", "1"]} +{"text": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", "tokens": 375, "pieces": ["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"]} +{"text": " ", "tokens": 24, "pieces": [" "]} +{"text": "........................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................", "tokens": 48, "pieces": ["........................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................................"]} +{"text": "abababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababab", "tokens": 750, "pieces": ["abababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababababab"]} +{"text": "\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n", "tokens": 188, "pieces": ["\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n"]} +{"text": "000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000", "tokens": 1000, "pieces": ["000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000", "000"]} +{"text": "!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!", "tokens": 188, "pieces": ["!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!"]} +{"text": "😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀", "tokens": 1000, "pieces": ["😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀"]} +{"text": "漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢", "tokens": 1000, "pieces": ["漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢漢"]} +{"text": " abc ! 
x  y ​z ‍‍ q", "tokens": 15, "pieces": [" abc", " ", "!", " ", "
x", " ", " y", " ​", "z", " ‍‍", " q"]} +{"text": "x…y \u000b\f z", "tokens": 8, "pieces": ["x", "…y", " \u000b\f", " z"]} +{"text": "\u0000\u0001\u0002  �", "tokens": 6, "pieces": ["\u0000\u0001\u0002", " ", " �"]} +{"text": "tab\tseparated\tvalues\n1\t2\t3\n", "tokens": 11, "pieces": ["tab", "\tseparated", "\tvalues", "\n", "1", "\t", "2", "\t", "3", "\n"]} +{"text": "MiXeD cAsE wOrDs AND ACRONYMS like NASA, HTTP/2, gRPC, iOS, macOS", "tokens": 29, "pieces": ["Mi", "Xe", "D", " c", "As", "E", " w", "Or", "Ds", " AND", " ACRONYMS", " like", " NASA", ",", " HTTP", "/", "2", ",", " g", "RPC", ",", " i", "OS", ",", " mac", "OS"]} +{"text": "snake_case_identifier camelCaseIdentifier PascalCaseIdentifier SCREAMING_SNAKE_CASE kebab-case", "tokens": 19, "pieces": ["snake", "_case", "_identifier", " camel", "Case", "Identifier", " Pascal", "Case", "Identifier", " SCREAMING", "_SNAKE", "_CASE", " kebab", "-case"]} +{"text": "x'sy x'ty x'rey x'vey x'my x'lly x'dy x'S x'T x'RE x'VE x'M x'LL x'D x'sS x'llL", "tokens": 44, "pieces": ["x's", "y", " x't", "y", " x're", "y", " x've", "y", " x'm", "y", " x'll", "y", " x'd", "y", " x'S", " x'T", " x'RE", " x'VE", " x'M", " x'LL", " x'D", " x's", "S", " x'll", "L"]} +{"text": "IT'SOK it'Dbe x'Sy x'Ty x'My x'Dy x'LLy x'VEy x'REy x'Ly x'Vy x'Ry 'Sx'Tx'Mx'LLx'VEx'REx'Dx", "tokens": 57, "pieces": ["IT'S", "OK", " it'D", "be", " x'S", "y", " x'T", "y", " x'M", "y", " x'D", "y", " x'LL", "y", " x'VE", "y", " x'RE", "y", " x", "'Ly", " x", "'Vy", " x", "'Ry", " '", "Sx'T", "x'M", "x'LL", "x'VE", "x'RE", "x'D", "x"]} +{"text": "'s't're've'm'll'd 'S'T'RE'VE'M'LL'D ''s '''s", "tokens": 22, "pieces": ["'s't", "'re've", "'m'll", "'d", " '", "S'T", "'RE'VE", "'M'LL", "'D", " ''", "s", " '''", "s"]} +{"text": "9'9 9's a'9 '9 ' 's' ' 's", "tokens": 18, "pieces": ["9", "'", "9", " ", "9", "'s", " a", "'", "9", " '", "9", " '", " '", "s", "'", " '", " '", "s"]} +{"text": "١٢٣٤ ½⅓¼ ⅣⅤ 𝟘𝟙𝟚𝟛𝟜𝟝𝟞𝟟𝟠𝟡 ①②③", "tokens": 48, "pieces": ["١٢٣", "٤", " ", "½⅓¼", " ", "ⅣⅤ", " ", "𝟘𝟙𝟚", "𝟛𝟜𝟝", "𝟞𝟟𝟠", "𝟡", " ", "①②③"]} +{"text": "camelCase PascalCase ABCdef ABCdeF ABC aB Ab ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyzABC", "tokens": 21, "pieces": ["camel", "Case", " Pascal", "Case", " ABCdef", " ABCde", "F", " ABC", " a", "B", " Ab", " ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz", "ABC"]} +{"text": "日本ABC ABC日本 日本語abc abc日本語 漢字Kanji kanji漢字 KANJI漢字kanji مرحباABC ABCمرحبا abcمرحبا", "tokens": 35, "pieces": ["日本", "ABC", " ABC日本", " 日本語abc", " abc日本語", " 漢字Kanji", " kanji漢字", " KANJI漢字kanji", " مرحبا", "ABC", " ABCمرحبا", " abcمرحبا"]} +{"text": "́ABC ́abc ́́A Á́ ÉA aÉ !!́a  ́A ẍY Ẍy", "tokens": 31, "pieces": ["́", "ABC", " ́abc", " ́́", "A", " Á́", " É", "A", " a", "É", " !!́", "a", " ", " ́", "A", " ẍ", "Y", " Ẍy"]} +{"text": "ᵃbc ᵃBC Aᵃbc Aᵃ ᵃ' ᵃ's Džungla aDžB ADžB ADžb DžDž Ljx İi ΣΊΣΥΦΟΣσ ΣσΣ", "tokens": 67, "pieces": ["ᵃbc", " ᵃ", "BC", " Aᵃbc", " Aᵃ", " ᵃ", "'", " ᵃ's", " Džungla", " a", "DžB", " ADžB", " ADžb", " DžDž", " Ljx", " İi", " ΣΊΣΥΦΟΣσ", " Σσ", "Σ"]} +{"text": "don'tx ABC's abc'S abc'ſ ABC'ſx IT'SOK it'Dbe 'sabc x's 's 'Sx'Tx x’s X'LLx X'Ll", "tokens": 40, "pieces": ["don't", "x", " ABC's", " abc'S", " abc'ſ", " ABC'ſ", "x", " IT'S", "OK", " it'D", "be", " '", "sabc", " x's", " '", "s", " '", "Sx'T", "x", " x", "’s", " X'LL", "x", " X'Ll"]} +{"text": "!ABC !AbC !!abc #camelCase (ABCdef)  ABC abc Abc \tABC\tabc", "tokens": 27, "pieces": ["!ABC", " !", "Ab", "C", " !!", "abc", " #", "camel", "Case", " (", "ABCdef", ")", " ", " ABC", " abc", " Abc", " ", "\tABC", "\tabc"]} +{"text": "!!/\n/x a/b !!\n/x /x // path/to/file.rs http://x.y/z?a=b/c \\/\\/ //\r\n//\n", "tokens": 29, "pieces": ["!!/\n/", "x", " a", "/b", " !!\n/", "x", " ", " /", "x", " ", " //", " path", "/to", "/file", ".rs", " http", "://", "x", ".y", "/z", "?a", "=b", "/c", " \\/\\/", " //\r\n//\n"]} +{"text": "x \n x \r\n \r\n y x \n a b \n\n c x\t\ty x\t\t end \n \n", "tokens": 23, "pieces": ["x", " \n", " x", " \r\n \r\n", " y", " x", " \n", " ", " a", " ", " b", " \n\n", " ", " c", " x", "\t", "\ty", " x", "\t\t", " end", " \n \n"]} +{"text": "12345 6 1abc abc1 ABC123abc 123ABC ١٢٣٤٥abc", "tokens": 22, "pieces": ["123", "45", " ", "6", " ", "1", "abc", " abc", "1", " ABC", "123", "abc", " ", "123", "ABC", " ", "١٢٣", "٤٥", "abc"]} +{"text": "Ⅳ٣٤٥٦<|endoftext|>9
Dž#$%", "tokens": 19, "pieces": ["Ⅳ٣٤", "٥٦", "<|", "endoftext", "|>", "9", "
Dž", "#$%"]} +{"text": "́!!ſİ'D​a'll 字0'MZſⅣ ḍ̇éfi㍿𐞁<|endoftext|>'reA'S#$%", "tokens": 40, "pieces": ["́", "!!", "ſ", "İ'D", "​a'll", " 字", "0", "'MZſ", "Ⅳ", " ḍ̇éfi", "㍿𐞁", "<|", "endoftext", "|>'", "re", "A'S", "#$%"]} +{"text": "ع字'T\r\ń½sꟲ㋿'VE'S<😀🏽!!12345678 ٣٤٥٦Džſḍ̇\réEOT­'ſ<|endoftext|><|fim_prefix|>ś
\tm", "tokens": 65, "pieces": ["ع字'T", "\r\n", "́", "½", "sꟲ", "㋿'", "VE", "'", "S", "<😀🏽!!", "123", "456", "78", " <", "EOT", ">", "٣٤٥", "٦", "Džſḍ̇", "\r", "é", "EOT", "­'", "ſ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "ś", "
", "\tm"]} +{"text": "9'Re\r\n'ſ  'T'Re \né#$%<½!!٣٤٥٦'ſ<\"عßZt\u000bDž'T<|fim_prefix|>ſ㋿\"éſ", "tokens": 51, "pieces": ["9", "'", "Re", "\r\n", "'ſ", "  ", " '", "T'Re", " \n", "é", "#$%<", "½", "!!", "٣٤٥", "٦", "'ſ", "<<", "META", "_START", ">\"", "عß", "Zt", "\u000bDž'T", "<|", "fim", "_prefix", "|>", "ſ", "㋿\"", "éſ"]} +{"text": "<|endoftext|>12345678#$%tعⅣ'T0'D<|endoftext|>é'M-'ſ'sß12345678ꟲ0(>\r\n'MZ'Sa'M", "tokens": 55, "pieces": ["<|", "endoftext", "|>", "123", "456", "78", "#$%", "tع", "Ⅳ", "'T", "0", "'D", "<|", "endoftext", "|>", "é'M", "-<", "EOT", ">'", "ſ's", "ß", "123", "456", "78", "ꟲ", "0", "(>\r\n", "'MZ'S", "a'M"]} +{"text": " \n'll\"‍ d \nß㍿\u000baⅣ😀🏽́ſ\n \n
'Reİ\tDž٣٤٥٦ع'llEOT.\nݽ٣٤٥٦>å\u000b<|fim_prefix|>\"𐞁", "tokens": 55, "pieces": [" \n", "'ll", "\"‍", " ", " d", " \n", "ß", "㍿", "\u000ba", "Ⅳ", "😀🏽́", "ſ", "\n \n", "
", "'Re", "İ", "\tDž", "٣٤٥", "٦", "ع'll", "EOT", ".\n", "İ", "½٣٤", "٥٦", ">å", "\u000b", "<|", "fim", "_prefix", "|>\"", "𐞁"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "å'remfi㍿s", "tokens": 9, "pieces": ["å're", "mfi", "㍿s"]} +{"text": "<|endoftext|>", "tokens": 7, "pieces": ["<|", "endoftext", "|>"]} +{"text": "🙂३'M🙂 12345678'VÉ,\u000bİ<|fim_prefix|> A'TDž 's!!", " ", "A'T", "Dž", " ", "'s", "!!<", "t", "\"!", "d", " \n", "m'M", "('", "ſ"]} +{"text": "\t", "tokens": 1, "pieces": ["\t"]} +{"text": "…<|fim_prefix|>\n.-åß\r\n\r\nEOT'llå½-fi!é'VE12345678EOT字'VE!🙂<|fim_prefix|>'ſ'D \r\r漢0…", "tokens": 52, "pieces": ["…", "<|", "fim", "_prefix", "|>\n", ".-", "åß", "\r\n\r\n", "EOT'll", "å", "½", "-fi", "!é'VE", "123", "456", "78", "EOT字'VE", "!🙂<|", "fim", "_prefix", "|>'", "ſ'D", " \r\r", "漢", "0", "…"]} +{"text": "a'S'T👍🏽> \n-s㍿.㍿#$%́\r\n\r\n", "tokens": 20, "pieces": ["a'S", "'T", "👍🏽>", " \n", "-s", "㍿.㍿#$%́\r\n\r\n"]} +{"text": " 漢#$%­…­ع­ß#$%\t0é", "tokens": 17, "pieces": [" 漢", "#$%­", "…", "­ع", "­ß", "#$%", "\t", "0", "é"]} +{"text": ",9'㍿-", "tokens": 10, "pieces": [",", "9", "'㍿-"]} +{"text": "٣٤٥٦Ⅳ漢İ'sع", "tokens": 10, "pieces": ["٣٤٥", "٦Ⅳ", "漢", "İ's", "ع"]} +{"text": ".éé'sZ>éEOTZ'0​e½ 0#$%́fiⅣ", "tokens": 24, "pieces": [".éé's", "Z", ">é", "EOTZ", "'", "0", "​e", "½", " ", " ", "0", "#$%́", "fi", "Ⅳ"]} +{"text": "t<Ⅳ…Dž'VE ꟲ٣٤٥٦éåع㍿'s字👍🏽ع ३EOTⅣ😀🏽e \nA\"", "tokens": 51, "pieces": ["t", "<", "Ⅳ", "…Dž'VE", "", " ", " ꟲ", "٣٤٥", "٦", "éåع", "㍿'<", "META", "_START", ">s字", "👍🏽", "ع", " ", " ", "३", "EOT", "Ⅳ", "😀🏽", "e", " \n", "A", "\""]} +{"text": "' . t\t<|fim_prefix|>㍿Dž३s😀🏽\t㍿EOTå𐞁​EOT
\t- \ns \n#$%ḍ̇é\r\n ee漢", "tokens": 55, "pieces": ["'", " .", " ", " t", "\t", "<|", "fim", "_prefix", "|>㍿", "Dž", "३", "s", "😀🏽", "\t", "㍿EOTå𐞁", "​EOT", "
", "\t", "-", " \n", "s", " \n", "#$%", "ḍ̇é", "\r\n", " ee漢"]} +{"text": "'ſEOTéⅣ\nİ'S!!!t<|endoftext|>éA<|fim_prefix|>Ⅳ'Z‍'Re'
<|endoftext|>\u000b<㋿ #$%漢A\"ꟲ㍿'T'T", "tokens": 65, "pieces": ["'ſ", "EOTé", "Ⅳ", "\n", "İ'S", "!!!", "t", "<|", "endoftext", "|>", "é", "A", "<|", "fim", "_prefix", "|>", "Ⅳ", "'Z", "‍'", "Re", "'", "
", "<|", "endoftext", "|>", "\u000b", "<㋿", " ", "#$%", "漢", "A", "\"ꟲ", "㍿'", "T'T"]} +{"text": "'Re😀🏽½(.>\u000bſa㍿<|fim_prefix|>>t!'ll३ꟲ \né\n0e\r\n\r\n…😀🏽½́dḍ̇𐞁\r\n\r\n<|fim_prefix|>.<|endoftext|>9", "tokens": 63, "pieces": ["'Re", "😀🏽", "½", "(.>", "\u000bſa", "㍿<|", "fim", "_prefix", "|>>", "t", "!'", "ll", "३", "ꟲ", " \n", "é", "\n", "0", "e", "\r\n\r\n", "…", "😀🏽", "½", "́dḍ̇𐞁", "\r\n\r\n", "<|", "fim", "_prefix", "|>.<|", "endoftext", "|>", "9"]} +{"text": "\r\nå👍🏽!!é", "tokens": 9, "pieces": ["\r\n", "å", "👍🏽!!", "é"]} +{"text": "dİ(.عع \n字㍿\nå(Z'ſ㍿\r\n\r\n,<|fim_prefix|>", "tokens": 29, "pieces": ["d", "İ", "(.", "عع", " \n", "字", "㍿\n", "å", "(Z'ſ", "㍿\r\n\r\n", ",<|", "fim", "_prefix", "|>"]} +{"text": "!", "tokens": 1, "pieces": ["!"]} +{"text": " \n\r\n.s <|endoftext|>aꟲ'sꟲ३…\r\n\r\n\u000bDž‍\t9🙂ſ", "tokens": 30, "pieces": [" \n\r\n", ".s", " <|", "endoftext", "|>", "aꟲ's", "ꟲ", "३", "…\r\n\r\n", "\u000bDž", "‍", "\t", "9", "🙂ſ"]} +{"text": "'re𐞁é३ fi", "tokens": 10, "pieces": ["'re𐞁é", "३", " ", " fi"]} +{"text": "12345678 #$%<|fim_prefix|>‍㍿'T😀🏽fi'll's'S12345678½é
,🙂٣٤٥٦#$%👍🏽🙂12345678<Ⅳ!\"'VE
", "tokens": 59, "pieces": ["123", "456", "78", " ", " #$%<|", "fim", "_prefix", "|>‍㍿'", "T", "😀🏽", "fi'll", "'s'S", "123", "456", "78½", "é", "
", ",🙂<", "META", "_START", ">", "٣٤٥", "٦", "#$%👍🏽🙂", "123", "456", "78", "<", "Ⅳ", "!\"'", "VE", "
"]} +{"text": "fi>İ𐞁…\u000b'D­\tß👍🏽 Ⅳß'Dé\r\n \nß 👍🏽", "tokens": 36, "pieces": ["fi", ">İ𐞁", "…", "\u000b", "'D", "­", "\t", "ß", "👍🏽", " ", " ", "Ⅳ", "ß'D", "é", "\r\n \n", "ß", " ", "👍🏽"]} +{"text": "#$% ßع́'T<A012345678 \n<|fim_prefix|> ㍿👍🏽'", "tokens": 62, "pieces": ["👍🏽>", "A", "012", "345", "678", " \n", "<|", "fim", "_prefix", "|>", " ", "㍿👍🏽'"]} +{"text": "ꟲ'll#$%(\r\n\r\nZ0\u000b👍🏽
'VE ,½'ll­EOT", "tokens": 23, "pieces": ["ꟲ'll", "#$%(\r\n\r\n", "Z", "0", "\u000b", "👍🏽", "
", "'VE", " ", ",", "½", "'ll", "­EOT"]} +{"text": "İ#$%\n9sßEOTd-!!<|endoftext|> 'ReAAfiⅣ'ſéſ🙂étſ\ne'ſ㋿é'VE\"ꟲ漢", "tokens": 47, "pieces": ["İ", "#$%\n", "9", "sß", "EOTd", "-!!<|", "endoftext", "|>", " ", "'Re", "AAfi", "Ⅳ", "'ſéſ", "🙂étſ", "\n", "e'ſ", "㋿é'VE", "\"ꟲ漢"]} +{"text": "<३…😀🏽m>'s<|endoftext|>\r\n\r\n३'ſ'S<|endoftext|><|fim_prefix|>Dž🙂ſⅣA㋿-'re#$%!é\r<|fim_prefix|>å9…s'VE", "tokens": 63, "pieces": ["<", "३", "…", "😀🏽", "m", ">'", "s", "<|", "endoftext", "|>\r\n\r\n", "३", "'ſ'S", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "Dž", "🙂ſ", "Ⅳ", "A", "㋿-'", "re", "#$%!", "é", "\r", "<|", "fim", "_prefix", "|>", "å", "9", "…s'VE"]} +{"text": " <|endoftext|>\r\t<|endoftext|>…ßſ\n#$%🙂㋿ḍ̇\r\n\r\n", "tokens": 34, "pieces": [" ", "<|", "endoftext", "|>\r", "\t", "<|", "endoftext", "|>", "…ßſ", "\n", "#$%🙂㋿", "ḍ̇", "\r\n\r\n"]} +{"text": " \n(\u000b!!\u000b\r\n\t́'t漢३!\nß \n㍿\t'T,-m-\u000b>ſ \ns३
<|endoftext|>㋿ \n'S'VE>\u000b🙂\r\n", "tokens": 47, "pieces": [" \n", "(", "\u000b", "!!", "\u000b\r\n", "\t́'t", "漢", "३", "!\n", "ß", " \n", "㍿", "\t", "'T", ",-", "m", "-", "\u000b", ">ſ", " \n", "s", "३", "
", "<|", "endoftext", "|>㋿", " \n", "'S'VE", ">", "\u000b", "🙂\r\n"]} +{"text": "d'Rea'0㍿İé👍🏽s.٣٤٥٦('S\",​
12345678…ꟲ.\r\n\r\n'T㋿ ½'D", "tokens": 51, "pieces": ["d'Re", "a", "'", "0", "㍿İé", "👍🏽", "s", ".", "٣٤٥", "٦", "('", "S", "<", "META", "_START", ">\",​", "
", "123", "456", "78", "…ꟲ", ".\r\n\r\n", "<", "EOT", ">'", "T", "㋿", " ", "½", "'D"]} +{"text": "'ſ'S…\" 'll㍿. ! (s!!\r\nß字㋿ḍ̇Z'ſ<|fim_prefix|>'Tß㍿ſ\n9\r\n#$%
㍿'T", "tokens": 58, "pieces": ["'ſ'S", "…", "\"", " '", "ll", "㍿.", " !", " ", "(<", "EOT", ">s", "!!\r\n", "ß字", "㋿ḍ̇", "Z'ſ", "<|", "fim", "_prefix", "|>'", "Tß", "㍿ſ", "\n", "9", "\r\n", "#$%", "
", "㍿'", "T"]} +{"text": "İm#$%🙂é'll'VE'VEfi \n\r\r0漢عt'llİd's\r'M­9", "tokens": 26, "pieces": ["İm", "#$%🙂", "é'll", "'VE'VE", "fi", " \n\r\r", "0", "漢عt'll", "İd's", "\r", "'M", "­", "9"]} +{"text": "́'T'ſ \ne\u000bⅣ\ns9ḍ̇'S,\r\n\r\néed're<|fim_prefix|> 👍🏽\u000b½'Re", "tokens": 33, "pieces": ["́'T", "'ſ", " \n", "e", "\u000b", "Ⅳ", "\n", "s", "9", "ḍ̇'S", ",\r\n\r\n", "éed're", "<|", "fim", "_prefix", "|>", " ", " 👍🏽", "\u000b", "½", "'Re"]} +{"text": "🙂'VE‍<​<|endoftext|>ⅣEOTdſ‍Dž-'D(", "tokens": 25, "pieces": ["🙂'", "VE", "‍<​<|", "endoftext", "|>", "Ⅳ", "EOTdſ", "‍Dž", "-'", "D", "("]} +{"text": "İ'Reİ𐞁'ſ\re", "tokens": 11, "pieces": ["İ'Re", "İ𐞁'ſ", "\r", "e"]} +{"text": "…åt\r'VE\nع​<|fim_prefix|>Ⅳß😀🏽s㍿<|fim_prefix|>,\r ­ꟲ…İ're!!,'T< 9EOT", "tokens": 55, "pieces": ["…åt", "\r", "'VE", "\n", "ع", "​<|", "fim", "_prefix", "|>", "Ⅳ", "ß", "😀🏽", "s", "㍿<|", "fim", "_prefix", "|>,\r", " ", " ­", "ꟲ", "…İ're", "!!,'", "T", "<", " ", "9", "EOT"]} +{"text": "a‍\"mé,\rßꟲ'llé,t…#$% 'M!!t㍿'VE<|endoftext|>t('ſ", "tokens": 37, "pieces": ["a", "‍\"", "mé", ",\r", "ßꟲ'll", "é", ",t", "…", "#$%", " '", "M", "!!", "t", "㍿'", "VE", "<|", "endoftext", "|>", "t", "('", "ſ"]} +{"text": "ع'S,", "tokens": 3, "pieces": ["ع'S", ","]} +{"text": "漢 a字", "tokens": 4, "pieces": ["漢", " a字"]} +{"text": "d'Reé're \n,!!<|fim_prefix|>😀🏽 \n​12345678m \n㍿ \r\n\r\nḍ̇'VE'S‍tZ>å#$%'S'D!,​,#$%\"٣٤٥٦A<漢,", "tokens": 57, "pieces": ["d'Re", "é're", " \n", ",!!<|", "fim", "_prefix", "|>😀🏽", " \n", "​", "123", "456", "78", "m", " \n", "㍿", " \r\n\r\n", "ḍ̇'VE", "'S", "‍t", "Z", ">å", "#$%'", "S'D", "!,​,#$%\"", "٣٤٥", "٦", "A", "<漢", ","]} +{"text": "'Sefi're\t­<|fim_prefix|>‍'Re\u000b🙂!12345678!! \na𐞁'S12345678EOT­A<|endoftext|>㍿'llİeé", "tokens": 51, "pieces": ["'Sefi're", "\t", "­<|", "fim", "_prefix", "|>‍'", "Re", "\u000b", "🙂!", "123", "456", "78", "!!", " \n", "a𐞁'S", "123", "456", "78", "EOT", "­A", "<|", "endoftext", "|>㍿'", "ll", "İeé"]} +{"text": " 𐞁‍३Dž́­!½\r\n\r\nZsA!'T", "tokens": 19, "pieces": [" 𐞁", "‍", "३", "Dž́", "­!", "½", "\r\n\r\n", "Zs", "A", "!'", "T"]} +{"text": "0‍'Re.٣٤٥٦'ſ 's\ta\r½\r\n>ée'Dع\u000b𐞁a'Dİ 0 🙂'D'så漢'D'D३é'M>", "tokens": 46, "pieces": ["0", "‍'", "Re", ".", "٣٤٥", "٦", "'ſ", " '", "s", "\ta", "\r", "½", "\r\n", ">ée'D", "ع", "\u000b𐞁a'D", "İ", " ", " ", "0", " ", "🙂'", "D's", "å漢'D", "'D", "३", "é'M", ">"]} +{"text": "عⅣ,9!!s …ع<
🙂,0 å\tDž👍🏽\r\n\r\nḍ̇ !!३ \n\r\n\r\n𐞁éfi'M", "tokens": 42, "pieces": ["ع", "Ⅳ", ",", "9", "!!", "s", " ", "…ع", "<", "
", "🙂,", "0", " å", "\tDž", "👍🏽\r\n\r\n", "ḍ̇", " ", "!!", "३", " \n\r\n\r\n", "𐞁éfi'M"]} +{"text": "!<|endoftext|>३t\"३,😀🏽\t'D𐞁12345678'½", "tokens": 30, "pieces": ["!<|", "endoftext", "|>", "३", "t", "\"", "३", ",<", "META", "_START", ">😀🏽", "\t", "'D𐞁", "123", "456", "78", "'", "½"]} +{"text": "Dž<|endoftext|>", "tokens": 9, "pieces": ["Dž", "<|", "endoftext", "|>"]} +{"text": "#$%३>
\r\n\r\n<|endoftext|>字٣٤٥٦fifiå\r
ZEOT\rå㋿‍#$%", "tokens": 37, "pieces": ["#$%", "३", ">", "
\r\n\r\n", "<|", "endoftext", "|>", "字", "٣٤٥", "٦", "fifiå", "\r", "
ZEOT", "\r", "å", "㋿‍#$%"]} +{"text": "𐞁㍿s9​…!ſḍ̇'Re.<|endoftext|>(ꟲs \n'll0 …ḍ̇ 'TDžfi<|fim_prefix|>0EOT​🙂½a0'sA\u000b", "tokens": 64, "pieces": ["𐞁", "㍿s", "9", "​", "…", "!ſḍ̇'Re", ".<", "META", "_START", "><|", "endoftext", "|>(", "ꟲs", " \n", "'ll", "0", " ", "…ḍ̇", " ", " '", "TDžfi", "<|", "fim", "_prefix", "|>", "0", "EOT", "​🙂", "½", "a", "0", "'s", "A", "\u000b"]} +{"text": ">09(!!ſADž- 'Sfi​\u000b'D'VE0!!\t'Se'VE's'D12345678''M", "tokens": 39, "pieces": [">", "09", "(!!", "ſ", "ADž", "-", " ", " '", "Sfi", "​", "\u000b", "'D'VE", "0", "!!<", "EOT", ">", "\t", "'Se'VE", "'", "s'D", "123", "456", "78", "''", "M"]} +{"text": "fi0m>-'sé \n\r‍9fi,Z\r\n½é9
㋿'re>'lĺéⅣ", "tokens": 44, "pieces": ["fi", "0", "m", ">-<", "EOT", ">'", "sé", " \n\r", "‍<", "EOT", ">", "9", "fi", ",Z", "\r\n", "½", "é", "", "9", "
", "㋿'", "re", ">'", "lĺé", "Ⅳ", ""]} +{"text": "😀🏽½ \r-\rEOTét#$%é\r'Tع>é İ'D.㍿<|fim_prefix|>½é½ ‍a\"DžDžAⅣ A<'VE𐞁", "tokens": 59, "pieces": ["😀🏽", "½", " \r", "-\r", "EOTét", "#$%", "é", "\r", "'Tع", ">é", " İ'D", ".㍿<|", "fim", "_prefix", "|>", "½", "é", "½", " ‍", "a", "\"DžDžA", "Ⅳ", " ", " A", "<<", "EOT", ">'", "VE𐞁"]} +{"text": "ꟲ'M😀🏽🙂­", "tokens": 13, "pieces": ["ꟲ", "'", "M", "😀🏽🙂­"]} +{"text": "㍿Z㍿åfié'ſZ㋿>'VEdeع \n \nm 👍🏽éå३é", "tokens": 37, "pieces": ["㍿Z", "㍿åfié'ſ", "Z", "㋿>'", "VEdeع", " \n \n", "m", " 👍🏽<", "META", "_START", ">éå", "३", "é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "…''re !m㋿ \n'T9\r\n\r\n<|fim_prefix|>m", "tokens": 21, "pieces": ["…", "''", "re", " ", "!m", "㋿", " \n", "'T", "9", "\r\n\r\n", "<|", "fim", "_prefix", "|>", "m"]} +{"text": "\t9EOT  's9'reİåt\n'D#$%s字>ꟲ", "tokens": 23, "pieces": ["\t", "9", "EOT", " ", " '", "s", "9", "'re", "İåt", "\n", "'D", "#$%", "s字", ">ꟲ"]} +{"text": "'D(mEOT½\u000bß\r\n\r\n,ḍ̇m's'T́>'ſ'D\r\na字0\t(ß'VE\r\u000b 🙂.عs9(", "tokens": 40, "pieces": ["'D", "(m", "EOT", "½", "\u000bß", "\r\n\r\n", ",ḍ̇m's", "'T́", ">'", "ſ'D", "\r\n", "a字", "0", "\t", "(", "ß'VE", "\r", "\u000b", " ", "🙂.", "عs", "9", "("]} +{"text": "å're㋿ḍ̇'reZ㋿́'S漢!!
(Aé'S३\r\n\r\n#$% \n're
'VE#$%fi\n,\u000b", "tokens": 38, "pieces": ["å're", "㋿ḍ̇'re", "Z", "㋿́'S", "漢", "!!", "
", "(Aé'S", "३", "\r\n\r\n", "#$%", " \n", "'re", "
", "'VE", "#$%", "fi", "\n", ",", "\u000b"]} +{"text": "!!é åéİ'Re㋿((!㋿", "tokens": 16, "pieces": ["!!", "é", " ", " åé", "İ'Re", "㋿((!㋿"]} +{"text": "'sDž\n字\r\n\r\nm#$%fi
漢'Ret½\u000bß'Tḍ̇9 ½ éEOT're'ſⅣ字३\tm", "tokens": 37, "pieces": ["'s", "Dž", "\n", "字", "\r\n\r\n", "m", "#$%", "fi", "
漢'Re", "t", "½", "\u000bß'T", "ḍ̇", "9", " ", "½", " ", " é", "EOT're", "'ſ", "Ⅳ", "字", "३", "\tm"]} +{"text": "ع'S\"", "tokens": 7, "pieces": ["ع", "'", "S", "\""]} +{"text": " \nZsⅣ\"sⅣ0é12345678<|fim_prefix|>>", "tokens": 19, "pieces": [" \n", "Zs", "Ⅳ", "\"s", "Ⅳ0", "é", "123", "456", "78", "<|", "fim", "_prefix", "|>>"]} +{"text": " \n
d㍿́12345678ſ'A㋿\" \né#$%\rfi<\r\n\r\n'lle", "tokens": 29, "pieces": [" \n", "
d", "㍿́", "123", "456", "78", "ſ", "'", "A", "㋿\"", " \n", "é", "#$%\r", "fi", "<\r\n\r\n", "'lle"]} +{"text": "'VEmDžd'Re\r\n'Re< ㍿ é ‍
 漢…'TZ t\r'Refi!", "tokens": 37, "pieces": ["'VEm", "Džd'Re", "\r\n", "'Re", "<", " ", " ㍿", " ", " é", " ", " ‍<", "EOT", ">", "
", " 漢", "…", "'TZ", " ", " t", "\r", "'Refi", "!"]} +{"text": "\t\"३!½#$%\"'Sḍ̇𐞁ꟲ… \nDž́ſéⅣ​👍🏽", "tokens": 49, "pieces": ["\t", "\"", "३", "!", "½", "字", "<", "META", "_START", ">#$%<", "META", "_START", ">\"'", "Sḍ̇𐞁ꟲ", "… \n", "Dž́ſé", "Ⅳ", "​👍🏽"]} +{"text": "'S(ß-'ll'T!

½>'VE'Td漢ع'Sd漢'VEA'så\r<|endoftext|>ⅣEOT'S 漢 '<|fim_prefix|>\t", "tokens": 52, "pieces": ["'S", "(ß", "-'", "ll'T", "!", "
", "
", "½", ">'", "VE'T", "d漢ع'S", "d漢'VE", "A's", "å", "\r", "<|", "endoftext", "|><", "META", "_START", ">", "Ⅳ", "EOT'S", " ", " 漢", " ", " '<|", "fim", "_prefix", "|>", "\t"]} +{"text": "🙂́Ⅳ
éßZ字,­漢'ſ漢'T<|fim_prefix|>", "tokens": 23, "pieces": ["🙂́", "Ⅳ", "
éß", "Z字", ",­", "漢'ſ", "漢'T", "<|", "fim", "_prefix", "|>"]} +{"text": "!!\tßß", "tokens": 4, "pieces": ["!!", "\tßß"]} +{"text": "Z!ḍ̇'Sꟲ<#$%a#$%a's'edⅣ\teeſmeİ9'Dm", "tokens": 28, "pieces": ["Z", "!ḍ̇'S", "ꟲ", "<#$%", "a", "#$%", "a's", "'ed", "Ⅳ", "\teeſme", "İ", "9", "'Dm"]} +{"text": "Dž\u000b12345678\"
tꟲ'Mt\"t½", "tokens": 17, "pieces": ["Dž", "\u000b", "123", "456", "78", "\"", "
tꟲ'M", "t", "\"t", "½"]} +{"text": "٣٤٥٦'ſ'M\r0EOT<'re9½<​㍿'ll'lla-s‍<|endoftext|>​tm½a aa½,å\"'D", "tokens": 44, "pieces": ["٣٤٥", "٦", "'ſ'M", "\r", "0", "EOT", "<'", "re", "9½", "<​㍿'", "ll'll", "a", "-s", "‍<|", "endoftext", "|>​", "tm", "½", "a", " aa", "½", ",å", "\"'", "D"]} +{"text": "123456780Afi", "tokens": 9, "pieces": ["", "123", "456", "780", "Afi"]} +{"text": "İ‍d'll\nEOT'Dd<|endoftext|>'S\u000b", "tokens": 18, "pieces": ["İ", "‍d'll", "\n", "EOT'D", "d", "<|", "endoftext", "|>'", "S", "\u000b"]} +{"text": "'M'ReeA12345678!😀🏽, Dž<|endoftext|>'s(\"ſ'\tſ0<|endoftext|>EOT ꟲ#$% \n😀🏽DžDž9عſe<'ReAⅣé", "tokens": 63, "pieces": ["'M'Re", "e", "A", "123", "456", "78", "!😀🏽,", " Dž", "<|", "endoftext", "|>'", "s", "(\"", "ſ", "'", "\tſ", "0", "<|", "endoftext", "|>", "EOT", " ", " ꟲ", "#$%", " \n", "😀🏽", "DžDž", "9", "عſe", "<'", "Re", "A", "Ⅳ", "é"]} +{"text": "Ⅳ'ḍ̇ 👍🏽0ḍ̇12345678EOT<|fim_prefix|>\r\nⅣ😀🏽'VE\"é!!'S\nß 'T.𐞁é
>ḍ̇\u000b<|fim_prefix|>,́s", "tokens": 66, "pieces": ["Ⅳ", "'<", "EOT", ">ḍ̇", " ", "👍🏽", "0", "ḍ̇", "123", "456", "78", "EOT", "<|", "fim", "_prefix", "|>\r\n", "Ⅳ", "😀🏽'", "VE", "\"é", "!!'", "S", "\n", "ß", " ", " '", "T", ".𐞁é", "
", ">ḍ̇", "\u000b", "<|", "fim", "_prefix", "|>,́", "s"]} +{"text": "!Z\r\n½\u000b\u000bsع\n<.<|fim_prefix|>
'VEsDž㋿d𐞁' \n<|fim_prefix|>İ#$%'MZ३ Ⅳ 'VE\r\n\r\nm9  \r\n", "tokens": 56, "pieces": ["!Z", "\r\n", "½", "\u000b", "\u000bsع", "\n", "<.<|", "fim", "_prefix", "|>", "
", "'VEs", "Dž", "㋿d𐞁", "'", " \n", "<|", "fim", "_prefix", "|>", "İ", "#$%'", "MZ", "३", " ", " ", "Ⅳ", " ", " '", "VE", "\r\n\r\n", "m", "9", "  \r\n"]} +{"text": "!åİ㋿('Re\r\n'VEḍ̇t<|endoftext|>\n0İ<9!0(", "tokens": 33, "pieces": ["!å", "İ", "㋿<", "META", "_START", ">('", "Re", "\r\n", "'VEḍ̇t", "<|", "endoftext", "|>\n", "0", "İ", "<", "9", "!", "0", "("]} +{"text": "fi<''Tde,\t… \n<😀🏽ḍ̇Dž👍🏽 é
\tſعḍ̇Ⅳ🙂t's\"㋿ mm(", "tokens": 47, "pieces": ["fi", "<''", "Tde", ",", "\t… \n", "<😀🏽", "ḍ̇", "Dž", "👍🏽", " é", "
", "\tſعḍ̇", "Ⅳ", "🙂t's", "\"㋿", " ", " mm", "("]} +{"text": "'re'Re're ,ß
​㋿'reEOTA!!㋿tDž#$%> \ne9🙂é…9å'red(s\r\n", "tokens": 40, "pieces": ["'re'Re", "'re", " ", " ,", "ß", "
", "​㋿'", "re", "EOTA", "!!㋿", "t", "Dž", "#$%>", " \n", "e", "9", "🙂é", "…", "9", "å're", "d", "(s", "\r\n"]} +{"text": "fia.9㍿.aꟲ 'llß", "tokens": 16, "pieces": ["fia", ".", "9", "㍿.", "aꟲ", " ", " '", "llß"]} +{"text": "\nDžé <|endoftext|>…字 ­e\r\n…<|endoftext|><'s ꟲ\t \t-!…<|fim_prefix|>𐞁 ſ\r\n\r\n'VEſḍ̇dع", "tokens": 59, "pieces": ["\n", "Džé", " <|", "endoftext", "|>", "…字", " ­", "e", "\r\n", "…", "<|", "endoftext", "|><'", "s", " ꟲ", "\t ", "\t", "-!", "…", "<|", "fim", "_prefix", "|>", "𐞁", " ſ", "\r\n\r\n", "'VEſḍ̇dع"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "½,字å<|fim_prefix|>🙂字\u000bé\n'Mß're.ſ½é'D<🙂", "字", "\u000bé", "\n", "'M", "ß're", ".ſ", "½", "é'D", "<<", "d漢"]} +{"text": "'S
 ḍ̇m㋿Ae字́ḍ̇㋿'TŹ>mAé
  <|endoftext|>'VE,å \n'ſ", "tokens": 47, "pieces": ["'S", "
", " ḍ̇m", "㋿Ae字́ḍ̇", "㋿'", "TŹ", ">m", "A", "é", "
 ", " ", "<|", "endoftext", "|>'", "VE", ",å", " \n", "'ſ"]} +{"text": "9ZDž<字é字<Dž½d‍'T!sⅣEOT \n‍\tEOTåa'ſAå٣٤٥٦İ𐞁", "tokens": 43, "pieces": ["9", "ZDž", "<字é字", "<Dž", "½", "d", "‍'", "T", "!s", "Ⅳ", "EOT", " \n", "‍", "\tEOTåa'ſ", "Aå", "٣٤٥", "٦", "İ𐞁"]} +{"text": "'ll>३ \n'12345678#$%\r!d‍́,-t😀🏽\r\n\r\n'D>́👍🏽", "tokens": 31, "pieces": ["'ll", ">", "३", " \n", "'", "123", "456", "78", "#$%\r", "!d", "‍́", ",-", "t", "😀🏽\r\n\r\n", "'D", ">́", "👍🏽"]} +{"text": "12345678'S as­<|fim_prefix|>Dž12345678>es!<|endoftext|>\t", "tokens": 28, "pieces": ["123", "456", "78", "'S", " as", "­<|", "fim", "_prefix", "|>", "Dž", "123", "456", "78", ">es", "!<|", "endoftext", "|>", "\t"]} +{"text": "Z12345678e'S\r\n<|fim_prefix|>é½-", "tokens": 19, "pieces": ["Z", "123", "456", "78", "e'S", "\r\n", "<|", "fim", "_prefix", "|>", "é", "½", "-"]} +{"text": "<|fim_prefix|>\r\nm𐞁\u000b漢߅'s'VE'Dßåß ,  'Re'D'MDž.'ll字'Sꟲ…ع", "tokens": 48, "pieces": ["<|", "fim", "_prefix", "|>\r\n", "m𐞁", "\u000b漢ß", "…", "'s'VE", "'Dß", "åß", " ", ",", " ", " <", "META", "_START", ">'", "Re'D", "'MDž", ".'", "ll字'S", "ꟲ", "…ع"]} +{"text": "'Dḍ̇'D\nfi!'s\r\n\r\nİ́t …,'re\tḍ̇<|endoftext|>", "tokens": 33, "pieces": ["'Dḍ̇'D", "\n", "fi", "!'", "s", "\r\n\r\n", "İ́t", " ", "…", ",'", "re", "\t", "ḍ̇", "<|", "endoftext", "|>"]} +{"text": "İ.,'VE'S
\r( !­ßſ'Sꟲs\tꟲ\u000b-'ſع,Aå", "tokens": 30, "pieces": ["İ", ".,'", "VE'S", "
\r", "(", " ", " !­", "ßſ'S", "ꟲs", "\tꟲ", "\u000b", "-'", "ſع", ",Aå"]} +{"text": "12345678­٣٤٥٦('Re0ſ<…ع🙂\u000btéⅣa're're", "tokens": 33, "pieces": ["123", "456", "78", "­", "٣٤٥", "٦", "('", "Re", "0", "ſ", "<", "…ع", "🙂", "\u000b", "t", "é", "Ⅳ", "a're", "'re"]} +{"text": "ꟲ字>t́㋿\r\n\r\n字👍🏽ds'Tß<|fim_prefix|>🙂 字", "tokens": 30, "pieces": ["ꟲ字", ">t́", "㋿\r\n\r\n", "字", "👍🏽", "ds'T", "ß", "<|", "fim", "_prefix", "|>🙂", " 字"]} +{"text": " a  ​'sꟲdés!!\"'VE<|endoftext|>'re👍🏽EOT'sZ're' ́A字é½𐞁's㋿9㍿ſ", "tokens": 54, "pieces": [" a", " ", " ", "​'", "sꟲ", "dés", "!!\"'", "VE", "<|", "endoftext", "|>'", "re", "👍🏽", "EOT's", "Z're", "'", " ́A字é", "½", "𐞁's", "㋿", "9", "㍿ſ"]} +{"text": "‍\n'sm,fiع́t,'s!!٣٤٥٦'字😀🏽👍🏽‍'re३!
", "tokens": 30, "pieces": ["‍\n", "'sm", ",fiع́t", ",'", "s", "!!", "٣٤٥", "٦", "'字", "😀🏽👍🏽‍'", "re", "३", "!", "
"]} +{"text": " …𐞁­'D­Ⅳ0EOT\r🙂字\r\nß'VE३é<|endoftext|>Z👍🏽😀🏽ß٣٤٥٦d𐞁ꟲ㋿<|fim_prefix|>㍿\n", "tokens": 69, "pieces": [" ", "…𐞁", "­'", "D", "­", "Ⅳ0", "EOT", "\r", "🙂字", "\r\n", "ß'VE", "३", "é", "<|", "endoftext", "|>", "Z", "👍🏽😀🏽", "ß", "٣٤٥", "٦", "d𐞁ꟲ", "㋿<|", "fim", "_prefix", "|>㍿<", "META", "_START", ">\n"]} +{"text": "-\r\n\r\nß🙂're<|fim_prefix|>EOT…ésfi<'ll…ꟲ12345678㋿e>", "tokens": 37, "pieces": ["-\r\n\r\n", "ß", "🙂<", "META", "_START", ">'", "re", "<|", "fim", "_prefix", "|>", "EOT", "…ésfi", "<'", "ll", "…ꟲ", "123", "456", "78", "㋿e", ">"]} +{"text": "字.EOTå \n ½ ‍'re'VE'llß
sſ漢'T'M'Då(#$%<|fim_prefix|><­'re\r\n\r\nZ½ß're<|fim_prefix|>'s é", "tokens": 55, "pieces": ["字", ".EOTå", " \n", " ", " ", "½", " ", "‍'", "re'VE", "'llß", "
sſ漢'T", "'M'D", "å", "(#$%<|", "fim", "_prefix", "|><­'", "re", "\r\n\r\n", "Z", "½", "ß're", "<|", "fim", "_prefix", "|><", "META", "_START", ">'", "s", " é"]} +{"text": "Ⅳ", "tokens": 5, "pieces": ["", "Ⅳ"]} +{"text": " 9ß(😀🏽㋿漢< <|endoftext|>''Mfi𐞁'D\r\n\r\n 12345678\r\n漢å<ع're,\r\nZ", "tokens": 41, "pieces": [" ", "9", "ß", "(😀🏽㋿", "漢", "<", " <|", "endoftext", "|>''", "Mfi𐞁'D", "\r\n\r\n", " ", "123", "456", "78", "\r\n", "漢å", "<ع're", ",\r\n", "Z"]} +{"text": "٣٤٥٦ß!!A\t👍🏽!!e.ꟲ'VE", "tokens": 19, "pieces": ["٣٤٥", "٦", "ß", "!!", "A", "\t", "👍🏽!!", "e", ".ꟲ'VE"]} +{"text": "'D\r\n\r\n'sſ𐞁<|fim_prefix|><|endoftext|>12345678<|endoftext|>é́́İ< ſDž'T \nfiⅣſA>Zꟲ㋿ع\r(>ſ' \n \n12345678٣٤٥٦A", "tokens": 73, "pieces": ["'D", "\r\n\r\n", "'sſ𐞁", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "123", "456", "78", "<|", "endoftext", "|>", "é́́", "İ", "<", " ", " ſ", "Dž'T", " \n", "fi", "Ⅳ", "ſ", "A", "><", "EOT", ">Zꟲ", "㋿ع", "\r", "(>", "ſ", "'", " \n \n", "123", "456", "78٣", "٤٥٦", "A"]} +{"text": "'sḍ̇-é३<|fim_prefix|>'T'sſ'Mé<|fim_prefix|>", "tokens": 25, "pieces": ["'sḍ̇", "-é", "३", "<|", "fim", "_prefix", "|>'", "T's", "ſ'M", "é", "<|", "fim", "_prefix", "|>"]} +{"text": ">漢d\"\nZ're­'S'D,\n", "tokens": 11, "pieces": [">漢d", "\"\n", "Z're", "­'", "S'D", ",\n"]} +{"text": "m 🙂'TAtⅣ\rd'Reſ'VE٣٤٥٦A३㍿'Reİ'Ss å😀🏽́!…\r(\n'M'S'ſ漢Ⅳ 漢…Ⅳ ", "tokens": 55, "pieces": ["m", " ", "🙂'", "TAt", "Ⅳ", "\r", "d'Re", "ſ'VE", "٣٤٥", "٦", "A", "३", "㍿'", "Re", "İ'S", "s", " å", "😀🏽́!", "…\r", "(\n", "'M'S", "'ſ漢", "Ⅳ", " ", " 漢", "…", "Ⅳ", " "]} +{"text": "🙂́ 12345678<|endoftext|>'M(", "tokens": 15, "pieces": ["🙂́", " ", "123", "456", "78", "<|", "endoftext", "|>'", "M", "("]} +{"text": "<'ſ½'ſ're½'VE\"\n
'S,ßꟲ\n!!ſ'ſ 
!'MßéZ", "tokens": 29, "pieces": ["<'", "ſ", "½", "'ſ're", "½", "'VE", "\"\n", "
", "'S", ",ßꟲ", "\n", "!!", "ſ'ſ", " ", "
", "!'", "Mßé", "Z"]} +{"text": "
d́EOT漢!<|fim_prefix|>.½  ३…\"Á'S<|endoftext|>", "tokens": 31, "pieces": ["
d́", "EOT漢", "!<|", "fim", "_prefix", "|>.", "½", "  ", " ", "३", "…", "\"Á'S", "<|", "endoftext", "|>"]} +{"text": "<|endoftext|>0ع👍🏽'Re字ع're🙂\r\n\r\n𐞁ḍ̇,", "tokens": 27, "pieces": ["<|", "endoftext", "|>", "0", "ع", "👍🏽'", "Re字ع're", "🙂\r\n\r\n", "𐞁ḍ̇", ","]} +{"text": "'re\n㋿\u000bs३\r\nḍ̇ꟲ's'llå-ḍ̇A \n​㍿'!'Re'rem㋿\r\n \nḍ̇㍿😀🏽½", "tokens": 49, "pieces": ["'re", "\n", "㋿", "\u000bs", "३", "\r\n", "ḍ̇ꟲ's", "'llå", "-ḍ̇", "A", " \n", "​㍿'!'", "Re're", "m", "㋿\r\n", " \n", "ḍ̇", "㍿😀🏽", "½"]} +{"text": "e🙂fi'S<|endoftext|>
met'M's½å🙂㋿👍🏽'M!, ß<|fim_prefix|>\r\n\r\n<|fim_prefix|> m,", "tokens": 45, "pieces": ["e", "🙂fi'S", "<|", "endoftext", "|>", "
met'M", "'s", "½", "å", "🙂㋿👍🏽'", "M", "!,", " ß", "<|", "fim", "_prefix", "|>\r\n\r\n", "<|", "fim", "_prefix", "|>", " m", ","]} +{"text": "<Aét<|fim_prefix|>İ<|fim_prefix|>,
", "tokens": 17, "pieces": ["<Aét", "<|", "fim", "_prefix", "|>", "İ", "<|", "fim", "_prefix", "|>,", "
"]} +{"text": "٣٤٥٦ß aAİa \t\r\n'Dꟲ🙂 ", "tokens": 21, "pieces": ["٣٤٥", "٦", "ß", " a", "Aİa", "", " \t\r\n", "'Dꟲ", "🙂", " "]} +{"text": "½#$%​é字12345678ß½fi'llDž­Ⅳ㋿(,Dž \n'S'VEd", "tokens": 31, "pieces": ["½", "#$%​", "é字", "123", "456", "78", "ß", "½", "fi'll", "Dž", "­", "Ⅳ", "㋿(,", "Dž", " \n", "'S'VE", "d"]} +{"text": ">'S(#$%'D…Z‍'M㋿\r\n>", "tokens": 17, "pieces": [">'", "S", "(#$%'", "D", "…Z", "‍'", "M", "㋿\r\n", ">"]} +{"text": "aⅣ12345678<|fim_prefix|>>m🙂 !!…'Re< \r\nA​😀🏽#$%-A'Re🙂​'Tt'Re(s٣٤٥٦Z­m", "tokens": 43, "pieces": ["a", "Ⅳ12", "345", "678", "<|", "fim", "_prefix", "|>>", "m", "🙂", " !!", "…", "'Re", "<", " \r\n", "A", "​😀🏽#$%-", "A'Re", "🙂​'", "Tt'Re", "(s", "٣٤٥", "٦", "Z", "­m"]} +{"text": "å.'M'VE
\tEOT漢漢🙂.12345678ع
EOT!!Zß  dع9ſ.<'S漢\n,👍🏽9🙂\u000b", "tokens": 41, "pieces": ["å", ".'", "M'VE", "
", "\tEOT漢漢", "🙂.", "123", "456", "78", "ع", "
EOT", "!!", "Zß", "  ", " dع", "9", "ſ", ".<'", "S漢", "\n", ",👍🏽", "9", "🙂", "\u000b"]} +{"text": "ḍ̇A'ſ<", "tokens": 7, "pieces": ["ḍ̇", "A'ſ", "<"]} +{"text": "㍿\u000b>e㋿#$%'S\r<|endoftext|>fi \ń(<|fim_prefix|>
.İDžs'M'Ret0
'S'MA12345678", "tokens": 46, "pieces": ["㍿", "\u000b", ">e", "㋿#$%'", "S", "\r", "<|", "endoftext", "|>", "fi", " \n", "́", "(<|", "fim", "_prefix", "|>", "
", ".İDžs'M", "'Ret", "0", "
", "'S'M", "A", "123", "456", "78"]} +{"text": "å", "tokens": 2, "pieces": ["å"]} +{"text": " ḍ̇Dž", "tokens": 7, "pieces": [" ", " ḍ̇", "Dž"]} +{"text": "'ſå३ß'MDž'M<|endoftext|>​12345678٣٤٥٦😀🏽 'VE'T'(ع\t(åⅣ", "tokens": 47, "pieces": ["'ſå", "३", "ß'M", "Dž'M", "<|", "endoftext", "|>​", "123", "456", "78", "", "٣٤٥", "٦", "😀🏽", " ", " '", "VE'T", "'(", "ع", "\t", "(å", "Ⅳ"]} +{"text": "EOT𐞁d-!!‍'D👍🏽ß'T漢0👍🏽字12345678'Re(…('T字 å(Z½éḍ̇0#$%'S ٣٤٥٦​", "tokens": 59, "pieces": ["EOT𐞁d", "-<", "META", "_START", ">!!‍'", "D", "👍🏽", "ß'T", "漢", "0", "👍🏽", "字", "123", "456", "78", "'Re", "(", "…", "('", "T字", " ", "å", "(Z", "½", "éḍ̇", "0", "#$%'", "S", " ", " ", "٣٤٥", "٦", "​"]} +{"text": "0'Re\r\n\r\n-mⅣ½a½́ ꟲ…", "tokens": 15, "pieces": ["0", "'Re", "\r\n\r\n", "-m", "Ⅳ½", "a", "½", "́", " ꟲ", "…"]} +{"text": "9EOT٣٤٥٦'ſe'VE٣٤٥٦字é\"'ll", "tokens": 21, "pieces": ["9", "EOT", "٣٤٥", "٦", "'ſe'VE", "٣٤٥", "٦", "字é", "\"'", "ll"]} +{"text": " EOT'ſfi'S-12345678.ꟲfi🙂🙂 'D½12345678\tEOTdm'll\n'!åå-'D", "tokens": 41, "pieces": [" EOT'ſ", "fi'S", "-", "123", "456", "78", ".ꟲfi", "🙂<", "EOT", ">🙂", " '", "D", "½12", "345", "678", "\tEOTdm'll", "\n", "'!", "åå", "-'", "D"]} +{"text": "'VE EOT'Mḍ̇'ſ're𐞁İ'ſ\"'reß \n­0­<|endoftext|>漢sDž 'Re­'VEع𐞁.\n३㍿Zdé'M", "tokens": 63, "pieces": ["'VE", " EOT'M", "ḍ̇", "'", "ſ're", "𐞁", "İ'ſ", "\"'", "re", "ß", " \n", "­", "0", "­<|", "endoftext", "|>", "漢s", "Dž", " ", "'Re", "­'", "VEع𐞁", ".\n", "३", "㍿Zdé'M"]} +{"text": "!!½!!s12345678!!.", "tokens": 8, "pieces": ["!!", "½", "!!", "s", "123", "456", "78", "!!."]} +{"text": "fi㍿t३…Dž12345678é३Z\u000bfi,0­d'ſ‍\r\n\r\nⅣ…fi ", "tokens": 42, "pieces": ["fi", "㍿t", "३", "…Dž", "123", "456", "78", "é", "३", "Z", "\u000bfi", ",", "0", "­<", "EOT", ">d'ſ", "‍\r\n\r\n", "", "Ⅳ", "…fi", " "]} +{"text": "dع'Rem\r'M>'S​'ſAdéDž́ſİ9ḍ̇,\nå-字­- 'Ms'M!!٣٤٥٦12345678३Aé́\t", "tokens": 46, "pieces": ["dع'Re", "m", "\r", "'M", ">'", "S", "​'", "ſ", "Adé", "Dž́ſ", "İ", "9", "ḍ̇", ",\n", "å", "-字", "­-", " '", "Ms'M", "!!", "٣٤٥", "٦12", "345", "678", "३", "Aé́", "\t"]} +{"text": "é🙂\t\n,Asé 🙂'lĺ'Ddß\"字 !!'S'D\u000b'll< \n(DžAå<|endoftext|>0!'re", "tokens": 48, "pieces": ["é", "🙂", "\t\n", ",Asé", " ", "🙂'", "lĺ'D", "dß", "\"字", " ", "!!'", "S'D", "\u000b", "'ll", "<", " \n", "(DžAå", "<|", "endoftext", "|>", "0", "!'", "re"]} +{"text": "é", "tokens": 2, "pieces": ["é"]} +{"text": " '​🙂-字'ſ<|endoftext|>'M㋿\"tt㍿d-ſ'Tꟲ'M½٣٤٥٦\r\n\r\n'ſ \n'llt'reZ٣٤٥٦0d", "tokens": 50, "pieces": [" '​🙂-", "字'ſ", "<|", "endoftext", "|>'", "M", "㋿\"", "tt", "㍿d", "-ſ'T", "ꟲ'M", "½٣٤", "٥٦", "\r\n\r\n", "'ſ", " \n", "'llt're", "Z", "٣٤٥", "٦0", "d"]} +{"text": "\"\nt 'ꟲß", "tokens": 11, "pieces": ["\"\n", "t", " ", "'<", "META", "_START", ">ꟲß"]} +{"text": "12345678'll\t​12345678.\r\n\r\nſ!tßfi‍\"'s< \r!!0é½\"('Dḍ̇et'S.", "tokens": 38, "pieces": ["123", "456", "78", "'ll", "\t", "​", "123", "456", "78", ".\r\n\r\n", "ſ", "!tßfi", "‍\"'", "s", "<", " \r", "!!", "0", "é", "½", "\"('", "Dḍ̇et'S", "."]} +{"text": "-m𐞁å", "tokens": 10, "pieces": ["-m𐞁å", ""]} +{"text": "\r\nZ🙂👍🏽­ \u000b", "tokens": 9, "pieces": ["\r\n", "Z", "🙂👍🏽­", " \u000b"]} +{"text": "\t👍🏽e12345678m'Sé>㋿'MEOT's\t㍿\né́'S\"!!0 ßå!d ,A٣٤٥٦ \n‍\n'Re", "tokens": 51, "pieces": ["\t", "👍🏽", "e", "123", "456", "78", "m'S", "é", ">㋿'", "MEOT's", "\t", "㍿\n", "é́'S", "\"!!", "0", " ", " ßå", "!d", " ", ",A", "٣٤٥", "٦", " \n", "‍\n", "'Re"]} +{"text": "𐞁( 😀🏽\n\r\n\r\n", "tokens": 11, "pieces": ["𐞁", "(", " ", "😀🏽\n\r\n\r\n"]} +{"text": "9tm,­ع-字३. 漢‍  Ⅳ'Re'Re\u000b9Z", "tokens": 22, "pieces": ["9", "tm", ",­", "ع", "-字", "३", ".", " ", " 漢", "‍", " ", " ", "Ⅳ", "'Re'Re", "\u000b", "9", "Z"]} +{"text": "‍👍🏽'M­́İḍ̇ 🙂(mDžA㋿\r\n 9'll ><|endoftext|>ſ\n!!🙂ét9🙂>'VE-9 ", "tokens": 51, "pieces": ["‍👍🏽'", "M", "­́İḍ̇", " ", " 🙂(", "m", "DžA", "㋿\r\n", " ", " ", "9", "'ll", " ><|", "endoftext", "|>", "ſ", "\n", "!!🙂", "ét", "9", "🙂>'", "VE", "-", "9", "", " "]} +{"text": "-0 ḍ̇", "tokens": 6, "pieces": ["-", "0", " ḍ̇"]} +{"text": "é字fifi!!é's👍🏽ſm<|fim_prefix|>(漢s\r\n\r\n
İ३
ݽ's9İaſs'sé'ſ \néİ😀🏽\t", "tokens": 47, "pieces": ["é字fifi", "!!", "é's", "👍🏽", "ſm", "<|", "fim", "_prefix", "|>(", "漢s", "\r\n\r\n", "
İ", "३", "
İ", "½", "'s", "9", "İaſs's", "é'ſ", " \n", "é", "İ", "😀🏽", "\t"]} +{"text": "عs", "tokens": 2, "pieces": ["عs"]} +{"text": "İ's\refi At's­9😀🏽åⅣ\r\n'd‍", "tokens": 21, "pieces": ["İ's", "\r", "efi", " At's", "­", "9", "😀🏽", "å", "Ⅳ", "\r\n", "'d", "‍"]} +{"text": "<\r漢\u000ba\"½t'T𐞁字0㋿\t३😀🏽'Re'T", "tokens": 26, "pieces": ["<\r", "漢", "\u000ba", "\"", "½", "t'T", "𐞁字", "0", "㋿", "\t", "३", "😀🏽'", "Re'T"]} +{"text": "👍🏽", "tokens": 3, "pieces": ["👍🏽"]} +{"text": " å👍🏽𐞁\r\n'' 0e!!'D🙂
", "tokens": 20, "pieces": [" å", "👍🏽", "𐞁", "\r\n", "''", " ", "0", "e", "!!'", "D", "🙂", "
"]} +{"text": "12345678Z", "tokens": 4, "pieces": ["123", "456", "78", "Z"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "fi…123456780Z0'M", "tokens": 9, "pieces": ["fi", "…", "123", "456", "780", "Z", "0", "'M"]} +{"text": "'ſ 'VE,\t३Ⅳ😀🏽", "tokens": 13, "pieces": ["'ſ", " ", " '", "VE", ",", "\t", "३Ⅳ", "😀🏽"]} +{"text": "字­'漢ꟲ\r\n­‍'S('S\"́sa\u000bßßt#$% <|fim_prefix|>👍🏽 ḍ̇t𐞁ḍ̇ …!ß𐞁㍿", "tokens": 61, "pieces": ["字", "­'", "漢ꟲ", "\r\n", "­‍'", "S", "('", "S", "\"́sa", "\u000b", "ßßt", "#$%", " <|", "fim", "_prefix", "|><", "EOT", ">👍🏽", " ḍ̇t𐞁ḍ̇", " ", "…", "!ß𐞁", "㍿"]} +{"text": "ſ́😀🏽㋿0'T­漢ع‍\t d12345678'VE\tⅣſ🙂㋿dDž't!!", "tokens": 37, "pieces": ["ſ́", "😀🏽㋿", "0", "'T", "­漢", "ع", "‍", "\t", " d", "123", "456", "78", "'VE", "\t", "Ⅳ", "ſ", "🙂㋿", "d", "Dž't", "!!"]} +{"text": "t…عꟲß́
👍🏽\u000b'MAdſ'VE'D", "tokens": 21, "pieces": ["t", "…عꟲß́", "
", "👍🏽", "\u000b", "'MAdſ'VE", "'D"]} +{"text": "İß漢e㋿ ३…<|endoftext|>9Á", "tokens": 22, "pieces": ["İß漢e", "㋿", " ", " ", "३", "…", "<|", "endoftext", "|>", "9", "Á"]} +{"text": "ß'ret(ßt'Re👍🏽0,\u000b!!s\rA'​𐞁​字ꟲع\u000b", "tokens": 36, "pieces": ["ß're", "t", "(ßt'Re", "👍🏽", "0", ",", "\u000b", "!!", "s", "\r", "A", "'​", "𐞁", "​字ꟲع", "", "\u000b"]} +{"text": "٣٤٥٦A'T\t  mꟲ9 >'T𐞁AaA\r\n>👍🏽'S'Dm
​", "tokens": 41, "pieces": ["٣٤٥", "٦", "A'T", "\t ", " mꟲ", "9", "", " >'", "T𐞁Aa", "A", "\r\n", ">👍🏽'", "S'D", "m", "
", "​<", "EOT", ">"]} +{"text": "Džḍ̇㍿(<|fim_prefix|>ḍ̇½!!\n('Re<|fim_prefix|>👍🏽 \n…字\n12345678A㋿ḍ̇d éİEOTt.…ſ\u000bå're​9", "tokens": 64, "pieces": ["Džḍ̇", "㍿(<|", "fim", "_prefix", "|>", "ḍ̇", "½", "!!\n", "('", "Re", "<|", "fim", "_prefix", "|>👍🏽", " \n", "…字", "\n", "123", "456", "78", "A", "㋿ḍ̇d", " é", "İEOTt", ".", "…ſ", "\u000bå're", "​", "9"]} +{"text": "\r\n\r\nⅣſ'redé,é \n<|endoftext|>ßſ\rꟲA,\n\r👍🏽(👍🏽😀🏽<|endoftext|>\"-\n𐞁…9", "tokens": 51, "pieces": ["\r\n\r\n", "Ⅳ", "ſ're", "dé", ",é", " \n", "<|", "endoftext", "|>", "ßſ", "\r", "ꟲ", "A", ",\n\r", "👍🏽(👍🏽😀🏽<|", "endoftext", "|>\"-\n", "𐞁", "…", "9"]} +{"text": "A0's", "tokens": 3, "pieces": ["A", "0", "'s"]} +{"text": "😀🏽Z\n'Tſfi\r\n'll>.'SßéEOT0 漢…😀🏽́Ⅳ́𐞁AéEOT\u000bt", "tokens": 39, "pieces": ["😀🏽", "Z", "\n", "'Tſfi", "\r\n", "'ll", ">.'", "Sßé", "EOT", "0", " 漢", "…", "😀🏽́", "Ⅳ", "́𐞁Aé", "EOT", "\u000bt"]} +{"text": "🙂漢!!İ🙂٣٤٥٦EOTß\r\n\t
#$%å0>-👍🏽\ŕ𐞁𐞁३İ", "tokens": 36, "pieces": ["🙂漢", "!!", "İ", "🙂", "٣٤٥", "٦", "EOTß", "\r\n", "\t", "
", "#$%", "å", "0", ">-👍🏽\r", "́𐞁𐞁", "३", "İ"]} +{"text": "Z'T're ㋿​½­‍å \n< <'T \nmm½", "tokens": 24, "pieces": ["Z'T", "'re", " ", "㋿​", "½", "­‍", "å", " \n", "<", " ", "<'", "T", " \n", "mm", "", "½"]} +{"text": "'VE​㍿­½<|endoftext|>", "tokens": 19, "pieces": ["'VE", "​㍿­<", "META", "_START", ">", "½", "<|", "endoftext", "|>"]} +{"text": "\r\nå\r ٣٤٥٦!!'ll\nsß're'ſ'S9sZ🙂\r\nß­s", "tokens": 30, "pieces": ["\r\n", "å", "\r", " ", " ", "٣٤٥", "٦", "!!'", "ll", "\n", "s", "ß're", "'ſ'S", "9", "s", "Z", "🙂\r\n", "ß", "­s"]} +{"text": "'VE🙂", "tokens": 3, "pieces": ["'VE", "🙂"]} +{"text": "é\u000bDž‍s­!", "tokens": 9, "pieces": ["é", "\u000bDž", "‍s", "­!"]} +{"text": "'SZ#$%Z½m🙂ſ字ع字\råeſ'll😀🏽å", "tokens": 23, "pieces": ["'SZ", "#$%", "Z", "½", "m", "🙂ſ字ع字", "\r", "åeſ'll", "😀🏽", "å"]} +{"text": "ḍ̇'T<|endoftext|>
9<|fim_prefix|>漢㋿‍0ß🙂 a\u000bA३", "tokens": 32, "pieces": ["ḍ̇'T", "<|", "endoftext", "|>", "
", "9", "<|", "fim", "_prefix", "|>", "漢", "㋿‍", "0", "ß", "🙂", " a", "\u000bA", "३"]} +{"text": "'ll-'ll0'Re字ZꟲⅣ漢👍🏽\r\n\r\n😀🏽\u000bZ字,dß", "tokens": 25, "pieces": ["'ll", "-'", "ll", "0", "'Re字", "Zꟲ", "Ⅳ", "漢", "👍🏽\r\n\r\n", "😀🏽", "\u000bZ字", ",dß"]} +{"text": "\r\n\r\nſDžZ're \nⅣ9A'ſ\r\n\r\n'VÉé٣٤٥٦𐞁ém!<'Re #$%EOT!!", "tokens": 39, "pieces": ["\r\n\r\n", "ſ", "DžZ're", " \n", "Ⅳ9", "A'ſ", "\r\n\r\n", "'VÉé", "٣٤٥", "٦", "𐞁ém", "!<'", "Re", " #$%", "EOT", "!!"]} +{"text": "'S<|fim_prefix|>ꟲ😀🏽s d'VE漢<|endoftext|>'llDžſ😀🏽", "tokens": 38, "pieces": ["'S", "<|", "fim", "_prefix", "|>", "ꟲ", "😀🏽", "s", " d'VE", "漢", "<|", "endoftext", "|>'", "ll", "Džſ", "😀🏽"]} +{"text": "'Sde㍿漢>🙂Dž!!<👍🏽 \n½ḍ̇'ll­fi12345678\"é<|fim_prefix|>", "tokens": 34, "pieces": ["'Sde", "㍿漢", ">🙂", "Dž", "!!<👍🏽", " \n", "½", "ḍ̇'ll", "­fi", "123", "456", "78", "\"é", "<|", "fim", "_prefix", "|>"]} +{"text": "\"\r\n'VE<|fim_prefix|>🙂\"", "tokens": 11, "pieces": ["\"\r\n", "'VE", "<|", "fim", "_prefix", "|>🙂\""]} +{"text": " ,<​ß'T<\r\n\r\nåeß' é<|endoftext|>>'ll-ß's㋿'VEß‍<|fim_prefix|>İt…-12345678'>٣٤٥٦ ß", "tokens": 54, "pieces": [" ,<​", "ß'T", "<\r\n\r\n", "åeß", "'", " é", "<|", "endoftext", "|>>'", "ll", "-ß's", "㋿'", "VEß", "‍<|", "fim", "_prefix", "|>", "İt", "…", "-", "123", "456", "78", "'>", "٣٤٥", "٦", " ß"]} +{"text": "t​!12345678'll😀🏽\r\n", "tokens": 11, "pieces": ["t", "​!", "123", "456", "78", "'ll", "😀🏽\r\n"]} +{"text": ".é ,<|fim_prefix|>ḍ̇\rEOT'Ret​åé\t'ſt㋿ſⅣ🙂漢e\r\n\r\n'SAa12345678 !­", "tokens": 45, "pieces": [".é", " ", ",<|", "fim", "_prefix", "|>", "ḍ̇", "\r", "EOT'Re", "t", "​åé", "\t", "'ſt", "㋿ſ", "Ⅳ", "🙂漢e", "\r\n\r\n", "'SAa", "123", "456", "78", " ", "!­"]} +{"text": " ́  're12345678dm", "tokens": 9, "pieces": [" ́", " ", " ", "'re", "123", "456", "78", "dm"]} +{"text": "ſéſ9́EOTEOT!!\" ßİ!!fia é!…s字e", "tokens": 26, "pieces": ["ſéſ", "9", "́", "EOTEOT", "!!\"", " ", " ß", "İ", "!!", "fia", " é", "!", "…s字e"]} +{"text": "३'s\r'M
㍿,\r\n\r\nعEOT 'll,'re\r\n'M'ſİ!", "tokens": 26, "pieces": ["३", "'s", "\r", "'M", "
", "㍿,\r\n\r\n", "ع", "EOT", " ", "'ll", ",'", "re", "\r\n", "'M'ſ", "İ", "!<", "EOT", ">"]} +{"text": "'ſm'S'llſ'Re漢éꟲ9", "tokens": 16, "pieces": ["'ſm'S", "'llſ'Re", "漢é", "ꟲ", "9"]} +{"text": "-.ꟲſ'\r'll'VE
å", "tokens": 13, "pieces": ["-.", "ꟲſ", "'\r", "'ll'VE", "
å"]} +{"text": "'S٣٤٥٦é0A🙂é<|endoftext|>s\r\n\r\n", "tokens": 23, "pieces": ["'S", "٣٤٥", "٦", "é", "0", "A", "🙂é", "<|", "endoftext", "|><", "EOT", ">s", "\r\n\r\n"]} +{"text": "\u000b", "tokens": 1, "pieces": ["\u000b"]} +{"text": ".\"-ßⅣEOT🙂aⅣé
", "tokens": 14, "pieces": [".\"-", "ß", "Ⅳ", "EOT", "🙂a", "Ⅳ", "é", "
"]} +{"text": "½\u000b🙂‍0!a ३9\r\n😀🏽߅.'S", "tokens": 19, "pieces": ["½", "\u000b", "🙂‍", "0", "!a", " ", "३9", "\r\n", "😀🏽", "ß", "…", ".'", "S"]} +{"text": "漢Ⅳḍ̇A((\r\r\n .<|endoftext|>
ſ'Re३٣٤٥٦ꟲm", "tokens": 29, "pieces": ["漢", "Ⅳ", "ḍ̇", "A", "((\r\r\n", " ", ".<|", "endoftext", "|>", "
ſ'Re", "३٣٤", "٥٦", "ꟲm"]} +{"text": "!!٣٤٥٦\r\n\n", "tokens": 6, "pieces": ["!!", "٣٤٥", "٦", "\r\n\n"]} +{"text": "٣٤٥٦字İ!!m'T…eé", "tokens": 17, "pieces": ["٣٤٥", "٦", "字", "İ", "!!", "m'T", "…eé", ""]} +{"text": ". 😀🏽d(ÁZ \nA🙂ß'Dİd \n½㍿é‍", "tokens": 24, "pieces": [".", " ", "😀🏽", "d", "(Á", "Z", " \n", "A", "🙂ß'D", "İd", " \n", "½", "㍿é", "‍"]} +{"text": "\r\n\r\n​Dž😀🏽s<|endoftext|>½0½½'re…", "tokens": 26, "pieces": ["\r\n\r\n", "​Dž", "😀🏽", "s", "<|", "endoftext", "|>", "½0", "", "½½", "'re", "…"]} +{"text": "Z'Dꟲ‍åéḍ̇mع\r", "tokens": 18, "pieces": ["Z'D", "ꟲ", "‍åéḍ̇mع", "\r"]} +{"text": "­>'S'ſ!!étß'll字𐞁m>ꟲ漢\nEOT٣٤٥٦'ll12345678(fi​9å0'D🙂́ Am", "tokens": 47, "pieces": ["­>'", "S'ſ", "!!", "étß'll", "字𐞁m", ">ꟲ漢", "\n", "EOT", "٣٤٥", "٦", "'ll", "123", "456", "78", "(fi", "​", "9", "å", "0", "'D", "🙂́", " ", " Am"]} +{"text": "s,'ll
…0's9é㍿é३🙂́Džſ0\r\n\r\n'sع'M 
Z!!ḍ̇", "tokens": 38, "pieces": ["s", ",'", "ll", "
", "…", "0", "'s", "9", "é", "㍿é", "३", "🙂́Džſ", "0", "\r\n\r\n", "'s", "ع'M", " ", "
Z", "!!", "ḍ̇"]} +{"text": "é ,İ<|endoftext|>عéé𐞁👍🏽​ſa👍🏽ZZ𐞁\u000ba0 ­9½0\t​\u000bfi''re'Re<|fim_prefix|>​\u000bå‍(<", "tokens": 60, "pieces": ["é", " ", ",İ", "<|", "endoftext", "|>", "عéé𐞁", "👍🏽​", "ſa", "👍🏽", "ZZ𐞁", "\u000ba", "0", " ", "­", "9½0", "\t", "​", "\u000bfi", "''", "re'Re", "<|", "fim", "_prefix", "|>​", "\u000bå", "‍(<"]} +{"text": "'Tḍ̇½ \n''Re<'re‍½.e12345678'D́é 𐞁fi🙂12345678\t٣٤٥٦fi漢'ſḍ̇12345678'D𐞁 漢é٣٤٥٦", "tokens": 63, "pieces": ["'Tḍ̇", "½", " \n", "''", "Re", "<'", "re", "‍", "½", ".e", "123", "456", "78", "'D́é", " ", " 𐞁fi", "🙂", "123", "456", "78", "\t", "٣٤٥", "٦", "fi漢'ſ", "ḍ̇", "", "123", "456", "78", "'D𐞁", " 漢é", "٣٤٥", "٦"]} +{"text": " s, ­
fi're \nEOT ꟲ३字é漢\"\"!😀🏽ع
s‍méݽ 12345678‍㍿ſ.", "tokens": 46, "pieces": [" ", " s", ",", " ", "­", "
fi're", " \n", "EOT", " ꟲ", "३", "字", "é漢", "\"\"!😀🏽", "ع", "
s", "‍mé", "İ", "½", " ", " ", "123", "456", "78", "‍㍿", "ſ", "."]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " s's…å,ſa\"t½𐞁're\rEOT½'llⅣfi\">\"A½İ-३t", "tokens": 40, "pieces": [" ", " s's", "…å", ",ſa", "\"t", "½", "𐞁're", "\r", "EOT", "½", "'ll", "Ⅳ", "fi", "\">\"", "A", "½", "İ", "-<", "META", "_START", ">", "३", "t"]} +{"text": "\r\n\r\n<,9Ⅳ३'Re'Re0\n12345678३ḍ̇\u000b\r\n t.\t<|endoftext|>A\r\n\r\n \n\ŕfi'Ss<|fim_prefix|>\r\n\r\n\n!!‍", "tokens": 47, "pieces": ["\r\n\r\n", "<,", "9Ⅳ३", "'Re'Re", "0", "\n", "123", "456", "78३", "ḍ̇", "\u000b\r\n", " t", ".", "\t", "<|", "endoftext", "|>", "A", "\r\n\r\n \n\r", "́fi'S", "s", "<|", "fim", "_prefix", "|>\r\n\r\n\n", "!!‍"]} +{"text": "Ⅳ́é \n​Ⅳع́ ", "tokens": 11, "pieces": ["Ⅳ", "́é", " \n", "​", "Ⅳ", "ع́", " "]} +{"text": ">ꟲ\r\n👍🏽٣٤٥٦\"!é​
's#$%½'Re𐞁\r\n\r\né,", "tokens": 32, "pieces": [">ꟲ", "\r\n", "👍🏽", "٣٤٥", "٦", "\"!", "é", "​<", "META", "_START", ">", "
", "'s", "#$%", "½", "'Re𐞁", "\r\n\r\n", "é", ","]} +{"text": "İ's👍🏽<|fim_prefix|><|fim_prefix|>'ll#$%'saḍ̇
ſ12345678 'S'VE\n", "tokens": 33, "pieces": ["İ's", "👍🏽<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>'", "ll", "#$%'", "saḍ̇", "
ſ", "123", "456", "78", " '", "S'VE", "\n"]} +{"text": "३\r\nꟲåéDž
(㋿éfiꟲEOT-a'VE \"'DéⅣ\n'\n
.A!!\" \n👍🏽 🙂…é\r\n\r\n\n<|endoftext|>", "tokens": 57, "pieces": ["३", "\r\n", "ꟲåé", "Dž", "
", "(㋿", "éfiꟲ", "EOT", "-a'VE", " ", "\"'", "Dé", "Ⅳ", "\n", "'\n", "
", ".A", "!!\"", " \n", "👍🏽", " 🙂", "…é", "\r\n\r\n\n", "<|", "endoftext", "|>"]} +{"text": "ß𐞁½,!!­½-😀🏽\r\n\r\n'ſ ‍#$%(,Z,\r\n,tt\u000b'ſ#$%", "tokens": 37, "pieces": ["ß𐞁", "½", ",!!­", "½", "-😀🏽\r\n\r\n", "'ſ", " ", "‍#$%(<", "META", "_START", "><", "EOT", ">,", "Z", ",\r\n", ",tt", "\u000b", "'ſ", "#$%"]} +{"text": "🙂३'ſ\rꟲ", "tokens": 8, "pieces": ["🙂", "३", "'ſ", "\r", "ꟲ"]} +{"text": "ع,'S", "tokens": 3, "pieces": ["ع", ",'", "S"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'VE'VE٣٤٥٦㋿İ'Re‍İ\r\n'VE'ss", "tokens": 20, "pieces": ["'VE'VE", "٣٤٥", "٦", "㋿İ'Re", "‍İ", "\r\n", "'VE's", "s"]} +{"text": " 'M", "tokens": 2, "pieces": [" '", "M"]} +{"text": "👍🏽 \nع><|fim_prefix|>(㍿", "tokens": 14, "pieces": ["👍🏽", " \n", "ع", "><|", "fim", "_prefix", "|>(㍿"]} +{"text": "½Z", "tokens": 2, "pieces": ["½", "Z"]} +{"text": "'s\r'T\r\nfi\tésEOT \nEOT​Z 12345678ع \né👍🏽…!!'llZ㋿́٣٤٥٦", "tokens": 39, "pieces": ["'s", "\r", "'T", "\r\n", "fi", "\tés", "EOT", " \n", "EOT", "​Z", " ", "123", "456", "78", "ع", " \n", "é", "👍🏽", "…", "!!'", "ll", "Z", "㋿́", "٣٤٥", "٦"]} +{"text": "(字'reſd", "tokens": 5, "pieces": ["(字're", "ſd"]} +{"text": "'s…EOT\r\nع \nſås(m#$%㍿'D", "tokens": 20, "pieces": ["'s", "…EOT", "\r\n", "ع", " \n", "ſås", "(m", "#$%㍿'", "D"]} +{"text": "
('VEꟲ㋿12345678 \n😀🏽'Rea\u000b \nⅣ\u000b漢'S \u000bß٣٤٥٦…३m'VE<|fim_prefix|>", "tokens": 45, "pieces": ["
", "('", "VEꟲ", "㋿", "123", "456", "78", " \n", "😀🏽'", "Rea", "\u000b \n", "Ⅳ", "\u000b漢'S", " ", "\u000bß", "٣٤٥", "٦", "…", "३", "m'VE", "<|", "fim", "_prefix", "|>"]} +{"text": "ع\t½0fiß<|endoftext|>-'D٣٤٥٦!", "tokens": 23, "pieces": ["ع", "\t", "½0", "fiß", "<|", "endoftext", "|>-'", "D", "٣٤٥", "٦", "!"]} +{"text": "㍿字tꟲ 'D 'ſéea\n३tDžع字'12345678漢0'S\"", "tokens": 31, "pieces": ["㍿字tꟲ", " ", "'D", " ", "'ſéea", "\n", "३", "t", "Džع字", "'", "123", "456", "78", "漢", "0", "'S", "\""]} +{"text": "ſ字'sⅣsⅣZ 12345678're½0'T>👍🏽­t'Séé́३,A㍿\r\n>é EOT \n…'VE'\r\n\r\n‍😀🏽", "tokens": 53, "pieces": ["ſ字's", "Ⅳ", "s", "Ⅳ", "Z", " ", " ", "123", "456", "78", "'re", "½0", "'T", ">👍🏽­", "t'S", "éé́", "३", ",A", "㍿\r\n", ">é", " ", " EOT", " \n", "…", "'VE", "'\r\n\r\n", "‍😀🏽"]} +{"text": "३!! \n\r\n \n🙂'D-😀🏽fi\rع12345678'.👍🏽३‍\té👍🏽0ꟲ㋿ſå", "tokens": 43, "pieces": ["३", "!!", " \n\r\n \n", "🙂'", "D", "-😀🏽", "fi", "\r", "ع", "123", "456", "78", "'.<", "META", "_START", ">👍🏽", "३", "‍", "\té", "👍🏽", "0", "ꟲ", "㋿ſå"]} +{"text": "漢𐞁é", "tokens": 6, "pieces": ["漢𐞁é"]} +{"text": "ſ'Mfi fi'ré'Reß'Re!", "tokens": 11, "pieces": ["ſ'M", "fi", " ", " fi're", "́'Re", "ß'Re", "!"]} +{"text": "'T\rfi­
\r\n\u000bEOTDžfi \n fis\r½'Re,
٣٤٥٦'T​(ꟲfi ½… ", "tokens": 37, "pieces": ["'T", "\r", "fi", "­", "
\r\n", "\u000bEOTDžfi", " \n", " fis", "\r", "½", "'Re", ",", "
", "٣٤٥", "٦", "'T", "​(", "ꟲfi", " ", "½", "… "]} +{"text": "㍿👍🏽d!\t\t́12345678", "tokens": 14, "pieces": ["㍿👍🏽", "d", "!", "\t", "\t́", "123", "456", "78"]} +{"text": "'D😀🏽'M𐞁A<ꟲꟲع'T,.㍿'T\n'll<|fim_prefix|>'D<|fim_prefix|>12345678e's\tꟲ'\u000b😀🏽İ(\"", "tokens": 60, "pieces": ["'D", "😀🏽'", "M𐞁", "A", "<ꟲꟲع'T", ",.㍿<", "META", "_START", ">'", "T", "\n", "'ll", "<|", "fim", "_prefix", "|>'", "D", "<|", "fim", "_prefix", "|>", "123", "456", "78", "e's", "\tꟲ", "'", "\u000b", "😀🏽", "İ", "(\""]} +{"text": "9'D'ſ𐞁\r\n\r\n'M字㋿å👍🏽sm'll\rfi(😀🏽'ſ𐞁#$%", "tokens": 35, "pieces": ["9", "'D'ſ", "𐞁", "\r\n\r\n", "'M字", "㋿å", "👍🏽", "sm'll", "\r", "fi", "(😀🏽'", "ſ𐞁", "#$%"]} +{"text": " ­'ll's!", "tokens": 5, "pieces": [" ­'", "ll's", "!"]} +{"text": "Džİ 'SA​0 …dꟲ'VE9'Re\r\n\r\nEOTꟲḍ̇ mmå-\"‍字ⅣA", "tokens": 39, "pieces": ["Džİ", " ", "'SA", "​", "0", " ", "…dꟲ'VE", "9", "'Re", "\r\n\r\n", "EOTꟲḍ̇", " mmå", "-\"‍", "字", "Ⅳ", "A"]} +{"text": "-'M.d(", "tokens": 4, "pieces": ["-'", "M", ".d", "("]} +{"text": "\t", "tokens": 1, "pieces": ["\t"]} +{"text": "\n👍🏽'Rem<|fim_prefix|>'M字e'Reée'D!!'Re-", "tokens": 22, "pieces": ["\n", "👍🏽'", "Rem", "<|", "fim", "_prefix", "|>'", "M字e'Re", "ée'D", "!!'", "Re", "-"]} +{"text": "\t#$%Ⅳ'M\r\n İſ ​'ll\u000b'reDžḍ̇e're(ßZ字㍿ſ<㋿<|fim_prefix|><|endoftext|>㋿ß12345678'", "tokens": 58, "pieces": ["\t", "#$%", "Ⅳ", "'M", "\r\n", " İſ", " ", "​'", "ll", "\u000b", "'re", "Džḍ̇e're", "(ß", "Z字", "㍿ſ", "<㋿<", "META", "_START", "><|", "fim", "_prefix", "|><|", "endoftext", "|>㋿", "ß", "123", "456", "78", "'"]} +{"text": "'ſ'D<|endoftext|>sfi'T<é \r\n\r\ndm\"½Dž́'T\t㋿EOTDž fi", "tokens": 36, "pieces": ["'ſ'D", "<|", "endoftext", "|>", "sfi'T", "<é", " \r\n\r\n", "dm", "\"", "½", "Dž́'T", "", "\t", "㋿EOTDž", " fi"]} +{"text": "'ReⅣs'VE🙂'D", "tokens": 9, "pieces": ["'Re", "Ⅳ", "s'VE", "🙂'", "D"]} +{"text": " å\u000bع\"><'VEé👍🏽 ", "tokens": 13, "pieces": [" ", " å", "\u000bع", "\"><'", "VEé", "👍🏽", " "]} +{"text": "t!!'ſ!!!\td½ſ,<|fim_prefix|>'s'D12345678>Z…́Džſ\u000bé𐞁\"'T'Re​'D", "tokens": 45, "pieces": ["t", "!!<", "EOT", ">'", "ſ", "!!!", "\td", "½", "ſ", ",<|", "fim", "_prefix", "|>'", "s'D", "123", "456", "78", ">Z", "…́Džſ", "\u000bé𐞁", "\"'", "T", "'", "Re", "​'", "D"]} +{"text": "Ⅳ…eé,'ſ'Re\r\nd😀🏽-\nß0Z(👍🏽'S12345678m½\"㍿㍿å٣٤٥٦​<|endoftext|>\te \nZ", "tokens": 56, "pieces": ["Ⅳ", "…eé", ",'", "ſ'Re", "\r\n", "d", "😀🏽-\n", "ß", "0", "Z", "(👍🏽'", "S", "123", "456", "78", "m", "½", "\"㍿㍿", "å", "٣٤٥", "٦", "​<|", "endoftext", "|>", "\te", " \n", "Z"]} +{"text": "'ll\r\nꟲd", "tokens": 6, "pieces": ["'ll", "\r\n", "ꟲd"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿‍s#$%\t\r\n\r\na<|endoftext|>٣٤٥٦åع\n 0㋿\r\n<​Ⅳ,​9!!'<😀🏽 \n", "tokens": 44, "pieces": ["㍿‍", "s", "#$%", "\t\r\n\r\n", "a", "<|", "endoftext", "|>", "٣٤٥", "٦", "åع", "\n", " ", " ", "0", "㋿\r\n", "<​", "Ⅳ", ",​", "9", "!!'<😀🏽", " \n"]} +{"text": "''Re 12345678漢\ntḍ̇0'DEOTEOT'VEꟲ-ꟲ­", "tokens": 27, "pieces": ["''", "Re", " ", "123", "456", "78", "漢", "\n", "tḍ̇", "0", "'DEOTEOT'VE", "ꟲ", "-ꟲ", "­"]} +{"text": "0", "tokens": 4, "pieces": ["", "0"]} +{"text": "<ß're'S🙂'Reſ𐞁 0 ㋿㍿0\r\n𐞁\u000bé>👍🏽 
 \nſfi\t\t́Z'Re'D", "tokens": 45, "pieces": ["<ß're", "'S", "🙂'", "Reſ𐞁", " ", "0", " ㋿㍿", "0", "\r\n", "𐞁", "\u000bé", ">👍🏽", " 
 \n", "ſfi", "\t", "\t́", "Z'Re", "'D"]} +{"text": "'Re\t#$%<|endoftext|>é\u000b'ḍ̇,tA‍éAß'T9're 're,㍿'D​𐞁a'D\"fi३", "tokens": 45, "pieces": ["'Re", "\t", "#$%<|", "endoftext", "|>", "é", "\u000b", "'ḍ̇", ",t", "A", "‍é", "Aß'T", "9", "'re", " ", " '", "re", ",㍿'", "D", "​𐞁a'D", "\"fi", "३"]} +{"text": "'S\r\nZ\rd!!  ​\t ", "tokens": 11, "pieces": ["'S", "\r\n", "Z", "\r", "d", "!!", " ", " ", "​", "\t "]} +{"text": "𐞁'ſt'Re", "tokens": 8, "pieces": ["𐞁'ſ", "t'Re"]} +{"text": "éd0 \n'ſ \nⅣİ12345678e,Dž漢me字😀🏽", "tokens": 24, "pieces": ["éd", "0", " \n", "'ſ", " \n", "Ⅳ", "İ", "123", "456", "78", "e", ",Dž漢me字", "😀🏽"]} +{"text": " !!<|endoftext|>'VE'T 字'", "tokens": 25, "pieces": [" ", "!!<|", "endoftext", "|>'", "VE", "'", "T", "", " 字", "'<", "META", "_START", ">"]} +{"text": "'s", "tokens": 1, "pieces": ["'s"]} +{"text": "漢!!'VE\t<|fim_prefix|>'Reem㍿\r\n🙂A'Me٣٤٥٦ꟲ\u000bİ👍🏽३ 👍🏽 'T", "tokens": 44, "pieces": ["漢", "!!'", "VE", "\t", "<|", "fim", "_prefix", "|>'", "Reem", "㍿\r\n", "🙂<", "META", "_START", ">A'M", "e", "٣٤٥", "٦", "ꟲ", "\u000bİ", "👍🏽", "३", " ", " 👍🏽", " ", "'T"]} +{"text": "'s!!‍'Re'llt-‍​-s12345678 9ḍ̇", "tokens": 25, "pieces": ["'s", "!!‍'", "Re'll", "t", "-‍​-", "s", "123", "456", "78", " ", " ", "9", "ḍ̇"]} +{"text": "é'VE!s\"Dž字s𐞁e\r\n\r\nZé's", "tokens": 26, "pieces": ["é'VE", "!<", "EOT", "><", "META", "_START", ">s", "\"Dž字s𐞁e", "\r\n\r\n", "Zé's"]} +{"text": " \tⅣ<|fim_prefix|>\n12345678🙂<|fim_prefix|>'Sé'Re<|fim_prefix|> ع\r\nfi𐞁'T'D𐞁́ꟲ…Ad fi's<9s<|endoftext|>😀🏽Ⅳ٣٤٥٦12345678", "tokens": 76, "pieces": [" ", "\t", "Ⅳ", "<|", "fim", "_prefix", "|>\n", "123", "456", "78", "🙂<|", "fim", "_prefix", "|>'", "Sé'Re", "<|", "fim", "_prefix", "|>", " ع", "\r\n", "fi𐞁'T", "'D𐞁́ꟲ", "…Ad", " fi's", "<", "9", "s", "<|", "endoftext", "|>😀🏽", "Ⅳ٣٤", "٥٦1", "234", "567", "8"]} +{"text": " 9👍🏽\r㋿>d㍿!\u000b㋿\n'VEdét\u000b>éé👍🏽३ ३😀🏽㍿aAéİ…m‍", "tokens": 50, "pieces": [" ", "9", "👍🏽\r", "㋿>", "d", "㍿!", "\u000b", "㋿\n", "'VEdét", "\u000b", ">éé", "👍🏽", "३", " ", "३", "😀🏽㍿", "a", "Aé", "İ", "…m", "‍"]} +{"text": "s(.<|endoftext|> !ḍ̇'Mꟲ'VEDž", "tokens": 26, "pieces": ["s", "(.<|", "endoftext", "|>", " ", "!<", "META", "_START", ">ḍ̇'M", "ꟲ'VE", "Dž"]} +{"text": "​'TDž \n…㍿\u000b ſt㍿åé½😀🏽é­9👍🏽<|endoftext|>ḍ̇'reⅣ#$%…ſA", "tokens": 59, "pieces": ["​'", "TDž", " \n", "…", "㍿", "\u000b ", " ſt", "㍿", "åé", "½", "😀🏽", "é", "­", "9", "👍🏽<|", "endoftext", "|>", "ḍ̇'re", "Ⅳ", "#$%", "…ſ", "A", ""]} +{"text": "!!!\tß'ſ漢s'Re9<|endoftext|>d'S -'!!​ 字t½\n‍ ḍ̇½३", "tokens": 39, "pieces": ["!!!", "\tß'ſ", "漢s'Re", "9", "<|", "endoftext", "|>", "d'S", " -'!!​", " ", " 字t", "½", "\n", "‍<", "EOT", ">", " ", " ḍ̇", "½३"]} +{"text": "'llA -​\u000b'VÉ字'ſß३Ⅳſ٣٤٥٦'😀🏽é<.Dž'VE.0<|fim_prefix|>ع…<|endoftext|>", "tokens": 52, "pieces": ["'ll", "A", " ", "-​", "\u000b", "'VÉ字'ſ", "ß", "३Ⅳ", "ſ", "٣٤٥", "٦", "'😀🏽", "é", "<.", "Dž'VE", ".", "0", "<|", "fim", "_prefix", "|>", "ع", "…", "<|", "endoftext", "|>"]} +{"text": "'reꟲd٣٤٥٦EOT9'D\n 字😀🏽㋿字m'VE😀🏽-0 -㍿😀🏽's !!", "tokens": 46, "pieces": ["'reꟲd", "٣٤٥", "٦", "EOT", "9", "'D", "\n", " ", " 字", "😀🏽<", "EOT", ">㋿", "字m'VE", "😀🏽-", "0", " -㍿😀🏽'", "s", " ", "!!"]} +{"text": "mt'S.A \n‍३!!<|fim_prefix|>🙂'sZ<|endoftext|>Aḍ̇.Dž㍿12345678s- 'ś12345678ß Ⅳ㍿ꟲåA#$%字å½m", "tokens": 69, "pieces": ["mt'S", ".A", " \n", "‍", "३", "!!<|", "fim", "_prefix", "|>🙂'", "s", "Z", "<|", "endoftext", "|>", "Aḍ̇", ".Dž", "㍿", "123", "456", "78", "s", "-", " '", "ś", "123", "456", "78", "ß", "", " ", "Ⅳ", "㍿ꟲå", "A", "#$%", "字å", "½", "m"]} +{"text": "👍🏽Ⅳ漢<|endoftext|>EOT́\t㋿\r\nع㍿", "tokens": 25, "pieces": ["👍🏽", "Ⅳ", "漢", "<|", "endoftext", "|>", "EOT́", "\t", "㋿\r\n", "ع", "㍿"]} +{"text": " , é12345678Afi​0
t\r\n\r\nİDž😀🏽<|endoftext|>'ll\u000b'S'll  ㋿­EOT­­\r", "tokens": 48, "pieces": [" ", ",", " é", "123", "456", "78", "Afi", "​", "0", "
t", "\r\n\r\n", "İDž", "😀🏽<|", "endoftext", "|><", "META", "_START", ">'", "ll", "\u000b", "'S'll", " ", " <", "EOT", ">", " ", " ㋿­", "EOT", "­­\r"]} +{"text": "\"\r\n\r\n>́ß ", "tokens": 5, "pieces": ["\"\r\n\r\n", ">́ß", " "]} +{"text": "字t'SEOTعZ㋿''ll'sEOTfi", "tokens": 16, "pieces": ["字t'S", "EOTع", "Z", "㋿''", "ll's", "EOTfi"]} +{"text": "!!​👍🏽漢aꟲa\r\n🙂́\tⅣḍ̇Ⅳßd🙂Ⅳé'VEEOT…​㍿㍿\r\n\r\n!!…🙂 tfiⅣ🙂'Reé", "tokens": 56, "pieces": ["!!​👍🏽", "漢aꟲa", "\r\n", "🙂́", "\t", "Ⅳ", "ḍ̇", "Ⅳ", "ßd", "🙂", "Ⅳ", "é'VE", "EOT", "…", "​㍿㍿\r\n\r\n", "!!", "…", "🙂", " ", " tfi", "Ⅳ", "🙂'", "Reé"]} +{"text": "
…\tfiḍ̇'Re0'é\u000b", "tokens": 13, "pieces": ["
…", "\tfiḍ̇'Re", "0", "'é", "\u000b"]} +{"text": "Džs㍿-ſ", "tokens": 8, "pieces": ["Džs", "㍿-", "ſ"]} +{"text": "'VEé <|fim_prefix|>é\r'M", "tokens": 15, "pieces": ["'VEé", " ", "<|", "fim", "_prefix", "|>", "é", "\r", "'M"]} +{"text": "ßfi'ſ'ſ𐞁d", "tokens": 11, "pieces": ["ßfi'ſ", "'ſ𐞁d"]} +{"text": "'ſ t\t fi\tEOT<|fim_prefix|>åafi\n!!m'\r\nع…'ſ ३ع'll…é👍🏽漢t'VEé \u000b­…漢<|endoftext|>", "tokens": 61, "pieces": ["'ſ", " t", "\t ", " <", "EOT", ">fi", "\tEOT", "<|", "fim", "_prefix", "|>", "åafi", "\n", "!!", "m", "'\r\n", "ع", "…", "'ſ", " ", " ", "३", "ع'll", "…é", "👍🏽", "漢t'VE", "é", " ", "\u000b", "­", "…漢", "<|", "endoftext", "|>"]} +{"text": "😀🏽\r\n12345678 \nⅣ's,\t.9'é­'M😀🏽\rs㍿'M‍å㍿'Msa…ß'ſ9t'Re\tꟲZ 'S're", "tokens": 55, "pieces": ["😀🏽\r\n", "123", "456", "78", " \n", "Ⅳ", "'s", ",", "\t", ".", "9", "'é", "­'", "M", "😀🏽\r", "s", "㍿'", "M", "‍å", "㍿'", "Msa", "…ß'ſ", "9", "t'Re", "\tꟲ", "Z", " '", "S're"]} +{"text": "EOT", "tokens": 2, "pieces": ["EOT"]} +{"text": "'S'D !9 ­३é, 'ReAßs", "tokens": 16, "pieces": ["'S'D", " ", "!", "9", " ", "­", "३", "é", ",", " ", "'Re", "Aßs"]} +{"text": " 'ſ're#$%0'De'عDž'Dعfi😀🏽Ⅳmå'sd\r\n\r\n'll\r'S\r\n\r\n", "tokens": 34, "pieces": [" ", " '", "ſ're", "#$%<", "EOT", ">", "0", "'De", "'ع", "Dž'D", "عfi", "😀🏽", "Ⅳ", "må's", "d", "\r\n\r\n", "'ll", "\r", "'S", "\r\n\r\n"]} +{"text": "\n\r\n\r\n<㍿½\"é's​­…㍿ꟲ३(𐞁㋿\r\n\r\n é
字å9\"aå\"", "tokens": 50, "pieces": ["\n\r\n\r\n", "<㍿", "½", "\"é's", "​­", "…", "㍿", "ꟲ", "", "३", "(𐞁", "㋿\r\n\r\n", " ", " é", "
字å", "9", "\"aå", "\""]} +{"text": ".'ſ'ſZt­ 'M-İḍ̇'S12345678#$%0'D㋿İfi\rEOTm'DZßع", "tokens": 36, "pieces": [".'", "ſ'ſ", "Zt", "­", " ", " '", "M", "-İḍ̇'S", "123", "456", "78", "#$%", "0", "'D", "㋿İfi", "\r", "EOTm'D", "Zßع"]} +{"text": "9½Dž\r\n\r\nZ!!‍A!\t㋿-åꟲ…fiém 'M'sa­㍿,😀🏽🙂 é\r\n👍🏽's'", "tokens": 50, "pieces": ["9½", "Dž", "\r\n\r\n", "Z", "!!‍", "A", "!", "\t", "㋿-", "åꟲ", "…", "fiém", " ", " '", "M's", "a", "­㍿,😀🏽🙂", " é", "\r\n", "👍🏽'", "s", "'"]} +{"text": "­'Tİ \n'ß\"…'VE'Mé​-<|endoftext|>\r\n\r\n​३'ſ.-a\"", "tokens": 31, "pieces": ["­'", "Tİ", " \n", "'ß", "\"", "…", "'VE'M", "é", "​-<|", "endoftext", "|>\r\n\r\n", "​", "३", "'ſ", ".-", "a", "\""]} +{"text": "\r\n12345678ßm'D
٣٤٥٦ḍ̇​ß,عs9<|fim_prefix|><|fim_prefix|><12345678'S­Ze9", "tokens": 44, "pieces": ["\r\n", "123", "456", "78", "ßm'D", "
", "٣٤٥", "٦", "ḍ̇", "​ß", ",", "عs", "9", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|><", "123", "456", "78", "'S", "­Ze", "9"]} +{"text": "'T\r\n'S\r\n\r\nꟲ,-  å12345678EOT㋿ \r\n\r\nⅣ<|endoftext|>t#$%👍🏽ßaé \n㋿", "tokens": 44, "pieces": ["'T", "\r\n", "'S", "\r\n\r\n", "ꟲ", ",-", " ", " å", "123", "456", "78", "EOT", "㋿", " \r\n\r\n", "Ⅳ", "<|", "endoftext", "|>", "t", "#$%👍🏽", "ßaé", " \n", "㋿"]} +{"text": "'s", "tokens": 1, "pieces": ["'s"]} +{"text": "ß字ع#$%e.ee\r\u000bm\n​字\u000bß\r\n", "tokens": 19, "pieces": ["ß字ع", "#$%", "e", ".ee", "\r", "", "\u000bm", "\n", "​字", "\u000bß", "\r\n"]} +{"text": "­
m'lle-A'MA😀🏽!!٣٤٥٦🙂-<|endoftext|>'Re'ſ'S12345678'D\t'll'safi ", "tokens": 40, "pieces": ["­", "
m'll", "e", "-A'M", "A", "😀🏽!!", "٣٤٥", "٦", "🙂-<|", "endoftext", "|>'", "Re'ſ", "'S", "123", "456", "78", "'D", "\t", "'ll's", "afi", " "]} +{"text": "aDž e'Sİ
d½s'VE …m'S…\nꟲ
ꟲ'Re'T…‍#$%å𐞁 'Re<|endoftext|>'VEEOT'Tꟲeé", "tokens": 63, "pieces": ["a", "Dž", " e'S", "İ", "
d", "½", "s'VE", " ", "…", "m'S", "…\n", "ꟲ", "
ꟲ'Re", "'T", "…", "‍#$%", "å𐞁", " '", "Re", "<|", "endoftext", "|>'", "VEEOT'T", "ꟲeé"]} +{"text": "å🙂\t'S\t𐞁ⅣdaEOT\t<|fim_prefix|>Ⅳ½(12345678fi're漢­㍿#$%\n ,>\n<|endoftext|>å(ꟲ're \nḍ̇", "tokens": 59, "pieces": ["å", "🙂", "\t", "'S", "\t𐞁", "Ⅳ", "da", "EOT", "\t", "<|", "fim", "_prefix", "|>", "Ⅳ½", "(", "123", "456", "78", "fi're", "漢", "­㍿#$%\n", " ", " ,>\n", "<|", "endoftext", "|>", "å", "(ꟲ're", " \n", "ḍ̇"]} +{"text": "\r\n", "tokens": 1, "pieces": ["\r\n"]} +{"text": "😀🏽
 .\n👍🏽'‍\n12345678t \n'VE\"", "tokens": 23, "pieces": ["😀🏽", "
", " ", ".\n", "👍🏽'‍\n", "123", "456", "78", "t", " \n", "'VE", "\"<", "META", "_START", ">"]} +{"text": "​👍🏽😀🏽.<|endoftext|>", "tokens": 14, "pieces": ["​👍🏽😀🏽.<|", "endoftext", "|>"]} +{"text": "9'D…é.㍿!dm 'Rem's漢Džd​9s\r \r\n字A'DⅣ㋿𐞁㋿<|endoftext|>'M漢a12345678<", "tokens": 59, "pieces": ["9", "'D", "…é", ".㍿!", "dm", " ", "'Rem's", "漢Džd", "​", "9", "s", "\r \r\n", "字", "A'D", "Ⅳ", "㋿𐞁", "㋿<|", "endoftext", "|>'", "M漢a", "123", "456", "78", "<"]} +{"text": "½'m", "tokens": 2, "pieces": ["½", "'m"]} +{"text": "'Re 
<'D
 t<ß \n'VEé',e<|fim_prefix|>३‍<|fim_prefix|>fi\t\r\n🙂9\"<|fim_prefix|>\r\n s…ع", "tokens": 48, "pieces": ["'Re", " ", "
", "<'", "D", "
 ", " t", "<ß", " \n", "'VEé", "',", "e", "<|", "fim", "_prefix", "|>", "३", "‍<|", "fim", "_prefix", "|>", "fi", "\t\r\n", "🙂", "9", "\"<|", "fim", "_prefix", "|>\r\n", "", " ", " s", "…ع"]} +{"text": "'s\r\n\t!!s'VEs'Dßå\nd😀🏽EOT \n  \n漢eé​🙂\">Z\t >", "tokens": 38, "pieces": ["'s", "\r\n", "\t", "!!", "s'VE", "s'D", "ßå", "\n", "d", "😀🏽", "EOT", " \n  \n", "漢eé", "​🙂\">", "Z", "\t ", " >"]} +{"text": "‍sİé́<|fim_prefix|>🙂'ſfi'Mعt٣٤٥٦ ſDž🙂👍🏽‍-mⅣ
́ 'S'漢", "tokens": 45, "pieces": ["‍s", "İé́", "<|", "fim", "_prefix", "|>🙂'", "ſfi'M", "عt", "", "٣٤٥", "٦", " ſ", "Dž", "🙂👍🏽‍-", "m", "Ⅳ", "
́", " '", "S", "'漢"]} +{"text": "'MⅣfiḍ̇eſ㍿漢0٣٤٥٦-m's\r\n'Dع ꟲ字字", "tokens": 29, "pieces": ["'M", "Ⅳ", "fiḍ̇eſ", "㍿漢", "0٣٤", "٥٦", "-m's", "\r\n", "'Dع", " ꟲ字字"]} +{"text": "ḍ̇ḍ̇‍٣٤٥٦e's!", "tokens": 14, "pieces": ["ḍ̇ḍ̇", "‍", "٣٤٥", "٦", "e's", "!"]} +{"text": " 're'ſ.‍ꟲ'llå'ſ'ſİDž>­ (At<", "tokens": 29, "pieces": [" '", "re'ſ", ".‍", "ꟲ'll", "å'ſ", "'ſ", "İDž", ">­", " ", "(<", "META", "_START", ">At", "<"]} +{"text": " -''D字9", "tokens": 10, "pieces": [" ", "-''", "D", "字", "9"]} +{"text": "(\r 's👍🏽é>\t…'Tعḍ̇'T<ꟲ'reDž½- 'ſ(t㍿\r \n-́", "tokens": 39, "pieces": ["(\r", " ", "'s", "👍🏽", "é", ">", "\t", "…", "'Tعḍ̇'T", "<ꟲ're", "Dž", "½", "-", " ", "'ſ", "(t", "㍿\r", " \n", "-́"]} +{"text": "­åEOT< ßå😀🏽t'ſt३\r\n\r\n'Reع \r
ꟲs<|fim_prefix|> ­", "tokens": 38, "pieces": ["­å", "EOT", "<", " ", "ßå", "😀🏽", "t'ſ", "t", "३", "\r\n\r\n", "'Reع", " \r", "
ꟲs", "<|", "fim", "_prefix", "|>", " ­"]} +{"text": "m\nEOT\u000b'VE,'D३㋿t३<|endoftext|>fi㍿ꟲmEOT漢㍿Z\"\rfi\r\nA\r漢İ👍🏽३,", "tokens": 50, "pieces": ["m", "\n", "EOT", "\u000b", "'VE", ",'", "D", "३", "㋿t", "३", "<|", "endoftext", "|>", "fi", "㍿ꟲm", "EOT漢", "㍿Z", "\"\r", "fi", "\r\n", "A", "\r", "漢", "İ", "👍🏽", "३", ","]} +{"text": " 👍🏽\u000b-\r\"\r\nsA Dž'VE", "tokens": 20, "pieces": [" ", " 👍🏽", "\u000b", "-<", "EOT", ">\r", "\"\r\n", "s", "A", " ", " Dž'VE"]} +{"text": "ꟲ--,eع㍿…ḍ̇\nZ \n\"sع \r\nſ'D
Z'M(", "tokens": 27, "pieces": ["ꟲ", "--,", "eع", "㍿", "…ḍ̇", "\n", "Z", " \n", "\"sع", " \r\n", "ſ'D", "
Z'M", "("]} +{"text": "'M'Re A𐞁ḍ̇,é\r𐞁… Z.\na٣٤٥٦12345678é'T", "tokens": 35, "pieces": ["'M'Re", " ", " A𐞁ḍ̇", ",é", "\r", "𐞁", "…", " Z", ".\n", "a", "٣٤٥", "٦12", "345", "678", "é'T"]} +{"text": "'S'T́", "tokens": 3, "pieces": ["'S'T", "́"]} +{"text": "0😀🏽­𐞁sſé 'MⅣEOT\u000b", "tokens": 20, "pieces": ["0", "😀🏽­", "𐞁sſé", " ", " '", "M", "Ⅳ", "EOT", "\u000b"]} +{"text": "Z<fi're9…fi\n!!\r!å9\r\n\r\nZEOTß\"12345678…㍿", "tokens": 28, "pieces": ["Z", "<fi're", "9", "…fi", "\n", "!!\r", "!å", "9", "\r\n\r\n", "ZEOTß", "\"", "123", "456", "78", "…", "㍿"]} +{"text": "\t\r\n\r\n Ⅳ \u000bꟲt'Sſꟲ'ſ\r\n
 ㋿'M\u000bfi㍿\né0!!㍿", "tokens": 42, "pieces": ["\t\r\n\r\n", " ", "Ⅳ", " ", "\u000bꟲt'S", "ſ", "ꟲ'ſ", "\r\n", "
", " ㋿'", "M", "\u000bfi", "㍿\n", "é", "0", "!!㍿"]} +{"text": "́A👍🏽 \n'Tع! 're'reİ'Dß,ß\u000b🙂㍿!!㍿
'T½t'M'll<|fim_prefix|>s'!t㋿🙂‍é\r\n\r\n㍿", "tokens": 55, "pieces": ["́", "A", "👍🏽", " \n", "'Tع", "!", " ", "'re're", "İ'D", "ß", ",ß", "\u000b", "🙂㍿!!㍿", "
", "'T", "½", "t'M", "'ll", "<|", "fim", "_prefix", "|>", "s", "'<", "META", "_START", ">!", "t", "㋿🙂‍", "é", "\r\n\r\n", "㍿"]} +{"text": "#$%ßEOT…½Dž ٣٤٥٦ \n'M,𐞁㋿s ", "tokens": 27, "pieces": ["#$%", "ß", "EOT", "…", "½", "Dž", " ", "٣٤٥", "٦", " \n", "'M", ",𐞁", "㋿s", " "]} +{"text": "0' \nḍ̇ßꟲ<\"
<|endoftext|>漢́amEOT‍Dž 🙂👍🏽 a", "tokens": 37, "pieces": ["0", "'", " \n", "ḍ̇ßꟲ", "<\"", "
", "<|", "endoftext", "|><", "META", "_START", ">漢́am", "EOT", "‍Dž", " ", " 🙂👍🏽", " a"]} +{"text": "½'ſå\t­'Daꟲ👍🏽fi㋿!!EOTé𐞁m👍🏽
 's'Mß're's \n 's\"\u000b'T\r\n", "tokens": 47, "pieces": ["½", "'ſå", "\t", "­'", "Daꟲ", "👍🏽", "fi", "㋿!!", "EOTé𐞁m", "👍🏽", "
", " ", "'s'M", "ß're", "'s", " \n", " ", " '", "s", "\"", "\u000b", "'T", "\r\n"]} +{"text": "­ع́ ٣٤٥٦'Sſ 9٣٤٥٦'T 漢٣٤٥٦-'T!!EOT​<|endoftext|>Ⅳ0\" ꟲ㋿
'ſ<'Mع", "tokens": 56, "pieces": ["­ع́", " ", "٣٤٥", "٦", "'Sſ", " ", "9٣٤", "٥٦", "'T", " 漢", "٣٤٥", "٦", "-'", "T", "!!", "EOT", "​<|", "endoftext", "|>", "Ⅳ0", "\"", " ꟲ", "㋿", "
", "'ſ", "<'", "Mع"]} +{"text": "'Re", "tokens": 1, "pieces": ["'Re"]} +{"text": "-!,'ll
́  .,.İ'\t#$%s're'VE(­\u000b!​<|endoftext|>'VE'D\r\n\u000bA'SDž", "tokens": 47, "pieces": ["-!,'", "ll", "
́", " ", " ", ".,.", "İ", "'", "\t", "#$%", "s", "'", "re'VE", "(­", "\u000b", "!<", "EOT", ">​<|", "endoftext", "|>'", "VE'D", "\r\n", "\u000bA'S", "Dž"]} +{"text": "
\u000b\tß(<|fim_prefix|>-  ㍿😀🏽'll'ſ", "tokens": 22, "pieces": ["
\u000b", "\tß", "(<|", "fim", "_prefix", "|>-", " ", " ㍿😀🏽'", "ll'ſ"]} +{"text": "½'३😀🏽((…'Ts>å", "tokens": 13, "pieces": ["½", "'", "३", "😀🏽((", "…", "'Ts", ">å"]} +{"text": "e½Z👍🏽ḍ̇12345678<|fim_prefix|>३\"s>\t!!'re'llDž<|endoftext|>ſ́'ll㋿0d😀🏽Dž", "tokens": 51, "pieces": ["e", "½", "Z", "👍🏽", "ḍ̇", "", "123", "456", "78", "<|", "fim", "_prefix", "|>", "३", "\"s", ">", "\t", "!!'", "re'll", "Dž", "<|", "endoftext", "|>", "ſ́'ll", "㋿", "0", "d", "😀🏽", "Dž"]} +{"text": "ḍ̇字'ſ<|fim_prefix|>字t'S#$%ßa'M 'T\u000b'M­AⅣ'< eé!!fi'reéé !ꟲ!!", "tokens": 48, "pieces": ["ḍ̇字'ſ", "<|", "fim", "_prefix", "|>", "字t'S", "#$%", "ßa'M", " ", "'T", "\u000b", "'M", "­A", "Ⅳ", "'<", " ", " eé", "!!", "fi're", "éé", " ", "!<", "EOT", ">ꟲ", "!!"]} +{"text": "İ㋿🙂\r٣٤٥٦عfi३Ⅳ'Re\u000b'VE‍🙂'D ३\t'ſ½🙂9'll٣٤٥٦ſ­å'S字.eéEOT\n'T !é字", "tokens": 52, "pieces": ["İ", "㋿🙂\r", "٣٤٥", "٦", "عfi", "३Ⅳ", "'Re", "\u000b", "'VE", "‍🙂'", "D", " ", "३", "\t", "'ſ", "½", "🙂", "9", "'ll", "٣٤٥", "٦", "ſ", "­å'S", "字", ".eé", "EOT", "\n", "'T", " ", " !", "é字"]} +{"text": "́fi.'S\t'VEDž㍿ \"<ꟲt-𐞁", "tokens": 23, "pieces": ["́fi", ".'", "S", "\t", "'VEDž", "㍿", " ", " \"<", "ꟲt", "-𐞁"]} +{"text": "9A\u000bZAéſſ\r\n\r\nꟲe !!\r\n\r\nİ", "tokens": 23, "pieces": ["9", "A", "\u000bZAé", "ſſ", "\r\n\r\n", "ꟲe", " ", "!!\r\n\r\n", "İ"]} +{"text": "fi½", "tokens": 2, "pieces": ["fi", "½"]} +{"text": "'s  …㋿'D'Re 'S\r\n's 9's­!İsEOTs12345678İ's'S😀🏽 fié", "tokens": 48, "pieces": ["'s", "  ", "…", "㋿<", "é", "​,>'", "D'Re", "", " ", "'S", "\r\n", "'s", " ", " ", "9", "'s", "­!", "İs", "EOTs", "123", "456", "78", "İ's", "'S", "😀🏽", " fié"]} +{"text": "!!\r\n\r\n' ß
>…<|endoftext|>'D<|endoftext|>字!!.😀🏽'sß­'Re'VE's\t 字'ſ.(", "tokens": 43, "pieces": ["!!\r\n\r\n", "'", " ß", "
", ">", "…", "<|", "endoftext", "|>'", "D", "<|", "endoftext", "|>", "字", "!!.😀🏽'", "sß", "­'", "Re'VE", "'s", "\t", " 字'ſ", ".("]} +{"text": " ع(", "tokens": 3, "pieces": [" ", " ع", "("]} +{"text": "字EOT.\r\nfi!ꟲ>.\r", "tokens": 11, "pieces": ["字", "EOT", ".\r\n", "fi", "!ꟲ", ">.\r"]} +{"text": " <'M0'VE ㍿…👍🏽é>9 漢12345678900EOT\r", "tokens": 31, "pieces": [" ", "<'", "M", "0", "'VE", " ", "㍿", "…", "👍🏽", "é", ">", "9", " ", " 漢", "123", "456", "789", "00", "EOT", "\r"]} +{"text": "'S,'ſ.<|fim_prefix|><|endoftext|>'ſ'sß0A\r9", "tokens": 22, "pieces": ["'S", ",'", "ſ", ".<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "ſ's", "ß", "0", "A", "\r", "9"]} +{"text": "عDžEOT!(👍🏽", "tokens": 9, "pieces": ["ع", "DžEOT", "!(👍🏽"]} +{"text": "<😀🏽'Afi \r\n0Dž-'MDž‍0ſ'll''re\n­ \nع٣٤٥٦👍🏽s
ßfi0 ३12345678​𐞁å\r\n\r\n", "tokens": 62, "pieces": ["<😀🏽'", "Afi", " \r\n", "0", "Dž", "-'", "MDž", "‍<", "META", "_START", ">", "0", "ſ", "'", "ll", "''", "re", "\n", "­", " \n", "ع", "٣٤٥", "٦", "👍🏽", "s", "
ßfi", "0", " ", "३12", "345", "678", "​", "𐞁å", "\r\n\r\n"]} +{"text": "'S-a​'reéß𐞁½ 'SZ漢fi ", "tokens": 17, "pieces": ["'S", "-a", "​'", "reéß𐞁", "½", " '", "SZ漢fi", " "]} +{"text": "\t㋿'VE<|endoftext|>", "tokens": 13, "pieces": ["\t", "㋿'", "VE", "<|", "endoftext", "|>"]} +{"text": "<#$%#$% a9EOT𐞁a \r\n\r\nt<'s  \n-ḍ̇fi٣٤٥٦åaådfi 0…-", "tokens": 44, "pieces": ["<#$%#$%", " a", "9", "EOT𐞁a", " \r\n\r\n", "t", "<'", "s", "  \n", "-ḍ̇fi", "٣٤٥", "٦", "å", "aådfi", " ", " ", "0", "…", "-"]} +{"text": "­😀🏽\"'ع👍🏽12345678's 'll🙂12345678<|fim_prefix|>😀🏽<.'M", "tokens": 36, "pieces": ["­😀🏽\"'<", "EOT", ">ع", "👍🏽", "123", "456", "78", "'s", " ", " '", "ll", "🙂", "123", "456", "78", "<|", "fim", "_prefix", "|>😀🏽<.'", "M"]} +{"text": ". ½.\u000b㍿d𐞁 12345678‍\n'DA", "tokens": 21, "pieces": [".", " ", "½", ".", "\u000b", "㍿d𐞁", " ", "123", "456", "78", "‍\n", "'DA"]} +{"text": "½a…😀🏽EOT'll-", "tokens": 11, "pieces": ["½", "a", "…", "😀🏽", "EOT'll", "-"]} +{"text": "\tA‍(😀🏽 \nعꟲ㍿(a!!㍿", "tokens": 25, "pieces": ["\t", "A", "‍(😀🏽", " \n", "عꟲ", "㍿(", "a", "!!㍿"]} +{"text": "́🙂'T\u000b!!a𐞁\te \",!!", "tokens": 15, "pieces": ["́", "🙂'", "T", "\u000b", "!!", "a𐞁", "\te", " ", "\",!!"]} +{"text": "\r\n\r\n'll👍🏽ts㋿m#$%.tA😀🏽㍿ 'ſعſ12345678‍ſfi>३é's'Ms\t\r\n\r\nDžع", "tokens": 49, "pieces": ["\r\n\r\n", "'ll", "👍🏽", "ts", "㋿m", "#$%.", "t", "A", "😀🏽㍿", " ", " '", "ſعſ", "123", "456", "78", "‍<", "EOT", ">ſfi", ">", "३", "é's", "'Ms", "\t\r\n\r\n", "Džع"]} +{"text": "!<|fim_prefix|>㍿", "tokens": 10, "pieces": ["!<|", "fim", "_prefix", "|>㍿"]} +{"text": "-漢é tm
,åß,'S", "tokens": 11, "pieces": ["-漢é", " tm", "
", ",åß", ",'", "S"]} +{"text": "\r\n\r\n३㍿!!", "tokens": 6, "pieces": ["\r\n\r\n", "३", "㍿!!"]} +{"text": "'D'smⅣé𐞁ꟲ'12345678EOTå'ſ12345678 \n'S㋿ \néⅣع👍🏽́fi
å é'Dſ-Z'", "sm", "Ⅳ", "é𐞁ꟲ", "'", "123", "456", "78", "EOTå'ſ", "123", "456", "78", " \n", "'S", "㋿", " \n", "é", "Ⅳ", "ع", "👍🏽́", "fi", "
å", " é'D", "ſ", "-Z", "<|endoftext|>s½,ꟲ\u000b9", "tokens": 28, "pieces": ["etſ", "9", "'re", "㋿\r\n", "½0३", "<|", "endoftext", "|>", "s", "½", ",ꟲ", "\u000b", "9"]} +{"text": "å\r\nsſ字Ⅳ​A­٣٤٥٦𐞁!EOTⅣfi\nßt's­'ll…d", "tokens": 34, "pieces": ["å", "\r\n", "sſ字", "Ⅳ", "​A", "­", "٣٤٥", "٦", "𐞁", "!EOT", "Ⅳ", "fi", "\n", "ßt's", "­'", "ll", "…d"]} +{"text": " ḍ̇\r\n>\r\n'll\r\n½Ⅳ㋿,Dž'MDž㋿A'VE\"\t'så…㋿é", "tokens": 38, "pieces": [" ", " ḍ̇", "\r\n", ">\r\n", "'ll", "\r\n", "½Ⅳ", "㋿,", "Dž'M", "Dž", "㋿A'VE", "\"", "\t", "'så", "…", "㋿é"]} +{"text": "
‍ésⅣ👍🏽s\r\n\r\n'VE\nḍ̇EOTd.  !!'T", "tokens": 26, "pieces": ["
", "‍és", "Ⅳ", "👍🏽", "s", "\r\n\r\n", "'VE", "\n", "ḍ̇", "EOTd", ".", "  ", " !!'", "T"]} +{"text": "'Sḍ̇.\r\n\r\n­३ ­!㍿", "tokens": 13, "pieces": ["'Sḍ̇", ".\r\n\r\n", "­", "३", " ", "­!㍿"]} +{"text": "'VE‍㍿\naå'S.", "tokens": 11, "pieces": ["'VE", "‍㍿\n", "aå'S", "."]} +{"text": "åDžſ'T😀🏽\r\n\r\n\n'SEOT\r\n\r\nḍ̇İ9 ㋿'Re㋿Dž''re'Ma㋿'T!a,\"\r\n\r\n\na'VE0'Reé", "tokens": 51, "pieces": ["å", "Džſ'T", "😀🏽\r\n\r\n\n", "'SEOT", "\r\n\r\n", "ḍ̇", "İ", "9", " ", "㋿'", "Re", "㋿Dž", "''", "re'M", "a", "㋿'", "T", "!a", ",\"\r\n\r\n\n", "a'VE", "0", "'Reé"]} +{"text": "tt\r\nDžm'a,<🙂\u000b­.‍9EOT<\rع\r
9! \tİع٣٤٥٦👍🏽Z'ſꟲåİ", "tokens": 48, "pieces": ["tt", "\r\n", "Džm", "'<", "EOT", ">a", ",<🙂", "\u000b", "­.‍", "9", "EOT", "<\r", "ع", "\r", "
", "9", "!", " ", "\tİع", "", "٣٤٥", "٦", "👍🏽", "Z'ſ", "ꟲå", "İ"]} +{"text": "\n\u000bfi<|fim_prefix|>, t'M-a\r\n\r\néⅣ'reAfi \r\n", "tokens": 25, "pieces": ["\n", "\u000bfi", "<|", "fim", "_prefix", "|>,", " ", " t'M", "-a", "\r\n\r\n", "é", "Ⅳ", "'re", "Afi", " \r\n"]} +{"text": "'sع!! \nعm👍🏽fiß ́A(<㋿'ſ'D!\t漢's­🙂,", "tokens": 31, "pieces": ["'sع", "!!", " \n", "عm", "👍🏽", "fiß", " ́", "A", "(<㋿'", "ſ'D", "!", "\t漢's", "­🙂,"]} +{"text": "́<|fim_prefix|>å12345678>½'S‍ع😀🏽'T‍12345678\nfi", "tokens": 32, "pieces": ["́", "<|", "fim", "_prefix", "|>", "å", "123", "456", "78", ">", "½", "'S", "‍ع", "😀🏽'", "T", "‍<", "EOT", ">", "123", "456", "78", "\n", "fi"]} +{"text": "\r\n\r\n DžZfi'ſ<|endoftext|> \n漢09\r'VE😀🏽Z.ß'ſ, \n𐞁\u000b'Re<|endoftext|>éß", "tokens": 51, "pieces": ["\r\n\r\n", "", " DžZfi'ſ", "<|", "endoftext", "|>", " \n", "漢", "09", "\r", "'VE", "😀🏽", "Z", ".ß'ſ", ",", " \n", "𐞁", "\u000b", "'Re", "<|", "endoftext", "|>", "éß"]} +{"text": "'D㍿'re'ſ's\naé​d>\ré३At​ \n-½ \nfiİfi字㍿'VE'D!Z\r\n\u000b'ſ", "tokens": 46, "pieces": ["'D", "㍿'", "re'ſ", "'s", "\n", "aé", "​", "d", ">\r", "é", "३", "At", "​", " \n", "-", "½", " \n", "fi", "İfi字", "㍿'", "VE'D", "!Z", "\r\n", "\u000b", "'ſ"]} +{"text": "🙂'VEſ👍🏽İ're­字( \n", "tokens": 13, "pieces": ["🙂'", "VEſ", "👍🏽", "İ're", "­字", "(", " \n"]} +{"text": "a!عDža​ 😀🏽…åع'M­d‍\"", "tokens": 19, "pieces": ["a", "!عDža", "​", " 😀🏽", "…åع'M", "­d", "‍\""]} +{"text": "9\r\n\"<> \n…(Ⅳ'M'S
'VE㋿>­٣٤٥٦'M0…字…", "tokens": 33, "pieces": ["9", "\r\n", "\"<<", "EOT", ">>", " \n", "…", "(", "Ⅳ", "'M'S", "
", "'VE", "㋿>­", "٣٤٥", "٦", "'M", "0", "…字", "…"]} +{"text": "'Re!­'३ d​-\u000b", "tokens": 10, "pieces": ["'Re", "!­'", "३", " d", "​-", "\u000b"]} +{"text": "½'re'VE…Z9ꟲ", "tokens": 11, "pieces": ["½", "'re'VE", "…Z", "9", "ꟲ"]} +{"text": "åⅣd👍🏽Ⅳ㍿dd,३́'Re<|fim_prefix|>\u000b \n>.", "tokens": 30, "pieces": ["å", "Ⅳ", "d", "👍🏽", "Ⅳ", "㍿dd", ",", "३", "́'Re", "<|", "fim", "_prefix", "|>", "\u000b", "", " \n", ">."]} +{"text": "𐞁İ㋿\u000b'Ḿ \nZ‍é't0éZA9३字fi", "tokens": 26, "pieces": ["𐞁", "İ", "㋿", "\u000b", "'Ḿ", " \n", "Z", "‍é't", "0", "é", "ZA", "9३", "字fi"]} +{"text": "'ll'Sİ(́s­字\r\n\r\nEOT'👍🏽'İ…>é'll 𐞁ݽ字'T'D🙂…sⅣé٣٤٥٦\" e\n", "tokens": 51, "pieces": ["'ll'S", "İ", "(́s", "­字", "\r\n\r\n", "EOT", "'👍🏽'", "İ", "…", ">é'll", " ", " 𐞁", "İ", "½", "字'T", "'D", "🙂", "…s", "Ⅳ", "é", "٣٤٥", "٦", "\"", " e", "\n"]} +{"text": " 字a(\r\nعⅣfi'M!d😀🏽12345678", "tokens": 17, "pieces": [" ", " 字a", "(\r\n", "ع", "Ⅳ", "fi'M", "!d", "😀🏽", "123", "456", "78"]} +{"text": "㍿<ß'VE>d's३'llm…​(ꟲ!'reé漢12345678're\u000b're'M'D", "tokens": 31, "pieces": ["㍿<", "ß'VE", ">d's", "३", "'llm", "…", "​(", "ꟲ", "!'", "reé漢", "123", "456", "78", "'re", "\u000b", "'re'M", "'D"]} +{"text": "e​ae!<ع", "tokens": 14, "pieces": ["e", "​", "a", "e", "!<", "ع"]} +{"text": "​字e𐞁EOTaſḍ̇#$% \"0\u000bſßm字… \n<|fim_prefix|>åEOT", "tokens": 41, "pieces": ["​字e𐞁", "EOTaſḍ̇", "#$%", " ", "\"", "0", "\u000bſßm", "字", "… \n", "<|", "fim", "_prefix", "|>", "å", "EOT"]} +{"text": "#$%!!Dž½३'VE.!e's Dž#$%Z'D​-🙂३'VEs'S,", "tokens": 32, "pieces": ["#$%!!", "Dž", "½३", "'VE", ".!", "e's", " ", "Dž", "#$%", "Z'D", "​-🙂", "३", "'VEs'S", ","]} +{"text": "-'VE<|fim_prefix|>'<ꟲfi'T٣٤٥٦عſ字字fi\t́字", "tokens": 26, "pieces": ["-'", "VE", "<|", "fim", "_prefix", "|>'<", "ꟲfi'T", "٣٤٥", "٦", "عſ字字fi", "\t́字"]} +{"text": "<'M's<|endoftext|>d(👍🏽٣٤٥٦字𐞁.t<漢'ſ\n​é<|endoftext|>'re9𐞁", "tokens": 46, "pieces": ["<'", "M's", "<|", "endoftext", "|>", "d", "(👍🏽", "٣٤٥", "٦", "字𐞁", ".t", "<漢'ſ", "\n", "​é", "<|", "endoftext", "|>'", "re", "9", "𐞁"]} +{"text": "\"ع३EOT", "tokens": 5, "pieces": ["\"ع", "३", "EOT"]} +{"text": "'ś\r!.,'M", "tokens": 7, "pieces": ["'ś", "\r", "!.,'", "M"]} +{"text": ">,<|endoftext|>9!!㋿½Ⅳ\r\n\r\nd'T9𐞁İİ\n­㍿​EOT‍\"'M><|endoftext|> \n\r", "tokens": 46, "pieces": [">,<|", "endoftext", "|>", "9", "!!㋿", "½Ⅳ", "\r\n\r\n", "d'T", "9", "𐞁", "İİ", "\n", "­㍿​", "EOT", "‍\"'", "M", "><|", "endoftext", "|>", " \n\r"]} +{"text": "s漢漢 ٣٤٥٦", "tokens": 8, "pieces": ["s漢漢", " ", "٣٤٥", "٦"]} +{"text": "'ſ \n字‍\rḍ̇Dž\"EOT'ſⅣ\"'M'S\u000b!!‍ßm's.fi­'ReⅣ", "tokens": 36, "pieces": ["'ſ", " \n", "字", "‍\r", "ḍ̇", "Dž", "\"EOT'ſ", "Ⅳ", "\"<", "EOT", ">'", "M'S", "\u000b", "!!‍", "ßm's", ".fi", "­'", "Re", "Ⅳ"]} +{"text": "t#$%#$%''s\r\n\r\nſZ\"<|endoftext|><٣٤٥٦<|fim_prefix|>!!-'Sé😀🏽٣٤٥٦d字#$%0s'Re漢é'Reeḍ̇s㋿ع", "tokens": 61, "pieces": ["t", "#$%#$%''", "s", "\r\n\r\n", "ſ", "Z", "\"<|", "endoftext", "|><", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>!!-'", "Sé", "😀🏽", "٣٤٥", "٦", "d字", "#$%", "0", "s'Re", "漢é", "'", "Reeḍ̇s", "㋿ع"]} +{"text": "\ta #$%,a\u000bt,Dža'll'Re-efi𐞁ſ٣٤٥٦
,éa㍿12345678👍🏽\r\n'D12345678d㋿12345678åEOT", "tokens": 56, "pieces": ["\ta", " #$%,", "a", "\u000bt", ",Dža'll", "'Re", "-efi𐞁ſ", "٣٤٥", "٦", "
", ",éa", "㍿", "123", "456", "78", "👍🏽\r\n", "'D", "123", "456", "78", "d", "㋿", "123", "456", "78", "å", "EOT"]} +{"text": "éḍ̇'T👍🏽 ‍ Ⅳ'ReZ…\"Ⅳİ!12345678ſd, \n 漢s\r\n\r\nDžå<|endoftext|>t(😀🏽 \n'Re!eé'red", "tokens": 57, "pieces": ["éḍ̇'T", "👍🏽", " ", "‍", " ", " ", "Ⅳ", "'Re", "Z", "…", "\"", "Ⅳ", "İ", "!", "123", "456", "78", "ſd", ",", " \n", " 漢s", "\r\n\r\n", "Džå", "<|", "endoftext", "|>", "t", "(😀🏽", " \n", "'Re", "!eé're", "d"]} +{"text": "İ
‍ḍ̇ꟲfi…", "tokens": 12, "pieces": ["İ", "
", "‍ḍ̇ꟲfi", "…"]} +{"text": "!!㍿́'ll𐞁㋿\r", "tokens": 15, "pieces": ["!!㍿́'", "ll𐞁", "㋿\r"]} +{"text": " 0<|fim_prefix|>'ll㋿'så­t😀🏽३'Taå\r\nZ㍿12345678𐞁​
ꟲİ ꟲ\r\n", "tokens": 46, "pieces": [" ", " ", "0", "<|", "fim", "_prefix", "|>'", "ll", "㋿'", "så", "­t", "😀🏽", "३", "'Taå", "\r\n", "Z", "㍿", "123", "456", "78", "𐞁", "​", "
ꟲ", "İ", " ꟲ", "\r\n"]} +{"text": "e𐞁㍿é 12345678'sß 12345678meZ㋿字0'Re!'MZ-ſ'T𐞁", "tokens": 41, "pieces": ["e𐞁", "㍿<", "META", "_START", ">é", " ", "123", "456", "78", "'sß", " ", "123", "456", "78", "me", "Z", "㋿字", "0", "'Re", "!'", "MZ", "-ſ'T", "𐞁"]} +{"text": " 0𐞁Z…#$%<|endoftext|>'D …👍🏽12345678á<|fim_prefix|>'re!!'Sfi \n😀🏽's٣٤٥٦'s'T<(", "tokens": 54, "pieces": [" ", "0", "𐞁", "Z", "…", "#$%<|", "endoftext", "|>'", "D", " ", "…", "👍🏽", "123", "456", "78", "á", "<|", "fim", "_prefix", "|>'", "re", "!!'", "Sfi", " \n", "😀🏽'", "s", "٣٤٥", "٦", "'s'T", "<("]} +{"text": "👍🏽A'reé'reİ\n३字字,åZt३ꟲ<|endoftext|> 'Mꟲ", "tokens": 36, "pieces": ["👍🏽", "A're", "é're", "İ", "\n", "३", "字", "字", ",å", "Zt", "३", "ꟲ", "<|", "endoftext", "|>", " ", "'Mꟲ"]} +{"text": "  ‍…\r漢ع're­ع \né'S👍🏽", "tokens": 17, "pieces": [" ", " ", "‍", "…\r", "漢ع're", "­ع", " \n", "é'S", "👍🏽"]} +{"text": "\r#$%<|endoftext|>\u000b𐞁'T𐞁#$%e's \u000b…!'T́漢㍿>\r>👍🏽\r
å9,fi'VE‍\r㍿ < 𐞁<|fim_prefix|>漢ſ", "tokens": 71, "pieces": ["\r", "#$%<|", "endoftext", "|>", "\u000b𐞁'T", "𐞁", "#$%", "e's", " \u000b", "…", "!'", "T́漢", "㍿>\r", ">👍🏽\r", "
å", "9", ",fi'VE", "‍\r", "㍿", " ", "<", " ", " 𐞁", "<|", "fim", "_prefix", "|>", "漢ſ"]} +{"text": "ſ9ع'Reé३'reſ'll\t'ſ㋿Ⅳ'Retfi­'re'VE​", "tokens": 26, "pieces": ["ſ", "9", "ع'Re", "é", "३", "'reſ'll", "\t", "'ſ", "㋿", "Ⅳ", "'Retfi", "­'", "re'VE", "​"]} +{"text": "EOTd90'llA.'ll\" ", "tokens": 10, "pieces": ["EOTd", "90", "'ll", "A", ".'", "ll", "\"", " "]} +{"text": "dA٣٤٥٦½td<­३å'Té12345678㍿字>\t'D <|fim_prefix|>ꟲ\n'MA'reå 'Dß🙂å 9", "tokens": 49, "pieces": ["d", "A", "٣٤٥", "٦½", "td", "<­", "३", "å'T", "é", "123", "456", "78", "㍿字", ">", "\t", "'D", " ", "<|", "fim", "_prefix", "|>", "ꟲ", "\n", "'MA're", "å", " ", "'Dß", "🙂å", " ", "9"]} +{"text": "'İDž\t½(漢", "tokens": 8, "pieces": ["'İDž", "\t", "½", "(漢"]} +{"text": ".ß𐞁‍İ \r\n", "tokens": 16, "pieces": [".", "ß𐞁", "‍", "İ", " \r\n"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'ſ ㋿ée👍🏽9\rEOT'\n<|endoftext|> ꟲ'S", " ꟲ'S", " \n​😀🏽 \nſDžſ'VE٣٤٥٦t\t<>", "tokens": 49, "pieces": ["'re", "🙂ḍ̇", "9", "​\n", "㋿'", "S", "­", "9", "éd漢", "
e", "İ", "", " \n", "​😀🏽", " \n", "ſ", "Džſ'VE", "٣٤٥", "٦", "<", "META", "_START", ">t", "\t", "<>"]} +{"text": " #$%'D🙂9,…'re", "tokens": 10, "pieces": [" ", "#$%'", "D", "🙂", "9", ",", "…", "'re"]} +{"text": "ſß!<", "tokens": 4, "pieces": ["ſß", "!<"]} +{"text": "ßd'Re", "tokens": 3, "pieces": ["ßd'Re"]} +{"text": "fi( 'M\t''ReEOTéa\r\n㍿𐞁½ ३ é😀🏽\u000b>😀🏽'Dع'Dé\"İ ", "tokens": 42, "pieces": ["fi", "(", " ", "'M", "\t", "''", "Re", "EOTéa", "\r\n", "㍿𐞁", "½", " ", "३", " é", "😀🏽", "\u000b", ">😀🏽'", "Dع'D", "é", "\"İ", " "]} +{"text": "㋿́!!-😀🏽'0ع\t'ſ'D!!#$%<é漢fi<­", "tokens": 40, "pieces": ["㋿́", "!!-😀🏽'", "0", "ع", "", "\t", "'ſ'D", "!!<", "åå", " ", " '", "D", "#$%<", "é漢fi", "<­"]} +{"text": ".'Re­>", "tokens": 4, "pieces": [".'", "Re", "­>"]} +{"text": " <|endoftext|>​ \n-", "tokens": 10, "pieces": [" <|", "endoftext", "|>​", " \n", "-"]} +{"text": "'ſ!m漢😀🏽m㋿ſ9( \n𐞁Zꟲ'llé'M‍'s#$%(åḍ̇㍿٣٤٥٦'Re \n\r\nZ", "tokens": 49, "pieces": ["'ſ", "!m漢", "😀🏽", "m", "㋿ſ", "9", "(", " \n", "𐞁Zꟲ'll", "é'M", "‍'", "s", "#$%(", "åḍ̇", "㍿", "٣٤٥", "٦", "'Re", " \n\r\n", "Z"]} +{"text": "t'ReEOTdA'M\r\n'D'VE\r\n'ſſ𐞁㋿'M'ſt>字", "tokens": 32, "pieces": ["t'Re", "EOTd", "A", "'", "M", "\r\n", "'D'VE", "\r\n", "'ſſ𐞁", "㋿'", "M'ſ", "t", ">字"]} +{"text": "'Mſ'T३d​\r\n<|fim_prefix|>'ſḍ̇'VE-Z-'Sİ㋿\r­ⅣEOT㋿\r\n\né😀🏽㍿'D>'M", "tokens": 47, "pieces": ["'Mſ'T", "३", "d", "​\r\n", "<|", "fim", "_prefix", "|>'", "ſḍ̇'VE", "-Z", "-'", "Sİ", "㋿\r", "­", "Ⅳ", "EOT", "㋿\r\n\n", "é", "😀🏽㍿'", "D", ">'", "M"]} +{"text": "漢 \r
  
'll½字Ⅳm\r\n\r\n,eémDž<|fim_prefix|>! 0​'S're-!!'D'D
ꟲ\"𐞁ḍ̇\u000b‍", "tokens": 59, "pieces": ["漢", " \r", "
  ", "
", "'ll", "½", "字", "Ⅳ", "m", "\r\n\r\n", ",<", "EOT", ">eém", "Dž", "<|", "fim", "_prefix", "|>!", " ", " ", "0", "​'", "S're", "-!!'", "D", "'", "D", "
ꟲ", "\"𐞁ḍ̇", "\u000b", "‍"]} +{"text": "\t12345678ßd\r\nm'Re​", "tokens": 10, "pieces": ["\t", "123", "456", "78", "ßd", "\r\n", "m'Re", "​"]} +{"text": ".!!㋿ḍ̇éⅣ'red३'Dåt", "tokens": 22, "pieces": [".!!㋿", "ḍ̇é", "Ⅳ", "'red", "", "३", "'Dåt"]} +{"text": "-‍\",
", "tokens": 4, "pieces": ["-‍\",", "
"]} +{"text": "'s'T DžEOT12345678'S", "tokens": 11, "pieces": ["'s'T", " DžEOT", "123", "456", "78", "'S"]} +{"text": "­㋿\r字\r\n\r\n㍿ (​é\r\n", "tokens": 20, "pieces": ["­㋿<", "META", "_START", ">\r", "字", "\r\n\r\n", "㍿", " ", "(​", "é", "\r\n"]} +{"text": "字!9", "tokens": 3, "pieces": ["字", "!", "9"]} +{"text": "e\nḍ̇字A-٣٤٥٦\"​
\"'DZⅣ𐞁'S<|endoftext|>> (İ \nm,><漢", "tokens": 48, "pieces": ["e", "\n", "ḍ̇字", "A", "-", "٣٤٥", "٦", "\"<", "EOT", ">​", "
", "\"'", "DZ", "", "Ⅳ", "𐞁'S", "<|", "endoftext", "|>>", " ", " (", "İ", " \n", "m", ",<", "META", "_START", ">><", "漢"]} +{"text": "é<‍㍿漢…\u000b​9'M字漢 \r\n㍿ꟲ<'så\rm漢fi‍!!㋿\"٣٤٥٦ſ", "tokens": 28, "pieces": ["é", "Ⅳ", "…", "!!", "m", "(d'VE", "漢", ">å", "\r", "m漢fi", "‍!!㋿\"", "٣٤٥", "٦", "ſ"]} +{"text": "… \nßⅣ12345678
ḍ̇mعßA​İ-é12345678🙂'M‍ 'M9EOT\r\n'll\r\n\r\n٣٤٥٦ſ ", "tokens": 43, "pieces": ["… \n", "ß", "Ⅳ12", "345", "678", "
ḍ̇mعß", "A", "​İ", "-é", "123", "456", "78", "🙂'", "M", "‍", " ", " '", "M", "9", "EOT", "\r\n", "'ll", "\r\n\r\n", "٣٤٥", "٦", "ſ", " "]} +{"text": "'ll🙂\rꟲ #$%, 'Re½​㋿…'D漢ꟲ'SAA e\t\tḍ̇'reİ👍🏽e😀🏽fi🙂s, 'VEt!", "tokens": 55, "pieces": ["'ll", "🙂\r", "ꟲ", " ", "#$%,", " '", "Re", "½", "​㋿", "…", "'D漢ꟲ'S", "AA", " ", " e", "\t", "\tḍ̇'re", "İ", "👍🏽", "e", "😀🏽<", "EOT", ">fi", "🙂s", ",", " ", " '", "VEt", "!"]} +{"text": "漢ß\t​", "tokens": 4, "pieces": ["漢ß", "\t", "​"]} +{"text": " ​må12345678#$%<|fim_prefix|>½Ⅳ㋿字'D字­!é\rꟲ'T", "tokens": 43, "pieces": ["", " ", "​må", "123", "456", "78", "#$%<|", "fim", "_prefix", "|>", "½", "<", "META", "_START", ">", "Ⅳ", "㋿字'D", "字", "­!", "é", "\r", "ꟲ'T"]} +{"text": "㍿
'D9𐞁½dß\r\n\r\n\u000bEOTd'sDžså'M<|endoftext|>ZDž 
.é'Re'TZ \n𐞁Ⅳ\t", "tokens": 57, "pieces": ["㍿", "
", "'D", "9", "𐞁", "½", "dß", "\r\n\r\n", "\u000bEOTd", "'", "s", "Džså'M", "<|", "endoftext", "|>", "ZDž", " ", "
", ".é'Re", "'TZ", " \n", "𐞁", "", "Ⅳ", "\t"]} +{"text": "EOTm'll're'reé ­ m字,<|fim_prefix|>'Re'tfi‍'VE🙂0é'M'VEt' ㋿-\n
'ſ'ſ‍9EOT𐞁'S", "tokens": 50, "pieces": ["EOTm'll", "'re're", "é", " ­", " m字", ",<|", "fim", "_prefix", "|>'", "Re't", "fi", "‍'", "VE", "🙂", "0", "é'M", "'VEt", "'", " ㋿-\n", "
", "'ſ'ſ", "‍", "9", "EOT𐞁'S"]} +{"text": "Ⅳ#$%ååß 'VE0fi㋿½'T'Re0-.12345678'>'s.🙂>ꟲ…", "tokens": 36, "pieces": ["Ⅳ", "#$%", "ååß", " ", "'VE", "0", "fi", "㋿", "½", "'T'Re", "0", "-.", "123", "456", "78", "'>'", "s", ".🙂>", "ꟲ", "…"]} +{"text": "<|endoftext|>fi-㍿'T​", "tokens": 15, "pieces": ["<|", "endoftext", "|>", "fi", "-㍿'", "T", "​"]} +{"text": "İſ٣٤٥٦09字'­#$%\t'VE's'­e\"३\r \n<|fim_prefix|>s㋿Dž'T ", "tokens": 37, "pieces": ["İſ", "٣٤٥", "٦09", "字", "'­#$%", "\t", "'VE's", "'­", "e", "\"", "३", "\r \n", "<|", "fim", "_prefix", "|>", "s", "㋿Dž'T", " "]} +{"text": "A​Aſſé-12345678.\u000b- ३½s'Tḍ̇#$%(ع 
ß\nDž'D'MDž 'seß", "tokens": 41, "pieces": ["A", "​A", "ſſé", "-", "123", "456", "78", ".", "\u000b", "-", " ", "३½", "s'T", "ḍ̇", "#$%(", "ع", " ", "
ß", "\n", "Dž'D", "'MDž", " ", "'seß"]} +{"text": "- \né'Sa'VEſ漢عſ𐞁..EOT<字,'s'ſ<\r\n\r\nt٣٤٥٦ꟲ\r\n\r\n👍🏽é­字9", "tokens": 47, "pieces": ["-", " \n", "é'S", "a'VE", "ſ漢عſ𐞁", "..", "EOT", "<字", ",'", "s'ſ", "<\r\n\r\n", "t", "٣٤٥", "٦", "ꟲ", "\r\n\r\n", "👍🏽", "é", "­字", "9", ""]} +{"text": "9\n \n 'D,Dž'ع<<|fim_prefix|><|fim_prefix|>​EOT>İ'!ع字éée字('ll‍👍🏽\t'VEḍ̇字🙂a字 ,", "tokens": 54, "pieces": ["9", "\n \n", " ", "'D", ",Dž", "'ع", "<<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|><", "EOT", ">​", "EOT", ">İ", "'!", "ع字éée字", "('", "ll", "‍👍🏽", "\t", "'VEḍ̇字", "🙂a字", " ", ","]} +{"text": "å<'re<|endoftext|>½ \n<|endoftext|>\r\n\r\n٣٤٥٦fi ​\r0'reZ \nEOTDž. ㍿­\r\n#$%İ, 9 \n\"'M'TA", "tokens": 55, "pieces": ["å", "<'", "re", "<|", "endoftext", "|>", "½", " \n", "<|", "endoftext", "|>\r\n\r\n", "٣٤٥", "٦", "fi", " ", " ​\r", "0", "'re", "Z", " \n", "EOTDž", ".", " ", "㍿­\r\n", "#$%", "İ", ",", " ", " ", "9", " \n", "\"'", "M'T", "A"]} +{"text": "\u000b<|endoftext|>\u000b㋿'s३ſ‍.'VE
\u000ba㍿Ⅳt​'Re\r ꟲ", "tokens": 43, "pieces": ["\u000b", "<|", "endoftext", "|>", "\u000b", "㋿'", "s", "३", "ſ", "‍.'", "VE", "
", "\u000ba", "㍿", "Ⅳ", "t", "​'", "Re", "\r", " ꟲ", ""]} +{"text": "EOT#$%😀🏽9㋿'T'D
<|endoftext|>'M,", "tokens": 24, "pieces": ["EOT", "#$%😀🏽", "9", "㋿'", "T'D", "
", "<|", "endoftext", "|>'", "M", ","]} +{"text": "'ſ ", "tokens": 6, "pieces": ["'ſ", "", " "]} +{"text": "éå'M字
>­​dé\r\n\r\n½\r\n, \u000bmfi.'llſ👍🏽 \n ع​t …é'T12345678fi('t\n½", "tokens": 42, "pieces": ["éå'M", "字", "
", ">­​", "dé", "\r\n\r\n", "½", "\r\n", ",", " ", "\u000bmfi", ".'", "llſ", "👍🏽", " \n", " ع", "​t", " ", "…é'T", "123", "456", "78", "fi", "('", "t", "\n", "½"]} +{"text": "٣٤٥٦å👍🏽t'M'!'ſm'Ś12345678…(ſ\u000b👍🏽漢's'Re'T!", "tokens": 36, "pieces": ["٣٤٥", "٦", "å", "👍🏽", "t'M", "'!'", "ſm'S", "́", "123", "456", "78", "…", "(ſ", "", "\u000b", "👍🏽", "漢's", "'Re'T", "!"]} +{"text": "'s­'S\r", "tokens": 5, "pieces": ["'s", "­'", "S", "\r"]} +{"text": "́㋿ 
Ⅳ字! ‍Dž'T\rß(Aß漢字🙂.'M'VEDže'VE'ét٣٤٥٦…(9\r\n\r\n'VE\r\n\r\n
A", "tokens": 49, "pieces": ["́", "㋿", " ", "
", "Ⅳ", "字", "!", " ", " ‍", "Dž'T", "\r", "ß", "(Aß漢字", "🙂.'", "M'VE", "Dže'VE", "'ét", "٣٤٥", "٦", "…", "(", "9", "\r\n\r\n", "'VE", "\r\n\r\n", "
A"]} +{"text": "DžéaA㋿ .,é
٣٤٥٦ßⅣ12345678", "tokens": 22, "pieces": ["Džéa", "A", "㋿", " .,", "é", "
", "٣٤٥", "٦", "ß", "Ⅳ12", "345", "678"]} +{"text": " 👍🏽!!'M\t!!>½s३<|fim_prefix|>åA३'D\tZ\r\n\r\n'Re>#$%ḍ̇Ⅳd\r\n\r\n\tdt!!< t \"ع", "tokens": 57, "pieces": [" ", " 👍🏽!!'", "M", "\t", "!!>", "½", "s", "३", "<|", "fim", "_prefix", "|>", "å", "A", "३", "'D", "\tZ", "\r\n\r\n", "'Re", ">#$%", "ḍ̇", "Ⅳ", "d", "\r\n\r\n", "\t", "dt", "!!<", " t", " ", " \"", "ع"]} +{"text": "㍿३عAſ,>#$%!!'ll12345678'", "tokens": 21, "pieces": ["㍿", "३", "ع", "A", "ſ", ",>#$%!!'", "ll", "123", "456", "78", "'"]} +{"text": "'S- !!\"😀🏽 aåé 'S
", "tokens": 21, "pieces": ["<", "EOT", ">'", "S", "-", " ", "!!\"😀🏽", " aåé", " ", "'S", "
"]} +{"text": "Dž( \n12345678sé#$%(12345678e😀🏽", "tokens": 17, "pieces": ["Dž", "(", " \n", "123", "456", "78", "sé", "#$%(", "123", "456", "78", "e", "😀🏽"]} +{"text": "'ReéDžé 😀🏽­㍿ 'lle", "tokens": 23, "pieces": ["'Reé", "Dž", "é", " 😀🏽­㍿", " ", "'lle", ""]} +{"text": "३!㋿m\t'D㍿\r\n\r\n.'llDž 'lls'é<", "tokens": 25, "pieces": ["३", "!㋿", "m", "", "\t", "'D", "㍿\r\n\r\n", ".'", "ll", "Dž", " ", "'lls", "'é", "<"]} +{"text": " \n'VE'Sd­ ع'Må\r\n\r\n'ree 'ReeZ's‍ ('re🙂 (\r\u000bꟲ", "tokens": 36, "pieces": [" \n", "'VE'S", "d", "­", " ع'M", "å", "\r\n\r\n", "'ree", " ", " '", "Ree", "Z's", "‍", " ", "('", "re", "🙂<", "META", "_START", ">", " ", "(\r", "\u000bꟲ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'sZ", "tokens": 2, "pieces": ["'s", "Z"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n'S!!‍字\r\n\r\ns漢A‍漢0!!‍dém0漢-́Dž \r\n\r\n<漢're‍'M12345678ḍ̇", "tokens": 40, "pieces": ["\r\n", "'S", "!!‍", "字", "\r\n\r\n", "s漢", "A", "‍漢", "0", "!!‍", "dém", "0", "漢", "-́", "Dž", " \r\n\r\n", "<漢're", "‍'", "M", "123", "456", "78", "ḍ̇"]} +{"text": "<|endoftext|>𐞁३'S,ſ́!!e\t­\r\n'Reḍ̇'s'Ss३ ㍿#$%Z'ſa½<(!9", "tokens": 46, "pieces": ["<|", "endoftext", "|>", "𐞁", "३", "'S", ",ſ́", "!!", "e", "\t", "­<", "META", "_START", ">\r\n", "'Reḍ̇'s", "'Ss", "३", " ", "㍿#$%", "Z'ſ", "a", "½", "<(!", "9"]} +{"text": "méꟲé\n漢's<|fim_prefix|>
9३ éa A'VE३!!漢 \na'VE>éEOT<|endoftext|>ḍ̇ A,字'll#$%ſ", "tokens": 58, "pieces": ["méꟲé", "\n", "漢's", "<|", "fim", "_prefix", "|>", "
", "9", "", "३", " éa", " ", " A'VE", "३", "!!", "漢", " \n", "a'VE", ">é", "EOT", "<|", "endoftext", "|>", "ḍ̇", " A", ",字'll", "#$%", "ſ"]} +{"text": "EOT३('ll㍿e😀🏽A'ſ­é<12345678
漢\u000b ́12345678", "tokens": 29, "pieces": ["EOT", "३", "('", "ll", "㍿e", "😀🏽", "A'ſ", "­é", "<", "123", "456", "78", "
漢", "\u000b", " ́", "123", "456", "78"]} +{"text": "!👍🏽'S'ſ \r\n'VE‍'!!'ḍ̇>Dž\r\n'ſ٣٤٥٦md é字ع!!<|endoftext|>0!!!'S
𐞁d", "tokens": 51, "pieces": ["!👍🏽'", "S'ſ", " \r\n", "'VE", "‍'!!'", "ḍ̇", ">Dž", "\r\n", "'ſ", "٣٤٥", "٦", "md", " é字ع", "!!<|", "endoftext", "|>", "0", "!!!'", "S", "
𐞁d"]} +{"text": "é'ſtA'ſ\r\n<|fim_prefix|>\t'ſ३漢\r\n\u000b\nét!'s\r\n\r\n-Dž<|fim_prefix|>ع>9'VEİé㍿", "tokens": 47, "pieces": ["é'ſ", "t", "A'ſ", "\r\n", "<|", "fim", "_prefix", "|>", "\t", "'ſ", "३", "漢", "\r\n\u000b\n", "ét", "!'", "s", "\r\n\r\n", "-Dž", "<|", "fim", "_prefix", "|>", "ع", ">", "9", "'VEİé", "㍿"]} +{"text": "<12345678👍🏽\r­‍", "tokens": 10, "pieces": ["<", "123", "456", "78", "👍🏽\r", "­‍"]} +{"text": "'s <'s!!ع\n-​\u000b Dž👍🏽Ⅳ's're<‍<|endoftext|>😀🏽\"'M 'll>٣٤٥٦ \tEOTm ", "tokens": 45, "pieces": ["'s", " ", "<'", "s", "!!", "ع", "\n", "-​", "\u000b", " Dž", "👍🏽", "Ⅳ", "'s're", "<‍<|", "endoftext", "|>😀🏽\"'", "M", " ", "'ll", ">", "٣٤٥", "٦", " ", "\tEOTm", " "]} +{"text": "\r\n\"🙂fi é‍٣٤٥٦ḍ̇å<|endoftext|>
!å​ 'Mé'T9!!!३㍿.d12345678'VE'D<(", "tokens": 48, "pieces": ["\r\n", "\"🙂", "fi", " ", " é", "‍", "٣٤٥", "٦", "ḍ̇å", "<|", "endoftext", "|>", "
", "!å", "​", " ", "'Mé'T", "9", "!!!", "३", "㍿.", "d", "123", "456", "78", "'VE'D", "<("]} +{"text": "́ ḍ̇\"㍿'VEsⅣſ(#$%ß  
\rAعt\t𐞁!!👍🏽­sfi!'Reå-fiEOT", "tokens": 45, "pieces": ["́", " ḍ̇", "\"㍿'", "VEs", "Ⅳ", "ſ", "(#$%", "ß", "  
\r", "Aعt", "\t𐞁", "!!👍🏽­", "sfi", "!'", "Reå", "-fi", "EOT"]} +{"text": "'T‍३Ⅳ", "tokens": 5, "pieces": ["'T", "‍", "३Ⅳ"]} +{"text": "!!tA'reꟲ½ع­३a!s'll\r\n<<|fim_prefix|>'S're<|endoftext|>👍🏽 A \n,fi\r\n,İ'ſ👍🏽d#$%t-", "tokens": 52, "pieces": ["!!", "t", "A're", "ꟲ", "½", "ع", "­", "३", "a", "!s'll", "\r\n", "<<|", "fim", "_prefix", "|>'", "S're", "<|", "endoftext", "|>👍🏽", " A", " \n", ",fi", "\r\n", ",İ'ſ", "👍🏽", "d", "#$%", "t", "-"]} +{"text": ">'s'll‍ꟲ\u000b​'s#$%'Td👍🏽", "tokens": 17, "pieces": [">'", "s'll", "‍ꟲ", "\u000b", "​'", "s", "#$%'", "Td", "👍🏽"]} +{"text": "\r\n\"​'ſ漢\u000bⅣ-٣٤٥٦t.", "tokens": 18, "pieces": ["\r\n", "\"​'", "ſ漢", "\u000b", "Ⅳ", "-", "٣٤٥", "٦", "t", "."]} +{"text": "9​å\r\n\r\n''VE३d,​́és're'T'ſmḍ̇, \n é
\"mſİ", "tokens": 48, "pieces": ["9", "​å", "\r\n\r\n", "''", "VE", "३", "d", ",​́", "és're", "'T'ſ", "mḍ̇", ",", " \n", " é", "
", "\"mſ", "İ"]} +{"text": "‍\r\n㍿Zİ9é.>fi'VEEOT ٣٤٥٦å😀🏽'D  \n#$%‍'re߅\r\n\r\n\u000b\"a 12345678", "tokens": 48, "pieces": ["‍<", "EOT", ">\r\n", "㍿Zİ", "9", "é", ".>", "fi'VE", "EOT", " ", "٣٤٥", "٦", "å", "😀🏽'", "D", "  \n", "#$%‍'", "reß", "…\r\n\r\n", "\u000b", "\"a", " ", "123", "456", "78"]} +{"text": "A.३-\u000b9字ꟲZ'll㋿(a!𐞁sEOT'll.'å!! ḍ̇", "tokens": 38, "pieces": ["A", ".", "३", "-", "\u000b", "9", "字ꟲ", "Z'll", "㋿(", "a", "!𐞁s", "EOT'll", ".'", "å", "!!", " ḍ̇"]} +{"text": "!!s'fifi👍🏽m…\u000ba\"Ⅳ're½.9İEOTꟲſe字ꟲ'ſ", "tokens": 34, "pieces": ["!!", "s", "'fifi", "👍🏽", "m", "…", "\u000ba", "\"", "Ⅳ", "'re", "½", ".", "9", "İEOTꟲſe字ꟲ'ſ"]} +{"text": "𐞁'Sß​<|fim_prefix|>ſ \n>㋿9ꟲA9 a", "tokens": 27, "pieces": ["𐞁'S", "ß", "​<|", "fim", "_prefix", "|>", "ſ", " \n", ">㋿", "9", "ꟲ", "A", "9", " ", " a"]} +{"text": "fi३\u000b're'VEع", "tokens": 7, "pieces": ["fi", "३", "\u000b", "'re'VE", "ع"]} +{"text": "👍🏽'séd<­A㋿
'Mm'M½
", "tokens": 19, "pieces": ["👍🏽'", "séd", "<­", "A", "㋿", "
", "'Mm'M", "½", "
"]} +{"text": "'VEs-漢>Z'", "tokens": 7, "pieces": ["'VEs", "-漢", ">Z", "'"]} +{"text": "İḍ̇t're½e0😀🏽'\"d", "tokens": 14, "pieces": ["İḍ̇t're", "½", "e", "0", "😀🏽'\"", "d"]} +{"text": "efi're🙂\r㋿(字
…", "tokens": 13, "pieces": ["efi're", "🙂\r", "㋿(", "字", "
…"]} +{"text": "9​é", "tokens": 9, "pieces": ["", "9", "​", "é"]} +{"text": "ß😀🏽å's", "tokens": 7, "pieces": ["ß", "😀🏽", "å's"]} +{"text": "‍Z<|fim_prefix|>३!!\n­", "tokens": 11, "pieces": ["‍Z", "<|", "fim", "_prefix", "|>", "३", "!!\n", "­"]} +{"text": "漢é.…", "tokens": 9, "pieces": ["漢é", ".", "…", ""]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "9.9 ㋿é(\r\nİ,s
", "tokens": 17, "pieces": ["9", ".", "9", " ", " ㋿<", "EOT", ">é", "(\r\n", "İ", ",s", "
"]} +{"text": "\u000b #$%A \n'VE'ſ🙂́'Tsé'ſt ", "tokens": 23, "pieces": ["\u000b", " ", "#$%", "A", " \n", "'VE'ſ", "🙂<", "META", "_START", ">́'T", "sé'ſ", "t", " "]} +{"text": "𐞁३ \n٣٤٥٦'ll­‍'ReꟲEOT'D👍🏽'M'll<'ll.d\"'Md'T漢(", "tokens": 38, "pieces": ["𐞁", "३", " \n", "٣٤٥", "٦", "'ll", "­‍'", "Reꟲ", "EOT'D", "👍🏽'", "M'll", "<'", "ll", ".d", "\"'", "Md'T", "漢", "("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "0fiß.'re\"\u000b 'ſfiß!!'ll\r\n'S​'D㋿\r\n\r\n…", "tokens": 30, "pieces": ["0", "fiß", ".'", "re", "\"", "\u000b", "", " ", " '", "ſfiß", "!!'", "ll", "\r\n", "'S", "​'", "D", "㋿\r\n\r\n", "…"]} +{"text": "ß३'VEع<|endoftext|>Ⅳ👍🏽ß,", "tokens": 19, "pieces": ["ß", "३", "'VEع", "<|", "endoftext", "|>", "Ⅳ", "👍🏽", "ß", ","]} +{"text": "t\r'T9­A'S🙂 #$%!'T\tt½'VE㋿
-m👍🏽\n
 \n", "tokens": 31, "pieces": ["t", "\r", "'T", "9", "­A'S", "🙂", " ", "#$%!'", "T", "", "\tt", "½", "'VE", "㋿", "
", "-m", "👍🏽\n", "
 \n"]} +{"text": "0ḍ̇Zdé.\r\n\r\nZ­<|endoftext|>as…'ll'M12345678<|fim_prefix|><\n‍ 'S👍🏽9!e'Re'Re", "tokens": 47, "pieces": ["0", "ḍ̇", "Zdé", ".\r\n\r\n", "Z", "­<|", "endoftext", "|>", "as", "…", "'ll'M", "123", "456", "78", "<|", "fim", "_prefix", "|><\n", "‍", " ", " '", "S", "👍🏽", "9", "!e'Re", "'Re"]} +{"text": "é漢ḍ̇", "tokens": 5, "pieces": ["é漢ḍ̇"]} +{"text": "ß٣٤٥٦'TZt३<|endoftext|>å.漢a\tm'TA A½e‍​ ㋿㋿'M
 \n<'Re \n٣٤٥٦́", "tokens": 53, "pieces": ["ß", "٣٤٥", "٦", "'TZt", "३", "<|", "endoftext", "|>", "å", ".漢a", "\tm'T", "A", " A", "½", "e", "‍​", " ", "㋿㋿'", "M", "
 \n", "<'", "Re", " \n", "٣٤٥", "٦", "́"]} +{"text": " ㍿ 'll A字<|fim_prefix|>'Re <|fim_prefix|>٣٤٥٦'M'Mꟲ\r\nå>Ⅳ.𐞁EOT‍'ll\n\rt!\nm<|fim_prefix|>", "tokens": 62, "pieces": [" ", " ㍿", " ", "'ll", " A字", "<|", "fim", "_prefix", "|>'", "Re", " ", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "'M'M", "ꟲ", "\r\n", "å", ">", "Ⅳ", ".𐞁", "EOT", "‍'", "ll", "\n\r", "t", "!\n", "m", "<|", "fim", "_prefix", "|>"]} +{"text": "d\nḍ̇fiß\r'M😀🏽'S'T㋿ \r\n\r\n㍿<­漢٣٤٥٦'ReDž\n<|endoftext|>a--!!漢fi…\u000bé\r३­", "tokens": 52, "pieces": ["d", "\n", "ḍ̇fiß", "\r", "'M", "😀🏽'", "S'T", "㋿", " \r\n\r\n", "㍿<­", "漢", "٣٤٥", "٦", "'Re", "Dž", "\n", "<|", "endoftext", "|>", "a", "--!!", "漢fi", "…", "\u000bé", "\r", "३", "­"]} +{"text": "́a!!", "tokens": 3, "pieces": ["́a", "!!"]} +{"text": "a<|endoftext|>𐞁\nꟲ'S\r\nfiEOT12345678🙂", "tokens": 25, "pieces": ["a", "<|", "endoftext", "|>", "𐞁", "\n", "ꟲ'S", "\r\n", "fi", "EOT", "123", "456", "78", "🙂"]} +{"text": "👍🏽'reé㍿ 🙂0\t㋿\r#$% 
m!!é#$%🙂's㍿s'VEeſ'M#$%'ll漢Z,a( ", "tokens": 45, "pieces": ["👍🏽'", "reé", "㍿", " 🙂", "0", "\t", "㋿\r", "#$%", " ", "
m", "!!", "é", "#$%🙂'", "s", "㍿s'VE", "eſ'M", "#$%'", "ll漢", "Z", ",a", "(", " "]} +{"text": "👍🏽…㋿İd ", "tokens": 11, "pieces": ["👍🏽", "…", "㋿İd", " "]} +{"text": " 字 \u000bmEOT​ḍ̇ İ٣٤٥٦👍🏽'ſDža'T\r>'TDžDž!!٣٤٥٦\"'VE", "tokens": 43, "pieces": [" 字", " ", "\u000bm", "EOT", "​ḍ̇", " İ", "٣٤٥", "٦", "👍🏽'", "ſ", "Dža'T", "\r", ">'", "TDžDž", "!!", "٣٤٥", "٦", "\"<", "EOT", ">'", "VE"]} +{"text": "ḍ̇  \n é<|endoftext|>'ll<|endoftext|>é…\t \né\r\nm.'DéA#$%\u000bⅣ­", "tokens": 42, "pieces": ["ḍ̇", "  \n", " é", "<|", "endoftext", "|>'", "ll", "<|", "endoftext", "|>", "é", "…\t \n", "é", "\r\n", "m", ".'", "Dé", "A", "#$%", "\u000b", "Ⅳ", "­"]} +{"text": "'­ßع<|endoftext|>'llss\"٣٤٥٦é", "tokens": 19, "pieces": ["'­", "ßع", "<|", "endoftext", "|>'", "llss", "\"", "٣٤٥", "٦", "é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "Ⅳ!!éé<|endoftext|>m
Dž'D ḍ̇\"३<|endoftext|>'👍🏽́'́'VEa!\n'll㋿́EOT<|endoftext|>!!‍9ḍ̇", "tokens": 60, "pieces": ["Ⅳ", "!!", "éé", "<|", "endoftext", "|>", "m", "
Dž'D", " ḍ̇", "\"", "३", "<|", "endoftext", "|>'👍🏽́'́'", "VEa", "!\n", "'ll", "㋿́", "EOT", "<|", "endoftext", "|>!!‍", "9", "ḍ̇"]} +{"text": " ٣٤٥٦ſ!!ع𐞁-漢å漢३İ‍'M\"ع!A#$%EOTe9\u000b<|endoftext|>‍'VE ٣٤٥٦12345678 ée9>\r\n\r\n\r\n\r\n'S", "tokens": 65, "pieces": [" ", " ", "٣٤٥", "٦", "ſ", "!!", "ع𐞁", "-漢å漢", "३", "İ", "‍'", "M", "\"ع", "!A", "#$%", "EOTe", "9", "\u000b", "<|", "endoftext", "|>‍'", "VE", " ", "٣٤٥", "٦12", "345", "678", " ée", "9", ">\r\n\r\n\r\n\r\n", "<", "EOT", ">'", "S"]} +{"text": "'D½!é0- \r\n\r\nd😀🏽 9­#$%Z…éꟲ e'Reé>漢'S9", "tokens": 34, "pieces": ["'D", "½", "!é", "0", "-", " \r\n\r\n", "d", "😀🏽", " ", " ", "9", "­#$%", "Z", "…éꟲ", " e'Re", "é", ">漢'S", "9"]} +{"text": " \r\né t! ' #$%12345678!e12345678! \r", "tokens": 24, "pieces": [" \r\n", "é", " t", "!", " ", "'", " ", " #$%", "123", "456", "78", "!e", "123", "456", "78", "!", " \r"]} +{"text": "EOT!'ſ­\r\nع", "tokens": 7, "pieces": ["EOT", "!'", "ſ", "­\r\n", "ع"]} +{"text": "㍿9#$%(d!!Á
ع٣٤٥٦\"\"eß\u000b \n!!𐞁𐞁٣٤٥٦​👍🏽'll 
", "tokens": 50, "pieces": ["㍿", "9", "#$%(", "d", "!!", "Á", "
ع", "", "٣٤٥", "٦", "\"\"", "eß", "\u000b \n", "!!", "𐞁", "𐞁", "٣٤٥", "٦", "​👍🏽'", "ll", " 
"]} +{"text": ", EOT\n<|endoftext|>'ReEOT
ſ'ſ㍿İſ\r\n\r\n㍿½🙂<|endoftext|>'\r\n\r\nⅣꟲm", "tokens": 47, "pieces": [",", " EOT", "\n", "<|", "endoftext", "|>'", "Re", "EOT", "
ſ'ſ", "㍿İſ", "\r\n\r\n", "㍿", "½", "🙂<|", "endoftext", "|>'\r\n\r\n", "Ⅳ", "ꟲ", "m"]} +{"text": "ß \n<|endoftext|>A'llEOT \r åé9é👍🏽㍿㍿'Ḿm\"ꟲḍ̇漢'ſ12345678'T३ \n're'('D\u000b", "tokens": 57, "pieces": ["ß", " \n", "<|", "endoftext", "|>", "A'll", "EOT", " \r", " åé", "9", "é", "👍🏽㍿㍿'", "Ḿm", "\"ꟲḍ̇漢'ſ", "123", "456", "78", "'T", "३", " \n", "'re", "'('", "D", "\u000b"]} +{"text": "\n're😀🏽‍éḍ̇9㋿", "tokens": 15, "pieces": ["\n", "'re", "😀🏽‍", "éḍ̇", "9", "㋿"]} +{"text": "漢\u000b\r\n\r\n
 漢'redİ\r\n", "tokens": 10, "pieces": ["漢", "\u000b\r\n\r\n", "
", " 漢're", "d", "İ", "\r\n"]} +{"text": "½!e", "tokens": 3, "pieces": ["½", "!e"]} +{"text": " 'VE​t३", "tokens": 6, "pieces": [" ", "'VE", "​t", "३"]} +{"text": "'M,d㋿\r>as>!'re\tſ́<|fim_prefix|>#$%-fi𐞁½㋿́\ré", "tokens": 34, "pieces": ["'M", ",d", "㋿\r", ">as", ">!'", "re", "\tſ́", "<|", "fim", "_prefix", "|>#$%-", "fi𐞁", "½", "㋿́", "\r", "é"]} +{"text": "e,😀🏽9\r\n#$%'Dt👍🏽s'ReⅣ", "tokens": 21, "pieces": ["e", ",😀🏽", "9", "\r\n", "#$%'", "Dt", "👍🏽", "s'Re", "Ⅳ", ""]} +{"text": "…\"\"½#$%😀🏽's!!\"ea< \n", "tokens": 16, "pieces": ["…", "\"\"", "½", "#$%😀🏽'", "s", "!!\"", "ea", "<", " \n"]} +{"text": "'TⅣ're㋿\r\n\r\n🙂'Re<|endoftext|> 'reꟲ ḍ̇m😀🏽字'S'ReåA", "tokens": 41, "pieces": ["'T", "Ⅳ", "'re", "㋿\r\n\r\n", "🙂<", "META", "_START", ">'", "Re", "<|", "endoftext", "|>", " ", "'reꟲ", " ", " ḍ̇m", "😀🏽", "字'S", "'Reå", "A"]} +{"text": "'<漢…🙂<|endoftext|> ßé\r\n\r\nḍ̇Dž", "tokens": 24, "pieces": ["'<", "漢", "…", "🙂<", "META", "_START", "><|", "endoftext", "|>", " ßé", "\r\n\r\n", "ḍ̇", "Dž"]} +{"text": "a'fi.EOT( 'T-\r\n\r\n12345678….", "tokens": 16, "pieces": ["a", "'fi", ".EOT", "(", " '", "T", "-\r\n\r\n", "123", "456", "78", "…", "."]} +{"text": "­'Re<㋿", "tokens": 7, "pieces": ["­'", "Re", "<㋿"]} +{"text": " \n>😀🏽 -ع'S३
🙂ß,!'s<-'re'Mmm'TdEOT-­\r\n\r\n", "tokens": 33, "pieces": [" \n", "><", "META", "_START", ">😀🏽", " -", "ع'S", "३", "
", "🙂ß", ",!'", "s", "<-'", "re'M", "mm'T", "d", "EOT", "-­\r\n\r\n"]} +{"text": "å'Refie <|fim_prefix|>'TEOT a३d'sſ' \n!'reDž(߅(Ⅳeع", "tokens": 35, "pieces": ["å'Re", "fie", " ", "<|", "fim", "_prefix", "|>'", "TEOT", " a", "३", "d's", "ſ", "'", " \n", "!'", "re", "Dž", "(ß", "…", "(", "Ⅳ", "eع"]} +{"text": "ع‍'ſDž12345678😀🏽 \n #$%(s<|endoftext|>'VE", "tokens": 26, "pieces": ["ع", "‍'", "ſ", "Dž", "123", "456", "78", "😀🏽", " \n", " ", " #$%(", "s", "<|", "endoftext", "|>'", "VE"]} +{"text": "‍👍🏽", "tokens": 4, "pieces": ["‍👍🏽"]} +{"text": "å 字­👍🏽ésm٣٤٥٦Dže12345678\r\n,9İ'Re
A", "tokens": 27, "pieces": ["å", " 字", "­👍🏽", "ésm", "٣٤٥", "٦", "Dže", "123", "456", "78", "\r\n", ",", "9", "İ'Re", "
A"]} +{"text": "#$%Zd (\r\n,'ſ漢\r\nDž#$%‍A𐞁", "tokens": 19, "pieces": ["#$%", "Zd", " ", "(\r\n", ",'", "ſ漢", "\r\n", "Dž", "#$%‍", "A𐞁"]} +{"text": "(a'D's<|fim_prefix|>(Z åt😀🏽EOTe!\r\na'TEOT…!", "tokens": 28, "pieces": ["(a'D", "'s", "<|", "fim", "_prefix", "|>(", "Z", " ", " åt", "😀🏽", "EOTe", "!\r\n", "a'T", "EOT", "…", "!"]} +{"text": "\u000b'De\ns", "tokens": 5, "pieces": ["\u000b", "'De", "\n", "s"]} +{"text": "s", "tokens": 1, "pieces": ["s"]} +{"text": "'Refi EOTſ 𐞁\u000b字!'ReⅣ'ſ", "tokens": 20, "pieces": ["'Refi", " ", " EOTſ", " ", " 𐞁", "\u000b字", "!'", "Re", "Ⅳ", "'ſ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "- 'llAſ İ.\r\n\r\n𐞁#$%½9\u000b㋿\u000b ​m'St‍å0ſ… ", "tokens": 41, "pieces": ["-", " '", "ll", "Aſ", " <", "EOT", ">İ", ".\r\n\r\n", "𐞁", "#$%", "½9", "\u000b", "㋿", "\u000b", " ", "​m'S", "t", "‍å", "0", "ſ", "… "]} +{"text": "\r\n'Rete\u000bfi'M\rꟲ9'll㍿ḍ̇ és'9Dž‍'T!​", "tokens": 30, "pieces": ["\r\n", "'Rete", "\u000bfi'M", "\r", "ꟲ", "9", "'ll", "㍿ḍ̇", " ", " és", "'", "9", "Dž", "‍'", "T", "!​"]} +{"text": "‍A'\r\n…\u000b#$%.9'T12345678
\u000b mſ\r\n\r\n", "tokens": 23, "pieces": ["‍A", "'\r\n", "…", "\u000b", "#$%.", "9", "'T", "123", "456", "78", "
\u000b", " m", "ſ", "\r\n\r\n"]} +{"text": "🙂A, 12345678…㋿­३٣٤٥٦12345678A३\r#$%<​­å", "tokens": 36, "pieces": ["🙂A", ",", " ", "123", "456", "78", "…", "㋿­", "३٣٤", "٥٦1", "234", "567", "8", "A", "३", "\r", "#$%<​­", "å"]} +{"text": "㍿é'M\t'S!!s\n,aİZé12345678‍­('M\"å😀🏽'M<|endoftext|>fiḍ̇d \n𐞁", "tokens": 46, "pieces": ["㍿é'M", "\t", "'S", "!!", "s", "\n", ",a", "İZé", "123", "456", "78", "‍­('", "M", "\"å", "😀🏽'", "M", "<|", "endoftext", "|>", "fiḍ̇d", " \n", "𐞁"]} +{"text": " '😀🏽ḍ̇<|endoftext|><|endoftext|>Ⅳ\" 'ḍ̇🙂'S'S'll m'VE Ⅳ \n'🙂字'ſ'VE\rEOT\r‍‍!!\r9", "tokens": 58, "pieces": [" ", " '😀🏽", "ḍ̇", "<|", "endoftext", "|><|", "endoftext", "|>", "Ⅳ", "\"", " ", " '", "ḍ̇", "🙂'", "S'S", "'ll", " m'VE", " ", "Ⅳ", " \n", "'🙂", "字'ſ", "'VE", "\r", "EOT", "\r", "‍‍!!\r", "9"]} +{"text": "0<|fim_prefix|>(Dž \n㍿\u000bDž<…​漢A(,.12345678٣٤٥٦\"字​'Re㍿\r\n\r\n‍㍿", "tokens": 44, "pieces": ["0", "<|", "fim", "_prefix", "|>(", "Dž", " \n", "㍿", "\u000bDž", "<", "…", "​漢", "A", "(,.", "123", "456", "78٣", "٤٥٦", "\"字", "​'", "Re", "㍿\r\n\r\n", "‍㍿"]} +{"text": "漢३DžⅣ\n'ſ३fi \n 'Re'MA<|fim_prefix|>\r\n\r\n\r\n-ꟲ½!!<|endoftext|>½ 's½\r\n३ḍ̇\r\n\r\nſ
𐞁Ⅳa ‍٣٤٥٦", "tokens": 64, "pieces": ["漢", "३", "Dž", "Ⅳ", "\n", "'", "ſ", "३", "fi", " \n", " ", " '", "Re'M", "A", "<|", "fim", "_prefix", "|>\r\n\r\n\r\n", "-ꟲ", "½", "!!<|", "endoftext", "|>", "½", " ", " '", "s", "½", "\r\n", "३", "ḍ̇", "\r\n\r\n", "ſ", "
𐞁", "Ⅳ", "a", " ", "‍", "٣٤٥", "٦"]} +{"text": "\t\r\n<am
>('S're\n㋿'Re\u000b‍\"́ \n<ß0
t 're''½>­\"'S'reDž-", "tokens": 41, "pieces": ["\t", "\r\n", "<<", "META", "_START", ">am", "
", ">('", "S're", "\n", "㋿'", "Re", "\u000b", "‍\"́", " \n", "<ß", "0", "
t", " ", "'re", "''", "½", ">­\"'", "S're", "Dž", "-"]} +{"text": "\r\n<𐞁9'M​ß🙂'D", "tokens": 13, "pieces": ["\r\n", "<𐞁", "9", "'M", "​ß", "🙂'", "D"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "…ꟲ 𐞁'Té​Z(é\u000bA\n\u000b!\"!!#$%  ſ\r0're 'T(٣٤٥٦<|fim_prefix|>\r're<|endoftext|>12345678Ⅳ㍿\r 😀🏽", "tokens": 69, "pieces": ["…ꟲ", " 𐞁'T", "é", "​Z", "(é", "\u000bA", "\n", "\u000b", "!\"!!#$%", " ", " ſ", "\r", "0", "'re", " ", "'T", "(", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>\r", "'", "re", "<|", "endoftext", "|>", "123", "456", "78Ⅳ", "㍿\r", " 😀🏽"]} +{"text": "#$%eİ\nḍ̇", "tokens": 8, "pieces": ["#$%", "e", "İ", "\n", "ḍ̇"]} +{"text": "
-­😀🏽", "tokens": 6, "pieces": ["
", "-­😀🏽"]} +{"text": "(Ⅳß 'ſ-<|endoftext|>", "tokens": 15, "pieces": ["(", "Ⅳ", "ß", " ", "'ſ", "-<|", "endoftext", "|>"]} +{"text": "mt0#$%sA\nع 'T\r\n\r\nfiå(Džḍ̇İ \n\r", "tokens": 31, "pieces": ["mt", "0", "#$%", "s", "A", "\n", "ع", " ", "'T", "\r\n\r\n", "fiå", "(<", "EOT", ">Džḍ̇", "İ", " \n\r"]} +{"text": "İ㋿#$%EOT𐞁३­#$%­-ꟲåḍ̇ⅣⅣḍ̇t's漢漢ꟲs'M'VEs\r\n½ İEOT٣٤٥٦A'll'M12345678", "tokens": 61, "pieces": ["İ", "㋿#$%", "EOT𐞁", "३", "­#$%­-", "ꟲåḍ̇", "ⅣⅣ", "ḍ̇t's", "漢漢ꟲs'M", "'VEs", "\r\n", "½", " İEOT", "٣٤٥", "٦", "A'll", "'M", "123", "456", "78"]} +{"text": "ſd\r\n\r\ne!!12345678'a'Dſ漢  EOT<|endoftext|>t!! -',\n🙂<|endoftext|> <|endoftext|>", "tokens": 44, "pieces": ["ſd", "\r\n\r\n", "e", "!!", "123", "456", "78", "'a'D", "ſ漢", " ", " EOT", "<|", "endoftext", "|>", "t", "!!", " ", " -',\n", "🙂<|", "endoftext", "|>", " ", "<|", "endoftext", "|>"]} +{"text": "!.Z'Dfié", "tokens": 6, "pieces": ["!.", "Z'D", "fié"]} +{"text": "​Zé<#$%12345678d
EOT३́\r\n\r\né­!EOTå'T'T३
\r\n12345678
", "tokens": 33, "pieces": ["​Zé", "<#$%", "123", "456", "78", "d", "
EOT", "३", "́", "\r\n\r\n", "é", "­!", "EOTå'T", "'T", "३", "
\r\n", "123", "456", "78", "
"]} +{"text": "ſ🙂'lléع́>😀🏽'VEe#$%😀🏽\tfiDžé'Re‍🙂ß'<Ⅳ9​'M㋿\r\nd'S \t>\n", "tokens": 55, "pieces": ["ſ", "🙂'", "lléع́", ">😀🏽'", "VEe", "#$%😀🏽", "\tfi", "Džé'Re", "‍🙂", "ß", "'<", "Ⅳ9", "​'", "M", "㋿\r\n", "d'S", " ", "", "\t", ">\n"]} +{"text": "İas👍🏽'ſé'llع\tDžé12345678!é!<İ\r\nfi 'VE", "tokens": 37, "pieces": ["İ", "as", "👍🏽'", "ſé'll", "ع", "\tDžé", "123", "456", "78", "!é", "!<", "İ", "\r\n", "fi", " ", "'VE"]} +{"text": "Zdꟲ(EOT'T\n", "tokens": 8, "pieces": ["Zdꟲ", "(EOT'T", "\n"]} +{"text": "字'D­'M

'll>İ(½'s३!㍿‍İ's#$%ééaß're.'T \n\"\r\nß­٣٤٥٦ꟲfi😀🏽", "tokens": 46, "pieces": ["字'D", "­'", "M", "
", "
", "'ll", ">İ", "(", "½", "'s", "३", "!㍿‍", "İ's", "#$%", "ééaß're", ".'", "T", " \n", "\"\r\n", "ß", "­", "٣٤٥", "٦", "ꟲfi", "😀🏽"]} +{"text": "9('VEß 'T<|fim_prefix|><|endoftext|>🙂\t३>!!å'D㋿(", "tokens": 42, "pieces": ["9", "('", "VEß", "", " ", "'", "T", "<|", "fim", "_prefix", "|><|", "endoftext", "|>🙂", "\t", "३", ">!!", "å'D", "㋿("]} +{"text": "½'D́e'sßEOT >'ReDžt", "tokens": 17, "pieces": ["½", "'", "D́e's", "ß", "EOT", " ", ">'", "Re", "Džt"]} +{"text": "‍­", "tokens": 2, "pieces": ["‍­"]} +{"text": "<|endoftext|> 'M㋿ß\r\nⅣ㋿​'Re\ne\n 12345678 'Reḍ̇'VE\"ḍ̇", "tokens": 45, "pieces": ["<|", "endoftext", "|>", " '", "M", "㋿ß", "\r\n", "", "Ⅳ", "㋿​'", "Re", "\n", "e", "\n", " ", " ", "123", "456", "78", " ", "'Reḍ̇'VE", "\"ḍ̇"]} +{"text": "é'SZß\r\n\r\n\u000b<|endoftext|>!!  \u000b >(ḍ̇Džds😀🏽'll'", "SZß", "\r\n\r\n", "\u000b", "<|", "endoftext", "|>!!", "  \u000b ", " >(", "ḍ̇", "Džds", "😀🏽'", "ll", "३#$%a'ſ'S­
's字m", "tokens": 20, "pieces": ["३", "\"<|", "endoftext", "|>", "३", "#$%", "a'ſ", "'S", "­", "
", "'s字m"]} +{"text": "'re٣٤٥٦‍ḍ̇'S'ſ漢m…३ſꟲ‍½  'ſ<|fim_prefix|>́'Re'Reع\té,ḍ̇.ß \n字'ſ", "tokens": 53, "pieces": ["'re", "٣٤٥", "٦", "‍ḍ̇'S", "'ſ漢m", "…", "३", "ſꟲ", "‍", "½", " ", " ", "'ſ", "<|", "fim", "_prefix", "|>́'", "Re'Re", "ع", "\té", ",ḍ̇", ".ß", " \n", "字'ſ", ""]} +{"text": "-d(\u000b\"३fi३ \n🙂\r'T'M", "tokens": 12, "pieces": ["-d", "(", "\u000b", "\"", "३", "fi", "३", " \n", "🙂\r", "'T'M"]} +{"text": "\r\n字!\n0٣٤٥٦👍🏽㍿ⅣZ𐞁ß㍿\" é,Z 
३'M", "tokens": 35, "pieces": ["\r\n", "字", "!\n", "0٣٤", "٥٦", "👍🏽㍿", "Ⅳ", "Z𐞁ß", "㍿\"", " é", ",Z", " ", "
", "३", "'M"]} +{"text": ">'d12345678\"…ßⅣ\n12345678Džع𐞁<|endoftext|><|endoftext|>\r'llEOTd½ 'T😀🏽0'​EOT\"٣٤٥٦㋿'ll12345678'll'll,'D", "tokens": 72, "pieces": ["><", "EOT", ">'", "d", "123", "456", "78", "\"", "…ß", "Ⅳ", "\n", "123", "456", "78", "Džع𐞁", "<|", "endoftext", "|><|", "endoftext", "|>\r", "'ll", "EOTd", "½", " ", " '", "T", "😀🏽", "0", "'​", "EOT", "\"", "٣٤٥", "٦", "㋿'", "ll", "123", "456", "78", "'ll'll", ",'", "D"]} +{"text": "­", "tokens": 1, "pieces": ["­"]} +{"text": "Džå\u000b'Re🙂é>'re", "tokens": 11, "pieces": ["Džå", "\u000b", "'Re", "🙂é", ">'", "re"]} +{"text": "٣٤٥٦d,\n'll'sعfi \r\n!!é!\t'Sfi😀🏽(­", "tokens": 23, "pieces": ["٣٤٥", "٦", "d", ",\n", "'ll's", "عfi", " \r\n", "!!", "é", "!", "\t", "'Sfi", "😀🏽(­"]} +{"text": "🙂'sAå \ts漢Z'll-éع", "tokens": 14, "pieces": ["🙂'", "s", "Aå", " ", "\ts漢", "Z'll", "-éع"]} +{"text": "'Re½\r\nm", "tokens": 4, "pieces": ["'Re", "½", "\r\n", "m"]} +{"text": ">'s!! 漢३<|endoftext|>Aعa>!!<|endoftext|>\r!s\r\n‍ Dž \nⅣ's#$% \r\né٣٤٥٦<|endoftext|>'D0Z\t", "tokens": 63, "pieces": [">'", "s", "!!<", "META", "_START", ">", " 漢", "३", "<|", "endoftext", "|>", "Aعa", ">!!<|", "endoftext", "|>\r", "!s", "\r\n", "‍", " ", " Dž", " \n", "Ⅳ", "'s", "#$%", " \r\n", "é", "٣٤٥", "٦", "<|", "endoftext", "|>'", "D", "0", "Z", "\t"]} +{"text": "'s,ßDž'Re'!!!e9'Då\r\nİ \n-㍿d\u000b'T", "tokens": 23, "pieces": ["'s", ",ß", "Dž'Re", "'!!!", "e", "9", "'Då", "\r\n", "İ", " \n", "-㍿", "d", "\u000b", "'T"]} +{"text": "\u000b'T9 😀🏽'Té\n're\t́ß!'M <…
­😀🏽Ⅳa", "tokens": 29, "pieces": ["\u000b", "'T", "9", " ", "😀🏽'", "Té", "\n", "'re", "\t́ß", "!'", "M", " ", "<", "…", "
", "­😀🏽", "Ⅳ", "a"]} +{"text": "ḍ̇ꟲ‍m ­ 'Re㋿,'M ㍿-EOT‍́ \nع\n'ḾDž \n\"é'Red½éEOT'S", "tokens": 44, "pieces": ["ḍ̇ꟲ", "‍m", " ", " ­", " ", "'Re", "㋿,'", "M", " ", " ㍿-", "EOT", "‍́", " \n", "ع", "\n", "'Ḿ", "Dž", " \n", "\"é'Re", "d", "½", "é", "EOT'S"]} +{"text": "'Re \n's­ع㋿字é \nEOT'Re\" 'M😀🏽EOTḍ̇0<|fim_prefix|>漢-!३½Ⅳ𐞁ꟲ's", "tokens": 52, "pieces": ["'Re", " \n", "'s", "­ع", "㋿字é", " \n", "EOT'Re", "\"", " ", "'M", "😀🏽", "EOTḍ̇", "0", "<|", "fim", "_prefix", "|>", "漢", "-!", "३½Ⅳ", "𐞁ꟲ's"]} +{"text": "fi!fi<‍A‍́😀🏽\r\n\r\n'll\"A漢'Re<|fim_prefix|><|fim_prefix|><‍㍿", "tokens": 32, "pieces": ["fi", "!fi", "<‍", "A", "‍́", "😀🏽\r\n\r\n", "'ll", "\"A漢'Re", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|><‍㍿"]} +{"text": "👍🏽'VE'Reé👍🏽d\" \n-\"𐞁d<|endoftext|>\r\n­EOT(12345678३'reA㋿", "tokens": 49, "pieces": ["👍🏽'", "VE'Re", "é", "👍🏽", "d", "\"", " \n", "-\"", "𐞁", "d", "<|", "endoftext", "|>\r\n", "­", "EOT", "(", "123", "456", "78३", "'re", "A", "㋿"]} +{"text": "Aå🙂.A㋿½EOTꟲ12345678İ'M's", "tokens": 25, "pieces": ["Aå", "🙂.", "A", "㋿<", "META", "_START", ">", "½", "EOTꟲ", "123", "456", "78", "İ'M", "'s"]} +{"text": "-Džt,#$%EOT's0ꟲ
👍🏽mé३ꟲ'S. ع字ß\"''re
å'S \u000bé ", "tokens": 44, "pieces": ["-Džt", ",#$%", "EOT's", "0", "ꟲ", "
", "👍🏽", "mé", "३", "ꟲ'S", ".<", "META", "_START", ">", " ", " ع字ß", "\"''", "re", "
å'S", " ", "\u000bé", " "]} +{"text": "dDž", "tokens": 11, "pieces": ["", "३", " ", " <", "META", "_START", ">d", "Dž"]} +{"text": "‍'ReDžfiİé'ſ\r'ſꟲA٣٤٥٦👍🏽e½\r\né.'M'", "tokens": 32, "pieces": ["‍'", "Re", "Džfi", "İé'ſ", "\r", "'ſꟲ", "A", "٣٤٥", "٦", "👍🏽", "e", "½", "\r\n", "é", ".'", "M", "'"]} +{"text": "Ⅳ ́0 \n>a

a½'<|fim_prefix|>𐞁ع(t\"<|fim_prefix|>'ſDžt", "tokens": 37, "pieces": ["", "Ⅳ", " ́", "0", " \n", ">a", "
", "
a", "½", "'<|", "fim", "_prefix", "|>", "𐞁ع", "(t", "\"<|", "fim", "_prefix", "|>'", "ſ", "Džt"]} +{"text": "\r\n\r\n​३<|endoftext|>­'ll\u000b s㋿a
", "tokens": 21, "pieces": ["\r\n\r\n", "​", "३", "<|", "endoftext", "|>­'", "ll", "\u000b", " s", "㋿a", "
"]} +{"text": "\u000b<|endoftext|>0\n9𐞁fi'Så\u000b‍>㍿é𐞁m İtéé (12345678", "tokens": 44, "pieces": ["\u000b", "<|", "endoftext", "|>", "0", "\n", "9", "𐞁fi'S", "å", "\u000b", "‍>㍿", "é𐞁m", " İ", "téé", " ", "(", "123", "456", "78"]} +{"text": "½'ſfi👍🏽‍\u000b\r\nEOT \r\n\r\nfi'M12345678\u000bdA9 \"٣٤٥٦<<|fim_prefix|> ", "tokens": 39, "pieces": ["", "½", "'ſfi", "👍🏽‍", "\u000b\r\n", "EOT", " \r\n\r\n", "fi'M", "123", "456", "78", "\u000bd", "A", "9", " \"", "٣٤٥", "٦", "<<|", "fim", "_prefix", "|>", " "]} +{"text": "(\n漢EOT😀🏽", "tokens": 7, "pieces": ["(\n", "漢", "EOT", "😀🏽"]} +{"text": "ſ㍿aZ ß'Re'é…​AعⅣéfi12345678½", "tokens": 24, "pieces": ["ſ", "㍿a", "Z", " ß'Re", "'é", "…", "​Aع", "Ⅳ", "éfi", "123", "456", "78½"]} +{"text": "İ'D㍿ \n字é‍ßd😀🏽ḍ̇a>'ſ漢 >12345678ß<>'ll fi'😀🏽<|fim_prefix|>", "tokens": 42, "pieces": ["İ'D", "㍿", " \n", "字é", "‍ßd", "😀🏽", "ḍ̇a", ">'", "ſ漢", " >", "123", "456", "78", "ß", "<>'", "ll", " fi", "'😀🏽<|", "fim", "_prefix", "|>"]} +{"text": "\"'M\t\r\ns12345678.éé\"\tß‍\"!!é- 
a#$%ß  \n 'VE", "tokens": 31, "pieces": ["'Re", "½", "👍🏽", "½", "'ſſ", "!<|", "endoftext", "|>\"!!", "é", "-", " ", "
a", "#$%", "ß", "  \n", " ", " '", "VE"]} +{"text": "㍿!A字Dž", "tokens": 8, "pieces": ["㍿!", "A字", "Dž"]} +{"text": "٣٤٥٦'Sꟲ\r\n\r\n😀🏽é(!>'re字é㋿e
-́​😀🏽́\r\néDž<|fim_prefix|>𐞁​​'llå\"😀🏽", "tokens": 57, "pieces": ["٣٤٥", "٦", "'Sꟲ", "\r\n\r\n", "😀🏽", "é", "(!>'", "re字é", "㋿e", "
", "-́", "​😀🏽́\r\n", "é", "Dž", "<|", "fim", "_prefix", "|>", "𐞁", "​​'", "llå", "\"😀🏽"]} +{"text": "…

…‍Ⅳ'D'ſ'll.½fi\r\n\r\n-9<|endoftext|> 漢!!", "tokens": 30, "pieces": ["…

", "…", "‍", "Ⅳ", "'D'ſ", "'ll", ".", "½", "fi", "\r\n\r\n", "-", "9", "<|", "endoftext", "|>", " ", " 漢", "!!"]} +{"text": "'D́
's​EOT٣٤٥٦İꟲ\r\n\r\nA 9dß'Sm<|endoftext|>😀🏽
's", "tokens": 49, "pieces": ["'D́", "
", "'s", "​EOT", "٣٤٥", "٦", "İꟲ", "\r\n\r\n", "A", " ", " ", "9", "d", "", "ß'S", "m", "<|", "endoftext", "|>😀🏽", "
", "'s"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ع\n'D(ß㍿‍-'D😀🏽ß½Z'll\u000b'reİ", "tokens": 24, "pieces": ["ع", "\n", "'D", "(ß", "㍿‍-'", "D", "😀🏽", "ß", "½", "Z", "'", "ll", "\u000b", "'re", "İ"]} +{"text": "Ⅳ!!👍🏽Ⅳ🙂ſ ㋿!>½\rß½ ''D<fi'ſ", "tokens": 33, "pieces": ["Ⅳ", "!!👍🏽", "Ⅳ", "🙂ſ", " ", "㋿!>", "½", "\r", "ß", "", "½", " ", " ''", "D", "<fi'ſ"]} +{"text": "́EOT'D<‍!'S9字🙂\nå<|fim_prefix|>.a\t\r\né!!0<|endoftext|>Ⅳİ'll!\" 0", "tokens": 40, "pieces": ["́", "EOT'D", "<‍!'", "S", "9", "字", "🙂\n", "å", "<|", "fim", "_prefix", "|>.", "a", "\t\r\n", "é", "!!", "0", "<|", "endoftext", "|>", "Ⅳ", "İ'll", "!\"", " ", "0"]} +{"text": "<9\u000b'ſmⅣ'MßtA😀🏽ꟲ'reع ع'M<,ſİé'ſ-\r\n9A­漢 9👍🏽.३mt", "tokens": 46, "pieces": ["<", "9", "\u000b", "'ſm", "Ⅳ", "'Mßt", "A", "😀🏽", "ꟲ're", "ع", " ع'M", "<,", "ſ", "İé'ſ", "-\r\n", "9", "A", "­漢", " ", "9", "👍🏽<", "EOT", ">.", "३", "mt"]} +{"text": "0'M", "tokens": 2, "pieces": ["0", "'M"]} +{"text": "åe𐞁
\r,ß'VEa12345678ḍ̇a'Ree.> 
́😀🏽'T㋿Z,'ſ\r\n 𐞁åé\t", "tokens": 54, "pieces": ["åe𐞁", "
\r", ",ß'VE", "a", "123", "456", "78", "ḍ̇a'Re", "e", ".>", " ", "
́", "😀🏽'", "T", "㋿Z", ",'", "ſ", "\r\n", " ", " 𐞁åé", "\t"]} +{"text": "'ſ́é \n½\n's𐞁>a​ꟲfi-Z'ſ\r\n㋿ é \n
Dž…́'M…­éⅣ😀🏽字ſ're'Re­d'VE", "tokens": 52, "pieces": ["'ſ́é", " \n", "½", "\n", "'s𐞁", ">a", "​ꟲfi", "-Z'ſ", "\r\n", "㋿", " é", " \n", "
Dž", "…́'M", "…", "­é", "Ⅳ", "😀🏽", "字ſ're", "'Re", "­d'VE"]} +{"text": "\r\n\r\nt 😀🏽ßfi'reé'ſe\u000b漢!ß 'Reſ'Mع<|endoftext|>ſ Afi-s字㋿", "tokens": 40, "pieces": ["\r\n\r\n", "t", " ", "😀🏽", "ßfi're", "é'ſ", "e", "\u000b漢", "!ß", " ", "'Reſ'M", "ع", "<|", "endoftext", "|>", "ſ", " ", " Afi", "-s字", "㋿"]} +{"text": "\n👍🏽!!9'Re😀🏽", "tokens": 10, "pieces": ["\n", "👍🏽!!", "9", "'Re", "😀🏽"]} +{"text": "<|fim_prefix|>½'½ 'D", "tokens": 14, "pieces": ["<|", "fim", "_prefix", "|>", "½", "'", "½", " ", "'", "D"]} +{"text": "٣٤٥٦Ⅳ", "tokens": 6, "pieces": ["٣٤٥", "٦Ⅳ"]} +{"text": "#$% \nꟲ\"A.😀🏽 9're( ٣٤٥٦ ㍿😀🏽", "tokens": 29, "pieces": ["#$%", " \n", "ꟲ", "\"A", ".😀🏽", " ", " ", "9", "'re", "(", " ", "٣٤٥", "٦", " ", "㍿😀🏽"]} +{"text": "'re'T 😀🏽½EOTſ\r\n\r\n㋿912345678<|endoftext|>'ReeA \n12345678Z,\n'reḍ̇ḍ̇\n", "tokens": 41, "pieces": ["'re'T", " ", "😀🏽", "½", "EOTſ", "\r\n\r\n", "㋿", "912", "345", "678", "<|", "endoftext", "|>'", "Ree", "A", " \n", "123", "456", "78", "Z", ",\n", "'reḍ̇ḍ̇", "\n"]} +{"text": "​.ſ", "tokens": 2, "pieces": ["​.", "ſ"]} +{"text": "㍿Ⅳ\téfia\t́,''M> İ३", "tokens": 17, "pieces": ["㍿", "Ⅳ", "\téfia", "\t́", ",''", "M", ">", " ", " İ", "३"]} +{"text": ">👍🏽e㋿\né​ ३('D", "tokens": 19, "pieces": [">👍🏽", "e", "㋿\n", "é", "​", " ", "३", "('", "D"]} +{"text": "\n'>'Ddé🙂12345678å'Mع0 9 ३㍿!!", "tokens": 25, "pieces": ["\n", "'>'", "Ddé", "🙂", "123", "456", "78", "å'M", "ع", "0", " ", " ", "9", " ", " ", "३", "㍿!!"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "İåfiå \n­\r\n\r\nt​.ꟲ­-'llꟲ\r\né!!DžA!d🙂\" a'D \né‍ßde(fi \n🙂", "tokens": 44, "pieces": ["İåfiå", " \n", "­\r\n\r\n", "t", "​.", "ꟲ", "­-'", "llꟲ", "\r\n", "é", "!!", "DžA", "!d", "🙂\"", " a'D", " \n", "é", "‍ßde", "(fi", " \n", "🙂"]} +{"text": "\n­\nꟲ½e\n😀🏽…عİ \n㍿!\"'Re", "tokens": 22, "pieces": ["\n", "­\n", "ꟲ", "½", "e", "\n", "😀🏽", "…ع", "İ", " \n", "㍿!\"'", "Re"]} +{"text": "e><<|fim_prefix|>å'Dع9<'VEé'D…½३! 'S😀🏽's👍🏽", "tokens": 33, "pieces": ["e", "><<|", "fim", "_prefix", "|>", "å'D", "ع", "9", "<'", "VEé'D", "…", "½३", "!", " ", "'S", "😀🏽'", "s", "👍🏽"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": " \u000b'M\nEOT  \n<|fim_prefix|>'SEOTſ'll'VE‍Dž'll字'\r9.'reع ", "tokens": 34, "pieces": [" ", "\u000b", "'M", "\n", "EOT", "  \n", "<|", "fim", "_prefix", "|>'", "SEOTſ'll", "'VE", "‍Dž'll", "字", "'\r", "9", ".'", "reع", " "]} +{"text": "t (12345678😀🏽字½ع'DⅣ👍🏽㍿\"👍🏽<|endoftext|>İfiꟲⅣ(", "tokens": 43, "pieces": ["t", " ", "(", "123", "456", "78", "😀🏽", "字", "½", "ع'D", "Ⅳ", "👍🏽㍿\"👍🏽<", "EOT", "><|", "endoftext", "|>", "İfiꟲ", "Ⅳ", "("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\n \n12345678👍🏽İ'llZ!!ſ\u000b\"<㋿Džtİ'ReİAa👍🏽'Så\t<½,. d'T", "tokens": 49, "pieces": ["\n \n", "123", "456", "78", "👍🏽", "İ'll", "Z", "!!", "ſ", "\u000b", "\"<㋿", "Dž", "t", "İ'Re", "İAa", "👍🏽'", "Så", "\t", "<", "½", ",.", " d", "'", "T"]} +{"text": "ḍ̇㍿ḍ̇", "tokens": 9, "pieces": ["ḍ̇", "㍿ḍ̇"]} +{"text": "'VE㍿EOT\nİ", "tokens": 9, "pieces": ["'VE", "㍿EOT", "\n", "İ"]} +{"text": "Ⅳ́ 's \n", "tokens": 6, "pieces": ["Ⅳ", "́", " '", "s", " \n"]} +{"text": "(", "tokens": 9, "pieces": ["å", "<|", "endoftext", "|>("]} +{"text": "!‍ \n'Tſ9<𐞁Ⅳ\rt'sfi!!", "tokens": 18, "pieces": ["!‍", " \n", "'Tſ", "9", "<𐞁", "Ⅳ", "\r", "t's", "fi", "!!"]} +{"text": "ḍ̇‍A'Mſİع́İ0fi", "tokens": 16, "pieces": ["ḍ̇", "‍A'M", "ſ", "İع́", "İ", "0", "fi"]} +{"text": "ß👍🏽 ꟲ.'D­字\"😀🏽'll>ع< ſ \r\n\r\n
𐞁😀🏽Z‍٣٤٥٦e", "tokens": 42, "pieces": ["ß", "👍🏽", " ꟲ", ".'", "D", "­字", "\"😀🏽'", "ll", ">", "ع", "<", " ſ", " \r\n\r\n", "
𐞁", "😀🏽", "Z", "‍", "٣٤٥", "٦", "e"]} +{"text": "㋿\"'sꟲ. åéé(😀🏽", "tokens": 17, "pieces": ["㋿\"'", "sꟲ", ".", " åéé", "(😀🏽"]} +{"text": "!", "tokens": 1, "pieces": ["!"]} +{"text": "<ḍ̇\" 'Re!!#$%​!
😀🏽'T", "tokens": 22, "pieces": ["<ḍ̇", "\"", " ", "'Re", "!!#$%​!", "
", "😀🏽'", "T"]} +{"text": "'VE'T­​A #$%t​ \r\n\r\nꟲm0å'३!\tfit<|fim_prefix|>\r12345678'D.'VE<|fim_prefix|>'re", "tokens": 46, "pieces": ["'VE'T", "­​", "A", " ", "#$%", "t", "​", " \r\n\r\n", "ꟲm", "0", "å", "'", "३", "!", "\tfit", "<|", "fim", "_prefix", "|>\r", "123", "456", "78", "'D", ".'", "VE", "<|", "fim", "_prefix", "|>'", "re"]} +{"text": "ZZß👍🏽٣٤٥٦d'D漢𐞁\r\n'll'ſ…😀🏽!!\"㍿", "tokens": 33, "pieces": ["ZZ", "ß", "👍🏽", "٣٤٥", "٦", "d'D", "漢𐞁", "\r\n", "'ll'ſ", "…", "😀🏽!!\"㍿"]} +{"text": "½'Mdꟲ \"9Afi's'D'ſåé
\r\n\n!ḍ̇\"\r\n'T0ḍ̇👍🏽漢 漢12345678👍🏽ع!\"", "tokens": 47, "pieces": ["½", "'Mdꟲ", " ", "\"", "9", "Afi", "'", "s'D", "'ſåé", "
\r\n\n", "!ḍ̇", "\"\r\n", "'T", "0", "ḍ̇", "👍🏽", "漢", " 漢", "123", "456", "78", "👍🏽", "ع", "!\""]} +{"text": "EOT\r\n\r\n'MZ'D<|endoftext|>ḍ̇'T\"'ſ<|endoftext|>", "tokens": 29, "pieces": ["EOT", "\r\n\r\n", "'", "MZ'D", "<|", "endoftext", "|>", "ḍ̇'T", "\"'", "ſ", "<|", "endoftext", "|>"]} +{"text": "İ\rß>t٣٤٥٦éfi", "tokens": 12, "pieces": ["İ", "\r", "ß", ">t", "٣٤٥", "٦", "éfi"]} +{"text": " ſ!!!\r\n\r\n㍿'TDž३é!'Ś'll9३ḍ̇", "tokens": 27, "pieces": [" ſ", "!!!\r\n\r\n", "㍿'", "TDž", "३", "é", "!'", "S", "́'ll", "9३", "ḍ̇"]} +{"text": "\"#$%é🙂fi", "tokens": 6, "pieces": ["\"#$%", "é", "🙂fi"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "(٣٤٥٦३𐞁\r-'D́'漢'ſعEOT.ſt\u000b'…-<|fim_prefix|>😀🏽٣٤٥٦\"12345678ꟲs𐞁\u000b<|fim_prefix|>ḍ̇EOT-,", "tokens": 70, "pieces": ["(", "٣٤٥", "٦३", "𐞁", "\r", "-'", "D́", "'漢'ſ", "ع", "EOT", ".ſt", "\u000b", "'", "…", "-<|", "fim", "_prefix", "|>😀🏽", "٣٤٥", "٦", "\"", "123", "456", "78", "ꟲs𐞁", "", "\u000b", "<|", "fim", "_prefix", "|>", "ḍ̇", "EOT", "-,"]} +{"text": "-­(Ⅳe>Zع,", "tokens": 13, "pieces": ["-­(", "Ⅳ", "e", ">Z", "ع", ","]} +{"text": "㋿\u000b㍿\r\nİ'\r😀🏽'ſ😀🏽a!!'ſ\t>.漢٣٤٥٦ 's㋿9.ås \n\r\n\r\n0 \n'll\"㋿<|endoftext|>m", "tokens": 60, "pieces": ["㋿", "\u000b", "㍿\r\n", "İ", "'\r", "😀🏽'", "ſ", "😀🏽", "a", "!!'", "ſ", "\t", ">.", "漢", "٣٤٥", "٦", " ", " <", "META", "_START", ">'", "s", "㋿", "9", ".ås", " \n\r\n\r\n", "0", " \n", "'ll", "\"㋿<|", "endoftext", "|>", "m"]} +{"text": "ſ eⅣt're漢<ḍ̇'Dd'DEOT're​", "tokens": 25, "pieces": ["ſ", " e", "Ⅳ", "t're", "漢", "<ḍ̇'D", "d", "'", "D", "EOT're", "​"]} +{"text": "\u000bEOT00<|endoftext|>‍Z,ḍ̇12345678<|fim_prefix|>sEOTs­am <|fim_prefix|>å👍🏽sA漢", "tokens": 50, "pieces": ["\u000bEOT", "00", "<|", "endoftext", "|>‍", "Z", ",<", "EOT", ">ḍ̇", "123", "456", "78", "<|", "fim", "_prefix", "|>", "s", "EOTs", "­am", " ", " <|", "fim", "_prefix", "|>", "å", "👍🏽", "s", "A漢"]} +{"text": "<|endoftext|>9<ꟲEOTⅣfi!tⅣ ß\r\n\r\n​-३'Re", "tokens": 29, "pieces": ["<|", "endoftext", "|>", "9", "<ꟲ", "EOT", "Ⅳ", "fi", "!t", "Ⅳ", " ", " ß", "\r\n\r\n", "​-", "३", "'Re"]} +{"text": "'D😀🏽 'D0٣٤٥٦\r9m🙂\r \n‍", "tokens": 18, "pieces": ["'D", "😀🏽", " '", "D", "0٣٤", "٥٦", "\r", "9", "m", "🙂\r", " \n", "‍"]} +{"text": "é'M🙂Dž
,<㍿'ll\"\t字aḍ̇字<|fim_prefix|>'ſ!!d🙂<|fim_prefix|>İ'VE .‍'s㍿'VE\t12345678İ\u000bⅣé", "tokens": 67, "pieces": ["é'M", "🙂<", "META", "_START", ">Dž", "
", ",<㍿'", "ll", "\"", "\t字aḍ̇字", "<|", "fim", "_prefix", "|>'", "ſ", "!!", "d", "🙂<|", "fim", "_prefix", "|>", "İ'VE", " ", " .‍'", "s", "㍿'", "VE", "\t", "123", "456", "78", "İ", "\u000b", "Ⅳ", "é"]} +{"text": "0, å漢 ſ", "tokens": 9, "pieces": ["0", ",", " å漢", " ", " ſ"]} +{"text": "'s<‍\n're‍🙂 \n#$% 'Re'Ms😀🏽s\r'llt…fiḍ̇'ll​ßİ0…\"<ß‍", "tokens": 43, "pieces": ["'s", "<‍\n", "'re", "‍🙂", " \n", "#$%", " ", "'Re'M", "s", "😀🏽", "s", "\r", "'ll", "t", "…fiḍ̇'ll", "​ß", "İ", "0", "…", "\"<", "ß", "‍"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "9'SDž<|fim_prefix|> 𐞁<|endoftext|>'ll !>㍿\u000b", "tokens": 30, "pieces": ["9", "'SDž", "<|", "fim", "_prefix", "|>", " 𐞁", "<|", "endoftext", "|>'", "ll", " ", " !>㍿", "\u000b"]} +{"text": "㍿EOT'ſ㋿\u000b\r\n\r\n 's \nd \nع🙂0!!ſDž'T>éd३é,Ⅳt<​‍İ12345678\r\n\r\n'S\r", "tokens": 51, "pieces": ["㍿EOT'ſ", "㋿", "\u000b\r\n\r\n", " ", "'s", " \n", "d", " \n", "ع", "🙂", "0", "!!", "ſ", "Dž'T", ">éd", "३", "é", ",", "Ⅳ", "t", "<​<", "EOT", ">‍", "İ", "123", "456", "78", "\r\n\r\n", "'S", "\r", ""]} +{"text": "㍿-㋿٣٤٥٦d🙂>t 's\r\n\r\nZ.'VEé\r\nDž å👍🏽‍ ", "tokens": 33, "pieces": ["㍿-㋿", "٣٤٥", "٦", "d", "🙂>", "t", " '", "s", "\r\n\r\n", "Z", ".'", "VEé", "\r\n", "Dž", " ", " å", "👍🏽‍", " "]} +{"text": "́('D…😀🏽ß\r\n 'll'Re'sⅣ\r\n", "tokens": 17, "pieces": ["́", "('", "D", "…", "😀🏽", "ß", "\r\n", " '", "ll'Re", "'s", "Ⅳ", "\r\n"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "½‍😀🏽'll'Mfiå< 12345678­👍🏽å
ḍ̇\r\ń㍿'Reå​ḍ̇", "tokens": 42, "pieces": ["", "½", "‍😀🏽'", "ll'M", "fiå", "<", " ", "123", "456", "78", "­👍🏽", "å", "
ḍ̇", "\r\n", "́", "㍿'", "Reå", "​ḍ̇"]} +{"text": ",ß\r\n'S<ſée‍tt", "tokens": 9, "pieces": [",ß", "\r\n", "'S", "<ſée", "‍tt"]} +{"text": "ſEOT<|fim_prefix|>\u000b0 ꟲ<|fim_prefix|>İ\t <|endoftext|>​<|endoftext|>t,9", "tokens": 41, "pieces": ["ſ", "EOT", "<|", "fim", "_prefix", "|>", "\u000b", "0", " ", " ꟲ", "<|", "fim", "_prefix", "|>", "İ", "\t ", " <|", "endoftext", "|>​<|", "endoftext", "|>", "t", ",", "9"]} +{"text": "\r\n\r\nع३", "tokens": 3, "pieces": ["\r\n\r\n", "ع", "३"]} +{"text": "#$%\r‍­Džḍ̇'T\nḍ̇🙂>!!é'VEé.-𐞁­­A🙂ſ'Sſ\t'ſع字>", "tokens": 47, "pieces": ["#$%\r", "‍­", "Džḍ̇'T", "\n", "ḍ̇", "🙂>!!", "é'VE", "é", ".-", "𐞁", "­­", "A", "🙂ſ'S", "ſ", "", "\t", "'", "ſع字", ">"]} +{"text": "𐞁\r\n½d \n'Reséé'ſſ\"­'llⅣİ😀🏽<|endoftext|>fi", "tokens": 34, "pieces": ["𐞁", "\r\n", "½", "d", " \n", "'Reséé'ſ", "ſ", "\"­'", "ll", "Ⅳ", "İ", "😀🏽<|", "endoftext", "|>", "fi"]} +{"text": "å<|fim_prefix|>'T<\n\na👍🏽\rſ#$%'ll\r<|endoftext|>-'T(\u000b'm0‍ß'ReDž", "tokens": 43, "pieces": ["å", "<|", "fim", "_prefix", "|>'", "T", "<\n\n", "a", "👍🏽\r", "ſ", "#$%'", "ll", "\r", "<|", "endoftext", "|>-'", "T", "(", "\u000b", "'m", "0", "‍ß", "'", "Re", "Dž"]} +{"text": "ßع👍🏽m-'ſ'('M", "tokens": 11, "pieces": ["ßع", "👍🏽", "m", "-'", "ſ", "'('", "M"]} +{"text": "‍EOTꟲ٣٤٥٦", "tokens": 10, "pieces": ["‍EOTꟲ", "٣٤٥", "٦"]} +{"text": "'M½!!\n​ \nⅣꟲ́ ㋿३Ⅳꟲ<|fim_prefix|>'  stfi<|fim_prefix|>'VEm då\n​", "tokens": 43, "pieces": ["'M", "½", "!!\n", "​", " \n", "Ⅳ", "ꟲ́", " ", " ㋿", "३Ⅳ", "ꟲ", "<|", "fim", "_prefix", "|>'", "  ", " stfi", "<|", "fim", "_prefix", "|>'", "VEm", " då", "\n", "​"]} +{"text": "👍🏽‍½t", "tokens": 10, "pieces": ["👍🏽‍<", "EOT", ">", "½", "t"]} +{"text": "\r\n,s!!( 'M ꟲ\"‍😀🏽 ​😀🏽EOT'reé", "tokens": 24, "pieces": ["\r\n", ",s", "!!(", " ", "'M", " ꟲ", "\"‍😀🏽", " ", "​😀🏽", "EOT're", "é"]} +{"text": "'M- é \n字Dž'll9…EOTḍ̇eİaⅣ\nfi'llA'S漢.٣٤٥٦½​9.…\n", "tokens": 44, "pieces": ["'M", "-", " é", " \n", "字", "Dž'll", "9", "…EOTḍ̇e", "İa", "Ⅳ", "\n", "fi'll", "A'S", "漢", ".", "٣٤٥", "٦½", "​<", "EOT", ">", "9", ".", "…\n"]} +{"text": "e!🙂'Reſ<ſ́👍🏽㍿'sm👍🏽 0eAm", "tokens": 25, "pieces": ["e", "!🙂'", "Reſ", "<ſ́", "👍🏽㍿'", "sm", "👍🏽", " ", "0", "e", "Am"]} +{"text": "\n0'<|endoftext|>'ll\u000b­(Ad́", "tokens": 16, "pieces": ["\n", "0", "'<|", "endoftext", "|>'", "ll", "\u000b", "­(", "Ad́"]} +{"text": "ſ\t", "tokens": 2, "pieces": ["ſ", "\t"]} +{"text": "‍½m ́​Dž", "tokens": 8, "pieces": ["‍", "½", "m", " ́", "​Dž"]} +{"text": "ḍ̇ fiEOT", "tokens": 10, "pieces": ["ḍ̇", " fi", "EOT"]} +{"text": ".a(İ
d'D…ع're字㍿😀🏽…'ll字😀🏽>#$%\r\n\r\nm<|fim_prefix|><|endoftext|>!'Mfi,-.é>­'S9dİ", "tokens": 55, "pieces": [".a", "(İ", "
d'D", "…ع're", "字", "㍿😀🏽", "…", "'ll字", "😀🏽>#$%\r\n\r\n", "m", "<|", "fim", "_prefix", "|><|", "endoftext", "|>!'", "Mfi", ",-.", "é", ">­'", "S", "9", "d", "İ"]} +{"text": "ß🙂ع('s'Da👍🏽å'T'M ́عfiZİ'M", "tokens": 21, "pieces": ["ß", "🙂ع", "('", "s'D", "a", "👍🏽", "å'T", "'M", " ́عfi", "Zİ'M"]} +{"text": "ع'VE'sZ'\tİ​ 😀🏽", "tokens": 17, "pieces": ["ع'VE", "'s", "Z", "'", "\t", "İ", "​", " ", "😀🏽"]} +{"text": "\r३'s'VEfiꟲ>", "tokens": 10, "pieces": ["\r", "३", "'s'VE", "fiꟲ", ">"]} +{"text": "👍🏽\tEOT漢'Re . tEOT\u000bß​\n\t!Ⅳ👍🏽\r\nع.a𐞁'VEZ
漢字é'ſ", "tokens": 48, "pieces": ["👍🏽", "\tEOT漢'Re", " ", " <", "META", "_START", ">", " .", " t", "EOT", "\u000bß", "​\n", "\t", "!", "Ⅳ", "👍🏽\r\n", "ع", ".a", "𐞁'VE", "Z", "
漢字é'ſ"]} +{"text": "d<|fim_prefix|>\r\n\r\n漢'Re\"t عéå-0 字''ſ'D́9
  EOT", "tokens": 29, "pieces": ["d", "<|", "fim", "_prefix", "|>\r\n\r\n", "漢'Re", "\"t", " عéå", "-", "0", " 字", "''", "ſ'D", "́", "9", "
 ", " EOT"]} +{"text": " t !", "tokens": 4, "pieces": [" ", " t", " ", " !"]} +{"text": " 👍🏽 Ⅳ\n", "tokens": 8, "pieces": [" ", " 👍🏽", " ", "Ⅳ", "\n"]} +{"text": "<|fim_prefix|>漢s'Me'T'll<|endoftext|><|endoftext|>e!!9㋿,'reⅣ'VEee‍'T>dfiḍ̇İ'lld३", "tokens": 57, "pieces": ["<|", "fim", "_prefix", "|>", "漢s'M", "e'T", "'ll", "<|", "endoftext", "|><|", "endoftext", "|>", "e", "!!", "9", "㋿,'", "re", "Ⅳ", "'VEee", "‍'", "T", ">dfi", "ḍ̇", "İ", "'", "lld", "३"]} +{"text": "½a\"\t😀🏽ſé٣٤٥٦漢\u000b,\t9Ⅳ'T\r\n", "tokens": 26, "pieces": ["½", "a", "\"", "\t", "😀🏽", "ſé", "٣٤٥", "٦", "漢", "\u000b", ",", "\t", "9", "", "Ⅳ", "'T", "\r\n"]} +{"text": "t! 😀🏽ſⅣ \n ꟲ  m'll漢㍿'漢!!12345678ſd́\t🙂ꟲ\"'Re<|endoftext|>", "tokens": 47, "pieces": ["t", "!", " 😀🏽", "ſ", "Ⅳ", " \n", " ", " ꟲ", "  ", " m'll", "漢", "㍿'", "漢", "!!", "123", "456", "78", "ſd́", "\t", "🙂ꟲ", "\"'", "Re", "<|", "endoftext", "|>"]} +{"text": "e\né'T<😀🏽ꟲ.\u000ba \nA'llfi'll\r\n\r\n'll,\rDžeع<|fim_prefix|>👍🏽ét'(t'Re'ſ#$%Z👍🏽​'reZ", "tokens": 54, "pieces": ["e", "\n", "é'T", "<😀🏽", "ꟲ", ".", "\u000ba", " \n", "A'll", "fi'll", "\r\n\r\n", "'ll", ",\r", "Džeع", "<|", "fim", "_prefix", "|>👍🏽", "ét", "'(", "t'Re", "'", "ſ", "#$%", "Z", "👍🏽​'", "re", "Z"]} +{"text": "́!!ⅣmEOT", "tokens": 7, "pieces": ["́", "!!", "Ⅳ", "m", "EOT"]} +{"text": "!👍🏽 İ\nß>", "tokens": 9, "pieces": ["!👍🏽", " İ", "\n", "ß", ">"]} +{"text": "ſ\r\n\r\n'll(𐞁're㋿'M", "tokens": 14, "pieces": ["ſ", "\r\n\r\n", "'ll", "(𐞁're", "㋿'", "M"]} +{"text": "-(½'Rea!\r\n('Re", "tokens": 7, "pieces": ["-(", "½", "'Rea", "!\r\n", "('", "Re"]} +{"text": "!Zع\r\n­'Da漢ZİDž' #$%\r\n", "tokens": 16, "pieces": ["!Zع", "\r\n", "­'", "Da漢", "ZİDž", "'", " ", "#$%\r\n"]} +{"text": "ſع \n'ſ#$%!é­s\u000b​ß<字m字𐞁éⅣ'T.<<|endoftext|>'ſ́½9­-(🙂'reſ\n12345678🙂", "tokens": 48, "pieces": ["ſع", " \n", "'ſ", "#$%!", "é", "­s", "\u000b", "​ß", "<字m字𐞁é", "Ⅳ", "'T", ".<<|", "endoftext", "|>'", "ſ́", "½9", "­-(🙂'", "reſ", "\n", "123", "456", "78", "🙂"]} +{"text": "åEOT \nA", "tokens": 6, "pieces": ["å", "EOT", " \n", "A"]} +{"text": "'sd'T  \n漢<>'llas(ḍ̇漢​㍿३字­<<|endoftext|>,…ß㍿㍿", "tokens": 41, "pieces": ["'sd'T", "  \n", "漢", "<>'", "llas", "(ḍ̇漢", "​㍿", "३", "字", "­<<|", "endoftext", "|>,", "…ß", "㍿㍿<", "META", "_START", ">"]} +{"text": "#$%\t½㍿tßEOT\u000b\r\nZ.'llé३>", "tokens": 19, "pieces": ["#$%", "\t", "½", "㍿tß", "EOT", "\u000b\r\n", "Z", ".'", "llé", "३", ">"]} +{"text": "!!👍🏽👍🏽eİ
­é'll𐞁'll!!<|fim_prefix|>0'Mé \r\n\r\ń㍿<|fim_prefix|>́'ſ'SⅣ-ZAt", "tokens": 50, "pieces": ["!!👍🏽👍🏽", "e", "İ", "
", "­é'll", "𐞁'll", "!!<|", "fim", "_prefix", "|>", "0", "'Mé", " \r\n\r\n", "́", "㍿<|", "fim", "_prefix", "|>́'", "ſ'S", "Ⅳ", "-ZAt"]} +{"text": " Dž𐞁 \n字EOT​ḍ̇é𐞁t'Re\u000bm \r\n\r\n'M12345678(<|endoftext|>fiZ.'VE,ꟲ㍿\r\n\r\n🙂'll", "tokens": 57, "pieces": [" ", " Dž𐞁", " \n", "字", "EOT", "​ḍ̇é𐞁t'Re", "\u000bm", "", " \r\n\r\n", "'M", "123", "456", "78", "(<|", "endoftext", "|>", "fi", "Z", ".'", "VE", ",ꟲ", "㍿\r\n\r\n", "🙂'", "ll"]} +{"text": "12345678
عEOTİ,'VE👍🏽Džfi,'VE'👍🏽Ⅳ", "tokens": 27, "pieces": ["", "123", "456", "78", "
ع", "EOTİ", ",'", "VE", "👍🏽", "Džfi", ",'", "VE", "'👍🏽", "Ⅳ"]} +{"text": "< 0s", "tokens": 4, "pieces": ["<", " ", "0", "s"]} +{"text": "ꟲ", "tokens": 3, "pieces": ["ꟲ"]} +{"text": "é'S漢.d'Z!!\ta\u000bA<Ⅳ­😀🏽 é'T\t㋿\"0'ſDž< ", "tokens": 40, "pieces": ["é'S", "漢", ".d", "'Z", "!!", "\ta", "\u000bA", "<", "Ⅳ", "­<", "META", "_START", ">😀🏽", " é'T", "", "\t", "㋿\"", "0", "'ſ", "Dž", "<", " "]} +{"text": "'re𐞁åm'D \n12345678漢ع<😀🏽\r\n\r\n٣٤٥٦'Seꟲ\r\n\r\n!!<|endoftext|>\t‍'VE<漢٣٤٥٦­ꟲ e'M", "tokens": 55, "pieces": ["'re𐞁åm'D", " \n", "123", "456", "78", "漢ع", "<😀🏽\r\n\r\n", "٣٤٥", "٦", "'Seꟲ", "\r\n\r\n", "!!<|", "endoftext", "|>", "\t", "‍'", "VE", "<漢", "٣٤٥", "٦", "­ꟲ", " e'M"]} +{"text": ",🙂\r\n\r\nعİ ßḍ̇😀🏽12345678 'ſ'", "tokens": 22, "pieces": [",🙂\r\n\r\n", "ع", "İ", " ßḍ̇", "😀🏽", "123", "456", "78", " '", "ſ", "'"]} +{"text": "s\n ­👍🏽90'Re𐞁A>", "tokens": 14, "pieces": ["s", "\n", " ­👍🏽", "90", "'Re𐞁", "A", ">"]} +{"text": "ꟲ(<|endoftext|>", "tokens": 10, "pieces": ["ꟲ", "(<|", "endoftext", "|>"]} +{"text": "‍eEOT12345678'Dꟲ👍🏽ꟲ漢e İ!'Ⅳ­.e٣٤٥٦字\u000b🙂0'Sع'ſ e'Dꟲte", "tokens": 46, "pieces": ["‍e", "EOT", "123", "456", "78", "'Dꟲ", "👍🏽", "ꟲ漢e", " İ", "!'", "Ⅳ", "­.", "e", "٣٤٥", "٦", "字", "\u000b", "🙂", "0", "'Sع'ſ", " ", " e'D", "ꟲte"]} +{"text": "A<|endoftext|>\rEOTé!!é>ḍ̇åé
\nmfi \r\n", "tokens": 27, "pieces": ["A", "<|", "endoftext", "|>\r", "EOTé", "!!", "é", ">ḍ̇åé", "
\n", "mfi", " \r\n"]} +{"text": "'sḍ̇<|fim_prefix|>'٣٤٥٦'re
…ꟲe ٣٤٥٦'VE's\u000b '", "٣٤٥", "٦", "'re", "
", "…ꟲe", " ", "٣٤٥", "٦", "'VE's", "\u000b", " <", "Zİ", "½", "\t", "(㋿", "A", "!ß", "\"", "३", "EOT", "!!🙂\r\n", "'re", "İ", " \n", " ", " !'", "T"]} +{"text": "!! 😀🏽- ​s​A\r\nع‍ß\t字'Re' ſ<'Re\"👍🏽\t>\n<|endoftext|>😀🏽ß\u000b", "tokens": 40, "pieces": ["!!", " ", "😀🏽-", " ​", "s", "​A", "\r\n", "ع", "‍ß", "\t字'Re", "'", " ſ", "<'", "Re", "\"👍🏽", "\t", ">\n", "<|", "endoftext", "|>😀🏽", "ß", "\u000b"]} +{"text": "🙂-<|endoftext|>‍𐞁(", "tokens": 15, "pieces": ["🙂-<|", "endoftext", "|>‍", "𐞁", "("]} +{"text": "0\n字\ré!👍🏽'ſ'llEOT0<|endoftext|>.😀🏽\rꟲⅣⅣⅣ㍿'T9🙂٣٤٥٦'VEꟲé­​Ⅳ
", "tokens": 58, "pieces": ["0", "\n", "字", "\r", "é", "!👍🏽'", "ſ'll", "EOT", "0", "<|", "endoftext", "|>.😀🏽\r", "ꟲ", "ⅣⅣⅣ", "㍿'", "T", "9", "🙂", "٣٤٥", "٦", "'VEꟲé", "­​", "Ⅳ", "
"]} +{"text": "A…fi٣٤٥٦ \r\n\r\n🙂!'s…'Red\r\nZ字#$%🙂12345678ꟲm!!漢'T½𐞁#$%", "tokens": 44, "pieces": ["A", "…fi", "٣٤٥", "٦", " \r\n\r\n", "🙂!'", "s", "…", "'Red", "\r\n", "Z字", "#$%🙂", "123", "456", "78", "ꟲm", "!!", "漢'T", "½", "𐞁", "#$%<", "EOT", ">"]} +{"text": "0!a.字9'M'llḍ̇𐞁9漢'ſ>fi\t'T", "tokens": 27, "pieces": ["0", "!a", ".字", "9", "'M", "'", "llḍ̇𐞁", "9", "漢'ſ", ">fi", "\t", "'T"]} +{"text": "👍🏽Ⅳ\"!İ0,aİ'D字Ⅳ<ꟲ\t½\u000b 😀🏽", "tokens": 32, "pieces": ["👍🏽", "Ⅳ", "\"!", "İ", "0", ",a", "İ'D", "字", "Ⅳ", "<ꟲ", "\t", "½", "<", "EOT", ">", "\u000b", " ", "😀🏽"]} +{"text": "s\"mA12345678å ́🙂'Tع👍🏽'M…(漢ꟲ- 'M\n9", "tokens": 36, "pieces": ["s", "\"m", "A", "123", "456", "78", "å", " ́", "🙂'", "T", "ع", "👍🏽'", "M", "…", "(漢ꟲ", "-", " ", "'M", "\n", "9"]} +{"text": ">ꟲ٣٤٥٦\n漢.ß漢Ⅳ0عå 'ſ'ReEOT'T
​-(12345678, 's🙂stꟲ", "tokens": 41, "pieces": [">ꟲ", "٣٤٥", "٦", "\n", "漢", ".ß漢", "Ⅳ0", "عå", " ", " '", "ſ'Re", "EOT'T", "
", "​-(", "123", "456", "78", ",", " ", " '", "s", "🙂stꟲ"]} +{"text": "ß>12345678🙂<|fim_prefix|>\"<|endoftext|>ع <|fim_prefix|>>0'SEOT  <|endoftext|><|fim_prefix|>漢३,é字é‍\r\n're漢e(d(#$%ßå\u000b", "tokens": 68, "pieces": ["ß", ">", "123", "456", "78", "🙂<|", "fim", "_prefix", "|>\"<|", "endoftext", "|>", "ع", " ", " <|", "fim", "_prefix", "|>>", "0", "'SEOT", " ", " ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "漢", "३", ",é字é", "‍\r\n", "'re漢e", "(d", "(#$%", "ßå", "\u000b", ""]} +{"text": "fi İDž\r\n
0.'T'VE'Dḍ̇\r\n\r\nt a­𐞁#$%🙂.'", "T'VE", "'Dḍ̇", "\r\n\r\n", "t", " a", "­𐞁", "#$%🙂<", "dé", "(𐞁", "½", "s", "\r\n\r\n", "s"]} +{"text": "!!<'ſå \r\n𐞁fiZ 12345678'reİ( \nḍ̇३#$%'S'ſ\r,m­㍿.\n\r\n​\n'VE \n'ſ", "tokens": 43, "pieces": ["!!<'", "ſå", " \r\n", "𐞁fi", "Z", " ", "123", "456", "78", "'re", "İ", "(", " \n", "ḍ̇", "३", "#$%'", "S'ſ", "\r", ",m", "­㍿.\n\r\n", "​\n", "'VE", " \n", "'ſ"]} +{"text": "å\r'VE\u000b\u000b0>m'sa字a0٣٤٥٦.'S0\t", "tokens": 23, "pieces": ["å", "\r", "'VE", "\u000b", "\u000b", "0", ">m's", "a字a", "0٣٤", "٥٦", ".'", "S", "0", "\t"]} +{"text": " Z!!'D<|fim_prefix|>­ å('VÉ", "tokens": 21, "pieces": [" Z", "!!'", "D", "<|", "fim", "_prefix", "|>­", " å", "('", "VÉ"]} +{"text": "… 'T'VE👍🏽!‍", "tokens": 12, "pieces": ["… ", " '", "T'VE", "👍🏽!‍"]} +{"text": "\r\n\r\n(𐞁ꟲ \u000b'VE㋿👍🏽", "tokens": 19, "pieces": ["\r\n\r\n", "(𐞁ꟲ", " ", "\u000b", "'VE", "㋿👍🏽"]} +{"text": "​s'llé", "tokens": 9, "pieces": ["​s'll", "é", ""]} +{"text": " \u000b 'VE‍m!!!éå. 're'M", "tokens": 19, "pieces": [" \u000b", " '", "VE", "‍<", "EOT", ">m", "!!!", "éå", ".", " '", "re'M"]} +{"text": "३Ⅳ0\n\r\n\r\n,å", "tokens": 11, "pieces": ["३Ⅳ0", "\n\r\n\r\n", ",å", ""]} +{"text": "ms's\r\nd\r\n\r\n12345678'S''Té(A字'VE\tet(
 a
é12345678'D𐞁\r\r\né", "tokens": 37, "pieces": ["ms's", "\r\n", "d", "\r\n\r\n", "123", "456", "78", "'S", "''", "Té", "(A字'VE", "\tet", "(", "
", " a", "
é", "123", "456", "78", "'D𐞁", "\r\r\n", "é"]} +{"text": " fi👍🏽 'll
< '
'ſ'S \ne👍🏽𐞁‍'sꟲḍ̇", "tokens": 36, "pieces": ["", " ", " fi", "👍🏽", " ", "'ll", "
", "<", " ", "'", "
", "'ſ'S", " \n", "e", "👍🏽", "𐞁", "‍'", "sꟲḍ̇"]} +{"text": "'re-㍿\nå-㍿\r\"𐞁!!½🙂
Dž𐞁#$%ſꟲ,\n!!😀🏽'll'Re𐞁\r\n…e👍🏽", "tokens": 53, "pieces": ["'re", "-㍿\n", "å", "-㍿\r", "\"𐞁", "!!", "½", "🙂", "
Dž𐞁", "#$%", "ſꟲ", ",\n", "!!😀🏽'", "ll'Re", "𐞁", "\r\n", "…e", "👍🏽"]} +{"text": "½A><|endoftext|>½'M>Ⅳع\t>字é🙂 \n! d\u000b ,​t㋿漢>\r\n", "tokens": 38, "pieces": ["½", "A", "><|", "endoftext", "|>", "½", "'M", ">", "Ⅳ", "ع", "\t", ">字é", "🙂", " \n", "!", " d", "\u000b ", " ,​", "t", "㋿漢", ">\r\n"]} +{"text": "'٣٤٥٦ßꟲ漢 'Re​ \n … ​'VE\"0İ12345678 …0'T𐞁", "tokens": 35, "pieces": ["'", "٣٤٥", "٦", "ßꟲ漢", " '", "Re", "​", " \n", " …", " ​'", "VE", "\"", "0", "İ", "123", "456", "78", " ", "…", "0", "'T𐞁"]} +{"text": "e漢'S ", "tokens": 4, "pieces": ["e漢'S", " "]} +{"text": "ḍ̇ß 
\n", "tokens": 11, "pieces": ["ḍ̇ß", " 
\n"]} +{"text": "ſA", "tokens": 2, "pieces": ["ſ", "A"]} +{"text": "\n­Dž٣٤٥٦,.
'D'ſ s'EOT'字e😀🏽0aEOT'M㍿'S\"㋿'VEع", "tokens": 42, "pieces": ["\n", "­Dž", "٣٤٥", "٦", ",.", "
", "'D'ſ", " s", "'EOT", "'字e", "😀🏽", "0", "a", "EOT'M", "㍿'", "S", "\"㋿'", "VE", "ع"]} +{"text": "字😀🏽9ḍ̇EOTZ'll\t३३'T
s­İ­12345678'VE漢(\r\n\r\n漢İ're㋿ع'VEİ漢👍🏽Aa", "tokens": 53, "pieces": ["字", "😀🏽", "9", "ḍ̇", "EOTZ'll", "\t", "३३", "'T", "
s", "­İ", "­", "123", "456", "78", "'VE漢", "(\r\n\r\n", "漢", "İ", "'", "re", "㋿ع'VE", "İ漢", "👍🏽<", "META", "_START", ">Aa"]} +{"text": "­\"…‍.
d#$%'ſ'S'll'VE.ſ12345678𐞁's<|fim_prefix|> 'ſ12345678d<|endoftext|>\"!!​ å𐞁DžⅣ\r", "tokens": 65, "pieces": ["­\"", "…", "‍.", "
d", "#$%'", "ſ'S", "'ll'VE", ".ſ", "123", "456", "78", "𐞁's", "<|", "fim", "_prefix", "|>", " ", " '", "ſ", "123", "456", "78", "d", "<|", "endoftext", "|>\"<", "EOT", "><", "META", "_START", ">!!​", " ", " å𐞁", "Dž", "Ⅳ", "\r"]} +{"text": "å'SéꟲDžⅣ३㍿m\r\n😀🏽Ⅳ'Reḍ̇㍿ ٣٤٥٦", "tokens": 37, "pieces": ["å'S", "éꟲ", "Dž", "Ⅳ३", "㍿m", "\r\n", "😀🏽", "Ⅳ", "'Reḍ̇", "㍿", " ", "٣٤٥", "٦"]} +{"text": "\r\n>ée'S<|endoftext|>ḍ̇A字<|fim_prefix|>fi", "tokens": 25, "pieces": ["\r\n", ">ée'S", "<|", "endoftext", "|>", "ḍ̇", "A字", "<|", "fim", "_prefix", "|>", "fi"]} +{"text": "\n'VE…\"ſß's'ſ<|fim_prefix|>\r", "tokens": 18, "pieces": ["\n", "'VE", "…", "\"ſß's", "'ſ", "<|", "fim", "_prefix", "|>\r"]} +{"text": "'Dm​'VEåع12345678!! 👍🏽\n \né", "tokens": 19, "pieces": ["'Dm", "​'", "VEåع", "123", "456", "78", "!!", " ", "👍🏽\n", " \n", "é"]} +{"text": "字-0'VE,­ع#$%'D'T\r\n-fi", "tokens": 19, "pieces": ["字", "-", "0", "'VE", ",­", "ع", "#$%'", "D'T", "\r\n", "-fi"]} +{"text": "'ll>🙂漢eḍ̇'M\n\r\n‍ 👍🏽", "tokens": 15, "pieces": ["'ll", ">🙂", "漢eḍ̇'M", "\n\r\n", "‍", " ", "👍🏽"]} +{"text": "12345678éſ漢!\n9­'ſ>ḍ̇'́ ", "tokens": 19, "pieces": ["123", "456", "78", "éſ漢", "!\n", "9", "­'", "ſ", ">ḍ̇", "'́", " "]} +{"text": "\t,漢>' dDž>\rDžZꟲ­'ll \n Ⅳ d's0'Re'D​", "tokens": 30, "pieces": ["\t", ",漢", ">'", " d", "Dž", ">\r", "DžZꟲ", "­'", "ll", " \n", " ", "Ⅳ", " ", " d's", "0", "'Re'D", "​"]} +{"text": "\r\n\r\n-'S !!-\u000b‍", "tokens": 8, "pieces": ["\r\n\r\n", "-'", "S", " ", " !!-", "\u000b", "‍"]} +{"text": "é​ꟲ'll‍'re…\rع😀🏽tꟲ\t\"a🙂m,\r\n\r\n
३½ \nt,'ſDžع👍🏽𐞁EOTfiß#$%Dž'll'M ", "tokens": 53, "pieces": ["é", "​ꟲ'll", "‍'", "re", "…\r", "ع", "😀🏽", "tꟲ", "\t", "\"a", "🙂m", ",\r\n\r\n", "
", "३½", " \n", "t", ",'", "ſ", "Džع", "👍🏽", "𐞁EOTfiß", "#$%", "Dž'll", "'M", " "]} +{"text": "́㍿'llt'll\"'S 字", "tokens": 11, "pieces": ["́", "㍿'", "llt'll", "\"'", "S", " 字"]} +{"text": "'M\t é>'D㋿İ", "tokens": 10, "pieces": ["'M", "\t", " é", ">'", "D", "㋿İ"]} +{"text": "\"é>t'D<|endoftext|>", "tokens": 12, "pieces": ["\"é", ">t'D", "<|", "endoftext", "|>"]} +{"text": "A😀🏽'ſ#$%'re'll#$%​ ㍿  ", "tokens": 22, "pieces": ["A", "😀🏽'", "ſ", "#$%'", "re", "'", "ll", "#$%​", " ㍿", "  "]} +{"text": "ꟲꟲfi< m漢9EOTs𐞁
'll㋿𐞁.a.Ⅳ… \n<|fim_prefix|>٣٤٥٦🙂½é'Reſ", "tokens": 57, "pieces": ["ꟲꟲfi", "<<", "EOT", ">", " m漢", "9", "EOTs𐞁", "
", "'ll", "㋿𐞁", ".a", ".", "Ⅳ", "… \n", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "🙂", "½", "é", "'", "Reſ"]} +{"text": "'S'D'se'ſ9Ⅳſ🙂㍿Z\r\n­\u000b", "tokens": 18, "pieces": ["'S'D", "'se'ſ", "9Ⅳ", "ſ", "🙂㍿", "Z", "\r\n", "­", "\u000b"]} +{"text": "!\n,(a…aEOT𐞁'VE0'", "VE", "0", "#$%\"", "tokens": 16, "pieces": ["'llſ", "\r", "123", "456", "78", "d", "<|", "endoftext", "|>#$%\""]} +{"text": "'", "tokens": 1, "pieces": ["'"]} +{"text": "㋿A漢s'Tع
'Mꟲḍ̇é(- åaſ​Džt\tA'S!漢 -漢 ‍<|endoftext|>😀🏽'D", "tokens": 50, "pieces": ["㋿A漢s'T", "ع", "
", "'Mꟲḍ̇é", "(-", " åaſ", "​Džt", "\tA'S", "!漢", " ", " -", "漢", " ", " ‍<|", "endoftext", "|>😀🏽'", "D"]} +{"text": "'ll'VE३ß 0字­ 'ReİeEOTⅣ\nZ​", "tokens": 21, "pieces": ["'ll'VE", "३", "ß", " ", "0", "字", "­", " ", " '", "Re", "İe", "EOT", "Ⅳ", "\n", "Z", "​"]} +{"text": "İAſ #$%㋿'M", "tokens": 11, "pieces": ["İAſ", " ", "#$%㋿'", "M"]} +{"text": "\"", "tokens": 1, "pieces": ["\""]} +{"text": "m#$%>'Mm.ZZ👍🏽㋿'T
㍿e\"㋿٣٤٥٦9\tḍ̇0½", "tokens": 39, "pieces": ["m", "#$%>'", "Mm", ".ZZ", "👍🏽<", "META", "_START", ">㋿'", "T", "
", "㍿e", "\"㋿", "٣٤٥", "٦9", "\tḍ̇", "0½"]} +{"text": " \n\t(½EOTßt漢\t㋿EOTß\r\n\r\n㍿'𐞁\u000ba½🙂", "tokens": 28, "pieces": [" \n", "\t", "(", "½", "EOTßt漢", "\t", "㋿EOTß", "\r\n\r\n", "㍿'", "𐞁", "\u000ba", "½", "🙂"]} +{"text": "9漢'SDž𐞁㍿ع'ſ३Ⅳ(é½👍🏽\t\r\n,", "tokens": 26, "pieces": ["9", "漢'S", "Dž𐞁", "㍿ع'ſ", "३Ⅳ", "(é", "½", "👍🏽", "\t\r\n", ","]} +{"text": "'ſ😀🏽#$%\r\n­<|fim_prefix|><> EOT>​éⅣ'D\r'D\tt­'s 'Re>\"‍EOT#$%\r\nḍ̇fi d\r\n\r\n😀🏽", "tokens": 54, "pieces": ["'ſ", "😀🏽#$%\r\n", "­<|", "fim", "_prefix", "|><>", " ", " EOT", ">​", "é", "Ⅳ", "'", "D", "\r", "'D", "\tt", "­'", "s", " ", " '", "Re", ">\"‍", "EOT", "#$%\r\n", "ḍ̇fi", " ", " d", "\r\n\r\n", "😀🏽"]} +{"text": "'Ree'sé(#$%Dž🙂́'ſ‍", "tokens": 14, "pieces": ["'Ree's", "é", "(#$%", "Dž", "🙂́'ſ", "‍"]} +{"text": "ع𐞁eA\r\n\r\nfi‍\" 'ſ'VE'sſ'Re.Dž𐞁å'll<|endoftext|>,😀🏽㋿👍🏽ḍ̇<|endoftext|>t .㍿<<🙂
", "tokens": 69, "pieces": ["ع𐞁e", "A", "\r\n\r\n", "fi", "‍\"", " ", "'ſ", "'", "VE's", "ſ'Re", ".Dž𐞁å'll", "<|", "endoftext", "|>,😀🏽㋿👍🏽", "ḍ̇", "<|", "endoftext", "|>", "t", " ", " <", "META", "_START", ">.㍿<<🙂", "
"]} +{"text": "'ll>'s
\r,𐞁<|fim_prefix|>Z afi9字're字 ḍ̇<|endoftext|>字\r\n\r\n'sA0 d\n#$%12345678‍s'VEm字…", "tokens": 60, "pieces": ["'ll", ">'", "s", "", "
\r", ",𐞁", "<|", "fim", "_prefix", "|>", "Z", " ", " afi", "9", "字're", "字", " ḍ̇", "<|", "endoftext", "|>", "字", "\r\n\r\n", "'s", "A", "0", " ", " d", "\n", "#$%", "123", "456", "78", "‍s'VE", "m字", "…"]} +{"text": "A\n-漢Dž👍🏽٣٤٥٦'ſ'D३ 漢 é", "tokens": 22, "pieces": ["A", "\n", "-漢", "Dž", "👍🏽", "٣٤٥", "٦", "'ſ'D", "३", " 漢", " é"]} +{"text": "'S#$%'ß(😀🏽,é½EOTꟲ'VE\r'SⅣ12345678<(ßt'D‍ 'M'Ret'VE", "tokens": 36, "pieces": ["'S", "#$%'", "ß", "(😀🏽,", "é", "½", "EOTꟲ'VE", "\r", "'S", "Ⅳ12", "345", "678", "<(", "ßt'D", "‍", " ", "'M'Re", "t'VE"]} +{"text": "'re字'M(\u000b're𐞁'Ś9Dž\r\n\r\nDž \n('reꟲḍ̇
\rs \n", "tokens": 34, "pieces": ["'re字'M", "(", "\u000b", "'re", "𐞁'S", "́", "9", "Dž", "\r\n\r\n", "Dž", " \n", "('", "reꟲḍ̇", "
\r", "s", " \n"]} +{"text": "​!!İA'M!!'sfimⅣ ꟲ👍🏽t👍🏽…é0é's>å 'ſ<|endoftext|>٣٤٥٦éA Dž(  Dž㍿́ḍ̇😀🏽", "tokens": 66, "pieces": ["​!!", "İA'M", "!!'", "sfim", "Ⅳ", " ꟲ", "👍🏽", "t", "👍🏽", "…é", "0", "é's", ">å", " ", "'ſ", "<|", "endoftext", "|>", "٣٤٥", "٦", "é", "A", " Dž", "(", " ", " Dž", "㍿́ḍ̇", "😀🏽"]} +{"text": "🙂\t😀🏽dA-😀🏽éⅣ́\r\n\r\n👍🏽 ſ́ \n- m\" 'Sd", "tokens": 31, "pieces": ["🙂", "\t", "😀🏽", "d", "A", "-😀🏽", "é", "Ⅳ", "́", "\r\n\r\n", "👍🏽", " ſ́", " \n", "-", " ", " m", "\"", " ", "'Sd"]} +{"text": "😀🏽!!éꟲEOT́ꟲ😀🏽'T𐞁½…👍🏽'EOT.ßa𐞁a", "tokens": 40, "pieces": ["😀🏽!!", "éꟲ", "EOT́ꟲ", "😀🏽'", "T𐞁", "½", "…", "👍🏽'", "EOT", ".ßa𐞁a"]} +{"text": "!!t.99٣٤٥٦Ⅳع,漢字\u000bİ9\u000b", "tokens": 18, "pieces": ["!!", "t", ".", "99٣", "٤٥٦", "Ⅳ", "ع", ",漢字", "\u000bİ", "9", "\u000b"]} +{"text": ".\"A\r>\u000bte<|endoftext|>#$%<|fim_prefix|>,eeå'Re👍🏽 s'Re 'Reſ\n>12345678#$%漢… \n<|fim_prefix|>\t", "tokens": 52, "pieces": [".\"", "A", "\r", ">", "\u000bte", "<|", "endoftext", "|>#$%<|", "fim", "_prefix", "|>,", "eeå'Re", "👍🏽", " s'Re", " ", " '", "Reſ", "\n", ">", "123", "456", "78", "#$%", "漢", "… \n", "<|", "fim", "_prefix", "|>", "\t"]} +{"text": "'VE're​​\u000b\n\n'> 'll\u000b𐞁 ß' t㍿m٣٤٥٦‍t0­ İ👍🏽Dž‍t\r", "tokens": 44, "pieces": ["'VE're", "​​", "\u000b\n\n", "'>", " '", "ll", "\u000b𐞁", " ß", "'", " ", " t", "㍿", "m", "٣٤٥", "٦", "‍t", "0", "­", " ", " İ", "👍🏽", "Dž", "‍t", "\r"]} +{"text": " ㍿㍿\nſ‍!'0👍🏽", "tokens": 15, "pieces": [" ", "㍿㍿\n", "ſ", "‍!'", "0", "👍🏽"]} +{"text": "ḍ̇ḍ̇'M字sḍ̇\t#$%t,'VEéß३0'Sé'll‍👍🏽,s", "tokens": 33, "pieces": ["ḍ̇ḍ̇'M", "字sḍ̇", "\t", "#$%", "t", ",'", "VEéß", "३0", "'Sé'll", "‍👍🏽,", "s"]} +{"text": "㍿sfi३Z
…😀🏽!!\"漢12345678'sfiZd字Dž𐞁Zå0.d9aß\"<|fim_prefix|>'12345678㋿字Z's're", "tokens": 55, "pieces": ["㍿sfi", "३", "Z", "", "
", "…", "😀🏽!!\"", "漢", "123", "456", "78", "'sfi", "Zd字", "Dž𐞁Zå", "0", ".d", "9", "aß", "\"<|", "fim", "_prefix", "|>'", "123", "456", "78", "㋿字", "Z's", "'re"]} +{"text": " \"'VE're 9ḍ̇éeⅣ'VE!!㍿\t.", "tokens": 22, "pieces": [" ", " \"'", "VE're", " ", "9", "ḍ̇ée", "Ⅳ", "'VE", "!!㍿", "\t", "."]} +{"text": "½å㋿🙂𐞁\r\n12345678's…", "tokens": 18, "pieces": ["½", "å", "㋿🙂", "𐞁", "\r\n", "123", "456", "78", "'s", "…"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "
ع…
séİ\r\n\r\n,dſéZ \n 'DEOT'Re", "tokens": 18, "pieces": ["
ع", "…", "
sé", "İ", "\r\n\r\n", ",dſé", "Z", " \n", " '", "DEOT'Re"]} +{"text": "9dZ\rZ'll \t!!(\t​\u000bfi0 sİe­#$%'ſm\u000bm", "tokens": 37, "pieces": ["9", "d", "Z", "\r", "Z", "A", "'", "ll", " ", "\t", "!!(", "\t", "​", "\u000bfi", "0", " s", "İe", "­#$%'", "ſm", "\u000bm"]} +{"text": "<0‍ZEOTm'D
'll12345678.Z'llZé<ſfiſ0Ⅳ>", "tokens": 24, "pieces": ["<", "0", "‍ZEOTm'D", "
", "'ll", "123", "456", "78", ".Z'll", "Zé", "<ſfiſ", "0Ⅳ", ">"]} +{"text": " å👍🏽12345678٣٤٥٦㍿ ३𐞁s \t\"३'lle'T'‍''VE 漢Ⅳ<|endoftext|> \n.👍🏽漢fí٣٤٥٦ſ12345678'VE", "tokens": 66, "pieces": [" ", " å", "👍🏽", "123", "456", "78٣", "٤٥٦", "㍿", " ", "३", "𐞁s", " ", "\t", "\"", "३", "'lle'T", "'‍''", "VE", " 漢", "Ⅳ", "<|", "endoftext", "|>", " \n", ".👍🏽", "漢fí", "٣٤٥", "٦", "ſ", "123", "456", "78", "'", "VE"]} +{"text": "Ⅳ \nḍ̇t<|endoftext|>At !,㍿!!٣٤٥٦🙂ⅣEOT#$%A\r\n\r\n", "At", " ", "!,㍿!!", "٣٤٥", "٦", "🙂", "Ⅳ", "EOT", "#$%", "A", "\r\n\r\n", ",½\"\té'reİ\t(漢#$%\r\n\r\nå'!!٣٤٥٦\r'M
Zé٣٤٥٦́İ ́­'३😀🏽å'<|fim_prefix|>", "tokens": 57, "pieces": ["<|", "fim", "_prefix", "|>,", "½", "\"", "\té're", "İ", "\t", "(漢", "#$%\r\n\r\n", "å", "'!!", "٣٤٥", "٦", "\r", "'M", "
Zé", "٣٤٥", "٦", "́", "İ", " ", " ́", "­'", "३", "😀🏽", "å", "'<|", "fim", "_prefix", "|>"]} +{"text": "\r\n…\u000b𐞁é-mſ…\r\n\r\nß<|endoftext|>d0 (字३½s字́'T\r\t'ſ'sⅣ́\tDž\r\n", "tokens": 44, "pieces": ["\r\n", "…", "\u000b𐞁é", "-mſ", "…\r\n\r\n", "ß", "<|", "endoftext", "|>", "d", "0", " (", "字", "३½", "s字́'T", "\r", "\t", "'ſ's", "Ⅳ", "́", "\tDž", "\r\n"]} +{"text": "#$%!EOTdé'S\u000bß \n३<|endoftext|>ß­s‍'ll0‍", "tokens": 33, "pieces": ["#$%!", "EOTdé'S", "\u000b", "ß", " \n", "३", "<|", "endoftext", "|>", "ß", "­", "s", "‍'", "ll", "0", "‍"]} +{"text": "İꟲ12345678", "tokens": 7, "pieces": ["İꟲ", "123", "456", "78"]} +{"text": "Dž(!!Zİ<­ 'Ree,
", "tokens": 14, "pieces": ["Dž", "(!!", "Zİ", "<­", " ", " '", "Ree", ",", "
"]} +{"text": "字 🙂s…Dž\r\n\r\n'ſ\tsİ𐞁\r'T\rm'VE-a9,e0!漢eİ'VE's s'Re字#$%‍<|endoftext|>", "tokens": 47, "pieces": ["(s", "!!", "ع", "​d", "'", "ſ", "\ts", "İ𐞁", "\r", "'T", "\r", "m'VE", "-a", "9", ",e", "0", "!漢e", "İ'VE", "'s", " s'Re", "字", "#$%‍<|", "endoftext", "|>"]} +{"text": "٣٤٥٦٣٤٥٦EOTé\u000b<|fim_prefix|>𐞁!9漢(字é…😀🏽eİ\"e \t!!A'reḍ̇'saZ\r\ntḍ̇­㍿\r\n\r\n𐞁", "tokens": 63, "pieces": ["٣٤٥", "٦٣٤", "٥٦", "EOTé", "\u000b", "<|", "fim", "_prefix", "|>", "𐞁", "!", "9", "漢", "(字é", "…", "😀🏽", "e", "İ", "\"e", " ", "\t", "!!", "A're", "ḍ̇'s", "a", "Z", "\r\n", "tḍ̇", "­㍿\r\n\r\n", "𐞁"]} +{"text": "éع(A>३#$%!
\n\t'llع\r\n\r\nⅣd'VEs(\n'Re", "tokens": 26, "pieces": ["éع", "(A", ">", "३", "#$%!", "
\n", "\t", "'llع", "\r\n\r\n", "Ⅳ", "d'VE", "s", "(\n", "'Re"]} +{"text": "عAZ٣٤٥٦#$%", "tokens": 13, "pieces": ["ع", "AZ", "٣٤٥", "٦", "#$%"]} +{"text": "\u000bd😀🏽DžA३٣٤٥٦\u000bs#$%\t㍿'S,عé‍m\u000b-s!!­Z'#$%", "tokens": 37, "pieces": ["\u000bd", "😀🏽", "DžA", "३٣٤", "٥٦", "\u000bs", "#$%", "\t", "㍿'", "S", ",عé", "‍m", "\u000b", "-s", "!!­", "Z", "'#$%"]} +{"text": "eé
<|fim_prefix|>'re'D'", "re'D", "'T㋿! \n…!\"'s٣٤٥٦٣٤٥٦'re ", "tokens": 32, "pieces": ["Ⅳ", "'VE字'VE", "'", "T", "㋿!", " \n", "…", "!\"'", "s", "٣٤٥", "٦٣٤", "٥٦", "'re", " "]} +{"text": "-🙂d𐞁🙂 <\r\n㍿t­#$%'T å<|endoftext|>🙂'lls", "tokens": 36, "pieces": ["𐞁", "🙂", " ", "<\r\n", "㍿t", "­#$%'", "T", " å", "<|", "endoftext", "|>🙂'", "lls"]} +{"text": "'ſ !㍿,\r\n\r\n<|fim_prefix|>ßefi,㍿'Dİ🙂\t\r\n<|endoftext|>.ß9🙂😀🏽", "tokens": 37, "pieces": ["'ſ", " !㍿,\r\n\r\n", "<|", "fim", "_prefix", "|>", "ßefi", ",㍿'", "Dİ", "🙂", "\t\r\n", "<|", "endoftext", "|>.", "ß", "9", "🙂😀🏽"]} +{"text": "漢漢eعés\rDž'VE Z", "tokens": 12, "pieces": ["漢漢eعés", "\r", "Dž'VE", " Z"]} +{"text": "!fit#$%‍Džfi İ🙂🙂 ㍿‍́'ſ<|fim_prefix|>", "tokens": 27, "pieces": ["!fit", "#$%‍", "Džfi", " İ", "🙂🙂", " ", "㍿‍́'", "ſ", "<|", "fim", "_prefix", "|>"]} +{"text": "Z👍🏽s#$%<|fim_prefix|>ḍ̇", "tokens": 16, "pieces": ["Z", "👍🏽", "s", "#$%<|", "fim", "_prefix", "|>", "ḍ̇"]} +{"text": "\n'VEZé'll,…😀🏽d<|endoftext|>㋿ß!<|endoftext|>½é\r'lléAt👍🏽.'VE'!!Ⅳ", "tokens": 49, "pieces": ["\n", "'VEZé'll", ",", "…", "😀🏽", "d", "<|", "endoftext", "|>㋿", "ß", "!<|", "endoftext", "|>", "½", "é", "\r", "'llé", "At", "👍🏽.'", "VE", "'!!", "Ⅳ"]} +{"text": "ſ½​­😀🏽½\n'VE('عⅣ…'reé'reee12345678😀🏽- <'ꟲ'ReDž12345678<  'VEt!! é\r漢's", "tokens": 53, "pieces": ["ſ", "½", "​­😀🏽", "½", "\n", "'VE", "('", "ع", "Ⅳ", "…", "'reé're", "ee", "123", "456", "78", "😀🏽-", " <'", "ꟲ'Re", "Dž", "123", "456", "78", "<", " ", " ", "'VEt", "!!", " é", "\r", "漢's"]} +{"text": "A­ꟲ<'ll
", "tokens": 8, "pieces": ["A", "­ꟲ", "<'", "ll", "
"]} +{"text": "Dž #$%㋿字㍿٣٤٥٦d\t… \n🙂's,-", "tokens": 25, "pieces": ["Dž", " ", "#$%㋿", "字", "㍿", "٣٤٥", "٦", "d", "\t… \n", "🙂'", "s", ",-"]} +{"text": "🙂'VE\r­'ſⅣꟲfié<|fim_prefix|>,(", "tokens": 24, "pieces": ["🙂'", "VE", "\r", "­<", "EOT", ">'", "ſ", "Ⅳ", "ꟲfié", "<|", "fim", "_prefix", "|>,("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "12345678Z<|endoftext|>\".\t> ½\u000b'M 12345678#$%㋿\u000bå漢at\u000bd<|endoftext|>!t👍🏽 > …s\tſ", "tokens": 56, "pieces": ["123", "456", "78", "Z", "<|", "endoftext", "|><", "EOT", ">\".", "\t", ">", " ", "½", "\u000b", "'M", " ", "123", "456", "78", "#$%㋿", "\u000bå漢at", "\u000bd", "<|", "endoftext", "|>!", "t", "👍🏽", " ", " >", " ", "…s", "\tſ"]} +{"text": "٣٤٥٦ Z漢́ 'VE‍'VEعع🙂३té\t\r\n\r\n>🙂'Sé 'S\r½ ", "tokens": 37, "pieces": ["٣٤٥", "٦", "", " Z漢́", " ", "'VE", "‍'", "VEعع", "🙂", "३", "té", "\t\r\n\r\n", "><", "EOT", ">🙂'", "Sé", " '", "S", "\r", "½", " "]} +{"text": " <|endoftext|><|endoftext|>..t<|fim_prefix|>'T#$%Dž<'D👍🏽ås'll'fi​'llꟲ'Sd", "tokens": 46, "pieces": [" ", "<|", "endoftext", "|><|", "endoftext", "|>..", "t", "<|", "fim", "_prefix", "|>'", "T", "#$%", "Dž", "<'", "D", "👍🏽", "ås'll", "'fi", "​'", "llꟲ'S", "d"]} +{"text": "(.Ⅳ
Dž9", "tokens": 7, "pieces": ["(.", "Ⅳ", "
Dž", "9"]} +{"text": "… -,-Atfi.عſ'VE'VE㋿㋿'\r> \n'字 ㋿<|fim_prefix|>'ll
'S㋿ 漢9e漢Z", "tokens": 50, "pieces": ["… ", " -,-", "Atfi", ".عſ'VE", "'VE", "㋿㋿'\r", ">", " \n", "'字", " ", " ㋿<|", "fim", "_prefix", "|>'", "ll", "
", "'S", "㋿", " 漢", "9", "e漢", "Z"]} +{"text": "́㋿<|fim_prefix|>🙂\r\n\r\nA
e
\"d12345678fiDž 's'Re <|endoftext|><|fim_prefix|>fi𐞁å㋿ع…🙂­३'(ß🙂\r\n\r\n", "A", "
e", "
", "\"d", "123", "456", "78", "fi", "Dž", " ", "'s'Re", " ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "fi𐞁å", "㋿ع", "…", "🙂­", "३", "'(", "ß", "é>#$% '", "tokens": 40, "pieces": ["!'", "T", "٣٤٥", "٦", "m", "123", "456", "78", "é", " ", "👍🏽", "é", " \n", "…é", ",'", "S", "(­", "İ'S", "\n", "<|", "endoftext", "|>", "é", ">#$%", " '"]} +{"text": "!👍🏽
é12345678'll'Re🙂…A#$%!🙂're\rİ>", "tokens": 31, "pieces": ["!👍🏽", "
é", "123", "456", "78", "'ll'Re", "🙂", "…A", "#$%!🙂'", "re", "\r", "İ", "><", "EOT", ">"]} +{"text": "DžⅣ \nſ 're…'-tfi'reⅣ \ń.dma", "tokens": 28, "pieces": ["Dž", "Ⅳ", " \n", "ſ", " ", " '", "re", "…", "'-", "t", "fi're", "Ⅳ", " \n", "́", ".dma", ""]} +{"text": "'(👍🏽e#$%'S😀🏽\tß-​Aa㍿\r\n\r\n eå<|fim_prefix|>\u000b!\n\t😀🏽🙂\tt'llZ㍿ ", "tokens": 47, "pieces": ["'(👍🏽", "e", "#$%'", "S", "😀🏽", "\tß", "-​", "Aa", "㍿\r\n\r\n", " ", " <", "META", "_START", ">eå", "<|", "fim", "_prefix", "|>", "\u000b", "!\n", "\t", "😀🏽🙂", "\tt'll", "Z", "㍿", " "]} +{"text": "字ꟲ fi​t12345678<'llEOT ­👍🏽\r\n
<|fim_prefix|>a㍿́-<|fim_prefix|>\t <́#$%Z,\nİ \n\r\n", "tokens": 48, "pieces": ["字ꟲ", " fi", "​t", "123", "456", "78", "<'", "ll", "EOT", " ­👍🏽\r\n", "
", "<|", "fim", "_prefix", "|>", "a", "㍿́", "-<|", "fim", "_prefix", "|>", "\t", " <́#$%", "Z", ",\n", "İ", " \n\r\n"]} +{"text": "㋿", "tokens": 3, "pieces": ["㋿"]} +{"text": "<|fim_prefix|>d字\r\n\r\n-('Reé'VEع 👍🏽٣٤٥٦\r\n\r\nſ𐞁\n٣٤٥٦İ!!>\r​😀🏽9", "tokens": 45, "pieces": ["<|", "fim", "_prefix", "|>", "d字", "\r\n\r\n", "-('", "Reé'VE", "ع", " ", "👍🏽", "٣٤٥", "٦", "\r\n\r\n", "ſ𐞁", "\n", "٣٤٥", "٦", "İ", "!!>\r", "​😀🏽", "9"]} +{"text": "
a'Dß,!'S<|endoftext|>\tⅣ'D'S½å😀🏽ع ­ßtع!\u000b!! \n­漢9㋿‍!", "tokens": 47, "pieces": ["
a'D", "ß", ",!'", "S", "<|", "endoftext", "|>", "\t", "Ⅳ", "'D'S", "½", "å", "😀🏽", "ع", " ", "­ß", "tع", "!", "\u000b", "!!", " \n", "­漢", "9", "㋿‍!"]} +{"text": "३ ß😀🏽<|endoftext|>½​́'S\"!!A𐞁漢😀🏽\r٣٤٥٦'VE 😀🏽\u000b​\n
Z\r\n\r\n!!㍿EOT<|fim_prefix|>A‍Z‍ (", "tokens": 62, "pieces": ["३", " ", " ß", "😀🏽<|", "endoftext", "|>", "½", "​́'S", "\"!!", "A𐞁漢", "😀🏽\r", "٣٤٥", "٦", "'VE", " ", "😀🏽", "\u000b", "​\n", "
Z", "\r\n\r\n", "!!㍿", "EOT", "<|", "fim", "_prefix", "|>", "A", "‍Z", "‍", " ("]} +{"text": ".a12345678å 'll 's!!𐞁 \n.t'Re३٣٤٥٦ⅣEOTEOT𐞁", "tokens": 34, "pieces": [".a", "123", "456", "78", "å", " ", " '", "ll", " ", " '", "s", "!!", "𐞁", " \n", ".t'Re", "३٣٤", "٥٦Ⅳ", "EOTEOT𐞁"]} +{"text": " \n٣٤٥٦", "tokens": 5, "pieces": [" \n", "٣٤٥", "٦"]} +{"text": ">ſ9!!İ,عß½te\"​é'Té漢m\u000b \t­ #$%9<İ\r's09a\r👍🏽", "tokens": 38, "pieces": [">ſ", "9", "!!", "İ", ",عß", "½", "te", "\"​", "é'T", "é漢m", "\u000b ", "\t", "­", " ", " #$%", "9", "<İ", "\r", "'s", "09", "a", "\r", "👍🏽"]} +{"text": "½​- ३𐞁𐞁éع字's9t'D🙂😀🏽#$%ß字٣٤٥٦s'M", "tokens": 40, "pieces": ["½", "​-<", "EOT", ">", " ", " ", "३", "𐞁𐞁éع字's", "9", "t'D", "🙂😀🏽#$%", "ß字", "٣٤٥", "٦", "s'M"]} +{"text": "३é.'Re-٣٤٥٦Dž#$%𐞁​́é<|fim_prefix|>ꟲ\".😀🏽<𐞁t12345678dd\t\t(ḍ̇- ", "tokens": 52, "pieces": ["३", "é", ".'", "Re", "-", "٣٤٥", "٦", "Dž", "#$%", "𐞁", "​́é", "<|", "fim", "_prefix", "|>", "ꟲ", "\".😀🏽<", "𐞁t", "123", "456", "78", "dd", "\t", "\t", "(ḍ̇", "-", " "]} +{"text": "9…३Ⅳ漢EOTe👍🏽𐞁s漢EOTå9fi'll", "tokens": 26, "pieces": ["9", "…", "३Ⅳ", "漢EOTe", "👍🏽", "𐞁s漢", "EOTå", "9", "fi'll"]} +{"text": "㋿\r…'VEEOTß12345678å'lléfi​
å'T㍿'s'M \r'TåDžd>漢 d ㍿", "tokens": 50, "pieces": ["㋿\r", "…", "'VEEOTß", "123", "456", "78", "å'll", "éfi", "​<", "EOT", ">", "
å'T", "㍿'", "s'M", " \r", "'Tå", "Džd", ">漢", " ", " d", " ", "㍿"]} +{"text": "ꟲ's'ſ…Zt\r\n\r\n<🙂ꟲ\teß'T'Dꟲ'll.𐞁dZdtDžEOT", "tokens": 43, "pieces": ["ꟲ", "'", "s'ſ", "…Zt", "\r\n\r\n", "<🙂", "ꟲ", "\teß'T", "'Dꟲ'll", ".𐞁d", "Zdt", "DžEOT", ""]} +{"text": "s<|fim_prefix|>​ſ\u000b-d<漢're\"👍🏽(…\r'ſ#$%\"'S<|fim_prefix|>a
'D\n\té-åİ'Ⅳd<|endoftext|>字\r<|fim_prefix|>", "tokens": 65, "pieces": ["s", "<|", "fim", "_prefix", "|>​", "ſ", "\u000b", "-d", "<漢're", "\"👍🏽(", "…\r", "'ſ", "#$%\"'", "S", "<|", "fim", "_prefix", "|>", "a", "
", "'D", "\n", "\té", "-å", "İ", "'<", "EOT", ">", "Ⅳ", "d", "<|", "endoftext", "|>", "字", "\r", "<|", "fim", "_prefix", "|>"]} +{"text": "́'re9\r\n\r\n>٣٤٥٦asß>​ع'ſ'VEe\t'S9, Z'Re­𐞁'Re", "tokens": 36, "pieces": ["́'re", "9", "\r\n\r\n", ">", "٣٤٥", "٦", "asß", ">​", "ع'ſ", "'VEe", "\t", "'S", "9", ",", " Z'Re", "­<", "META", "_START", ">𐞁'Re"]} +{"text": "😀🏽👍🏽\t", "tokens": 7, "pieces": ["😀🏽👍🏽", "\t"]} +{"text": "a㋿'ll!!\r\n\r\né\"! ' -9'reA३ß <#$%'VE\"0!#$%Dž漢'T\r\n", "tokens": 33, "pieces": ["a", "㋿'", "ll", "!!\r\n\r\n", "é", "\"!", " '", " ", "-", "9", "'re", "A", "३", "ß", " ", "<#$%'", "VE", "\"", "0", "!#$%", "Dž漢'T", "\r\n"]} +{"text": " \nßDž😀🏽's,m-'ſEOT𐞁𐞁", "tokens": 25, "pieces": [" \n", "ß", "Dž", "😀🏽<", "EOT", ">'", "s", ",m", "-'", "ſ", "EOT𐞁𐞁"]} +{"text": "
\r㍿<|endoftext|>'ll(<|fim_prefix|> fi​eⅣ٣٤٥٦", "tokens": 29, "pieces": ["
\r", "㍿<|", "endoftext", "|>'", "ll", "(<|", "fim", "_prefix", "|>", " fi", "​e", "Ⅳ٣٤", "٥٦"]} +{"text": "İZ \n12345678fi'Re'sſ
字, <|fim_prefix|>'D,12345678\"a ,㋿", "tokens": 30, "pieces": ["İZ", " \n", "123", "456", "78", "fi'Re", "'sſ", "
字", ",", " ", "<|", "fim", "_prefix", "|>'", "D", ",", "123", "456", "78", "\"a", " ,㋿"]} +{"text": "EOT漢!fi,İ'se'Re㍿३'VE#$%t0Z\u000bⅣZ", "tokens": 31, "pieces": ["EOT漢", "!fi", ",İ's", "e'Re", "㍿", "३", "'VE", "#$%<", "EOT", ">t", "0", "Z", "\u000b", "Ⅳ", "Z"]} +{"text": " \u000b🙂<|fim_prefix|> 
12345678'TtⅣꟲ>fi३…㍿sⅣEOT'VE½étİé", "tokens": 42, "pieces": [" ", "\u000b", "🙂<|", "fim", "_prefix", "|>", " ", "
", "123", "456", "78", "'Tt", "Ⅳ", "ꟲ", ">fi", "३", "…", "㍿s", "Ⅳ", "EOT'VE", "½", "ét", "İé"]} +{"text": "٣٤٥٦0\r\n12345678<|endoftext|>m'D \n٣٤٥٦'S३ſ\t漢'Re३'M12345678Ⅳ'VE're.字 .<|fim_prefix|>\u000b<|fim_prefix|>m", "tokens": 59, "pieces": ["٣٤٥", "٦0", "\r\n", "123", "456", "78", "<|", "endoftext", "|>", "m'D", " \n", "٣٤٥", "٦", "'S", "", "३", "ſ", "\t漢'Re", "३", "'M", "123", "456", "78Ⅳ", "'VE're", ".字", " ", ".<|", "fim", "_prefix", "|>", "\u000b", "<|", "fim", "_prefix", "|>", "m"]} +{"text": "㋿ß(12345678\tmع'S", "tokens": 11, "pieces": ["㋿ß", "(", "123", "456", "78", "\tmع'S"]} +{"text": "ꟲ​½s\u000b㋿#$%A\"​ Dž!!ḍ̇Z字>fi'S\r\n\rå!e漢.ßa.👍🏽>'", "tokens": 42, "pieces": ["ꟲ", "​", "½", "s", "\u000b", "㋿#$%", "A", "\"​", " Dž", "!!", "ḍ̇", "Z字", ">fi'S", "\r\n\r", "å", "!e漢", ".ßa", ".👍🏽>'"]} +{"text": "‍\t\t'DéA \n <|fim_prefix|>", "tokens": 15, "pieces": ["‍", "\t", "\t", "'Dé", "A", " \n", " ", " <|", "fim", "_prefix", "|>"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " Dž#$%'T'T'M漢🙂<|fim_prefix|><'Té'D㋿👍🏽…fi🙂0Dž9👍🏽're😀🏽
éع\téa-.", "tokens": 59, "pieces": [" ", " Dž", "#$%'", "T'T", "'M漢", "🙂<|", "fim", "_prefix", "|><'", "Té'D", "㋿<", "EOT", ">👍🏽", "…fi", "🙂", "0", "Dž", "9", "👍🏽'", "re", "😀🏽", "
éع", "\téa", "-."]} +{"text": "​\r!,ad字t…𐞁>🙂'VE\r\n\r\nſ Dž, 'ſ'Re \n're're㍿'S0<漢'D,", "tokens": 39, "pieces": ["​\r", "!,", "ad字t", "…𐞁", ">🙂'", "VE", "\r\n\r\n", "ſ", " Dž", ",", " ", " '", "ſ'Re", " \n", "'re're", "㍿'", "S", "0", "<漢'D", ","]} +{"text": "Z12345678.\"३\u000bİ字fi!!​\r\n\r\n🙂!!'ſ​㍿
'D!!t😀🏽½ſ٣٤٥٦
fi", "tokens": 43, "pieces": ["Z", "123", "456", "78", ".\"", "३", "\u000bİ字fi", "!!​\r\n\r\n", "🙂!!<", "EOT", ">'", "ſ", "​<", "EOT", ">㍿", "
", "'D", "!!", "t", "😀🏽", "½", "ſ", "٣٤٥", "٦", "
fi"]} +{"text": "\r\n\r\ns\t>\"étfiém'VEa're字\"Z\u000b m#$%İ\r<|endoftext|>s🙂ꟲ\te>", "tokens": 37, "pieces": ["\r\n\r\n", "s", "\t", ">\"", "étfiém'VE", "a're", "字", "\"Z", "\u000b", " m", "#$%", "İ", "\r", "<|", "endoftext", "|>", "s", "🙂ꟲ", "\te", ">"]} +{"text": "'VE#$%ꟲ'll٣٤٥٦ſꟲ åⅣ.…<|fim_prefix|>٣٤٥٦ea'M!!३Ⅳå'D\r\n​\r\n'S㋿½'ll½ع'MZEOT'", "tokens": 61, "pieces": ["'VE", "#$%", "ꟲ'll", "٣٤٥", "٦", "ſꟲ", " å", "Ⅳ", ".", "…", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "ea'M", "!!", "३Ⅳ", "å'D", "\r\n", "​\r\n", "'S", "㋿", "½", "'ll", "½", "ع'M", "ZEOT", "'"]} +{"text": "​Ⅳ EOT<|endoftext|>as!㋿éſd'DDž é,㍿(d​
\r\né a🙂ſ", "tokens": 46, "pieces": ["​", "Ⅳ", " EOT", "<|", "endoftext", "|>", "as", "!㋿", "éſd'D", "Dž", " ", " é", ",㍿(", "d", "​", "
\r\n", "é", " ", " a", "🙂ſ"]} +{"text": "㍿<|endoftext|>fitZ 𐞁'VE'll'D…\t🙂<Dž>A!!'VEEOT'sع字's \"(…😀🏽'T!!'ſ​Z", "tokens": 59, "pieces": ["㍿<|", "endoftext", "|>", "fit", "Z", " ", " 𐞁'VE", "'ll'D", "…", "\t", "🙂<", "Dž", ">A", "!!'", "VEEOT's", "ع字's", " ", "\"(", "…", "😀🏽'", "T", "!!'", "ſ", "​Z"]} +{"text": "ḍ̇½'Sa'́9Ⅳfi
 \n\u000bfi e,d…", "tokens": 25, "pieces": ["ḍ̇", "½", "'Sa", "'́", "9", "", "Ⅳ", "fi", "
 \n", "\u000bfi", " ", " e", ",d", "…"]} +{"text": "漢 éå🙂é \n(ſs#$%'s e( ,㋿🙂ꟲ\n㍿
", "tokens": 35, "pieces": ["漢", " éå", "🙂é", " \n", "(ſs", "#$%'", "s", " ", " e", "(", " ", ",㋿🙂", "ꟲ", "\n", "㍿<", "EOT", ">", "
"]} +{"text": "s'VEß𐞁ſ9 mfi!!㍿<|fim_prefix|>s9  ", "tokens": 39, "pieces": ["s'VE", "ß𐞁ſ", "9", " ", " mfi", "!!㍿<|", "fim", "_prefix", "|>", "s", "", "9", "  "]} +{"text": ">-ſEOT'T<<|endoftext|>\r fi漢👍🏽 a\"å㍿ ​EOT\n're-Z \n<|endoftext|>sm٣٤٥٦字", "tokens": 51, "pieces": [">-", "ſ", "EOT'T", "<<|", "endoftext", "|>\r", "", " fi漢", "👍🏽", " ", " a", "\"å", "㍿", " ", " ​", "EOT", "\n", "'re", "-Z", " \n", "<|", "endoftext", "|>", "sm", "٣٤٥", "٦", "字"]} +{"text": "\"𐞁ſ-٣٤٥٦ßémİ!३٣٤٥٦á>‍\r\n\r\n<|fim_prefix|>#$%ß're
🙂a<|fim_prefix|><|endoftext|>12345678d\",", "tokens": 55, "pieces": ["\"𐞁ſ", "-", "٣٤٥", "٦", "ßém", "İ", "!", "३٣٤", "٥٦", "á", ">‍\r\n\r\n", "<|", "fim", "_prefix", "|>#$%", "ß're", "
", "🙂a", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "123", "456", "78", "d", "\","]} +{"text": "😀🏽…,‍('T𐞁#$%
Ⅳع\t'字's're\r\n\r\n'll👍🏽ع .½< 😀🏽é३'VE漢漢'S\nⅣع…A#$%", "tokens": 54, "pieces": ["😀🏽", "…", ",‍('", "T𐞁", "#$%", "
", "Ⅳ", "ع", "\t", "'字's", "'re", "\r\n\r\n", "'ll", "👍🏽", "ع", " ", " .", "½", "<", " ", " 😀🏽", "é", "३", "'VE漢漢'S", "\n", "Ⅳ", "ع", "…A", "#$%"]} +{"text": "Z \n>३d'S<|endoftext|>12345678👍🏽😀🏽'lĺ(­…㍿\t㋿㋿字fie0'A'll漢s's🙂👍🏽😀🏽'VE<🙂fi", "tokens": 61, "pieces": ["Z", " \n", ">", "३", "d'S", "<|", "endoftext", "|>", "123", "456", "78", "👍🏽😀🏽'", "lĺ", "(­", "…", "㍿", "\t", "㋿㋿", "字fie", "0", "'A'll", "漢s's", "🙂👍🏽😀🏽'", "VE", "<🙂", "fi"]} +{"text": "é>EOT'VE'VE 'VEⅣ
 'll३0d12345678👍🏽Ⅳ字å字Aḍ̇½<|fim_prefix|>Ⅳ< <|endoftext|>åEOT'VE", "'VE", " ", "'VE", "Ⅳ", "
", " '", "ll", "३0", "d", "123", "456", "78", "👍🏽", "Ⅳ", "字å字", "Aḍ̇", "½", "<|", "fim", "_prefix", "|>", "Ⅳ", "<", " ", "<|", "endoftext", "|>", "å", "!字<|fim_prefix|>#$%عté12345678😀🏽İ३d­>ꟲ-'sa३½­…é", "tokens": 53, "pieces": ["٣٤٥", "٦", " fi", "!<", "META", "_START", ">字", "<|", "fim", "_prefix", "|>#$%", "عté", "123", "456", "78", "😀🏽", "İ", "३", "d", "­>", "ꟲ", "-'", "sa", "३", "", "½", "­", "…é"]} +{"text": "éEOTe.( ٣٤٥٦<|fim_prefix|>e😀🏽🙂 😀🏽><|endoftext|>é٣٤٥٦Z \n> 'llfié<|fim_prefix|>👍🏽漢ꟲ .'M", "tokens": 66, "pieces": ["é", "EOT", "e", ".(", " ", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "e", "😀🏽🙂", " 😀🏽><|", "endoftext", "|>", "é", "٣٤٥", "٦", "Z", " \n", "><", "META", "_START", ">", " ", "'llfié", "<|", "fim", "_prefix", "|>👍🏽", "漢ꟲ", " ", " .'", "M"]} +{"text": "ßꟲs𐞁🙂<|fim_prefix|>e\"\r\n\r\n \n", "tokens": 22, "pieces": ["ßꟲs𐞁", "🙂<", "META", "_START", "><|", "fim", "_prefix", "|>", "e", "\"\r\n\r\n", " \n"]} +{"text": "<|fim_prefix|>'T #$%‍'ll<|fim_prefix|>'T \t0字ꟲé😀🏽é\n'🙂( m½é'M\u000b<|fim_prefix|>㋿>", "tokens": 52, "pieces": ["<|", "fim", "_prefix", "|>'", "T", " ", "#$%‍'", "ll", "<|", "fim", "_prefix", "|>'", "T", " ", "\t", "0", "字ꟲé", "😀🏽", "é", "\n", "'🙂(", " m", "½", "é'M", "\u000b", "<|", "fim", "_prefix", "|>㋿>"]} +{"text": "́​ꟲ", "tokens": 5, "pieces": ["́", "​ꟲ"]} +{"text": "'lla३́ꟲ  🙂
", "tokens": 13, "pieces": ["'lla", "३", "́ꟲ", "", " ", " 🙂", "
"]} +{"text": "0٣٤٥٦İ's(12345678\u000b0'M ", "tokens": 52, "pieces": ["0", "", "٣٤٥", "٦", "İ's", "(", "123", "456", "78", "\u000b", "0", "'M", " "]} +{"text": "\u000bſ9́ \n'Re𐞁\r漢<|fim_prefix|>", "tokens": 18, "pieces": ["\u000bſ", "9", "́", " \n", "'Re𐞁", "\r", "漢", "<|", "fim", "_prefix", "|>"]} +{"text": "\r\n'D\r\nİ ('M٣٤٥٦ ع", "tokens": 12, "pieces": ["\r\n", "'D", "\r\n", "İ", " ('", "M", "٣٤٥", "٦", " ع"]} +{"text": "Ⅳ\"ꟲ\nß😀🏽9👍🏽é㍿ع㍿Džع!!𐞁 'S9😀🏽9'Ssꟲ'S‍<|endoftext|>\r\n३-
<|fim_prefix|>'EOTḍ̇३a", "tokens": 73, "pieces": ["Ⅳ", "\"ꟲ", "\n", "ß", "😀🏽", "9", "👍🏽", "é", "㍿ع", "㍿Džع", "!!", "𐞁", " ", "'S", "9", "😀🏽", "9", "'Ssꟲ'S", "‍<|", "endoftext", "|>\r\n", "३", "-", "
", "<|", "fim", "_prefix", "|>'", "EOTḍ̇", "३", "a"]} +{"text": "🙂́\u000bß!!Ⅳ,\r(漢'sꟲ½字\r\nA,\t'Tå𐞁३!!👍🏽Ⅳ\nt'Se", "tokens": 47, "pieces": ["🙂́", "", "\u000bß", "!!", "Ⅳ", ",\r", "(漢's", "ꟲ", "½", "字", "\r\n", "A", ",", "\t", "'Tå𐞁", "३", "!!👍🏽", "Ⅳ", "\n", "t'S", "e"]} +{"text": "'D\"se🙂ع12345678e", "tokens": 9, "pieces": ["'D", "\"se", "🙂ع", "123", "456", "78", "e"]} +{"text": "<( \n (", "tokens": 12, "pieces": ["<(<", "META", "_START", ">", " \n", "", " ", "("]} +{"text": "İ'MEOT'ſ\u000b३\r\n\r\nع'M\r\n>\t ", "tokens": 18, "pieces": ["İ'M", "EOT'ſ", "", "\u000b", "३", "\r\n\r\n", "ع'M", "\r\n", ">", "\t "]} +{"text": " -'S\r\né漢0ſ​'s'T'ſ '\u000b'D,㋿'M​td㍿<|endoftext|>", "tokens": 42, "pieces": [" -<", "META", "_START", ">'", "S", "\r\n", "é漢", "0", "ſ", "​<", "EOT", ">'", "s'T", "'ſ", " ", " '", "\u000b", "'D", ",㋿'", "M", "​td", "㍿<|", "endoftext", "|>"]} +{"text": "\nꟲ㍿<|endoftext|> 'M'sm½EOT  sꟲ ع🙂!\n!!Ⅳ-<|fim_prefix|>.m'M\r㍿'Rem", "tokens": 52, "pieces": ["\n", "ꟲ", "㍿<|", "endoftext", "|>", " ", " '", "M's", "m", "½", "EOT", " ", " sꟲ", " ع", "🙂!\n", "!!", "Ⅳ", "-<|", "fim", "_prefix", "|>.", "m'M", "\r", "㍿'", "Rem"]} +{"text": "ع 'Res", "tokens": 8, "pieces": ["ع", "", " ", "'Res"]} +{"text": "9! \ń'SDž३'Ss'D0(Z🙂ſDž!!ꟲß0…fi", "tokens": 26, "pieces": ["9", "!", " \n", "́'S", "Dž", "३", "'Ss'D", "0", "(Z", "🙂ſ", "Dž", "!!", "ꟲß", "0", "…fi"]} +{"text": "\"́½ \nⅣAéꟲ👍🏽<A( ́d漢👍🏽'ſ'S12345678\t'D,Dž12345678३", "tokens": 40, "pieces": ["\"́", "½", " \n", "Ⅳ", "Aéꟲ", "👍🏽<", "A", "(", " ", " ́d漢", "👍🏽'", "ſ'S", "123", "456", "78", "\t", "'D", ",Dž", "123", "456", "78३"]} +{"text": "\r\nå, !!fi ́字½12345678\n're𐞁'Re <|endoftext|>👍🏽tEOT👍🏽 emⅣe", "tokens": 43, "pieces": ["\r\n", "å", ",", " ", " !!", "fi", " ́字", "½12", "345", "678", "\n", "'re𐞁'Re", " ", "<|", "endoftext", "|>👍🏽", "t", "EOT", "👍🏽", " ", " em", "Ⅳ", "e"]} +{"text": "\réße字0's
३ḍ̇0EOT", "tokens": 20, "pieces": ["\r", "éß", "e字", "0", "'s", "
", "३", "ḍ̇", "0", "EOT"]} +{"text": "A'ſ🙂EOT\"ḍ̇Ⅳ㍿́٣٤٥٦", "tokens": 23, "pieces": ["A'ſ", "🙂EOT", "\"ḍ̇", "Ⅳ", "㍿́", "٣٤٥", "٦"]} +{"text": "३.'M\t", "tokens": 4, "pieces": ["३", ".'", "M", "\t"]} +{"text": "'Re'llİ< \r\n\r\n\nEOT.d'll!'S,", "tokens": 19, "pieces": ["'Re'll", "İ", "<", " \r\n\r\n\n", "EOT", ".d", "'", "ll", "!'", "S", ","]} +{"text": "1234567812345678'D'M\"㍿🙂'S'S<|endoftext|> >\n.'S.é𐞁​🙂 ſ \n<\r\nḍ̇0👍🏽!३,", "tokens": 52, "pieces": ["123", "456", "781", "234", "567", "8", "'D'M", "\"㍿🙂'", "S'S", "<|", "endoftext", "|>", " >\n", ".'", "S", ".é", "𐞁", "​🙂", " ſ", " \n", "<\r\n", "ḍ̇", "0", "👍🏽!", "३", ","]} +{"text": "'Sß'll
‍ß-é'D.A३!!字're'll(0a'\"ⅣEOT‍🙂\r\n㍿😀🏽A#$%EOT \n!!fíḍ̇", "tokens": 48, "pieces": ["'Sß'll", "
", "‍ß", "-é'D", ".A", "३", "!!", "字're", "'ll", "(", "0", "a", "'\"", "Ⅳ", "EOT", "‍🙂\r\n", "㍿😀🏽", "A", "#$%", "EOT", " \n", "!!", "fíḍ̇"]} +{"text": "👍🏽Z'ſ \r\n-\r\n!字fi \n,s \n㍿,<|fim_prefix|>12345678🙂漢عemsḍ̇'e\"<|endoftext|>\tⅣ", "tokens": 45, "pieces": ["👍🏽", "Z'ſ", " \r\n", "-\r\n", "!字fi", " \n", ",s", " \n", "㍿,<|", "fim", "_prefix", "|>", "123", "456", "78", "🙂漢عemsḍ̇", "'e", "\"<|", "endoftext", "|>", "\t", "Ⅳ"]} +{"text": "!!\t\r\n\r\n
😀🏽\r0\u000b😀🏽ع <|endoftext|>0 ٣٤٥٦\u000b​s३<|endoftext|> ́'VEEOT mmå-३字३  漢", "tokens": 60, "pieces": ["!!", "\t\r\n\r\n", "
", "😀🏽\r", "0", "\u000b", "😀🏽", "ع", " <|", "endoftext", "|>", "0", " ", "٣٤٥", "٦", "\u000b", "​s", "३", "<|", "endoftext", "|>", " ́'VE", "EOT", " mmå", "-", "३", "字", "", "३", " ", " 漢"]} +{"text": "ع½śdDž!!,Z'llt३́A", "tokens": 15, "pieces": ["ع", "½", "śd", "Dž", "!!,", "Z'll", "t", "३", "́", "A"]} +{"text": "<|endoftext|>'ll12345678 \n漢're३㍿'S-\"'re…<|endoftext|>'Śḍ̇!!", "tokens": 38, "pieces": ["<|", "endoftext", "|>'", "ll", "123", "456", "78", " \n", "漢're", "३", "㍿'", "S", "-\"'", "re", "…", "<|", "endoftext", "|>'", "Śḍ̇", "!!"]} +{"text": "ḍ̇0!\t​👍🏽9ſ(dⅣt'VE­0'D<漢\r\n\r\nA<|fim_prefix|>EOTfi", "tokens": 34, "pieces": ["ḍ̇", "0", "!", "\t", "​👍🏽", "9", "ſ", "(d", "Ⅳ", "t'VE", "­", "0", "'D", "<漢", "\r\n\r\n", "A", "<|", "fim", "_prefix", "|>", "EOTfi"]} +{"text": "‍ \"\rsß½ꟲß😀🏽…漢'M㍿<́-t‍漢Zعd<|endoftext|>", "tokens": 37, "pieces": ["‍", " ", " \"\r", "sß", "½", "ꟲß", "😀🏽", "…漢'M", "㍿<́-", "t", "‍漢Zعd", "<|", "endoftext", "|>"]} +{"text": " 're!ém<|fim_prefix|>​\r\n<|endoftext|>\rع​#$%'llḍ̇㋿𐞁A!>\rå­12345678ſéꟲ٣٤٥٦Am\r\n\r\nع'ſ'", "tokens": 64, "pieces": [" '", "re", "!ém", "<|", "fim", "_prefix", "|>​\r\n", "<|", "endoftext", "|>\r", "ع", "​#$%'", "llḍ̇", "㋿𐞁", "A", "!>\r", "å", "­", "123", "456", "78", "ſéꟲ", "٣٤٥", "٦", "Am", "\r\n\r\n", "ع'ſ", "'"]} +{"text": "'sEOTefi'\r\n\r\ń­ß<|endoftext|>t'reⅣ🙂", "tokens": 21, "pieces": ["'s", "EOTefi", "'\r\n\r\n", "́", "­ß", "<|", "endoftext", "|>", "t're", "Ⅳ", "🙂"]} +{"text": ">(😀🏽'M\téⅣ\u000b'VE!! \n(\tm\t12345678<.t<|fim_prefix|>'re​, ́ꟲ!!​\r\nDž'sḍ̇", "tokens": 51, "pieces": [">(😀🏽'", "M", "\té", "Ⅳ", "\u000b", "'VE", "!!", " \n", "(", "\tm", "\t", "123", "456", "78", "<.", "t", "<|", "fim", "_prefix", "|>'", "re", "​,", " ", " ́ꟲ", "!!​<", "EOT", ">\r\n", "Dž's", "ḍ̇"]} +{"text": "🙂'ſⅣ'Re漢́㍿'Mmd-t!!🙂-!!'Sİ'ſ३㋿\r𐞁'S'T<|endoftext|>m'VEſ…dع'rem٣٤٥٦", "tokens": 59, "pieces": ["🙂'", "ſ", "Ⅳ", "'Re漢́", "㍿<", "META", "_START", ">'", "Mmd", "-t", "!!🙂-!!'", "Sİ'ſ", "३", "㋿\r", "𐞁'S", "'T", "<|", "endoftext", "|>", "m'VE", "ſ", "…dع're", "m", "٣٤٥", "٦"]} +{"text": "-👍🏽­'ll12345678!३㋿ß‍ 'så\r\nⅣ\r\n\r\n
\r\n ㍿३ m \n\t", "tokens": 34, "pieces": ["-👍🏽­'", "ll", "123", "456", "78", "!", "३", "㋿ß", "‍", " '", "så", "\r\n", "Ⅳ", "\r\n\r\n
\r\n", " ", "㍿", "३", " m", " \n", "\t"]} +{"text": "ḍ̇!m㍿s\r ,𐞁ع fí", "tokens": 20, "pieces": ["ḍ̇", "!m", "㍿s", "\r", " ", ",𐞁ع", " ", " fí"]} +{"text": ",İ\r\n\u000b", "tokens": 4, "pieces": [",İ", "\r\n", "\u000b"]} +{"text": "‍#$%<'S'ReⅣee𐞁'M \r\n9-٣٤٥٦…,'re٣٤٥٦'M'VE 'T٣٤٥٦A-\"-'VEſſ-", "tokens": 55, "pieces": ["‍#$%<'", "S'Re", "Ⅳ", "ee𐞁'M", " \r\n", "9", "-", "٣٤٥", "٦", "…", ",'", "re", "٣٤٥", "٦", "'M'VE", "", " ", " '", "T", "٣٤٥", "٦", "A", "-\"-'", "VEſſ", "-"]} +{"text": "<|endoftext|>३ <‍İ'så३'Mta's'M'ſ're \r\n\r\nꟲ\n😀🏽'Reꟲå\ndꟲ9Z\r\u000b \n\r\nté­İ", "tokens": 55, "pieces": ["<|", "endoftext", "|>", "३", " <‍<", "EOT", ">İ's", "å", "३", "'Mta's", "'M'ſ", "'re", " \r\n\r\n", "ꟲ", "\n", "😀🏽'", "Reꟲå", "\n", "dꟲ", "9", "Z", "\r\u000b \n\r\n", "té", "­İ"]} +{"text": "<|endoftext|>é‍å \n३!😀🏽 \n'VE\t\r\n\r\n\néßå🙂\r\n\r\nfia", "tokens": 43, "pieces": ["㋿<", "META", "_START", "><|", "endoftext", "|>", "é", "‍", "å", " \n", "३", "!😀🏽", " \n", "'VE", "\t\r\n\r\n\n", "éßå", "🙂\r\n\r\n", "fia"]} +{"text": "字s12345678", "tokens": 5, "pieces": ["字s", "123", "456", "78"]} +{"text": "ḍ̇m㋿\n👍🏽,😀🏽 漢 \n­'T漢\t́'MåEOT're9t­\rß! Ⅳ‍'Re\r", "tokens": 43, "pieces": ["\u000b́'M", "İA", "123", "456", "78", "EOT", "<|", "fim", "_prefix", "|>'", "T漢", "\t́'M", "å", "EOT're", "", "9", "t", "­\r", "ß", "!", " ", "Ⅳ", "‍'", "Re", "\r"]} +{"text": "漢٣٤٥٦d<|fim_prefix|>'sEOT #$%​<|endoftext|> 'Sfiع!\"ſ漢", "tokens": 33, "pieces": ["漢", "٣٤٥", "٦", "d", "<|", "fim", "_prefix", "|>'", "s", "EOT", " ", "#$%​<|", "endoftext", "|>", " ", "'Sfiع", "!\"", "ſ漢"]} +{"text": "'🙂ſ३'VE99
t…🙂字EOT0. \n​عEOTtꟲDžfí'llİ‍\r\né", "tokens": 36, "pieces": ["'🙂", "ſ", "३", "'VE", "99", "
t", "…", "🙂字", "EOT", "0", ".", " \n", "​عEOTtꟲ", "Džfí'll", "İ", "‍\r\n", "é"]} +{"text": "'S​'ll‍
İ", "tokens": 7, "pieces": ["'S", "​'", "ll", "‍", "
İ"]} +{"text": "å'Så ३é👍🏽'VE><\r\n\r\n😀🏽𐞁'aEOT\r\n\r\nß'll0Dže​Aꟲ", "tokens": 44, "pieces": ["å'S", "å", " ", " ", "३", "é", "👍🏽'", "VE", "><\r\n\r\n", "😀🏽", "𐞁", "'<", "META", "_START", ">a", "EOT", "\r\n\r\n", "ß'll", "0", "Dže", "​Aꟲ"]} +{"text": "''Mع>🙂ꟲ\r\n 9e\"字
", "tokens": 18, "pieces": ["''", "Mع", ">🙂", "ꟲ", "\r\n", " ", "9", "e", "\"字", "
"]} +{"text": ",
!<|endoftext|>‍㍿…'M‍㍿ \r\n'Mtİḍ̇fi👍🏽.\r\n\r\né>'re\"\r\n", "tokens": 37, "pieces": [",", "
", "!<|", "endoftext", "|>‍㍿", "…", "'M", "‍㍿", " \r\n", "'Mt", "İḍ̇fi", "👍🏽.\r\n\r\n", "é", ">'", "re", "\"\r\n"]} +{"text": "'M'D‍012345678'll12345678<|fim_prefix|>\u000bA ḍ̇!Ⅳ,'ll‍A>İé \n👍🏽're9''ll漢-.a\n㋿é", "tokens": 54, "pieces": ["'M'D", "‍", "012", "345", "678", "'ll", "123", "456", "78", "<|", "fim", "_prefix", "|>", "\u000bA", " ḍ̇", "!", "Ⅳ", ",'", "ll", "‍A", ">İé", " \n", "👍🏽'", "re", "9", "''", "ll", "漢", "-.", "a", "\n", "㋿é"]} +{"text": "'VE-'llḍ̇fi'DⅣ.​,́", "tokens": 14, "pieces": ["'VE", "-'", "llḍ̇fi'D", "Ⅳ", ".​,́"]} +{"text": "<|endoftext|>tm😀🏽!!'T,'sfi,漢,'ſfiZé<|fim_prefix|>,,!!
\t", "tokens": 34, "pieces": ["<|", "endoftext", "|>", "tm", "😀🏽!!'", "T", ",'", "sfi", ",漢", ",'", "ſfi", "Zé", "<|", "fim", "_prefix", "|>,,!!", "
\t"]} +{"text": "́'Reꟲ \n'M>. 9 EOT#$%👍🏽t字 \nt​  ٣٤٥٦(<|fim_prefix|>'re\t", "tokens": 36, "pieces": ["́'Re", "ꟲ", " \n", "'M", ">.", " ", "9", " EOT", "#$%👍🏽", "t字", " \n", "t", "​", " ", " ", "٣٤٥", "٦", "(<|", "fim", "_prefix", "|>'", "re", "\t"]} +{"text": "\t're…'VEſ字ea🙂\re٣٤٥٦'T字'reé'llꟲ'ſ½ \nⅣa", "tokens": 32, "pieces": ["\t", "'re", "…", "'VEſ字ea", "🙂\r", "e", "٣٤٥", "٦", "'T字're", "é'll", "ꟲ'ſ", "½", " \n", "Ⅳ", "a"]} +{"text": "\tⅣe 0漢\u000b'D, ,漢Ⅳع😀🏽é́
'D \n\t.‍dꟲİ‍#$%A'T", "tokens": 44, "pieces": ["\t", "Ⅳ", "e", " ", "0", "漢", "\u000b", "'", "D", ",", " ", ",漢", "Ⅳ", "ع", "😀🏽", "é́", "
", "'D", "", " \n", "\t", ".‍", "dꟲ", "İ", "‍#$%", "A'T"]} +{"text": "字!!'s 're'D\r\n\r\n(​👍🏽\r\n\r\n𐞁'll fiDž123456780‍'Tå漢٣٤٥٦", "tokens": 36, "pieces": ["字", "!!'", "s", " ", " '", "re'D", "\r\n\r\n", "(​👍🏽\r\n\r\n", "𐞁'll", " fi", "Dž", "123", "456", "780", "‍'", "Tå漢", "٣٤٥", "٦"]} +{"text": "Ⅳ. ⅣEOT \raſ漢<👍🏽-!e.٣٤٥٦.m", "tokens": 26, "pieces": ["Ⅳ", ".", " ", "Ⅳ", "EOT", " \r", "aſ漢", "<👍🏽-!", "e", ".", "٣٤٥", "٦", ".m"]} +{"text": "-#$%'T…Z<|fim_prefix|>\"9 \n½<|fim_prefix|>'M", "tokens": 23, "pieces": ["-#$%'", "T", "…Z", "<|", "fim", "_prefix", "|>\"", "9", " \n", "½", "<|", "fim", "_prefix", "|>'", "M"]} +{"text": "ꟲ३'T .
<''T'ReAe'S<|endoftext|>​a<|endoftext|>𐞁\r\n >EOT're😀🏽Z'T\"#$%ßZ'll\r\n\r\n㍿", "tokens": 58, "pieces": ["ꟲ", "३", "'T", " ", " .", "
", "<''", "T'Re", "Ae'S", "<|", "endoftext", "|>​", "a", "<|", "endoftext", "|>", "𐞁", "\r\n", " ", ">EOT're", "😀🏽", "Z'T", "\"#$%", "ß", "Z'll", "\r\n\r\n", "㍿"]} +{"text": ">m­​ \n'Ds\u000bé㋿́", "tokens": 13, "pieces": [">m", "­​", " \n", "'Ds", "\u000bé", "㋿́"]} +{"text": "<‍\n'MZfiß\tfi!'S\r…'s!!tǻ漢عꟲعå'ſEOT", "tokens": 31, "pieces": ["<‍\n", "'MZfiß", "\tfi", "!'", "S", "\r", "…", "'s", "!!", "tǻ漢عꟲعå'ſ", "EOT"]} +{"text": "字😀🏽👍🏽‍ع\u000b'S㍿<|endoftext|>a!'ll", "tokens": 24, "pieces": ["字", "😀🏽👍🏽‍", "ع", "\u000b", "'S", "㍿<|", "endoftext", "|>", "a", "!'", "ll"]} +{"text": "ꟲ<|endoftext|>‍😀🏽​('ſ<|endoftext|>\t ​\u000b'D!!㍿'s\r\n\r\n>\"字 \n‍", "tokens": 43, "pieces": ["ꟲ", "<|", "endoftext", "|>‍😀🏽<", "META", "_START", ">​('", "ſ", "<|", "endoftext", "|>", "\t ", " ​", "\u000b", "'D", "!!㍿'", "s", "\r\n\r\n", ">\"", "字", " \n", "‍"]} +{"text": "m \n漢'S \"İ漢𐞁", "tokens": 15, "pieces": ["m", " \n", "漢", "'", "S", " ", " \"", "İ漢𐞁"]} +{"text": " EOTé٣٤٥٦a#$%عEOT#$%ع\n🙂 aZ​EOT\r\nع字m'Re𐞁Ⅳ'SA'S  \u000b
!EOT'll字", "tokens": 51, "pieces": [" EOTé", "٣٤٥", "٦", "a", "#$%", "ع", "EOT", "#$%", "ع", "\n", "🙂", " ", " a", "Z", "​EOT", "\r\n", "ع字m'Re", "𐞁", "Ⅳ", "'SA'S", "  \u000b", "
", "!EOT'll", "字"]} +{"text": "́عé३'re!!.'­a !0s…\"åé's😀🏽9", "tokens": 24, "pieces": ["́عé", "३", "'re", "!!.'­", "a", " !", "0", "s", "…", "\"åé's", "😀🏽", "9"]} +{"text": "ḍ̇é\t>\r.d‍'Re字 \tß!\t \n漢'fi t", "tokens": 25, "pieces": ["ḍ̇é", "\t", ">\r", ".d", "‍'", "Re字", " ", "\tß", "!<", "EOT", ">", "\t \n", "漢", "'fi", " t"]} +{"text": "Z'S0A\r\n\r\n\r\n\r\nß,㋿", "tokens": 10, "pieces": ["Z'S", "0", "A", "\r\n\r\n\r\n\r\n", "ß", ",㋿"]} +{"text": "½é½😀🏽#$%­😀🏽0 9ſ(\r\r>𐞁'Sß-<|fim_prefix|>👍🏽\n👍🏽İ\"عZ'Z,
", "tokens": 48, "pieces": ["½", "é", "½", "😀🏽#$%­😀🏽", "0", " ", "9", "ſ", "(\r\r", ">𐞁'S", "ß", "-<|", "fim", "_prefix", "|>👍🏽\n", "👍🏽", "İ", "\"ع", "Z", "'Z", ",", "
"]} +{"text": "!DžEOT's𐞁t ́\t🙂a٣٤٥٦ḍ̇ꟲAZ .m0 !!İ'Dt\r'll<|endoftext|>Z\nZ#$%'s­㋿‍#$%", "tokens": 62, "pieces": ["!DžEOT's", "𐞁t", " ́", "\t", "🙂a", "٣٤٥", "٦", "ḍ̇ꟲ", "AZ", " ", " <", "EOT", ">.", "m", "0", " ", " !!", "İ'D", "t", "\r", "'ll", "<|", "endoftext", "|>", "Z", "\n", "Z", "#$%'", "s", "­㋿‍#$%"]} +{"text": "<(d İs 9!!\r\n\r\n<|endoftext|>s…\r\n'ſ'D'ſ", "tokens": 24, "pieces": ["<(", "d", " ", " İs", " ", "9", "!!\r\n\r\n", "<|", "endoftext", "|>", "s", "…\r\n", "'ſ'D", "'ſ"]} +{"text": "́½0", "tokens": 3, "pieces": ["́", "½0"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㋿é­", "tokens": 5, "pieces": ["㋿é", "­"]} +{"text": "😀🏽aع'VE\"9㍿#$%😀🏽
eEOT!!'sſ<|fim_prefix|>'Re\u000bZ漢'T>­𐞁३ e'ſ‍", "tokens": 51, "pieces": ["😀🏽<", "EOT", ">aع'VE", "\"", "9", "㍿#$%😀🏽", "
e", "EOT", "!!'", "sſ", "<|", "fim", "_prefix", "|>'", "Re", "\u000bZ漢'T", ">­", "𐞁", "३", " e'ſ", "‍"]} +{"text": "#$% #$%Zꟲ漢 EOT'ſ'reA­'VE漢😀🏽३😀🏽s9( >\r \n\u000b
m<|fim_prefix|>< \n٣٤٥٦​e\r\n٣٤٥٦
", "tokens": 56, "pieces": ["#$%", " ", "#$%", "Zꟲ漢", " EOT'ſ", "'re", "A", "­'", "VE漢", "😀🏽", "३", "😀🏽", "s", "9", "(", " >\r", " \n", "\u000b", "
m", "<|", "fim", "_prefix", "|><", " \n", "٣٤٥", "٦", "​e", "\r\n", "٣٤٥", "٦", "
"]} +{"text": "'s 
\nİ,'VE -a \n\r12345678#$%'S", "tokens": 21, "pieces": ["'s", "", " 
\n", "İ", ",'", "VE", " ", "-a", " \n\r", "123", "456", "78", "#$%'", "S"]} +{"text": "'Mdꟲ字", "tokens": 6, "pieces": ["'Mdꟲ字"]} +{"text": "…ſ 12345678…㍿  😀🏽m!!ḍ̇", "tokens": 22, "pieces": ["…ſ", " ", "123", "456", "78", "…", "㍿", " ", " ", "😀🏽", "m", "!!", "ḍ̇"]} +{"text": "12345678e.fi're#$%­ \n,漢𐞁👍🏽", "tokens": 20, "pieces": ["123", "456", "78", "e", ".fi're", "#$%­", " \n", ",漢𐞁", "👍🏽"]} +{"text": ". 漢sss\u000b\r\n\r\nſs\n😀🏽\t'ret EOT!!‍㍿,\r're\u000b'Re\"Ⅳ'M ", "tokens": 33, "pieces": [".", " 漢sss", "\u000b\r\n\r\n", "ſs", "\n", "😀🏽", "\t", "'ret", " EOT", "!!‍㍿,\r", "'re", "\u000b", "'Re", "\"", "Ⅳ", "'M", " "]} +{"text": "!!(!!🙂A㋿㍿0㍿", "tokens": 15, "pieces": ["!!(!!🙂", "A", "㋿㍿", "0", "㍿"]} +{"text": "𐞁Ź\r\n\r\n're'Tſ'D\td'T're9­​e🙂Zع\n'D­\r\n\u000b😀🏽½.ع字😀🏽İEOT½", "tokens": 44, "pieces": ["𐞁Ź", "\r\n\r\n", "'re'T", "ſ'D", "\td'T", "'re", "9", "­<", "EOT", ">​", "e", "🙂Zع", "\n", "'D", "­\r\n", "\u000b", "😀🏽", "½", ".ع字", "😀🏽", "İEOT", "½"]} +{"text": "İ>", "tokens": 2, "pieces": ["İ", ">"]} +{"text": "é𐞁漢", "tokens": 6, "pieces": ["é𐞁漢"]} +{"text": "!! \n½👍🏽😀🏽#$%<|endoftext|><|fim_prefix|>.㋿<|endoftext|>\u000b", "tokens": 34, "pieces": ["!!", " \n", "½", "👍🏽😀🏽#$%<|", "endoftext", "|><|", "fim", "_prefix", "|>.㋿<|", "endoftext", "|>", "\u000b"]} +{"text": "'ll\r\n<|endoftext|>İ­\r\n\r\n's t's'reét\r٣٤٥٦dd(>m<|endoftext|>'D㍿é", "tokens": 43, "pieces": ["'ll", "\r\n", "<|", "endoftext", "|>", "İ", "­\r\n\r\n", "'s", " t's", "'reét", "\r", "٣٤٥", "٦", "dd", "(><", "EOT", ">m", "<|", "endoftext", "|>'", "D", "㍿é"]} +{"text": "𐞁𐞁'T", "tokens": 9, "pieces": ["𐞁𐞁'T"]} +{"text": "12345678'T'ſ#$%d​EOTⅣ-…<|fim_prefix|>ꟲ's12345678Džé'llm'VE(Aåİ́", "tokens": 43, "pieces": ["123", "456", "78", "'T'ſ", "#$%", "d", "​EOT", "Ⅳ", "-", "…", "<|", "fim", "_prefix", "|>", "ꟲ's", "123", "456", "78", "Džé'll", "m'VE", "(Aå", "İ́"]} +{"text": "👍🏽12345678­  \n0å'S​́ſ0Dž9s12345678é\"ſEOT'llḍ̇-éDžfi're'S", "tokens": 44, "pieces": ["👍🏽", "123", "456", "78", "­", " ", "", " \n", "0", "å'S", "​́ſ", "0", "Dž", "9", "s", "123", "456", "78", "é", "\"ſ", "EOT'll", "ḍ̇", "-é", "Džfi're", "'S"]} +{"text": "'llåfi😀🏽e\r\n\t d<|fim_prefix|>", "tokens": 18, "pieces": ["'llåfi", "😀🏽", "e", "\r\n", "\t", " d", "<|", "fim", "_prefix", "|>"]} +{"text": "'Re\r\n\r\n<|endoftext|>#$%( \n", "tokens": 15, "pieces": ["'Re", "\r\n\r\n", "<|", "endoftext", "|>#$%(", " \n"]} +{"text": ",­s.\t\u000b㋿
!!😀🏽'M<|endoftext|>#$%㋿漢e's🙂\t३e'T'm!!.,EOT
­mꟲZ", "tokens": 55, "pieces": [",<", "META", "_START", ">­", "s", ".", "\t", "\u000b", "㋿", "
", "!!😀🏽'", "M", "<|", "endoftext", "|>#$%㋿", "漢e's", "🙂", "\t", "३", "e'T", "'m", "!!.,", "EOT", "
", "­", "mꟲ", "Z"]} +{"text": "​>ꟲꟲ㋿\r\nſ👍🏽'Re'T ½'M(", "tokens": 23, "pieces": ["​>", "ꟲꟲ", "㋿\r\n", "ſ", "👍🏽'", "Re'T", " ", "½", "'M", "("]} +{"text": "Aꟲſ
!dfiEOT12345678", "tokens": 14, "pieces": ["Aꟲſ", "
", "!dfi", "EOT", "123", "456", "78"]} +{"text": "字'S<|endoftext|>9('llꟲ,d-\r\n\r\n12345678té\r\n\r\n٣٤٥٦'S9🙂\"-ꟲA9 \n>'llſ­EOT", "tokens": 43, "pieces": ["字'S", "<|", "endoftext", "|>", "9", "('", "llꟲ", ",d", "-\r\n\r\n", "123", "456", "78", "té", "\r\n\r\n", "٣٤٥", "٦", "'S", "9", "🙂\"-", "ꟲ", "A", "9", " \n", ">'", "llſ", "­EOT"]} +{"text": " ३fi㍿(\re'reſ\"'VE0😀🏽!!𐞁12345678<|endoftext|>ع ZZ'Reé­a9🙂́'ſ­s​", "tokens": 45, "pieces": [" ", " ", "३", "fi", "㍿(\r", "e're", "ſ", "\"'", "VE", "0", "😀🏽!!", "𐞁", "123", "456", "78", "<|", "endoftext", "|>", "ع", " ZZ'Re", "é", "­a", "9", "🙂́'ſ", "­s", "​"]} +{"text": "‍'re!!'Re\rfiDž½🙂'Dḍ̇d'ſ…'s\r\n\r\n\r\n('", "tokens": 25, "pieces": ["‍'", "re", "!!'", "Re", "\r", "fi", "Dž", "½", "🙂'", "Dḍ̇d'ſ", "…", "'s", "\r\n\r\n\r\n", "('"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "!
\t​İ'M ㋿'Dm>ꟲ
'llEOTꟲ ꟲꟲ\"fi\r\nm!ß(…\u000b…𐞁", "tokens": 45, "pieces": ["!", "
", "\t", "​İ'M", " ㋿'", "Dm", ">ꟲ", "
", "'ll", "EOTꟲ", " ꟲꟲ", "\"fi", "\r\n", "m", "!ß", "(", "…\u000b", "…𐞁"]} +{"text": "٣٤٥٦'D \n#$%\t0-aå😀🏽0‍!e㍿🙂#$%fi-…٣٤٥٦é", "tokens": 34, "pieces": ["٣٤٥", "٦", "'D", " \n", "#$%", "\t", "0", "-aå", "😀🏽", "0", "‍!", "e", "㍿🙂#$%", "fi", "-", "…", "٣٤٥", "٦", "é"]} +{"text": "'re\r\n​!! \n​漢A\r\n㍿!!\r\nع.😀🏽'ſ<|fim_prefix|>\rZ<\r\n", "tokens": 33, "pieces": ["'re", "\r\n", "​!!", " \n", "​漢", "A", "\r\n", "㍿!!\r\n", "ع", ".😀🏽'", "ſ", "<|", "fim", "_prefix", "|>\r", "Z", "<\r\n"]} +{"text": "­ꟲ9'M.Z 😀🏽'VE's😀🏽>\r\n\r\nß .
 \ń'sḍ̇'VEm", "tokens": 36, "pieces": ["­ꟲ", "9", "'M", ".Z", " ", "😀🏽'", "VE's", "😀🏽>\r\n\r\n", "ß", " .", "
 \n", "́", "'", "sḍ̇", "'", "VEm"]} +{"text": "Dž \r\n\r\nß㍿'re<|fim_prefix|>👍🏽'Re ​0", "tokens": 23, "pieces": ["Dž", " \r\n\r\n", "ß", "㍿'", "re", "<|", "fim", "_prefix", "|>👍🏽'", "Re", " ​", "0"]} +{"text": "İe'll", "tokens": 3, "pieces": ["İe'll"]} +{"text": "ſ<|endoftext|>
é0'Re#$%daꟲ 'S٣٤٥٦'Re½漢\r\n ٣٤٥٦́ ㍿A'll㋿(", "tokens": 51, "pieces": ["ſ", "<|", "endoftext", "|>", "
é", "0", "'Re", "#$%<", "EOT", ">daꟲ", " ", " '", "S", "٣٤٥", "٦", "'Re", "½", "漢", "\r\n", " ", " ", "٣٤٥", "٦", "́", " ", "㍿A'll", "㋿("]} +{"text": "٣٤٥٦-tꟲ9EOTſ\u000b́Zm0-.<|endoftext|>\r\n ́ <'s<|endoftext|>
", "tokens": 45, "pieces": ["٣٤٥", "٦", "-tꟲ", "9", "EOT", "ſ", "\u000b́Zm", "0", "-.<|", "endoftext", "|>\r\n", " ", " ́", " ", " <'", "s", "<|", "endoftext", "|>", "
"]} +{"text": "a'd12345678𐞁", "tokens": 9, "pieces": ["a'd", "123", "456", "78", "𐞁"]} +{"text": "ß\r'ſ\u000b٣٤٥٦>Ⅳ㍿Ⅳséåſ𐞁\u000bİ'VE字EOT'Re😀🏽 0😀🏽eaⅣ", "tokens": 60, "pieces": ["ß", "\r", "'ſ", "\u000b", "٣٤٥", "٦", ">", "Ⅳ", "㍿", "Ⅳ", "séå", "ſ𐞁", "\u000bİ'VE", "字", "EOT'Re", "😀🏽", " ", "0", "😀🏽", "ea", "Ⅳ"]} +{"text": "\r\n\r\n<|endoftext|>!!\u000b'Dḍ̇å\r …a<|fim_prefix|>!!#$%👍🏽ꟲⅣ-\tt.\r\n
\"​𐞁\"\r\n\r\n'Z'ſ'S#$%", "tokens": 63, "pieces": ["\r\n\r\n", "<|", "endoftext", "|>!!", "\u000b", "'Dḍ̇å", "\r", " ", "…a", "<|", "fim", "_prefix", "|>!!#$%👍🏽", "ꟲ", "Ⅳ", "-", "\tt", ".<", "EOT", ">\r\n", "
", "\"​", "𐞁", "\"\r\n\r\n", "'Z'ſ", "'S", "#$%<", "META", "_START", ">"]} +{"text": "\r'T½\n٣٤٥٦'\"m\"𐞁", "tokens": 22, "pieces": ["\r", "'T", "½", "\n", "٣٤٥", "٦", "'\"", "m", "\"𐞁", ""]} +{"text": "\r\nⅣ\r'Reع㋿​d🙂m<|endoftext|>12345678dḍ̇\"'sſ'Re½t#$%🙂e#$% 'll", "tokens": 45, "pieces": ["\r\n", "Ⅳ", "\r", "'Reع", "㋿​", "d", "🙂m", "<|", "endoftext", "|>", "123", "456", "78", "dḍ̇", "\"'", "sſ", "'", "Re", "½", "t", "#$%🙂", "e", "#$%", " ", "'ll"]} +{"text": "🙂!!\n<|fim_prefix|>​'M0<|endoftext|>…‍fieſaet😀🏽​aZ  'll👍🏽A", "tokens": 40, "pieces": ["🙂!!\n", "<|", "fim", "_prefix", "|>​'", "M", "0", "<|", "endoftext", "|>", "…", "‍fieſaet", "😀🏽​", "a", "Z", " ", " ", "'ll", "👍🏽", "A"]} +{"text": "#$%>. 😀🏽e\u000b'll👍🏽EOT\"٣٤٥٦eꟲ!!>٣٤٥٦12345678<\"ſa'Re'S\tDž12345678'D", "tokens": 45, "pieces": ["#$%>.", " 😀🏽", "e", "\u000b", "'ll", "👍🏽", "EOT", "\"", "٣٤٥", "٦", "eꟲ", "!!>", "٣٤٥", "٦12", "345", "678", "<\"", "ſa'Re", "'S", "\tDž", "123", "456", "78", "'D"]} +{"text": "㍿\nع\r'ſ912345678Ⅳå,'M'Ré", "tokens": 19, "pieces": ["㍿\n", "ع", "\r", "'ſ", "912", "345", "678", "Ⅳ", "å", ",'", "M'Re", "́"]} +{"text": "t'Dm🙂<\n0\r\n\r\n(d-🙂😀🏽-'SA😀🏽", "tokens": 19, "pieces": ["t'D", "m", "🙂<\n", "0", "\r\n\r\n", "(d", "-🙂😀🏽-'", "SA", "😀🏽"]} +{"text": "Ⅳ9små're३ >​\"fi's<|fim_prefix|>!٣٤٥٦d‍İ#$%<|fim_prefix|>", "tokens": 35, "pieces": ["Ⅳ9", "små're", "३", " >​\"", "fi's", "<|", "fim", "_prefix", "|>!", "٣٤٥", "٦", "d", "‍İ", "#$%<|", "fim", "_prefix", "|>"]} +{"text": "㋿ Z", "tokens": 5, "pieces": ["㋿", " Z"]} +{"text": "\"'\r\nİ<|fim_prefix|>. \nå.(\u000bfi🙂ḍ̇  ٣٤٥٦İ𐞁åſ👍🏽", "tokens": 38, "pieces": ["\"'\r\n", "İ", "<|", "fim", "_prefix", "|>.", " \n", "å", ".(", "\u000bfi", "🙂ḍ̇", " ", " ", "٣٤٥", "٦", "İ𐞁åſ", "👍🏽"]} +{"text": "é \r𐞁½🙂३", "tokens": 14, "pieces": ["é", " \r", "𐞁", "½", "🙂<", "META", "_START", ">", "३"]} +{"text": "å👍🏽'Dع  ́𐞁㍿'re👍🏽㍿½漢<|fim_prefix|> 😀🏽'ſå e-㍿३", "tokens": 52, "pieces": ["å", "👍🏽'", "D", "ع", " ", " ́𐞁", "㍿'", "re", "👍🏽㍿", "½", "漢", "<|", "fim", "_prefix", "|>", " ", "😀🏽'", "ſå", " ", " e", "-㍿", "३"]} +{"text": "漢٣٤٥٦\r<|endoftext|>👍🏽٣٤٥٦½t…'VE'VE", "tokens": 28, "pieces": ["漢", "٣٤٥", "٦", "\r", "<|", "endoftext", "|>👍🏽", "٣٤٥", "٦½", "t", "…", "'VE'VE"]} +{"text": " \nİ-<漢\r\"\u000b
'S\r<|fim_prefix|>👍🏽sععt́㍿sꟲ𐞁\t'9\t😀🏽", "tokens": 52, "pieces": [" \n", "İ", "-<", "漢", "\r", "\"", "\u000b", "
", "'S", "\r", "<|", "fim", "_prefix", "|><", "META", "_START", ">👍🏽", "sععt́", "㍿s", "ꟲ𐞁", "\t", "'", "9", "\t", "😀🏽"]} +{"text": " 👍🏽 #$%-​<ꟲİ'ſ㍿EOT><,!me👍🏽A\n.d😀🏽Z'VE'VE>m'sZ", "tokens": 44, "pieces": [" ", " 👍🏽", " #$%-​<", "META", "_START", "><", "ꟲ", "İ'ſ", "㍿EOT", "><,!", "me", "👍🏽", "A", "\n", ".d", "😀🏽", "Z'VE", "'VE", ">m's", "Z"]} +{"text": "ſ 'Re-e\nDž\n", "tokens": 11, "pieces": ["ſ", " ", "'Re", "-e", "\n", "Dž", "\n", ""]} +{"text": "ḍ̇ſésA'é'lls'ſ­'M'll'sa😀🏽s", "tokens": 26, "pieces": ["ḍ̇ſés", "A", "'é'll", "s'ſ", "­'", "M'll", "'sa", "😀🏽", "s"]} +{"text": "'​\u000b'Mꟲs字👍🏽t\n𐞁\"İZ'ſ½\r ſ'T🙂㋿😀🏽٣٤٥٦<|fim_prefix|>é½ A​漢'VE9🙂é", "tokens": 64, "pieces": ["'​", "\u000b", "'Mꟲs字", "👍🏽", "t", "\n", "𐞁", "\"İZ'ſ", "½", "\r", " ſ'T", "🙂㋿😀🏽", "٣٤٥", "٦", "<|", "fim", "_prefix", "|><", "META", "_START", ">é", "½", " A", "​<", "META", "_START", ">漢'VE", "9", "🙂é"]} +{"text": "
sé'M'VEsDžé.(‍\"'ſ", "tokens": 13, "pieces": ["
sé'M", "'VEs", "Džé", ".(‍\"'", "ſ"]} +{"text": "'T12345678-\nd!!mé,9'll 'D\" ḍ̇m'ſ'T…<|endoftext|>12345678A字#$%(fi12345678漢", "tokens": 43, "pieces": ["'T", "123", "456", "78", "-\n", "d", "!!", "mé", ",", "9", "'ll", " '", "D", "\"", " ḍ̇m'ſ", "'T", "…", "<|", "endoftext", "|>", "123", "456", "78", "A字", "#$%(", "fi", "123", "456", "78", "漢"]} +{"text": "<|endoftext|>Dž", "tokens": 9, "pieces": ["<|", "endoftext", "|>", "Dž"]} +{"text": "‍<|endoftext|>><|fim_prefix|>'ſ \nd's'llmⅣ'ſ'VE٣٤٥٦'VE\u000b'VE#$%dⅣꟲ字‍\r", "tokens": 49, "pieces": ["‍<|", "endoftext", "|>><", "EOT", "><|", "fim", "_prefix", "|>'", "ſ", " \n", "d's", "'llm", "Ⅳ", "'ſ'VE", "٣٤٥", "٦", "'VE", "\u000b", "'VE", "#$%", "d", "Ⅳ", "ꟲ字", "‍\r"]} +{"text": "'VEå9fi 'll-🙂!!
‍ééA\t\rfié'D", "tokens": 28, "pieces": ["'VEå", "9", "fi", " ", " '", "ll", "-🙂!!", "
", "‍éé", "A", "\t\r", "fié", "'", "D"]} +{"text": "\rDž#$%0\r\n㍿́12345678å😀🏽'ſ👍🏽", "tokens": 24, "pieces": ["\r", "Dž", "#$%", "0", "\r\n", "㍿́", "123", "456", "78", "å", "😀🏽'", "ſ", "👍🏽"]} +{"text": " ́<'Re912345678!'re<|fim_prefix|>0\"EOT​t#$%9Z12345678­'ll\"'re🙂😀🏽  0<'ſ", "tokens": 45, "pieces": [" ", " ́", "<'", "Re", "912", "345", "678", "!'", "re", "<|", "fim", "_prefix", "|><", "META", "_START", ">", "0", "\"EOT", "​t", "#$%", "9", "Z", "123", "456", "78", "­'", "ll", "\"'", "re", "🙂😀🏽", " ", " ", "0", "<'", "ſ"]} +{"text": "\r\n12345678#$%​<|endoftext|>>…EOTⅣ'Séå,́\tA\nßZ 😀🏽٣٤٥٦'M!#$%", "tokens": 43, "pieces": ["\r\n", "123", "456", "78", "#$%​<|", "endoftext", "|>>", "…EOT", "Ⅳ", "'Séå", ",́", "\tA", "\n", "ß", "Z", " ", "😀🏽", "٣٤٥", "٦", "'M", "!#$%"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'!e'ſ!漢'M're0\"İ\r\n\r\n\"😀🏽ꟲeåé>s", "tokens": 25, "pieces": ["'!", "e'ſ", "!漢'M", "'re", "0", "\"İ", "\r\n\r\n", "\"😀🏽", "ꟲeåé", ">s"]} +{"text": "Dž<|endoftext|>!!ſ …t0 å'ſ", "tokens": 21, "pieces": ["Dž", "<|", "endoftext", "|>!!", "ſ", " ", "…t", "0", " å'ſ"]} +{"text": "😀🏽\r's12345678Dž'漢​…(s'0\r\n\r\n…\r٣٤٥٦İ漢<|endoftext|>'Re\u000b ", "tokens": 38, "pieces": ["😀🏽\r", "'s", "123", "456", "78", "Dž", "'漢", "​", "…", "(s", "'", "0", "\r\n\r\n…\r", "٣٤٥", "٦", "İ漢", "<|", "endoftext", "|>'", "Re", "\u000b "]} +{"text": "'D!!m'Sé​ß \"‍Z'T'ssİ‍Dž<\t>Dž", "tokens": 23, "pieces": ["'D", "!!", "m'S", "é", "​ß", " ", " \"‍", "Z'T", "'ss", "İ", "‍Dž", "<", "\t", ">Dž"]} +{"text": "ꟲ(9 \n
m👍🏽0Ⅳ\r'T 𐞁!\r\n\r\n>­ſ \u000b>\t'd'S🙂Ⅳé<\"", "tokens": 39, "pieces": ["ꟲ", "(", "9", " \n", "
m", "👍🏽", "0Ⅳ", "\r", "'T", " ", " 𐞁", "!\r\n\r\n", ">­", "ſ", " ", "\u000b", ">", "\t", "'d'S", "🙂", "Ⅳ", "é", "<\""]} +{"text": "𐞁'S#$%字ſ\u000b", "tokens": 14, "pieces": ["𐞁", "'", "S", "#$%", "字ſ", "\u000b"]} +{"text": ".🙂ꟲ<|endoftext|>s𐞁३12345678!ß(½ét!é  t\r\n३.", "tokens": 36, "pieces": [".🙂", "ꟲ", "<|", "endoftext", "|>", "s𐞁", "३12", "345", "678", "!ß", "(", "½", "ét", "!é", " ", " t", "\r\n", "३", "."]} +{"text": " ́s'VE.'S٣٤٥٦३٣٤٥٦-sſ'Ms…9're😀🏽t́Afi㋿‍'Re<½'M字0-'re!!EOT'Ⅳå", "tokens": 54, "pieces": [" ́s'VE", ".'", "S", "٣٤٥", "٦३٣", "٤٥٦", "-sſ'M", "s", "…", "9", "'re", "😀🏽", "t́", "Afi", "㋿‍'", "Re", "<", "½", "'M字", "0", "-'", "re", "!!", "EOT", "'", "Ⅳ", "å"]} +{"text": "é('D 9ḿ'ſ\r'S", "tokens": 11, "pieces": ["é", "('", "D", " ", "9", "ḿ'ſ", "\r", "'S"]} +{"text": " Zſ'\"", "tokens": 3, "pieces": [" Zſ", "'\""]} +{"text": "漢​'Re<|endoftext|>Ⅳfi!!", "tokens": 18, "pieces": ["漢", "​'", "Re", "<|", "endoftext", "|>", "Ⅳ", "fi", "!!"]} +{"text": "'M字㋿éİß!!㋿ḍ̇Asꟲfi'T's漢!(ḍ̇Ⅳ😀🏽0åå9é", "tokens": 44, "pieces": ["'M字", "㋿", "é", "İß", "!!㋿", "ḍ̇", "Asꟲfi'T", "'s漢", "!(", "ḍ̇", "Ⅳ", "😀🏽", "0", "åå", "9", "é"]} +{"text": "\te字ꟲ….ſ", "tokens": 9, "pieces": ["\te字ꟲ", "…", ".ſ"]} +{"text": "éd…­ ㋿٣٤٥٦​'t㋿İ.'Ts0𐞁
🙂'lltع !!fi👍🏽!!́ ", "tokens": 42, "pieces": ["éd", "…", "­", " ㋿", "٣٤٥", "٦", "​'", "t", "㋿İ", ".'", "Ts", "0", "𐞁", "
", "🙂'", "lltع", " ", " !!", "fi", "👍🏽!!́", " "]} +{"text": "🙂'Re🙂'll​漢'D(\r\n\r\nİ!👍🏽's'M\n-😀🏽Z\r\r\n<<12345678İa\r\n\r\n", "tokens": 33, "pieces": ["🙂'", "Re", "🙂'", "ll", "​漢'D", "(\r\n\r\n", "İ", "!👍🏽'", "s'M", "\n", "-😀🏽", "Z", "\r\r\n", "<<", "123", "456", "78", "İa", "\r\n\r\n"]} +{"text": "'ll字é'M\r\n\r\n\n<|endoftext|>'D'🙂Džéİ<'reå'Reé<|endoftext|> 'T\r'D…
s12345678>< sḍ̇𐞁٣٤٥٦'ll<fi\r\n\r\n", "tokens": 66, "pieces": ["'ll字é'M", "\r\n\r\n\n", "<|", "endoftext", "|>'", "D", "'🙂", "Džé", "İ", "<'", "reå'Re", "é", "<|", "endoftext", "|>", " ", "'T", "\r", "'", "D", "…", "
s", "123", "456", "78", "><", " ", " sḍ̇𐞁", "٣٤٥", "٦", "'ll", "<fi", "\r\n\r\n"]} +{"text": "
㍿'DA\r\n\r\n'!!>.'ſꟲ\neDž'Re\u000bع-🙂éé,m-\raİ ", "tokens": 35, "pieces": ["
", "㍿'", "DA", "\r\n\r\n", "'!!>.'", "ſꟲ", "\n", "e", "Dž'Re", "\u000bع", "-🙂", "éé", ",m", "-\r", "a", "İ", " "]} +{"text": "é0'Re٣٤٥٦", "tokens": 8, "pieces": ["é", "0", "'Re", "٣٤٥", "٦"]} +{"text": "d👍🏽 '\u000b…字'll٣٤٥٦ ſ'D", "tokens": 22, "pieces": ["d", "👍🏽", " ", " '", "\u000b", "…字'll", "٣٤٥", "٦", "", " ſ'D"]} +{"text": "🙂­㋿́s​å​'redfiİ'D
\r\n\r\ne0'T,e…'re漢é!!", "tokens": 29, "pieces": ["🙂­㋿́", "s", "​å", "​'", "redfi", "İ'D", "
\r\n\r\n", "e", "0", "'T", ",e", "…", "'re漢é", "!!"]} +{"text": "'0! \nⅣ9å㍿𐞁Z12345678é's…㍿!!('T
", "tokens": 32, "pieces": ["'", "0", "!", " \n", "Ⅳ9", "å", "㍿𐞁", "Z", "123", "456", "78", "é's", "…", "㍿!!('", "T", "
"]} +{"text": "m🙂'Ré", "tokens": 5, "pieces": ["m", "🙂'", "Ré"]} +{"text": "٣٤٥٦ع́ḍ̇,👍🏽(́\r\nm's­'VE
👍🏽\"\r\nſ\u000b sZ٣٤٥٦.'Te𐞁\r'Da,ḍ̇\u000bꟲ!!", "tokens": 52, "pieces": ["٣٤٥", "٦", "ع́ḍ̇", ",👍🏽(́\r\n", "m's", "­'", "VE", "
", "👍🏽\"\r\n", "ſ", "\u000b", " s", "Z", "٣٤٥", "٦", ".'", "Te𐞁", "\r", "'Da", ",ḍ̇", "\u000bꟲ", "!!"]} +{"text": "12345678'MsA𐞁e٣٤٥٦🙂 ", "tokens": 21, "pieces": ["123", "456", "78", "'Ms", "A𐞁e", "٣٤٥", "٦", "🙂<", "EOT", ">", " "]} +{"text": "fi­A字mⅣ'D𐞁'T👍🏽😀🏽're!#$%٣٤٥٦\nd", "tokens": 30, "pieces": ["fi", "­A字m", "Ⅳ", "'D𐞁'T", "👍🏽😀🏽'", "re", "!#$%", "٣٤٥", "٦", "\n", "d"]} +{"text": "<'ReA\r\ns's'ret㍿́𐞁.12345678-'S🙂>字\r\n\r\ns㍿\r\ne\r\na12345678👍🏽\r\n", "tokens": 57, "pieces": ["<'", "Re", "A", "\r\n", "s", "'", "s're", "t", "㍿́𐞁", ".", "123", "456", "78", "-'", "S", "🙂>", "字", "\r\n\r\n", "s", "㍿\r\n", "e", "\r\n", "a", "123", "456", "78", "👍🏽\r\n"]} +{"text": "m'ReⅣ​㋿…'VEعßd'S'Re 漢12345678Dž \té \n's.\rfi \n.", "tokens": 37, "pieces": ["m'Re", "Ⅳ", "​㋿", "…", "'VEعßd'S", "'Re", " 漢", "123", "456", "78", "Dž", " ", "\té", " \n", "'s", ".\r", "fi", " \n", ".<", "EOT", ">"]} +{"text": "​>", "tokens": 2, "pieces": ["​>"]} +{"text": "!!t३#$% \n\r\n\r㍿#$%9ß'ḍ̇­mdé
́.", "tokens": 29, "pieces": ["!!", "t", "३", "#$%", " \n\r\n\r", "㍿#$%", "9", "ß", "'ḍ̇", "­mdé", "
́", "."]} +{"text": "m t 'reḍ̇​<|endoftext|>0<|endoftext|>👍🏽", "tokens": 31, "pieces": ["m", " ", " t", " ", "'reḍ̇", "​<|", "endoftext", "|>", "0", "<|", "endoftext", "|>👍🏽<", "META", "_START", ">"]} +{"text": " s ,İꟲع''s'S字t漢-😀🏽12345678\r<|fim_prefix|>ꟲ's\ré's'Dfi<|endoftext|>ḍ̇'ſDž'll,字", "tokens": 62, "pieces": [" s", " ,", "İꟲع", "'<", "EOT", ">'", "s'S", "字t漢", "-<", "EOT", ">😀🏽", "123", "456", "78", "\r", "<|", "fim", "_prefix", "|>", "ꟲ's", "\r", "é's", "'Dfi", "<|", "endoftext", "|>", "ḍ̇'ſ", "Dž'll", ",字"]} +{"text": "
m<|endoftext|><|endoftext|>㋿<|endoftext|>­\"(½\u000b\r'S\"(a'Sḍ̇é½'ſå😀🏽a­🙂­ \nſ'Re👍🏽,\u000bd 👍🏽", "tokens": 68, "pieces": ["
m", "<|", "endoftext", "|><|", "endoftext", "|>㋿<|", "endoftext", "|>­<", "EOT", ">\"(", "½", "\u000b\r", "'S", "\"(", "a'S", "ḍ̇é", "½", "'ſå", "😀🏽", "a", "­🙂­", " \n", "ſ'Re", "👍🏽,", "\u000bd", " ", "👍🏽"]} +{"text": "d‍३", "tokens": 3, "pieces": ["d", "‍", "३"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\u000b,\r٣٤٥٦'ll ‍<-!­.\r", "tokens": 15, "pieces": ["\u000b", ",\r", "٣٤٥", "٦", "'ll", " ", "‍<-!­.\r"]} +{"text": "😀🏽'D​​字", "tokens": 7, "pieces": ["😀🏽'", "D", "​​", "字"]} +{"text": "'S­ \nⅣ>", "tokens": 6, "pieces": ["'S", "­", " \n", "Ⅳ", ">"]} +{"text": "́İ㋿ 字é!'Re😀🏽#$%字 ‍é''VE\tⅣd\rA ", "tokens": 28, "pieces": ["­m", "\r", "\"<('", "Re", " s", ">😀🏽#$%", "字", " ", " ‍", "é", "''", "VE", "\t", "Ⅳ", "d", "\r", "A", " "]} +{"text": "'VE<|endoftext|>\r\n\r\ntaEOT<\u000b🙂​A-́𐞁-'ſ-👍🏽'́d​ſ​e'M<9\r\n-'s'Déet́9", "tokens": 51, "pieces": ["'VE", "<|", "endoftext", "|>\r\n\r\n", "ta", "EOT", "<", "\u000b", "🙂​", "A", "-́", "𐞁", "-'", "ſ", "-👍🏽'́", "d", "​ſ", "​e'M", "<", "9", "\r\n", "-'", "s'D", "éet́", "9"]} +{"text": "३\t 'VE12345678'S \nſDž\r\néd'M'S…s\n >t<|fim_prefix|>👍🏽<|fim_prefix|>́'VE1234567812345678<|endoftext|>ꟲ३", "tokens": 64, "pieces": ["३", "\t", " ", "'VE", "123", "456", "78", "'S", " \n", "ſ", "Dž", "\r\n", "éd'M", "'S", "…s", "\n", " ", ">", "t", "<|", "fim", "_prefix", "|>👍🏽<|", "fim", "_prefix", "|>́'", "VE", "123", "456", "781", "234", "567", "8", "<|", "endoftext", "|>", "ꟲ", "३"]} +{"text": "0'M ḍ̇🙂 ", "tokens": 8, "pieces": ["0", "'M", " ḍ̇", "🙂", " "]} +{"text": "-<|fim_prefix|>Dž́㋿-'TⅣ🙂३ \n漢're-e'VE'M(\"𐞁字", "tokens": 35, "pieces": ["-<|", "fim", "_prefix", "|>", "Dž́", "㋿-'", "T", "Ⅳ", "🙂", "३", " \n", "漢're", "-e'VE", "'M", "(\"", "𐞁字"]} +{"text": ".㋿㍿", "tokens": 7, "pieces": [".㋿㍿"]} +{"text": ",0'll#$%ꟲ\t-… ३ A ſİ'S'Smfi\n
", "tokens": 26, "pieces": [",", "0", "'ll", "#$%", "ꟲ", "\t", "-", "…", " ", "३", " ", " A", " ſ", "İ'S", "'Smfi", "\n", "
"]} +{"text": "ع́​  <字 \n'VÉ#$%\u000b('D å३½'TDžA-EOT ꟲßEOTA'S", "tokens": 36, "pieces": ["ع́", "​", " ", " ", "<字", " \n", "'VÉ", "#$%", "\u000b", "('", "D", " å", "३½", "'TDžA", "-EOT", " ꟲß", "EOTA'S"]} +{"text": "ع‍'s'S٣٤٥٦Z字'VEİ<'reéé!!EOT12345678…½<'VE\tåaⅣ…éZ'S\n
A-​\r\n\r\n", "tokens": 48, "pieces": ["ع", "‍'", "s'S", "٣٤٥", "٦", "Z字'VE", "İ", "<'", "re", "éé", "!!", "EOT", "123", "456", "78", "…", "½", "<'", "VE", "\tåa", "Ⅳ", "…é", "Z'S", "\n", "
A", "-​\r\n\r\n"]} +{"text": "­ ½d\u000bedZ're-m字#$%ḍ̇'re
ḍ̇́漢(>(İ👍🏽<|endoftext|>𐞁<|fim_prefix|>㋿'T字'Re\r\nİ0字", "tokens": 56, "pieces": ["­", " ", "½", "d", "\u000bed", "Z're", "-m字", "#$%", "ḍ̇'re", "
ḍ̇́漢", "(>(", "İ", "👍🏽<|", "endoftext", "|>", "𐞁", "<|", "fim", "_prefix", "|>㋿'", "T字'Re", "\r\n", "İ", "0", "字"]} +{"text": "0‍é٣٤٥٦<|endoftext|><'Re३ -\r\n٣٤٥٦́\r\te ,''T9㍿👍🏽字​<|fim_prefix|>", "tokens": 49, "pieces": ["0", "‍é", "٣٤٥", "٦", "<|", "endoftext", "|><", "META", "_START", "><'", "Re", "३", " ", " -\r\n", "٣٤٥", "٦", "́", "\r", "\te", " ", " ,''", "T", "9", "㍿👍🏽", "字", "​<|", "fim", "_prefix", "|>"]} +{"text": "<|endoftext|><|fim_prefix|>'S🙂a!३\"'T字\ré0,123456780 \t­''VE㋿<|endoftext|> é", "tokens": 45, "pieces": ["<|", "endoftext", "|><|", "fim", "_prefix", "|>'", "S", "🙂a", "!", "३", "\"'", "T字", "\r", "é", "0", ",", "123", "456", "780", " ", "\t", "­''", "VE", "㋿<|", "endoftext", "|>", " é"]} +{"text": "!'M\u000b<|endoftext|>å\u000b", "tokens": 13, "pieces": ["!'", "M", "\u000b", "<|", "endoftext", "|>", "å", "\u000b"]} +{"text": ",'ll\nd漢ſ", "tokens": 6, "pieces": [",'", "ll", "\n", "d漢ſ"]} +{"text": "eEOT'Tå\r'sḍ̇'ß\"EOT\n㋿", "tokens": 22, "pieces": ["e", "EOT'T", "å", "\r", "'sḍ̇", "'ß", "\"<", "META", "_START", ">EOT", "\n", "㋿"]} +{"text": "'sfißḍ̇Aé\n\u000bAé'Mm<|fim_prefix|>å<|fim_prefix|>٣٤٥٦", "tokens": 33, "pieces": ["'sfißḍ̇", "Aé", "\n", "\u000bAé'M", "m", "<|", "fim", "_prefix", "|>", "å", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦"]} +{"text": "'M漢#$%ß..'\r\n0\u000b(́\"\r\nééfiⅣDž \n\u000bAḍ̇As", "tokens": 30, "pieces": ["'M漢", "#$%", "ß", "..'\r\n", "0", "\u000b", "(́", "\"\r\n", "ééfi", "Ⅳ", "Dž", "", " \n", "\u000bAḍ̇", "As"]} +{"text": "é \nſ", "tokens": 4, "pieces": ["é", " \n", "ſ"]} +{"text": "t Z㋿,12345678. EOT'VE
३­ Ⅳ㋿<|endoftext|>ßEOT!EOT'S", "tokens": 40, "pieces": ["t", " Z", "㋿,", "123", "456", "78", ".", " EOT'VE", "
", "३", "­", " ", " ", "Ⅳ", "㋿<|", "endoftext", "|>", "ß", "EOT", "!EOT'S"]} +{"text": "٣٤٥٦<|endoftext|>😀🏽३😀🏽'DA'reß 'Re. \n 9éå're'ſ<'reع!! <‍me👍🏽'llm", "tokens": 52, "pieces": ["٣٤٥", "٦", "<|", "endoftext", "|>😀🏽", "३", "😀🏽'", "DA're", "ß", " ", "'Re", ".<", "EOT", ">", " \n", " ", "9", "éå're", "'ſ", "<'", "reع", "!!", " ", "<‍", "me", "👍🏽'", "llm"]} +{"text": "'M Zſtaé9́ Ⅳ'S٣٤٥٦\r\n\r\n'ſ're'Re…", "tokens": 25, "pieces": ["'M", " Zſtaé", "9", "́", " ", "Ⅳ", "'S", "٣٤٥", "٦", "\r\n\r\n", "'ſ're", "'Re", "…"]} +{"text": "!!漢EOT字<漢", "tokens": 9, "pieces": ["!!", "漢", "EOT字", "<漢"]} +{"text": "\n \n🙂​>‍'s字㍿s
're½EOT'Re'reé. \n", "tokens": 23, "pieces": ["\n \n", "🙂​>‍'", "s字", "㍿s", "
", "'re", "½", "EOT'Re", "'reé", ".", " \n"]} +{"text": "\r\n\r\n'D́é-'lld!𐞁\u000bm㋿ع'D", "tokens": 22, "pieces": ["\r\n\r\n", "'D́é", "-'", "lld", "!<", "META", "_START", ">𐞁", "\u000bm", "㋿ع'D"]} +{"text": "​…३(​9t ", "tokens": 9, "pieces": ["​", "…", "३", "(​", "9", "t", " "]} +{"text": "🙂\r\nd0'VE🙂'M字
㋿!!'!!'D🙂", "tokens": 23, "pieces": ["🙂\r\n", "d", "0", "'VE", "🙂'", "M字", "
", "㋿!!'<", "EOT", ">!!'", "D", "🙂"]} +{"text": "-.m' \r\nt<|fim_prefix|>👍🏽👍🏽a😀🏽½!!'VEDž\r ", "tokens": 30, "pieces": ["-.", "m", "'", " \r\n", "t", "<|", "fim", "_prefix", "|>👍🏽👍🏽", "a", "😀🏽", "½", "!!'", "VEDž", "\r", " "]} +{"text": " \n<'M#$%\n\r\nİéA🙂 漢å\" ع", "tokens": 21, "pieces": [" \n", "<'", "M", "#$%\n\r\n", "İé", "A", "🙂", " 漢å", "\"", " ع"]} +{"text": "9'Dİ👍🏽㍿🙂m 
e fi<|fim_prefix|>㋿t'M'll \n!Ⅳ'VEİ-'reé½ !fi9\t\n \n<|endoftext|>", "tokens": 52, "pieces": ["9", "'Dİ", "👍🏽㍿🙂", "m", " ", "
e", " ", " fi", "<|", "fim", "_prefix", "|>㋿", "t'M", "'ll", " \n", "!", "Ⅳ", "'VEİ", "-'", "reé", "½", " ", " !", "fi", "9", "\t\n \n", "<|", "endoftext", "|>"]} +{"text": "́…́\"Dž ", "tokens": 11, "pieces": ["́", "…́", "\"Dž", " "]} +{"text": "\r\n\r\nds,㋿'VE", "tokens": 8, "pieces": ["\r\n\r\n", "ds", ",㋿'", "VE"]} +{"text": "-a<|endoftext|>'s12345678!!s . ", "tokens": 21, "pieces": ["-a", "<|", "endoftext", "|>'", "s", "123", "456", "78", "!!", "s", "", " ", ".", " "]} +{"text": "12345678‍'s(t's 𐞁éعꟲ٣٤٥٦>0\r\nİ12345678!!३\",,fie'rea12345678Džꟲé' .EOT", "tokens": 60, "pieces": ["123", "456", "78", "‍'", "s", "(t's", " 𐞁éع", "ꟲ", "٣٤٥", "٦", ">", "0", "\r\n", "İ", "123", "456", "78", "!!", "३", "\",<", "META", "_START", ">,", "fie", "'", "rea", "123", "456", "78", "Džꟲé", "'", " .", "EOT"]} +{"text": "a'llEOT-ع𐞁're٣٤٥٦㋿dé<|fim_prefix|>at,é字#$%-0👍🏽😀🏽.½ḍ̇<|fim_prefix|>ß<𐞁
㍿ (", "tokens": 61, "pieces": ["a'll", "EOT", "-ع𐞁're", "٣٤٥", "٦", "㋿dé", "<|", "fim", "_prefix", "|>", "at", ",é字", "#$%-", "0", "👍🏽😀🏽.", "½", "ḍ̇", "<|", "fim", "_prefix", "|>", "ß", "<𐞁", "
", "㍿", " ", "("]} +{"text": "a-ßt𐞁e<|endoftext|>éfi字're
\tß 'VE\r\n\r\n \n㋿'TfiEOTdm𐞁\rß́'M", "tokens": 50, "pieces": ["a", "-ßt𐞁e", "<|", "endoftext", "|>", "éfi字", "'", "re", "
", "\tß", " ", "'VE", "\r\n\r\n \n", "㋿'", "Tfi", "EOTdm𐞁", "\r", "ß́'M"]} +{"text": "ꟲ!!٣٤٥٦\r\ne's't\n字 EOTmⅣ ß漢0dZ.ßsꟲ \" 'refié", "tokens": 42, "pieces": ["ꟲ", "!!", "٣٤٥", "٦", "\r\n", "e's", "'t", "\n", "字", " EOTm", "Ⅳ", " ß漢", "0", "d", "Z", ".ßsꟲ", " ", "\"", " ", " '", "re", "fié"]} +{"text": "-😀🏽12345678å.0fiEOT'Tع!㋿ Z>字 'Tꟲ#$%\r\nt'VEع
\u000bßḍ̇'S12345678'S \n\t\u000b٣٤٥٦", "tokens": 58, "pieces": ["-😀🏽", "123", "456", "78", "å", ".", "0", "fi", "EOT'T", "ع", "!㋿", " Z", ">字", " ", " '", "Tꟲ", "#$%\r\n", "t'VE", "ع", "
", "\u000bßḍ̇", "'", "S", "123", "456", "78", "'S", " \n", "\t", "\u000b", "٣٤٥", "٦"]} +{"text": "\nİ\"\n'T.e9'reé'ſZé👍🏽,12345678  👍🏽'll漢 Ⅳ'VEta's'VE \n's'Tعꟲ", "tokens": 44, "pieces": ["\n", "İ", "\"\n", "'T", ".e", "9", "'reé'ſ", "Zé", "👍🏽,", "123", "456", "78", " ", " ", "👍🏽'", "ll漢", " ", "Ⅳ", "'VEta's", "'VE", " \n", "'s'T", "عꟲ"]} +{"text": "'VE'sDž½fi㍿漢ſ<\rDž\r\n<\u000b\"\u000b漢Aḍ̇­ad🙂", "tokens": 29, "pieces": ["'VE's", "Dž", "½", "fi", "㍿漢ſ", "<\r", "Dž", "\r\n", "<", "\u000b", "\"", "\u000b漢Aḍ̇", "­ad", "🙂"]} +{"text": "\u000b!\ra漢 EOT.‍!\u000b​​a👍🏽 \r\n'́", "tokens": 23, "pieces": ["\u000b", "!\r", "a漢", " EOT", ".‍!", "\u000b", "​​", "a", "👍🏽", " \r\n", "'́"]} +{"text": " Z!!½-,#$%fi'MDž 漢ꟲ>'reꟲ0t'reéḍ̇12345678(12345678ſ३#$%'ſdAet12345678t३", "tokens": 55, "pieces": [" Z", "!!", "½", "-,#$%", "fi'M", "Dž", " ", " 漢", "ꟲ", ">'", "reꟲ", "0", "t're", "éḍ̇", "123", "456", "78", "(", "123", "456", "78", "ſ", "३", "#$%'", "ſd", "A", "et", "123", "456", "78", "t", "३"]} +{"text": "㋿\t\r\n12345678é\u000b!!'ſ'MA,\r\n\r\n\"'ll'EOT٣٤٥٦'MZ", "tokens": 28, "pieces": ["㋿", "\t\r\n", "123", "456", "78", "é", "\u000b", "!!'", "ſ'M", "A", ",\r\n\r\n", "\"<", "META", "_START", ">'", "ll", "'EOT", "٣٤٥", "٦", "'MZ"]} +{"text": "'Re
‍\t\r\n\r\nİ ㍿d \nmİ#$%ḍ̇字#$%eZ(12345678\r\n\r\n t \n'ReA‍\r\n's#$%ⅣA<|endoftext|>'VE'VE", "tokens": 55, "pieces": ["'Re", "
", "‍", "\t\r\n\r\n", "İ", " ", " ㍿", "d", " \n", "m", "İ", "#$%", "ḍ̇字", "#$%", "e", "Z", "(", "123", "456", "78", "\r\n\r\n", " ", " t", " \n", "'Re", "A", "‍\r\n", "'s", "#$%", "Ⅳ", "A", "<|", "endoftext", "|>'", "VE'VE"]} +{"text": "'lld㍿
'red٣٤٥٦👍🏽aDž'reꟲ<<|fim_prefix|>\r\n'Re'VEa\"<|fim_prefix|>é''🙂EOT'0're'ſ'T", "tokens": 52, "pieces": ["'lld", "㍿", "
", "'red", "٣٤٥", "٦", "👍🏽", "a", "Dž're", "ꟲ", "<<", "EOT", "><|", "fim", "_prefix", "|>\r\n", "'Re'VE", "a", "\"<|", "fim", "_prefix", "|>", "é", "''🙂", "EOT", "'", "0", "'re'ſ", "'T"]} +{"text": " EOT 漢\"<\"٣٤٥٦ꟲ12345678!\u000b👍🏽'M!A'VE", "tokens": 28, "pieces": [" ", " EOT", " 漢", "\"<\"", "٣٤٥", "٦", "ꟲ", "123", "456", "78", "!", "\u000b", "👍🏽'", "M", "!A'VE"]} +{"text": "<|fim_prefix|> 'reſZ‍🙂\r\n\r\nİfi\r\n'S👍🏽ſİ…'S ", "tokens": 29, "pieces": ["<|", "fim", "_prefix", "|>", " ", "'reſ", "Z", "‍🙂\r\n\r\n", "İfi", "\r\n", "'S", "👍🏽", "ſ", "İ", "…", "'", "S", " "]} +{"text": "de३ ", "tokens": 7, "pieces": ["de", "", "३", " "]} +{"text": "\n𐞁㍿Dž ㋿a'll!!ſ字…Ⅳ'sé12345678", "tokens": 33, "pieces": ["\n", "𐞁", "㍿Dž", " ", " ㋿", "a'll", "!!", "ſ字", "…", "Ⅳ", "'sé", "", "123", "456", "78"]} +{"text": "<|fim_prefix|>e😀🏽tZ​\t#$%عå३ ㍿ß😀🏽İ9", "tokens": 34, "pieces": ["<|", "fim", "_prefix", "|>", "e", "😀🏽", "t", "Z", "​", "\t", "#$%", "عå", "३", " ", "㍿<", "META", "_START", ">ß", "😀🏽", "İ", "9"]} +{"text": "#$%!!½ ٣٤٥٦'D'VEa
\t ٣٤٥٦('VE! \"​.", "tokens": 27, "pieces": ["#$%!!", "½", " ", " ", "٣٤٥", "٦", "'D'VE", "a", "
\t", " ", "٣٤٥", "٦", "('", "VE", "!", " ", "\"​."]} +{"text": "9<'re'M'T😀🏽('ſ'M.'VE\r\n\r\n👍🏽éa\u000bⅣعé<|fim_prefix|>>'D\tZDž  ", "tokens": 42, "pieces": ["9", "<'", "re'M", "'T", "😀🏽('", "ſ'M", ".'", "VE", "\r\n\r\n", "👍🏽", "éa", "", "\u000b", "Ⅳ", "عé", "<|", "fim", "_prefix", "|>>'", "D", "\tZDž", "  "]} +{"text": "'‍ſ#$%…३\r\n\r\ne𐞁'T!٣٤٥٦-sé é​'s𐞁d\r\n\r\nſå<|endoftext|>字 \n👍🏽DžⅣ½\",Z👍🏽", "tokens": 63, "pieces": ["'‍", "ſ", "#$%", "…", "३", "\r\n\r\n", "e𐞁'T", "!", "٣٤٥", "٦", "-sé", " é", "​'", "s𐞁d", "\r\n\r\n", "ſå", "<|", "endoftext", "|>", "字", " \n", "👍🏽", "Dž", "Ⅳ½", "\",", "Z", "👍🏽"]} +{"text": "9Z…🙂३ß \n(<|endoftext|>\t9'D٣٤٥٦EOT ꟲ.é
å'll漢İA", "tokens": 40, "pieces": ["9", "Z", "…", "🙂", "३", "ß", " \n", "(<|", "endoftext", "|>", "\t", "", "9", "'D", "٣٤٥", "٦", "EOT", " ", " ꟲ", ".é", "
å'll", "漢", "İA"]} +{"text": "
-\u000b'Tß'M 𐞁😀🏽\r'll३'Ⅳ…9Zm9Ae'Tİ漢<éßå", "tokens": 43, "pieces": ["
", "-<", "EOT", ">", "\u000b", "'Tß'M", " 𐞁", "😀🏽\r", "'ll", "३", "'", "Ⅳ", "…", "9", "Zm", "9", "Ae'T", "İ漢", "<éßå"]} +{"text": "\r\n'é\r\nع㋿'reEOT a'll \r0'D\u000bİ#$%🙂\r\n>㍿>å\r's😀🏽́ 'S A ", "tokens": 47, "pieces": ["\r\n", "'é", "\r\n", "ع", "㋿'", "re", "EOT", " a'll", " \r", "0", "'D", "\u000bİ", "#$%🙂\r\n", ">㍿>", "å", "\r", "'s", "😀🏽<", "META", "_START", ">́", " ", "'S", " A", " "]} +{"text": ".A<|fim_prefix|>'ſfiDž­'eée٣٤٥٦'VE<|endoftext|>👍🏽mİ!!
", "tokens": 39, "pieces": [".A", "<|", "fim", "_prefix", "|>'", "ſfi", "Dž", "­'", "e", "ée", "٣٤٥", "٦", "'VE", "<|", "endoftext", "|>👍🏽", "m", "İ", "!!", "
"]} +{"text": "!漢‍<|fim_prefix|>A'D'ſ 𐞁- .  a", "tokens": 24, "pieces": ["!漢", "‍<|", "fim", "_prefix", "|>", "A'D", "'ſ", " ", " 𐞁", "-", " .", " ", " a"]} +{"text": "İé<|fim_prefix|>'re'T½ꟲ!!t Z ​ꟲEOT\r\n(fi's#$%sDž́ ㋿…'T!m\u000b", "tokens": 48, "pieces": ["İé", "<|", "fim", "_prefix", "|>'", "re'T", "½", "ꟲ", "!!", "t", " Z", " ", "​ꟲ", "EOT", "\r\n", "(fi's", "#$%", "s", "Dž́", " ", "㋿", "…", "'T", "!m", "\u000b"]} +{"text": "'re漢<|endoftext|>½ (😀🏽a#$%0<|fim_prefix|>å𐞁<|endoftext|>'Tſ\r\n\r\n‍Ⅳ\né<٣٤٥٦('ſ'T 9​漢३é,\u000b12345678İ\u000b", "tokens": 71, "pieces": ["'re漢", "<|", "endoftext", "|>", "½", " ", "(😀🏽", "a", "#$%", "0", "<|", "fim", "_prefix", "|>", "å𐞁", "<|", "endoftext", "|>'", "Tſ", "\r\n\r\n", "‍", "Ⅳ", "\n", "é", "<", "٣٤٥", "٦", "('", "ſ'T", " ", " ", "9", "​漢", "३", "é", ",", "\u000b", "123", "456", "78", "İ", "", "\u000b"]} +{"text": "#$%
(㍿Z!!s👍🏽\"EOT<\t𐞁,'T!'DⅣ'Re👍🏽३ḍ̇", "tokens": 36, "pieces": ["#$%", "
", "(㍿", "Z", "!!", "s", "👍🏽\"", "EOT", "<", "\t𐞁", ",'", "T", "!'", "D", "Ⅳ", "'Re", "👍🏽", "३", "ḍ̇"]} +{"text": "mm 're𐞁då'ReⅣ🙂  >\r\n'ſ", "tokens": 19, "pieces": ["mm", " ", " '", "re𐞁då'Re", "Ⅳ", "🙂", " ", " ", ">\r\n", "'ſ"]} +{"text": "٣٤٥٦'ll\r\nZ字9½́#$%!t", "tokens": 14, "pieces": ["٣٤٥", "٦", "'ll", "\r\n", "Z字", "9½", "́", "#$%!", "t"]} +{"text": "fi'Dt.'llḍ̇'så.​\u000b'D㍿ḍ̇'ll'T >Ⅳ \n#$%'D!'ll\r", "tokens": 36, "pieces": ["fi'D", "t", ".<", "META", "_START", ">'", "llḍ̇'s", "å", ".​", "\u000b", "'D", "㍿ḍ̇'ll", "'T", " ", ">", "Ⅳ", " \n", "#$%'", "D", "!'", "ll", "\r"]} +{"text": "'Reİfi\t'VE12345678字's'Mꟲſé(>'S ٣٤٥٦,😀🏽ſa\r\nmDža…ⅣA'\r\n", "tokens": 43, "pieces": ["'Re", "İfi", "\t", "'VE", "123", "456", "78", "字's", "'Mꟲſé", "(>'", "S", " ", "٣٤٥", "٦", ",😀🏽", "ſa", "\r\n", "m", "Dža", "…", "Ⅳ", "A", "'\r\n"]} +{"text": "!!EOT½​\r\n\r\n\t½'ſ'll's½
\"d½'\r\n\r\n٣٤٥٦t½ ḍ̇(<|fim_prefix|>.ꟲm'D​sm!😀🏽9 ", "tokens": 46, "pieces": ["!!", "EOT", "½", "​\r\n\r\n", "\t", "½", "'ſ'll", "'s", "½", "
", "\"d", "½", "'\r\n\r\n", "٣٤٥", "٦", "t", "½", " ḍ̇", "(<|", "fim", "_prefix", "|>.", "ꟲm'D", "​sm", "!😀🏽", "9", " "]} +{"text": "🙂fiع#$%\ŕ", "tokens": 7, "pieces": ["🙂fiع", "#$%\r", "́"]} +{"text": "EOTİ\n#$%٣٤٥٦fi9́'S'VEZa12345678字s0>'re'Re<<|fim_prefix|>́", "tokens": 39, "pieces": ["EOTİ", "\n", "#$%<", "EOT", ">", "٣٤٥", "٦", "fi", "9", "́'S", "'VEZa", "123", "456", "78", "字s", "0", ">'", "re'Re", "<<|", "fim", "_prefix", "|>́"]} +{"text": "…漢'ſeZꟲ'VE­'Sa 0\u000b\r", "tokens": 19, "pieces": ["…漢'ſ", "e", "Zꟲ'VE", "­'", "Sa", " ", "0", "\u000b\r"]} +{"text": "­ \nm!", "tokens": 4, "pieces": ["­", " \n", "m", "!"]} +{"text": "'DAm½é,'ſ,-e'D😀🏽'S,'T0é<
 ㍿e㍿'D'Re٣٤٥٦#$%\"𐞁'DAa", "tokens": 50, "pieces": ["'DAm", "½", "é", ",'", "ſ", ",-", "e'D", "😀🏽'", "S", ",'", "T", "0", "é", "<", "
", " ", "㍿e", "㍿'", "D'Re", "٣٤٥", "٦", "#$%\"", "𐞁'D", "Aa"]} +{"text": "\r𐞁<|endoftext|>9dd!'T\r\n\r\n'sDž#$%9ß́\u000b<|fim_prefix|>\u000bḍ̇‍ ", "tokens": 38, "pieces": ["\r", "𐞁", "<|", "endoftext", "|>", "9", "dd", "!'", "T", "\r\n\r\n", "'s", "Dž", "#$%", "9", "ß́", "\u000b", "<|", "fim", "_prefix", "|>", "\u000bḍ̇", "‍", " "]} +{"text": "-𐞁!!ḍ̇‍. 0ꟲ '字a‍fiEOT\"ſ \ns'DZ're ٣٤٥٦'s", "tokens": 36, "pieces": ["-𐞁", "!!", "ḍ̇", "‍.", " ", "0", "ꟲ", " ", "'字a", "‍fi", "EOT", "\"ſ", " \n", "s'D", "Z're", " ", "٣٤٥", "٦", "'s"]} +{"text": "'VE!!!😀🏽\tEOT'M-t>
\re0\"t\r\n\r\n'SEOT́
Dž字‍(", "tokens": 31, "pieces": ["'VE", "!!!😀🏽", "\tEOT'M", "-t", ">", "
\r", "e", "0", "\"t", "\r\n\r\n", "'SEOT́", "
Dž字", "‍<", "EOT", ">("]} +{"text": "EOT'ſ​'M", "tokens": 7, "pieces": ["EOT'ſ", "​'", "M"]} +{"text": "𐞁👍🏽👍🏽Aß0½'MDže!!
EOT'llع<|fim_prefix|>é", "tokens": 32, "pieces": ["𐞁", "👍🏽👍🏽", "Aß", "0½", "'MDže", "!!", "
EOT'll", "ع", "<|", "fim", "_prefix", "|>", "é"]} +{"text": "#$%ع'TEOTs'S#$%ßd­é​ 'Re-#$%́", "tokens": 30, "pieces": ["#$%", "ع'T", "EOTs'S", "#$%", "ßd", "­é", "​", " ", " <", "META", "_START", ">'", "Re", "-#$%́<", "EOT", ">"]} +{"text": "ꟲ字​\r'll!!", "tokens": 8, "pieces": ["ꟲ字", "​\r", "'ll", "!!"]} +{"text": "……a㍿m
𐞁𐞁,0fiſ‍d .>👍🏽 -é,\u000b́Dž字👍🏽㍿\n\u000b", "tokens": 50, "pieces": ["…", "…a", "㍿m", "
𐞁𐞁", ",", "0", "fiſ", "‍d", " ", " .>👍🏽", " ", "-", "é", ",", "\u000b́Dž字", "👍🏽㍿\n", "\u000b"]} +{"text": "d½Džtaع​s#$%t'reع'VE👍🏽're", "tokens": 23, "pieces": ["d", "½", "Džt", "aع", "​s", "#$%", "t're", "ع'VE", "👍🏽'", "re"]} +{"text": "ꟲaſ \n'D-", "tokens": 8, "pieces": ["ꟲaſ", " \n", "'D", "-"]} +{"text": "😀🏽(३'ſZ\r\nſ  t \n<|endoftext|> Ⅳꟲḍ̇İ12345678e<|fim_prefix|>٣٤٥٦\n'S t's İd👍🏽d", "tokens": 60, "pieces": ["😀🏽(", "३", "'ſ", "Z", "\r\n", "ſ", " ", " t", " \n", "<|", "endoftext", "|>", " ", " ", "Ⅳ", "ꟲḍ̇", "İ", "123", "456", "78", "e", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "\n", "'S", " t's", " İd", "👍🏽", "d"]} +{"text": ",…'llḍ̇s\"<\"३a(#$%\"s👍🏽字é'D!!
", "tokens": 26, "pieces": [",", "…", "'llḍ̇s", "\"<<", "META", "_START", ">\"", "३", "a", "(#$%\"", "s", "👍🏽", "字é'D", "!!", "
"]} +{"text": "👍🏽 ㍿½'VE#$%­'Re३ ſ'D­t'M३👍🏽A<,'lla<|fim_prefix|>ADž🙂sd", "tokens": 40, "pieces": ["👍🏽", " ", "㍿", "½", "'VE", "#$%­'", "Re", "३", " ſ'D", "­t'M", "३", "👍🏽", "A", "<,'", "lla", "<|", "fim", "_prefix", "|>", "ADž", "🙂sd"]} +{"text": "A𐞁\r\n\r\n#$%0㋿EOT-عſ'MmⅣd#$%'D12345678d… ", "tokens": 32, "pieces": ["A𐞁", "\r\n\r\n", "#$%", "0", "㋿EOT", "-عſ'M", "m", "Ⅳ", "d", "#$%'", "D", "123", "456", "78", "d", "… "]} +{"text": "<|endoftext|>…'㋿<|endoftext|>t> fim's㋿(…e👍🏽<|fim_prefix|>t ,ꟲ", "tokens": 49, "pieces": ["<|", "endoftext", "|>", "…", "'㋿<|", "endoftext", "|>", "t", ">", " ", " fim's", "㋿(", "…e", "👍🏽<|", "fim", "_prefix", "|>", "t", " ,", "ꟲ"]} +{"text": "
字́ſ'M'D<|endoftext|> (́
­,9#$%\"'s'Tſḍ̇>9‍d \n\"३fi!!å'S!td<|fim_prefix|>", "tokens": 48, "pieces": ["
字́ſ'M", "'D", "<|", "endoftext", "|>", " (́", "
", "­,", "9", "#$%\"'", "s'T", "ſḍ̇", ">", "9", "‍d", " \n", "\"", "३", "fi", "!!", "å'S", "!td", "<|", "fim", "_prefix", "|>"]} +{"text": "<|endoftext|>12345678sfi(a d", "tokens": 15, "pieces": ["<|", "endoftext", "|>", "123", "456", "78", "sfi", "(a", " d"]} +{"text": "Džß'll😀🏽EOT٣٤٥٦ſZ0<|fim_prefix|>-", "tokens": 26, "pieces": ["Džß'll", "😀🏽", "EOT", "٣٤٥", "٦", "ſ", "Z", "0", "<|", "fim", "_prefix", "|>-"]} +{"text": "字'VE'T'M'll(\t<'ſſ ‍!!!!d#$%!!", "tokens": 21, "pieces": ["字'VE", "'T'M", "'ll", "(", "\t", "<'", "ſſ", " ", " ‍!!!!", "d", "#$%!!"]} +{"text": "<|fim_prefix|>!!…́'㍿a­\"㋿\r\n -<|endoftext|>>eḍ̇🙂字'Re->
.s", "tokens": 39, "pieces": ["<|", "fim", "_prefix", "|>!!", "…́", "'㍿", "a", "­\"㋿\r\n", " -<|", "endoftext", "|>>", "eḍ̇", "🙂字'Re", "->", "
", ".s"]} +{"text": "(9漢'S're\r\ń's'ſ\r\n\r\n
('Dd<​12345678'D٣٤٥٦½#$%t­
ſ'S\r\nꟲ", "tokens": 40, "pieces": ["(", "9", "漢'S", "'re", "\r\n", "́'s", "'ſ", "\r\n\r\n", "
", "('", "Dd", "<​", "123", "456", "78", "'D", "٣٤٥", "٦½", "#$%", "t", "­", "
ſ'S", "\r\n", "ꟲ"]} +{"text": "🙂m(('s.½  ſ.'Tfiåa‍-9\t\"#$%#$%​ꟲ-'Re \n'Tꟲ\n㍿字-‍#$%", "tokens": 44, "pieces": ["🙂m", "(('", "s", ".", "½", " ", " ſ", ".'", "Tfiåa", "‍-", "9", "\t", "\"#$%#$%​", "ꟲ", "-'", "Re", " \n", "'Tꟲ", "\n", "㍿字", "-‍#$%"]} +{"text": "(\r<|fim_prefix|>½'re'll​aß", "tokens": 14, "pieces": ["(\r", "<|", "fim", "_prefix", "|>", "½", "'re'll", "​aß"]} +{"text": "\r\n٣٤٥٦d,\r\n\r\n", "tokens": 7, "pieces": ["\r\n", "٣٤٥", "٦", "d", ",\r\n\r\n"]} +{"text": "é\t­ \n0(12345678👍🏽٣٤٥٦ ḍ̇", "tokens": 20, "pieces": ["é", "\t", "­", " \n", "0", "(", "123", "456", "78", "👍🏽", "٣٤٥", "٦", " ḍ̇"]} +{"text": "㍿ßém'Tå't🙂<|endoftext|>! 'M३.09Dž'ReZ!!👍🏽\u000bfid(e​漢 
'ſ12345678éA'T", "tokens": 47, "pieces": ["㍿ßém'T", "å't", "🙂<|", "endoftext", "|>!", " ", "'M", "३", ".", "09", "Dž'Re", "Z", "!!👍🏽", "\u000bfid", "(e", "​漢", " ", "
", "'ſ", "123", "456", "78", "é", "A'T"]} +{"text": "(½İ<|fim_prefix|>!!a' ", "tokens": 13, "pieces": ["(", "½", "İ", "<|", "fim", "_prefix", "|>!!", "a", "'", " "]} +{"text": "\r\n½ſ㋿ ㋿👍🏽dſ½㋿", "tokens": 19, "pieces": ["\r\n", "½", "ſ", "㋿", " ", "㋿👍🏽", "dſ", "½", "㋿"]} +{"text": "å½EOT's漢EOT‍ Dž' \n9é \n😀🏽é\rع>ḍ̇'VEß'Dmsİḍ̇", "tokens": 42, "pieces": ["å", "½", "EOT's", "漢", "EOT", "‍", " Dž", "'", " \n", "9", "é", " \n", "😀🏽", "é", "\r", "ع", ">ḍ̇'VE", "ß", "'", "Dms", "İḍ̇"]} +{"text": "Ⅳ12345678ḍ̇字,㍿0عs\u000b0fi'Ré.eßADž​漢ſ", "tokens": 29, "pieces": ["Ⅳ12", "345", "678", "ḍ̇字", ",㍿", "0", "عs", "\u000b", "0", "fi'Re", "́", ".eß", "ADž", "​漢ſ"]} +{"text": "é'Maa'lls😀🏽
,Dž \nع,", "tokens": 16, "pieces": ["é'M", "aa'll", "s", "😀🏽", "
", ",Dž", " \n", "ع", ","]} +{"text": "'sAé… \ne'T½12345678s", "tokens": 14, "pieces": ["'s", "Aé", "… \n", "e'T", "½12", "345", "678", "s"]} +{"text": "12345678<|fim_prefix|>😀🏽t字é'ع㋿👍🏽é\t fid'VE\r\" s\r\nß", "tokens": 38, "pieces": ["123", "456", "78", "<|", "fim", "_prefix", "|>😀🏽", "t字é", "'ع", "㋿👍🏽", "é", "\t", " fid'VE", "\r", "\"", " ", " s", "\r\n", "ß"]} +{"text": "ꟲDž,'VE", "tokens": 7, "pieces": ["ꟲ", "Dž", ",'", "VE"]} +{"text": "Dž<|endoftext|>-'𐞁'll
𐞁å\u000b<😀🏽#$%\t A\u000bé'S'Ree12345678𐞁‍éA #$%ß'VE", "tokens": 58, "pieces": ["Dž", "<|", "endoftext", "|>-'", "𐞁'll", "
𐞁å", "\u000b", "<😀🏽#$%", "\t", " A", "\u000bé'S", "'Ree", "123", "456", "78", "𐞁", "‍é", "A", " <", "EOT", ">#$%", "ß'VE"]} +{"text": "\té", "tokens": 2, "pieces": ["\té"]} +{"text": " ḍ̇…㋿ꟲ३,𐞁İ\n-'M字mfi𐞁😀🏽ع👍🏽́!!!!d\"\u000b'VE ३ ( ", "tokens": 48, "pieces": [" ḍ̇", "…", "㋿ꟲ", "३", ",𐞁", "İ", "\n", "-'", "M字mfi𐞁", "😀🏽", "ع", "👍🏽́!!!!", "d", "\"", "\u000b", "'VE", " ", "३", " ", " (", " "]} +{"text": "\r\n0'M\r😀🏽e½
\r\n\r\n\n(.#$% \nß'llعⅣéſⅣs ㍿\r\n\r\n#$%\"EOT'S 👍🏽🙂'VE\n'ſ", "tokens": 54, "pieces": ["\r\n", "0", "'", "M", "\r", "😀🏽", "e", "½", "
\r\n\r\n\n", "(.#$%", " \n", "ß'll", "ع", "Ⅳ", "éſ", "", "Ⅳ", "s", " ", "㍿\r\n\r\n", "#$%\"", "EOT'S", " ", "👍🏽🙂'", "VE", "\n", "'ſ"]} +{"text": "𐞁㋿!\n'DA\n\tßé‍Ⅳ>İ.<|fim_prefix|> ", "tokens": 30, "pieces": ["𐞁", "㋿!\n", "'DA", "\n", "\tßé", "‍", "Ⅳ", ">İ", ".<|", "fim", "_prefix", "|>", " "]} +{"text": "́㋿ḍ̇!字'ſ EOT​fiⅣ\t漢𐞁'sfi'T'M ‍ \n㍿é'Re-0", "tokens": 42, "pieces": ["́", "㋿ḍ̇", "!字'ſ", " EOT", "​fi", "Ⅳ", "\t漢𐞁's", "fi'T", "'M", " ", " ‍", " \n", "㍿é'Re", "-<", "EOT", ">", "0"]} +{"text": "'Re‍sEOT#$%字'ſ,😀🏽d漢 \nA \n!!'M're…!!'T ㍿ İ(a 𐞁s\rß٣٤٥٦\t'VE'Mİa𐞁𐞁", "tokens": 61, "pieces": ["'Re", "‍s", "EOT", "#$%", "字'ſ", ",😀🏽", "d漢", " \n", "A", " \n", "!!'", "M're", "…", "!!'", "T", " ㍿", " İ", "(a", " 𐞁s", "\r", "ß", "٣٤٥", "٦", "\t", "'VE'M", "İa𐞁𐞁"]} +{"text": "𐞁#$%dDž漢", "tokens": 13, "pieces": ["𐞁", "#$%<", "EOT", ">d", "Dž漢"]} +{"text": "<|fim_prefix|>ḍ̇'Séſ\n're\r\n\r\n", "tokens": 16, "pieces": ["<|", "fim", "_prefix", "|>", "ḍ̇'S", "éſ", "\n", "'re", "\r\n\r\n"]} +{"text": "ſ٣٤٥٦A'T'll'ſ٣٤٥٦​ İ漢t", "tokens": 23, "pieces": ["ſ", "", "٣٤٥", "٦", "A'T", "'ll'ſ", "٣٤٥", "٦", "​", " İ漢t"]} +{"text": "­\r'M'VE\u000b0𐞁'D\r\n\r\né😀🏽'…\u000b<Dž!!…
\" d👍🏽ßm'ſ9½-", "tokens": 50, "pieces": ["­\r", "'M'VE", "\u000b", "0", "𐞁'D", "\r\n\r\n", "é", "😀🏽'", "…", "\u000b", "<Dž", "!!", "…", "
", "\"<", "é", "<|", "fim", "_prefix", "|>", " d", "👍🏽", "ßm'ſ", "9½", "-"]} +{"text": "9", "tokens": 1, "pieces": ["9"]} +{"text": "(\"字Džé(ꟲ-", "tokens": 14, "pieces": ["(\"", "字Džé", "(<", "EOT", ">ꟲ", "-"]} +{"text": "\r\n­m½ \r\n \n#$%ꟲ><|endoftext|> 0d.'s漢😀🏽é", "tokens": 29, "pieces": ["\r\n", "­m", "½", " \r\n \n", "#$%", "ꟲ", "><|", "endoftext", "|>", " ", "0", "d", ".'", "s漢", "😀🏽", "é"]} +{"text": "ſEOT\r\n9\r \na", "tokens": 8, "pieces": ["ſ", "EOT", "\r\n", "9", "\r \n", "a"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "👍🏽\n'D", "tokens": 5, "pieces": ["👍🏽\n", "'D"]} +{"text": "'S'S\n
-Z
३.½'ſ'VE½ İⅣ\n\t𐞁\">\n A'VEs㋿\r!!'S🙂!'DⅣ", "tokens": 45, "pieces": ["'S'S", "\n", "
", "-Z", "
", "", "३", ".", "½", "'ſ'VE", "½", " İ", "Ⅳ", "\n", "\t𐞁", "\">\n", " ", " A'VE", "s", "㋿\r", "!!'", "S", "🙂!'", "D", "Ⅳ"]} +{"text": "
\r\n\r\n㍿\r\n\r\n\r\n#$%ꟲ", "tokens": 11, "pieces": ["
\r\n\r\n", "㍿\r\n\r\n\r\n", "#$%", "ꟲ"]} +{"text": "\t'Re#$%ꟲ👍🏽 'ꟲ'llé\té,'ſß", "tokens": 27, "pieces": ["\t", "'Re", "#$%", "ꟲ", "👍🏽<", "EOT", ">", " ", "'ꟲ'll", "é", "\té", ",'", "ſß"]} +{"text": "<
\r-('VEDž\r'S\ré'Ree\r\n😀🏽's\né\r\nm\t", "tokens": 28, "pieces": ["<", "
\r", "-('", "VEDž", "\r", "'S", "\r", "é'Re", "e", "\r\n", "😀🏽'", "s", "\n", "é", "\r\n", "m", "\t"]} +{"text": "<|fim_prefix|>𐞁\r\n'll\t's9fi12345678'12345678…-0\n > \n<|fim_prefix|>-é
😀🏽's'S!!🙂\n㋿‍\n0ḍ̇", "tokens": 57, "pieces": ["<|", "fim", "_prefix", "|>", "𐞁", "\r\n", "'ll", "\t", "'s", "9", "fi", "123", "456", "78", "'", "123", "456", "78", "…", "-", "0", "\n", " ", ">", " \n", "<|", "fim", "_prefix", "|>-", "é", "
", "😀🏽'", "s'S", "!!🙂\n", "㋿‍\n", "0", "ḍ̇"]} +{"text": "𐞁🙂𐞁ꟲⅣDž½a0,㍿éḍ̇½'s٣٤٥٦٣٤٥٦", "tokens": 43, "pieces": ["𐞁", "🙂𐞁ꟲ", "Ⅳ", "Dž", "½", "a", "0", ",<", "EOT", "><", "EOT", ">㍿", "éḍ̇", "½", "'s", "٣٤٥", "٦٣٤", "٥٦"]} +{"text": "''re​> \n\r\n\r\n>字s'M­fi㋿'s\r\n\r\n\r\nſ0.t", "tokens": 21, "pieces": ["''", "re", "​>", " \n\r\n\r\n", ">字s'M", "­fi", "㋿'", "s", "\r\n\r\n\r\n", "ſ", "0", ".t"]} +{"text": "9 fiZꟲA👍🏽'M­ع<|endoftext|>🙂#$%­ḍ̇t🙂  字Z ½ꟲ\r'<|endoftext|>́tDž­", "tokens": 52, "pieces": ["9", " fi", "Zꟲ", "A", "👍🏽'", "M", "­ع", "<|", "endoftext", "|>🙂#$%­", "ḍ̇t", "🙂", " ", " 字", "Z", " ", " ", "½", "ꟲ", "\r", "'<|", "endoftext", "|>́", "t", "Dž", "­"]} +{"text": "𐞁'S(㍿!!e'll'Tsfi㍿𐞁EOT're", "tokens": 29, "pieces": ["𐞁'S", "(㍿!!", "e", "'", "ll'T", "sfi", "㍿𐞁", "EOT're"]} +{"text": "aⅣEOT字㍿́\"​ع㍿EOT'D\r<|fim_prefix|>'Sé\r𐞁'Dعİ㍿''VEå‍'Ree0 \u000bDž", "tokens": 57, "pieces": ["a", "Ⅳ", "EOT字", "㍿́", "\"​", "ع", "㍿EOT'D", "\r", "<|", "fim", "_prefix", "|><", "META", "_START", ">'", "Sé", "\r", "𐞁'D", "ع", "İ", "㍿''", "VE", "å", "‍'", "Ree", "0", " ", "\u000bDž"]} +{"text": "å,🙂<|endoftext|>. \n​.\r\n字\r\nİ\r\n\r\n'MA㋿½9", "tokens": 56, "pieces": ["å", ",🙂<|", "endoftext", "|>.", " \n", "​.\r\n", "字", "\r\n", "İ", "\r\n\r\n", "'MA", "㋿", "½9"]} +{"text": "
<|fim_prefix|>Dž", "tokens": 9, "pieces": ["
", "<|", "fim", "_prefix", "|>", "Dž"]} +{"text": "-\n
(\r\n\r\n\r\n fi'seß\r漢㋿s ­", "tokens": 16, "pieces": ["-\n", "
", "(\r\n\r\n\r\n", " fi's", "eß", "\r", "漢", "㋿s", " ­"]} +{"text": "'VE٣٤٥٦👍🏽's
e😀🏽ß🙂>ß <AA'S!!㍿㋿ '‍㋿é>ḍ̇fi\r\n\r\n'ſ", "tokens": 46, "pieces": ["'VE", "٣٤٥", "٦", "👍🏽'", "s", "
e", "😀🏽", "ß", "🙂>", "ß", " <", "AA'S", "!!㍿㋿", " ", " '‍㋿", "é", ">ḍ̇fi", "\r\n\r\n", "'ſ"]} +{"text": "'D'ſ…'S9", "tokens": 7, "pieces": ["'D'ſ", "…", "'S", "9"]} +{"text": "\rm३ 'S🙂漢㍿.㍿,'EOT!漢 EOTå", "tokens": 23, "pieces": ["\r", "m", "३", " ", "'S", "🙂漢", "㍿.㍿,'", "EOT", "!漢", " EOTå"]} +{"text": "­İ㍿!!🙂 ​EOT'S0㋿éßå㋿ \n", "tokens": 25, "pieces": ["­İ", "㍿!!🙂", " ", " ​", "EOT'S", "0", "㋿éßå", "㋿", " \n"]} +{"text": "s", "tokens": 1, "pieces": ["s"]} +{"text": "'llⅣ''<|endoftext|>a½\r\n\r\n m !d!!aA 'Dß㋿Z<|fim_prefix|>>'red", "tokens": 41, "pieces": ["'ll", "Ⅳ", "''<|", "endoftext", "|>", "a", "½", "\r\n\r\n", " ", " m", " ", "!d", "!!", "a", "A", " ", "'D", "ß", "㋿Z", "<|", "fim", "_prefix", "|>>'", "red"]} +{"text": "𐞁𐞁t\r\n-́'Te'Dſ…<|fim_prefix|> 9ḍ̇", "tokens": 29, "pieces": ["𐞁𐞁t", "\r\n", "-́'T", "e'D", "ſ", "…", "<|", "fim", "_prefix", "|>", " ", "9", "ḍ̇"]} +{"text": "'Re", "tokens": 1, "pieces": ["'Re"]} +{"text": " #$%'ſé('ll ‍­'Re'ReDža", "tokens": 16, "pieces": [" ", "#$%'", "ſé", "('", "ll", " ", " ‍­'", "Re'Re", "Dža"]} +{"text": "٣٤٥٦\u000b<|fim_prefix|> fi㋿", "tokens": 16, "pieces": ["٣٤٥", "٦", "\u000b", "<|", "fim", "_prefix", "|>", " fi", "㋿"]} +{"text": "ꟲعae\r\n\r\n'ſ
字(-\t!!!!½åßéA!​'Reſ", "tokens": 24, "pieces": ["ꟲعae", "\r\n\r\n", "'ſ", "
字", "(-", "\t", "!!!!", "½", "åßé", "A", "!​'", "Reſ"]} +{"text": "-३ #$%!!👍🏽", "tokens": 10, "pieces": ["-", "३", " ", " #$%!!👍🏽"]} +{"text": "𐞁'\t- 'M.𐞁​", "tokens": 20, "pieces": ["𐞁", "'", "\t", "-", " ", " '", "M", ".𐞁", "​"]} +{"text": " å <|fim_prefix|>EOT字d‍㍿‍-12345678're\"t0👍🏽<|endoftext|>s>\r\n\r\n9Ⅳad'VE're\"­", "tokens": 48, "pieces": [" ", " å", " ", "<|", "fim", "_prefix", "|>", "EOT字d", "‍㍿‍-", "123", "456", "78", "'re", "\"t", "0", "👍🏽<|", "endoftext", "|>", "s", ">\r\n\r\n", "9Ⅳ", "ad'VE", "'re", "\"­"]} +{"text": "ſ", "tokens": 1, "pieces": ["ſ"]} +{"text": "å٣٤٥٦Z😀🏽>'T\r\n\r\n३ß\"½éZ㋿‍ḍ̇9s'D字\n漢ſdt३e", "tokens": 40, "pieces": ["å", "٣٤٥", "٦", "Z", "😀🏽<", "EOT", ">>'", "T", "\r\n\r\n", "३", "ß", "\"", "½", "é", "Z", "㋿‍", "ḍ̇", "9", "s'D", "字", "\n", "漢ſdt", "३", "e"]} +{"text": "㍿-.'s.e ­'VE ­\u000b\r\nå", "tokens": 16, "pieces": ["㍿-.'", "s", ".e", " ", "­'", "VE", " ­", "\u000b\r\n", "å"]} +{"text": "'T'T", "tokens": 2, "pieces": ["'T'T"]} +{"text": "-.('", "tokens": 2, "pieces": ["-.('"]} +{"text": "Z'ſ'VEé'T…ß'ſ\n're ­.>𐞁́å'VE😀🏽 -d\u000bA'T", "tokens": 36, "pieces": ["Z'ſ", "'VEé'T", "…ß'ſ", "\n", "'re", " ", "­.>", "𐞁́å'VE", "😀🏽", " ", "-d", "\u000bA'T"]} +{"text": "ḍ̇३३㋿e'M'Re\r\n
\"d \u000b'D-漢 漢e\r\n\r\n\u000b㍿a㍿", "tokens": 35, "pieces": ["ḍ̇", "", "३३", "㋿e'M", "'Re", "\r\n", "
", "\"d", " ", "\u000b", "'D", "-漢", " 漢e", "\r\n\r\n", "\u000b", "㍿a", "㍿"]} +{"text": "'ll'VE'VEå👍🏽é9\r\n\r\n<漢", "tokens": 22, "pieces": ["'ll'VE", "'", "VEå", "👍🏽", "é", "", "9", "\r\n\r\n", "<漢"]} +{"text": "عⅣ'M'Md٣٤٥٦0!
're ‍<>måm३\"ع-㍿㋿'re🙂0's'Ś​m<|endoftext|>½ \n­m-🙂", "tokens": 51, "pieces": ["ع", "Ⅳ", "'M'M", "d", "٣٤٥", "٦0", "!", "
", "'re", " ‍<>", "måm", "३", "\"ع", "-㍿㋿'", "re", "🙂", "0", "'s'S", "́", "​m", "<|", "endoftext", "|>", "½", " \n", "­m", "-🙂"]} +{"text": "\u000b字'éfi'VEDžⅣ\tDžda\r\n\r\n 'VE½<|fim_prefix|><|endoftext|>🙂-'ll漢Ⅳ", "tokens": 41, "pieces": ["\u000b字", "'éfi'VE", "Dž", "Ⅳ", "\tDžda", "\r\n\r\n", " ", " <", "EOT", ">'", "VE", "½", "<|", "fim", "_prefix", "|><|", "endoftext", "|>🙂-'", "ll漢", "Ⅳ"]} +{"text": "İeꟲA", "tokens": 6, "pieces": ["İeꟲ", "A"]} +{"text": "9'Reعd(𐞁\n㋿३a\r\n\r\nⅣ(A \n'S字\"a
fi\u000b're-!", "tokens": 38, "pieces": ["9", "'Reعd", "(𐞁", "\n", "㋿", "३", "a", "\r\n\r\n", "Ⅳ", "(A", " \n", "'S字", "\"a", "
fi", "\u000b", "'re", "-!<", "META", "_START", ">"]} +{"text": "Dž٣٤٥٦#$%", "tokens": 8, "pieces": ["Dž", "٣٤٥", "٦", "#$%"]} +{"text": "㍿\r\n\r\nd e>…Aſé<|endoftext|>\tḍ̇\r\n\r\n\r", "tokens": 27, "pieces": ["㍿\r\n\r\n", "d", " e", ">", "…Aſé", "<|", "endoftext", "|>", "\tḍ̇", "\r\n\r\n\r"]} +{"text": "!fi>'re'VEſ­å'Re.EOTſa 
㋿'ſ\r\n'Mß'M'Sem\r\n\r\n12345678ḍ̇ 𐞁!!­é12345678́👍🏽12345678", "tokens": 58, "pieces": ["!fi", ">'", "re'VE", "ſ", "­å'Re", ".EOTſa", " ", "
", "㋿'", "ſ", "\r\n", "'Mß", "'", "M'S", "em", "\r\n\r\n", "123", "456", "78", "ḍ̇", " ", " 𐞁", "!!­", "é", "123", "456", "78", "́", "👍🏽", "123", "456", "78"]} +{"text": "e.́عſe​\r\nt -㋿½\n'VE('re0 0 ㍿'T ßA𐞁!!ḍ̇<|fim_prefix|>字‍½t㋿", "tokens": 59, "pieces": ["e", ".́عſe", "​\r\n", "t", " ", " -㋿", "½", "\n", "'VE", "('", "re", "", "0", " ", " ", "0", " ", " ㍿'", "T", " ", " ß", "A𐞁", "!!", "ḍ̇", "<|", "fim", "_prefix", "|>", "字", "‍", "½", "t", "㋿"]} +{"text": "#$% 's‍'ſém ſ'T0½'M㋿éZ", "tokens": 20, "pieces": ["#$%", " ", " '", "s", "‍'", "ſém", " ſ'T", "0½", "'M", "㋿é", "Z"]} +{"text": " m٣٤٥٦ \n­'D㋿", "tokens": 16, "pieces": [" m", "٣٤٥", "٦", " \n", "­'", "D", "㋿"]} +{"text": "d٣٤٥٦́­́३<\r\n..‍'S-.㍿́
İ<'T", "tokens": 30, "pieces": ["d", "٣٤٥", "٦", "́", "­́", "३", "<\r\n", "..<", "EOT", ">‍'", "S", "-.㍿́", "
İ", "<'", "T"]} +{"text": "'S\u000bs're>Ⅳt'VE<'Re><|endoftext|>​'Té
<|fim_prefix|>'s\r\n0½.'VE\u000b'M#$%'M​ſZeḍ̇.< !!d're", "tokens": 52, "pieces": ["'S", "\u000bs're", ">", "Ⅳ", "t'VE", "<'", "Re", "><|", "endoftext", "|>​'", "Té", "
", "<|", "fim", "_prefix", "|>'", "s", "\r\n", "0½", ".'", "VE", "\u000b", "'M", "#$%'", "M", "​ſ", "Zeḍ̇", ".<", " ", "!!", "d're"]} +{"text": "
́عé'll‍ḍ̇!!'ſ're\r'VE…‍'VEſe'VE😀🏽🙂sع", "tokens": 34, "pieces": ["
́عé'll", "‍ḍ̇", "!!'", "ſ're", "\r", "'VE", "…", "‍'", "VEſe'VE", "😀🏽🙂", "sع"]} +{"text": "!!́ fie 
12345678-\u000bé0're(", "tokens": 19, "pieces": ["!!́", " fie", " ", "
", "123", "456", "78", "-", "\u000bé", "0", "'re", "("]} +{"text": "\r\n.́'Dt<\" ٣٤٥٦'Sd'Md'll😀🏽 ḍ̇İ9\t<|fim_prefix|>ém​A 漢å", "tokens": 43, "pieces": ["\r\n", ".́'D", "t", "<\"", " ", "٣٤٥", "٦", "'Sd'M", "d'll", "😀🏽", " ", " ḍ̇", "İ", "9", "\t", "<|", "fim", "_prefix", "|>", "ém", "​A", " 漢å"]} +{"text": "Zfi
åZ👍🏽.'VE're…é123456780ع㋿<|endoftext|>İ'!(9'(㋿
👍🏽'Re ㋿>", "tokens": 53, "pieces": ["Zfi", "
å", "Z", "👍🏽.'", "VE're", "", "…é", "123", "456", "780", "ع", "㋿<|", "endoftext", "|>", "İ", "'!(", "9", "'(㋿", "
", "👍🏽'", "Re", " ", "㋿>"]} +{"text": "'M'llEOT'll", "tokens": 9, "pieces": ["'M", "'", "ll", "EOT'll"]} +{"text": "…éd EOTع're'DDž>㍿>İ\r\n\r\n( <­½é-'ſع're​ ", "tokens": 30, "pieces": ["…éd", " ", " EOTع're", "'DDž", ">㍿>", "İ", "\r\n\r\n", "(", " ", "<­", "½", "é", "-'", "ſع're", "​", " "]} +{"text": "ßſßfi'St", "tokens": 6, "pieces": ["ßſßfi'S", "t"]} +{"text": "Dž0t'VEm(fi\r", "tokens": 10, "pieces": ["Dž", "0", "t'VE", "m", "(fi", "\r"]} +{"text": "<#$%'Re \n \n字\r\n\r\n…́<|fim_prefix|>12345678EOT­9Dž-'' ​m𐞁३'T'T\n", "tokens": 41, "pieces": ["<#$%'", "Re", " \n \n", "字", "\r\n\r\n", "…́", "<|", "fim", "_prefix", "|>", "123", "456", "78", "EOT", "­", "9", "Dž", "-''", " ​", "m𐞁", "३", "'T'T", "\n"]} +{"text": "<|fim_prefix|>!(EOTm👍🏽Z\"", "tokens": 18, "pieces": ["<|", "fim", "_prefix", "|>!(", "EOTm", "👍🏽", "Z", "\"<", "EOT", ">"]} +{"text": "\r\n\r\nḍ̇'M३'Dع😀🏽><|endoftext|>!>\r'Re漢'VE", "tokens": 25, "pieces": ["\r\n\r\n", "ḍ̇'M", "३", "'Dع", "😀🏽><|", "endoftext", "|>!>\r", "'Re漢'VE"]} +{"text": "३'S🙂é,​'reİ㍿ꟲa-<|endoftext|>éZ­-😀🏽\u000b", "tokens": 33, "pieces": ["३", "'S", "🙂é", ",​'", "re", "İ", "㍿ꟲa", "-<|", "endoftext", "|>", "é", "Z", "­-😀🏽", "\u000b"]} +{"text": " 'VEعAßꟲ…ſ'M​­d\ré're!\t!#$%\n😀🏽", "tokens": 32, "pieces": [" ", "'VEعAßꟲ", "…ſ'M", "​­", "d", "\r", "é're", "!", "\t", "!#$%<", "EOT", ">\n", "😀🏽"]} +{"text": "'ſ…'llåé'Re.…m'reſ(\"e'Ⅳéİ­'D#$%a
'S½½३́\rعA३ḍ̇㍿", "tokens": 54, "pieces": ["'ſ", "…", "'llåé'Re", ".", "…m're", "ſ", "(\"", "e", "'", "Ⅳ", "é", "İ", "­'", "D", "#$%", "a", "
", "'S", "½½३", "́", "\r", "ع", "A", "३", "ḍ̇", "㍿"]} +{"text": ",'ſ(,'é'Re😀🏽字A㍿㍿(🙂m>'ſ٣٤٥٦å12345678३‍('T㍿‍", "tokens": 49, "pieces": [",'", "ſ", "(,'", "é", "'", "Re", "😀🏽", "字", "A", "㍿㍿(<", "EOT", ">🙂", "m", ">'", "ſ", "٣٤٥", "٦", "å", "123", "456", "78३", "‍('", "T", "㍿‍"]} +{"text": "<🙂٣٤٥٦ⅣEOTEOTt'ſ12345678 \n , ́\"\u000b,\r\n\r\n字'VEfi.'s#$%", "tokens": 32, "pieces": ["<🙂", "٣٤٥", "٦Ⅳ", "EOTEOTt'ſ", "123", "456", "78", " \n", " ,", " ́", "\"", "\u000b", ",\r\n\r\n", "字'VE", "fi", ".'", "s", "#$%"]} +{"text": ".ß -<|fim_prefix|>99\r\n\n👍🏽ḍ̇㋿t㋿<|fim_prefix|>'refi", "tokens": 32, "pieces": [".ß", " -<|", "fim", "_prefix", "|>", "99", "\r\n\n", "👍🏽", "ḍ̇", "㋿t", "㋿<|", "fim", "_prefix", "|>'", "refi"]} +{"text": "­'s12345678.İ‍fit'D(Aé\t'T9…a३İ", "tokens": 24, "pieces": ["­'", "s", "123", "456", "78", ".İ", "‍fit'D", "(Aé", "\t", "'T", "9", "…a", "३", "İ"]} +{"text": "​'Sſ漢t e,.'S ½ \nsſé'M㋿'llſ'Dž\t \"🙂tZḍ̇…", "tokens": 42, "pieces": ["​'", "Sſ", "漢t", " e", ",.'", "S", " ", " ", "½", " \n", "sſé'M", "㋿'", "llſ", "'Dž", "\t", " ", "\"🙂", "t", "Zḍ̇", "…"]} +{"text": ",9\tḍ̇", "tokens": 6, "pieces": [",", "9", "\tḍ̇"]} +{"text": "٣٤٥٦عßé\r\n…,ḍ̇\n㍿'T> <|fim_prefix|>ع<|endoftext|>s<", "tokens": 45, "pieces": ["'Dḍ̇", "\t字", "\n", "s'M", "a漢", " ", "㋿'", "re", "\t", ",🙂", "0", ".", "
é", ">", " ", "<|", "fim", "_prefix", "|>", "ع", "<|", "endoftext", "|>", "s", "<"]} +{"text": "!‍'D9Dž'llḍ̇​A㋿३\r३té'llå'D'Re…­३0漢<|fim_prefix|>9(­½<|fim_prefix|>́ ́t", "tokens": 52, "pieces": ["!‍'", "D", "9", "Dž'll", "ḍ̇", "​A", "㋿", "३", "\r", "३", "té'll", "å'D", "'Re", "…", "­", "३0", "漢", "<|", "fim", "_prefix", "|>", "9", "(­", "½", "<|", "fim", "_prefix", "|>́", " ́t"]} +{"text": "12345678<|fim_prefix|>\r\nå👍🏽#$%é漢12345678Ⅳ٣٤٥٦t🙂ع𐞁'sİ٣٤٥٦…漢👍🏽ḍ̇EOT're", "tokens": 52, "pieces": ["123", "456", "78", "<|", "fim", "_prefix", "|>\r\n", "å", "👍🏽#$%", "é漢", "123", "456", "78Ⅳ", "٣٤٥", "٦", "t", "🙂ع𐞁's", "İ", "٣٤٥", "٦", "…漢", "👍🏽", "ḍ̇", "EOT're"]} +{"text": "ßعs'㍿\"\r\n\r\n字👍🏽0afi'ſ12345678tİ<9 😀🏽\r\n\r\n-", "tokens": 32, "pieces": ["ßعs", "'㍿\"\r\n\r\n", "字", "👍🏽", "0", "afi'ſ", "123", "456", "78", "t", "İ", "<", "9", " ", "😀🏽\r\n\r\n", "-"]} +{"text": "'M字t12345678…#$%­,\n \nDž9字m\r\n", "tokens": 19, "pieces": ["'M字t", "123", "456", "78", "…", "#$%­,\n", " \n", "Dž", "9", "字m", "\r\n"]} +{"text": "字's👍🏽\u000b\r\n\r\nع<\u000b३ \nDž>㋿åßfi \n", "tokens": 49, "pieces": ["字's", "👍🏽", "\u000b\r\n\r\n", "ع", "<", "\u000b", "", "३", " \n", "Dž", ">㋿", "åßfi", " \n"]} +{"text": "'D'll fi👍🏽EOT‍s👍🏽'VEEOT a😀🏽\r!!
mZ\r\n(­😀🏽ḍ̇EOT\r\n \n", "tokens": 50, "pieces": ["'D'll", "", " fi", "👍🏽", "EOT", "‍s", "👍🏽'", "VEEOT", " ", " <", "EOT", ">a", "😀🏽\r", "!!", "
m", "Z", "\r\n", "(­<", "META", "_START", ">😀🏽", "ḍ̇", "EOT", "\r\n \n"]} +{"text": "ßꟲ‍­<|fim_prefix|>12345678\r\n\r\n", "tokens": 16, "pieces": ["ßꟲ", "‍­<|", "fim", "_prefix", "|>", "123", "456", "78", "\r\n\r\n"]} +{"text": "́", "tokens": 1, "pieces": ["́"]} +{"text": "…é­<|endoftext|>d字,\nع'M9ḍ̇'Re<'sd", "tokens": 26, "pieces": ["…é", "­<|", "endoftext", "|>", "d字", ",\n", "ع'M", "9", "ḍ̇'Re", "<'", "sd"]} +{"text": "\r\n\r\n", "tokens": 1, "pieces": ["\r\n\r\n"]} +{"text": "🙂 12345678Z𐞁‍'ſZ٣٤٥٦'re😀🏽👍🏽#$%>'ſ90‍12345678m's", "tokens": 39, "pieces": ["🙂", " <", "EOT", ">", "123", "456", "78", "Z𐞁", "‍'", "ſ", "Z", "٣٤٥", "٦", "'re", "😀🏽👍🏽#$%>'", "ſ", "90", "‍", "123", "456", "78", "m's"]} +{"text": "!fi.''D​'S​aé🙂'ſ🙂‍eé'Re½!!\u000b!!'T㍿t", "tokens": 36, "pieces": ["!fi", ".<", "EOT", ">''", "D", "​'", "S", "​aé", "🙂'", "ſ", "🙂‍", "eé'Re", "½", "!!<", "META", "_START", ">", "\u000b", "!!'", "T", "㍿t"]} +{"text": "'S'", "tokens": 6, "pieces": ["'", "S", "'"]} +{"text": "\"ſİ#$%½-½ 𐞁'Re'ſ e-\r\n'M'll
e👍🏽a'ſ", "tokens": 29, "pieces": ["\"ſ", "İ", "#$%", "½", "-", "½", " 𐞁'Re", "'ſ", " e", "-\r\n", "'M'll", "
e", "👍🏽", "a'ſ"]} +{"text": "👍🏽'VE\"fi-t's'll漢ße'Tåefis\r\r\n're
ع!å㍿å𐞁 \né😀🏽9A(é'T㋿\"-", "tokens": 48, "pieces": ["👍🏽'", "VE", "\"fi", "-t's", "'ll漢ße'T", "åefis", "\r\r\n", "'re", "
ع", "!å", "㍿å𐞁", " \n", "é", "😀🏽", "9", "A", "(é'T", "㋿\"-"]} +{"text": "𐞁​㍿", "tokens": 8, "pieces": ["𐞁", "​㍿"]} +{"text": "ꟲEOT­<|endoftext|>m​ſ \n👍🏽<|fim_prefix|>EOT>#$%t \n'llfi😀🏽Ze,", "tokens": 46, "pieces": ["ꟲ", "EOT", "­<", "META", "_START", "><|", "endoftext", "|>", "m", "​ſ", " \n", "👍🏽<|", "fim", "_prefix", "|>", "EOT", "><", "EOT", ">#$%", "t", " \n", "'llfi", "😀🏽", "Ze", ","]} +{"text": "a\r\n\r\n‍ s'ſ­é \n!!t字0\t,'S", "tokens": 17, "pieces": ["a", "\r\n\r\n", "‍", " s'ſ", "­é", " \n", "!!", "t字", "0", "\t", ",'", "S"]} +{"text": "'M!<٣٤٥٦\nDž'ſ-12345678👍🏽'Sß字Z३\r\né.'T#$%-'t𐞁​㍿<é\u000bⅣ​", "tokens": 48, "pieces": ["'M", "!<", "٣٤٥", "٦", "\n", "Dž'ſ", "-", "123", "456", "78", "👍🏽'", "Sß字", "Z", "३", "\r\n", "é", ".'", "T", "#$%-'", "t𐞁", "​㍿<", "é", "\u000b", "Ⅳ", "​"]} +{"text": "​ 'll\r\n're", "tokens": 9, "pieces": ["​", " ", " '", "ll", "\r\n", "'re"]} +{"text": " \"e'T漢<|endoftext|>Dž(…'VE㋿ſİ\r\n\r\n'reḍ̇'Re.", "tokens": 49, "pieces": [" ", "\"e'T", "漢", "<|", "endoftext", "|>", "Dž", "(", "…", "'VE", "㋿ſ", "İ", "\r\n\r\n", "A", "'", "reḍ̇'Re", "."]} +{"text": "३٣٤٥٦½", "tokens": 6, "pieces": ["३٣٤", "٥٦½"]} +{"text": "ß'Re\"fi ३İ \n<<|endoftext|>12345678🙂'ſ12345678
٣٤٥٦", "tokens": 33, "pieces": ["ß'Re", "\"fi", " ", " <", "META", "_START", ">", "३", "İ", " \n", "<<|", "endoftext", "|>", "123", "456", "78", "🙂'", "ſ", "123", "456", "78", "
", "٣٤٥", "٦"]} +{"text": "<|fim_prefix|>'s\t ḍ̇'re🙂 ß漢's\r\n\r\nZ9'Resḍ̇\n'T\"漢\"Ⅳ", "tokens": 38, "pieces": ["<|", "fim", "_prefix", "|>'", "s", "\t ", " ḍ̇'re", "🙂", " ß漢's", "\r\n\r\n", "Z", "9", "'Resḍ̇", "\n", "'T", "\"漢", "\"", "Ⅳ"]} +{"text": "‍‍㍿\u000b<|fim_prefix|> 𐞁t-​'aA>ḍ̇ 'llß٣٤٥٦da'e㍿ꟲſ", " 𐞁t", "-​'", "a", "A", ">ḍ̇", " ", "'llß", "٣٤٥", "٦", "da", "'e", "㍿ꟲſ", "<|fim_prefix|>'Refié𐞁½\u000b३ 'Dſḍ̇12345678 
", "tokens": 31, "pieces": ["‍😀🏽><|", "fim", "_prefix", "|>'", "Refié𐞁", "½", "\u000b", "३", " '", "Dſḍ̇", "123", "456", "78", " 
"]} +{"text": "\t'T İ​Z😀🏽", "tokens": 9, "pieces": ["\t", "'T", " İ", "​Z", "😀🏽"]} +{"text": "'D!!Ⅳs're ­😀🏽'Re٣٤٥٦́ ㋿٣٤٥٦٣٤٥٦ Dž'M<|endoftext|>
", "tokens": 50, "pieces": ["'D", "!!", "Ⅳ", "​<", "EOT", ">s're", " ", " ­😀🏽'", "Re", "٣٤٥", "٦", "́", " ㋿", "٣٤٥", "٦٣٤", "٥٦", " Dž'M", "<|", "endoftext", "|>", "
"]} +{"text": "ßåéḍ̇ꟲfi👍🏽'll\"", "tokens": 24, "pieces": ["ßåé", "<", "EOT", ">ḍ̇ꟲfi", "👍🏽'", "ll", "\""]} +{"text": "\"t'reع漢́٣٤٥٦Ⅳ\t٣٤٥٦<́s'Sع​m\tⅣ.\r'Déé", "tokens": 31, "pieces": ["\"t're", "ع漢́", "٣٤٥", "٦Ⅳ", "\t", "٣٤٥", "٦", "<́s'S", "ع", "​m", "\t", "Ⅳ", ".\r", "'Déé"]} +{"text": "('M\r字…-३ \u000b…
é<|fim_prefix|>'M.'Re", "tokens": 23, "pieces": ["('", "M", "\r", "字", "…", "-", "३", " \u000b…", "
é", "<|", "fim", "_prefix", "|>'", "M", ".'", "Re"]} +{"text": "'VE.<|endoftext|>fi字\t​\r\n\r\n字​ꟲ\r Dž́𐞁'D12345678٣٤٥٦'T<\r\n'ſ12345678👍🏽漢
𐞁'll­12345678 ع½😀🏽#$% \r\n\r\n", "tokens": 69, "pieces": ["'VE", ".<|", "endoftext", "|>", "fi字", "\t", "​\r\n\r\n", "字", "​ꟲ", "\r", " ", " Dž́𐞁'D", "123", "456", "78٣", "٤٥٦", "'T", "<\r\n", "'ſ", "123", "456", "78", "👍🏽", "漢", "
𐞁'll", "­", "123", "456", "78", " ", " ع", "½", "😀🏽#$%", " \r\n\r\n"]} +{"text": "漢!!'Re !!", "tokens": 6, "pieces": ["漢", "!!'", "Re", " ", " !!"]} +{"text": "-EOT", "tokens": 5, "pieces": ["-", "EOT"]} +{"text": "­
ꟲfi㋿٣٤٥٦👍🏽(३'M\"'T('re😀🏽>Afi12345678", "tokens": 36, "pieces": ["­", "
ꟲfi", "㋿", "٣٤٥", "٦", "👍🏽(", "३", "'M", "\"'", "T", "('", "re", "😀🏽>", "Afi", "", "123", "456", "78"]} +{"text": "㍿'s'ſ字\u000b'Re😀🏽,'T́­ 𐞁
!!fi ,漢漢३\t३ß́\rⅣ\r'ſ'ſå\r\r\n\r\n👍🏽", "tokens": 49, "pieces": ["㍿'", "s'ſ", "字", "\u000b", "'Re", "😀🏽,'", "T́", "­", " 𐞁", "
", "!!", "fi", " ", ",漢漢", "३", "\t", "३", "ß́", "\r", "Ⅳ", "\r", "'ſ'ſ", "å", "\r\r\n\r\n", "👍🏽"]} +{"text": "\t12345678🙂İ­Dža­< A", "tokens": 14, "pieces": ["\t", "123", "456", "78", "🙂İ", "­Dža", "­<", " A"]} +{"text": "a<|fim_prefix|>\r\n\r\n\r\n\r\n
­👍🏽 m'så'VE", "tokens": 19, "pieces": ["a", "<|", "fim", "_prefix", "|>\r\n\r\n\r\n\r\n", "
", "­👍🏽", " m's", "å'VE"]} +{"text": " \r\n\r\n😀🏽‍'VE-İ\naA<字<|endoftext|>#$%a🙂é>ع!! ꟲ​dſ½'T", "tokens": 47, "pieces": [" \r\n\r\n", "😀🏽‍'", "VE", "-İ", "\n", "a", "A", "<字", "<|", "endoftext", "|>#$%", "a", "🙂é", ">ع", "!!<", "EOT", ">", " ꟲ", "​dſ", "½", "'T"]} +{"text": "'D३a漢𐞁🙂12345678<|fim_prefix|>>㍿'VE٣٤٥٦ EOT'T\t🙂e字(", "tokens": 39, "pieces": ["'D", "३", "a漢𐞁", "🙂", "123", "456", "78", "<|", "fim", "_prefix", "|>>㍿<", "META", "_START", ">'", "VE", "٣٤٥", "٦", " EOT'T", "\t", "🙂e字", "("]} +{"text": "́  \r\nfi", "tokens": 5, "pieces": ["́", "  \r\n", "fi"]} +{"text": "'Sİt­.'D's𐞁\r\n-'ll'll.İ́\r\n\r\n'\n字éåt \nß", "tokens": 30, "pieces": ["'Sİt", "­.'", "D's", "𐞁", "\r\n", "-'", "ll'll", ".<", "EOT", ">İ́", "\r\n\r\n", "'\n", "字éåt", " \n", "ß"]} +{"text": "🙂t\"ꟲ\r\nꟲİ<|endoftext|> ḍ̇é 'T​𐞁漢t ſع12345678fiع", "tokens": 40, "pieces": ["🙂t", "\"ꟲ", "\r\n", "ꟲ", "İ", "<|", "endoftext", "|>", " ḍ̇é", " ", "'T", "​𐞁漢t", " ſع", "123", "456", "78", "fiع"]} +{"text": "'Re'D'VEDž.!!İꟲ", "tokens": 12, "pieces": ["'Re'D", "'VEDž", ".!!", "İꟲ"]} +{"text": "fiDž­s㍿'
㋿'reſé👍🏽漢𐞁<|fim_prefix|>ḍ̇tDž'sa", "tokens": 42, "pieces": ["fi", "Dž", "­s", "㍿'", "
", "㋿'", "reſé", "👍🏽<", "META", "_START", ">漢𐞁", "<|", "fim", "_prefix", "|>", "ḍ̇t", "Dž's", "a"]} +{"text": "éꟲ漢٣٤٥٦ 'M\tfi\u000b're字", "tokens": 17, "pieces": ["éꟲ漢", "٣٤٥", "٦", " ", " '", "M", "\tfi", "\u000b", "'re字"]} +{"text": " …ſe'9漢­字Ⅳ😀🏽㋿\u000b\r\n\r\nd'D🙂… d#$%'Reḍ̇e㍿Zt…\u000bİ>e\r­a\r", "tokens": 52, "pieces": [" ", "…ſe", "'", "9", "漢", "­字", "", "Ⅳ", "😀🏽㋿", "\u000b\r\n\r\n", "d'D", "🙂", "…", " d", "#$%'", "Reḍ̇e", "㍿Zt", "…", "\u000bİ", ">e", "\r", "­a", "\r"]} +{"text": ">,<|endoftext|>İ漢 \r\n\r\ns字!ſ0漢Z\u000bEOT'VE
A👍🏽,", "tokens": 29, "pieces": [">,<|", "endoftext", "|>", "İ漢", " \r\n\r\n", "s字", "!ſ", "0", "漢", "Z", "\u000bEOT'VE", "
A", "👍🏽,"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'TdⅣfi­t\t-­'s<\"'VE'M 'Re'VE漢ſ
'D
‍㍿9#$%ſ'd
\r", "tokens": 42, "pieces": ["'Td", "Ⅳ", "fi", "­t", "\t", "-­'", "s", "<<", "META", "_START", ">\"'", "VE'M", " ", "'Re'VE", "漢ſ", "
", "'D", "
", "‍㍿", "9", "#$%", "ſ'd", "
\r"]} +{"text": "eḍ̇ ,'M0tEOT \n'T\"👍🏽t漢
!!㍿\r\nſİ😀🏽<|endoftext|>'reZ's", "tokens": 39, "pieces": ["eḍ̇", " ,'", "M", "0", "t", "EOT", " \n", "'T", "\"👍🏽", "t漢", "
", "!!㍿\r\n", "ſ", "İ", "😀🏽<|", "endoftext", "|>'", "re", "Z's"]} +{"text": "𐞁'M'ſZ'T'Då\u000bßfi'VE'S\n\n漢'S\r\nſ'SEOT\" Ⅳ", "tokens": 33, "pieces": ["𐞁'M", "'ſ", "Z'T", "'Då", "\u000bßfi'VE", "'", "S", "\n\n", "漢'S", "\r\n", "ſ'S", "EOT", "\"", " ", "Ⅳ"]} +{"text": "😀🏽  -0'M \n㋿9", "tokens": 13, "pieces": ["😀🏽", " ", " ", "-", "0", "'M", " \n", "㋿", "9"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "½'Da'D-<|fim_prefix|>'re'D​m'å३\r\n\r\n३Afi(, e.'re'Reḍ̇'T", "tokens": 37, "pieces": ["½", "'Da'D", "-<|", "fim", "_prefix", "|>'", "re'D", "​m", "'å", "३", "\r\n\r\n", "३", "Afi", "(,", " ", " e", ".'", "re'Re", "ḍ̇", "'", "T"]} +{"text": "\rm‍㍿0('Séع<|endoftext|>­ ", "tokens": 20, "pieces": ["\r", "m", "‍㍿", "0", "('", "Séع", "<|", "endoftext", "|>­", " "]} +{"text": "'ſaEOT \n\t🙂é漢're​Dž🙂㍿(dm㍿éfiEOT..́…,(0.👍🏽…İ­!㋿", "tokens": 46, "pieces": ["'ſa", "EOT", " \n", "\t", "🙂é漢're", "​Dž", "🙂㍿(", "dm", "㍿éfi", "EOT", "..́", "…", ",(", "0", ".👍🏽", "…İ", "­!㋿"]} +{"text": ",'VEéDž", "tokens": 6, "pieces": [",'", "VEé", "Dž"]} +{"text": "'llÁ㋿12345678 é𐞁Aa9𐞁fi㋿ḍ̇'ſſ'D\nḍ̇ß́>('VE'VE,㋿", "tokens": 48, "pieces": ["'ll", "Á", "㋿", "123", "456", "78", " ", " é𐞁", "Aa", "9", "𐞁fi", "㋿ḍ̇'ſ", "ſ'D", "\n", "ḍ̇ß́", ">('", "VE'VE", ",㋿"]} +{"text": "'ll-ḍ̇ \r'ſd!! ३,𐞁½ſ éḍ̇sḍ̇é'ſ!\t<|fim_prefix|>\t­
字😀🏽ß<,'re…\u000b.A𐞁'M", "tokens": 62, "pieces": ["'ll", "-ḍ̇", " \r", "'ſd", "!!", " ", "३", ",𐞁", "½", "ſ", " éḍ̇sḍ̇é'ſ", "!", "\t", "<|", "fim", "_prefix", "|>", "\t", "­", "
字", "😀🏽", "ß", "<,'", "re", "…", "\u000b", ".A𐞁'M"]} +{"text": "🙂s\n\r­'re EOT!漢🙂漢'reß'D<.'ſ Ⅳ'T9\n's(m're12345678>A<|endoftext|>#$%", "tokens": 46, "pieces": ["🙂s", "\n\r", "­'", "re", " EOT", "!漢", "🙂<", "EOT", ">漢're", "ß'D", "<.'", "ſ", " ", "Ⅳ", "'T", "9", "\n", "'s", "(m're", "123", "456", "78", ">A", "<|", "endoftext", "|>#$%"]} +{"text": "fiİ​ſ'VEſ0-0 \n, 9<|fim_prefix|>漢½", "tokens": 25, "pieces": ["fi", "İ", "​ſ'VE", "ſ", "0", "-", "0", " \n", ",", " ", "9", "<|", "fim", "_prefix", "|>", "漢", "½"]} +{"text": "\r\n-\rs३\t́'T#$%\r\ne​½'ll<|fim_prefix|>", "tokens": 24, "pieces": ["\r\n", "-\r", "s", "३", "\t́'T", "#$%<", "META", "_START", ">\r\n", "e", "​", "½", "'ll", "<|", "fim", "_prefix", "|>"]} +{"text": "漢\"9'ſ're", "tokens": 6, "pieces": ["漢", "\"", "9", "'ſ're"]} +{"text": "(Zs\u000b 
.<'VE\r\n\r\n㍿🙂㍿‍\t(🙂½\r're
 😀🏽🙂!Z \n😀🏽", "tokens": 39, "pieces": ["(Zs", "\u000b ", "
", ".<'", "VE", "\r\n\r\n", "㍿🙂㍿‍", "\t", "(🙂", "½", "\r", "'re", "
 ", " 😀🏽🙂!", "Z", "", " \n", "😀🏽"]} +{"text": "é<|fim_prefix|>d", "tokens": 9, "pieces": ["é", "<|", "fim", "_prefix", "|>", "d"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "12345678A🙂𐞁👍🏽.<|fim_prefix|>\u000b's\u000b!!!'\n\rEOTdꟲsta'S9fi<Ⅳ\u000b ع\r\n\r\n½ 'S", "tokens": 44, "pieces": ["123", "456", "78", "A", "🙂𐞁", "👍🏽.<|", "fim", "_prefix", "|>", "\u000b", "'s", "\u000b", "!!!'\n\r", "EOTdꟲsta'S", "9", "fi", "<", "Ⅳ", "\u000b ", " ع", "\r\n\r\n", "½", " ", "'S"]} +{"text": "(­ꟲ'llZ😀🏽́½㋿< \nå\rßå \n", "tokens": 25, "pieces": ["(­", "ꟲ'll", "Z", "😀🏽́", "½", "㋿<", " \n", "å", "\r", "ßå", " \n"]} +{"text": "!!#$%Ⅳḍ̇漢'llfißeDž'D ‍㍿.(字<|fim_prefix|>ee👍🏽12345678eDžé'T!'VE㋿ m", "tokens": 51, "pieces": ["!!#$%", "Ⅳ", "ḍ̇漢'll", "fiße", "Dž'D", " ", "‍㍿.(", "字", "<|", "fim", "_prefix", "|>", "ee", "👍🏽", "123", "456", "78", "e", "Džé'T", "!'", "VE", "㋿", " ", " m"]} +{"text": "\r'😀🏽'Re  İ'ſ\"🙂 𐞁\r\n\t#$% ­.a!\t  \n\u000bſ", "tokens": 33, "pieces": ["\r", "'😀🏽'", "Re", " ", " İ'ſ", "\"🙂", " 𐞁", "\r\n", "\t", "#$%", " ", "­.", "a", "!", "\t  \n", "\u000bſ"]} +{"text": "ßعe're9'sßſ漢'VE'VE½\"👍🏽'ReZ😀🏽'VE12345678ḍ̇👍🏽
㋿Z…\t (\r\n'reⅣ'T'12345678'S
", "tokens": 55, "pieces": ["ßعe're", "9", "'sßſ漢'VE", "'VE", "½", "\"👍🏽'", "Re", "Z", "😀🏽'", "VE", "123", "456", "78", "ḍ̇", "👍🏽", "
", "㋿Z", "…\t", " ", "(\r\n", "'re", "Ⅳ", "'T", "'", "123", "456", "78", "'S", "
"]} +{"text": "ḍ̇
ꟲ'T'VE
'ſ.Dž\r\n'…𐞁 !'D\r\n\r\n'Mt字 ſ​🙂,­s\r\n\r\nꟲ'Mt(㋿½३", "tokens": 53, "pieces": ["ḍ̇", "
ꟲ'T", "'VE", "
", "'ſ", ".", "Dž", "\r\n", "'", "…𐞁", " ", "!'", "D", "\r\n\r\n", "'Mt字", " ſ", "​🙂,­", "s", "\r\n\r\n", "ꟲ'M", "t", "(㋿", "½३"]} +{"text": "åAⅣ🙂ßfi'", "tokens": 9, "pieces": ["å", "A", "Ⅳ", "🙂ßfi", "'"]} +{"text": "'s(\nd👍🏽'll A(\n.", "tokens": 12, "pieces": ["'s", "(\n", "d", "👍🏽'", "ll", " A", "(\n", "."]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "#$%😀🏽𐞁A\r\n\"'re0‍s㍿­㋿'s
0½­fi\r\n>́", "tokens": 36, "pieces": ["#$%😀🏽", "𐞁", "A", "\r\n", "\"'", "re", "0", "‍s", "㍿­㋿'", "s", "
", "0½", "­fi", "\r\n", ">́"]} +{"text": "'se'M!,‍eع\r\n,9", "tokens": 10, "pieces": ["'se'M", "!,‍", "eع", "\r\n", ",", "9"]} +{"text": "<|fim_prefix|>,EOT<…-dfie'reZå 912345678> 'Tfi'D​٣٤٥٦!!🙂EOT\r\n\r\n'ſfi­\r٣٤٥٦‍'re㍿", "tokens": 52, "pieces": ["<|", "fim", "_prefix", "|>,", "EOT", "<", "…", "-dfie're", "Zå", " ", " ", "912", "345", "678", ">", " ", "'Tfi'D", "​", "٣٤٥", "٦", "!!🙂", "EOT", "\r\n\r\n", "'ſfi", "­\r", "٣٤٥", "٦", "‍'", "re", "㍿"]} +{"text": "ع!'Dß", "tokens": 8, "pieces": ["ع", "!'", "Dß", ""]} +{"text": "漢tZ<'ReDž-㍿('Re0'M\u000b-fi½\r\n!!Z", "tokens": 22, "pieces": ["漢t", "Z", "<'", "Re", "Dž", "-㍿('", "Re", "0", "'M", "\u000b", "-fi", "½", "\r\n", "!!", "Z"]} +{"text": "'s'ſs, -\r'séZDž<|fim_prefix|> \ns३a
👍🏽½!👍🏽's
!!\r㋿tİꟲ \nsåꟲ\u000b", "tokens": 52, "pieces": ["'s'ſ", "s", ",", " ", "-\r", "'sé", "ZDž", "<|", "fim", "_prefix", "|>", " \n", "s", "३", "a", "
", "👍🏽", "½", "!👍🏽'", "s", "
", "!!\r", "㋿t", "İꟲ", " \n", "såꟲ", "\u000b"]} +{"text": "t‍!!fi 'VE", "tokens": 7, "pieces": ["t", "‍!!", "fi", " ", "'VE"]} +{"text": "㋿ \"'s​EOT#$%\ne", "tokens": 30, "pieces": ["'reعfi", ".­", "eعé", "\tDžA're", "Dž'S", "e", "!<", "EOT", ">'", "s", "​EOT", "#$%\n", "e"]} +{"text": "½Dž \n½,e'D٣٤٥٦𐞁\r\n\r\ne
<|endoftext|>\u000bſ 0<('Tfi", "tokens": 52, "pieces": ["½", "Dž", " \n", "½", ",e'D", "٣٤٥", "٦", "𐞁", "\r\n\r\n", "e", "
", "<|", "endoftext", "|>", "\u000bſ", " ", "0", "<('", "Tfi"]} +{"text": "​A ​٣٤٥٦́a're'VE‍ ㍿å0'VÉ9're!!'D'S0\"!½'Re", "tokens": 35, "pieces": ["​A", " ", "​", "٣٤٥", "٦", "́a're", "'VE", "‍", " ", " ㍿", "å", "0", "'VÉ", "9", "'re", "!!'", "D'S", "0", "\"!", "½", "'Re"]} +{"text": "'S½漢́ \r\n'S.,ḍ̇\"'SZ t<|fim_prefix|>😀🏽e'T㋿dé㋿Ⅳ9'ſ字fi!", "tokens": 45, "pieces": ["'S", "½", "漢́", " \r\n", "'S", ".,", "ḍ̇", "\"'", "SZ", " ", " t", "<|", "fim", "_prefix", "|>😀🏽", "e'T", "㋿dé", "㋿", "Ⅳ9", "'", "ſ字fi", "!"]} +{"text": "🙂fi‍',
é​ \n12345678𐞁…'VEA'Re", "tokens": 22, "pieces": ["🙂fi", "‍',", "
é", "​", " \n", "123", "456", "78", "𐞁", "…", "'VEA'Re"]} +{"text": "'Re漢fi\"12345678'ſ'ſ<|endoftext|>㍿𐞁३字9\u000b\" ſ漢m'VEİ<|endoftext|>,'llé 9DžDž٣٤٥٦\n", "tokens": 60, "pieces": ["'Re漢fi", "\"", "123", "456", "78", "'ſ'ſ", "<|", "endoftext", "|>㍿", "𐞁", "३", "字", "9", "\u000b", "\"", " ſ漢m'VE", "İ", "<|", "endoftext", "|>,'", "llé", " ", "9", "DžDž", "٣٤٥", "٦", "\n"]} +{"text": "عꟲ<|endoftext|>é\r\nİDž'Re​ꟲ>-漢字ع́!!ß'ree \n'👍🏽­İß𐞁٣٤٥٦", "tokens": 46, "pieces": ["عꟲ", "<|", "endoftext", "|>", "é", "\r\n", "İDž'Re", "​ꟲ", ">-", "漢字ع́", "!!", "ß're", "e", " \n", "'👍🏽­", "İß𐞁", "٣٤٥", "٦"]} +{"text": "…́㍿'Re 'llt ३ \nmZ''re'ſ", "tokens": 22, "pieces": ["…́", "㍿'", "Re", " '", "llt", " ", "३", " \n", "m", "Z", "'<", "EOT", ">'", "re'ſ"]} +{"text": " ㍿Dž <|fim_prefix|>m's٣٤٥٦-éDž ع‍fi\"٣٤٥٦'Re'𐞁𐞁,.\r\n'Re漢", "tokens": 43, "pieces": [" ", "㍿Dž", " <|", "fim", "_prefix", "|>", "m's", "٣٤٥", "٦", "-é", "Dž", " ع", "‍fi", "\"", "٣٤٥", "٦", "'Re", "'𐞁𐞁", ",.\r\n", "'Re漢"]} +{"text": "'ReEOT'llꟲDžé0\tfi\"🙂'S'ſ'DZés'll' tḍ̇.", "tokens": 33, "pieces": ["'Re", "EOT'll", "ꟲDžé", "0", "\tfi", "\"🙂<", "EOT", ">'", "S'ſ", "'DZés'll", "'", " tḍ̇", "."]} +{"text": "\u000b<-<|fim_prefix|>#$%ſ'D'ſEOTAḍ̇
12345678!'Re'ſİ'VE're's'll'T0'S'Re\r\n<'re,>٣٤٥٦😀🏽½½\"\t…ع", "tokens": 57, "pieces": ["\u000b", "<-<|", "fim", "_prefix", "|>#$%", "ſ'D", "'ſ", "EOTAḍ̇", "
", "123", "456", "78", "!'", "Re'ſ", "İ'VE", "'re's", "'ll'T", "0", "'S'Re", "\r\n", "<'", "re", ",>", "٣٤٥", "٦", "😀🏽", "½½", "\"", "\t", "…ع"]} +{"text": "9 9İ३ſ\n'ree<|fim_prefix|>'llḍ̇…𐞁9m>ع'D!!\nse‍​å12345678漢'ſ
.Ⅳ", "tokens": 46, "pieces": ["9", " ", "9", "İ", "३", "ſ", "\n", "'ree", "<|", "fim", "_prefix", "|>'", "llḍ̇", "…𐞁", "9", "m", ">ع'D", "!!\n", "se", "‍​", "å", "123", "456", "78", "漢'ſ", "
", ".", "Ⅳ"]} +{"text": "\n99İ​a'Re漢 éſ\r\n\r\nDž", "tokens": 13, "pieces": ["\n", "99", "İ", "​a'Re", "漢", " éſ", "\r\n\r\n", "Dž"]} +{"text": "́\"\r\n\r\n'VE.́Z٣٤٥٦😀🏽a<|endoftext|>\"'ſfié  'S\r\n\r\n's12345678'T‍eDžéع", "tokens": 47, "pieces": ["́", "\"\r\n\r\n", "'VE", ".́", "Z", "٣٤٥", "٦", "😀🏽", "a", "<|", "endoftext", "|>\"'", "ſfié", " ", " <", "META", "_START", ">", " ", " ", "'S", "\r\n\r\n", "'s", "123", "456", "78", "'T", "‍e", "Džéع"]} +{"text": "٣٤٥٦<|fim_prefix|>''s㍿Ⅳ́  \n३ſ🙂\r\n\r\n\n㍿'s\r\nå", "tokens": 33, "pieces": ["٣٤٥", "٦", "<|", "fim", "_prefix", "|>''", "s", "㍿", "Ⅳ", "́", "  \n", "३", "ſ", "🙂\r\n\r\n\n", "㍿'", "s", "\r\n", "å"]} +{"text": "!३ \n", "tokens": 3, "pieces": ["!", "३", " \n"]} +{"text": "𐞁'Z㋿㋿Ⅳ‍!'Re  ", "tokens": 18, "pieces": ["𐞁", "'Z", "㋿㋿", "Ⅳ", "‍!'", "Re", "  "]} +{"text": "-
", "tokens": 2, "pieces": ["-", "
"]} +{"text": "‍­ꟲ,<㋿'s🙂", "tokens": 12, "pieces": ["‍­", "ꟲ", ",<㋿'", "s", "🙂"]} +{"text": "\rİⅣ漢0\"​éétḍ̇\"İ\u000b!!'reⅣ 12345678\t㋿0<|endoftext|>12345678…́\r\n\u000bſ\"!éⅣ-", "tokens": 53, "pieces": ["\r", "İ", "Ⅳ", "漢", "0", "\"​", "éétḍ̇", "\"İ", "\u000b", "!!'", "re", "Ⅳ", " ", "123", "456", "78", "\t", "㋿", "0", "<|", "endoftext", "|>", "123", "456", "78", "…́", "\r\n", "\u000bſ", "\"!", "é", "Ⅳ", "-"]} +{"text": "a<|fim_prefix|>", "tokens": 7, "pieces": ["a", "<|", "fim", "_prefix", "|>"]} +{"text": "é>Ź👍🏽­'T‍ …ß<İ'st\r\né­㋿\r\n😀🏽Ⅳfie\u000b", "tokens": 38, "pieces": ["é", ">Ź", "👍🏽­'", "T", "‍", " ", "…ß", "<İ's", "t", "\r\n", "é", "­㋿\r\n", "😀🏽", "Ⅳ", "fie", "\u000b"]} +{"text": "s\"å!Ⅳİe\r\n\r\n𐞁9\r㋿½\n'M-é ­\u000b#$%", "tokens": 43, "pieces": ["s", "\"å", "!", "Ⅳ", "İe", "\r\n\r\n", "𐞁", "9", "\r", "㋿", "½", "\n", "'M", "-é", " ", " <", "s漢'S", " ", "'s", "­", " \n", "'MDžꟲ", "­>­", "\u000b", "#$%"]} +{"text": "…'re\n\r9Ⅳ \n㋿åå\u000b's'Tꟲ­ꟲ\n'T\r\n\r\n12345678 😀🏽's'll12345678e½", "tokens": 47, "pieces": ["…", "'re", "\n\r", "9Ⅳ", " \n", "㋿åå", "\u000b", "'s'T", "ꟲ", "­ꟲ", "\n", "'T", "\r\n\r\n", "123", "456", "78", " ", " 😀🏽<", "META", "_START", ">'", "s'll", "123", "456", "78", "e", "½"]} +{"text": "字漢\u000b‍fit'VE\r'S'😀🏽ع…\t", "tokens": 18, "pieces": ["字漢", "\u000b", "‍fit'VE", "\r", "'S", "'😀🏽", "ع", "…\t"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": ".0३'MA३ḍ̇\r<|endoftext|>12345678
㍿t", "tokens": 25, "pieces": [".", "0३", "'MA", "३", "ḍ̇", "\r", "<|", "endoftext", "|>", "123", "456", "78", "
", "㍿t"]} +{"text": "\r\n\r\n'll㍿>\n字é<|endoftext|>\t\t12345678>-Dž", "tokens": 24, "pieces": ["\r\n\r\n", "'ll", "㍿>\n", "字é", "<|", "endoftext", "|>", "\t", "\t", "123", "456", "78", ">-", "Dž"]} +{"text": "­ …!ḍ̇m𐞁'Tعſ\r\n\r\nſ́ 0m", "tokens": 29, "pieces": ["­", " ", "…", "!ḍ̇m𐞁'T", "عſ", "\r\n\r\n", "ſ́", "", " ", "0", "m"]} +{"text": "字\"e9ع'M­tſ👍🏽>ßDž­", "tokens": 16, "pieces": ["字", "\"e", "9", "ع'M", "­tſ", "👍🏽>", "ß", "Dž", "­"]} +{"text": "ع…A'D🙂'reEOT#$%! 'M'VE ٣٤٥٦'T‍-👍🏽#$%'Re\n\n 字", "tokens": 38, "pieces": ["ع", "…A'D", "🙂'", "re", "EOT", "#$%!<", "META", "_START", ">", " ", " '", "M'VE", " ", "٣٤٥", "٦", "'T", "‍-👍🏽#$%'", "Re", "\n\n", " ", " 字"]} +{"text": "'Tt𐞁ḍ̇ḍ̇३عEOT'Sſ'd.s​t🙂ݽ‍漢t<|fim_prefix|>", "tokens": 34, "pieces": ["'Tt𐞁ḍ̇ḍ̇", "३", "ع", "EOT'S", "ſ'd", ".s", "​t", "🙂İ", "½", "‍漢t", "<|", "fim", "_prefix", "|>"]} +{"text": " 😀🏽é'll!३'VEZs\r\n\r\n\t\n<|endoftext|>\n\r're", "tokens": 22, "pieces": [" 😀🏽", "é'll", "!", "३", "'VEZs", "\r\n\r\n\t\n", "<|", "endoftext", "|>\n\r", "'re"]} +{"text": "mdİa½a'VE-d​#$% ३", "tokens": 14, "pieces": ["md", "İa", "½", "a'VE", "-d", "​#$%", " ", " ", "३"]} +{"text": "\u000b", "tokens": 1, "pieces": ["\u000b"]} +{"text": "😀🏽'VE㍿.--\u000b're㍿eé'S<|fim_prefix|>🙂𐞁'S-Áİ<|endoftext|>\u000b12345678 Z'Ms\r\n0ſ字", "tokens": 52, "pieces": ["😀🏽'", "VE", "㍿.--", "\u000b", "'re", "㍿eé'S", "<|", "fim", "_prefix", "|>🙂", "𐞁'S", "-Á", "İ", "<|", "endoftext", "|>", "\u000b", "123", "456", "78", " Z'M", "s", "\r\n", "0", "ſ字"]} +{"text": "'D́عe
ꟲt'Re\"­.!!‍‍é
#$%ḍ̇#$%ꟲ>s漢e…mEOTꟲ…", "tokens": 41, "pieces": ["'D́عe", "
ꟲt'Re", "\"­.!!‍‍", "é", "
", "#$%", "ḍ̇", "#$%", "ꟲ", ">s漢e", "…m", "EOTꟲ", "…"]} +{"text": "a0­\r\n\r\n (s(字mع t'ſ'T!!", "tokens": 17, "pieces": ["a", "0", "­\r\n\r\n", " ", " (", "s", "(字mع", " t'ſ", "'T", "!!"]} +{"text": "‍'sꟲßs!!٣٤٥٦0ꟲ!!́㍿ſ'D\r\n½!!\r\n é\t😀🏽́́'S
…fiⅣ", "tokens": 48, "pieces": ["‍'", "sꟲßs", "!!", "٣٤٥", "٦0", "ꟲ", "!!́㍿", "ſ", "'", "D", "\r\n", "½", "!!\r\n", " ", " é", "\t", "😀🏽́́'", "S", "
", "…fi", "Ⅳ"]} +{"text": "\r\n\r\n<|fim_prefix|>\"‍-''", "tokens": 10, "pieces": ["\r\n\r\n", "<|", "fim", "_prefix", "|>\"‍-''"]} +{"text": " <|fim_prefix|>!!0ſ'Re'ſ<|endoftext|> \n\n ㋿'ſ'Reå'll\r\n9३\"-字.d<Ⅳå", "tokens": 43, "pieces": [" ", "<|", "fim", "_prefix", "|>!!", "0", "ſ'Re", "'ſ", "<|", "endoftext", "|>", " \n\n", " ", " ㋿'", "ſ'Re", "å'll", "\r\n", "9३", "\"-", "字", ".d", "<", "Ⅳ", "å"]} +{"text": "(A <|fim_prefix|>🙂'll,漢 å㋿'VE9'VEDž.𐞁​A\"\r'se", "tokens": 52, "pieces": ["(A", "", " ", "<|", "fim", "_prefix", "|><", "META", "_START", ">🙂<", "EOT", ">'", "ll", ",漢", " ", " å", "㋿'", "VE", "9", "'VEDž", ".𐞁", "​<", "META", "_START", ">A", "\"\r", "'se"]} +{"text": "e字'ſ㋿'ſſḍ̇Ⅳt字ſ'Ś…😀🏽\u000b\nİa\n​🙂'M,٣٤٥٦fi \u000b漢", "tokens": 50, "pieces": ["e字'ſ", "㋿'", "ſſḍ̇", "Ⅳ", "t字", "ſ'S", "́", "…", "😀🏽", "\u000b\n", "İa", "\n", "​🙂'", "M", ",", "٣٤٥", "٦", "fi", " ", "\u000b漢"]} +{"text": " '\u000bḍ̇'llß👍🏽­㋿Ⅳſ𐞁漢½漢­३#$%(😀🏽\"½ſⅣDž'VE'VEé a", "tokens": 50, "pieces": [" ", "'", "\u000bḍ̇'ll", "ß", "👍🏽­㋿", "Ⅳ", "ſ", "𐞁漢", "½", "漢", "­", "३", "#$%(😀🏽\"", "½", "ſ", "Ⅳ", "Dž'VE", "'VEé", " ", " a"]} +{"text": "<|fim_prefix|>!!ß're'reDž#$%12345678\rmå㋿'D\nḍ̇-12345678fi㍿'Re​12345678é\t", "tokens": 45, "pieces": ["<|", "fim", "_prefix", "|>!!", "ß're", "'re", "Dž", "#$%", "123", "456", "78", "\r", "må", "㋿'", "D", "\n", "ḍ̇", "-", "123", "456", "78", "fi", "㍿'", "Re", "​", "123", "456", "78", "é", "\t"]} +{"text": "fi㋿s'ſ́'T'<|fim_prefix|>-İe'll Dž߅'Re ㍿EOT #$%㍿…\u000bZ'VE'S
ſ\r\n\r\n (", "tokens": 55, "pieces": ["fi", "㋿s'ſ", "́'T", "'<|", "fim", "_prefix", "|>-<", "META", "_START", ">İe'll", " Džß", "…", "'Re", " ", " ㍿", "EOT", " ", " #$%㍿", "…", "\u000bZ'VE", "'S", "
ſ", "\r\n\r\n", " ", "("]} +{"text": " 🙂#$%#$%9'VE!!'VE…eDž㋿d\"'SEOT 漢're
'#$%!'Ret0\u000b'S EOT字åعa912345678\r\n\r\n\u000béDž<|fim_prefix|>", "tokens": 26, "pieces": ["é", "<", "META", "_START", ">a", "912", "345", "678", "\r\n\r\n", "\u000bé", "Dž", "<|", "fim", "_prefix", "|>"]} +{"text": "👍🏽\r0s३🙂漢ée \n𐞁DžsⅣ(\n'ſ'll½,ſ<|fim_prefix|>İ́>ḍ̇<👍🏽!!'ll'S٣٤٥٦字'VE>", "tokens": 55, "pieces": ["👍🏽\r", "0", "s", "३", "🙂漢ée", " \n", "𐞁Džs", "Ⅳ", "(\n", "'ſ'll", "½", ",ſ", "<|", "fim", "_prefix", "|>", "İ́", ">ḍ̇", "<👍🏽!!'", "ll'S", "٣٤٥", "٦", "字'VE", ">"]} +{"text": "\n!m- \n'll'Tt'VE…Dž👍🏽ع३'ſß\r\nſ å漢漢\u000bA! ́㋿0ع", "tokens": 46, "pieces": ["\n", "!m", "-", " \n", "'", "ll'T", "t'VE", "…Dž", "👍🏽", "ع", "३", "'ſß", "\r\n", "ſ", " å漢漢", "\u000bA", "!", " ", " ́", "㋿", "0", "ع"]} +{"text": " 'T!!'sDž", "tokens": 7, "pieces": [" '", "T", "!!'", "s", "Dž"]} +{"text": "…\u000b's Ⅳß", "tokens": 11, "pieces": ["…", "\u000b", "'s", "", " ", "Ⅳ", "ß"]} +{"text": "#$%Ⅳ\r\n9 Ⅳ0'll𐞁
,'M'll‍'Dḍ̇Dž㍿٣٤٥٦ \n३,Z🙂é\r\n\r\ns", "tokens": 45, "pieces": ["#$%", "Ⅳ", "\r\n", "9", " ", " <", "META", "_START", ">", "Ⅳ0", "'ll𐞁", "
", ",'", "M'll", "‍'", "Dḍ̇", "Dž", "㍿", "٣٤٥", "٦", " \n", "३", ",Z", "🙂é", "\r\n\r\n", "s"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿'ll<|fim_prefix|>s're're(<'S'Re㍿ A​'D", "tokens": 29, "pieces": ["㍿'", "ll", "<|", "fim", "_prefix", "|>", "s're", "'re", "(<'", "S'Re", "㍿", " A", "​'", "D"]} +{"text": " 'S('ſ……'D'll🙂'٣٤٥٦­🙂<|endoftext|>​t <|fim_prefix|>> #$%‍ßⅣ're\r
9३'D'Mİ́漢s", "tokens": 52, "pieces": [" '", "S", "('", "ſ", "…", "…", "'D'll", "🙂'", "٣٤٥", "٦", "­🙂<|", "endoftext", "|>​", "t", " ", "<|", "fim", "_prefix", "|>>", " ", "#$%‍", "ß", "Ⅳ", "'re", "\r", "
", "9३", "'D'M", "İ́漢s"]} +{"text": "👍🏽s​<|endoftext|>'S \nm!!å٣٤٥٦\"fi", "tokens": 27, "pieces": ["👍🏽", "s", "​<|", "endoftext", "|>'", "S", " \n", "m", "!!", "å", "٣٤٥", "٦", "\"<", "META", "_START", ">fi"]} +{"text": "́'Réd-𐞁\t㍿mdé'́
é३", "tokens": 21, "pieces": ["́'Re", "́d", "-𐞁", "\t", "㍿mdé", "'́", "
é", "३"]} +{"text": "ß​'D<'D\"🙂fi 'M…é‍ \n12345678𐞁字㍿ \n9ß😀🏽té0fi३ḍ̇'VE", "tokens": 47, "pieces": ["ß", "​'", "D", "<'", "D", "\"🙂", "fi", " ", " '", "M", "…é", "‍", " \n", "123", "456", "78", "𐞁", "字", "㍿", " \n", "9", "ß", "😀🏽", "té", "0", "fi", "३", "ḍ̇'VE"]} +{"text": "½-ع12345678\n漢é'T'S\u000b𐞁dåå🙂عİ\r 
'ſḍ̇.'VE
 ", "tokens": 36, "pieces": ["½", "-ع", "123", "456", "78", "\n", "漢é'T", "'S", "\u000b𐞁dåå", "🙂ع", "İ", "\r", " ", "
", "'ſḍ̇", ".'", "VE", "
 "]} +{"text": ".😀🏽'Sm're'ſ<|fim_prefix|>Aḍ̇.­.0'llⅣ<ß.<|endoftext|>漢𐞁😀🏽'res\u000b३‍", "tokens": 48, "pieces": [".😀🏽'", "Sm're", "'ſ", "<|", "fim", "_prefix", "|>", "Aḍ̇", ".­.", "0", "'ll", "Ⅳ", "<ß", ".<|", "endoftext", "|>", "漢𐞁", "😀🏽'", "res", "\u000b", "३", "‍"]} +{"text": "Z\u000b", "tokens": 2, "pieces": ["Z", "\u000b"]} +{"text": "s­\r\n'Mm…EOTa'VE😀🏽- \n…𐞁<|endoftext|>é'Tm㍿ 'Ret<|fim_prefix|>12345678'T३🙂!é'llEOT㋿s'VE\r\n\r\n​", "tokens": 66, "pieces": ["s", "­\r\n", "'Mm", "…EOT", "a'VE", "😀🏽-", " \n", "…𐞁", "<|", "endoftext", "|>", "é'T", "m", "㍿", " '", "Ret", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'T", "३", "🙂!", "é'll", "EOT", "㋿s'VE", "\r\n\r\n", "​"]} +{"text": "字 0,\u000bZé('VE's\"\r\n\r\nꟲ漢'ReDž\nfiع㍿12345678(㍿", "tokens": 34, "pieces": ["字", " ", "0", ",", "\u000bZé", "('", "VE's", "\"\r\n\r\n", "ꟲ漢", "'", "Re", "Dž", "\n", "fiع", "㍿", "123", "456", "78", "(㍿"]} +{"text": "'Ms-09ḍ̇DžⅣ d0'<ſß \n", "tokens": 22, "pieces": ["'Ms", "-", "09", "ḍ̇", "Dž", "Ⅳ", " ", " d", "0", "'<<", "META", "_START", ">ſß", " \n"]} +{"text": "'ſ𐞁\r're\r\n\r\n\n", "tokens": 15, "pieces": ["'ſ𐞁", "\r", "'", "re", "\r\n\r\n\n"]} +{"text": "'VE eéA'Sa😀🏽
Ⅳ-'M𐞁,  >'VEtd½\t'Mḍ̇‍\r\n\r\n !ꟲ \t३", "tokens": 46, "pieces": ["'VE", " e", "é", "A'S", "a", "😀🏽", "
", "Ⅳ", "-'", "M𐞁", ",", " ", " ", ">'", "VEtd", "½", "\t", "'Mḍ̇", "‍\r\n\r\n", " ", " !", "ꟲ", " ", "\t", "३"]} +{"text": "İ'S0
", "tokens": 4, "pieces": ["İ'S", "0", "
"]} +{"text": "<|fim_prefix|><|fim_prefix|>ß 's'Sm😀🏽fi'Tḍ̇\u000bs\"‍'ſ \nmEOT‍", "tokens": 40, "pieces": ["<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", "ß", " ", " '", "s'S", "m", "😀🏽", "fi'T", "ḍ̇", "\u000bs", "\"‍'", "ſ", " \n", "m", "EOT", "‍"]} +{"text": "'Reḍ̇'s!𐞁d'VEſ👍🏽㍿a('St'ſ㋿𐞁  <|fim_prefix|>\r\n½Dž!!字<|endoftext|>'re字12345678'Reaa aEOT", "tokens": 63, "pieces": ["'Reḍ̇'s", "!𐞁d'VE", "ſ", "👍🏽㍿", "a", "('", "St'ſ", "㋿𐞁", " ", " ", "<|", "fim", "_prefix", "|>\r\n", "½", "Dž", "!!", "字", "<|", "endoftext", "|>'", "re字", "123", "456", "78", "'Reaa", " ", " a", "EOT"]} +{"text": "'VE­s\"'M\r\n\r\nZ\r\n\r\n'VE'D9ꟲ", "tokens": 23, "pieces": ["'VE", "­s", "\"'", "M", "\r\n\r\n", "Z", "\r\n\r\n", "'VE'D", "9", "ꟲ", ""]} +{"text": ".d🙂", "tokens": 2, "pieces": [".d", "🙂"]} +{"text": "EOTEOTEOTß<|fim_prefix|>㋿ ́fi‍\nİ!! 
½ß", "tokens": 29, "pieces": ["EOTEOTEOTß", "<|", "fim", "_prefix", "|>㋿", " ", " ́fi", "‍\n", "İ", "!!", " ", "
", "½", "ß"]} +{"text": "'sdDžsſ ㍿ \n🙂\t!'T\u000bse.‍\n<|fim_prefix|>🙂ſ㍿ ­'D'S'Re٣٤٥٦Ⅳ'S", "tokens": 45, "pieces": ["'sd", "Džsſ", " ", " ㍿", " \n", "🙂", "\t", "!'", "T", "\u000bse", ".‍\n", "<|", "fim", "_prefix", "|>🙂", "ſ", "㍿", " ", " ­'", "D'S", "'Re", "٣٤٥", "٦Ⅳ", "'S"]} +{"text": "'ll😀🏽eEOT. \n \r'sعع́\u000b\rⅣé", "tokens": 21, "pieces": ["'ll", "😀🏽", "e", "EOT", ".", " \n \r", "'sعع́", "\u000b\r", "Ⅳ", "é"]} +{"text": "<|endoftext|>😀🏽\r\n!İ\t \n'T'#$%m ­😀🏽m9<mꟲå\t \ne\n㍿", "tokens": 46, "pieces": ["<|", "endoftext", "|>😀🏽\r\n", "!İ", "\t \n", "'T", "'#$%", "m", " ", " ­😀🏽", "m", "9", "<<", "EOT", ">mꟲå", "\t \n", "e", "\n", "㍿"]} +{"text": "'reع\tfi''MEOT㍿9\r\né'T‍eع\r\n\r\n😀🏽<|fim_prefix|>\r\n>​\n\"", "tokens": 37, "pieces": ["'reع", "\tfi", "''", "MEOT", "㍿", "9", "\r\n", "é'T", "‍eع", "\r\n\r\n", "😀🏽<|", "fim", "_prefix", "|><", "EOT", ">\r\n", ">​\n", "\""]} +{"text": "'ſ>\t", "tokens": 4, "pieces": ["'ſ", ">", "\t"]} +{"text": "'ſ\n𐞁'Re,-<|endoftext|>'VEſ…<|fim_prefix|>EOT​'Mſ'S0٣٤٥٦12345678👍🏽'llİ12345678!!\tſé'ſ‍漢'ſå\u000b're'T'D\t㍿½", "tokens": 72, "pieces": ["'ſ", "\n", "𐞁'Re", ",-<|", "endoftext", "|>'", "VEſ", "…", "<|", "fim", "_prefix", "|>", "EOT", "​'", "Mſ'S", "0٣٤", "٥٦1", "234", "567", "8", "👍🏽'", "ll", "İ", "123", "456", "78", "!!", "\tſé'ſ", "‍漢'ſ", "å", "\u000b", "'re'T", "'D", "\t", "㍿", "½"]} +{"text": "㋿𐞁é'VE­­-å‍'漢,EOT字A३<|fim_prefix|>​🙂🙂'VE''s…t​'re'St㍿ !!", "tokens": 51, "pieces": ["㋿𐞁é'VE", "­­-", "å", "‍'", "漢", ",EOT字", "A", "३", "<|", "fim", "_prefix", "|>​🙂🙂'", "VE", "''", "s", "…t", "​'", "re'S", "t", "㍿", " ", "!!"]} +{"text": "'ll字", "tokens": 2, "pieces": ["'ll字"]} +{"text": "\t㍿s㋿#$%12345678,e#$%'T'Re٣٤٥٦😀🏽👍🏽s
!", "tokens": 31, "pieces": ["\t", "㍿s", "㋿#$%", "123", "456", "78", ",e", "#$%'", "T'Re", "٣٤٥", "٦", "😀🏽👍🏽", "s", "
", "!"]} +{"text": "t'ſ‍\t🙂éd\r㋿😀🏽a\r\n\r\n9 'llA'll-(0‍٣٤٥٦­ 字#$%(å'ſ
Ⅳ'SⅣa", "tokens": 54, "pieces": ["t'ſ", "‍", "\t", "🙂éd", "\r", "㋿😀🏽", "a", "\r\n\r\n", "9", " ", "'ll", "A'll", "-(", "0", "‍", "٣٤٥", "٦", "­", " 字", "#$%<", "META", "_START", ">(", "å'ſ", "
", "Ⅳ", "'S", "Ⅳ", "a"]} +{"text": "'\r\n\r​ḍ̇­a. EOT'VE­\r\n\r\n'Re​㋿ 👍🏽<|fim_prefix|>>㋿>\r\n\r\n\r\n!Dž'VE 'llſ'ré\t'ſ'sḍ̇,", "tokens": 55, "pieces": ["'\r\n\r", "​ḍ̇", "­a", ".", " ", " EOT'VE", "­\r\n\r\n", "'Re", "​㋿", " ", " 👍🏽<|", "fim", "_prefix", "|>>㋿>\r\n\r\n\r\n", "!Dž", "'", "VE", " ", "'llſ're", "́", "\t", "'ſ's", "ḍ̇", ","]} +{"text": "👍🏽AéEOT\"Z\ré́\u000bAعaZ𐞁<|fim_prefix|>9d३", "tokens": 31, "pieces": ["👍🏽", "Aé", "EOT", "\"Z", "\r", "é́", "\u000bAعa", "Z𐞁", "<|", "fim", "_prefix", "|>", "9", "d", "३"]} +{"text": "9‍'T字'reⅣt'Re'Dꟲå0'reꟲé's.ع\"'ll字㍿>", "tokens": 33, "pieces": ["9", "‍'", "T字're", "Ⅳ", "t'Re", "'Dꟲå", "0", "'reꟲé's", ".ع", "\"'", "ll字", "㍿>"]} +{"text": "#$%㋿é", "tokens": 6, "pieces": ["#$%㋿", "é"]} +{"text": "é🙂‍'re
å'VE३s.…'T漢.!#$%12345678㋿٣٤٥٦m㋿#$%\r", "tokens": 46, "pieces": ["é", "🙂‍'", "re", "", "
å'VE", "३", "s", ".", "…", "'T漢", ".!#$%", "123", "456", "78", "㋿", "٣٤٥", "٦", "m", "㋿<", "EOT", ">#$%\r"]} +{"text": "👍🏽!!", "tokens": 4, "pieces": ["👍🏽!!"]} +{"text": "'sḍ̇Z\r\n\r\n½İ🙂\r\n\r\n'Tß", "tokens": 29, "pieces": ["ße", "0", "Aꟲ'D", "𐞁", "\u000b", "
", "Ⅳ", "'S", "<'", "Re", "-s", "<|", "fim", "_prefix", "|>🙂\r\n\r\n", "'Tß"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'S>éZſ<|endoftext|>\">३-ع#$%Z!'VE", "tokens": 21, "pieces": ["'S", ">é", "Zſ", "<|", "endoftext", "|>\">", "३", "-ع", "#$%", "Z", "!'", "VE"]} +{"text": "d'ſ<\r🙂<|endoftext|>\nǻ#$%ع́ \n\n½\tå'Re‍🙂ḍ̇'S,m👍🏽EOTfi​㍿\u000b😀🏽<|fim_prefix|>", "tokens": 52, "pieces": ["d'ſ", "<\r", "🙂<|", "endoftext", "|>\n", "ǻ", "#$%", "ع́", " \n\n", "½", "\tå'Re", "‍🙂", "ḍ̇'S", ",m", "👍🏽", "EOTfi", "​㍿", "\u000b", "😀🏽<|", "fim", "_prefix", "|>"]} +{"text": "e…é'VE9漢é!!<|fim_prefix|>", "tokens": 17, "pieces": ["e", "…é'VE", "9", "漢é", "!!<|", "fim", "_prefix", "|>"]} +{"text": "ḍ̇
\t३", "tokens": 6, "pieces": ["ḍ̇", "
", "\t", "३"]} +{"text": "\t 字é㍿İ<|endoftext|>'VEḍ̇Ⅳ", "tokens": 26, "pieces": ["\t", " 字é", "㍿<", "EOT", ">İ", "<|", "endoftext", "|>'", "VEḍ̇", "Ⅳ"]} +{"text": "Ⅳ​😀🏽İ\"'D🙂'T​9d\råe12345678\r\n\r\n \nع's", "tokens": 34, "pieces": ["Ⅳ", "​😀🏽", "İ", "\"'", "D", "🙂'", "T", "​<", "EOT", ">", "9", "d", "\r", "åe", "123", "456", "78", "\r\n\r\n \n", "ع's"]} +{"text": "<|endoftext|>👍🏽
́\"Z0عZꟲ'Re🙂…<٣٤٥٦'MDž㍿'D½<|fim_prefix|>(ⅣZd́ ٣٤٥٦㋿ß'Ma'\r\n'ReⅣ㍿\r\n\r\n.", "tokens": 68, "pieces": ["<|", "endoftext", "|>👍🏽", "
́", "\"Z", "0", "عZꟲ'Re", "🙂", "…", "<", "٣٤٥", "٦", "'MDž", "㍿'", "D", "½", "<|", "fim", "_prefix", "|>(", "Ⅳ", "Zd́", " ", "٣٤٥", "٦", "㋿ß'M", "a", "'\r\n", "'Re", "Ⅳ", "㍿\r\n\r\n", "."]} +{"text": "​9ſA㋿\r\n\r\n😀🏽Dž字'VE\tİ'reḍ̇#$%<|endoftext|>\"#$%‍\n", "tokens": 38, "pieces": ["​", "9", "ſ", "A", "㋿\r\n\r\n", "😀🏽", "Dž字'VE", "\tİ're", "ḍ̇", "#$%<|", "endoftext", "|><", "META", "_START", ">\"#$%‍\n"]} +{"text": "(<|fim_prefix|>", "tokens": 6, "pieces": ["(<|", "fim", "_prefix", "|>"]} +{"text": "ſ'ſ'T­Z. \n
<|endoftext|>!!३9m٣٤٥٦9 \r\n𐞁
ßſ's'३'ſ
'VEḍ̇", "tokens": 44, "pieces": ["ſ'ſ", "'T", "­Z", ".", " \n", "
", "<|", "endoftext", "|>!!", "३9", "m", "٣٤٥", "٦9", " \r\n", "𐞁", "
ßſ's", "'", "३", "'ſ", "
", "'VEḍ̇"]} +{"text": "'s३'S \n ſ!'D­🙂㋿\rḍ̇e#$% \n,ss '", "tokens": 24, "pieces": ["'s", "३", "'S", " \n", " ſ", "!'", "D", "­🙂㋿\r", "ḍ̇e", "#$%", " \n", ",ss", " '"]} +{"text": "'VE\u000b\t\t漢\u000b<|endoftext|>åꟲ😀🏽.'sm's!!'ſ́३", "tokens": 30, "pieces": ["'VE", "\u000b\t", "\t漢", "\u000b", "<|", "endoftext", "|>", "åꟲ", "😀🏽.'", "sm's", "!!'", "ſ́", "३"]} +{"text": ".\"<\t'S> ſ!!eå👍🏽'M'Re!😀🏽 \n!!…ß", "tokens": 38, "pieces": [".\"<", "\t", "'S", ">", " ſ", "!!", "eå", "👍🏽'", "M'Re", "!😀🏽", " \n", "!!", "…ß"]} +{"text": "!!'ſ'ſ\r\n->'ś'D0é0㍿<'ſ'VE漢\"é­", "tokens": 29, "pieces": ["!!<", "EOT", ">'", "ſ'ſ", "\r\n", "->'", "ś'D", "0", "é", "0", "㍿<'", "ſ'VE", "漢", "\"é", "­"]} +{"text": "12345678ſ'M\n'T㍿d\t", "tokens": 12, "pieces": ["123", "456", "78", "ſ'M", "\n", "'T", "㍿d", "\t"]} +{"text": " ſfi🙂0
½0\u000b
\u000b ‍\r\nⅣ
𐞁12345678'VE😀🏽e'S>fi\r", "tokens": 39, "pieces": [" ", " ſfi", "🙂", "0", "
", "½0", "\u000b
\u000b ", " ‍\r\n", "Ⅳ", "
𐞁", "123", "456", "78", "'VE", "😀🏽", "e'S", ">fi", "\r"]} +{"text": "'VEå(ḍ̇0#$%
's Aꟲ", "tokens": 18, "pieces": ["'VEå", "(ḍ̇", "0", "#$%", "
", "'s", " Aꟲ"]} +{"text": "Dž're
漢ß(\" (é(åfi㍿'VE", "tokens": 22, "pieces": ["Dž're", "
漢", "ß", "(\"", " ", "(é", "(åfi", "㍿'", "VE"]} +{"text": "<|endoftext|>m​㋿", "tokens": 12, "pieces": ["<|", "endoftext", "|>", "m", "​㋿"]} +{"text": "ḍ̇#$% \nt'e㍿\n \n'", "tokens": 21, "pieces": ["ḍ̇", "#$%", " \n", "t", "'", "e", "㍿\n", " \n", "'"]} +{"text": "­́ع0'''ſsḍ̇", "tokens": 41, "pieces": ["­́", "ع", "0", "'''", "ſsḍ̇"]} +{"text": "'T'D'SꟲaDž字字'Mmfim ́ß\"
eعm!…m!٣٤٥٦‍s\r\n\r\n0👍🏽A", "tokens": 57, "pieces": ["'T'D", "'Sꟲa", "Dž字字'M", "a", "Dž", "mfim", " <", "META", "_START", ">́ß", "\"", "
eعm", "!", "…m", "!", "٣٤٥", "٦", "‍s", "\r\n\r\n", "0", "👍🏽", "A"]} +{"text": "'Reḍ̇ß𐞁Ⅳ'ſ \n'De­ع0ḍ̇.'\r\nå\u000b ", "tokens": 28, "pieces": ["'Reḍ̇ß𐞁", "Ⅳ", "'ſ", " \n", "'De", "­ع", "0", "ḍ̇", ".'\r\n", "å", "\u000b "]} +{"text": "İ's\nⅣ12345678Džd9\r\n
's<|fim_prefix|>(Dž
'Re'VE'll ß'M‍\n''s漢é३'ſaéⅣḍ̇EOT", "tokens": 52, "pieces": ["İ's", "\n", "Ⅳ12", "345", "678", "Džd", "9", "\r\n", "
", "'s", "<|", "fim", "_prefix", "|>(", "Dž", "
", "'Re'VE", "'ll", " ß'M", "‍\n", "''", "s漢é", "३", "'ſaé", "Ⅳ", "ḍ̇", "EOT"]} +{"text": "amé'VEſZ𐞁'M🙂.e字", "tokens": 15, "pieces": ["amé'VE", "ſ", "Z𐞁'M", "🙂.", "e字"]} +{"text": "ꟲ0s.​EOTꟲ\r\n\r\n\r\nع字 0Z 🙂ſ<|fim_prefix|> \nfi👍🏽İ,A\nİt!å‍", "tokens": 41, "pieces": ["ꟲ", "0", "s", ".​", "EOTꟲ", "\r\n\r\n\r\n", "ع字", " ", "0", "Z", " ", "🙂ſ", "<|", "fim", "_prefix", "|>", " \n", "fi", "👍🏽", "İ", ",A", "\n", "İt", "!å", "‍"]} +{"text": "'T9A!'D𐞁's
\"٣٤٥٦\t٣٤٥٦'M'T<<'Så३'re­ .EOT­' eꟲm<|endoftext|> ,'Så", "tokens": 63, "pieces": ["å", " ", " '", "D're", "\r\n\r\n", "漢", "<|", "fim", "_prefix", "|>'", "s", "
", "\"", "٣٤٥", "٦", "\t", "٣٤٥", "٦", "'M'T", "<<'", "Så", "३", "'re", "­", " ", ".EOT", "­<", "EOT", ">'", " eꟲm", "<|", "endoftext", "|>", " ", ",'", "Så"]} +{"text": "\r🙂'Tß字#$%\r\n\r\n\t㍿\r\n\r\n.\u000bİ㋿
12345678'T", "tokens": 25, "pieces": ["\r", "🙂'", "Tß字", "#$%\r\n\r\n", "\t", "㍿\r\n\r\n", ".", "\u000bİ", "㋿", "
", "123", "456", "78", "'T"]} +{"text": "'M
 ß!ſ'Red('ll­\n'S!!Ⅳs#$%.'ss'VE­
'''Ḿa're'M'D", "tokens": 39, "pieces": ["'M", "
", " ß", "!ſ'Re", "d", "('", "ll", "­\n", "'S", "!!", "Ⅳ", "s", "#$%.'", "s", "s'VE", "­", "
", "'''", "M", "́a're", "'M'D"]} +{"text": "<|endoftext|><字fiŹ\n12345678㍿EOT#$%Džé👍🏽
\n'S…'D'M'ſ​ع\n<‍", "tokens": 42, "pieces": ["<|", "endoftext", "|><", "字fi", "Ź", "\n", "123", "456", "78", "㍿EOT", "#$%", "Džé", "👍🏽", "
\n", "'S", "…", "'D'M", "'ſ", "​ع", "\n", "<‍"]} +{"text": "<|endoftext|>ß\r\nİ#$%9ß'ſ'T½", "tokens": 18, "pieces": ["<|", "endoftext", "|>", "ß", "\r\n", "İ", "#$%", "9", "ß'ſ", "'T", "½"]} +{"text": "9,\tſé٣٤٥٦é'reſ'ſZDžsİDž'S-'s漢0!
're<|endoftext|>𐞁's
é٣٤٥٦㍿!!\r\n 's", "tokens": 57, "pieces": ["9", ",", "\tſé", "٣٤٥", "٦", "é're", "ſ'ſ", "ZDžs", "İDž'S", "-'", "s漢", "", "0", "!", "
", "'re", "<|", "endoftext", "|>", "𐞁's", "
é", "٣٤٥", "٦", "㍿!!\r\n", " ", "'s"]} +{"text": "½m½́🙂Ze \nt‍́ 
sꟲå‍'Reé", "tokens": 22, "pieces": ["½", "m", "½", "́", "🙂Ze", " \n", "t", "‍́", " ", "
sꟲå", "‍'", "Reé"]} +{"text": "12345678 \n'Re漢(.e字're‍#$%'VE!!aEOT😀🏽­'S🙂\"'VE\re漢\r\n\r\neZ", "tokens": 33, "pieces": ["123", "456", "78", " \n", "'Re漢", "(.", "e字're", "‍#$%'", "VE", "!!", "a", "EOT", "😀🏽­'", "S", "🙂\"'", "VE", "\r", "e漢", "\r\n\r\n", "e", "Z"]} +{"text": "'ll字,", "tokens": 3, "pieces": ["'ll字", ","]} +{"text": "s,'M\rſå", "tokens": 7, "pieces": ["s", ",'", "M", "\r", "ſå"]} +{"text": "s'\"'ſeع …<'Reꟲ'T­é\r\n\r\n", "tokens": 19, "pieces": ["s", "'\"'", "ſeع", " ", "…", "<'", "Reꟲ'T", "­é", "\r\n\r\n"]} +{"text": "😀🏽#$%\r\n…12345678'Re<\".漢m٣٤٥٦عå'ſe\u000bt12345678å", "tokens": 38, "pieces": ["😀🏽#$%\r\n", "…", "", "123", "456", "78", "'Re", "<\".", "漢m", "٣٤٥", "٦", "ع", "å'ſ", "e", "\u000bt", "123", "456", "78", "å"]} +{"text": "'VE'lla  \r\n\r\n'llt<", "tokens": 9, "pieces": ["'VE'll", "a", "  \r\n\r\n", "'llt", "<"]} +{"text": "éé­İ'S‍<", "tokens": 8, "pieces": ["éé", "­İ'S", "‍<"]} +{"text": "'ll'st漢EOT​Aeİ𐞁0éḍ̇EOT­", "tokens": 23, "pieces": ["'ll's", "t漢", "EOT", "​Ae", "İ𐞁", "0", "éḍ̇", "EOT", "­"]} +{"text": ">\n𐞁sⅣ𐞁#$%t'VE#$%<|endoftext|>-<|endoftext|>ſ­>'D­३d'VE\t👍🏽𐞁tꟲ\n12345678'D#$%ⅣZ'll㍿ 漢a'M", "tokens": 76, "pieces": [">\n", "𐞁s", "Ⅳ", "𐞁", "#$%", "t'VE", "#$%<|", "endoftext", "|>-<|", "endoftext", "|>", "ſ", "­>'", "D", "­", "३", "d'VE", "\t", "👍🏽", "𐞁tꟲ", "\n", "123", "456", "78", "'D", "#$%", "Ⅳ", "Z'll", "㍿", " 漢a'M"]} +{"text": "'Ḿ's \né! ,'T
's \"'D9ع'llDžEOT­A ½漢…", "tokens": 31, "pieces": ["'Ḿ's", " \n", "é", "!", " ,'", "T", "
", "'s", " ", "\"'", "D", "9", "ع'll", "DžEOT", "­A", " ", "½", "漢", "…"]} +{"text": "ḍ̇ſe\t \n߅'re're'Sfi'عİ­​​'ll🙂ḍ̇㍿'lle.><|endoftext|>(0,漢ḍ̇½e​\u000b", "tokens": 47, "pieces": ["ḍ̇ſe", "\t \n", "ß", "…", "'re're", "'Sfi", "'ع", "İ", "­​​'", "ll", "🙂ḍ̇", "㍿'", "lle", ".><|", "endoftext", "|>(", "0", ",漢ḍ̇", "½", "e", "​", "\u000b"]} +{"text": "३ßZ字0́'Re'T'ſſ12345678३>'remé\r\n­…fit \n!İ….\n'ſ#$% ", "tokens": 35, "pieces": ["३", "ß", "Z字", "0", "́'Re", "'T'ſ", "ſ", "123", "456", "78३", ">'", "remé", "\r\n", "­", "…fit", " \n", "!İ", "…", ".\n", "'ſ", "#$%", " "]} +{"text": "'VE字<|endoftext|><|endoftext|>\nAع㋿", "tokens": 21, "pieces": ["'VE字", "<|", "endoftext", "|><|", "endoftext", "|>\n", "Aع", "㋿"]} +{"text": "!é \nⅣa!­\"!m\r\n#$%'字𐞁'ع㍿<|endoftext|>'se( ꟲḍ̇Džd漢!. 'S!!d", "tokens": 53, "pieces": ["!é", " \n", "Ⅳ", "a", "!­\"!", "m", "\r\n", "#$%'", "字𐞁", "'ع", "㍿<|", "endoftext", "|>'", "se", "(", " ꟲḍ̇", "Džd漢", "!.", " ", " '", "S", "!!", "d", ""]} +{"text": "!'SDžİⅣ\r\n\r\n\r!㋿-é\ntéé🙂İ​\r\nſ👍🏽!‍​㋿
३e‍ a", "tokens": 44, "pieces": ["!'", "SDžİ", "Ⅳ", "\r\n\r\n\r", "!㋿<", "META", "_START", ">-", "é", "\n", "téé", "🙂İ", "​\r\n", "ſ", "👍🏽!‍​㋿", "
", "३", "e", "‍", " a"]} +{"text": "\u000b'S", "tokens": 2, "pieces": ["\u000b", "'S"]} +{"text": "<\"é ſ ­ꟲ Ⅳ🙂'ſ३ع٣٤٥٦aé𐞁'M
'T<|endoftext|>", "tokens": 38, "pieces": ["<\"", "é", " ſ", " ­", "ꟲ", " ", "Ⅳ", "🙂'", "ſ", "३", "ع", "٣٤٥", "٦", "aé𐞁'M", "
", "'T", "<|", "endoftext", "|>"]} +{"text": "!.eعt\r\n\r\n\"<|fim_prefix|>fi EOT\r\n\r\nꟲ
'Reḍ̇ sfi,漢mtm\t", "tokens": 39, "pieces": ["!.<", "META", "_START", ">eعt", "\r\n\r\n", "\"<", "EOT", "><|", "fim", "_prefix", "|>", "fi", " EOT", "\r\n\r\n", "ꟲ", "
", "'Reḍ̇", " sfi", ",漢mtm", "\t"]} +{"text": "ßé're912345678t", "tokens": 7, "pieces": ["ßé're", "912", "345", "678", "t"]} +{"text": "İ\"!!", "tokens": 3, "pieces": ["İ", "\"!!"]} +{"text": " .\u000bå🙂t😀🏽<|fim_prefix|>​​,0'(é'ſm'Re", "tokens": 26, "pieces": [" ", ".", "\u000bå", "🙂t", "😀🏽<|", "fim", "_prefix", "|>​​,", "0", "'(", "é'ſ", "m'Re"]} +{"text": "Ⅳ'ſİ😀🏽e0𐞁 😀🏽#$%t'S>३9㍿٣٤٥٦'㋿'De'Re'!<|endoftext|>e½​‍ \r\n\r\n'…​ß ' ", "tokens": 63, "pieces": ["Ⅳ", "'ſ", "İ", "😀🏽", "e", "0", "𐞁", " 😀🏽#$%", "t'S", ">", "३9", "㍿", "٣٤٥", "٦", "'㋿'", "De'Re", "'!<|", "endoftext", "|>", "e", "½", "​‍", " \r\n\r\n", "'<", "EOT", ">", "…", "​ß", " '", " "]} +{"text": "ſ", "tokens": 1, "pieces": ["ſ"]} +{"text": "​३ ('reḍ̇漢ḍ̇m-'𐞁​'D-ꟲ\r\n\r\nEOT字Ⅳ㋿𐞁 \n> \u000b", "tokens": 45, "pieces": ["​", "३", " ", " ('", "reḍ̇漢ḍ̇m", "-'", "𐞁", "​'", "D", "-ꟲ", "\r\n\r\n", "EOT字", "Ⅳ", "㋿𐞁", " \n", "><", "META", "_START", ">", " \u000b"]} +{"text": "­Ⅳ-ſ! 'sß", "tokens": 12, "pieces": ["­", "Ⅳ", "-", "ſ", "!", " '", "sß"]} +{"text": "漢­­9🙂🙂𐞁😀🏽0 d'TZ", "tokens": 17, "pieces": ["漢", "­­", "9", "🙂🙂", "𐞁", "😀🏽", "0", " ", " d'T", "Z"]} +{"text": "İé0<|endoftext|>İ'll🙂́fi ​s-efi<|endoftext|>'ll­́३,", "tokens": 39, "pieces": ["İé", "0", "<|", "endoftext", "|>", "İ'll", "🙂́", "fi", " ", "​s", "-efi", "<|", "endoftext", "|>'", "ll", "­́", "३", ","]} +{"text": "12345678're 漢́é\n0mßḍ̇!!Ze12345678're'VE\tm'M\r\né'll😀🏽Ⅳt½\t👍🏽ḍ̇fi'T<|endoftext|>­㍿Z'Re", "tokens": 53, "pieces": ["\u000b́'re", "-.‍", "eß", "Ze", "123", "456", "78", "'re'VE", "\tm'M", "\r\n", "é'll", "😀🏽", "Ⅳ", "t", "½", "\t", "👍🏽", "ḍ̇fi'T", "<|", "endoftext", "|>­㍿", "Z'Re"]} +{"text": "(ع👍🏽ſⅣ'e…t㋿fifi३'T🙂dAém'ss\r\nꟲ㍿9 \n!…Z\" 're٣٤٥٦'re\u000bt!!", "tokens": 52, "pieces": ["(ع", "👍🏽", "ſ", "Ⅳ", "'e", "…t", "㋿fi", "fi", "३", "'T", "🙂d", "Aém's", "s", "\r\n", "ꟲ", "㍿", "9", " \n", "!", "…Z", "\"", " '", "re", "٣٤٥", "٦", "'re", "\u000bt", "!!"]} +{"text": "𐞁漢å <|endoftext|>'VE㋿AZ\u000bm🙂İ\r\n\r\n'sع12345678å🙂\t#$%'-ß'Reع> d…é", "tokens": 47, "pieces": ["𐞁漢å", " ", "<|", "endoftext", "|>'", "VE", "㋿AZ", "\u000bm", "🙂İ", "\r\n\r\n", "'sع", "123", "456", "78", "å", "🙂", "\t", "#$%'-", "ß'Re", "ع", ">", " d", "…é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\n<|endoftext|>'VE½'re𐞁\u000b漢12345678eé<㍿漢Z'M'VE
'M👍🏽s'D ḍ̇-㍿<|fim_prefix|>", "tokens": 52, "pieces": ["\n", "<|", "endoftext", "|>'", "VE", "½", "'re𐞁", "\u000b漢", "123", "456", "78", "eé", "<㍿", "漢", "Z'M", "'VE", "
", "'M", "👍🏽", "s'D", " ḍ̇", "-㍿<|", "fim", "_prefix", "|>"]} +{"text": "12345678\r\n\r\nſ<½12345678ḍ̇'ſع<'ReA'D(\n漢‍ #$%​٣٤٥٦🙂­!! ​\r\n\r\n<<|endoftext|>>!!ⅣDžſſ!字", "tokens": 58, "pieces": ["123", "456", "78", "\r\n\r\n", "ſ", "<", "½12", "345", "678", "ḍ̇'ſ", "ع", "<'", "Re", "A'D", "(\n", "漢", "‍", " ", " #$%​", "٣٤٥", "٦", "🙂­!!", " ​\r\n\r\n", "<<|", "endoftext", "|>>!!", "Ⅳ", "Džſſ", "!字"]} +{"text": "
s𐞁tⅣ(٣٤٥٦åZ½​'D ḍ̇<‍ ‍'VEeعfi\r\ń½'S9'T\tſ漢​'re'Ms漢㍿'M", "tokens": 56, "pieces": ["
s𐞁t", "Ⅳ", "(", "٣٤٥", "٦", "å", "Z", "½", "​'", "D", " ḍ̇", "<<", "META", "_START", ">‍", " ‍'", "VEeعfi", "\r\n", "́", "½", "'S", "9", "'T", "\tſ漢", "​'", "re'M", "s漢", "㍿'", "M"]} +{"text": "fiꟲḍ̇", "tokens": 7, "pieces": ["fiꟲḍ̇"]} +{"text": "字😀🏽
İⅣA​é<|fim_prefix|>DžEOT😀🏽!<\u000b㋿\" ", "tokens": 32, "pieces": ["字", "😀🏽", "
İ", "Ⅳ", "A", "​é", "<|", "fim", "_prefix", "|>", "DžEOT", "😀🏽!<", "\u000b", "㋿\"", " "]} +{"text": "0㍿<ḍ̇İ'T!… \r\n \n'M٣٤٥٦İ́漢\rtꟲ…éİ٣٤٥٦é're 'M‍ß.", "tokens": 49, "pieces": ["0", "㍿<", "ḍ̇", "İ'T", "!", "… \r\n \n", "'M", "٣٤٥", "٦", "İ́漢", "\r", "tꟲ", "…é", "İ", "٣٤٥", "٦", "é're", " ", " '", "M", "‍ß", "."]} +{"text": "0'llt ", "tokens": 4, "pieces": ["0", "'llt", " "]} +{"text": "🙂'9ß(å\u000ba'Re\r\nḍ̇.\rꟲ㋿ع9'VEsZDž…'M\r\n\r\n\rİA'Mfia.३ḍ̇", "tokens": 51, "pieces": ["🙂<", "META", "_START", ">'<", "META", "_START", ">", "9", "ß", "(å", "\u000ba'Re", "\r\n", "ḍ̇", ".\r", "ꟲ", "㋿ع", "9", "'VEs", "ZDž", "…", "'M", "\r\n\r\n\r", "İA'M", "fia", ".", "३", "ḍ̇"]} +{"text": "   \n'VEt<|fim_prefix|>é㋿Džts,'lĺ'Re\r\n<,", "tokens": 27, "pieces": ["   \n", "'VEt", "<|", "fim", "_prefix", "|>", "é", "㋿Džts", ",'", "lĺ'Re", "\r\n", "<,"]} +{"text": "(- 'Té𐞁ǻ…EOTعꟲ'Re\u000b,٣٤٥٦㍿s\u000b", "tokens": 35, "pieces": ["(-", " ", " '", "Té𐞁ǻ", "…EOTعꟲ", "'", "Re", "\u000b", ",", "٣٤٥", "٦", "㍿s", "\u000b"]} +{"text": "\"Ⅳ-𐞁-e!'llåİ 漢Zfi㋿", "tokens": 21, "pieces": ["\"", "Ⅳ", "-𐞁", "-e", "!'", "llå", "İ", " ", " 漢Zfi", "㋿"]} +{"text": "d<|endoftext|>'TⅣ\r😀🏽ſⅣd‍!㋿", "tokens": 24, "pieces": ["d", "<|", "endoftext", "|>'", "T", "Ⅳ", "\r", "😀🏽", "ſ", "Ⅳ", "d", "‍!㋿"]} +{"text": "👍🏽#$% EOT<|endoftext|>,عſéſ\t\r½'sd…😀🏽é½<\t\n'VEé👍🏽", "tokens": 41, "pieces": ["👍🏽#$%", " ", " EOT", "<|", "endoftext", "|>,", "عſéſ", "\t\r", "½", "'sd", "…", "😀🏽", "é", "½", "<", "\t\n", "'VEé", "👍🏽"]} +{"text": "(åع🙂ع🙂Dž­'S!'s're­㋿'ſfi😀🏽½'D9", "tokens": 27, "pieces": ["(åع", "🙂ع", "🙂Dž", "­'", "S", "!'", "s're", "­㋿'", "ſfi", "😀🏽", "½", "'D", "9"]} +{"text": "\t٣٤٥٦'T३\rع…𐞁fi٣٤٥٦🙂'll'll
", "tokens": 25, "pieces": ["\t", "٣٤٥", "٦", "'T", "३", "\r", "ع", "…𐞁fi", "٣٤٥", "٦", "🙂'", "ll'll", "
"]} +{"text": "'­\"́́'ll're🙂👍🏽 字

…0ꟲ'llⅣ\" \n३ ,\u000b'Re \n", "tokens": 36, "pieces": ["'­\"́́'", "ll're", "🙂👍🏽", " 字", "
", "", "
", "…", "0", "ꟲ'll", "Ⅳ", "\"", " \n", "३", " ", ",", "\u000b", "'Re", " \n"]} +{"text": "
Z<|fim_prefix|>!!\n‍é\u000b㋿ \n'Re  0d(d㍿Z'lls12345678 👍🏽'T ​́> >👍🏽 ", "tokens": 58, "pieces": ["
Z", "<|", "fim", "_prefix", "|>!!\n", "‍<", "META", "_START", ">é", "\u000b", "㋿<", "EOT", ">", " \n", "'Re", " ", " ", "0", "d", "(d", "㍿Z'll", "s", "123", "456", "78", " ", "👍🏽'", "T", " ", " ​́>", " >👍🏽", " "]} +{"text": "'S漢​'Sİḍ̇‍\r\nA", "tokens": 15, "pieces": ["'S漢", "​<", "META", "_START", ">'", "Sİḍ̇", "‍\r\n", "A"]} +{"text": "㍿ <<|fim_prefix|>å\r\n<|fim_prefix|>㍿'T\r\nſ!", "tokens": 27, "pieces": ["㍿", " ", " <<|", "fim", "_prefix", "|>", "å", "\r\n", "<|", "fim", "_prefix", "|>㍿'", "T", "\r\n", "ſ", "!"]} +{"text": ".a
'll'llm<|endoftext|>Z\r\"-'ſ😀🏽>字'sⅣ<|endoftext|>d\tİ­", "tokens": 36, "pieces": [".a", "
", "'ll'll", "m", "<|", "endoftext", "|>", "Z", "\r", "\"-'", "ſ", "😀🏽>", "字's", "Ⅳ", "<|", "endoftext", "|>", "d", "\tİ", "­"]} +{"text": "😀🏽'Re🙂'sعm'VE🙂's('re㋿daa'Dß🙂\u000bİ'll!! åع'så'St12345678\r\n\r\n\neé", "tokens": 43, "pieces": ["😀🏽'", "Re", "🙂'", "sعm'VE", "🙂'", "s", "('", "re", "㋿daa'D", "ß", "🙂", "\u000bİ'll", "!!", " åع's", "å'S", "t", "123", "456", "78", "\r\n\r\n\n", "eé"]} +{"text": "'D're\r\n123456789㍿t‍.​<|endoftext|>!ß㍿\r\nDž<|fim_prefix|>㋿ḍ̇́'M…", "tokens": 46, "pieces": ["'D're", "\r\n", "123", "456", "789", "㍿t", "‍.​<|", "endoftext", "|>!", "ß", "㍿<", "EOT", ">\r\n", "Dž", "<|", "fim", "_prefix", "|>㋿", "ḍ̇́'M", "…"]} +{"text": "t́e9 \t'VE漢ꟲ12345678 Džİ'ZZ\rع𐞁é.😀🏽İ​\u000b­(0'\rZ…m!!EOT ́😀🏽", "tokens": 53, "pieces": ["t́e", "9", " ", "\t", "'VE漢ꟲ", "123", "456", "78", " Džİ", "'ZZ", "\r", "ع𐞁é", ".😀🏽", "İ", "​", "\u000b", "­(", "0", "'\r", "Z", "…m", "!!", "EOT", " ́", "😀🏽"]} +{"text": "EOT٣٤٥٦!!<|fim_prefix|>fifiaß́\n'VE𐞁\n­🙂 \n(t㋿'D\n\r\né", "tokens": 38, "pieces": ["EOT", "٣٤٥", "٦", "!!<|", "fim", "_prefix", "|>", "fifiaß́", "\n", "'VE𐞁", "\n", "­🙂", " \n", "(t", "㋿'", "D", "\n\r\n", "é"]} +{"text": "! \n…#$%\r\nſ😀🏽३\"́!…😀🏽́é‍#$%é.ém'll'S", "tokens": 37, "pieces": ["!", " \n", "…", "#$%<", "META", "_START", ">\r\n", "ſ", "😀🏽", "३", "\"́", "!", "…", "😀🏽́", "é", "‍#$%", "é", ".ém'll", "'S"]} +{"text": "İ", "tokens": 1, "pieces": ["İ"]} +{"text": "'Md३🙂'llİ🙂\"! ‍Dž<½\r\n\r\n'ſ­s漢!! ع 12345678\"½\n👍🏽\r\n३½.𐞁३", "tokens": 45, "pieces": ["'Md", "३", "🙂'", "ll", "İ", "🙂<", "META", "_START", ">\"!", " ", "‍Dž", "<", "½", "\r\n\r\n", "'ſ", "­s漢", "!!", " ع", " ", "123", "456", "78", "\"", "½", "\n", "👍🏽\r\n", "३½", ".𐞁", "३"]} +{"text": "\u000b(-<|endoftext|>.३ 👍🏽#$%'VE'res\r\n", "tokens": 20, "pieces": ["\u000b", "(-<|", "endoftext", "|>.", "३", " ", "👍🏽#$%'", "VE're", "s", "\r\n"]} +{"text": "🙂 \n😀🏽<|endoftext|>​('🙂mt'sm\"ع́é­m'S9!㍿𐞁é'ßAm'ſ'S½'ſ
", "tokens": 53, "pieces": ["🙂", " \n", "😀🏽<|", "endoftext", "|>​('🙂", "m", "t's", "m", "\"ع́é", "­m'S", "9", "!㍿", "𐞁é", "'ß", "Am'ſ", "'S", "½", "'ſ", "
"]} +{"text": "-,漢<|endoftext|>漢\r\n𐞁<|fim_prefix|>𐞁𐞁<|endoftext|>​ع'D㋿'ReⅣ-\"'ſ'Z'll​'Tå  ​Dž", "tokens": 69, "pieces": ["<|", "endoftext", "|>", "漢", "\r\n", "𐞁", "<|", "fim", "_prefix", "|>", "𐞁𐞁", "<|", "endoftext", "|>​", "ع'D", "㋿'", "Re", "Ⅳ", "-\"'", "ſ", "'Z'll", "​'", "Tå", "  ", " ​", "Dž", ""]} +{"text": "ع漢EOT३<|fim_prefix|>ع'llİ  é<|endoftext|>sß!!🙂㋿- ꟲå0ع'Dfi३>٣٤٥٦ 'S\r\n\r\n\rfi", "tokens": 54, "pieces": ["ع漢", "EOT", "३", "<|", "fim", "_prefix", "|>", "ع'll", "İ", " ", " é", "<|", "endoftext", "|>", "sß", "!!🙂㋿-", " ꟲå", "0", "ع'D", "fi", "३", ">", "٣٤٥", "٦", " ", "'S", "\r\n\r\n\r", "fi"]} +{"text": "'ſEOT\u000b​t́,漢字,ḍ̇🙂d'T'llİḍ̇‍'VE😀🏽12345678s漢٣٤٥٦漢İ  ", "tokens": 42, "pieces": ["'ſ", "EOT", "\u000b", "​t́", ",漢字", ",ḍ̇", "🙂d'T", "'ll", "İḍ̇", "‍'", "VE", "😀🏽", "123", "456", "78", "s漢", "٣٤٥", "٦", "漢", "İ", "  "]} +{"text": "Z'Så(ḍ̇'VEe‍", "tokens": 12, "pieces": ["Z'S", "å", "(ḍ̇'VE", "e", "‍"]} +{"text": "sßꟲsa\u000baſtع㍿́ع३漢\r\n\tEOTßſé\"­\u000bع'S ́ß", "tokens": 37, "pieces": ["sßꟲsa", "\u000baſtع", "㍿́ع", "३", "漢", "\r\n", "\tEOTß", "ſé", "\"­", "\u000bع'S", " ́ß"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "Džd३\"İ12345678'll字'llfi\r\ńé३Dž'VE'VE.a<'lldt", "tokens": 31, "pieces": ["Džd", "३", "\"İ", "123", "456", "78", "'ll字'll", "fi", "\r\n", "́é", "३", "Dž'VE", "'VE", ".a", "<'", "lldt"]} +{"text": "'S <|endoftext|>é'S𐞁<|endoftext|>​ſ…­ ''re", "tokens": 31, "pieces": ["'S", " ", " <|", "endoftext", "|>", "é'S", "𐞁", "<|", "endoftext", "|>​", "ſ", "…", "­", " ", "''", "re"]} +{"text": "'s\t'ſZ \"​d👍🏽>!><İſ𐞁'İ<|fim_prefix|>'D- 🙂.'ll'S𐞁sEOT😀🏽dEOT'M Z", "tokens": 52, "pieces": ["'s", "\t", "'ſ", "Z", " ", "\"​", "d", "👍🏽>!><", "İſ𐞁", "'İ", "<|", "fim", "_prefix", "|>'", "D", "-", " ", "🙂.'", "ll'S", "𐞁s", "EOT", "😀🏽", "d", "EOT'M", " Z"]} +{"text": " \u000be,ſs😀🏽字\"ع<|fim_prefix|>'re… \ns\r\r\n ", "tokens": 28, "pieces": [" ", "\u000be", ",ſ", "s", "😀🏽", "字", "\"ع", "<|", "fim", "_prefix", "|>'", "re", "… \n", "s", "\r\r\n", " "]} +{"text": " \t<<|fim_prefix|> \n>㋿ḍ̇½t🙂'T👍🏽\t<|endoftext|>\r\n\r\n 
🙂", "tokens": 35, "pieces": [" ", "\t", "<<|", "fim", "_prefix", "|>", " \n", ">㋿", "ḍ̇", "½", "t", "🙂'", "T", "👍🏽", "\t", "<|", "endoftext", "|>\r\n\r\n", " ", "
", "🙂"]} +{"text": " 0\t's'M<,👍🏽'ſ\t's…👍🏽0", "tokens": 21, "pieces": [" ", " ", "0", "\t", "'s'M", "<,👍🏽'", "ſ", "\t", "'s", "…", "👍🏽", "0"]} +{"text": "ḍ̇\nİ12345678३㋿", "tokens": 12, "pieces": ["ḍ̇", "\n", "İ", "123", "456", "78३", "㋿"]} +{"text": "'M 9<|endoftext|>fi('𐞁a<𐞁😀🏽½½a-'VE\tm\r é ½𐞁ⅣDžéd<|fim_prefix|>…\t'sZİع
Ⅳ", "tokens": 68, "pieces": ["'M", " ", "9", "<|", "endoftext", "|>", "fi", "('", "𐞁a", "<𐞁", "😀🏽", "½½", "a", "-'", "VE", "\tm", "\r", " ", " é", " ", "½", "𐞁", "Ⅳ", "Džéd", "<|", "fim", "_prefix", "|>", "…", "\t", "'s", "Z", "İع", "
", "Ⅳ"]} +{"text": "
m \n!!ſⅣ‍ß𐞁's#$%ع-d'́\t!!<|endoftext|>aDžfi㍿😀🏽#$%#$% ḍ̇‍,", "tokens": 49, "pieces": ["
m", " \n", "!!", "ſ", "Ⅳ", "‍ß𐞁's", "#$%", "ع", "-d", "'́", "\t", "!!<|", "endoftext", "|>", "a", "Džfi", "㍿😀🏽#$%#$%", " ", " ḍ̇", "‍,"]} +{"text": "́a- ३ḍ̇'reꟲ㋿> 'll字ع😀🏽'M\t-'M\r", "tokens": 33, "pieces": ["́a", "-<", "EOT", ">", " ", "३", "ḍ̇'re", "ꟲ", "㋿>", " '", "ll字ع", "😀🏽'", "M", "\t", "-'", "M", "\r"]} +{"text": "'S
('S🙂Z<'VE tm", "tokens": 10, "pieces": ["'S", "
", "('", "S", "🙂Z", "<'", "VE", " ", " tm"]} +{"text": "-0! \u000b'M0EOT(t'Re e𐞁>dEOT\"​eꟲmm٣٤٥٦!!ḍ̇٣٤٥٦👍🏽12345678", "tokens": 48, "pieces": ["-", "0", "!", " ", "\u000b", "'M", "0", "EOT", "(t'Re", " e", "𐞁", ">d", "EOT", "\"​", "eꟲmm", "٣٤٥", "٦", "!!", "ḍ̇", "٣٤٥", "٦", "👍🏽", "123", "456", "78"]} +{"text": "٣٤٥٦́a9><A​\t㍿A!!'<fi>é,0㋿漢Ⅳ'Re‍å字m🙂 'VE", "tokens": 39, "pieces": ["٣٤٥", "٦", "́a", "9", "><", "A", "​", "\t", "㍿A", "!!'<", "fi", ">é", ",", "0", "㋿漢", "Ⅳ", "'Re", "‍å字m", "🙂", " ", " '", "VE"]} +{"text": "ع", "tokens": 1, "pieces": ["ع"]} +{"text": " \n0㍿𐞁Dž㋿é字'Re\tḍ̇\r\nꟲ\r 'S'D½", "tokens": 37, "pieces": [" \n", "0", "㍿𐞁", "Dž", "㋿é字'Re", "\tḍ̇", "\r\n", "ꟲ", "\r", " ", "'S'D", "½", ""]} +{"text": "å'S dfi㍿fi '0Z-'re३
A'ſ<|endoftext|>Ⅳeå…(ſ\r\nm\r\n\r\n<|endoftext|>Dž漢عḍ̇\r\n\r\n-", "tokens": 64, "pieces": ["å'S", "", " ", " dfi", "㍿fi", " ", "'", "0", "Z", "-<", "EOT", ">'", "re", "३", "
A", "'", "ſ", "<|", "endoftext", "|>", "Ⅳ", "eå", "…", "(ſ", "\r\n", "m", "\r\n\r\n", "<|", "endoftext", "|>", "Dž漢عḍ̇", "\r\n\r\n", "-"]} +{"text": "<'M'M!!(ꟲ½12345678!İ🙂'VEDž३#$%\"㍿'D 'Re>må'D\t
漢<|endoftext|><|endoftext|>'s<|fim_prefix|>'T", "tokens": 57, "pieces": ["<'", "M'M", "!!(", "ꟲ", "½12", "345", "678", "!İ", "🙂'", "VEDž", "३", "#$%\"㍿'", "D", " '", "Re", ">må'D", "\t", "
漢", "<|", "endoftext", "|><|", "endoftext", "|>'", "s", "<|", "fim", "_prefix", "|>'", "T"]} +{"text": "Ⅳ'M.'Tḍ̇­-'ſt㍿㋿é­ḍ̇0'D9#$%<|fim_prefix|>\r\n\r\nEOT'Re 'Re😀🏽,dꟲ𐞁#$%é", "tokens": 58, "pieces": ["Ⅳ", "'", "M", ".'", "Tḍ̇", "­-'", "ſt", "㍿㋿", "é", "­ḍ̇", "0", "'D", "9", "#$%<|", "fim", "_prefix", "|>\r\n\r\n", "EOT'Re", " '", "Re", "😀🏽,", "dꟲ𐞁", "#$%", "é"]} +{"text": "'Mİ åA\r\n㋿'ſm \n\t>👍🏽!!!ḍ̇İ9𐞁12345678​EOT 字漢fi'ſ9३½ !!\nꟲ", "tokens": 52, "pieces": ["'Mİ", " ", " å", "A", "\r\n", "㋿'", "ſm", " \n", "\t", ">👍🏽!!!", "ḍ̇", "İ", "9", "𐞁", "123", "456", "78", "​EOT", " 字漢fi'ſ", "9३½", " ", " !!\n", "ꟲ"]} +{"text": "'Re - 𐞁å'ſꟲeå\r\n's 🙂­ع🙂ع\r\n\r\nꟲ#$%
>ḍ̇é", "tokens": 37, "pieces": ["'Re", " ", "-", " 𐞁å'ſ", "ꟲeå", "\r\n", "'s", " ", "🙂­", "ع", "🙂ع", "\r\n\r\n", "ꟲ", "#$%", "
", ">ḍ̇é"]} +{"text": "­(- sådA😀🏽éDž#$%<|endoftext|>字", "tokens": 23, "pieces": ["­(-", " såd", "A", "😀🏽", "é", "Dž", "#$%<|", "endoftext", "|>", "字"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n\r\n-'Sİe \n'MⅣ́👍🏽‍字9🙂漢fi're­a12345678字'M<|endoftext|>\r\n\r\n漢 …,\"Džß", "tokens": 52, "pieces": ["\r\n\r\n", "-'", "Sİe", " \n", "'M", "", "Ⅳ", "́", "👍🏽‍", "字", "9", "🙂<", "EOT", ">漢fi're", "­a", "123", "456", "78", "字'M", "<|", "endoftext", "|>\r\n\r\n", "漢", " ", "…", ",\"", "Džß"]} +{"text": "İs㋿‍fí👍🏽9‍>", "tokens": 14, "pieces": ["İs", "㋿‍", "fí", "👍🏽", "9", "‍>"]} +{"text": "ḍ̇㋿é'ſ'ſ'D12345678 ß‍!", "tokens": 23, "pieces": ["ḍ̇", "㋿é", "'", "ſ'ſ", "'D", "123", "456", "78", " ", " ß", "‍!"]} +{"text": "Z.'ſ (
sd'ſ's\r\n\r\n\r<👍🏽'Re'D>'Re\n\r\n\r\nZ 'Res‍
\t.ع", "tokens": 31, "pieces": ["Z", ".'", "ſ", " (", "
sd'ſ", "'s", "\r\n\r\n\r", "<👍🏽'", "Re'D", ">'", "Re", "\n\r\n\r\n", "Z", " ", " '", "Res", "‍", "
", "\t", ".ع"]} +{"text": "\r\n\r\n\r\n\r\nAd\t'Sع ", "tokens": 7, "pieces": ["\r\n\r\n\r\n\r\n", "Ad", "\t", "'Sع", " "]} +{"text": "'ſꟲ​seaع#$%fi'll0'ſZꟲ<|fim_prefix|>9<ꟲ'D'D", "tokens": 32, "pieces": ["'ſꟲ", "​seaع", "#$%", "fi'll", "0", "'ſ", "Zꟲ", "<|", "fim", "_prefix", "|>", "9", "<ꟲ'D", "'D"]} +{"text": "éß'll'T㋿ \r\n\r\n\n <|fim_prefix|>‍́<|endoftext|>㍿ꟲ\"'字\t ­👍🏽\t'Re'D‍ \nß", "tokens": 53, "pieces": ["éß'll", "'T", "㋿", " \r\n\r\n\n", " ", "<|", "fim", "_prefix", "|>‍́<|", "endoftext", "|>㍿", "ꟲ", "\"'<", "EOT", ">字", "\t", " ­👍🏽", "\t", "'Re", "'", "D", "‍", " \n", "ß"]} +{"text": "0ſ9Aß\rꟲéⅣ\r㋿\u000bß'Reé\t­<|endoftext|>9\r\n 's'­'M", "tokens": 44, "pieces": ["0", "ſ", "9", "Aß", "\r", "ꟲé", "<", "EOT", ">", "Ⅳ", "\r", "㋿", "\u000bß'Re", "é", "\t", "­<|", "endoftext", "|>", "9", "\r\n", " ", "'s", "'­'", "M"]} +{"text": "12345678​\r\n漢 \n'Reḍ̇9EOT('\r\n\r\nA'Re.ḍ̇   ſ9'D𐞁字\"e🙂12345678ꟲ'reſe३㋿!!<|fim_prefix|>,å", "tokens": 60, "pieces": ["123", "456", "78", "​\r\n", "漢", " \n", "'Reḍ̇", "9", "EOT", "('\r\n\r\n", "A'Re", ".ḍ̇", "  ", " ſ", "", "9", "'D𐞁字", "\"e", "🙂", "123", "456", "78", "ꟲ're", "ſe", "३", "㋿!!<|", "fim", "_prefix", "|>,", "å"]} +{"text": "३<|endoftext|>!́𐞁३漢#$% 漢>
a9‍", "tokens": 26, "pieces": ["३", "<|", "endoftext", "|>!́", "𐞁", "३", "漢", "#$%", " ", " 漢", ">", "
a", "9", "‍"]} +{"text": "­'Re-…de!!9…'T'D\n\r\néⅣm🙂\ts\ta <'ll", "tokens": 24, "pieces": ["­'", "Re", "-", "…de", "!!", "9", "…", "'T'D", "\n\r\n", "é", "Ⅳ", "m", "🙂", "\ts", "\ta", " <'", "ll"]} +{"text": "\u000b́٣٤٥٦<|fim_prefix|>\"", "tokens": 38, "pieces": ["å", "", "\u000b́", "", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>\""]} +{"text": "ſ'ſ> d'ſ …İ'ſm's#$%. 9 ", "tokens": 21, "pieces": ["ſ'ſ", ">", " ", " d'ſ", " ", "…İ'ſ", "m's", "#$%.", " ", "9", " "]} +{"text": "('M🙂\r½'ſdEOTDžAfi <#$%", "tokens": 43, "pieces": ["('", "M", "🙂\r", "½", "'ſd", "EOTDžA", "", "fi", " ", "<#$%"]} +{"text": " \nd½!'ReDžİ'VEd٣٤٥٦🙂e \n\n'Tß 'S
dDž", "tokens": 29, "pieces": [" \n", "d", "½", "!'", "Re", "Džİ'VE", "d", "٣٤٥", "٦", "🙂e", " \n\n", "'Tß", " ", "'S", "
", "d", "Dž"]} +{"text": ">\u000b㍿#$%'fi\t'll!!é9>,­'VE'​'Re  'll(\u000b'S字👍🏽\n \n're٣٤٥٦ !!\r ع", "tokens": 47, "pieces": [">", "\u000b", "㍿#$%'", "fi", "\t", "'ll", "!!", "é", "9", ">,­'", "VE", "'​'", "Re", "  ", " '", "ll", "(", "\u000b", "'S", "字", "👍🏽\n", " \n", "'re", "٣٤٥", "٦", " ", "!!\r", " ع"]} +{"text": "­३(!", "tokens": 3, "pieces": ["­", "३", "(!"]} +{"text": " t ‍Ⅳa'D'reåd㋿İdZ…½Ⅳ\nعꟲ\td!!", "tokens": 33, "pieces": [" t", " ", "‍", "Ⅳ", "a'D", "'reåd", "㋿İd", "Z", "…", "½", "", "Ⅳ", "\n", "عꟲ", "\td", "!!"]} +{"text": "…a's<'Mع😀🏽'M…9ꟲ'ſ\"A\ta\u000bḍ̇('T-12345678e \n 🙂<|endoftext|>́'D", "tokens": 50, "pieces": ["…a's", "<'", "Mع", "😀🏽'", "M", "…", "9", "ꟲ'ſ", "\"A", "\ta", "\u000bḍ̇", "(<", "META", "_START", ">'", "T", "-", "123", "456", "78", "e", " \n", " ", "🙂<|", "endoftext", "|>́'", "D"]} +{"text": "d½t'll㋿\r\n
>字字mع,12345678d'Sé३'VE!!><|endoftext|>EOT́å½12345678'A,A­½\r\n👍🏽é", "tokens": 59, "pieces": ["d", "½", "t'll", "㋿\r\n", "
", ">字字mع", ",", "123", "456", "78", "d'S", "é", "३", "'", "VE", "!!><|", "endoftext", "|>", "EOT́å", "½12", "345", "678", "'A", ",A", "­", "½", "\r\n", "👍🏽", "é", ""]} +{"text": "'ſDž…㍿#$%>'re,fi \ńß ( \n漢#$%.", "tokens": 24, "pieces": ["'ſ", "Dž", "…", "㍿#$%>'", "re", ",fi", " \n", "́ß", " ", " (", " \n", "漢", "#$%."]} +{"text": "e ३عm('s漢 \n \né漢m#$%Aé'Sع12345678e\"ſ३'ll😀🏽Z'Re>👍🏽३…­\r\n\r\n😀🏽", "tokens": 49, "pieces": ["e", "", " ", " ", "३", "عm", "('", "s漢", " \n \n", "é漢m", "#$%", "Aé'S", "ع", "123", "456", "78", "e", "\"ſ", "३", "'ll", "😀🏽", "Z'Re", ">👍🏽", "३", "…", "­\r\n\r\n", "😀🏽"]} +{"text": "\r\n\r\n'ſ\r\n\r\n!👍🏽\t \n㋿#$%.å\r\n're漢\"! ", "tokens": 27, "pieces": ["\r\n\r\n", "'ſ", "\r\n\r\n", "!👍🏽", "\t \n", "㋿#$%.", "å", "\r\n", "'re漢", "\"!", " "]} +{"text": "​fi漢.\"ع\nعa'lls'ſ", "tokens": 12, "pieces": ["​fi漢", ".\"", "ع", "\n", "عa'll", "s'ſ"]} +{"text": "\r'S'D a字fi­<9EOT½…>d,\r😀🏽٣٤٥٦😀🏽'M0ḿ𐞁med's𐞁​", "tokens": 44, "pieces": ["\r", "'S'D", " a字fi", "­<", "9", "EOT", "½", "…", ">d", ",\r", "😀🏽", "٣٤٥", "٦", "😀🏽'", "M", "0", "ḿ𐞁med's", "𐞁", "​"]} +{"text": "٣٤٥٦'VE\r\n\r\n \r\n\r\n\r\n\r\ń\r\n𐞁(́'ſ​३!!\u000b Z漢,ع ­'EOT́😀🏽Aſ­Dž <|fim_prefix|> >fi", "tokens": 54, "pieces": ["٣٤٥", "٦", "'VE", "\r\n\r\n \r\n\r\n\r\n\r\n", "́", "\r\n", "𐞁", "(́'ſ", "​", "३", "!!", "\u000b", " Z漢", ",ع", " ", "­'", "EOT́", "😀🏽", "Aſ", "­Dž", " <|", "fim", "_prefix", "|>", " ", " >", "fi"]} +{"text": "(å \n‍(-12345678'D٣٤٥٦é<… İ 🙂字,٣٤٥٦EOT'VE,ſ>字'll'll<|endoftext|>fi", "tokens": 48, "pieces": ["(å", " \n", "‍(-", "123", "456", "78", "'D", "٣٤٥", "٦", "é", "<", "…", " İ", " ", "🙂字", ",", "٣٤٥", "٦", "EOT'VE", ",ſ", ">", "字'll", "'ll", "<|", "endoftext", "|>", "fi"]} +{"text": "𐞁👍🏽.㋿('Sfi\nd\r\n<|fim_prefix|>…ꟲ字(​ḍ̇<|fim_prefix|>\u000b𐞁e漢", "tokens": 50, "pieces": ["𐞁", "👍🏽.㋿('", "Sfi", "\n", "d", "\r\n", "<|", "fim", "_prefix", "|><", "EOT", ">", "…ꟲ字", "(​", "ḍ̇", "<|", "fim", "_prefix", "|>", "\u000b𐞁e漢"]} +{"text": "\u000b漢😀🏽\t \r\n\r\n'T!½🙂A9,éfie🙂(\r\n\r\n\r\n!Ⅳ٣٤٥٦>a\u000b'­!!㍿", "tokens": 35, "pieces": ["\u000b漢", "😀🏽", "\t \r\n\r\n", "'T", "!", "½", "🙂A", "9", ",éfie", "🙂(\r\n\r\n\r\n", "!", "Ⅳ٣٤", "٥٦", ">a", "\u000b", "'­!!㍿"]} +{"text": "<|fim_prefix|>🙂, 😀🏽<|endoftext|> !!'llDž字㍿\u000b''M㋿'llⅣ#$%!! 漢 0", "tokens": 46, "pieces": ["<|", "fim", "_prefix", "|>🙂,", " ", " 😀🏽<|", "endoftext", "|>", " ", "!!'", "ll", "Dž字", "㍿", "\u000b", "''", "M", "㋿'", "ll", "Ⅳ", "#$%!!", " 漢", " ", "0"]} +{"text": "'T're#$%İ-!!Ź9'VE<|fim_prefix|>٣٤٥٦'ll,As­d'ſ", "tokens": 32, "pieces": ["'T're", "#$%", "İ", "-!!", "Ź", "9", "'VE", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "'", "ll", ",As", "­d'ſ"]} +{"text": "٣٤٥٦ß٣٤٥٦­d'Dž é >''ſm\t <|fim_prefix|>fi½\t'T \nſ d", "tokens": 36, "pieces": ["٣٤٥", "٦", "ß", "٣٤٥", "٦", "­d", "'Dž", " é", " ", ">''", "ſm", "\t", " ", "<|", "fim", "_prefix", "|>", "fi", "½", "\t", "'T", " \n", "ſ", " d"]} +{"text": "'Dع12345678­ſ㋿<|fim_prefix|>㍿. ‍½ꟲs. ꟲ'M'llß(
<|endoftext|>\ts!
漢", "tokens": 54, "pieces": ["'Dع", "123", "456", "78", "­ſ", "㋿<|", "fim", "_prefix", "|>㍿.", " ‍", "½", "ꟲs", ".", " ꟲ'M", "'llß", "(", "
", "<|", "endoftext", "|>", "\t", "<", "EOT", ">s", "!", "
漢"]} +{"text": "\tß ​𐞁<|endoftext|> ㋿\rⅣ>'s\"->,'D
​
A\n Ⅳ!!s\r\n🙂's👍🏽(漢áع", "tokens": 54, "pieces": ["\tß", " ", "​𐞁", "<|", "endoftext", "|>", " ", "㋿<", "EOT", ">\r", "Ⅳ", ">'", "s", "\"->,'", "D", "
", "​", "
A", "\n", " ", "Ⅳ", "!!", "s", "\r\n", "🙂'", "s", "👍🏽(", "漢áع"]} +{"text": "㋿ e👍🏽s'Refi(👍🏽<|endoftext|>'S½><#$% \r\nA \r­🙂🙂
fi 0ꟲ'D 👍🏽(<|endoftext|>漢", "tokens": 58, "pieces": ["㋿", " ", "e", "👍🏽", "s'Re", "fi", "(👍🏽<|", "endoftext", "|>'", "S", "½", "><#$%", " \r\n", "A", " \r", "­🙂🙂", "
fi", " ", "0", "ꟲ'D", " ", "👍🏽(<|", "endoftext", "|>", "漢"]} +{"text": "🙂㋿½", "tokens": 5, "pieces": ["🙂㋿", "½"]} +{"text": " 😀🏽㍿ 𐞁'llß\rßé12345678aع'S-\r's\r'#$%!\"eß𐞁", "tokens": 37, "pieces": [" ", " 😀🏽㍿", " 𐞁'll", "ß", "\r", "ßé", "123", "456", "78", "aع'S", "-\r", "'s", "\r", "'#$%!\"", "eß𐞁"]} +{"text": "‍㍿'Mſ㍿½ ​9s漢ß漢afi
ſå12345678\r\n\r\n­ꟲ…­AA字td漢​eع\"<|fim_prefix|>'ll", "tokens": 54, "pieces": ["‍㍿'", "Mſ", "㍿", "½", " ", " ​", "9", "s漢ß漢afi", "
ſå", "123", "456", "78", "\r\n\r\n", "­ꟲ", "…", "­AA字td漢", "​eع", "\"<|", "fim", "_prefix", "|>'", "ll"]} +{"text": "12345678'D́ſ\"عfi'llA ㋿'Me㋿'ſZع\r\n\r\n<
\"! ­'D\n é😀🏽ع字sé", "tokens": 44, "pieces": ["123", "456", "78", "'D́ſ", "\"عfi'll", "A", " ㋿'", "Me", "㋿'", "ſ", "Zع", "\r\n\r\n", "<", "
", "\"!<", "EOT", ">", " ", " ­'", "D", "\n", " é", "😀🏽", "ع字sé"]} +{"text": "å…9㍿ḍ̇A<|endoftext|>­
İꟲmEOT'Mİét'VE'VE🙂\r字ſ", "tokens": 41, "pieces": ["å", "…", "9", "㍿ḍ̇", "A", "<|", "endoftext", "|>­", "
İꟲm", "EOT'M", "İét'VE", "'VE", "🙂\r", "字ſ"]} +{"text": "½fi­9½​m㋿😀🏽\r\nḍ̇#$%字𐞁'ſ#$%字Ae \nZ", "tokens": 33, "pieces": ["½", "fi", "­", "9½", "​m", "㋿😀🏽\r\n", "ḍ̇", "#$%", "字𐞁'ſ", "#$%", "字Ae", " \n", "Z"]} +{"text": "🙂sé<|fim_prefix|>'ll 'M漢!!!!Dž㍿٣٤٥٦ Zḍ̇<|endoftext|>'ſ <|endoftext|>A,İ#$%", "tokens": 47, "pieces": ["🙂sé", "<|", "fim", "_prefix", "|>'", "ll", " '", "M漢", "!!!!", "Dž", "㍿", "٣٤٥", "٦", " Zḍ̇", "<|", "endoftext", "|>'", "ſ", " <|", "endoftext", "|>", "A", ",İ", "#$%"]} +{"text": "('Re'ſ'Re", "tokens": 8, "pieces": ["('", "Re'ſ", "'", "Re"]} +{"text": " å\r\n\r\n<ßfifié<|fim_prefix|>'reA'DžⅣ́!!#$%'<|fim_prefix|>ع㍿", "tokens": 40, "pieces": [" å", "\r\n\r\n", "<ßfifié", "<|", "fim", "_prefix", "|>'", "re", "A", "'Dž", "Ⅳ", "́", "!!#$%'<|", "fim", "_prefix", "|><", "META", "_START", ">ع", "㍿"]} +{"text": "\r\n\r\n­
३\n字'ſ'T𐞁\u000b's\rsd<|endoftext|>fifiDž!'s ", "tokens": 33, "pieces": ["\r\n\r\n", "­", "
", "३", "\n", "字'ſ", "'T𐞁", "\u000b", "'s", "\r", "sd", "<|", "endoftext", "|>", "fifi", "Dž", "!'", "s", " "]} +{"text": "\u000b ,é\r\n\r\n\r 'S
😀🏽-'re𐞁\r\n\r\n\r\n\r\n", "tokens": 21, "pieces": ["\u000b ", " ,", "é", "\r\n\r\n\r", " ", " '", "S", "
", "😀🏽-'", "re𐞁", "\r\n\r\n\r\n\r\n"]} +{"text": "‍-ß
te👍🏽…d-́\r\n<|endoftext|>🙂!!\n<|endoftext|>fi\r\n\r\n​Ⅳ\n", "tokens": 39, "pieces": ["‍-", "ß", "
", "te", "👍🏽", "…d", "-́", "\r\n", "<|", "endoftext", "|>🙂!!\n", "<|", "endoftext", "|>", "fi", "\r\n\r\n", "​", "Ⅳ", "\n"]} +{"text": "\ntſ\t'll३e'ſ'T'',🙂0Dž'TsDž'llḍ̇'D३½👍🏽‍-", "tokens": 34, "pieces": ["\n", "tſ", "\t", "'ll", "३", "e'ſ", "'T", "'',🙂", "0", "Dž'T", "s", "Dž'll", "ḍ̇'D", "३", "", "½", "👍🏽‍-"]} +{"text": "m \t<㋿#$%́12345678🙂'S漢 \n's‍㍿'VE​ßs12345678t'SZ0,,9'ſ<|endoftext|>!12345678'ſ३🙂<|fim_prefix|><>A", "tokens": 62, "pieces": ["m", " ", "\t", "<㋿#$%́", "123", "456", "78", "🙂'", "S漢", " \n", "'s", "‍㍿'", "VE", "​ßs", "123", "456", "78", "t'S", "Z", "0", ",,", "9", "'ſ", "<|", "endoftext", "|>!", "123", "456", "78", "'ſ", "३", "🙂<|", "fim", "_prefix", "|><>", "A"]} +{"text": "aḍ̇ꟲ!!é(\n'Re ㋿!!>!'\u000bé\r\n\r\n'ſ \nA,é\néfi9 👍🏽,<👍🏽ꟲd'Tḍ̇", "tokens": 53, "pieces": ["aḍ̇ꟲ", "!!", "é", "(\n", "'Re", " ㋿!!>!<", "META", "_START", ">'", "\u000bé", "\r\n\r\n", "'ſ", " \n", "A", ",é", "\n", "éfi", "9", " ", " 👍🏽,<👍🏽", "ꟲ", "d'T", "ḍ̇"]} +{"text": "'s漢0'D 𐞁.>漢….\"\r½'M'sé<|endoftext|>>12345678ع12345678𐞁", "tokens": 38, "pieces": ["'s漢", "0", "'D", " 𐞁", ".>", "漢", "…", ".\"\r", "½", "'M's", "é", "<|", "endoftext", "|>>", "123", "456", "78", "ع", "123", "456", "78", "𐞁"]} +{"text": "!!'S\"'ſfi. '", "tokens": 12, "pieces": ["!!'", "S", "\"<", "META", "_START", ">'", "ſfi", ".", " ", "'"]} +{"text": "<ع's", "tokens": 3, "pieces": ["<ع's"]} +{"text": "㍿漢te
'D m \n३<|endoftext|>ⅣDž,.३ \nß字ſß㍿''>9ع.", "tokens": 45, "pieces": ["㍿漢t", "e", "
", "'D", " m", " \n", "३", "<|", "endoftext", "|>", "Ⅳ", "Dž", ",.", "३", " \n", "ß字ſß", "㍿''>", "9", "ع", "."]} +{"text": ">🙂😀🏽'Rees're( '12345678‍9​\"'D👍🏽
", "tokens": 34, "pieces": [">🙂😀🏽'", "Rees", "'", "re", "(", " '", "123", "456", "78", "‍", "9", "​\"'", "D", "👍🏽", "
"]} +{"text": " ३'Re\r\n\r9‍
ſ\r\nع\r\n\r\nå'T'Z𐞁'VE ­‍.0!é👍🏽fiEOT\nt🙂EOT><­ꟲ ", "tokens": 55, "pieces": [" ", " ", "३", "'Re", "\r\n\r", "9", "‍", "
", "ſ", "\r\n", "ع", "\r\n\r\n", "å'T", "'Z𐞁'VE", " ", " ­‍.", "0", "!", "é", "👍🏽", "fi", "EOT", "\n", "t", "🙂EOT", "><­", "ꟲ", " "]} +{"text": "Z'VEEOT<|endoftext|>㋿'s😀🏽9#$%😀🏽d 'Re((Džꟲ>< ع'\u000bé'VE㍿#$%'ReEOT​'T!! ३-'M", "tokens": 59, "pieces": ["Z'VE", "EOT", "<|", "endoftext", "|>㋿'", "s", "😀🏽", "9", "#$%😀🏽", "d", " ", "'Re", "((", "Džꟲ", "><", " ع", "'", "\u000bé'VE", "㍿#$%'", "Re", "EOT", "​'", "T", "!!", " ", "३", "-'", "M"]} +{"text": "mꟲ\"aé,ḍ̇'VE'Ma‍d👍🏽a…(字\rd'VE9é'Re<|endoftext|> mꟲ're‍½Džꟲ‍,'s 漢", "tokens": 56, "pieces": ["mꟲ", "\"aé", ",ḍ̇'VE", "'Ma", "‍d", "👍🏽", "a", "…", "(字", "\r", "d'VE", "9", "é'Re", "<|", "endoftext", "|>", " mꟲ're", "‍", "½", "Džꟲ", "‍,'", "s", " 漢"]} +{"text": "t<|fim_prefix|>9㍿½EOTå.Z'T‍漢Ⅳ'ſe𐞁", "tokens": 33, "pieces": ["t", "<|", "fim", "_prefix", "|>", "9", "㍿", "½", "EOTå", ".", "Z'T", "‍漢", "Ⅳ", "'ſe𐞁"]} +{"text": "\r\n\r\n0'S,!!aEOTéå're\r\n\r\n", "tokens": 14, "pieces": ["\r\n\r\n", "0", "'S", ",!!", "a", "EOTéå're", "\r\n\r\n"]} +{"text": "٣٤٥٦ \néåfiع🙂𐞁>'Reḍ̇'M'lleA'llA", "tokens": 29, "pieces": ["٣٤٥", "٦", " \n", "éåfi", "ع", "🙂𐞁", ">'", "Reḍ̇'M", "'lle", "A'll", "A"]} +{"text": " \n­\u000b­! ḍ̇​!Dž\"𐞁!!12345678­\u000bs٣٤٥٦é'så㍿\r\ńZ…", "tokens": 42, "pieces": [" \n", "­", "\u000b", "­!", " ḍ̇", "​!", "Dž", "\"𐞁", "!!", "123", "456", "78", "­", "\u000bs", "٣٤٥", "٦", "é's", "å", "㍿\r\n", "́", "Z", "…"]} +{"text": "'s<|fim_prefix|>İ!漢<|fim_prefix|>(ꟲ'reé å𐞁", "tokens": 29, "pieces": ["'s", "<|", "fim", "_prefix", "|>", "İ", "!漢", "<|", "fim", "_prefix", "|>(", "ꟲ're", "é", " ", " å𐞁"]} +{"text": " 😀🏽<|endoftext|>‍👍🏽\".>\r\n𐞁Dž\n", "tokens": 27, "pieces": [" ", " 😀🏽<|", "endoftext", "|>‍👍🏽\".>\r\n", "𐞁", "Dž", "\n"]} +{"text": "\r\nḍ̇\r\n\r\nA'D\n….­", "tokens": 12, "pieces": ["\r\n", "ḍ̇", "\r\n\r\n", "A'D", "\n", "…", ".­"]} +{"text": "'D ㍿'refi!!<|fim_prefix|>\n👍🏽…𐞁…'re0<|endoftext|>>𐞁'VE‍́ \n'll", "tokens": 46, "pieces": ["'D", " ", " ㍿'", "refi", "!!<|", "fim", "_prefix", "|>\n", "👍🏽", "…𐞁", "…", "'re", "0", "<|", "endoftext", "|>>", "𐞁'VE", "‍́", " \n", "'ll"]} +{"text": "\rå", "tokens": 3, "pieces": ["\r", "å"]} +{"text": "㍿İ'ſ#$%́\r\n\r\n\r\n\r\n\u000b字Z
.a'llEOT 9 <'D\"!é'D'Sm'S‍Z'Dꟲİa#$%<|fim_prefix|> \n", "tokens": 52, "pieces": ["㍿İ'ſ", "#$%́\r\n\r\n\r\n\r\n", "\u000b字", "Z", "
", ".a'll", "EOT", " ", " ", "9", " ", "<'", "D", "\"!", "é'D", "'Sm", "'", "S", "‍Z'D", "ꟲİa", "#$%<|", "fim", "_prefix", "|>", " \n"]} +{"text": "'re>'ſ<|fim_prefix|>́
é ", "tokens": 14, "pieces": ["'re", ">'", "ſ", "<|", "fim", "_prefix", "|>́", "
é", " "]} +{"text": "\r\n\r\n'Td(İ'T>dꟲ\r\nm", "tokens": 12, "pieces": ["\r\n\r\n", "'Td", "(İ'T", ">dꟲ", "\r\n", "m"]} +{"text": ".'Re're'S'Ss'ſAⅣé٣٤٥٦#$%
!-㍿'D 'VEa's㋿'De\"
  ", "tokens": 44, "pieces": [".'", "Re're", "'S'S", "s'ſ", "A", "Ⅳ", "é", "٣٤٥", "٦", "#$%", "
", "!-㍿'", "D", " ", "'VEa's", "㋿'", "D", "e", "\"", "
  "]} +{"text": "३A#$%0m", "tokens": 10, "pieces": ["३", "A", "#$%", "0", "m"]} +{"text": ",\u000b-'ll \n'𐞁Aée'll9fi㋿Dž​\rs字", "tokens": 31, "pieces": [",", "\u000b", "-'", "ll", " \n", "'<", "EOT", ">𐞁Aée'll", "9", "fi", "㋿", "Dž", "​\r", "s字"]} +{"text": "é<|fim_prefix|>\r\n\r\n㍿ ‍é", "tokens": 17, "pieces": ["é", "<|", "fim", "_prefix", "|>\r\n\r\n", "㍿", " ", "‍é"]} +{"text": "<|endoftext|>", "tokens": 7, "pieces": ["<|", "endoftext", "|>"]} +{"text": "ſ عfi\r‍-EOT🙂‍'S
 ­'ꟲ#$%\re\"ꟲ#$%<|endoftext|>👍🏽\"09
Ⅳ <|endoftext|>>🙂é", "tokens": 65, "pieces": ["ſ", " عfi", "\r", "‍-", "EOT", "🙂‍'", "S", "
 ", " ­'", "ꟲ", "#$%\r", "e", "\"<", "META", "_START", ">ꟲ", "#$%<|", "endoftext", "|><", "META", "_START", ">👍🏽\"", "09", "
", "Ⅳ", " ", "<|", "endoftext", "|>><", "EOT", ">🙂", "é"]} +{"text": "㋿fi\tDž \n'Tḍ̇éa🙂 \n!!ß😀🏽\r<'reŹ0​\u000b­efi#$%'Sa'S٣٤٥٦
d", "tokens": 43, "pieces": ["㋿fi", "\tDž", " \n", "'Tḍ̇éa", "🙂", " \n", "!!", "ß", "😀🏽\r", "<'", "re", "Ź", "0", "​", "\u000b", "­efi", "#$%'", "Sa'S", "٣٤٥", "٦", "
d"]} +{"text": "0a'såEOTEOTḍ̇'Tß's'VE'D,<|fim_prefix|>…'Dž'S<\r…fim👍🏽ḍ̇३", "tokens": 42, "pieces": ["0", "a's", "å", "EOTEOTḍ̇'T", "ß's", "'VE'D", ",<|", "fim", "_prefix", "|>", "…", "'Dž'S", "<\r", "…fim", "👍🏽", "ḍ̇", "३"]} +{"text": "㍿!!A'D!'ſ\n٣٤٥٦‍é", "tokens": 16, "pieces": ["㍿!!", "A'D", "!'", "ſ", "\n", "٣٤٥", "٦", "‍é"]} +{"text": "m😀🏽½'T𐞁​ß🙂 \n,ß­ \n(́'ſDž­३'EOT>'ſ\rع'M㋿tfiZ'S🙂 ", "tokens": 46, "pieces": ["m", "😀🏽", "½", "'T𐞁", "​<", "EOT", ">ß", "🙂", " \n", ",ß", "­", " \n", "(́'ſ", "Dž", "­", "३", "'EOT", ">'", "ſ", "\r", "ع'M", "㋿tfi", "Z'S", "🙂", " "]} +{"text": "åé ½ é\t#$%\r\n'D're'D('st å<|fim_prefix|>t'Td
d \n,㍿'s9'Mḍ̇", "tokens": 45, "pieces": ["åé", " ", "½", " ", " é", "\t", "#$%\r\n", "'D're", "'D", "('", "st", " å", "<|", "fim", "_prefix", "|>", "t'T", "d", "
d", " \n", ",㍿'", "s", "", "9", "'Mḍ̇"]} +{"text": "<ꟲ<|endoftext|>å
'll'ſé字d<|endoftext|>\neeḍ̇\r\n\r\n\r\t\t \"\"td \n're", "tokens": 39, "pieces": ["<ꟲ", "<|", "endoftext", "|>", "å", "
", "'ll'ſ", "é字d", "<|", "endoftext", "|>\n", "eeḍ̇", "\r\n\r\n\r", "\t\t", " ", "\"\"", "td", " \n", "'re"]} +{"text": ",‍e😀🏽
('M٣٤٥٦㋿fi㋿", "fi", ",fi👍🏽…e'Reع 'ſ'Re Ⅳ'M'S ", "tokens": 32, "pieces": [".漢", "
d", "㋿🙂<", "META", "_START", ">,", "fi", "👍🏽", "…e'Re", "ع", " ", "'ſ'Re", " ", " ", "Ⅳ", "'M'S", " "]} +{"text": "EOT d\r ſ‍
d३ \n㋿12345678Zé", "tokens": 20, "pieces": ["EOT", " d", "\r", " ſ", "‍", "
d", "३", " \n", "㋿", "123", "456", "78", "Zé"]} +{"text": "-AEOT!!fi \n🙂<漢,½३é字'ś 字'ſtsd漢12345678", "tokens": 27, "pieces": ["-AEOT", "!!", "fi", " \n", "🙂<", "漢", ",", "½३", "é字's", "́", " 字'ſ", "tsd漢", "123", "456", "78"]} +{"text": "​(fi#$%­\n\r<|endoftext|><\r\n'VE漢 𐞁٣٤٥٦mDž'ſ́'TEOT<|fim_prefix|>ſ", "tokens": 46, "pieces": ["​(", "fi", "#$%­\n\r", "<|", "endoftext", "|><\r\n", "'VE漢", " 𐞁", "٣٤٥", "٦", "m", "Dž'ſ", "́'T", "EOT", "<|", "fim", "_prefix", "|>", "ſ"]} +{"text": "9字٣٤٥٦t٣٤٥٦ ­​\r\n३", "tokens": 16, "pieces": ["9", "字", "٣٤٥", "٦", "t", "٣٤٥", "٦", " ", "­​\r\n", "३"]} +{"text": "ع'M!½<|fim_prefix|>EOTeꟲ \nſ!<👍🏽½३a!", "tokens": 29, "pieces": ["ع'M", "!", "½", "<|", "fim", "_prefix", "|><", "EOT", ">EOTeꟲ", " \n", "ſ", "!<👍🏽", "½३", "a", "!"]} +{"text": "'re!!İ٣٤٥٦İ( ,\tZ#$%
İ😀🏽…İ.'Re 'ReEOT#$%'T#$%字<|fim_prefix|>m­İ", "tokens": 48, "pieces": ["'re", "!!", "İ", "٣٤٥", "٦", "İ", "(", " ", ",", "\tZ", "#$%", "
İ", "😀🏽", "…İ", ".'", "Re", " ", "'Re", "EOT", "#$%'", "T", "#$%", "字", "<|", "fim", "_prefix", "|><", "EOT", ">m", "­İ"]} +{"text": "s\u000b‍'re 'ſfi0ma.​9́\t.\n字'Ré'M'll'Sİ,
EOT…👍🏽\t'VE​s,", "tokens": 47, "pieces": ["٣٤٥", "٦", "ß", "<|", "fim", "_prefix", "|>", " '", "ſfi", "0", "ma", ".​", "9", "́", "\t", ".\n", "字'Re", "́'M", "'", "ll'S", "İ", ",", "
EOT", "…", "👍🏽", "\t", "'VE", "​s", ","]} +{"text": "'S\"Ⅳ.\ta३ A(㋿'VEß!!é'VE😀🏽-'VE٣٤٥٦🙂 e'VE'\t0ß👍🏽字#$% \r\n", "tokens": 50, "pieces": ["'S", "\"", "Ⅳ", ".", "\ta", "३", " A", "(㋿'", "VEß", "!!", "é'VE", "😀🏽-'", "VE", "٣٤٥", "٦", "🙂", " e'VE", "'", "\t", "0", "ß", "👍🏽", "字", "#$%", " \r\n"]} +{"text": "12345678ſ​ #$%ꟲ !'VEm", "tokens": 16, "pieces": ["123", "456", "78", "ſ", "​", " ", " #$%", "ꟲ", " !'", "VEm"]} +{"text": "!  \n字' 'ſ\r\nå", "tokens": 54, "pieces": ["!", "  \n", "字", "'<", "META", "_START", ">A", "", " ", " '", "ſ", "\r\n", "å"]} +{"text": "'ſ\tꟲ𐞁m9\"…\u000b­‍éع'reⅣ,‍٣٤٥٦('sDžZ'm👍🏽👍🏽", "tokens": 51, "pieces": ["'ſ", "\tꟲ𐞁m", "9", "\"", "…", "\u000b", "­<", "META", "_START", ">‍", "éع", "'", "re", "Ⅳ", ",<", "EOT", ">‍", "٣٤٥", "٦", "('", "s", "DžZ'm", "👍🏽👍🏽"]} +{"text": "'sA 'T'sfi9­'Re ḍ̇e'M­\r\" EOT\u000btDž", "tokens": 34, "pieces": ["'s", "A", " <", "EOT", ">'", "T's", "fi", "9", "­'", "Re", " ", " ḍ̇e'M", "­\r", "\"", " EOT", "\u000bt", "Dž", ""]} +{"text": "'S'VE😀🏽'D,½㍿a <<|endoftext|>'reß㍿👍🏽!'re
é…-
½ع𐞁\"fi,
́é>'D-😀🏽0,㋿\t", "tokens": 64, "pieces": ["'S'VE", "😀🏽'", "D", ",", "½", "㍿a", " ", " <<|", "endoftext", "|>'", "reß", "㍿👍🏽!'", "re", "
é", "…", "-", "
", "½", "ع𐞁", "\"fi", ",", "
́é", ">'", "D", "-😀🏽", "0", ",㋿", "\t"]} +{"text": "'S‍𐞁٣٤٥٦३Džaé t<|fim_prefix|>A\"-'Rea\r\né\r\n\u000b \u000b9'S'S>", "tokens": 41, "pieces": ["'S", "‍𐞁", "٣٤٥", "٦३", "Džaé", " t", "<|", "fim", "_prefix", "|>", "A", "\"-<", "EOT", ">'", "Rea", "\r\n", "é", "\r\n", "\u000b ", "\u000b", "9", "'S'S", ">"]} +{"text": "e'VE'Då'Sḍ̇👍🏽", "tokens": 20, "pieces": ["e'VE", "'Då'S", "ḍ̇", "👍🏽<", "EOT", "><", "EOT", ">"]} +{"text": "㍿\r\n'll٣٤٥٦at㋿>Dž😀🏽\n ", "tokens": 21, "pieces": ["㍿\r\n", "'ll", "٣٤٥", "٦", "at", "㋿>", "Dž", "😀🏽\n", " "]} +{"text": "'śé0३😀🏽 \n#$%,#$%<|endoftext|>-İ
!३", "tokens": 25, "pieces": ["'śé", "0३", "😀🏽", " \n", "#$%,#$%<|", "endoftext", "|>-", "İ", "
", "!", "३"]} +{"text": "ſéⅣ\r\n🙂Dž㍿", "tokens": 12, "pieces": ["ſé", "Ⅳ", "\r\n", "🙂Dž", "㍿"]} +{"text": ".́\", fi\r\n\r\n३9 é!!㋿aé👍🏽!!'s३éꟲ\r!!'ſḍ̇ß🙂", "tokens": 35, "pieces": [".́", "\",", " fi", "\r\n\r\n", "३9", " ", " é", "!!㋿", "aé", "👍🏽!!'", "s", "३", "éꟲ", "\r", "!!'", "ſḍ̇ß", "🙂"]} +{"text": "m'VEs's'll<A👍🏽㋿'ſ\r\r\n\r\n.12345678\tع\r\n\r\n字🙂ḍ̇½'reſ\"!!!!́Ⅳfiꟲ<|fim_prefix|>\r‍s", "tokens": 51, "pieces": ["m'VE", "s's", "'ll", "<A", "👍🏽㋿'", "ſ", "\r\r\n\r\n", ".", "123", "456", "78", "\tع", "\r\n\r\n", "字", "🙂ḍ̇", "½", "'reſ", "\"!!!!́", "Ⅳ", "fiꟲ", "<|", "fim", "_prefix", "|>\r", "‍s"]} +{"text": " \n\tDžḍ̇EOTe ㋿'Re𐞁\r'VEaDž<|endoftext|>9#$%ꟲå字're
!\r\n😀🏽'reAaḍ̇ Dž 'M漢\r\n'ſ're", "tokens": 70, "pieces": [" \n", "\tDžḍ̇", "EOTe", " ", " <", "META", "_START", ">㋿'", "Re𐞁", "\r", "'VEa", "Dž", "<|", "endoftext", "|>", "9", "#$%", "ꟲå字're", "
", "!\r\n", "😀🏽'", "re", "Aaḍ̇", " ", " Dž", " ", "'M漢", "\r\n", "'ſ're"]} +{"text": "漢'Ré'reEOT​m", "tokens": 8, "pieces": ["漢'Re", "́'re", "EOT", "​m"]} +{"text": "‍é'Dd……!e३>.<Ⅳ३३", "tokens": 25, "pieces": ["‍", "é'D", "d", "…", "…", "!<", "EOT", ">e", "३", ">.<", "Ⅳ३३"]} +{"text": "'TDž're0're\r\n'Re
Ⅳḍ̇ ㋿‍ß-'TdEOTfi å!(fiḍ̇ \n", "tokens": 34, "pieces": ["'TDž're", "0", "'re", "\r\n", "'Re", "
", "Ⅳ", "ḍ̇", " ", "㋿‍", "ß", "-'", "Td", "EOTfi", " ", " å", "!(", "fiḍ̇", " \n"]} +{"text": "'s½,­३d‍ꟲ\t漢㋿𐞁.t… \r\n\r\n9fi字­٣٤٥٦👍🏽fi👍🏽('s12345678\r\n\r\nEOT \n'ع#$%'ll½", "tokens": 55, "pieces": ["'s", "½", ",­", "३", "d", "‍ꟲ", "\t漢", "㋿𐞁", ".t", "… \r\n\r\n", "9", "fi字", "­", "٣٤٥", "٦", "👍🏽", "fi", "👍🏽('", "s", "123", "456", "78", "\r\n\r\n", "EOT", " \n", "'ع", "#$%'", "ll", "½"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\nå,a字字'Re'S😀🏽12345678 ſfiḍ̇'Re#$%㋿'D", "tokens": 32, "pieces": ["\r\n", "å", ",", "a字字'Re", "'S", "😀🏽", "123", "456", "78", " ſfiḍ̇'Re", "#$%㋿'", "D"]} +{"text": "\r\néDž👍🏽漢", "tokens": 8, "pieces": ["\r\n", "é", "Dž", "👍🏽", "漢"]} +{"text": "A\r\n\r\n\r\n㍿عtZⅣ​d'retſ!<|fim_prefix|> 'VE,", "tokens": 34, "pieces": ["A", "\r\n\r\n\r\n", "㍿عt", "Z", "Ⅳ", "​<", "EOT", ">d're", "tſ", "!<", "EOT", "><", "META", "_START", "><|", "fim", "_prefix", "|>", " '", "VE", ","]} +{"text": " \n", "tokens": 1, "pieces": [" \n"]} +{"text": "t-\t", "tokens": 3, "pieces": ["t", "-", "\t"]} +{"text": "'M12345678é<|endoftext|>'Sḍ̇㍿é!ſḍ̇a'ſfi're#$%́", "tokens": 34, "pieces": ["'M", "123", "456", "78", "é", "<|", "endoftext", "|>'", "Sḍ̇", "㍿é", "!ſḍ̇a'ſ", "fi're", "#$%́"]} +{"text": "'T ع(㋿\"­İ", "tokens": 13, "pieces": ["'T", " ع", "(㋿<", "META", "_START", ">\"­", "İ"]} +{"text": "字'lĺ­!㍿'ſ<|fim_prefix|>½a0\t<|endoftext|>३😀🏽\r\n\r\n.m​'ſfiåEOT'‍'M\r\n\r\n字\r\n\r\n㋿s're", "tokens": 61, "pieces": ["字'll", "́", "­!㍿'", "ſ", "<|", "fim", "_prefix", "|>", "½", "a", "0", "", "\t", "<|", "endoftext", "|>", "३", "😀🏽\r\n\r\n", ".m", "​'", "ſfiå", "EOT", "'‍'", "M", "\r\n\r\n", "字", "\r\n\r\n", "㋿s", "'", "re"]} +{"text": ",…<|fim_prefix|>­ 'D­\r\ne\u000b'SⅣ<|fim_prefix|>.\u000b😀🏽s'VEa\t𐞁'Rea Ⅳ", "tokens": 46, "pieces": [",", "…", "<|", "fim", "_prefix", "|>­", " ", "'D", "­\r\n", "e", "\u000b", "'S", "Ⅳ", "<|", "fim", "_prefix", "|>.", "\u000b", "😀🏽", "s'VE", "a", "", "\t𐞁'Re", "a", " ", "Ⅳ"]} +{"text": "­'s#$%\u000bİ", "tokens": 7, "pieces": ["­'", "s", "#$%", "\u000bİ"]} +{"text": "㋿'M​", "tokens": 6, "pieces": ["㋿'", "M", "​"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "字12345678㋿ß\u000b\u000b \n½t字'S9½\r🙂å३0½< ſ\r'sſ'sع\r-d‍ 'refi's", "tokens": 51, "pieces": ["字", "123", "456", "78", "㋿ß", "\u000b\u000b \n", "½", "t字'S", "9½", "\r", "🙂å", "३0½", "<", " ", " ſ", "\r", "'s", "ſ's", "ع", "\r", "-d", "‍", " ", "'re", "fi's"]} +{"text": "0‍٣٤٥٦
0­éß漢d漢
漢's-ꟲås\r字'll-.İsA ​́", "tokens": 34, "pieces": ["0", "‍", "٣٤٥", "٦", "
", "0", "­éß漢d漢", "
漢's", "-ꟲås", "\r", "字'll", "-.", "İs", "A", " ​́"]} +{"text": ". é३'re.>३'Mḍ̇'re👍🏽<|fim_prefix|>'T́漢12345678'VE\r\n\r\n👍🏽­'ſDž", "tokens": 39, "pieces": [".", " é", "३", "'re", ".>", "३", "'Mḍ̇'re", "👍🏽<|", "fim", "_prefix", "|>'", "T́漢", "123", "456", "78", "'VE", "\r\n\r\n", "👍🏽­'", "ſ", "Dž"]} +{"text": "\"ⅣEOT'VE
'VE!!…\u000bd'S>\"", "tokens": 17, "pieces": ["\"", "Ⅳ", "EOT'VE", "
", "'VE", "!!", "…", "\u000bd'S", ">\""]} +{"text": "­\nŹ\r\n", "tokens": 4, "pieces": ["­\n", "Ź", "\r\n"]} +{"text": ">'T㍿fi#$%­字Ⅳ0'M'İmⅣEOT-👍🏽 A!EOTⅣ\n0\r\n'VEe\u000b ع\u000b漢🙂­", "tokens": 53, "pieces": [">'", "T", "㍿", "fi", "#$%­", "字", "Ⅳ0", "'M", "'<", "EOT", ">İm", "Ⅳ", "EOT", "-👍🏽", " ", " A", "!EOT", "Ⅳ", "\n", "0", "\r\n", "'VEe", "\u000b ", " ع", "\u000b漢", "🙂­"]} +{"text": "tA \"s's\n's ­<\n𐞁,AEOT㍿A\r\n\r\n½漢\rEOT'EOTs.", "tokens": 32, "pieces": ["t", "A", " ", "\"s's", "\n", "'s", " ", "­<\n", "𐞁", ",AEOT", "㍿A", "\r\n\r\n", "½", "漢", "\r", "EOT", "'EOTs", "."]} +{"text": "Džſ😀🏽😀🏽<>​<‍é 'ſEOT<|endoftext|>‍­'Sfi
", "tokens": 36, "pieces": ["Džſ", "😀🏽😀🏽<>​<‍", "é", " ", " '", "ſ", "EOT", "<|", "endoftext", "|>‍­'", "Sfi", "", "
"]} +{"text": " Zåİ12345678字", "tokens": 8, "pieces": [" Zå", "İ", "123", "456", "78", "字"]} +{"text": "#$%٣٤٥٦İé'Se‍fi!!Dž<'ſ ½", "tokens": 25, "pieces": ["#$%", "٣٤٥", "٦", "İé'S", "e", "‍fi", "!!", "Dž", "<'", "ſ", " <", "META", "_START", ">", "½", ""]} +{"text": "👍🏽s", "tokens": 4, "pieces": ["👍🏽", "s"]} +{"text": "#$%\r\n\r\ns٣٤٥٦<|fim_prefix|>३\u000bm9'Re\ta½<|fim_prefix|>\"\nZ​!!\n#$%ß\rm's\"'T", "tokens": 42, "pieces": ["#$%\r\n\r\n", "s", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "३", "\u000bm", "9", "'Re", "\ta", "", "½", "<|", "fim", "_prefix", "|>\"\n", "Z", "​!!\n", "#$%", "ß", "\r", "m's", "\"'", "T"]} +{"text": "'re'T'S\ń\r\n​fi<ع!字12345678\n9'M‍😀🏽́'re\u000b漢😀🏽́'", "re", "\u000b漢", "
'rea \néEOT😀🏽", "tokens": 29, "pieces": ["e字", "<字're", "٣٤٥", "٦ⅣⅣ", "<|", "fim", "_prefix", "|>", "
", "'rea", " \n", "é", "EOT", "😀🏽"]} +{"text": "٣٤٥٦'ſ<|endoftext|>0EOT👍🏽0  ­…å\r\"'T\r\nḍ̇éa-'re<|endoftext|>'S\r\n\r\n", "tokens": 48, "pieces": ["٣٤٥", "٦", "'ſ", "<|", "endoftext", "|>", "0", "EOT", "👍🏽", "0", " ", " ", "­", "…å", "\r", "\"'", "T", "\r\n", "ḍ̇éa", "-'", "re", "<|", "endoftext", "|>'", "S", "\r\n\r\n"]} +{"text": "Dž​fi're'll", "tokens": 6, "pieces": ["Dž", "​fi're", "'ll"]} +{"text": "\r\nå…,漢
fiß\" \n\r-𐞁'VE>\tm㋿'s
>- \n㍿", "tokens": 33, "pieces": ["\r\n", "å", "…", ",漢", "
fiß", "\"", " \n\r", "-𐞁'VE", ">", "\tm", "㋿'", "s", "
", ">-", " \n", "㍿"]} +{"text": "ma'0½३'D'rea'ſ\u000b😀🏽漢'Ḿ'D!…𐞁 \nee٣٤٥٦ǻ>", "tokens": 35, "pieces": ["ma", "'", "0½३", "'D're", "a'ſ", "\u000b", "😀🏽", "漢'M", "́'D", "!", "…𐞁", " \n", "ee", "٣٤٥", "٦", "ǻ", ">"]} +{"text": "​ع<|endoftext|>㍿ \nZ\u000b\tß'Re\r\n\r\n३EOT", "tokens": 22, "pieces": ["​ع", "<|", "endoftext", "|>㍿", " \n", "Z", "\u000b", "\tß'Re", "\r\n\r\n", "३", "EOT"]} +{"text": "'s\tḍ̇ t
 ㋿-٣٤٥٦ße🙂😀🏽ß12345678😀🏽m<|endoftext|>٣٤٥٦字ſ(\r\néfi0é'T", "tokens": 54, "pieces": ["'s", "\tḍ̇", " t", "
 ", " ㋿<", "META", "_START", ">-", "٣٤٥", "٦", "ße", "🙂😀🏽", "ß", "123", "456", "78", "😀🏽", "m", "<|", "endoftext", "|>", "٣٤٥", "٦", "字ſ", "(\r\n", "éfi", "0", "é'T"]} +{"text": "A'VE<|fim_prefix|>", "tokens": 9, "pieces": ["A'VE", "<|", "fim", "_prefix", "|>"]} +{"text": "fi‍Aa'S-9㍿!­EOT.'M>\r\né", "tokens": 22, "pieces": ["fi", "‍", "Aa'S", "-", "9", "㍿!­", "EOT", ".'", "M", ">\r\n", "é"]} +{"text": "ſ'VEḍ̇३å漢Ⅳ0m🙂👍🏽ꟲ㍿", "tokens": 24, "pieces": ["ſ'VE", "ḍ̇", "३", "å漢", "Ⅳ0", "m", "🙂👍🏽", "ꟲ", "㍿"]} +{"text": "'ſ'll<|endoftext|>< tm३ Z-<|endoftext|>'St.m", "tokens": 25, "pieces": ["'ſ'll", "<|", "endoftext", "|><", " ", " tm", "३", " Z", "-<|", "endoftext", "|>'", "St", ".m"]} +{"text": "\"-Z<|endoftext|>'rea\n  Zéé'Taé0'Méå Dž<|fim_prefix|>ⅣDž9's", "tokens": 40, "pieces": ["\"-", "Z", "<|", "endoftext", "|>'", "rea", "\n", " ", " Zéé'T", "aé", "0", "'Méå", " Dž", "<|", "fim", "_prefix", "|>", "Ⅳ", "Dž", "9", "'s"]} +{"text": "Ⅳ😀🏽\r\ndé½\r\n\r\né'D's(", "tokens": 14, "pieces": ["Ⅳ", "😀🏽\r\n", "dé", "½", "\r\n\r\n", "é'D", "'s", "("]} +{"text": " (🙂're漢İⅣ \nß'Tİ'T'M", "tokens": 50, "pieces": [" ", "(🙂'", "re漢", "İ", "Ⅳ", "", " \n", "ß'T", "İ'T", "'M"]} +{"text": "\r\n\r\n!!\"\t ́\tſEOT\t漢𐞁\"ع's㋿(­३'MDž\t\nع\té>d'Sİ\"\n", "tokens": 39, "pieces": ["\r\n\r\n", "!!\"", "\t", " ́", "\tſ", "EOT", "\t漢𐞁", "\"<", "EOT", ">ع's", "㋿(­", "३", "'MDž", "\t\n", "ع", "\té", ">d'S", "İ", "\"\n"]} +{"text": "\r\n\r\nİ字'T㍿å \n‍.fi\r!!!t!<,Z​́Z", "tokens": 22, "pieces": ["\r\n\r\n", "İ字'T", "㍿å", " \n", "‍.", "fi", "\r", "!!!", "t", "!<,", "Z", "​́", "Z"]} +{"text": "s 'D٣٤٥٦Z\" 's<|endoftext|>\t#$%\n \n'llé'MZm𐞁'D'T", "tokens": 45, "pieces": ["s", " ", " '", "D", "٣٤٥", "٦", "Z", "\"", " ", " '", "s", "<|", "endoftext", "|>", "\t", "#$%\n", " \n", "'llé", "'", "MZm𐞁'D", "'", "T", ""]} +{"text": "🙂<|fim_prefix|>🙂'Re12345678\n>Dž#$%'s12345678!é字عa👍🏽A字a漢\u000b0\t#$%", "tokens": 56, "pieces": ["🙂<", "EOT", "><", "mé", "!!'", "M", "<|", "endoftext", "|><|", "fim", "_prefix", "|>🙂'", "Re", "123", "456", "78", "\n", ">Dž", "#$%'", "s", "123", "456", "78", "!é字عa", "👍🏽", "A字a漢", "\u000b", "0", "\t", "#$%"]} +{"text": "‍½'D'D‍sDž㋿-'s\"́é'D d\u000bعEOT​㋿", "tokens": 27, "pieces": ["‍", "½", "'D'D", "‍s", "Dž", "㋿-'", "s", "\"́é'D", " d", "\u000bع", "EOT", "​㋿"]} +{"text": "å'Tt<|fim_prefix|>字𐞁𐞁\u000b'TⅣ12345678é字", "tokens": 32, "pieces": ["å'T", "t", "<|", "fim", "_prefix", "|>", "字𐞁𐞁", "\u000b", "'T", "Ⅳ12", "345", "678", "é字"]} +{"text": ">😀🏽ꟲ", "tokens": 7, "pieces": [">😀🏽", "ꟲ"]} +{"text": "ß🙂 !İ'ſ<|endoftext|>'re-Zß9ḍ̇\n\téع٣٤٥٦(éEOT>㍿é#$%-m12345678ſDžİ", "tokens": 48, "pieces": ["ß", "🙂", " ", "!İ'ſ", "<|", "endoftext", "|>'", "re", "-Zß", "9", "ḍ̇", "\n", "\téع", "٣٤٥", "٦", "(é", "EOT", ">㍿", "é", "#$%-", "m", "123", "456", "78", "ſ", "Džİ"]} +{"text": "9'́<|fim_prefix|>🙂\"!​ꟲꟲ\r\n\r\n'reße0Z'D漢", "tokens": 25, "pieces": ["9", "'́", "<|", "fim", "_prefix", "|>🙂\"!​", "ꟲꟲ", "\r\n\r\n", "'reße", "0", "Z'D", "漢"]} +{"text": "!0're,'s'D.'VE#$%👍🏽😀🏽İ​🙂'Sé 'MEOT<٣٤٥٦'s\u000bfi0<|endoftext|>عé", "tokens": 43, "pieces": ["!", "0", "'re", ",'", "s'D", ".'", "VE", "#$%👍🏽😀🏽", "İ", "​🙂'", "Sé", " ", " '", "MEOT", "<", "٣٤٥", "٦", "'s", "\u000bfi", "0", "<|", "endoftext", "|>", "عé"]} +{"text": "\u000be㋿👍🏽'D٣٤٥٦'reع'M'll
12345678Ⅳa漢㍿>Ⅳ.-'D'VEs ३­\n", "tokens": 42, "pieces": ["\u000be", "㋿👍🏽'", "D", "٣٤٥", "٦", "'reع'M", "'ll", "
", "123", "456", "78Ⅳ", "a漢", "㍿>", "Ⅳ", ".-'", "D'VE", "s", " ", " ", "३", "­\n"]} +{"text": "é>'12345678㋿½٣٤٥٦३'Re\"t !!‍ ́", "tokens": 25, "pieces": ["é", ">'", "123", "456", "78", "㋿", "½٣٤", "٥٦३", "'Re", "\"t", " ", "!!‍", " ́", ""]} +{"text": "Ź's𐞁e字A9­ ßA漢tå\rd'MDž<", "tokens": 24, "pieces": ["Ź's", "𐞁e字", "A", "9", "­", " ß", "A漢tå", "\r", "d'M", "Dž", "<"]} +{"text": "\r\n\u000bEOTa'sEOT½0​½'VE-\r\t(👍🏽\u000bꟲ\"İ३\néİ <|endoftext|>'M0", "tokens": 45, "pieces": ["\r\n", "\u000bEOTa's", "EOT", "½0", "​", "½", "'VE", "-\r", "\t", "(👍🏽<", "META", "_START", ">", "\u000bꟲ", "\"İ", "३", "\n", "é", "İ", " ", "<|", "endoftext", "|>'", "M", "0"]} +{"text": "㋿'T ae🙂ſ0٣٤٥٦👍🏽m,‍\ré'VE Ⅳ'Så­>\n'Re३'Re<​.㍿'Tß'T👍🏽", "tokens": 47, "pieces": ["㋿'", "T", " ae", "🙂ſ", "0٣٤", "٥٦", "👍🏽", "m", ",‍\r", "é'VE", " ", "Ⅳ", "'Så", "­>\n", "'Re", "३", "'Re", "<​.㍿'", "Tß'T", "👍🏽"]} +{"text": "<|fim_prefix|>", "tokens": 6, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "a!!EOT'", "tokens": 5, "pieces": ["a", "!!", "EOT", "'"]} +{"text": "ſ🙂å漢ſéfi ", "tokens": 14, "pieces": ["ſ", "🙂å漢ſé", "fi", " "]} +{"text": "́́
0Dž‍㍿sع'Re'sm'Re
<|fim_prefix|>३å३字👍🏽\n", "tokens": 32, "pieces": ["́́", "
", "0", "Dž", "‍㍿", "sع'Re", "'sm'Re", "
", "<|", "fim", "_prefix", "|>", "३", "å", "३", "字", "👍🏽\n"]} +{"text": "t\né\t'å \n​,'M\"\u000b½m㋿ \n<‍ß12345678d Z'VE'ſA٣٤٥٦'M#$%éDž", "tokens": 43, "pieces": ["t", "\n", "é", "\t", "'å", " \n", "​,'", "M", "\"", "\u000b", "½", "m", "㋿", " \n", "<‍", "ß", "123", "456", "78", "d", " Z'VE", "'ſ", "A", "٣٤٥", "٦", "'M", "#$%", "é", "Dž"]} +{"text": "'T<'S!!\r\nZDž.㍿ås'em\n'T 12345678'T9🙂 \n's㍿\"'D9's\r.ſ12345678👍🏽", "tokens": 43, "pieces": ["'T", "<'", "S", "!!\r\n", "ZDž", ".㍿", "ås", "'em", "\n", "'T", " ", "123", "456", "78", "'T", "9", "🙂", " \n", "'s", "㍿\"'", "D", "9", "'s", "\r", ".ſ", "123", "456", "78", "👍🏽"]} +{"text": "me<|fim_prefix|>#$%é​", "tokens": 12, "pieces": ["me", "<|", "fim", "_prefix", "|>#$%", "é", "​"]} +{"text": "\tst…\reßé 'ſé'S\tḍ̇", "tokens": 17, "pieces": ["\tst", "…\r", "eßé", " ", " '", "ſé'S", "\tḍ̇"]} +{"text": "İaé<|endoftext|>#$%<'ſs\n", "tokens": 16, "pieces": ["İaé", "<|", "endoftext", "|>#$%<'", "ſs", "\n"]} +{"text": "é\n eZ's­👍🏽 ​'T.mm🙂d<|endoftext|> 's\"!é­ ‍\".", "tokens": 33, "pieces": ["é", "\n", " e", "Z's", "­👍🏽", " ", " ​'", "T", ".mm", "🙂d", "<|", "endoftext", "|>", " '", "s", "\"!", "é", "­", " ", " ‍\"."]} +{"text": "🙂​​\u000bⅣ-<|fim_prefix|>'VE½½'ſ…å'll\"😀🏽 \n३Ata<½ \n<", "tokens": 34, "pieces": ["🙂​​", "\u000b", "Ⅳ", "-<|", "fim", "_prefix", "|>'", "VE", "½½", "'ſ", "…å'll", "\"😀🏽", " \n", "३", "Ata", "<", "½", " \n", "<"]} +{"text": " e'VE👍🏽­EOT<|fim_prefix|>漢ßå​.'re<|fim_prefix|>​😀🏽‍½ع𐞁,é 9­㍿
.…\n ٣٤٥٦aA<|fim_prefix|>s ", "tokens": 68, "pieces": [" ", " e'VE", "👍🏽­", "EOT", "<|", "fim", "_prefix", "|>", "漢ßå", "​.'", "re", "<|", "fim", "_prefix", "|>​😀🏽‍", "½", "ع𐞁", ",é", " ", "9", "­㍿", "
", ".", "…\n", " ", "٣٤٥", "٦", "a", "A", "<|", "fim", "_prefix", "|>", "s", " "]} +{"text": "é३!!-a12345678!!", "tokens": 9, "pieces": ["é", "३", "!!-", "a", "123", "456", "78", "!!"]} +{"text": " \n're.", "tokens": 7, "pieces": [" \n", "'", "re", "."]} +{"text": "'T'DA३𐞁'T", "tokens": 9, "pieces": ["'T'D", "A", "३", "𐞁'T"]} +{"text": " <|endoftext|>", "tokens": 8, "pieces": [" ", "<|", "endoftext", "|>"]} +{"text": "👍🏽…­漢\r\nⅣ åZAZ'T<|fim_prefix|>\n\nsZm\" EOTZ\r\n\r\n
'll 
Z👍🏽漢㋿٣٤٥٦😀🏽EOTꟲ0're½‍", "tokens": 58, "pieces": ["👍🏽", "…", "­漢", "\r\n", "Ⅳ", " å", "ZAZ'T", "<|", "fim", "_prefix", "|>\n\n", "s", "Zm", "\"", " ", " EOTZ", "\r\n\r\n", "
", "'ll", " ", "
Z", "👍🏽", "漢", "㋿", "٣٤٥", "٦", "😀🏽", "EOTꟲ", "0", "'re", "½", "‍"]} +{"text": "🙂as\nå\r.å​Dž(\t​!!Ⅳd‍>㋿#$%​\u000b字'll<|endoftext|>fi!,", "tokens": 47, "pieces": ["🙂<", "META", "_START", ">as", "\n", "å", "\r", ".å", "​Dž", "(", "\t", "​!!", "Ⅳ", "d", "‍<", "EOT", ">>㋿#$%​", "\u000b字'll", "<|", "endoftext", "|>", "fi", "!,"]} +{"text": "\"٣٤٥٦å٣٤٥٦'s éZ!!👍🏽( 12345678İ́ <|endoftext|>\r\n\r\n< …\r\n\r\n'T'٣٤٥٦㋿😀🏽<|endoftext|>٣٤٥٦te'MmꟲA", "tokens": 67, "pieces": ["\"", "٣٤٥", "٦", "å", "٣٤٥", "٦", "'s", " ", " é", "Z", "!!👍🏽(", " ", "123", "456", "78", "İ́", " <|", "endoftext", "|>\r\n\r\n", "<", " …\r\n\r\n", "'T", "'", "٣٤٥", "٦", "㋿😀🏽<|", "endoftext", "|>", "٣٤٥", "٦", "te'M", "mꟲ", "A"]} +{"text": ">\u000b😀🏽漢́s-'M'sEOT 漢9#$%ß👍🏽,", "tokens": 23, "pieces": [">", "\u000b", "😀🏽", "漢́s", "-'", "M's", "EOT", " ", " 漢", "9", "#$%", "ß", "👍🏽,"]} +{"text": "́> DžDžEOTfi\r\n🙂a'Re >EOTAEOT.Z0<|fim_prefix|><|endoftext|>.\"é ss'D0‍ å…", "tokens": 47, "pieces": ["́", ">", " ", " DžDžEOTfi", "\r\n", "🙂a'Re", " ", ">EOTAEOT", ".Z", "0", "<|", "fim", "_prefix", "|><|", "endoftext", "|>.\"", "é", " ss'D", "0", "‍", " ", " å", "…"]} +{"text": "İ'T'sd!å½ḍ̇㋿ع're👍🏽é漢ß\n'VE​Ⅳ\t㍿'VE!,٣٤٥٦'ſ", "tokens": 55, "pieces": ["İ'T", "'sd", "!<", "m漢ß", "<|", "endoftext", "|>", "å", "½", "ḍ̇", "㋿ع're", "👍🏽", "é漢ß", "\n", "'VE", "​", "Ⅳ", "\t", "㍿'", "VE", "!,", "٣٤٥", "٦", "'ſ"]} +{"text": "EOT𐞁afi  \taDžfí 字‍ß \n", "tokens": 19, "pieces": ["EOT𐞁afi", "  ", "\ta", "Džfí", " 字", "‍ß", " \n"]} +{"text": "'VE\n'ſ'Deéḍ̇\"𐞁Z'ſ ,12345678 \n
Zꟲ 9😀🏽.…!!!eDž<-, ", "tokens": 44, "pieces": ["'VE", "\n", "'ſ'D", "eéḍ̇", "\"𐞁", "Z'ſ", " ,", "123", "456", "78", " \n", "
Zꟲ", " ", "9", "😀🏽.", "…", "!!!", "e", "Dž", "<-,", " "]} +{"text": "EOTe'T'ſ𐞁<|endoftext|>\n0ZZع(\u000b>½m ", "tokens": 29, "pieces": ["EOTe'T", "'ſ𐞁", "<|", "endoftext", "|>\n", "0", "Z", "Zع", "(", "\u000b", ">", "½", "m", " "]} +{"text": "sZ‍<|fim_prefix|>'Re!'Té \r🙂ea'D'D<㍿é́ e'D'T'Tſ(EOT😀🏽\r\n\r\n", "tokens": 41, "pieces": ["s", "Z", "‍<|", "fim", "_prefix", "|>'", "Re", "!'", "Té", " \r", "🙂ea'D", "'D", "<㍿", "é́", " <", "EOT", ">e'D", "'T'T", "ſ", "(EOT", "😀🏽\r\n\r\n"]} +{"text": "'T<İ'S
​('sA३'lled(", "tokens": 37, "pieces": ["ſe", "-", "٣٤٥", "٦", "'s", "A", "३", "'lled", "("]} +{"text": "''Re'D\t​eſ́.<12345678'S'VE\"ḍ̇'Re​'Reḍ̇s<fiDž''Te🙂ꟲſ'D(​'VEé9́", "tokens": 50, "pieces": ["''", "Re'D", "\t", "​eſ́", ".<", "123", "456", "78", "'S'VE", "\"ḍ̇'Re", "​'", "Reḍ̇s", "<fi", "Dž", "''", "Te", "🙂ꟲſ", "'", "D", "(​'", "VEé", "9", "́"]} +{"text": ",\t\"\re>Aé!!\r\n\r\n'ś#$%'reEOT9!'S'llEOT\u000b", "tokens": 24, "pieces": [",", "\t", "\"\r", "e", ">Aé", "!!\r\n\r\n", "'ś", "#$%'", "re", "EOT", "9", "!'", "S'll", "EOT", "\u000b"]} +{"text": "#$%İa'Dß \n
­
㋿…'Re ", "tokens": 17, "pieces": ["#$%", "İa'D", "ß", " \n", "
", "­", "
", "㋿", "…", "'Re", " "]} +{"text": "'s 'DEOTé0'll👍🏽mⅣ\u000b'Re'D\n're", "tokens": 22, "pieces": ["'s", " ", "'DEOTé", "0", "'ll", "👍🏽", "m", "Ⅳ", "\u000b", "'Re", "'", "D", "\n", "'re"]} +{"text": "a'Ds'VEⅣ're 'VE 9​'Re", "tokens": 21, "pieces": ["a'D", "s'VE", "", "Ⅳ", "'re", " ", "'VE", " ", " ", "9", "​'", "Re"]} +{"text": "'s­👍🏽…fi\u000b漢,\r\n'reع!!-fis", "tokens": 17, "pieces": ["'s", "­👍🏽", "…fi", "\u000b漢", ",\r\n", "'reع", "!!-", "fis"]} +{"text": ".m  \u000b㋿d'ſ́'D𐞁\r\n'Mså​9EOT㋿\"9.
عa", "tokens": 35, "pieces": [".m", "  ", "\u000b", "㋿d'ſ", "́'D", "𐞁", "\r\n", "'Mså", "​", "9", "EOT", "㋿\"", "9", ".", "
عa"]} +{"text": " \nßA'M 9'S ­'Re're're漢漢's<|fim_prefix|>éé\rDž…ß#$%👍🏽­
'D\r\t#$%Z,'VE३#$%,'re-", "tokens": 56, "pieces": [" \n", "ß", "A'M", " ", " ", "9", "'S", " ", "­'", "Re're", "'re漢漢's", "<|", "fim", "_prefix", "|>", "éé", "\r", "Dž", "…ß", "#$%👍🏽­", "
", "'D", "\r", "\t", "#$%", "Z", ",'", "VE", "३", "#$%,'", "re", "-"]} +{"text": "㍿ß½m\u000b<㋿'ſ 'A ſ漢d\råtEOT
‍>", "tokens": 31, "pieces": ["㍿ß", "½", "m", "\u000b", "<㋿'", "ſ", " '", "A", " ſ", "漢d", "\r", "åt", "EOT", "
", "‍>"]} +{"text": "\"é३🙂'Reḍ̇!A're", "tokens": 12, "pieces": ["\"é", "३", "🙂'", "Reḍ̇", "!A're"]} +{"text": "३A\"m<|fim_prefix|><12345678t'M<|endoftext|>'s㋿٣٤٥٦…\r\n\r\n12345678ś", "tokens": 38, "pieces": ["३", "A", "\"m", "<|", "fim", "_prefix", "|><", "123", "456", "78", "t'M", "<|", "endoftext", "|>'", "s", "㋿", "٣٤٥", "٦", "…\r\n\r\n", "123", "456", "78", "ś"]} +{"text": "'ll🙂'٣٤٥٦éEOTfi漢EOT<|fim_prefix|>\r\n\r\n \n'VEDž\n<|fim_prefix|>", "tokens": 33, "pieces": ["'ll", "🙂'", "٣٤٥", "٦", "é", "EOTfi漢", "EOT", "<|", "fim", "_prefix", "|>\r\n\r\n", " \n", "'VEDž", "\n", "<|", "fim", "_prefix", "|>"]} +{"text": "'re३Dž's!! \n𐞁'll'SEOT9!!㍿éDž\r0é", "tokens": 27, "pieces": ["'re", "३", "Dž's", "!!", " \n", "𐞁'll", "'SEOT", "9", "!!㍿", "é", "Dž", "\r", "0", "é"]} +{"text": "'D-
'12345678́<|endoftext|>#$%٣٤٥٦㍿<|endoftext|>'VE\"åDž,é
\r…'𐞁​́Dž
字 Dž", "tokens": 56, "pieces": ["'D", "-", "
", "'", "123", "456", "78", "́", "<|", "endoftext", "|>#$%", "٣٤٥", "٦", "㍿<|", "endoftext", "|>'", "VE", "\"å", "Dž", ",é", "
\r", "…", "'𐞁", "​́", "Dž", "
字", " Dž"]} +{"text": "​‍\r\n\r\n\"EOT\n\r\n🙂 . ㋿ع́éعع>İsعİ́ 'ſ…́é字", "tokens": 36, "pieces": ["​‍\r\n\r\n", "\"EOT", "\n", "\r\n", "🙂", " ", ".", " ", "㋿ع́éعع", ">İsع", "İ́", " ", "'ſ", "…́é字"]} +{"text": "\n\n#$% \n\r\tع \nⅣmعs३٣٤٥٦
<\r\n.ḍ̇👍🏽", "tokens": 28, "pieces": ["\n\n", "#$%", " \n\r", "\tع", " \n", "Ⅳ", "mعs", "३٣٤", "٥٦", "
", "<\r\n", ".ḍ̇", "👍🏽"]} +{"text": "('VE㋿fiḍ̇ ", "tokens": 14, "pieces": ["('", "VE", "㋿<", "META", "_START", ">fiḍ̇", " "]} +{"text": "m't", "tokens": 2, "pieces": ["m't"]} +{"text": "9\r#$% 'Re𐞁\rt#$%'SEOTİ𐞁é'0sḍ̇!!'ſtꟲ9a<|fim_prefix|>\"㋿'Re㋿, ٣٤٥٦👍🏽-e👍🏽Ⅳ㋿​", "tokens": 72, "pieces": ["9", "\r", "#$%", " ", "'Re𐞁", "\r", "t", "#$%'", "SEOTİ𐞁é", "'", "0", "sḍ̇", "!!'", "ſtꟲ", "9", "a", "<|", "fim", "_prefix", "|>\"㋿'", "Re", "㋿,", " ", "٣٤٥", "٦", "👍🏽-", "e", "👍🏽", "Ⅳ", "㋿​"]} +{"text": "𐞁​\n12345678.𐞁0d𐞁㍿३e½A३'re", "tokens": 28, "pieces": ["𐞁", "​\n", "123", "456", "78", ".𐞁", "0", "d𐞁", "㍿", "३", "e", "½", "A", "३", "'re"]} +{"text": "ßé'Re ́#$%'Sm
0'12345678's0\r\ńⅣEOT㍿ \naſİꟲ漢(", "tokens": 35, "pieces": ["ßé'Re", " ", " ́", "#$%'", "Sm", "
", "0", "'", "123", "456", "78", "'s", "0", "\r\n", "́", "Ⅳ", "EOT", "㍿", " \n", "aſ", "İꟲ漢", "("]} +{"text": "t\rſ#$%'Mꟲ'ſ​!!EOT0", "tokens": 16, "pieces": ["t", "\r", "ſ", "#$%'", "Mꟲ'ſ", "​!!", "EOT", "0"]} +{"text": "‍afi\n0漢", "tokens": 6, "pieces": ["‍afi", "\n", "0", "漢"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'Smع'll\r\n'VE\r\n\r\n😀🏽\t\nſ漢㍿
'Re.ع\r\n\r\n", "😀🏽", "\t\n", "ſ漢", "㍿", "
", "'Re", ".ع", "½ḍ̇Dž​(ADž'T9𐞁.'Ts㋿<|fim_prefix|>‍Dž½>ſ,Z'Ms.٣٤٥٦d>'D're'ſ😀🏽9<|endoftext|>a", "tokens": 63, "pieces": ["", "½", "ḍ̇", "Dž", "​(", "ADž'T", "9", "𐞁", ".'", "Ts", "㋿<|", "fim", "_prefix", "|>‍", "Dž", "½", ">ſ", ",Z'M", "s", ".", "٣٤٥", "٦", "d", ">'", "D're", "'ſ", "😀🏽", "9", "<|", "endoftext", "|>", "a"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " İ­İſDž\u000b", "tokens": 8, "pieces": [" İ", "­İſ", "Dž", "\u000b"]} +{"text": "३\r\n\r\n㋿A٣٤٥٦eعs!!'s㋿½Ⅳ'Té\r\n\r\n\"'D!<|fim_prefix|>عſ\r\n\r\n\r\ń́\r,\n \n<३", "tokens": 53, "pieces": ["३", "\r\n\r\n", "㋿A", "٣٤٥", "٦", "eعs", "!!'", "s", "㋿", "½Ⅳ", "'Té", "\r\n\r\n", "\"'", "D", "!<|", "fim", "_prefix", "|>", "عſ", "\r\n\r\n\r\n", "́́", "\r", ",\n", " \n", "<", "३", ""]} +{"text": "Z>sé\t\r\nſß<|endoftext|>…!😀🏽0#$%😀🏽é <|endoftext|> \n99'VE
å٣٤٥٦-fi漢Ⅳ‍'D're㋿'Re,", "tokens": 65, "pieces": ["Z", ">sé", "\t\r\n", "ſß", "<|", "endoftext", "|>", "…", "!😀🏽", "0", "#$%😀🏽", "é", " <|", "endoftext", "|>", " \n", "99", "'VE", "
å", "", "٣٤٥", "٦", "-fi漢", "Ⅳ", "‍'", "D're", "㋿'", "Re", ","]} +{"text": "é 👍🏽عdꟲ'T'Re", "tokens": 12, "pieces": ["é", " ", "👍🏽", "عdꟲ'T", "'Re"]} +{"text": "'Da\u000b#$% \nZ're <|endoftext|>…​Z 0.\u000b㋿e", "tokens": 27, "pieces": ["'Da", "\u000b", "#$%", " \n", "Z're", " <|", "endoftext", "|>", "…", "​Z", " ", "0", ".", "\u000b", "㋿e"]} +{"text": "\u000b३𐞁'M…­\n㋿", "tokens": 13, "pieces": ["\u000b", "३", "𐞁'M", "…", "­\n", "㋿"]} +{"text": "'s\n'SZ㍿", "tokens": 7, "pieces": ["'s", "\n", "'SZ", "㍿"]} +{"text": "👍🏽!!‍\"ß\"ḍ̇\r\n\r\n३fi", "tokens": 14, "pieces": ["👍🏽!!‍\"", "ß", "\"ḍ̇", "\r\n\r\n", "३", "fi"]} +{"text": "éé'll \nA !-", "tokens": 10, "pieces": ["éé'll", " \n", "A", " ", " !-"]} +{"text": "'Re 'ſtḍ̇… 9漢ß字!#$%𐞁…a12345678㋿9", "tokens": 36, "pieces": ["'Re", " ", " '", "ſtḍ̇", "… ", " ", "9", "漢ß字", "!#$%", "𐞁", "…a", "", "123", "456", "78", "㋿", "9"]} +{"text": "#$%字 Dž\r", "tokens": 7, "pieces": ["#$%", "字", " Dž", "\r"]} +{"text": "\" \n㋿<|fim_prefix|>EOT\r\n\r\n ", "tokens": 15, "pieces": ["\"", " \n", "㋿<|", "fim", "_prefix", "|>", "EOT", "\r\n\r\n", " "]} +{"text": "\r\n\r\n㋿👍🏽!'VEß12345678'll\r'ſ!!ß'DⅣ", "tokens": 26, "pieces": ["\r\n\r\n", "㋿👍🏽!'", "VEß", "123", "456", "78", "'ll", "\r", "'ſ", "!!<", "META", "_START", ">ß'D", "Ⅳ"]} +{"text": "EOT😀🏽👍🏽<|fim_prefix|>12345678t'D0!ꟲ0fi…12345678👍🏽é㋿12345678.", "tokens": 46, "pieces": ["EOT", "😀🏽👍🏽<|", "fim", "_prefix", "|>", "123", "456", "78", "t'D", "0", "!ꟲ", "", "0", "fi", "…", "123", "456", "78", "👍🏽", "é", "㋿", "123", "456", "78", "."]} +{"text": "<|endoftext|> <|endoftext|>0 'reع'Ma½", "tokens": 22, "pieces": ["<|", "endoftext", "|>", " ", " <|", "endoftext", "|>", "0", " ", "'reع'M", "a", "½"]} +{"text": "\r's‍(. .\n👍🏽0A<|endoftext|>'D\ré­\r\n\r\ń​'D'T", "tokens": 32, "pieces": ["\r", "'s", "‍(.", " ", ".\n", "👍🏽", "0", "A", "<|", "endoftext", "|>'", "D", "\r", "é", "­\r\n\r\n", "́", "​'", "D'T"]} +{"text": "'ReDž-…🙂漢Z漢
 'll.\r३é​'VE\u000b\"'reé'S 'VE漢0ß👍🏽😀🏽ع\t‍😀🏽ꟲ", "tokens": 56, "pieces": ["'Re", "Dž", "-", "…", "🙂漢Z漢", "
", " ", "'", "ll", ".\r", "३", "é", "​'", "VE", "\u000b", "\"'", "reé'S", " '", "VE漢", "0", "ß", "👍🏽😀🏽", "ع", "\t", "‍<", "EOT", ">😀🏽", "ꟲ"]} +{"text": "0<>'Tİ.ḍ̇åDž\r\n\r\nعA", "tokens": 16, "pieces": ["0", "<>'", "Tİ", ".ḍ̇å", "Dž", "\r\n\r\n", "ع", "A"]} +{"text": "s9ḍ̇字𐞁'll'३A é Z‍½s>'Mİa'Sé#$%té'<|endoftext|>", "tokens": 38, "pieces": ["s", "9", "ḍ̇字𐞁'll", "'", "३", "A", " é", " Z", "‍", "½", "s", ">'", "Mİa'S", "é", "#$%", "té", "'<|", "endoftext", "|>"]} +{"text": "𐞁å
t'llZéDž…e'TعEOT#$%#$%\u000b​. EOTa> ́'(ß<|endoftext|>㋿'re😀🏽-.", "tokens": 59, "pieces": ["𐞁å", "
t'll", "Zé", "Dž", "…e'T", "ع", "EOT", "#$%#$%", "\u000b", "​.", " EOTa", ">", " ", " ́", "'(", "ß", "<|", "endoftext", "|>㋿'", "re", "😀🏽<", "EOT", ">-."]} +{"text": "Dž٣٤٥٦EOT\r٣٤٥٦ 字å𐞁👍🏽ſ0​ '㋿<|fim_prefix|>𐞁#$%字'llA\t#$%å \né<|fim_prefix|>㍿#$%s'VEع're", "tokens": 74, "pieces": ["Dž", "٣٤٥", "٦", "EOT", "\r", "٣٤٥", "٦", " 字å𐞁", "👍🏽", "ſ", "0", "​", " ", " '㋿<|", "fim", "_prefix", "|>", "𐞁", "#$%", "字'll", "A", "\t", "#$%", "å", " \n", "é", "<|", "fim", "_prefix", "|>㍿#$%", "s'VE", "ع're", ""]} +{"text": "…\n½ >'D\r\n\r\n\t­-é\t9​'ll٣٤٥٦mſ漢-Dž٣٤٥٦İEOT'½\u000b\r\n٣٤٥٦漢.ḍ̇ß½", "tokens": 52, "pieces": ["…\n", "½", " >'", "D", "\r\n\r\n", "\t", "­-", "é", "\t", "9", "​'", "ll", "٣٤٥", "٦", "mſ漢", "-Dž", "٣٤٥", "٦", "İEOT", "'", "½", "\u000b\r\n", "٣٤٥", "٦", "漢", ".ḍ̇ß", "½"]} +{"text": " \n'Re㋿👍🏽éZ½aſ́-ſ'll'Re🙂(<|fim_prefix|>'\rfi'VEDžİ㋿d!é>", "tokens": 39, "pieces": [" \n", "'Re", "㋿👍🏽", "é", "Z", "½", "aſ́", "-ſ'll", "'Re", "🙂(<|", "fim", "_prefix", "|>'\r", "fi'VE", "Džİ", "㋿d", "!é", ">"]} +{"text": " \n''VEḍ̇'T", "tokens": 11, "pieces": [" \n", "''", "VE", "ḍ̇'T"]} +{"text": "!!#$%\n0é٣٤٥٦å.(<
12345678'ſa\n漢Aḍ̇e t('ſå'll\"Z'VE \"\té#$%mt", "tokens": 49, "pieces": ["!!#$%<", "EOT", ">\n", "0", "é", "٣٤٥", "٦", "å", ".(<", "
", "123", "456", "78", "'ſa", "\n", "漢Aḍ̇e", " t", "('", "ſå'll", "\"Z'VE", " \"", "\té", "#$%", "mt"]} +{"text": "'e߅!å''MⅣ㋿'VE", "tokens": 16, "pieces": ["'eß", "…", "!å", "''", "M", "Ⅳ", "㋿'", "VE"]} +{"text": "🙂fi<|fim_prefix|>👍🏽é'D'Re\r'ſ", "tokens": 18, "pieces": ["🙂fi", "<|", "fim", "_prefix", "|>👍🏽", "é'D", "'Re", "\r", "'ſ"]} +{"text": " ́\r\ń\n😀🏽'ſfi'Re0éß", "tokens": 15, "pieces": [" ́", "\r\n", "́", "\n", "😀🏽'", "ſfi'Re", "0", "éß"]} +{"text": "㍿'re'SA字", "tokens": 8, "pieces": ["㍿'", "re'S", "A字"]} +{"text": "é漢\"m#$%!!\r\n\r\n३👍🏽EOT\t漢 9 'D\"٣٤٥٦\r\n\r\n\r\nꟲ#$%́'M- \nعDž \n😀🏽9!", "tokens": 56, "pieces": ["é漢", "\"m", "#$%!!<", "EOT", ">\r\n\r\n", "३", "👍🏽", "EOT", "\t", "漢", " ", " ", "9", " '", "D", "\"", "٣٤٥", "٦", "\r\n\r\n\r\n", "ꟲ", "#$%́'", "M", "-", " \n", "ع", "Dž", " \n", "😀🏽", "9", "!<", "EOT", ">"]} +{"text": "d-\te d👍🏽,½‍åå٣٤٥٦", "tokens": 19, "pieces": ["d", "-", "\te", " d", "👍🏽,", "½", "‍åå", "٣٤٥", "٦"]} +{"text": "\r\nEOTś٣٤٥٦'D0'S👍🏽字", "tokens": 18, "pieces": ["\r\n", "EOT", "ś", "٣٤٥", "٦", "'D", "0", "'S", "👍🏽", "字"]} +{"text": ",é½", "tokens": 3, "pieces": [",é", "½"]} +{"text": "eEOT👍🏽'ſ <|endoftext|>", "tokens": 16, "pieces": ["e", "EOT", "👍🏽'", "ſ", " ", " <|", "endoftext", "|>"]} +{"text": "ḍ̇\tß𐞁\r'VE\r\n12345678'llſ\u000bé३🙂.­'VEꟲ<|fim_prefix|>\u000b㍿9Z㍿\r\n\r\n0ſ >́'Mfi,'S漢\u000b'<|fim_prefix|>", "tokens": 53, "pieces": ["-", "9", "'ll", "‍!", "漢t", "Dž", "<|", "endoftext", "|>'", "VEꟲ", "<|", "fim", "_prefix", "|>", "\u000b", "㍿", "9", "Z", "㍿\r\n\r\n", "0", "ſ", " ", ">́'M", "fi", ",'", "S漢", "\u000b", "'<|", "fim", "_prefix", "|>"]} +{"text": "'D!'ll㍿aZ>­EOT­", "EOT", " \t,'ll<'SⅣ漢'!!(a½#$%é㋿eEOT😀🏽t#$%9 \t٣٤٥٦<|endoftext|>'Re a", "tokens": 63, "pieces": ["EOTs'VE", "#$%<|", "fim", "_prefix", "|>", " ", "\t", ",'", "ll", "<'", "S", "Ⅳ", "漢", "'!!(", "a", "½", "#$%", "é", "㋿e", "EOT", "😀🏽", "t", "#$%", "9", " ", "\t", "٣٤٥", "٦", "<|", "endoftext", "|>'", "Re", " a"]} +{"text": "\"09\r\n'T", "tokens": 4, "pieces": ["\"", "09", "\r\n", "'T"]} +{"text": "́😀🏽EOT\r<|fim_prefix|>Z'T٣٤٥٦é㍿ß
 ½\"''VE
12345678عtZ<|endoftext|>", "tokens": 50, "pieces": ["́", "😀🏽", "EOT", "\r", "<|", "fim", "_prefix", "|>", "Z'T", "٣٤٥", "٦", "é", "㍿ß", "
", " ", "½", "\"''", "VE", "
", "123", "456", "78", "عt", "Z", "<|", "endoftext", "|>"]} +{"text": "d!< \n㋿t😀🏽12345678㋿…e<(‍>ع'Tſ", "tokens": 30, "pieces": ["d", "!<", " \n", "㋿t", "😀🏽", "123", "456", "78", "㋿", "…e", "<(‍>", "ع'T", "ſ"]} +{"text": "🙂", "tokens": 1, "pieces": ["🙂"]} +{"text": "dfi​'DⅣéZ- \nſ", "tokens": 13, "pieces": ["dfi", "​'", "D", "Ⅳ", "é", "Z", "-", " \n", "ſ"]} +{"text": "'Tm👍🏽0🙂12345678́!!½s \n٣٤٥٦\re#$%㋿漢EOT", "tokens": 29, "pieces": ["'Tm", "👍🏽", "0", "🙂", "123", "456", "78", "́", "!!", "½", "s", " \n", "٣٤٥", "٦", "\r", "e", "#$%㋿", "漢", "EOT"]} +{"text": "㍿mméZ-ع,DžDž'S½\r\n\r\n'T㋿\"tm<ꟲ'Re\r\n\r\nEOT<|endoftext|>​", "tokens": 41, "pieces": ["㍿mmé", "Z", "-ع", ",<", "EOT", ">DžDž'S", "½", "\r\n\r\n", "'T", "㋿\"", "tm", "<ꟲ'Re", "\r\n\r\n", "EOT", "<|", "endoftext", "|>​"]} +{"text": "d\t𐞁m\tém३½👍🏽 \"\tⅣ-'ſ'M", "tokens": 23, "pieces": ["d", "\t𐞁m", "\tém", "३½", "👍🏽", " ", " \"", "\t", "Ⅳ", "-'", "ſ'M"]} +{"text": "  m३!!'Re ́e!!ßḍ̇ß İ\n'Té<|endoftext|>9 ", "tokens": 32, "pieces": [" ", " m", "३", "!!'", "Re", " ́", "e", "!!", "ßḍ̇ß", " İ", "\n", "'Té", "<|", "endoftext", "|>", "9", " "]} +{"text": "ḍ̇s\r\r\n\r\n<\t\n🙂>0ß'Re\r\n\r\nAEOTdAt9'T.😀🏽!fi'llé<|endoftext|>'T‍ꟲ
", "tokens": 51, "pieces": ["ḍ̇s", "\r\r\n\r\n", "<", "\t\n", "🙂>", "0", "ß'Re", "\r\n\r\n", "AEOTd", "At", "9", "'T", ".😀🏽<", "META", "_START", ">!", "fi'll", "é", "<|", "endoftext", "|><", "EOT", ">'", "T", "‍ꟲ", "
"]} +{"text": "\n\r\n#$% ́\u000bm.\t<|endoftext|> ꟲa A­㍿́㋿(́'ll'😀🏽t​", "tokens": 48, "pieces": ["\n\r\n", "#$%", " ́", "\u000b", "m", ".", "\t", "<|", "endoftext", "|>", " ꟲa", " ", " A", "­㍿́㋿(́'", "ll", "'<", "META", "_START", ">😀🏽", "t", "​"]} +{"text": "ſ\"½é>‍#$%🙂're ½ \n<👍🏽am漢­.12345678 \n\t'ſ", "tokens": 29, "pieces": ["ſ", "\"", "½", "é", ">‍#$%🙂'", "re", " ", "½", " \n", "<👍🏽", "am漢", "­.", "123", "456", "78", " \n", "\t", "'ſ"]} +{"text": "㍿'VEå'VE \nEOT <|fim_prefix|>\nåaå'S\n\r‍", "tokens": 27, "pieces": ["㍿'", "VEå'VE", " \n", "EOT", " ", " <|", "fim", "_prefix", "|>\n", "åaå'S", "\n\r", "‍"]} +{"text": "'Re'll#$%!-'s<­d'Re​a0😀🏽 'S漢\t", "tokens": 24, "pieces": ["'Re'll", "#$%!-<", "META", "_START", ">'", "s", "<­", "d'Re", "​a", "0", "😀🏽", " '", "S漢", "\t"]} +{"text": "fiDž😀🏽m''Re!! t>a'å\t(́ݽꟲ-DžAt㍿İ's😀🏽9‍m\ta \"", "tokens": 41, "pieces": ["fi", "Dž", "😀🏽", "m", "''", "Re", "!!", " t", ">a", "'å", "\t", "(́", "İ", "½", "ꟲ", "-DžAt", "㍿İ's", "😀🏽", "9", "‍m", "\ta", " ", "\""]} +{"text": "­'D're", "tokens": 4, "pieces": ["­'", "D're"]} +{"text": "mḍ̇9ḍ̇fi'M", "tokens": 17, "pieces": ["m", "ḍ̇", "9", "ḍ̇fi'M", ""]} +{"text": "́ \n㍿\r\n\r\n👍🏽👍🏽!!Aa㋿½Dže­🙂\r\n\r\n'T !!dZt👍🏽\u000b>ع'S\"mt", "tokens": 47, "pieces": ["́", " \n", "㍿\r\n\r\n", "👍🏽👍🏽!!", "Aa", "㋿", "½", "Dže", "­🙂\r\n\r\n", "'T", " ", "!!", "d", "Zt", "👍🏽", "\u000b", ">", "ع'S", "\"mt"]} +{"text": "👍🏽#$%\t Z<|endoftext|>'Reḍ̇,0😀🏽! \n!́½\r\n\r\n\tſ'​ꟲ\"\u000bß‍EOT\"​(ß👍🏽", "tokens": 49, "pieces": ["👍🏽#$%", "\t ", " Z", "<|", "endoftext", "|>'", "Reḍ̇", ",", "0", "😀🏽!", " \n", "!́", "½", "\r\n\r\n", "\tſ", "'​", "ꟲ", "\"", "\u000bß", "‍EOT", "\"​(", "ß", "👍🏽"]} +{"text": "
'Sé‍Ⅳ<|fim_prefix|>漢9\r\n\r\n​ßs#$%- …​漢1234567812345678-Ⅳ#$%,,㍿é'saß(字'ſ\u000b", "tokens": 57, "pieces": ["
", "'", "Sé", "‍", "Ⅳ", "<|", "fim", "_prefix", "|>", "漢", "9", "\r\n\r\n", "​ßs", "#$%-", " ", "…", "​漢", "123", "456", "781", "234", "567", "8", "-", "Ⅳ", "#$%,,㍿<", "META", "_START", ">é's", "aß", "(字'ſ", "\u000b"]} +{"text": "(d\r\n\r\n,'Re\u000bꟲ'VE <|endoftext|>
ḍ̇'ſ 'Re\r\n\r\nİ字👍🏽'M漢 ٣٤٥٦'漢‍ſ​ \n<|fim_prefix|>", "tokens": 57, "pieces": ["(d", "\r\n\r\n", ",'", "Re", "\u000bꟲ'VE", " ", "<|", "endoftext", "|>", "
ḍ̇'ſ", " ", " <", "META", "_START", ">'", "Re", "\r\n\r\n", "İ字", "👍🏽'", "M漢", " ", " ", "٣٤٥", "٦", "'漢", "‍ſ", "​", " \n", "<|", "fim", "_prefix", "|>"]} +{"text": " ३", "tokens": 2, "pieces": [" ", "३"]} +{"text": "​'S<|fim_prefix|>s🙂0-'ſع­t\"㋿ \r\n", "tokens": 22, "pieces": ["​'", "S", "<|", "fim", "_prefix", "|>", "s", "🙂", "0", "-'", "ſع", "­t", "\"㋿", " \r\n"]} +{"text": "ⅣEOT ", "tokens": 8, "pieces": ["Ⅳ", "EOT", "", " "]} +{"text": "EOT", "tokens": 2, "pieces": ["EOT"]} +{"text": "a'ſ12345678🙂漢​ſḍ̇Z-‍m\n'ſDž", "tokens": 22, "pieces": ["a'ſ", "123", "456", "78", "🙂漢", "​ſḍ̇", "Z", "-‍", "m", "\n", "'ſ", "Dž"]} +{"text": "EOT字'VE", "tokens": 9, "pieces": ["EOT", "字'VE"]} +{"text": "­'re\"\r\n𐞁ع.a'DEOT\"‍\u000b'Re", "tokens": 22, "pieces": ["­'", "re", "\"\r", "\n", "𐞁ع", ".a'D", "EOT", "\"‍", "\u000b", "'Re"]} +{"text": "…,e'ſd''VEA٣٤٥٦​DždEOT漢'ſ'll\rſ㍿", "tokens": 34, "pieces": ["…", ",e'ſ", "d", "''", "VE", "A", "٣٤٥", "٦", "​", "Džd", "EOT漢'ſ", "'ll", "\r", "ſ", "㍿"]} +{"text": "㋿,", "tokens": 4, "pieces": ["㋿,"]} +{"text": "ḍ̇㍿٣٤٥٦<|fim_prefix|>t \"İ''Tfí‍å‍'retd'İé\"d", "tokens": 37, "pieces": ["ḍ̇", "㍿", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "t", " \"", "İ", "''", "Tfí", "‍å", "‍'", "ret", "d", "'İé", "\"d"]} +{"text": "'s🙂'D 
عé 's'VE.İ'İ..12345678A\r㋿é ½'ſ'VE-ꟲ #$%<|endoftext|>", "tokens": 67, "pieces": ["'s", "🙂'", "D", " ", "
", "عé", " ", " '", "s'VE", ".İ", "'İ", "..", "123", "456", "78", "A", "\r", "㋿é", " ", "½", "'ſ'VE", "-ꟲ", " ", " #$%<|", "endoftext", "|>"]} +{"text": "mA🙂​㍿漢'll( eAß12345678\t#$%३t's 'T!!!!><😀🏽漢Dž're", "tokens": 37, "pieces": ["m", "A", "🙂​㍿", "漢", "'", "ll", "(", " e", "Aß", "123", "456", "78", "\t", "#$%", "३", "t's", " ", "'T", "!!!!><😀🏽", "漢", "Dž're"]} +{"text": "('S'å's\n‍Ⅳ½\r's'VE 字", "tokens": 16, "pieces": ["('", "S", "'å's", "\n", "‍", "Ⅳ½", "\r", "'s'VE", " ", " 字"]} +{"text": "fi'D㋿", "tokens": 5, "pieces": ["fi'D", "㋿"]} +{"text": "'re", "tokens": 1, "pieces": ["'re"]} +{"text": "'D9…٣٤٥٦!!t
🙂字🙂漢-e'Re 'M​\u000b'ſ", "tokens": 23, "pieces": ["'D", "9", "…", "٣٤٥", "٦", "!!", "t", "
", "🙂字", "🙂漢", "-e'Re", " ", "'M", "​", "\u000b", "'ſ"]} +{"text": "\nḍ̇fiEOT12345678㍿👍🏽㋿Džع\u000b३<'re𐞁'VE'Mßḍ̇\r>𐞁 90'T½ع𐞁 ‍é<|endoftext|>'reDž३12345678.", "tokens": 70, "pieces": ["\n", "ḍ̇fi", "EOT", "123", "456", "78", "㍿👍🏽㋿", "Džع", "\u000b", "३", "<'", "re𐞁'VE", "'Mßḍ̇", "\r", ">𐞁", " ", "90", "'T", "½", "ع𐞁", " ", "‍é", "<|", "endoftext", "|>'", "re", "Dž", "३12", "345", "678", "."]} +{"text": " e'Re​é9㍿s​'Mİ,Ⅳ12345678ع'sAs Ⅳ\t.12345678", "tokens": 35, "pieces": [" e'Re", "​é", "9", "㍿<", "EOT", ">s", "​'", "Mİ", ",", "Ⅳ12", "345", "678", "ع's", "As", " ", "Ⅳ", "\t", ".", "123", "456", "78"]} +{"text": "'EOT\"…𐞁9é🙂're9,>a,'VEmꟲ-ß٣٤٥٦", "tokens": 30, "pieces": ["'EOT", "\"", "…𐞁", "9", "é", "🙂'", "re", "9", ",>", "a", ",'", "VEmꟲ", "-ß", "٣٤٥", "٦"]} +{"text": "ꟲ­ !!­'S's😀🏽…­Z#$%'re sDž'll12345678'Sعs٣٤٥٦́३t­ ", "tokens": 40, "pieces": ["ꟲ", "­", " ", "!!­'", "S's", "😀🏽", "…", "­Z", "#$%'", "re", " s", "Dž'll", "123", "456", "78", "'Sعs", "٣٤٥", "٦", "́", "३", "t", "­", " "]} +{"text": "s\"é<|endoftext|>👍🏽!漢>ḍ̇\u000b\tⅣ\u000b12345678𐞁\r\n字\u000bm!㍿('VE'ſ0 \n㍿​'M\r\n\r\n'M12345678fi㋿tⅣ", "tokens": 64, "pieces": ["s", "\"é", "<|", "endoftext", "|>👍🏽!", "漢", ">ḍ̇", "\u000b", "\t", "Ⅳ", "\u000b", "123", "456", "78", "𐞁", "\r\n", "字", "\u000bm", "!㍿('", "VE'ſ", "0", " \n", "㍿​'", "M", "\r\n\r\n", "'M", "123", "456", "78", "fi", "㋿t", "Ⅳ"]} +{"text": "12345678\n(ß㋿Á12345678'S½<|endoftext|>", "tokens": 25, "pieces": ["123", "456", "78", "\n", "(ß", "㋿Á", "123", "456", "78", "'S", "½", "<|", "endoftext", "|>"]} +{"text": "‍㍿", "tokens": 4, "pieces": ["‍㍿"]} +{"text": "é#$%9s\r\nſ'VE<|endoftext|>'ll👍🏽sfi'Re'reå", "tokens": 27, "pieces": ["é", "#$%", "9", "s", "\r\n", "ſ'VE", "<|", "endoftext", "|>'", "ll", "👍🏽", "sfi'Re", "'reå"]} +{"text": "Ⅳꟲåå(å<|fim_prefix|>!sİ!\t \n'ſ'ſ३ éDž
́é㍿Dž\r\ńs٣٤٥٦<|endoftext|><|fim_prefix|>", "tokens": 58, "pieces": ["Ⅳ", "ꟲåå", "(å", "<|", "fim", "_prefix", "|>!", "s", "İ", "!", "\t \n", "'ſ'ſ", "३", " é", "Dž", "
́é", "㍿Dž", "\r\n", "́s", "٣٤٥", "٦", "<|", "endoftext", "|><|", "fim", "_prefix", "|>"]} +{"text": "𐞁\n½٣٤٥٦A㋿0\"å!fi字́'ll ", "tokens": 23, "pieces": ["𐞁", "\n", "½٣٤", "٥٦", "A", "㋿", "0", "\"å", "!fi字́'ll", " "]} +{"text": "ꟲⅣ-ꟲaé漢­!'Sḍ̇EOT!İas", "tokens": 23, "pieces": ["ꟲ", "Ⅳ", "-ꟲaé漢", "­!'", "Sḍ̇", "EOT", "!İas"]} +{"text": "Dž'T<|fim_prefix|>字㋿́漢'Dm12345678 ", "tokens": 21, "pieces": ["Dž'T", "<|", "fim", "_prefix", "|>", "字", "㋿́漢'D", "m", "123", "456", "78", " "]} +{"text": ">", "tokens": 1, "pieces": [">"]} +{"text": "'re>'m#$%…!!𐞁\t\n9ſ㍿EOTİ'漢", "tokens": 23, "pieces": ["'re", ">'", "m", "#$%", "…", "!!", "𐞁", "\t\n", "9", "ſ", "㍿EOTİ", "'漢"]} +{"text": "s#$% fi<|fim_prefix|>\u000b 漢A😀🏽ßfi'M'T'T'MZßAꟲ>\n0 !#$%٣٤٥٦\r\nDž'VE٣٤٥٦é'll'ret", "tokens": 55, "pieces": ["s", "#$%", " ", " fi", "<|", "fim", "_prefix", "|>", "\u000b", " 漢", "A", "😀🏽", "ßfi'M", "'T'T", "'MZß", "Aꟲ", ">\n", "0", " !#$%", "٣٤٥", "٦", "\r\n", "Dž'VE", "٣٤٥", "٦", "é'll", "'ret"]} +{"text": "0…'T's\u000bZfi", "tokens": 8, "pieces": ["0", "…", "'T's", "\u000bZfi"]} +{"text": "e漢dé\r\n\r\n'M", "tokens": 10, "pieces": ["e漢", "dé", "\r\n\r\n", "'M"]} +{"text": "😀🏽", "tokens": 3, "pieces": ["😀🏽"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "-é'S<Ⅳ३ ع\tſ!! \n", "tokens": 12, "pieces": ["-é'S", "<", "Ⅳ३", " ع", "\tſ", "!!", " \n"]} +{"text": "<", "tokens": 1, "pieces": ["<"]} +{"text": "Z㋿0!!½'ll('Tt'remDžEOT9>'S \u000bDž. \u000bⅣfia३'re漢'M'Re", "tokens": 36, "pieces": ["Z", "㋿", "0", "!!", "½", "'ll", "('", "Tt're", "m", "DžEOT", "9", ">'", "S", " ", "\u000bDž", ".", " ", "\u000b", "Ⅳ", "fia", "३", "'re漢'M", "'Re"]} +{"text": "😀🏽!‍>'D \na‍ A", "tokens": 12, "pieces": ["😀🏽!‍>'", "D", " \n", "a", "‍", " A"]} +{"text": "Z<|fim_prefix|>𐞁'llé", "tokens": 17, "pieces": ["Z", "<|", "fim", "_prefix", "|>", "𐞁", "'", "llé"]} +{"text": "AEOT<Dž's👍🏽'VE'VEİ\r\n\r\nEOT! \n३9t", "tokens": 23, "pieces": ["AEOT", "<Dž's", "👍🏽'", "VE'VE", "İ", "\r\n\r\n", "EOT", "!", " \n", "३9", "t"]} +{"text": "\r\n½\ne", "tokens": 4, "pieces": ["\r\n", "½", "\n", "e"]} +{"text": "'DEOTeꟲs\rEOT㋿­'ſ12345678'", "tokens": 26, "pieces": ["'", "DEOTeꟲs", "\r", "EOT", "㋿­<", "EOT", ">'", "ſ", "123", "456", "78", "'"]} +{"text": "­ſ٣٤٥٦å12345678", "tokens": 11, "pieces": ["­ſ", "٣٤٥", "٦", "å", "123", "456", "78"]} +{"text": "12345678'VEA½sDž字", "tokens": 11, "pieces": ["123", "456", "78", "'VEA", "½", "s", "Dž字"]} +{"text": "३'M
(…ZꟲⅣ'VE字½𐞁 ‍", "tokens": 25, "pieces": ["३", "'M", "
", "(", "…Zꟲ", "Ⅳ", "'VE字", "½", "𐞁", " ", "‍"]} +{"text": "३'(é𐞁३!½'Sfié!#$%'ſ㍿ß '(ſ'Refi,!𐞁'llt'ſ 🙂 ß😀🏽𐞁és#$%'", "tokens": 50, "pieces": ["३", "'(", "é𐞁", "३", "!", "½", "'Sfié", "!#$%'", "ſ", "㍿ß", " ", "'(", "ſ'Re", "fi", ",!", "𐞁'll", "t'ſ", " ", "🙂", " ß", "😀🏽", "𐞁és", "#$%'"]} +{"text": "👍🏽𐞁é\r\n\r\n!!'ReſßعZ'T!!İ字ſEOT́-é३ 漢'll字漢ſmⅣ,👍🏽m(", "tokens": 41, "pieces": ["👍🏽", "𐞁é", "\r\n\r\n", "!!'", "Reſßع", "Z'T", "!!", "İ字ſ", "EOT́", "-é", "३", " 漢'll", "字漢ſm", "Ⅳ", ",👍🏽", "m", "("]} +{"text": " \t<|fim_prefix|>𐞁'll'Tdꟲḍ̇🙂'T\r\né́e​Ⅳ", "tokens": 32, "pieces": [" ", "\t", "<|", "fim", "_prefix", "|>", "𐞁'll", "'Tdꟲḍ̇", "🙂'", "T", "\r\n", "é́e", "​", "Ⅳ"]} +{"text": "ſZß漢!!ꟲ!!㋿㋿ Ⅳ", "tokens": 22, "pieces": ["ſ", "Z", "ß漢", "!!", "ꟲ", "!!㋿㋿", " ", "Ⅳ"]} +{"text": "!ḍ̇​dsåm\r\n\r\n><|endoftext|>,'Re<|fim_prefix|>'llfis\r\n\r\n½", "tokens": 33, "pieces": ["!ḍ̇", "​dsåm", "\r\n\r\n", "><|", "endoftext", "|>,<", "EOT", ">'", "Re", "<|", "fim", "_prefix", "|>'", "llfis", "\r\n\r\n", "½"]} +{"text": "tEOT'D('Re'M><|endoftext|>\u000b‍,0İ's're'T\tA", "tokens": 23, "pieces": ["t", "EOT'D", "('", "Re'M", "><|", "endoftext", "|>", "\u000b", "‍,", "0", "İ's", "'re'T", "\tA"]} +{"text": "㋿'M'll-ḍ̇é
9Ⅳfi!!><|endoftext|>é字 #$%​0'Re!!İ'll9<字<…ſå", "tokens": 49, "pieces": ["㋿'", "M'll", "-ḍ̇é", "
", "", "9Ⅳ", "fi", "!!><|", "endoftext", "|>", "é字", " ", "#$%​", "0", "'Re", "!!", "İ'll", "9", "<字", "<", "…ſå"]} +{"text": "Z'\u000b٣٤٥٦ſ
\u000b''D'D‍m½\nZ……<'T >ém'MaA\u000b\r\n \n å\u000bå½\r#$%9", "tokens": 48, "pieces": ["Z", "'", "\u000b", "٣٤٥", "٦", "ſ", "
", "\u000b", "''", "D'D", "‍m", "½", "\n", "Z", "…", "…", "<'", "T", " ", " >", "ém'M", "a", "A", "\u000b\r\n \n", " ", " å", "\u000bå", "", "½", "\r", "#$%", "9"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "😀🏽İDž٣٤٥٦\r're''S\r😀🏽é.0s🙂'ع½ \n's\"(😀🏽\u000b'ſ!!😀🏽İZ漢ſ عé-", "tokens": 50, "pieces": ["😀🏽", "İDž", "٣٤٥", "٦", "\r", "'", "re", "''", "S", "\r", "😀🏽", "é", ".", "0", "s", "🙂'", "ع", "½", " \n", "'s", "\"(😀🏽", "\u000b", "'ſ", "!!😀🏽", "İZ漢ſ", " عé", "-"]} +{"text": "\r\nEOT
å(٣٤٥٦\t#$%'lld ٣٤٥٦9#$%㋿- ,
 \n\rtḍ̇>​Z", "tokens": 47, "pieces": ["\r\n", "EOT", "
å", "(", "٣٤٥", "٦", "\t", "#$%'", "lld", " ", "٣٤٥", "٦9", "#$%㋿-", " ", " ,", "
 \n\r", "tḍ̇", ">​", "Z", ""]} +{"text": "9​ ſ𐞁Ⅳ'ſꟲDž😀🏽عßZ㋿afi…'re'lla\t …字́३Ⅳ'st", "tokens": 44, "pieces": ["9", "​", " ", " ſ𐞁", "Ⅳ", "'ſꟲ", "Dž", "😀🏽", "عß", "Z", "㋿afi", "…", "'re'll", "a", "\t ", "…字́", "३Ⅳ", "'st"]} +{"text": "👍🏽!!\n\nⅣAm,٣٤٥٦,A,Zß'ſ 12345678\n9ḍ̇Ⅳß \n🙂é 👍🏽<|fim_prefix|><|endoftext|>", "tokens": 51, "pieces": ["👍🏽!!\n\n", "Ⅳ", "Am", ",", "٣٤٥", "٦", ",A", ",Zß'ſ", " ", "123", "456", "78", "\n", "9", "ḍ̇", "Ⅳ", "ß", " \n", "🙂é", " ", " 👍🏽<|", "fim", "_prefix", "|><|", "endoftext", "|>"]} +{"text": ".३'S'S\"a'D­٣٤٥٦
'llAA𐞁0字9", "tokens": 22, "pieces": [".", "३", "'S'S", "\"a'D", "­", "٣٤٥", "٦", "
", "'ll", "AA𐞁", "0", "字", "9"]} +{"text": "\r\n'!!é'll<|endoftext|>\t'MⅣt漢‍‍𐞁A'ſe>​ EOT'Reſ'-𐞁Z's", "tokens": 45, "pieces": ["\r\n", "'!!", "é'll", "<|", "endoftext", "|>", "\t", "'M", "Ⅳ", "t漢", "‍‍", "𐞁", "A'ſ", "e", ">​", " EOT'Re", "ſ", "'-", "𐞁", "Z's"]} +{"text": "\t<|endoftext|>👍🏽ß'Re漢a9>éå'D12345678İDž…9\t㍿At", "tokens": 37, "pieces": ["\t", "<|", "endoftext", "|>👍🏽", "ß'Re", "漢a", "9", ">éå'D", "123", "456", "78", "İDž", "…", "9", "\t", "㍿At"]} +{"text": "漢Dž!é'M‍'VEꟲ\t㋿'ſ…'så're㋿İ9fi!!<|endoftext|>", "tokens": 38, "pieces": ["漢", "Dž", "!é'M", "‍'", "VEꟲ", "\t", "㋿'", "ſ", "…", "'så're", "㋿İ", "9", "fi", "!!<|", "endoftext", "|>"]} +{"text": "́
fi'Rem\u000b!å'Reſ㋿ …", "tokens": 20, "pieces": ["́", "
fi'Re", "m", "\u000b", "!å'Re", "ſ", "㋿", " …"]} +{"text": "<|endoftext|>a(,'DDž\r\n 'T'Så‍ a ㍿\r<|endoftext|>", "tokens": 35, "pieces": ["<|", "endoftext", "|>", "a", "(,'", "DDž", "\r\n", " '", "T'S", "å", "‍", " a", " ", " ㍿\r", "<|", "endoftext", "|>"]} +{"text": "‍<<|fim_prefix|>9", "tokens": 8, "pieces": ["‍<<|", "fim", "_prefix", "|>", "9"]} +{"text": "​‍ع٣٤٥٦ḍ̇👍🏽!३ Dž​
\n(字 ٣٤٥٦t,\r\n'M'T's'Dfi𐞁-ḍ̇dß𐞁", "tokens": 51, "pieces": ["​‍", "ع", "٣٤٥", "٦", "ḍ̇", "👍🏽!", "३", " Dž", "​", "
\n", "(字", " ", "٣٤٥", "٦", "t", ",\r\n", "'M'T", "'s'D", "fi𐞁", "-", "ḍ̇dß𐞁"]} +{"text": "́'reⅣß'll👍🏽\r\n\r\n​👍🏽'\r\n\r\nß㍿", "tokens": 22, "pieces": ["́", "'", "re", "Ⅳ", "ß'll", "👍🏽\r\n\r\n", "​👍🏽'\r\n\r\n", "ß", "㍿"]} +{"text": "å… 'M,eع'Ḿ\r‍🙂\" å‍İع𐞁m'TtEOT'S  'D", "tokens": 33, "pieces": ["å", "…", " ", "'M", ",eع'M", "́", "\r", "‍🙂\"", " å", "‍İع𐞁m'T", "t", "EOT'S", " ", " ", "'D"]} +{"text": "'ReA…,'s\n ", "tokens": 8, "pieces": ["'Re", "A", "…", ",'", "s", "\n", " "]} +{"text": "aé é're👍🏽a\r\n\r\n字ḍ̇‍12345678('Re's0‍>éé,", "tokens": 33, "pieces": ["aé", " ", " é're", "👍🏽", "a", "\r\n\r\n", "字ḍ̇", "‍", "123", "456", "78", "('", "Re's", "0", "‍>", "éé", ","]} +{"text": "0́…½ße‍a𐞁٣٤٥٦<|endoftext|>
", "tokens": 24, "pieces": ["0", "́", "…", "½", "ße", "‍a𐞁", "٣٤٥", "٦", "<|", "endoftext", "|>", "
"]} +{"text": "fimİ<|endoftext|>'M<|fim_prefix|>३ \n
.字Aİt漢é'Re12345678", "tokens": 32, "pieces": ["fim", "İ", "<|", "endoftext", "|>'", "M", "<|", "fim", "_prefix", "|>", "३", " \n", "
", ".字Aİt漢é'Re", "123", "456", "78"]} +{"text": "٣٤٥٦ſ A\u000bİſ're<|fim_prefix|>ع'll…\r\n\r\n!!'Dİ9'Mعåd३ 0‍‍\r\n\r\nd'D­", "tokens": 46, "pieces": ["٣٤٥", "٦", "ſ", " ", " A", "\u000bİſ're", "<|", "fim", "_prefix", "|>", "ع'll", "…\r\n\r\n", "!!'", "Dİ", "9", "'M", "عåd", "३", " ", "0", "‍‍\r\n\r\n", "d'D", "­"]} +{"text": "ꟲe0\r😀🏽 \nḍ̇'S
<|endoftext|>EOT\u000b!!Aİ\n'Reſ", "tokens": 31, "pieces": ["ꟲe", "0", "\r", "😀🏽", " \n", "ḍ̇'S", "
", "<|", "endoftext", "|>", "EOT", "\u000b", "!!", "Aİ", "\n", "'Reſ"]} +{"text": "('T \n𐞁 \nå👍🏽aꟲ!!­'ſ\r\n\r\na'VE‍'Se㍿\"<|fim_prefix|>>İ'S,", "tokens": 40, "pieces": ["('", "T", " \n", "𐞁", " \n", "å", "👍🏽", "aꟲ", "!!­'", "ſ", "\r\n\r\n", "a'VE", "‍'", "Se", "㍿\"<|", "fim", "_prefix", "|>>", "İ'S", ","]} +{"text": "'T'T👍🏽…<|fim_prefix|>'VE\té㋿", "tokens": 19, "pieces": ["'T'T", "👍🏽", "…", "<|", "fim", "_prefix", "|>'", "VE", "\té", "㋿"]} +{"text": "­'re…'T,a…३½ \n-\r \na'VE-ꟲ", "tokens": 22, "pieces": ["­'", "re", "…", "'T", ",a", "…", "३½", " \n", "-\r", " \n", "a'VE", "-ꟲ"]} +{"text": "İ!!é \nⅣ\n…\r\n\r\n
m ", "tokens": 18, "pieces": ["İ", "!!", "é", " \n", "", "Ⅳ", "\n…\r\n\r\n", "
m", " "]} +{"text": "ع😀🏽'ſ㋿.t åfiZ", "tokens": 15, "pieces": ["ع", "😀🏽'", "ſ", "㋿.", "t", " åfi", "Z"]} +{"text": "'s\r\n!td'M'D!​́ \u000b \n'll😀🏽'Re!!>́9́", "9", "EOT'sع\r\n\r\ndꟲé<㍿efi 🙂'T…'Re", "tokens": 43, "pieces": [".-", "İé", "#$%", "é", "-!!", "e", " 漢", "­", " ", "\u000b", ">EOT's", "ع", "\r\n\r\n", "dꟲé", "<㍿", "efi", " ", "🙂'", "T", "…", "'", "Re"]} +{"text": "\"\u000b're!!\nsſ👍🏽e
…ſ<|endoftext|>'D́s!!é9३\"㍿,
é😀🏽Ⅳ\t<\net-漢fiDžt", "tokens": 50, "pieces": ["\"", "\u000b", "'re", "!!\n", "sſ", "👍🏽", "e", "
", "…ſ", "<|", "endoftext", "|>'", "D́s", "!!", "é", "9३", "\"㍿,", "
é", "😀🏽", "Ⅳ", "\t", "<\n", "et", "-漢fi", "Džt"]} +{"text": "Ⅳ३㋿'ll'S𐞁 fi\u000b<|fim_prefix|>漢'ſ#$% \nm t‍'VE<|endoftext|> ३ 
", "tokens": 44, "pieces": ["Ⅳ३", "㋿'", "ll'S", "𐞁", " fi", "\u000b", "<|", "fim", "_prefix", "|>", "漢'ſ", "#$%", " \n", "m", " t", "‍'", "VE", "<|", "endoftext", "|>", " ", "३", " 
"]} +{"text": "< ꟲ…'s#$%\u000b🙂İs\r\n\r\n㋿dßm>\"​\r\n\r\nⅣ'D𐞁-​'Śtß\r­!漢d…३👍🏽", "tokens": 55, "pieces": ["<", " ", " ꟲ", "…", "'s", "#$%", "\u000b", "🙂İs", "\r\n\r\n", "㋿", "dßm", ">\"​\r\n\r\n", "Ⅳ", "'D𐞁", "-​'", "Śtß", "\r", "­<", "EOT", ">!", "漢d", "…", "३", "👍🏽"]} +{"text": "٣٤٥٦㋿0am<|fim_prefix|>åé
#$%Ⅳ9eⅣ12345678 \n…mßعe㍿ع'", "tokens": 48, "pieces": ["٣٤٥", "٦", "㋿", "0", "am", "<|", "fim", "_prefix", "|>", "åé", "", "
", "#$%", "Ⅳ9", "e", "Ⅳ12", "345", "678", " \n", "…mßعe", "㍿ع", "'"]} +{"text": "<½'Re'm12345678\nm \n漢Džſ", "tokens": 14, "pieces": ["<", "½", "'Re'm", "123", "456", "78", "\n", "m", " \n", "漢Džſ"]} +{"text": "漢<|endoftext|> 'S'sꟲZꟲas<|endoftext|>A㍿́fiſ0'M​912345678ſ", "tokens": 40, "pieces": ["漢", "<|", "endoftext", "|>", " ", "'S's", "ꟲZꟲas", "<|", "endoftext", "|>", "A", "㍿́fiſ", "0", "'M", "​", "912", "345", "678", "ſ"]} +{"text": " 'D३éDžİ'Sd
EOT. t", "tokens": 15, "pieces": [" '", "D", "३", "é", "Džİ'S", "d", "
EOT", ".", " t"]} +{"text": "'ll😀🏽'DfiZ'VE  -ꟲ
\rꟲe…
", "tokens": 25, "pieces": ["'ll", "😀🏽'", "Dfi", "Z'VE", " ", " ", "-ꟲ", "
\r", "ꟲe", "…
"]} +{"text": "३!\u000bds'll'M\r\n", "tokens": 10, "pieces": ["३", "!", "\u000bds'll", "'M", "\r\n"]} +{"text": "𐞁'll½ḍ̇", "tokens": 9, "pieces": ["𐞁'll", "½", "ḍ̇"]} +{"text": "!!ꟲß ß'D(12345678👍🏽\td \r\n\r\nİAſ½dd㋿😀🏽 é0éå<|endoftext|>é३", "tokens": 50, "pieces": ["!!", "ꟲß", " ", " ß'D", "(", "123", "456", "78", "👍🏽", "\td", " ", " <", "META", "_START", ">\r\n\r\n", "İAſ", "½", "dd", "㋿😀🏽", " é", "0", "éå", "<|", "endoftext", "|>", "é", "३"]} +{"text": "''llma㍿'s<|fim_prefix|>ꟲDž\n!​(-#$%…Z👍🏽eé tꟲ.\"\r…Dž😀🏽'll ", "tokens": 50, "pieces": ["''", "llma", "㍿'", "s", "<|", "fim", "_prefix", "|>", "ꟲ", "Dž", "\n", "!​(-#$%", "…Z", "👍🏽", "eé", " tꟲ", ".\"\r", "…Dž", "😀🏽'", "ll", " "]} +{"text": "\r\nt'Re३'re!!\"㍿éß'M\r\n\r\n'Rea½㍿'M<|endoftext|>'ſ'D'Mß🙂Zd字'Re9e \nmé'!!'Re's", "tokens": 47, "pieces": ["\r\n", "t'Re", "३", "'re", "!!\"㍿", "éß'M", "\r\n\r\n", "'Rea", "½", "㍿'", "M", "<|", "endoftext", "|>'", "ſ'D", "'Mß", "🙂Zd字'Re", "9", "e", " \n", "mé", "'!!'", "Re's"]} +{"text": "ß>'Refi12345678ßꟲé🙂\"'S٣٤٥٦!३!𐞁\u000b 👍🏽", "tokens": 32, "pieces": ["ß", ">'", "Refi", "123", "456", "78", "ßꟲé", "🙂\"'", "S", "٣٤٥", "٦", "!", "३", "!𐞁", "\u000b ", " 👍🏽"]} +{"text": "'SZ­t. (𐞁 \n'Reé
EOT\r'll…\r\n㍿ſ!!!İ ſ,\r\nfi\r㍿㍿\"<|fim_prefix|>ع", "tokens": 50, "pieces": ["'SZ", "­t", ".", " ", "(𐞁", " \n", "'Reé", "
EOT", "\r", "'ll", "…\r\n", "㍿ſ", "!!!", "İ", " ", " ſ", ",\r\n", "fi", "\r", "㍿㍿\"<|", "fim", "_prefix", "|>", "ع", ""]} +{"text": "#$%>é!!", "tokens": 6, "pieces": ["#$%>", "é", "!!"]} +{"text": "İḍ̇́9m'T12345678'S12345678s½३́é\r\n\r\n👍🏽'ſ㋿ꟲd 'Reḍ̇‍́😀🏽'Dß're EOT! \r\n\r\n", "tokens": 56, "pieces": ["İḍ̇́", "9", "m'T", "123", "456", "78", "'S", "123", "456", "78", "s", "½३", "́é", "\r\n\r\n", "👍🏽'", "ſ", "㋿ꟲd", " ", "'Reḍ̇", "‍́", "😀🏽'", "Dß're", " EOT", "!", " \r\n\r\n"]} +{"text": ">漢ß🙂're‍'VE!!<|fim_prefix|>ḍ̇", "tokens": 19, "pieces": [">漢ß", "🙂'", "re", "‍'", "VE", "!!<|", "fim", "_prefix", "|>", "ḍ̇"]} +{"text": "'lld'Re!𐞁>!漢Dž12345678\t漢٣٤٥٦㍿", "tokens": 33, "pieces": ["'lld'Re", "!𐞁", ">!", "漢", "Dž", "123", "456", "78", "\t漢", "٣٤٥", "٦", "㍿"]} +{"text": "é<|fim_prefix|>Dž٣٤٥٦½👍🏽½'Re9'll𐞁d12345678́ſa ß>\",'TaA \n😀🏽ß!<\r\n\r\nå \n漢<|endoftext|>'llA", "tokens": 64, "pieces": ["é", "<|", "fim", "_prefix", "|>", "Dž", "٣٤٥", "٦½", "👍🏽", "½", "'Re", "9", "'ll𐞁d", "123", "456", "78", "́ſa", " ", " ß", ">\",'", "Ta", "A", " \n", "😀🏽", "ß", "!<\r\n\r\n", "å", " \n", "漢", "<|", "endoftext", "|>'", "ll", "A"]} +{"text": "­\t Z<|endoftext|>EOTé(amd'M㋿ 字a \nḍ̇ é'VEſAḍ̇'ſß'ſ' <'s'D<", "tokens": 47, "pieces": ["­", "\t", " Z", "<|", "endoftext", "|>", "EOTé", "(amd'M", "㋿", " 字a", " \n", "ḍ̇", " é'VE", "ſ", "Aḍ̇'ſ", "ß'ſ", "'", " ", "<'", "s'D", "<"]} +{"text": "!!m𐞁", "tokens": 6, "pieces": ["!!", "m𐞁"]} +{"text": "Z३\u000b\r'am! \n,<|endoftext|>㋿🙂EOT㋿字#$%!…٣٤٥٦字<|fim_prefix|>!\rſ'S'D½​½ſ>\u000b
\r\n\r\n0", "tokens": 53, "pieces": ["Z", "३", "\u000b\r", "'am", "!", " \n", ",<|", "endoftext", "|>㋿🙂", "EOT", "㋿字", "#$%!", "…", "٣٤٥", "٦", "字", "<|", "fim", "_prefix", "|>!\r", "ſ'S", "'D", "½", "​", "½", "ſ", ">", "\u000b
\r\n\r\n", "0"]} +{"text": "\r🙂'D!!'lla-­🙂\u000b.'S ​dfi\rꟲİ㋿-\nⅣ\n é 0", "tokens": 33, "pieces": ["\r", "🙂'", "D", "!!'", "lla", "-­🙂", "\u000b", ".'", "S", " ", "​dfi", "\r", "ꟲ", "İ", "㋿-\n", "Ⅳ", "\n", " é", " ", "0"]} +{"text": "\n'll\r'ſ\n're0👍🏽d", "tokens": 15, "pieces": ["\n", "'ll", "\r", "'ſ", "\n", "'re", "0", "👍🏽", "d"]} +{"text": "!!A

😀🏽\r\n३'M'Reꟲ㍿éDž, A<|endoftext|>", "tokens": 31, "pieces": ["!!", "A", "
", "
", "😀🏽\r\n", "३", "'M'Re", "ꟲ", "㍿é", "Dž", ",", " ", " A", "<|", "endoftext", "|>"]} +{"text": "𐞁0-'ſ'VE<|fim_prefix|>a", "tokens": 16, "pieces": ["𐞁", "0", "-'", "ſ'VE", "<|", "fim", "_prefix", "|>", "a"]} +{"text": "éA😀🏽
EOT㍿ 😀🏽'VE! ½å'lléⅣ, <\u000b#$%漢.\tts\"'T<|fim_prefix|>\r,ßé'SédADž", "tokens": 57, "pieces": ["é", "A", "😀🏽", "
EOT", "㍿", " ", "😀🏽'", "VE", "!", " ", " ", "½", "å'll", "é", "Ⅳ", ",", " ", "<", "\u000b", "#$%", "漢", ".", "\tts", "\"'", "T", "<|", "fim", "_prefix", "|>\r", ",ßé'S", "éd", "ADž"]} +{"text": "'T३​漢'll0٣٤٥٦ꟲ ", "tokens": 14, "pieces": ["'T", "३", "​漢'll", "0٣٤", "٥٦", "ꟲ", " "]} +{"text": "㍿m,㍿t\t
<|endoftext|>0tt123456780t'ſß're\u000be㍿٣٤٥٦'llⅣ'ſ½dåꟲ३!!9ḍ̇🙂", "tokens": 55, "pieces": ["㍿m", ",㍿", "t", "\t", "
", "<|", "endoftext", "|>", "0", "tt", "123", "456", "780", "t'ſ", "ß're", "\u000be", "㍿", "٣٤٥", "٦", "'ll", "Ⅳ", "'ſ", "½", "dåꟲ", "३", "!!", "9", "ḍ̇", "🙂"]} +{"text": "!!t,'reDž12345678,s \r\n\r\n<|endoftext|>'Sḍ̇A'S,🙂é'VE12345678éß
\r\n\r\n\r\n\r\n'S\r\nDž'T<|fim_prefix|>é३ſⅣ!!३
İ🙂", "tokens": 60, "pieces": ["!!", "t", ",'", "re", "Dž", "123", "456", "78", ",s", " \r\n\r\n", "<|", "endoftext", "|>'", "Sḍ̇", "A'S", ",🙂", "é'VE", "123", "456", "78", "éß", "
\r\n\r\n\r\n\r\n", "'S", "\r\n", "Dž'T", "<|", "fim", "_prefix", "|>", "é", "३", "ſ", "Ⅳ", "!!", "३", "
İ", "🙂"]} +{"text": "'M0\r\n\r\n\r0ßEOT'reA
", "tokens": 11, "pieces": ["'M", "0", "\r\n\r\n\r", "0", "ß", "EOT're", "A", "
"]} +{"text": "å‍A's9'ſ🙂漢's
\u000b\t
a\"👍🏽عع-0Džfi\u000bt", "tokens": 29, "pieces": ["å", "‍A's", "9", "'ſ", "🙂漢's", "
\u000b\t", "
a", "\"👍🏽", "عع", "-", "0", "Džfi", "\u000bt"]} +{"text": "'set", "tokens": 2, "pieces": ["'set"]} +{"text": "…𐞁fiꟲ", "tokens": 10, "pieces": ["…𐞁fiꟲ"]} +{"text": "'re漢\r\n\u000b'Re㍿ee!!fisḍ̇​३\n<|endoftext|>😀🏽,t'ſ", "tokens": 32, "pieces": ["'re漢", "\r\n", "\u000b", "'Re", "㍿ee", "!!", "fisḍ̇", "​", "३", "\n", "<|", "endoftext", "|>😀🏽,", "t'ſ"]} +{"text": "🙂 \n𐞁'T­'re(İ'reDž-'reDžs½\r\n>🙂😀🏽ḍ̇'re\r漢́", "tokens": 34, "pieces": ["🙂", " \n", "𐞁'T", "­'", "re", "(İ're", "Dž", "-'", "re", "Džs", "½", "\r\n", ">🙂😀🏽", "ḍ̇'re", "\r", "漢́"]} +{"text": "'lĺ -d㋿\n!!'Re\r\n.é'D­å𐞁\u000bꟲ\r\n\r\n'S!!<|fim_prefix|>­e٣٤٥٦", "tokens": 41, "pieces": ["'lĺ", " ", " -", "d", "㋿\n", "!!'", "Re", "\r\n", ".é'D", "­å𐞁", "\u000bꟲ", "\r\n\r\n", "'S", "!!<|", "fim", "_prefix", "|>­", "e", "٣٤٥", "٦"]} +{"text": "m'T­>d\u000bA'sⅣ fia🙂\u000b👍🏽fi👍🏽ſtع", "tokens": 25, "pieces": ["m'T", "­>", "d", "\u000bA's", "Ⅳ", " fia", "🙂", "\u000b", "👍🏽", "fi", "👍🏽", "ſtع"]} +{"text": "'!\u000b㋿.9३t\u000ba.'Te 'D٣٤٥٦Ⅳ​ İ\"m 'Re'Re𐞁 <|endoftext|> ㋿\rꟲ😀🏽", "tokens": 57, "pieces": ["'!", "\u000b", "㋿.", "9३", "t", "\u000ba", ".'", "Te", " ", "'D", "٣٤٥", "٦Ⅳ", "​", " İ", "\"m", " ", " '", "Re'Re", "𐞁", " ", "<|", "endoftext", "|>", " ㋿\r", "ꟲ", "😀🏽"]} +{"text": "\n'Re㍿\nm‍'½ḍ̇Dž \n…  ३.é", "tokens": 27, "pieces": ["\n", "'Re", "㍿\n", "m", "‍'", "½", "ḍ̇", "Dž", " \n", "…  ", " ", "३", ".é"]} +{"text": "ḍ̇㍿'S🙂>\"<|endoftext|>A'३é!! <|fim_prefix|>'३\u000b'T\r\" e🙂", "tokens": 44, "pieces": ["ḍ̇", "㍿'", "S", "🙂>\"<|", "endoftext", "|>", "A", "'", "३", "é", "!!<", "EOT", ">", " ", "<|", "fim", "_prefix", "|>'", "३", "\u000b", "'T", "\r", "\"", " ", " e", "🙂"]} +{"text": "३𐞁m㋿AⅣ", "tokens": 12, "pieces": ["३", "𐞁m", "㋿A", "Ⅳ"]} +{"text": "fi ", "tokens": 2, "pieces": ["fi", " "]} +{"text": " s're​EOT'reꟲ½'sfié́", "tokens": 19, "pieces": [" ", " s're", "​EOT're", "ꟲ", "½", "'sfié́"]} +{"text": " \t­'D🙂'ſ漢d", "tokens": 35, "pieces": [" ", "\t", "­'", "D", "🙂'", "ſ", "<", "EOT", ">漢d"]} +{"text": "字ꟲDž́🙂<|endoftext|>tA's12345678­<|fim_prefix|>'ſ́Z👍🏽mAś's'D'VE#$%'VE\téß㍿ß'M", "tokens": 59, "pieces": ["字ꟲDž́", "🙂<|", "endoftext", "|>", "t", "A's", "123", "456", "78", "­<|", "fim", "_prefix", "|>'", "ſ́", "Z", "👍🏽", "m", "As", "́'s", "'", "D'VE", "#$%'", "VE", "\téß", "㍿ß'M"]} +{"text": "0
३'ll🙂'll", "tokens": 7, "pieces": ["0", "
", "३", "'ll", "🙂'", "ll"]} +{"text": "\r!!'T!!ꟲe٣٤٥٦fi<|fim_prefix|>👍🏽-#$%́𐞁\r\n#$%㍿'sé㍿", "tokens": 43, "pieces": ["\r", "!!'", "T", "!!", "ꟲe", "٣٤٥", "٦", "fi", "<|", "fim", "_prefix", "|>👍🏽-#$%́", "𐞁", "\r\n", "#$%㍿'", "sé", "㍿"]} +{"text": "'D", "tokens": 4, "pieces": ["'", "D"]} +{"text": "ꟲ.­Dž's\r\n 'VE \n٣٤٥٦é!!9fi-𐞁\r\u000b12345678\r\n.EOT12345678İ'ſaſA-㍿ \n0", "tokens": 59, "pieces": ["ꟲ", ".­", "Dž's", "\r\n", " '", "VE", " \n", "٣٤٥", "٦", "é", "!!", "9", "fi", "-𐞁", "\r", "\u000b", "123", "456", "78", "\r\n", ".EOT", "", "123", "456", "78", "İ'ſ", "a", "ſ", "A", "-㍿", " \n", "0"]} +{"text": "😀🏽s.​㍿,…å-e'VE9
fi  ", "tokens": 21, "pieces": ["😀🏽", "s", ".​㍿,", "…å", "-e'VE", "9", "
fi", "  "]} +{"text": "\u000b12345678<|endoftext|>", "tokens": 11, "pieces": ["\u000b", "123", "456", "78", "<|", "endoftext", "|>"]} +{"text": "ḍ̇Dž's.\n,㍿𐞁㍿s9'Ta \n<|endoftext|>́👍🏽e'Re'lls \n'll.d
.…m", "tokens": 46, "pieces": ["ḍ̇", "Dž's", ".\n", ",㍿", "𐞁", "㍿s", "9", "'Ta", " \n", "<|", "endoftext", "|>́👍🏽", "e'Re", "'lls", " \n", "'ll", ".d", "
", ".", "…m"]} +{"text": "(㋿#$%'M\r 'DmEOT're'reeå­\u000b,å字aİ👍🏽‍‍fiaḍ̇'re'VE \n,'T!!'så\r", "tokens": 48, "pieces": ["(㋿#$%'", "M", "\r", " ", "'Dm", "EOT're", "'reeå", "­", "\u000b", ",å字", "a", "İ", "👍🏽‍‍", "fiaḍ̇'re", "'VE", " \n", ",'", "T", "!!'", "så", "\r"]} +{"text": ".mİéå😀🏽\r½𐞁ßİ३…<عé'Re\r\r\n\r\n ß​åddع0", "tokens": 34, "pieces": [".m", "İéå", "😀🏽\r", "½", "𐞁ß", "İ", "३", "…", "<عé'Re", "\r\r\n\r\n", " ", " ß", "​åddع", "0"]} +{"text": "#$%\nſd३!!३ḍ̇㋿12345678'll, \n'ſ!! <|fim_prefix|>'ſé㍿,0…'ll३'TDž!٣٤٥٦Z😀🏽00AZ", "tokens": 58, "pieces": ["#$%<", "META", "_START", ">\n", "ſd", "३", "!!", "३", "ḍ̇", "㋿", "123", "456", "78", "'ll", ",", " \n", "'ſ", "!!", " <|", "fim", "_prefix", "|>'", "ſé", "㍿,", "0", "…", "'ll", "३", "'TDž", "!", "٣٤٥", "٦", "Z", "😀🏽", "00", "AZ"]} +{"text": "\t'DZ(><,!!عdDžm\nat-👍🏽\tß😀🏽३­́½'Sfimع", "tokens": 42, "pieces": ["\t", "'DZ", "(><,!!", "عd", "Džm", "\n", "at", "-👍🏽", "\tß", "😀🏽", "३", "­́", "½", "'Sfi", "漢", "mع"]} +{"text": "ß٣٤٥٦ꟲ fi
a", "tokens": 11, "pieces": ["ß", "٣٤٥", "٦", "ꟲ", " fi", "
a"]} +{"text": " 'Ḿ​t é㋿Ⅳ𐞁 ſ漢d‍fim<|fim_prefix|>", "tokens": 30, "pieces": [" ", "'Ḿ", "​t", " é", "㋿", "Ⅳ", "𐞁", " ſ漢d", "‍fim", "<|", "fim", "_prefix", "|>"]} +{"text": "'Re('S\n", "tokens": 4, "pieces": ["'Re", "('", "S", "\n"]} +{"text": "Dž👍🏽Ⅳ\u000bfi'VE㍿at 字eḍ̇\"字…\t字 é'Reſ", "tokens": 34, "pieces": ["Dž", "👍🏽", "Ⅳ", "\u000bfi'VE", "㍿at", "", " ", " 字eḍ̇", "\"字", "…", "\t字", " é'Re", "ſ"]} +{"text": "'D<|fim_prefix|>d𐞁\r'll'Re٣٤٥٦३㍿'ll('ſ́\u000bs…'S\r\n\r\n👍🏽'D'<'VE'SZ09'D'S㋿<|endoftext|>", "tokens": 57, "pieces": ["'D", "<|", "fim", "_prefix", "|>", "d𐞁", "\r", "'ll'Re", "٣٤٥", "٦३", "㍿'", "ll", "('", "ſ́", "\u000bs", "…", "'S", "\r\n\r\n", "👍🏽'", "D", "'<'", "VE'S", "Z", "09", "'D'S", "㋿<|", "endoftext", "|>"]} +{"text": "ḍ̇0‍🙂​​'Re0㍿>", "tokens": 14, "pieces": ["ḍ̇", "0", "‍🙂​​'", "Re", "0", "㍿>"]} +{"text": "é'M\n\r'M'DA\"", "tokens": 9, "pieces": ["é'M", "\n\r", "'M'D", "A", "\""]} +{"text": "e㍿' EOTéåDž fiß's'll🙂\u000b'Tİ\r\nİ!", "tokens": 27, "pieces": ["e", "㍿'", " EOTéå", "Dž", " fiß", "'", "s'll", "🙂", "\u000b", "'Tİ", "\r\n", "İ", "!"]} +{"text": "\r0'VE
'VEZعs \t'VE㋿İ ḍ̇́-té😀🏽A'ſ\n\r \n0é'S​", "tokens": 42, "pieces": ["\r", "0", "'VE", "", "
", "'VEZعs", " ", "\t", "'VE", "㋿İ", " ḍ̇́", "-té", "😀🏽", "A'ſ", "\n\r \n", "0", "é'S", "​"]} +{"text": " ſt \r\nś'M'D", "tokens": 9, "pieces": [" ſt", " \r\n", "ś'M", "'D"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "9'M漢漢३!!é½''Dß字'll''ſ\r㍿Ⅳ", "tokens": 21, "pieces": ["9", "'M漢漢", "३", "!!", "é", "½", "''", "Dß字'll", "''", "ſ", "\r", "㍿", "Ⅳ"]} +{"text": "́", "tokens": 1, "pieces": ["́"]} +{"text": "->\r\n\r\n­'Re​ꟲ<|endoftext|>‍m…‍㋿0
eDž'S字\"<|endoftext|>Ⅳé㍿‍ \nع'M' >́'VEfi", "tokens": 55, "pieces": ["->\r\n\r\n", "­'", "Re", "​ꟲ", "<|", "endoftext", "|>‍", "m", "…", "‍㋿", "0", "
e", "Dž'S", "字", "\"<|", "endoftext", "|>", "Ⅳ", "é", "㍿‍", " \n", "ع'M", "'", " ", ">́'VE", "fi"]} +{"text": "0㍿'é're.!!ß0", "tokens": 12, "pieces": ["0", "㍿'", "é're", ".!!", "ß", "0"]} +{"text": "'ſ'T🙂 'M\u000b½a'ſ‍字!!m(!!😀🏽ع", "tokens": 32, "pieces": ["'ſ'T", "🙂", " ", " '", "M", "", "\u000b", "½", "a'ſ", "‍<", "EOT", "><", "EOT", ">字", "!!", "m", "(!!😀🏽", "ع"]} +{"text": "\r\ń12345678Ⅳ😀🏽'Re😀🏽<|fim_prefix|>漢", "tokens": 25, "pieces": ["\r\n", "́", "123", "456", "78", "", "Ⅳ", "😀🏽'", "Re", "😀🏽<|", "fim", "_prefix", "|>", "漢"]} +{"text": "\u000b'Re-<|endoftext|>𐞁-👍🏽'D\u000b!!🙂\u000b\r", "tokens": 28, "pieces": ["\u000b", "'Re", "-<|", "endoftext", "|>", "𐞁", "-👍🏽'", "D", "\u000b", "!!🙂", "\u000b\r"]} +{"text": "'s\t#$%EOT0漢. .(t‍s>-'re\rİ12345678EOT-'", "re", "\r", "İ", "123", "456", "78", "EOT", "'ll👍🏽>,", "tokens": 17, "pieces": [",t", "㍿‍\r\n\r\n", "m", "…", "'", "ll", "👍🏽>,"]} +{"text": "'ll>३\r\r\n\r\ń", "tokens": 6, "pieces": ["'ll", ">", "३", "\r\r\n\r\n", "́"]} +{"text": "​", "tokens": 1, "pieces": ["​"]} +{"text": "'D३😀🏽<|fim_prefix|><'T\r\n\r\nDž!\"tDž\r\n漢. (a<|fim_prefix|>\r\n\r\n-㋿\n.\né<|fim_prefix|>DžⅣ字ad!!٣٤٥٦", "tokens": 55, "pieces": ["'D", "३", "😀🏽<|", "fim", "_prefix", "|><'", "T", "\r\n\r\n", "Dž", "!\"", "t", "Dž", "\r\n", "漢", ".", " ", "(a", "<|", "fim", "_prefix", "|>\r\n\r\n", "-㋿\n", ".\n", "é", "<|", "fim", "_prefix", "|>", "Dž", "Ⅳ", "字ad", "!!", "٣٤٥", "٦"]} +{"text": "\r", "tokens": 1, "pieces": ["\r"]} +{"text": "!!", "tokens": 1, "pieces": ["!!"]} +{"text": ".𐞁​㍿\"e३'M \ń \nfifi\t 😀🏽mt <|fim_prefix|>", "tokens": 33, "pieces": [".𐞁", "​㍿\"", "e", "३", "'M", " \n", "́", "", " \n", "fifi", "\t ", " 😀🏽", "mt", " <|", "fim", "_prefix", "|>"]} +{"text": "Z'MamA 'ſd½㋿㋿a\"d字 Ⅳ<|endoftext|>́İ'ſ'VE'ſ\n12345678㍿ 👍🏽𐞁İ", "tokens": 56, "pieces": ["Z'M", "am", "A", " ", "'ſd", "½", "㋿㋿<", "META", "_START", ">a", "\"d字", " ", " ", "Ⅳ", "<|", "endoftext", "|>́", "İ'ſ", "'VE'ſ", "\n", "123", "456", "78", "㍿", " ", "👍🏽", "𐞁", "İ"]} +{"text": "٣٤٥٦!Zſꟲ-…'ll ", "tokens": 15, "pieces": ["٣٤٥", "٦", "!Zſꟲ", "-", "…", "'ll", " "]} +{"text": "'ll‍ İdİ\naEOTd!\r\n\r\n\r\n\r\n\u000b'M'VEꟲ㍿<|endoftext|>0 \n<|endoftext|>'ſ­㋿'ſ", "tokens": 46, "pieces": ["'ll", "‍", " ", " İd", "İ", "\n", "a", "EOTd", "!\r\n\r\n\r\n\r\n", "\u000b", "'M'VE", "ꟲ", "㍿<|", "endoftext", "|>", "0", " \n", "<|", "endoftext", "|>'", "ſ", "­㋿'", "ſ"]} +{"text": "fi\r\n\r\n…", "tokens": 4, "pieces": ["fi", "\r\n\r\n", "…"]} +{"text": "ſ>'s'Sa9ع٣٤٥٦\r\n\r\n-Ⅳ\nß", "tokens": 17, "pieces": ["ſ", ">'", "s'S", "a", "9", "ع", "٣٤٥", "٦", "\r\n\r\n", "-", "Ⅳ", "\n", "ß"]} +{"text": "٣٤٥٦<|fim_prefix|>0\u000b​(-३é字'S!!!!字fi'ſ<|endoftext|>0 's\u000bZ㍿'Red\r>👍🏽s\n👍🏽'll'll
e", "tokens": 60, "pieces": ["٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "0", "\u000b", "​(-", "३", "é字'S", "!!!!", "字fi'ſ", "<|", "endoftext", "|>", "0", " ", "'s", "\u000bZ", "㍿'", "Red", "\r", ">👍🏽", "s", "\n", "👍🏽'", "ll'll", "
e"]} +{"text": "­t漢 (\nfi >t\rs!!'T're!!½'refi ­­!!\nḍ̇😀🏽#$%­EOT'T>", "tokens": 32, "pieces": ["­t漢", " (\n", "fi", " >", "t", "\r", "s", "!!'", "T're", "!!", "½", "'refi", " ", "­­!!\n", "ḍ̇", "😀🏽#$%­", "EOT'T", ">"]} +{"text": "'S<|fim_prefix|>.㋿\"👍🏽½​ḍ̇\u000b>9", "tokens": 25, "pieces": ["'S", "<|", "fim", "_prefix", "|>.㋿\"👍🏽", "½", "​ḍ̇", "", "\u000b", ">", "9"]} +{"text": "٣٤٥٦😀🏽㍿-0!!'VE''S!!Z(-0'!!́३
", "tokens": 29, "pieces": ["٣٤٥", "٦", "😀🏽㍿-", "0", "!!'", "VE", "''", "S", "!!<", "EOT", ">Z", "(-", "0", "'!!́", "३", "
"]} +{"text": "'ſ㍿!İ're t\r\n‍½ſfi\u000b'Dé!漢­½‍'ſ\"ḍ̇éA<'Ss,", "tokens": 41, "pieces": ["'ſ", "㍿!", "İ're", " t", "\r\n", "‍<", "EOT", ">", "½", "ſfi", "\u000b", "'Dé", "!漢", "­", "½", "‍'", "ſ", "\"ḍ̇é", "A", "<'", "Ss", ","]} +{"text": ".ḍ̇åİ\"!!ḍ̇ꟲ漢  ­A\nſ>३12345678#$%a३\u000b'ſ9,", "tokens": 38, "pieces": [".ḍ̇å", "İ", "\"!!", "ḍ̇ꟲ漢", " ", " ­", "A", "\n", "ſ", ">", "३12", "345", "678", "#$%", "a", "३", "\u000b", "'ſ", "9", ","]} +{"text": "㋿12345678>٣٤٥٦<|fim_prefix|>,'Re<|fim_prefix|>>́😀🏽ADž9'VE \n're", "tokens": 37, "pieces": ["㋿", "123", "456", "78", ">", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>,'", "Re", "<|", "fim", "_prefix", "|>>́😀🏽", "ADž", "9", "'VE", " \n", "'re"]} +{"text": "Džſ­< \n \rme12345678dعſ­#$%-ḿ…\" (Dž 's‍fi!٣٤٥٦<|endoftext|>\r\n'll
… \n're", "tokens": 53, "pieces": ["Džſ", "­<", " \n \r", "me", "123", "456", "78", "dعſ", "­#$%-", "ḿ", "…", "\"", " ", "(Dž", " ", " '", "s", "‍fi", "!", "٣٤٥", "٦", "<|", "endoftext", "|>\r\n", "'ll", "
… \n", "'re"]} +{"text": "İſ,㍿!a0'M‍'ll漢'VE<|endoftext|>Dž३…!Dž ḍ̇'Dé'll🙂ع३aåß \n,½'Re\r\n\r\n漢'll'ß", "tokens": 56, "pieces": ["İſ", ",㍿!", "a", "0", "'M", "‍'", "ll漢", "'", "VE", "<|", "endoftext", "|>", "Dž", "३", "…", "!Dž", " ḍ̇'D", "é'll", "🙂ع", "३", "aåß", " \n", ",", "½", "'Re", "\r\n\r\n", "漢'll", "'ß"]} +{"text": "٣٤٥٦A㋿EOT​\r\n\r\n\"‍'Sa'VE٣٤٥٦Z!ꟲ🙂 'ſ'll", "tokens": 31, "pieces": ["٣٤٥", "٦", "A", "㋿EOT", "​\r\n\r\n", "\"‍'", "Sa'VE", "٣٤٥", "٦", "Z", "!ꟲ", "🙂", " '", "ſ'll"]} +{"text": "\t(s(", "tokens": 3, "pieces": ["\t", "(s", "("]} +{"text": "ADž𐞁a-😀🏽é\r 漢
​'\u000b𐞁EOTḍ̇'VE…>㍿\t
字🙂!'s­A#$%'VE漢0ß'M're㍿", "tokens": 60, "pieces": ["ADž𐞁a", "-😀🏽", "é", "\r", " 漢", "
", "​'", "\u000b", "𐞁EOTḍ̇'VE", "…", ">㍿", "\t", "
字", "🙂!'", "s", "­A", "#$%'", "VE漢", "0", "ß'M", "'re", "㍿"]} +{"text": "-!🙂'll٣٤٥٦ '.㍿Dž!!eſe", "tokens": 20, "pieces": ["-!🙂'", "ll", "٣٤٥", "٦", " ", "'.㍿", "Dž", "!!", "eſe"]} +{"text": " 😀🏽9İ\t㋿…ß m\t", "tokens": 14, "pieces": [" 😀🏽", "9", "İ", "\t", "㋿", "…ß", " m", "\t"]} +{"text": "…㋿🙂tm🙂字>EOT 漢'T­\r\r\n\r\n ٣٤٥٦éé", "tokens": 24, "pieces": ["…", "㋿🙂", "tm", "🙂字", ">EOT", " 漢'T", "­\r\r\n\r\n", " ", "٣٤٥", "٦", "éé"]} +{"text": "'ſ\r漢
<|endoftext|>😀🏽\rå😀🏽<|endoftext|>a‍<|fim_prefix|> ع\u000bå", "tokens": 44, "pieces": ["'ſ", "\r", "漢", "
", "<|", "endoftext", "|>😀🏽\r", "å", "😀🏽<", "EOT", "><|", "endoftext", "|>", "a", "‍<|", "fim", "_prefix", "|>", " ع", "\u000bå"]} +{"text": "ß'T'🙂EOTZdEOT🙂 ½!!'D<|endoftext|>fi0\" \n㋿٣٤٥٦-é\"'VE 'S", "tokens": 40, "pieces": ["ß'T", "'🙂", "EOTZd", "EOT", "🙂", " ", " ", "½", "!!'", "D", "<|", "endoftext", "|>", "fi", "0", "\"", " \n", "㋿", "٣٤٥", "٦", "-é", "\"'", "VE", " ", "'S"]} +{"text": "İ\n's12345678ß<|fim_prefix|>\r\n\r\ns漢 -٣٤٥٦#$% 'Re0!!!㋿\tDž", "tokens": 43, "pieces": ["İ", "\n", "'s", "123", "456", "78", "ß", "<|", "fim", "_prefix", "|>\r\n\r\n", "s漢", " -", "٣٤٥", "٦", "#$%", " ", " '", "Re", "0", "!!!㋿", "\t", "Dž", ""]} +{"text": "\"́ع'T\rḍ̇é㍿👍🏽ḍ̇\r\n\r\n​🙂's#$%,½'T<|endoftext|>ßDž ३'M​å<|endoftext|>''T", "tokens": 56, "pieces": ["\"́ع'T", "\r", "ḍ̇é", "㍿👍🏽", "ḍ̇", "\r\n\r\n", "​🙂'", "s", "#$%,", "½", "'T", "<|", "endoftext", "|>", "ß", "Dž", " ", "३", "'M", "​å", "<|", "endoftext", "|>''", "T"]} +{"text": "12345678ééİ Dž<|endoftext|>!!𐞁३㋿👍🏽漢‍<|fim_prefix|><|endoftext|>s", "tokens": 49, "pieces": ["123", "456", "78", "éé", "İ", " ", " Dž", "<|", "endoftext", "|>!!", "𐞁", "", "३", "㋿👍🏽", "漢", "‍<|", "fim", "_prefix", "|><|", "endoftext", "|>", "s"]} +{"text": "㍿🙂A<­> éع123456780\r\n𐞁12345678'VE'S漢<|fim_prefix|>éDžİ'D're…Zeé>٣٤٥٦ \t'VE>fiéß", "tokens": 61, "pieces": ["㍿🙂", "A", "<­>", " éع", "123", "456", "780", "\r\n", "𐞁", "123", "456", "78", "'VE'S", "漢", "<|", "fim", "_prefix", "|><", "EOT", ">é", "Džİ'D", "'re", "…Zeé", ">", "٣٤٥", "٦", " ", "\t", "'VE", ">fiéß"]} +{"text": "👍🏽a'VE9ſعDž ٣٤٥٦­Dž\r\n३fiß‍‍'De<|fim_prefix|>ſع漢<|fim_prefix|>\"­A", "tokens": 52, "pieces": ["👍🏽", "a'VE", "9", "ſ", "ع", "Dž", " ", " ", "٣٤٥", "٦", "­Dž", "\r\n", "३", "fiß", "‍‍'", "De", "<|", "fim", "_prefix", "|>", "ſع漢", "<|", "fim", "_prefix", "|>\"<", "META", "_START", ">­", "A"]} +{"text": "å,\r\n\r\nع½㋿\r\né३-'M<|fim_prefix|>'M­ 'Téßİ'st३'s\"tA\r\n\r\nd\u000b'sḍ̇ſ٣٤٥٦\r\n", "tokens": 47, "pieces": ["å", ",\r\n\r\n", "ع", "½", "㋿\r\n", "é", "३", "-'", "M", "<|", "fim", "_prefix", "|>'", "M", "­", " '", "Téß", "İ's", "t", "३", "'s", "\"t", "A", "\r\n\r\n", "d", "\u000b", "'sḍ̇ſ", "٣٤٥", "٦", "\r\n"]} +{"text": "ß 'M#$%'s ३㍿ sdDž'll'sEOT
\r\n\r\n<|endoftext|> ‍\u000bEOT\u000b\n㍿Dž éEOT'S㋿ \né", "tokens": 53, "pieces": ["ß", " '", "M", "#$%'", "s", " ", "३", "㍿", " sd", "Dž'll", "'s", "EOT", "
\r\n\r\n", "<|", "endoftext", "|>", " ", "‍", "\u000bEOT", "\u000b\n", "㍿Dž", " <", "EOT", ">é", "EOT'S", "㋿", " \n", "é"]} +{"text": "٣٤٥٦åſ ‍és'ſ'lĺ12345678'T#$%٣٤٥٦!ع ́­", "tokens": 34, "pieces": ["٣٤٥", "٦", "åſ", " ", "‍és'ſ", "'lĺ", "123", "456", "78", "'T", "#$%<", "META", "_START", ">", "٣٤٥", "٦", "!ع", " ", " ́", "­"]} +{"text": ">EOT'reꟲß\nſ'S's'llm\u000b👍🏽'㋿'lĺ\t㋿Ⅳ>漢👍🏽 A", "tokens": 38, "pieces": [">EOT're", "ꟲß", "\n", "ſ'S", "'s'll", "m", "\u000b", "👍🏽'㋿'", "lĺ", "\t", "㋿", "Ⅳ", ">漢", "👍🏽", " ", " A"]} +{"text": "'re٣٤٥٦EOTEOT<३😀🏽're-Ⅳ'D­ß\té३​漢٣٤٥٦-'S>ḍ̇ſ0sſ9-d0İ", "tokens": 44, "pieces": ["'re", "٣٤٥", "٦", "EOTEOT", "<", "३", "😀🏽'", "re", "-", "Ⅳ", "'D", "­ß", "\té", "३", "​漢", "٣٤٥", "٦", "-'", "S", ">ḍ̇ſ", "0", "sſ", "9", "-d", "0", "İ"]} +{"text": "12345678\tſ", "tokens": 5, "pieces": ["123", "456", "78", "\tſ"]} +{"text": " EOT\"Ⅳ😀🏽'ſ", "tokens": 11, "pieces": [" ", " EOT", "\"", "Ⅳ", "😀🏽'", "ſ"]} +{"text": "½ßs\u000b\"😀🏽'llDžA .9İ'Re#$%12345678é", "tokens": 24, "pieces": ["½", "ßs", "\u000b", "\"😀🏽'", "ll", "DžA", " ", ".", "9", "İ'Re", "#$%", "123", "456", "78", "é"]} +{"text": " 👍🏽\u000b३ !!​!!漢<|fim_prefix|>\n
fi a½\t12345678'Re' \u000b12345678", "tokens": 53, "pieces": ["", " ", " 👍🏽", "\u000b", "", "३", " ", " !!​!!", "漢", "<|", "fim", "_prefix", "|>\n", "
fi", " a", "½", "\t", "123", "456", "78", "'Re", "'", " ", "\u000b", "123", "456", "78", "<", "EOT", ">"]} +{"text": "'re😀🏽EOTß\ns'll\n#$%0<́é>ع<Ⅳ'ع", "tokens": 44, "pieces": ["'re", "😀🏽", "EOTß", "\n", "s'll", "\n", "#$%", "0", "<<", "EOT", ">́é", ">ع", "<", "Ⅳ", "'ع"]} +{"text": "(字as'ſ!ꟲ
9m👍🏽\"ⅣZḿ\"'Tİá12345678Dž'll😀🏽0\nDž!㋿ḍ̇Aa'll'M٣٤٥٦ꟲ 
12345678", "tokens": 53, "pieces": ["Dž", "123", "456", "78", "'re'll", "㋿d", "9", "\"👍🏽<", "META", "_START", ">Dž'll", "😀🏽", "0", "\n", "Dž", "!㋿", "ḍ̇", "Aa'll", "'M", "٣٤٥", "٦", "ꟲ", " ", "
", "123", "456", "78"]} +{"text": "aa‍!Z㍿👍🏽!!>٣٤٥٦ ꟲDžA! ½ꟲéé.Dždafi𐞁Dž­ \n…A-👍🏽, 😀🏽𐞁👍🏽", "tokens": 66, "pieces": ["aa", "‍!", "Z", "㍿👍🏽!!>", "٣٤٥", "٦", " ꟲ", "DžA", "!", " ", "½", "ꟲéé", ".Dždafi", "𐞁", "Dž", "­", " \n", "…A", "-👍🏽,", " ", " 😀🏽", "𐞁", "👍🏽"]} +{"text": "<|endoftext|>½३㋿#$%a9,.åDžDž'S ß漢12345678….'TDž", "tokens": 37, "pieces": ["<|", "endoftext", "|>", "½३", "㋿#$%", "a", "9", ",.", "å", "DžDž'S", " ", " ß漢", "123", "456", "78", "…", ".'", "TDž"]} +{"text": "dⅣfifi'MⅣ३‍'s\n­'ll (!!'s é\"-", "tokens": 25, "pieces": ["d", "Ⅳ", "fifi'M", "Ⅳ३", "‍'", "s", "\n", "­'", "ll", " ", "(!!'", "s", " ", " é", "\"-"]} +{"text": "㋿🙂 ḍ̇#$%\r\n ٣٤٥٦½ꟲ🙂'ſß\u000b\rعa‍<|endoftext|> A \na­Z'll\r\n'T\r\n\r\nsⅣ'ſſEOT #$%​", "tokens": 64, "pieces": ["㋿🙂", " ḍ̇", "#$%\r\n", " ", " ", "٣٤٥", "٦½", "ꟲ", "🙂'", "ſß", "\u000b\r", "عa", "‍<|", "endoftext", "|>", " <", "EOT", ">A", " \n", "a", "­Z'll", "\r\n", "'T", "\r\n\r\n", "s", "Ⅳ", "'ſſ", "EOT", " <", "EOT", ">#$%​"]} +{"text": "漢ع Ⅳ🙂​'ll<|endoftext|> 'ḍ̇​İ…३\r\n٣٤٥٦", "tokens": 31, "pieces": ["漢ع", " ", "Ⅳ", "🙂​'", "ll", "<|", "endoftext", "|>", " ", " '", "ḍ̇", "​İ", "…", "३", "\r\n", "٣٤٥", "٦"]} +{"text": "s​🙂9ſ0\r's👍🏽'VE 9(A", "tokens": 17, "pieces": ["s", "​🙂", "9", "ſ", "0", "\r", "'s", "👍🏽'", "VE", " ", "9", "(A"]} +{"text": ".'S'VE字漢'lĺaDž!!éEOT\u000b< A㋿aådع'SEOT\u000b​­<'Re", "tokens": 37, "pieces": [".'", "S'VE", "字漢'll", "́a", "Dž", "!!", "é", "EOT", "\u000b", "<", " A", "㋿aådع", "'", "SEOT", "\u000b", "​­<'", "Re"]} +{"text": "fi!! ", "tokens": 3, "pieces": ["fi", "!!", " "]} +{"text": "\"‍'ſ'Ma9'VE>\r\n'llfi", "tokens": 17, "pieces": ["\"‍'", "ſ'M", "a", "9", "'", "VE", ">\r\n", "'ll", "fi"]} +{"text": "mß'll'M 9عd9e'retEOT\r\n\r\n३'ſ9
㋿", "tokens": 23, "pieces": ["mß'll", "'M", " ", "9", "عd", "9", "e're", "t", "EOT", "\r\n\r\n", "३", "'ſ", "9", "
", "㋿"]} +{"text": "🙂0!!!‍½'a½½>s字‍dfi-'\r'ſ'reⅣZß'Re…", "tokens": 31, "pieces": ["🙂", "0", "!!!‍", "½", "'a", "½½", ">s字", "‍dfi", "-'\r", "'ſ", "'", "re", "Ⅳ", "Zß'Re", "…"]} +{"text": "!!eⅣ!<|fim_prefix|>", "tokens": 11, "pieces": ["!!", "e", "Ⅳ", "!<|", "fim", "_prefix", "|>"]} +{"text": "a'ſ ", "tokens": 4, "pieces": ["a'ſ", " "]} +{"text": "<|endoftext|>½és're‍", "tokens": 23, "pieces": ["<|", "endoftext", "|>", "½", "és're", "‍"]} +{"text": "'M's​\r­ 're'S'sdfi're'VE(​Ⅳ", "tokens": 18, "pieces": ["'M's", "​\r", "­", " ", "'re'S", "'sdfi're", "'VE", "(​", "Ⅳ"]} +{"text": "
३EOT éfi'll\r!…İ", "tokens": 13, "pieces": ["
", "३", "EOT", " ", " éfi'll", "\r", "!", "…İ"]} +{"text": "'Ts,m", "tokens": 3, "pieces": ["'Ts", ",m"]} +{"text": "𐞁Z12345678é\"\u000b'ſ ́٣٤٥٦ꟲZ", "tokens": 31, "pieces": ["𐞁", "Z", "123", "456", "78", "é", "\"<", "EOT", ">", "\u000b", "'ſ", " ", " <", "EOT", ">́", "٣٤٥", "٦", "ꟲ", "Z"]} +{"text": "'M're Ⅳ𐞁­\"", "tokens": 11, "pieces": ["'M're", " ", "Ⅳ", "𐞁", "­\""]} +{"text": "
㋿\r\n \nés…", "tokens": 9, "pieces": ["
", "㋿\r\n", " \n", "és", "…"]} +{"text": "<(٣٤٥٦.m0字İ're٣٤٥٦m\r\n\r\n\r\n 00Z'M \n'ſZ­Z's㋿ \n🙂 \n\nDž<|fim_prefix|>mİ's<|endoftext|>", "tokens": 51, "pieces": ["<(", "٣٤٥", "٦", ".m", "0", "字", "İ're", "٣٤٥", "٦", "m", "\r\n\r\n\r\n", " ", "00", "Z'M", " \n", "'ſ", "Z", "­Z's", "㋿", " \n", "🙂", " \n\n", "Dž", "<|", "fim", "_prefix", "|>", "m", "İ's", "<|", "endoftext", "|>"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'ll३\u000b<|endoftext|>🙂\"!…Z字'T㋿s 'T字!👍🏽…🙂…㍿,fi\"ḍ̇!ع\nå ​Ⅳ ½\"t", "tokens": 60, "pieces": ["'ll", "३", "\u000b", "<|", "endoftext", "|>🙂\"!", "…Z字'T", "㋿s", "", " ", "'T字", "!👍🏽", "…", "🙂", "…", "㍿,", "fi", "\"ḍ̇", "!ع", "\n", "å", " ​", "Ⅳ", " ", "½", "\"t"]} +{"text": " ꟲ \u000b12345678 \r
👍🏽𐞁", "tokens": 19, "pieces": [" ꟲ", " ", "\u000b", "123", "456", "78", " \r", "
", "👍🏽", "𐞁"]} +{"text": "½9🙂ß", "tokens": 4, "pieces": ["½9", "🙂ß"]} +{"text": "'Re9́<\n漢t\"½#$%\u000b\u000b字٣٤٥٦('M'M#$%'‍ \"'S字​<|fim_prefix|> 'VE\u000b👍🏽'TZ​d", "tokens": 44, "pieces": ["'Re", "9", "́", "<\n", "漢t", "\"", "½", "#$%", "\u000b", "\u000b字", "٣٤٥", "٦", "('", "M'M", "#$%'‍", " ", "\"'", "S字", "​<|", "fim", "_prefix", "|>", " '", "VE", "\u000b", "👍🏽'", "TZ", "​d"]} +{"text": "!\r\n​  0ſ'T'reḍ̇​ ㋿#$%9EOT​", "tokens": 23, "pieces": ["!\r\n", "​", " ", " ", "0", "ſ'T", "'reḍ̇", "​", " ", " ㋿#$%", "9", "EOT", "​"]} +{"text": "'VE'S.\r\n\r\n٣٤٥٦<|fim_prefix|>\r'VEA\nAZ\u000b.", "tokens": 27, "pieces": ["'VE'S", ".\r\n\r\n", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>\r", "'VEA", "\n", "AZ", "\u000b", "."]} +{"text": "'VE!.12345678éA'fiea \nḍ̇\t'Re0'm#$%𐞁!!‍EOT<|endoftext|> A,३Dž", "tokens": 46, "pieces": ["'VE", "!.", "123", "456", "78", "é", "A", "'<", "EOT", ">fiea", " \n", "ḍ̇", "\t", "'Re", "0", "'m", "#$%", "𐞁", "!!‍", "EOT", "<|", "endoftext", "|>", " A", ",", "३", "Dž"]} +{"text": ".9aA!!åé字#$%Ⅳé 
EOT​mm(…‍'字字EOT \n👍🏽\r\n'.'VE🙂", "tokens": 41, "pieces": [".", "9", "a", "A", "!!", "åé字", "#$%", "Ⅳ", "é", " ", "
EOT", "​mm", "(", "…", "‍'", "字字", "EOT", " \n", "👍🏽\r\n", "'.'", "VE", "🙂"]} +{"text": "\r\n👍🏽.sesEOT㋿\"'De!'VE9", "tokens": 18, "pieces": ["\r\n", "👍🏽.", "ses", "EOT", "㋿\"'", "De", "!'", "VE", "9"]} +{"text": "\t!!…'Re<|endoftext|> å12345678Ⅳ漢<'re!!mEOT\tEOTß <|endoftext|>EOTm-𐞁A ḍ̇漢ⅣZ!é'ſ😀🏽EOT", "tokens": 68, "pieces": ["\t", "!!", "…", "'Re", "<|", "endoftext", "|>", " å", "123", "456", "78Ⅳ", "漢", "<'", "re", "!!", "m", "EOT", "\tEOT", "ß", " ", "<|", "endoftext", "|>", "EOTm", "-𐞁", "A", " ḍ̇漢", "Ⅳ", "Z", "!é'ſ", "😀🏽", "EOT"]} +{"text": "\u000b\rm­s'T𐞁Ⅳ're…३é>Ⅳ‍'s'ſ(\r\nfi<字tſEOT½éd́-'reDž\"<|endoftext|>m㋿'M", "tokens": 55, "pieces": ["\u000b\r", "m", "­s'T", "𐞁", "Ⅳ", "'re", "…", "३", "é", ">", "Ⅳ", "‍'", "s'ſ", "(\r\n", "fi", "<字tſ", "EOT", "½", "éd́", "-'", "re", "Dž", "\"<|", "endoftext", "|>", "m", "㋿'", "M"]} +{"text": "Ⅳ#$%'Ree\u000b½<|endoftext|>ß​½A½s 'sEOT𐞁\r\u000b½ḍ̇'٣٤٥٦", "tokens": 40, "pieces": ["Ⅳ", "#$%'", "Ree", "\u000b", "½", "<|", "endoftext", "|>", "ß", "​", "½", "A", "½", "s", " ", "'s", "EOT𐞁", "\r", "\u000b", "½", "ḍ̇", "'", "٣٤٥", "٦"]} +{"text": "ſé٣٤٥٦0‍t­Ⅳ½ꟲ'llⅣe\"½ ", "tokens": 24, "pieces": ["ſé", "٣٤٥", "٦0", "‍t", "­", "Ⅳ½", "ꟲ'll", "Ⅳ", "e", "\"", "½", " "]} +{"text": "
>'T½émå9e>😀🏽ém İéEOT \n#$%Ⅳ'Rea<|endoftext|>å12345678𐞁ZEOT 字t'T", "tokens": 52, "pieces": ["
", ">'", "T", "½", "émå", "9", "e", ">😀🏽", "ém", " ", " İé", "EOT", " \n", "#$%", "Ⅳ", "'Rea", "<|", "endoftext", "|>", "å", "123", "456", "78", "𐞁", "ZEOT", " 字t'T"]} +{"text": "'D.é㋿!\r12345678٣٤٥٦\"fi\ts'M३'Dع㋿\r\u000bⅣm\n-𐞁㍿", "tokens": 54, "pieces": ["字ß'Re", "\n", "𐞁åḍ̇漢å", "\t", "'S", "!!'", "ſ", "9", "<|", "fim", "_prefix", "|>", "\ts'M", "३", "'Dع", "㋿<", "EOT", ">\r", "\u000b", "Ⅳ", "m", "\n", "-𐞁", "㍿"]} +{"text": "\n're'reDžfi
Dž  \r\n", "tokens": 11, "pieces": ["\n", "'re're", "Džfi", "
Dž", "  \r\n"]} +{"text": "!EOT.'ſ \n'T're'VE'Dſ  字 ", "tokens": 22, "pieces": ["!EOT", ".'", "ſ", " \n", "'T", "'", "re'VE", "'Dſ", " ", " 字", " <", "EOT", ">"]} +{"text": "'sm!  t\"٣٤٥٦漢Zé'll 'll!٣٤٥٦ -00㋿́\u000b \n m ́'Ddع .Dž𐞁", "tokens": 44, "pieces": ["'sm", "!", "  ", " t", "\"", "٣٤٥", "٦", "漢Zé'll", " '", "ll", "!", "٣٤٥", "٦", " ", " -", "00", "㋿́", "\u000b \n", " m", " ́'D", "dع", " .", "Dž𐞁"]} +{"text": "'S 'ſ \nt
​fi\nſ㍿9'VE12345678Z㍿<|endoftext|>Ⅳ👍🏽's<\n\r\n\r\n", "tokens": 46, "pieces": ["'S", " ", " '", "ſ", " \n", "t", "
", "​", "fi", "\n", "ſ", "㍿", "9", "'VE", "123", "456", "78", "Z", "㍿<|", "endoftext", "|>", "Ⅳ", "👍🏽'", "s", "<<", "META", "_START", ">\n\r\n\r\n"]} +{"text": "漢👍🏽.Džſ0!!\r\n'T9å åaåEOTİꟲ㍿‍㋿\t\tḍ̇", "tokens": 41, "pieces": ["漢", "👍🏽.", "Džſ", "0", "!!\r\n", "'T", "9", "å", " ", "åaå", "EOTİꟲ", "㍿‍㋿", "\t", "\tḍ̇"]} +{"text": "Z\r\n\r\n\r'ſ'MZ\r,12345678㍿㍿", "tokens": 18, "pieces": ["Z", "\r\n\r\n\r", "'ſ'M", "Z", "\r", ",", "123", "456", "78", "㍿㍿"]} +{"text": " \rd'ſa😀🏽're३<㋿\r\n\r\nꟲZ<🙂's <|fim_prefix|>mꟲ👍🏽字", "tokens": 40, "pieces": [" \r", "d'ſ", "a", "😀🏽'", "re", "३", "<㋿\r\n\r\n", "ꟲ", "Z", "<🙂'", "s", " ", " <|", "fim", "_prefix", "|>", "mꟲ", "👍🏽", "字"]} +{"text": "ßİm(\r\n'lléA're\rs'DDž!fi३عdA'D-
​e㍿!!‍", "tokens": 30, "pieces": ["ß", "İm", "(\r\n", "'llé", "A're", "\r", "s'D", "Dž", "!fi", "३", "عd", "A'D", "-", "
", "​e", "㍿!!‍"]} +{"text": "㍿ꟲ0dßt's漢 \n🙂😀🏽­ꟲ's👍🏽'Dtm<|endoftext|>é٣٤٥٦#$%!字éع'll👍🏽A", "tokens": 57, "pieces": ["㍿ꟲ", "0", "dßt's", "漢", " \n", "🙂😀🏽­", "ꟲ's", "👍🏽'", "Dtm", "<|", "endoftext", "|>", "é", "٣٤٥", "٦", "#$%!<", "EOT", ">字éع", "'", "ll", "👍🏽", "A"]} +{"text": "\u000b<|endoftext|>A‍ß0#$%ⅣtZ>
're Dž-ع!e\tß'ſ é عß,< 
ḍ̇", "tokens": 47, "pieces": ["\u000b", "<|", "endoftext", "|>", "A", "‍ß", "0", "#$%", "Ⅳ", "t", "Z", ">", "
", "'re", " Dž", "-ع", "!e", "\tß'ſ", " é", " ", " ع", "ß", ",<", " ", "
ḍ̇"]} +{"text": "éḍ̇\r\n\t0!'T\ts字㋿Dž\u000bع'reEOT'Re!.Z'ſ", "tokens": 27, "pieces": ["éḍ̇", "\r\n", "\t", "0", "!'", "T", "\ts字", "㋿Dž", "\u000bع're", "EOT'Re", "!.", "Z'ſ"]} +{"text": "ḍ̇ée!㍿\"'D👍🏽\"fidA'M'VE'!!🙂½\r\n\r\n>'s's
Zḍ̇'ll9漢m٣٤٥٦Z#$%(­", "tokens": 50, "pieces": ["ḍ̇é", "e", "!㍿\"'", "D", "👍🏽\"", "fid", "A'M", "'VE", "'!!🙂", "½", "\r\n\r\n", ">'", "s's", "
Zḍ̇'ll", "9", "漢m", "٣٤٥", "٦", "Z", "#$%(­"]} +{"text": "0'lléa😀🏽 \r\n\r\n字… e'Tt> e 'Té​12345678EOTZ㋿
.½0A㍿字𐞁́漢m'sa", "tokens": 49, "pieces": ["0", "'lléa", "😀🏽", " \r\n\r\n", "字", "…", " e'T", "t", ">", " e", " ", " '", "Té", "​", "123", "456", "78", "EOTZ", "㋿", "
", ".", "½0", "A", "㍿字𐞁́漢m's", "a"]} +{"text": "<|fim_prefix|>漢 fi𐞁ع#$%.😀🏽#$%#$%12345678½ꟲ'Re!å'Reع😀🏽\n'sfi\ńA½\t.> 'M's🙂", "tokens": 53, "pieces": ["<|", "fim", "_prefix", "|>", "漢", " fi𐞁ع", "#$%.😀🏽#$%#$%", "123", "456", "78½", "ꟲ'Re", "!å'Re", "ع", "😀🏽\n", "'sfi", "\n", "́", "A", "½", "\t", ".>", " ", "'M's", "🙂"]} +{"text": "½'s
  12345678 \nſEOT\r㍿Zḍ̇!EOTſ'T\r\nß 𐞁", "tokens": 35, "pieces": ["½", "'s", "
 ", " ", "123", "456", "78", " \n", "ſ", "EOT", "\r", "㍿Zḍ̇", "!EOTſ'T", "\r\n", "ß", " 𐞁"]} +{"text": "'ſ<|fim_prefix|>'Rea12345678'll­٣٤٥٦Z0३9'Res", "tokens": 29, "pieces": ["'ſ", "<|", "fim", "_prefix", "|>'", "Rea", "123", "456", "78", "'ll", "­", "٣٤٥", "٦", "Z", "0३9", "'Res", ""]} +{"text": "३12345678e-Zİe Ⅳ'(\r\n\r\n‍\r\n\r\né \n漢㍿\nA㍿å<
 \nEOT­Ⅳ,\"9<|endoftext|>\t9\u000b\r", "tokens": 50, "pieces": ["३12", "345", "678", "e", "-Zİe", " ", " ", "Ⅳ", "'(\r\n\r\n", "‍\r\n\r\n", "é", " \n", "漢", "㍿\n", "A", "㍿å", "<", "
 \n", "EOT", "­", "Ⅳ", ",\"", "9", "<|", "endoftext", "|>", "\t", "9", "\u000b\r"]} +{"text": "é'VE''Ta 
<.­👍🏽\nİ\u000b#$%d٣٤٥٦'ſ'll '漢'Re‍½😀🏽'VE٣٤٥٦", "tokens": 41, "pieces": ["é'VE", "''", "Ta", " ", "
", "<.­👍🏽\n", "İ", "\u000b", "#$%", "d", "٣٤٥", "٦", "'ſ'll", " '", "漢'Re", "‍", "½", "😀🏽'", "VE", "٣٤٥", "٦"]} +{"text": "'Mfi \r\n12345678EOT  ٣٤٥٦éZİſ …,👍🏽!", "tokens": 27, "pieces": ["'Mfi", " \r\n", "123", "456", "78", "EOT", "  ", " ", "٣٤٥", "٦", "é", "Zİſ", " ", "…", ",👍🏽!"]} +{"text": "'🙂😀🏽a\"- 0 '𐞁'VEßé'D३👍🏽İ'D👍🏽ßA'VEt ㋿ \tİé", "tokens": 49, "pieces": ["'🙂😀🏽", "a", "\"-", " ", " ", "0", " ", " <", "EOT", ">'", "𐞁'VE", "ßé'D", "३", "👍🏽", "İ'D", "👍🏽", "ß", "A'VE", "t", " ", "㋿", " ", "\tİé"]} +{"text": "👍🏽aßa\" .å", "tokens": 10, "pieces": ["👍🏽", "aßa", "\"", " .", "å"]} +{"text": " \nꟲ½\te\u000b३ع'Re漢'T9Ⅳ'>३-<|fim_prefix|><­-'D'll'D'sİ<|endoftext|>12345678EOT
", "tokens": 47, "pieces": [" \n", "ꟲ", "½", "\te", "\u000b", "३", "ع'Re", "漢'T", "9Ⅳ", "'>", "३", "-<", "EOT", "><|", "fim", "_prefix", "|><­-'", "D'll", "'D's", "İ", "<|", "endoftext", "|>", "123", "456", "78", "EOT", "
"]} +{"text": "İåeEOT\n‍👍🏽‍😀🏽 \r\n\r\n12345678", "tokens": 19, "pieces": ["İåe", "EOT", "\n", "‍👍🏽‍😀🏽", " \r\n\r\n", "123", "456", "78"]} +{"text": "#$%'ſ\r\n 90😀🏽<|fim_prefix|>'ſ \n \n­㋿då", "tokens": 25, "pieces": ["#$%'", "ſ", "\r\n", " ", " ", "90", "😀🏽<|", "fim", "_prefix", "|>'", "ſ", " \n \n", "­㋿", "då"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'s'<|endoftext|>EOTعdİ\"å\ta", "tokens": 16, "pieces": ["'s", "'<|", "endoftext", "|>", "EOTعd", "İ", "\"å", "\ta"]} +{"text": "'A🙂 \n's\r12345678're", "tokens": 10, "pieces": ["'A", "🙂", " \n", "'s", "\r", "123", "456", "78", "'re"]} +{"text": "ꟲ<'M>", "tokens": 6, "pieces": ["ꟲ", "<'", "M", ">"]} +{"text": "👍🏽fißå<İ㍿#$%'Re٣٤٥٦0́٣٤٥٦Z<|endoftext|>‍<|endoftext|>ß👍🏽ßfi,", "tokens": 52, "pieces": ["👍🏽<", "META", "_START", ">fißå", "<İ", "㍿#$%'", "Re", "٣٤٥", "٦0", "́", "٣٤٥", "٦", "Z", "<|", "endoftext", "|>‍<|", "endoftext", "|>", "ß", "👍🏽", "ßfi", ","]} +{"text": "​<|fim_prefix|>\r'T㋿'VE字.​EOT,'T३", "tokens": 21, "pieces": ["​<|", "fim", "_prefix", "|>\r", "'T", "㋿'", "VE字", ".​", "EOT", ",'", "T", "३"]} +{"text": "\u000b", "tokens": 5, "pieces": ["", "\u000b"]} +{"text": "12345678㋿Džé½>\"!!e'Re…'M٣٤٥٦EOTİ😀🏽", "tokens": 31, "pieces": ["123", "456", "78", "㋿Džé", "½", ">\"!!", "e'Re", "…", "'M", "٣٤٥", "٦", "EOTİ", "😀🏽"]} +{"text": "\"å\u000bⅣ.e é३", "tokens": 9, "pieces": ["\"å", "\u000b", "Ⅳ", ".e", " ", " é", "३"]} +{"text": "👍🏽ZDž<|endoftext|> ", "tokens": 14, "pieces": ["👍🏽", "ZDž", "<|", "endoftext", "|>", " "]} +{"text": "​\r\nḍ̇́å'T Z漢0

'Re\r\nå٣٤٥٦\t½㋿ꟲ'T㋿ ㍿😀🏽 'VEß'refi>\nḍ̇m's12345678-ß\r\n", "tokens": 59, "pieces": ["​\r\n", "ḍ̇́å'T", " Z漢", "0", "
", "
", "'Re", "\r\n", "å", "٣٤٥", "٦", "\t", "½", "㋿ꟲ'T", "㋿", " ", "㍿😀🏽", " '", "VEß're", "fi", ">\n", "ḍ̇m's", "123", "456", "78", "-ß", "\r\n"]} +{"text": "½ 12345678('ſ😀🏽<|endoftext|>𐞁😀🏽\r\n\r\n(!'Sſ<|fim_prefix|>Z's👍🏽 👍🏽,", "tokens": 49, "pieces": ["½", " ", " ", "123", "456", "78", "('", "ſ", "😀🏽<|", "endoftext", "|>", "𐞁", "😀🏽\r\n\r\n", "(!'", "Sſ", "<|", "fim", "_prefix", "|>", "Z's", "👍🏽", " ", "👍🏽,"]} +{"text": "'llⅣ३>½12345678ḍ̇́é>a>'D(\r\n\r\n,12345678'M", "tokens": 25, "pieces": ["'ll", "Ⅳ३", ">", "½12", "345", "678", "ḍ̇́é", ">a", ">'", "D", "(\r\n\r\n", ",", "123", "456", "78", "'M"]} +{"text": "İ字㋿㋿Z'Sع
ḍ̇EOT.'D
㋿e 🙂12345678\u000b,ßå'Reé🙂, \ne'ſ३'s>‍\"ꟲ's'", "tokens": 55, "pieces": ["İ字", "㋿㋿", "Z'S", "ع", "
ḍ̇", "EOT", ".'", "D", "
", "㋿e", " 🙂", "123", "456", "78", "\u000b", ",<", "META", "_START", ">ßå'Re", "é", "🙂,", " \n", "e'ſ", "३", "'s", ">‍\"", "ꟲ's", "'"]} +{"text": "m👍🏽", "tokens": 4, "pieces": ["m", "👍🏽"]} +{"text": "字ⅣDž​ -0漢12345678", "tokens": 13, "pieces": ["字", "Ⅳ", "Dž", "​", " ", " -", "0", "漢", "123", "456", "78"]} +{"text": "'DEOT' \n 0字Z३", "tokens": 10, "pieces": ["'DEOT", "'", " \n", " ", "0", "字", "Z", "३"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "字Ⅳm​d‍½'re👍🏽,\nDžs㍿𐞁Z12345678\r\n", "tokens": 40, "pieces": ["字", "Ⅳ", "m", "​d", "‍", "½", "'re", "👍🏽,\n", "Džs", "㍿𐞁", "Z", "123", "456", "78", "\r\n"]} +{"text": "'ſ<|endoftext|>t'Tdaع👍🏽字\rZ'Mꟲ\r 漢\t0fi<|endoftext|>'MⅣe',३­At­​Ⅳé½><|fim_prefix|>", "tokens": 59, "pieces": ["'ſ", "<|", "endoftext", "|>", "t'T", "daع", "👍🏽", "字", "\r", "Z'M", "ꟲ", "\r", " ", " 漢", "\t", "0", "fi", "<|", "endoftext", "|>'", "M", "Ⅳ", "e", "',", "३", "­At", "­​", "Ⅳ", "é", "½", "><|", "fim", "_prefix", "|>"]} +{"text": "İ'Ré Ⅳ ſ'reZsm\ń 'Tḍ̇'reEOT ​  \n𐞁\r'Sa'Té٣٤٥٦­́٣٤٥٦DžⅣA", "tokens": 51, "pieces": ["İ'Re", "́", " ", "Ⅳ", " ſ're", "Zsm", "\n", "́", " '", "Tḍ̇'re", "EOT", " ", "​", "  \n", "𐞁", "\r", "'", "Sa'T", "é", "٣٤٥", "٦", "­́", "٣٤٥", "٦", "Dž", "Ⅳ", "A"]} +{"text": "…字\u000btå>'T字'VEḍ̇½'", "T字'VE", "ḍ̇", "½", "İ'sé\r\n\r\n­ßs\r\n\r\nfi<9're\n'MéⅣ<|fim_prefix|>#$%.\"'Te३ \r\n\r\n> e٣٤٥٦\u000bed", "tokens": 52, "pieces": ["Ź's", "123", "456", "78", "Z", "İ's", "é", "\r\n\r\n", "­ßs", "\r\n\r\n", "fi", "<", "9", "'re", "\n", "'Mé", "Ⅳ", "<|", "fim", "_prefix", "|>#$%.\"'", "Te", "३", " \r\n\r\n", ">", " e", "٣٤٥", "٦", "\u000bed"]} +{"text": "'s'll ३Džad'll-  ㍿'T'ſ\"Ⅳ 字漢 ३ !!İa\r\n\r\n३0s", "tokens": 35, "pieces": ["'s'll", " ", " ", "३", "Džad'll", "-", " ", " ", "㍿'", "T'ſ", "\"", "Ⅳ", " ", " 字漢", " ", "३", " ", "!!", "İa", "\r\n\r\n", "३0", "s"]} +{"text": " ‍㍿'VEEOT-\t🙂́12345678\r\n\r\n'Re#$%😀🏽​🙂-\r\n\r\n😀🏽're#$%ع,'VEEOT㍿👍🏽é𐞁m'D 👍🏽٣٤٥٦\"(ſ", "tokens": 62, "pieces": [" ", "‍㍿'", "VEEOT", "-", "\t", "🙂́", "123", "456", "78", "\r\n\r\n", "'Re", "#$%😀🏽​🙂-\r\n\r\n", "😀🏽'", "re", "#$%", "ع", ",'", "VEEOT", "㍿👍🏽", "é𐞁m'D", " ", "👍🏽", "٣٤٥", "٦", "\"(", "ſ"]} +{"text": ".a\r\n
'Sꟲ㋿fiDž\n9fi 9🙂'Re👍🏽 d……३", "tokens": 30, "pieces": [".a", "\r\n", "
", "'Sꟲ", "㋿fi", "Dž", "\n", "9", "fi", " ", "9", "🙂'", "Re", "👍🏽", " d", "…", "…", "३"]} +{"text": "d👍🏽fi!m漢漢!!‍EOT\r\n<|fim_prefix|>'Re­'ll㍿ 🙂\r\n\r\n,'Dḍ̇e­ ­étع́'llß<|endoftext|>\r -", "tokens": 59, "pieces": ["d", "👍🏽", "fi", "!m漢", "漢", "!!‍", "EOT", "\r\n", "<|", "fim", "_prefix", "|>'", "Re", "­'", "ll", "㍿", " ", "🙂\r\n\r\n", ",'", "Dḍ̇e", "­", " ", "­étع́'ll", "ß", "<|", "endoftext", "|>\r", " ", "-"]} +{"text": "<|endoftext|>'S!eⅣ\r\naéEOT'ſ\r\n\r\n­<|endoftext|>sß 👍🏽ⅣéEOT​", "tokens": 42, "pieces": ["<|", "endoftext", "|>'", "S", "!e", "Ⅳ", "\r\n", "aé", "EOT'ſ", "\r\n\r\n", "­<|", "endoftext", "|>", "sß", " ", "👍🏽", "Ⅳ", "é", "EOT", "​"]} +{"text": "\ŕ<|fim_prefix|>'D'M 
ḍ̇s \n😀🏽#$%Aéİ ㋿\n ", "tokens": 31, "pieces": ["\r", "́", "<|", "fim", "_prefix", "|>'", "D'M", " ", "
ḍ̇s", " \n", "😀🏽#$%", "Aé", "İ", " ", "㋿\n", " "]} +{"text": "ꟲ\"Zḍ̇ååfiꟲ", "tokens": 16, "pieces": ["ꟲ", "\"Zḍ̇ååfiꟲ"]} +{"text": "\r\n\r\nm'll're'VE\u000bEOTd\t🙂're٣٤٥٦ꟲ漢\t字#$%ḍ̇\"de", "tokens": 34, "pieces": ["\r\n\r\n", "m'll", "'re'VE", "\u000b", "EOTd", "\t", "🙂'", "re", "٣٤٥", "٦", "ꟲ漢", "\t字", "#$%", "ḍ̇", "\"de"]} +{"text": ".漢\t'㋿12345678ßZEOT'VE <\r \n…㍿(fis㍿'VEDž's-ſ12345678٣٤٥٦", "tokens": 44, "pieces": [".漢", "\t", "'㋿", "123", "456", "78", "ß", "ZEOT'VE", " ", "<\r", " \n", "…", "㍿(", "fis", "㍿'", "VEDž's", "-ſ", "123", "456", "78٣", "٤٥٦"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'TEOT\"-ſ㋿ \n㋿", "tokens": 12, "pieces": ["'TEOT", "\"-", "ſ", "㋿", " \n", "㋿"]} +{"text": ".'llåm0'T­‍'VE're m½\ŕ\t", "tokens": 17, "pieces": [".'", "llåm", "0", "'T", "­‍'", "VE're", " ", " m", "½", "\r", "́", "\t"]} +{"text": "'Re😀🏽३'Mt", "tokens": 11, "pieces": ["'Re", "😀🏽", "३", "'Mt", ""]} +{"text": "ⅣZå\r!('llDž\r\n
ßſ're 𐞁'sꟲ", "tokens": 24, "pieces": ["Ⅳ", "Zå", "\r", "!('", "ll", "Dž", "\r\n", "
ßſ're", " 𐞁's", "ꟲ"]} +{"text": "👍🏽d́.\r\n३ \n\" \t😀🏽e Z'ḍ̇EOT👍🏽>", "tokens": 27, "pieces": ["👍🏽", "d́", ".\r\n", "३", " \n", "\"", " ", "\t", "😀🏽", "e", " ", " Z", "'ḍ̇", "EOT", "👍🏽>"]} +{"text": "9m㋿12345678's'S'Ts'Re'Re\t<|fim_prefix|>.\"\r\n", "tokens": 22, "pieces": ["9", "m", "㋿", "123", "456", "78", "'s'S", "'Ts'Re", "'Re", "\t", "<|", "fim", "_prefix", "|>.\"\r\n"]} +{"text": "EOT!字,'ſ字\r😀🏽éİ#$%٣٤٥٦३m🙂\".é'S🙂ḍ̇ſ \n \n\" 9EOTe", "tokens": 45, "pieces": ["EOT", "!字", ",'", "ſ字", "\r", "😀🏽", "é", "İ", "#$%", "٣٤٥", "٦३", "m", "🙂\"<", "META", "_START", ">.", "é'S", "🙂ḍ̇ſ", " \n \n", "\"", " ", "9", "EOTe"]} +{"text": " \na\r\n\r\n9'll<'VE'Re漢
­\r\nⅣ\u000b㋿字🙂(>A", "tokens": 27, "pieces": [" \n", "a", "\r\n\r\n", "9", "'", "ll", "<'", "VE'Re", "漢", "
", "­\r\n", "Ⅳ", "\u000b", "㋿字", "🙂(>", "A"]} +{"text": "㋿ſ Ⅳ ><|fim_prefix|>
Z🙂
d\t", "tokens": 23, "pieces": ["㋿", "ſ", " ", "Ⅳ", " ", "><|", "fim", "_prefix", "|>", "
Z", "🙂", "
d", "\t"]} +{"text": "
((½Ⅳ 𐞁‍#$%(é'T<#$%éEOTZ12345678's\n‍\u000b.ſe<|endoftext|>㋿​́>㋿<|fim_prefix|>
字Dž٣٤٥٦", "tokens": 66, "pieces": ["
", "((", "½Ⅳ", " 𐞁", "‍#$%(", "é'T", "<#$%", "é", "EOTZ", "123", "456", "78", "'s", "\n", "‍", "\u000b", ".ſe", "<|", "endoftext", "|>㋿​́>㋿<|", "fim", "_prefix", "|>", "
字", "Dž", "٣٤٥", "٦"]} +{"text": "­漢EOT'D३é \r\n\r\n‍'s㍿ \n'll\n12345678ßEOTtd İع‍e#$% İ'VEßA", "tokens": 37, "pieces": ["­漢", "EOT'D", "३", "é", " \r\n\r\n", "‍'", "s", "㍿", " \n", "'ll", "\n", "123", "456", "78", "ß", "EOTtd", " İع", "‍e", "#$%", " ", " İ'VE", "ß", "A"]} +{"text": "fi'S,\r\n\"​<😀🏽", "tokens": 9, "pieces": ["fi'S", ",\r\n", "\"​<😀🏽"]} +{"text": "ſ­ EOT'ſ
'Re👍🏽😀🏽Dž\u000bad🙂s", "tokens": 24, "pieces": ["ſ", "­", " EOT'ſ", "", "
", "'Re", "👍🏽😀🏽", "Dž", "\u000bad", "🙂s"]} +{"text": "<|endoftext|>İ\n…-👍🏽㍿m's ­Džt <|fim_prefix|>12345678'T<|fim_prefix|>dḍ̇a", "tokens": 47, "pieces": ["<|", "endoftext", "|>", "İ", "\n", "…", "-👍🏽㍿", "m's", " ", "­Džt", " ", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'T", "<|", "fim", "_prefix", "|>", "dḍ̇a"]} +{"text": "('D\r'ſ", "tokens": 5, "pieces": ["('", "D", "\r", "'ſ"]} +{"text": "!!aée\"漢 漢 .㍿ ", "tokens": 17, "pieces": ["!!", "aé", "e", "\"漢", " ", " 漢", " .㍿", " "]} +{"text": "å½­12345678\t,ع,as'VE\r ٣٤٥٦ḍ̇३ \n㍿٣٤٥٦ ſå½𐞁ßḍ̇\n12345678ǻ", "tokens": 53, "pieces": ["å", "½", "­", "123", "456", "78", "\t", ",ع", ",as'VE", "\r", " ", " ", "٣٤٥", "٦", "ḍ̇", "३", " \n", "㍿", "٣٤٥", "٦", " ſå", "½", "𐞁ßḍ̇", "\n", "123", "456", "78", "ǻ"]} +{"text": "é'Tm
.́字
're​ \né'll", "tokens": 15, "pieces": ["é'T", "m", "
", ".́字", "
", "'re", "​", " \n", "é'll"]} +{"text": "㋿9EOT½,'re'S!'ſ<|endoftext|>🙂३İ're'T\r\n…ꟲ#$%'Sḍ̇😀🏽12345678'ſḍ̇\r's𐞁!!\n
漢#$%", "tokens": 58, "pieces": ["㋿", "9", "EOT", "½", ",'", "re'S", "!'", "ſ", "<|", "endoftext", "|>🙂", "३", "İ're", "'T", "\r\n", "…ꟲ", "#$%'", "Sḍ̇", "😀🏽", "123", "456", "78", "'ſḍ̇", "\r", "'s𐞁", "!!\n", "
漢", "#$%"]} +{"text": ".fi ꟲt12345678\n㍿ḍ̇\r\n\r\ń\n", "㍿ḍ̇", "\r\n\r\n", "́", "EOTA\"e>  \"㍿", "tokens": 19, "pieces": ["İ", "<|", "endoftext", "|>", "EOTA", "\"e", ">", " ", " \"㍿"]} +{"text": "ſ<|endoftext|>9A-'re>½>́ ", "tokens": 17, "pieces": ["ſ", "<|", "endoftext", "|>", "9", "A", "-'", "re", ">", "½", ">́", " "]} +{"text": "A \n(Ⅳ'VEſ(ZZå\r\n\r\nå\" s㍿ḍ̇㍿9e.!!́\n", "tokens": 33, "pieces": ["A", " \n", "(", "Ⅳ", "'VEſ", "(ZZå", "\r\n\r\n", "å", "\"", " ", " s", "㍿ḍ̇", "㍿", "9", "e", ".!!́\n"]} +{"text": "ꟲDž !<\r\n're -", "tokens": 11, "pieces": ["ꟲ", "Dž", " !<\r\n", "'re", " ", " -"]} +{"text": "ß9<|fim_prefix|>'sⅣ'sḍ̇éſ\t#$%A'Mé­9\u000bés> ​漢's😀🏽'll ½!!\r\n'M'T‍\n", "tokens": 51, "pieces": ["ß", "9", "<|", "fim", "_prefix", "|>'", "s", "Ⅳ", "'sḍ̇éſ", "\t", "#$%", "A'M", "é", "­", "9", "\u000bés", ">", " ", "​<", "EOT", ">漢's", "😀🏽'", "ll", " ", "½", "!!\r\n", "'M'T", "‍\n"]} +{"text": "́,字́\"d's'S\r\n ½9'D ḍ̇½🙂ꟲ'́漢!!🙂", "tokens": 27, "pieces": ["́", ",字́", "\"d's", "'S", "\r\n", " ", "½9", "'D", " ", " ḍ̇", "½", "🙂ꟲ", "'́漢", "!!🙂"]} +{"text": "#$%ZꟲAam!9Ⅳß½́12345678!!", "tokens": 19, "pieces": ["#$%", "ZꟲAam", "!", "9Ⅳ", "ß", "½", "́", "123", "456", "78", "!!"]} +{"text": "
\r\n\r\n㍿<0\r\nß'9<|endoftext|>ſ", "tokens": 19, "pieces": ["
\r\n\r\n", "㍿<", "0", "\r\n", "ß", "'", "9", "<|", "endoftext", "|>", "ſ"]} +{"text": "
ع…-å'S9👍🏽EOT́\"'ḍ̇.'é­!́EOT<|fim_prefix|>字åå ", "tokens": 41, "pieces": ["
ع", "…", "-å'S", "9", "👍🏽", "EOT́", "\"'", "ḍ̇", ".'", "é", "­<", "META", "_START", ">!́", "EOT", "<|", "fim", "_prefix", "|>", "字åå", " "]} +{"text": "<|fim_prefix|>𐞁'ſ9ḍ̇
字ꟲع éEOT𐞁 #$%!!'ſ​३mEOT\r\nZ(​'s㍿#$%!!\"", "tokens": 57, "pieces": ["<|", "fim", "_prefix", "|>", "𐞁'ſ", "9", "ḍ̇", "
字ꟲع", " é", "EOT𐞁", " ", "#$%!!<", "EOT", ">'", "ſ", "​", "३", "m", "EOT", "\r\n", "Z", "(​'", "s", "㍿#$%!!\""]} +{"text": "Ⅳ½​m\r\nA'T…\u000bdDžaİ <|fim_prefix|>‍'llⅣ‍\"\u000b'S-", "tokens": 40, "pieces": ["Ⅳ½", "​m", "\r\n", "A'T", "…", "\u000bd", "Dža", "İ", " ", "<|", "fim", "_prefix", "|>‍'", "ll", "", "Ⅳ", "‍\"", "\u000b", "'S", "-"]} +{"text": "ſe… EOT", "tokens": 7, "pieces": ["ſe", "…", " EOT"]} +{"text": "३('VE漢\r\n\r\n 👍🏽-́a‍#$%<\u000b>😀🏽𐞁ſꟲ'Re
ꟲte'llİ", "tokens": 37, "pieces": ["३", "('", "VE漢", "\r\n\r\n", " ", " 👍🏽-́", "a", "‍#$%<", "\u000b", ">😀🏽", "𐞁ſꟲ'Re", "
ꟲte'll", "İ"]} +{"text": "½‍\r­'reåⅣİé㍿,'s'S", "tokens": 17, "pieces": ["½", "‍\r", "­'", "reå", "Ⅳ", "İé", "㍿,'", "s'S"]} +{"text": "((
Dž'… 0d\r>aé're'll<|endoftext|>'D'llع<|fim_prefix|>'>عé're9\t \nAa", "tokens": 46, "pieces": ["((", "
Dž", "'", "…", " ", "0", "d", "\r", ">aé're", "'ll", "<|", "endoftext", "|>'", "D'll", "ع", "<|", "fim", "_prefix", "|>'>", "عé're", "", "9", "\t \n", "Aa"]} +{"text": "ع'll😀🏽're'Re㋿'M字 ", "tokens": 19, "pieces": ["ع'll", "😀🏽'", "re'Re", "㋿'", "M", "字", " "]} +{"text": "fi㋿(d< 0\"Z漢ß‍😀🏽漢''ſé<|fim_prefix|>\r\n!9\"#$%…­EOT👍🏽\t👍🏽'TDž-", "tokens": 49, "pieces": ["fi", "㋿(", "d", "<", " ", "0", "\"Z漢ß", "‍😀🏽", "漢", "''", "ſé", "<|", "fim", "_prefix", "|>\r\n", "!", "9", "\"#$%", "…", "­EOT", "👍🏽", "\t", "👍🏽'", "TDž", "-"]} +{"text": "ſm🙂…Ⅳ𐞁½́st!'re", "tokens": 16, "pieces": ["ſm", "🙂", "…", "Ⅳ", "𐞁", "½", "́st", "!'", "re"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n\r\n😀🏽ع́字🙂ß३'Re\"𐞁\"'ll \n㋿12345678🙂\n's\n #$%!!ꟲ'ſ#$%!!ſ\r\n👍🏽's're'T", "tokens": 51, "pieces": ["\r\n\r\n", "😀🏽", "ع́字", "🙂ß", "३", "'Re", "\"𐞁", "\"'", "ll", " \n", "㋿", "123", "456", "78", "🙂\n", "'s", "\n", " ", " #$%!!", "ꟲ'ſ", "#$%!!", "ſ", "\r\n", "👍🏽'", "s're", "'T"]} +{"text": "㍿㍿<👍🏽\"🙂 déEOTåḍ̇\r\n\r\n's\t<|fim_prefix|>><'S漢㋿å're \n½'D-", "tokens": 50, "pieces": ["㍿㍿<👍🏽\"🙂", " ", " dé", "EOTå", "ḍ̇", "\r\n\r\n", "'s", "\t", "<|", "fim", "_prefix", "|>><'", "S漢", "㋿", "å're", " \n", "½", "'D", "-"]} +{"text": "'VE.́ …\rⅣİ'Re\"é٣٤٥٦'VE\u000b字٣٤٥٦ḍ̇👍🏽0!!A३é12345678mat\r\n'S👍🏽å漢‍'sDžA'S", "tokens": 61, "pieces": ["'VE", ".́", " …\r", "Ⅳ", "İ'Re", "\"é", "٣٤٥", "٦", "'VE", "\u000b字", "٣٤٥", "٦", "ḍ̇", "👍🏽", "0", "!!", "A", "३", "é", "123", "456", "78", "mat", "\r\n", "'S", "👍🏽", "å漢", "‍'", "s", "DžA'S"]} +{"text": "!!字é<|fim_prefix|>'D\r(Ⅳ'Dꟲ,🙂…EOT३'D( \n \n­ \n'lléas
'Ree\r\n- ", "tokens": 41, "pieces": ["!!", "字é", "<|", "fim", "_prefix", "|>'", "D", "\r", "(", "Ⅳ", "'Dꟲ", ",🙂", "…EOT", "३", "'D", "(", " \n \n", "­", " \n", "'lléas", "
", "'Ree", "\r\n", "-", " "]} +{"text": "…", "tokens": 2, "pieces": ["…"]} +{"text": "'re#$%Dž'😀🏽㋿ع \nå0­
'lla字''s'S😀🏽'ſ \u000bZ-é \"'ll­👍🏽<|fim_prefix|>ꟲ!!\n", "tokens": 56, "pieces": ["'re", "#$%", "Dž", "'😀🏽㋿", "ع", " \n", "å", "0", "­", "
", "'lla字", "''", "s'S", "😀🏽'", "ſ", " ", "\u000bZ", "-é", " ", "\"'", "ll", "­👍🏽<|", "fim", "_prefix", "|>", "ꟲ", "!!\n"]} +{"text": "'D…\u000b🙂a🙂 EOT!!٣٤٥٦12345678\r'Té३\r\n\r\n9'VE", "tokens": 25, "pieces": ["'D", "…", "\u000b", "🙂a", "🙂", " EOT", "!!", "٣٤٥", "٦12", "345", "678", "\r", "'Té", "३", "\r\n\r\n", "9", "'VE"]} +{"text": "'Så\u000be𐞁'ſ㍿'s㋿İ'S㍿d\t㋿'S'ſ're's😀🏽३#$%İⅣfi fifidꟲ'", "ſ", "㍿'", "s", "㋿İ'S", "㍿d", "\t", "㋿'", "S'ſ", "'re's", "😀🏽", "३", "#$%", "İ", "Ⅳ", "fi", " fifidꟲ", "éå㍿'s\té\r٣٤٥٦å­å'Re\r\n🙂٣٤٥٦12345678­Ⅳ", "tokens": 39, "pieces": ["t", "-a'D", ">éå", "㍿'", "s", "\té", "\r", "٣٤٥", "٦", "å", "­å'Re", "\r\n", "🙂", "٣٤٥", "٦12", "345", "678", "­", "Ⅳ"]} +{"text": "½'Re'ſ9>d \n<'llé0dß12345678'D­ſ#$%'s", "tokens": 25, "pieces": ["½", "'", "Re'ſ", "9", ">d", " \n", "<'", "llé", "0", "dß", "123", "456", "78", "'D", "­ſ", "#$%'", "s"]} +{"text": "'VE", "tokens": 6, "pieces": ["'VE", ""]} +{"text": "'re's३'Re12345678eA'VE\nA,'s#$%\n𐞁\"Dž", "tokens": 28, "pieces": ["'re's", "३", "'Re", "123", "456", "78", "e", "A'VE", "\n", "A", ",'", "s", "#$%\n", "𐞁", "\"Dž"]} +{"text": "\n'VE🙂½<|endoftext|><\n'M!'Mꟲ­𐞁ꟲ're\u000bfis! 9'Da\r\na👍🏽😀🏽're( 12345678'", "tokens": 57, "pieces": ["\n", "'VE", "🙂", "½", "<|", "endoftext", "|><\n", "'M", "!'", "Mꟲ", "­<", "EOT", ">𐞁ꟲ're", "\u000bfis", "!", " ", "9", "'Da", "\r\n", "a", "👍🏽😀🏽'", "re", "(", " ", " ", "123", "456", "78", "'"]} +{"text": "fi \t👍🏽12345678're​ꟲ👍🏽EOT٣٤٥٦'ll٣٤٥٦ å('.", "tokens": 31, "pieces": ["fi", " ", "\t", "👍🏽", "123", "456", "78", "'re", "​ꟲ", "👍🏽", "EOT", "٣٤٥", "٦", "'ll", "٣٤٥", "٦", " å", "('."]} +{"text": "\"\rd㋿\t \t 'S\tså'VE🙂", "tokens": 15, "pieces": ["\"\r", "d", "㋿", "\t \t", " ", "'S", "\tså'VE", "🙂"]} +{"text": "m३\r\n 𐞁👍🏽\r's,s😀🏽́­ \r\n\r\n🙂𐞁", "tokens": 25, "pieces": ["m", "३", "\r\n", " 𐞁", "👍🏽\r", "'s", ",s", "😀🏽́­", " \r\n\r\n", "🙂𐞁"]} +{"text": "0ſ\r!EOT​é'Ḿ\r'M'S​et'reſ\u000b", "tokens": 22, "pieces": ["", "0", "ſ", "\r", "!EOT", "​é'M", "́", "\r", "'M'S", "​et're", "ſ", "\u000b"]} +{"text": "#$%,d
EOT漢\n​Ⅳꟲe'M", "tokens": 16, "pieces": ["#$%,", "d", "
EOT漢", "\n", "​", "Ⅳ", "ꟲe'M"]} +{"text": "><|endoftext|>!ḍ̇'Ré.漢>'M", "tokens": 20, "pieces": ["><|", "endoftext", "|>!", "ḍ̇'Re", "́", ".漢", ">'", "M", ""]} +{"text": "Zéſ--Dž's\r\n\r\nꟲ́'VE字漢're!!s\r\n'-<", "EOT", ">'", "s", "\r\n\r\n", "ꟲ́'VE", "字漢're", "!!", "s", "\r\n", "'-<", "EOT", "🙂", " ", " ", "#$%"]} +{"text": " 👍🏽
!!\t!!'sꟲ́<ꟲ🙂é'VEA\"Z𐞁'-İ<|endoftext|>", "
", "!!", "\t", "!!'", "sꟲ́", "<ꟲ", "🙂é'VE", "A", "\"Z𐞁", "'-", "İ", "<|", "endoftext", "|><", "s", "9", "​"]} +{"text": ",tع'D \nḍ̇㍿'Re\r\n 𐞁'VEd👍🏽​'ll'ReA𐞁 \n🙂", "tokens": 35, "pieces": [",tع'D", " \n", "ḍ̇", "㍿'", "Re", "\r\n", " 𐞁'VE", "d", "👍🏽​'", "ll'Re", "A𐞁", " \n", "🙂"]} +{"text": "́'VEDž\nⅣé𐞁‍'M", "tokens": 16, "pieces": ["́'VE", "Dž", "\n", "Ⅳ", "é𐞁", "‍'", "M"]} +{"text": "é \nEOT
<'‍'Ttå!", "tokens": 13, "pieces": ["é", " \n", "EOT", "
", "<'‍'", "Ttå", "!"]} +{"text": "!!字\r\n\r\n'ſ'D Z㋿'T( Ⅳ0eé㋿a're \n", "tokens": 27, "pieces": ["!!", "字", "\r\n\r\n", "'ſ'D", " Z", "㋿'", "T", "(", " ", " ", "Ⅳ0", "eé", "㋿a're", " \n"]} +{"text": "aſ'D👍🏽­㍿aḍ̇\u000b​! \n'S<|endoftext|>s're9fi>", "tokens": 31, "pieces": ["aſ'D", "👍🏽­㍿", "aḍ̇", "\u000b", "​!", " \n", "'S", "<|", "endoftext", "|>", "s're", "9", "fi", ">"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "Ⅳ३A٣٤٥٦0éſ\u000b\r𐞁 Džé½ḍ̇sd,'Sa\r\n\r\n३é", "tokens": 32, "pieces": ["Ⅳ३", "A", "٣٤٥", "٦0", "éſ", "\u000b\r", "𐞁", " Džé", "½", "ḍ̇sd", ",'", "Sa", "\r\n\r\n", "३", "é"]} +{"text": "t d​ß'ſfi-!😀🏽a३'sḍ̇12345678Z'll­'Re#$%
 ,s'Så'Té'ſ(\r\n\r\ntſm'\r\n", "tokens": 47, "pieces": ["t", " ", " d", "​ß'ſ", "fi", "-!😀🏽", "a", "३", "'sḍ̇", "123", "456", "78", "Z'll", "­'", "Re", "#$%", "
 ", " ,", "s'S", "å'T", "é'ſ", "(\r\n\r\n", "tſm", "'\r\n"]} +{"text": "​漢<'DtEOT­ḍ̇<𐞁<字́ 'M're0", "tokens": 25, "pieces": ["​", "漢", "<'", "Dt", "EOT", "­ḍ̇", "<𐞁", "<字́", " '", "M're", "0"]} +{"text": ">dé,!!-\rEOT- \nt Ⅳ \n\tḍ̇Z字", "tokens": 29, "pieces": [">", "dé", ",!!-\r", "EOT", "-", " \n", "t", " ", " ", "Ⅳ", " \n", "", "\tḍ̇", "Z字"]} +{"text": "\r\n
'Sſ٣٤٥٦!!…DžZ३'Tع#$%e", "tokens": 24, "pieces": ["\r\n", "
", "'Sſ", "٣٤٥", "٦", "!!", "…", "DžZ", "३", "'Tع", "#$%", "e"]} +{"text": "fi🙂'👍🏽'M \n'Ta🙂t \u000b >Afi", "🙂'👍🏽'", "M", " \n", "'Ta", "🙂t", " \u000b", " ", ">A", "#$%å'ſ𐞁<|fim_prefix|>ß!!🙂….ꟲ\"", "tokens": 48, "pieces": [".", "123", "456", "78", "'𐞁", "३", "ع'M", "(,", "½", "́", "\n", "३", ">#$%", "å'ſ", "𐞁", "<|", "fim", "_prefix", "|>", "ß", "!!🙂", "…", ".ꟲ", "\""]} +{"text": "<|fim_prefix|>> \n𐞁e", "tokens": 12, "pieces": ["<|", "fim", "_prefix", "|>>", " \n", "𐞁e"]} +{"text": "'Dt-ḍ̇३'ſ'dZ.m\r\n‍\r\n\r\n,Dž\r''M'VE", "tokens": 23, "pieces": ["'Dt", "-ḍ̇", "३", "'ſ'd", "Z", ".m", "\r\n", "‍\r\n\r\n", ",Dž", "\r", "''", "M'VE"]} +{"text": "'VE'TA<\r\n'S'VEfi漢-#$%\u000b½sééd.½ 'S.'ll>🙂A's('T​漢'T", "tokens": 36, "pieces": ["'VE'T", "A", "<\r\n", "'S'VE", "fi漢", "-#$%", "\u000b", "½", "sééd", ".", "½", " ", "'S", ".'", "ll", ">🙂", "A's", "('", "T", "​漢'T"]} +{"text": "'S\"𐞁\r\n\r\n字‍,
\"🙂‍.aa!9'Dta'M​\t\nZ\t …m­İ𐞁,Za㋿'re", "tokens": 47, "pieces": ["'S", "\"", "𐞁", "\r\n\r\n", "字", "‍,", "
", "\"🙂‍.", "aa", "!", "9", "'Dta'M", "​", "\t\n", "Z", "\t ", "…m", "­İ", "𐞁", ",Za", "㋿'", "re"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㋿å㍿12345678(-<|endoftext|>>!!\r\n'ſ'VEſ
'sZ<|endoftext|>😀🏽9ßİ‍…EOTⅣع's(ſ <|fim_prefix|>", "tokens": 69, "pieces": ["㋿å", "㍿", "123", "456", "78", "(-<|", "endoftext", "|>><", "EOT", ">!!\r\n", "'ſ", "'", "VEſ", "
", "'s", "Z", "<|", "endoftext", "|>😀🏽", "9", "ß", "İ", "‍", "…EOT", "Ⅳ", "ع's", "(ſ", " <|", "fim", "_prefix", "|>"]} +{"text": " '‍\r\n9-'ree'rea\r\n\r\n३𐞁(́Dž…'Ree😀🏽\"🙂'D0,fi", "tokens": 37, "pieces": [" ", " '‍\r\n", "9", "-'", "ree're", "a", "\r\n\r\n", "३", "𐞁", "(́", "Dž", "", "…", "'Ree", "😀🏽\"🙂'", "D", "0", ",fi"]} +{"text": "🙂0漢(Dž'VE<|fim_prefix|> \"ḍ̇­Z👍🏽٣٤٥٦漢(å字EOT\"12345678's\n'D", "tokens": 43, "pieces": ["🙂", "0", "漢", "(Dž'VE", "<|", "fim", "_prefix", "|>", " ", "\"ḍ̇", "­Z", "👍🏽", "٣٤٥", "٦", "漢", "(å字", "EOT", "\"", "123", "456", "78", "'s", "\n", "'D"]} +{"text": "<|fim_prefix|><|endoftext|>'Re…'T#$% d<|endoftext|><|fim_prefix|>12345678", "tokens": 34, "pieces": ["<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "Re", "…", "'T", "#$%", " d", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "123", "456", "78"]} +{"text": " ㋿𐞁İ\nmḍ̇'re<|fim_prefix|>'Sḍ̇mfi9<٣٤٥٦'Dß're \n ḍ̇㍿\r‍'👍🏽<|endoftext|>🙂ſ'S", "tokens": 60, "pieces": [" ", "㋿𐞁", "İ", "\n", "mḍ̇'re", "<|", "fim", "_prefix", "|>'", "Sḍ̇mfi", "9", "<", "٣٤٥", "٦", "'Dß're", " \n", " ḍ̇", "㍿\r", "‍'👍🏽<|", "endoftext", "|>🙂", "ſ'S"]} +{"text": "'D're½½As9>'M,😀🏽.'Re ́🙂'Ms's३­\r\n­9'", "re", "½½", "As", "9", ">'", "M", ",😀🏽.'", "Re", " ́", "🙂'", "Ms's", "३", "­\r\n", "­", "9", "é́t𐞁Ⅳ#$%'VE'sfi,ſḍ̇'M-,EOT \nA's", "tokens": 44, "pieces": ["İ", "‍​", "
", "㋿", " 漢", "‍\r", "३", "!!́<", "EOT", ">é́t𐞁", "Ⅳ", "#$%'", "VE's", "fi", ",ſḍ̇'M", "-,", "EOT", " \n", "A's"]} +{"text": "'re…0!!٣٤٥٦ßdZ‍'sZ", "tokens": 16, "pieces": ["'re", "…", "0", "!!", "٣٤٥", "٦", "ßd", "Z", "‍'", "s", "Z"]} +{"text": "fiA \nEOT\t½'ll,s A9", "tokens": 12, "pieces": ["fi", "A", " \n", "EOT", "\t", "½", "'ll", ",s", " A", "9"]} +{"text": "𐞁!!漢٣٤٥٦ ㋿'S\r\n\r\nſ", "tokens": 18, "pieces": ["𐞁", "!!", "漢", "٣٤٥", "٦", " ", "㋿'", "S", "\r\n\r\n", "ſ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "!\r\n\r\n.\u000b­ſ.åe", "tokens": 9, "pieces": ["!\r\n\r\n", ".", "\u000b", "­ſ", ".åe"]} +{"text": " \t9'Mſ9!!\"\"9'llꟲ\u000bİ㋿!!fiZع'ſⅣⅣſ", "tokens": 29, "pieces": [" ", "\t", "9", "'Mſ", "9", "!!\"\"", "9", "'llꟲ", "\u000bİ", "㋿!!", "fi", "Zع'ſ", "ⅣⅣ", "ſ"]} +{"text": "EOT३ 12345678'T'Reḍ̇ \n'Dm'sd😀🏽\t㍿,t 'Tع\rå", "tokens": 41, "pieces": ["EOT", "३", "", " ", "123", "456", "78", "'", "T'Re", "ḍ̇", " \n", "'Dm's", "d", "😀🏽", "\t", "㍿,", "t", "", " ", "'Tع", "\r", "å"]} +{"text": "\u000bAé
e\r'T <|endoftext|>t!!e'D٣٤٥٦𐞁👍🏽Ⅳ\rſ'lltDž('", "tokens": 44, "pieces": ["\u000bAé", "
e", "\r", "'T", " ", " <|", "endoftext", "|>", "t", "!!<", "EOT", ">e'D", "٣٤٥", "٦", "𐞁", "👍🏽", "Ⅳ", "\r", "ſ'll", "t", "Dž", "('"]} +{"text": "\nḍ̇㋿㍿😀🏽漢a", "tokens": 15, "pieces": ["\n", "ḍ̇", "㋿㍿😀🏽", "漢a"]} +{"text": "s ㍿'reAt!!𐞁½d㋿EOT…d!!㍿㍿e
éİ.​‍ßZ
३0Ⅳ, é'S'VE漢३0's", "tokens": 54, "pieces": ["s", " ", "㍿'", "re", "At", "!!", "𐞁", "½", "d", "㋿EOT", "…d", "!!㍿㍿", "e", "
é", "İ", ".​‍", "ß", "Z", "
", "३0Ⅳ", ",", " é'S", "'VE漢", "३0", "'s"]} +{"text": " <㍿ع😀🏽Z𐞁ſ
mm​dß'Re漢\r\n\r\n \n\"…\"", "tokens": 27, "pieces": [" <㍿", "ع", "😀🏽", "Z𐞁ſ", "
mm", "​dß'Re", "漢", "\r\n\r\n \n", "\"", "…", "\""]} +{"text": " \ném'S½'s
字­'ſ\r\n\r\n‍ſ٣٤٥٦\r\n\r\n
'VE​éDž㍿\u000bååع\r\n'se\rعt''ll", "tokens": 48, "pieces": [" \n", "ém'S", "½", "'s", "
字", "­<", "EOT", ">'", "ſ", "\r\n\r\n", "‍ſ", "٣٤٥", "٦", "\r\n\r\n", "
", "'VE", "​é", "Dž", "㍿", "\u000bååع", "\r\n", "'se", "\r", "عt", "''", "ll"]} +{"text": "字mſ🙂'", "tokens": 5, "pieces": ["字mſ", "🙂'"]} +{"text": "🙂 ", "tokens": 2, "pieces": ["🙂", " "]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'s<|endoftext|>㋿\"👍🏽\rEOT́!e", "tokens": 21, "pieces": ["'s", "<|", "endoftext", "|>㋿\"👍🏽\r", "EOT́", "!e"]} +{"text": ".㋿at字<'Sm'M9字e
\u000bع!👍🏽ß㍿..-d ḍ̇🙂.'S\r'ſ\r\n\r\nA'ſ12345678'VE
é'ſ\n", "tokens": 52, "pieces": [".㋿", "at字", "<'", "Sm'M", "9", "字e", "
", "\u000bع", "!👍🏽", "ß", "㍿..-", "d", " ", " ḍ̇", "🙂.'", "S", "\r", "'ſ", "\r\n\r\n", "A'ſ", "123", "456", "78", "'VE", "
é'ſ", "\n"]} +{"text": "ḍ̇e'ſ👍🏽<|fim_prefix|><|endoftext|>s!­ é😀🏽ſ½…'Mꟲ\u000b -㋿ 'S\r#$%fi字-<\u000b\r(-\reé㋿", "tokens": 68, "pieces": ["ḍ̇e'ſ", "👍🏽<|", "fim", "_prefix", "|><|", "endoftext", "|>", "s", "!­", " ", "é", "😀🏽", "ſ", "½", "…", "'Mꟲ", "\u000b", " -㋿", " ", " '", "S", "\r", "#$%", "fi字", "-<", "\u000b\r", "(-\r", "eé", "㋿"]} +{"text": "fid <|endoftext|>!­ 'D३Z🙂<|endoftext|>#$%字'ſé", "tokens": 34, "pieces": ["fid", " ", " <|", "endoftext", "|><", "META", "_START", ">!­", " ", "'D", "३", "Z", "🙂<|", "endoftext", "|>#$%", "字'ſ", "é"]} +{"text": "'ll !!​<|fim_prefix|>ع‍'ll‍‍''S.EOT#$%'
\u000b👍🏽Dž'\r\n\r\nİ́", "tokens": 38, "pieces": ["'ll", " ", " !!​<", "EOT", "><|", "fim", "_prefix", "|>", "ع", "‍'", "ll", "‍‍''", "S", ".EOT", "#$%'", "
", "\u000b", "👍🏽", "Dž", "'\r\n\r\n", "İ́"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "<|fim_prefix|>a­fi\"'S'ſ३<0'Sé>'VE㋿a.'Re'T\r\nſDža🙂0\"ſ", "tokens": 37, "pieces": ["<|", "fim", "_prefix", "|>", "a", "­fi", "\"'", "S'ſ", "३", "<", "0", "'Sé", ">'", "VE", "㋿a", ".'", "Re'T", "\r\n", "ſ", "Dža", "🙂", "0", "\"ſ"]} +{"text": "㍿12345678'S!!#$%fi'ß\t㋿åe.­😀🏽'VE\r\n\r\nDž‍漢٣٤٥٦İ㍿ >9'Re\u000b9!\u000b0 \nß‍'Re'ſſ", "tokens": 57, "pieces": ["㍿", "123", "456", "78", "'S", "!!#$%", "fi", "'ß", "\t", "㋿åe", ".­😀🏽'", "VE", "\r\n\r\n", "Dž", "‍漢", "٣٤٥", "٦", "İ", "㍿", " ", " >", "9", "'Re", "\u000b", "9", "!", "\u000b", "0", " \n", "ß", "‍'", "Re'ſ", "ſ"]} +{"text": "tm'll\t12345678<,३A🙂12345678Dž'll𐞁İå𐞁\u000b\u000b
", "tokens": 31, "pieces": ["tm'll", "\t", "123", "456", "78", "<,", "३", "A", "🙂", "123", "456", "78", "Dž'll", "𐞁İå𐞁", "\u000b\u000b
"]} +{"text": "'Reßå>12345678å­m😀🏽", "tokens": 15, "pieces": ["'Reßå", ">", "123", "456", "78", "å", "­m", "😀🏽"]} +{"text": "​ ſå㍿ꟲ<|fim_prefix|>-EOT<|fim_prefix|>'T,(㍿ 
ßعé'Mḍ̇.", "tokens": 44, "pieces": ["​", " ſå", "㍿ꟲ", "<|", "fim", "_prefix", "|>-", "EOT", "<|", "fim", "_prefix", "|>'", "T", ",(㍿", " ", "", "
ßعé'M", "ḍ̇", "."]} +{"text": " ㋿㋿\u000bꟲ", "tokens": 11, "pieces": [" ", "㋿㋿", "\u000bꟲ"]} +{"text": " ꟲ३'ſ'ſ\n'll!!'M'S1234567812345678🙂\r\nZ\r", "tokens": 25, "pieces": [" ꟲ", "३", "'ſ'ſ", "\n", "'ll", "!!'", "M'S", "123", "456", "781", "234", "567", "8", "🙂\r\n", "Z", "\r"]} +{"text": "fi​'D\"Džmſ 
>३ꟲ\r\n'ſa", "tokens": 20, "pieces": ["fi", "​'", "D", "\"Džmſ", " ", "
", ">", "३", "ꟲ", "\r\n", "'ſa"]} +{"text": "é>…‍́漢́ꟲ09…12345678d- <|endoftext|> \n\r'' EOTå- ,'sEOT#$% \n<|endoftext|>9!!…😀🏽", "tokens": 66, "pieces": ["é", ">", "…", "‍́漢́ꟲ", "09", "…", "123", "456", "78", "d", "-", " ", "<|", "endoftext", "|>", " \n\r", "''", " ", " EOTå", "-<", "META", "_START", ">", " ", ",'", "s", "EOT", "#$%", " \n", "<|", "endoftext", "|>", "9", "!!", "…", "😀🏽"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "å'lls<|fim_prefix|>'M'-12345678 \n'D é", "tokens": 20, "pieces": ["å'll", "s", "<|", "fim", "_prefix", "|>'", "M", "'-", "123", "456", "78", " \n", "'D", " ", " é"]} +{"text": "३漢t!!!!es\u000b\u000b\r\n​'ſ<|fim_prefix|>'TZ\"å!!!!", "es", "\u000b\u000b\r\n", "​'", "ſ", "<|", "fim", "_prefix", "|>'", "TZ", "\"å", "99 ㍿'D🙂-", "tokens": 14, "pieces": ["🙂", " ", "", "99", " ㍿'", "D", "🙂-"]} +{"text": "'T \n'T‍mZ'llmA!\r\n\r\n(ds\t", "tokens": 13, "pieces": ["'T", " \n", "'T", "‍m", "Z'll", "m", "A", "!\r\n\r\n", "(ds", "\t"]} +{"text": "­
Dž\"ßİ#$%'M'sꟲ(\r\n\r\n<ꟲ'S'M😀🏽d३….​½ꟲ٣٤٥٦", "tokens": 41, "pieces": ["­", "
Dž", "\"", "ß", "İ", "#$%'", "M's", "ꟲ", "(\r\n\r\n", "<ꟲ'S", "'M", "😀🏽", "d", "३", "…", ".​", "½", "ꟲ", "٣٤٥", "٦"]} +{"text": "'Da‍㍿,'Reß12345678'DeZ-'re'D's,-12345678'VE'll३(", "tokens": 28, "pieces": ["'Da", "‍㍿,'", "Reß", "123", "456", "78", "'De", "Z", "-'", "re'D", "'s", ",-", "123", "456", "78", "'VE'll", "३", "("]} +{"text": "​ \n'VEſ", "tokens": 5, "pieces": ["​", " \n", "'VEſ"]} +{"text": "-ß'Re३å .", "tokens": 7, "pieces": ["-ß'Re", "३", "å", " ."]} +{"text": "åḍ̇!!<|fim_prefix|>\r's'S‍'sſ٣٤٥٦'D12345678.#$%-!fi fi< 'é‍😀🏽३å> ", "tokens": 47, "pieces": ["åḍ̇", "!!<|", "fim", "_prefix", "|>\r", "'s'S", "‍'", "sſ", "٣٤٥", "٦", "'D", "123", "456", "78", ".#$%-!", "fi", " fi", "<", " ", " '", "é", "‍😀🏽", "३", "å", ">", " "]} +{"text": "ß😀🏽ſ'llt'T''T'M0́!a'S,㍿٣٤٥٦>
…s>́'re\n's", "tokens": 34, "pieces": ["ß", "😀🏽", "ſ'll", "t'T", "''", "T'M", "0", "́", "!a'S", ",㍿", "٣٤٥", "٦", ">", "
", "…s", ">́'re", "\n", "'s"]} +{"text": "
\r\n
İ0'Red#$% 12345678åe \nع'VE'Re d'T漢> 'll'Tİ́㍿३𐞁ß12345678 ㋿", "tokens": 52, "pieces": ["
\r\n", "
İ", "0", "'Red", "#$%<", "EOT", ">", " ", "123", "456", "78", "åe", " \n", "ع'VE", "'Re", " d'T", "漢", ">", " '", "ll'T", "İ́", "㍿", "३", "𐞁ß", "123", "456", "78", " ", " ㋿"]} +{"text": "\u000bعع<|endoftext|>㍿ ​<|endoftext|>EOT'ſtd㍿'s‍'M\r(#$%9\n<|fim_prefix|>! \n'ſ9٣٤٥٦'T\tⅣ", "tokens": 60, "pieces": ["\u000bعع", "<|", "endoftext", "|>㍿", " ", " ​<|", "endoftext", "|>", "EOT'ſ", "td", "㍿'", "s", "‍'", "M", "\r", "(#$%", "9", "\n", "<|", "fim", "_prefix", "|>!", " \n", "'ſ", "9٣٤", "٥٦", "'T", "\t", "Ⅳ"]} +{"text": "eEOT''ress'VEm>'S- 'M字\r\n\r\n'Sḍ̇İ0're'Re
́dEOT", "''", "ress'VE", "m", ">'", "S", "-", " '", "M字", "\r\n\r\n", "'Sḍ̇", "İ", "0", "'re'Re", "
́d", "EOT‍ß\r
\n字<\u000bꟲḍ̇Dž'D́ésⅣ!\u000b<…(<|endoftext|>0's'Re 're'Sſa", "tokens": 60, "pieces": ["'re", "0", "A", ".㋿<", "EOT", ">EOT", "‍ß", "\r
\n", "字", "<", "\u000bꟲḍ̇", "Dž'D", "́és", "Ⅳ", "!", "\u000b", "<", "…", "(<|", "endoftext", "|>", "0", "'", "s'Re", " <", "META", "_START", ">'", "re'S", "ſa"]} +{"text": "\" é", "tokens": 4, "pieces": ["\"", " ", " é"]} +{"text": "😀🏽's…½\t३#$%ḍ̇'T\r'Reé-0İ㍿<|fim_prefix|>Dž \n ", "tokens": 55, "pieces": ["ꟲع", " ß", "Ae", "(\r\n\r\n", "eß", "㍿", " ", " ḍ̇'s", "ꟲ", "<|", "endoftext", "|>", "ḍ̇'T", "\r", "'Reé", "-", "0", "İ", "㍿<|", "fim", "_prefix", "|>", "Dž", " \n", " "]} +{"text": "عe 'D३'s'reéa#$%9字\r!'s", "tokens": 17, "pieces": ["عe", " ", "'D", "३", "'s're", "éa", "#$%", "9", "字", "\r", "!'", "s"]} +{"text": "'ſⅣⅣ12345678>Dž​٣٤٥٦३", "tokens": 19, "pieces": ["'ſ", "ⅣⅣ1", "234", "567", "8", ">Dž", "​", "٣٤٥", "٦३"]} +{"text": "\rſ'ſ.ſ", "tokens": 6, "pieces": ["\r", "ſ'ſ", ".ſ"]} +{"text": "​(ꟲ'M🙂'VE\re'VE\n12345678İ's!!>٣٤٥٦Z\r\n\r\n\tA३\r\nZ㍿\r,­٣٤٥٦-\r !!'VE<|endoftext|><|endoftext|>", "tokens": 61, "pieces": ["​(", "ꟲ'M", "🙂'", "VE", "\r", "e'VE", "\n", "123", "456", "78", "İ's", "!!>", "٣٤٥", "٦", "Z", "\r\n\r\n", "\tA", "३", "\r\n", "Z", "㍿\r", ",­", "٣٤٥", "٦", "-\r", " ", "!!'", "VE", "<|", "endoftext", "|><|", "endoftext", "|>"]} +{"text": "ⅣⅣZé'T \tm…'T<|fim_prefix|>ḍ̇- \n㋿Z'sé漢३​m \nd", "tokens": 36, "pieces": ["ⅣⅣ", "Zé'T", " ", "\tm", "…", "'T", "<|", "fim", "_prefix", "|>", "ḍ̇", "-", " \n", "㋿Z's", "é漢", "३", "​m", " \n", "d"]} +{"text": "!!ꟲ'Ts, #$%,…🙂>​aİ\r\n\r\n
ée

é‍,a'VE​aß\r\n\t9t<|fim_prefix|>\rſ٣٤٥٦<|fim_prefix|>Džae", "tokens": 57, "pieces": ["ſ", "\n", "😀🏽'", "re", "<|", "fim", "_prefix", "|>🙂>​", "a", "İ", "\r\n\r\n", "
ée", "
", "
é", "‍,", "a'VE", "​aß", "\r\n", "\t", "9", "t", "<|", "fim", "_prefix", "|>\r", "ſ", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "Džae"]} +{"text": "<|endoftext|>㍿­ås½ ع\r\n(ع𐞁👍🏽-'ll!'llm㋿Z", "tokens": 36, "pieces": ["<|", "endoftext", "|>㍿­", "ås", "½", " ", " ع", "\r\n", "(ع𐞁", "👍🏽-'", "ll", "!'", "llm", "㋿Z"]} +{"text": "<|fim_prefix|>İd㍿ß ­🙂!>,\r\n\r\n𐞁٣٤٥٦\" ,ꟲ…tå a½é<|endoftext|>,٣٤٥٦é>", "tokens": 56, "pieces": ["<|", "fim", "_prefix", "|>", "İd", "㍿ß", " ­🙂!>,\r\n\r\n", "𐞁", "٣٤٥", "٦", "\"", " ,", "ꟲ", "…tå", " a", "½", "é", "<|", "endoftext", "|>,", "٣٤٥", "٦", "é", ">"]} +{"text": "> ́!\t're'T㍿ſ㍿EOT\r\n\r\n­'ReEOTⅣ'M", "tokens": 29, "pieces": [">", " ́", "!<", "META", "_START", ">", "\t", "'re'T", "㍿ſ", "㍿EOT", "\r\n\r\n", "­'", "Re", "EOT", "Ⅳ", "'M"]} +{"text": "m-½𐞁Zꟲ漢㋿'D<|fim_prefix|>.<|fim_prefix|>'re'T0''lle\ńmsd12345678!!'Reſ(Ⅳ", "tokens": 48, "pieces": ["m", "-", "½", "𐞁Zꟲ漢", "㋿'", "D", "<|", "fim", "_prefix", "|>.<|", "fim", "_prefix", "|>'", "re'T", "0", "''", "lle", "\n", "́msd", "123", "456", "78", "!!'", "Reſ", "(", "Ⅳ"]} +{"text": "ꟲ👍🏽٣٤٥٦ Dž‍9'M㋿fi<|endoftext|>'ſs\té(㋿", "tokens": 43, "pieces": ["ꟲ", "👍🏽<", "META", "_START", ">", "٣٤٥", "٦", " Dž", "‍<", "META", "_START", ">", "9", "'M", "㋿fi", "<|", "endoftext", "|>'", "ſs", "\té", "(㋿"]} +{"text": "𐞁'VE\r\n🙂‍字字(", "tokens": 12, "pieces": ["𐞁'VE", "\r\n", "🙂‍", "字字", "("]} +{"text": "́­'Re>Áå.'VE½\n'Tt9åⅣ\"fi🙂A", "tokens": 27, "pieces": ["́", "­'", "Re", "><", "EOT", ">Áå", ".'", "VE", "½", "\n", "'Tt", "9", "å", "Ⅳ", "\"fi", "🙂A"]} +{"text": "<ſå½EOTeDžé ", "tokens": 12, "pieces": ["<ſå", "½", "EOTe", "Džé", " "]} +{"text": "éİ(\r㍿ \né३Z!!9\u000bⅣ\u000b<|fim_prefix|>'ſ ḍ̇漢-\tm'Té", "tokens": 37, "pieces": ["é", "İ", "(\r", "㍿", " \n", "é", "३", "Z", "!!", "9", "", "\u000b", "Ⅳ", "\u000b", "<|", "fim", "_prefix", "|>'", "ſ", " ḍ̇漢", "-", "\tm'T", "é"]} +{"text": "'ll \"\u000b 🙂३<‍🙂12345678Dž <㋿e漢'ſⅣ0ع<|endoftext|>a㍿", "tokens": 39, "pieces": ["'ll", " ", " \"", "\u000b ", " 🙂", "३", "<‍🙂", "123", "456", "78", "Dž", " ", " <㋿", "e漢'ſ", "Ⅳ0", "ع", "<|", "endoftext", "|>", "a", "㍿"]} +{"text": "!!'ſa'Re\rA!A9m", "tokens": 11, "pieces": ["!!'", "ſa'Re", "\r", "A", "!A", "9", "m"]} +{"text": "ع12345678", "tokens": 4, "pieces": ["ع", "123", "456", "78"]} +{"text": "​字ع're\n ㍿'VEé'llfi", "tokens": 14, "pieces": ["​字ع're", "\n", " ㍿'", "VEé'll", "fi"]} +{"text": " \n.عİ12345678d'fia\n​३ḍ̇m!>'re#$%(​👍🏽𐞁
'S­'llſſ\n👍🏽 \nDž'Re 'sfi", "tokens": 56, "pieces": [" \n", ".ع", "İ", "123", "456", "78", "d", "'fia", "\n", "​", "३", "ḍ̇m", "!><", "META", "_START", ">", "३", "'", "re", "#$%(​👍🏽", "𐞁", "
", "'S", "­'", "llſſ", "\n", "👍🏽", " \n", "Dž'Re", " ", "'sfi"]} +{"text": "d", "tokens": 1, "pieces": ["d"]} +{"text": "\t㋿", "tokens": 4, "pieces": ["\t", "㋿"]} +{"text": ">  ㋿'llZ\t<|endoftext|><|endoftext|>'VE­ḍ̇!!'s­-\r'", "tokens": 35, "pieces": [">", " ", " ", "㋿'", "ll", "Z", "\t", "<|", "endoftext", "|><|", "endoftext", "|>'", "VE", "­ḍ̇", "!!'", "s", "­-\r", "'"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ꟲDž㍿𐞁d<|endoftext|><|fim_prefix|>< (عA字‍‍'T👍🏽's's#$%!!<Ⅳß ٣٤٥٦>'Dع ", "tokens": 56, "pieces": ["ꟲ", "Dž", "㍿𐞁d", "<|", "endoftext", "|><|", "fim", "_prefix", "|><", " ", " (", "عA字", "‍‍'", "T", "👍🏽'", "s's", "#$%!!<", "Ⅳ", "ß", " ", "٣٤٥", "٦", ">'", "Dع", " "]} +{"text": "m \n's9İé(!>३㋿­<|fim_prefix|> !漢𐞁ḍ̇s#$%åß12345678'D", "tokens": 46, "pieces": ["m", " \n", "'s", "9", "İé", "(!>", "३", "㋿­<|", "fim", "_prefix", "|>", " ", " <", "META", "_START", ">!", "漢𐞁", "ḍ̇s", "#$%", "åß", "123", "456", "78", "'D"]} +{"text": "!é🙂\r\n३s𐞁'llEOT>
㋿>,́<𐞁عḍ̇\"عté0sam", "tokens": 35, "pieces": ["!é", "🙂\r\n", "३", "s𐞁'll", "EOT", ">", "
", "㋿>,́<", "𐞁عḍ̇", "\"عté", "0", "sam"]} +{"text": "fiZ'ſ!\nß\u000bsea0'ſ
Ⅳ🙂👍🏽 \n‍
('M t('DA!éⅣ‍ m\té9d😀🏽'ſ", "tokens": 48, "pieces": ["fi", "Z'ſ", "!\n", "ß", "\u000bsea", "0", "'ſ", "
", "Ⅳ", "🙂👍🏽", " \n", "‍", "
", "('", "M", " t", "('", "DA", "!é", "Ⅳ", "‍", " m", "\té", "9", "d", "😀🏽'", "ſ"]} +{"text": "12345678\n'…><|fim_prefix|>A(d'ſ!'ll<|fim_prefix|>'s", "tokens": 26, "pieces": ["123", "456", "78", "\n", "'", "…", "><|", "fim", "_prefix", "|>", "A", "(d'ſ", "!'", "ll", "<|", "fim", "_prefix", "|>'", "s"]} +{"text": "\r\n Z>'ſåſ\"ſ½­-(< ع\r\n\r\n
", "tokens": 18, "pieces": ["\r\n", " Z", ">'", "ſåſ", "\"ſ", "½", "­-(<", " ع", "\r\n\r\n", "
"]} +{"text": "!😀🏽🙂EOT'VEſ", "tokens": 10, "pieces": ["!😀🏽🙂", "EOT'VE", "ſ"]} +{"text": "é#$%m\r\n\r\nß𐞁>\t(‍A\r\n\r\n'ſḍ̇३'S🙂İ(#$%", "tokens": 32, "pieces": ["é", "#$%", "m", "\r\n\r\n", "ß", "𐞁", ">", "\t", "(‍", "A", "\r\n\r\n", "'ſḍ̇", "३", "'S", "🙂İ", "(#$%"]} +{"text": ".'Sḍ̇'S…", "tokens": 12, "pieces": [".'", "Sḍ̇'S", "", "…"]} +{"text": "ع …A🙂'T. \n!'Re🙂é12345678Z\r\n", "tokens": 24, "pieces": ["ع", "", " ", "…A", "🙂'", "T", ".", " \n", "!'", "Re", "🙂é", "123", "456", "78", "Z", "\r\n"]} +{"text": "\t३‍A🙂Ⅳ㋿İḍ̇
<\r\n\r\n½<|fim_prefix|>ꟲ'DعADž'D'VE<|endoftext|>'ſ", "tokens": 43, "pieces": ["\t", "३", "‍A", "🙂", "Ⅳ", "㋿İḍ̇", "
", "<\r\n\r\n", "½", "<|", "fim", "_prefix", "|>", "ꟲ'D", "ع", "ADž'D", "'VE", "<|", "endoftext", "|>'", "ſ"]} +{"text": "maſ\r\n\r\n'TDž'D Dž\rⅣ'T12345678å\t
Dž漢(!ع(!! ½EOT\r\nḍ̇!! <|fim_prefix|>", "tokens": 50, "pieces": ["maſ", "\r\n\r\n", "'TDž'D", " ", " Dž", "\r", "Ⅳ", "'T", "123", "456", "78", "å", "\t", "
Dž漢", "(!", "ع", "(!!", " ", " ", "½", "EOT", "\r\n", "ḍ̇", "!!<", "EOT", ">", " ", "<|", "fim", "_prefix", "|>"]} +{"text": "​<|fim_prefix|>\t'lls漢Ⅳ >'T٣٤٥٦! Z#$%\"'re>
ع'M'ſEOT's👍🏽'\r\n\r\n‍​s\n­<|endoftext|>​!!ḍ̇0,", "tokens": 59, "pieces": ["​<|", "fim", "_prefix", "|>", "\t", "'lls漢", "Ⅳ", " ", ">'", "T", "٣٤٥", "٦", "!", " Z", "#$%\"'", "re", ">", "
ع'M", "'ſ", "EOT's", "👍🏽'\r\n\r\n", "‍​", "s", "\n", "­<|", "endoftext", "|>​!!", "ḍ̇", "0", ","]} +{"text": "½!t३\n .d㍿'Res​12345678 \n\r\n.漢…ع…ß'Re漢'VE<|endoftext|>٣٤٥٦ß
.\u000b'Séd
漢", "tokens": 60, "pieces": ["½", "!t", "३", "\n", " ", " .", "d", "㍿'", "Res", "​", "123", "456", "78", " \n\r\n", ".", "漢", "…ع", "…ß'Re", "漢'VE", "<|", "endoftext", "|>", "٣٤٥", "٦", "ß", "", "
", ".", "\u000b", "'Séd", "
漢"]} +{"text": "…s👍🏽e३'VE''reZ  \n", "tokens": 14, "pieces": ["٣٤٥", "٦", "<|", "endoftext", "|>", "Z", "  \n"]} +{"text": "m>'S12345678\r\nḍ̇m‍ḍ̇漢d9<|fim_prefix|>ds, ( EOT!'D9é, a
#$%𐞁d́🙂", "tokens": 52, "pieces": ["m", ">'", "S", "123", "456", "78", "\r\n", "ḍ̇m", "‍ḍ̇漢d", "9", "<|", "fim", "_prefix", "|>", "ds", ",", " (", " ", " EOT", "!<", "META", "_START", ">'", "D", "9", "é", ",", " a", "
", "#$%", "𐞁d́", "🙂"]} +{"text": "(", "tokens": 1, "pieces": ["("]} +{"text": ">​-<|fim_prefix|>㍿'Tt(< 'D🙂 Z\r\n\r\n
(s'VE.#$%Dž​'T㋿\t​
Dž'-", "tokens": 47, "pieces": [">​-<|", "fim", "_prefix", "|>㍿'", "Tt", "(<", " ", "'D", "🙂", " Z", "\r\n\r\n", "
", "(s'VE", ".#$%", "Dž", "​'", "T", "㋿", "\t", "​", "
Dž", "'-"]} +{"text": "'Ḿ9", "tokens": 3, "pieces": ["'Ḿ", "9"]} +{"text": "'s'S>", "tokens": 3, "pieces": ["'s'S", ">"]} +{"text": "‍㋿Ⅳ", "tokens": 6, "pieces": ["‍㋿", "Ⅳ"]} +{"text": "٣٤٥٦\r\n\r\n­å ­0​½½9字e \u000b", "tokens": 20, "pieces": ["٣٤٥", "٦", "\r\n\r\n", "­å", " ­", "0", "​", "½½9", "字e", " \u000b"]} +{"text": "ßⅣ> m'VEfi ḍ̇", "tokens": 13, "pieces": ["ß", "Ⅳ", ">", " ", " m'VE", "fi", " ḍ̇"]} +{"text": "'M<|fim_prefix|>ع\t<ꟲ'S><|endoftext|>‍'ſ\u000b́
EOTé­
 \n9", "tokens": 37, "pieces": ["'M", "<|", "fim", "_prefix", "|>", "ع", "\t", "<", "ꟲ'S", "><|", "endoftext", "|>‍'", "ſ", "\u000b́", "
EOTé", "­", "
 \n", "9"]} +{"text": "\rß\rZa\tm 👍🏽0 🙂漢EOT'll're''A\u000b< \naa-'llå\r\nḍ̇t­<12345678'Mſ", "tokens": 40, "pieces": ["\r", "ß", "\r", "Za", "\tm", " ", "👍🏽", "0", " ", "🙂漢", "EOT'll", "'re", "''", "A", "\u000b", "<", " \n", "aa", "-'", "llå", "\r\n", "ḍ̇t", "­<", "123", "456", "78", "'Mſ"]} +{"text": " ßd٣٤٥٦EOT\r\n\r\n9-🙂'ſEOT\r\u000b<|fim_prefix|>'🙂 !!字mZꟲ<|endoftext|>!!٣٤٥٦'ſ!'ſ!!ſ", "tokens": 51, "pieces": [" ßd", "٣٤٥", "٦", "EOT", "\r\n\r\n", "9", "-🙂'", "ſ", "EOT", "\r", "\u000b", "<|", "fim", "_prefix", "|>'🙂", " !!", "字m", "Zꟲ", "<|", "endoftext", "|>!!", "٣٤٥", "٦", "'ſ", "!'", "ſ", "!!", "ſ"]} +{"text": "0're<…,12345678­Džå'Tḍ̇'Re👍🏽\nDž漢𐞁", "tokens": 34, "pieces": ["0", "'re", "<", "…", ",", "123", "456", "78", "­", "Džå'T", "ḍ̇'Re", "👍🏽\n", "Dž漢𐞁"]} +{"text": "-'VE…<|fim_prefix|>'́́'0e‍s,''ll'ſ\t…s0👍🏽é", "tokens": 34, "pieces": ["-'", "VE", "…", "<|", "fim", "_prefix", "|>'́́'", "0", "e", "‍s", ",''", "ll'ſ", "\t", "…s", "0", "👍🏽", "é"]} +{"text": "ḍ̇  ३'Tḍ̇漢 ſ's…'D…  <\rعİétḍ̇å🙂0é\u000bع<|endoftext|>'T\u000b३", "tokens": 54, "pieces": ["ḍ̇", " <", "EOT", ">", " ", "३", "'Tḍ̇漢", " ſ's", "…", "'D", "… ", " ", "<\r", "عİétḍ̇å", "🙂", "0", "é", "\u000bع", "<|", "endoftext", "|>'", "T", "\u000b", "३", ""]} +{"text": "٣٤٥٦(é👍🏽e9 \n(😀🏽0\r'Sſ're漢é\r\né👍🏽,0 ſds's㍿9ḍ̇'re(🙂ſa ", "tokens": 47, "pieces": ["٣٤٥", "٦", "(é", "👍🏽", "e", "9", " \n", "(😀🏽", "0", "\r", "'Sſ're", "漢é", "\r\n", "é", "👍🏽,", "0", " ſds's", "㍿", "9", "ḍ̇'re", "(🙂", "ſa", " "]} +{"text": "ßİ0​'ſ#$%​\"㋿́'Sꟲ<ع-٣٤٥٦", "tokens": 26, "pieces": ["ß", "İ", "0", "​'", "ſ", "#$%​\"㋿́'", "Sꟲ", "<ع", "-", "٣٤٥", "٦"]} +{"text": "😀🏽<|fim_prefix|>s!'M!!😀🏽\u000bea", "tokens": 18, "pieces": ["😀🏽<|", "fim", "_prefix", "|>", "s", "!'", "M", "!!😀🏽", "\u000bea"]} +{"text": ".½\r\n𐞁ſ㋿½👍🏽Dž''ll㋿'S's'ſ½0 mfi Z,s'Redİ", "tokens": 38, "pieces": [".", "½", "\r\n", "𐞁ſ", "㋿", "½", "👍🏽", "Dž", "''", "ll", "㋿'", "S's", "'ſ", "½0", " mfi", " Z", ",s'Re", "d", "İ"]} +{"text": "㍿", "tokens": 3, "pieces": ["㍿"]} +{"text": "é12345678fi 𐞁'ſ<|fim_prefix|>३!!İ\t0Z!d.<|fim_prefix|>🙂#$%>…!", "tokens": 43, "pieces": ["é", "123", "456", "78", "fi", " ", " 𐞁'ſ", "<|", "fim", "_prefix", "|><", "META", "_START", ">", "३", "!!", "İ", "\t", "0", "Z", "!d", ".<|", "fim", "_prefix", "|>🙂#$%>", "…", "!"]} +{"text": "#$%㍿'Mm
,!!0'S'ſ'VE'D", "tokens": 22, "pieces": ["#$%㍿'", "M", "m", "
", ",!!", "0", "'S'ſ", "'VE'D"]} +{"text": "<|endoftext|>'S字m'VEt́9#$%''ret\r'll­", "tokens": 22, "pieces": ["<|", "endoftext", "|>'", "S字m'VE", "t́", "9", "#$%''", "ret", "\r", "'ll", "­"]} +{"text": "mⅣé", "tokens": 4, "pieces": ["m", "Ⅳ", "é"]} +{"text": "DžEOT ('Re'll½🙂㋿ⅣÁ(👍🏽>́", "tokens": 29, "pieces": ["DžEOT", " ", " (<", "EOT", ">'", "Re'll", "½", "🙂㋿", "Ⅳ", "Á", "(👍🏽><", "META", "_START", ">́"]} +{"text": "३ A(­'D s Dž🙂Dž \n㍿ 'ſ \t🙂A ३#$%Dž0'T'0\u000b", "tokens": 38, "pieces": [" ", "😀🏽", "A", "<|", "endoftext", "|>", " \n", "㍿", " ", "'ſ", " ", "\t", "🙂A", " ", "३", "#$%", "Dž", "", "0", "'T", "'", "0", "\u000b"]} +{"text": "A<|endoftext|>fiß9ꟲ<|endoftext|>e ½ >㋿İ😀🏽' ㍿😀🏽́\u000b३\n\u000b9d'M'Tße", "tokens": 53, "pieces": ["A", "<|", "endoftext", "|>", "fiß", "9", "ꟲ", "<|", "endoftext", "|>", "e", " ", "½", " ", ">㋿", "İ", "😀🏽'", " ", "㍿😀🏽́", "\u000b", "३", "\n", "\u000b", "9", "d'M", "'Tße"]} +{"text": "…½é\r\ne\t9 #$%'ll(s漢'ſ<|endoftext|>A<|fim_prefix|>'S<|endoftext|>'D>EOT㋿٣٤٥٦🙂<½'M½té>'T", "tokens": 59, "pieces": ["…", "½", "é", "\r\n", "e", "\t", "9", " ", " #$%'", "ll", "(s漢'ſ", "<|", "endoftext", "|>", "A", "<|", "fim", "_prefix", "|>'", "S", "<|", "endoftext", "|>'", "D", ">EOT", "㋿", "٣٤٥", "٦", "🙂<", "½", "'M", "½", "té", ">'", "T"]} +{"text": " Zma<‍és'VE'sm'S \n👍🏽ꟲ9😀🏽", "tokens": 25, "pieces": [" Zma", "<‍", "és'VE", "'sm'S", " \n", "👍🏽", "ꟲ", "9", "😀🏽"]} +{"text": "ß'VE<|fim_prefix|>d\t!!ḍ̇å\ńſ\"́é​ 'D'VE㋿!'sé😀🏽EOT 'lld😀🏽́'s", "tokens": 52, "pieces": ["ß'VE", "<|", "fim", "_prefix", "|>", "d", "\t", "!!", "ḍ̇å", "\n", "́ſ", "\"́é", "​", " ", "'D'VE", "㋿!'", "sé", "😀🏽", "EOT", " ", "'lld", "😀🏽́'", "s", ""]} +{"text": "é<|endoftext|>EOT́<|fim_prefix|> \n( fiß-㋿'ſ<|fim_prefix|>,'s \n#$%-
\r", "tokens": 42, "pieces": ["é", "<|", "endoftext", "|>", "EOT́", "<|", "fim", "_prefix", "|>", " \n", "(", " fiß", "-㋿'", "ſ", "<|", "fim", "_prefix", "|>,'", "s", " \n", "#$%-", "
\r"]} +{"text": "
é٣٤٥٦'D \n'll\u000bå'MEOT 'll'ſⅣ'\rfifid😀🏽­…字😀🏽३'ll'Dd,ḍ̇", "tokens": 46, "pieces": ["
é", "٣٤٥", "٦", "'D", " \n", "'ll", "\u000bå'M", "EOT", " ", " '", "ll'ſ", "Ⅳ", "'\r", "fifid", "😀🏽­", "…字", "😀🏽", "३", "'ll'D", "d", ",ḍ̇"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿½<ſ'll \n(½'ll字'Re!!!!<…ꟲ½a\tA'VE'lltⅣDžm'Re ", "tokens": 35, "pieces": ["㍿", "½", "<ſ'll", " \n", "(", "½", "'ll字'Re", "!!!!<", "…ꟲ", "½", "a", "\tA'VE", "'llt", "Ⅳ", "Džm'Re", " "]} +{"text": "'s…\u000b\r-'s'll's'ſ​\r\n\r\n'Sm\u000b½ !\ré9😀🏽s ſ", "tokens": 30, "pieces": ["'s", "…\u000b\r", "-'", "s'll", "'s'ſ", "​\r\n\r\n", "'Sm", "\u000b", "½", " !\r", "é", "9", "😀🏽<", "EOT", ">s", " ſ"]} +{"text": "😀🏽ſ'ſİ
字ém'Re \n​<́३ 'T", "tokens": 24, "pieces": ["😀🏽", "ſ'ſ", "İ", "
字", "ém'Re", " \n", "​<́", "३", " ", "'T"]} +{"text": "'VE<#$%'re<|fim_prefix|>!fié٣٤٥٦'s>fis\"<|fim_prefix|> ३'ll\r\n12345678𐞁>,", "tokens": 42, "pieces": ["'VE", "<#$%'", "re", "<|", "fim", "_prefix", "|>!", "fié", "٣٤٥", "٦", "'s", ">fis", "\"<|", "fim", "_prefix", "|>", " ", "३", "'ll", "\r\n", "123", "456", "78", "𐞁", ">,"]} +{"text": " 
a éꟲ👍🏽
ꟲ'ſ 'VEtZ🙂½漢", "tokens": 30, "pieces": [" ", "
a", " éꟲ", "👍🏽", "
ꟲ'ſ", " '", "VEt", "<", "EOT", ">Z", "🙂", "½", "漢"]} +{"text": "Ⅳİ''ll\t…å😀🏽 \n\u000b!عDž", "tokens": 19, "pieces": ["Ⅳ", "İ", "''", "ll", "\t", "…å", "😀🏽", " \n", "\u000b", "!ع", "Dž"]} +{"text": "\r\nfi 👍🏽!!,😀🏽İ字EOT\r\n\r\nⅣⅣ.ḍ̇…🙂Ⅳ
(('Ss t́Ⅳ \nZ\t 's\r\n\r\n!!", "tokens": 44, "pieces": ["\r\n", "fi", " ", "👍🏽!!,😀🏽", "İ字", "EOT", "\r\n\r\n", "ⅣⅣ", ".ḍ̇", "…", "🙂", "Ⅳ", "
", "(('", "Ss", " ", " t́", "Ⅳ", " \n", "Z", "\t ", " '", "s", "\r\n\r\n", "!!"]} +{"text": "é", "tokens": 1, "pieces": ["é"]} +{"text": "­-👍🏽A😀🏽s\t.12345678½ \r\nḍ̇''Re​‍.
sḍ̇a", "tokens": 30, "pieces": ["­-👍🏽", "A", "😀🏽", "s", "\t", ".", "123", "456", "78½", " \r\n", "ḍ̇", "''", "Re", "​‍.", "
sḍ̇a"]} +{"text": "A‍👍🏽\t​ås字 ع'Ds", "tokens": 14, "pieces": ["A", "‍👍🏽", "\t", "​ås字", " ع'D", "s"]} +{"text": "9", "tokens": 1, "pieces": ["9"]} +{"text": "Dž", "tokens": 2, "pieces": ["Dž"]} +{"text": "́,İt'VE!!å'ſⅣ🙂'ſ's9\r\nſ𐞁ḍ̇ <|fim_prefix|>Ź٣٤٥٦'Dİfiꟲß,s0a 字", "tokens": 55, "pieces": ["́", ",İt'VE", "!!", "å'ſ", "Ⅳ", "🙂'", "ſ's", "9", "\r\n", "ſ", "𐞁ḍ̇", " ", "<|", "fim", "_prefix", "|>", "Ź", "٣٤٥", "٦", "'Dİfiꟲß", ",s", "0", "a", " ", " 字"]} +{"text": "'S<|fim_prefix|><<|fim_prefix|>…-\r\n\"é\rZ'llع(fi", "tokens": 28, "pieces": ["'S", "<|", "fim", "_prefix", "|><<|", "fim", "_prefix", "|><", "META", "_START", ">", "…", "-\r\n", "\"é", "\r", "Z'll", "ع", "(fi"]} +{"text": "s'½ \nééſ'll👍🏽'ſ😀🏽fi\r\n9é‍'VE\r… 𐞁\r'Dİſ\"é 字İé'M'DA🙂,ſ", "tokens": 52, "pieces": ["s", "'", "½", " \n", "ééſ'll", "👍🏽'", "ſ", "😀🏽", "fi", "\r\n", "9", "é", "‍'", "VE", "\r", "…", "", " 𐞁", "\r", "'Dİſ", "\"é", " ", " 字İé'M", "'DA", "🙂,", "ſ"]} +{"text": "'re fi e'D('D", "tokens": 8, "pieces": ["'re", " ", " fi", " e'D", "('", "D"]} +{"text": "9're字­'D字'🙂12345678å>🙂's३ſ \n", "tokens": 21, "pieces": ["9", "'re字", "­'", "D字", "'🙂", "123", "456", "78", "å", ">🙂'", "s", "३", "ſ", " \n"]} +{"text": "12345678😀🏽!字\r\n'Tḍ̇٣٤٥٦­½'T'VEⅣ\"", "tokens": 29, "pieces": ["123", "456", "78", "😀🏽!", "字", "\r\n", "'Tḍ̇", "٣٤٥", "٦", "­", "½", "'T'VE", "", "Ⅳ", "\""]} +{"text": ".👍🏽\r\n\r\nd­‍<|endoftext|>😀🏽 Ⅳ<|endoftext|>12345678m\r\n\r\n'S漢ꟲ", "tokens": 38, "pieces": [".👍🏽\r\n\r\n", "d", "­‍<|", "endoftext", "|>😀🏽", " ", "Ⅳ", "<|", "endoftext", "|>", "123", "456", "78", "m", "\r\n\r\n", "'S漢ꟲ"]} +{"text": ".>#$%Z12345678…㍿", "tokens": 13, "pieces": [".>#$%", "Z", "123", "456", "78", "…", "㍿"]} +{"text": "#$% \n'Re \n<́(ꟲ㍿​fiéfi\t​Ⅳ'S ́ßemd\t0\n", "tokens": 33, "pieces": ["#$%", " \n", "'Re", " \n", "<́", "(<", "EOT", ">ꟲ", "㍿​", "fiéfi", "\t", "​", "Ⅳ", "'S", " ́ßemd", "\t", "0", "\n"]} +{"text": "(…'lleá#$%EOT!! ​9ß­'res\"ⅣⅣ-'M३e#$%'s12345678'll<|fim_prefix|>…Džfi‍åZ<|endoftext|>", "tokens": 60, "pieces": ["(", "…", "'lleá", "#$%", "EOT", "!!", " ", "​", "9", "ß", "­'", "res", "\"", "ⅣⅣ", "-'", "M", "३", "e", "#$%'", "s", "123", "456", "78", "'ll", "<|", "fim", "_prefix", "|>", "…Džfi", "‍å", "Z", "<|", "endoftext", "|>"]} +{"text": "
‍३'re0\r\nDž­🙂a👍🏽­㍿-ḍ̇Dž", "tokens": 28, "pieces": ["
", "‍", "३", "'re", "0", "\r\n", "Dž", "­🙂", "a", "👍🏽­<", "EOT", ">㍿-", "ḍ̇", "Dž"]} +{"text": "\té'ſ \nḍ̇aꟲ''VE'VEé 漢…' A'Re's.", "tokens": 32, "pieces": ["\té'ſ", " \n", "ḍ̇aꟲ", "''", "VE'VE", "é", " ", " 漢", "…", "'", " A'Re", "'s", ".<", "META", "_START", ">"]} +{"text": "12345678'stⅣع.… ,EOT字İ漢㋿ !!'M🙂㋿
", "tokens": 32, "pieces": ["123", "456", "78", "'st", "Ⅳ", "ع", ".", "…", " ", ",EOT字İ漢", "㋿", " ", "!!<", "META", "_START", ">'", "M", "🙂㋿", "
"]} +{"text": " \n.Aİ,½👍🏽're\t'VE𐞁عéDžſ(#$%'reſ ㋿😀🏽ع!å𐞁‍s👍🏽å­Z'", "tokens": 58, "pieces": [" \n", ".A", "İ", ",", "½", "👍🏽'", "re", "\t", "'VE𐞁عé", "Džſ", "(#$%'", "reſ", " ", " ㋿😀🏽", "ع", "!å𐞁", "‍s", "👍🏽", "å", "­Z", "'"]} +{"text": "ḍ̇ꟲms12345678e'S9٣٤٥٦字ḍ̇\"ع㍿İ'M'llſ\r\n 🙂a<|endoftext|>fi漢!३<|fim_prefix|> \u000b٣٤٥٦>", "tokens": 62, "pieces": ["ḍ̇ꟲms", "123", "456", "78", "e'S", "9", "", "٣٤٥", "٦", "字ḍ̇", "\"ع", "㍿İ'M", "'llſ", "\r\n", " ", " 🙂", "a", "<|", "endoftext", "|>", "fi漢", "!", "३", "<|", "fim", "_prefix", "|>", " ", "\u000b", "٣٤٥", "٦", ">"]} +{"text": "ßåع0ع12345678½㋿ Džꟲtſ-sꟲ#$%'ſ'ſt''D sDž\"a‍Ⅳ字12345678 \n", "tokens": 49, "pieces": ["ßå", "ع", "0", "ع", "123", "456", "78½", "㋿", " Džꟲtſ", "-sꟲ", "#$%'", "ſ'ſ", "t", "''", "D", " ", " s", "Dž", "\"a", "‍", "Ⅳ", "字", "123", "456", "78", " \n"]} +{"text": "m<|endoftext|>Ⅳ\r\n\r\n!\r\n\r\n'漢  'S'Ses'D're𐞁'VE.aDžm(", "tokens": 36, "pieces": ["m", "<|", "endoftext", "|>", "Ⅳ", "\r\n\r\n", "!\r\n\r\n", "'漢", " ", " ", "'S'S", "es'D", "'re𐞁'VE", ".a", "Džm", "("]} +{"text": "\téa\t0t漢ꟲEOT\u000bEOT́e­-\u000b​", "tokens": 24, "pieces": ["\téa", "\t", "0", "t漢ꟲ", "EOT", "\u000bEOT́e", "­<", "EOT", ">-", "\u000b", "​"]} +{"text": "<|endoftext|>½EOTfiḍ̇​½́
\"ḍ̇#$%Dž​Aꟲ\"<", "tokens": 32, "pieces": ["<|", "endoftext", "|>", "½", "EOTfiḍ̇", "​", "½", "́", "
", "\"ḍ̇", "#$%", "Dž", "​Aꟲ", "\"<"]} +{"text": "٣٤٥٦0(😀🏽'lld0\r.<İ'Re'Re#$%<|endoftext|>#$%<|endoftext|>ß­'Re>sa", "tokens": 44, "pieces": ["٣٤٥", "٦0", "(😀🏽'", "lld", "0", "\r", ".<", "EOT", "><", "İ'Re", "'Re", "#$%<|", "endoftext", "|>#$%<|", "endoftext", "|>", "ß", "­'", "Re", ">sa"]} +{"text": "😀🏽12345678㋿ḍ̇.𐞁\r\n<|endoftext|>,ß -😀🏽é.'S9\r\n\r\n !!漢", "tokens": 38, "pieces": ["😀🏽", "123", "456", "78", "㋿ḍ̇", ".𐞁", "\r\n", "<|", "endoftext", "|>,", "ß", " -😀🏽", "é", ".'", "S", "9", "\r\n\r\n", " ", "!!", "漢"]} +{"text": "👍🏽\u000b ſ>‍!!字٣٤٥٦\"ḍ̇'३ß9é३\"e\nß0
!Ad  A9\n", "tokens": 40, "pieces": ["👍🏽", "\u000b ", " ſ", ">‍!!", "字", "٣٤٥", "٦", "\"ḍ̇", "'", "३", "ß", "9", "é", "३", "\"e", "\n", "ß", "0", "
", "!Ad", " ", " A", "9", "\n"]} +{"text": "t12345678 'ſfiḍ̇ſfiⅣ㍿>", "tokens": 18, "pieces": ["t", "123", "456", "78", " '", "ſfiḍ̇ſfi", "Ⅳ", "㍿>"]} +{"text": "('Mꟲ字'Dß-<|endoftext|>٣٤٥٦\u000bd\r\nſ😀🏽\n…aémA…㍿½👍🏽३s
 ", "tokens": 63, "pieces": [" ", " <'", "s", "Z", " ", "‍", "½", "e'ſ", "e", "
e", "9", " \n", "<|", "fim", "_prefix", "|>'", "Dß", "-<|", "endoftext", "|>", "٣٤٥", "٦", "\u000bd", "\r\n", "ſ", "😀🏽\n", "…aém", "A", "…", "㍿", "½", "👍🏽", "३", "s", "
 "]} +{"text": " ſ>!'Re㋿'M's'T,İ!ع9-(­å㍿ع-'ll𐞁'Re​d😀🏽ꟲ !!😀🏽", "tokens": 49, "pieces": [" ſ", ">!'", "Re", "㋿'", "M's", "'T", ",İ", "!ع", "9", "-(­", "å", "㍿ع", "-'", "ll", "𐞁'Re", "​d", "😀🏽", "ꟲ", " ", "!!😀🏽"]} +{"text": "३- \n٣٤٥٦é ſ!!'ll'ſ \n>'å­'ll½'M𐞁́Dž​<|endoftext|>!fißſa12345678<|endoftext|>́9'", "å", "­'", "ll", "½", "'M𐞁́", "Dž", "​<|", "endoftext", "|>!", "fißſa", "123", "456", "78", "<|", "endoftext", "|>́", "9", "字<|endoftext|>(0​\rA 0İ㍿'D㋿ḍ̇㋿\"Z", "tokens": 44, "pieces": [" \n", "9", "👍🏽", "é're", "字", "<|", "endoftext", "|>(", "0", "​\r", "A", " ", "0", "İ", "㍿'", "D", "㋿ḍ̇", "㋿\"", "Z"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'😀🏽ع!𐞁 s\"😀🏽字٣٤٥٦٣٤٥٦İ,٣٤٥٦ꟲ#$%,\r\n\r\n字a", "tokens": 39, "pieces": ["'😀🏽", "ع", "!𐞁", " s", "\"😀🏽", "字", "٣٤٥", "٦٣٤", "٥٦", "İ", ",", "٣٤٥", "٦", "ꟲ", "#$%,\r\n\r\n", "字a"]} +{"text": "\rDž'ſ're12345678‍'Re'T\"\r'M
ع\u000b\tt", "tokens": 24, "pieces": ["Z漢", "-'", "VEDž", " ", " 🙂,", "9", "字", "9", "…", "\"\r", "'M", "
ع", "\u000b", "\tt"]} +{"text": "<ḍ̇漢ꟲ'T<('Reꟲع\r\n\r\nDžⅣ'll㍿!!Dž \n'T'Dعfi<|fim_prefix|>漢é­m'ſ㋿ \n㋿0\u000b'refi", "tokens": 63, "pieces": ["<ḍ̇漢", "ꟲ'T", "<('", "Reꟲع", "\r\n\r\n", "Dž", "Ⅳ", "'ll", "㍿!!", "Dž", " \n", "'T'D", "عfi", "<|", "fim", "_prefix", "|>", "漢é", "­m'ſ", "㋿", " \n", "㋿", "0", "\u000b", "'", "refi"]} +{"text": "0'Tt漢漢 <|endoftext|><|fim_prefix|><|fim_prefix|>'T\"m'S½\u000bt'ſḍ̇>ßſ'Tt>s éZ 12345678'Ma­'Re😀🏽", "tokens": 56, "pieces": ["0", "'Tt漢漢", " ", "<|", "endoftext", "|><|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>'", "T", "\"m'S", "½", "\u000bt'ſ", "ḍ̇", ">ßſ'T", "t", ">s", " é", "Z", " ", "123", "456", "78", "'Ma", "­'", "Re", "😀🏽"]} +{"text": "'VE'll\r\n\r\n's'M're­'ss'D३ß㍿", "tokens": 16, "pieces": ["'VE'll", "\r\n\r\n", "'s'M", "'re", "­'", "ss'D", "३", "ß", "㍿"]} +{"text": "​㍿㍿ 字३३Dž!12345678s'VEt漢­EOT", "tokens": 25, "pieces": ["​㍿㍿", " ", " 字", "३३", "Dž", "!", "123", "456", "78", "s'VE", "t漢", "­EOT"]} +{"text": "'S‍s😀🏽'Re ㋿émfi0", "tokens": 18, "pieces": ["'S", "‍s", "😀🏽'", "Re", " ", " ㋿", "émfi", "0"]} +{"text": "#$%dé'ſ́Z'MEOT.'VE>", "tokens": 14, "pieces": ["#$%", "dé'ſ", "́", "Z'M", "EOT", ".'", "VE", ">"]} +{"text": "t>Dž'Re aEOT­\r\n\r\n字9EOT-!(\u000b'Mm-", "tokens": 23, "pieces": ["t", ">Dž'Re", " a", "EOT", "­\r\n\r\n", "字", "9", "EOT", "-!(", "\u000b", "'Mm", "-"]} +{"text": "字㍿ !('T \r9'll😀🏽'D EOT‍<|endoftext|>(Dž-'D​\r\nEOT", "tokens": 43, "pieces": ["字", "㍿", " ", "!('", "T", " \r", "9", "'ll", "😀🏽'", "D", "", " EOT", "‍<|", "endoftext", "|><", "EOT", ">(", "Dž", "-'", "D", "​\r\n", "EOT"]} +{"text": ">漢㋿d t😀🏽‍Dž \n'TsdEOT\"<|endoftext|>ع🙂\r👍🏽(ß9's,", "tokens": 36, "pieces": [">漢", "㋿d", " t", "😀🏽‍", "Dž", " \n", "'Tsd", "EOT", "\"<|", "endoftext", "|>", "ع", "🙂\r", "👍🏽(", "ß", "9", "'s", ","]} +{"text": "㋿\u000b‍EOTåt字\r\né\r #$%ꟲ>㋿é'S 'm.𐞁", "tokens": 35, "pieces": ["㋿", "\u000b", "‍EOTåt字", "\r\n", "é", "\r", " ", " #$%", "ꟲ", ">㋿", "é'S", " '", "m", ".𐞁"]} +{"text": "\r\n 'ſḍ̇​'Re,", "tokens": 13, "pieces": ["\r\n", " <", "META", "_START", ">'", "ſḍ̇", "​'", "Re", ","]} +{"text": "'ſ!
\"'S>'sſ'Re.
 ½<|fim_prefix|> \nd'<'ſé'll", "tokens": 31, "pieces": ["'ſ", "!", "
", "\"'", "S", ">'", "s", "ſ'Re", ".", "
", " ", "½", "<|", "fim", "_prefix", "|>", " \n", "d", "'<'", "ſé'll"]} +{"text": "(Ź\n<|fim_prefix|>́'VEA<|fim_prefix|>", "tokens": 19, "pieces": ["(Ź", "\n", "<|", "fim", "_prefix", "|>́'", "VEA", "<|", "fim", "_prefix", "|>"]} +{"text": "😀🏽'VEfiḍ̇\r\n\r\nßéſ#$%Am!'re½ 'Re­३\nⅣ (>'reعe\r漢", "tokens": 37, "pieces": ["😀🏽'", "VEfiḍ̇", "\r\n\r\n", "ßéſ", "#$%", "Am", "!'", "re", "½", " ", "'Re", "­", "३", "\n", "Ⅳ", " (>'", "reعe", "\r", "漢"]} +{"text": "'s㍿m'VE>ßß'ſ (Ⅳ‍'Dßß", "'", "ſ", " ", "(", "Ⅳ", "‍'", "D", "字", "tokens": 37, "pieces": ["३", "𐞁", "-", "\t", "…", "㍿Dž字", "!!👍🏽", "ꟲ'VE", "é", "#$%\r\n", "<'", "s", "<|", "endoftext", "|>", "字"]} +{"text": "㋿9e,🙂å\r\n­'M>.eꟲ.'st'ſ३0Z'M.", "tokens": 32, "pieces": ["㋿", "9", "e", ",🙂", "å", "\r\n", "­'", "M", ">.", "eꟲ", ".'", "st'ſ", "३", "", "0", "Z'M", "."]} +{"text": " \n'Dž.'ssé
9́(#$%'M'S㍿Ⅳ漢éé's \n­\r\u000b'M😀🏽", "tokens": 44, "pieces": [" \n", "'Dž", ".'", "ssé", "
", "9", "́", "(#$%'", "M'S", "㍿", "Ⅳ", "漢éé's", " \n", "­\r", "", "\u000b", "'", "M", "😀🏽"]} +{"text": "ḍ̇é漢", "tokens": 5, "pieces": ["ḍ̇é漢"]} +{"text": "Z12345678\u000b('D,é>'<|endoftext|>'\r\n漢'sm<|endoftext|>Zt", "tokens": 33, "pieces": ["Z", "123", "456", "78", "\u000b", "('", "D", ",é", ">'<|", "endoftext", "|>'\r\n", "漢's", "m", "<|", "endoftext", "|>", "Zt"]} +{"text": "!!'re'VE'll're ع'D \u000b9!!", "tokens": 14, "pieces": ["!!'", "re'VE", "'ll're", " ع'D", " ", "\u000b", "9", "!!"]} +{"text": "
å å‍>👍🏽\r\n<|endoftext|>ݽe­ꟲDž-ḍ̇!!ꟲfi<|fim_prefix|>\t'VEs <é\r\n\r\n'll٣٤٥٦", "tokens": 55, "pieces": ["
å", " ", " å", "‍>👍🏽\r\n", "<|", "endoftext", "|>", "İ", "½", "e", "­ꟲ", "Dž", "-ḍ̇", "!!", "ꟲfi", "<|", "fim", "_prefix", "|>", "\t", "'VEs", " <", "é", "\r\n\r\n", "'ll", "٣٤٥", "٦"]} +{"text": "\r\n'VE>dd́字'Re'lle's\né  t३ 漢'ſ!!#$%漢३\n\"
'👍🏽😀🏽EOT", "tokens": 38, "pieces": ["\r\n", "'VE", ">dd́字'Re", "'lle's", "\n", "é", "  ", " t", "३", " 漢'ſ", "!!#$%", "漢", "३", "\n", "\"", "
", "'👍🏽😀🏽", "EOT"]} +{"text": "etd12345678 ", "tokens": 9, "pieces": ["etd", "123", "456", "78", "", " "]} +{"text": "\n½'reİİ\"\n 'M>'ſ's\u000b'VE㋿d 🙂a\nEOTé​", "tokens": 38, "pieces": ["\n", "½", "'re", "İİ", "\"\n", " '", "M", ">'", "ſ's", "\u000b", "'VE", "㋿d", "", " ", "🙂a", "\n", "EOTé", "​"]} +{"text": ".\r\n\r\n٣٤٥٦", "tokens": 5, "pieces": [".\r\n\r\n", "٣٤٥", "٦"]} +{"text": "å字( å .s- m३9( ́!​<'Tm", "tokens": 38, "pieces": ["å字", "(", " å", " ", " .", "s", "-", " m", "३9", "(<", "EOT", ">", " ́", "!​<<", "META", "_START", "><", "EOT", ">'", "Tm"]} +{"text": "#$%ع\"\n", "tokens": 4, "pieces": ["#$%", "ع", "\"\n"]} +{"text": " \nſꟲ-👍🏽(mm#$%!!(\t
'D\" 
'Ré#$%>", "tokens": 37, "pieces": [" \n", "ſ", "ꟲ", "-👍🏽(", "mm", "#$%!!(", "\t", "
", "'D", "\"", " ", " <", "EOT", ">", "
", "'Re", "́", "#$%>"]} +{"text": "!! \nßåå", "tokens": 7, "pieces": ["!!", " \n", "ßåå"]} +{"text": "ꟲ'VE­́Ⅳte!-'M#$%", "tokens": 15, "pieces": ["ꟲ'VE", "­́", "Ⅳ", "te", "!-'", "M", "#$%"]} +{"text": "​㋿0\r'M漢é́Dždåe s \nå😀🏽'Resd \n३>'ll\u000b 'refi'lla'VE½\n", "tokens": 40, "pieces": ["​㋿", "0", "\r", "'M漢é́", "Dždåe", " s", " \n", "å", "😀🏽'", "Resd", " \n", "३", ">'", "ll", "\u000b ", " '", "refi'll", "a'VE", "½", "\n"]} +{"text": "e'Re9٣٤٥٦​٣٤٥٦'T", "tokens": 13, "pieces": ["e'Re", "9٣٤", "٥٦", "​", "٣٤٥", "٦", "'T"]} +{"text": "🙂m", "tokens": 2, "pieces": ["🙂m"]} +{"text": " 
́'ll字İ", "tokens": 6, "pieces": [" ", "
́'ll", "字", "İ"]} +{"text": "å'Mꟲ'S\t٣٤٥٦३…e­ ­å३é<|endoftext|>  \r\t.", "tokens": 35, "pieces": ["å'M", "ꟲ'S", "\t", "٣٤٥", "٦३", "…e", "­", " ", "­å", "३", "é", "<|", "endoftext", "|>", "  \r", "\t", "."]} +{"text": "٣٤٥٦t𐞁 \n\r\n!éDž>'T'S.'s0'ſ ", "tokens": 24, "pieces": ["٣٤٥", "٦", "t𐞁", " \n\r\n", "!é", "Dž", ">'", "T'S", ".'", "s", "0", "'ſ", " "]} +{"text": "٣٤٥٦'reſ \n\r\n<Ⅳ \r३0
s½'VE \n#$%٣٤٥٦\u000b𐞁", "tokens": 41, "pieces": ["٣٤٥", "٦", "'reſ", " \n\r\n", "<", "Ⅳ", " ", "\r", "३0", "
s", "½", "'VE", " \n", "#$%", "٣٤٥", "٦", "\u000b", "𐞁"]} +{"text": "' ḍ̇…d'é<|endoftext|>EOT \u000b㋿'Re'll­ \n😀🏽½𐞁<|endoftext|>#$%é9'VE9's(İ", "tokens": 55, "pieces": ["'", " ḍ̇", "…d", "'é", "<|", "endoftext", "|>", "EOT", " ", "\u000b", "㋿'", "Re'll", "­", " \n", "😀🏽", "½", "𐞁", "<|", "endoftext", "|>#$%", "é", "9", "'", "VE", "9", "'s", "(İ"]} +{"text": "​'漢00
…㍿!😀🏽‍-'𐞁 😀🏽\r\n\r\n👍🏽A漢字<<𐞁m\r\n", "tokens": 42, "pieces": ["​'", "漢", "00", "
", "", "…", "㍿!😀🏽‍-'", "𐞁", " ", " 😀🏽\r\n\r\n", "👍🏽", "A漢字", "<<", "𐞁m", "\r\n"]} +{"text": "fi‍(", "tokens": 3, "pieces": ["fi", "‍("]} +{"text": "­!ꟲ字é㍿m  ㍿éAⅣEOTfi 'ſ𐞁a \n‍ſ é\r\n\r\n\"å'S", "tokens": 48, "pieces": ["­!", "ꟲ字é", "㍿m", " ", " ", "㍿é", "A", "Ⅳ", "EOTfi", " ", "'ſ𐞁a", " \n", "‍<", "EOT", ">ſ", " é", "\r\n\r\n", "\"å'S"]} +{"text": "Z \n!!ß(<|endoftext|>EOT३ ḍ̇>åEOTß!字ꟲ½字.㍿!!é 'VE\t", "tokens": 43, "pieces": ["Z", " \n", "!!", "ß", "(<|", "endoftext", "|>", "EOT", "३", " ", "ḍ̇", ">å", "EOTß", "!字ꟲ", "½", "字", ".㍿!!", "é", " '", "VE", "\t"]} +{"text": "'s\t#$%‍‍  \n Dž ٣٤٥٦ 0.fiꟲ<\u000b'D…'re12345678ꟲ…㍿ع'T<|fim_prefix|>\tå­'T'D>!!'ſm", "tokens": 59, "pieces": ["'s", "\t", "#$%‍‍", "  \n", " Dž", " ", "٣٤٥", "٦", " ", "0", ".fiꟲ", "<", "\u000b", "'D", "…", "'re", "123", "456", "78", "ꟲ", "…", "㍿ع'T", "<|", "fim", "_prefix", "|>", "\tå", "­'", "T'D", ">!!'", "ſm"]} +{"text": " 'VEع'reeⅣ \n\r\n٣٤٥٦३ ‍ EOT-. 😀🏽'sa", "tokens": 29, "pieces": [" ", "'VEع're", "e", "Ⅳ", " \n", "\r\n", "٣٤٥", "٦३", " ‍", " ", " EOT", "-.", " ", "😀🏽'", "sa"]} +{"text": "㍿EOTſ<12345678𐞁\tßa .​t漢<|fim_prefix|><ع😀🏽\u000b\n
 字Z0Dž<|fim_prefix|>", "tokens": 47, "pieces": ["㍿EOTſ", "<", "123", "456", "78", "𐞁", "\tßa", " ", " .​", "t漢", "<|", "fim", "_prefix", "|><", "ع", "😀🏽", "\u000b\n", "
", " 字", "Z", "0", "Dž", "<|", "fim", "_prefix", "|>"]} +{"text": "\"'T'D'Re ́'sİ<|fim_prefix|>\r३\r\nꟲaßſ½́½'ll", "tokens": 31, "pieces": ["\"'", "T'D", "'Re", " ", " <", "META", "_START", ">́'s", "İ", "<|", "fim", "_prefix", "|>\r", "३", "\r\n", "ꟲaßſ", "½", "́", "½", "'ll"]} +{"text": "<'s\nİ३#$%İd ‍ſ字㋿㍿ſ-字> 😀🏽́ß\na", "tokens": 30, "pieces": ["<'", "s", "\n", "İ", "३", "#$%", "İd", " ‍", "ſ字", "㋿㍿", "ſ", "-字", ">", " ", " 😀🏽́", "ß", "\n", "a"]} +{"text": "字 ", "tokens": 2, "pieces": ["字", " "]} +{"text": "-'lltİ12345678ꟲd½İ0ßḍ̇t-< 👍🏽", "tokens": 28, "pieces": ["-'", "llt", "İ", "123", "456", "78", "ꟲd", "½", "İ", "0", "ß", "ḍ̇t", "-<", " ", " 👍🏽"]} +{"text": " ㋿\r<|fim_prefix|> 0é!!.㋿'ſḍ̇é'll 'T#$%\t\r\n\r\n12345678fi\"\"Džé'Reع<", "tokens": 44, "pieces": [" ", "㋿\r", "<|", "fim", "_prefix", "|>", " ", "0", "é", "!!.㋿'", "ſḍ̇é'll", " ", " '", "T", "#$%", "\t\r\n\r\n", "123", "456", "78", "fi", "\"\"", "Džé'Re", "ع", "<"]} +{"text": "'llå,\r­ \u000b㍿\r\n\r\n\r'👍🏽\r\n\r\né", "tokens": 20, "pieces": ["'llå", ",\r", "­", " ", "\u000b", "㍿\r\n\r\n\r", "'👍🏽\r\n\r\n", "é"]} +{"text": "'ſ", "tokens": 2, "pieces": ["'ſ"]} +{"text": " 're'éé㍿ḍ̇DžZ(m😀🏽
عé\rⅣ'M㋿ḍ̇İé½\u000b", "tokens": 37, "pieces": [" '", "re", "'éé", "㍿ḍ̇", "DžZ", "(m", "😀🏽", "
عé", "\r", "Ⅳ", "'M", "㋿ḍ̇", "İé", "½", "\u000b"]} +{"text": "'ſⅣt𐞁­\n👍🏽🙂a🙂'M<|fim_prefix|>字Dž𐞁ع́㋿", "Ⅳ", "t𐞁", "­\n", "👍🏽🙂", "a", "🙂'", "M", "<|", "fim", "_prefix", "|>", "字Dž𐞁ع́", "㋿<", "é漢", "…", ",e", "(𐞁"]} +{"text": "​.'ś‍­.'ll9m٣٤٥٦ßⅣ'ſḍ̇'Re\tmß>𐞁#$%Ⅳ(\t !ꟲAZ 0d0", "tokens": 46, "pieces": ["​.'", "ś", "‍­.'", "ll", "9", "m", "٣٤٥", "٦", "ß", "Ⅳ", "'ſḍ̇'Re", "\tmß", ">𐞁", "#$%", "Ⅳ", "(", "\t ", " !", "ꟲ", "AZ", " ", "0", "d", "0"]} +{"text": "́!'DZ Dž<|fim_prefix|> \n'Re", "tokens": 14, "pieces": ["́", "!'", "DZ", " Dž", "<|", "fim", "_prefix", "|>", " \n", "'Re"]} +{"text": "Ⅳ\r🙂'M9-́ \n!!İ­\t!字'D<|fim_prefix|>ꟲe'VE \n's😀🏽mعꟲet 👍🏽'D字🙂'Re", "tokens": 53, "pieces": ["Ⅳ", "\r", "🙂'", "M", "9", "-́", " \n", "!!", "İ", "­", "\t", "!字'D", "<|", "fim", "_prefix", "|>", "ꟲe'VE", " \n", "'s", "😀🏽", "mعꟲet", " ", "👍🏽'", "D字", "🙂<", "EOT", ">'", "Re"]} +{"text": " ,> Z٣٤٥٦(‍'s…\r'Re>…é\u000bİꟲ'S'Re!\ndḍ̇'DéZ字 ſ'Re0ع'S \u000bİ", "tokens": 49, "pieces": [" ", ",>", " ", " Z", "٣٤٥", "٦", "(‍'", "s", "…\r", "'Re", ">", "…é", "\u000bİꟲ", "'", "S'Re", "!\n", "dḍ̇'D", "é", "Z字", " ſ'Re", "0", "ع'S", " ", "\u000bİ"]} +{"text": "'MZ‍!!'S३ß<|fim_prefix|>(<|fim_prefix|>e'red٣٤٥٦'VE \n'VEe0👍🏽 ", "tokens": 40, "pieces": ["'MZ", "‍!!'", "S", "३", "ß", "<|", "fim", "_prefix", "|>(<|", "fim", "_prefix", "|>", "e're", "d", "٣٤٥", "٦", "'VE", " \n", "'VEe", "0", "👍🏽", " "]} +{"text": "#$%‍ å𐞁é㍿ <EOTd'ſ12345678\r!漢­'Tfi\n.'Sm#$%", "tokens": 44, "pieces": ["#$%‍", " å𐞁é", "㍿", " ", "<<", "META", "_START", ">EOTd'ſ", "123", "456", "78", "\r", "!漢", "<", "META", "_START", ">­'", "Tfi", "\n", ".'", "Sm", "#$%"]} +{"text": ",漢½m'T३👍🏽Dž\r\n
́-s\r\n\r\n<|fim_prefix|>a३éİ­🙂عfi­ſ'Mfi٣٤٥٦\n", "tokens": 39, "pieces": [",漢", "½", "m'T", "३", "👍🏽", "Dž", "\r\n", "
́", "-s", "\r\n\r\n", "<|", "fim", "_prefix", "|>", "a", "३", "é", "İ", "­🙂", "عfi", "­ſ'M", "fi", "٣٤٥", "٦", "\n"]} +{"text": "…#$%", "tokens": 4, "pieces": ["…", "#$%"]} +{"text": "<|fim_prefix|>", "tokens": 6, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "👍🏽fi \nes'T é0å😀🏽ع\t12345678İd́A­ḍ̇<|fim_prefix|>\rAß! ", "tokens": 43, "pieces": ["👍🏽", "fi", " \n", "es'T", " é", "0", "å", "😀🏽", "ع", "\t", "", "123", "456", "78", "İd́", "A", "­ḍ̇", "<|", "fim", "_prefix", "|>\r", "Aß", "!", " "]} +{"text": " ­<", "tokens": 3, "pieces": [" ", "­<"]} +{"text": "\"e🙂>ع-é12345678Džḍ̇.t👍🏽!३", "tokens": 21, "pieces": ["\"e", "🙂>", "ع", "-é", "123", "456", "78", "Džḍ̇", ".t", "👍🏽!", "३"]} +{"text": "३\r\nå!\nZ㋿ſ'Re,𐞁'll(\n'D!'re'ſ 字m", "tokens": 30, "pieces": ["३", "\r\n", "å", "!\n", "Z", "㋿ſ'Re", ",𐞁'll", "(\n", "'D", "!<", "META", "_START", ">'", "re'ſ", " ", " 字m"]} +{"text": "<|fim_prefix|>字𐞁'VEİs \nعⅣ'ReéfiⅣDž\u000bfi", "tokens": 32, "pieces": ["<|", "fim", "_prefix", "|>", "字", "𐞁'VE", "İs", " \n", "ع", "Ⅳ", "'Reéfi", "Ⅳ", "Dž", "\u000bfi"]} +{"text": "m-<\r\n\r\n'ſ\u000b\"", "tokens": 8, "pieces": ["m", "-<\r\n\r\n", "'ſ", "\u000b", "\""]} +{"text": "'ſ're㍿m'S'Reꟲ\r
🙂.12345678sfi,'<|fim_prefix|>9ꟲ\u000b漢\nt>字\r\nsſ'Red'TmZEOT\t'T", "tokens": 56, "pieces": ["'ſ're", "㍿m'S", "'Reꟲ", "\r", "
", "🙂.", "123", "456", "78", "sfi", ",'<|", "fim", "_prefix", "|>", "9", "ꟲ", "\u000b漢", "\n", "t", ">字", "\r\n", "sſ'Re", "d'T", "m", "ZEOT", "\t", "'T"]} +{"text": ", ", "tokens": 2, "pieces": [",", " "]} +{"text": "EOT🙂!EOT're,ß'Ms\r\n9\r\n\r\nZ#$%\"'sEOT'ſ \n'll😀🏽㍿'M- ½'ſ!a'D'll0\n", "tokens": 45, "pieces": ["EOT", "🙂!", "EOT're", ",ß'M", "s", "\r\n", "9", "\r\n\r\n", "Z", "#$%\"'", "s", "EOT'ſ", " \n", "'ll", "😀🏽㍿'", "M", "-", " ", " ", "½", "'ſ", "!a'D", "'ll", "0", "\n"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " \n Ⅳ0e 🙂", "tokens": 11, "pieces": [" \n", " ", "Ⅳ", "", "0", "e", " 🙂"]} +{"text": "d<|endoftext|>'sİ!'VEß,́Dž!tⅣ'M<ꟲ字 \n\u000b!!㍿٣٤٥٦ !!<\rDž'T''VE<|endoftext|>(漢㋿", "tokens": 60, "pieces": ["d", "<|", "endoftext", "|>'", "s", "İ", "!'", "VEß", ",́", "Dž", "!t", "Ⅳ", "'M", "<ꟲ字", " \n", "\u000b", "!!㍿", "٣٤٥", "٦", " ", "!!<\r", "Dž'T", "''", "VE", "<|", "endoftext", "|>(", "漢", "㋿"]} +{"text": "#$%\r\n​aé  !३'é\r­\rZ字!!é0٣٤٥٦'Re'M ", "tokens": 29, "pieces": ["#$%\r\n", "​aé", "", "  ", " !", "३", "'é", "\r", "­\r", "Z字", "!!", "é", "0٣٤", "٥٦", "'Re'M", " "]} +{"text": "e!EOTع­<|endoftext|>0t½½ع́'ſ", "tokens": 24, "pieces": ["e", "!EOTع", "­<|", "endoftext", "|>", "0", "t", "½½", "ع́'ſ"]} +{"text": "éA㍿ḍ̇e'VEDž\r 12345678½é're'Re0\t\r\n\r\ń½!!\r\n\r\n👍🏽'Dfi'S<|endoftext|>s\r\n-ꟲa<|fim_prefix|>'ſ 'res", "tokens": 64, "pieces": ["é", "A", "㍿ḍ̇e'VE", "Dž", "\r", " ", "123", "456", "78½", "é're", "'Re", "0", "\t\r\n\r\n", "́", "½", "!!\r\n\r\n", "👍🏽'", "Dfi'S", "<|", "endoftext", "|>", "s", "\r\n", "-ꟲa", "<|", "fim", "_prefix", "|>'", "ſ", "", " ", "'res"]} +{"text": "\"‍٣٤٥٦é0!!ꟲA", "tokens": 14, "pieces": ["\"‍", "٣٤٥", "٦", "é", "0", "!!", "ꟲ", "A"]} +{"text": "és", "tokens": 2, "pieces": ["és"]} +{"text": "Ⅳع­𐞁#$%\t😀🏽fi ꟲfi'ſ!!t'D'MZꟲ", "tokens": 41, "pieces": ["Ⅳ", "ع", "­<", "META", "_START", ">𐞁", "#$%", "\t", "😀🏽", "fi", " ꟲfi'ſ", "!!", "t", "'", "D'M", "Zꟲ"]} +{"text": "🙂DžA9fi'D𐞁EOT.0‍<漢,'M>३㍿‍漢'fi>'S", "tokens": 38, "pieces": ["🙂DžA", "9", "<", "META", "_START", ">fi'D", "𐞁", "EOT", ".", "0", "‍<", "漢", ",'", "M", ">", "३", "㍿‍", "漢", "'fi", ">'", "S"]} +{"text": "​ <|endoftext|>字(​ 'ſEOT'Mß‍㍿<|endoftext|>字ſ 'sꟲ12345678Ⅳ'Dꟲ३'reⅣs'll'll\n ㋿A a
㍿३ \n", "tokens": 66, "pieces": ["​", " ", " <|", "endoftext", "|>", "字", "(​", " '", "ſ", "EOT'M", "ß", "‍㍿<|", "endoftext", "|>", "字ſ", " ", "'sꟲ", "123", "456", "78Ⅳ", "'Dꟲ", "३", "'re", "Ⅳ", "s'll", "'ll", "\n", " ㋿", "A", " a", "
", "㍿", "३", " \n"]} +{"text": "𐞁's㍿dſ'ſ 👍🏽\t#$%Ⅳſ<
🙂\"m‍🙂ſ", "tokens": 30, "pieces": ["𐞁's", "㍿dſ'ſ", " ", "👍🏽", "\t", "#$%", "Ⅳ", "ſ", "<", "
", "🙂\"", "m", "‍🙂", "ſ"]} +{"text": "\r", "tokens": 1, "pieces": ["\r"]} +{"text": "m\"'T<  👍🏽'T ꟲ字㋿é! #$%\u000b!!٣٤٥٦漢'M", "tokens": 36, "pieces": ["m", "\"'", "T", "<<", "META", "_START", ">", " ", " ", "👍🏽'", "T", " ꟲ字", "㋿é", "!", " ", " #$%", "\u000b", "!!", "٣٤٥", "٦", "漢'M"]} +{"text": "<'så漢'M'ſ🙂٣٤٥٦\nfi😀🏽d #$%(", "tokens": 22, "pieces": ["<'", "så漢'M", "'ſ", "🙂", "٣٤٥", "٦", "\n", "fi", "😀🏽", "d", " ", " #$%("]} +{"text": "d'S
 ", "tokens": 4, "pieces": ["d'S", "
 "]} +{"text": " EOT́😀🏽­d\tA123456780.'VE😀🏽éåé9<|endoftext|>­ß'S'EOTDž'llßİ", "tokens": 44, "pieces": [" EOT́", "😀🏽­", "d", "\tA", "123", "456", "780", ".'", "VE", "😀🏽", "é", "åé", "9", "<|", "endoftext", "|>­", "ß'S", "'EOTDž'll", "ß", "İ"]} +{"text": "👍🏽İ<|fim_prefix|>", "tokens": 10, "pieces": ["👍🏽", "İ", "<|", "fim", "_prefix", "|>"]} +{"text": "<|endoftext|>\r\n-d\r\n\r\n<|endoftext|> \n'D>½…\t🙂", "tokens": 24, "pieces": ["<|", "endoftext", "|>\r\n", "-d", "\r\n\r\n", "<|", "endoftext", "|>", " \n", "'D", ">", "½", "…", "\t", "🙂"]} +{"text": "ع'reå", "tokens": 4, "pieces": ["ع're", "å"]} +{"text": "EOT'll're åt\r\nZ 9<|fim_prefix|>'M…>­㍿!!fiعEOT\u000b'Re", "tokens": 33, "pieces": ["EOT'll", "'re", " åt", "\r\n", "Z", " ", "9", "<|", "fim", "_prefix", "|>'", "M", "…", ">­㍿!!", "fiع", "EOT", "\u000b", "'Re"]} +{"text": "㋿ß漢…Z㍿\"ß<#$%
\r\n\tZEOTZ9́>AZ'M­'Re \r é字!!é", "tokens": 43, "pieces": ["㋿ß漢", "…Z", "㍿\"", "ß", "<#$%", "
\r\n", "\tZEOTZ", "9", "́", "><", "EOT", ">AZ'M", "­'", "Re", " \r", " é字", "!!", "é"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿'ſ'VE'VE'Re'Mḍ̇ ­​\u000b(ḍ̇​'Re㍿㋿ \u000ba'reé\r㍿\n,12345678 𐞁", "tokens": 59, "pieces": ["㍿'", "ſ'VE", "'VE'Re", "'", "Mḍ̇", " ", "­​", "\u000b", "(<", "EOT", ">ḍ̇", "​'", "Re", "㍿㋿", " ", "\u000ba're", "é", "\r", "㍿\n", ",", "123", "456", "78", " ", " 𐞁"]} +{"text": "​
字>㍿😀🏽\"'D'reéDž 're‍é'Re\"'T٣٤٥٦12345678٣٤٥٦Z㍿😀🏽\"'", "D're", "é", "Dž", " ", " '", "re", "‍é'Re", "\"'", "T", "٣٤٥", "٦12", "345", "678", "٣٤٥", "٦", "Z", "'ſ s漢́\t>'Re<­", "tokens": 20, "pieces": ["ß", "#$%🙂", "३", "a'VE", "\r\n", ">'", "ſ", " s漢́", "\t", ">'", "Re", "<­"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "é\rDž12345678 'll\tḍ̇'T\"!'llعs'll'll ㋿", "tokens": 25, "pieces": ["é", "\r", "Dž", "123", "456", "78", " ", "'ll", "\tḍ̇'T", "\"!'", "llعs'll", "'ll", " ", "㋿"]} +{"text": ">A#$%åEOT's字ß‍ \t9'ſ>'M 'S'Res<|endoftext|><|endoftext|>İ㍿é9Džé's漢'", "tokens": 53, "pieces": [">A", "#$%", "å", "EOT's", "字ß", "‍", " ", "\t", "9", "'ſ", ">'", "M", " ", " <", "META", "_START", ">'", "S'Re", "s", "<|", "endoftext", "|><|", "endoftext", "|>", "İ", "㍿é", "9", "Džé's", "漢", "'"]} +{"text": "Ⅳ!!0\u000bé\r\n\r\nA!!Džꟲ'9…9'D㍿\n0Z12345678'M#$%
\r\nfi'reß字", "tokens": 56, "pieces": ["Ⅳ", "!!", "0", "\u000bé", "\r\n\r\n", "A", "!!", "Džꟲ", "'", "9", "…", "9", "'D", "㍿\n", "", "0", "Z", "123", "456", "78", "'M", "#$%", "
\r\n", "fi're", "ß", "字"]} +{"text": "عſ३es9​㍿\r\n'D\r\n\r\n", "tokens": 21, "pieces": ["عſ", "३", "es", "9", "​㍿\r\n", "'D", "\r\n\r\n"]} +{"text": "éDž.­ …>…\r\n­", "tokens": 14, "pieces": ["é", "Dž", ".­", " ", "…", ">", "…\r\n", "­"]} +{"text": "​㋿'T'D㋿㍿>字 ㍿", "tokens": 19, "pieces": ["​㋿'", "T'D", "㋿㍿>", "字", " ", "㍿"]} +{"text": "\r!!'re's‍!m(字­
\téع́d😀🏽EOTaDž\"e'VE\rå0\nꟲA漢>\n9… 'Déعع", "tokens": 49, "pieces": ["\r", "!!'", "re's", "‍!", "m", "(字", "­", "
", "\téع́d", "😀🏽", "EOTa", "Dž", "\"e'VE", "\r", "å", "0", "\n", "ꟲA漢", ">\n", "9", "…", " ", "'Déعع"]} +{"text": "A\r​a\t🙂 'Dع​t𐞁'llfiİ", "tokens": 18, "pieces": ["A", "\r", "​a", "\t", "🙂", " ", "'Dع", "​t𐞁'll", "fi", "İ"]} +{"text": "EOT😀🏽<|endoftext|>‍\r\n\"Džꟲ'\rd.<9\r\r\n\r\n!\n…9 ٣٤٥٦", "tokens": 39, "pieces": ["EOT", "😀🏽<|", "endoftext", "|>‍\r\n", "\"Dž", "ꟲ", "'\r", "d", ".<", "9", "\r\r\n\r\n", "!\n", "…", "9", " ", "٣٤٥", "٦"]} +{"text": "'re 👍🏽字\r!!<|endoftext|>sⅣ'T( Z\r\n😀🏽٣٤٥٦0🙂👍🏽'M<|endoftext|>\r漢ḍ̇\r\n\r\n İ<9Ⅳ'Ds", "tokens": 58, "pieces": ["'re", " ", "👍🏽", "字", "\r", "!!<|", "endoftext", "|>", "s", "Ⅳ", "'T", "(", " Z", "\r\n", "😀🏽", "٣٤٥", "٦0", "🙂👍🏽'", "M", "<|", "endoftext", "|>\r", "漢ḍ̇", "\r\n\r\n", " İ", "<", "9Ⅳ", "'Ds"]} +{"text": "\u000b\tſ漢é", "tokens": 6, "pieces": ["\u000b", "\tſ漢é"]} +{"text": "Dž#$% t'Re‍ ꟲ", "tokens": 10, "pieces": ["Dž", "#$%", " t'Re", "‍", " ꟲ"]} +{"text": "Zꟲ½e#$%e", "tokens": 9, "pieces": ["Zꟲ", "½", "e", "#$%", "e"]} +{"text": " …‍​'s'ſ🙂'll <|fim_prefix|>ع\t!é٣٤٥٦ḍ̇e…漢Dž", "tokens": 58, "pieces": [" ", "…", "‍​'", "s'ſ", "🙂'", "ll", " <|", "fim", "_prefix", "|>", "ع", "", "\t", "!é", "٣٤٥", "٦", "ḍ̇e", "…漢", "Dž"]} +{"text": "'T<|fim_prefix|>12345678'\r\n
á İ\" İ 's", "tokens": 21, "pieces": ["'T", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'\r\n", "
á", " İ", "\"", " ", " İ", " ", "'s"]} +{"text": "'s\t9EOT'ſ<😀🏽​>0漢sſ ꟲ'VEǻ‍", "tokens": 26, "pieces": ["'s", "\t", "9", "EOT'ſ", "<😀🏽​>", "0", "漢sſ", " ꟲ'VE", "ǻ", "‍"]} +{"text": ">'M", "tokens": 5, "pieces": ["><", "META", "_START", ">'", "M"]} +{"text": ".!<|endoftext|> ३#$%", "tokens": 12, "pieces": [".!<|", "endoftext", "|>", " ", "३", "#$%"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'T​å३t𐞁å½­​0a", "tokens": 17, "pieces": ["'T", "​å", "३", "t𐞁å", "½", "­​", "0", "a"]} +{"text": "‍", "tokens": 1, "pieces": ["‍"]} +{"text": "\u000bétA\r\nå \n​0'T EOT9\r\n\r\n \nt\"'Sa", "tokens": 21, "pieces": ["\u000bét", "A", "\r\n", "å", " \n", "​", "0", "'T", " EOT", "9", "\r\n\r\n \n", "t", "\"'", "Sa"]} +{"text": "(-‍
'S­Z漢Z🙂12345678 'VEs'Re漢👍🏽​>  ​́s😀🏽Ⅳ'D
Z'Z", "tokens": 48, "pieces": ["(-‍", "
", "'", "S", "­<", "META", "_START", ">Z漢", "Z", "🙂<", "EOT", ">", "123", "456", "78", " ", "'VEs'Re", "漢", "👍🏽​>", " ", " ​́", "s", "😀🏽", "Ⅳ", "'D", "
Z", "'Z"]} +{"text": "…fi'll's #$%㋿std#$%.Ⅳ😀🏽👍🏽'T㍿​ß'M", "tokens": 39, "pieces": ["…fi'll", "'s", " ", "#$%㋿", "s", "td", "#$%.", "Ⅳ", "😀🏽👍🏽'", "T", "㍿​", "ß'M"]} +{"text": "㍿é0\r\n\r\n㋿d '…12345678漢㋿Z ſ 'Re ( 'reDž\n\r\n\r\n", "tokens": 41, "pieces": ["㍿", "é", "0", "\r\n\r\n", "㋿d", " ", "'", "…", "123", "456", "78", "漢", "㋿Z", " ſ", " ", " '", "Re", " ", " (", " ", " '", "re", "Dž", "\n\r\n\r\n"]} +{"text": "'llDž12345678aé<'T-'S ३eEOTſ\r'ſ\u000bꟲEOT‍ .DžⅣZ!!-'", "S", " ", "३", "e", "EOTſ", "\r", "'ſ", "\u000bꟲ", "EOT", "‍", " .", "Dž", "Ⅳ", "Z", "!!<", "EOTtع'D", " é", ",ǻ", "A", "३"]} +{"text": "#$%‍½İ'T㋿‍漢e ", "tokens": 13, "pieces": ["#$%‍", "½", "İ'T", "㋿‍", "漢e", " "]} +{"text": "'VE!Z३fi३9Dž 'T
'T>‍'Re'Re ,.㋿'s#$%'T\r\n㍿9½<|endoftext|>'lle \n漢 字​", "tokens": 48, "pieces": ["'VE", "!Z", "३", "fi", "३9", "Dž", " ", "'T", "
", "'T", ">‍'", "Re'Re", " ", ",.㋿'", "s", "#$%'", "T", "\r\n", "㍿", "9½", "<|", "endoftext", "|>'", "lle", " \n", "漢", " 字", "​"]} +{"text": "é👍🏽ꟲ𐞁 漢Z'ſſ'll!!<|fim_prefix|>\r\nſ­ \n­\r\n\r\n'll٣٤٥٦\tfi12345678ḍ̇é \n\"12345678İt'Sḍ̇🙂-", "tokens": 57, "pieces": ["é", "👍🏽", "ꟲ𐞁", " 漢", "Z'ſ", "ſ'll", "!!<|", "fim", "_prefix", "|>\r\n", "ſ", "­", " \n", "­\r\n\r\n", "'ll", "٣٤٥", "٦", "\tfi", "123", "456", "78", "ḍ̇é", " \n", "\"", "123", "456", "78", "İt'S", "ḍ̇", "🙂-"]} +{"text": "-㍿́#$%ꟲ-㍿\r\n\r\n\r\n'D 
‍ꟲ ꟲعé#$%
​🙂ßع👍🏽'ſ'D​<|endoftext|>!'ll\r\n\r\nDž½ßd", "tokens": 57, "pieces": ["-㍿́#$%", "ꟲ", "-㍿\r\n\r\n\r\n", "'D", " ", "
", "‍ꟲ", " ꟲعé", "#$%", "
", "​🙂", "ßع", "👍🏽'", "ſ'D", "​<|", "endoftext", "|>!'", "ll", "\r\n\r\n", "Dž", "½", "ßd"]} +{"text": "'sſ 's
…ع-٣٤٥٦>e! ٣٤٥٦
㍿<|endoftext|>
<fi😀🏽'VE>İé\n<ꟲ­ßſ🙂12345678'DZ ㋿ Ⅳ", "tokens": 58, "pieces": ["\n\n", ">-", "٣٤٥", "٦", ">e", "!", " ", "٣٤٥", "٦", "
", "㍿<|", "endoftext", "|>", "
", "<fi", "😀🏽'", "VE", ">İé", "\n", "<ꟲ", "­ßſ", "🙂", "123", "456", "78", "'DZ", " ", "㋿", " ", "Ⅳ"]} +{"text": "\td#$%#$%>\r\n,t\t0<|fim_prefix|>\nع३½\r\n", "tokens": 25, "pieces": ["\td", "#$%#$%><", "EOT", ">\r\n", ",t", "\t", "0", "<|", "fim", "_prefix", "|>\n", "ع", "", "३½", "\r\n"]} +{"text": "́-­'T'lld0.e'M,'Se\r\n\r\n\r\n!!㍿'reDž字", "tokens": 25, "pieces": ["́", "-­'", "T'll", "d", "0", ".e'M", ",'", "Se", "\r\n", "\r\n\r\n", "!!㍿'", "re", "Dž字"]} +{"text": ".'sEOT\r\n\r\nⅣå!s", "tokens": 11, "pieces": [".'", "s", "EOT", "\r\n\r\n", "Ⅳ", "å", "!s"]} +{"text": "३İ'll३漢", "tokens": 5, "pieces": ["३", "İ'll", "३", "漢"]} +{"text": "Z \n‍A'VE\r\nméå\t㋿\té​‍\"é'ſt㋿!", "tokens": 29, "pieces": ["Z", " \n", "‍A'VE", "\r\n", "méå", "\t", "㋿", "\té", "​‍\"", "é'ſ", "t", "㋿!"]} +{"text": "fi'ſ\r\n\r\n!!. \"ع", "tokens": 8, "pieces": ["fi'ſ", "\r\n\r\n", "!!.", " ", " \"", "ع"]} +{"text": "𐞁ꟲ're\n<\"İém­'re", "tokens": 18, "pieces": ["𐞁ꟲ're", "\n", "<\"", "İém", "­'", "re"]} +{"text": "\r\nå0​👍🏽A're'TEOT", "tokens": 13, "pieces": ["\r\n", "å", "0", "​👍🏽", "A're", "'TEOT"]} +{"text": "\n>'Sfí㋿ed
'Re<|fim_prefix|>aZ<|fim_prefix|>漢
", "tokens": 27, "pieces": ["\n", ">'", "Sfí", "㋿ed", "
", "'Re", "<|", "fim", "_prefix", "|>", "a", "Z", "<|", "fim", "_prefix", "|>", "漢", "
"]} +{"text": "123456789
ꟲ­<|fim_prefix|>㋿٣٤٥٦­'ſ'S!ß'Smḍ̇ḍ̇", "tokens": 35, "pieces": ["123", "456", "789", "
ꟲ", "­<|", "fim", "_prefix", "|>㋿", "٣٤٥", "٦", "­'", "ſ'S", "!ß'S", "mḍ̇ḍ̇"]} +{"text": "३", "tokens": 1, "pieces": ["३"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿​٣٤٥٦12345678'Dع🙂9ſ😀🏽ß㋿İ'Ret३Zع9İ'Reſ‍字‍!३​'s<|fim_prefix|><|fim_prefix|>𐞁12345678​\n'llm", "tokens": 66, "pieces": ["㍿​", "٣٤٥", "٦12", "345", "678", "'Dع", "🙂", "9", "ſ", "😀🏽", "ß", "㋿İ'Re", "t", "३", "Zع", "9", "İ'Re", "ſ", "‍字", "‍!", "३", "​'", "s", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", "𐞁", "123", "456", "78", "​\n", "'llm"]} +{"text": "字as'ś'sA漢12345678å", "tokens": 12, "pieces": ["字as's", "́'s", "A漢", "123", "456", "78", "å"]} +{"text": "<|endoftext|>s012345678<|endoftext|>0>\"\r\n\r\n㋿İ𐞁.漢,'Ta'Dd\n \n㋿'S½漢\r\n \n漢", "tokens": 47, "pieces": ["<|", "endoftext", "|>", "s", "012", "345", "678", "<|", "endoftext", "|>", "0", ">\"\r\n\r\n", "㋿İ𐞁", ".漢", ",'", "Ta'D", "d", "\n \n", "㋿'", "S", "½", "漢", "\r\n \n", "漢"]} +{"text": " ㋿३ßda'T!!", "tokens": 9, "pieces": [" ㋿", "३", "ßda'T", "!!"]} +{"text": "<Dž 𐞁㋿\u000bع‍", "tokens": 14, "pieces": ["<Dž", " 𐞁", "㋿", "\u000bع", "‍"]} +{"text": "🙂e٣٤٥٦'llZ<|fim_prefix|>'T", "tokens": 15, "pieces": ["🙂e", "٣٤٥", "٦", "'ll", "Z", "<|", "fim", "_prefix", "|>'", "T"]} +{"text": "m,'\"e字㋿字\u000b'll\u000b0å\r'M9ds𐞁A\r\n 're12345678ḍ̇éaEOT\r½dZ\r\n\r\n \n>e,9é", "tokens": 49, "pieces": ["m", ",'\"", "e字", "㋿字", "\u000b", "'ll", "\u000b", "0", "å", "\r", "'M", "9", "ds𐞁", "A", "\r\n", " ", " '", "re", "123", "456", "78", "ḍ̇éa", "EOT", "\r", "½", "d", "Z", "\r\n\r\n \n", ">e", ",", "9", "é"]} +{"text": "12345678'M😀🏽.٣٤٥٦'T'Re३e(<㍿åd('T'll\r\n\r\n'VE㋿🙂m​,ḍ̇ \n#$%ß \n‍<|endoftext|>ß'T½'
Dž#$%", "tokens": 65, "pieces": ["123", "456", "78", "'M", "😀🏽.", "٣٤٥", "٦", "'T'Re", "३", "e", "(<㍿", "åd", "('", "T", "'", "ll", "\r\n\r\n", "'VE", "㋿🙂", "m", "​,", "ḍ̇", " \n", "#$%", "ß", " \n", "‍<|", "endoftext", "|>", "ß'T", "½", "'", "
Dž", "#$%"]} +{"text": "a\r\n\r\nEOT \nḍ̇>12345678's<|fim_prefix|>'ſ.EOT12345678'Rema'D𐞁'T<|endoftext|>fi.", "tokens": 43, "pieces": ["a", "\r\n\r\n", "EOT", " \n", "ḍ̇", ">", "123", "456", "78", "'s", "<|", "fim", "_prefix", "|>'", "ſ", ".EOT", "123", "456", "78", "'Rema'D", "𐞁'T", "<|", "endoftext", "|>", "fi", "."]} +{"text": "!!!!fi漢İAe.ḍ̇t漢㍿​!!'reda'M.'ſ‍é> 'VÉ", "tokens": 38, "pieces": ["!!<", "EOT", ">!!", "fi漢", "İ", "Ae", ".ḍ̇t漢", "㍿​!!'", "reda'M", ".'", "ſ", "‍é", ">", " ", " '", "VÉ"]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "­>\"ådſ  ", "tokens": 7, "pieces": ["­>\"", "ådſ", "  "]} +{"text": "'D'D 'Re12345678'Re'T ZEOT ع㍿\t's㍿­ſ, ", "tokens": 26, "pieces": ["'D'D", " ", " '", "Re", "123", "456", "78", "'Re'T", " ZEOT", " ع", "㍿", "\t", "'s", "㍿­", "ſ", ",", " "]} +{"text": "'Re'S,-👍🏽 'T
é #$%", "tokens": 15, "pieces": ["'Re'S", ",-👍🏽", " ", " '", "T", "
é", " ", "#$%"]} +{"text": "😀🏽\u000b'llA\r\n\r\n\r\n<|endoftext|>(㍿.'s👍🏽½dDž\"", "tokens": 27, "pieces": ["😀🏽", "\u000b", "'ll", "A", "\r\n\r\n\r\n", "<|", "endoftext", "|>(㍿.'", "s", "👍🏽", "½", "d", "Dž", "\""]} +{"text": "9'Re0İ0'M\n<|endoftext|>'T😀🏽٣٤٥٦0", "tokens": 23, "pieces": ["9", "'Re", "0", "İ", "0", "'M", "\n", "<|", "endoftext", "|>'", "T", "😀🏽", "٣٤٥", "٦0"]} +{"text": "<|endoftext|>s㋿\r\n\r\n👍🏽éad.'sſ'S", "tokens": 28, "pieces": ["<|", "endoftext", "|><", "EOT", ">s", "㋿\r\n\r\n", "👍🏽", "éad", ".'", "sſ", "'", "S"]} +{"text": "a \u000b'ſꟲfi<'VE'Re\r\n", "tokens": 16, "pieces": ["a", " ", "\u000b", "'ſꟲfi", "<'", "VE'Re", "\r\n"]} +{"text": "\r\n\r\n­Dž's'M'S<́'VE'Re\t '३'🙂é👍🏽\u000b<|endoftext|>Ⅳ३‍㋿'re", "tokens": 39, "pieces": ["\r\n\r\n", "­Dž's", "'M'S", "<́'VE", "'Re", "\t", " ", "'", "३", "'🙂", "é", "👍🏽", "\u000b", "<|", "endoftext", "|>", "Ⅳ३", "‍㋿'", "re"]} +{"text": "é漢a t'Z\tع> #$%\tİ…'ReDž…٣٤٥٦'D!!é\nⅣmAt sZ😀🏽'Mfi'll", "tokens": 51, "pieces": ["é漢a", " ", " t", "'Z", "\tع", ">", " ", " #$%", "\tİ", "…", "'Re", "Dž", "…", "٣٤٥", "٦", "'D", "!!", "é", "\n", "Ⅳ", "m", "At", " s", "Z", "😀🏽'", "Mfi'll"]} +{"text": "٣٤٥٦ß'Re…!'re\tZmé字9fi!'TEOT -'ſ's(漢a", "tokens": 26, "pieces": ["٣٤٥", "٦", "ß'Re", "…", "!'", "re", "\tZmé字", "9", "fi", "!'", "TEOT", " -'", "ſ's", "(漢a"]} +{"text": "<|endoftext|>m½!!‍0 <|fim_prefix|>-\r12345678'T'Zs12345678. 'sZ‍<|endoftext|>d'VEd漢'S'VEⅣ
\u000bé漢'Re \n", "tokens": 63, "pieces": ["<|", "endoftext", "|>", "m", "½", "!!‍", "0", " ", "<|", "fim", "_prefix", "|>-\r", "", "123", "456", "78", "'T", "'Zs", "123", "456", "78", ".", " ", "'s", "Z", "‍<|", "endoftext", "|>", "d'VE", "d漢'S", "'VE", "Ⅳ", "
", "\u000bé漢'Re", " \n"]} +{"text": "ſ‍A(ſ३\t'D \tⅣ's\r\"EOT.٣٤٥٦ \nİ \"", "tokens": 24, "pieces": ["ſ", "‍A", "(ſ", "३", "\t", "'D", " ", "\t", "Ⅳ", "'s", "\r", "\"EOT", ".", "٣٤٥", "٦", " \n", "İ", " \""]} +{"text": " (!'Re'Re<|fim_prefix|>\rع\u000bDž're ß‍'T99d!!åⅣ,‍'D!!㍿
a'll\nŹ'D ", "tokens": 51, "pieces": [" ", "(!'", "Re", "'", "Re", "<|", "fim", "_prefix", "|>\r", "ع", "\u000bDž're", " ß", "‍'", "T", "99", "d", "!!", "å", "Ⅳ", ",‍'", "D", "!!㍿", "
a'll", "\n", "Ź'D", " "]} +{"text": "'D're㋿0!", "tokens": 7, "pieces": ["'D're", "㋿", "0", "!"]} +{"text": "Ⅳ'ſ‍<|fim_prefix|>9t'Re\rḍ̇🙂'D\r\n\r\nſ'Re字! ع'D\"… e'll\n!İ !<|fim_prefix|> tA\"fi", "tokens": 50, "pieces": ["Ⅳ", "'ſ", "‍<|", "fim", "_prefix", "|>", "9", "t'Re", "\r", "ḍ̇", "🙂'", "D", "\r\n\r\n", "ſ'Re", "字", "!", " ع'D", "\"", "…", " e'll", "\n", "!İ", " !<|", "fim", "_prefix", "|>", " t", "A", "\"fi"]} +{"text": "🙂\u000bİ12345678\r\n12345678", "tokens": 10, "pieces": ["🙂", "\u000bİ", "123", "456", "78", "\r\n", "123", "456", "78"]} +{"text": "åe\u000b0m <\t'T㍿…'VE𐞁e9<|fim_prefix|>'sſ!!\n\r\n㋿Ⅳ'D <0é", "tokens": 50, "pieces": ["åe", "\u000b", "0", "m", " <", "\t", "'T", "㍿", "…", "'VE𐞁e", "9", "<|", "fim", "_prefix", "|>'", "sſ", "!!<", "EOT", ">\n\r\n", "㋿", "Ⅳ", "'D", " ", "<", "0", "é"]} +{"text": "'ll'Ddm>㋿'İ're'S-'S>𐞁 …A", "tokens": 25, "pieces": ["'ll'D", "dm", ">㋿'", "İ're", "'S", "-'", "S", ">", "𐞁", " ", "…A"]} +{"text": "12345678'M\r,A 字'sm\r\n\r\n m>𐞁 \n عſe٣٤٥٦", "tokens": 31, "pieces": ["123", "456", "78", "'M", "\r", ",A", " ", " 字's", "m", "\r\n\r\n", " m", ">𐞁", " \n", " عſe", "٣٤٥", "٦"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿ 'DZ. \n!\r\n\r\n\n-👍🏽'ſ\"'D\"㍿ \n'D,'ſ", "tokens": 28, "pieces": ["㍿", " ", "'", "DZ", ".", " \n", "!\r\n\r\n\n", "-👍🏽'", "ſ", "\"'", "D", "\"㍿", " \n", "'D", ",'", "ſ"]} +{"text": "ZaعeéEOT🙂(!!'T٣٤٥٦ع 0éⅣ\n'ſ<|endoftext|>'re👍🏽\r\n\r\nå.< 字", "tokens": 50, "pieces": ["Zaعeé", "EOT", "🙂(!!<", "META", "_START", ">'", "T", "٣٤٥", "٦", "ع", " ", "0", "é", "Ⅳ", "\n", "'ſ", "<", "EOT", "><|", "endoftext", "|>'", "re", "👍🏽\r\n\r\n", "å", ".<", " ", " 字"]} +{"text": "İm12345678\r''12345678'T-t ß12345678
<|fim_prefix|>‍", "tokens": 28, "pieces": ["İm", "123", "456", "78", "\r", "''", "123", "456", "78", "'T", "-t", " ß", "123", "456", "78", "
", "<|", "fim", "_prefix", "|>‍"]} +{"text": "ḍ̇ea\" 字ع\r
👍🏽
👍🏽😀🏽åḍ̇३A\u000b<|endoftext|>'M­'TZ'ſ👍🏽(‍㋿३0", "tokens": 51, "pieces": ["ḍ̇ea", "\"", " 字ع", "\r", "
", "👍🏽", "
", "👍🏽😀🏽", "åḍ̇", "३", "A", "\u000b", "<|", "endoftext", "|>'", "M", "­'", "TZ'ſ", "👍🏽(‍㋿", "३0"]} +{"text": "'D\n½½㍿\r\nꟲ́12345678 a\t#$%٣٤٥٦\n!d0'ſ12345678٣٤٥٦㍿ßé😀🏽.'", "tokens": 48, "pieces": ["'D", "\n", "½½", "㍿<", "META", "_START", ">\r\n", "ꟲ́", "123", "456", "78", " a", "\t", "#$%", "٣٤٥", "٦", "\n", "!d", "0", "'ſ", "123", "456", "78٣", "٤٥٦", "㍿ßé", "😀🏽.'"]} +{"text": "Dž>Aa㍿́<(ḍ̇İ#$%\r' \n e!!!!9<'T", "tokens": 28, "pieces": ["Dž", ">Aa", "㍿́", "<(", "ḍ̇", "İ", "#$%\r", "'<", "META", "_START", ">", " \n", " e", "!!!!", "9", "<'", "T"]} +{"text": "#$%!字\"👍🏽㋿a'Re\t​A​İ'M\rſ'T\u000bİ字<'D!!
", "tokens": 28, "pieces": ["#$%!", "字", "\"👍🏽㋿", "a'Re", "\t", "​A", "​İ'M", "\r", "ſ'T", "\u000bİ字", "<'", "D", "!!", "
"]} +{"text": "ꟲ're 😀🏽\"'", "tokens": 9, "pieces": ["ꟲ're", " ", " 😀🏽\"'"]} +{"text": " …'s \nßEOTé", "tokens": 10, "pieces": [" ", "…", "'s", " \n", "ß", "EOTé"]} +{"text": "́(", "tokens": 2, "pieces": ["́", "("]} +{"text": "\r\n\r\n
å<|fim_prefix|>EOT'Ss \r\nt…㋿'ll!!'VE're 0Zt👍🏽ع'D", "tokens": 37, "pieces": ["\r\n\r\n", "
å", "<|", "fim", "_prefix", "|>", "EOT'S", "s", " \r\n", "t", "…", "㋿'", "ll", "!!'", "VE're", " ", "0", "Zt", "👍🏽", "ع'D"]} +{"text": "'Dß字9\r\n\r\n३…d'VEfi👍🏽Z…ſ(\tDže३", "tokens": 28, "pieces": ["'Dß字", "9", "\r\n\r\n", "३", "…d'VE", "fi", "👍🏽", "Z", "…ſ", "(", "\tDže", "", "३"]} +{"text": "\"ḍ̇ع\"!,!! ''T 12345678漢́\"漢 <|endoftext|>🙂'ſ's㋿👍🏽>‍", "tokens": 53, "pieces": ["\"ḍ̇", "ع", "\"!,!!", " ", "''", "T", " ", "123", "456", "78", "漢́", "\"漢", " <|", "endoftext", "|>🙂'", "ſ's", "㋿👍🏽>‍"]} +{"text": "'\r\nع'Dé'<|endoftext|>'Re12345678\r\n\r\nfi#$%t\r\n\r\n字EOT<<|fim_prefix|> \n-'VE'Re​ \n", "tokens": 39, "pieces": ["'\r\n", "ع'D", "é", "'<|", "endoftext", "|>'", "Re", "123", "456", "78", "\r\n\r\n", "fi", "#$%", "t", "\r\n\r\n", "字", "EOT", "<<|", "fim", "_prefix", "|>", " \n", "-'", "VE'Re", "​", " \n", ""]} +{"text": "'ReEOT𐞁ḍ̇<|fim_prefix|>e½İ\u000b½å", "tokens": 23, "pieces": ["'Re", "EOT𐞁ḍ̇", "<|", "fim", "_prefix", "|>", "e", "½", "İ", "\u000b", "½", "å"]} +{"text": "ſe​", "tokens": 3, "pieces": ["ſe", "​"]} +{"text": "'Re😀🏽'\r😀🏽12345678\tß\r\n\r\n <|fim_prefix|>́ſ<|endoftext|>#$%('s\r\n\r\n,!!ꟲ𐞁👍🏽­́ß ", "tokens": 52, "pieces": ["'Re", "😀🏽'\r", "😀🏽", "123", "456", "78", "\tß", "\r\n\r\n", " ", "<|", "fim", "_prefix", "|>́", "ſ", "<|", "endoftext", "|>#$%('", "s", "\r\n\r\n", ",!!", "ꟲ𐞁", "👍🏽­́", "ß", " "]} +{"text": "'D👍🏽́'s'", "s", "#$%m", "tokens": 49, "pieces": ["\u000b", "#$%<", "d字́'ſ", "ß", "m"]} +{"text": "å 's \"<|fim_prefix|>12345678'll. \n - \u000bé½ß'Dſ\"!!ꟲm😀🏽'Rea12345678Ⅳ", "tokens": 43, "pieces": ["å", " ", " '", "s", " ", "\"<|", "fim", "_prefix", "|>", "123", "456", "78", "'ll", ".", " \n", " -", " ", "\u000bé", "½", "ß'D", "ſ", "\"!!", "ꟲm", "😀🏽'", "Rea", "123", "456", "78Ⅳ"]} +{"text": "'ḍ̇ ३'ReA३\r\n
s( .​9
-\n½…½­😀🏽'D(fi", "tokens": 32, "pieces": ["'ḍ̇", " ", " ", "३", "'Re", "A", "३", "\r\n", "
s", "(", " ", " .​", "9", "
", "-\n", "½", "…", "½", "­😀🏽'", "D", "(fi"]} +{"text": "'Mḍ̇s(sfiZ٣٤٥٦d'D‍漢\r\n\r\n!12345678३EOT'S0́㋿漢 ½", "tokens": 33, "pieces": ["'Mḍ̇s", "(sfi", "Z", "٣٤٥", "٦", "d'D", "‍漢", "\r\n\r\n", "!", "123", "456", "78३", "EOT'S", "0", "́", "㋿漢", " ", "½"]} +{"text": "漢! \r\ń😀🏽'😀🏽 'SZ\"t,ع'ſ𐞁<|endoftext|>Z#$%\"", "tokens": 36, "pieces": ["漢", "!", " \r\n", "́", "😀🏽'😀🏽", " <", "META", "_START", ">'", "SZ", "\"t", ",ع'ſ", "𐞁", "<|", "endoftext", "|>", "Z", "#$%\""]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ſ
s👍🏽㍿­å<|fim_prefix|>ßa", "tokens": 23, "pieces": ["ſ", "
s", "👍🏽㍿­", "å", "<|", "fim", "_prefix", "|>", "ß", "a"]} +{"text": "Ⅳſİ'S'll'D'llé-ß'S ​́­m", "tokens": 17, "pieces": ["Ⅳ", "ſ", "İ'S", "'ll'D", "'llé", "-ß'S", " ​́­", "m"]} +{"text": "0EOT-३s𐞁 \nḍ̇'M\r\n㋿fi!𐞁a'D''se<|fim_prefix|>\"İ<|fim_prefix|>A­ḍ̇", "tokens": 47, "pieces": ["0", "EOT", "-", "३", "s𐞁", " \n", "ḍ̇'M", "\r\n", "㋿fi", "!𐞁a'D", "''", "se", "<|", "fim", "_prefix", "|>\"", "İ", "<|", "fim", "_prefix", "|>", "A", "­ḍ̇"]} +{"text": "­\neḍ̇fi\"s.'S漢  'VE😀🏽é\r\n\r\n!!s Z-ع(mé\r\ń's. t३!", "tokens": 35, "pieces": ["­\n", "eḍ̇fi", "\"s", ".'", "S漢", " ", " ", "'VE", "😀🏽", "é", "\r\n\r\n", "!!", "s", " ", " Z", "-ع", "(mé", "\r\n", "́'s", ".", " t", "३", "!"]} +{"text": "漢#$%9fiⅣ½㍿s😀🏽fi\r", "tokens": 17, "pieces": ["漢", "#$%", "9", "fi", "Ⅳ½", "㍿s", "😀🏽", "fi", "\r"]} +{"text": "<|endoftext|>\r\n\r\nꟲ\u000b'VE\n…Ⅳ𐞁 \n👍🏽9'll12345678​", "tokens": 38, "pieces": ["<|", "endoftext", "|>\r\n\r\n", "ꟲ", "\u000b", "'VE", "\n", "", "…", "Ⅳ", "𐞁", " \n", "👍🏽", "9", "'ll", "123", "456", "78", "​"]} +{"text": "'sDž'Re", "tokens": 7, "pieces": ["'s", "Dž", "'", "Re"]} +{"text": "🙂 ­'M d'ſ 😀🏽㍿eå<0'll\"'Dm", "tokens": 24, "pieces": ["🙂", " ", " ­'", "M", " d'ſ", " 😀🏽㍿", "e", "å", "<", "0", "'ll", "\"'", "Dm"]} +{"text": "\n…'Re'll👍🏽maé٣٤٥٦åİ'VE㋿🙂٣٤٥٦<|endoftext|>ée'Re\t\r\n!'Mfi
字🙂", "tokens": 57, "pieces": ["é'll", "'lle", "İ", " <|", "fim", "_prefix", "|>", "…", "'Re'll", "👍🏽", "maé", "٣٤٥", "٦", "å", "İ'VE", "㋿🙂", "٣٤٥", "٦", "<|", "endoftext", "|>", "ée'Re", "\t\r\n", "!'", "Mfi", "
字", "🙂"]} +{"text": "Džå0‍.'M\u000b'Se‍漢!!'T\u000bⅣ́", "tokens": 24, "pieces": ["Džå", "0", "‍.'", "M", "\u000b", "'Se", "‍漢", "!!'", "T", "\u000b", "Ⅳ", "́"]} +{"text": "Z\"", "tokens": 2, "pieces": ["Z", "\""]} +{"text": "aſ'ſ(e>'T Z12345678
<'Reé👍🏽s!!٣٤٥٦'D\r\ns👍🏽dDž\"ß'ع\rꟲa", "tokens": 49, "pieces": ["aſ'ſ", "(e", ">'", "T", " Z", "123", "456", "78", "
", "<'", "Reé", "👍🏽", "s", "!!", "٣٤٥", "٦", "'D", "\r\n", "s", "👍🏽", "d", "Dž", "\"ß", "'ع", "\r", "ꟲa"]} +{"text": ".t(ta'M‍😀🏽'ré🙂'Sé12345678 é", "tokens": 24, "pieces": [".t", "(ta", "'", "M", "‍😀🏽'", "ré", "🙂'", "Sé", "123", "456", "78", " é"]} +{"text": "'M å​'ſ漢é!", "tokens": 10, "pieces": ["'M", " å", "​'", "ſ漢é", "!"]} +{"text": "e…d\r\n\r\n½ 'S'M‍\"\tİe- \n!!'re३(\n字 ​fi 's ,\n9ꟲ­", "tokens": 36, "pieces": ["e", "…d", "\r\n\r\n", "½", " ", "'S'M", "‍\"", "\tİe", "-", " \n", "!!'", "re", "३", "(<", "META", "_START", ">\n", "字", " ", "​fi", " ", "'s", " ,\n", "9", "ꟲ", "­"]} +{"text": "#$%\r \n½ a'S­'D", "tokens": 11, "pieces": ["#$%\r", " \n", "½", " a'S", "­'", "D"]} +{"text": "🙂aß㍿<|fim_prefix|>'VE 𐞁#$%ſ'll#$%\r\nA<|endoftext|>", "tokens": 33, "pieces": ["🙂aß", "㍿<|", "fim", "_prefix", "|>'", "VE", " ", " 𐞁", "#$%", "ſ'll", "#$%\r\n", "A", "<|", "endoftext", "|>"]} +{"text": "𐞁 å!ae(\r\ns\t字'M'll'TA😀🏽٣٤٥٦ع\t́😀🏽éZİ\u000b", "tokens": 34, "pieces": ["𐞁", " å", "!ae", "(\r\n", "s", "\t字'M", "'ll'T", "A", "😀🏽", "٣٤٥", "٦", "ع", "\t́", "😀🏽", "é", "Zİ", "\u000b"]} +{"text": "0m٣٤٥٦Ⅳ'T", "tokens": 9, "pieces": ["0", "m", "٣٤٥", "٦Ⅳ", "'T"]} +{"text": "å'M", "tokens": 3, "pieces": ["å'M"]} +{"text": "å'Ret'Retꟲ𐞁
m🙂🙂\n𐞁e", "tokens": 23, "pieces": ["å'Re", "t'Re", "tꟲ𐞁", "
m", "🙂🙂\n", "𐞁e"]} +{"text": "'re're>åſ㍿​\u000bé'M0'M\r", "tokens": 16, "pieces": ["'re're", ">åſ", "㍿​", "\u000bé'M", "0", "'M", "\r"]} +{"text": "e \u000b'S.a", "tokens": 5, "pieces": ["e", " ", "\u000b", "'S", ".a"]} +{"text": " <  \n漢d're​eİe\"\t㋿
ع́\t(", "tokens": 21, "pieces": [" ", "<", "  \n", "漢d're", "​e", "İe", "\"", "\t", "㋿", "
ع́", "\t", "("]} +{"text": "!!'s'S'Tt", "tokens": 9, "pieces": ["!!'", "s'S", "'Tt"]} +{"text": ",'ſḍ̇<|fim_prefix|>d é字'ſ㋿㋿e", "tokens": 32, "pieces": [",'", "ſḍ̇", "<|", "fim", "_prefix", "|>", "d", " ", " é", "字'ſ", "㋿<", "META", "_START", ">㋿", "e"]} +{"text": "Dž 'D𐞁
👍🏽#$%𐞁!٣٤٥٦Z!!t", "tokens": 30, "pieces": ["Dž", " '", "D𐞁", "
", "👍🏽#$%", "𐞁", "!<", "EOT", ">", "٣٤٥", "٦", "Z", "!!", "t"]} +{"text": "\t,Ⅳ", "tokens": 4, "pieces": ["\t", ",", "Ⅳ"]} +{"text": " ​'re \nعt𐞁'Dta \nåé'D漢👍🏽så㍿́DžDžt Z'S< ", "tokens": 42, "pieces": [" ", " ​'", "re", " \n", "عt𐞁'D", "ta", " \n", "åé'D", "漢", "👍🏽", "så", "㍿́DžDžt", " ", " Z'S", "<", " "]} +{"text": "<|fim_prefix|>…'s.𐞁عfi字 >\r\n're", "tokens": 23, "pieces": ["<|", "fim", "_prefix", "|>", "…", "'s", ".𐞁عfi字", "", " ", " >\r\n", "'re"]} +{"text": "'VE9👍🏽", "tokens": 6, "pieces": ["'VE", "9", "👍🏽"]} +{"text": "Ⅳfi(\"'D\r\nİ>A漢😀🏽'S .mEOTZ
……!!9字ḍ̇𐞁<ꟲ\r\n<", "tokens": 48, "pieces": ["Ⅳ", "fi", "(\"'", "D", "\r\n", "İ", ">A漢", "😀🏽'", "S", " ", ".m", "EOTZ", "
", "", "…", "…", "!!", "9", "字ḍ̇𐞁", "<ꟲ", "\r\n", "<"]} +{"text": "'M#$%٣٤٥٦(#$%.m㍿㋿9#$%'T㋿\r\n\r\n!!Aḍ̇", "tokens": 30, "pieces": ["'M", "#$%", "٣٤٥", "٦", "(#$%.", "m", "㍿㋿", "9", "#$%'", "T", "㋿\r\n\r\n", "!!", "Aḍ̇"]} +{"text": "'s<|fim_prefix|>'ReZ😀🏽㋿漢s𐞁Ⅳḍ̇t İ ́e👍🏽…ḍ̇fi \n 漢ß'VE", "tokens": 47, "pieces": ["'s", "<|", "fim", "_prefix", "|>'", "Re", "Z", "😀🏽㋿", "漢s𐞁", "Ⅳ", "ḍ̇t", " ", " İ", " ́e", "👍🏽", "…ḍ̇fi", " \n", " 漢ß'VE"]} +{"text": "'re'Dİ字9​.🙂​㋿\t-
İé12345678‍
<|endoftext|>㍿12345678👍🏽\r㍿Zfi'VE> é", "tokens": 52, "pieces": ["'re'D", "İ字", "9", "​.🙂​㋿", "\t", "-", "
İé", "123", "456", "78", "‍", "
", "<|", "endoftext", "|>㍿", "123", "456", "78", "👍🏽\r", "㍿Zfi'VE", ">", " é", ""]} +{"text": "0 \n­  …'re漢­'T\r٣٤٥٦9s9d. 're\n'sⅣ\r\n'T<|fim_prefix|>", "tokens": 41, "pieces": ["0", " \n", "­", "  ", "…", "'re漢", "­'", "T", "\r", "٣٤٥", "٦9", "s", "", "9", "d", ".", " '", "re", "\n", "'s", "Ⅳ", "\r\n", "'T", "<|", "fim", "_prefix", "|>"]} +{"text": "🙂㋿'re字'll,s\r½ ḍ̇𐞁漢'D<|fim_prefix|>٣٤٥٦٣٤٥٦'re'Re,👍🏽!!­\r9", "tokens": 48, "pieces": ["🙂㋿'", "re字'll", ",<", "EOT", ">s", "\r", "½", " ḍ̇𐞁漢'D", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦٣٤", "٥٦", "'re'Re", ",👍🏽!!­\r", "9"]} +{"text": "'VE", "tokens": 2, "pieces": ["'VE"]} +{"text": "
ß 's'Re#$% \r\n㍿́m 'M12345678!!
🙂<|fim_prefix|>ß>é'M'D!!DžݽEOT", "tokens": 30, "pieces": ["!ß", "🙂", "٣٤٥", "٦", "!", "
", "'T'M", "ß", ">é'M", "'D", "!!", "Džİ", "", "½", "EOT"]} +{"text": "mZas !Ⅳ>
'S㍿0\r­ꟲ,#$%漢'T'll
-½'ſ'D!!Zté\"漢9字字's
½", "tokens": 42, "pieces": ["m", "Zas", " ", "!", "Ⅳ", ">", "
", "'S", "㍿", "0", "\r", "­ꟲ", ",#$%", "漢'T", "'ll", "
", "-", "½", "'ſ'D", "!!", "Zté", "\"漢", "9", "字字's", "
", "½"]} +{"text": "<'Re­ 'T㋿'VE0 \nEOT\n mééİ.\r\n\r\né<|endoftext|>å<|endoftext|>🙂!12345678\"İ'S-Dž", "tokens": 49, "pieces": ["<'", "Re", "­", " ", " '", "T", "㋿'", "VE", "0", " \n", "EOT", "\n", " méé", "İ", ".\r\n\r\n", "é", "<|", "endoftext", "|>", "å", "<|", "endoftext", "|>🙂!", "123", "456", "78", "\"İ'S", "-Dž"]} +{"text": "­'Tå'D're\r\n\r\n-é,12345678.<,٣٤٥٦ſ٣٤٥٦\t́\"'VE½m\u000bꟲ", "tokens": 40, "pieces": ["­'", "T", "å'D", "'re", "\r\n\r\n", "-é", ",", "123", "456", "78", ".<,", "٣٤٥", "٦", "ſ", "", "٣٤٥", "٦", "\t́", "\"'", "VE", "½", "m", "\u000bꟲ"]} +{"text": "'M🙂12345678e\r\n9,12345678İ'reعA", "tokens": 16, "pieces": ["'M", "🙂", "123", "456", "78", "e", "\r\n", "9", ",", "123", "456", "78", "İ're", "ع", "A"]} +{"text": " \"-m'३㍿字🙂 \tꟲ‍ḍ̇éḍ̇#$%­å,'s", "tokens": 37, "pieces": [" ", " \"-", "m", "'<", "EOT", ">", "३", "㍿字", "🙂", " ", "\tꟲ", "‍", "ḍ̇éḍ̇", "#$%­", "å", ",'", "s"]} +{"text": "㍿\rEOTḍ̇'S ßİ​d,½٣٤٥٦Ⅳ' ٣٤٥٦漢'Dé 'S𐞁ḍ̇mfi\u000b\rm'Re ½<|fim_prefix|><|endoftext|>> \n'", "tokens": 71, "pieces": ["㍿\r", "EOTḍ̇'S", " ß", "İ", "​d", ",<", "EOT", ">", "½٣٤", "٥٦Ⅳ", "'", " ", " ", "٣٤٥", "٦", "漢'D", "é", " ", "'S𐞁ḍ̇mfi", "\u000b\r", "m'Re", " ", " ", "½", "<|", "fim", "_prefix", "|><|", "endoftext", "|>>", " \n", "'"]} +{"text": "\r\n\r\né< (ḍ̇ ", "tokens": 10, "pieces": ["\r\n\r\n", "é", "<", " ", "(ḍ̇", " "]} +{"text": "t,!㍿😀🏽éé", "tokens": 20, "pieces": ["t", "Ⅳ", "İ𐞁'T", "A", ".🙂'", "ll", "Ⅳ", "́", ">éé"]} +{"text": "é́٣٤٥٦Džmſ\r𐞁<|fim_prefix|>­s\r\n9ß \n å'Re…", "tokens": 32, "pieces": ["é́", "٣٤٥", "٦", "Džmſ", "\r", "𐞁", "<|", "fim", "_prefix", "|>­", "s", "\r\n", "9", "ß", " \n", " å'Re", "…"]} +{"text": "!!­d(ſe<|endoftext|>Ⅳſ'D\r\n\r\n#$%dd漢㋿(ع(漢EOT.'s½字", "tokens": 35, "pieces": ["!!­", "d", "(ſe", "<|", "endoftext", "|>", "Ⅳ", "ſ'D", "\r\n\r\n", "#$%", "dd漢", "㋿(", "ع", "(漢", "EOT", ".'", "s", "½", "字"]} +{"text": "ꟲ'Re fi漢!!å>'D", "tokens": 16, "pieces": ["ꟲ'Re", " ", " fi漢", "!!", "å", ">'", "D"]} +{"text": "12345678­ꟲ!! 漢́'reéZ's", "tokens": 20, "pieces": ["123", "456", "78", "­ꟲ", "!!", " 漢́'re", "é", "Z's", ""]} +{"text": "́éé.㋿İé३​><…漢­🙂'VE\t \n\n!Ⅳ", "tokens": 27, "pieces": ["́éé", ".㋿", "İé", "३", "​><", "…漢", "­🙂'", "VE", "\t \n\n", "!", "Ⅳ"]} +{"text": "eⅣ😀🏽'D're३'s's9-0!", "tokens": 16, "pieces": ["e", "Ⅳ", "😀🏽'", "D're", "३", "'s's", "9", "-", "0", "!"]} +{"text": "0 \n(½å<|fim_prefix|> m عd'llé!Z字EOTå😀🏽㍿é.İ A \nfi
.'ßfi​Z'VEfi \n", "tokens": 50, "pieces": ["0", " \n", "(", "½", "å", "<|", "fim", "_prefix", "|>", " m", " عd'll", "é", "!Z字EOTå", "😀🏽㍿", "é", ".İ", " A", " \n", "fi", "
", ".'", "ßfi", "​Z'VE", "fi", " \n"]} +{"text": "<|endoftext|> \n…Aİ‍ ,", "tokens": 15, "pieces": ["<|", "endoftext", "|>", " \n", "…Aİ", "‍", " ", " ,"]} +{"text": "a'S!", "tokens": 3, "pieces": ["a'S", "!"]} +{"text": "👍🏽\r\n\r\n<|fim_prefix|>字'M.ḍ̇ꟲ", "tokens": 19, "pieces": ["👍🏽\r\n\r\n", "<|", "fim", "_prefix", "|>", "字'M", ".ḍ̇ꟲ"]} +{"text": " ½字\nm\r\n\r\n'T<>>mſaå㋿ 'T!𐞁ع<|endoftext|>\r\n\r\na're,123456789'S \n-", "tokens": 45, "pieces": ["", " ", "½", "字", "\n", "m", "\r\n\r\n", "'T", "<>>", "mſaå", "㋿", " ", "'T", "!𐞁ع", "<|", "endoftext", "|>\r\n\r\n", "a're", ",", "123", "456", "789", "'S", " \n", "-"]} +{"text": "'Re<|endoftext|>Dž\n٣٤٥٦ḍ̇ta!fi…a٣٤٥٦ -Ⅳ😀🏽é\n#$%漢👍🏽   😀🏽'T'Mſ'ſé", "tokens": 60, "pieces": ["'Re", "<|", "endoftext", "|>", "Dž", "\n", "٣٤٥", "٦", "ḍ̇ta", "!fi", "…a", "٣٤٥", "٦", " ", "-", "Ⅳ", "😀🏽", "é", "\n", "#$%", "漢", "👍🏽", "  ", " ", "😀🏽'", "T", "'", "Mſ'ſ", "é"]} +{"text": "́㍿'Re9d#$%#$%ꟲAꟲ<|fim_prefix|>\r\n\n<|endoftext|>EOTZ㍿m", "tokens": 40, "pieces": ["́", "㍿'", "Re", "9", "d", "#$%#$%", "ꟲAꟲ", "<|", "fim", "_prefix", "|>\r\n\n", "<|", "endoftext", "|>", "EOTZ", "㍿m"]} +{"text": "'Tt㍿😀🏽efi'T12345678sİ 'S​(​'S'T", "tokens": 24, "pieces": ["'Tt", "㍿😀🏽", "efi'T", "123", "456", "78", "s", "İ", " '", "S", "​(​'", "S'T"]} +{"text": "9'<ⅣⅣ 0'll d'ſDž", "tokens": 19, "pieces": ["9", "'<", "ⅣⅣ", " ", " ", "0", "'", "ll", " d'ſ", "Dž"]} +{"text": "\r\n.d<|fim_prefix|>字­,ꟲꟲ!EOT's", "tokens": 21, "pieces": ["\r\n", ".d", "<|", "fim", "_prefix", "|>", "字", "­,", "ꟲꟲ", "!EOT's"]} +{"text": "ſ.'ſ<|endoftext|>👍🏽EOT👍🏽A Ⅳ👍🏽9'ſ å…\"m'M>\u000b.<'VEm", "tokens": 47, "pieces": ["ſ", ".'", "ſ", "<|", "endoftext", "|>👍🏽", "EOT", "👍🏽", "A", " ", "Ⅳ", "👍🏽", "9", "'ſ", " ", "<", "EOT", ">å", "…", "\"m'M", ">", "\u000b", ".<'", "VEm"]} +{"text": " 'll12345678'D\u000bſ\t'Re𐞁…㍿e'ſ­EOT𐞁.A'ReEOT\n", "tokens": 40, "pieces": [" ", " '", "ll", "123", "456", "78", "'", "D", "\u000bſ", "\t", "'Re𐞁", "…", "㍿e'ſ", "­EOT𐞁", ".A'Re", "EOT", "\n"]} +{"text": "s३漢㍿\r\n\r\nꟲع
DžEOT\"ſ३ḍ̇mDž", "tokens": 28, "pieces": ["s", "३", "漢", "㍿\r\n\r\n", "ꟲع", "
DžEOT", "\"ſ", "३", "ḍ̇m", "Dž"]} +{"text": " ſ…'MåDžḍ̇\t'Re,<|endoftext|>ⅣA字 \n İfi.\tétſsꟲ<|endoftext|>㍿>ꟲ", "tokens": 53, "pieces": [" ", " ſ", "…", "'Må", "Džḍ̇", "\t", "'Re", ",<|", "endoftext", "|>", "Ⅳ", "A字", " \n", " ", " İfi", ".", "\tétſsꟲ", "<|", "endoftext", "|>㍿>", "ꟲ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "<|endoftext|> ", "tokens": 8, "pieces": ["<|", "endoftext", "|>", " "]} +{"text": "12345678're'½'ſ😀🏽漢‍(ſ m𐞁漢->Ⅳ😀🏽'ſ(字\u000bm9Džعḍ̇漢ḍ̇ \n½ß ", "tokens": 48, "pieces": ["123", "456", "78", "'re", "'", "½", "'ſ", "😀🏽", "漢", "‍(", "ſ", " m𐞁漢", "->", "Ⅳ", "😀🏽'", "ſ", "(字", "\u000bm", "9", "Džعḍ̇漢ḍ̇", " \n", "½", "ß", " "]} +{"text": "'Re -e𐞁EOT9'VE<|endoftext|>\r'D㋿e!\r\n😀🏽‍́ḍ̇…ع字é<|fim_prefix|>'ſ", "tokens": 54, "pieces": ["'Re", " ", "-e", "<", "META", "_START", ">𐞁", "EOT", "9", "'VE", "<|", "endoftext", "|>\r", "'D", "㋿e", "!\r\n", "😀🏽‍́", "ḍ̇", "…ع字é", "<|", "fim", "_prefix", "|>'", "ſ"]} +{"text": "<|endoftext|>>a#$%dZEOT m\naع", "tokens": 17, "pieces": ["<|", "endoftext", "|>>", "a", "#$%", "d", "ZEOT", " m", "\n", "aع"]} +{"text": "🙂‍DžDž㍿字\u000b's字'é're", "tokens": 16, "pieces": ["🙂‍", "DžDž", "㍿字", "\u000b", "'s字", "'é're"]} +{"text": "ع,<|fim_prefix|>'e \n>ḍ̇\u000bd'D12345678're a#$%…Dž𐞁\nå㍿\u000bⅣ!!\"½\n\"​#$%​Ⅳ½́́t", "tokens": 56, "pieces": ["ع", ",<|", "fim", "_prefix", "|>'", "e", " \n", ">ḍ̇", "\u000bd'D", "123", "456", "78", "'re", " ", " a", "#$%", "…Dž𐞁", "\n", "å", "㍿", "\u000b", "Ⅳ", "!!\"", "½", "\n", "\"​#$%​", "Ⅳ½", "́́t"]} +{"text": "-.'Sfi🙂9İ­#$%Ⅳ9\"m\r'll३s\r\n\r\nfiZe३'ll", "tokens": 32, "pieces": ["-<", "META", "_START", ">.'", "Sfi", "🙂", "9", "İ", "­#$%", "Ⅳ9", "\"m", "\r", "'", "ll", "३", "s", "\r\n\r\n", "fi", "Ze", "३", "'ll"]} +{"text": "​\rA(٣٤٥٦d👍🏽<ع'DعAé'Re\r\n\r\n \t'ſt#$%Z\"Dž‍.'Sa\r 𐞁 \n\r\n", "tokens": 43, "pieces": ["​\r", "A", "(", "٣٤٥", "٦", "d", "👍🏽<", "ع'D", "عAé'Re", "\r\n\r\n", " ", "\t", "'ſt", "#$%", "Z", "\"Dž", "‍.'", "Sa", "\r", " 𐞁", " \n\r\n"]} +{"text": "ꟲ,'M", "tokens": 5, "pieces": ["ꟲ", ",'", "M"]} +{"text": "㋿ fi,ع'll12345678'reé\r\n,\ns'", "tokens": 18, "pieces": ["㋿", " fi", ",ع'll", "123", "456", "78", "'reé", "\r\n", ",\n", "s", "'"]} +{"text": "'re-Ⅳ\"ḍ̇ع!!s", "tokens": 11, "pieces": ["'re", "-", "Ⅳ", "\"ḍ̇ع", "!!", "s"]} +{"text": "'lléꟲ\n12345678ḍ̇Ⅳ0㍿'VE'D12345678e㍿ß㋿'sß-½s\u000b\t'll…#$%\r Z'TtEOT 𐞁,ꟲd", "tokens": 65, "pieces": ["'lléꟲ", "\n", "123", "456", "78", "ḍ̇", "Ⅳ0", "㍿'", "VE'D", "123", "456", "78", "e", "㍿ß", "㋿'", "sß", "-", "½", "s", "\u000b", "\t", "'ll", "…", "#$%\r", " <", "EOT", ">Z'T", "t", "EOT", " 𐞁", ",ꟲd"]} +{"text": "\u000b
åZ‍ \r\n\r\n\r\n\r\nEOT \n‍tع…'Re\r\n!ⅣſⅣ!!😀🏽9<(\r\n>'​fi\u000b🙂", "tokens": 36, "pieces": ["\u000b", "
å", "Z", "‍", " \r\n\r\n\r\n\r\n", "EOT", " \n", "‍tع", "…", "'Re", "\r\n", "!", "Ⅳ", "ſ", "Ⅳ", "!!😀🏽", "9", "<(\r\n", ">'​", "fi", "\u000b", "🙂"]} +{"text": " …३\r'M
!漢- 
12345678t<|endoftext|>
'D,\na'Red ٣٤٥٦漢\"é㋿ s‍>\r\n\r\ns9DžZ½😀🏽", "tokens": 53, "pieces": [" ", "…", "३", "\r", "'M", "
", "!漢", "-", " ", "
", "123", "456", "78", "t", "<|", "endoftext", "|>", "
", "'D", ",\n", "a'Re", "d", " ", "٣٤٥", "٦", "漢", "\"é", "㋿", " s", "‍>\r\n\r\n", "s", "9", "DžZ", "½", "😀🏽"]} +{"text": "'ll𐞁\r…'ll!Ⅳ-<|endoftext|>ß \r\n\r\n\r\n\r\nét㋿\n'Tt12345678", "tokens": 35, "pieces": ["'ll𐞁", "\r", "…", "'ll", "!", "Ⅳ", "-<|", "endoftext", "|>", "ß", " \r\n\r\n\r\n\r\n", "ét", "㋿\n", "'Tt", "123", "456", "78"]} +{"text": "'s\u000b-'s‍EOT#$%ſ\u000bßEOT\u000b(åعfi", "tokens": 19, "pieces": ["'s", "\u000b", "-'", "s", "‍EOT", "#$%", "ſ", "\u000bß", "EOT", "\u000b", "(åعfi"]} +{"text": "\t< '漢e漢12345678'Re'D'D912345678\u000b éad ḍ̇İmm!!12345678३㋿\r\n", "tokens": 36, "pieces": ["\t", "<", " ", " '", "漢e漢", "123", "456", "78", "'Re'D", "'D", "912", "345", "678", "\u000b ", " éad", " ḍ̇", "İmm", "!!", "123", "456", "78३", "㋿\r\n"]} +{"text": "Z09٣٤٥٦0é'T're 'T", "tokens": 12, "pieces": ["Z", "09٣", "٤٥٦", "0", "é'T", "'re", " ", "'T"]} +{"text": "㍿٣٤٥٦\nt㍿字\r\n\r\n漢٣٤٥٦\n<'re‍>\t㍿㍿Ⅳ\r\n!'VE.9e'.ع", "tokens": 41, "pieces": ["㍿", "٣٤٥", "٦", "\n", "t", "㍿字", "\r\n\r\n", "漢", "٣٤٥", "٦", "\n", "<'", "re", "‍>", "\t", "㍿㍿", "Ⅳ", "\r\n", "!'", "VE", ".", "9", "e", "'.", "ع"]} +{"text": "!! ‍é½३A'VEſ<|fim_prefix|>ع​é😀🏽'T\r<|fim_prefix|> d!!a'll٣٤٥٦½A<字s😀🏽#$%d½㋿e#$%", "tokens": 62, "pieces": ["!!", " ", "‍é", "½३", "A'VE", "ſ", "<|", "fim", "_prefix", "|>", "ع", "​é", "😀🏽'", "T", "\r", "<|", "fim", "_prefix", "|>", " d", "!!", "a", "'", "ll", "٣٤٥", "٦½", "A", "<字s", "😀🏽#$%", "d", "½", "㋿e", "#$%"]} +{"text": "ع​A(t12345678>'S'T😀🏽a'Re٣٤٥٦\r\nd", "tokens": 28, "pieces": ["ع", "​A", "(<", "META", "_START", ">t", "123", "456", "78", ">'", "S'T", "😀🏽", "a'Re", "٣٤٥", "٦", "\r\n", "d"]} +{"text": "!!\"
#$%\"'lléß
ß 12345678->t👍🏽\r\n\r\n're漢 ‍<|fim_prefix|>d0 \n 0's㋿e#$%0😀🏽", "tokens": 55, "pieces": ["!!\"", "
", "#$%\"'", "lléß", "
ß", " ", "123", "456", "78", "->", "t", "👍🏽\r\n\r\n", "'re漢", " ", "‍<|", "fim", "_prefix", "|>", "d", "0", " \n", " ", "0", "'s", "㋿e", "#$%<", "META", "_START", ">", "0", "😀🏽"]} +{"text": "'Da<३éſd'Ret'Re𐞁字字'DEOT\r\n\r\nEOT\"½'S­ -\r.tå\r\n\r\n​​ع㋿😀🏽", "tokens": 41, "pieces": ["'Da", "<", "३", "éſd'Re", "t'Re", "𐞁字字'D", "EOT", "\r\n\r\n", "EOT", "\"", "½", "'S", "­", " ", "-\r", ".tå", "\r\n\r\n", "​​", "ع", "㋿😀🏽"]} +{"text": "\r\n'llA", "tokens": 3, "pieces": ["\r\n", "'ll", "A"]} +{"text": ">🙂\"'Rea", "tokens": 5, "pieces": [">🙂\"'", "Rea"]} +{"text": "'D'ſ'٣٤٥٦ <\u000b漢㋿0'llA.\t'D🙂'T٣٤٥٦ ٣٤٥٦-<|endoftext|>!! <|fim_prefix|><|endoftext|>字Džé㍿ع\r\nd\u000b're㍿​ḍ̇", "tokens": 76, "pieces": ["'D'ſ", "'", "٣٤٥", "٦", " ", " <", "\u000b漢", "㋿", "0", "'ll", "A", ".", "\t", "'D", "🙂'", "T", "٣٤٥", "٦", " ", " ", "٣٤٥", "٦", "-<|", "endoftext", "|>!!", " ", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "字Džé", "㍿ع", "\r\n", "d", "\u000b", "'re", "㍿​", "ḍ̇"]} +{"text": "' 字'D٣٤٥٦dſZ \u000b­ 字é'reéé
ḍ̇fi😀🏽İḍ̇é‍३md𐞁<|endoftext|>­👍🏽​'VE'‍\r", "tokens": 61, "pieces": ["'", " 字'D", "٣٤٥", "٦", "dſ", "Z", " ", "\u000b", "­", " 字é're", "éé", "
", "ḍ̇fi", "😀🏽", "İḍ̇é", "‍", "३", "md𐞁", "<|", "endoftext", "|>­👍🏽​'", "VE", "'‍\r"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "😀🏽 ㍿'ſ😀🏽👍🏽ḍ̇'Dİ'S字\r9😀🏽漢字fi<|endoftext|>s(\t­\r(", "tokens": 43, "pieces": ["😀🏽", " ", "㍿'", "ſ", "😀🏽👍🏽", "ḍ̇'D", "İ'S", "字", "\r", "9", "😀🏽", "漢字fi", "<|", "endoftext", "|>", "s", "(", "\t", "­\r", "("]} +{"text": "åm½s\"'Re'D字\r\n''ſ字ع0s12345678😀🏽A\u000b<|endoftext|>\"
EOT're", "tokens": 35, "pieces": ["åm", "½", "s", "\"'", "Re'D", "字", "\r\n", "''", "ſ字ع", "0", "s", "123", "456", "78", "😀🏽", "A", "\u000b", "<|", "endoftext", "|>\"", "
EOT're"]} +{"text": " \ń👍🏽1234567812345678", "tokens": 11, "pieces": [" \n", "́", "👍🏽", "123", "456", "781", "234", "567", "8"]} +{"text": "\r\"\"é\r\n\r\n#$%😀🏽-𐞁…\r\n de…t­s>EOT!!", "tokens": 27, "pieces": ["\r", "\"\"", "é", "\r\n\r\n", "#$%😀🏽-", "𐞁", "…\r\n", " ", " de", "…t", "­s", ">EOT", "!!"]} +{"text": "́ḍ̇", "tokens": 4, "pieces": ["́ḍ̇"]} +{"text": "''字9ꟲ 'VE\u000b😀🏽\n'ſ!0Zß٣٤٥٦m 😀🏽<<|fim_prefix|>字å 'Re\r\n३'T'D<", "字å", " ", " '", "Re", "\r\n", "३", "'T'D", "<<", "d", "👍🏽", "İ", "…", "("]} +{"text": "ع<", "tokens": 2, "pieces": ["ع", "<"]} +{"text": "
\t\u000bḍ̇'\"", "tokens": 7, "pieces": ["
\t", "\u000bḍ̇", "'\""]} +{"text": "\"'M३ds😀🏽'T​\r\n.\t d🙂٣٤٥٦
३'Ss\r\n\r\nfi'T\r\n\r\n\u000b", "tokens": 28, "pieces": ["\"'", "M", "३", "ds", "😀🏽'", "T", "​\r\n", ".", "\t ", " d", "🙂", "٣٤٥", "٦", "
", "३", "'Ss", "\r\n\r\n", "fi'T", "\r\n\r\n", "\u000b"]} +{"text": "0‍'SZ(12345678㋿🙂'D12345678
\"ⅣEOT'T\r\n\r\n字<|endoftext|>#$%<|endoftext|>'D\u000b<|fim_prefix|>​ A'Refi­Dž0'<ſfi-,🙂", "tokens": 64, "pieces": ["0", "‍'", "SZ", "(", "123", "456", "78", "㋿🙂'", "D", "123", "456", "78", "
", "\"", "Ⅳ", "EOT'T", "\r\n\r\n", "字", "<|", "endoftext", "|>#$%<|", "endoftext", "|>'", "D", "\u000b", "<|", "fim", "_prefix", "|>​", " A'Re", "fi", "­Dž", "0", "'<", "ſfi", "-,🙂"]} +{"text": "< \"'Re<|fim_prefix|>", "tokens": 15, "pieces": ["<", " ", "\"'", "Re", "<|", "fim", "_prefix", "|><", "META", "_START", ">"]} +{"text": "½\u000b'ſADž.\r­><|fim_prefix|>!!½s ß🙂", "tokens": 22, "pieces": ["½", "\u000b", "'ſ", "ADž", ".\r", "­><|", "fim", "_prefix", "|>!!", "½", "s", " ß", "🙂"]} +{"text": "0.३é!90🙂d9'VE​字 t.'S३😀🏽0ß字<\rA٣٤٥٦🙂字m<<|fim_prefix|>👍🏽", "tokens": 48, "pieces": ["0", ".", "३", "é", "!", "90", "🙂d", "9", "'VE", "​字", " t", ".'", "S", "३", "😀🏽", "0", "ß字", "<\r", "A", "٣٤٥", "٦", "🙂字m", "<<|", "fim", "_prefix", "|>👍🏽"]} +{"text": "ḍ̇漢'Dé​#$%㋿'re''re<>\r\né😀🏽12345678", "tokens": 27, "pieces": ["ḍ̇漢'D", "é", "​#$%㋿'", "re", "''", "re", "<>\r\n", "é", "😀🏽", "123", "456", "78"]} +{"text": "'ll३-té½e'…\n​ß\u000bⅣ😀🏽Ⅳ'M9!३'", "tokens": 25, "pieces": ["'ll", "३", "-té", "½", "e", "'", "…\n", "​ß", "\u000b", "Ⅳ", "😀🏽", "Ⅳ", "'M", "9", "!", "३", "'"]} +{"text": "'llع😀🏽 -'s Džع99( 'Re9até9٣٤٥٦㍿漢->\"EOTéEOT½!!ßa12345678​
'VE'Re", "tokens": 51, "pieces": ["'llع", "😀🏽", " -'", "s", " Džع", "99", "(", " '", "Re", "9", "até", "9٣٤", "٥٦", "㍿漢", "->\"", "EOTé", "EOT", "½", "!!", "ßa", "123", "456", "78", "​", "
", "'VE'Re"]} +{"text": "ع'VE字", "tokens": 4, "pieces": ["ع'VE", "字"]} +{"text": "٣٤٥٦𐞁12345678'…​'VE𐞁(ß", "tokens": 26, "pieces": ["٣٤٥", "٦", "𐞁", "123", "456", "78", "'<", "META", "_START", ">", "…", "​'", "VE𐞁", "(ß"]} +{"text": "12345678字(İ 字😀🏽३\u000bm𐞁<|fim_prefix|>👍🏽 \"fi'T'Msꟲ́½​\"漢fi\u000b<|endoftext|>㍿fi", "tokens": 57, "pieces": ["123", "456", "78", "字", "(İ", " ", " 字", "😀🏽", "३", "\u000bm𐞁", "<|", "fim", "_prefix", "|>👍🏽", " ", " \"", "fi'T", "'Msꟲ́", "½", "​\"<", "EOT", ">漢fi", "\u000b", "<|", "endoftext", "|>㍿", "fi"]} +{"text": "㍿12345678­ꟲ'VE'DꟲDž'Sḍ̇ḍ̇.'T'\r\n\r\n'll\rDž \n'llZåſ<|endoftext|>𐞁.", "tokens": 49, "pieces": ["㍿", "123", "456", "78", "­ꟲ'VE", "'Dꟲ", "Dž'S", "ḍ̇ḍ̇", ".'", "T", "'\r\n\r\n", "'ll", "\r", "Dž", " \n", "'ll", "Zåſ", "<|", "endoftext", "|>", "𐞁", "."]} +{"text": "<ꟲ३é<|fim_prefix|>0(​😀🏽", "tokens": 19, "pieces": ["<ꟲ", "३", "é", "<|", "fim", "_prefix", "|>", "0", "(​😀🏽"]} +{"text": " 's\"'ll\r'M… 'D'\t­ ‍A㋿\r\n\r\n \n.", "tokens": 27, "pieces": [" '", "s", "\"'", "ll", "\r", "'M", "… ", " '", "D", "'", "\t", "­", " ", "‍A", "㋿\r\n\r\n", " \n", "."]} +{"text": " 'ReEOTİ́é­s𐞁'Dع٣٤٥٦'M0½㋿\n'VEa", "tokens": 28, "pieces": [" '", "Re", "EOTİ́é", "­s𐞁'D", "ع", "٣٤٥", "٦", "'M", "0½", "㋿\n", "'VEa"]} +{"text": "­\"A", "tokens": 3, "pieces": ["­\"", "A"]} +{"text": "!!\r\na㋿'reéDž\ré!!!!t字tḍ̇!!'DDž .\r\u000be漢,", "tokens": 40, "pieces": ["!!\r\n", "a", "㋿'", "reé", "Dž", "\r", "é", "!!!!", "t字tḍ̇", "!!<", "m", "'", "DDž", " ", ".\r", "\u000be漢", ","]} +{"text": " 12345678!((\rm‍e ½ſ'VE 'S३\t‍.9<", "tokens": 23, "pieces": [" ", "123", "456", "78", "!((\r", "m", "‍e", " ", "½", "ſ'VE", " ", " '", "S", "३", "\t", "‍.", "9", "<"]} +{"text": "!<|endoftext|>12345678\ns½0<\"٣٤٥٦​
A㍿ \n
İ字é​", "tokens": 33, "pieces": ["!<|", "endoftext", "|>", "123", "456", "78", "\n", "s", "½0", "<\"", "٣٤٥", "٦", "​", "
A", "㍿", " \n", "
İ字é", "​"]} +{"text": "🙂ꟲ\u000bt\n
​㋿㍿.\r\"​12345678㋿字e\"'ſ漢'T\t!ſ​
're<…​", "tokens": 50, "pieces": ["🙂ꟲ", "", "\u000bt", "\n", "
", "​㋿㍿.\r", "\"​", "123", "456", "78", "㋿字e", "\"'", "ſ漢'T", "\t", "!ſ", "​", "
", "'re", "<", "…", "​"]} +{"text": "ßEOT½ 0½عZ '‍‍'VEé\u000b", "tokens": 18, "pieces": ["ß", "EOT", "½", " ", "0½", "ع", "Z", " ", "'‍‍'", "VEé", "\u000b"]} +{"text": "åİa\t字9㋿9…<|fim_prefix|>, ‍'re…\t'12345678-'D<|fim_prefix|>漢 \n!'M 'D\t字…åe", "tokens": 51, "pieces": ["å", "İa", "\t字", "9", "㋿", "9", "…", "<|", "fim", "_prefix", "|>,", " ", "‍'", "re", "…", "\t", "'", "123", "456", "78", "-'", "D", "<|", "fim", "_prefix", "|>", "漢", " \n", "!'", "M", " ", "'D", "\t字", "…åe"]} +{"text": "ḍ̇\nḍ̇​٣٤٥٦", "tokens": 12, "pieces": ["ḍ̇", "\n", "ḍ̇", "​", "٣٤٥", "٦"]} +{"text": "fi ,ß\"İZ09\u000b \né\"'Reİꟲ\"ß'T́!!('ree-'S'reḍ̇Z
'VE", "tokens": 35, "pieces": ["fi", " ", " ,", "ß", "\"İZ", "09", "\u000b \n", "é", "\"'", "Re", "İꟲ", "\"ß'T", "́", "!!('", "ree", "-'", "S're", "ḍ̇", "Z", "
", "'VE"]} +{"text": "漢ꟲ🙂İ👍🏽\r\n\r\n字' A😀🏽(ꟲ𐞁<|endoftext|>AZéa👍🏽٣٤٥٦\r", "tokens": 48, "pieces": ["漢ꟲ", "🙂İ", "👍🏽\r\n\r\n", "字", "'", " A", "😀🏽(", "ꟲ𐞁", "<|", "endoftext", "|>", "AZéa", "👍🏽", "٣٤٥", "٦", "\r"]} +{"text": "12345678​
\n‍­9½­'re<|endoftext|>İ👍🏽fi'ſé½'ſ​\u000b㋿0
字é<|endoftext|>d३0A㍿­٣٤٥٦12345678>EOT-🙂", "tokens": 67, "pieces": ["123", "456", "78", "​", "
\n", "‍­", "9½", "­'", "re", "<|", "endoftext", "|>", "İ", "👍🏽", "fi'ſ", "é", "½", "'ſ", "​", "\u000b", "㋿", "0", "
字é", "<|", "endoftext", "|>", "d", "३0", "A", "㍿­", "٣٤٥", "٦12", "345", "678", ">EOT", "-🙂"]} +{"text": "عt\"é<|fim_prefix|>Dž!!é㋿'VEé½\n-٣٤٥٦", "tokens": 30, "pieces": ["عt", "\"é", "<|", "fim", "_prefix", "|>", "Dž", "!!", "é", "㋿'", "VEé", "½", "\n", "-", "٣٤٥", "٦"]} +{"text": "İ㋿'T'VE'SⅣ0EOT٣٤٥٦,.٣٤٥٦A½", "tokens": 25, "pieces": ["İ", "㋿'", "T'VE", "'S", "Ⅳ0", "EOT", "٣٤٥", "٦", ",.", "٣٤٥", "٦", "A", "½"]} +{"text": "ße'D३Z‍", "tokens": 5, "pieces": ["ße'D", "३", "Z", "‍"]} +{"text": "é<|fim_prefix|>9dåſ!!Z ", "tokens": 15, "pieces": ["é", "<|", "fim", "_prefix", "|>", "9", "dåſ", "!!", "Z", " "]} +{"text": " 'SEOTDž\r\n\u000b
‍ع😀🏽\r\nع'sḍ̇å
", "tokens": 25, "pieces": [" '", "SEOTDž", "\r\n", "\u000b", "
", "‍ع", "😀🏽\r\n", "ع", "'", "sḍ̇å", "
"]} +{"text": "𐞁㋿ḍ̇\n ½ſḍ̇m漢", "tokens": 22, "pieces": ["𐞁", "㋿ḍ̇", "\n", " ", "½", "ſḍ̇m漢", ""]} +{"text": "'T३!!9'T'MⅣ\r\n\r\n🙂½.ꟲ-Z…'s0漢عع\r'Re(ع'll123456780㋿(<|endoftext|>\r\n३é", "tokens": 43, "pieces": ["'T", "३", "!!", "9", "'T'M", "Ⅳ", "\r\n\r\n", "🙂", "½", ".ꟲ", "-Z", "…", "'s", "0", "漢عع", "\r", "'Re", "(ع'll", "123", "456", "780", "㋿(<|", "endoftext", "|>\r\n", "३", "é"]} +{"text": "(
\"
>", "tokens": 5, "pieces": ["(", "
", "\"", "
", ">"]} +{"text": "漢́\r\n\r\n\t́字字a 'Dé", "tokens": 15, "pieces": ["漢́", "\r\n\r\n", "\t", "́字字a", " ", " '", "Dé"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " Dž 'refi\r'\u000b<|endoftext|>Z\t0", "tokens": 19, "pieces": [" Dž", " '", "refi", "\r", "'", "\u000b", "<|", "endoftext", "|>", "Z", "\t", "0"]} +{"text": "'Re㋿İ-éEOTAſa‍字\"㍿…İع​'VE'VE㋿Aḍ̇ꟲ 's.👍🏽>\r\n\r\n३ \n\r\n…12345678ad \n㍿", "tokens": 60, "pieces": ["'", "Re", "㋿İ", "-é", "EOTAſa", "‍字", "\"㍿", "…İع", "​'", "VE'VE", "㋿Aḍ̇ꟲ", " '", "s", ".👍🏽>\r\n\r\n", "३", " \n\r\n", "…", "123", "456", "78", "ad", " \n", "㍿"]} +{"text": "\",\r\n\r\nḍ̇\r\n\r\n<|endoftext|>(tİ<|fim_prefix|>'re\r\n\r\n🙂'llea\u000bAe", "tokens": 34, "pieces": ["\",\r\n\r\n", "ḍ̇", "\r\n\r\n", "<|", "endoftext", "|>(", "t", "İ", "<|", "fim", "_prefix", "|>'", "re", "\r\n\r\n", "🙂'", "llea", "\u000b", "Ae"]} +{"text": "sZ0's😀🏽
३", "tokens": 9, "pieces": ["s", "Z", "0", "'s", "😀🏽", "
", "३"]} +{"text": "\rs字 \n漢'D​m𐞁e0,Z0½", "tokens": 17, "pieces": ["\r", "s字", " \n", "漢'D", "​m𐞁e", "0", ",Z", "0½"]} +{"text": "12345678 \r\n'Re>ſ\ne A\t­ \n…mAfi#$%éå'", "tokens": 26, "pieces": ["123", "456", "78", " \r\n", "'Re", ">ſ", "\n", "e", " A", "\t", "­", " \n", "…m", "Afi", "#$%", "éå", "'"]} +{"text": "<|fim_prefix|> \na\r\n\r\n'.'𐞁🙂३'re.0\r\n\r\n👍🏽,<|endoftext|>< éDžé'T𐞁㍿\r'll.𐞁👍🏽\u000bİ‍", "tokens": 57, "pieces": ["<|", "fim", "_prefix", "|>", " \n", "a", "\r\n\r\n", "'.'", "𐞁", "🙂", "३", "'re", ".", "0", "\r\n\r\n", "👍🏽,<|", "endoftext", "|><", " é", "Džé'T", "𐞁", "㍿\r", "'ll", ".𐞁", "👍🏽", "\u000bİ", "‍"]} +{"text": "‍'ſ
 'ſ​!!.EOT عİ#$%\r\n\r\n漢's'Sfi", "tokens": 21, "pieces": ["‍'", "ſ", "
 ", " '", "ſ", "​!!.", "EOT", " ", " ع", "İ", "#$%\r\n\r\n", "漢's", "'Sfi"]} +{"text": "! \n#$%ع'llmꟲDž'VE\t\r\n\r\n-½­'VE𐞁EOT're's🙂", "tokens": 29, "pieces": ["!", " \n", "#$%", "ع'll", "mꟲ", "Dž'VE", "\t\r\n\r\n", "-", "½", "­'", "VE𐞁", "EOT're", "'s", "🙂"]} +{"text": "fi\nⅣ'Re㍿!!­", "tokens": 10, "pieces": ["fi", "\n", "Ⅳ", "'Re", "㍿!!­"]} +{"text": "'M!!a0'VE\r\n\r\nß漢(Ⅳm ع>(!", "tokens": 17, "pieces": ["'M", "!!", "a", "0", "'VE", "\r\n\r\n", "ß漢", "(", "Ⅳ", "m", " ", " ع", ">(!"]} +{"text": "'VE#$%efi're.'re'ſ'T'VEḍ̇㋿'S㋿<|endoftext|>٣٤٥٦👍🏽", "tokens": 39, "pieces": ["'VE", "#$%", "efi're", ".'", "re'ſ", "'T'VE", "ḍ̇", "㋿'", "S", "㋿<|", "endoftext", "|>", "٣٤٥", "٦", "👍🏽"]} +{"text": "<|endoftext|>'ſ 'T0'll𐞁<|endoftext|>'S<|endoftext|>'s👍🏽🙂👍🏽9d🙂ßعZ, 's<|endoftext|>😀🏽'T's\u000bŹ \n<<<\ré𐞁", "tokens": 75, "pieces": ["<|", "endoftext", "|>'", "ſ", " ", " '", "T", "0", "'ll𐞁", "<|", "endoftext", "|>'", "S", "<|", "endoftext", "|>'", "s", "👍🏽🙂👍🏽", "9", "d", "🙂ßع", "Z", ",", " ", " '", "s", "<|", "endoftext", "|>😀🏽'", "T's", "\u000bŹ", " \n", "<<<\r", "é𐞁"]} +{"text": "fi'ſ.#$%\r\n\r\n'Md‍åsfié !!å\r12345678! <|fim_prefix|>é\"漢字漢\né½> \r㍿", "tokens": 46, "pieces": ["fi'ſ", ".#$%\r\n\r\n", "'Md", "‍åsfié", " ", "!!", "å", "\r", "123", "456", "78", "!", " ", "<|", "fim", "_prefix", "|>", "é", "\"漢字漢", "\n", "é", "½", ">", " \r", "㍿"]} +{"text": "e'ReⅣm 'T#$%👍🏽\u000b,", "tokens": 14, "pieces": ["e'Re", "Ⅳ", "m", " ", "'T", "#$%👍🏽", "\u000b", ","]} +{"text": "'ſع9ꟲfi", "tokens": 8, "pieces": ["'ſع", "9", "ꟲfi"]} +{"text": "(å字téعⅣé,", "tokens": 10, "pieces": ["(å字téع", "Ⅳ", "é", ","]} +{"text": "<́'Re9'㍿ ½\"<|fim_prefix|>\r\n.<|endoftext|>  're\r\n\r\nms٣٤٥٦é😀🏽'VE漢ßdd́>sḍ̇é'Sm", "tokens": 49, "pieces": ["<́'Re", "9", "'㍿", " ", "½", "\"<|", "fim", "_prefix", "|>\r\n", ".<|", "endoftext", "|>", " ", " ", "'re", "\r\n\r\n", "ms", "٣٤٥", "٦", "é", "😀🏽'", "VE漢ßdd́", ">sḍ̇é'S", "m"]} +{"text": ">‍\n'ſ- Z字'ſſ12345678åꟲعs'llDž\r\n\r\nfiⅣß'Mé👍🏽Dž漢㋿👍🏽'S\u000b'Så", "tokens": 51, "pieces": [">‍\n", "'ſ", "-", " Z字'ſ", "ſ", "123", "456", "78", "åꟲعs'll", "Dž", "\r\n\r\n", "fi", "Ⅳ", "ß'M", "é", "👍🏽", "Dž漢", "㋿👍🏽'", "S", "\u000b", "'Så"]} +{"text": "漢ꟲ…\r\n\r\nés'Sİ12345678ſaع'MEOTⅣ
'Tꟲ", "tokens": 26, "pieces": ["漢ꟲ", "…\r\n\r\n", "és'S", "İ", "123", "456", "78", "ſaع'M", "EOT", "Ⅳ", "
", "'Tꟲ"]} +{"text": "Ⅳ 'VE>ḍ̇\n'S\n🙂<|endoftext|>.Dž''MZ㍿'ll\r\n\r\né‍12345678٣٤٥٦😀🏽'llé'Mt㋿'ll\n\tEOTß", "tokens": 57, "pieces": ["Ⅳ", " ", " '", "VE", ">ḍ̇", "\n", "'S", "\n", "🙂<|", "endoftext", "|>.", "Dž", "''", "MZ", "㍿'", "ll", "\r\n\r\n", "é", "‍", "123", "456", "78٣", "٤٥٦", "😀🏽'", "llé'M", "t", "㋿'", "ll", "\n", "\tEOTß"]} +{"text": "ḍ̇fi", "tokens": 4, "pieces": ["ḍ̇fi"]} +{"text": "åDžßDžⅣ.'Ré-\"12345678fifi!!<|endoftext|>", "tokens": 29, "pieces": ["å", "Džß", "Dž", "Ⅳ", ".'", "Re", "́", "-\"", "123", "456", "78", "fifi", "!!<|", "endoftext", "|>"]} +{"text": "½<|fim_prefix|>​12345678a", "tokens": 12, "pieces": ["½", "<|", "fim", "_prefix", "|>​", "123", "456", "78", "a"]} +{"text": "e EOTß👍🏽<'VE𐞁(<|fim_prefix|>#$%<ḍ̇'Re're#$%#$%å'ſſ́\rå字 \n>'M're\t", "tokens": 53, "pieces": ["e", " ", " EOTß", "👍🏽<'", "VE𐞁", "<", "EOT", ">(<|", "fim", "_prefix", "|>#$%<", "ḍ̇'Re", "'re", "#$%#$%", "å'ſ", "ſ́", "\r", "å字", " \n", ">'", "M're", "\t"]} +{"text": "
ḍ̇'s\r\n0 \n㍿'T!'T,'ll🙂<|endoftext|>Dž!'s🙂's'll'VE's'\u000b", "tokens": 38, "pieces": ["
ḍ̇'s", "\r\n", "0", " \n", "㍿'", "T", "!'", "T", ",'", "ll", "🙂<|", "endoftext", "|>", "Dž", "!'", "s", "🙂'", "s'll", "'VE's", "'", "\u000b"]} +{"text": "\u000b\t👍🏽'll#$%m<٣٤٥٦‍\t 'D😀🏽<\n​­EOT३.㋿!㋿Z", "tokens": 37, "pieces": ["\u000b", "\t", "👍🏽'", "ll", "#$%", "m", "<", "٣٤٥", "٦", "‍", "\t ", " '", "D", "😀🏽<\n", "​­", "EOT", "३", ".㋿!㋿", "Z"]} +{"text": "(9'll", "tokens": 3, "pieces": ["(", "9", "'ll"]} +{"text": "🙂
…ſd#$%\u000b'VEé'Re\u000ba(<|endoftext|>mÁ🙂字​'S \r\n'llß'llⅣ", "tokens": 38, "pieces": ["🙂", "
", "…ſd", "#$%", "\u000b", "'VEé'Re", "\u000ba", "(<|", "endoftext", "|>", "m", "Á", "🙂字", "​'", "S", " \r\n", "'llß'll", "Ⅳ"]} +{"text": " EOT'Re\u000b0字", "tokens": 6, "pieces": [" EOT'Re", "\u000b", "0", "字"]} +{"text": "Z. 
9عt 'Rea🙂字m!<㍿ꟲḍ̇fi'D ḍ̇9're('ſß\r\n\r\nss", "tokens": 44, "pieces": ["Z", ".<", "EOT", ">", " ", "
", "9", "عt", " '", "Rea", "🙂字m", "!<㍿", "ꟲḍ̇fi'D", " ḍ̇", "9", "'re", "('", "ſß", "\r\n\r\n", "ss", ""]} +{"text": "ⅣEOT<|endoftext|>fi'll<|fim_prefix|>e \tZ 'T\"ꟲ字,EOT漢Dž\"\r\n'ſ\r\nſ><\r\n\r\n'S­👍🏽\råé#$%", "tokens": 55, "pieces": ["Ⅳ", "EOT", "<|", "endoftext", "|>", "fi'll", "<|", "fim", "_prefix", "|>", "e", " ", "\tZ", " ", "'T", "\"ꟲ字", ",EOT漢", "Dž", "\"\r\n", "'ſ", "\r\n", "ſ", "><\r\n\r\n", "'S", "­👍🏽\r", "åé", "#$%"]} +{"text": "0'ſ㋿'T >\n㍿'re字!!'reDž\t👍🏽 dd-​‍0İ ", "tokens": 31, "pieces": ["0", "'ſ", "㋿'", "T", " ", ">\n", "㍿'", "re字", "!!'", "re", "Dž", "\t", "👍🏽", " dd", "-​‍", "0", "İ", " "]} +{"text": "\r-'>12345678\"!!漢㍿٣٤٥٦­AmAع", "tokens": 21, "pieces": ["\r", "-'>", "123", "456", "78", "\"!!", "漢", "㍿", "٣٤٥", "٦", "­Am", "Aع"]} +{"text": "\u000b٣٤٥٦\n'D😀🏽Z'S\"'ſ٣٤٥٦​9\r\n\r\n(#$%!!🙂ع<>İt३'ḍ̇  ́Ⅳ㍿㍿", "tokens": 50, "pieces": ["\u000b", "٣٤٥", "٦", "\n", "'D", "😀🏽", "Z'S", "\"'", "ſ", "٣٤٥", "٦", "​", "9", "\r\n\r\n", "(#$%<", "EOT", ">!!🙂", "ع", "<>", "İt", "३", "'ḍ̇", " ", " ́", "Ⅳ", "㍿㍿"]} +{"text": "'s\r\n", "tokens": 5, "pieces": ["'", "s", "\r\n"]} +{"text": "… 'ſ​>Dž,'s<'Re<|fim_prefix|>fißß<'S'VE<|fim_prefix|>t🙂Ⅳ\u000b\r'VE<|fim_prefix|>e,a<|endoftext|>字AİDž", "tokens": 68, "pieces": ["… ", " '", "ſ", "​>", "Dž", ",'", "s", "<'", "Re", "<|", "fim", "_prefix", "|>", "fißß", "<'", "S'VE", "<|", "fim", "_prefix", "|>", "t", "🙂<", "EOT", ">", "Ⅳ", "\u000b\r", "'VE", "<|", "fim", "_prefix", "|>", "e", ",a", "<|", "endoftext", "|>", "字", "AİDž"]} +{"text": "漢's'T𐞁\u000b\t12345678'D9ſḍ̇9'VE'S㋿Z👍🏽d­漢éd", "tokens": 38, "pieces": ["漢's", "'T𐞁", "\u000b", "\t", "123", "456", "78", "'", "D", "9", "ſḍ̇", "9", "'VE'S", "㋿Z", "👍🏽", "d", "­漢éd"]} +{"text": "é٣٤٥٦­#$%ḍ̇½ḍ̇ ㍿ḍ̇\t­½٣٤٥٦'Re𐞁EOTع<|fim_prefix|><|fim_prefix|>!Ⅳfié<|fim_prefix|>\r…0#$%​'M \nt're", "tokens": 77, "pieces": ["é", "٣٤٥", "٦", "­#$%", "ḍ̇", "½", "ḍ̇", " ", " ㍿", "ḍ̇", "\t", "­", "½٣٤", "٥٦", "'Re𐞁", "EOTع", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>!", "Ⅳ", "fié", "<|", "fim", "_prefix", "|>\r", "…", "0", "#$%​'", "M", " \n", "t're"]} +{"text": "\n­Ⅳ\r\n\r\n 'M,
\r\n\r\n,é e'Sfi👍🏽", "tokens": 20, "pieces": ["\n", "­", "Ⅳ", "\r\n\r\n", " ", "'M", ",", "
", "\r\n\r\n", ",é", " e'S", "fi", "👍🏽"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\u000b(́'res ꟲm\r\néé\r\n\r\nḍ̇!.'S字
\r\nfia'ſDž'T's­'VEm३d字'll", "tokens": 34, "pieces": ["👍🏽", "123", "456", "78३", "­​<", "META", "_START", ">.'", "S字", "
\r\n", "fia'ſ", "Dž'T", "'s", "­'", "VEm", "३", "d字'll"]} +{"text": "<\r\n\r\nå<'Tfi9#$%Dž\r\n😀🏽🙂‍😀🏽\t,'T.'s\u000b字e#$% \u000b t9٣٤٥٦\t'ſ \n", "tokens": 50, "pieces": ["<<", "META", "_START", ">\r\n\r\n", "å", "<'", "Tfi", "9", "#$%<", "EOT", ">Dž", "\r\n", "😀🏽🙂‍😀🏽", "\t", ",'", "T", ".'", "s", "\u000b字e", "#$%", " \u000b ", " t", "9٣٤", "٥٦", "\t", "'ſ", " \n"]} +{"text": ">\ń٣٤٥٦-a‍<|endoftext|>\u000bé'T12345678Dž0eß\r\nfi\r\n\r\n ß́́ ⅣDž<|fim_prefix|>t​…åZ​a<|fim_prefix|>", "tokens": 64, "pieces": [">\n", "́", "٣٤٥", "٦", "-a", "‍<|", "endoftext", "|>", "\u000bé'T", "123", "456", "78", "Dž", "0", "eß", "\r\n", "fi", "\r\n\r\n", " ß́", "́", " ", "Ⅳ", "Dž", "<|", "fim", "_prefix", "|>", "t", "​", "…å", "Z", "​a", "<|", "fim", "_prefix", "|>"]} +{"text": "­..😀🏽're漢\t\t\r\n😀🏽\u000bß'VE𐞁é漢<|endoftext|>​Ⅳ'Ś…<|fim_prefix|>(Ⅳ\tm३", "tokens": 49, "pieces": ["­..😀🏽'", "re漢", "\t\t\r\n", "😀🏽", "\u000bß'VE", "𐞁é漢", "<|", "endoftext", "|>​", "Ⅳ", "'", "Ś", "…", "<|", "fim", "_prefix", "|>(", "Ⅳ", "\tm", "३"]} +{"text": "😀🏽漢😀🏽 's-\"\r'ſDž0#$%!!👍🏽! Dž㋿'<|endoftext|>'VEع m\r字Ⅳ٣٤٥٦>Dž½<​<|fim_prefix|>", "tokens": 71, "pieces": ["😀🏽", "漢", "😀🏽", " ", " '", "s", "-\"\r", "'ſ", "Dž", "0", "#$%!!👍🏽!<", "META", "_START", ">", " Dž", "㋿'<|", "endoftext", "|>'", "VEع", " ", " m", "\r", "字", "Ⅳ٣٤", "٥٦", ">Dž", "½", "<​<", "EOT", "><|", "fim", "_prefix", "|><", "EOT", ">"]} +{"text": "‍#$%éİ
३>0\"'VEé", "tokens": 13, "pieces": ["‍#$%", "é", "İ", "
", "३", ">", "0", "\"'", "VEé"]} +{"text": "‍,'res'VE‍DžⅣ's٣٤٥٦\r\n\r\n'Re 🙂'M\r\n\r\n\rdſ \nZ½é", "tokens": 29, "pieces": ["‍,'", "res'VE", "‍Dž", "Ⅳ", "'s", "٣٤٥", "٦", "\r\n\r\n", "'Re", " ", " 🙂'", "M", "\r\n\r\n\r", "dſ", " \n", "Z", "½", "é"]} +{"text": "'re!ع\rmDž😀🏽#$%…é
\nİ\"é'VEß(sd'VE <漢­ꟲ\r\n\r\nså'M", "tokens": 38, "pieces": ["'re", "!ع", "\r", "m", "Dž", "😀🏽#$%", "…é", "
\n", "İ", "\"é'VE", "ß", "(sd'VE", " ", "<漢", "­ꟲ", "\r\n\r\n", "så'M"]} +{"text": "'D, \r\n\r\nعꟲ'ſ", "tokens": 9, "pieces": ["'D", ",", " \r\n\r\n", "عꟲ'ſ"]} +{"text": "漢<|fim_prefix|>aéع \na'VE­‍ſDž字­t'\te𐞁0́'VEsİ
ß字'll!İ(-ꟲ", "tokens": 51, "pieces": ["漢", "<|", "fim", "_prefix", "|>", "aéع", " \n", "a'VE", "­‍", "ſ", "Dž", "字", "­t", "'", "\te𐞁", "0", "́'VE", "s", "İ", "
ß字'll", "!İ", "(-<", "EOT", ">ꟲ"]} +{"text": "ßİꟲ😀🏽'\r\ne<|fim_prefix|>Dž٣٤٥٦t(ſ'ReA", "tokens": 27, "pieces": ["ß", "İꟲ", "😀🏽'\r\n", "e", "<|", "fim", "_prefix", "|>", "Dž", "٣٤٥", "٦", "t", "(ſ'Re", "A"]} +{"text": "'ll字,aß'Re½ A", "tokens": 8, "pieces": ["'ll字", ",aß'Re", "½", " A"]} +{"text": "Ⅳ's漢İ
(𐞁0ée\r\n\r\nd½\rḍ̇9!9
​,٣٤٥٦\r\nع'ſ३'Re", "tokens": 38, "pieces": ["Ⅳ", "'s漢", "İ", "
", "(𐞁", "0", "ée", "\r\n\r\n", "d", "½", "\r", "ḍ̇", "9", "!", "9", "
", "​,", "٣٤٥", "٦", "\r\n", "ع'ſ", "३", "'Re"]} +{"text": "e'M 9 (…'Re!Z<|fim_prefix|>'D", "tokens": 21, "pieces": ["e'M", " ", "9", " ", "(", "…", "'Re", "!", "Z", "<|", "fim", "_prefix", "|>'", "D"]} +{"text": "Z \r😀🏽‍'VE字A😀🏽-\t#$%'>́Dž\r-\r\n're漢'ſ", "tokens": 28, "pieces": ["Z", " \r", "😀🏽‍'", "VE字", "A", "😀🏽-", "\t", "#$%'>́", "Dž", "\r", "-\r\n", "'re漢'ſ"]} +{"text": "é
٣٤٥٦ 'M🙂fim 'S<", "tokens": 15, "pieces": ["é", "
", "٣٤٥", "٦", " ", " '", "M", "🙂fim", " '", "S", "<"]} +{"text": "å
🙂ⅣDž\n👍🏽s's'res \n's'll!fi😀🏽​a're", "tokens": 27, "pieces": ["å", "
", "🙂", "Ⅳ", "Dž", "\n", "👍🏽", "s's", "'res", " \n", "'s'll", "!fi", "😀🏽​", "a're"]} +{"text": "9EOT‍ 're\"<|endoftext|>३t'T\n.İ\n'T́s🙂<'D\ta", "tokens": 28, "pieces": ["9", "EOT", "‍", " ", "'re", "\"<|", "endoftext", "|>", "३", "t'T", "\n", ".İ", "\n", "'T́s", "🙂<'", "D", "\ta"]} +{"text": "'ſ'llⅣ", "tokens": 8, "pieces": ["'ſ'll", "Ⅳ", ""]} +{"text": "🙂ß𐞁'll<|endoftext|> \n<|endoftext|>A<|fim_prefix|>'M字㍿'Re's!!!fiEOTⅣ", "tokens": 48, "pieces": ["🙂ß𐞁'll", "<|", "endoftext", "|>", " \n", "<|", "endoftext", "|>", "A", "<|", "fim", "_prefix", "|>'", "M字", "㍿'", "Re's", "!<", "META", "_START", ">!!", "fi", "EOT", "Ⅳ"]} +{"text": "\u000bé‍'re
é́漢  t<|endoftext|>'Re12345678\"­ ́\u000b字e\r'll", "tokens": 47, "pieces": ["\u000bé", "‍'", "re", "
é́漢", " ", " t", "<|", "endoftext", "|>'", "Re", "123", "456", "78", "\"­", " ́", "<", "EOT", ">㋿<", "META", "_START", ">", "\u000b字e", "\r", "'ll"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "fi'Dİ'reéåع\"Ⅳ(téİ", "tokens": 17, "pieces": ["fi'D", "İ're", "é", "åع", "\"", "Ⅳ", "(té", "İ"]} +{"text": "9å'reⅣ're­\u000b'Téd , 字😀🏽'Såع٣٤٥٦ḍ̇\r,½ꟲ漢‍ꟲ…Ⅳ's", "tokens": 57, "pieces": ["9", "å're", "Ⅳ", "'re", "­", "\u000b", "'Téd", " ", ",", " 字", "😀🏽'", "Såع", "", "٣٤٥", "٦", "ḍ̇", "\r", ",", "½", "ꟲ漢", "‍ꟲ", "…", "Ⅳ", "'s"]} +{"text": "㋿‍!!​'re-EOT\r\n\r\n-​,'ll​éd\n!ß­ m'Rea'ſ \r'Red…0", "tokens": 41, "pieces": ["㋿‍!!​'", "re", "-EOT", "\r\n\r\n", "-​,'", "ll", "​éd", "\n", "!", "ß", "­", " m'Re", "a'ſ", " \r", "'Re", "d", "…", "0"]} +{"text": "'M'ſ'D½\t…𐞁….\n12345678İ
.'M‍㍿<|fim_prefix|>", "tokens": 35, "pieces": ["'M'ſ", "'D", "½", "\t", "…𐞁", "…", ".\n", "123", "456", "78", "İ", "
", ".'", "M", "‍㍿<|", "fim", "_prefix", "|>"]} +{"text": "ß\ra
9😀🏽'ſ>…😀🏽Ⅳ­'T<㋿İꟲ‍'ſ 0'll.t", "tokens": 36, "pieces": ["ß", "\r", "a", "
", "9", "😀🏽'", "ſ", ">", "…", "😀🏽", "Ⅳ", "­'", "T", "<㋿", "İꟲ", "‍'", "ſ", " ", "0", "'ll", ".t"]} +{"text": "'sſ'Re\"\n123456780'ſa ", "tokens": 11, "pieces": ["'sſ'Re", "\"\n", "123", "456", "780", "'ſa", " "]} +{"text": "\r\n 漢t ḍ̇😀🏽sꟲ'Dt\rfi\" 漢td ", "tokens": 25, "pieces": ["\r\n", " 漢t", " ḍ̇", "😀🏽", "sꟲ'D", "t", "\r", "fi", "\"", " ", " 漢td", " "]} +{"text": "ع㋿Z'M𐞁 \n…'ll'VE,e.Džs", "tokens": 21, "pieces": ["ع", "㋿Z'M", "𐞁", " \n", "…", "'ll'VE", ",e", ".Džs"]} +{"text": " 'DéA<\r\n\r\n>Dž!d字#$%'D!½e‍­s!<\réAعfi'Re", "tokens": 29, "pieces": [" '", "Dé", "A", "<\r\n\r\n", ">Dž", "!d字", "#$%'", "D", "!", "½", "e", "‍­", "s", "!<\r", "é", "Aعfi'Re"]} +{"text": "(aſİ'ſ
字A\"३e9!\r ½
's​\r\n\r\n'VE", "tokens": 23, "pieces": ["(aſ", "İ'ſ", "
字", "A", "\"", "३", "e", "9", "!\r", " ", " ", "½", "
", "'s", "​\r\n\r\n", "'VE"]} +{"text": "A \nꟲ'T\"fis
's𐞁🙂å\r\na\r\n\r\n", "tokens": 21, "pieces": ["A", " \n", "ꟲ'T", "\"fis", "
", "'s𐞁", "🙂å", "\r\n", "a", "\r\n\r\n"]} +{"text": "m 'VE-'.👍🏽EOTé\r\nétå", "tokens": 17, "pieces": ["m", " ", "'VE", "-'.👍🏽", "EOTé", "\r\n", "étå"]} +{"text": "'S'D-ſ‍#$%'Dß́'S s!'MZé>é", "tokens": 19, "pieces": ["'S'D", "-ſ", "‍#$%'", "Dß́'S", " ", " s", "!'", "MZé", ">é"]} +{"text": "'ll,'VEs'VE're​a​\"-Zع'Séå…ꟲ字", "tokens": 24, "pieces": ["'ll", ",'", "VEs'VE", "'re", "​a", "​\"-", "Zع'S", "éå", "…ꟲ字"]} +{"text": "\n \n s mfi !\t'Dt!!ꟲ,dİ12345678㍿…'Reع<|endoftext|>​漢ع😀🏽'ſ  m", "tokens": 44, "pieces": ["\n \n", " ", " s", " ", " mfi", " !", "\t", "'Dt", "!!", "ꟲ", ",d", "İ", "123", "456", "78", "㍿", "…", "'Reع", "<|", "endoftext", "|>​", "漢ع", "😀🏽'", "ſ", " ", " m"]} +{"text": "㍿ßßm'DDž ٣٤٥٦'Re
 𐞁(½\n \n٣٤٥٦<|endoftext|>å'Mß́0\"…,'Re'M(fi<|fim_prefix|>EOT㋿Ⅳ٣٤٥٦", "tokens": 69, "pieces": ["㍿ßßm'D", "Dž", " ", " ", "٣٤٥", "٦", "'Re", "
", " 𐞁", "(", "½", "\n \n", "٣٤٥", "٦", "<|", "endoftext", "|>", "å'M", "ß́", "0", "\"", "…", ",'", "Re'M", "(", "fi", "<|", "fim", "_prefix", "|>", "EOT", "㋿", "Ⅳ٣٤", "٥٦"]} +{"text": "(ꟲé <|fim_prefix|> \n<|fim_prefix|> 🙂😀🏽fi0a>\n٣٤٥٦éꟲ😀🏽́é", " \n", "<|", "fim", "_prefix", "|>", " ", " 🙂😀🏽", "fi", "0", "a", ">\n", "٣٤٥", "٦", "éꟲ", "😀🏽́", "é", "½12345678…😀🏽Ⅳ<|endoftext|>\"३ع㍿té
d​m½", "tokens": 49, "pieces": ["…", "'lléع", " \n", "<'", "re", "Ⅳ", "<ß", "<|", "fim", "_prefix", "|>", "½12", "345", "678", "…", "😀🏽", "Ⅳ", "<|", "endoftext", "|>\"", "३", "ع", "㍿té", "
d", "​m", "½"]} +{"text": "٣٤٥٦Ⅳ ​Ⅳ'reſ́'Re٣٤٥٦éé<<|endoftext|> <|fim_prefix|>عåⅣ😀🏽 ḍ̇'D'D
٣٤٥٦㍿'s'VEs'Re👍🏽0'VEå\r\n>字", "tokens": 74, "pieces": ["٣٤٥", "٦Ⅳ", " ", " ​", "Ⅳ", "'reſ́'Re", "٣٤٥", "٦", "éé", "<<|", "endoftext", "|>", " ", " <|", "fim", "_prefix", "|>", "عå", "Ⅳ", "😀🏽", " ", " ḍ̇'D", "'D", "
", "٣٤٥", "٦", "㍿'", "s'VE", "s'Re", "👍🏽", "0", "'VEå", "\r\n", ">字"]} +{"text": "#$%!!㋿Z½٣٤٥٦ḍ̇\tAtعd㋿漢'Téع9A'é\né­#$%Dž…", "tokens": 44, "pieces": ["#$%!!㋿", "Z", "½٣٤", "٥٦", "ḍ̇", "\tAtعd", "㋿<", "META", "_START", ">漢'T", "éع", "9", "A", "'é", "\n", "é", "­#$%", "Dž", "…"]} +{"text": "m
ḍ̇ع're're漢 \n\r\nAⅣ\n'res<|fim_prefix|>‍عé12345678", "tokens": 30, "pieces": ["m", "
ḍ̇ع're", "'re漢", " \n\r\n", "A", "Ⅳ", "\n", "'res", "<|", "fim", "_prefix", "|>‍", "عé", "123", "456", "78"]} +{"text": "é'SDž'sعß'T.", "tokens": 9, "pieces": ["é'S", "Dž's", "عß'T", "."]} +{"text": "\" --Dž-‍mA\r\n\n\" \n're", "tokens": 15, "pieces": ["\"", " ", "--", "Dž", "-‍", "m", "A", "\r\n\n", "\"", " \n", "'re"]} +{"text": "\u000bmé0‍­'S㍿­३𐞁Dž‍", "tokens": 19, "pieces": ["\u000bmé", "0", "‍­'", "S", "㍿­", "३", "𐞁", "Dž", "‍"]} +{"text": "!!字㍿ Ⅳ #$% !!漢<|fim_prefix|>😀🏽,
\"ꟲEOT  Z\"'Tعå٣٤٥٦0>😀🏽-tfi\t㋿😀🏽", "tokens": 62, "pieces": ["!!", "字", "㍿", " ", "Ⅳ", " ", " #$%", " ", "!!", "漢", "<|", "fim", "_prefix", "|>😀🏽,", "
", "\"ꟲ", "EOT", " ", " Z", "\"'", "Tعå", "٣٤٥", "٦0", "><", "EOT", ">😀🏽-", "tfi", "\t", "㋿😀🏽"]} +{"text": "EOT'RedⅣZ🙂('Re'D\rDž'VE\t\t.'T㋿\u000b​㍿​😀🏽é0٣٤٥٦\r\nZ-'re🙂fi漢½'re.ع", "tokens": 50, "pieces": ["EOT'Re", "d", "Ⅳ", "Z", "🙂('", "Re'D", "\r", "Dž'VE", "\t", "\t", ".'", "T", "㋿", "\u000b", "​㍿​😀🏽", "é", "0٣٤", "٥٦", "\r\n", "Z", "-'", "re", "🙂fi漢", "½", "'re", ".ع"]} +{"text": "#$%½sé", "tokens": 4, "pieces": ["#$%", "½", "sé"]} +{"text": "\u000bat<İ…ß🙂عꟲ字're३,'T٣٤٥٦'Re…A\r\n ́t>\r\n's​
éDž9'Dع‍", "tokens": 48, "pieces": ["\u000bat", "<İ", "", "…ß", "🙂عꟲ字're", "३", ",'", "T", "٣٤٥", "٦", "'Re", "…A", "\r\n", " ", " ́t", ">\r\n", "'s", "​<", "EOT", ">", "
é", "Dž", "9", "'Dع", "‍"]} +{"text": "a漢\u000bİ​\n\r's!'Re'D👍🏽İ\r\n'VEİ'St ", "tokens": 21, "pieces": ["a漢", "\u000bİ", "​\n\r", "'s", "!'", "Re'D", "👍🏽", "İ", "\r\n", "'VEİ'S", "t", " "]} +{"text": "𐞁d'Sع㋿éݽ'S\u000bſ'T\"9ſ123456780\r\nß'smA,½ ", "tokens": 31, "pieces": ["𐞁d'S", "ع", "㋿é", "İ", "½", "'S", "\u000bſ'T", "\"", "9", "ſ", "123", "456", "780", "\r\n", "ß's", "m", "A", ",", "½", " "]} +{"text": "́m#$%'T ('VE\r,Ⅳ😀🏽!!½ ́\t- 0'D३å‍>eſd", "tokens": 37, "pieces": ["́m", "#$%'", "T", " ", "('", "VE", "\r", ",", "Ⅳ", "😀🏽!!", "½", " ", " ́", "\t", "-", " ", "0", "'D", "३", "å", "‍>", "eſd"]} +{"text": " \n ", "tokens": 2, "pieces": [" \n", " "]} +{"text": "\t \n<|endoftext|>\r\n9,\r\n12345678EOT", "tokens": 15, "pieces": ["\t \n", "<|", "endoftext", "|>\r\n", "9", ",\r\n", "123", "456", "78", "EOT"]} +{"text": "'Dſ'S9३İ ,'T🙂\u000b're\t'e'Re<fi​🙂<|fim_prefix|>'D,‍<|fim_prefix|> -12345678字12345678é ꟲ", "tokens": 52, "pieces": ["'Dſ'S", "9", "", "३", "İ", " ,'", "T", "🙂", "\u000b", "'re", "\t", "'e'Re", "<fi", "​🙂<|", "fim", "_prefix", "|>'", "D", ",‍<|", "fim", "_prefix", "|>", " ", "-", "123", "456", "78", "字", "123", "456", "78", "é", " ꟲ"]} +{"text": "(㋿\u000b0\r\n\r\n'M", "tokens": 8, "pieces": ["(㋿", "\u000b", "0", "\r\n\r\n", "'M"]} +{"text": "\t
'ſⅣ㍿ 字'D >", "tokens": 18, "pieces": ["\t", "
", "'ſ", "Ⅳ", "㍿", " ", "字'D", " ", ">"]} +{"text": "m\r㍿( Dž\n३-­\"\t's", "tokens": 19, "pieces": ["m", "\r", "㍿(", " Dž", "\n", "३", "-­\"<", "EOT", ">", "\t", "'s"]} +{"text": "३!İ🙂(,字d0​漢a", "tokens": 12, "pieces": ["३", "!İ", "🙂(,", "字d", "0", "​漢a"]} +{"text": "!!e \n<#$% a >ع \n​३\u000b'VE­\r", "tokens": 18, "pieces": ["!!", "e", " \n", "<#$%", " a", " >", "ع", " \n", "​", "३", "\u000b", "'VE", "­\r"]} +{"text": ",'Dž", "tokens": 3, "pieces": [",'", "Dž"]} +{"text": "A'ſ \n", "tokens": 4, "pieces": ["A'ſ", " \n"]} +{"text": "ſEOT0‍-éZع\r", "tokens": 10, "pieces": ["ſ", "EOT", "0", "‍-", "é", "Zع", "\r"]} +{"text": "'re EOTŹ'३عś\r.ع
… ß👍🏽", "tokens": 22, "pieces": ["'re", " EOTŹ", "'", "३", "عś", "\r", ".ع", "
…", " ß", "👍🏽"]} +{"text": "'DEOTA'Re\r\n\u000b \n😀🏽ae,ſDžꟲ.㍿'llİ३𐞁A'M'VE…é'll9<|endoftext|><|endoftext|>'", "tokens": 53, "pieces": ["'DEOTA'Re", "\r\n\u000b \n", "😀🏽", "ae", ",ſ", "Džꟲ", ".㍿'", "ll", "İ", "३", "𐞁", "A'M", "'VE", "…é'll", "9", "<|", "endoftext", "|><|", "endoftext", "|>'"]} +{"text": "Ⅳ('s\"!!'S#$%'llé'll\u000b ㋿!a's<|fim_prefix|>('re <|fim_prefix|>\r\né
½'Re\t½½'s!́'ll३ \n ", "tokens": 55, "pieces": ["Ⅳ", "('", "s", "\"!!'", "S", "#$%'", "ll", "é'll", "\u000b", " ㋿!", "a's", "<|", "fim", "_prefix", "|>('", "re", " ", "<|", "fim", "_prefix", "|>\r\n", "é", "
", "½", "'Re", "\t", "½½", "'s", "!́'ll", "३", " \n", " "]} +{"text": "e12345678", "tokens": 4, "pieces": ["e", "123", "456", "78"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "", "tokens": 0, "pieces": []} +{"text": "fi0‍ḍ̇<|endoftext|>ḍ̇>9 '<|fim_prefix|>\nß́'VE\r\n​(𐞁𐞁", "tokens": 43, "pieces": ["fi", "0", "‍ḍ̇", "<|", "endoftext", "|>", "ḍ̇", ">", "9", " ", " '<|", "fim", "_prefix", "|>\n", "ß́'VE", "\r\n", "​(", "𐞁𐞁"]} +{"text": "­-'D9👍🏽👍🏽é\"👍🏽e's<|fim_prefix|>12345678İt åé३'D >('VE\n", "tokens": 43, "pieces": ["­-<", "EOT", ">'", "D", "9", "👍🏽👍🏽", "é", "\"👍🏽", "e's", "<|", "fim", "_prefix", "|>", "123", "456", "78", "İt", " åé", "३", "'D", " ", ">('", "VE", "\n"]} +{"text": "'lls\r\n\r\n \"'TA'Re٣٤٥٦sé 😀🏽", "tokens": 17, "pieces": ["'lls", "\r\n\r\n", " \"'", "TA'Re", "٣٤٥", "٦", "sé", " ", "😀🏽"]} +{"text": "s\n'VE\rEOT\n", "tokens": 8, "pieces": ["s", "\n", "'VE", "\r", "EOT", "\n"]} +{"text": "fiⅣ
>", "tokens": 5, "pieces": ["fi", "Ⅳ", "
", ">"]} +{"text": "½!!عDžⅣⅣİ😀🏽'T \néß'Dm'Tعß!, ع>", "tokens": 30, "pieces": ["½", "!!", "ع", "Dž", "ⅣⅣ", "İ", "😀🏽'", "T", " \n", "éß'D", "m'T", "عß", "!,", " ع", ">"]} +{"text": " \r\n\r\nⅣ'Rea", "tokens": 5, "pieces": [" \r\n\r\n", "Ⅳ", "'Rea"]} +{"text": "#$%å's🙂㍿'Re\t😀🏽Z ­e‍'s漢're>!-'sß٣٤٥٦'s<|fim_prefix|>㍿0!! \ń<|endoftext|>👍🏽9m­A", "tokens": 68, "pieces": ["#$%", "å's", "🙂㍿'", "Re", "\t", "😀🏽", "Z", "", " ", "­e", "‍'", "s漢're", ">!-'", "sß", "٣٤٥", "٦", "'s", "<|", "fim", "_prefix", "|>㍿", "0", "!!", " \n", "́", "<|", "endoftext", "|>👍🏽", "9", "m", "­<", "META", "_START", ">A"]} +{"text": "A>㋿Ⅳ#$%­­'VEs\rİ'Md  \r\n…<'s.ſ٣٤٥٦EOT\"9​< t\u000b", "tokens": 43, "pieces": ["A", ">㋿", "Ⅳ", "#$%­­'", "VEs", "\r", "İ'M", "d", "  \r\n", "…", "<'", "s", ".", "ſ", "٣٤٥", "٦", "EOT", "\"", "9", "​<", " t", "\u000b"]} +{"text": "Aé-'\n", "tokens": 8, "pieces": ["Aé", "-'\n"]} +{"text": "<|fim_prefix|>a'Re<𐞁漢
½", "tokens": 19, "pieces": ["<|", "fim", "_prefix", "|>", "a'Re", "<<", "META", "_START", ">𐞁漢", "
", "½"]} +{"text": "<|endoftext|>٣٤٥٦Džéś9A 🙂s!<,㍿12345678\n'M'VE<|endoftext|>😀🏽", "tokens": 47, "pieces": ["<|", "endoftext", "|>", "٣٤٥", "٦", "Džéś", "9", "A", " ", "🙂s", "!<,㍿", "123", "456", "78", "\n", "'", "M'VE", "<|", "endoftext", "|>😀🏽"]} +{"text": "é🙂漢𐞁Ⅳfi'VEd.­ \n…\u000b㋿ \n𐞁㋿d\"EOT--'ſZ\tm", "tokens": 50, "pieces": ["é", "🙂漢", "𐞁", "Ⅳ", "fi'VE", "d", ".­", " \n", "…", "\u000b", "㋿", " \n", "𐞁", "㋿d", "\"EOT", "--'", "ſ", "Z", "\tm"]} +{"text": "'M'ret👍🏽ꟲ\u000bt12345678'M -٣٤٥٦!'re", "tokens": 23, "pieces": ["'M're", "t", "👍🏽", "ꟲ", "\u000bt", "123", "456", "78", "'M", " ", " -", "٣٤٥", "٦", "!'", "re"]} +{"text": "\r\n\r\nİ'VE٣٤٥٦­éZ\t \n", "tokens": 15, "pieces": ["\r\n\r\n", "İ'VE", "٣٤٥", "٦", "­é", "Z", "\t \n"]} +{"text": "<|endoftext|>'s🙂𐞁​‍d", "tokens": 16, "pieces": ["<|", "endoftext", "|>'", "s", "🙂𐞁", "​‍", "d"]} +{"text": "'sßꟲ ,12345678㍿🙂're<|fim_prefix|>'MAſ'll㋿½e!,'ll‍ꟲⅣ>", "tokens": 45, "pieces": ["'", "sßꟲ", " ,", "123", "456", "78", "㍿🙂'", "re", "<|", "fim", "_prefix", "|>'", "MAſ'll", "㋿", "½", "e", "!,'", "ll", "‍ꟲ", "Ⅳ", ">"]} +{"text": "<|endoftext|>\"'sfi", "tokens": 10, "pieces": ["<|", "endoftext", "|>\"'", "sfi"]} +{"text": "(m12345678㍿३'Re(İ'D( #$%sZ'llé0🙂‍­9", "tokens": 25, "pieces": ["(m", "123", "456", "78", "㍿", "३", "'Re", "(İ'D", "(", " ", "#$%", "s", "Z'll", "é", "0", "🙂‍­", "9"]} +{"text": "ſ(字\t​!ſA٣٤٥٦😀🏽  ३'TfiA…'s㋿<'VE", "tokens": 32, "pieces": ["ſ", "(字", "\t", "​!", "ſ", "A", "٣٤٥", "٦", "😀🏽", " ", " ", "३", "'Tfi", "A", "…", "'s", "㋿<'", "VE"]} +{"text": "' d字😀🏽­Z<|fim_prefix|>Ⅳ
fi", "tokens": 19, "pieces": ["'", " ", " d字", "😀🏽­", "Z", "<|", "fim", "_prefix", "|>", "Ⅳ", "
fi"]} +{"text": "😀🏽㍿", "tokens": 6, "pieces": ["😀🏽㍿"]} +{"text": "\t\n<|endoftext|>m,\r\n\r\n\u000b🙂\"<|endoftext|>\na", "tokens": 23, "pieces": ["\t\n", "<|", "endoftext", "|><", "EOT", ">m", ",\r\n\r\n", "\u000b", "🙂\"<|", "endoftext", "|>\n", "a"]} +{"text": "s,\"…!!'res‍Dž ḍ̇漢ⅣDž12345678'D'M😀🏽👍🏽 \n'D 
", "tokens": 35, "pieces": ["s", ",\"", "…", "!!'", "res", "‍Dž", " ", " ḍ̇漢", "Ⅳ", "Dž", "123", "456", "78", "'D'M", "😀🏽👍🏽", " \n", "'D", " 
"]} +{"text": "ḍ̇㋿ع'M''M m漢\r 'D0s'TZEOTfi­", "tokens": 26, "pieces": ["ḍ̇", "㋿ع'M", "''", "M", " m漢", "\r", " ", "'D", "0", "s'T", "ZEOTfi", "­"]} +{"text": "🙂åe㋿'D….ḍ̇́\n.DžⅣḍ̇😀🏽'VE'S漢😀🏽A!!d(a𐞁 \nZ<|fim_prefix|>٣٤٥٦EOT", "tokens": 60, "pieces": ["🙂<", "EOT", ">åe", "㋿'", "D", "…", ".ḍ̇́", "\n", ".Dž", "Ⅳ", "ḍ̇", "😀🏽'", "VE'S", "漢", "😀🏽", "A", "!!", "d", "(a𐞁", " \n", "Z", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "EOT"]} +{"text": "‍ aé<|endoftext|>A٣٤٥٦ſ\u000bꟲ\nt'll㋿ß\"!!\nséå- ع'M漢\r\n\r\n 'DEOT㍿12345678½A", "tokens": 50, "pieces": ["‍", " aé", "<|", "endoftext", "|>", "A", "٣٤٥", "٦", "ſ", "\u000bꟲ", "\n", "t'll", "㋿ß", "\"!!\n", "séå", "-", " ع'M", "漢", "\r\n\r\n", " '", "DEOT", "㍿", "123", "456", "78½", "A"]} +{"text": "dét\u000b \n#$%\t#$%'S12345678\r\n𐞁\u000b'T‍ 'ree㍿㍿\"İA. \nd\r\n\r\nééaß\n", "tokens": 45, "pieces": ["dét", "\u000b \n", "#$%", "\t", "#$%'", "S", "123", "456", "78", "\r\n", "𐞁", "\u000b", "'T", "‍", " ", " '", "ree", "㍿㍿\"", "İA", ".", " \n", "d", "\r\n\r\n", "éé", "aß", "\n"]} +{"text": "12345678İ\u000b'VE\u000b<|fim_prefix|>99,👍🏽A'D😀🏽½ 👍🏽'll३-á'Reß'll", "tokens": 37, "pieces": ["123", "456", "78", "İ", "\u000b", "'VE", "\u000b", "<|", "fim", "_prefix", "|>", "99", ",👍🏽", "A'D", "😀🏽", "½", " ", " 👍🏽'", "ll", "३", "-á'Re", "ß'll"]} +{"text": " ㋿\"", "tokens": 6, "pieces": [" ", " ㋿\""]} +{"text": "!!é's.m>'M🙂İ'<|fim_prefix|>\r\n( \u000b's ́,٣٤٥٦.'T'll
½m<'VEd'ſ", "tokens": 37, "pieces": ["!!", "é's", ".m", ">'", "M", "🙂İ", "'<|", "fim", "_prefix", "|>\r\n", "(", " ", "\u000b", "'s", " ́", ",", "٣٤٥", "٦", ".'", "T'll", "
", "½", "m", "<'", "VEd'ſ"]} +{"text": "-\n😀🏽(dd३!!'ſå'T\r\n\u000bſ'VE'VEa…\u000b​ 're", "tokens": 27, "pieces": ["-\n", "😀🏽(", "dd", "३", "!!'", "ſå'T", "\r\n", "\u000bſ'VE", "'VEa", "…", "\u000b", "​", " '", "re"]} +{"text": "\r\n\r\n<|endoftext|>12345678­‍\r\nع'sé'T>Ⅳ
ꟲ!́<|endoftext|>", "tokens": 35, "pieces": ["\r\n\r\n", "<|", "endoftext", "|>", "123", "456", "78", "­‍\r\n", "ع's", "é'T", ">", "Ⅳ", "
ꟲ", "!́", "<|", "endoftext", "|>"]} +{"text": "'D'VEt\r㋿\r", "tokens": 12, "pieces": ["'D'VE", "t", "\r", "㋿\r"]} +{"text": "'Re", "tokens": 9, "pieces": ["'Re", "å", ""]} +{"text": "t'll٣٤٥٦­\ra A ḍ̇🙂-𐞁'reⅣDžåé㍿\r\n\r\n'Re!\n👍🏽\r\n\r\n\r'S👍🏽å३‍\u000b㍿
­", "tokens": 55, "pieces": ["t'll", "٣٤٥", "٦", "­\r", "a", " A", " ḍ̇", "🙂-", "𐞁're", "Ⅳ", "Džåé", "㍿\r\n\r\n", "'Re", "!\n", "👍🏽\r\n\r\n\r", "'S", "👍🏽", "å", "३", "‍", "\u000b", "㍿", "
", "­"]} +{"text": "!!Ⅳ🙂Dž\r\n\r\n're \ńd漢½'S🙂må\"'S٣٤٥٦!🙂ß\t‍é", "tokens": 30, "pieces": ["!!", "Ⅳ", "🙂Dž", "\r\n\r\n", "'re", " \n", "́d漢", "½", "'S", "🙂må", "\"'", "S", "٣٤٥", "٦", "!🙂", "ß", "\t", "‍é"]} +{"text": ".­é 12345678'Re½👍🏽ꟲ字,!ſ'ſDž
👍🏽🙂३\n ", "tokens": 31, "pieces": [".­", "é", " ", "123", "456", "78", "'Re", "½", "👍🏽", "ꟲ字", ",!", "ſ'ſ", "Dž", "
", "👍🏽🙂", "३", "\n", " "]} +{"text": "eعé٣٤٥٦🙂 >­é ꟲ🙂\r\n\r\n\r", "tokens": 20, "pieces": ["eعé", "٣٤٥", "٦", "🙂", " ", " >­", "é", " ", " ꟲ", "🙂\r\n\r\n\r"]} +{"text": "åd<|endoftext|>'́\r\n\r\n's㋿😀🏽ḍ̇😀🏽'Re🙂å0'D‍#$%'VEḍ̇🙂 (🙂sd#$%éa\"\u000b­ſİ\r\n\r\n", "tokens": 54, "pieces": ["åd", "<|", "endoftext", "|>'́\r\n\r\n", "'s", "㋿😀🏽", "ḍ̇", "😀🏽'", "Re", "🙂å", "0", "'D", "‍#$%'", "VEḍ̇", "🙂", " ", "(🙂", "sd", "#$%", "éa", "\"", "\u000b", "­ſ", "İ", "\r\n\r\n"]} +{"text": "\rDžZmZ'Dt!'s!㍿", "tokens": 13, "pieces": ["\r", "DžZm", "Z'D", "t", "!'", "s", "!㍿"]} +{"text": "ݽ#$%ß👍🏽'll­'ſ\n字'Dé㍿a'३'D 😀🏽\t <|endoftext|>Dž३fiꟲEOTtdé…>'ſ", "tokens": 51, "pieces": ["İ", "½", "#$%", "ß", "👍🏽'", "ll", "­'", "ſ", "\n", "字'D", "é", "㍿a", "'", "३", "'D", " 😀🏽", "\t ", " <|", "endoftext", "|>", "Dž", "३", "fiꟲ", "EOTtdé", "…", ">'", "ſ"]} +{"text": "EOT ́<|fim_prefix|>a'llßع \n!,😀🏽é😀🏽-İ'sß#$%㋿é", "tokens": 34, "pieces": ["EOT", " ́", "<|", "fim", "_prefix", "|>", "a'll", "ßع", " \n", "!,😀🏽", "é", "😀🏽-", "İ's", "ß", "#$%㋿", "é"]} +{"text": "½<😀🏽'Re", "tokens": 7, "pieces": ["½", "<😀🏽'", "Re"]} +{"text": "३'Re<|endoftext|>😀🏽!३ḍ̇!!.😀🏽\u000b'M>㍿", "tokens": 27, "pieces": ["३", "'Re", "<|", "endoftext", "|>😀🏽!", "३", "ḍ̇", "!!.😀🏽", "\u000b", "'M", ">㍿"]} +{"text": "((", "tokens": 1, "pieces": ["(("]} +{"text": "'sée 'S‍\t.,Džm½….‍㍿Ⅳ 𐞁'll٣٤٥٦عa🙂
-ꟲ\r\n\r\n३'ſ<'T's!'-EOT
", "tokens": 52, "pieces": ["'sée", " '", "S", "‍", "\t", ".,", "Džm", "½", "…", ".‍㍿", "Ⅳ", " 𐞁'll", "٣٤٥", "٦", "عa", "🙂", "
", "-ꟲ", "\r\n\r\n", "३", "'ſ", "<'", "T's", "!'-", "EOT", "
"]} +{"text": "\t字…'s字em sae\" 'ſ12345678é'reİ'ss㋿<|fim_prefix|>#$% e漢'VE.fi>…é½ꟲ…", "tokens": 54, "pieces": ["Ⅳ", "'Re", "\u000b\n", "'ſ", "\tt", "", " sae", "\"", " ", " '", "ſ", "123", "456", "78", "é're", "İ's", "s", "㋿<|", "fim", "_prefix", "|>#$%", " e漢'VE", ".fi", ">", "…é", "½", "ꟲ", "…"]} +{"text": "ḍ̇Dž<|endoftext|>'MEOTA ſ'lls٣٤٥٦<|endoftext|>Zꟲ,a,\t\r\n\r\n\r\n", "tokens": 39, "pieces": ["ḍ̇", "Dž", "<|", "endoftext", "|>'", "MEOTA", " ", " ſ'll", "s", "٣٤٥", "٦", "<|", "endoftext", "|>", "Zꟲ", ",a", ",", "\t\r\n\r\n\r\n"]} +{"text": "EOT…㋿'T,t9'Me\u000b( A!!('ſⅣfi🙂're!0(ḍ̇'T'D㍿
's'll", "tokens": 40, "pieces": ["EOT", "…", "㋿'", "T", ",t", "9", "'Me", "\u000b", "(", " A", "!!('", "ſ", "Ⅳ", "fi", "🙂'", "re", "!", "0", "(ḍ̇'T", "'D", "㍿", "
", "'s'll"]} +{"text": "m<<|fim_prefix|>!!å🙂!!<|fim_prefix|>fi😀🏽éꟲſ'Re३ \n0😀🏽\ńå½
\r\n\r\n!!a'VE-ع>", "tokens": 49, "pieces": ["m", "<<|", "fim", "_prefix", "|>!!", "å", "🙂!!<|", "fim", "_prefix", "|>", "fi", "😀🏽", "éꟲſ'Re", "३", " \n", "0", "😀🏽\n", "́å", "½", "
\r\n\r\n", "!!", "a'VE", "-ع", ">"]} +{"text": "<|endoftext|>ß \nfi'Re9­'T\t 'Må\"㋿", "tokens": 23, "pieces": ["<|", "endoftext", "|>", "ß", " \n", "fi'Re", "9", "­'", "T", "\t", " '", "Må", "\"㋿"]} +{"text": "Ⅳ(d­\r>éⅣ", "tokens": 9, "pieces": ["Ⅳ", "(d", "­\r", ">é", "Ⅳ"]} +{"text": "ßİZ,Aé\"t 'VE‍'>'T're½!㍿\tm.<|endoftext|>ḍ̇,'VEd0!Zm!!e漢😀🏽t­å.­", "tokens": 48, "pieces": ["ß", "İZ", ",Aé", "\"t", " '", "VE", "‍'>'", "T're", "½", "!㍿", "\tm", ".<|", "endoftext", "|>", "ḍ̇", ",'", "VEd", "0", "!Zm", "!!", "e漢", "😀🏽", "t", "­å", ".­"]} +{"text": "'se'reeDžZ- e Ⅳ½Ⅳ👍🏽🙂tat𐞁ḍ̇'VE9<ꟲſİ𐞁(‍étꟲ", "tokens": 49, "pieces": ["'se're", "e", "DžZ", "-", " e", " ", "Ⅳ½Ⅳ", "👍🏽🙂", "tat𐞁ḍ̇'VE", "9", "<ꟲſ", "İ𐞁", "(‍", "étꟲ"]} +{"text": "ꟲ'ſ<|fim_prefix|>'M½ß\r\n\r\né½å​٣٤٥٦\u000b…'T​Ⅳsع🙂0", "tokens": 35, "pieces": ["ꟲ'ſ", "<|", "fim", "_prefix", "|>'", "M", "½", "ß", "\r\n\r\n", "é", "½", "å", "​", "٣٤٥", "٦", "\u000b", "…", "'T", "​", "Ⅳ", "sع", "🙂", "0"]} +{"text": "é're ३.‍dßع½𐞁< <\r're'T's(ḍ̇𐞁 \r\n\r\n\n\r'S12345678'\r\n\r\né\r\n", "tokens": 39, "pieces": ["é're", " ", "३", ".‍", "dßع", "½", "𐞁", "<", " ", "<\r", "'re'T", "'s", "(ḍ̇𐞁", " \r\n\r\n\n\r", "'S", "123", "456", "78", "'\r\n\r\n", "é", "\r\n"]} +{"text": "a
. 漢- -<|fim_prefix|>ع字٣٤٥٦fiꟲ'ſ漢'T<㋿'T#$%a#$%dⅣ#$%😀🏽", "tokens": 47, "pieces": ["a", "
", ".", " 漢", "-", " ", " -<|", "fim", "_prefix", "|>", "ع字", "٣٤٥", "٦", "fiꟲ'ſ", "漢'T", "<㋿'", "T", "#$%", "a", "#$%", "d", "Ⅳ", "#$%😀🏽"]} +{"text": "㋿d's'Dž'sDž<|endoftext|>'ll'VE,(­åſ𐞁ßDž\u000bfi>-‍t å", "tokens": 45, "pieces": ["㋿d's", "'Dž's", "Dž", "<|", "endoftext", "|>'", "ll'VE", ",(­", "åſ𐞁ß", "Dž", "\u000bfi", ">-‍", "t", " ", " å", ""]} +{"text": "'re#$%½a'S\t-< \n12345678'VE\r\n\r\n
…,\"t  ", "tokens": 24, "pieces": ["'re", "#$%", "½", "a'S", "\t", "-<", " \n", "123", "456", "78", "'", "VE", "\r\n\r\n", "
", "…", ",\"", "t", "  "]} +{"text": "३(漢'ſ­㍿ >", "tokens": 11, "pieces": ["३", "(漢'ſ", "­㍿", " ", ">"]} +{"text": "
'T'ſ٣٤٥٦\"\u000bꟲ<|fim_prefix|>'M'llſ\r\n'D\r\n\r\n'ſ٣٤٥٦", "tokens": 31, "pieces": ["
", "'T'ſ", "٣٤٥", "٦", "\"", "\u000bꟲ", "<|", "fim", "_prefix", "|>'", "M'll", "ſ", "\r\n", "'D", "\r\n\r\n", "'ſ", "٣٤٥", "٦"]} +{"text": "İ​é'S\r𐞁\r\n\r\n<…​!!9'Re", "tokens": 23, "pieces": ["İ", "​", "é'S", "\r", "𐞁", "\r\n\r\n", "<", "…", "​!!", "9", "'Re"]} +{"text": "\r\n\r\n0fi0-\u000b", "tokens": 8, "pieces": ["", "0", "fi", "0", "-", "\u000b"]} +{"text": "'ſꟲZ<|endoftext|>\u000b", "tokens": 14, "pieces": ["'ſꟲ", "Z", "<|", "endoftext", "|>", "\u000b"]} +{"text": "dt!!Z \n'T'D0\rEOT'Re<|fim_prefix|>aß\u000baⅣ !字İ'DAſ's\"ḍ̇'S漢Ae३a\r\n\r\nå'Re", "tokens": 47, "pieces": ["dt", "!!", "Z", " \n", "'T'D", "0", "\r", "EOT'Re", "<|", "fim", "_prefix", "|>", "aß", "\u000ba", "Ⅳ", " !", "字", "İ'D", "Aſ's", "\"ḍ̇'S", "漢Ae", "३", "a", "\r\n\r\n", "å'Re"]} +{"text": "'S 'T‍\t12345678m(\"字", "tokens": 11, "pieces": ["'S", " ", "'T", "‍", "\t", "123", "456", "78", "m", "(\"", "字"]} +{"text": "\r\n'ſ<漢'VE>,'Re\u000btfié漢
ḍ̇
'Mḍ̇ß's#$%\rع \nA!!", "tokens": 38, "pieces": ["\r\n", "'ſ", "<漢'VE", ">,'", "Re", "\u000btfié漢", "
ḍ̇", "
", "'Mḍ̇", "ß's", "#$%\r", "ع", " \n", "A", "!!"]} +{"text": "!字㋿字ß<|fim_prefix|><🙂ḍ̇\r漢 ſfi…\r\n\r\né​EOT'Mḍ̇m
ḿDžé­½<|endoftext|>字9", "tokens": 58, "pieces": ["!", "字", "㋿字ß", "<|", "fim", "_prefix", "|><🙂", "ḍ̇", "\r", "漢", " ſfi", "…\r\n\r\n", "é", "​EOT'M", "ḍ̇m", "
ḿ", "Džé", "­", "½", "<|", "endoftext", "|>", "字", "", "9"]} +{"text": "'M👍🏽字'ḍ̇d'D\r\n\r\nfi \n​åİZ'S<|fim_prefix|>'ſ'ſ\r\n\r\neA㍿漢're漢d<|fim_prefix|>9\n#$%😀🏽İ", "tokens": 56, "pieces": ["'M", "👍🏽", "字", "'ḍ̇d'D", "\r\n\r\n", "fi", " \n", "​å", "İ", "Z'S", "<|", "fim", "_prefix", "|>'", "ſ'ſ", "\r\n\r\n", "e", "A", "㍿漢're", "漢d", "<|", "fim", "_prefix", "|>", "9", "\n", "#$%😀🏽", "İ"]} +{"text": "'reع\r\n\nA'T😀🏽", "tokens": 31, "pieces": ["'", "reع", "\r\n", "\n", "A'T", "😀🏽"]} +{"text": " !!12345678\t'MZ'T.\t'VE𐞁\r!!٣٤٥٦!! ", "tokens": 28, "pieces": [" ", "!!", "123", "456", "78", "\t", "'MZ", "'", "T", ".", "\t", "'VE𐞁", "\r", "!!", "٣٤٥", "٦", "!!", " "]} +{"text": "!!ꟲ漢 𐞁½👍🏽9'!!漢,'re漢 ​é\r字!'VEefi\n<|fim_prefix|>d\r\nꟲ", "tokens": 43, "pieces": ["!!", "ꟲ漢", " 𐞁", "½", "👍🏽", "9", "'!!", "漢", ",'", "re漢", " ", "​é", "\r", "字", "!'", "VEefi", "\n", "<|", "fim", "_prefix", "|>", "d", "\r\n", "ꟲ"]} +{"text": " \n\u000b", "tokens": 2, "pieces": [" \n", "\u000b"]} +{"text": "<'Reé<|endoftext|>.'ReEOT \n", "tokens": 21, "pieces": ["<<", "EOT", ">'", "Reé", "<|", "endoftext", "|><", "META", "_START", ">.'", "Re", "EOT", " \n"]} +{"text": "m\n­ع é́'S'Mt!!''ll\r\n\r\n<'ſ,'", "tokens": 19, "pieces": ["m", "\n", "­ع", " ", " é́'S", "'Mt", "!!''", "ll", "\r\n\r\n", "<'", "ſ", ",'"]} +{"text": "t 'M
fi.EOTꟲḍ̇\t​\u000be'SA'Re\u000b,", "tokens": 22, "pieces": ["t", " '", "M", "
fi", ".EOTꟲḍ̇", "\t", "​", "\u000be'S", "A'Re", "\u000b", ","]} +{"text": "ßm'D .🙂\u000b<|endoftext|>A'VE'T३9<|fim_prefix|>ḍ̇ḍ̇‍s<|fim_prefix|><|fim_prefix|> #$%12345678å<|endoftext|>\r0tع(字\r", "tokens": 70, "pieces": ["ßm'D", " ", ".🙂", "\u000b", "<|", "endoftext", "|>", "A'VE", "'T", "३9", "<|", "fim", "_prefix", "|>", "ḍ̇ḍ̇", "‍s", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", " ", "#$%", "123", "456", "78", "å", "<|", "endoftext", "|><", "EOT", ">\r", "0", "tع", "(字", "\r"]} +{"text": "0'VE'll 'VE ,'sEOT12345678字<字㍿㋿!'Mꟲ!!Dž\r.åm'S0\tⅣDž#$%s½'ll🙂㍿'ll…㋿", "tokens": 59, "pieces": ["0", "'VE'll", " ", "'VE", " ", " ,'", "s", "EOT", "123", "456", "78", "字", "<字", "㍿㋿!'", "Mꟲ", "!!", "Dž", "\r", ".åm'S", "0", "\t", "Ⅳ", "Dž", "#$%", "s", "½", "'ll", "🙂㍿'", "ll", "…", "㋿"]} +{"text": "<-EOT​ \n㍿漢'VEdꟲ123456789", "tokens": 22, "pieces": ["<-", "EOT", "​", " \n", "㍿", "漢'VE", "dꟲ", "123", "456", "789"]} +{"text": "\r .٣٤٥٦漢㋿'ll👍🏽t🙂'D\t㋿,'VE..'ReA'ſ!!<|fim_prefix|> \u000b <(", "tokens": 46, "pieces": ["\r", " ", ".", "٣٤٥", "٦", "漢", "㋿'", "ll", "👍🏽", "t", "🙂'", "D", "\t", "㋿,'", "VE", "..'", "Re", "A'ſ", "!!<|", "fim", "_prefix", "|>", " \u000b", " <("]} +{"text": "…då!('T👍🏽,字漢㋿'ll12345678½\u000båå,👍🏽\"", "tokens": 32, "pieces": ["…då", "!('", "T", "👍🏽,", "字漢", "㋿'", "ll", "123", "456", "78½", "\u000båå", ",👍🏽\""]} +{"text": "ꟲéꟲ<|endoftext|>😀🏽s<|fim_prefix|>'Daéعt­'12345678 ३d​\t३'M#$%#$%\tDž09​👍🏽字<|fim_prefix|>漢fiꟲ", "tokens": 64, "pieces": ["ꟲéꟲ", "<|", "endoftext", "|>😀🏽", "s", "<|", "fim", "_prefix", "|>'", "Daéعt", "­'", "123", "456", "78", " ", "३", "d", "​", "\t", "३", "'M", "#$%#$%", "\tDž", "09", "​👍🏽", "字", "<|", "fim", "_prefix", "|>", "漢fiꟲ"]} +{"text": "éfiꟲß'ſ\r\n\r\n'VE\u000b\r ­'llḍ̇𐞁12345678ßꟲ½12345678İA漢'S12345678fi9🙂'Mt!!s", "tokens": 56, "pieces": ["éfiꟲ", "ß'ſ", "\r\n\r\n", "'VE", "\u000b\r", " ­'", "llḍ̇𐞁", "123", "456", "78", "ßꟲ", "½12", "345", "678", "İA漢'S", "123", "456", "78", "fi", "9", "🙂'", "Mt", "!!", "s"]} +{"text": "#$%\"Džé\r\nEOTⅣ\u000b \n.éaEOTḍ̇㍿ꟲd𐞁㍿Džſ‍s\"İ㋿\"fi<|fim_prefix|>'D\r\n\r\n", "tokens": 57, "pieces": ["#$%\"", "Dž", "é", "\r\n", "EOT", "Ⅳ", "\u000b \n", ".éa", "EOTḍ̇", "㍿ꟲd𐞁", "㍿Džſ", "‍s", "\"İ", "㋿\"", "fi", "<|", "fim", "_prefix", "|>'", "D", "\r\n\r\n"]} +{"text": "\tt'S'SEOT​Z.eꟲ'll'D½İ३字ḿ'!!tå é'll'Dİ(#$%­𐞁'll٣٤٥٦​t'Re\t …", "tokens": 49, "pieces": ["\tt'S", "'SEOT", "​Z", ".eꟲ'll", "'D", "½", "İ", "३", "字ḿ", "'!!", "tå", " ", " é'll", "'Dİ", "(#$%­", "𐞁'll", "٣٤٥", "٦", "​t'Re", "\t …"]} +{"text": "'VE>'ll٣٤٥٦", "tokens": 8, "pieces": ["'VE", ">'", "ll", "٣٤٥", "٦"]} +{"text": "­漢عe𐞁>ts e'!!'Red漢're!d!!ع\rsfi👍🏽", "tokens": 33, "pieces": ["­漢", "عe𐞁", ">", "ts", " ", " e", "'!!'", "Red漢're", "!d", "!!", "ع", "\r", "sfi", "👍🏽"]} +{"text": "㋿३\r\n<|fim_prefix|>Ⅳ'T'Mḍ̇ZZ\r\t \n\",'S'VE", "tokens": 30, "pieces": ["㋿", "३", "\r\n", "<", "META", "_START", "><|", "fim", "_prefix", "|>", "Ⅳ", "'T'M", "ḍ̇", "ZZ", "\r\t \n", "\",'", "S'VE"]} +{"text": "…'ſ\r\n\r\n\r\rⅣ ​'S>'T‍<|fim_prefix|>9ß\r'ſ''VE.!!'VEm'Dé…\r\n㍿ſꟲé\n\raEOT", "tokens": 58, "pieces": ["…", "'ſ", "\r\n\r\n\r\r", "", "Ⅳ", " ", "​'", "S", ">'", "T", "‍<|", "fim", "_prefix", "|><", "EOT", ">", "9", "ß", "\r", "'ſ", "''", "VE", ".!!'", "VEm'D", "é", "…\r\n", "㍿ſꟲé", "\n\r", "a", "EOT"]} +{"text": "#$%\r\n字'St‍'VE", "tokens": 8, "pieces": ["#$%\r\n", "字'S", "t", "‍'", "VE"]} +{"text": "Z​m'sß٣٤٥٦'ſ㋿ \n'́", "tokens": 17, "pieces": ["Z", "​m's", "ß", "٣٤٥", "٦", "'ſ", "㋿", " \n", "'́"]} +{"text": "ſع\u000bⅣ㋿!! -'re३fi<fi", "tokens": 17, "pieces": ["ſع", "\u000b", "Ⅳ", "㋿!!", " ", " -'", "re", "३", "fi", "<fi"]} +{"text": "😀🏽\"ع-́", "tokens": 7, "pieces": ["😀🏽\"", "ع", "-́"]} +{"text": "ꟲ'sⅣ 0́́'T!!㋿#$%३At<", "tokens": 21, "pieces": ["ꟲ's", "Ⅳ", " ", "0", "́́'T", "!!㋿#$%", "३", "At", "<"]} +{"text": "> 'S\r\n\r\n'll''VEꟲꟲ\r\n­AZ'Re\r\n漢9<½\r\n<‍İ\r\n­‍\n12345678'M.<|fim_prefix|>\t
tt", "tokens": 44, "pieces": [">", " '", "S", "\r\n\r\n", "'ll", "''", "VEꟲꟲ", "\r\n", "­AZ'Re", "\r\n", "漢", "9", "<", "½", "\r\n", "<‍", "İ", "\r\n", "­‍\n", "123", "456", "78", "'M", ".<|", "fim", "_prefix", "|>", "\t", "
tt"]} +{"text": "𐞁", "tokens": 4, "pieces": ["𐞁"]} +{"text": "\t​", "tokens": 2, "pieces": ["\t", "​"]} +{"text": "\u000b́\r\naع'S\r\n'S\u000bḍ̇'S'Re㋿ 'S'ſ,👍🏽'll12345678s 😀🏽\t\u000bm 'ḍ̇#$%", "tokens": 48, "pieces": ["\u000b́", "\r\n", "aع'S", "\r\n", "'S", "\u000bḍ̇'S", "'Re", "㋿", " ", "'S'ſ", ",👍🏽'", "ll", "123", "456", "78", "s", " ", " 😀🏽", "\t", "\u000bm", " ", "'<", "EOT", ">ḍ̇", "#$%"]} +{"text": "\"é're<|endoftext|>", "tokens": 10, "pieces": ["\"é're", "<|", "endoftext", "|>"]} +{"text": "👍🏽ꟲ字Ⅳ𐞁9e!!㋿'Mfi㍿'M9!", "tokens": 29, "pieces": ["👍🏽", "ꟲ字", "Ⅳ", "𐞁", "9", "e", "!!㋿'", "Mfi", "㍿'", "M", "9", "!"]} +{"text": "İé́'S'S'VE‍\"'D\r\n\r\n👍🏽 daAḍ̇ \nA'DDž㍿", "tokens": 31, "pieces": ["İé́'S", "'S'VE", "‍\"'", "D", "\r\n\r\n", "👍🏽", " ", " da", "Aḍ̇", " \n", "A", "'", "DDž", "㍿"]} +{"text": "Dž\r\n\r\n.é A<|fim_prefix|>es.ع漢'ſ<|endoftext|>ß<|fim_prefix|>ß<|endoftext|><́>(🙂ع<|fim_prefix|>'ll字", "tokens": 56, "pieces": ["Dž", "\r\n\r\n", ".é", " A", "<|", "fim", "_prefix", "|>", "es", ".ع", "漢'ſ", "<|", "endoftext", "|>", "ß", "<|", "fim", "_prefix", "|>", "ß", "<|", "endoftext", "|><́>(🙂", "ع", "<|", "fim", "_prefix", "|>'", "ll字"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'Re字\n#$%!
\r\n\r\n<|fim_prefix|>㍿ꟲA9'…
<|fim_prefix|>\r\n\r\n३EOT 
 \né'D0'M‍", "tokens": 42, "pieces": ["'Re字", "\n", "#$%!", "
\r\n\r\n", "<|", "fim", "_prefix", "|>㍿", "ꟲ", "A", "9", "'", "…", "
", "<|", "fim", "_prefix", "|>\r\n\r\n", "३", "EOT", " 
 \n", "é'D", "0", "'M", "‍"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ſ#$%'SA!! \n\"‍!!ḍ̇\r\n­å(,#$%😀🏽-…e \r(!! 9ß m\tꟲ('Re0é🙂㍿-", "tokens": 52, "pieces": ["ſ", "#$%'", "SA", "!!", " \n", "\"‍!!", "ḍ̇", "\r\n", "­å", "(,#$%😀🏽<", "EOT", ">-", "…e", " \r", "(!!", " ", "9", "ß", " m", "\tꟲ", "('", "Re", "0", "é", "🙂㍿-"]} +{"text": "ꟲé>\"'ll'S( ḍ̇\r<|endoftext|> -<🙂", "tokens": 49, "pieces": ["ꟲé", ">\"'", "ll'S", "(", " ", " ḍ̇", "\r", "<|", "endoftext", "|>", " ", "-<", "ta", "…", "‍\r\n", "́a'S", "å", "Ⅳ", "a", "!!\r", "🙂<|", "endoftext", "|><🙂"]} +{"text": "\"\nsDžte\u000b\u000bé㋿½
", "tokens": 14, "pieces": ["\"\n", "s", "Džte", "\u000b", "\u000bé", "㋿", "½", "
"]} +{"text": "<|endoftext|>>em\u000b\ŕßé(Ⅳ㋿'M'MA éḍ̇!fi!\r\n\r\n'T's.🙂'ReZ'Re…#$%", "tokens": 43, "pieces": ["<|", "endoftext", "|>>", "em", "\u000b\r", "́ßé", "(", "Ⅳ", "㋿'", "M'M", "A", " éḍ̇", "!fi", "!\r\n\r\n", "'T's", ".🙂'", "Re", "Z'Re", "…", "#$%"]} +{"text": "Z字9#$%!!efim123456780<|endoftext|>", "tokens": 23, "pieces": ["Z字", "9", "#$%!!", "efi", "m", "123", "456", "780", "<|", "endoftext", "|>"]} +{"text": "👍🏽'Re㍿ſß߅'re'reꟲ\t\"字ſ\"12345678", "tokens": 23, "pieces": ["\r", "<|", "endoftext", "|>", "…", "'re're", "ꟲ", "\t", "\"字ſ", "\"", "123", "456", "78"]} +{"text": "'T'll'VE (‍ſ३½👍🏽𐞁'VE😀🏽Ⅳ'\n👍🏽㍿​'D'lla#$% ", "tokens": 38, "pieces": ["'T'll", "'VE", " (‍", "ſ", "३½", "👍🏽", "𐞁'VE", "😀🏽", "Ⅳ", "'\n", "👍🏽㍿​'", "D'll", "a", "#$%", " "]} +{"text": "d…-\r\n\r\n'S!漢's…!m­­'T12345678'M'Reع
\u000b'S<|fim_prefix|>ع\r", "tokens": 33, "pieces": ["d", "…", "-\r\n\r\n", "'S", "!漢's", "…", "!m", "­­'", "T", "123", "456", "78", "'M'Re", "ع", "
", "\u000b", "'S", "<|", "fim", "_prefix", "|>", "ع", "\r"]} +{"text": "'D👍🏽𐞁<|fim_prefix|>Ⅳḍ̇'M字<|endoftext|>३", "tokens": 31, "pieces": ["'D", "👍🏽", "𐞁", "<|", "fim", "_prefix", "|>", "Ⅳ", "ḍ̇'M", "字", "<|", "endoftext", "|>", "३"]} +{"text": "9㋿e'TEOTafi😀🏽!!<字漢'M<'VE‍åꟲ fi\u000b😀🏽é​'M#$%", "tokens": 43, "pieces": ["9", "㋿e'T", "EOTafi", "😀🏽<", "EOT", ">!!<", "字漢'M", "<'", "VE", "‍åꟲ", " ", " fi", "\u000b", "😀🏽", "é", "​'", "M", "#$%"]} +{"text": "\"ḍ̇!ع٣٤٥٦EOT ,s>­'VE-İ'llⅣ", "tokens": 28, "pieces": ["\"ḍ̇", "!ع", "٣٤٥", "٦", "EOT", " ", " ,", "s", ">­'", "VE", "-<", "META", "_START", ">İ'll", "Ⅳ"]} +{"text": "(\r\n\r\n'MfiⅣ!!ſ\rfim‍9A<|fim_prefix|>'VEA ", "tokens": 23, "pieces": ["(\r\n\r\n", "'Mfi", "Ⅳ", "!!", "ſ", "\r", "fim", "‍", "9", "A", "<|", "fim", "_prefix", "|>'", "VEA", " "]} +{"text": "e­-\n12345678字", "tokens": 7, "pieces": ["e", "­-\n", "123", "456", "78", "字"]} +{"text": "Aé \nß9'TEOT𐞁#$%>\"912345678
12345678\n\"'s", "tokens": 25, "pieces": ["Aé", " \n", "ß", "9", "'TEOT𐞁", "#$%>\"", "912", "345", "678", "
", "123", "456", "78", "\n", "\"'", "s"]} +{"text": "'VE's \n\r\nİ ३ \nm<|endoftext|>㋿'llZ'll­İ\"9e-ꟲ'Re", "tokens": 37, "pieces": ["'VE", "'", "s", " \n\r\n", "İ", " ", "३", " \n", "m", "<|", "endoftext", "|>㋿'", "ll", "Z'll", "­İ", "\"", "9", "e", "-ꟲ'Re"]} +{"text": "漢\r\n\r\n\r\n\r\n<|endoftext|>EOT
<|endoftext|>\n٣٤٥٦字́'ſ'S< …\u000b\t'a \n'Re 'T𐞁'Re", "tokens": 50, "pieces": ["漢", "\r\n\r\n\r\n\r\n", "<|", "endoftext", "|>", "EOT", "", "
", "<|", "endoftext", "|>\n", "٣٤٥", "٦", "字́'ſ", "'S", "<", " …\u000b", "\t", "'a", "", " \n", "'Re", " ", "'T𐞁'Re"]} +{"text": "́…'D\"s>'lla", "tokens": 7, "pieces": ["́", "…", "'D", "\"s", ">'", "lla"]} +{"text": "s\nét-'M \n'De'ſfiḍ̇'ſ(İa½ EOT🙂ed㋿😀🏽12345678a\té­​\rZe…'", "tokens": 51, "pieces": ["s", "\n", "ét", "-'", "M", " \n", "'De'ſ", "fiḍ̇'ſ", "(İa", "½", " EOT", "🙂ed", "㋿😀🏽", "123", "456", "78", "a", "\té", "­​\r", "Ze", "…", "'"]} +{"text": "½字
字<|endoftext|>-dZ'Re 'ſ'VE‍字#$%0EOT'İ <|endoftext|>>(…A漢‍‍٣٤٥٦.!!.a're", "tokens": 50, "pieces": ["½", "字", "
字", "<|", "endoftext", "|>-", "d", "Z'Re", " ", "'ſ'VE", "‍字", "#$%", "0", "EOT", "'İ", " ", " <|", "endoftext", "|>>(", "…A漢", "‍‍", "٣٤٥", "٦", ".!!.", "a're"]} +{"text": "Z\r\n\r\né<|fim_prefix|>Z字㍿
å'M'Tſm½'Re\rⅣ \n's,9DžéDž字EOT漢ḍ̇ a12345678're ‍0\u000b漢", "tokens": 52, "pieces": ["Z", "\r\n\r\n", "é", "<|", "fim", "_prefix", "|>", "Z字", "㍿", "
å'M", "'Tſm", "½", "'Re", "\r", "Ⅳ", " \n", "'s", ",", "9", "Džé", "Dž字EOT漢ḍ̇", " a", "123", "456", "78", "'re", " ‍", "0", "\u000b漢"]} +{"text": "‍ 'ꟲ'T'Re<#$%'S<٣٤٥٦sDž\t \né\n's\r\nm0👍🏽😀🏽🙂 \r\n\r\nⅣ12345678ßt", "tokens": 42, "pieces": ["‍", " ", "'ꟲ'T", "'Re", "<#$%'", "S", "<", "٣٤٥", "٦", "s", "Dž", "\t \n", "é", "\n", "'s", "\r\n", "m", "0", "👍🏽😀🏽🙂", " \r\n\r\n", "Ⅳ12", "345", "678", "ßt"]} +{"text": "aae 're", "tokens": 3, "pieces": ["aae", " '", "re"]} +{"text": "s'ſع \r\n", "tokens": 5, "pieces": ["s'ſ", "ع", " \r\n"]} +{"text": "३\n🙂.a‍  012345678🙂 .… 'VE
½fiſ😀🏽", "tokens": 26, "pieces": ["३", "\n", "🙂.", "a", "‍", " ", " ", "012", "345", "678", "🙂", " ", " .", "…", " ", "'VE", "
", "½", "fiſ", "😀🏽"]} +{"text": "<|fim_prefix|>́.…😀🏽Ⅳ漢👍🏽éḍ̇字-EOTZ'ſḍ̇'#$%s😀🏽३!(ع  ́𐞁👍🏽(ZZ½-", "tokens": 59, "pieces": ["<|", "fim", "_prefix", "|>́.<", "EOT", ">", "…", "😀🏽", "Ⅳ", "漢", "👍🏽", "éḍ̇字", "-EOTZ'ſ", "ḍ̇", "'#$%", "s", "😀🏽", "३", "!(", "ع", " ", " ́𐞁", "👍🏽(", "ZZ", "½", "-"]} +{"text": ">­ ‍Džḍ̇  \t٣٤٥٦,\t12345678'S\u000b\tEOT'S\"‍ \nt'll Dž", "tokens": 34, "pieces": [">­", " ", " ‍", "Džḍ̇", "  ", "\t", "٣٤٥", "٦", ",", "\t", "123", "456", "78", "'S", "\u000b", "\tEOT'S", "\"‍", " \n", "t'll", " Dž"]} +{"text": "å mع'S're're!!#$%'VEé\n‍", "tokens": 16, "pieces": ["å", " mع'S", "'re're", "!!#$%'", "VEé", "\n", "‍"]} +{"text": "٣٤٥٦३ ꟲe.…Zd​Dž\n㍿ ß́٣٤٥٦'re ㋿'㋿ ś\nḍ̇", "tokens": 46, "pieces": ["٣٤٥", "٦३", " ꟲe", ".", "…Zd", "​Dž", "\n", "㍿", " ß́", "٣٤٥", "٦", "'re", " ", "㋿'㋿", " <", "META", "_START", ">ś", "\n", "ḍ̇"]} +{"text": " \n\t'MZ", "tokens": 7, "pieces": ["", " \n", "\t", "'MZ"]} +{"text": "ß٣٤٥٦㍿ ‍'s><㍿ع's漢
 İꟲ­\t!漢‍'Re \nß'VE", "tokens": 36, "pieces": ["ß", "٣٤٥", "٦", "㍿", " ", " ‍'", "s", "><㍿", "ع's", "漢", "
 ", " İꟲ", "­", "\t", "!漢", "‍'", "Re", " \n", "ß'VE"]} +{"text": "İ३", "tokens": 2, "pieces": ["İ", "३"]} +{"text": " \ń字aİ!'re \r\n\r\né-\rmm,\u000b'S\r\n\u000b​!½'D\"'ſꟲs३\n0​ع ꟲEOTé🙂", "tokens": 45, "pieces": [" \n", "́字a", "İ", "!'", "re", " \r\n\r\n", "é", "-\r", "mm", ",", "\u000b", "'S", "\r\n", "\u000b", "​!", "½", "'D", "\"'", "ſꟲs", "३", "\n", "0", "​ع", " ꟲEOTé", "🙂"]} +{"text": "३'D ' åZ12345678ع'S\"३ḍ̇🙂>\tḍ̇Za\"å \n're-é‍𐞁", "tokens": 39, "pieces": ["३", "'D", " ", "'", " å", "Z", "", "123", "456", "78", "ع'S", "\"", "३", "ḍ̇", "🙂>", "\tḍ̇", "Za", "\"å", " \n", "'re", "-é", "‍𐞁"]} +{"text": "😀🏽İⅣ½…ꟲm ꟲ", "tokens": 20, "pieces": ["😀🏽", "İ", "Ⅳ½", "…ꟲm", " ", "ꟲ"]} +{"text": "🙂0EOTḍ̇漢(,\réꟲ٣٤٥٦éA<<|fim_prefix|> \n\r\n\r\n!!", "tokens": 30, "pieces": ["🙂", "0", "EOTḍ̇漢", "(,\r", "éꟲ", "٣٤٥", "٦", "é", "A", "<<|", "fim", "_prefix", "|>", " \n\r\n\r\n", "!!"]} +{"text": "…é(٣٤٥٦é<|endoftext|>‍>fiſ\r\n0
('D<'D!!!!s're!!🙂\r\n'ſ'VE½३\r\n\r\né!'VE", "tokens": 44, "pieces": ["…é", "(", "٣٤٥", "٦", "é", "<|", "endoftext", "|>‍>", "fiſ", "\r\n", "0", "
", "('", "D", "<'", "D", "!!!!", "s're", "!!🙂\r\n", "'ſ'VE", "½३", "\r\n\r\n", "é", "!'", "VE"]} +{"text": "d!㍿<|endoftext|>‍12345678d­​漢😀🏽é<|fim_prefix|>  ſs\"ḍ̇‍", "tokens": 41, "pieces": ["d", "!㍿<|", "endoftext", "|>‍", "123", "456", "78", "d", "­​", "漢", "😀🏽", "é", "<|", "fim", "_prefix", "|>", "  ", " ſs", "\"ḍ̇", "‍"]} +{"text": "㋿​<|endoftext|>½'ReꟲEOTs", "tokens": 19, "pieces": ["㋿​<|", "endoftext", "|>", "½", "'Reꟲ", "EOTs"]} +{"text": "㍿.ZDž𐞁'VE…½", "tokens": 16, "pieces": ["㍿.", "ZDž𐞁'VE", "…", "½"]} +{"text": "'SAd­EOT😀🏽ع!a/b,", "tokens": 14, "pieces": ["'SAd", "­EOT", "😀🏽", "ع", "!a", "/b", ","]} +{"text": "İ'Sé㍿…0ſ'M", "tokens": 11, "pieces": ["İ'S", "é", "㍿", "…", "0", "ſ'M"]} +{"text": "㍿İ ­🙂Ⅳ…!!('s/\r\n½٣٤٥٦𐞁字", "tokens": 25, "pieces": ["㍿İ", " ", "­🙂", "Ⅳ", "…", "!!('", "s", "/\r\n", "½٣٤", "٥٦", "𐞁字"]} +{"text": "#$%EOT!mİſ!!aB́😀🏽/🙂>aB\r字𐞁aa", "tokens": 27, "pieces": ["#$%", "EOT", "!m", "İſ", "!!", "a", "B́", "😀🏽/🙂>", "a", "B", "\r", "字𐞁aa"]} +{"text": "HTTPServer's,", "tokens": 4, "pieces": ["HTTPServer's", ","]} +{"text": "\u000b9-ßß'T३㍿\n/𐞁camelCaset'A/ \n
😀🏽é'Re 0Z😀🏽漢字Z'M0\r\n'T\u000b‍<😀🏽", "tokens": 49, "pieces": ["\u000b", "9", "-ßß'T", "३", "㍿\n/", "𐞁camel", "Caset", "'A", "/", " \n", "
", "😀🏽", "é'Re", " ", "0", "Z", "😀🏽", "漢字", "Z'M", "0", "\r\n", "'T", "\u000b", "‍<😀🏽"]} +{"text": "\r\n\r\n \n ३Ze", "tokens": 5, "pieces": ["\r\n\r\n \n", " ", "३", "Ze"]} +{"text": "eAABC<\r#$%\r \n EOTꟲ0aBᵃ字< ㍿
<|endoftext|>👍🏽ḍ̇!!fiaİ\r\n\r\nEOTᵃaBABC\r\n\r\n#$%'retBßHTTPServer", "tokens": 61, "pieces": ["e", "AABC", "<\r", "#$%\r", " \n", " EOTꟲ", "0", "a", "Bᵃ字", "<", " ", "㍿", "
", "<|", "endoftext", "|>👍🏽", "ḍ̇", "!!", "fia", "İ", "\r\n\r\n", "EOTᵃa", "BABC", "\r\n\r\n", "#$%'", "ret", "Bß", "HTTPServer"]} +{"text": "ꟲ٣٤٥٦t", "tokens": 8, "pieces": ["ꟲ", "٣٤٥", "٦", "t"]} +{"text": "́åm'll12345678👍🏽ḍ̇12345678́fi字HTTPServer mDžungla\r\naBABC12345678٣٤٥٦/-'Sé字́<|endoftext|>Z#$%<|fim_prefix|>", "tokens": 61, "pieces": ["́åm'll", "123", "456", "78", "👍🏽", "ḍ̇", "123", "456", "78", "́fi字", "HTTPServer", " m", "Džungla", "\r\n", "a", "BABC", "123", "456", "78٣", "٤٥٦", "/-'", "Sé字́", "<|", "endoftext", "|>", "Z", "#$%<|", "fim", "_prefix", "|>"]} +{"text": "'re\t \n!! \nå…!!ḍ̇<|fim_prefix|>\r\n\r\naB'Mds\n/>\t …ع<ſꟲiOS𐞁camelCase/\r\n٣٤٥٦>,'  camelCaseDžunglaé😀🏽", "tokens": 61, "pieces": ["'re", "\t \n", "!!", " \n", "å", "…", "!!", "ḍ̇", "<|", "fim", "_prefix", "|>\r\n\r\n", "a", "B'M", "ds", "\n", "/>", "\t ", "…ع", "<ſꟲi", "OS𐞁camel", "Case", "/\r\n", "٣٤٥", "٦", ">,'", " ", " camel", "Case", "Džunglaé", "😀🏽"]} +{"text": "camelCase३字iOS/!!-Aſß-!!'ResDžunglaDž​́<\t", "tokens": 26, "pieces": ["camel", "Case", "३", "字i", "OS", "/!!-", "Aſß", "-!!'", "Res", "Džungla", "Dž", "​́", "<", "\t"]} +{"text": "fiiOS…'re३HTTPServer\n­'é", "tokens": 14, "pieces": ["fii", "OS", "…", "'re", "३", "HTTPServer", "\n", "­'", "é"]} +{"text": "'re\n \nſ!!'re \n 
Džungla'ſ-'Tå maB", "tokens": 24, "pieces": ["'re", "\n \n", "ſ", "!!'", "re", "", " \n", " ", "
Džungla'ſ", "-'", "Tå", " ", " ma", "B"]} +{"text": "aB\r/.㍿字'SDžungla漢(''Re\u000b/a/bDžEOTe<|fim_prefix|>a/b12345678<|fim_prefix|>\r\na/b㍿<\r\n\r\n\n\n/\t#$%'T'<|endoftext|>é㍿\r\n ", "tokens": 72, "pieces": ["a", "B", "\r", "/.㍿", "字'S", "Džungla漢", "(''", "Re", "\u000b", "/a", "/b", "DžEOTe", "<|", "fim", "_prefix", "|>", "a", "/b", "123", "456", "78", "<|", "fim", "_prefix", "|>\r\n", "a", "/b", "㍿<\r\n\r\n\n\n/", "\t", "#$%'", "T", "'<|", "endoftext", "|>", "é", "㍿\r\n", " "]} +{"text": "ḍ̇a٣٤٥٦\ta/b,,iOS/\r\n😀🏽Z0/\r\nİDžé,<|endoftext|>e'D\" \n 
३ꟲ", "tokens": 63, "pieces": ["ḍ̇a", "٣٤٥", "٦", "\ta", "/b", ",,", "i", "OS", "/\r\n", "😀🏽", "Z", "0", "/\r\n", "İDžé", ",<", "a", "/b", "‍\n", "\t", "㍿!!", "s", "Ab", "
", "!!", "t", "<<|", "endoftext", "|><|", "endoftext", "|>", "e'D", "\"", " \n", " ", "
", "३", "ꟲ"]} +{"text": "㍿A<.camelCaseİ\r\n\r\n éZ/\r\n\r\n\r\nع'll½\u000ba/b\naDžungla<|endoftext|>'reAb漢\n/(HTTPServer  
<|fim_prefix|>'re, ", "tokens": 51, "pieces": ["㍿A", "<.", "camel", "Case", "İ", "\r\n\r\n", " é", "Z", "/\r\n\r\n\r\n", "ع'll", "½", "\u000ba", "/b", "\n", "a", "Džungla", "<|", "endoftext", "|>'", "re", "Ab漢", "\n", "/(", "HTTPServer", "  ", "
", "<|", "fim", "_prefix", "|>'", "re", ",", " "]} +{"text": "<|endoftext|>Ⅳ \n å字\td're漢😀🏽ᵃ#$%ᵃ\r\nع<9
", "tokens": 32, "pieces": ["<|", "endoftext", "|>", "Ⅳ", " \n", " å字", "\td're", "漢", "😀🏽", "ᵃ", "#$%", "ᵃ", "\r\n", "ع", "<", "9", "
"]} +{"text": "Džunglaꟲ\n<|endoftext|>ß㍿Džungla\"0, 👍🏽ᵃ12345678\na/bḍ̇\u000bdİع'T/\r\ń🙂👍🏽dA Z🙂㋿m!", "tokens": 66, "pieces": ["Džunglaꟲ", "\n", "<|", "endoftext", "|>", "ß", "㍿Džungla", "\"", "0", ",<", "EOT", ">", " ", "👍🏽", "ᵃ", "123", "456", "78", "\n", "a", "/bḍ̇", "\u000bd", "İع'T", "/\r\n", "́", "🙂👍🏽", "d", "A", " Z", "🙂㋿", "m", "!"]} +{"text": "ḍ̇", "tokens": 3, "pieces": ["ḍ̇"]} +{"text": "👍🏽>\u000b>😀🏽é
Dž,'M漢'D\n\"ꟲ Z\r<|fim_prefix|>'ſⅣ Džunglaḍ̇½", "tokens": 44, "pieces": ["👍🏽>", "\u000b", ">😀🏽", "é", "
Dž", ",'", "M漢'D", "\n", "\"ꟲ", " ", " Z", "\r", "<|", "fim", "_prefix", "|>'", "ſ", "Ⅳ", " Džunglaḍ̇", "½"]} +{"text": "camelCased-m٣٤٥٦AHTTPServer", "tokens": 15, "pieces": ["camel", "Cased", "-m", "٣٤٥", "٦", "A", "HTTPServer"]} +{"text": "㋿漢'M'VEiOS<|endoftext|>", "tokens": 16, "pieces": ["㋿漢'M", "'VEi", "OS", "<|", "endoftext", "|>"]} +{"text": "/,HTTPServeŕ ḍ̇iOS٣٤٥٦'DHTTPServer'M\r\n're'reꟲ👍🏽'reaBB<|endoftext|>\u000be​ſ'måꟲſſd👍🏽m½", "tokens": 55, "pieces": ["/,", "HTTPServeŕ", " ḍ̇i", "OS", "٣٤٥", "٦", "'DHTTPServer'M", "\r\n", "'re're", "ꟲ", "👍🏽'", "rea", "BB", "<|", "endoftext", "|>", "\u000be", "​ſ'm", "åꟲſſd", "👍🏽", "m", "½"]} +{"text": "Abḍ̇\r\n", "tokens": 5, "pieces": ["Abḍ̇", "\r\n"]} +{"text": "ḍ̇\r\nİ​m#$% ३­HTTPServeré'Re", "tokens": 16, "pieces": ["ḍ̇", "\r\n", "İ", "​m", "#$%", " ", "३", "­HTTPServeré'Re"]} +{"text": "​३0.‍ZⅣ'Td'sßſs㋿ſ㋿ABC🙂𐞁\"", "tokens": 34, "pieces": ["​", "३0", ".‍", "Z", "Ⅳ", "'Td's", "ßſs", "㋿ſ", "㋿ABC", "🙂𐞁", "\"<", "EOT", ">"]} +{"text": "s12345678!'ll \n 'ReⅣs B(㍿  'Tß'DABCAb>HTTPServer\r#$%\r\nDžunglaEOT\r\n\r\n12345678字", "tokens": 44, "pieces": ["s", "123", "456", "78", "!'", "ll", " \n", " '", "Re", "Ⅳ", "s", " B", "(㍿", " ", " ", "'Tß'D", "ABCAb", ">HTTPServer", "\r", "#$%\r\n", "Džungla", "EOT", "\r\n\r\n", "123", "456", "78", "字"]} +{"text": "A'D's /ḍ̇ABCEOTſḍ̇s", "tokens": 15, "pieces": ["A'D", "'s", " /", "ḍ̇", "ABCEOTſḍ̇s"]} +{"text": "<|endoftext|>Ⅳm're0a/b
méé's\n/Dž'Dd\n漢漢<|endoftext|> é \naB🙂३'ReB‍m'ſ㍿.camelCase", "tokens": 59, "pieces": ["<|", "endoftext", "|>", "Ⅳ", "m're", "0", "a", "/b", "
m", "éé's", "\n", "/Dž'D", "d", "\n", "漢漢", "<|", "endoftext", "|>", " é", " \n", "a", "B", "🙂", "३", "'Re", "B", "‍m'ſ", "㍿.", "camel", "Case"]} +{"text": "éAb½dḍ̇d'T\rꟲᵃ>HTTPServera/b­'VEé३<…\n/<|endoftext|>A<|fim_prefix|>", "tokens": 46, "pieces": ["é", "Ab", "½", "dḍ̇", "d'T", "\r", "ꟲᵃ", ">HTTPServera", "/b", "­'", "VEé", "३", "<", "…\n", "/<|", "endoftext", "|>", "A", "<|", "fim", "_prefix", "|>"]} +{"text": "㍿\ré ſéſ́ \r\n\r\n!ع(fi-ع
㋿camelCase#$%HTTPServerİ é…字 \n \"", "tokens": 39, "pieces": ["㍿\r", "é", " ſéſ́", " \r\n\r\n", "!ع", "(fi", "-ع", "
", "㋿camel", "Case", "#$%", "HTTPServer", "İ", " é", "", "…字", " \n", " \""]} +{"text": "A Bé ꟲBAb,de🙂
", "tokens": 13, "pieces": ["A", " Bé", " ", " ꟲBAb", ",de", "🙂", "
"]} +{"text": "\u000b<𐞁\"ᵃå,'T㍿㋿ABC\n/<|endoftext|>​'T👍🏽ꟲꟲİ'VE'\r\n\r\n,
a/b", "tokens": 52, "pieces": ["\u000b", "<𐞁", "\"ᵃå", ",'", "T", "㍿㋿", "ABC", "\n", "/<|", "endoftext", "|><", "META", "_START", ">​'", "T", "👍🏽", "ꟲꟲ", "İ'VE", "'\r\n\r\n", ",", "
a", "/b"]} +{"text": "
'll/㍿ع<|fim_prefix|>/\r\niOS\n/…\r\n\r\n éå­ḍ̇'SDžungla𐞁9", "tokens": 43, "pieces": ["
", "'ll", "/㍿", "ع", "<|", "fim", "_prefix", "|><", "META", "_START", ">/\r\n", "i", "OS", "\n", "/", "…\r\n\r\n", " éå", "­ḍ̇'S", "Džungla𐞁", "9"]} +{"text": "a'll9Z ", "tokens": 5, "pieces": ["a'll", "9", "Z", " "]} +{"text": "Dž'M字fi.'reİ'ſ  #$%", "tokens": 14, "pieces": ["Dž'M", "字fi", ".'", "re", "İ'ſ", "  ", " #$%"]} +{"text": "/ß'T½㍿<\rcamelCase<|endoftext|>'re", "tokens": 19, "pieces": ["/ß'T", "½", "㍿<\r", "camel", "Case", "<|", "endoftext", "|>'", "re"]} +{"text": "ß\u000ba/b<|fim_prefix|>½'T…!İ(𐞁漢‍\ta👍🏽'M\r<|endoftext|>t", "tokens": 38, "pieces": ["ß", "\u000ba", "/b", "<|", "fim", "_prefix", "|>", "½", "'T", "…", "!İ", "(𐞁漢", "‍", "\ta", "👍🏽'", "M", "\r", "<|", "endoftext", "|>", "t"]} +{"text": "👍🏽\n/!!iOS12345678", "tokens": 14, "pieces": ["👍🏽\n/", "!!", "i", "OS", "123", "456", "78", ""]} +{"text": "'VE 'll🙂ḍ̇\rs", "tokens": 15, "pieces": ["'VE", " ", " '", "ll", "🙂ḍ̇", "\r", "s", ""]} +{"text": "camelCase½Z…HTTPServer-'llſB's<|fim_prefix|>\n/\r\n \n fifi'.٣٤٥٦at'ſ", "tokens": 31, "pieces": ["camel", "Case", "½", "Z", "…HTTPServer", "-'", "llſ", "B's", "<|", "fim", "_prefix", "|>\n/\r\n", " \n", " fifi", "'.", "٣٤٥", "٦", "at'ſ"]} +{"text": "e'S#$%👍🏽iOSⅣ😀🏽/\r\n­'DžunglaEOT#$%'DiOS/#$%sB\r'res", "tokens": 38, "pieces": ["e'S", "#$%👍🏽", "i", "OS", "Ⅳ", "😀🏽/\r\n", "­'", "Džungla", "EOT", "#$%'", "Di", "OS", "/#$%", "s", "B", "\r", "'res"]} +{"text": " 9㋿a/b㍿'Re३字\n/#$%B
<|fim_prefix|>…a", "/b", "㍿'", "Re", "३", "字", "\n", "/#$%", "B", "
", "<|", "fim", "_prefix", "|>", "…", "mع㍿EOT\n<|fim_prefix|> \nEOT\r\nDžungla\"\r\nⅣ😀🏽A'Re.<'ll<|endoftext|> eDžungla👍🏽漢12345678B\n'Re e​ \n ", "tokens": 62, "pieces": ["mع", "㍿EOT", "\n", "<|", "fim", "_prefix", "|>", " \n", "EOT", "\r\n", "Džungla", "\"\r\n", "Ⅳ", "😀🏽", "A'Re", ".<'", "ll", "<|", "endoftext", "|>", " e", "Džungla", "👍🏽", "漢", "123", "456", "78", "B", "\n", "'Re", " e", "​", " \n", " "]} +{"text": "́٣٤٥٦ \n ‍\u000béaBiOS", "tokens": 12, "pieces": ["́", "٣٤٥", "٦", " \n", " ‍", "\u000béa", "Bi", "OS"]} +{"text": "t🙂.#$%> camelCase AHTTPServerté…!Ⅳ𐞁12345678Zsm!‍ ⅣB\n\n/", "tokens": 39, "pieces": ["t", "🙂.#$%>", " ", " camel", "Case", " ", " AHTTPServerté", "…", "!", "Ⅳ", "𐞁", "123", "456", "78", "Zsm", "!‍", " ", " ", "Ⅳ", "B", "\n\n", "/"]} +{"text": "́e-👍🏽Ab'Re'ſ'Re.𐞁­𐞁ß's//\r\n٣٤٥٦(🙂0/ \n \r‍
", "tokens": 41, "pieces": ["́e", "-👍🏽<", "EOT", ">Ab'Re", "'ſ'Re", ".𐞁", "­𐞁ß's", "//\r\n", "٣٤٥", "٦", "(🙂", "0", "/", " \n \r", "‍", "
"]} +{"text": "d<'D", "tokens": 3, "pieces": ["d", "<'", "D"]} +{"text": "fi12345678ABC/\r\nᵃAb#$%İ12345678BiOS\n/
/½́
A <|fim_prefix|>́", "tokens": 33, "pieces": ["fi", "123", "456", "78", "ABC", "/\r\n", "ᵃAb", "#$%", "İ", "123", "456", "78", "Bi", "OS", "\n", "/", "
", "/", "½", "́", "
A", " <|", "fim", "_prefix", "|>́"]} +{"text": "Ⅳ́0é'́\r\nDž🙂​\r\n\r\n<|fim_prefix|>\n👍🏽​", "tokens": 30, "pieces": ["Ⅳ", "́", "", "0", "é", "'́", "\r\n", "Dž", "🙂<", "EOT", ">​\r\n\r\n", "<|", "fim", "_prefix", "|>\n", "👍🏽​"]} +{"text": " \n é/\n\r9'M
 \n ㍿ddḍ̇é\u000bع\r\n\r\n", "tokens": 21, "pieces": [" \n", " é", "/\n\r", "9", "'M", "
 \n", " ㍿", "ddḍ̇é", "\u000bع", "\r\n\r\n"]} +{"text": "'VE<|fim_prefix|>½t>\u000bİd­é㍿\n/漢å", "tokens": 28, "pieces": ["'VE", "<|", "fim", "_prefix", "|>", "½", "t", ">", "\u000bİd", "­é", "㍿\n/", "漢å"]} +{"text": "\r\n\r\nᵃcamelCaseZ12345678'D'Tm字㋿å\r\n\r\n'Re!!'ll\r\n12345678३🙂\nABCA漢-​㍿\n#$%HTTPServer!!'Tß​<|fim_prefix|>漢å!!/\r\n", "tokens": 63, "pieces": ["\r\n\r\n", "ᵃcamel", "Case", "Z", "123", "456", "78", "'D'T", "m字", "㋿å", "\r\n\r\n", "'Re", "!!'", "ll", "\r\n", "123", "456", "78३", "🙂\n", "ABCA漢", "-​㍿\n", "#$%", "HTTPServer", "!!'", "Tß", "​<|", "fim", "_prefix", "|>", "漢å", "!!/\r\n"]} +{"text": "'s\n/", "tokens": 3, "pieces": ["'s", "\n", "/"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'re漢", "tokens": 2, "pieces": ["'re漢"]} +{"text": "ſ'a9Džungla", "tokens": 7, "pieces": ["ſ", "'a", "9", "Džungla"]} +{"text": "\n/\r\nm/'ſ​d'M٣٤٥٦ABC d٣٤٥٦ع\r!\r!ᵃ𐞁 (Dž'\r\n\r\nſ> ​é ", "tokens": 44, "pieces": ["\n", "/\r\n", "m", "/'", "ſ", "​", "d'M", "٣٤٥", "٦", "ABC", " d", "٣٤٥", "٦", "ع", "\r", "!\r", "!ᵃ𐞁", " ", "(Dž", "'\r\n\r\n", "ſ", ">", " ", "​é", " "]} +{"text": ",'reZꟲᵃ/", "tokens": 10, "pieces": [",'", "re", "Zꟲᵃ", "/"]} +{"text": "\rⅣ'ſDžHTTPServer'Reå12345678㋿9ſ㋿ABC🙂BåaDžungla‍!camelCase12345678<|endoftext|>/' ㋿\"'Ts", "tokens": 53, "pieces": ["\r", "Ⅳ", "'ſ", "DžHTTPServer'Re", "å", "123", "456", "78", "㋿", "9", "ſ", "㋿ABC", "🙂Båa", "Džungla", "‍!", "camel", "Case", "123", "456", "78", "<|", "endoftext", "|>/'", " ", "㋿\"'", "Ts"]} +{"text": "\n'a/b‍<|fim_prefix|>½字EOT३", "tokens": 15, "pieces": ["\n", "'a", "/b", "‍<|", "fim", "_prefix", "|>", "½", "字", "EOT", "३"]} +{"text": " \n ع\r\n\r\nå😀🏽#$%  'M‍İ ㍿\r\n​'S­/\r\ncamelCase'S>…#$%a/b\n/Bé", "tokens": 42, "pieces": [" \n", " ع", "\r\n\r\n", "å", "😀🏽#$%", " ", " ", "'M", "‍İ", " ", " ㍿\r\n", "​'", "S", "­/\r\n", "camel", "Case'S", ">", "…", "#$%", "a", "/b", "\n", "/Bé"]} +{"text": " (#$%\"é.camelCase(-t \r\n\r\n>㍿\n/\r\n \n 's㋿a/b\t<|fim_prefix|>٣٤٥٦ᵃt\t > ", "tokens": 44, "pieces": [" ", " (#$%\"", "é", ".camel", "Case", "(-", "t", " \r\n\r\n", ">㍿\n/\r\n", " \n", " '", "s", "㋿a", "/b", "\t", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "ᵃt", "\t", " >", " "]} +{"text": "Z㋿EOT㋿'sdZⅣ're(dꟲ㍿'½عß", "tokens": 26, "pieces": ["Z", "㋿EOT", "㋿'", "sd", "Z", "Ⅳ", "'re", "(dꟲ", "㍿'", "½", "عß"]} +{"text": "ABC/\r\na👍🏽ᵃcamelCase½mß́camelCase👍🏽t \n<|fim_prefix|>a 12345678/漢Za/b́m\r>aB
m ́'M'ſ
 >…\u000b'll", "tokens": 56, "pieces": ["ABC", "/\r\n", "a", "👍🏽", "ᵃcamel", "Case", "½", "mß́camel", "Case", "👍🏽", "t", " \n", "<|", "fim", "_prefix", "|>", "a", " ", "123", "456", "78", "/漢Za", "/b́m", "\r", ">a", "B", "
m", " ́'M", "'ſ", "
", " ", ">", "…", "\u000b", "'ll"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r\n३㋿é㋿", "tokens": 10, "pieces": ["\r\n", "३", "㋿é", "㋿"]} +{"text": " \r\n\r\n!ḍ̇ABC'VEå𐞁<|fim_prefix|>'s٣٤٥٦ꟲé<|endoftext|>a/btHTTPServeŕ'VEḍ̇!!\"t\t \n ,camelCase'VE>camelCase\n👍🏽", "tokens": 72, "pieces": [" \r\n\r\n", "!ḍ̇", "ABC'VE", "å𐞁", "<|", "fim", "_prefix", "|>'", "s", "٣٤٥", "٦", "ꟲ", "é", "<|", "endoftext", "|>", "a", "/bt", "HTTPServeŕ'VE", "ḍ̇", "!!\"<", "EOT", ">t", "\t \n", " ,", "camel", "Case'VE", ">camel", "Case", "\n", "👍🏽"]} +{"text": ".㋿!a \n !aBḍ̇‍👍🏽12345678Dž", "tokens": 25, "pieces": [".<", "EOT", ">㋿!", "a", " \n", " !", "a", "Bḍ̇", "‍👍🏽", "123", "456", "78", "Dž"]} +{"text": "عZ \n\nİ\u000b. …'s/İ,👍🏽", "tokens": 16, "pieces": ["ع", "Z", " \n\n", "İ", "\u000b", ".", " ", "…", "'s", "/İ", ",👍🏽"]} +{"text": "s​", "tokens": 2, "pieces": ["s", "​"]} +{"text": "३EOT٣٤٥٦d\t're0iOSABCAbſaB\r\n\r\nİ\t!<|endoftext|>㍿ ع'llꟲaB३\"aB​.ᵃ9 t", "tokens": 52, "pieces": ["३", "EOT", "٣٤٥", "٦", "d", "\t", "'re", "0", "i", "OSABCAbſa", "B", "\r\n\r\n", "İ", "\t", "!<|", "endoftext", "|>㍿<", "EOT", ">", " ع'll", "ꟲa", "B", "३", "\"a", "B", "​.", "ᵃ", "9", " t"]} +{"text": "…‍ß\t字​…aB<|endoftext|> 'ſعAb /😀🏽", "tokens": 24, "pieces": ["漢", "<|", "endoftext", "|><|", "endoftext", "|>", " ", "'ſع", "Ab", " ", "/😀🏽"]} +{"text": "iOS\r\n\r\n 'Džungla<|endoftext|>'D
EOTſ", "tokens": 21, "pieces": ["i", "OS", "\r\n\r\n", " ", "'Džungla", "<|", "endoftext", "|>'", "D", "
EOTſ"]} +{"text": "!😀🏽字'M‍Dž0 12345678fi½…a", "tokens": 19, "pieces": ["!😀🏽", "字'M", "‍Dž", "0", " ", "123", "456", "78", "fi", "½", "…a"]} +{"text": "'S'Mm'\"#$%'re\t-👍🏽㍿\r\n\r\n.Z'ſEOT AbAb<|fim_prefix|>/. \n 字iOS‍ḍ̇字9", "tokens": 45, "pieces": ["'S'M", "m", "'\"#$%'", "re", "\t", "-<", "META", "_START", ">👍🏽㍿\r\n\r\n", ".Z'ſ", "EOT", " ", " Ab", "Ab", "<|", "fim", "_prefix", "|>/.", " \n", " 字i", "OS", "‍ḍ̇字", "9"]} +{"text": " 's\n/㋿'Re👍🏽/\r\n' fi'S/DžunglacamelCase \n's(́m<|fim_prefix|>ᵃ👍🏽\r,'Re'M­𐞁\r\n\r\n\"字­🙂e<|endoftext|>漢Z12345678", "tokens": 68, "pieces": [" ", "'s", "\n", "/㋿'", "Re", "👍🏽/\r\n", "'", " fi'S", "/Džunglacamel", "Case", " \n", "'s", "(́m", "<|", "fim", "_prefix", "|>", "ᵃ", "👍🏽\r", ",'", "Re'M", "­𐞁", "\r\n\r\n", "\"字", "­🙂", "e", "<|", "endoftext", "|>", "漢", "Z", "123", "456", "78"]} +{"text": "camelCase<ꟲ㍿\"ع😀🏽😀🏽", "tokens": 20, "pieces": ["camel", "Case", "<ꟲ", "㍿\"", "ع", "😀🏽😀🏽"]} +{"text": "é'D‍<|fim_prefix|>ḍ̇å½ſ​…ḍ̇'S'VE\r\n\r\n'll字a0ᵃ0Dž३0 \n ß \u000b𐞁ABCDž", "tokens": 50, "pieces": ["é'D", "‍<|", "fim", "_prefix", "|>", "ḍ̇å", "½", "ſ", "​", "…ḍ̇'S", "'VE", "\r\n\r\n", "'ll字a", "0", "ᵃ", "0", "Dž", "३0", " \n", " ß", " ", "\u000b𐞁", "ABCDž"]} +{"text": "!/\r\n‍s­Dža/b😀🏽", "tokens": 12, "pieces": ["!/\r\n", "‍s", "­Dža", "/b", "😀🏽"]} +{"text": "é's‍…٣٤٥٦t<|fim_prefix|>'VE٣٤٥٦,a/bDžungla\nAb/\r\n/\r\n\tacamelCase9éABC", "tokens": 39, "pieces": ["é's", "‍", "…", "٣٤٥", "٦", "t", "<|", "fim", "_prefix", "|>'", "VE", "٣٤٥", "٦", ",a", "/b", "Džungla", "\n", "Ab", "/\r\n/\r\n", "\tacamel", "Case", "9", "é", "ABC"]} +{"text": " ⅣéZİ­'ſå\r\n\r\nAbAb🙂٣٤٥٦'ſᵃꟲ😀🏽ᵃ/ 12345678/\r\nABC𐞁a𐞁ع字12345678😀🏽ع", "tokens": 65, "pieces": [" ", "Ⅳ", "é", "Zİ", "­'", "ſå", "\r\n\r\n", "Ab", "Ab", "🙂", "٣٤٥", "٦", "'ſᵃꟲ", "😀🏽", "ᵃ", "/<", "META", "_START", ">", " ", " ", "123", "456", "78", "/\r\n", "ABC𐞁a𐞁ع字", "123", "456", "78", "😀🏽", "ع", ""]} +{"text": ">'t", "tokens": 2, "pieces": [">'", "t"]} +{"text": "ꟲꟲe!aB­\r漢,é​éHTTPServerDžungla ‍ #$%< #$%ᵃ👍🏽/\r\nHTTPServer🙂 -9­DžZ𐞁", "tokens": 54, "pieces": ["ꟲꟲe", "!a", "B", "­\r", "漢", ",é", "​é", "HTTPServer", "Džungla", " ‍", " #$%<", " ", " #$%", "ᵃ", "👍🏽/\r\n", "HTTPServer", "🙂", " ", "-", "9", "­DžZ𐞁"]} +{"text": "!'s'>'Re9>३🙂Džunglaß字字aB\r<|endoftext|>ADžDž'll<|endoftext|>ᵃaB!<|fim_prefix|>ḍ̇́å \n٣٤٥٦", "tokens": 62, "pieces": ["!'", "s", "'>'", "Re", "9", ">", "३", "🙂Džunglaß字字a", "B", "\r", "<|", "endoftext", "|>", "ADžDž'll", "<|", "endoftext", "|>", "ᵃa", "B", "!<|", "fim", "_prefix", "|>", "ḍ̇́å", " \n", "٣٤٥", "٦"]} +{"text": "aⅣ<|endoftext|>åß㋿/ \n  \rß-\r\n! \nع'reſDžungla <|fim_prefix|>İ,‍字", "tokens": 45, "pieces": ["a", "Ⅳ", "<|", "endoftext", "|>", "åß", "㋿/<", "EOT", ">", " \n  \r", "ß", "-\r\n", "!", " \n", "ع're", "ſ", "Džungla", " ", "<|", "fim", "_prefix", "|>", "İ", ",‍", "字"]} +{"text": "'ſ", "tokens": 2, "pieces": ["'ſ"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ABCß/\n/>'aB㋿e0\nع'\n𐞁 \n😀🏽.é𐞁/\r\n é­ İ'll<|endoftext|>!", "tokens": 44, "pieces": ["ABCß", "/\n/", ">'", "a", "B", "㋿e", "0", "\n", "ع", "'\n", "𐞁", " \n", "😀🏽.", "é𐞁", "/\r\n", " é", "­", " İ'll", "<|", "endoftext", "|>!"]} +{"text": ".és٣٤٥٦İfi'D\n", "tokens": 10, "pieces": [".és", "٣٤٥", "٦", "İfi'D", "\n"]} +{"text": "/\r\n'a/b'ſZDž \n𐞁ꟲ'T<,,<|endoftext|>㍿'DABCꟲ👍🏽ᵃ\u000b३ \nA", "tokens": 49, "pieces": ["/\r\n", "'", "a", "/b'ſ", "ZDž", " \n", "𐞁ꟲ'T", "<,,<|", "endoftext", "|>㍿'", "DABCꟲ", "👍🏽", "ᵃ", "\u000b", "३", " \n", "A"]} +{"text": "😀🏽ꟲtAb/\r\n/\r\n३٣٤٥٦Ⅳ​>/EOTcamelCaseA٣٤٥٦", "tokens": 28, "pieces": ["😀🏽", "ꟲt", "Ab", "/\r\n/\r\n", "३٣٤", "٥٦Ⅳ", "​>/", "EOTcamel", "Case", "A", "٣٤٥", "٦"]} +{"text": "å \n å 字A㋿\ra/bß'Re!­\n/!㋿", "tokens": 24, "pieces": ["å", " \n", " å", " 字", "A", "㋿\r", "a", "/bß'Re", "!­\n/", "!㋿"]} +{"text": "a/bé9㍿عå'M👍🏽 iOS/ᵃ", "tokens": 23, "pieces": ["a", "/bé", "9", "㍿عå'M", "👍🏽", " i", "OS", "/<", "META", "_START", ">ᵃ"]} +{"text": "ß/\r\nt(‍'S!'TiOS0a/bİDž㋿ \n , å'EOT<|endoftext|>a/bm'\r½", "tokens": 51, "pieces": ["ß", "/\r\n", "t", "(‍'", "S", "!'", "Ti", "OS", "0", "a", "/b", "İDž", "㋿", " \n", " ,", " ", "å", "'EOT", "<|", "endoftext", "|>", "a", "/bm", "'\r", "½"]} +{"text": "'ſ'sß'SsABC😀🏽\t'T'll\r\n३'٣٤٥٦ḍ̇\r\n'T漢/İ12345678aBEOT", "tokens": 34, "pieces": ["'ſ's", "ß'S", "s", "ABC", "😀🏽", "\t", "'T'll", "\r\n", "३", "'", "٣٤٥", "٦", "ḍ̇", "\r\n", "'T漢", "/İ", "123", "456", "78", "a", "BEOT"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " 字0a 'VE\t'VE 𐞁'ſ\u000b𐞁​", "tokens": 22, "pieces": [" 字", "0", "a", " ", "'VE", "\t", "'VE", " 𐞁'ſ", "\u000b𐞁", "​"]} +{"text": "\r\n<|endoftext|>", "tokens": 8, "pieces": ["\r\n", "<|", "endoftext", "|>"]} +{"text": "fiEOT🙂.,aB\n/\n/ 🙂, <|fim_prefix|>'Mdſ camelCaseEOT🙂…👍🏽\u000b٣٤٥٦a!!EOT'Re👍🏽fiå'll<|endoftext|>‍m", "tokens": 56, "pieces": ["fi", "EOT", "🙂.,", "a", "B", "\n", "/\n/", " ", "🙂,", " <|", "fim", "_prefix", "|>'", "Mdſ", " camel", "Case", "EOT", "🙂", "…", "👍🏽", "\u000b", "٣٤٥", "٦", "a", "!!", "EOT'Re", "👍🏽", "fiå'll", "<|", "endoftext", "|>‍", "m"]} +{"text": "​12345678m!!éİéiOSABC'll >Ⅳmİ<|fim_prefix|>\r!Ab /camelCase're'M12345678½'M \n /\r\nm-'ſ HTTPServer🙂", "tokens": 50, "pieces": ["​", "123", "456", "78", "m", "!!", "é", "İéi", "OSABC'll", " ", ">", "Ⅳ", "m", "İ", "<|", "fim", "_prefix", "|>\r", "!Ab", " ", " /", "camel", "Case're", "'", "M", "123", "456", "78½", "'M", " \n", " /\r\n", "m", "-'", "ſ", " HTTPServer", "🙂"]} +{"text": "ABC're\r\n\r\n'TEOT'VEa/b(​字\r\n\t", "tokens": 15, "pieces": ["ABC're", "\r\n\r\n", "'TEOT'VE", "a", "/b", "(​", "字", "\r\n", "\t"]} +{"text": "㍿字EOTİfi'ſ(", "tokens": 11, "pieces": ["㍿字EOTİfi'ſ", "("]} +{"text": "( Džunglá​㋿İ'VE'!", "tokens": 20, "pieces": ["(<", "META", "_START", ">", " ", " Džunglá", "​㋿", "İ'VE", "'!"]} +{"text": " \nfi‍字é‍\"HTTPServer­-é'fi \n/\n're\u000b㋿d'Ree", "tokens": 29, "pieces": [" <", "EOT", ">", " \n", "fi", "‍字é", "‍\"", "HTTPServer", "­-", "é", "'fi", " \n", "/\n", "'re", "\u000b", "㋿d'Re", "e"]} +{"text": "/At'T㋿/\r\n
0HTTPServerEOT½ \tß‍𐞁\" ABC‍…'M'D/\r\nAb,​a/b'Mع-𐞁\t", "tokens": 48, "pieces": ["/At'T", "㋿/\r\n", "
", "0", "HTTPServer", "EOT", "", "½", " ", "\tß", "‍𐞁", "\"", " ", " ABC", "‍", "…", "'M'D", "/\r\n", "Ab", ",​", "a", "/b'M", "ع", "-𐞁", "\t"]} +{"text": "'ſAé\r\n\r\na'S👍🏽<|fim_prefix|>‍漢​'ll", "tokens": 25, "pieces": ["'ſ", "Aé", "\r\n\r\n", "a'S", "👍🏽<|", "fim", "_prefix", "|>‍<", "EOT", ">漢", "​'", "ll"]} +{"text": " ㍿>'VE\t'Té'VEZعDžsA\t>…<Džungla>", "tokens": 27, "pieces": [" ㍿>'", "VE", "\t", "'Té'VE", "ZعDžs", "A", "\t", ">", "…", "<Džungla", ">"]} +{"text": "HTTPServer('ſHTTPServer३camelCaseİBDž\rå>tA'M#$%/å٣٤٥٦9ꟲé ta/\r\n<|fim_prefix|>s.ꟲ\r\n/", "tokens": 54, "pieces": ["HTTPServer", "('", "ſ", "HTTPServer", "३", "camel", "Case", "İBDž", "\r", "å", ">t", "A'M", "#$%/", "å", "٣٤٥", "٦9", "ꟲé", " ta", "/\r\n", "<|", "fim", "_prefix", "|>", "s", ".ꟲ", "\r\n", "/"]} +{"text": "a\n/Ab", "tokens": 41, "pieces": ["a", "\n", "/Ab", ""]} +{"text": "
/(em'T's(漢'Re  \n e 're\n/Z'ſå\r\n\r\n#$%#$%٣٤٥٦'Re
camelCase­'VE🙂", "tokens": 39, "pieces": ["
", "/(", "em'T", "'s", "(漢'Re", "  \n", " e", " ", "'re", "\n", "/Z'ſ", "å", "\r\n\r\n", "#$%#$%", "٣٤٥", "٦", "'Re", "
camel", "Case", "­'", "VE", "🙂"]} +{"text": "<|fim_prefix|>0 <|fim_prefix|>Ⅳ́ZcamelCaseⅣ‍'Re-<|endoftext|> ſ0a/bmᵃ", "tokens": 42, "pieces": ["<|", "fim", "_prefix", "|>", "0", " <|", "fim", "_prefix", "|>", "Ⅳ", "́Zcamel", "Case", "Ⅳ", "‍'", "Re", "-<|", "endoftext", "|>", " ", " ſ", "0", "a", "/bmᵃ"]} +{"text": "字éİ#$%ſZꟲᵃa iOS,camelCase👍🏽a​a0a0\u000bEOT😀🏽é", "tokens": 37, "pieces": ["字é", "İ", "#$%", "ſ", "Zꟲᵃa", " ", " i", "OS", ",camel", "Case", "👍🏽", "a", "​a", "0", "a", "0", "\u000bEOT", "😀🏽", "é"]} +{"text": "\u000ba/bAdeᵃᵃ(漢½
fi٣٤٥٦a/b", "tokens": 44, "pieces": ["\u000ba", "/b", "Adeᵃᵃ", "(漢", "", "½", "
fi", "٣٤٥", "٦", "a", "/b"]} +{"text": "İ'VEİdDžungla​!!Abd\r'll'S'D'VE\n/", "tokens": 21, "pieces": ["İ'VE", "İd", "Džungla", "​!!", "Abd", "\r", "'ll'S", "'D'VE", "\n", "/"]} +{"text": "Ab½#$%'ll­(.𐞁३\r'll­😀🏽-d'D३½ع漢­'😀🏽
ꟲ./\r\n \naB\n\n0'Re \nd<|endoftext|>", "tokens": 55, "pieces": ["Ab", "½", "#$%'", "ll", "­(.<", "EOT", ">𐞁", "३", "\r", "'ll", "­😀🏽-", "d'D", "३½", "ع漢", "­'😀🏽", "
ꟲ", "./\r\n", " \n", "a", "B", "\n\n", "0", "'Re", " \n", "d", "<|", "endoftext", "|>"]} +{"text": "fi😀🏽\r\n\r\n<ع #$%\u000b \n 
", "tokens": 16, "pieces": ["fi", "😀🏽\r\n\r\n", "<", "ع", " ", "#$%", "\u000b \n", " 
"]} +{"text": "#$%'Me.Džunglaa'T٣٤٥٦a…fi\r\n\r\na'Tḍ̇字éAbAbå'Té're're-'\r\n\r\nscamelCasefi", "tokens": 41, "pieces": ["#$%'", "Me", ".Džunglaa'T", "٣٤٥", "٦", "a", "…fi", "\r\n\r\n", "a'T", "ḍ̇字é", "Ab", "Abå'T", "é're", "'re", "-'\r\n\r\n", "scamel", "Casefi"]} +{"text": "#$% \nḍ̇ZsⅣ<|endoftext|>Ⅳ㍿😀🏽", "tokens": 25, "pieces": ["#$%", " \n", "ḍ̇", "Zs", "Ⅳ", "<|", "endoftext", "|>", "Ⅳ", "㍿😀🏽"]} +{"text": "a/b­😀🏽'TAع㍿>'re٣٤٥٦a/b\"𐞁a/b0 😀🏽ß'Re<|fim_prefix|>", "tokens": 40, "pieces": ["a", "/b", "­😀🏽'", "TAع", "㍿>'", "re", "٣٤٥", "٦", "a", "/b", "\"𐞁a", "/b", "0", " 😀🏽", "ß'Re", "<|", "fim", "_prefix", "|>"]} +{"text": "sB,'D", "tokens": 4, "pieces": ["s", "B", ",'", "D"]} +{"text": "dt漢m#$%B𐞁fiḍ̇٣٤٥٦\n'sꟲDžunglaa👍🏽'VEſ字A㍿ſſB\u000be>\r\n'DİſtiOS ᵃcamelCase>iOS", "tokens": 63, "pieces": ["dt漢m", "#$%<", "EOT", ">B𐞁fiḍ̇", "٣٤٥", "٦", "\n", "'sꟲ", "Džunglaa", "👍🏽'", "VEſ字", "A", "㍿ſſ", "B", "\u000be", ">\r\n", "'Dİſti", "OS", " ", " ᵃcamel", "Case", ">i", "OS"]} +{"text": "'T-A😀🏽ſ'TEOT \n \r\n\r\n\n/\nİꟲ漢>fimaBſ३", "tokens": 57, "pieces": ["'T", "-A", "😀🏽", "ſ'T", "EOT", "", " \n \r\n\r\n\n", "/\n", "İꟲ漢", ">fima", "B", "", "ſ", "३"]} +{"text": "d\r\nꟲⅣ", "tokens": 7, "pieces": ["d", "\r\n", "ꟲ", "Ⅳ"]} +{"text": "'MiOS(\r\n\r\na/b'Mḍ̇éé\nİ‍fi\r\n\r\nßḍ̇ſⅣ'\"-İ(\u000b'VE(", "tokens": 56, "pieces": ["'Mi", "OS", "(\r\n\r\n", "a", "/b'M", "ḍ̇", "", "éé", "\n", "İ", "‍fi", "\r\n\r\n", "ßḍ̇", "ſ", "Ⅳ", "'\"-", "İ", "(", "\u000b", "'VE", "("]} +{"text": ",…\r‍Dž​'reé0\rDžungla'Re'Dfi/\r\n३­t'Rett'T'TDž字
٣٤٥٦>'VEa/b'Re\u000béé½Džİ'M0", "tokens": 49, "pieces": [",", "…\r", "‍Dž", "​'", "reé", "0", "\r", "Džungla'Re", "'Dfi", "/\r\n", "३", "­t'Re", "tt'T", "'TDž字", "
", "٣٤٥", "٦", ">'", "VEa", "/b'Re", "\u000béé", "½", "Džİ'M", "0"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "t#$%\t㋿EOT", "tokens": 9, "pieces": ["t", "#$%", "\t", "㋿EOT"]} +{"text": "90're \n(#$%-Dž .ᵃ
'll'ſſ­'DDžungla.漢åaBed👍🏽 \n ع!!0\"/字", "tokens": 44, "pieces": ["90", "'re", "", " \n", "(#$%-", "Dž", " ", ".ᵃ", "
", "'ll'ſ", "ſ", "­'", "DDžungla", ".漢åa", "Bed", "👍🏽", " \n", " ع", "!!", "0", "\"/", "字"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "0>
A😀🏽'ſ٣٤٥٦!/\r\n!\r\n\r\n >é'VEⅣB'Ss😀🏽'sAb…-ᵃḍ̇iOS…/\r\ncamelCase\r\n#$%", "tokens": 52, "pieces": ["0", ">", "
A", "😀🏽'", "ſ", "٣٤٥", "٦", "!/\r\n", "!\r\n\r\n", " ", " >", "é'VE", "Ⅳ", "B'S", "s", "😀🏽'", "s", "Ab", "…", "-ᵃḍ̇i", "OS", "…", "/\r\n", "camel", "Case", "\r\n", "#$%"]} +{"text": "Džfi0é\n/½'T#$%Ⅳ>…é\t‍d𐞁 <|endoftext|>!ß#$%tİİa‍/ع>…'S'D🙂", "tokens": 56, "pieces": ["Džfi", "0", "é", "\n", "/", "½", "'T", "#$%", "Ⅳ", ">", "…", "é", "\t", "‍d𐞁", " ", "<|", "endoftext", "|>!", "ß", "#$%", "t", "İİa", "‍/", "ع", ">", "…", "'S'D", "🙂"]} +{"text": "'M\r\n\u000b\ta/b\n//\r\n\r\nع㍿,Abᵃ\r", "tokens": 18, "pieces": ["'M", "\r\n", "\u000b", "\ta", "/b", "\n", "//\r\n\r\n", "ع", "㍿,", "Abᵃ", "\r"]} +{"text": "‍  ſABCå\"\naé👍🏽 å", "\"\n", "aé", "👍🏽", " ", " ꟲ😀🏽camelCase,åDž\n/sé👍🏽Ab\rع'llcamelCase ,-iOS\rDžungla३…ꟲ㍿'re/aB", "tokens": 60, "pieces": ["9", "漢", "<|", "endoftext", "|>", "é", "ꟲ", "😀🏽", "camel", "Case", ",å", "Dž", "\n", "/sé", "👍🏽", "Ab", "\r", "ع'll", "camel", "Case", " ", " ,-", "i", "OS", "\r", "Džungla", "३", "…ꟲ", "㍿'", "re", "/a", "B"]} +{"text": "0<|fim_prefix|>👍🏽'VE'Re!'re,Džunglaſ\r 'ſ>'Reſt
㋿Z!!'S", "tokens": 37, "pieces": ["0", "<|", "fim", "_prefix", "|>👍🏽'", "VE'Re", "!'", "re", ",Džunglaſ", "\r", " ", " '", "ſ", ">'", "Reſt", "
", "㋿Z", "!!'", "S"]} +{"text": "<\" \n 👍🏽HTTPServer'llꟲeém😀🏽👍🏽.<ᵃⅣ\u000b>𐞁s('M\u000bcamelCaseå字éDžd're a/b-㍿'D''T", "tokens": 62, "pieces": ["<\"<", "META", "_START", ">", " \n", " 👍🏽", "HTTPServer'll", "ꟲeém", "😀🏽👍🏽.<", "ᵃ", "Ⅳ", "\u000b", ">𐞁s", "('", "M", "\u000bcamel", "Caseå字é", "Džd're", " ", " a", "/b", "-㍿'", "D", "''", "T"]} +{"text": "𐞁🙂'T<|endoftext|>\nDžungla \n'ReHTTPServer\raHTTPServerB ABC< <|endoftext|>aBⅣ​ écamelCase \n \r\n\r\n9A漢Dž½.t<
fi
\n", "tokens": 61, "pieces": ["𐞁", "🙂'", "T", "<|", "endoftext", "|>\n", "Džungla", " \n", "'Re", "HTTPServer", "\r", "a", "HTTPServer", "B", " ", " ABC", "<", " ", " <|", "endoftext", "|>", "a", "B", "Ⅳ", "​", " écamel", "Case", " \n \r\n\r\n", "9", "A漢", "Dž", "½", ".t", "<", "
fi", "
\n"]} +{"text": "t​ée ", "tokens": 4, "pieces": ["t", "​ée", " "]} +{"text": "iOS \n­< !! \n<㋿  \rſcamelCaseⅣꟲ३camelCase- عiOS'S  ㋿é", "tokens": 38, "pieces": ["i", "OS", " \n", "­<", " ", "!!", " \n", "<㋿", "  \r", "ſcamel", "Case", "Ⅳ", "ꟲ", "३", "camel", "Case", "-", " عi", "OS'S", " ", " ", "㋿é"]} +{"text": "<|fim_prefix|>ḍ̇camelCase'se'TeHTTPServerEOT \r👍🏽d😀🏽'VEDž३", "tokens": 44, "pieces": ["<|", "fim", "_prefix", "|>", "ḍ̇camel", "Case's", "e'T", "e", "HTTPServer", "EOT", "", " \r", "👍🏽", "d", "😀🏽'", "VE", "Dž", "३"]} +{"text": "HTTPServer'reeé\r\n \nß", "tokens": 8, "pieces": ["HTTPServer're", "eé", "\r\n \n", "ß"]} +{"text": ">'VE0ᵃ>", "tokens": 11, "pieces": [">'", "VE", "", "0", "ᵃ", ">"]} +{"text": "/\r\né'Dß<|fim_prefix|>🙂'T00\r\n\r\n\"ع'TcamelCase-\u000b12345678ꟲa/b", "tokens": 34, "pieces": ["/\r\n", "é'D", "ß", "<|", "fim", "_prefix", "|>🙂'", "T", "00", "\r\n\r\n", "\"ع", "'", "Tcamel", "Case", "-", "\u000b", "123", "456", "78", "ꟲa", "/b"]} +{"text": "Džungla12345678>HTTPServer/\r\n​'re'VE.㋿12345678​ B'll- Zm​'é'SعⅣ'DⅣ/\r\nعB́B'D'Re\u000b٣٤٥٦", "tokens": 52, "pieces": ["Džungla", "123", "456", "78", ">HTTPServer", "/\r\n", "​'", "re'VE", ".㋿", "123", "456", "78", "​", " B'll", "-", " Zm", "​'", "é'S", "ع", "Ⅳ", "'D", "Ⅳ", "/\r\n", "عB́", "B'D", "'Re", "\u000b", "٣٤٥", "٦"]} +{"text": "\t𐞁🙂Dž\nDž
#$%́ ‍ \n \r\n\r\n'MAbmdABC-", "tokens": 32, "pieces": ["\t𐞁", "🙂Dž", "\n", "Dž", "
", "#$%́<", "META", "_START", ">", " ", " ‍", " \n \r\n\r\n", "'MAb", "md", "ABC", "-"]} +{"text": "m½'D\n😀🏽\"🙂‍'Re 'D…éÁ३ꟲEOT\n/\r'M\u000b\n<́", "tokens": 36, "pieces": ["m", "½", "'D", "\n", "😀🏽\"🙂‍'", "Re", " ", "'D", "…é", "Á", "३", "ꟲ", "EOT", "\n", "/\r", "'M", "\u000b", "\n", "<́"]} +{"text": "<éBABC…👍🏽 'Ds'S0tHTTPServerABC​\r\n'll' 0<|endoftext|>", "tokens": 32, "pieces": ["<é", "BABC", "…", "👍🏽", " ", " '", "Ds'S", "0", "t", "HTTPServer", "ABC", "​\r\n", "'ll", "'", " ", " ", "0", "<|", "endoftext", "|>"]} +{"text": "ßm\t\r\n\r\n
iOSDžungla \n a/b", "tokens": 13, "pieces": ["ßm", "\t\r\n\r\n", "
i", "OSDžungla", " \n", " a", "/b"]} +{"text": "‍\r\n\r\naBع'M㍿t\n 'SZ/\r\n/\r\n½é \n're. (
字", "tokens": 25, "pieces": ["‍\r\n\r\n", "a", "Bع'M", "㍿t", "\n", " ", " '", "SZ", "/\r\n/\r\n", "½", "é", " \n", "'re", ".", " ", "(", "
字"]} +{"text": "漢ſå\r\n\r\n<|fim_prefix|>\r\n😀🏽EOT<|fim_prefix|>½ iOS \n", "tokens": 27, "pieces": ["漢ſå", "\r\n\r\n", "<|", "fim", "_prefix", "|>\r\n", "😀🏽", "EOT", "<|", "fim", "_prefix", "|>", "½", " ", " i", "OS", " \n"]} +{"text": "'sᵃ㋿å'DABC'Teå<-'D½½9å", "tokens": 27, "pieces": ["'sᵃ", "㋿å'D", "ABC'T", "eå", "<-'", "D", "½", "", "½9", "å"]} +{"text": "İ<|fim_prefix|>éeaBa/b\n/iOS'D \n EOT,Ab'll!🙂d\r\n\r\n\r\n\r\nᵃcamelCase \n ABC٣٤٥٦", "tokens": 37, "pieces": ["İ", "<|", "fim", "_prefix", "|>", "éea", "Ba", "/b", "\n", "/i", "OS'D", " \n", " EOT", ",Ab'll", "!🙂", "d", "\r\n\r\n\r\n\r\n", "ᵃcamel", "Case", " \n", " ABC", "٣٤٥", "٦"]} +{"text": "ⅣaBİ<|endoftext|>'M…#$%'VEDž ", "tokens": 21, "pieces": ["Ⅳ", "a", "Bİ", "<|", "endoftext", "|>'", "M", "…", "#$%'", "VEDž", " "]} +{"text": "a/b३t0/\r\n \n\r\n\r\n­\tḍ̇ᵃ'Re><|fim_prefix|>'THTTPServer字­camelCase \n'M ३'T'T", "tokens": 37, "pieces": ["a", "/b", "३", "t", "0", "/\r\n", " \n\r\n\r\n", "­", "\tḍ̇ᵃ'Re", "><|", "fim", "_prefix", "|>'", "THTTPServer字", "­camel", "Case", " \n", "'M", " ", " ", "३", "'T'T"]} +{"text": "​12345678😀🏽<|endoftext|>Dž \n ", "tokens": 28, "pieces": ["i", "OS", "🙂", " ", " B", "/\r\n", "\t", "㋿>", "123", "456", "78", "😀🏽<|", "endoftext", "|>", "Dž", " \n", " "]} +{"text": "ᵃ-\u000b12345678ſiOSa/b9<|endoftext|>'D'll'Dž'll'sſ#$%éſ<|endoftext|>9(EOTfi é#$%ée­­𐞁'llZ ㋿", "tokens": 63, "pieces": ["ᵃ", "-", "\u000b", "123", "456", "78", "ſi", "OSa", "/b", "9", "<|", "endoftext", "|>'", "D'll", "'Dž'll", "'sſ", "#$%", "éſ", "<|", "endoftext", "|>", "9", "(EOTfi", " é", "#$%", "ée", "­­", "𐞁'll", "Z", " ", " ㋿"]} +{"text": "a/b(Dž३<|fim_prefix|>…😀🏽9éᵃ", "tokens": 22, "pieces": ["a", "/b", "(Dž", "३", "<|", "fim", "_prefix", "|>", "…", "😀🏽", "9", "éᵃ"]} +{"text": "-𐞁/ḍ̇!İ'VEDžunglaéfi'Td
(́(ع/", "tokens": 27, "pieces": ["-𐞁", "/ḍ̇", "!İ'VE", "Džunglaéfi'T", "d", "
", "(́", "(ع", "/"]} +{"text": "12345678-\n/aB­/!<|endoftext|>", "tokens": 17, "pieces": ["123", "456", "78", "-\n/", "a", "B", "­/!<|", "endoftext", "|>"]} +{"text": "å㋿d…é'/  \n İ\"é\u000b漢-😀🏽s <|endoftext|>½é­ABC,
 ḍ̇\n HTTPServerHTTPServer", "tokens": 52, "pieces": ["å", "㋿d", "", "…é", "'/", "  \n", " İ", "\"é", "\u000b漢", "-😀🏽", "s", " ", "<|", "endoftext", "|>", "½", "é", "­ABC", ",", "
", " ḍ̇", "\n", " ", " HTTPServer", "HTTPServer"]} +{"text": "İßs
ꟲa/b", "tokens": 9, "pieces": ["İßs", "
ꟲa", "/b"]} +{"text": "'ll'sé<|endoftext|>३å'M>!", "tokens": 16, "pieces": ["'ll's", "é", "<|", "endoftext", "|>", "३", "å'M", ">!"]} +{"text": "sZAb  \n0ſB(\u000bd𐞁\r\n\r\n>'SDž /३عⅣd'M(aBé,…\n<|endoftext|>'S'll0Ab👍🏽
", "tokens": 50, "pieces": ["s", "ZAb", "  \n", "0", "ſ", "B", "(", "\u000bd𐞁", "\r\n\r\n", ">'", "SDž", " ", "/", "३", "ع", "Ⅳ", "d'M", "(a", "Bé", ",", "…\n", "<|", "endoftext", "|>'", "S'll", "0", "Ab", "👍🏽", "
"]} +{"text": "‍e'śع\"\r 12345678ß-́,9#$%\n/ſ \n -d‍字'SaB\na/bcamelCaseİ\n/", "tokens": 40, "pieces": ["‍e's", "́ع", "\"\r", " ", " ", "123", "456", "78", "ß", "-́", ",", "9", "#$%\n/", "ſ", " \n", " -", "d", "‍字'S", "a", "B", "\n", "a", "/bcamel", "Case", "İ", "\n", "/"]} +{"text": "Dž \n 𐞁🙂<|endoftext|>'s9<|endoftext|>\u000b字'T́(\"9\r\n\r\n'siOS字<|fim_prefix|><|endoftext|>'Re'M\r\nd<|fim_prefix|>'reé🙂", "tokens": 61, "pieces": ["Dž", " \n", " 𐞁", "🙂<|", "endoftext", "|>'", "s", "9", "<|", "endoftext", "|>", "\u000b字'T", "́", "(\"", "9", "\r\n\r\n", "'si", "OS字", "<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "Re'M", "\r\n", "d", "<|", "fim", "_prefix", "|>'", "reé", "🙂"]} +{"text": " EOT\u000bfidZ#$%HTTPServer/iOSᵃ㍿́\r\"ݽ…\nm.Ab'M'S٣٤٥٦Džungla(éꟲ'll", "tokens": 45, "pieces": [" EOT", "\u000bfid", "Z", "#$%", "HTTPServer", "/i", "OSᵃ", "㍿́", "\r", "\"İ", "½", "…\n", "m", ".Ab'M", "'S", "٣٤٥", "٦", "Džungla", "(éꟲ'll"]} +{"text": "ᵃ\r\n.#$%!!٣٤٥٦\ndAB0\n/́ꟲ !!𐞁B(ſABC字HTTPServer<|fim_prefix|>\r \nABCé\n😀🏽HTTPServera/b\t'12345678", "tokens": 63, "pieces": ["ᵃ", "\r\n", ".#$%!!", "٣٤٥", "٦", "\n", "d", "AB", "0", "\n", "/́ꟲ", " !!<", "META", "_START", ">𐞁", "B", "(ſ", "ABC字HTTPServer", "<|", "fim", "_prefix", "|>\r", " \n", "ABCé", "\n", "😀🏽", "HTTPServera", "/b", "\t", "'", "123", "456", "78"]} +{"text": "( \n㋿td🙂å'så'M\n/camelCase½\n,ABCå\n0/", "tokens": 26, "pieces": ["(", " \n", "㋿td", "🙂å's", "å'M", "\n", "/camel", "Case", "½", "\n", ",ABCå", "\n", "0", "/"]} +{"text": "\n/>-HTTPServer\nAb'VE'S#$%<|endoftext|>camelCaseß'VEعDžungla>ꟲéß\r'(", "tokens": 50, "pieces": ["\n", "/>-", "HTTPServer", "\n", "Ab'VE", "'S", "#$%<|", "endoftext", "|>", "camel", "Caseß'VE", "ع", "Džungla", ">ꟲéß", "\r", "'(<", "META", "_START", ">"]} +{"text": "ꟲa/b", "tokens": 5, "pieces": ["ꟲa", "/b"]} +{"text": "09HTTPServer's㋿iOS#$%ᵃ,ⅣaB'siOS", "tokens": 22, "pieces": ["09", "HTTPServer's", "㋿i", "OS", "#$%", "ᵃ", ",", "Ⅳ", "a", "B's", "i", "OS"]} +{"text": "é", "tokens": 1, "pieces": ["é"]} +{"text": "Džungla㋿'s", "tokens": 9, "pieces": ["Džungla", "㋿'", "s"]} +{"text": "字…‍HTTPServerA‍' \n 'D\n\r\n a/b…\n😀🏽12345678Bfi'Re🙂s\"㋿!😀🏽\r\n­", "tokens": 48, "pieces": ["字", "", "…", "‍HTTPServer", "A", "‍'", " \n", " '", "D", "\n\r\n", " a", "/b", "…\n", "😀🏽", "123", "456", "78", "Bfi'Re", "🙂s", "\"㋿!😀🏽\r\n", "­"]} +{"text": " #$%éḍ̇\t३́afi, Ab\r\n\r\na٣٤٥٦\",­'re!!ABC\r\né!\r٣٤٥٦'re\r\r\n\r\n­ᵃ'VE\té<|fim_prefix|>", "tokens": 51, "pieces": [" ", "#$%", "éḍ̇", "\t", "३", "́afi", ",", " Ab", "\r\n\r\n", "a", "٣٤٥", "٦", "\",­'", "re", "!!", "ABC", "\r\n", "é", "!\r", "٣٤٥", "٦", "'re", "\r\r\n\r\n", "­ᵃ'VE", "\té", "<|", "fim", "_prefix", "|>"]} +{"text": "İAé,", "tokens": 4, "pieces": ["İAé", ","]} +{"text": "‍ <|fim_prefix|>,'D.'ſ/\r.a0m\n!'reAb\"👍🏽/Ab!!Ab३​\tABC", "tokens": 36, "pieces": ["‍", " ", " <|", "fim", "_prefix", "|>,'", "D", ".'", "ſ", "/\r", ".a", "0", "m", "\n", "!'", "re", "Ab", "\"👍🏽<", "META", "_START", ">/", "Ab", "!!", "Ab", "३", "​", "\tABC"]} +{"text": "ⅣcamelCase<|endoftext|>\" \r\n<|fim_prefix|>'rea/bᵃtfi'reᵃB👍🏽٣٤٥٦ABCe'DBᵃ\rm\"\r\r\n d عDžungla\r\n0'Re𐞁 \n!ſ", "tokens": 66, "pieces": ["Ⅳ", "camel", "Case", "<|", "endoftext", "|>\"", " \r\n", "<|", "fim", "_prefix", "|>'", "rea", "/bᵃtfi're", "ᵃ", "B", "👍🏽", "٣٤٥", "٦", "ABCe'D", "Bᵃ", "\r", "m", "\"\r\r\n", " d", " عDžungla", "\r\n", "0", "'Re𐞁", " \n", "!ſ"]} +{"text": "İ漢\"aZ½‍<|fim_prefix|>d12345678B're", "tokens": 18, "pieces": ["İ漢", "\"a", "Z", "½", "‍<|", "fim", "_prefix", "|>", "d", "123", "456", "78", "B're"]} +{"text": " 0字…½\n'ſ/\r\né½t12345678漢ḍ̇.漢漢>é-/\r\n'ḍ̇\"Ⅳ'D👍🏽\n#$%<|endoftext|>", "tokens": 49, "pieces": [" ", "0", "字", "…", "½", "\n", "'ſ", "/\r\n", "é", "½", "t", "123", "456", "78", "漢ḍ̇", ".漢漢", ">é", "-/\r\n", "'ḍ̇", "\"", "Ⅳ", "'D", "👍🏽\n", "#$%<|", "endoftext", "|>"]} +{"text": "
å'VE‍'D \n
ß'VE \n👍🏽٣٤٥٦​𐞁'TAb𐞁…tHTTPServer👍🏽", "tokens": 40, "pieces": ["
å'VE", "‍'", "D", " \n", "
ß'VE", " \n", "👍🏽", "٣٤٥", "٦", "​𐞁'T", "Ab𐞁", "…t", "HTTPServer", "👍🏽"]} +{"text": "'ſDžes\n'ſHTTPServer'S㋿e
  \n HTTPServer…\n
iOSZ㍿ß\n\u000b/👍🏽'reع漢 \n ḍ̇-'VE½", "tokens": 49, "pieces": ["'ſ", "Džes", "\n", "'ſ", "HTTPServer'S", "㋿e", "
  \n", " HTTPServer", "…\n", "
i", "OSZ", "㍿ß", "\n", "\u000b", "/👍🏽'", "reع漢", " \n", " ḍ̇", "-'", "VE", "½"]} +{"text": "'T½\"😀🏽'TBd/'re\r\n\r\nſ é٣٤٥٦'Re…<
", "tokens": 24, "pieces": ["'T", "½", "\"😀🏽'", "TBd", "/'", "re", "\r\n\r\n", "ſ", " ", " é", "٣٤٥", "٦", "'Re", "…", "<", "
"]} +{"text": "-\u000ba0Z9é '½.", "tokens": 11, "pieces": ["-", "\u000ba", "0", "Z", "9", "é", " '", "½", "."]} +{"text": "३'S\nİa/b​A. s \n ㋿/å.e!'Mḍ̇'DiOS", "tokens": 28, "pieces": ["३", "'S", "\n", "İa", "/b", "​A", ".", " s", " \n", " ㋿/", "å", ".e", "!'", "Mḍ̇'D", "i", "OS"]} +{"text": "…/Džunglaİ
é\te \né \n eB <\rAᵃ'ßBa\n/\n/½\t३sAb𐞁camelCase're/ 🙂", "tokens": 50, "pieces": ["…", "/<", "EOT", ">Džungla", "İ", "
é", "\te", " \n", "é", " \n", " e", "B", "", " ", " <\r", "Aᵃ", "'ß", "Ba", "\n", "/\n/", "½", "\t", "३", "s", "Ab𐞁camel", "Case're", "/", " 🙂"]} +{"text": "­'ReZ
字Z", "tokens": 7, "pieces": ["­'", "Re", "Z", "
字", "Z"]} +{"text": "ABC\n/漢'll\n\r\n\r\n🙂́ᵃAꟲ'll'Réd>字-#$%😀🏽e½/\r\naBa/b'Re'", "ll'Re", "́d", ">字", "-#$%😀🏽", "e", "½", "/\r\n", "a", "Ba", "/b'Re", "d9a\r\u000bꟲ'Re
漢éAb're<|endoftext|>\n0
EOTAd٣٤٥٦😀🏽", "tokens": 62, "pieces": ["a'D", "å", "‍", " ", "'VEa", "/b'll", "漢", " ", "\"", "…", "㋿ꟲ", "\n", "Dž", "d", "9", "a", "\r", "\u000bꟲ'Re", "
漢é", "Ab're", "<|", "endoftext", "|>\n", "0", "
EOTAd", "٣٤٥", "٦", "😀🏽"]} +{"text": "12345678İ \nꟲEOTaعZ İ'ſ‍ \n12345678Džunglaé㍿eDž<|fim_prefix|>‍‍٣٤٥٦s\téⅣ12345678Z字aB(㋿fiAbé", "tokens": 65, "pieces": ["123", "456", "78", "İ", " \n", "ꟲEOTaع", "Z", " İ'ſ", "‍", " \n", "123", "456", "78", "Džunglaé", "㍿e", "Dž", "<|", "fim", "_prefix", "|>‍‍", "٣٤٥", "٦", "s", "\té", "Ⅳ12", "345", "678", "Z字a", "B", "(㋿", "fi", "Abé"]} +{"text": "DžABC\r\n<|fim_prefix|>🙂‍Ab㍿٣٤٥٦Dž <|fim_prefix|>B#$%HTTPServer", "tokens": 34, "pieces": ["DžABC", "\r\n", "<|", "fim", "_prefix", "|>🙂‍", "Ab", "㍿", "٣٤٥", "٦", "Dž", " ", " <|", "fim", "_prefix", "|>", "B", "#$%", "HTTPServer"]} +{"text": "'sDžunglaHTTPServer  \n \n/漢'VE👍🏽ⅣAb0éⅣaå३12345678🙂'M,\u000b\r…٣٤٥٦Džungla३'Re(ſ'D", "tokens": 56, "pieces": ["'s", "Džungla", "HTTPServer", "  \n \n", "/漢'VE", "👍🏽", "Ⅳ", "Ab", "0", "é", "Ⅳ", "aå", "३12", "345", "678", "🙂'", "M", ",", "\u000b\r", "…", "٣٤٥", "٦", "Džungla", "३", "'", "Re", "(ſ'D"]} +{"text": "'Mİ‍'sA.-½ ꟲ'VEḍ̇éᵃ \nAb12345678", "tokens": 26, "pieces": ["'Mİ", "‍'", "s", "A", ".-", "½", " ꟲ'VE", "ḍ̇éᵃ", " \n", "Ab", "123", "456", "78"]} +{"text": "A A𐞁å㋿aß#$%a/\r\n,
s½>😀🏽(½
漢EOTABCع\u000b'S'-İ
m👍🏽<|endoftext|>\r\naᵃ", "tokens": 54, "pieces": ["A", " A𐞁å", "㋿aß", "#$%", "a", "/\r\n", ",", "
s", "½", ">😀🏽(", "½", "
漢EOTABCع", "\u000b", "'S", "'-", "İ", "
m", "👍🏽<|", "endoftext", "|>\r\n", "aᵃ"]} +{"text": "#$%'VE(İ \n Ab­ 0a'Re - 'S'Re­/\r\nB", "tokens": 22, "pieces": ["#$%'", "VE", "(İ", " \n", " Ab", "­", " ", " ", "0", "a'Re", " ", "-", " ", " '", "S'Re", "­/\r\n", "B"]} +{"text": "!!٣٤٥٦'Re \n👍🏽ꟲDžunglaa/b'ſßcamelCase(\r.'DaZs", "tokens": 33, "pieces": ["!!", "٣٤٥", "٦", "'Re", " \n", "👍🏽", "ꟲDžunglaa", "/b'ſ", "ßcamel", "Case", "(\r", ".'", "Da", "Zs"]} +{"text": "sⅣ\n/ꟲa🙂½'M\n", "tokens": 19, "pieces": ["s", "", "Ⅳ", "\n", "/ꟲa", "🙂", "½", "'M", "\n"]} +{"text": "­​字Dž\t'Tß\r\niOS's'MAbİ >é'll/\r\nAbiOSa.​DžB😀🏽́", "tokens": 35, "pieces": ["­​", "字", "Dž", "\t", "'Tß", "\r\n", "i", "OS's", "'MAb", "İ", " ", ">é'll", "/\r\n", "Abi", "OSa", ".​", "DžB", "😀🏽́"]} +{"text": "B\t/\r\n𐞁a/b<|fim_prefix|>Dž#$%\u000b\n/mcamelCase", "tokens": 27, "pieces": ["B", "\t", "/\r\n", "𐞁a", "/b", "<|", "fim", "_prefix", "|>", "Dž", "#$%", "\u000b\n", "/mcamel", "Case"]} +{"text": "!#$%'VEع<|fim_prefix|>HTTPServer\n<­\r\n\r\n‍字😀🏽é­,B<|fim_prefix|>३ \ne👍🏽\r\n\r\nd>", "tokens": 55, "pieces": ["!#$%'", "VEع", "<|", "fim", "_prefix", "|>", "HTTPServer", "\n", "<­\r\n\r\n", "‍字", "😀🏽", "é", "­,", "B", "<|", "fim", "_prefix", "|>", "३", " \n", "e", "👍🏽\r\n\r\n", "d", ">"]} +{"text": "½tİ\u000b
", "tokens": 5, "pieces": ["½", "t", "İ", "\u000b
"]} +{"text": "'VE!!\n/ع😀🏽", "tokens": 8, "pieces": ["'VE", "!!\n/", "ع", "😀🏽"]} +{"text": "
 \n ABC\tfi\"'VEAb㍿/s🙂ABCa/b‍ 'M0t\r\n'sEOT", "tokens": 29, "pieces": ["
 \n", " ABC", "\tfi", "\"'", "VEAb", "㍿/", "s", "🙂ABCa", "/b", "‍", " ", "'M", "0", "t", "\r\n", "'s", "EOT"]} +{"text": "\r\n\r\n#$%!!​½A'M 'TiOS", "tokens": 12, "pieces": ["\r\n\r\n", "#$%!!​", "½", "A'M", " ", " '", "Ti", "OS"]} +{"text": "9éDžunglaEOTtda/bİ12345678<|fim_prefix|>#$%ß𐞁m\r\n/\r\nع9camelCases\u000b'SDžunglaḍ̇éꟲEOT<|fim_prefix|>aſ,/", "tokens": 65, "pieces": ["9", "é", "Džungla", "EOTt", "da", "/b", "İ", "123", "456", "78", "<|", "fim", "_prefix", "|>#$%", "ß𐞁m", "\r\n", "/\r\n", "ع", "9", "camel", "Cases", "\u000b", "'SDžunglaḍ̇éꟲ", "EOT", "<|", "fim", "_prefix", "|>", "aſ", ",/"]} +{"text": "EOT0 0Ⅳꟲ#$%字
Z'ſ'Ta/bḍ̇'llß", "tokens": 25, "pieces": ["EOT", "0", " ", "0Ⅳ", "ꟲ", "#$%", "字", "
Z'ſ", "'Ta", "/bḍ̇'ll", "ß"]} +{"text": " \nDžunglam12345678\n/Ⅳ12345678EOT'lliOS'ſß", "tokens": 23, "pieces": [" \n", "Džunglam", "123", "456", "78", "\n", "/", "Ⅳ12", "345", "678", "EOT'll", "i", "OS'ſ", "ß"]} +{"text": " \nEOTDž‍字\u000b👍🏽😀🏽字🙂", "tokens": 16, "pieces": [" \n", "EOTDž", "‍字", "\u000b", "👍🏽😀🏽", "字", "🙂"]} +{"text": "ꟲé12345678aB \n 12345678EOT \n<३'VE00…Ⅳ字iOSé'Reꟲ're's'siOSt 0aeعé's", "tokens": 54, "pieces": ["ꟲé", "123", "456", "78", "a", "B", " \n", " ", "123", "456", "78", "EOT", " \n", "<", "३", "'VE", "", "00", "…", "Ⅳ", "字i", "OSé'Re", "ꟲ're", "'s's", "i", "OSt", " ", "0", "aeعé's", ""]} +{"text": "'ll'Sé/a/b(d eꟲ/\r\nAå​'reAb'M#$%'sss\rḍ̇ ㋿/\r\nm", "tokens": 33, "pieces": ["'ll'S", "é", "/a", "/b", "(d", " eꟲ", "/\r\n", "Aå", "​'", "re", "Ab'M", "#$%'", "sss", "\r", "ḍ̇", " ", " ㋿/\r\n", "m"]} +{"text": "'ll \n/­", "tokens": 9, "pieces": ["'ll", " \n", "/­<", "EOT", ">"]} +{"text": "\"­t'll\"ééfi0Z's\n '㍿aB\"'re #$%…'ll\n/m漢३\u000b'D𐞁.ᵃ!!\t㍿Džungla😀🏽𐞁", "tokens": 61, "pieces": ["\"­<", "META", "_START", ">t'll", "\"ééfi", "0", "Z's", "\n", " ", " '㍿", "a", "B", "\"'", "re", " #$%", "…", "'ll", "\n", "/m漢", "३", "\u000b", "'D𐞁", ".ᵃ", "!!", "\t", "㍿Džungla", "😀🏽", "𐞁"]} +{"text": "sßAb'M\n\"\r\n\r\n<漢\tAb🙂DžA\t㍿\n/ ", "tokens": 21, "pieces": ["sß", "Ab'M", "\n", "\"\r\n\r\n", "<漢", "\tAb", "🙂DžA", "\t", "㍿\n/", " "]} +{"text": "'s \n 👍🏽㍿ḍ̇🙂", "tokens": 12, "pieces": ["'s", " \n", " 👍🏽㍿", "ḍ̇", "🙂"]} +{"text": "(½ Z9s\r#$%12345678m'S
­\nDžunglafi𐞁ᵃmHTTPServer 'Re㍿㍿ \n B!!!'reZعİ<|endoftext|>", "tokens": 54, "pieces": ["(", "½", " Z", "9", "s", "\r", "#$%", "123", "456", "78", "m'S", "
", "­\n", "Džunglafi𐞁ᵃm", "HTTPServer", " '", "Re", "㍿㍿", " \n", " B", "!!!'", "re", "Zع", "İ", "<|", "endoftext", "|>"]} +{"text": "\réA a/ba­…٣٤٥٦\n12345678漢", "tokens": 19, "pieces": ["\r", "é", "A", " a", "/ba", "­", "…", "٣٤٥", "٦", "\n", "123", "456", "78", "漢"]} +{"text": "\r\n\r\n'T#$%< \n 'VE ", "tokens": 9, "pieces": ["\r\n\r\n", "'T", "#$%<", " \n", " '", "VE", " "]} +{"text": "\n/字iOSع½\r\n\r\n
camelCase#$%٣٤٥٦å<|fim_prefix|>­\r'SfiⅣ", "tokens": 34, "pieces": ["\n", "/字i", "OSع", "½", "\r\n\r\n", "
", "camel", "Case", "#$%", "٣٤٥", "٦", "å", "<|", "fim", "_prefix", "|>­\r", "'Sfi", "Ⅳ"]} +{"text": "'s漢𐞁İiOScamelCasedİe #$%d'MABCs \n 'reİa/b-ع'reHTTPServer \n!३camelCase<|fim_prefix|>😀🏽é.", "tokens": 52, "pieces": ["'s漢𐞁", "İi", "OScamel", "Cased", "İe", " ", " #$%", "d'M", "ABCs", " \n", " '", "re", "İa", "/b", "-", "ع're", "HTTPServer", " \n", "!", "३", "camel", "Case", "<|", "fim", "_prefix", "|>😀🏽", "é", "."]} +{"text": "12345678Džungla'D ½a \n …<12345678ᵃ/aBDžungla\u000b㍿camelCase🙂mع\u000b<|fim_prefix|>'Tm字iOS<|endoftext|><|endoftext|>iOSſABC\tꟲ😀🏽\n/Ab\t", "tokens": 79, "pieces": ["123", "456", "78", "Džungla'D", " ", "½", "a", " \n", " ", "…", "<", "123", "456", "78", "ᵃ", "/a", "BDžungla", "\u000b", "㍿camel", "Case", "🙂mع", "\u000b", "<|", "fim", "_prefix", "|>'", "Tm字i", "OS", "<|", "endoftext", "|><|", "endoftext", "|>", "i", "OSſ", "ABC", "\tꟲ", "😀🏽\n/", "Ab", "\t"]} +{"text": "漢'ſ", "tokens": 3, "pieces": ["漢'ſ"]} +{"text": "… camelCasefiİ<३#$%㍿ḍ̇dB😀🏽", "tokens": 22, "pieces": ["… ", " camel", "Casefi", "İ", "<", "३", "#$%㍿", "ḍ̇d", "B", "😀🏽"]} +{"text": "\u000b\n/!\r\n\r\niOS'ſ12345678/\r\n ㍿'M<|fim_prefix|>Ab\r😀🏽ſ'll­", "tokens": 33, "pieces": ["\u000b\n", "/!\r\n\r\n", "i", "OS'ſ", "123", "456", "78", "/\r\n", " ㍿'", "M", "<|", "fim", "_prefix", "|>", "Ab", "\r", "😀🏽", "ſ'll", "­"]} +{"text": "'Taſ​'ll0३'S'ſ'Sſ!!😀🏽,!!\u000bHTTPServer/​mé
9𐞁ß𐞁\r\n\r\n's'ᵃ🙂s", "tokens": 44, "pieces": ["'Taſ", "​'", "ll", "0३", "'S'ſ", "'Sſ", "!!😀🏽,!!", "\u000bHTTPServer", "/​", "mé", "
", "9", "𐞁ß𐞁", "\r\n\r\n", "'s", "'ᵃ", "🙂s"]} +{"text": "/\r\n👍🏽­​İİß're\n/'llZmB\n/́ſ́aBİع😀🏽're …'ſ ' .㋿camelCaseHTTPServerfit'sm", "tokens": 49, "pieces": ["/\r\n", "👍🏽­​", "İİß're", "\n", "/'", "ll", "Zm", "B", "\n", "/́ſ́a", "Bİع", "😀🏽'", "re", " ", "…", "'ſ", " ", " '", " ", ".㋿", "camel", "Case", "HTTPServerfit's", "m"]} +{"text": "ꟲ!!m'D,.,ع🙂!!iOS#$%\r\n…\"\rAb'SaBABC👍🏽>\u000b👍🏽t#$%Dž½", "tokens": 38, "pieces": ["ꟲ", "!!", "m'D", ",.,", "ع", "🙂!!", "i", "OS", "#$%\r\n", "…", "\"\r", "Ab'S", "a", "BABC", "👍🏽>", "\u000b", "👍🏽", "t", "#$%", "Dž", "½"]} +{"text": "Ab0\r\nſ🙂é,B-,fi'll'VE'BEOT
12345678\n/HTTPServerEOT👍🏽ع \n ⅣcamelCase🙂字🙂ßEOT\u000beiOS…🙂\"٣٤٥٦", "tokens": 53, "pieces": ["Ab", "0", "\r\n", "ſ", "🙂é", ",B", "-,", "fi'll", "'VE", "'BEOT", "
", "123", "456", "78", "\n", "/HTTPServer", "EOT", "👍🏽", "ع", " \n", " ", "Ⅳ", "camel", "Case", "🙂字", "🙂ß", "EOT", "\u000bei", "OS", "…", "🙂\"", "٣٤٥", "٦"]} +{"text": "EOT'reİ😀🏽 ㋿aB
HTTPServer <३'D", "tokens": 20, "pieces": ["EOT're", "İ", "😀🏽", " ㋿", "a", "B", "
HTTPServer", " ", "<", "३", "'D"]} +{"text": "Ⅳ\r\u000b漢
<|endoftext|> \n 'll½­å Džungla åⅣᵃiOSḍ̇́ſ!‍aB/\r\nAbع ́aBe\u000b'ſ㋿
", "tokens": 61, "pieces": ["Ⅳ", "\r", "\u000b漢", "
", "<|", "endoftext", "|><", "META", "_START", ">", " \n", " '", "ll", "½", "­å", " ", " Džungla", " å", "Ⅳ", "ᵃi", "OSḍ̇́ſ", "!‍", "a", "B", "/\r\n", "Abع", " ́a", "Be", "\u000b", "'ſ", "㋿", "
"]} +{"text": "\nZ!!é#$%٣٤٥٦ ㋿\t​­/
ABC㋿", "tokens": 23, "pieces": ["\n", "Z", "!!", "é", "#$%", "٣٤٥", "٦", " ", "㋿", "\t", "​­/", "
ABC", "㋿"]} +{"text": " \n Zm\r\n'sé漢!!#$%a/bEOTꟲ<|endoftext|>'sß👍🏽😀🏽\n/dcamelCase'T'D9s>\niOS\n/ꟲ'VE's/\rᵃDž…iOS", "tokens": 67, "pieces": [" \n", " Zm", "\r\n", "'sé漢", "!!#$%<", "META", "_START", ">a", "/b", "EOTꟲ", "<|", "endoftext", "|>'", "sß", "👍🏽😀🏽\n/", "dcamel", "Case'T", "'D", "9", "s", ">\n", "i", "OS", "\n", "/ꟲ'VE", "'s", "/\r", "ᵃ", "Dž", "…i", "OS"]} +{"text": "\rsAعsEOT
 \tſ \n \n  \n a/b\t'VEe>'VEDžungla.e'ReZ…'<|fim_prefix|>0BⅣ\"\ta", "tokens": 47, "pieces": ["\r", "s", "Aعs", "EOT", "
 ", "\t", "ſ", " \n \n  \n", " a", "/b", "\t", "'VEe", ">'", "VEDžungla", ".e'Re", "Z", "…", "'<|", "fim", "_prefix", "|>", "0", "B", "Ⅳ", "\"", "\ta"]} +{"text": "Ⅳ", "tokens": 6, "pieces": ["Ⅳ", ""]} +{"text": "ſiOSéEOT  \nsß\r\n\r\n字½!!㋿(é😀🏽
>", "tokens": 24, "pieces": ["ſi", "OSé", "EOT", "  \n", "sß", "\r\n\r\n", "字", "½", "!!㋿(", "é", "😀🏽", "
", ">"]} +{"text": "​½ꟲ㋿.dⅣ'M३1234567812345678iOS‍…'Refi'llİ", "tokens": 29, "pieces": ["​", "½", "ꟲ", "㋿.", "d", "Ⅳ", "'M", "३12", "345", "678", "123", "456", "78", "i", "OS", "‍", "…", "'Refi'll", "İ"]} +{"text": "㋿é
'Mſ#$%s#$%
👍🏽 😀🏽३İ‍'VEع-HTTPServer字­", "tokens": 31, "pieces": ["㋿é", "
", "'Mſ", "#$%", "s", "#$%", "
", "👍🏽", " ", " 😀🏽", "३", "İ", "‍'", "VEع", "-HTTPServer字", "­"]} +{"text": "३fi9­HTTPServer", "tokens": 6, "pieces": ["३", "fi", "9", "­HTTPServer"]} +{"text": " Abꟲ\"ABC", "tokens": 7, "pieces": [" Abꟲ", "\"ABC"]} +{"text": "字́d/Ⅳḍ̇'a/bZ,ᵃ
'St​'ll‍३s𐞁İ", "tokens": 30, "pieces": ["字́d", "/", "Ⅳ", "ḍ̇", "'a", "/b", "Z", ",ᵃ", "
", "'St", "​'", "ll", "‍", "३", "s𐞁", "İ"]} +{"text": "漢ꟲéİſfiZ\u000b㍿.\rDž३iOS>\n
EOT㍿ \n'T", "tokens": 30, "pieces": ["漢ꟲé", "İſfi", "Z", "\u000b", "㍿.\r", "Dž", "३", "i", "OS", ">\n", "
EOT", "㍿", " \n", "'T"]} +{"text": ">😀🏽's🙂>'re\r\ń/\r\n<|endoftext|>'DB\r\n9…'sA\n/(eß'D/a…- \n m\r\nem
ABC😀🏽'", "s", "🙂>'", "re", "\r\n", "́", "/\r\n", "<|", "endoftext", "|>'", "DB", "\r\n", "9", "…", "'s", "A", "\n", "/(", "eß'D", "/", "a", "…", "-", " \n", " ", " m", "\r\n", "em", "
ABC", "12345678'T\n'VE𐞁𐞁\u000b<ß<|fim_prefix|>diOSaBABC…Ⅳa\r\n​", "tokens": 50, "pieces": [" ", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'T", "\n", "'VE", "𐞁𐞁", "\u000b", "<ß", "<|", "fim", "_prefix", "|>", "d", "i", "OSa", "BABC", "…", "Ⅳ", "a", "\r\n", "​"]} +{"text": "å😀🏽عcamelCase /\r\n ḍ̇ \n Ab漢", "tokens": 23, "pieces": ["å", "😀🏽", "ع", "camel", "Case", " ", " /\r\n", " ", " ḍ̇", " \n", " Ab漢"]} +{"text": "ḍ̇\u000b­Z/Džungla٣٤٥٦!\n … >EOT<|fim_prefix|>
'ReAbéعABC\n/'VE½A\u000be(👍🏽", "tokens": 45, "pieces": ["ḍ̇", "\u000b", "­Z", "/Džungla", "٣٤٥", "٦", "!\n", " …", " ", ">EOT", "<|", "fim", "_prefix", "|>", "
", "'Re", "Abéع", "ABC", "\n", "/'", "VE", "½", "A", "\u000be", "(👍🏽"]} +{"text": "'S>m!B>#$%!!0ſAbABCİ👍🏽é'D ​ B \n\u000bDžungla\n/ aBB \n/ \r'M", "tokens": 38, "pieces": ["'S", ">m", "!B", ">#$%!!", "0", "ſ", "Ab", "ABCİ", "👍🏽", "é'D", " ", "​", " B", " \n", "\u000bDžungla", "\n", "/", " ", " a", "BB", " \n", "/", " \r", "'M"]} +{"text": "/😀🏽-İ-åع'ſ\r\n dcamelCase😀🏽's㋿(㍿㍿'et…a<|endoftext|>/\r\n!", "tokens": 51, "pieces": ["/😀🏽-", "İ", "-åع'ſ", "\r\n", " dcamel", "Case", "😀🏽'", "s", "㋿(㍿㍿'<", "META", "_START", ">et", "…a", "<|", "endoftext", "|>/\r\n", "!<", "META", "_START", ">"]} +{"text": "Zع'fi\r\n\r\n\r\nABCe0\rZ३ficamelCase'DDžungla'MsEOTd\r0ſꟲ'ReDžungla9…Ab½DžunglaAbDžungla½", "tokens": 50, "pieces": ["Zع", "'fi", "\r\n\r\n\r\n", "ABCe", "0", "\r", "Z", "३", "ficamel", "Case'D", "Džungla'M", "s", "EOTd", "\r", "0", "ſꟲ'Re", "Džungla", "9", "…Ab", "½", "Džungla", "Ab", "Džungla", "½"]} +{"text": "ABCaBß 'VEe'sDžunglaᵃ", "tokens": 16, "pieces": ["ABCa", "Bß", " ", "'VEe's", "Džunglaᵃ"]} +{"text": "Z 🙂é!!漢‍\r\n漢👍🏽B", "tokens": 13, "pieces": ["Z", " ", " 🙂", "é", "!!", "漢", "‍\r\n", "漢", "👍🏽", "B"]} +{"text": "EOTDžungla (iOS'😀🏽 \n /\r\n,\rt\r\n\r\n🙂B'Ⅳ0EOTa/b\r\n\r\n!‍", "tokens": 38, "pieces": ["EOTDžungla", " ", " (", "i", "OS", "'😀🏽", " \n", " /\r\n", ",\r", "t", "\r\n\r\n", "🙂B", "'", "Ⅳ0", "EOTa", "/b", "\r\n\r\n", "!‍<", "EOT", ">"]} +{"text": "\u000bEOT.s
ſ👍🏽👍🏽İع. \n >🙂㍿,s'T𐞁iOSḍ̇‍BDžungla😀🏽Džungla>İ🙂DžunglacamelCase'D'ſ!d㋿", "tokens": 68, "pieces": ["\u000bEOT", ".s", "
ſ", "👍🏽👍🏽", "İع", ".", " \n", " >🙂㍿,", "s", "'", "T𐞁i", "OSḍ̇", "‍BDžungla", "😀🏽", "Džungla", ">İ", "🙂Džunglacamel", "Case'D", "'", "ſ", "!d", "㋿"]} +{"text": "''S9ABC㋿,\n/9\r'Ś(ABCⅣ́", "tokens": 20, "pieces": ["''", "S", "", "9", "ABC", "㋿,\n/", "9", "\r", "'Ś", "(ABC", "Ⅳ", "́"]} +{"text": "漢㋿​'VE​ßḍ̇<㋿iOS…İ㍿ fiA  ́㋿ \n Ab'VE", "tokens": 37, "pieces": ["漢", "㋿​'", "VE", "​ßḍ̇", "<㋿", "i", "OS", "…İ", "㍿", " fi", "A", " ", " ́", "㋿", " \n", " Ab'VE"]} +{"text": "d字\r\n\r\n𐞁camelCaseHTTPServer'ſ́9aåå<|fim_prefix|>< .éḍ̇dⅣßfi", "tokens": 36, "pieces": ["d字", "\r\n\r\n", "𐞁camel", "Case", "HTTPServer'ſ", "́", "9", "aåå", "<|", "fim", "_prefix", "|><", " ", " .", "éḍ̇d", "Ⅳ", "ßfi"]} +{"text": "'VEⅣåst \n fi\"( \nİ٣٤٥٦0camelCase٣٤٥٦'D'TA/\r\nBdt字३", "tokens": 31, "pieces": ["'VE", "Ⅳ", "åst", " \n", " fi", "\"(", " \n", "İ", "٣٤٥", "٦0", "camel", "Case", "٣٤٥", "٦", "'D'T", "A", "/\r\n", "Bdt字", "३"]} +{"text": "字\rcamelCase\r\n\r\néåt#$%fi\n/,!HTTPServer#$% 'ss", "tokens": 26, "pieces": ["字", "\r", "camel", "Case", "\r\n\r\n", "éåt", "#$%", "fi", "\n", "/<", "EOT", ">,!", "HTTPServer", "#$%", " ", " '", "ss"]} +{"text": "\t𐞁'll….\rⅣ9\u000bᵃ٣٤٥٦𐞁iOS\n\t…​'M ३\t'S👍🏽 d'", "tokens": 44, "pieces": ["\t𐞁'll", "…", ".\r", "Ⅳ9", "\u000bᵃ", "٣٤٥", "٦", "𐞁i", "OS", "\n", "\t", "…", "​'", "M", " ", "३", "\t", "'S", "👍🏽", " d", "'"]} +{"text": "fi漢👍🏽", "tokens": 5, "pieces": ["fi漢", "👍🏽"]} +{"text": "字<|endoftext|>\r\n>𐞁\t\r\nß AbZ \n 'Sd\r­/\r\n​ \n ", "tokens": 29, "pieces": ["字", "<|", "endoftext", "|>\r\n", ">𐞁", "\t\r\n", "ß", " Ab", "Z", "", " \n", " '", "Sd", "\r", "­/\r\n", "​", " \n", " "]} +{"text": "​Džungla", "tokens": 5, "pieces": ["​Džungla"]} +{"text": "\r\n  éfi㍿Z३Dž", "tokens": 12, "pieces": ["\r\n", " ", " éfi", "㍿Z", "३", "Dž"]} +{"text": "9\u000b'VE'll", "tokens": 5, "pieces": ["9", "\u000b", "'VE'll"]} +{"text": " 🙂!​\n'Mm\n/Bé'D>s\r\n字'M㋿\nEOT's/\r\n fi('Ta/bmd", "tokens": 28, "pieces": [" 🙂!​\n", "'Mm", "\n", "/Bé'D", ">s", "\r\n", "字'M", "㋿\n", "EOT's", "/\r\n", " fi", "('", "Ta", "/bmd"]} +{"text": "/\r\n‍å>B's字'VE'𐞁iOS-ꟲå३Džunglae‍ḍ̇​fi>.", "tokens": 39, "pieces": ["/\r\n", "‍å", ">B's", "字'VE", "'𐞁i", "OS", "-ꟲå", "३", "Džunglae", "‍ḍ̇", "​fi", ">."]} +{"text": " \nABCiOS३İ\nABCDžs", "tokens": 11, "pieces": [" \n", "ABCi", "OS", "३", "İ", "\n", "ABCDžs"]} +{"text": "e字,\r\n\r\ncamelCase'llcamelCaseåHTTPServereعtEOT'ReABCé>", "tokens": 22, "pieces": ["e字", ",\r\n\r\n", "camel", "Case'll", "camel", "Caseå", "HTTPServereعt", "EOT'Re", "ABCé", ">"]} +{"text": "009<|endoftext|>a'll\" 'reḍ̇Džungla /\r\n émİꟲ'M", "tokens": 35, "pieces": ["009", "<|", "endoftext", "|>", "a'll", "\"<", "META", "_START", ">", " ", " '", "reḍ̇", "Džungla", " ", "/\r\n", " ", " ém", "İꟲ'M"]} +{"text": "\n漢'VE \n ", "tokens": 6, "pieces": ["\n", "漢'VE", " \n", " "]} +{"text": "\n/\t\nEOTꟲ0㋿édéB㋿,", "tokens": 23, "pieces": ["\n", "/<", "META", "_START", ">", "\t\n", "EOTꟲ", "0", "㋿édé", "B", "㋿,"]} +{"text": "ᵃ漢 \n­\n/aBédéaB \n#$%fiZ\r\n!!HTTPServer'ſ'llcamelCase", "tokens": 26, "pieces": ["ᵃ漢", " \n", "­\n/", "a", "Bédéa", "B", " \n", "#$%", "fi", "Z", "\r\n", "!!", "HTTPServer'ſ", "'llcamel", "Case"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'llİᵃ9😀🏽s‍'ſⅣ㋿<|endoftext|>ᵃ're🙂𐞁'ſ…\u000beعßB're", "tokens": 44, "pieces": ["'ll", "İᵃ", "9", "😀🏽", "s", "‍'", "ſ", "Ⅳ", "㋿<|", "endoftext", "|>", "ᵃ're", "🙂𐞁'ſ", "…", "\u000beعß", "B're"]} +{"text": "'Re!camelCase字👍🏽(å-!\"٣٤٥٦\u000b/½/​ع<ᵃ \n <|endoftext|>'Se 12345678'll🙂<|fim_prefix|>éDžunglaé'MHTTPServer!!‍!", "tokens": 62, "pieces": ["'Re", "!camel", "Case字", "👍🏽(", "å", "-!\"", "٣٤٥", "٦", "\u000b", "/", "½", "/​", "ع", "<ᵃ", " \n", " <|", "endoftext", "|>'", "Se", " ", "123", "456", "78", "'ll", "🙂<|", "fim", "_prefix", "|>", "é", "Džunglaé'M", "HTTPServer", "!!‍!"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "'VE㋿", "tokens": 5, "pieces": ["'VE", "㋿"]} +{"text": "
iOS''Refi👍🏽\nfi👍🏽\r\n\r\n!! \n ٣٤٥٦'ll٣٤٥٦t!!…😀🏽ᵃmİⅣ", "tokens": 41, "pieces": ["
i", "OS", "''", "Refi", "👍🏽\n", "fi", "👍🏽\r\n\r\n", "!!", " \n", " ", "٣٤٥", "٦", "'ll", "٣٤٥", "٦", "t", "!!", "…", "😀🏽", "ᵃm", "İ", "Ⅳ"]} +{"text": "…'Me<|fim_prefix|>\r\n\r\n 𐞁camelCase३\u000biOS12345678d('D", "tokens": 27, "pieces": ["…", "'Me", "<|", "fim", "_prefix", "|>\r\n\r\n", " 𐞁camel", "Case", "३", "\u000bi", "OS", "123", "456", "78", "d", "('", "D"]} +{"text": "Džungla\rEOT Džungla-…aB\r\n ", "tokens": 20, "pieces": ["Džungla", "\r", "EOT", " ", " Džungla", "-", "…a", "B", "\r\n", " "]} +{"text": "İå'Sm-!fi'S\u000bſBiOS​ \n >camelCase\"912345678\r\n\r\n\tZ", "tokens": 24, "pieces": ["İå'S", "m", "-!", "fi'S", "\u000bſ", "Bi", "OS", "​", " \n", " >", "camel", "Case", "\"", "912", "345", "678", "\r\n\r\n", "\tZ"]} +{"text": "‍9‍­/\r\n-sé,…!!d(\r\n \nſ#$%漢é#$%!㋿ᵃ½ABC\r\t\r\nZed\n/𐞁 ​ ", "tokens": 45, "pieces": ["‍", "9", "‍­/\r\n", "-sé", ",", "…", "!!", "d", "(\r\n", " \n", "ſ", "#$%", "漢é", "#$%!㋿", "ᵃ", "½", "ABC", "\r\t\r\n", "Zed", "\n", "/𐞁", "", " ​", " "]} +{"text": "HTTPServer'DßAb漢m'll9Dž!0漢'ſḍ̇\u000b-ſa/bfiB\r\n\r\n‍'TDž (Ⅳ/\r\n", "tokens": 41, "pieces": ["HTTPServer'D", "ß", "Ab漢m'll", "9", "Dž", "!", "0", "漢'ſ", "ḍ̇", "\u000b", "-ſa", "/bfi", "B", "\r\n\r\n", "‍'", "TDž", " ", " (", "Ⅳ", "/\r\n"]} +{"text": "!å('T ½a/b​0\r\n.DžunglaHTTPServer'ſ㍿fi㋿!字!​漢😀🏽\tᵃZ!Z字 ", "tokens": 45, "pieces": ["!å", "('", "T", " ", "½", "a", "/b", "​", "0", "\r\n", ".Džungla", "HTTPServer'ſ", "㍿fi", "㋿!", "字", "!​", "漢", "😀🏽", "\tᵃ", "Z", "!Z字", " "]} +{"text": "ع EOT!!0ſ'VEA", "tokens": 9, "pieces": ["ع", " EOT", "!!", "0", "ſ'VE", "A"]} +{"text": "\r<|fim_prefix|>'s \n DžåAb\t­\n३B'sAbs\n😀🏽…", "tokens": 27, "pieces": ["\r", "<|", "fim", "_prefix", "|>'", "s", " \n", " Džå", "Ab", "\t", "­\n", "३", "B's", "Abs", "\n", "😀🏽", "…"]} +{"text": "m👍🏽e\r\n\r\n9‍a/bⅣ<🙂́(ع(<|fim_prefix|>­- \n \"m<|endoftext|>>㋿\t
­-", " \n", " \"", "m", "<|", "endoftext", "|>>㋿", "\t", "
", "㍿ⅣHTTPServer \nABC", "tokens": 24, "pieces": [" \n", "!!", "İ", "\r\n\r\n", "٣٤٥", "٦", " ", "<|", "fim", "_prefix", "|>㍿", "Ⅳ", "HTTPServer", " \n", "ABC"]} +{"text": "'ll,🙂camelCase…­😀🏽-m'D'D ⅣB'Sᵃ😀🏽'HTTPServer", "tokens": 29, "pieces": ["'ll", ",🙂", "camel", "Case", "…", "­😀🏽-", "m'D", "'D", " ", "Ⅳ", "B'S", "ᵃ", "😀🏽'", "HTTPServer"]} +{"text": "é.Ⅳé
Ⅳ> \n12345678a/b👍🏽Dž\r\n\r\n\r\nBs𐞁<'T>…9,9字 'reᵃᵃ!>­ſ\"Džungla漢", "tokens": 58, "pieces": ["é", ".", "Ⅳ", "é", "
", "Ⅳ", ">", " \n", "123", "456", "78", "a", "/b", "👍🏽", "Dž", "\r\n\r\n\r\n", "Bs𐞁", "<'", "T", ">", "…", "9", ",", "9", "字", " '", "reᵃᵃ", "!>­", "ſ", "\"Džungla漢"]} +{"text": "aEOT'ſ'ſ s \n iOS \n ᵃ'refiABC३\n'D", "tokens": 22, "pieces": ["a", "EOT'ſ", "'ſ", " s", " \n", " i", "OS", " \n", " ᵃ're", "fi", "ABC", "३", "\n", "'D"]} +{"text": "tiOS'SA\r\n漢9‍ \t12345678'M٣٤٥٦>𐞁३٣٤٥٦\n/!!d漢'VEHTTPServer'reAABC", "tokens": 48, "pieces": ["ti", "OS'S", "A", "\r\n", "漢", "9", "‍", " ", "\t", "123", "456", "78", "'M", "٣٤٥", "٦", ">𐞁", "३٣٤", "٥٦", "\n", "/!!", "d漢'VE", "HTTPServer're", "AABC"]} +{"text": "'S'T‍12345678字ABC", "tokens": 8, "pieces": ["'S'T", "‍", "123", "456", "78", "字", "ABC"]} +{"text": "㍿​'DB!!<Dž😀🏽'D漢\n//tAb  ᵃ'ReHTTPServerfi\r\n字saBa字iOS'll#$%Z½㋿", "tokens": 47, "pieces": ["㍿​'", "DB", "!!<", "Dž", "😀🏽'", "D漢", "\n", "/<", "META", "_START", ">/", "t", "Ab", " ", " ᵃ'Re", "HTTPServerfi", "\r\n", "字sa", "Ba字i", "OS'll", "#$%", "Z", "½", "㋿"]} +{"text": "s,\niOS!#$%A\r…😀🏽,a/bZ's­mZḍ̇३'VE\r\n/'ll\u000b.HTTPServereعİm㋿'S'M'Re", "tokens": 45, "pieces": ["s", ",\n", "i", "OS", "!#$%", "A", "\r", "…", "😀🏽,", "a", "/b", "Z's", "­m", "Zḍ̇", "३", "'VE", "\r\n", "/'", "ll", "\u000b", ".HTTPServereع", "İm", "㋿'", "S'M", "'Re"]} +{"text": "ßt 👍🏽́ع'VE's٣٤٥٦!!12345678<|endoftext|>\r/\r\n½ ḍ̇", "tokens": 31, "pieces": ["ßt", " 👍🏽́", "ع'VE", "'s", "٣٤٥", "٦", "!!", "123", "456", "78", "<|", "endoftext", "|>\r/\r\n", "½", " ḍ̇"]} +{"text": "\n/12345678camelCase/٣٤٥٦EOT'ſ's\n/<\ra/biOSİ!!😀🏽a\r\n\r\n\r\n!!åééa㍿a/b", "tokens": 44, "pieces": ["\n", "/", "123", "456", "78", "camel", "Case", "/", "٣٤٥", "٦", "EOT'ſ", "'s", "\n", "/<\r", "a", "/bi", "OSİ", "!!😀🏽", "a", "\r\n\r\n\r\n", "!!", "åééa", "㍿a", "/b"]} +{"text": "👍🏽½Ⅳ
'M \n字 'T ,½́/é'ſ'", "tokens": 21, "pieces": ["👍🏽", "½Ⅳ", "
", "'M", " \n", "字", " ", "'T", " ", " ,", "½", "́", "/é'ſ", "'"]} +{"text": "\rZ'sEOTHTTPServerfi0½AbHTTPServera/b<|endoftext|>…'Re\n<|fim_prefix|>-", "tokens": 36, "pieces": ["\r", "Z's", "EOTHTTPServer", "fi", "0½", "Ab", "HTTPServera", "/b", "<|", "endoftext", "|>", "…", "'Re", "\n", "<|", "fim", "_prefix", "|>-"]} +{"text": "s'D'rét漢é<|endoftext|>…HTTPServerHTTPServerABC9!!'re½'D😀🏽,ß漢Z", "tokens": 35, "pieces": ["s'D", "'rét漢é", "<|", "endoftext", "|>", "…HTTPServer", "HTTPServer", "ABC", "9", "!!'", "re", "½", "'D", "😀🏽,", "ß漢", "Z"]} +{"text": "😀🏽'VEt३ \t9'Reé́!,𐞁́<|endoftext|><|fim_prefix|>㋿9>­́'ReZ're/\r\n 👍🏽'ſ
é'T㍿\r\n\r\n", "tokens": 61, "pieces": ["😀🏽'", "VEt", "३", " ", "\t", "9", "'Reé́", "!,", "𐞁́", "<|", "endoftext", "|><|", "fim", "_prefix", "|>㋿", "9", ">­́'", "Re", "Z're", "/\r\n", " ", "👍🏽'", "ſ", "
é'T", "㍿\r\n\r\n"]} +{"text": "
<|endoftext|>", "tokens": 8, "pieces": ["
", "<|", "endoftext", "|>"]} +{"text": "s.👍🏽é'M漢ꟲ'ß \n #$%0\r\n‍(", "tokens": 25, "pieces": ["s", ".👍🏽", "é'M", "漢ꟲ", "'ß", " \n", " #$%", "0", "\r\n", "‍("]} +{"text": "å漢EOT👍🏽🙂12345678\n/#$% \n'S Džungla'D🙂Ⅳ12345678fi🙂EOT'", "tokens": 35, "pieces": ["å漢", "EOT", "👍🏽🙂", "123", "456", "78", "\n", "/#$%", " \n", "'S", " Džungla'D", "🙂", "Ⅳ12", "345", "678", "fi", "🙂EOT", "'"]} +{"text": "३ſ'VEع\"'VE𐞁­'s٣٤٥٦'SZ/!!'ſꟲ<'M 
/\r\nHTTPServerfi", "tokens": 35, "pieces": ["३", "ſ'VE", "ع", "\"'", "VE𐞁", "­'", "s", "٣٤٥", "٦", "'SZ", "/!!'", "ſꟲ", "<'", "M", " ", "
", "/\r\n", "HTTPServerfi"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\u000b're'SA'Re𐞁9\r\n\n/,m\r\n\r\ncamelCase𐞁३é\r\n\r\nع­​…\t'SA
aBſ…ḍ̇Z/㍿Z‍a/b", "tokens": 53, "pieces": ["\u000b", "'re'S", "A'Re", "𐞁", "9", "\r\n\n", "/,", "m", "\r\n\r\n", "camel", "Case𐞁", "३", "é", "\r\n\r\n", "ع", "­​", "…", "\t", "'SA", "
", "a", "Bſ", "…ḍ̇", "Z", "/㍿", "Z", "‍a", "/b"]} +{"text": "-‍.<|endoftext|>́Ⅳꟲa/b字'S#$%m'T
½", "tokens": 25, "pieces": ["-‍.<|", "endoftext", "|>́", "Ⅳ", "ꟲa", "/b字'S", "#$%", "m'T", "
", "½"]} +{"text": "'T,,e0/ \n' B'S'SDžunglaABCZ\u000b.\r\n9\ts\r\n\r\n'M㋿㋿12345678mſ!!
aB́>-
/\r\n👍🏽", "tokens": 48, "pieces": ["'T", ",,", "e", "0", "/", " \n", "'<", "EOT", ">", " ", " B'S", "'SDžungla", "ABCZ", "\u000b", ".\r\n", "9", "\ts", "\r\n\r\n", "'M", "㋿㋿", "123", "456", "78", "mſ", "!!", "
a", "B́", ">-", "
", "/\r\n", "👍🏽"]} +{"text": "ſe­'VE½'re\"́㍿'.0>d,٣٤٥٦Ab'\r\n ZZs're漢Ⅳ­A \n\u000bZ0aB​/\r\nᵃEOT", "tokens": 44, "pieces": ["ſe", "­'", "VE", "½", "'re", "\"́", "㍿'.", "0", ">d", ",", "٣٤٥", "٦", "Ab", "'\r\n", " ", " ZZs're", "漢", "Ⅳ", "­A", " \n", "\u000bZ", "0", "a", "B", "​/\r\n", "ᵃ", "EOT"]} +{"text": "…,😀🏽\n//\r\nHTTPServer!㋿Z'ſ-", "tokens": 18, "pieces": ["…", ",😀🏽\n//\r\n", "HTTPServer", "!㋿", "Z'ſ", "-"]} +{"text": "Dž<|endoftext|>…HTTPServer-ꟲ'VE 字 ßaBDžع9\u000bEOT9'.ꟲ\r\nssaB  Dž", "tokens": 44, "pieces": ["Dž", "<|", "endoftext", "|>", "…HTTPServer", "-ꟲ'VE", " 字", " ßa", "BDžع", "9", "\u000bEOT", "9", "'.", "ꟲ", "\r\n", "ssa", "B", " ", " Dž"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "/\r\nİe‍#$%'ScamelCase'llEOT/'Dé
字字fi\tA\"'MABC/ع'reꟲ­ꟲ­aEOT𐞁字Džꟲßſ", "tokens": 52, "pieces": ["/\r\n", "İe", "‍#$%'", "Scamel", "Case'll", "EOT", "/'", "Dé", "
", "字字fi", "\tA", "\"'", "MABC", "/ع're", "ꟲ", "­ꟲ", "­a", "EOT𐞁字Džꟲßſ"]} +{"text": "'VEᵃİ're.'s, !'ſ\r\r\n\r\n \n 9'T-\r\nm​camelCase/\r\n \n \r\n㋿٣٤٥٦字'ſ'VE>'s \n", "tokens": 46, "pieces": ["'VEᵃ", "İ're", ".'", "s", ",", " ", "!'", "ſ", "\r\r\n\r\n \n", " ", "9", "'T", "-\r\n", "m", "​camel", "Case", "/\r\n", " \n \r\n", "㋿", "٣٤٥", "٦", "字'ſ", "'VE", ">'", "s", " \n", ""]} +{"text": "'M\rå<|fim_prefix|> \n9ḍ̇e/ᵃᵃcamelCase\t‍'D<|endoftext|>漢(", "tokens": 44, "pieces": ["'M", "\r", "å", "<|", "fim", "_prefix", "|>", " \n", "9", "ḍ̇e", "/ᵃᵃcamel", "Case", "\t", "‍'", "D", "<|", "endoftext", "|>", "漢", "("]} +{"text": "d-Ab \nḍ̇­aB漢' …㋿>", "tokens": 17, "pieces": ["d", "-Ab", " \n", "ḍ̇", "­a", "B漢", "'", " ", "…", "㋿>"]} +{"text": "𐞁camelCase<|fim_prefix|>é'll ", "tokens": 15, "pieces": ["𐞁camel", "Case", "<|", "fim", "_prefix", "|>", "é'll", " "]} +{"text": "<👍🏽camelCasecamelCase", "tokens": 8, "pieces": ["<👍🏽", "camel", "Casecamel", "Case"]} +{"text": "EOT-.́½ꟲ!!漢ABCd/\r\n!Džungla'lléABC'Reſſ'Mfim\"'Z'saBEOT'll‍Z", "tokens": 40, "pieces": ["EOT", "-.́", "½", "ꟲ", "!!", "漢ABCd", "/\r\n", "!Džungla'll", "é", "ABC'Re", "ſſ'M", "fi", "m", "\"'", "Z's", "a", "BEOT'll", "‍Z"]} +{"text": "å 'lla/b\n/\r\n㍿\r'  \n Dž<|fim_prefix|>BAb字😀🏽Džungla\n/d-'Då\r'sᵃcamelCaseDž,\n'MfiZ\rA'ſå", "tokens": 61, "pieces": ["å", " '", "lla", "/b", "\n", "/\r\n", "㍿<", "META", "_START", ">\r", "'", "  \n", " Dž", "<|", "fim", "_prefix", "|>", "BAb字", "😀🏽", "Džungla", "\n", "/d", "-'", "Då", "\r", "'sᵃcamel", "Case", "Dž", ",\n", "'Mfi", "Z", "\r", "A'ſ", "å"]} +{"text": "👍🏽's", "tokens": 5, "pieces": ["👍🏽'", "s"]} +{"text": ".'Re😀🏽Dž", "tokens": 7, "pieces": [".'", "Re", "😀🏽", "Dž"]} +{"text": "camelCaseᵃé🙂a'll mᵃABC<|fim_prefix|>", "tokens": 20, "pieces": ["camel", "Caseᵃé", "🙂a'll", " mᵃ", "ABC", "<|", "fim", "_prefix", "|>"]} +{"text": "عع\r\naBs\r\n\r\n'Re", "tokens": 10, "pieces": ["عع", "\r\n", "a", "Bs", "\r\n\r\n", "'Re"]} +{"text": "
漢a ß\r\n\r\n9㋿Ⅳ'Re'sABC‍\tABC<'Re㍿㋿漢,'re\"", "tokens": 30, "pieces": ["
漢a", " ß", "\r\n\r\n", "9", "㋿", "Ⅳ", "'Re's", "ABC", "‍", "\tABC", "<'", "Re", "㍿㋿", "漢", ",'", "re", "\""]} +{"text": "'ſ३fié- ́!B'T𐞁\n/!<👍🏽0'T0DžcamelCase\"!'​ \n'S .\r<|fim_prefix|>åmABC😀🏽 ", "tokens": 59, "pieces": ["'ſ", "३", "fié", "-", " ́", "!<", "META", "_START", ">B'T", "𐞁", "\n", "/!<👍🏽", "0", "'T", "0", "Džcamel", "Case", "\"!'​", " \n", "'S", " ", ".\r", "<|", "fim", "_prefix", "|>", "åm", "ABC", "😀🏽", " "]} +{"text": "B㍿İ/s𐞁
", "tokens": 11, "pieces": ["B", "㍿İ", "/s𐞁", "
"]} +{"text": "…<|fim_prefix|>HTTPServer'Re३' \n 'sABC<|fim_prefix|>s \n ',", "tokens": 29, "pieces": ["…", "<|", "fim", "_prefix", "|>", "HTTPServer'Re", "३", "'", " \n", " '", "s", "ABC", "<|", "fim", "_prefix", "|>", "s", " \n", " '<", "META", "_START", ">,"]} +{"text": "!!é\u000ba/bß'llfi字漢(Abé\u000b\r…‍d٣٤٥٦ \r'S.é'llEOT.AbcamelCase-\u000b'Re", "tokens": 39, "pieces": ["!!", "é", "\u000ba", "/bß'll", "fi字漢", "(Abé", "\u000b\r", "…", "‍d", "٣٤٥", "٦", " \r", "'S", ".é'll", "EOT", ".Abcamel", "Case", "-", "\u000b", "'Re"]} +{"text": "㋿.Dž
 /\r\n\t३/\r\niOS‍𐞁㍿́\r\n\r\n٣٤٥٦'0😀🏽<|fim_prefix|>㋿ß's㍿'ll ", "tokens": 57, "pieces": ["㋿.", "Dž", "
", " ", "/\r\n", "\t", "३", "/\r\n", "i", "OS", "‍𐞁", "㍿́", "\r\n\r\n", "٣٤٥", "٦", "'", "0", "😀🏽<|", "fim", "_prefix", "|>㋿", "ß", "'", "s", "㍿'", "ll", " "]} +{"text": ">,iOSHTTPServer\"(­ᵃt😀🏽", "tokens": 14, "pieces": [">,", "i", "OSHTTPServer", "\"(­", "ᵃt", "😀🏽"]} +{"text": "'ll", "tokens": 1, "pieces": ["'ll"]} +{"text": " \n…'MaBt'T'M字'lld\tEOT!! ‍Ⅳ\r\n\r\nEOT😀🏽", "tokens": 28, "pieces": [" \n", "…", "'Ma", "Bt'T", "'M字'll", "d", "\tEOT", "!!", " ", " ‍", "Ⅳ", "\r\n\r\n", "EOT", "😀🏽<", "META", "_START", ">"]} +{"text": "iOS½,a/b',EOT\"㋿
'sEOT", "tokens": 16, "pieces": ["i", "OS", "½", ",a", "/b", "',", "EOT", "\"㋿", "
", "'s", "EOT"]} +{"text": "ABC<|fim_prefix|>#$% <|endoftext|>İéſAb'VEZ㍿0㋿㍿t9'VE😀🏽a/b🙂Džungla0'ReDžungla漢#$%'T🙂😀🏽\t", "tokens": 67, "pieces": ["ABC", "<|", "fim", "_prefix", "|>#$%", " ", " <|", "endoftext", "|>", "İéſ", "Ab'VE", "Z", "㍿", "0", "㋿㍿<", "META", "_START", ">t", "9", "'VE", "😀🏽", "a", "/b", "🙂Džungla", "0", "'Re", "Džungla漢", "#$%'", "T", "🙂😀🏽", "\t"]} +{"text": "㋿<|endoftext|>0ḿs\"-<|endoftext|>mᵃ #$%\"\taB're㍿s/'Re漢👍🏽Ⅳ'Re३. Źᵃ", "tokens": 51, "pieces": ["㋿<|", "endoftext", "|>", "0", "ḿs", "\"-<|", "endoftext", "|>", "mᵃ", " ", "#$%\"", "\ta", "B're", "㍿s", "/'", "Re漢", "👍🏽", "Ⅳ", "'Re", "३", ".", " Źᵃ"]} +{"text": "ḍ̇\t\n/'re!!  å😀🏽åmt!\né<|fim_prefix|>ع-\"''T\t'T́/\r\n!/#$%😀🏽-㋿㋿ d", "tokens": 53, "pieces": ["ḍ̇", "\t\n", "/'", "re", "!!", " ", " å", "😀🏽", "åmt", "!\n", "é", "<|", "fim", "_prefix", "|>", "ع", "-\"''", "T", "\t", "'T́", "/\r\n", "!/#$%😀🏽-㋿㋿", " d", ""]} +{"text": " ḍ̇…fiBe\n/\r\nAbⅣ\n/'ll", "tokens": 16, "pieces": [" ḍ̇", "…fi", "Be", "\n", "/\r\n", "Ab", "Ⅳ", "\n", "/'", "ll"]} +{"text": "HTTPServer\n/\n-éaB #$%9½.>'T \n\r\n\r\n", "tokens": 21, "pieces": ["HTTPServer", "\n", "/\n", "-éa", "B", " #$%", "9½", ".>'", "T", " \n", "\r\n\r\n"]} +{"text": "ع-३٣٤٥٦\r\n\r\n012345678t३Ⅳ,漢ſiOS𐞁'Tİå'S12345678EOT'D𐞁ma/béⅣ'D😀🏽İ0漢/\r\n a/b
å", "tokens": 58, "pieces": ["ع", "-", "३٣٤", "٥٦", "\r\n\r\n", "012", "345", "678", "t", "३Ⅳ", ",漢ſi", "OS𐞁'T", "İå'S", "123", "456", "78", "EOT'D", "𐞁ma", "/bé", "Ⅳ", "'D", "😀🏽", "İ", "0", "漢", "/\r\n", " ", " a", "/b", "
å"]} +{"text": "Džİ ᵃ‍'MEOT'll!>…'sAba/bع \n0٣٤٥٦­/ \nß'sA ‍.\r,‍", "tokens": 38, "pieces": ["Džİ", " ᵃ", "‍'", "MEOT'll", "!>", "…", "'s", "Aba", "/bع", " \n", "0٣٤", "٥٦", "­/", " \n", "ß's", "A", " ", "‍.\r", ",‍"]} +{"text": " e'M'Stſ'ſ ­Džungla​<|endoftext|>ꟲHTTPServer'D/\r\n​<|fim_prefix|>🙂ꟲ(.>/\r\n'Re'll\té𐞁字 \nABC0're", "tokens": 59, "pieces": [" e'M", "'Stſ'ſ", " ", "­Džungla", "​<|", "endoftext", "|>", "ꟲHTTPServer'D", "/\r\n", "​<|", "fim", "_prefix", "|>🙂", "ꟲ", "(.>/\r\n", "'", "Re'll", "\té𐞁字", " \n", "ABC", "0", "'re"]} +{"text": "'ll(
> \n t🙂\n/>\r\n,éꟲ aHTTPServer>éAb'VE…", "tokens": 26, "pieces": ["'ll", "(", "
", ">", " \n", " t", "🙂\n/", ">\r\n", ",éꟲ", " a", "HTTPServer", ">é", "Ab'VE", "…"]} +{"text": "!!½𐞁<­ᵃ\"ع ", "tokens": 14, "pieces": ["!!", "½", "𐞁", "<­", "ᵃ", "\"ع", " "]} +{"text": "\"(­­‍‍漢#$%'D\r\n\r\n \n 'llfi0eİ!aB\rå", "tokens": 22, "pieces": ["\"(­­‍‍", "漢", "#$%'", "D", "\r\n\r\n \n", " '", "llfi", "0", "e", "İ", "!a", "B", "\r", "å"]} +{"text": "'M'T漢ABĆ㍿a/bHTTPServer½a/bém#$%A㍿ABC>
Ⅳ३'ſ𐞁'>aBABCB㍿ \n tABC sa/bm'reᵃ12345678", "tokens": 59, "pieces": ["'M'T", "漢", "ABC", "́", "㍿a", "/b", "HTTPServer", "½", "a", "/bém", "#$%", "A", "㍿ABC", ">", "
", "Ⅳ३", "'ſ𐞁", "'>", "a", "BABCB", "㍿", " \n", " t", "ABC", " sa", "/bm're", "ᵃ", "123", "456", "78"]} +{"text": "ع 
é́ꟲ-'M'S\"\t12345678>sİå", "tokens": 20, "pieces": ["ع", " ", "
é́ꟲ", "-'", "M'S", "\"", "\t", "123", "456", "78", ">s", "İå"]} +{"text": " \n /\r\n­'Dß👍🏽t\n/Ⅳ#$% B9​B\"Ⅳ'Re½A<|endoftext|>字́", "tokens": 37, "pieces": [" \n", " /\r\n", "­'", "Dß", "👍🏽", "t", "\n", "/", "Ⅳ", "#$%", " B", "9", "​B", "\"", "Ⅳ", "'Re", "½", "A", "<|", "endoftext", "|>", "字́"]} +{"text": "e/\r\n \n ३sᵃsDžunglaDžß \n  'Re!ḍ̇́'ſ'D'sABCABC<|endoftext|>ABC0\r\n\r\n\r\n…!'VE👍🏽𐞁'Re'ſé,", "tokens": 63, "pieces": ["e", "/\r\n", " \n", " ", "३", "sᵃs", "Džungla", "Džß", " \n", " ", " ", "'Re", "!ḍ̇́'ſ", "'D's", "ABCABC", "<|", "endoftext", "|>", "ABC", "0", "\r\n\r\n\r\n", "…", "!'", "VE", "👍🏽", "𐞁", "'", "Re'ſ", "é", ","]} +{"text": "\n/B'VE३'a/b\rİ\r­字  Džunglaſ
ꟲé/\r\nHTTPServer>\">", "tokens": 30, "pieces": ["\n", "/B'VE", "३", "'a", "/b", "\r", "İ", "\r", "­字", " ", " Džunglaſ", "
ꟲé", "/\r\n", "HTTPServer", ">\">"]} +{"text": "'MHTTPServer\r\n\r\n㍿téᵃt㍿ع!se0\r\nᵃ/\r\nße'Mé😀🏽‍\t́Dž😀🏽,#$%", "tokens": 44, "pieces": ["'MHTTPServer", "\r\n\r\n", "㍿téᵃt", "㍿ع", "!", "se", "0", "\r\n", "ᵃ", "/\r\n", "ße'M", "é", "😀🏽‍", "\t́", "Dž", "😀🏽,#$%"]} +{"text": " EOT'll'ſ́ⅣDžungla
 \n(.a/b\n/字s", "tokens": 24, "pieces": [" EOT'll", "'ſ́", "", "Ⅳ", "Džungla", "
 \n", "(.", "a", "/b", "\n", "/字s"]} +{"text": "‍'re\r\n\r\n字'lla-'s<12345678'D'MiOSé­ \n", "tokens": 24, "pieces": ["‍'", "re", "\r\n\r\n", "字'll", "a", "-'", "s", "<", "123", "456", "78", "'D'M", "i", "OSé", "­", " \n"]} +{"text": "A'VE'M/ 👍🏽Džungla12345678'ſ9m \n \n\n/daDžungla'T'S<|fim_prefix|>ſaB<|endoftext|>m's𐞁‍Z🙂é'S'reBfi", "tokens": 71, "pieces": ["A'VE", "'", "M", "/", " ", "👍🏽", "Džungla", "123", "456", "78", "'ſ", "", "9", "m", " \n \n\n", "/da", "Džungla'T", "'S", "<|", "fim", "_prefix", "|>", "ſ", "a", "B", "<|", "endoftext", "|>", "m's", "𐞁", "‍Z", "🙂é'S", "'re", "Bfi"]} +{"text": "……''re\r\n٣٤٥٦-B<|fim_prefix|>😀🏽(-­'ll(
iOS<|fim_prefix|>'s👍🏽
\r\ne\r\n\r\n​a/b 𐞁,", "tokens": 52, "pieces": ["…", "…", "''", "re", "\r\n", "٣٤٥", "٦", "-B", "<|", "fim", "_prefix", "|>😀🏽(-­'", "ll", "(", "
i", "OS", "<|", "fim", "_prefix", "|>'", "s", "👍🏽", "
\r\n", "e", "\r\n\r\n", "​a", "/b", " 𐞁", ","]} +{"text": "­", "tokens": 1, "pieces": ["­"]} +{"text": "ß\r\n🙂m'T9camelCase'Re\u000bAb'M ", "tokens": 13, "pieces": ["ß", "\r\n", "🙂m'T", "9", "camel", "Case'Re", "\u000bAb'M", " "]} +{"text": "åⅣ<|endoftext|>", "tokens": 11, "pieces": ["å", "Ⅳ", "<|", "endoftext", "|>"]} +{"text": " \n camelCaseß\rḍ̇.٣٤٥٦ſ३३EOT", "tokens": 18, "pieces": [" \n", " camel", "Caseß", "\r", "ḍ̇", ".", "٣٤٥", "٦", "ſ", "३३", "EOT"]} +{"text": "٣٤٥٦aB/\r\n½½'VEDž/\r\nm字Am👍🏽9dᵃ\n,\"camelCase'M漢'sß,A'Tع𐞁Z's fiꟲå'D", "tokens": 56, "pieces": ["٣٤٥", "٦", "a", "B", "/\r\n", "½½", "'VEDž", "/\r\n", "m字", "Am", "👍🏽", "9", "dᵃ", "\n", ",\"", "camel", "Case'M", "漢's", "ß", ",A'T", "ع𐞁", "Z's", " ", " fiꟲå'D"]} +{"text": "/\r\nfi,0", "tokens": 8, "pieces": ["/\r\n", "fi", ",", "0"]} +{"text": "ᵃ-İe­🙂s'Tå're\r\n\r\nA\r\n\r\n👍🏽ḍ̇>", "tokens": 23, "pieces": ["ᵃ", "-İe", "­🙂", "s'T", "å're", "\r\n\r\n", "A", "\r\n\r\n", "👍🏽", "ḍ̇", ">"]} +{"text": " \n B ٣٤٥٦👍🏽👍🏽,'Dᵃ<|endoftext|><|fim_prefix|>iOS'll/fi'D'VEDžungla \n ꟲsDžZ👍🏽𐞁", "tokens": 58, "pieces": [" \n", " B", " ", "٣٤٥", "٦", "👍🏽👍🏽,'", "Dᵃ", "<|", "endoftext", "|><|", "fim", "_prefix", "|>", "i", "OS'll", "/fi'D", "'VEDžungla", " \n", " ꟲs", "DžZ", "👍🏽", "𐞁"]} +{"text": "d½\n/iOS .B-\n'́字iOS<|fim_prefix|>😀🏽camelCase‍å३½ꟲ'Sfi a/b,e'reAb/\r\ńDžcamelCase'S", "tokens": 51, "pieces": ["d", "½", "\n", "/i", "OS", " ", " <", "META", "_START", ">.", "B", "-\n", "'́字i", "OS", "<|", "fim", "_prefix", "|>😀🏽", "camel", "Case", "‍å", "३½", "ꟲ'S", "fi", " a", "/b", ",e're", "Ab", "/\r\n", "́Džcamel", "Case'S"]} +{"text": "'M''T(\u000bé\r\n dⅣ're‍'a(𐞁㍿३a/b\n(m\r\n\r\n/\r\n\r.‍>/\r\n \n /\r\n'ReDžZ", "tokens": 44, "pieces": ["'M", "''", "T", "(", "\u000bé", "\r\n", " d", "Ⅳ", "'re", "‍'", "a", "(", "𐞁", "㍿", "३", "a", "/b", "\n", "(m", "\r\n\r\n", "/\r\n\r", ".‍>/\r\n", " \n", " /\r\n", "'Re", "DžZ"]} +{"text": "́ \r\n", "tokens": 2, "pieces": ["́", " \r\n"]} +{"text": "#$%\n\n.ع́'Rea٣٤٥٦\r\né-/\r\né ", "tokens": 18, "pieces": ["#$%\n\n", ".ع́'Re", "a", "٣٤٥", "٦", "\r\n", "é", "-/\r\n", "é", " "]} +{"text": "<|fim_prefix|>", "tokens": 6, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "ABCDžunglaaB9camelCase
å>s👍🏽 'TADžungla­\r.'T👍🏽,", "tokens": 36, "pieces": ["ABCDžunglaa", "B", "9", "camel", "Case", "
å", ">s", "👍🏽", " '", "TADžungla", "­\r", ".'", "T", "👍🏽,"]} +{"text": " \t'D#$%Ab/…", "tokens": 9, "pieces": [" ", "\t", "'D", "#$%", "Ab", "/", "…"]} +{"text": "'DHTTPServermḍ̇<Džungla \"åA漢𐞁㋿Džungla>ſEOT٣٤٥٦ⅣA'sAsfiḍ̇sZ\ts0", "tokens": 58, "pieces": ["'DHTTPServermḍ̇", "<Džungla", " ", " \"", "å", "A漢𐞁", "㋿Džungla", ">ſ", "EOT", "٣٤٥", "٦Ⅳ", "A's", "As", "fiḍ̇s", "Z", "\ts", "", "0"]} +{"text": "mİ漢 🙂㍿\n/e'ſ …
camelCaseaBḍ̇Ⅳ", "tokens": 25, "pieces": ["m", "İ漢", " ", "🙂㍿\n/", "e'ſ", " …", "
camel", "Casea", "Bḍ̇", "Ⅳ"]} +{"text": "𐞁(👍🏽漢Dž… 's>ꟲåfi३'T
d!!>\n/t\tß½ⅣAås\na/b'M३!漢a/b", "tokens": 56, "pieces": ["𐞁", "(👍🏽", "漢", "Dž", "…", "", " ", "'s", ">ꟲå", "fi", "३", "'T", "
d", "!!>\n/", "t", "\tß", "½Ⅳ", "Aås", "\n", "a", "/b'M", "३", "!漢a", "/b"]} +{"text": "12345678 \n ḍ̇'S'", "S", "'ḍ̇\r\n'M0字!!(🙂'reEOTAba/bḍ̇m\n/ ٣٤٥٦\n/-a/b<|fim_prefix|>ß👍🏽åDžungla\r\n\r\n#$%a/b<", "tokens": 59, "pieces": ["<|", "fim", "_prefix", "|>'", "ḍ̇", "\r\n", "'M", "0", "字", "!!(🙂'", "re", "EOTAba", "/bḍ̇m", "\n", "/", " ", "٣٤٥", "٦", "\n", "/-", "a", "/b", "<|", "fim", "_prefix", "|>", "ß", "👍🏽", "å", "Džungla", "\r\n\r\n", "#$%", "a", "/b", "<"]} +{"text": "𐞁<|endoftext|>'½\r\n\r\nḍ̇ ABC字!!ᵃ😀🏽12345678\ré!aB\nAb'T🙂\t\"Dž\t字Džع३\nDž<|endoftext|>t🙂camelCase㋿ḍ̇", "tokens": 68, "pieces": ["𐞁", "<|", "endoftext", "|>'", "½", "\r\n\r\n", "ḍ̇", " ABC字", "!!", "ᵃ", "😀🏽", "123", "456", "78", "\r", "é", "!a", "B", "\n", "Ab'T", "🙂", "\t", "\"Dž", "\t字Džع", "३", "\n", "Dž", "<|", "endoftext", "|>", "t", "🙂camel", "Case", "㋿ḍ̇"]} +{"text": "'VE'Re٣٤٥٦. \n 'THTTPServerA'll३ m🙂#$%𐞁fiiOS
漢\nEOT㍿éiOSa/b0a'
 \n 
", "tokens": 50, "pieces": ["'VE'Re", "٣٤٥", "٦", ".", " \n", " '", "THTTPServer", "A'll", "३", " ", " m", "🙂#$%", "𐞁fii", "OS", "", "
漢", "\n", "EOT", "㍿éi", "OSa", "/b", "0", "a", "'", "
 \n", " 
"]} +{"text": "३Z 9/<|endoftext|>", "tokens": 15, "pieces": ["", "३", "Z", " ", "9", "/<|", "endoftext", "|>"]} +{"text": "'T.½AbİABCḍ̇३iOS'ᵃ'M", "tokens": 17, "pieces": ["'T", ".", "½", "Ab", "İABCḍ̇", "३", "i", "OS", "'ᵃ'M"]} +{"text": "tⅣ'ReiOSḍ̇'M<|endoftext|>,㍿A#$%0\tZ.'‍½a/b9… /\r\n HTTPServerḍ̇/\r\n're''VEHTTPServer \n漢 Ⅳ", "tokens": 57, "pieces": ["t", "Ⅳ", "'Rei", "OSḍ̇'M", "<|", "endoftext", "|>,㍿", "A", "#$%", "0", "", "\tZ", ".'‍", "½", "a", "/b", "9", "… ", " /\r\n", " HTTPServerḍ̇", "/\r\n", "'re", "''", "VEHTTPServer", " \n", "漢", " ", "Ⅳ"]} +{"text": "'D\tiOSt'\u000b\reſa 👍🏽 \n .ᵃ'TeAḍ̇a/b!!", "tokens": 31, "pieces": ["'D", "\ti", "OSt", "'", "\u000b\r", "eſa", " ", " 👍🏽", " \n", " <", "META", "_START", ">.", "ᵃ'T", "e", "Aḍ̇a", "/b", "!!"]} +{"text": "३'s<ḍ̇ ſ0åß𐞁 🙂camelCase12345678", "tokens": 27, "pieces": ["३", "'", "s", "<ḍ̇", " ", " ſ", "0", "åß𐞁", " ", "🙂camel", "Case", "123", "456", "78"]} +{"text": "عEOT<|endoftext|>'Re\r­½a'ReDžungla0ḍ̇aå  'S‍́ Dž🙂", "tokens": 36, "pieces": ["ع", "EOT", "<|", "endoftext", "|>'", "Re", "\r", "­", "½", "a'Re", "Džungla", "0", "ḍ̇aå", " ", " ", "'S", "‍́", " ", " Dž", "🙂"]} +{"text": "‍㍿ᵃé\"<|endoftext|><|fim_prefix|>\ttᵃ​'Ta<|endoftext|>'½ \té字ᵃEOTABC<|fim_prefix|>'s", "tokens": 56, "pieces": ["‍㍿", "ᵃé", "\"<|", "endoftext", "|><|", "fim", "_prefix", "|>", "\ttᵃ", "​'", "Ta", "<|", "endoftext", "|>'", "½", " ", "\té字ᵃ", "EOTABC", "<|", "fim", "_prefix", "|>'", "s"]} +{"text": "'Bd\"ßsa/b .BAb…>é‍es\n/ \nAb9ḍ̇\r\nBd㍿٣٤٥٦EOT/\r\nABC", "tokens": 41, "pieces": ["'Bd", "\"ßsa", "/b", " ", " .", "BAb", "…", ">é", "‍es", "\n", "/", " \n", "Ab", "9", "ḍ̇", "\r\n", "Bd", "㍿", "٣٤٥", "٦", "EOT", "/\r\n", "ABC"]} +{"text": "
fiåꟲ  ꟲ12345678ᵃ…-‍B'Dm>å#$%­/İ ß'ſ'llA.!! /", "tokens": 46, "pieces": ["", "
fiåꟲ", " ", " ꟲ", "123", "456", "78", "ᵃ", "…", "-‍", "B'D", "m", ">å", "#$%­/", "İ", " ß'ſ", "'ll", "A", ".!!", " ", " /"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "عꟲ​/\r\n s🙂'VE👍🏽٣٤٥٦0Džungla㋿/\r\nAb㋿'TDž/0A३٣٤٥٦漢ß9ꟲ", "tokens": 49, "pieces": ["عꟲ", "​/\r\n", " s", "🙂'", "VE", "👍🏽", "٣٤٥", "٦0", "Džungla", "㋿/\r\n", "Ab", "㋿'", "TDž", "/", "0", "A", "३٣٤", "٥٦", "漢ß", "9", "ꟲ"]} +{"text": "'ſ/!!漢/\r\nm👍🏽sfi9ſ\r\n👍🏽's\t!!३camelCaseſfi\rꟲ😀🏽>ś㍿éAb \n ,\t㋿", "tokens": 48, "pieces": ["'ſ", "/!!", "漢", "/\r\n", "m", "👍🏽", "sfi", "9", "ſ", "\r\n", "👍🏽'", "s", "\t", "!!", "३", "camel", "Caseſfi", "\r", "ꟲ", "😀🏽>", "ś", "㍿é", "Ab", " \n", " ,", "\t", "㋿"]} +{"text": "camelCase'llⅣe >12345678\r\n \n ㍿㍿'VEcamelCase​ABC'res𐞁漢ḍ̇ ", "tokens": 37, "pieces": ["camel", "Case'll", "Ⅳ", "e", " ", ">", "123", "456", "78", "\r\n \n", " ㍿㍿'", "VEcamel", "Case", "​ABC're", "s𐞁漢ḍ̇", " "]} +{"text": "eDžungla‍<|fim_prefix|><", "tåé", "Džß", "३", "́t", "-a", "\t", "…ḍ̇𐞁", "9", "/\r\n", "ḍ̇", "👍🏽", "\u000b", "…İ", "
"]} +{"text": "'ll dm/\r\nDž́𐞁<|fim_prefix|>\n/😀🏽́Dž字\n字\r(Z", "tokens": 29, "pieces": ["'ll", " dm", "/\r\n", "Dž́𐞁", "<|", "fim", "_prefix", "|>\n/", "😀🏽́", "Dž字", "\n", "字", "\r", "(Z"]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "٣٤٥٦'T​🙂s​!
'DžHTTPServerZḍ̇\u000b'ſ!'M \u000ba/b​👍🏽-​éDžunglaB \n,/ᵃ३", "tokens": 46, "pieces": ["٣٤٥", "٦", "'T", "​🙂", "s", "​!", "
", "'DžHTTPServer", "Zḍ̇", "\u000b", "'ſ", "!'", "M", " ", "\u000ba", "/b", "​👍🏽-​", "é", "Džungla", "B", " \n", ",/", "ᵃ", "३"]} +{"text": "🙂é\r(
'T>12345678­ 'll Á\reaBḍ̇ \n 'VEdé(Z<‍Ⅳꟲ/😀🏽😀🏽'VE\u000bDž\r\tå𐞁", "tokens": 58, "pieces": ["🙂é", "\r", "(", "
", "'T", ">", "123", "456", "78", "­", " '", "ll", " ", " Á", "\r", "ea", "Bḍ̇", " \n", " '", "VEdé", "(Z", "<‍", "Ⅳ", "ꟲ", "/😀🏽😀🏽'", "VE", "\u000bDž", "\r", "\tå𐞁"]} +{"text": "\u000b\n/'Mſ'ſ'd", "tokens": 8, "pieces": ["\u000b\n", "/'", "Mſ'ſ", "'d"]} +{"text": "<|fim_prefix|>Dž><|fim_prefix|> \n/\r\naiOŚİ\r\n\r'M/\rå'ſ'Tع\n ᵃعABC'Re", "tokens": 39, "pieces": ["<|", "fim", "_prefix", "|>", "Dž", "><|", "fim", "_prefix", "|>", " \n", "/\r\n", "ai", "OŚ", "İ", "\r\n\r", "'M", "/\r", "å'ſ", "'Tع", "\n", " ᵃع", "ABC'Re"]} +{"text": "HTTPServerB३EOTⅣtm12345678'll!!", "tokens": 17, "pieces": ["HTTPServer", "B", "", "३", "EOT", "Ⅳ", "tm", "123", "456", "78", "'ll", "!!"]} +{"text": "EOTcamelCase
३å🙂m-\u000b🙂9Ab٣٤٥٦\u000bDžungla㍿!!iOS\n/<|endoftext|>'TaB٣٤٥٦/\r\n\r\n'sa/b…fi<|fim_prefix|>\n‍EOT", "tokens": 60, "pieces": ["EOTcamel", "Case", "
", "३", "å", "🙂m", "-", "\u000b", "🙂", "9", "Ab", "٣٤٥", "٦", "\u000bDžungla", "㍿!!", "i", "OS", "\n", "/<|", "endoftext", "|>'", "Ta", "B", "٣٤٥", "٦", "/\r\n\r\n", "'sa", "/b", "…fi", "<|", "fim", "_prefix", "|>\n", "‍EOT"]} +{"text": "DžunglaABC𐞁\r /\r\n𐞁 \n t👍🏽Ab'sdméé9'M!!é…ꟲ\t-/\r\n'éé", "tokens": 41, "pieces": ["Džungla", "ABC𐞁", "\r", " /\r\n", "𐞁", " \n", " t", "👍🏽", "Ab's", "dméé", "9", "'M", "!!", "é", "…ꟲ", "\t", "-/\r\n", "'éé"]} +{"text": "ß,ᵃs\r\n
B<9'll \n 🙂é9Z
>iOS<|endoftext|>\r漢ß<|endoftext|>👍🏽٣٤٥٦㋿", "tokens": 53, "pieces": ["ß", ",ᵃs", "\r\n", "
B", "<", "9", "'ll", " \n", " <", "EOT", ">🙂", "é", "9", "Z", "
", ">i", "OS", "<|", "endoftext", "|>\r", "漢ß", "<|", "endoftext", "|>👍🏽", "٣٤٥", "٦", "㋿"]} +{"text": "camelCase'Re,t'rem'ſꟲ🙂\r\nſ 👍🏽fi\r/iOSZ(", "tokens": 31, "pieces": ["camel", "Case'Re", ",t're", "m'ſ", "ꟲ", "🙂<", "META", "_START", ">\r\n", "ſ", " ", "👍🏽", "fi", "\r", "/i", "OSZ", "("]} +{"text": "'re'VE'/\r\n#$%Ⅳ\r\n'DZfiعEOT'Re
t 字😀🏽<|endoftext|>", "tokens": 31, "pieces": ["'re'VE", "'/\r\n", "#$%", "Ⅳ", "\r\n", "'DZfiع", "EOT'Re", "
t", " 字", "😀🏽<|", "endoftext", "|>"]} +{"text": "éHTTPServerᵃ👍🏽\r\n\r\n<Ⅳ're\"0\u000b9👍🏽​ع! ­'T\r\n\r\n ३'VE<|fim_prefix|>​\n/‍\n0\u000b٣٤٥٦Džungla", "tokens": 57, "pieces": ["é", "HTTPServerᵃ", "👍🏽\r\n\r\n", "<", "Ⅳ", "'re", "\"", "0", "\u000b", "9", "👍🏽​", "ع", "!", " ", "­'", "T", "\r\n\r\n", " ", "३", "'VE", "<|", "fim", "_prefix", "|>​\n/", "‍\n", "0", "\u000b", "٣٤٥", "٦", "Džungla"]} +{"text": "' ..!0eſB \n ㋿'ſ's'T'漢\r\nZ\n/Abaé dAbEOT<|endoftext|>́👍🏽", "tokens": 40, "pieces": ["'", " ", "..!", "0", "eſ", "B", " \n", " ㋿'", "ſ's", "'T", "'漢", "\r\n", "Z", "\n", "/Abaé", " ", " d", "Ab", "EOT", "<|", "endoftext", "|>́👍🏽"]} +{"text": "'M/\r\nZHTTPServer <ᵃ \n e\n/fi.d9fi㋿'M字camelCasemd/👍🏽Ⅳ㍿字é😀🏽\r\r\t'T'll", "tokens": 46, "pieces": ["'M", "/\r\n", "ZHTTPServer", " ", "<ᵃ", " \n", " e", "\n", "/fi", ".d", "9", "fi", "㋿'", "M字camel", "Casemd", "/👍🏽", "Ⅳ", "㍿字é", "😀🏽\r\r", "\t", "'T'll"]} +{"text": "🙂ᵃ9ḍ̇‍AbⅣ३߅ſ\"", "tokens": 41, "pieces": ["🙂", "ᵃ", "9", "ḍ̇", "‍Ab", "Ⅳ३", "ß", "…ſ", "\""]} +{"text": "­𐞁EOTDžHTTPServerEOTعa\r\n\r\n", "tokens": 16, "pieces": ["­𐞁EOTDžHTTPServer", "EOTعa", "\r\n\r\n"]} +{"text": "éع!", "tokens": 3, "pieces": ["éع", "!"]} +{"text": "åaDžunglaacamelCase \n'D㋿/\r\n'sfí\n(\r!!EOTfi#$%iOS's'👍🏽(eEOT漢camelCase'S<\"㋿\r\n\u000b'T ", "tokens": 56, "pieces": ["åa", "Džungla", "acamel", "Case", " \n", "'D", "㋿/\r\n", "'sfí", "\n", "(\r", "!!", "EOTfi", "#$%", "i", "OS's", "'👍🏽(", "e", "EOT漢camel", "Case'S", "<\"㋿\r\n", "\u000b", "'T", " "]} +{"text": "\n\r\na/bAb\r
字­ſEOT漢0A", "tokens": 14, "pieces": ["\n\r\n", "a", "/b", "Ab", "\r", "
字", "­ſ", "EOT漢", "0", "A"]} +{"text": "ꟲs​ſ\t ́'ReaBaDž/'ReéⅣAbficamelCase𐞁aå\r\n\r\n<|endoftext|>İ\t \n m\u000b", "tokens": 42, "pieces": ["ꟲs", "​ſ", "\t", " ́'Re", "a", "Ba", "Dž", "/'", "Reé", "Ⅳ", "Abficamel", "Case𐞁aå", "\r\n\r\n", "<|", "endoftext", "|>", "İ", "\t \n", " m", "\u000b"]} +{"text": "㍿(iOS\n/ᵃ㋿ #$%aİétaB 👍🏽A!!('ll#$%!!HTTPServerᵃ́ HTTPServer‍'re", "tokens": 44, "pieces": ["㍿(", "i", "OS", "\n", "/ᵃ", "㋿", " ", "#$%", "a", "İéta", "B", " ", " 👍🏽", "A", "!!('", "ll", "#$%!!", "HTTPServerᵃ́", " HTTPServer", "‍'", "re"]} +{"text": "३Ⅳ!!camelCase'reaAbå​\r👍🏽 😀🏽́ḍ̇ⅣⅣ", "tokens": 28, "pieces": ["३Ⅳ", "!!", "camel", "Case're", "a", "Abå", "​\r", "👍🏽", " ", "😀🏽́", "ḍ̇", "ⅣⅣ"]} +{"text": "İ9mA㍿ \u000b\r\n\r\n<|endoftext|>Z'M'Reİ \n mZ🙂m12345678Ⅳ😀🏽٣٤٥٦\"12345678EOT(EOTB0\n𐞁😀🏽t'llå", "tokens": 64, "pieces": ["İ", "9", "m", "A", "㍿", " \u000b\r\n\r\n", "<|", "endoftext", "|>", "Z'M", "'Re", "İ", " \n", " m", "Z", "🙂m", "123", "456", "78Ⅳ", "😀🏽<", "EOT", ">", "٣٤٥", "٦", "\"", "123", "456", "78", "EOT", "(EOTB", "0", "\n", "𐞁", "😀🏽", "t'll", "å"]} +{"text": "'re'res!'s'll'D'Reé漢''ſ 'Re \n aB\r", "tokens": 18, "pieces": ["'re're", "s", "!'", "s'll", "'D'Re", "é漢", "''", "ſ", " '", "Re", " \n", " a", "B", "\r"]} +{"text": "> \ń­\r\n\r\n12345678/…HTTPServerᵃm\n 'T…,ſ\n/ꟲ/\r\n \nå㋿.DžunglaHTTPServerᵃ\t", "tokens": 55, "pieces": [">", " \n", "́", "­\r\n\r\n", "123", "456", "78", "/", "…HTTPServerᵃm", "\n", " ", " '", "T", "…", ",ſ", "\n", "/ꟲ", "/\r\n", " \n", "å", "㋿.", "Džungla", "HTTPServerᵃ", "\t", ""]} +{"text": ">å,camelCasesA字Z<|fim_prefix|>s🙂fi're'iOS\n/𐞁­m'VE'D'ſ🙂½", "tokens": 35, "pieces": [">å", ",camel", "Cases", "A字", "Z", "<|", "fim", "_prefix", "|>", "s", "🙂fi're", "'i", "OS", "\n", "/𐞁", "­m'VE", "'D'ſ", "🙂", "½"]} +{"text": "\r\n\r\n\"!! fi'S🙂0İ0\u000b👍🏽('S\"EOT", "tokens": 18, "pieces": ["\r\n\r\n", "\"!!", " fi'S", "🙂", "0", "İ", "0", "\u000b", "👍🏽('", "S", "\"EOT"]} +{"text": "ꟲ🙂/\r\nå912345678㍿३ABCſع\n'Tİ'T/\r\n/", "tokens": 26, "pieces": ["ꟲ", "🙂/\r\n", "å", "912", "345", "678", "㍿", "३", "ABCſع", "\n", "'Tİ'T", "/\r\n/", ""]} +{"text": "(tB(<|fim_prefix|>'D\"'re㍿\rDž\r\n'reB.'ſ'M", "tokens": 23, "pieces": ["(t", "B", "(<|", "fim", "_prefix", "|>'", "D", "\"'", "re", "㍿\r", "Dž", "\r\n", "'re", "B", ".'", "ſ'M"]} +{"text": "'漢iOS\n/ \néd漢,é\r\"m\rſDžunglaaBéß(İ漢!!ſع\r, 'SDžungla camelCase", "tokens": 42, "pieces": ["'漢i", "OS", "\n", "/", " \n", "éd漢", ",é", "\r", "\"m", "\r", "ſ", "Džunglaa", "Béß", "(İ漢", "!!", "ſع", "\r", ",", " ", " '", "SDžungla", " camel", "Case"]} +{"text": "t(12345678dEOT👍🏽", "tokens": 11, "pieces": ["t", "(", "123", "456", "78", "d", "EOT", "👍🏽"]} +{"text": "ꟲ字🙂s", "tokens": 6, "pieces": ["ꟲ字", "🙂s"]} +{"text": "Abå\r\n\r\nDžungla<|endoftext|>A're'DⅣéaB!aBZiOS,\rDžungla#$%A0e\n ́Ⅳ\r\n\n/🙂.㋿iOS", "tokens": 54, "pieces": ["Abå", "\r\n\r\n", "Džungla", "<|", "endoftext", "|>", "A're", "'D", "Ⅳ", "éa", "B", "!a", "BZi", "OS", ",\r", "Džungla", "#$%", "A", "0", "e", "\n", " ́", "Ⅳ", "\r\n\n", "/🙂.㋿", "i", "OS"]} +{"text": "'T½漢>éḍ̇'re#$%a/bdB'S", "tokens": 20, "pieces": ["'T", "½", "漢", ">éḍ̇'re", "#$%<", "EOT", ">a", "/bd", "B'S"]} +{"text": "0'\r\n\r\n", "tokens": 2, "pieces": ["0", "'\r\n\r\n"]} +{"text": "'T٣٤٥٦12345678/ ,㍿\"'TBⅣ㍿", "tokens": 21, "pieces": ["'T", "٣٤٥", "٦12", "345", "678", "/", " ", " ,㍿\"'", "TB", "Ⅳ", "㍿"]} +{"text": "m!<|endoftext|>.é🙂> 'ſ \n !!å", "tokens": 19, "pieces": ["m", "!<|", "endoftext", "|>.", "é", "🙂>", " '", "ſ", " \n", " !!", "å"]} +{"text": "t-\r\n\r\nDž‍", "tokens": 6, "pieces": ["t", "-\r\n\r\n", "Dž", "‍"]} +{"text": "<|endoftext|>\r\n\r\nſ 'D(​Džungla‍EOTİ 9B. \n camelCase!!t…>…'ll'Mḍ̇­EOT.½\t\u000b", "tokens": 51, "pieces": ["<|", "endoftext", "|>\r\n\r\n", "ſ", " ", "'D", "(​", "Džungla", "‍EOT", "İ", " ", " ", "9", "B", ".", " \n", " camel", "Case", "!!", "t", "…", ">", "…", "'ll'M", "ḍ̇", "­EOT", ".", "½", "\t\u000b"]} +{"text": "iOS٣٤٥٦\tt/\r\nᵃ<|fim_prefix|>.Ⅳ👍🏽🙂12345678\u000bmDžB/\r\n", "tokens": 35, "pieces": ["i", "OS", "٣٤٥", "٦", "\tt", "/\r\n", "ᵃ", "<|", "fim", "_prefix", "|>.", "Ⅳ", "👍🏽🙂", "123", "456", "78", "\u000bm", "DžB", "/\r\n"]} +{"text": ".\r\nA<́9d's 'll😀🏽é\r\n\r\n aB\rZcamelCase'sm>0<|endoftext|>s'reABC'Reḍ̇‍\r\n\r\n字Džungla\t\u000ba/bAb", "tokens": 50, "pieces": [".\r\n", "A", "<́", "9", "d's", " ", "'ll", "😀🏽", "é", "\r\n\r\n", " a", "B", "\r", "Zcamel", "Case's", "m", ">", "0", "<|", "endoftext", "|>", "s're", "ABC'Re", "ḍ̇", "‍\r\n\r\n", "字Džungla", "\t", "\u000ba", "/b", "Ab"]} +{"text": " !漢fiEOT㍿a/bm'ree\r\n\r\n!!<|endoftext|>\nع…", "tokens": 25, "pieces": [" !", "漢fi", "EOT", "㍿a", "/bm're", "e", "\r\n\r\n", "!!<|", "endoftext", "|>\n", "ع", "…"]} +{"text": "s​\n/å,漢\u000b", "tokens": 8, "pieces": ["s", "​\n/", "å", ",漢", "\u000b"]} +{"text": "字٣٤٥٦'ſßſ/\r\na/bꟲ!!'T<|endoftext|>🙂å\r\n\r\naå", "tokens": 31, "pieces": ["字", "٣٤٥", "٦", "'ſßſ", "/\r\n", "a", "/bꟲ", "!!'", "T", "<|", "endoftext", "|>🙂", "å", "\r\n\r\n", "aå"]} +{"text": "ZaB >٣٤٥٦<|endoftext|>'llB\rZ(<|endoftext|>HTTPServerEOT>0😀🏽'D'T(HTTPServer,ß \n a/bⅣ\u000baB B\r\n/'llße", "tokens": 57, "pieces": ["Za", "B", " ", " >", "٣٤٥", "٦", "<|", "endoftext", "|>'", "ll", "B", "\r", "Z", "(<|", "endoftext", "|>", "HTTPServer", "EOT", ">", "0", "😀🏽'", "D'T", "(HTTPServer", ",ß", " \n", " a", "/b", "Ⅳ", "\u000ba", "B", " B", "\r\n", "/'", "llße"]} +{"text": "s'", "tokens": 2, "pieces": ["s", "'"]} +{"text": "<|endoftext|>e\n­'llé!12345678ſ'S're-𐞁٣٤٥٦\r\n\n ('sᵃ", "tokens": 35, "pieces": ["<|", "endoftext", "|>", "e", "\n", "­'", "llé", "!", "123", "456", "78", "ſ'S", "'re", "-𐞁", "٣٤٥", "٦", "\r\n\n", " ('", "sᵃ"]} +{"text": "DžunglaDžungla(- a/b👍🏽\n .👍🏽0ع\n!!'ReéiOS ­ \naB ", "tokens": 37, "pieces": ["Džungla", "Džungla", "(-", " ", " a", "/b", "👍🏽\n", " ", ".👍🏽", "0", "ع", "\n", "!!'", "Reéi", "OS", " ", " ­", " \n", "a", "B", " "]} +{"text": "å#$%a ḍ̇å'VE'S‍<漢‍<😀🏽\r\n'S\t. \tiOSAb \ns's٣٤٥٦'ll/\r\nع.ḍ̇'Re 9a/bAbABC'S", "tokens": 52, "pieces": ["å", "#$%", "a", " ḍ̇å'VE", "'S", "‍<", "漢", "‍<😀🏽\r\n", "'S", "\t", ".", " ", "\ti", "OSAb", " \n", "s's", "٣٤٥", "٦", "'ll", "/\r\n", "ع", ".ḍ̇'Re", " ", "9", "a", "/b", "Ab", "ABC'S"]} +{"text": "t!'Refis\t'ſ#$%\r\na Dž㍿0\r\n\r\nİ12345678ABCß字㋿HTTPServer𐞁Dž!!(", "tokens": 40, "pieces": ["t", "!'", "Refis", "\t", "'ſ", "#$%\r\n", "a", " ", " Dž", "㍿", "0", "\r\n\r\n", "İ", "123", "456", "78", "ABCß字", "㋿HTTPServer𐞁", "Dž", "!!("]} +{"text": "٣٤٥٦३", "tokens": 5, "pieces": ["٣٤٥", "٦३"]} +{"text": "ß Ⅳ\n/'VE\r\nZ-#$%㋿🙂 \n ㍿iOS \n 'T㋿ع'ReAb!́é", "tokens": 38, "pieces": ["ß", " ", " ", "Ⅳ", "\n", "/'", "VE", "\r\n", "Z", "-#$%㋿🙂", " \n", " ", " ㍿", "i", "OS", " \n", " '", "T", "㋿ع'Re", "Ab", "!́é"]} +{"text": "'ſ-½\n \niOSé  >0/\r\n /'sa'Re'\t\"!! B­عDž \n9\té \n", "tokens": 33, "pieces": ["'ſ", "-", "½", "\n \n", "i", "OSé", " ", " ", ">", "0", "/\r\n", " ", " /'", "sa'Re", "'", "\t", "\"!!", " B", "­ع", "Dž", " \n", "9", "\té", " \n"]} +{"text": "0'TåAb", "tokens": 5, "pieces": ["0", "'Tå", "Ab"]} +{"text": "漢é\r­Ab'ReᵃiOS!!ᵃſ<|fim_prefix|>𐞁 éß
\r", "­Ab'Re", "ᵃi", "OS", "!!", "ᵃſ", "<|", "fim", "_prefix", "|>", "𐞁", " ", " éß", "
", ".\r0'VE㍿'T­<|endoftext|>/!!", "tokens": 43, "pieces": [" '", "S", "!!", " ", " '", "M", ",𐞁ß", "
", "'VEAi", "OS", "‍", "0", "𐞁", ">.\r", "0", "'VE", "㍿'", "T", "­<|", "endoftext", "|>/!!"]} +{"text": "HTTPServer!!a/bDž😀🏽DžⅣ<\tét\n/ \n <|fim_prefix|>AbaBé٣٤٥٦iOS'M𐞁३9! \n /
s", "tokens": 48, "pieces": ["HTTPServer", "!!", "a", "/b", "Dž", "😀🏽", "Dž", "Ⅳ", "<", "\tét", "\n", "/", " \n", " <|", "fim", "_prefix", "|>", "Aba", "Bé", "٣٤٥", "٦", "i", "OS'M", "𐞁", "३9", "!", " \n", " /", "
s"]} +{"text": "HTTPServerA>'reḍ̇ \"'S\r\n\r\n漢aBA< \n B/​­iOS­漢-a/baB​㍿㋿'ſ'S'VEß", "tokens": 46, "pieces": ["HTTPServer", "A", ">'", "reḍ̇", " ", "\"'", "S", "\r\n\r\n", "漢a", "BA", "<", " \n", " <", "META", "_START", ">B", "/​­", "i", "OS", "­漢", "-a", "/ba", "B", "​㍿㋿'", "ſ'S", "'VEß"]} +{"text": "‍9ꟲ \ńᵃfiİ㍿ \nfi'VEå​Aᵃ0漢camelCase'reſAbEOT", "tokens": 35, "pieces": ["‍", "9", "ꟲ", " \n", "́ᵃfi", "İ", "㍿", " \n", "fi'VE", "å", "​Aᵃ", "0", "漢camel", "Case're", "ſ", "Ab", "EOT"]} +{"text": "a/b'Tꟲ/\r\nABC0字", "tokens": 14, "pieces": ["a", "/b'T", "ꟲ", "/\r\n", "ABC", "", "0", "字"]} +{"text": "'VE'VEᵃ𐞁'ſcamelCase‍<|endoftext|>så½İ", "tokens": 30, "pieces": ["'VE'VE", "ᵃ𐞁'ſ", "camel", "Case", "‍<|", "endoftext", "|>", "s", "å", "½", "İ"]} +{"text": "Dž'ſ!!ABC'lladé\r\n\r\n㍿Ab're", "tokens": 15, "pieces": ["Dž'ſ", "!!", "ABC'll", "adé", "\r\n\r\n", "㍿Ab're"]} +{"text": "ḍ̇", "tokens": 3, "pieces": ["ḍ̇"]} +{"text": "HTTPServers\nꟲ🙂>a㍿", "tokens": 15, "pieces": ["HTTPServers", "\n", "ꟲ", "🙂>", "a", "㍿"]} +{"text": "😀🏽Ab/\r\n!ß㍿ 'ſ \n/.", " '", "ſ", " \n", "/.<", "m", " HTTPServer", "😀🏽", " "]} +{"text": "'ſ'llaBᵃ\n/", "tokens": 10, "pieces": ["'ſ'll", "a", "Bᵃ", "\n", "/"]} +{"text": "½ \n ​-0字å \n.'ſ٣٤٥٦Dž9 \n'ſ'Rea😀🏽 \t…t漢#$%Ab🙂ḍ̇", "tokens": 39, "pieces": ["½", " \n", " ​-", "0", "字å", " \n", ".'", "ſ", "٣٤٥", "٦", "Dž", "9", " \n", "'ſ'Re", "a", "😀🏽", " \t", "…t漢", "#$%", "Ab", "🙂ḍ̇"]} +{"text": "EOTⅣ9", "tokens": 8, "pieces": ["EOT", "Ⅳ9"]} +{"text": "EOT,\r<|endoftext|>😀🏽३!Bİ‍ᵃiOSعAb ᵃ'ſ/<|endoftext|> ḍ̇İ'ſᵃ'M/ꟲ😀🏽ع\r'red!<|endoftext|>㍿aB", "tokens": 77, "pieces": ["EOT", ",\r", "<|", "endoftext", "|>😀🏽", "३", "!Bİ", "‍ᵃi", "OSعAb", " ᵃ'ſ", "/<|", "endoftext", "|>", " ḍ̇", "İ'ſ", "ᵃ'M", "/ꟲ", "😀🏽", "ع", "\r", "'red", "!<|", "endoftext", "|><", "EOT", ">㍿", "a", "B"]} +{"text": " éſd,<|endoftext|>å12345678Džt!!\nعa😀🏽EOT­\"'s½…'Ⅳ­'D \n ,'ſ\n-fi0", "tokens": 49, "pieces": [" éſd", ",<|", "endoftext", "|>", "å", "123", "456", "78", "Džt", "!!\n", "عa", "😀🏽", "EOT", "­\"'", "s", "½", "…", "'", "Ⅳ", "­'", "D", " \n", " ,'", "ſ", "\n", "-fi", "0"]} +{"text": "<|fim_prefix|>", "tokens": 6, "pieces": ["<|", "fim", "_prefix", "|>"]} +{"text": "/mEOT​ ABC", "tokens": 9, "pieces": ["/m", "EOT", "​", " ", " ABC"]} +{"text": "\r\n\r\nfi ½a/b㋿\" \u000bZt's漢३é", "tokens": 20, "pieces": ["\r\n\r\n", "fi", " ", " ", "½", "a", "/b", "㋿\"", " ", "\u000bZt's", "漢", "३", "é"]} +{"text": "t…٣٤٥٦Z'Reİ", "tokens": 10, "pieces": ["t", "…", "٣٤٥", "٦", "Z'Re", "İ"]} +{"text": " 'Re㍿😀🏽\" \n㋿iOS(12345678İA½,‍🙂‍'S's. …a/b-'sİ­<|fim_prefix|>İA", "tokens": 48, "pieces": [" ", " '", "Re", "㍿😀🏽\"", " \n", "㋿i", "OS", "(", "123", "456", "78", "İA", "½", ",‍🙂‍'", "S's", ".", " ", "…a", "/b", "-'", "s", "İ", "­<|", "fim", "_prefix", "|>", "İA"]} +{"text": "a\r\n\r\nA​<|endoftext|>'s's'T३'M.sAABCDž👍🏽Džungla\r\n𐞁Džunglaع\u000bEOT​'ll\r
HTTPServer🙂½", "tokens": 50, "pieces": ["a", "\r\n\r\n", "A", "​<|", "endoftext", "|>'", "s's", "'T", "३", "'M", ".s", "AABCDž", "👍🏽", "Džungla", "\r\n", "𐞁Džunglaع", "\u000bEOT", "​'", "ll", "\r", "
HTTPServer", "🙂", "½"]} +{"text": "Ⅳ٣٤٥٦'VE0#$%🙂9-#$%é!EOT𐞁ḍ̇e fiAb", "tokens": 31, "pieces": ["Ⅳ٣٤", "٥٦", "'VE", "0", "#$%🙂", "9", "-#$%", "é", "!EOT𐞁ḍ̇e", " fi", "Ab"]} +{"text": "👍🏽字/\r\nBDž \n#$%\"eßß½\"", "tokens": 16, "pieces": ["👍🏽", "字", "/\r\n", "BDž", " \n", "#$%\"", "eßß", "½", "\""]} +{"text": "'T-'Re𐞁", "tokens": 7, "pieces": ["'T", "-'", "Re𐞁"]} +{"text": "!!\r\n'Ret \n\r\ns#$%Ⅳ", "tokens": 14, "pieces": ["!!\r\n", "'Ret", " \n\r\n", "s", "#$%", "Ⅳ", ""]} +{"text": "!!ſ𐞁👍🏽ABC\nḍ̇a/b<", "tokens": 17, "pieces": ["!!", "ſ𐞁", "👍🏽", "ABC", "\n", "ḍ̇a", "/b", "<"]} +{"text": "å DžunglaEOT!!ḍ̇-Ab>.'sdaB'reḍ̇\"d\r\n\r\n #$%A\n/", "tokens": 31, "pieces": ["å", " Džungla", "EOT", "!!", "ḍ̇", "-Ab", ">.'", "sda", "B're", "ḍ̇", "\"d", "\r\n\r\n", " #$%", "A", "\n", "/"]} +{"text": "'M'ſABC'​👍🏽 ", "tokens": 10, "pieces": ["'M'ſ", "ABC", "'​👍🏽", " "]} +{"text": "\n'D'ſ/ \n'M­dḍ̇>㋿<|fim_prefix|>ᵃ漢\" \n\r\n0​ſſ​-", "tokens": 34, "pieces": ["\n", "'D'ſ", "/", " \n", "'M", "­dḍ̇", ">㋿<|", "fim", "_prefix", "|>", "ᵃ漢", "\"", " \n\r\n", "0", "​ſſ", "​-"]} +{"text": "Z'ſ'عᵃfi", "tokens": 9, "pieces": ["Z'ſ", "'عᵃfi"]} +{"text": "ſ!'ſ/\r\n \n𐞁eaB👍🏽́ da", "tokens": 17, "pieces": ["ſ", "!'", "ſ", "/\r\n", " \n", "𐞁ea", "B", "👍🏽́", " da"]} +{"text": "A漢 12345678ḍ̇HTTPServer㍿\n/'Re\r\n\r\n‍B9ßaB'S!!👍🏽'M!!'llAꟲ­👍🏽Dž9m½ع.字m\u000bA", "tokens": 53, "pieces": ["A漢", " ", "123", "456", "78", "ḍ̇", "HTTPServer", "㍿\n/", "'Re", "\r\n\r\n", "‍B", "9", "ßa", "B'S", "!!👍🏽'", "M", "!!'", "ll", "Aꟲ", "­👍🏽", "Dž", "9", "m", "½", "ع", ".字m", "\u000bA"]} +{"text": "camelCasee \n ٣٤٥٦\u000b ABC'T", "tokens": 13, "pieces": ["camel", "Casee", " \n", " ", "٣٤٥", "٦", "\u000b", " ABC'T"]} +{"text": "'s'Må\u000b ́Ab'T(‍", "tokens": 11, "pieces": ["'s'M", "å", "\u000b", " ́Ab'T", "(‍"]} +{"text": " \n\n/#$%ḍ̇'Re'Ta/b㋿>ḍ̇ \u000bع'll'ſ
 \t!३éA \n camelCasea/bå'ſB0B…< \n ,,a", "tokens": 56, "pieces": [" \n\n", "/#$%", "ḍ̇'Re", "'", "Ta", "/b", "㋿>", "ḍ̇", " ", "\u000bع'll", "'ſ", "
 ", "\t", "!", "३", "é", "A", " \n", " camel", "Casea", "/bå'ſ", "B", "0", "B", "…", "<", " \n", " ,,", "a"]} +{"text": " \n …<|fim_prefix|>\"𐞁!㍿'ss", "tokens": 20, "pieces": [" \n", " ", "…", "<|", "fim", "_prefix", "|>\"", "𐞁", "!㍿'", "ss"]} +{"text": "ᵃ😀🏽 é,…fi", "tokens": 13, "pieces": ["ᵃ", "😀🏽", " é", ",", "…fi"]} +{"text": "é're-\r\n", "tokens": 3, "pieces": ["é're", "-\r\n"]} +{"text": "12345678e½'M\rABC's㍿<\"𐞁éſ㋿👍🏽Ab iOS'T\rⅣ('ll'llABCABCBiOS­>Ⅳ", "tokens": 46, "pieces": ["123", "456", "78", "e", "½", "'M", "\r", "ABC's", "㍿<\"", "𐞁éſ", "㋿👍🏽", "Ab", " i", "OS'T", "\r", "Ⅳ", "('", "ll'll", "ABCABCBi", "OS", "­>", "Ⅳ"]} +{"text": "\r<٣٤٥٦\"İ́", "tokens": 9, "pieces": ["\r", "<", "٣٤٥", "٦", "\"İ́"]} +{"text": "<|fim_prefix|>.aBA𐞁é‍mAbABCåé'reꟲaaBåDž字‍9!é'llA!\rDžᵃ'Ret", "tokens": 53, "pieces": ["<|", "fim", "_prefix", "|>.", "a", "BA𐞁é", "‍m", "Ab", "ABCå", "é're", "ꟲaa", "Bå", "Dž字", "‍", "9", "!é'll", "A", "!\r", "Džᵃ'Re", "t"]} +{"text": "fi0\n/🙂camelCase<ᵃ𐞁", "tokens": 21, "pieces": ["fi", "0", "\n", "/<", "META", "_START", ">🙂", "camel", "Case", "<ᵃ𐞁", ""]} +{"text": "​ß\r\nå漢å\n/>< \n 'T३EOT'٣٤٥٦ABC漢t9s !!३", "tokens": 33, "pieces": ["​ß", "\r\n", "å漢å", "\n/", "><", " \n", " '", "T", "३", "EOT", "'", "٣٤٥", "٦", "ABC漢t", "9", "s", " ", "!!", "३"]} +{"text": "(d㍿ e.'s‍fiåİ\nB/\r\n \n­ (ß", "tokens": 24, "pieces": ["(d", "㍿", " e", ".'", "s", "‍fi", "å", "İ", "\n", "B", "/\r\n", " \n", "­", " ", "(ß"]} +{"text": "'VE\r\n!!🙂\r\nⅣB​Ab12345678#$%'TdDžſABC", "tokens": 21, "pieces": ["'VE", "\r\n", "!!🙂\r\n", "Ⅳ", "B", "​Ab", "123", "456", "78", "#$%'", "Td", "Džſ", "ABC"]} +{"text": "𐞁're'sDžungladᵃ9\r­t-…/𐞁ḍ̇<\n(𐞁aB字, \nå\u000b<|fim_prefix|><|endoftext|>'​‍ꟲ", "tokens": 61, "pieces": ["𐞁're", "'s", "Džungladᵃ", "9", "\r", "­t", "-", "…", "/𐞁ḍ̇", "<\n", "(<", "META", "_START", ">𐞁a", "B字", ",", " \n", "å", "\u000b", "<|", "fim", "_prefix", "|><|", "endoftext", "|>'​‍", "ꟲ"]} +{"text": "İ'VE", "tokens": 6, "pieces": ["İ", "'", "VE"]} +{"text": "Ab …ꟲ٣٤٥٦­HTTPServer…<|fim_prefix|>ſ", "tokens": 23, "pieces": ["Ab", " ", "…ꟲ", "٣٤٥", "٦", "­HTTPServer", "…", "<|", "fim", "_prefix", "|>", "ſ"]} +{"text": ">å\u000bAABC,HTTPServerǻ-(½EOTZ'D🙂Džungla字ᵃİ👍🏽\t\n३🙂👍🏽m 'll'S\u000bſ/\r\ncamelCase
", "tokens": 50, "pieces": [">å", "\u000bAABC", ",HTTPServerǻ", "-(", "½", "EOTZ'D", "🙂Džungla字ᵃ", "İ", "👍🏽", "\t\n", "३", "🙂👍🏽", "m", " ", "'ll'S", "\u000bſ", "/\r\n", "camel", "Case", "
"]} +{"text": " 'T\"'T<|endoftext|>9Dž/\r\nABC😀🏽éDž字>!!!HTTPServer ㍿\r\n\r\nHTTPServer٣٤٥٦'Re12345678Džungla½", "tokens": 53, "pieces": [" '", "T", "\"'", "T", "<|", "endoftext", "|>", "9", "Dž", "/\r\n", "ABC", "😀🏽", "é", "Dž字", ">!!!", "HTTPServer", " ", " ㍿\r\n\r\n", "HTTPServer", "٣٤٥", "٦", "'Re", "123", "456", "78", "Džungla", "½"]} +{"text": "<|endoftext|>عAbⅣ\r\n\r\nméßⅣ\n/ABC𐞁ßſcamelCase'M𐞁(ficamelCaseᵃiOSB\r\n\r\nſ½é'Saa", "tokens": 51, "pieces": ["<|", "endoftext", "|>", "عAb", "Ⅳ", "\r\n\r\n", "méß", "Ⅳ", "\n", "/ABC𐞁ßſ", "camel", "Case'M", "𐞁", "(ficamel", "Caseᵃi", "OSB", "\r\n\r\n", "ſ", "½", "é'S", "aa"]} +{"text": "a\r🙂'ree字<|endoftext|>aB­./漢", "tokens": 20, "pieces": ["a", "\r", "🙂'", "ree字", "<|", "endoftext", "|><", "EOT", ">a", "B", "­./", "漢"]} +{"text": "A٣٤٥٦'DZ'VE㋿mB/\r\n, \n👍🏽Z㍿ßfi'T३\n/DžunglaDž('M…\n/9…\r", "tokens": 46, "pieces": ["A", "٣٤٥", "٦", "'DZ'VE", "㋿m", "B", "/\r\n", ",", " \n", "👍🏽", "Z", "㍿ßfi'T", "३", "\n", "/Džungla", "Dž", "('", "M", "…\n", "/", "9", "…\r"]} +{"text": "'Se½fi٣٤٥٦३İ😀🏽 ,-́fia\n!<|fim_prefix|>fiAb!/Z.écamelCaseB½'D㋿HTTPServera'VEå漢", "tokens": 47, "pieces": ["'Se", "½", "fi", "٣٤٥", "٦३", "İ", "😀🏽", " ,-́", "fia", "\n", "!<|", "fim", "_prefix", "|>", "fi", "Ab", "!/", "Z", ".écamel", "Case", "B", "½", "'D", "㋿HTTPServera'VE", "å漢"]} +{"text": "e👍🏽'­
\rsḍ̇ \r\n\r\n/\r\n/DžAb<|endoftext|>\r\nعⅣ
iOSfi\r\n👍🏽d३ Dž\r9tEOT½e#$%a/bDžungla", "tokens": 56, "pieces": ["e", "👍🏽'­", "
\r", "sḍ̇", " \r\n\r\n", "/\r\n/", "DžAb", "<|", "endoftext", "|>\r\n", "ع", "Ⅳ", "
i", "OSfi", "\r\n", "👍🏽", "d", "३", " Dž", "\r", "9", "t", "EOT", "½", "e", "#$%", "a", "/b", "Džungla"]} +{"text": "́​'Re'D", "tokens": 5, "pieces": ["́", "​'", "Re'D"]} +{"text": " \n …d㍿
<ꟲé.ع're're", "tokens": 19, "pieces": [" \n", " ", "…d", "㍿", "
", "<ꟲé", ".ع're", "'re"]} +{"text": "å'DABC's'D𐞁<|endoftext|>́😀🏽,​\r\n\r\n<|endoftext|>ḍ̇e​­👍🏽aBéعaB'Reꟲ㍿", "tokens": 60, "pieces": ["å'D", "ABC's", "'D𐞁", "<|", "endoftext", "|>́😀🏽,​\r\n\r\n", "<|", "endoftext", "|>", "ḍ̇e", "​­<", "META", "_START", ">👍🏽", "a", "Béعa", "B'Re", "ꟲ", "㍿"]} +{"text": "9 ३́/\r\n\t", "tokens": 6, "pieces": ["9", " ", "३", "́", "/\r\n", "\t"]} +{"text": "'re'字!é'ſaB漢٣٤٥٦'Re٣٤٥٦é", "tokens": 21, "pieces": ["'re", "'字", "!é'ſ", "a", "B漢", "٣٤٥", "٦", "'Re", "٣٤٥", "٦", "é"]} +{"text": "́a/b\u000b\r\n\r\n!!é 12345678'll\t,Ab'Sſſm-.\n/a/bAbⅣ<|fim_prefix|>/\r\n(>mcamelCase#$%a/b\t'HTTPServer>漢\t", "tokens": 50, "pieces": ["́a", "/b", "\u000b\r\n\r\n", "!!", "é", " ", "123", "456", "78", "'ll", "\t", ",Ab'S", "ſſm", "-.\n/", "a", "/b", "Ab", "Ⅳ", "<|", "fim", "_prefix", "|>/\r\n", "(>", "mcamel", "Case", "#$%", "a", "/b", "\t", "'HTTPServer", ">漢", "\t"]} +{"text": "عdABs'dcamelCase/\"DžunglaaBt𐞁>'ſ12345678aBZ.#$%éꟲ½\u000bs<", "tokens": 46, "pieces": ["عd", "ABs'd", "camel", "Case", "/\"", "Džunglaa", "Bt𐞁", ">'", "ſ", "123", "456", "78", "a", "BZ", ".<", "META", "_START", ">#$%", "éꟲ", "½", "\u000bs", "<<", "EOT", ">"]} +{"text": "9t<|fim_prefix|>́\n/-12345678dꟲ<|endoftext|>ſa'M½٣٤٥٦ \n \n३d>d", "tokens": 38, "pieces": ["9", "t", "<|", "fim", "_prefix", "|>́\n/", "-", "123", "456", "78", "dꟲ", "<|", "endoftext", "|>", "ſa'M", "½٣٤", "٥٦", " \n \n", "३", "d", ">d"]} +{"text": "\ta/bİ٣٤٥٦<|fim_prefix|>é漢>'Sſ<\t'D\u000bAb<|endoftext|>tḍ̇'MéDž \nḍ̇漢ᵃDžunglá", "tokens": 51, "pieces": ["\ta", "/b", "İ", "٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "é漢", ">'", "Sſ", "<", "\t", "'D", "\u000bAb", "<|", "endoftext", "|>", "tḍ̇'M", "é", "Dž", " \n", "ḍ̇漢ᵃ", "Džunglá"]} +{"text": "ſ'T/!aBaEOT३aB'Re/\"a/bſ", "tokens": 16, "pieces": ["ſ'T", "/!", "a", "Ba", "EOT", "३", "a", "B'Re", "/\"", "a", "/bſ"]} +{"text": "'T.!Z
t漢ᵃ\r\n\r\n9👍🏽/\r\na/bع\réſZZ'Re,", "tokens": 24, "pieces": ["'T", ".!", "Z", "
t漢ᵃ", "\r\n\r\n", "9", "👍🏽/\r\n", "a", "/bع", "\r", "éſ", "ZZ'Re", ","]} +{"text": "ABCEOT>/\r\n'D're", "tokens": 9, "pieces": ["ABC", "EOT", ">/\r\n", "'D're"]} +{"text": "dſ9HTTPServerEOTꟲ/", "tokens": 13, "pieces": ["dſ", "9", "HTTPServer", "EOTꟲ", "/"]} +{"text": "00ådDžungla'👍🏽…Dž're<\u000bZiOS𐞁'ſ'VEé9s'DDž A<|fim_prefix|>'T𐞁\n/ß s'M('ſ", "tokens": 59, "pieces": ["00", "åd", "Džungla", "'👍🏽", "…Dž're", "<", "\u000bZi", "OS𐞁'ſ", "'VEé", "9", "s", "'", "DDž", " ", " A", "<|", "fim", "_prefix", "|>'", "T𐞁", "\n", "/ß", " s'M", "('", "ſ"]} +{"text": " ḍ̇", "tokens": 4, "pieces": [" ḍ̇"]} +{"text": "'s㍿ 0aB\tHTTPServerEOTfi🙂 dé́ 'D!!\r\n\r\nt'M\r\nåDž.\n/HTTPServer", "tokens": 33, "pieces": ["'s", "㍿", " ", "0", "a", "B", "\tHTTPServer", "EOTfi", "🙂", " ", " dé́", " ", " '", "D", "!!\r\n\r\n", "t'M", "\r\n", "å", "Dž", ".\n/", "HTTPServer"]} +{"text": ">'S\n/A'StEOT㍿३(å👍🏽\r/\r\n'll'M'/\r\n\r\n\r\n\r\n(.'D​  😀🏽ᵃ🙂!!Aſ \nB ", "tokens": 51, "pieces": [">'", "S", "\n", "/A'S", "t", "EOT", "㍿", "३", "(å", "👍🏽\r/\r\n", "'ll'M", "'/\r\n\r\n\r\n\r\n", "(.'", "D", "​", " ", " <", "EOT", ">", " ", "😀🏽", "ᵃ", "🙂<", "META", "_START", ">!!", "Aſ", " \n", "B", " "]} +{"text": "­é", "tokens": 2, "pieces": ["­é"]} +{"text": "<|fim_prefix|>\t𐞁字fi
", "tokens": 14, "pieces": ["<|", "fim", "_prefix", "|>", "\t𐞁字fi", "
"]} +{"text": ",ᵃ字aBa/bZABC字'漢­👍🏽字0,字 <|fim_prefix|>eABC㍿aB
", "tokens": 36, "pieces": [",ᵃ字a", "Ba", "/b", "ZABC字", "'漢", "­👍🏽", "字", "0", ",字", " ", "<|", "fim", "_prefix", "|>", "e", "ABC", "㍿a", "B", "
"]} +{"text": "㋿३'S \n'siOS\n/٣٤٥٦camelCase'ſ­Ab'SAaBsé", "tokens": 26, "pieces": ["㋿", "३", "'S", " \n", "'si", "OS", "\n", "/", "٣٤٥", "٦", "camel", "Case'ſ", "­Ab'S", "Aa", "Bsé"]} +{"text": "漢!>'scamelCasé", "tokens": 7, "pieces": ["漢", "!>'", "scamel", "Casé"]} +{"text": "-9Džunglaåm/e'M-Ab\u000b \nAᵃ\u000b", "tokens": 35, "pieces": ["-", "9", "Džunglaåm", "/e'M", "-Ab", "", "\u000b \n", "Aᵃ", "\u000b"]} +{"text": " (camelCase -#$%'Ta99!é#$%\n/'Dé㋿Džungla
\r\n\r\nᵃ<|fim_prefix|>aB㍿-  \r\n/​\n\r\n\r\neABCa/b
", "tokens": 55, "pieces": [" (", "camel", "Case", " ", "-#$%'", "Ta", "99", "!é", "#$%\n/", "'Dé", "㋿Džungla", "
\r\n\r\n", "ᵃ", "<|", "fim", "_prefix", "|>", "a", "B", "㍿-", "  \r\n", "/<", "META", "_START", ">​\n\r\n\r\n", "e", "ABCa", "/b", "
"]} +{"text": "​é😀🏽ḍ̇Dž.\r\n\r\n's½\u000b'T́", "tokens": 16, "pieces": ["​é", "😀🏽", "ḍ̇", "Dž", ".\r\n\r\n", "'s", "½", "\u000b", "'T́"]} +{"text": ">å(
<㋿'s٣٤٥٦-(a/b#$%İ'll#$%'T'S\"", "tokens": 26, "pieces": [">å", "(", "
", "<㋿'", "s", "٣٤٥", "٦", "-(", "a", "/b", "#$%", "İ'll", "#$%'", "T'S", "\""]} +{"text": "\r\nḍ̇\r\n12345678 \n (Ⅳd<|endoftext|>ABC'll…éDžZ'll‍­'T\u000b0­ \n ٣٤٥٦(éAEOTeå!́𐞁ḍ̇…m<|endoftext|>", "tokens": 68, "pieces": ["\r\n", "ḍ̇", "\r\n", "123", "456", "78", " \n", " (", "Ⅳ", "d", "<|", "endoftext", "|>", "ABC'll", "…é", "DžZ'll", "‍­'", "T", "\u000b", "0", "­", " \n", " ", "٣٤٥", "٦", "(é", "AEOTeå", "!́𐞁ḍ̇", "…m", "<|", "endoftext", "|>"]} +{"text": "é 'ſ㋿Z'VE<|fim_prefix|>\n/12345678'Re
s<㋿eᵃ\t漢\r\n\r\n漢ᵃ\r\n12345678ᵃ­åsZⅣ", "tokens": 52, "pieces": ["é", " ", " '", "ſ", "㋿Z'VE", "<|", "fim", "_prefix", "|>\n/", "123", "456", "78", "'Re", "
s", "<㋿", "eᵃ", "\t漢", "\r\n\r\n", "漢ᵃ", "\r\n", "123", "456", "78", "ᵃ", "­ås", "Z", "Ⅳ"]} +{"text": "𐞁𐞁½'s", "tokens": 10, "pieces": ["𐞁𐞁", "½", "'s"]} +{"text": "/\r\n ", "tokens": 2, "pieces": ["/\r\n", " "]} +{"text": "漢>0're", "tokens": 4, "pieces": ["漢", ">", "0", "'re"]} +{"text": "'SEOT\n/ᵃßEOTa/bZ㍿ \nع,عHTTPServer‍  ‍m'DA'Reعع", "tokens": 33, "pieces": ["'SEOT", "\n", "/ᵃß", "EOTa", "/b", "Z", "㍿", " \n", "ع", ",عHTTPServer", "‍", " ", " ", "‍m'D", "A'Re", "عع"]} +{"text": "!ßⅣḍ̇ꟲ'ſ㍿", "tokens": 15, "pieces": ["!ß", "Ⅳ", "ḍ̇ꟲ'ſ", "㍿"]} +{"text": "ꟲ12345678\rſ<|fim_prefix|>٣٤٥٦​Džunglaé𐞁Dž#$%012345678\r\n\r\n٣٤٥٦!Džungla😀🏽!é\n/½12345678'D ㍿\r\nd\r\n\r\nİ
Ab-'😀🏽fi", "tokens": 78, "pieces": ["ꟲ", "123", "456", "78", "\r", "ſ", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "​Džunglaé𐞁", "Dž", "#$%", "012", "345", "678", "\r\n\r\n", "٣٤٥", "٦", "!Džungla", "😀🏽!", "é", "\n", "/", "½12", "345", "678", "'D", " ", "㍿\r\n", "d", "\r\n\r\n", "İ", "", "
Ab", "-'😀🏽", "fi"]} +{"text": "𐞁'Re\n/Ab𐞁ABC३ß㍿ \n३ABC'ſ字字m're 'ReDžungla s👍🏽ḍ̇DžunglasDž9<|endoftext|>\t\n/३\r\n(𐞁ḍ̇'M'VEABC", "tokens": 71, "pieces": ["𐞁'Re", "\n", "/Ab𐞁", "ABC", "३", "ß", "㍿", " \n", "३", "ABC'ſ", "字字m're", " ", "'Re", "Džungla", " s", "👍🏽", "ḍ̇", "Džunglas", "Dž", "9", "<|", "endoftext", "|>", "\t\n", "/", "३", "\r\n", "(𐞁ḍ̇'M", "'VEABC"]} +{"text": "iOS(ABCfiDžunglae🙂EOT­Džungla㍿#$%12345678camelCase'ſ#$%#$%'MaB٣٤٥٦'Reé'llſſ", "tokens": 45, "pieces": ["i", "OS", "(ABCfi", "Džunglae", "🙂EOT", "­Džungla", "㍿#$%", "123", "456", "78", "camel", "Case'ſ", "#$%#$%'", "Ma", "B", "٣٤٥", "٦", "'Reé'll", "ſſ"]} +{"text": "٣٤٥٦㋿a\r\naa('llcamelCaseDžunglaDž\n/a ZiOS🙂३ABCa/b> \n0'ſ\n/-#$%𐞁'll\r\n\u000b'D>㍿\r\n\r\n", "tokens": 59, "pieces": ["٣٤٥", "٦", "㋿a", "\r\n", "aa", "('", "llcamel", "Case", "Džungla", "Dž", "\n", "/a", " ", " <", "META", "_START", ">Zi", "OS", "🙂", "३", "ABCa", "/b", ">", " \n", "0", "'ſ", "\n", "/-#$%", "𐞁'll", "\r\n", "\u000b", "'D", ">㍿\r\n\r\n"]} +{"text": "😀🏽Džꟲ0Dž'rea👍🏽 \nꟲ'D́A­Z 9As,#$% DžHTTPServer.\"Ⅳ", "tokens": 41, "pieces": ["😀🏽", "Džꟲ", "0", "Dž're", "a", "👍🏽", " \n", "ꟲ'D", "́", "A", "­Z", " ", " ", "9", "As", ",#$%", " DžHTTPServer", ".\"", "Ⅳ"]} +{"text": "३ſ<|endoftext|>ⅣBB! \n\t'Dm…🙂-/\r\nⅣaB 're👍🏽camelCase😀🏽ḍ̇aBéßEOT👍🏽😀🏽½#$%'sع'Då", "tokens": 63, "pieces": ["३", "ſ", "<|", "endoftext", "|>", "Ⅳ", "BB", "!", " \n", "\t", "'Dm", "…", "🙂-/\r\n", "Ⅳ", "a", "B", " '", "re", "👍🏽", "camel", "Case", "😀🏽", "ḍ̇a", "Béß", "EOT", "👍🏽😀🏽<", "EOT", ">", "½", "#$%'", "sع'D", "å"]} +{"text": "9< \n!m12345678a\r\n٣٤٥٦iOS \n
عé'ſ­,ꟲcamelCaseA३.\"'Re!\u000b,ᵃ٣٤٥٦'ſ'M\t", "tokens": 52, "pieces": ["9", "<", " \n", "!m", "123", "456", "78", "a", "\r\n", "٣٤٥", "٦", "i", "OS", "", " \n", "
عé'ſ", "­,", "ꟲcamel", "Case", "A", "३", ".\"'", "Re", "!", "\u000b", ",ᵃ", "٣٤٥", "٦", "'ſ'M", "\t"]} +{"text": "m\n…½\r३ Džungla!/ ع,<> 'S<|fim_prefix|>aB'Ma/bsaBİ.iOS", "tokens": 35, "pieces": ["m", "\n", "…", "½", "\r", "३", " Džungla", "!/", " ع", ",<>", " '", "S", "<|", "fim", "_prefix", "|>", "a", "B'M", "a", "/bsa", "Bİ", ".i", "OS"]} +{"text": ">>å'sZ(", "tokens": 6, "pieces": [">>", "å's", "Z", "("]} +{"text": "0㋿🙂!!- \naBa/b \n/,
9ꟲ-a<|fim_prefix|>Dž 9fiAiOS…'VE'D<|fim_prefix|>'T12345678 \n \n é12345678t\r ", "tokens": 57, "pieces": ["0", "㋿🙂!!-", " \n", "a", "Ba", "/b", " \n", "/,", "
", "9", "ꟲ", "-a", "<|", "fim", "_prefix", "|>", "Dž", " ", "9", "fi", "Ai", "OS", "…", "'VE'D", "<|", "fim", "_prefix", "|>'", "T", "123", "456", "78", " \n \n", " é", "123", "456", "78", "t", "\r", " "]} +{"text": "'S \n عd'VE३(0-㍿", "tokens": 13, "pieces": ["'S", " \n", " عd'VE", "३", "(", "0", "-㍿"]} +{"text": "('Re's👍🏽12345678'VE‍EOTع.12345678!!漢 \n'ſA漢,a/b\r\n\r\n 
…'S\"A\r\n…ᵃsEOTA'", "tokens": 51, "pieces": ["('", "Re's", "👍🏽", "123", "456", "78", "'VE", "‍EOTع", ".", "123", "456", "78", "!!", "漢", " \n", "'ſ", "A漢", ",a", "/b", "\r\n\r\n", " ", "
", "", "…", "'S", "\"A", "\r\n", "…ᵃs", "EOTA", "'"]} +{"text": "­ e­'s\r\n\r\n", "tokens": 7, "pieces": ["­", " ", " e", "­'", "s", "\r\n\r\n"]} +{"text": "!\r\n\r\nåßDžunglaEOT漢HTTPServer漢camelCase<|endoftext|>\t \n ABC\r\n🙂𐞁\u000b<|fim_prefix|>\"'T'Re'sta 'DaBBé(‍\t👍🏽", "tokens": 62, "pieces": ["!\r\n\r\n", "åß", "Džungla", "EOT漢HTTPServer漢camel", "Case", "<|", "endoftext", "|>", "\t \n", " ABC", "\r\n", "🙂𐞁", "\u000b", "<|", "fim", "_prefix", "|>\"'", "T'Re", "'sta", " ", "'", "Da", "BB", "é", "(‍", "\t", "👍🏽"]} +{"text": " \n \r\n\r\n", "tokens": 2, "pieces": [" \n \r\n\r\n"]} +{"text": "٣٤٥٦aB́B\n/'ſ'sa\n/👍🏽9é🙂e!!𐞁३HTTPServer,A", "tokens": 37, "pieces": ["٣٤٥", "٦", "a", "B́", "B", "\n", "/'", "ſ's", "a", "\n", "/👍🏽", "9", "é", "🙂e", "!!", "𐞁", "३", "HTTPServer", ",A"]} +{"text": "ع<|endoftext|>­iOS漢", "tokens": 12, "pieces": ["ع", "<|", "endoftext", "|>­", "i", "OS漢"]} +{"text": "'VE \n 'S
s\n/'T㍿.aB9camelCase12345678A", "tokens": 23, "pieces": ["'VE", " \n", " ", "'S", "
s", "\n", "/'", "T", "㍿.", "a", "B", "9", "camel", "Case", "123", "456", "78", "A"]} +{"text": "ᵃİ,㍿e9!!㍿\r!!", "tokens": 19, "pieces": ["ᵃ", "İ", ",㍿", "e", "9", "!!㍿\r", "!!"]} +{"text": "e<|endoftext|>t12345678ꟲé㍿fi'MA\na/b 'ABC(́å", "tokens": 33, "pieces": ["e", "<|", "endoftext", "|>", "t", "123", "456", "78", "ꟲé", "㍿fi'M", "A", "\n", "a", "/b", " ", "'ABC", "(́å"]} +{"text": "عHTTPServer\n/", "tokens": 5, "pieces": ["عHTTPServer", "\n", "/"]} +{"text": "ⅣiOS㋿'sABC'M<|endoftext|>.iOS >sßa/bAb\rḍ̇३!!­٣٤٥٦'DZ<𐞁/\r\n(#$%,½<'re𐞁", "tokens": 58, "pieces": ["Ⅳ", "i", "OS", "㋿'", "s", "ABC'M", "<|", "endoftext", "|>.", "i", "OS", " >", "sßa", "/b", "Ab", "\r", "ḍ̇", "३", "!!­", "٣٤٥", "٦", "'DZ", "<𐞁", "/\r\n", "(#$%,", "½", "<'", "re𐞁"]} +{"text": "-字ḍ̇s-aABCDžaB12345678d<|endoftext|>HTTPServerAḍ̇\r\nZ0\nB\"ᵃeé\"", "tokens": 41, "pieces": ["-字ḍ̇s", "-a", "ABCDža", "B", "123", "456", "78", "d", "<|", "endoftext", "|>", "HTTPServer", "Aḍ̇", "\r\n", "Z", "0", "\n", "B", "\"ᵃeé", "\""]} +{"text": "édḍ̇ABC\u000bḍ̇ᵃ9<|fim_prefix|>\r㍿'reHTTPServerét\"İå🙂d ßaB㋿‍٣٤٥٦.aB'D
\r\n\r\n", "tokens": 54, "pieces": ["édḍ̇", "ABC", "\u000bḍ̇ᵃ", "9", "<|", "fim", "_prefix", "|>\r", "㍿'", "re", "HTTPServerét", "\"İå", "🙂d", " ", " ßa", "B", "㋿‍", "٣٤٥", "٦", ".a", "B'D", "
\r\n\r\n"]} +{"text": "🙂㋿Zꟲ½,‍", "tokens": 11, "pieces": ["🙂㋿", "Zꟲ", "½", ",‍"]} +{"text": " \n
३'ſꟲ", "tokens": 8, "pieces": [" \n", "
", "३", "'ſꟲ"]} +{"text": " 12345678Z\tİ㍿\n/'VE#$%…!.'Džunglas\r\n\r\n-s'Re‍'MHTTPServerfi<0عZ३㋿EOTA𐞁", "tokens": 58, "pieces": [" ", "123", "456", "78", "Z", "\tİ", "㍿\n/", "'VE", "#$%", "…", "!.'", "Džunglas", "<", "META", "_START", ">\r\n\r\n", "-s", "'", "Re", "‍'", "MHTTPServerfi", "<", "0", "ع", "Z", "३", "㋿EOTA𐞁"]} +{"text": "'reaBſ<|endoftext|>😀🏽0tiOSAſHTTPServer", "tokens": 25, "pieces": ["'rea", "Bſ", "<|", "endoftext", "|>😀🏽", "0", "ti", "OS", "Aſ", "HTTPServer"]} +{"text": "a/३camelCaseAZ'VE", "tokens": 9, "pieces": ["a", "/", "३", "camel", "Case", "AZ'VE"]} +{"text": "🙂<|endoftext|>EOT", "tokens": 10, "pieces": ["🙂<|", "endoftext", "|>", "EOT"]} +{"text": "a're𐞁'llEOTAd​Dž'T<|fim_prefix|>/\r\n … /\r\n'Abꟲ३e#$%'Rea/b-", "tokens": 44, "pieces": ["a're", "𐞁'll", "EOTAd", "​Dž'T", "<|", "fim", "_prefix", "|>/\r\n", " ", "…", "", " ", "/\r\n", "'Abꟲ", "३", "e", "#$%'", "Rea", "/b", "-"]} +{"text": "İ\t🙂𐞁#$%dm‍m,Ab漢\r\n\t👍🏽 \n 'sm", "tokens": 23, "pieces": ["İ", "\t", "🙂𐞁", "#$%", "dm", "‍m", ",Ab漢", "\r\n", "\t", "👍🏽", " \n", " '", "sm"]} +{"text": "\" \n 're\t-'D\nd \n 'SᵃZ\r\n\r\nع'ſ \n😀🏽'Då\n/ 'T''D", "tokens": 33, "pieces": ["\"", " \n", " '", "re", "\t", "-'", "D", "\n", "d", " \n", " '", "Sᵃ", "Z", "\r\n\r\n", "ع'ſ", " \n", "😀🏽'", "Då", "\n", "/", " ", "'T", "''", "D"]} +{"text": "!!'ſ…\n#$%…\tİ𐞁½( ⅣHTTPServer12345678iOSZ \r\n\r\nt!!Ⅳ
9\n're", "tokens": 39, "pieces": ["!!'", "ſ", "…\n", "#$%", "…", "\tİ𐞁", "½", "(", " ", "Ⅳ", "HTTPServer", "123", "456", "78", "i", "OSZ", " \r\n\r\n", "t", "!!", "Ⅳ", "
", "9", "\n", "'re"]} +{"text": "\tcamelCase'S/\r\nABC'ReiOS㋿'Res \r\n12345678…\r \n  -iOS漢🙂٣٤٥٦é…㍿ \r\n\r\nA's!!-camelCase\rABC<|fim_prefix|>>/ḍ̇!!", "tokens": 60, "pieces": ["\tcamel", "Case'S", "/\r\n", "ABC'Re", "i", "OS", "㋿'", "Res", " \r\n", "123", "456", "78", "…\r \n", " ", " ", "-i", "OS漢", "🙂", "٣٤٥", "٦", "é", "…", "㍿", " \r\n\r\n", "A's", "!!-", "camel", "Case", "\r", "ABC", "<|", "fim", "_prefix", "|>>/", "ḍ̇", "!!"]} +{"text": "'T å'VE \n \n 'T\nꟲſ ßé'M‍\rAb😀🏽/\r\nA३/\r\nع !!é㍿", "tokens": 37, "pieces": ["'T", " å'VE", " \n \n", " '", "T", "\n", "ꟲſ", " ", " ßé'M", "‍\r", "Ab", "😀🏽/\r\n", "A", "३", "/\r\n", "ع", " ", "!!", "é", "㍿"]} +{"text": "s㍿>٣٤٥٦12345678\r\n\r\n ​\r\n's字s字😀🏽's𐞁Džungla㋿ABC'ſDžungla🙂t'VE\n/eDž \n'SeiOS<字'SAbع", "tokens": 66, "pieces": ["s", "㍿>", "٣٤٥", "٦12", "345", "678", "\r\n\r\n", " ​\r\n", "'s字s字", "😀🏽'", "s𐞁", "Džungla", "㋿ABC'ſ", "Džungla", "🙂t'VE", "\n", "/e", "Dž", " \n", "'Sei", "OS", "<<", "META", "_START", ">字'S", "Abع"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "‍! \n're​.", "tokens": 5, "pieces": ["‍!", " \n", "'re", "​."]} +{"text": "㍿'TiOSiOS!'re'S >ſ ꟲ'T٣٤٥٦'Reſ漢'ReᵃiOS/\r\n-\r\n‍ع٣٤٥٦́0😀🏽ABC \n/\r\n\r\n", "tokens": 48, "pieces": ["㍿'", "Ti", "OSi", "OS", "!'", "re'S", " ", ">ſ", " ꟲ'T", "٣٤٥", "٦", "'Reſ漢'Re", "ᵃi", "OS", "/\r\n", "-\r\n", "‍ع", "٣٤٥", "٦", "́", "0", "😀🏽", "ABC", " \n", "/\r\n\r\n"]} +{"text": "\"!\"½字½'reZéfiDžungla", "tokens": 13, "pieces": ["\"!\"", "½", "字", "½", "'re", "Zéfi", "Džungla"]} +{"text": "afi\t#$%<|endoftext|>Ab/\r\n EOT \n ꟲ'll\t👍🏽…aBaB'Ret㍿#$%ḍ̇", "tokens": 40, "pieces": ["afi", "\t", "#$%<|", "endoftext", "|>", "Ab", "/\r\n", " EOT", " \n", " ꟲ'll", "\t", "👍🏽", "…a", "Ba", "B'Re", "t", "㍿#$%", "ḍ̇"]} +{"text": "é-\r\n\r\n\u000b字('re'D \n ٣٤٥٦३\r\n\r\n!عB!iOSé-edABC\n/ \ncamelCase.漢mcamelCasedé", "tokens": 42, "pieces": ["é", "-\r\n\r\n", "\u000b字", "('", "re'D", " \n", " ", "٣٤٥", "٦", "", "३", "\r\n\r\n", "!ع", "B", "!i", "OSé", "-ed", "ABC", "\n", "/", " \n", "camel", "Case", ".漢mcamel", "Casedé"]} +{"text": "́sDžunglaat \n 漢DžHTTPServer㋿ \n ABCꟲm", "tokens": 21, "pieces": ["́s", "Džunglaat", " \n", " 漢DžHTTPServer", "㋿", " \n", " ABCꟲm"]} +{"text": "½'M!!!", "tokens": 3, "pieces": ["½", "'M", "!!!"]} +{"text": "👍🏽 'll<|endoftext|>'D 𐞁0Dž­ſßZiOS👍🏽\"३å.ḍ̇­ ­-\r \ns'M", "tokens": 48, "pieces": ["👍🏽", " ", "'ll", "<|", "endoftext", "|>'", "D", " <", "EOT", ">𐞁", "0", "Dž", "­ſß", "Zi", "OS", "👍🏽\"", "३", "å", ".ḍ̇", "­", " ", " ­-\r", " \n", "s'M"]} +{"text": "<|fim_prefix|>>字'M9a/béiOS12345678B ㋿å'll'Mt!!ḍ̇ḍ̇́<|endoftext|>!", "tokens": 44, "pieces": ["<|", "fim", "_prefix", "|>>", "字'M", "9", "a", "/béi", "OS", "123", "456", "78", "B", " ", " ㋿", "å'll", "'Mt", "!!", "ḍ̇ḍ̇́", "<|", "endoftext", "|>!"]} +{"text": "ع\u000b/ \n #$%", "tokens": 10, "pieces": ["ع", "\u000b", "/", " \n", " <", "META", "_START", ">#$%"]} +{"text": "'réécamelCase/\r\n\"/\r\na/b'reABCfi<ع#$%aB\t", "tokens": 20, "pieces": ["'réécamel", "Case", "/\r\n", "\"/\r\n", "a", "/b're", "ABCfi", "<ع", "#$%", "a", "B", "\t"]} +{"text": "٣٤٥٦<|fim_prefix|>٣٤٥٦iOSḍ̇Dž'T\n/…Džungla…é", "tokens": 35, "pieces": ["٣٤٥", "٦", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "i", "OSḍ̇", "Dž'T", "\n", "/", "…Džungla", "…é"]} +{"text": "ABCcamelCase'DcamelCase 😀🏽字12345678ꟲ🙂 漢", "tokens": 20, "pieces": ["ABCcamel", "Case'D", "camel", "Case", " ", " 😀🏽", "字", "123", "456", "78", "ꟲ", "🙂", " 漢"]} +{"text": "ſ漢/iOS٣٤٥٦>\r㋿Z12345678'MDžungla\t'TDžungla\"'VEⅣEOTZ\n/½ta/b\r\n a'D… <|endoftext|>é'St\u000b", "tokens": 63, "pieces": ["ſ漢", "/i", "OS", "٣٤٥", "٦", ">\r", "㋿Z", "123", "456", "78", "'MDžungla", "\t", "'TDžungla", "\"'", "VE", "Ⅳ", "EOTZ", "\n", "/", "½", "ta", "/b", "\r\n", " a'D", "…", " ", "<|", "endoftext", "|>", "é'S", "t", "\u000b"]} +{"text": "\"-", "tokens": 1, "pieces": ["\"-"]} +{"text": "/", "tokens": 1, "pieces": ["/"]} +{"text": "e's 'll'ſZ\u000bDž\u000b<|endoftext|>'M \n 's/\r\nås'VEm,\r\n\r\nåḍ̇/😀🏽<|endoftext|>,s", "tokens": 50, "pieces": ["e's", " ", "'ll'ſ", "Z", "\u000bDž", "\u000b", "<|", "endoftext", "|>'", "M", " \n", " '", "s", "/\r\n", "ås'VE", "m", ",\r\n\r\n", "å", "ḍ̇", "/😀🏽<|", "endoftext", "|>,", "s"]} +{"text": "😀🏽", "tokens": 3, "pieces": ["😀🏽"]} +{"text": "/Ⅳḍ̇(ABC \r.­½​iOS…\r\n\r\n\"\n <12345678ß.HTTPServerZ ᵃ-'S\ta#$%", "tokens": 42, "pieces": ["/", "Ⅳ", "ḍ̇", "(ABC", " \r", ".­", "½", "​i", "OS", "…\r\n\r\n", "\"\n", " ", " <", "123", "456", "78", "ß", ".HTTPServer", "Z", " ᵃ", "-<", "META", "_START", ">'", "S", "\ta", "#$%"]} +{"text": "aⅣİAb", "tokens": 5, "pieces": ["a", "Ⅳ", "İAb"]} +{"text": " (t'M​ꟲ \n(́ǻEOT \n e'sEOT \n å<|fim_prefix|>camelCase ㍿\t\r\n\r\n\r\n", "tokens": 38, "pieces": [" ", "(t'M", "​ꟲ", " \n", "(́ǻ", "EOT", " \n", " e's", "EOT", " \n", " å", "<|", "fim", "_prefix", "|>", "camel", "Case", " ", "㍿", "\t\r\n\r\n\r\n"]} +{"text": "ⅣEOTt\t/\r\n­'VE𐞁\t<३-,'s\n/㋿'VE'ſ\n<|fim_prefix|>9३'Dع'M 'S
Ⅳ0aB३ABC\r\n​s🙂", "tokens": 62, "pieces": ["Ⅳ", "EOTt", "\t", "/\r\n", "­'", "VE𐞁", "\t", "<", "३", "-,'", "s", "\n", "/㋿'", "VE'ſ", "\n", "<|", "fim", "_prefix", "|>", "9३", "'Dع'M", " ", "'", "S", "", "
", "Ⅳ0", "a", "B", "३", "ABC", "\r\n", "​s", "🙂"]} +{"text": "å­𐞁<|fim_prefix|>'M٣٤٥٦ddsdcamelCase,iOSꟲé…\n/㋿écamelCase'Re'T'", "tokens": 42, "pieces": ["å", "­𐞁", "<|", "fim", "_prefix", "|>'", "M", "٣٤٥", "٦", "ddsdcamel", "Case", ",i", "OSꟲé", "…\n", "/㋿", "écamel", "Case'Re", "'T", "'"]} +{"text": "‍ſ'reé\u000bḍ̇e🙂…ſ're….㋿é'T𐞁‍#$%٣٤٥٦ 
-é", "tokens": 43, "pieces": ["‍ſ're", "é", "\u000bḍ̇e", "🙂", "…ſ're", "…", ".㋿", "é'T", "𐞁", "‍#$%", "٣٤٥", "٦", " ", "
", "-é"]} +{"text": "<|fim_prefix|>漢-\r\nZ㍿!!'ſDžungla\rDžungla'Mt…Aḍ̇​iOS
é'T𐞁(DžéeABCé", "tokens": 50, "pieces": ["<|", "fim", "_prefix", "|>", "漢", "-\r\n", "Z", "㍿!!'", "ſ", "Džungla", "\r", "Džungla'M", "t", "…Aḍ̇", "​i", "OS", "
é'T", "𐞁", "(Džée", "ABCé"]} +{"text": ".漢", "tokens": 2, "pieces": [".漢"]} +{"text": "ᵃm", "tokens": 4, "pieces": ["ᵃm"]} +{"text": "́😀🏽DžHTTPServer.\n/İ< /\r\nḍ̇/\r\n're🙂'T'ſⅣ字a/b\u000b", "tokens": 29, "pieces": ["́", "😀🏽", "DžHTTPServer", ".\n/", "İ", "<", " /\r\n", "ḍ̇", "/\r\n", "'re", "🙂'", "T'ſ", "Ⅳ", "字a", "/b", "\u000b"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "㍿½٣٤٥٦a/bå'Reå'T", "tokens": 16, "pieces": ["㍿", "½٣٤", "٥٦", "a", "/bå'Re", "å'T"]} +{"text": "ꟲ,
 \nmß漢\u000b\na/bB́a/b/\r\n\r\nİfiİ", "tokens": 21, "pieces": ["ꟲ", ",", "
 \n", "mß漢", "\u000b\n", "a", "/b", "B́a", "/b", "/\r\n\r\n", "İfi", "İ"]} +{"text": " \n ३'ſ \n", "tokens": 6, "pieces": [" \n", " ", "३", "'ſ", " \n"]} +{"text": "9\n/DžunglaHTTPServer/漢ᵃ'TiOSHTTPServer߅'S", "tokens": 27, "pieces": ["", "9", "\n", "/Džungla", "HTTPServer", "/漢ᵃ'T", "i", "OSHTTPServerß", "…", "'S"]} +{"text": "fi's", "tokens": 2, "pieces": ["fi's"]} +{"text": "\r'Re ́/'MaBB😀🏽#$%<|endoftext|>\"d🙂aB", "tokens": 23, "pieces": ["\r", "'Re", " ́", "/'", "Ma", "BB", "😀🏽#$%<|", "endoftext", "|>\"", "d", "🙂a", "B"]} +{"text": "\tDžungla🙂漢字½ \n/\"'Mß\r\n\r\n\r\n/\r\n㋿å\"é.", "tokens": 25, "pieces": ["\tDžungla", "🙂漢字", "½", " \n", "/\"'", "Mß", "\r\n\r\n\r\n", "/\r\n", "㋿å", "\"é", "."]} +{"text": "iOS㍿a", "tokens": 6, "pieces": ["i", "OS", "㍿a"]} +{"text": "ᵃB\"HTTPServerᵃ, \r\n\r\né<|endoftext|>İ­EOT\r\nåḍ̇'DⅣ,'Re…٣٤٥٦\"a/b(.'re-'D
'Tåd 😀🏽\r'", "tokens": 60, "pieces": ["ᵃ", "B", "\"HTTPServerᵃ", ",", " \r\n\r\n", "é", "<|", "endoftext", "|>", "İ", "­EOT", "\r\n", "åḍ̇'D", "Ⅳ", ",'", "Re", "…", "٣٤٥", "٦", "\"a", "/b", "(.'", "re", "-'", "D", "
", "'Tåd", " ", "😀🏽\r", "'"]} +{"text": "'re!!EOT'llAbs'VEå🙂…<',12345678a/b ​㍿\n/
 HTTPServer(\n'M​", "tokens": 33, "pieces": ["'re", "!!", "EOT'll", "Abs'VE", "å", "🙂", "…", "<',", "123", "456", "78", "a", "/b", " ​㍿\n/", "
", " HTTPServer", "(\n", "'M", "​"]} +{"text": "eEOTfi' \n\" 'M", "tokens": 9, "pieces": ["e", "EOTfi", "'", " \n", "\"", " ", "'M"]} +{"text": "‍'SAbt㋿ !<㍿HTTPServer 'T/éåA/\r\n㍿३/\r\nع>!!<|fim_prefix|>\rt'TaB.9", "tokens": 51, "pieces": ["‍'", "SAbt", "㋿", " <", "META", "_START", ">!<㍿", "HTTPServer", "", " ", "'T", "/éå", "A", "/\r\n", "㍿", "३", "/\r\n", "ع", ">!!<|", "fim", "_prefix", "|>\r", "t'T", "a", "B", ".", "9"]} +{"text": "ḍ̇३9HTTPServer ,camelCaseåḍ̇İ\"'ſDžungla12345678 \n <|endoftext|><|fim_prefix|>éſ,Ⅳ𐞁åHTTPServerꟲ½fi'Reع #$%å ", "tokens": 67, "pieces": ["ḍ̇", "३9", "HTTPServer", " ", ",camel", "Caseåḍ̇", "İ", "\"'", "ſ", "Džungla", "123", "456", "78", " \n", " <|", "endoftext", "|><|", "fim", "_prefix", "|>", "éſ", ",", "Ⅳ", "𐞁å", "HTTPServerꟲ", "½", "fi'Re", "ع", " ", " #$%", "å", " "]} +{"text": "< \n㍿'M😀🏽<㍿'Re\n/ḍ̇(\r\n\r\n<|endoftext|>ſ \n,'ſ", "tokens": 37, "pieces": ["<", " \n", "㍿<", "EOT", ">'", "M", "😀🏽<㍿'", "Re", "\n", "/ḍ̇", "(\r\n\r\n", "<|", "endoftext", "|>", "ſ", " \n", ",'", "ſ"]} +{"text": "ᵃ/ \né'll­iOSḍ̇HTTPServera½字漢👍🏽iOSé's<|fim_prefix|>iOS(…éHTTPServer‍!!0,ſa/b/\r\n
", "tokens": 52, "pieces": ["ᵃ", "/", " \n", "é'll", "­i", "OS", "ḍ̇", "HTTPServera", "½", "字漢", "👍🏽", "i", "OSé's", "<|", "fim", "_prefix", "|>", "i", "OS", "(", "…é", "HTTPServer", "‍!!", "0", ",ſa", "/b", "/\r\n", "
"]} +{"text": "­Ab㍿'ll字㋿>'re\t'reZ'漢'T\r\n\r\nfi'S…\n/'VE<|fim_prefix|>9s'Sé٣٤٥٦", "tokens": 42, "pieces": ["­Ab", "㍿'", "ll字", "㋿>'", "re", "\t", "'re", "Z", "'漢'T", "\r\n\r\n", "fi'S", "…\n", "/'", "VE", "<|", "fim", "_prefix", "|>", "9", "s'S", "é", "٣٤٥", "٦"]} +{"text": "é'ſZ -ß,\r\n\r\ncamelCase/\r\n\" !!𐞁\r\n\u000b\r\nZ \n<|endoftext|>#$%٣٤٥٦​𐞁 \n", "tokens": 42, "pieces": ["é'ſ", "Z", " ", "-ß", ",\r\n\r\n", "camel", "Case", "/\r\n", "\"", " ", " !!", "𐞁", "\r\n\u000b\r\n", "Z", " \n", "<|", "endoftext", "|>#$%", "٣٤٥", "٦", "​𐞁", " \n"]} +{"text": "👍🏽iOS'SABC", "tokens": 7, "pieces": ["👍🏽", "i", "OS'S", "ABC"]} +{"text": "<漢ḍ̇/\r\n㋿fi字\u000b-", "tokens": 13, "pieces": ["<漢ḍ̇", "/\r\n", "㋿fi字", "\u000b", "-"]} +{"text": " \nåḍ̇ſ\"", "tokens": 14, "pieces": [" \n", "/,'", "S", "!<", "EOT", ">ḍ̇ſ", "\""]} +{"text": "d\u000b字ABC٣٤٥٦👍🏽", "tokens": 11, "pieces": ["d", "\u000b字", "ABC", "٣٤٥", "٦", "👍🏽"]} +{"text": "㋿\t", "tokens": 4, "pieces": ["㋿", "\t"]} +{"text": "ᵃ", "tokens": 6, "pieces": ["ᵃ"]} +{"text": "字'll\r\n /d \nDž Džungla.12345678​ !!Dž'VEḍ̇/\r\n­'ſAåfi'VE", "tokens": 46, "pieces": ["字'll", "\r\n", " /", "d", " \n", "Dž", " Džungla", ".", "123", "456", "78", "​", " <", "EOT", ">!!", "Dž", "'", "VEḍ̇", "/\r\n", "­'", "ſ", "Aåfi'VE", ""]} +{"text": "'Må‍३'VE<|fim_prefix|>
Z'é<\r\n\r\n'S!fié‍", "tokens": 23, "pieces": ["'Må", "‍", "३", "'VE", "<|", "fim", "_prefix", "|>", "
Z", "'é", "<\r\n\r\n", "'S", "!fié", "‍"]} +{"text": "å /\r\n/!Ⅳ \n !٣٤٥٦", "tokens": 14, "pieces": ["å", " /\r\n/", "!", "Ⅳ", " \n", " !", "٣٤٥", "٦"]} +{"text": "'T 漢 \n 'Dḍ̇mDž<|fim_prefix|> ​
…ſ<३  \u000b>Z'D 👍🏽ea­iOS0漢Džungla 'ſⅣZ.", "tokens": 51, "pieces": ["'T", " ", " 漢", " \n", " '", "Dḍ̇m", "Dž", "<|", "fim", "_prefix", "|>", " ", " ​", "
", "…ſ", "<", "३", "  ", "\u000b", ">Z'D", " ", "👍🏽", "ea", "­i", "OS", "0", "漢Džungla", " '", "ſ", "Ⅳ", "Z", "."]} +{"text": "HTTPServerß>camelCase12345678e İDžungla'Re\t", "tokens": 20, "pieces": ["HTTPServerß", ">", "camel", "Case", "123", "456", "78", "e", " İDžungla'Re", "\t"]} +{"text": ",\r\n\r\n<|endoftext|>å ABC字'S \n'VE㍿Dž(HTTPServer \n'S🙂é​'S'ſeaé /\r\n \r\n'll字", "tokens": 50, "pieces": [",\r\n\r\n", "<|", "endoftext", "|>", "å", " ", " ABC字", "'", "S", " \n", "'VE", "㍿Dž", "(HTTPServer", " \n", "'S", "🙂é", "​'", "S'ſ", "eaé", " ", " /\r\n", " ", " <", "META", "_START", ">\r\n", "'ll字"]} +{"text": "㍿iOSEOT<|fim_prefix|>a/b'T/\r\nABC'll'll'VEt'Sſet'SAb0-Džunglaſ'ſ'S\n字字>\r\ndé🙂å㋿👍🏽𐞁0Dž", "tokens": 61, "pieces": ["㍿i", "OSEOT", "<|", "fim", "_prefix", "|>", "a", "/b'T", "/\r\n", "ABC", "'", "ll'll", "'VEt'S", "ſet'S", "Ab", "0", "-Džunglaſ'ſ", "'S", "\n", "字字", ">\r\n", "dé", "🙂å", "㋿👍🏽", "𐞁", "0", "Dž"]} +{"text": "'M½'ſ'll
\u000b iOSDž!!\t/\r\n字Z㍿ḍ̇<|fim_prefix|>ع㍿!", "tokens": 33, "pieces": ["'M", "½", "'ſ'll", "
\u000b", " i", "OSDž", "!!", "\t", "/\r\n", "字", "Z", "㍿ḍ̇", "<|", "fim", "_prefix", "|>", "ع", "㍿!"]} +{"text": ">😀🏽ᵃ<|fim_prefix|>ꟲABC/\r\n \n ½\r\n> \n\r🙂's३ \n 漢ꟲ\n/'ſ(\niOS\r\n\t, \n\r\n", "tokens": 47, "pieces": [">😀🏽", "ᵃ", "<|", "fim", "_prefix", "|>", "ꟲ", "ABC", "/\r\n", " \n", " ", " ", "½", "\r\n", ">", " \n\r", "🙂'", "s", "३", " \n", " 漢ꟲ", "\n", "/'", "ſ", "(\n", "i", "OS", "\r\n", "\t", ",", " \n\r\n"]} +{"text": "㍿'S\n/-㋿
fi‍e'ſ'se‍é", "tokens": 24, "pieces": ["㍿'", "S", "\n", "/-㋿", "
fi", "‍e'ſ", "'", "se", "‍é"]} +{"text": "🙂Džungla \n éABCABC \n/\r\n漢'T㋿👍🏽Ab😀🏽३\r\n'll\"'ſBaB<㋿㋿å\nd-iOS#$%漢aBHTTPServer", "tokens": 50, "pieces": ["🙂Džungla", " \n", " é", "ABCABC", " \n", "/\r\n", "漢'T", "㋿👍🏽", "Ab", "😀🏽", "३", "\r\n", "'ll", "\"'", "ſ", "Ba", "B", "<㋿㋿", "å", "\n", "d", "-i", "OS", "#$%", "漢a", "BHTTPServer"]} +{"text": "/\r\n \n'VE/ \ne!\n/½'ſ/\r\n \n fi٣٤٥٦12345678
9-\ncamelCase\n/'sA\n!", "tokens": 19, "pieces": ["e'M", "", "123", "456", "78", "
", "9", "-\n", "camel", "Case", "\n", "/'", "s", "A", "\n", "!"]} +{"text": "ꟲEOT👍🏽字>dABC\t字's<'ſfi'S­㋿'Mß'D \r", "tokens": 28, "pieces": ["ꟲ", "EOT", "👍🏽", "字", ">d", "ABC", "\t字's", "<'", "ſfi'S", "­㋿'", "Mß'D", " \r"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": " 'S'Re(", "tokens": 4, "pieces": [" ", "'S'Re", "("]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "​ \n \n'VE Ⅳ(a/bEOT<|endoftext|>!!#$%𐞁🙂३\r字/\r\n\"ß🙂'sḍ̇́", "tokens": 51, "pieces": ["​", " \n \n", "'VE", " ", " ", "Ⅳ", "(a", "/b", "EOT", "<|", "endoftext", "|>!!<", "META", "_START", ">#$%", "𐞁", "🙂", "३", "\r", "字", "/\r\n", "\"ß", "🙂'", "sḍ̇́"]} +{"text": "㋿å's", "tokens": 6, "pieces": ["㋿å's"]} +{"text": " ", "tokens": 1, "pieces": [" "]} +{"text": "'ś३'re\r\r\n\r\n㋿.ꟲé㍿'M \n ſ<|endoftext|>å9Z0Z㍿>ABCAbⅣ'SaB'll'reB", "tokens": 49, "pieces": ["'ś", "३", "'re", "\r\r\n\r\n", "㋿.", "ꟲé", "㍿'", "M", " \n", " ſ", "<|", "endoftext", "|>", "å", "9", "Z", "0", "Z", "㍿>", "ABCAb", "Ⅳ", "'Sa", "B'll", "'re", "B"]} +{"text": "ABCⅣ-ᵃ \n 'THTTPServeréHTTPServer٣٤٥٦ \n<字'ſfi m𐞁é'llå1234567812345678", "tokens": 42, "pieces": ["ABC", "Ⅳ", "-ᵃ", " \n", " '", "THTTPServeré", "HTTPServer", "٣٤٥", "٦", " \n", "<字'ſ", "fi", " ", " m𐞁é'll", "å", "123", "456", "781", "234", "567", "8"]} +{"text": "#$%mİع㍿a/bfi𐞁fi३\r\n\r\n>'llHTTPServer'ſé<|fim_prefix|>…ſeZt0!! 漢\u000b\r\n½ \n.İ㋿fi漢 ", "tokens": 62, "pieces": ["#$%", "m", "İع", "㍿a", "/bfi𐞁fi", "३", "\r\n\r\n", ">'", "ll", "HTTPServer'ſ", "é", "<|", "fim", "_prefix", "|>", "…ſe", "Zt", "0", "!!", " ", " 漢", "\u000b\r\n", "½", "", " \n", ".İ", "㋿fi漢", " "]} +{"text": " (.", "tokens": 2, "pieces": [" ", "(."]} +{"text": "㋿'ll…३😀🏽ABCEOT½'Re,'M'VE'reZAb 'ſ'VE('ll.ꟲ\t/'Dé-é\"!", "tokens": 43, "pieces": ["㋿'", "ll", "…", "३", "😀🏽", "ABCEOT", "½", "'Re", ",'", "M'VE", "'re", "ZAb", " ", "'ſ'VE", "('", "ll", ".ꟲ", "\t", "/'", "Dé", "-", "é", "\"!"]} +{"text": "'VEDžunglaA'ſ'S'Meé/\r\n.<|endoftext|>Ⅳ>As٣٤٥٦12345678½  'ſå12345678' sſ", "tokens": 50, "pieces": ["'VEDžungla", "A'ſ", "'S'M", "eé", "/\r\n", ".<", "META", "_START", "><|", "endoftext", "|>", "Ⅳ", ">As", "٣٤٥", "٦12", "345", "678", "½", " ", " ", "'ſå", "123", "456", "78", "'", " ", " sſ"]} +{"text": "'s.'ll🙂ſ/!aB", "tokens": 9, "pieces": ["'s", ".'", "ll", "🙂ſ", "/!", "a", "B"]} +{"text": ">(𐞁EOT>aa ! !A漢9å \n 'Dᵃå \n ", "tokens": 33, "pieces": [">(", "𐞁", "EOT", ">aa", " <", "META", "_START", ">!", " !", "A漢", "9", "å", " \n", " '", "Dᵃå", " \n", " <", "META", "_START", ">"]} +{"text": "aiOS \n ‍‍DžacamelCaseecamelCase
camelCaseḍ̇㋿​a/b\rt \n  camelCase‍t<|fim_prefix|>", "tokens": 40, "pieces": ["ai", "OS", " \n", " ‍‍", "Džacamel", "Caseecamel", "Case", "
camel", "Caseḍ̇", "㋿​", "a", "/b", "\r", "t", " \n", " ", " camel", "Case", "‍t", "<|", "fim", "_prefix", "|>"]} +{"text": "́'sfiع<|fim_prefix|>(!!٣٤٥٦Ⅳ>ḍ̇Z", "tokens": 26, "pieces": ["́'s", "fiع", "<|", "fim", "_prefix", "|>(!!", "٣٤٥", "٦Ⅳ", ">ḍ̇", "Z"]} +{"text": "\r\n\r\néHTTPServer👍🏽\r\n\r\nacamelCase‍­,(0\rHTTPServer👍🏽\n/å0,ß \n Dž🙂s \n ㍿", "tokens": 40, "pieces": ["\r\n\r\n", "é", "HTTPServer", "👍🏽\r\n\r\n", "acamel", "Case", "‍­,(", "0", "\r", "HTTPServer", "👍🏽\n/", "å", "0", ",ß", " \n", " Dž", "🙂s", " \n", " ㍿"]} +{"text": "­
m…'T٣٤٥٦Ⅳ!'S!", "tokens": 15, "pieces": ["­", "
m", "…", "'T", "٣٤٥", "٦Ⅳ", "!'", "S", "!"]} +{"text": "<|fim_prefix|>.\u000baB \n/Abß٣٤٥٦a/b
  ㍿å\tع>ß\r\niOS'll㍿EOT", "tokens": 41, "pieces": ["<|", "fim", "_prefix", "|>.", "\u000ba", "B", " \n", "/Abß", "٣٤٥", "٦", "a", "/b", "
 ", " ", "㍿å", "\tع", ">ß", "\r\n", "i", "OS'll", "㍿EOT"]} +{"text": "9.Z<|fim_prefix|>😀🏽ᵃt'D", "tokens": 20, "pieces": ["9", ".Z", "<|", "fim", "_prefix", "|>😀🏽", "ᵃt'D", ""]} +{"text": "'D३(字'S9½/\r\n
\ré'T", "tokens": 12, "pieces": ["'D", "३", "(字'S", "9½", "/\r\n", "
\r", "é'T"]} +{"text": "'VE-fiAb𐞁,😀🏽 \nHTTPServerİᵃ/(B", "tokens": 22, "pieces": ["'VE", "-fi", "Ab𐞁", ",😀🏽", " \n", "HTTPServer", "İᵃ", "/(", "B"]} +{"text": "'Reé½('ABCa'D/\r\t👍🏽 \nİfi<​\">ßEOTꟲ9ßsd🙂/'re\r\n\r\n<12345678", "tokens": 41, "pieces": ["'Reé", "½", "('", "ABCa'D", "/\r", "\t", "👍🏽", " \n", "İfi", "<​\">", "ß", "EOTꟲ", "9", "ßsd", "🙂/'", "re", "\r\n\r\n", "<", "123", "456", "78", ""]} +{"text": "é\r\r\n\r0'ſ'#$%<|fim_prefix|>a", "tokens": 17, "pieces": ["é", "\r\r\n\r", "0", "'ſ", "'#$%<|", "fim", "_prefix", "|>", "a"]} +{"text": "/\r\n\r\nDžungla'M'ReAb\n/'ll㋿
३٣٤٥٦ 'SaBZé\tEOT12345678ع\nEOTſ#$%éiOS-tsa/bDž(㍿!!0", "tokens": 56, "pieces": ["/\r\n\r\n", "Džungla'M", "'Re", "Ab", "\n", "/'", "ll", "㋿", "
", "३٣٤", "٥٦", " ", "'Sa", "BZé", "\tEOT", "", "123", "456", "78", "ع", "\n", "EOTſ", "#$%", "éi", "OS", "-tsa", "/b", "Dž", "(㍿!!", "0"]} +{"text": "fi𐞁é‍t.", "tokens": 9, "pieces": ["fi𐞁é", "‍t", "."]} +{"text": "-…#$%.'<Džungla'StEOTEOT9'T½(\u000b!!\n/e#$%iOS \n", "tokens": 29, "pieces": ["-", "…", "#$%.'<", "Džungla'S", "t", "EOTEOT", "9", "'T", "½", "(", "\u000b", "!!\n/", "e", "#$%", "i", "OS", " \n"]} +{"text": "\r\n!!HTTPServer\t\"­sd\"½'VE'D're<|fim_prefix|>!!-ع", "tokens": 26, "pieces": ["\r\n", "!!", "HTTPServer", "\t", "\"­", "sd", "\"", "½", "'VE'D", "'re", "<|", "fim", "_prefix", "|>!!-", "ع"]} +{"text": "😀🏽​­'ſ's. \r\n\r\néßm 're", "tokens": 17, "pieces": ["😀🏽​­'", "ſ's", ".", " \r\n\r\n", "éßm", " ", "'re"]} +{"text": "Z0‍e", "tokens": 11, "pieces": ["Z", "", "0", "‍", "e"]} +{"text": "\n𐞁ꟲ'sa/bA >a/be12345678 \nA\r<|endoftext|>a\tDž.…d㋿", "tokens": 43, "pieces": ["\n", "𐞁ꟲ's", "a", "/b", "A", " >", "a", "/be", "123", "456", "78", " \n", "A", "\r", "<|", "endoftext", "|>", "a", "\tDž", ".", "…d", "㋿"]} +{"text": "🙂 \n iOS\"camelCaseA", "tokens": 8, "pieces": ["🙂", " \n", " i", "OS", "\"camel", "Case", "A"]} +{"text": "camelCaseDž'Re \nAb字 a \n!!éZé!
 \n 'A,/😀🏽!! 's٣٤٥٦ſ/ ٣٤٥٦", "tokens": 46, "pieces": ["camel", "Case", "Dž'Re", " \n", "Ab字", " a", " \n", "!!", "é", "Zé", "!", "
 \n", " '", "A", ",/😀🏽!!", " '", "s", "٣٤٥", "٦", "ſ", "/", " ", " ", "٣٤٥", "٦"]} +{"text": "/İ>‍漢 \nfi​ /\r\nß(‍👍🏽'DⅣعᵃ \n aDžungla  \n s", "tokens": 33, "pieces": ["/İ", ">‍", "漢", " \n", "fi", "​", " ", "/\r\n", "ß", "(‍👍🏽'", "D", "Ⅳ", "عᵃ", " \n", " a", "Džungla", "  \n", " s"]} +{"text": " /\r\nعHTTPServer३./\r\n9३'ſ 'VEAb<|fim_prefix|>Aé '३\r\n३㋿\nB9\r\na \n​\" ,عfiEOT-​", "tokens": 48, "pieces": [" ", "/\r\n", "عHTTPServer", "३", "./\r\n", "9३", "'ſ", " ", "'VEAb", "<|", "fim", "_prefix", "|>", "Aé", " ", " '", "३", "\r\n", "३", "㋿\n", "B", "9", "\r\n", "a", " \n", "​\"", " ", " ,", "عfi", "EOT", "-​"]} +{"text": "👍🏽\r\n٣٤٥٦'S<|endoftext|>fiABCᵃ𐞁,sa/b\u000b\r\n.Džéꟲᵃ🙂", "tokens": 44, "pieces": ["👍🏽\r\n", "٣٤٥", "٦", "'S", "<|", "endoftext", "|>", "fi", "ABCᵃ𐞁", ",sa", "/b", "\u000b", "\r\n", ".Džéꟲᵃ", "🙂"]} +{"text": "ABC'ReaB \nfi !.'ll  -ꟲ\r\n\r\n
é\u000b//dZ👍🏽ḍ̇'Mßs ​t🙂 \na/b'SB-‍ ", "tokens": 44, "pieces": ["ABC'Re", "a", "B", " \n", "fi", " !.'", "ll", " ", " -", "ꟲ", "\r\n\r\n", "
é", "\u000b", "//", "d", "Z", "👍🏽", "ḍ̇'M", "ßs", " ", "​t", "🙂", " \n", "a", "/b", "'", "SB", "-‍", " "]} +{"text": "a/béZ\rDžunglaᵃ😀🏽 \n\u000b're\rABC", "tokens": 20, "pieces": ["a", "/bé", "Z", "\r", "Džunglaᵃ", "😀🏽", " \n", "\u000b", "'re", "\r", "ABC"]} +{"text": "\u000bé0㍿fiB'S\t­'ll'camelCase \n'M", "tokens": 18, "pieces": ["\u000bé", "0", "㍿fi", "B'S", "\t", "­'", "ll", "'camel", "Case", " \n", "'M"]} +{"text": "\n-", "tokens": 2, "pieces": ["\n", "-"]} +{"text": "字é😀🏽ABC'ſ \n \n…'T‍Dž‍a/b३m/!'T'T🙂HTTPServer½㋿", " \n", "…", "'T", "‍Dž", "‍a", "/b", "३", "m", "/!'", "T'T", "🙂HTTPServer", "½", "㋿<", "Abꟲi", "OS", "/\r\n", "e're", " ", "㍿HTTPServers", "३", "/\r\n\r\n"]} +{"text": "<|endoftext|> 漢'T३/\r\nع('St字\r\n\r\nع㍿ꟲ٣٤٥٦é#$%漢'VEcamelCase", "tokens": 37, "pieces": ["<|", "endoftext", "|>", " 漢'T", "३", "/\r\n", "ع", "('", "St字", "\r\n\r\n", "ع", "㍿ꟲ", "٣٤٥", "٦", "é", "#$%", "漢'VE", "camel", "Case"]} +{"text": " \n'Mfi字 ㍿'a/ba/ḍ̇camelCasem\r \n!!'ſ!eå\r\n('ſå\t\r\n\r\n'D👍🏽\r\n\r\n\r\n\r\nAb", "tokens": 43, "pieces": [" \n", "'Mfi字", " ", "㍿'", "a", "/ba", "/ḍ̇camel", "Casem", "\r \n", "!!'", "ſ", "!eå", "\r\n", "('", "ſå", "\t\r\n\r\n", "'D", "👍🏽\r\n\r\n\r\n\r\n", "Ab"]} +{"text": "\r\n\r\nİZ🙂'İİa/b", "tokens": 11, "pieces": ["\r\n\r\n", "İZ", "🙂'", "İİ", "a", "/b"]} +{"text": "ß9.½'Re字//\r\n🙂a/baBعfi½camelCase/\r\n", "tokens": 21, "pieces": ["ß", "9", ".", "½", "'Re字", "//\r\n", "🙂a", "/ba", "Bعfi", "½", "camel", "Case", "/\r\n"]} +{"text": "\u000bZİ", "tokens": 5, "pieces": ["\u000b", "Zİ"]} +{"text": ">HTTPServerDžungla-HTTPServer aBAbm!!", "tokens": 15, "pieces": [">HTTPServer", "Džungla", "-HTTPServer", " a", "BAbm", "!!"]} +{"text": "iOS\"Bḍ̇ttᵃᵃ12345678ḍ̇9İ३🙂İ‍/.ſ'VE'ſ'SiOS'T'ꟲ'‍mᵃ. s", "tokens": 52, "pieces": ["i", "OS", "\"Bḍ̇ttᵃᵃ", "123", "456", "78", "ḍ̇", "9", "İ", "३", "🙂İ", "‍/.", "ſ'VE", "'ſ'S", "i", "OS'T", "'ꟲ", "'‍", "mᵃ", ".", " s"]} +{"text": "字<|fim_prefix|>", "tokens": 7, "pieces": ["字", "<|", "fim", "_prefix", "|>"]} +{"text": "🙂 're İ'S­ſ\r\r\n\r\nś'ſ‍0 \nſ'Re ½(aB're'Reé-ꟲ'ReDžungla‍EOT… (", "tokens": 42, "pieces": ["🙂", " '", "re", " İ'S", "­ſ", "\r\r\n\r\n", "ś'ſ", "‍", "0", " \n", "ſ'Re", " ", "½", "(a", "B're", "'Reé", "-ꟲ'Re", "Džungla", "‍EOT", "…", " ", "("]} +{"text": "BcamelCase'HTTPServer", "tokens": 9, "pieces": ["Bcamel", "Case", "'<", "META", "_START", ">HTTPServer"]} +{"text": "'ll'EOT\r٣٤٥٦ꟲa/b'VE\r\n<|endoftext|>'ſḍ̇<|fim_prefix|>'S''llAb .'re٣٤٥٦ḍ̇", "tokens": 53, "pieces": ["'ll", "'EOT", "\r", "٣٤٥", "٦", "ꟲa", "/b'VE", "\r\n", "<|", "endoftext", "|>'", "ſḍ̇", "<|", "fim", "_prefix", "|><", "META", "_START", ">'", "S", "''", "ll", "Ab", " ", ".'", "re", "٣٤٥", "٦", "ḍ̇"]} +{"text": "#$% \n ३'reꟲ", "tokens": 9, "pieces": ["#$%", " \n", " ", "३", "'reꟲ"]} +{"text": " \n's'Re\n/漢३​\t \n…", "tokens": 11, "pieces": [" \n", "'s'Re", "\n", "/漢", "३", "​", "\t \n", "…"]} +{"text": "Ⅳ㋿漢-İꟲᵃ\"aB
字#$%<|fim_prefix|>\r\n\r\n sİ/\r\n'll>'DEOT<­aDžunglaB,٣٤٥٦aAb(/å\t", "tokens": 53, "pieces": ["Ⅳ", "㋿漢", "-İꟲᵃ", "\"a", "B", "
字", "#$%<|", "fim", "_prefix", "|>\r\n\r\n", " s", "İ", "/\r\n", "'ll", ">'", "DEOT", "<­", "a", "Džungla", "B", ",", "٣٤٥", "٦", "a", "Ab", "(/", "å", "\t"]} +{"text": "EOT㍿\ré\"ſ9<|endoftext|>Džungla's'S'll٣٤٥٦fi<|fim_prefix|>Ⅳ\n­ 0camelCaseé12345678e ' \naB<|fim_prefix|> AåDžungla0㋿‍", "tokens": 73, "pieces": ["EOT", "㍿\r", "é", "\"ſ", "9", "<|", "endoftext", "|>", "Džungla's", "'S'll", "٣٤٥", "٦", "fi", "<|", "fim", "_prefix", "|>", "Ⅳ", "\n", "­", " ", "0", "camel", "Caseé", "123", "456", "78", "e", " ", " '", " \n", "a", "B", "<|", "fim", "_prefix", "|>", " Aå", "Džungla", "0", "㋿‍"]} +{"text": "👍🏽㍿\r\n\r\n\u000bé12345678㋿camelCaseⅣa/bZ<\r\n\r\n漢fi", "tokens": 57, "pieces": ["é", "Dž", "\"", "123", "456", "78", "/\r\n", " ", "'M", "<|", "fim", "_prefix", "|>㋿", "camel", "Case", "Ⅳ", "a", "/b", "Z", "<\r\n\r\n", "漢fi", ""]} +{"text": "㋿字'VE!!Ⅳ'T­#$%㍿漢camelCase", "tokens": 16, "pieces": ["'re", "\r\n", " 𐞁t", "/\r\n", ">㍿", "漢camel", "Case"]} +{"text": " \n 0a<​EOT٣٤٥٦‍( \n'reiOS'Reꟲ(字'MAb\n/'re­,#$%tEOTé😀🏽 ㍿a/b", "tokens": 48, "pieces": [" \n", " ", " ", "0", "a", "<​", "EOT", "٣٤٥", "٦", "‍(", " \n", "'rei", "OS'Re", "ꟲ", "(字'M", "Ab", "\n", "/'", "re", "­,#$%", "t", "EOTé", "😀🏽", " ", "㍿a", "/b"]} +{"text": "HTTPServerſ'T0", "tokens": 5, "pieces": ["HTTPServerſ'T", "0"]} +{"text": "'M😀🏽٣٤٥٦<́é'D 'ſ㋿ ½a/bᵃ'Re…Ⅳ-ZſaB,​‍'\r\naBEOT‍ᵃ0,'ſ", "tokens": 49, "pieces": ["'M", "😀🏽", "٣٤٥", "٦", "<́é'D", " '", "ſ", "㋿", " ", " ", "½", "a", "/bᵃ'Re", "…", "Ⅳ", "-Zſa", "B", ",​‍'\r\n", "a", "BEOT", "‍ᵃ", "0", ",'", "ſ"]} +{"text": "B\n'Re🙂 ABC-B", "tokens": 7, "pieces": ["B", "\n", "'Re", "🙂", " ", " ABC", "-B"]} +{"text": "-camelCaseeⅣé.३字\"deع'll😀🏽漢३'aB'Da/b\"camelCase(عſm漢
", "tokens": 33, "pieces": ["-camel", "Casee", "Ⅳ", "é", ".", "३", "字", "\"deع'll", "😀🏽", "漢", "३", "'a", "B'D", "a", "/b", "\"camel", "Case", "(عſm漢", "
"]} +{"text": "eAbé३'T12345678 \n", "tokens": 10, "pieces": ["e", "Abé", "३", "'T", "123", "456", "78", " \n"]} +{"text": "Z12345678é\n/0 m İ​EOT>!\r\n\n/'ll‍-'s", "tokens": 23, "pieces": ["Z", "123", "456", "78", "é", "\n", "/", "0", " ", " m", " İ", "​EOT", ">!\r\n\n/", "'ll", "‍-'", "s"]} +{"text": "'å㋿/\r\nAḍ̇dⅣ
d👍🏽!\tå<​३'İ ­,'re", "tokens": 36, "pieces": ["'å", "㋿/\r\n", "Aḍ̇d", "Ⅳ", "
d", "👍🏽!", "\tå", "<<", "META", "_START", "><", "EOT", ">​", "३", "'İ", " ", " ­,'", "re"]} +{"text": "Ⅳ Džungla,'VEꟲ㍿", "tokens": 22, "pieces": ["Ⅳ", " Džungla", ",'", "VEꟲ", "㍿"]} +{"text": "'ſ३㍿\u000b'Re​'reꟲᵃ😀🏽0٣٤٥٦\r\nmm\nAb/\r\n'T½'Re<|fim_prefix|>عéعcamelCase.'ſ \r\n\r\n12345678AAb‍<|endoftext|>\nfi", "tokens": 68, "pieces": ["'", "ſ", "३", "㍿", "\u000b", "'Re", "​'", "reꟲᵃ", "😀🏽", "0٣٤", "٥٦", "\r\n", "mm", "\n", "Ab", "/\r\n", "'T", "½", "'Re", "<|", "fim", "_prefix", "|>", "عéعcamel", "Case", ".<", "EOT", ">'", "ſ", " \r\n\r\n", "123", "456", "78", "AAb", "‍<|", "endoftext", "|>\n", "fi"]} +{"text": "\raBcamelCasetḍ̇aficamelCaseع'ſßfiع­٣٤٥٦t/ 'Re
'VEaB'll'ſ \n EOT'aBdd \n Aعa", "tokens": 47, "pieces": ["\r", "a", "Bcamel", "Casetḍ̇aficamel", "Caseع'ſ", "ßfiع", "­", "٣٤٥", "٦", "t", "/", " ", "'Re", "
", "'VEa", "B'll", "'ſ", " \n", " EOT", "'a", "Bdd", " \n", " Aعa"]} +{"text": "'s9\n/éſ🙂\n/漢عHTTPServer漢㍿iOSع<ß'ZdiOS.\r\n\r\n're ḍ̇ḍ̇\tß\ns<|endoftext|>\n/'VEſ", "tokens": 50, "pieces": ["'s", "9", "\n", "/éſ", "🙂\n/", "漢عHTTPServer漢", "㍿i", "OSع", "<ß", "'Zdi", "OS", ".\r\n\r\n", "'re", " ḍ̇ḍ̇", "\tß", "\n", "s", "<|", "endoftext", "|>\n/", "'VEſ"]} +{"text": "'sm'S'reع
\n\r­ᵃm٣٤٥٦३'ll>漢'T'llHTTPServeråſ!! 漢𐞁Ab's", "tokens": 37, "pieces": ["'sm'S", "'reع", "
\n\r", "­ᵃm", "٣٤٥", "٦३", "'ll", ">漢'T", "'ll", "HTTPServeråſ", "!!", " 漢𐞁Ab's"]} +{"text": "9a/b​½12345678 \n 123456780'M<|endoftext|>㋿'TDžungla\t🙂a/bDž \n\neDž<|endoftext|>ḍ̇!HTTPServers\r\nDžunglaع9
ꟲ/ \nDžungla'Mé", "tokens": 76, "pieces": ["9", "a", "/b", "​", "½12", "345", "678", " \n", " ", "123", "456", "780", "'M", "<|", "endoftext", "|>㋿'", "TDžungla", "\t", "🙂a", "/b", "Dž", " \n\n", "e", "Dž", "<|", "endoftext", "|>", "ḍ̇", "!HTTPServers", "\r\n", "Džunglaع", "9", "
ꟲ", "/", " \n", "Džungla'M", "é"]} +{"text": "ع㍿́å'Ⅳ👍🏽aBms\r🙂Z​!<'VE9t㋿<", "tokens": 31, "pieces": ["ع", "㍿́", "å", "'", "Ⅳ", "👍🏽", "a", "Bms", "\r", "🙂Z", "​!<'", "VE", "9", "t", "㋿<"]} +{"text": "m/d <|endoftext|>'M<|endoftext|>", "tokens": 17, "pieces": ["m", "/d", " <|", "endoftext", "|>'", "M", "<|", "endoftext", "|>"]} +{"text": "'iOS\tDžungla!\r\n\"-!'VE\r\n\r\nEOT \n/\r\n\r\n>tع99漢\t'ſDžᵃ'aé's…'٣٤٥٦\r\n\r\n-fi0", "tokens": 45, "pieces": ["'i", "OS", "\tDžungla", "!\r\n", "\"-!'", "VE", "\r\n\r\n", "EOT", " \n", "/\r\n\r\n", ">tع", "99", "漢", "\t", "'ſ", "Džᵃ", "'aé's", "…", "'", "٣٤٥", "٦", "\r\n\r\n", "-fi", "0"]} +{"text": "m 'Re12345678a/bAma0", "tokens": 11, "pieces": ["m", " ", "'Re", "123", "456", "78", "a", "/b", "Ama", "0"]} +{"text": " \nå<㋿aaB\n/ABCiOSaB-😀🏽'T\r<\n/fifi-३åḍ̇<|endoftext|>\r\nd 
'Re٣٤٥٦'re", "tokens": 57, "pieces": [" \n", "å", "<㋿<", "META", "_START", ">aa", "B", "\n", "/ABCi", "OSa", "B", "-😀🏽'", "T", "\r", "<\n/", "fifi", "-", "३", "åḍ̇", "<|", "endoftext", "|>\r\n", "d", "", " ", "
", "'Re", "٣٤٥", "٦", "'re"]} +{"text": " \n ", "tokens": 2, "pieces": [" \n", " "]} +{"text": " \n ३a/b\r\nZiOSé字!!Dž", "tokens": 14, "pieces": [" \n", " ", "३", "a", "/b", "\r\n", "Zi", "OSé字", "!!", "Dž"]} +{"text": "'M\"AſⅣİⅣ'ss12345678ſ'TaBaB0", "tokens": 20, "pieces": ["'M", "\"Aſ", "Ⅳ", "İ", "Ⅳ", "'ss", "123", "456", "78", "ſ'T", "a", "Ba", "B", "0"]} +{"text": "mع'll\nDž😀🏽 \n ..<|endoftext|>\t٣٤٥٦ \r\n\r\niOS's
!'TEOTEOTꟲ'> 'M😀🏽 /'Re​", "tokens": 56, "pieces": ["mع", "'", "ll", "\n", "Dž", "😀🏽", " \n", " <", "META", "_START", ">..<|", "endoftext", "|>", "\t", "٣٤٥", "٦", " ", " <", "META", "_START", ">\r\n\r\n", "i", "OS's", "
", "!'", "TEOTEOTꟲ", "'>", " ", "'M", "😀🏽", " ", " /'", "Re", "​"]} +{"text": "漢.٣٤٥٦a/baBDž>३ß<|fim_prefix|>漢Džungla- ><|fim_prefix|>9", "३", "ß", "<|", "fim", "_prefix", "|>", "漢Džungla", "-", " ", " ><|", "fim", "_prefix", "|>", "9", "fiå' \n fi/\r\nDž siOSḍ̇'VEé½'ll'ſ \n é'T\n \n'🙂ßé字'T字HTTPServerᵃ", "tokens": 58, "pieces": ["!!'", "Tع", "\n", "/'", "s", "👍🏽\r\n", "<|", "endoftext", "|>", "fiå", "'", " \n", " fi", "/\r\n", "Dž", " si", "OSḍ̇'VE", "é", "½", "'ll'ſ", " \n", " é'T", "\n \n", "'🙂", "ßé字'T", "字HTTPServerᵃ"]} +{"text": "'VEaB\n<​a/bHTTPServer!e\n/fiſ́🙂d", "tokens": 23, "pieces": ["'VEa", "B", "\n", "<​", "a", "/b", "HTTPServer", "!e", "\n", "/fiſ", "́", "🙂d"]} +{"text": "
AbHTTPServerAbA", "tokens": 6, "pieces": ["
Ab", "HTTPServer", "Ab", "A"]} +{"text": "‍٣٤٥٦‍<|fim_prefix|>漢'S.ſᵃ!Bt,B½
''VEB   'T0å\nAbcamelCaseꟲſ \n", "tokens": 42, "pieces": ["‍", "٣٤٥", "٦", "‍<|", "fim", "_prefix", "|>", "漢'S", ".ſᵃ", "!Bt", ",B", "½", "
", "''", "VEB", "  ", " ", "'T", "0", "å", "\n", "Abcamel", "Caseꟲſ", " \n"]} +{"text": "½fiB\n!!#$%㋿ssEOT/‍Dž", "tokens": 17, "pieces": ["½", "fi", "B", "\n", "!!#$%㋿", "ss", "EOT", "/‍", "Dž"]} +{"text": "aB (iOS\r\n㋿,\n<İ 👍🏽㋿Džع\r\ns㍿ع'ſ'reİßⅣ­", "tokens": 39, "pieces": ["a", "B", " (", "i", "OS", "\r\n", "㋿,\n", "<İ", " 👍🏽㋿", "Dž", "ع", "\r\n", "s", "㍿ع'ſ", "'re", "İß", "Ⅳ", "­"]} +{"text": "'\"éaaB٣٤٥٦fi-…>\u000b½/\r\n'S\r\n\r\n/\r\n!!Ⅳ'TmiOSA 'VE ꟲ\"ع", "tokens": 39, "pieces": ["'\"", "éaa", "B", "٣٤٥", "٦", "fi", "-", "…", ">", "\u000b", "½", "/\r\n", "'S", "\r\n\r\n", "/\r\n", "!!", "Ⅳ", "'Tmi", "OSA", " ", "'VE", " ", " ꟲ", "\"ع"]} +{"text": "​.ⅣHTTPServer fi…­\nḍ̇'M<|endoftext|>٣٤٥٦'ll'M//㍿㋿ḍ̇Ⅳa\n/  \rß٣٤٥٦B!!𐞁0fi​😀🏽३", "tokens": 62, "pieces": ["​.", "Ⅳ", "HTTPServer", " fi", "…", "­\n", "ḍ̇'M", "<|", "endoftext", "|>", "٣٤٥", "٦", "'ll'M", "//㍿㋿", "ḍ̇", "Ⅳ", "a", "\n", "/", "  \r", "ß", "٣٤٥", "٦", "B", "!!", "𐞁", "0", "fi", "​😀🏽", "३"]} +{"text": " e<|endoftext|>>🙂\r's'llAb åDžungla0", "tokens": 22, "pieces": [" ", " e", "<|", "endoftext", "|>>🙂\r", "'s'll", "Ab", " ", " å", "Džungla", "0"]} +{"text": "'ſ‍👍🏽\r\n\r\nmHTTPServer\r\nAb", "tokens": 12, "pieces": ["'ſ", "‍👍🏽\r\n\r\n", "m", "HTTPServer", "\r\n", "Ab"]} +{"text": " ‍\"d'D
'M å𐞁's.ḍ̇/\r\n३'½/ꟲB­fi/'Re'Re-\u000b/\r\n", "tokens": 36, "pieces": [" ", "‍\"", "d'D", "
", "'M", " å𐞁's", ".ḍ̇", "/\r\n", "३", "'", "½", "/ꟲ", "B", "­fi", "/'", "Re'Re", "-", "\u000b", "/\r\n"]} +{"text": "EOT​é ́\r \n Dž,🙂\néꟲ iOS'Re", "tokens": 24, "pieces": ["EOT", "​é", " ", " ́", "\r \n", " Dž", ",🙂\n", "éꟲ", " i", "OS'Re"]} +{"text": "İA'٣٤٥٦字\taBḍ̇𐞁‍\u000b'M\n/\r\n'½(#$%ß", "tokens": 28, "pieces": ["İA", "'", "٣٤٥", "٦", "字", "\ta", "Bḍ̇𐞁", "‍", "\u000b", "'M", "\n", "/\r\n", "'", "½", "(#$%", "ß"]} +{"text": " \n B#$%́\r\n a/b­ \nm\u000b fi", "tokens": 15, "pieces": [" \n", " B", "#$%́\r\n", " a", "/b", "­", " \n", "m", "\u000b", " fi"]} +{"text": "Dž👍🏽
're'Rea/bBᵃ­Dž​m𐞁-0--aABC😀🏽mta/bfi \n ḍ̇ ½३éd ", "tokens": 50, "pieces": ["Dž", "👍🏽", "
", "'re'Re", "a", "/b", "Bᵃ", "­Dž", "​m𐞁", "-", "0", "--", "a", "ABC", "😀🏽", "mta", "/bfi", " \n", " ḍ̇", " ", "½", "", "३", "éd", " "]} +{"text": "㍿s9iOSſ- B!s🙂e/0'\rع́éa/bEOT 字Džungla😀🏽'S", "tokens": 38, "pieces": ["㍿s", "9", "i", "OSſ", "-", " ", " B", "!s", "🙂e", "/", "0", "'\r", "ع́éa", "/b", "EOT", " 字Džungla", "😀🏽'", "S"]} +{"text": "'iOS0 \r!!<|fim_prefix|> 12345678\r<|fim_prefix|>éEOT!!'M/\r\n'Dfi\r\n\r\nHTTPServer\n're", "tokens": 38, "pieces": ["'i", "OS", "0", " \r", "!!<|", "fim", "_prefix", "|>", " ", "123", "456", "78", "\r", "<|", "fim", "_prefix", "|>", "é", "EOT", "!!'", "M", "/\r\n", "'Dfi", "\r\n\r\n", "HTTPServer", "\n", "'re"]} +{"text": "😀🏽>aB>İ/'S12345678 <|endoftext|>
a/bAa\t'VE<|fim_prefix|>'VE字fi३́'llé9ß<|endoftext|>Ⅳ<|endoftext|>\r\n\r\nß🙂'MEOT'M", "tokens": 65, "pieces": ["😀🏽>", "a", "B", ">İ", "/'", "S", "123", "456", "78", " <|", "endoftext", "|>", "
a", "/b", "Aa", "\t", "'VE", "<|", "fim", "_prefix", "|>'", "VE字fi", "३", "́'ll", "é", "9", "ß", "<|", "endoftext", "|>", "Ⅳ", "<|", "endoftext", "|>\r\n\r\n", "ß", "🙂'", "MEOT'M"]} +{"text": "'smDž'VEİé", "tokens": 9, "pieces": ["'sm", "Dž'VE", "İé"]} +{"text": ">-aDžAZ>漢\u000bع👍🏽'Mꟲ'Mté", "tokens": 21, "pieces": [">-", "a", "DžAZ", ">漢", "\u000bع", "👍🏽'", "Mꟲ'M", "té"]} +{"text": "㋿३éABC(!'T½\"👍🏽9/\r\n'VE's #$%\ra/b
camelCase's​​ iOS🙂‍", "tokens": 40, "pieces": ["㋿", "३", "é", "ABC", "(!'", "T", "½", "\"👍🏽", "9", "/\r\n", "'VE's", " ", "#$%\r", "a", "/b", "
camel", "Case's", "​​", " i", "OS", "🙂‍"]} +{"text": "ⅣAb ​'s٣٤٥٦ \u000b½½ꟲ/\r\n>", "tokens": 20, "pieces": ["Ⅳ", "Ab", " ", "​'", "s", "٣٤٥", "٦", " ", "\u000b", "½½", "ꟲ", "/\r\n", ">"]} +{"text": "🙂BABC㍿Ab'llZAZ\t…>9字​", "tokens": 18, "pieces": ["🙂BABC", "㍿Ab'll", "ZAZ", "\t", "…", ">", "9", "字", "​"]} +{"text": " \n\r'Dſ'S", "tokens": 5, "pieces": [" \n\r", "'Dſ'S"]} +{"text": "a/b/\r\n🙂٣٤٥٦!.‍<|fim_prefix|>>  \nſDž\n/camelCaseå𐞁 ", "tokens": 36, "pieces": ["a", "/b", "/\r\n", "🙂<", "EOT", ">", "٣٤٥", "٦", "!.‍<|", "fim", "_prefix", "|>>", "  \n", "ſ", "Dž", "\n", "/camel", "Caseå𐞁", " "]} +{"text": "9Džungla\u000b\u000b12345678İB'VE-'M漢'VE\r​😀🏽é'M٣٤٥٦d<|endoftext|>'S><|fim_prefix|>9 \n AbfiEOTDž
! \n 'M'T", "tokens": 63, "pieces": ["9", "Džungla", "\u000b", "\u000b", "123", "456", "78", "İB'VE", "-'", "M漢'VE", "\r", "​😀🏽", "é'M", "٣٤٥", "٦", "d", "<|", "endoftext", "|>'", "S", "><|", "fim", "_prefix", "|>", "9", " \n", " Abfi", "EOTDž", "
", "!", " \n", " '", "M'T", ""]} +{"text": "å­٣٤٥٦å.\" ᵃa/b\"\n/½a/b😀🏽 \u000b \n's<|endoftext|>'s \n\u000b 9½‍­İ'㋿m
字🙂ts", "tokens": 55, "pieces": ["å", "­", "٣٤٥", "٦", "å", ".\"", " ᵃa", "/b", "\"\n/", "½", "a", "/b", "😀🏽", " \u000b \n", "'s", "<|", "endoftext", "|>'", "s", " \n", "\u000b", " ", "9½", "‍­", "İ", "'㋿", "m", "", "
字", "🙂ts"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": ", ('M(\n/'re \n ٣٤٥٦\r\n<​字👍🏽\n \n 😀🏽a/bᵃ>'ſ<|fim_prefix|>漢́", "tokens": 44, "pieces": [",", " ", " ('", "M", "(\n/", "'re", " \n", " ", "٣٤٥", "٦", "\r\n", "<​", "字", "👍🏽\n", "", " \n", " 😀🏽", "a", "/bᵃ", ">'", "ſ", "<|", "fim", "_prefix", "|>", "漢́"]} +{"text": "<|endoftext|>'Sع½<३-Džungla'ScamelCase<|endoftext|>­\r<|fim_prefix|><|fim_prefix|>Ⅳ0-ABCHTTPServer('ll‍½-\n…\r\n\r\n字,\r\n\r\n<<|fim_prefix|>́0", "tokens": 71, "pieces": ["<|", "endoftext", "|>'", "S", "ع", "½", "<", "३", "-Džungla'S", "camel", "Case", "<|", "endoftext", "|>­\r", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>", "Ⅳ0", "-ABCHTTPServer", "('", "ll", "‍", "½", "-\n", "…\r\n\r\n", "字", ",\r\n\r\n", "<<|", "fim", "_prefix", "|>́", "0"]} +{"text": "<|endoftext|>…👍🏽\"'Re字Za/b ㍿ #$%<|fim_prefix|>ß👍🏽fi're.ſⅣ\n/३EOT,漢aع\r åſ‍👍🏽camelCase漢d\n!", "tokens": 63, "pieces": ["<|", "endoftext", "|>", "…", "👍🏽\"'", "Re字", "Za", "/b", " ", "㍿", " ", "#$%<|", "fim", "_prefix", "|>", "ß", "👍🏽", "fi're", ".ſ", "Ⅳ", "\n", "/", "३", "EOT", ",漢aع", "\r", " åſ", "‍👍🏽", "camel", "Case漢d", "\n", "!"]} +{"text": " <|endoftext|>\r\n'T12345678a/bß'Re\"½\u000b're/́dᵃ'S­😀🏽 漢'M\ncamelCase
ᵃ­ \n ٣٤٥٦ \n", "tokens": 49, "pieces": [" ", " <|", "endoftext", "|>\r\n", "'T", "123", "456", "78", "a", "/bß'Re", "\"", "½", "\u000b", "'re", "/́dᵃ'S", "­😀🏽", " 漢'M", "\n", "camel", "Case", "
ᵃ", "­", " \n", " ", "٣٤٥", "٦", " \n"]} +{"text": "Abå.!\n/Ab/\">\n/\r\nDžungla !HTTPServerABC­!!<|fim_prefix|>é", "tokens": 32, "pieces": ["Abå", ".!\n/", "Ab", "/\">\n/\r\n", "Džungla", " ", "!HTTPServer", "ABC", "­!!<|", "fim", "_prefix", "|>", "é"]} +{"text": "/\r\n'T' 12345678İDžunglaDž'ſBᵃ😀🏽!'TBBß\r\n\r\n#$%EOT<́漢字 ", "tokens": 37, "pieces": ["/\r\n", "'T", "'", " ", "123", "456", "78", "İDžungla", "Dž'ſ", "Bᵃ", "😀🏽!'", "TBBß", "\r\n\r\n", "#$%", "EOT", "<́漢字", " "]} +{"text": "-…12345678㋿'", "tokens": 10, "pieces": ["-", "…", "123", "456", "78", "㋿'"]} +{"text": "Džᵃ\".EOTABC'll'rea/b12345678\n<|fim_prefix|>'ll㍿Džungla字​-e…ᵃعiOSعḍ̇­'ll٣٤٥٦'s😀🏽Ab…", "tokens": 61, "pieces": ["Džᵃ", "\".", "EOTABC'll", "'rea", "/b", "123", "456", "78", "\n", "<|", "fim", "_prefix", "|>'", "ll", "㍿Džungla字", "​-", "e", "…ᵃعi", "OSعḍ̇", "­'", "ll", "٣٤٥", "٦", "'s", "😀🏽", "Ab", "…"]} +{"text": " 9a‍ İ 👍🏽camelCaseİ'!ꟲᵃ-,tt'VEås\"'Re 'reᵃ…'s'S>", "tokens": 45, "pieces": [" ", " ", "9", "a", "‍", " İ", " ", "👍🏽", "camel", "Case", "İ", "'!", "ꟲᵃ", "-,", "tt'VE", "ås", "\"<", "META", "_START", ">'", "Re", " ", " '", "reᵃ", "…", "'s'S", ">"]} +{"text": "Džunglaſ😀🏽>ᵃ'TcamelCaseⅣ \nⅣ", "tokens": 20, "pieces": ["Džunglaſ", "😀🏽>", "ᵃ'T", "camel", "Case", "Ⅳ", " \n", "Ⅳ"]} +{"text": "åt​ſⅣiOS  \r\niOSEOTiOS'Re-\t'ſ<|endoftext|>\r(㍿ \n iOS'Re\n'ſ漢'T ᵃ \n ", "tokens": 49, "pieces": ["åt", "​ſ", "Ⅳ", "i", "OS", "  \r\n", "i", "OSEOTi", "OS'Re", "-", "\t", "'ſ", "<|", "endoftext", "|>\r", "(㍿", " \n", " i", "OS'Re", "\n", "'ſ漢'T", " ᵃ", " \n", " "]} +{"text": "́३३Ab<🙂", "tokens": 9, "pieces": ["́", "३३", "Ab", "<🙂"]} +{"text": "'s'lls😀🏽/ \n 0å\u000b", "tokens": 13, "pieces": ["'s'll", "s", "😀🏽/", " \n", " ", "0", "å", "\u000b"]} +{"text": " \u000bs<|endoftext|>'ll>́\nt,'M,\u000b​\r", "tokens": 21, "pieces": [" ", "\u000bs", "<|", "endoftext", "|>'", "ll", ">́", "\n", "t", ",'", "M", ",", "\u000b", "​\r"]} +{"text": " ́!!Z 'DHTTPServerm㍿9'D½/", "tokens": 17, "pieces": [" ", " ́", "!!", "Z", " ", "'DHTTPServerm", "㍿", "9", "'D", "½", "/"]} +{"text": "a/bᵃ½.'D/\r\n\n", "tokens": 10, "pieces": ["a", "/bᵃ", "½", ".'", "D", "/\r\n\n"]} +{"text": "'ſ½\"ḍ̇漢/.-t\t'DBe\"\u000bß \n ३ßᵃ\tſ9ꟲ́Ⅳ­(ß's\u000b𐞁́
#$%ᵃ字B", "tokens": 51, "pieces": ["'ſ", "½", "\"ḍ̇漢", "/.-", "t", "\t", "'DBe", "\"", "\u000bß", " \n", " ", "३", "ßᵃ", "\tſ", "9", "ꟲ́", "Ⅳ", "­(", "ß's", "\u000b𐞁́", "
", "#$%", "ᵃ字", "B"]} +{"text": "'Aß 'T…'Refit('llaB \naB́ \n /३\r
,\r\n\r\n're​camelCase", "tokens": 27, "pieces": ["'Aß", " ", "'T", "…", "'Refit", "('", "lla", "B", " \n", "a", "B́", " \n", " /", "३", "\r", "
", ",\r\n\r\n", "'re", "​camel", "Case"]} +{"text": "!!\n/漢İ३\"('VEfiꟲd㋿9iOS,'Ree>", "tokens": 23, "pieces": ["!!\n/", "漢", "İ", "३", "\"('", "VEfiꟲd", "㋿", "9", "i", "OS", ",'", "Ree", ">"]} +{"text": "\r\nZ's'D\rſ'VE<|fim_prefix|>'D \n >", "tokens": 17, "pieces": ["\r\n", "Z's", "'D", "\r", "ſ'VE", "<|", "fim", "_prefix", "|>'", "D", " \n", " >"]} +{"text": "'re \n a'reİe's𐞁𐞁", "tokens": 22, "pieces": ["'re", " \n", " a're", "İ", "e's", "𐞁𐞁", ""]} +{"text": "'ſ mfié0
éᵃé\n/­12345678t‍mEOT#$%ꟲ\r३३३aaBZ-‍👍🏽ßB", "tokens": 46, "pieces": ["'ſ", " mfié", "0", "
éᵃé", "\n", "/­", "123", "456", "78", "t", "‍m", "EOT", "#$%", "ꟲ", "\r", "३३३", "aa", "BZ", "-<", "META", "_START", ">‍👍🏽", "ß", "B"]} +{"text": "'ſ'Msß's🙂!/㍿'sEOTåfi'saBsaḍ̇AİꟲåDžungla0(‍HTTPServer", "tokens": 43, "pieces": ["'ſ", "'", "Msß's", "🙂!/㍿'", "s", "EOTåfi's", "a", "Bsaḍ̇", "Aİꟲå", "Džungla", "0", "(‍", "HTTPServer"]} +{"text": " \n !́ſAbå>-\rḍ̇👍🏽BHTTPServercamelCaseḍ̇'T,\tt-字", "tokens": 30, "pieces": [" \n", " !́", "ſ", "Ab", "å", ">-\r", "ḍ̇", "👍🏽", "BHTTPServercamel", "Caseḍ̇'T", ",", "\tt", "-字"]} +{"text": "EOTعiOŚDžungla㋿👍🏽字🙂½t0'ſßⅣB!!
<|endoftext|>Dž👍🏽🙂Džḍ̇ .", "tokens": 51, "pieces": ["EOTعi", "OŚDžungla", "㋿👍🏽", "字", "🙂", "½", "t", "0", "'", "ſß", "Ⅳ", "B", "!!", "
", "<|", "endoftext", "|>", "Dž", "👍🏽🙂", "Džḍ̇", " ", " ."]} +{"text": "A<|fim_prefix|>B'ree㍿字ABC/\r\ncamelCase0a/\r\n漢\ta٣٤٥٦३", "tokens": 28, "pieces": ["A", "<|", "fim", "_prefix", "|>", "B're", "e", "㍿字", "ABC", "/\r\n", "camel", "Case", "0", "a", "/\r\n", "漢", "\ta", "٣٤٥", "٦३"]} +{"text": " aHTTPServer\r\n\n,ZaB'll ㋿\r㍿'VE A's'VE12345678're😀🏽'S( \n ABC‍ſ\t", "tokens": 47, "pieces": [" a", "HTTPServer", "\r\n\n", ",Za", "B'll", "", " ", " ㋿\r", "㍿'", "VE", " <", "META", "_START", ">A's", "'VE", "123", "456", "78", "'re", "😀🏽'", "S", "(", " \n", " ABC", "‍ſ", "\t"]} +{"text": "-a/b/\r\n- \n !é!\n/­>.é​\" \n \n -漢'S\r'Re'D!Džungla.'M३", "tokens": 31, "pieces": ["-a", "/b", "/\r\n", "-", " \n", " !", "é", "!\n/", "­>.", "é", "​\"", " \n \n", " -", "漢'S", "\r", "'Re'D", "!Džungla", ".'", "M", "३"]} +{"text": "\r\n🙂-/İDžunglaꟲ​ \n're🙂\u000bA'M'éé३\naé0𐞁fi12345678#$%HTTPServer\n/'re½\"ß", "tokens": 49, "pieces": ["\r\n", "🙂<", "EOT", ">-/", "İDžunglaꟲ", "​", " \n", "'re", "🙂", "\u000bA'M", "'éé", "३", "\n", "aé", "0", "𐞁fi", "123", "456", "78", "#$%", "HTTPServer", "\n", "/'", "re", "½", "\"ß"]} +{"text": "e-,𐞁t'ſⅣ​a/b>'ll'ſ<|fim_prefix|>!e12345678\n٣٤٥٦\tⅣt字ABC'reⅣ'reée'\r\n'
EOTHTTPServer'iOS<|fim_prefix|><|endoftext|>B", "tokens": 67, "pieces": ["e", "-,", "𐞁t'ſ", "Ⅳ", "​a", "/b", ">'", "ll'ſ", "<|", "fim", "_prefix", "|>!", "e", "123", "456", "78", "\n", "٣٤٥", "٦", "\t", "Ⅳ", "t字", "ABC're", "Ⅳ", "'reée", "'\r\n", "'", "
EOTHTTPServer", "'i", "OS", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "B"]} +{"text": "-\n/ABC \n 👍🏽́́ \n \naB/\r\n👍🏽'Dſ'ſ३camelCase0٣٤٥٦‍siOSDž#$%'😀🏽<\rꟲ<|endoftext|>'VE \n'ſa'T0'Re", "tokens": 64, "pieces": ["-\n/", "ABC", " \n", " ", " 👍🏽́́", " \n \n", "a", "B", "/\r\n", "👍🏽'", "Dſ'ſ", "", "३", "camel", "Case", "0٣٤", "٥٦", "‍si", "OSDž", "#$%'😀🏽<\r", "ꟲ", "<|", "endoftext", "|>'", "VE", " \n", "'ſa'T", "0", "'Re"]} +{"text": "'reeé٣٤٥٦>٣٤٥٦ABĆ'ſ--'VEt0<|fim_prefix|>\r\n\r\n​
\tDž<­fi字''s", "tokens": 38, "pieces": ["'reeé", "٣٤٥", "٦", ">", "٣٤٥", "٦", "ABĆ'ſ", "--'", "VEt", "0", "<|", "fim", "_prefix", "|>\r\n\r\n", "​", "
", "\tDž", "<­", "fi字", "''", "s"]} +{"text": "Džungla'llABC㍿ 9. 𐞁ᵃmſ𐞁Z", "tokens": 31, "pieces": ["Džungla", "'", "ll", "ABC", "㍿", " ", "9", ".", " 𐞁ᵃmſ𐞁", "Z"]} +{"text": "Dž12345678\u000b\"/ß\r\n𐞁'reé½Ab\n/0're'ſ>tAbß­a/a/b/AbHTTPServer\u000b'S", "tokens": 40, "pieces": ["Dž", "", "123", "456", "78", "\u000b", "\"/", "ß", "\r\n", "𐞁're", "é", "½", "Ab", "\n", "/", "0", "'re'ſ", ">t", "Abß", "­a", "/a", "/b", "/Ab", "HTTPServer", "\u000b", "'S"]} +{"text": "Ⅳ \n#$%İA<|fim_prefix|>sDžt!​'re\nfi", "tokens": 23, "pieces": ["Ⅳ", " \n", "#$%", "İA", "<|", "fim", "_prefix", "|>", "s", "Džt", "!​'", "re", "\n", "fi"]} +{"text": "''Tfi㋿!!", "tokens": 25, "pieces": ["''", "Tfi", "㋿!!"]} +{"text": "9 \n𐞁 \u000b  é-ḍ̇ \n aB'S'lla'T\u000b👍🏽🙂ABC漢åİ<|fim_prefix|>.漢'Ms", "tokens": 41, "pieces": ["9", " \n", "𐞁", " \u000b  ", " é", "-ḍ̇", " \n", " a", "B'S", "'lla'T", "\u000b", "👍🏽🙂", "ABC漢å", "İ", "<|", "fim", "_prefix", "|>.", "漢'M", "s"]} +{"text": "camelCase'M‍ݽ‍fi<|endoftext|> ㋿>aBs㍿\r\r\nع", "tokens": 27, "pieces": ["camel", "Case'M", "‍İ", "½", "‍fi", "<|", "endoftext", "|>", " ", "㋿>", "a", "Bs", "㍿\r\r\n", "ع"]} +{"text": "ḍ̇a/b#$%👍🏽
ᵃteåA½émⅣ", "tokens": 44, "pieces": ["", "ḍ̇a", "/b", "#$%👍🏽", "
ᵃteå", "A", "½", "ém", "Ⅳ"]} +{"text": "'ſ漢​\r\n!d", "tokens": 7, "pieces": ["'ſ漢", "​\r\n", "!d"]} +{"text": "\t…👍🏽ḍ̇!!", "tokens": 17, "pieces": ["\t", "", "…", "👍🏽", "ḍ̇", "!!"]} +{"text": "\rDž \n s👍🏽's字-mع🙂👍🏽/‍!å12345678🙂\r\n\r\n…/'Reḍ̇ <|endoftext|>👍🏽<|fim_prefix|><|fim_prefix|>​,Bå", "tokens": 67, "pieces": ["\r", "Dž", " \n", " s", "👍🏽'", "s字", "-mع", "🙂👍🏽/‍!", "å", "123", "456", "78", "🙂\r\n\r\n", "…", "/'", "Reḍ̇", " ", " <|", "endoftext", "|>👍🏽<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>​,", "Bå"]} +{"text": "ſ<|endoftext|>a'S😀🏽'EOTe(😀🏽İEOT9éDž", "tokens": 28, "pieces": ["ſ", "<|", "endoftext", "|>", "a'S", "😀🏽'", "EOTe", "(😀🏽", "İEOT", "9", "é", "Dž"]} +{"text": "\r\n\r\n\tḍ̇aB\r\u000b‍\u000b  ­a/b''S'Me12345678३ABC​…< te're\n/ ", "tokens": 34, "pieces": ["\r\n\r\n", "\tḍ̇a", "B", "\r", "\u000b", "‍", "\u000b ", " ", "­a", "/b", "''", "S'M", "e", "123", "456", "78३", "ABC", "​", "…", "<", " ", " te're", "\n", "/", " "]} +{"text": "'re!!/ \n", "tokens": 4, "pieces": ["'re", "!!/", " \n"]} +{"text": "‍𐞁漢'ſ😀🏽\r\n\r\nḍ̇sß Džungla'M\n!/\r\n", "tokens": 26, "pieces": ["‍𐞁漢'ſ", "😀🏽\r\n\r\n", "ḍ̇sß", " Džungla'M", "\n", "!/\r\n"]} +{"text": "٣٤٥٦'D're'\t \n 'fi'TⅣ३EOT!#$%", "tokens": 19, "pieces": ["٣٤٥", "٦", "'D're", "'", "\t \n", " '", "fi'T", "Ⅳ३", "EOT", "!#$%"]} +{"text": "‍e'T ३‍/\r\n", "tokens": 8, "pieces": ["‍e'T", " ", " ", "३", "‍/\r\n"]} +{"text": "🙂'㋿.Z'­>é𐞁​", "tokens": 17, "pieces": ["🙂'㋿.", "Z", "'­>", "é𐞁", "​"]} +{"text": "iOS㍿", "tokens": 5, "pieces": ["i", "OS", "㍿"]} +{"text": "EOT‍Ⅳé'M0
عiOSé漢dcamelCase0", "tokens": 18, "pieces": ["EOT", "‍", "Ⅳ", "é'M", "0", "
عi", "OSé漢dcamel", "Case", "0"]} +{"text": "/🙂\t<|endoftext|><|endoftext|>‍Ab#$%​
ABC-ḍ̇𐞁HTTPServer
́\ré'Re\u000b½'ll#$%‍'VE ½!!<|fim_prefix|>!!é're<|endoftext|>漢'll", "tokens": 68, "pieces": ["/🙂", "\t", "<|", "endoftext", "|><|", "endoftext", "|>‍", "Ab", "#$%​", "
ABC", "-ḍ̇𐞁", "HTTPServer", "
́", "\r", "é'Re", "\u000b", "½", "'ll", "#$%‍'", "VE", " ", "½", "!!<|", "fim", "_prefix", "|>!!", "é're", "<|", "endoftext", "|>", "漢'll"]} +{"text": "'VE ḍ̇é‍ḍ̇ /\r\n-\rDž字,ḍ̇'S٣٤٥٦\t<|endoftext|>­.d'RemHTTPServer \n ½ßfiåع 'Re<|fim_prefix|>ss'Sd𐞁ABC", "tokens": 68, "pieces": ["'VE", " ", " ḍ̇é", "‍ḍ̇", " /\r\n", "-\r", "Dž字", ",ḍ̇'S", "٣٤٥", "٦", "\t", "<|", "endoftext", "|>­.", "d'Re", "m", "HTTPServer", " \n", " ", "½", "ßfiåع", " ", "'Re", "<|", "fim", "_prefix", "|>", "ss'S", "d𐞁", "ABC"]} +{"text": "😀🏽s'Re's'D>'reta/bꟲHTTPServerm,!​ \n", "tokens": 19, "pieces": ["😀🏽", "s'Re", "'s'D", ">'", "reta", "/bꟲ", "HTTPServerm", ",!​", " \n"]} +{"text": "camelCase<|fim_prefix|><|endoftext|>'s٣٤٥٦EOT//HTTPServer'ReBDž ABCßİⅣDž", "tokens": 35, "pieces": ["camel", "Case", "<|", "fim", "_prefix", "|><|", "endoftext", "|>'", "s", "٣٤٥", "٦", "EOT", "//", "HTTPServer'Re", "BDž", " ABCß", "İ", "Ⅳ", "Dž"]} +{"text": "éع‍e\r\ne٣٤٥٦'re३ßaéꟲ漢'D.‍Džſ𐞁­-12345678👍🏽ꟲḍ̇(Ⅳ/\r\n,/fi ß½
", "tokens": 53, "pieces": ["éع", "‍e", "\r\n", "e", "٣٤٥", "٦", "'re", "३", "ßaéꟲ漢'D", ".‍", "Džſ𐞁", "­-", "123", "456", "78", "👍🏽", "ꟲḍ̇", "(", "Ⅳ", "/\r\n", ",/", "fi", " ß", "½", "
"]} +{"text": ",\r\n\r\nDžAb­\r\n\u000b12345678漢 \n .㍿́Džéᵃ's👍🏽,dé<|endoftext|>/
ßḍ̇<|fim_prefix|><|endoftext|>\nm'ſ \n ", "tokens": 64, "pieces": [",\r\n\r\n", "DžAb", "­\r\n", "\u000b", "123", "456", "78", "漢", " \n", " .㍿́", "Džé", "ᵃ's", "👍🏽,", "dé", "<|", "endoftext", "|>/", "
ßḍ̇", "<|", "fim", "_prefix", "|><|", "endoftext", "|>\n", "m'ſ", " \n", " "]} +{"text": "\n/漢-İ9ꟲ're'Re<|endoftext|>#$%-㍿<|endoftext|>ABCfi.'llABCDžungla/\r\n'ſ'VEꟲ,漢,漢½Dž\n𐞁m… <|endoftext|>ꟲ", "tokens": 76, "pieces": ["\n", "/漢", "-İ", "9", "ꟲ're", "'Re", "<|", "endoftext", "|>#$%-㍿<|", "endoftext", "|>", "ABCfi", ".'", "ll", "ABCDžungla", "/\r\n", "'ſ'VE", "ꟲ", ",", "漢", ",漢", "½", "Dž", "\n", "𐞁m", "…", " ", "<|", "endoftext", "|>", "ꟲ"]} +{"text": "🙂\"'ſßcamelCase
-'s!\"", "tokens": 13, "pieces": ["🙂\"'", "ſßcamel", "Case", "", "
", "-'", "s", "!\""]} +{"text": "İ́Ⅳ​iOS're'٣٤٥٦㍿ \u000b\r\n\r\n‍'VE ", "tokens": 33, "pieces": ["Ⅳ", "'Da字", "<|", "endoftext", "|>", "Ⅳ", "​i", "OS're", "'", "٣٤٥", "٦", "㍿", " \u000b\r\n\r\n", "‍'", "VE", " "]} +{"text": "ꟲa/b<|fim_prefix|>B\r字0<'s/\r\n!!𐞁\u000b\n/", "tokens": 26, "pieces": ["ꟲa", "/b", "<|", "fim", "_prefix", "|>", "B", "\r", "字", "0", "<'", "s", "/\r\n", "!!", "𐞁", "\u000b\n", "/"]} +{"text": "Ⅳ́\u000bEOTe'T s‍ \n'VE𐞁<|fim_prefix|>٣٤٥٦ABC0 \t/\r\n!/\r\n‍/ ㍿'s'sDž !!😀🏽", "tokens": 57, "pieces": ["å'ſ", "!", "Ⅳ३", ">", "\u000bEOTe'T", " s", "‍", " \n", "'VE𐞁", "<|", "fim", "_prefix", "|>", "٣٤٥", "٦", "ABC", "0", " ", "\t", "/\r\n", "!/\r\n", "‍/", " ", "㍿'", "s's", "Dž", " ", "!!😀🏽"]} +{"text": "B'T0fié \né<|fim_prefix|>/‍iOSDž
", "tokens": 25, "pieces": ["B'T", "0", "fié", " \n", "é", "<|", "fim", "_prefix", "|>/‍", "i", "OSDž", "
"]} +{"text": "9字'VEe", "tokens": 9, "pieces": ["9", "字'VE", "e", ""]} +{"text": "\ra/b/'S'sfiİ'D's<|endoftext|>0​Ź३B ('M漢㍿ABC𐞁'Mع
'DAſ'>/å㋿ᵃ", "tokens": 50, "pieces": ["\r", "a", "/b", "/'", "S's", "fi", "İ'D", "'s", "<|", "endoftext", "|>", "0", "​Ź", "३", "B", " ('", "M漢", "㍿ABC𐞁'M", "ع", "
", "'DAſ", "'>/", "å", "㋿ᵃ"]} +{"text": "'s'Td9 🙂åꟲéé३Ⅳ३12345678'Re\u000b\r\n \nZ\n/!!'llḍ̇camelCase", "tokens": 37, "pieces": ["'s'T", "d", "9", " ", " 🙂", "åꟲéé", "३Ⅳ३", "123", "456", "78", "'Re", "\u000b\r\n \n", "Z", "\n", "/!!'", "llḍ̇camel", "Case"]} +{"text": "fiZDžungla­<'re, !aB(A\r\n\t ٣٤٥٦😀🏽‍", "tokens": 29, "pieces": ["fi", "ZDžungla", "­<'", "re", ",", " !", "a", "B", "(A", "\r\n", "\t", "", " ", "٣٤٥", "٦", "😀🏽‍"]} +{"text": "(a/bmßa/baB'aBé<|fim_prefix|>Džungla / (.'sDž(>iOS…𐞁½ABC.\r\nfia/b!!iOS‍9 \r\n\r\nḍ̇ ", "tokens": 58, "pieces": ["(a", "/bmßa", "/b", "a", "B", "'a", "Bé", "<|", "fim", "_prefix", "|>", "Džungla", " ", " /", " ", " (.'", "s", "Dž", "(>", "i", "OS", "…𐞁", "½", "ABC", ".\r\n", "fia", "/b", "!!", "i", "OS", "‍", "9", " \r\n\r\n", "ḍ̇", " "]} +{"text": "​𐞁\n/ \n's\r\n\r\nmḍ̇'s\tAİß\"Ab#$%'ll👍🏽's㋿㋿\n!! d/عfi\n\"'re\rḍ̇
0", "tokens": 61, "pieces": ["​𐞁", "\n", "/<", "EOT", ">", " \n", "'s", "\r\n\r\n", "mḍ̇'s", "\t", "Aİß", "\"Ab", "#$%'", "ll", "👍🏽'", "s", "㋿㋿\n", "!!", " d", "/عfi", "\n", "\"'", "re", "\r", "ḍ̇", "
", "0"]} +{"text": "ſſEOT漢३t٣٤٥٦Ab-ſ́<|fim_prefix|>'ſiOS\"'ll\n\r\n\r\nſعaB'DiOS㍿🙂", "tokens": 42, "pieces": ["ſſ", "EOT漢", "३", "t", "٣٤٥", "٦", "Ab", "-ſ́", "<|", "fim", "_prefix", "|>'", "ſi", "OS", "\"'", "ll", "\n\r\n\r\n", "ſعa", "B'D", "i", "OS", "㍿🙂"]} +{"text": "३!!a!Džunglaꟲ\r\nDžungla'…e½é漢ſé'ſEOTéİ'VE>Abꟲ \n\r\n\r\ńAiOS㋿👍🏽t'ſ", "tokens": 56, "pieces": ["३", "!!", "a", "!Džunglaꟲ", "\r\n", "Džungla", "'", "…e", "½", "é", "漢ſé'ſ", "EOTé", "İ'VE", ">Abꟲ", " \n\r\n\r\n", "́Ai", "OS", "㋿👍🏽", "t'ſ"]} +{"text": "'re🙂\"\naB!!EOT\"åaBiOSḍ̇<-字ع/\r\nDž​é'ſꟲع\nꟲEOTBt ", "tokens": 38, "pieces": ["'re", "🙂\"\n", "a", "B", "!!", "EOT", "\"åa", "Bi", "OSḍ̇", "<-", "字ع", "/\r\n", "Dž", "​é'ſ", "ꟲع", "\n", "ꟲEOTBt", " "]} +{"text": "​é 字३é9é \n é\n/-'D", "tokens": 21, "pieces": ["​é", " <", "META", "_START", ">", " ", " 字", "३", "é", "9", "é", " \n", " é", "\n", "/-'", "D"]} +{"text": "\r\n٣٤٥٦>

\rEOTme Z<|endoftext|>/ḍ̇㍿字 \n'T", "tokens": 29, "pieces": ["\r\n", "٣٤٥", "٦", ">", "

\r", "EOTme", " Z", "<|", "endoftext", "|>/", "ḍ̇", "㍿字", " \n", "'T"]} +{"text": " İ\r\n\r\nHTTPServerAb\n/-/'ll㍿ß😀🏽EOT ́㋿'M㋿㋿m<|endoftext|>sé½\r\n…", "tokens": 46, "pieces": [" ", " İ", "\r\n\r\n", "HTTPServer", "Ab", "\n", "/-/'", "ll", "㍿ß", "😀🏽", "EOT", " ", " ́", "㋿'", "M", "㋿㋿", "m", "<|", "endoftext", "|>", "sé", "½", "\r\n", "…"]} +{"text": "'M!Dž㋿\n\n/İ ‍iOS'MᵃDžungla!!tß\na\u000b\n\nß -'VE\n/fiaBé9å<|fim_prefix|>12345678'ſé", "tokens": 56, "pieces": ["'M", "!Dž", "㋿\n\n/", "İ", " ", "‍i", "OS'M", "ᵃDžungla", "!!", "tß", "\n", "a", "\u000b\n\n", "ß", " ", " -'", "VE", "\n", "/fia", "Bé", "9", "å", "<|", "fim", "_prefix", "|>", "123", "456", "78", "'ſé"]} +{"text": ",ᵃcamelCase", "tokens": 6, "pieces": [",ᵃcamel", "Case"]} +{"text": "Ab
-ꟲ'D'\u000b>­#$%t ㍿iOS'M iOS\r\n\"🙂iOS👍🏽-ßa/b0 𐞁#$%\t", "tokens": 50, "pieces": ["Ab", "
", "-ꟲ'D", "'", "\u000b", "><", "META", "_START", ">­#$%", "t", " ㍿", "i", "OS'M", " i", "OS", "\r\n", "\"🙂", "i", "OS", "👍🏽-", "ßa", "/b", "0", " ", "𐞁", "#$%", "\t"]} +{"text": "ᵃ\r\n­३ZHTTPServerABC\n/<|endoftext|>", "tokens": 18, "pieces": ["ᵃ", "\r\n", "­", "३", "ZHTTPServer", "ABC", "\n", "/<|", "endoftext", "|>"]} +{"text": "d३>'Re-\u000b", "tokens": 6, "pieces": ["d", "३", ">'", "Re", "-", "\u000b"]} +{"text": "9 ㋿", "tokens": 5, "pieces": ["9", " ", "㋿"]} +{"text": "\u000bB\t㋿/s", "tokens": 8, "pieces": ["\u000bB", "\t", "㋿/", "s"]} +{"text": "ꟲ😀🏽Dž/ABCaBAßiOS'Rea", "tokens": 18, "pieces": ["ꟲ", "😀🏽", "Dž", "/ABCa", "BAßi", "OS'Re", "a"]} +{"text": "éḍ̇<٣٤٥٦HTTPServer,, ", "tokens": 13, "pieces": ["éḍ̇", "<", "٣٤٥", "٦", "HTTPServer", ",,", " "]} +{"text": "'T
İ'VEⅣß", "tokens": 8, "pieces": ["'T", "
İ'VE", "Ⅳ", "ß"]} +{"text": "12345678DžAb\u000be(.\tABCiOSß​ꟲå漢'll­<|fim_prefix|>­é㋿ \n ", "tokens": 36, "pieces": ["123", "456", "78", "DžAb", "\u000be", "(.", "\tABCi", "OSß", "​ꟲå漢'll", "­<|", "fim", "_prefix", "|>­", "é", "㋿", " \n", " "]} +{"text": "ꟲ ع\n<|endoftext|> \n 'S'VE#$%'T👍🏽/\r\né", "tokens": 33, "pieces": ["ꟲ", " ", " ع", "\n", "<|", "endoftext", "|>", " \n", " '", "S'VE", "#$%'", "T", "👍🏽/\r\n", "<", "EOT", ">é"]} +{"text": "\n/Z \n ", "tokens": 4, "pieces": ["\n", "/Z", " \n", " "]} +{"text": "!\n/'Reå😀🏽\r\n\n/Ⅳ'Me字fi'VEſ漢 \n!a/bDž­🙂.𐞁-", "tokens": 34, "pieces": ["!\n/", "'Reå", "😀🏽\r\n\n/", "Ⅳ", "'Me字fi'VE", "ſ漢", " \n", "!a", "/b", "Dž", "­🙂.", "𐞁", "-"]} +{"text": "\rEOT​Z<|endoftext|>́#$%/", "tokens": 16, "pieces": ["\r", "EOT", "​Z", "<|", "endoftext", "|>́#$%/"]} +{"text": "-字‍\r‍!­'re३ß½a/b'٣٤٥٦0iOSA🙂ḍ̇ꟲ㍿camelCase/🙂ḍ̇‍'ll\tſḍ̇>'re३-a /\r\n", "tokens": 60, "pieces": ["-字", "‍\r", "‍!­'", "re", "३", "ß", "½", "a", "/b", "'", "٣٤٥", "٦0", "i", "OSA", "🙂ḍ̇ꟲ", "㍿camel", "Case", "/🙂", "ḍ̇", "‍'", "ll", "\tſḍ̇", ">'", "re", "३", "-<", "META", "_START", ">a", " ", "/\r\n"]} +{"text": "😀🏽㋿­.éHTTPServer/\r\n9eABC!iOSİ'S12345678é'S-
  \n 's/​/-", "tokens": 34, "pieces": ["😀🏽㋿­.", "é", "HTTPServer", "/\r\n", "9", "e", "ABC", "!i", "OSİ'S", "123", "456", "78", "é'S", "-", "
  \n", " '", "s", "/​/-"]} +{"text": "éAb\r\n/>BEOT!!sDžå0're\n'S\r\n\r\n\n/😀🏽 \n <|endoftext|>\u000b<'T'Re'TaBß \n<|endoftext|>/ \n㍿\u000b", "tokens": 56, "pieces": ["é", "Ab", "\r\n", "/>", "BEOT", "!!", "s", "Džå", "0", "'re", "\n", "'S", "\r\n\r\n\n", "/😀🏽", " \n", " <|", "endoftext", "|>", "\u000b", "<'", "T'Re", "'Ta", "Bß", " \n", "<|", "endoftext", "|>/", " \n", "㍿", "\u000b"]} +{"text": " 0,é'sé\r\n\r\nABC's\r\n\rⅣte!A ㋿", "tokens": 22, "pieces": [" ", " ", "0", ",é's", "é", "\r\n\r\n", "ABC's", "\r\n\r", "Ⅳ", "te", "!A", " ", "㋿"]} +{"text": "ma/b字Džungla😀🏽B/\r\n🙂 td𐞁 \"#$%ABC'ssB\"'T\r\n'M", "tokens": 30, "pieces": ["ma", "/b字", "Džungla", "😀🏽", "B", "/\r\n", "🙂", " td𐞁", " \"#$%", "ABC's", "s", "B", "\"'", "T", "\r\n", "'M"]} +{"text": "٣٤٥٦0㍿ \n", "tokens": 9, "pieces": ["٣٤٥", "٦0", "㍿", " \n"]} +{"text": "'T\"å \n aBDžZ ́İ
'Re\r\n'S' 𐞁½\nå 漢s\t0\r\n<‍'ll'll٣٤٥٦㍿tHTTPServerfi", "tokens": 52, "pieces": ["'T", "\"å", " \n", " a", "BDžZ", " ́", "İ", "
", "'Re", "\r\n", "'S", "'", " ", " 𐞁", "½", "\n", "å", " 漢s", "\t", "0", "\r\n", "<<", "EOT", ">‍'", "ll'll", "٣٤٥", "٦", "㍿t", "HTTPServerfi"]} +{"text": "İå㋿", "tokens": 6, "pieces": ["İå", "㋿"]} +{"text": "'VE're'ſſEOT\r\n\r\n", "tokens": 12, "pieces": ["'VE're", "'", "ſſ", "EOT", "\r\n\r\n"]} +{"text": "Bs ع>ſ\r\nd字漢", "tokens": 9, "pieces": ["Bs", " ع", ">ſ", "\r\n", "d字漢"]} +{"text": "Abḍ̇½mfi'retDž\r\ńss\r\n🙂", "tokens": 16, "pieces": ["Abḍ̇", "½", "mfi're", "t", "Dž", "\r\n", "́ss", "\r\n", "🙂"]} +{"text": "𐞁 \n ㋿😀🏽 \n aB0 \n EOTEOTꟲ áſſ٣٤٥٦​́! 'M\u000b ㍿", "tokens": 43, "pieces": ["𐞁", " \n", " ㋿😀🏽", " \n", " a", "B", "0", " \n", " EOTEOTꟲ", " áſſ", "٣٤٥", "٦", "​́", "!", " ", " '", "M", "\u000b", " ", "㍿"]} +{"text": "㍿('T\r<|endoftext|>,é'T'Std<…", "tokens": 20, "pieces": ["㍿('", "T", "\r", "<|", "endoftext", "|>,", "é'T", "'Std", "<", "…"]} +{"text": "(.'re're٣٤٥٦ \n 'sEOTⅣ'll㍿aBⅣ!'Re\n\t .", "tokens": 28, "pieces": ["(.'", "re're", "٣٤٥", "٦", " \n", " '", "s", "EOT", "Ⅳ", "'ll", "㍿a", "B", "Ⅳ", "!'", "Re", "\n", "\t", " ."]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "", "tokens": 0, "pieces": []} +{"text": "ZaB/\r\nZmm​' 's'M!́'s'ſ\r\n👍🏽İAbfiع👍🏽Z'ReéHTTPServer( Z字ᵃ9½!!aaBAb\n/'re", "tokens": 48, "pieces": ["Za", "B", "/\r\n", "Zmm", "​'", " ", " '", "s'M", "!́'s", "'ſ", "\r\n", "👍🏽", "İAbfiع", "👍🏽", "Z'Re", "é", "HTTPServer", "(", " ", " Z字ᵃ", "9½", "!!", "aa", "BAb", "\n", "/'", "re"]} +{"text": ">ß'S12345678>👍🏽Abᵃ!fi👍🏽<|endoftext|>\r12345678­a/b🙂\"sHTTPServer\u000b#$%/ ß\t", "tokens": 44, "pieces": [">ß'S", "123", "456", "78", ">👍🏽", "Abᵃ", "!fi", "👍🏽<|", "endoftext", "|>\r", "123", "456", "78", "­a", "/b", "🙂\"", "s", "HTTPServer", "\u000b", "#$%/", " ß", "\t"]} +{"text": "t'!!Džungladß12345678", "tokens": 11, "pieces": ["t", "'!!", "Džungladß", "123", "456", "78"]} +{"text": " 'Re👍🏽 \n 'Dḍ̇ 'ſ'll \r\n\r\n\"🙂́'!!!", "tokens": 22, "pieces": [" ", " '", "Re", "👍🏽", " \n", " '", "Dḍ̇", " ", "'ſ'll", " \r\n\r\n", "\"🙂́'!!!"]} +{"text": "fi<|endoftext|>ſ'så\r\n<|endoftext|>𐞁𐞁m½e𐞁d́½Ab/<|fim_prefix|>'Td'MaB0​(㍿'ſ'D'S'Re'Re­ſ́", "tokens": 70, "pieces": ["fi", "<|", "endoftext", "|>", "ſ's", "å", "\r\n", "<|", "endoftext", "|>", "𐞁𐞁m", "½", "e𐞁", "d́", "½", "Ab", "/<|", "fim", "_prefix", "|>'", "Td'M", "a", "B", "0", "​(㍿'", "ſ'D", "'S'Re", "'Re", "­ſ́"]} +{"text": "aB\r\n\r\n㍿İ\r\n\r\n‍EOT \n d Z\r\n\r\n… \n \n's'Re0\tİ\n/Ab'll字ß\n/'D", "tokens": 33, "pieces": ["a", "B", "\r\n\r\n", "㍿İ", "\r\n\r\n", "‍EOT", " \n", " d", " Z", "\r\n\r\n… \n \n", "'s'Re", "0", "\tİ", "\n", "/Ab'll", "字ß", "\n", "/'", "D"]} +{"text": "'D!12345678𐞁\r\n\r\na'DA'VE😀🏽'S(­
🙂mß\"٣٤٥٦ Džungla\u000b\"", "tokens": 38, "pieces": ["'D", "!", "123", "456", "78", "𐞁", "\r\n\r\n", "a'D", "A'VE", "😀🏽'", "S", "(­", "
", "🙂mß", "\"", "٣٤٥", "٦", " Džungla", "\u000b", "\""]} +{"text": "ea/b're👍🏽عåtaB<|endoftext|>As½㋿-Ab🙂,Zfi\n\r\n\r\n‍㍿'re<|fim_prefix|>­0\n/#$%İ're", "As", "½", "㋿-", "Ab", "🙂,", "Zfi", "\n\r\n\r\n", "‍㍿'", "re", "<|", "fim", "_prefix", "|>­", "0", "\n", "/#$%", "İ're", ",e're're", "tokens": 18, "pieces": [" ", " Abéᵃꟲ", "<|", "fim", "_prefix", "|>,", "e're", "'re"]} +{"text": " 'llA<|endoftext|>Ab'll'sDž 'M#$%'D9#$%mᵃ/‍ſ'Re", "tokens": 34, "pieces": [" ", "'ll", "A", "<|", "endoftext", "|>", "Ab'll", "'s", "Dž", "", " ", "'M", "#$%'", "D", "9", "#$%", "mᵃ", "/‍", "ſ'Re"]} +{"text": "ꟲ\n9\r\n\r\né!'DiOS👍🏽//s\r'reᵃḍ̇'Ree(aAb'DᵃB\r 𐞁🙂!!", "tokens": 40, "pieces": ["ꟲ", "\n", "9", "\r\n\r\n", "é", "!'", "Di", "OS", "👍🏽//", "s", "\r", "'reᵃḍ̇'Re", "e", "(a", "Ab'D", "ᵃ", "B", "\r", " 𐞁", "🙂!!"]} +{"text": "‍AbſcamelCase𐞁0\r\n\r\n12345678ḍ̇e!ḍ̇Bᵃ'Rea/b", "tokens": 29, "pieces": ["‍Abſcamel", "Case𐞁", "0", "\r\n\r\n", "123", "456", "78", "ḍ̇e", "!ḍ̇", "Bᵃ'Re", "a", "/b"]} +{"text": "'re…㋿​", "tokens": 7, "pieces": ["'re", "…", "㋿​"]} +{"text": " EOT're's'D'VEİ
s\t\u000b.'llDž,😀🏽", "tokens": 20, "pieces": [" EOT're", "'s'D", "'VEİ", "
s", "\t", "\u000b", ".'", "ll", "Dž", ",😀🏽"]} +{"text": "👍🏽ḍ̇'D", "tokens": 7, "pieces": ["👍🏽", "ḍ̇'D"]} +{"text": "/Aa/b 'M🙂é字😀🏽'ſ\t漢 字­😀🏽m३'D\u000bm​ 字'M\"ꟲ'M<|endoftext|> 'D<|endoftext|>å", "tokens": 59, "pieces": ["/Aa", "/b", " ", "'M", "🙂é字", "😀🏽'", "ſ", "\t漢", " ", " 字", "­😀🏽", "m", "३", "'", "D", "\u000bm", "​", " 字'M", "\"ꟲ'M", "<|", "endoftext", "|>", " '", "D", "<|", "endoftext", "|>", "å"]} +{"text": "é½\r\n\r\n\r\n<|fim_prefix|>́Dž३'TcamelCase", "tokens": 17, "pieces": ["é", "½", "\r\n\r\n\r\n", "<|", "fim", "_prefix", "|>́", "Dž", "३", "'Tcamel", "Case"]} +{"text": "d٣٤٥٦'ſ\u000b​>३
'M", "tokens": 16, "pieces": ["d", "٣٤٥", "٦", "'ſ", "\u000b", "​>", "३", "
", "'M", ""]} +{"text": "\n'reع字BſéficamelCase👍🏽'ABC,9aB \n𐞁
\n/fi're", "tokens": 36, "pieces": ["\n", "'reع字", "Bſéficamel", "Case", "👍🏽<", "META", "_START", ">'", "ABC", ",", "9", "a", "B", " \n", "𐞁", "
\n", "/fi're"]} +{"text": "/\r\n's<|endoftext|>'s½…fia/b!!t'ſ9'\r\n é'SaBm\u000bZ'0é
<|fim_prefix|>d\rA㋿", "tokens": 49, "pieces": ["/\r\n", "'s", "<|", "endoftext", "|>'", "s", "½", "…fia", "/b", "!!", "t", "'", "ſ", "9", "'\r\n", " ", " é'S", "a", "Bm", "\u000bZ", "'", "0", "é", "
", "<|", "fim", "_prefix", "|>", "d", "\r", "A", "㋿"]} +{"text": "ḍ̇漢fiḍ̇0--ꟲ\r​12345678ßİ å<|fim_prefix|>DžunglaHTTPServer'Re\u000bAb👍🏽('e'T'T\n!A㍿㍿­", "tokens": 55, "pieces": ["ḍ̇漢fiḍ̇", "0", "--", "ꟲ", "\r", "​", "123", "456", "78", "ß", "İ", " å", "<|", "fim", "_prefix", "|>", "Džungla", "HTTPServer'Re", "\u000bAb", "👍🏽('", "e'T", "'T", "\n", "!A", "㍿㍿­"]} +{"text": "a/b- 𐞁\n/ t.fi e", "tokens": 15, "pieces": ["a", "/b", "-", " 𐞁", "\n", "/", " t", ".fi", " e"]} +{"text": "ᵃ!!'Re--. Z", "tokens": 10, "pieces": ["ᵃ", "!!'", "Re", "--.", " Z"]} +{"text": "İZDžungla㋿(å㍿iOS🙂aß३", "tokens": 21, "pieces": ["İZDžungla", "㋿(", "å", "㍿i", "OS", "🙂aß", "३"]} +{"text": "३å'M<|fim_prefix|>EOT'S
!!👍🏽é! ", "tokens": 21, "pieces": ["३", "å'M", "<|", "fim", "_prefix", "|>", "EOT'S", "
", "!!👍🏽", "é", "!", " "]} +{"text": "­‍\"\n/EOT'll'D'llⅣ", "tokens": 10, "pieces": ["­‍\"\n/", "EOT'll", "'D'll", "Ⅳ"]} +{"text": "'\n‍'ll\n/t👍🏽 \n HTTPServer٣٤٥٦'T字🙂9\n", "tokens": 29, "pieces": ["'\n", "‍'", "ll", "\n", "/t", "👍🏽<", "EOT", ">", " \n", " <", "META", "_START", ">HTTPServer", "٣٤٥", "٦", "'T字", "🙂", "9", "\n"]} +{"text": "a/bİ\tİta/b'DcamelCase'll'ſ \n \r\n\r\n", "tokens": 15, "pieces": ["a", "/b", "İ", "\tİta", "/b'D", "camel", "Case'll", "'ſ", " \n \r\n\r\n"]} +{"text": "\"́/ᵃ㋿​ >٣٤٥٦", "tokens": 19, "pieces": ["\"́", "/ᵃ", "㋿<", "EOT", ">​", " >", "٣٤٥", "٦"]} +{"text": "'VEABC's(e#$%<|endoftext|>𐞁'Da", "tokens": 20, "pieces": ["'VEABC's", "(e", "#$%<|", "endoftext", "|>", "𐞁'D", "a"]} +{"text": "t\r!!'sEOT'llع'T🙂😀🏽字​, ‍\r\n\r\nḍ̇\r\n\r\nB/'re<'ſm", "tokens": 40, "pieces": ["t", "\r", "!!'", "s", "EOT'll", "ع", "'", "T", "🙂😀🏽", "字", "​,", " ", "‍<", "EOT", ">\r\n\r\n", "ḍ̇", "\r\n\r\n", "B", "/'", "re", "<'", "ſm"]} +{"text": "ع٣٤٥٦\r漢Ab<|fim_prefix|><|fim_prefix|>><|endoftext|> \n 
👍🏽d½㍿<'sZ'M'll./\r\n're é", "tokens": 50, "pieces": ["ع", "٣٤٥", "٦", "\r", "漢Ab", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>><|", "endoftext", "|>", " \n", " ", "
", "👍🏽", "d", "", "½", "㍿<'", "s", "Z'M", "'ll", "./\r\n", "'re", " ", " é"]} +{"text": "Dž'MAb", "tokens": 4, "pieces": ["Dž'M", "Ab"]} +{"text": "<\rꟲ٣٤٥٦a/bå'refiDžunglaᵃ0camelCaseḍ̇''ll'SⅣ'sⅣé<|endoftext|>/\r\n", "tokens": 45, "pieces": ["<\r", "ꟲ", "٣٤٥", "٦", "a", "/bå're", "fi", "Džunglaᵃ", "0", "camel", "Caseḍ̇", "''", "ll'S", "Ⅳ", "'s", "Ⅳ", "é", "<|", "endoftext", "|>/\r\n"]} +{"text": "/\r\n  -\u000b(<|fim_prefix|>emHTTPServer#$%ḿ½​#$%", "tokens": 22, "pieces": ["/\r\n", " ", " ", "-", "\u000b", "(<|", "fim", "_prefix", "|>", "em", "HTTPServer", "#$%", "ḿ", "½", "​#$%"]} +{"text": "é/0/\r\n!!a'll \nꟲ(Ab/٣٤٥٦'reå🙂<|fim_prefix|>'Re EOT'M\r\n\r\n/\r\n­ABC#$%ßHTTPServer'Deſ", "tokens": 46, "pieces": ["é", "/", "0", "/\r\n", "!!", "a'll", " \n", "ꟲ", "(Ab", "/", "٣٤٥", "٦", "'reå", "🙂<|", "fim", "_prefix", "|>'", "Re", " EOT'M", "\r\n\r\n", "/\r\n", "­ABC", "#$%", "ß", "HTTPServer'D", "eſ"]} +{"text": "'reABC'reHTTPServerHTTPServerfi
'M 'Tİ\"
'M…A'S\ré٣٤٥٦\r\n\r\nfi­\r0Zᵃ́å!!'ſcamelCase.camelCase#$%>'Så", "tokens": 51, "pieces": ["'re", "ABC're", "HTTPServer", "HTTPServerfi", "
", "'M", " ", "'Tİ", "\"", "
", "'M", "…A'S", "\r", "é", "٣٤٥", "٦", "\r\n\r\n", "fi", "­\r", "0", "Zᵃ́å", "!!'", "ſcamel", "Case", ".camel", "Case", "#$%>'", "Så"]} +{"text": "'t \n 'ſع'D½ع漢12345678­عå😀🏽😀🏽'", "tokens": 33, "pieces": ["'t", " \n", " '", "ſع'D", "½", "ع漢", "123", "456", "78", "­ع", "å", "😀🏽😀🏽'"]} +{"text": ".٣٤٥٦DžABC <😀🏽‍,B9\"Ae, \n ", "tokens": 22, "pieces": [".", "٣٤٥", "٦", "DžABC", " ", " <😀🏽‍,", "B", "9", "\"Ae", ",", " \n", " "]} +{"text": "\r\n\rße\u000bⅣ字-\r\n\r\n'Mḍ̇#$%12345678.३',ᵃ👍🏽då'ᵃ<-\r\n\r\nꟲ-'ſ's👍🏽(", "tokens": 45, "pieces": ["\r\n\r", "ße", "\u000b", "Ⅳ", "字", "-\r\n\r\n", "'Mḍ̇", "#$%", "123", "456", "78", ".", "३", "',", "ᵃ", "👍🏽", "då", "'ᵃ", "<-\r\n\r\n", "ꟲ", "-'", "ſ's", "👍🏽("]} +{"text": "Båß\r!!ꟲ  ᵃ😀🏽😀🏽\"0- 'M", "tokens": 24, "pieces": ["Båß", "\r", "!!", "ꟲ", " ", " ᵃ", "😀🏽😀🏽\"", "0", "-", " ", "'M"]} +{"text": "'ReABCå'DAa İ12345678/\r\n9", "tokens": 13, "pieces": ["'Re", "ABCå'D", "Aa", " İ", "123", "456", "78", "/\r\n", "9"]} +{"text": "'VEİßꟲ((​😀🏽<|fim_prefix|>AZ'Dß<|fim_prefix|>'ſ,'re#$%", "tokens": 36, "pieces": ["'VEİßꟲ", "((​😀🏽<|", "fim", "_prefix", "|>", "AZ'D", "ß", "<|", "fim", "_prefix", "|>'", "ſ", ",'", "re", "#$%"]} +{"text": ",\n \ndſ'D😀🏽#$%/\r\nss/e>字fia/b/Džungla\n/ 🙂", "tokens": 27, "pieces": [",\n", " \n", "dſ'D", "😀🏽#$%/\r\n", "ss", "/e", ">字fia", "/b", "/Džungla", "\n", "/", " ", "🙂"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "\r​!!'resA…İ٣٤٥٦Ⅳꟲ''s½\tZ\"9३३\t\"İ\r\n\r\n'll(\n", "tokens": 32, "pieces": ["\r", "​!!'", "res", "A", "…İ", "٣٤٥", "٦Ⅳ", "ꟲ", "''", "s", "½", "\tZ", "\"", "9३३", "\t", "\"İ", "\r\n\r\n", "'ll", "(\n"]} +{"text": "så  \n ſⅣ/\r\n\u000b\"🙂're ㋿ \n - \n ZA Ab
d,🙂🙂'Re'Ma/b字<|fim_prefix|>'VEⅣ😀🏽\r\n\r\n\r\n<|fim_prefix|>İ\u000b<12345678, st", "tokens": 54, "pieces": ["'𐞁", "Z", "A", " ", " Ab", "
d", ",🙂🙂'", "Re'M", "a", "/b字", "<|", "fim", "_prefix", "|>'", "VE", "Ⅳ", "😀🏽\r\n\r\n\r\n", "<|", "fim", "_prefix", "|>", "İ", "\u000b", "<", "123", "456", "78", ",", " ", " st"]} +{"text": "camelCasedåABCDž𐞁'VÉ'Mع😀🏽", "tokens": 20, "pieces": ["camel", "Casedå", "ABCDž𐞁'VE", "́'M", "ع", "😀🏽"]} +{"text": "Z\r\n😀🏽ABC'S .'Sddḍ̇'Res½'SiOS\nABC<|endoftext|>a/b12345678-İ \r\n\r\n👍🏽 ꟲ'T\"EOTé", "tokens": 52, "pieces": ["Z", "\r\n", "😀🏽", "ABC'S", " ", " .'", "Sddḍ̇'Re", "s", "", "½", "'Si", "OS", "\n", "ABC", "<|", "endoftext", "|>", "a", "/b", "123", "456", "78", "-İ", " \r\n\r\n", "👍🏽", " ꟲ'T", "\"EOTé"]} +{"text": "\tİB'ſ😀🏽<|endoftext|>İHTTPServer\r\nt٣٤٥٦'VE's㍿
🙂\r\n​'Re\n", "İHTTPServer", "\r\n", "t", "٣٤٥", "٦", "'VE's", "㍿", "
", "🙂\r\n", "​'", "Re", "\n", "३字Džungla'S㋿'Re/…㍿tABCå'S<|fim_prefix|>३,́.aB'VE \n‍e​ 𐞁", "tokens": 51, "pieces": ["", "३", "字Džungla'S", "㋿'", "Re", "/", "…", "㍿t", "ABC", "å'S", "<|", "fim", "_prefix", "|>", "३", ",́", ".a", "B'VE", " \n", "‍e", "​", " 𐞁"]} +{"text": " \tZaBcamelCaset'ſ,ſ(mßå/12345678㍿½
Z'red", "tokens": 30, "pieces": [" ", "\t", "Za", "Bcamel", "Caset'ſ", ",ſ", "(mßå", "/", "123", "456", "78", "㍿", "½", "
Z're", "d"]} +{"text": "'🙂camelCase<|fim_prefix|>", "tokens": 10, "pieces": ["'🙂", "camel", "Case", "<|", "fim", "_prefix", "|>"]} +{"text": "camelCaseḍ̇-ſ/\r\nåaB'ſ\u000b㍿㋿camelCase
'‍'.Z/\r\n-iOS#$%\t'S're㍿ḍ̇ fi‍\"", "tokens": 57, "pieces": ["camel", "Caseḍ̇", "-ſ", "/\r\n", "å", "a", "B'ſ", "\u000b", "㍿㋿", "camel", "Case", "
", "'‍'.", "Z", "/\r\n", "-<", "META", "_START", ">i", "OS", "#$%", "\t", "'S're", "㍿ḍ̇", " ", " fi", "‍\""]} +{"text": "aB's''  A'/\r\nA'.", "tokens": 11, "pieces": ["a", "B's", "''", " ", " A", "'/\r\n", "A", "'."]} +{"text": "'ſ  'llå‍ſABC'S😀🏽12345678-㋿Bm /\r\n…AbB㋿\r\n\r\n​é 'Ś‍\rDžungla字å", "tokens": 48, "pieces": ["'ſ", " ", " ", "'llå", "‍ſ", "ABC'S", "😀🏽", "123", "456", "78", "-㋿", "Bm", " ", " /\r\n", "…Ab", "B", "㋿\r\n\r\n", "​é", " ", "'Ś", "‍\r", "Džungla字å"]} +{"text": "'S\r\n​‍ 'så<|endoftext|>٣٤٥٦å!!\t/\r\n
A!ḍ̇scamelCase(éꟲé½/\r\nꟲß<|endoftext|>å🙂\n/#$%ß", "tokens": 60, "pieces": ["'S", "\r\n", "​‍", " ", " '", "så", "<|", "endoftext", "|>", "٣٤٥", "٦", "å", "!!", "\t", "/\r\n", "
A", "!ḍ̇scamel", "Case", "(éꟲé", "½", "/\r\n", "ꟲß", "<|", "endoftext", "|>", "å", "🙂\n/", "#$%", "ß"]} +{"text": " \u000b/\r\n'll‍ſsß,३eéaB'ré'Sm㋿ /\r\n\n/Z😀🏽ſ\"iOS'㋿ᵃcamelCase𐞁 \n Aa/bEOTB!", "tokens": 59, "pieces": [" ", "\u000b", "/\r\n", "'ll", "‍ſsß", ",", "३", "eéa", "B're", "́'S", "m", "㋿", " ", "/\r\n\n/", "Z", "😀🏽", "ſ", "\"i", "OS", "'㋿", "ᵃcamel", "Case𐞁", " \n", " Aa", "/b", "EOTB", "!"]} +{"text": "\n\taB,s9d Ⅳ\nع\n🙂…ABC…é/\r\nm㍿tt'ſ9é­#$%a's<|fim_prefix|>åABCعꟲḍ̇𐞁", "tokens": 55, "pieces": ["\n", "\ta", "B", ",s", "9", "d", " ", "Ⅳ", "\n", "ع", "\n", "🙂", "…ABC", "…é", "/\r\n", "m", "㍿tt'ſ", "9", "é", "­#$%", "a's", "<|", "fim", "_prefix", "|>", "å", "ABCعꟲḍ̇𐞁"]} +{"text": "(iOS\r\n fi/'T字", "tokens": 15, "pieces": ["(i", "OS", "\r\n", "", " fi", "/'", "T字"]} +{"text": "𐞁'D'ſ🙂/­", "tokens": 17, "pieces": ["𐞁", "'", "D'ſ", "🙂<", "EOT", ">/­"]} +{"text": " åEOT camelCase­A\r\nm\r\nd", "tokens": 14, "pieces": [" ", " å", "EOT", " ", " camel", "Case", "­A", "\r\n", "m", "\r\n", "d"]} +{"text": "12345678'T'S,", "tokens": 6, "pieces": ["123", "456", "78", "'T'S", ","]} +{"text": "camelCase‍!'Re<|endoftext|>0Be('", "Re", "<|", "endoftext", "|>", "0", "Be", "(<", "m"]} +{"text": "İ'T \n/ABCm٣٤٥٦🙂ꟲ", "tokens": 14, "pieces": ["İ'T", " \n", "/ABCm", "٣٤٥", "٦", "🙂ꟲ"]} +{"text": "!0DžcamelCase٣٤٥٦a/bm 字a/b
ſ\r\n.'Ⅳ\r\n\r\n \na/b!İ\nA's'll<|fim_prefix|><|fim_prefix|>'Re'sꟲ<ꟲABCꟲEOT'D", "tokens": 64, "pieces": ["!", "0", "Džcamel", "Case", "٣٤٥", "٦", "a", "/bm", " ", " 字a", "/b", "", "
ſ", "\r\n", ".'", "Ⅳ", "\r\n\r\n \n", "a", "/b", "!İ", "\n", "A's", "'ll", "<|", "fim", "_prefix", "|><|", "fim", "_prefix", "|>'", "Re's", "ꟲ", "<ꟲABCꟲ", "EOT'D"]} +{"text": " \n🙂!!", "tokens": 3, "pieces": [" \n", "🙂!!"]} +{"text": "m'VE 'll
.'Re<'ll m", "tokens": 16, "pieces": ["m'VE", " ", " '", "ll", "
", ".'", "Re", "<'", "ll", "", " ", " m"]} +{"text": "9d#$%Ⅳ!\t12345678𐞁\n/\r😀🏽12345678ع\ta,a/b'D!!漢#$%\"ſ", "tokens": 34, "pieces": ["9", "d", "#$%", "Ⅳ", "!", "\t", "123", "456", "78", "𐞁", "\n", "/\r", "😀🏽", "123", "456", "78", "ع", "\ta", ",a", "/b'D", "!!", "漢", "#$%\"", "ſ"]} +{"text": "e'M! \n\r\nſḍ̇'DDž́ABCABC A'Ret'Sa's<ꟲݽs", "tokens": 29, "pieces": ["e'M", "!", " \n\r\n", "ſḍ̇'D", "Dž́", "ABCABC", " A'Re", "t'S", "a's", "<ꟲ", "İ", "½", "s"]} +{"text": "<|endoftext|> /-🙂<", "tokens": 11, "pieces": ["<|", "endoftext", "|>", " ", "/-🙂<"]} +{"text": "s\u000b", "tokens": 2, "pieces": ["s", "\u000b"]} +{"text": " \n 'D'reİ𐞁a/b <|fim_prefix|>½́dm<|endoftext|>字\n/é0fi字\u000bHTTPServer漢'T ­🙂/9å ᵃ\u000b", "tokens": 56, "pieces": [" \n", " '", "D're", "İ𐞁a", "/b", " ", "<|", "fim", "_prefix", "|>", "½", "́dm", "<|", "endoftext", "|>", "字", "\n", "/é", "0", "fi字", "\u000bHTTPServer漢'T", " ", " <", "EOT", ">­🙂/", "9", "å", " ᵃ", "\u000b"]} +{"text": "‍­", "tokens": 2, "pieces": ["‍­"]} +{"text": "iOS٣٤٥٦\r\n\r\n\r,\r \u000b's٣٤٥٦ßß­ \nABC漢😀🏽🙂'VE字0३a'sḍ̇\u000bHTTPServera३字\r\n\r\n<|endoftext|>
😀🏽٣٤٥٦'D㍿", "tokens": 63, "pieces": ["i", "OS", "٣٤٥", "٦", "\r\n\r\n\r", ",\r", " ", "\u000b", "'s", "٣٤٥", "٦", "ßß", "­", " \n", "ABC漢", "😀🏽🙂'", "VE字", "0३", "a's", "ḍ̇", "\u000bHTTPServera", "३", "字", "\r\n\r\n", "<|", "endoftext", "|>", "
", "😀🏽", "٣٤٥", "٦", "'D", "㍿"]} +{"text": "\n/<|fim_prefix|>EOT 'M'D٣٤٥٦EOT.A'DⅣ\t!!< \n/d<0,👍🏽<|endoftext|>t-BaBZABC", "tokens": 47, "pieces": ["\n", "/<|", "fim", "_prefix", "|>", "EOT", " '", "M'D", "٣٤٥", "٦", "EOT", ".A'D", "Ⅳ", "\t", "!!<", " \n", "/d", "<", "0", ",👍🏽<|", "endoftext", "|>", "t", "-Ba", "BZABC"]} +{"text": "EOTdsa/b🙂\r \n\r\n\r\n0'M", "tokens": 11, "pieces": ["EOTdsa", "/b", "🙂\r", " \n\r\n\r\n", "0", "'M"]} +{"text": "é​𐞁/\r\n-,fiA٣٤٥٦<|endoftext|>m​é\r>-d'Re'ſ>aBDž'\nABC", "tokens": 37, "pieces": ["é", "​𐞁", "/\r\n", "-,", "fi", "A", "٣٤٥", "٦", "<|", "endoftext", "|>", "m", "​é", "\r", ">-", "d'Re", "'ſ", ">a", "BDž", "'\n", "ABC"]} +{"text": "ſ>'sḿ'rea👍🏽\"iOS12345678", "tokens": 15, "pieces": ["ſ", ">'", "sḿ're", "a", "👍🏽\"", "i", "OS", "123", "456", "78"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "AcamelCase", "tokens": 3, "pieces": ["Acamel", "Case"]} +{"text": "Zs😀🏽\r\nd-09 !\r0EOT'McamelCase\u000b­३\r\n'Re \n <|fim_prefix|>👍🏽é", "tokens": 37, "pieces": ["Zs", "😀🏽\r\n", "d", "-", "09", "", " ", " !\r", "0", "EOT'M", "camel", "Case", "\u000b", "­", "३", "\r\n", "'Re", " \n", " <|", "fim", "_prefix", "|>👍🏽", "é"]} +{"text": "Z9Afi٣٤٥٦\r…å३", "tokens": 14, "pieces": ["Z", "9", "Afi", "٣٤٥", "٦", "\r", "…å", "३"]} +{"text": "#$% 12345678'll ३'s<|fim_prefix|><|endoftext|>𐞁
'RefiHTTPServerſ \n Bİ​ ‍åſa'T12345678eå>'re
\u000b\n/ /\r\n", "tokens": 57, "pieces": ["#$%", " ", "123", "456", "78", "'ll", " ", "३", "'s", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "𐞁", "
", "'Refi", "HTTPServerſ", " \n", " Bİ", "​", " ", " ‍", "åſa'T", "123", "456", "78", "eå", ">'", "re", "
\u000b\n", "/", " ", " /\r\n"]} +{"text": "㍿ 'VE😀🏽½/>'ſ字m'T字Dž㋿ḍ̇'ſ!'ſ½İ ḍ̇‍fi🙂tss\t👍🏽\"'!!", "tokens": 50, "pieces": ["㍿", " ", "'VE", "😀🏽", "½", "/>'", "ſ字m'T", "字", "Dž", "㋿ḍ̇'ſ", "!'", "ſ", "½", "İ", " ḍ̇", "‍fi", "🙂tss", "\t", "👍🏽\"'!!"]} +{"text": "/a/b \"𐞁Bd''VE0Ⅳᵃᵃ'Mé'Sعm#$%'VE's👍🏽…m\n!ᵃ<|fim_prefix|>aB½/\r\n 'T,
", "tokens": 56, "pieces": ["/a", "/b", " \"", "𐞁Bd", "''", "VE", "0Ⅳ", "ᵃᵃ'M", "é'S", "عm", "#$%'", "VE's", "👍🏽", "…m", "\n", "!ᵃ", "<|", "fim", "_prefix", "|><", "EOT", ">a", "B", "½", "/\r\n", " '", "T", ",", "
"]} +{"text": "<|fim_prefix|>aa/b<😀🏽'M½/\r\n­\r'ſABC​é‍!!­\r\nHTTPServeŕ.Z漢s/ \n ", "tokens": 37, "pieces": ["<|", "fim", "_prefix", "|>", "aa", "/b", "<😀🏽'", "M", "½", "/\r\n", "­\r", "'ſ", "ABC", "​é", "‍!!­\r\n", "HTTPServeŕ", ".Z漢s", "/", " \n", " "]} +{"text": "0­٣٤٥٦Džungla𐞁'ſAeABC're'VEع.(t'rea/b'ſ(…", "tokens": 33, "pieces": ["0", "­", "٣٤٥", "٦", "Džungla𐞁'ſ", "Ae", "ABC're", "'VEع", ".(", "t're", "a", "/b'ſ", "(", "…"]} +{"text": "m iOS'TA12345678\u000b
'Re\r\n\r\ncamelCaseé٣٤٥٦DžDž0,́!!👍🏽å'D ½s
9s㋿ABCDž'ſAb(­", "tokens": 61, "pieces": ["m", " i", "OS'T", "A", "123", "456", "78", "\u000b", "
", "'Re", "\r\n\r\n", "camel", "Caseé", "٣٤٥", "٦", "DžDž", "0", ",́", "!!👍🏽", "å", "<", "META", "_START", ">'", "D", " ", " ", "½", "s", "
", "9", "s", "㋿", "ABCDž'ſ", "Ab", "(­"]} +{"text": "é'­
#$%é\n/٣٤٥٦a dDžungla", "tokens": 21, "pieces": ["é", "'­", "
", "#$%", "é", "\n", "/", "٣٤٥", "٦", "a", " ", " d", "Džungla"]} +{"text": "‍#$%👍🏽 \nİ/ſ㋿<\r\n\r\n\r\nİ,éaBADžungla\r\n\r\nAßABCaB\r\n('S'McamelCase…㋿😀🏽\t(9", "tokens": 48, "pieces": ["‍#$%👍🏽", " \n", "İ", "/ſ", "㋿<\r\n\r\n\r\n", "İ", ",éa", "BADžungla", "\r\n\r\n", "Aß", "ABCa", "B", "\r\n", "('", "S'M", "camel", "Case", "…", "㋿😀🏽", "\t", "(", "9"]} +{"text": "\ré\r\nA /\r'TEOT!0\"'VEꟲḍ̇㍿​‍㋿fi'T😀🏽.HTTPServer", "tokens": 38, "pieces": ["\r", "é", "\r\n", "A", " /\r", "'TEOT", "!", "0", "\"'", "VEꟲḍ̇", "㍿​‍㋿", "fi'T", "😀🏽.", "HTTPServer"]} +{"text": "'Re", "tokens": 1, "pieces": ["'Re"]} +{"text": "DžunglafiABC漢 ''M
\r\n\r\nfia/bᵃé\"'Re…٣٤٥٦/\r\nDž​'s!!m,B", "tokens": 36, "pieces": ["Džunglafi", "ABC漢", " ", "''", "M", "
\r\n\r\n", "fia", "/bᵃé", "\"'", "Re", "…", "٣٤٥", "٦", "/\r\n", "Dž", "​'", "s", "!!", "m", ",B"]} +{"text": "漢éßtiOS½\n/t/A<|endoftext|>​é.\t \n Dž㋿.>'S㋿
> A\tåaB>'Dfi9", "tokens": 46, "pieces": ["漢éßti", "OS", "½", "\n", "/t", "/A", "<|", "endoftext", "|>​", "é", ".", "\t \n", " Dž", "㋿.>'", "S", "㋿", "
", ">", " A", "\tåa", "B", ">'", "Dfi", "9"]} +{"text": "iOS're(A<|endoftext|>sEOT  \u000b😀🏽.'Reſ \n­㍿#$%Ⅳİꟲ!<|endoftext|>½ß漢HTTPServer!!<|endoftext|>12345678é\u000b/<|endoftext|>", "tokens": 73, "pieces": ["i", "OS're", "(A", "<|", "endoftext", "|>", "s", "EOT", "  ", "\u000b", "😀🏽.'", "Reſ", " \n", "­㍿#$%", "Ⅳ", "İꟲ", "!<|", "endoftext", "|>", "½", "ß漢", "HTTPServer", "!!<|", "endoftext", "|>", "123", "456", "78", "é", "\u000b", "/<|", "endoftext", "|>"]} +{"text": "ABCDžungla३…\n", "tokens": 9, "pieces": ["ABCDžungla", "३", "…\n"]} +{"text": "t\u000b'll!camelCase…EOT,m'S/\r\nA \n ABC'TeⅣ/​漢s'Mİ-👍🏽‍'TcamelCase", "tokens": 42, "pieces": ["t", "\u000b", "'ll", "!camel", "Case", "…EOT", ",m'S", "/\r\n", "A", " \n", " ABC'T", "e", "Ⅳ", "/​<", "META", "_START", "><", "META", "_START", ">漢s'M", "İ", "-👍🏽‍'", "Tcamel", "Case"]} +{"text": "…㍿Ⅳ,Džİ \ncamelCase'!!0́½ß \né\r< \n/", "tokens": 27, "pieces": ["…", "㍿", "Ⅳ", ",Džİ", " \n", "camel", "Case", "'!!", "0", "́", "½", "ß", " \n", "é", "\r", "<", " \n", "/"]} +{"text": "12345678ſ!ABC0aBsdEOTHTTPServerA Ab-漢½/må", "tokens": 23, "pieces": ["123", "456", "78", "ſ", "!ABC", "0", "a", "Bsd", "EOTHTTPServer", "A", " Ab", "-漢", "½", "/må"]} +{"text": "e", "tokens": 1, "pieces": ["e"]} +{"text": "", "tokens": 0, "pieces": []} +{"text": "aB", "tokens": 2, "pieces": ["a", "B"]} +{"text": "漢 \n \"
A½ḍ̇<|fim_prefix|><|endoftext|>Džع9", "tokens": 28, "pieces": ["漢", "", " \n", " \"", "
A", "½", "ḍ̇", "<|", "fim", "_prefix", "|><|", "endoftext", "|>", "Džع", "9"]} +{"text": "ꟲ", "tokens": 3, "pieces": ["ꟲ"]} +{"text": "ḍ̇", "tokens": 3, "pieces": ["ḍ̇"]} +{"text": "e३'s'Re,é,at🙂/\r\n½iOS'VEfi-🙂aB\r\nſDžungla \n㋿((", "tokens": 34, "pieces": ["e", "३", "'s'Re", ",é", ",at", "🙂/\r\n", "½", "i", "OS'VE", "fi", "-🙂", "a", "B", "\r\n", "ſ", "Džungla", " \n", "㋿<", "EOT", ">(("]} +{"text": "ع​/\r\n \n İ", "tokens": 5, "pieces": ["ع", "​/\r\n", " \n", " İ"]} +{"text": " 𐞁😀🏽​", "tokens": 10, "pieces": [" ", " 𐞁", "😀🏽​"]} +{"text": " ‍‍#$%\tétaB''M", "tokens": 10, "pieces": [" ", " ‍‍#$%", "\téta", "B", "''", "M"]} +{"text": "9㍿es", "tokens": 5, "pieces": ["9", "㍿es"]} +{"text": "\u000b9'Re­\t''ſ'llḍ̇m\u000bDžunglaḍ̇ \n a", "tokens": 23, "pieces": ["\u000b", "9", "'Re", "­", "\t", "''", "ſ'll", "ḍ̇m", "\u000bDžunglaḍ̇", " \n", " ", " a"]} +{"text": "camelCase9Dž12345678camelCaseAb'sAbABCcamelCase'/\r\nDžungla!!>Dž­e…A", "tokens": 31, "pieces": ["camel", "Case", "9", "Dž", "123", "456", "78", "camel", "Case", "Ab's", "Ab", "ABCcamel", "Case", "'/\r\n", "Džungla", "!!>", "Dž", "­e", "…A"]} +{"text": "İ'VE,", "tokens": 4, "pieces": ["İ'VE", ","]} +{"text": "aB🙂…ABCع'ſ'DHTTPServer😀🏽as#$%\r'VE'a/b٣٤٥٦३d <|endoftext|>㍿#$%'Re's 😀🏽", "tokens": 47, "pieces": ["a", "B", "🙂", "…ABCع'ſ", "'DHTTPServer", "😀🏽", "as", "#$%\r", "'VE", "'a", "/b", "٣٤٥", "٦३", "d", " <|", "endoftext", "|>㍿#$%'", "Re's", " ", " 😀🏽"]} +{"text": "aB ­ ㍿ét,#$%㍿s's\r\nḍ̇ß\r\n!9ᵃ", "tokens": 27, "pieces": ["a", "B", " ­", " ", "㍿ét", ",#$%㍿", "s's", "\r\n", "ḍ̇ß", "\r\n", "!", "9", "ᵃ"]} +{"text": "\n/åéHTTPServer", "tokens": 6, "pieces": ["\n", "/åé", "HTTPServer"]} +{"text": "\ré字\n", "tokens": 4, "pieces": ["\r", "é字", "\n"]} +{"text": "EOTİ(́HTTPServer'Re> a
(\r\nfi 
/ iOS'D", "tokens": 21, "pieces": ["EOTİ", "(́HTTPServer'Re", ">", " a", "
", "(\r\n", "fi", " ", "
", "/", " i", "OS'D"]} +{"text": "EOT- \r\n\r\n\nt'MB​ſ'Reaé-
㍿Džungla. 0㍿é", "tokens": 31, "pieces": ["EOT", "-", " \r\n\r\n\n", "t'M", "B", "​ſ'Re", "aé", "-", "
", "㍿Džungla", ".", " ", " ", "0", "㍿é"]} +{"text": "a12345678<|endoftext|>!!édDž\rAb \nB#$% 𐞁🙂𐞁#$%'ll0'ſ'D🙂'Re123456780👍🏽're \n", "tokens": 50, "pieces": ["a", "123", "456", "78", "<|", "endoftext", "|>!!", "éd", "Dž", "\r", "Ab", " \n", "B", "#$%", " 𐞁", "🙂𐞁", "#$%'", "ll", "0", "'ſ'D", "🙂'", "Re", "123", "456", "780", "👍🏽'", "re", " \n"]} +{"text": "\r>ABC٣٤٥٦\"٣٤٥٦½ 'sⅣEOT0­­ \n a(/\r\n🙂漢.- DžunglaB
 \n عß👍🏽ع/'s", "tokens": 50, "pieces": ["\r", "><", "META", "_START", ">ABC", "٣٤٥", "٦", "\"", "٣٤٥", "٦½", " ", "'s", "Ⅳ", "EOT", "0", "­­", " \n", " a", "(/\r\n", "🙂漢", ".-", " Džungla", "B", "
 \n", " عß", "👍🏽", "ع", "/'", "s"]} +{"text": " 👍🏽
‍'ll.Džungla'ReABC12345678𐞁­Z\n/‍​'Re𐞁'M,", "tokens": 39, "pieces": [" ", " 👍🏽", "
", "‍'", "ll", ".Džungla'Re", "ABC", "123", "456", "78", "𐞁", "­Z", "\n", "/‍​<", "META", "_START", ">'", "Re𐞁'M", ","]} +{"text": "éⅣ0<|fim_prefix|>'fitdAtfi'reİ🙂a/bfiſ㋿\r\n\r\n😀🏽ß
字 ß<'M", "tokens": 39, "pieces": ["é", "Ⅳ0", "<|", "fim", "_prefix", "|>'", "fitd", "Atfi're", "İ", "🙂a", "/bfiſ", "㋿\r\n\r\n", "😀🏽", "ß", "
字", " <", "META", "_START", ">ß", "<'", "M"]} +{"text": "字 \ndſ, ­(½\n<|fim_prefix|>ABC\"\r 'VE३'re'TiOS/½\"'ll३,'(>", "tokens": 38, "pieces": ["字", " \n", "dſ", ",", " ", "­(", "½", "\n", "<|", "fim", "_prefix", "|>", "ABC", "\"\r", " '", "VE", "३", "'re'T", "i", "OS", "/", "½", "\"'", "ll", "३", ",'(>"]} +{"text": "\naZᵃ12345678'D9< \n aBᵃⅣiOSḍ̇", "tokens": 25, "pieces": ["\n", "a", "Zᵃ", "123", "456", "78", "'D", "9", "<", " \n", " a", "Bᵃ", "Ⅳ", "i", "OSḍ̇"]} +{"text": "Ab \n ́‍/\reß/\r\ne 'Re/½\r'TⅣEOT", "tokens": 21, "pieces": ["Ab", " \n", " ́", "‍/\r", "eß", "/\r\n", "e", " '", "Re", "/", "½", "\r", "'T", "Ⅳ", "EOT"]} +{"text": "ſ/ HTTPServerꟲ", "tokens": 8, "pieces": ["ſ", "/", " HTTPServerꟲ"]} +{"text": "ABCB'Re\n(ᵃ㋿\r\n 🙂/\r\n.!!'s'ſ́éiOSZ\n0", "tokens": 27, "pieces": ["ABCB'Re", "\n", "(ᵃ", "㋿\r\n", " ", " 🙂/\r\n", ".!!'", "s'ſ", "́éi", "OSZ", "\n", "0"]} diff --git a/litellm-rust/crates/token-counter/tests/token_counter.rs b/litellm-rust/crates/token-counter/tests/token_counter.rs index 0e71a5c03cf..12c59768952 100644 --- a/litellm-rust/crates/token-counter/tests/token_counter.rs +++ b/litellm-rust/crates/token-counter/tests/token_counter.rs @@ -1,4 +1,5 @@ use rstest::rstest; +use serde::Deserialize; use litellm_token_counter::{CountableRequest, Error, InputTokenCount, TokenCounter}; @@ -174,3 +175,147 @@ fn tool_choice_and_system_discount_change_the_count() { fn loading_a_bad_tokenizer_is_a_load_error() { assert!(matches!(TokenCounter::from_json("{}"), Err(Error::Load(_)))); } + +/// A tiktoken encoding: its fixture directory, the vendored rank file Python +/// loads, the constructor, and the model `generate.py` counted the requests for. +#[derive(Clone, Copy)] +struct TiktokenEncoding { + fixtures: &'static str, + rank_file: &'static str, + load: fn(&str) -> Result, + model: &'static str, +} + +const CL100K: TiktokenEncoding = TiktokenEncoding { + fixtures: "cl100k", + rank_file: "9b5ad71b2ce5302211f9c61530b329a4922fc6a4", + load: TokenCounter::from_cl100k_ranks, + model: "gpt-4", +}; + +const O200K: TiktokenEncoding = TiktokenEncoding { + fixtures: "o200k", + rank_file: "fb374d419588a4632f3f557e76b4b70aebbca790", + load: TokenCounter::from_o200k_ranks, + model: "gpt-4o", +}; + +fn tiktoken_counter(encoding: TiktokenEncoding) -> TokenCounter { + let path = format!( + "{}/../../../litellm/litellm_core_utils/tokenizers/{}", + env!("CARGO_MANIFEST_DIR"), + encoding.rank_file + ); + let ranks = std::fs::read_to_string(&path).expect("rank file is in the repo"); + (encoding.load)(&ranks).expect("ranks load") +} + +fn tiktoken_fixture(encoding: TiktokenEncoding, name: &str) -> String { + let path = format!( + "{}/tests/fixtures/{}/{name}", + env!("CARGO_MANIFEST_DIR"), + encoding.fixtures + ); + std::fs::read_to_string(&path).expect("fixture generated by tests/fixtures/generate.py") +} + +#[derive(Deserialize)] +struct TextFixture { + text: String, + tokens: usize, +} + +#[derive(Deserialize)] +struct RequestFixture { + body: String, + input_tokens: usize, +} + +/// Reference counts come from `tiktoken.get_encoding(name)`; see +/// `tests/fixtures/generate.py`. +#[rstest] +#[case::cl100k(CL100K)] +#[case::o200k(O200K)] +fn tiktoken_text_counts_match_tiktoken(#[case] encoding: TiktokenEncoding) { + let counter = tiktoken_counter(encoding); + let fixtures: Vec = tiktoken_fixture(encoding, "texts.jsonl") + .lines() + .map(|line| serde_json::from_str(line).expect("fixture line is json")) + .collect(); + assert!(fixtures.len() > 3000); + let mismatches: Vec<_> = fixtures + .iter() + .filter_map(|fixture| { + let count = counter.count_text(&fixture.text).expect("text counts"); + (count != fixture.tokens).then(|| (fixture.text.clone(), fixture.tokens, count)) + }) + .collect(); + assert!( + mismatches.is_empty(), + "(text, tiktoken, rust): {mismatches:?}" + ); +} + +/// Reference counts come from the proxy's admission counter +/// (`_count_input_tokens(body, model)`), so this pins the shared message, +/// tool and reply-priming accounting on the tiktoken paths as well. +#[rstest] +#[case::cl100k(CL100K)] +#[case::o200k(O200K)] +fn tiktoken_request_counts_match_python_admission_counter(#[case] encoding: TiktokenEncoding) { + let counter = tiktoken_counter(encoding); + let fixtures: Vec = tiktoken_fixture(encoding, "requests.jsonl") + .lines() + .map(|line| serde_json::from_str(line).expect("fixture line is json")) + .collect(); + let counts: Vec = fixtures + .iter() + .map(|fixture| { + let request = CountableRequest::parse(fixture.body.as_bytes()).expect("fixture parses"); + let count = counter.count_request(&request).expect("fixture counts"); + assert_eq!(count.model.as_deref(), Some(encoding.model)); + assert_eq!(count.input_tokens, fixture.input_tokens, "{}", fixture.body); + count.input_tokens + }) + .collect(); + assert!(counts.last().is_some_and(|tokens| *tokens >= 50_000)); +} + +#[rstest] +#[case::cl100k(CL100K)] +#[case::o200k(O200K)] +fn tiktoken_shares_the_message_accounting_with_the_anthropic_path( + #[case] encoding: TiktokenEncoding, +) { + let counter = tiktoken_counter(encoding); + let count = |body: &str| { + counter + .count_request(&CountableRequest::parse(body.as_bytes()).expect("parses")) + .expect("counts") + .input_tokens + }; + let text = |text: &str| counter.count_text(text).expect("counts"); + let base = count(r#"{"model":"m","messages":[{"role":"user","content":"hi"}]}"#); + assert_eq!(base, 3 + text("user") + text("hi") + 3); + assert_eq!( + count(r#"{"model":"m","messages":[{"role":"user","name":"al","content":"hi"}]}"#), + base + text("al") + 1 + ); + assert_eq!( + count(r#"{"model":"m","messages":[{"role":"user","content":"hi"}],"tool_choice":"none"}"#), + base + 1 + ); +} + +#[rstest] +#[case::empty("")] +#[case::not_base64("!!!! 0")] +#[case::missing_rank("YQ==")] +#[case::rank_not_a_number("YQ== x")] +#[case::single_byte_tokens_missing("YWI= 0")] +fn loading_a_bad_rank_file_is_a_load_error( + #[case] rank_file: &str, + #[values(CL100K, O200K)] encoding: TiktokenEncoding, +) { + assert!(matches!((encoding.load)(rank_file), Err(Error::Ranks(_)))); +} diff --git a/litellm/litellm_core_utils/default_encoding.py b/litellm/litellm_core_utils/default_encoding.py index c3b6a008411..71b30614d8d 100644 --- a/litellm/litellm_core_utils/default_encoding.py +++ b/litellm/litellm_core_utils/default_encoding.py @@ -1,4 +1,5 @@ import os +from pathlib import Path from typing import Final import litellm @@ -14,6 +15,20 @@ except (ImportError, AttributeError): filename = pkg_resources.resource_filename(__name__, "litellm_core_utils/tokenizers") +CL100K_BASE_RANK_FILE: Final = "9b5ad71b2ce5302211f9c61530b329a4922fc6a4" +O200K_BASE_RANK_FILE: Final = "fb374d419588a4632f3f557e76b4b70aebbca790" + + +def cl100k_base_rank_file() -> str: + """The vendored tiktoken `cl100k_base` rank file (`base64(token) rank` lines).""" + return Path(filename, CL100K_BASE_RANK_FILE).read_text(encoding="ascii") + + +def o200k_base_rank_file() -> str: + """The vendored tiktoken `o200k_base` rank file (`base64(token) rank` lines).""" + return Path(filename, O200K_BASE_RANK_FILE).read_text(encoding="ascii") + + # Always default TIKTOKEN_CACHE_DIR to the bundled tokenizers directory # unless the user explicitly overrides it via CUSTOM_TIKTOKEN_CACHE_DIR. # This keeps tiktoken fully offline-capable by default (see #1071). diff --git a/litellm/litellm_core_utils/token_counter.py b/litellm/litellm_core_utils/token_counter.py index 1de2533d514..3b128899f45 100644 --- a/litellm/litellm_core_utils/token_counter.py +++ b/litellm/litellm_core_utils/token_counter.py @@ -381,7 +381,7 @@ class _MessageCountParams: from litellm.utils import print_verbose actual_model: Final = _fix_model_name(model) - if actual_model == "gpt-3.5-turbo-0301": + if uses_legacy_message_accounting(model): self.tokens_per_message = 4 # every message follows <|start|>{role/name}\n{content}<|end|>\n self.tokens_per_name = -1 # if there's a name, the role is omitted elif actual_model in litellm.open_ai_chat_completion_models or actual_model in litellm.azure_llms: @@ -615,7 +615,7 @@ def _get_exact_count_function( ) -> TokenCounterFunction: """ Get the function to count tokens based on the model and custom tokenizer.""" - from litellm.utils import _select_tokenizer, print_verbose + from litellm.utils import _select_tokenizer if model is not None or custom_tokenizer is not None: tokenizer_json: Final = custom_tokenizer or _select_tokenizer(model) @@ -627,15 +627,7 @@ def _get_exact_count_function( return count_tokens elif tokenizer_json["type"] == "openai_tokenizer": - model_to_use: Final = _fix_model_name(model) - try: - if "gpt-4o" in model_to_use: - encoding = tiktoken.get_encoding("o200k_base") - else: - encoding = tiktoken.encoding_for_model(model_to_use) - except KeyError: - print_verbose("Warning: model not found. Using cl100k_base encoding.") - encoding = tiktoken.get_encoding("cl100k_base") + encoding: Final = openai_tokenizer_encoding(model) def encode_length(text: str) -> int: return len(encoding.encode(text, disallowed_special=())) @@ -651,6 +643,25 @@ def _get_exact_count_function( return _get_tiktoken_count_function(encode_length) +def openai_tokenizer_encoding(model: str) -> tiktoken.Encoding: + """The tiktoken encoding `token_counter` uses for a model on the `openai_tokenizer` path.""" + from litellm.utils import print_verbose + + model_to_use: Final = _fix_model_name(model) + if "gpt-4o" in model_to_use: + return tiktoken.get_encoding("o200k_base") + try: + return tiktoken.encoding_for_model(model_to_use) + except KeyError: + print_verbose("Warning: model not found. Using cl100k_base encoding.") + return tiktoken.get_encoding("cl100k_base") + + +def uses_legacy_message_accounting(model: str) -> bool: + """Whether `token_counter` prices messages with the `gpt-3.5-turbo-0301` constants (4 per message, -1 per name).""" + return _fix_model_name(model) == "gpt-3.5-turbo-0301" + + def _fix_model_name(model: str) -> str: """We normalize some model names to others""" if model in litellm.azure_llms: diff --git a/litellm/proxy/spend_tracking/budget_reservation.py b/litellm/proxy/spend_tracking/budget_reservation.py index ef21551ca93..eedec0619db 100644 --- a/litellm/proxy/spend_tracking/budget_reservation.py +++ b/litellm/proxy/spend_tracking/budget_reservation.py @@ -34,7 +34,7 @@ from litellm.proxy.common_utils.user_api_key_cache import ( ) from litellm.proxy.utils import PrismaClient, ProxyLogging from litellm.router import Router -from litellm.rust_bridge.token_counter import count_anthropic_input_tokens, uses_anthropic_tokenizer +from litellm.rust_bridge.token_counter import RustTokenizer, count_input_tokens, rust_tokenizer from litellm.types.proxy.model_access_group_budget import ModelAccessGroupBudget from litellm.types.router import DeploymentTypedDict @@ -1365,8 +1365,9 @@ async def count_request_input_tokens( Tokenizing is the reservation path's dominant CPU cost and is O(prompt), so counting a large prompt inline stalls every other request on the worker. - Models on the Anthropic tokenizer are counted from the raw body by the Rust - bridge when it is enabled, which parses and tokenizes with the GIL released. + Models whose tokenizer the Rust bridge ports (Anthropic, tiktoken cl100k_base + and o200k_base) are counted from the raw body by the bridge when it is enabled, once per + distinct tokenizer, which parses and tokenizes with the GIL released. Everything it declines is counted in Python, large prompts in a worker thread. The counts are reused by both the max-cost and the input-cost estimate. @@ -1374,23 +1375,31 @@ async def count_request_input_tokens( models: Final = _get_request_models(request_body=request_body, route=route, llm_router=llm_router) if not models: return MappingProxyType({}) - rust_count: Final = ( - await count_anthropic_input_tokens(raw_body) - if raw_body is not None and any(uses_anthropic_tokenizer(model) for model in models) - else None + tokenizers: Final[Mapping[str, RustTokenizer | None]] = MappingProxyType( + {model: rust_tokenizer(model) for model in models} + ) + distinct_tokenizers: Final[tuple[RustTokenizer, ...]] = tuple( + dict.fromkeys(tokenizer for tokenizer in tokenizers.values() if tokenizer is not None) + ) + rust_counts_by_tokenizer: Final[Mapping[RustTokenizer, int]] = MappingProxyType( + { + tokenizer: count.input_tokens + for tokenizer in distinct_tokenizers + if raw_body is not None and (count := await count_input_tokens(raw_body, tokenizer)) is not None + } ) rust_counts: Final = MappingProxyType( { - model: rust_count.input_tokens - for model in models - if rust_count is not None and uses_anthropic_tokenizer(model) + model: rust_counts_by_tokenizer[tokenizer] + for model, tokenizer in tokenizers.items() + if tokenizer is not None and tokenizer in rust_counts_by_tokenizer } ) python_models: Final = tuple(model for model in models if model not in rust_counts) - if not python_models: - return rust_counts python_counts: Final = ( - _count_input_tokens_for_models(request_body=request_body, models=python_models) + MappingProxyType({}) + if not python_models + else _count_input_tokens_for_models(request_body=request_body, models=python_models) if _approximate_input_size(request_body) < TOKENIZE_OFF_EVENT_LOOP_MIN_CHARS else await asyncio.to_thread( _count_input_tokens_for_models, @@ -1398,6 +1407,7 @@ async def count_request_input_tokens( models=python_models, ) ) + verbose_proxy_logger.debug("input token counts: rust=%s python=%s", dict(rust_counts), dict(python_counts)) return MappingProxyType({**rust_counts, **python_counts}) diff --git a/litellm/rust_bridge/token_counter.py b/litellm/rust_bridge/token_counter.py index ee755b317d8..d36234f56c1 100644 --- a/litellm/rust_bridge/token_counter.py +++ b/litellm/rust_bridge/token_counter.py @@ -5,17 +5,20 @@ from __future__ import annotations from collections.abc import Awaitable from dataclasses import dataclass from functools import lru_cache -from typing import Final, Protocol, cast # noqa: TID251 # native extension exposes dynamically typed callables +from typing import Final, Literal, Protocol, cast # noqa: TID251 # native extension exposes untyped callables from pydantic import TypeAdapter import litellm from litellm._logging import verbose_logger +from litellm.litellm_core_utils.default_encoding import cl100k_base_rank_file, o200k_base_rank_file +from litellm.litellm_core_utils.token_counter import openai_tokenizer_encoding, uses_legacy_message_accounting from litellm.rust_bridge.bindings import NativeBinding from litellm.rust_bridge.configuration import rust_enabled from litellm.rust_bridge.runtime import BridgeErrorContext, RustHandled, aattempt -from litellm.utils import claude_json_str -from litellm.utils import uses_anthropic_tokenizer as _python_uses_anthropic_tokenizer +from litellm.utils import claude_json_str, huggingface_tokenizer_kind + +RustTokenizer = Literal["anthropic", "cl100k_base", "o200k_base"] class RustTokenCounter(Protocol): @@ -27,6 +30,12 @@ class RustTokenCounterFactory(Protocol): def __call__(self, tokenizer_json: str) -> RustTokenCounter: raise NotImplementedError + def from_cl100k_ranks(self, rank_file: str) -> RustTokenCounter: + raise NotImplementedError + + def from_o200k_ranks(self, rank_file: str) -> RustTokenCounter: + raise NotImplementedError + @dataclass(frozen=True, slots=True) class InputTokenCount: @@ -50,18 +59,41 @@ def _as_factory(value: object) -> RustTokenCounterFactory | None: TOKEN_COUNTER: Final = NativeBinding("TokenCounter", validate=_as_factory) -def uses_anthropic_tokenizer(model: str) -> bool: - if litellm.disable_token_counter is True or litellm.disable_hf_tokenizer_download is True: - return False - return _python_uses_anthropic_tokenizer(model) +def rust_tokenizer(model: str) -> RustTokenizer | None: + """The Rust counter for the tokenizer `litellm.token_counter` selects for `model`, `None` when Python must count. + + Mirrors `_select_tokenizer_helper`: the Anthropic tokenizer has a Rust port, the other HuggingFace + downloads do not, and of the tiktoken encodings `cl100k_base` and `o200k_base` do (p50k/r50k do not). Rust + prices every message with the default constants, so the legacy `gpt-3.5-turbo-0301` accounting stays in + Python.""" + if litellm.disable_token_counter is True: + return None + kind: Final = None if litellm.disable_hf_tokenizer_download is True else huggingface_tokenizer_kind(model) + if kind == "anthropic": + return "anthropic" + if kind is not None or uses_legacy_message_accounting(model): + return None + match openai_tokenizer_encoding(model).name: + case "cl100k_base": + return "cl100k_base" + case "o200k_base": + return "o200k_base" + case _: + return None @lru_cache(maxsize=4) -def _anthropic_counter(factory: RustTokenCounterFactory) -> RustTokenCounter: - return factory(claude_json_str) +def _counter(factory: RustTokenCounterFactory, tokenizer: RustTokenizer) -> RustTokenCounter: + match tokenizer: + case "anthropic": + return factory(claude_json_str) + case "cl100k_base": + return factory.from_cl100k_ranks(cl100k_base_rank_file()) + case "o200k_base": + return factory.from_o200k_ranks(o200k_base_rank_file()) -async def count_anthropic_input_tokens(body: bytes) -> InputTokenCount | None: +async def count_input_tokens(body: bytes, tokenizer: RustTokenizer) -> InputTokenCount | None: if not rust_enabled(): return None factory: Final = TOKEN_COUNTER.load() @@ -69,11 +101,14 @@ async def count_anthropic_input_tokens(body: bytes) -> InputTokenCount | None: return None try: attempt: Final = await aattempt( - native_call=lambda: _anthropic_counter(factory).acount_request(body), + native_call=lambda: _counter(factory, tokenizer).acount_request(body), adapt=_INPUT_TOKEN_COUNT.validate_python, - context=BridgeErrorContext(route="token_counter", provider="anthropic", model=""), + context=BridgeErrorContext(route="token_counter", provider=tokenizer, model=""), ) except (RuntimeError, ValueError) as error: - verbose_logger.debug("Rust token counter failed, counting in Python: %s", error) + verbose_logger.debug("Rust token counter (%s) failed, counting in Python: %s", tokenizer, error) return None - return attempt.value if isinstance(attempt, RustHandled) else None + if not isinstance(attempt, RustHandled): + return None + verbose_logger.debug("Rust token counter (%s) counted %d input tokens", tokenizer, attempt.value.input_tokens) + return attempt.value diff --git a/litellm/utils.py b/litellm/utils.py index bb1bce66d9b..1a77655a5a4 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2227,25 +2227,39 @@ def uses_anthropic_tokenizer(model: str) -> bool: return model in litellm.anthropic_models and "claude-3" not in model -def _return_huggingface_tokenizer(model: str) -> SelectTokenizerResponse | None: +HuggingFaceTokenizerKind = Literal["cohere", "anthropic", "llama2", "llama3"] + + +def huggingface_tokenizer_kind(model: str) -> HuggingFaceTokenizerKind | None: + """Which HuggingFace tokenizer `token_counter` selects for a model; `None` means tiktoken.""" if model in litellm.cohere_models and "command-r" in model: - # cohere - cohere_tokenizer: Final = Tokenizer.from_pretrained("Xenova/c4ai-command-r-v01-tokenizer") - return {"type": "huggingface_tokenizer", "tokenizer": cohere_tokenizer} - # anthropic - elif uses_anthropic_tokenizer(model): - claude_tokenizer: Final = Tokenizer.from_str(claude_json_str) - return {"type": "huggingface_tokenizer", "tokenizer": claude_tokenizer} - # llama2 - elif "llama-2" in model.lower() or "replicate" in model.lower(): - tokenizer = Tokenizer.from_pretrained("hf-internal-testing/llama-tokenizer") - return {"type": "huggingface_tokenizer", "tokenizer": tokenizer} - # llama3 - elif "llama-3" in model.lower(): - tokenizer = Tokenizer.from_pretrained("Xenova/llama-3-tokenizer") - return {"type": "huggingface_tokenizer", "tokenizer": tokenizer} - else: + return "cohere" + if uses_anthropic_tokenizer(model): + return "anthropic" + if "llama-2" in model.lower() or "replicate" in model.lower(): + return "llama2" + if "llama-3" in model.lower(): + return "llama3" + return None + + +def _return_huggingface_tokenizer(model: str) -> SelectTokenizerResponse | None: + kind: Final = huggingface_tokenizer_kind(model) + if kind is None: return None + return {"type": "huggingface_tokenizer", "tokenizer": _load_huggingface_tokenizer(kind)} + + +def _load_huggingface_tokenizer(kind: HuggingFaceTokenizerKind) -> Tokenizer: + match kind: + case "cohere": + return Tokenizer.from_pretrained("Xenova/c4ai-command-r-v01-tokenizer") + case "anthropic": + return Tokenizer.from_str(claude_json_str) + case "llama2": + return Tokenizer.from_pretrained("hf-internal-testing/llama-tokenizer") + case "llama3": + return Tokenizer.from_pretrained("Xenova/llama-3-tokenizer") def encode(model="", text="", custom_tokenizer: dict | None = None): diff --git a/tests/test_litellm/proxy/spend_tracking/test_budget_reservation.py b/tests/test_litellm/proxy/spend_tracking/test_budget_reservation.py index ee062b14b96..3e0acf917aa 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_budget_reservation.py +++ b/tests/test_litellm/proxy/spend_tracking/test_budget_reservation.py @@ -1,5 +1,8 @@ +from __future__ import annotations + import json import math +from types import MappingProxyType from typing import Final import pytest @@ -207,8 +210,13 @@ def test_deployment_pricing_update_invalidates_cached_estimate() -> None: ANTHROPIC_TOKENIZER_MODEL: Final = "claude-sonnet-4-5-20250929" +CL100K_MODEL: Final = "gpt-4" +O200K_MODEL: Final = "gpt-4o" RUST_COUNTED_BODY: Final = {"model": ANTHROPIC_TOKENIZER_MODEL, "max_tokens": 16, "messages": ANTHROPIC_MESSAGES} RUST_INPUT_TOKENS: Final = 4_321 +RUST_INPUT_TOKENS_BY_TOKENIZER: Final = MappingProxyType( + {"anthropic": RUST_INPUT_TOKENS, "cl100k_base": 1_234, "o200k_base": 2_345} +) class _FakeDeclined(Exception): @@ -225,33 +233,57 @@ class _FakeNative: class _RecordingCounter: - bodies: Final[list[bytes]] = [] + """Stands in for one native counter; records `(tokenizer, body)` on the shared factory.""" - def __init__(self, tokenizer_json: str) -> None: - pass + def __init__(self, factory: _RecordingFactory, tokenizer: rust_token_counter.RustTokenizer) -> None: + self.factory = factory + self.tokenizer = tokenizer async def acount_request(self, body: bytes) -> object: - self.bodies.append(body) - return {"model": ANTHROPIC_TOKENIZER_MODEL, "input_tokens": RUST_INPUT_TOKENS} + self.factory.calls.append((self.tokenizer, body)) + return {"model": "", "input_tokens": RUST_INPUT_TOKENS_BY_TOKENIZER[self.tokenizer]} + + +class _RecordingFactory: + """Stands in for the native `TokenCounter` class: called with tokenizer JSON, or `from_*_ranks`.""" + + def __init__(self) -> None: + self.calls: list[tuple[rust_token_counter.RustTokenizer, bytes]] = [] + + def __call__(self, tokenizer_json: str) -> _RecordingCounter: + return _RecordingCounter(self, "anthropic") + + def from_cl100k_ranks(self, rank_file: str) -> _RecordingCounter: + return _RecordingCounter(self, "cl100k_base") + + def from_o200k_ranks(self, rank_file: str) -> _RecordingCounter: + return _RecordingCounter(self, "o200k_base") class _DecliningCounter: - def __init__(self, tokenizer_json: str) -> None: - pass - async def acount_request(self, body: bytes) -> object: raise _FakeDeclined("unsupported content block") +class _DecliningFactory: + def __call__(self, tokenizer_json: str) -> _DecliningCounter: + return _DecliningCounter() + + def from_cl100k_ranks(self, rank_file: str) -> _DecliningCounter: + return _DecliningCounter() + + def from_o200k_ranks(self, rank_file: str) -> _DecliningCounter: + return _DecliningCounter() + + @pytest.fixture def rust_counter(monkeypatch: pytest.MonkeyPatch): monkeypatch.setattr(bindings, "get_native_bridge", lambda: _FakeNative()) - rust_token_counter._anthropic_counter.cache_clear() + rust_token_counter._counter.cache_clear() configuration.reset_rust_configuration() - _RecordingCounter.bodies.clear() yield rust_token_counter.TOKEN_COUNTER.reset() - rust_token_counter._anthropic_counter.cache_clear() + rust_token_counter._counter.cache_clear() configuration.reset_rust_configuration() @@ -270,8 +302,9 @@ def rust_counter(monkeypatch: pytest.MonkeyPatch): async def test_rust_count_replaces_python_tokenizing_on_every_llm_route( rust_counter: None, route: str, request_body: dict ) -> None: + factory: Final = _RecordingFactory() litellm.rust(True) - rust_token_counter.TOKEN_COUNTER.override(_RecordingCounter) + rust_token_counter.TOKEN_COUNTER.override(factory) raw_body: Final = json.dumps(request_body).encode() counts: Final = await count_request_input_tokens( @@ -279,53 +312,136 @@ async def test_rust_count_replaces_python_tokenizing_on_every_llm_route( ) assert dict(counts) == {ANTHROPIC_TOKENIZER_MODEL: RUST_INPUT_TOKENS} - assert _RecordingCounter.bodies == [raw_body] + assert factory.calls == [("anthropic", raw_body)] @pytest.mark.asyncio -async def test_rust_decline_falls_back_to_python_count(rust_counter: None) -> None: +@pytest.mark.parametrize("model", (CL100K_MODEL, "azure/gpt-35-turbo", "gemini/gemini-2.5-pro", "my-router-alias")) +async def test_tiktoken_cl100k_models_are_counted_by_rust(rust_counter: None, model: str) -> None: + factory: Final = _RecordingFactory() litellm.rust(True) - rust_token_counter.TOKEN_COUNTER.override(_DecliningCounter) + rust_token_counter.TOKEN_COUNTER.override(factory) + body: Final = {"model": model, "messages": ANTHROPIC_MESSAGES} + raw_body: Final = json.dumps(body).encode() + + counts: Final = await count_request_input_tokens( + request_body=body, route="/v1/chat/completions", llm_router=None, raw_body=raw_body + ) + + assert dict(counts) == {model: RUST_INPUT_TOKENS_BY_TOKENIZER["cl100k_base"]} + assert factory.calls == [("cl100k_base", raw_body)] + + +@pytest.mark.asyncio +@pytest.mark.parametrize("model", (O200K_MODEL, "gpt-5", "o3", "gpt-4.1", "chatgpt-4o-latest")) +async def test_tiktoken_o200k_models_are_counted_by_rust(rust_counter: None, model: str) -> None: + factory: Final = _RecordingFactory() + litellm.rust(True) + rust_token_counter.TOKEN_COUNTER.override(factory) + body: Final = {"model": model, "messages": ANTHROPIC_MESSAGES} + raw_body: Final = json.dumps(body).encode() + + counts: Final = await count_request_input_tokens( + request_body=body, route="/v1/chat/completions", llm_router=None, raw_body=raw_body + ) + + assert dict(counts) == {model: RUST_INPUT_TOKENS_BY_TOKENIZER["o200k_base"]} + assert factory.calls == [("o200k_base", raw_body)] + + +@pytest.mark.asyncio +async def test_multi_model_request_counts_once_per_tokenizer_and_python_for_the_rest(rust_counter: None) -> None: + factory: Final = _RecordingFactory() + litellm.rust(True) + rust_token_counter.TOKEN_COUNTER.override(factory) + models: Final = ( + CL100K_MODEL, + ANTHROPIC_TOKENIZER_MODEL, + "gemini/gemini-2.5-pro", + O200K_MODEL, + "gpt-5", + "replicate/meta/llama-2-70b-chat", + ) + body: Final = {"model": list(models), "messages": ANTHROPIC_MESSAGES} + raw_body: Final = json.dumps(body).encode() python_counts: Final = await count_request_input_tokens( - request_body=RUST_COUNTED_BODY, route="/v1/messages", llm_router=None + request_body=body, route="/v1/chat/completions", llm_router=None ) counts: Final = await count_request_input_tokens( - request_body=RUST_COUNTED_BODY, + request_body=body, route="/v1/chat/completions", llm_router=None, raw_body=raw_body + ) + + assert factory.calls == [("cl100k_base", raw_body), ("anthropic", raw_body), ("o200k_base", raw_body)] + assert dict(counts) == { + CL100K_MODEL: RUST_INPUT_TOKENS_BY_TOKENIZER["cl100k_base"], + "gemini/gemini-2.5-pro": RUST_INPUT_TOKENS_BY_TOKENIZER["cl100k_base"], + ANTHROPIC_TOKENIZER_MODEL: RUST_INPUT_TOKENS, + O200K_MODEL: RUST_INPUT_TOKENS_BY_TOKENIZER["o200k_base"], + "gpt-5": RUST_INPUT_TOKENS_BY_TOKENIZER["o200k_base"], + "replicate/meta/llama-2-70b-chat": python_counts["replicate/meta/llama-2-70b-chat"], + } + assert counts["replicate/meta/llama-2-70b-chat"] not in RUST_INPUT_TOKENS_BY_TOKENIZER.values() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("model", (ANTHROPIC_TOKENIZER_MODEL, CL100K_MODEL, O200K_MODEL)) +async def test_rust_decline_falls_back_to_python_count(rust_counter: None, model: str) -> None: + litellm.rust(True) + rust_token_counter.TOKEN_COUNTER.override(_DecliningFactory()) + body: Final = {**RUST_COUNTED_BODY, "model": model} + python_counts: Final = await count_request_input_tokens(request_body=body, route="/v1/messages", llm_router=None) + + counts: Final = await count_request_input_tokens( + request_body=body, route="/v1/messages", llm_router=None, - raw_body=json.dumps(RUST_COUNTED_BODY).encode(), + raw_body=json.dumps(body).encode(), ) assert dict(counts) == dict(python_counts) - assert counts[ANTHROPIC_TOKENIZER_MODEL] != RUST_INPUT_TOKENS + assert counts[model] not in RUST_INPUT_TOKENS_BY_TOKENIZER.values() @pytest.mark.asyncio async def test_disabled_rust_never_sees_the_raw_body(rust_counter: None) -> None: + factory: Final = _RecordingFactory() litellm.rust(False) - rust_token_counter.TOKEN_COUNTER.override(_RecordingCounter) + rust_token_counter.TOKEN_COUNTER.override(factory) + body: Final = {"model": [ANTHROPIC_TOKENIZER_MODEL, CL100K_MODEL, O200K_MODEL], "messages": ANTHROPIC_MESSAGES} counts: Final = await count_request_input_tokens( - request_body=RUST_COUNTED_BODY, - route="/v1/messages", + request_body=body, + route="/v1/chat/completions", llm_router=None, - raw_body=json.dumps(RUST_COUNTED_BODY).encode(), + raw_body=json.dumps(body).encode(), ) - assert _RecordingCounter.bodies == [] - assert counts[ANTHROPIC_TOKENIZER_MODEL] != RUST_INPUT_TOKENS + assert factory.calls == [] + assert set(counts) == {ANTHROPIC_TOKENIZER_MODEL, CL100K_MODEL, O200K_MODEL} + assert not set(counts.values()) & set(RUST_INPUT_TOKENS_BY_TOKENIZER.values()) @pytest.mark.asyncio -async def test_non_anthropic_tokenizer_models_stay_in_python(rust_counter: None) -> None: +@pytest.mark.parametrize("model", ("replicate/meta/llama-2-70b-chat", "meta-llama/Llama-3-8b", "text-davinci-003")) +async def test_models_without_a_rust_tokenizer_stay_in_python( + rust_counter: None, monkeypatch: pytest.MonkeyPatch, model: str +) -> None: + monkeypatch.setattr( + litellm, "open_ai_chat_completion_models", litellm.open_ai_chat_completion_models | {"text-davinci-003"} + ) + factory: Final = _RecordingFactory() litellm.rust(True) - rust_token_counter.TOKEN_COUNTER.override(_RecordingCounter) - body: Final = {"model": "gpt-4o", "messages": ANTHROPIC_MESSAGES} + rust_token_counter.TOKEN_COUNTER.override(factory) + body: Final = {"model": model, "messages": ANTHROPIC_MESSAGES} + python_counts: Final = await count_request_input_tokens( + request_body=body, route="/v1/chat/completions", llm_router=None + ) counts: Final = await count_request_input_tokens( request_body=body, route="/v1/chat/completions", llm_router=None, raw_body=json.dumps(body).encode() ) - assert _RecordingCounter.bodies == [] - assert counts["gpt-4o"] != RUST_INPUT_TOKENS + assert factory.calls == [] + assert dict(counts) == dict(python_counts) + assert counts[model] not in RUST_INPUT_TOKENS_BY_TOKENIZER.values() diff --git a/tests/test_litellm/rust_bridge/test_token_counter.py b/tests/test_litellm/rust_bridge/test_token_counter.py index 2df91204390..71aa79cc4bb 100644 --- a/tests/test_litellm/rust_bridge/test_token_counter.py +++ b/tests/test_litellm/rust_bridge/test_token_counter.py @@ -8,16 +8,26 @@ cases need the extension and are skipped when it is not built. from __future__ import annotations import json +from types import MappingProxyType from typing import Final import pytest +import tiktoken +from tokenizers import Tokenizer import litellm +from litellm.constants import TIKTOKEN_ENCODE_CHUNK_SIZE_CHARS +from litellm.litellm_core_utils.token_counter import openai_tokenizer_encoding from litellm.proxy.spend_tracking.budget_reservation import _count_input_tokens from litellm.rust_bridge import bindings, configuration from litellm.rust_bridge import token_counter as bridge +from litellm.utils import claude_json_str MODEL: Final = "claude-sonnet-4-5-20250929" +CL100K_MODEL: Final = "gpt-4" +O200K_MODEL: Final = "gpt-4o" +TOKENIZERS: Final[tuple[bridge.RustTokenizer, ...]] = ("anthropic", "cl100k_base", "o200k_base") +RANK_FILE_LINES: Final = MappingProxyType({"cl100k_base": 100_256, "o200k_base": 199_998}) BODY: Final = json.dumps({"model": MODEL, "messages": [{"role": "user", "content": "hello"}]}).encode() @@ -44,69 +54,122 @@ class _RecordingCounter: return {"model": MODEL, "input_tokens": 42} -class _DecliningCounter: - def __init__(self, tokenizer_json: str) -> None: - pass +class _RecordingFactory: + """Stands in for the native `TokenCounter` class: callable for tokenizer JSON, `from_*_ranks` for rank files.""" + + def __init__(self) -> None: + self.counters: list[_RecordingCounter] = [] + self.rank_files: list[str] = [] + + def __call__(self, tokenizer_json: str) -> _RecordingCounter: + counter = _RecordingCounter(tokenizer_json) + self.counters.append(counter) + return counter + + def from_cl100k_ranks(self, rank_file: str) -> _RecordingCounter: + self.rank_files.append(rank_file) + return self("cl100k_base") + + def from_o200k_ranks(self, rank_file: str) -> _RecordingCounter: + self.rank_files.append(rank_file) + return self("o200k_base") + + +class _RaisingCounter: + def __init__(self, error: Exception) -> None: + self.error = error async def acount_request(self, body: bytes) -> object: - raise _FakeDeclined("request has no messages") + raise self.error -class _FailingCounter: - def __init__(self, tokenizer_json: str) -> None: - pass +class _RaisingFactory: + """Every counter it builds, for either tokenizer, raises `error` on count.""" - async def acount_request(self, body: bytes) -> object: - raise RuntimeError("encode failed") + def __init__(self, error: Exception) -> None: + self.error = error + + def __call__(self, tokenizer_json: str) -> _RaisingCounter: + return _RaisingCounter(self.error) + + def from_cl100k_ranks(self, rank_file: str) -> _RaisingCounter: + return _RaisingCounter(self.error) + + def from_o200k_ranks(self, rank_file: str) -> _RaisingCounter: + return _RaisingCounter(self.error) @pytest.fixture(autouse=True) def _reset_bridge(monkeypatch: pytest.MonkeyPatch): bridge.TOKEN_COUNTER.reset() - bridge._anthropic_counter.cache_clear() + bridge._counter.cache_clear() configuration.reset_rust_configuration() monkeypatch.setattr(bindings, "get_native_bridge", lambda: _FakeNative()) yield bridge.TOKEN_COUNTER.reset() - bridge._anthropic_counter.cache_clear() + bridge._counter.cache_clear() configuration.reset_rust_configuration() @pytest.mark.asyncio -async def test_disabled_bridge_never_constructs_a_counter() -> None: - constructed: list[str] = [] - - def factory(tokenizer_json: str) -> _RecordingCounter: - constructed.append(tokenizer_json) - return _RecordingCounter(tokenizer_json) - +@pytest.mark.parametrize("tokenizer", TOKENIZERS) +async def test_disabled_bridge_never_constructs_a_counter(tokenizer: bridge.RustTokenizer) -> None: + factory: Final = _RecordingFactory() litellm.rust(False) bridge.TOKEN_COUNTER.override(factory) - assert await bridge.count_anthropic_input_tokens(BODY) is None - assert constructed == [] + assert await bridge.count_input_tokens(BODY, tokenizer) is None + assert factory.counters == [] @pytest.mark.asyncio async def test_enabled_bridge_returns_typed_count_and_reuses_one_counter() -> None: - counters: list[_RecordingCounter] = [] - - def factory(tokenizer_json: str) -> _RecordingCounter: - counter = _RecordingCounter(tokenizer_json) - counters.append(counter) - return counter - + factory: Final = _RecordingFactory() litellm.rust(True) bridge.TOKEN_COUNTER.override(factory) - first: Final = await bridge.count_anthropic_input_tokens(BODY) - second: Final = await bridge.count_anthropic_input_tokens(BODY) + first: Final = await bridge.count_input_tokens(BODY, "anthropic") + second: Final = await bridge.count_input_tokens(BODY, "anthropic") assert first == bridge.InputTokenCount(model=MODEL, input_tokens=42) assert second == first - assert len(counters) == 1 - assert counters[0].bodies == [BODY, BODY] - assert json.loads(counters[0].tokenizer_json)["model"]["type"] == "BPE" + assert len(factory.counters) == 1 + assert factory.counters[0].bodies == [BODY, BODY] + assert json.loads(factory.counters[0].tokenizer_json)["model"]["type"] == "BPE" + + +@pytest.mark.asyncio +@pytest.mark.parametrize("tokenizer", ("cl100k_base", "o200k_base")) +async def test_tiktoken_counter_is_built_from_the_vendored_rank_file_once(tokenizer: bridge.RustTokenizer) -> None: + factory: Final = _RecordingFactory() + litellm.rust(True) + bridge.TOKEN_COUNTER.override(factory) + + first: Final = await bridge.count_input_tokens(BODY, tokenizer) + second: Final = await bridge.count_input_tokens(BODY, tokenizer) + + assert first == second == bridge.InputTokenCount(model=MODEL, input_tokens=42) + assert len(factory.rank_files) == 1 + assert factory.rank_files[0].startswith("IQ== 0\n") + assert factory.rank_files[0].count("\n") == RANK_FILE_LINES[tokenizer] + assert factory.counters[0].tokenizer_json == tokenizer + assert factory.counters[0].bodies == [BODY, BODY] + + +@pytest.mark.asyncio +async def test_each_tokenizer_gets_its_own_cached_counter() -> None: + factory: Final = _RecordingFactory() + litellm.rust(True) + bridge.TOKEN_COUNTER.override(factory) + + await bridge.count_input_tokens(BODY, "anthropic") + await bridge.count_input_tokens(BODY, "cl100k_base") + await bridge.count_input_tokens(BODY, "o200k_base") + await bridge.count_input_tokens(BODY, "anthropic") + await bridge.count_input_tokens(BODY, "o200k_base") + + assert [counter.tokenizer_json for counter in factory.counters][1:] == ["cl100k_base", "o200k_base"] + assert [len(counter.bodies) for counter in factory.counters] == [2, 1, 2] @pytest.mark.asyncio @@ -114,38 +177,131 @@ async def test_missing_native_module_falls_back(monkeypatch: pytest.MonkeyPatch) litellm.rust(True) monkeypatch.setattr(bindings, "get_native_bridge", lambda: None) - assert await bridge.count_anthropic_input_tokens(BODY) is None + assert [await bridge.count_input_tokens(BODY, tokenizer) for tokenizer in TOKENIZERS] == [None, None, None] @pytest.mark.asyncio -async def test_declined_request_falls_back() -> None: +@pytest.mark.parametrize("tokenizer", TOKENIZERS) +async def test_declined_request_falls_back(tokenizer: bridge.RustTokenizer) -> None: litellm.rust(True) - bridge.TOKEN_COUNTER.override(_DecliningCounter) + bridge.TOKEN_COUNTER.override(_RaisingFactory(_FakeDeclined("request has no messages"))) - assert await bridge.count_anthropic_input_tokens(BODY) is None + assert await bridge.count_input_tokens(BODY, tokenizer) is None @pytest.mark.asyncio -async def test_runtime_failure_falls_back() -> None: +@pytest.mark.parametrize("tokenizer", TOKENIZERS) +async def test_runtime_failure_falls_back(tokenizer: bridge.RustTokenizer) -> None: litellm.rust(True) - bridge.TOKEN_COUNTER.override(_FailingCounter) + bridge.TOKEN_COUNTER.override(_RaisingFactory(RuntimeError("encode failed"))) - assert await bridge.count_anthropic_input_tokens(BODY) is None + assert await bridge.count_input_tokens(BODY, tokenizer) is None @pytest.mark.parametrize( ("model", "expected"), - ((MODEL, True), ("claude-3-5-sonnet-20241022", False), ("gpt-4o", False), ("my-router-alias", False)), + ( + (MODEL, "anthropic"), + ("claude-3-5-sonnet-20241022", "cl100k_base"), + ("gpt-4", "cl100k_base"), + ("gpt-4-turbo", "cl100k_base"), + ("gpt-3.5-turbo", "cl100k_base"), + ("azure/gpt-35-turbo", "cl100k_base"), + ("gemini/gemini-2.5-pro", "cl100k_base"), + ("mistral/mistral-large-latest", "cl100k_base"), + ("my-router-alias", "cl100k_base"), + ("azure/gpt-4o", "cl100k_base"), + ("command-r-plus", "cl100k_base"), + ("gpt-4o", "o200k_base"), + ("gpt-4o-mini", "o200k_base"), + ("gpt-4o-2024-08-06", "o200k_base"), + ("chatgpt-4o-latest", "o200k_base"), + ("gpt-4.1", "o200k_base"), + ("gpt-5", "o200k_base"), + ("gpt-5-mini", "o200k_base"), + ("o1", "o200k_base"), + ("o3", "o200k_base"), + ("o3-mini", "o200k_base"), + ("o4-mini", "o200k_base"), + ("replicate/meta/llama-2-70b-chat", None), + ("meta-llama/Llama-3-8b", None), + ), ) -def test_uses_anthropic_tokenizer_mirrors_python_tokenizer_selection(model: str, expected: bool) -> None: - assert bridge.uses_anthropic_tokenizer(model) is expected +def test_rust_tokenizer_mirrors_python_tokenizer_selection(model: str, expected: bridge.RustTokenizer | None) -> None: + assert bridge.rust_tokenizer(model) == expected -@pytest.mark.parametrize("flag", ("disable_hf_tokenizer_download", "disable_token_counter")) -def test_uses_anthropic_tokenizer_respects_python_opt_outs(monkeypatch: pytest.MonkeyPatch, flag: str) -> None: - monkeypatch.setattr(litellm, flag, True) +@pytest.mark.parametrize( + ("model", "python_encoding"), + (("text-davinci-003", "p50k_base"), ("gpt-oss-120b", "o200k_harmony")), +) +def test_rust_tokenizer_declines_tiktoken_encodings_rust_does_not_have( + monkeypatch: pytest.MonkeyPatch, model: str, python_encoding: str +) -> None: + monkeypatch.setattr(litellm, "open_ai_chat_completion_models", litellm.open_ai_chat_completion_models | {model}) - assert bridge.uses_anthropic_tokenizer(MODEL) is False + assert openai_tokenizer_encoding(model).name == python_encoding + assert bridge.rust_tokenizer(model) is None + + +def test_rust_tokenizer_declines_the_cohere_tokenizer_download(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "cohere_models", litellm.cohere_models | {"command-r-plus"}) + + assert bridge.rust_tokenizer("command-r-plus") is None + + +@pytest.mark.parametrize("legacy_model", ("gpt-3.5-turbo-0301", "gpt-35-turbo-0301")) +def test_rust_tokenizer_declines_legacy_message_accounting_python_prices_differently( + monkeypatch: pytest.MonkeyPatch, legacy_model: str +) -> None: + monkeypatch.setattr( + litellm, "open_ai_chat_completion_models", litellm.open_ai_chat_completion_models | {"gpt-3.5-turbo-0301"} + ) + monkeypatch.setattr(litellm, "azure_llms", {**litellm.azure_llms, "gpt-35-turbo-0301": "azure"}) + messages: Final = [{"role": "user", "name": "bob", "content": "hello there"}] + + assert litellm.token_counter(model=legacy_model, messages=messages) != litellm.token_counter( + model=CL100K_MODEL, messages=messages + ) + assert bridge.rust_tokenizer(legacy_model) is None + assert bridge.rust_tokenizer(CL100K_MODEL) == "cl100k_base" + + +@pytest.mark.parametrize("model", (MODEL, CL100K_MODEL, O200K_MODEL, "gpt-5", "o3")) +def test_rust_tokenizer_names_the_encoding_python_actually_counts_with(model: str) -> None: + text: Final = ( + "Hello, world! camelCase ABCdef \u00e9\u00e8 12345 \u3053\u3093\u306b\u3061\u306f <|endoftext|>\r\n" * 9 + ) + python_count: Final = litellm.token_counter(model=model, text=text) + cl100k_count: Final = len(tiktoken.get_encoding("cl100k_base").encode(text, disallowed_special=())) + o200k_count: Final = len(tiktoken.get_encoding("o200k_base").encode(text, disallowed_special=())) + assert cl100k_count != o200k_count + match bridge.rust_tokenizer(model): + case "cl100k_base": + assert python_count == cl100k_count + case "o200k_base": + assert python_count == o200k_count + case "anthropic": + assert python_count == len(Tokenizer.from_str(claude_json_str).encode(text).ids) + assert python_count not in {cl100k_count, o200k_count} + case None: + pytest.fail(f"{model} must have a Rust tokenizer") + + +def test_disabled_hf_download_routes_anthropic_models_to_cl100k_like_python(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "disable_hf_tokenizer_download", True) + + assert bridge.rust_tokenizer(MODEL) == "cl100k_base" + assert bridge.rust_tokenizer("meta-llama/Llama-3-8b") == "cl100k_base" + assert bridge.rust_tokenizer(O200K_MODEL) == "o200k_base" + + +def test_disabled_token_counter_declines_every_model(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "disable_token_counter", True) + + assert bridge.rust_tokenizer(MODEL) is None + assert bridge.rust_tokenizer(CL100K_MODEL) is None + assert bridge.rust_tokenizer(O200K_MODEL) is None PARITY_REQUESTS: Final[tuple[dict[str, object], ...]] = ( @@ -182,7 +338,16 @@ PARITY_REQUESTS: Final[tuple[dict[str, object], ...]] = ( }, { "model": MODEL, - "messages": [{"role": "user", "content": "x " * 20_000}], + "messages": [{"role": "user", "content": "x " * 500}], + }, + { + "model": MODEL, + "messages": [ + { + "role": "user", + "content": "I'VE got 1234567 things; it's \"fine\"...\r\n\r\n caf\u00e9 \u0645\u0631\u062d\u0628\u0627 \U0001f600 <|endoftext|>", + } + ], }, {"model": MODEL, "prompt": "Write a haiku about ships.", "max_tokens": 20}, {"model": MODEL, "prompt": ["first prompt", "second prompt"]}, @@ -190,7 +355,7 @@ PARITY_REQUESTS: Final[tuple[dict[str, object], ...]] = ( "model": MODEL, "instructions": "be terse", "input": [ - {"role": "user", "content": [{"type": "input_text", "text": "Summarise caf\u00e9 menus \u2014 \"ok\"?\n"}]}, + {"role": "user", "content": [{"type": "input_text", "text": 'Summarise caf\u00e9 menus \u2014 "ok"?\n'}]}, {"role": "assistant", "content": "Sure."}, ], }, @@ -202,27 +367,63 @@ PARITY_REQUESTS: Final[tuple[dict[str, object], ...]] = ( ) +PARITY_MODELS: Final[tuple[tuple[str, bridge.RustTokenizer], ...]] = ( + (MODEL, "anthropic"), + (CL100K_MODEL, "cl100k_base"), + (O200K_MODEL, "o200k_base"), + ("gpt-5", "o200k_base"), +) + + @pytest.mark.asyncio +@pytest.mark.parametrize(("model", "tokenizer"), PARITY_MODELS) @pytest.mark.parametrize("request_body", PARITY_REQUESTS) async def test_native_count_matches_python_budget_counter( - monkeypatch: pytest.MonkeyPatch, request_body: dict[str, object] + monkeypatch: pytest.MonkeyPatch, request_body: dict[str, object], model: str, tokenizer: bridge.RustTokenizer ) -> None: native: Final = pytest.importorskip("litellm.rust_bridge._native") monkeypatch.setattr(bindings, "get_native_bridge", lambda: native) litellm.rust(True) + body: Final = json.dumps(request_body).replace(MODEL, model) - rust_count: Final = await bridge.count_anthropic_input_tokens(json.dumps(request_body).encode()) - python_count: Final = _count_input_tokens(request_body=request_body, model=MODEL) + rust_count: Final = await bridge.count_input_tokens(body.encode(), tokenizer) + python_count: Final = _count_input_tokens(request_body=json.loads(body), model=model) assert rust_count is not None - assert rust_count.model == request_body.get("model") + assert rust_count.model == json.loads(body).get("model") assert rust_count.input_tokens == python_count +@pytest.mark.asyncio +@pytest.mark.parametrize(("model", "tokenizer"), ((CL100K_MODEL, "cl100k_base"), (O200K_MODEL, "o200k_base"))) +async def test_tiktoken_counts_long_text_exactly_where_python_chunks( + monkeypatch: pytest.MonkeyPatch, model: str, tokenizer: bridge.RustTokenizer +) -> None: + """Python encodes tiktoken text in fixed-size chunks (drift of up to one token per chunk boundary); Rust does not.""" + native: Final = pytest.importorskip("litellm.rust_bridge._native") + monkeypatch.setattr(bindings, "get_native_bridge", lambda: native) + litellm.rust(True) + text: Final = "x " * 20_000 + body: Final = {"model": model, "messages": [{"role": "user", "content": text}]} + encoding: Final = tiktoken.get_encoding(tokenizer) + exact: Final = 3 + len(encoding.encode("user")) + len(encoding.encode(text)) + 3 + chunks: Final = -(-len(text) // TIKTOKEN_ENCODE_CHUNK_SIZE_CHARS) + + rust_count: Final = await bridge.count_input_tokens(json.dumps(body).encode(), tokenizer) + python_count: Final = _count_input_tokens(request_body=body, model=model) + + assert rust_count is not None + assert rust_count.input_tokens == exact + assert python_count is not None + assert exact < python_count <= exact + chunks + + DECLINED_REQUESTS: Final[tuple[dict[str, object], ...]] = ( { "model": MODEL, - "messages": [{"role": "user", "content": [{"type": "image_url", "image_url": {"url": "data:image/png;base64,AA"}}]}], + "messages": [ + {"role": "user", "content": [{"type": "image_url", "image_url": {"url": "data:image/png;base64,AA"}}]} + ], }, {"model": MODEL, "prompt": 1.5}, {"model": MODEL, "documents": [{"score": 0.5}]}, @@ -231,12 +432,13 @@ DECLINED_REQUESTS: Final[tuple[dict[str, object], ...]] = ( @pytest.mark.asyncio +@pytest.mark.parametrize("tokenizer", TOKENIZERS) @pytest.mark.parametrize("request_body", DECLINED_REQUESTS) async def test_native_declines_shapes_python_prices_differently( - monkeypatch: pytest.MonkeyPatch, request_body: dict[str, object] + monkeypatch: pytest.MonkeyPatch, request_body: dict[str, object], tokenizer: bridge.RustTokenizer ) -> None: native: Final = pytest.importorskip("litellm.rust_bridge._native") monkeypatch.setattr(bindings, "get_native_bridge", lambda: native) litellm.rust(True) - assert await bridge.count_anthropic_input_tokens(json.dumps(request_body).encode()) is None + assert await bridge.count_input_tokens(json.dumps(request_body).encode(), tokenizer) is None From e4706fa4090ba2f6d88532e63428ff32994a64df Mon Sep 17 00:00:00 2001 From: yucheng-berri Date: Fri, 11 Sep 2026 16:49:48 -0700 Subject: [PATCH 134/157] fix(hide-secrets): restore credential coverage lost to the 4.5 entropy limit (#40190) * fix(hide-secrets): restore credential coverage lost to the 4.5 entropy limit Shannon entropy is bounded by log2(length), so the 4.5 limit #39879 shipped cannot score any value shorter than 23 characters, and it catches a random 32-character base64 credential only about two thirds of the time. A line like REDIS_PASSWORD=aB3dE6gH9jK2mN5p therefore reaches the provider in the clear. Add a keyword plugin that yields the credential-shaped value assigned to a credential-named key, reusing detect_secrets' own maintained denylist so camelCase, snake_case and SCREAMING_CASE all work with no local word list, and re-run the assignment-quoting transform detect_secrets skips once its first pass has matched. The entropy limits are untouched, so #39879's false-positive fix still holds. * fix(hide-secrets): read the assignments in a prompt that is mostly prose configparser aborts the whole parse on the first line it cannot read, so a message like "Here is my config, can you review it?" followed by REDIS_PASSWORD=... lost every assignment to that one prose line. Hand the parser only the lines it can read, dedent the assignments inside a pasted config, and keep each key distinct by line number so a config naming api_key once per model keeps every value instead of only the last. * fix(hide-secrets): drop the plugin docstrings and pin the block-scalar shapes * fix(hide-secrets): keep a comment or an indented header from closing an open value * fix(hide-secrets): drop the explanatory comments from the new scan helpers * fix(hide-secrets): accept punctuation in a credential value The value filter only allowed the URL-safe Base64 alphabet, so a password such as hunter2!brahms or p@ssw0rd!2026 passed through unredacted while the upstream keyword plugin had already matched it. The filter now rejects only whitespace and brackets, which keeps function calls, subscripts and sentences out while letting symbol-heavy passwords through. * fix(hide-secrets): redact every credential on a line and skip timestamps and plain urls replaces the inherited first-match scan with finditer over every keyword match, drops iso 8601 timestamps and userinfo-free urls from credential values, and threads the parser's open-option state through itertools.accumulate instead of rebinding it Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(hide-secrets): scan the first token of an assignment and ignore surrounding punctuation Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(hide-secrets): drop the unreachable configparser error fallback Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(hide-secrets): keep prose after a credential key out of the keyword detector A bare value followed by ordinary words (secret_sauce: Worcestershire sauce) is prose, so the synthetic assignment is only built when the value stands alone or is followed by a shell operator, comment, or another assignment Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(hide-secrets): scan the first token of shell-style assignments regardless of what follows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(hide-secrets): keep spaced assignments in scope when shell text follows the value Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(hide-secrets): drop docstrings that restate the test names Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(hide-secrets): stop reading a comparison operator as a trailing assignment Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(hide-secrets): keep dashed flags as assignment trailers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../enterprise_callbacks/secret_detection.py | 136 +++- .../secrets_plugins/credential_keyword.py | 63 ++ .../test_secret_detection.py | 598 +++++++++++++++++- 3 files changed, 764 insertions(+), 33 deletions(-) create mode 100644 enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/credential_keyword.py diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py b/enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py index bfbfd7bfb15..f0f85178672 100644 --- a/enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py +++ b/enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py @@ -11,10 +11,14 @@ import sys sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -import functools +import configparser +import contextlib +import itertools +import re import tempfile +from collections.abc import Generator, Iterator, Sequence from contextvars import ContextVar -from typing import TYPE_CHECKING, ClassVar, Literal, Optional +from typing import TYPE_CHECKING, ClassVar, Final, Literal, Optional from litellm._logging import verbose_proxy_logger from litellm.caching.caching import DualCache @@ -433,12 +437,101 @@ _default_detect_secrets_config = { "name": "ZendeskSecretKeyDetector", "path": _custom_plugins_path + "/zendesk_secret_key.py", }, + { + "name": "CredentialKeywordDetector", + "path": _custom_plugins_path + "/credential_keyword.py", + }, {"name": "Base64HighEntropyString", "limit": 4.5}, {"name": "HexHighEntropyString", "limit": 3.0}, ], } +_CONFIG_SECTION: Final = "litellm-prompt" + +_ASSIGNMENT_LINE: Final = re.compile(r"[^\s\[#;:=][^:=]*[:=]") + +_SHELL_ASSIGNMENT: Final = re.compile(r"(?P[^\s\[#;:=](?:[^:=]*[^\s:=])?)=(?P\S+)") + +_SHELL_OPERATORS: Final = ";&|" + +_SHELL_TRAILER: Final = re.compile(r"\\|#.*|-*\w[\w.-]*=\S*") + +_SCAN_SUFFIX: Final = ".py" + + +@contextlib.contextmanager +def _temp_file(text: str) -> Generator[str, None, None]: + temp_file: Final = tempfile.NamedTemporaryFile(suffix=_SCAN_SUFFIX, delete=False) + try: + temp_file.write(text.encode("utf-8")) + temp_file.close() + yield temp_file.name + finally: + temp_file.close() + os.remove(temp_file.name) + + +def _scan_lines(lines: Sequence[str]) -> frozenset[tuple[str, str]]: + from detect_secrets import SecretsCollection + + secrets: Final = SecretsCollection() + with _temp_file("\n".join(lines)) as path: + secrets.scan_file(path) + + return frozenset( + (found_secret.secret_value, found_secret.type) + for file in secrets.files + for found_secret in secrets[file] + if found_secret.secret_value is not None + ) + + +def _classify_line(state: tuple[bool, str | None], numbered: tuple[int, str]) -> tuple[bool, str | None]: + open_option: Final = state[0] + number, line = numbered + stripped: Final = line.strip() + if not stripped or stripped[0] in "#;": + return open_option, None + shell_assignment: Final = _SHELL_ASSIGNMENT.match(stripped) + if shell_assignment is not None: + return True, f"{shell_assignment['key']}_{number}={shell_assignment['value']}" + assignment: Final = _ASSIGNMENT_LINE.match(stripped) + if assignment is not None: + return True, f"{assignment.group()[:-1].strip()}_{number}{stripped[assignment.end() - 1 :]}" + if line[0].isspace() and open_option: + return True, line + return False, None + + +def _parseable_lines(text: str) -> Iterator[str]: + states: Final = itertools.accumulate(enumerate(text.splitlines()), _classify_line, initial=(False, None)) + return (line for _, line in states if line is not None) + + +def _lone_value(line: str) -> str | None: + tokens: Final = line.split() + if not tokens or '"' in tokens[0]: + return None + value: Final = tokens[0].rstrip(_SHELL_OPERATORS) + if len(tokens) == 1 or value != tokens[0] or _SHELL_TRAILER.fullmatch(tokens[1]) is not None: + return value + return None + + +def _quoted_assignments(text: str) -> tuple[str, ...]: + parser: Final = configparser.ConfigParser(interpolation=None) + parser.optionxform = str # pyright: ignore[reportAttributeAccessIssue] # configparser types optionxform as a method + parser.read_string(f"[{_CONFIG_SECTION}]\n" + "\n".join(_parseable_lines(text))) + return tuple( + f'{key} = "{value}"' + for section in parser + for key, values in parser.items(section) + for line in values.splitlines() + if (value := _lone_value(line)) is not None + ) + + class _ENTERPRISE_SecretDetection(CustomGuardrail): # Keeps proxied traffic on async_pre_call_hook (the unified apply_guardrail # path skips should_run_check and never sees data["prompt"]). @@ -449,35 +542,21 @@ class _ENTERPRISE_SecretDetection(CustomGuardrail): super().__init__(**kwargs) def scan_message_for_secrets(self, message_content: str): - from detect_secrets import SecretsCollection from detect_secrets.settings import transient_settings - temp_file = tempfile.NamedTemporaryFile(delete=False) - temp_file.write(message_content.encode("utf-8")) - temp_file.close() - - secrets = SecretsCollection() - detect_secrets_config = ( self.user_defined_detect_secrets_config or _default_detect_secrets_config ) with transient_settings(detect_secrets_config): - secrets.scan_file(temp_file.name) - - os.remove(temp_file.name) + found: Final = _scan_lines( + (*message_content.splitlines(), *_quoted_assignments(message_content)) + ) return [ - {"type": found_secret.type, "value": found_secret.secret_value} - for file in sorted(secrets.files) - for found_secret in sorted( - secrets[file], - key=lambda secret: ( - -len(secret.secret_value or ""), - secret.type, - secret.secret_value or "", - ), + {"type": secret_type, "value": value} + for value, secret_type in sorted( + found, key=lambda pair: (-len(pair[0]), pair[1], pair[0]) ) - if found_secret.secret_value is not None ] def redact_text(self, text: str, source: str = "message") -> str: @@ -490,15 +569,16 @@ class _ENTERPRISE_SecretDetection(CustomGuardrail): if counts is not None: for secret in detected_secrets: counts[secret["type"]] = counts.get(secret["type"], 0) + 1 - secret_types = [secret["type"] for secret in detected_secrets] + secret_types: Final = sorted( + dict.fromkeys(secret["type"] for secret in detected_secrets) + ) verbose_proxy_logger.warning( - f"Detected and redacted secrets in {source}: {secret_types}" + "Detected and redacted secrets in %s: %s", source, secret_types ) - return functools.reduce( - lambda redacted, secret: redacted.replace(secret["value"], "[REDACTED]"), - detected_secrets, - text, + pattern: Final = re.compile( + "|".join(re.escape(secret["value"]) for secret in detected_secrets) ) + return pattern.sub("[REDACTED]", text) async def should_run_check(self, user_api_key_dict: UserAPIKeyAuth) -> bool: if user_api_key_dict.permissions is not None: diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/credential_keyword.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/credential_keyword.py new file mode 100644 index 00000000000..b69e347ded5 --- /dev/null +++ b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/credential_keyword.py @@ -0,0 +1,63 @@ +import re +from collections.abc import Generator, Mapping +from string import punctuation +from typing import Final + +from detect_secrets.plugins.keyword import ( + QUOTES_REQUIRED_DENYLIST_REGEX_TO_GROUP, + KeywordDetector, +) + +_CREDENTIAL_VALUE: Final = re.compile(r"[^\s()\[\]]+") +_ENVIRONMENT_REFERENCE: Final = re.compile(r"os\.environ/\w+", re.IGNORECASE) +_ENVIRONMENT_VARIABLE_NAME: Final = re.compile(r"[A-Z][A-Z0-9]*(?:_[A-Z0-9]+)+") +_LOWERCASE_WORD_SEQUENCE: Final = re.compile(r"[a-z]+(?:[-._/][a-z]+)+") +_ISO_8601_TIMESTAMP: Final = re.compile( + r"\d{4}-\d{2}-\d{2}(?:T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:?\d{2})?)?" +) +_URL_WITHOUT_USERINFO_OR_QUERY: Final = re.compile(r"[A-Za-z][A-Za-z0-9+.-]*://[^\s@?]*") +_BENIGN_VALUES: Final = ( + _ENVIRONMENT_REFERENCE, + _ENVIRONMENT_VARIABLE_NAME, + _LOWERCASE_WORD_SEQUENCE, + _ISO_8601_TIMESTAMP, + _URL_WITHOUT_USERINFO_OR_QUERY, +) + + +class CredentialKeywordDetector(KeywordDetector): # pyright: ignore[reportUntypedBaseClass] # detect_secrets ships no type information + secret_type = "Credential Keyword" + + def __init__(self, minimum_length: int = 12, keyword_exclude: str | None = None) -> None: + if ( + not isinstance(minimum_length, int) # pyright: ignore[reportUnnecessaryIsInstance] # the value comes from an operator's YAML + or minimum_length < 1 + ): + raise ValueError(f"minimum_length must be a positive integer, got {minimum_length!r}") + super().__init__(keyword_exclude=keyword_exclude) + self.minimum_length = minimum_length + + def _is_credential(self, value: str) -> bool: + core: Final = value.strip(punctuation) + return ( + len(value) >= self.minimum_length + and _CREDENTIAL_VALUE.fullmatch(value) is not None + and all(benign.fullmatch(core) is None for benign in _BENIGN_VALUES) + ) + + def analyze_string( + self, + string: str, + denylist_regex_to_group: Mapping[re.Pattern[str], int] | None = None, + ) -> Generator[str, None, None]: + if self.keyword_exclude is not None and self.keyword_exclude.search(string): + return + regex_to_group: Final = ( + QUOTES_REQUIRED_DENYLIST_REGEX_TO_GROUP if denylist_regex_to_group is None else denylist_regex_to_group + ) + yield from ( + match.group(group) + for regex, group in regex_to_group.items() + for match in regex.finditer(string) + if self._is_credential(match.group(group)) + ) diff --git a/tests/test_litellm/enterprise/enterprise_callbacks/test_secret_detection.py b/tests/test_litellm/enterprise/enterprise_callbacks/test_secret_detection.py index c1c3569c0e2..bf9eeeb9968 100644 --- a/tests/test_litellm/enterprise/enterprise_callbacks/test_secret_detection.py +++ b/tests/test_litellm/enterprise/enterprise_callbacks/test_secret_detection.py @@ -10,12 +10,15 @@ Covers the three defects from the ticket: handling live only on the native path). """ +import tempfile import time import pytest from litellm_enterprise.enterprise_callbacks.secret_detection import ( _ENTERPRISE_SecretDetection, + _default_detect_secrets_config, + _masked_entity_count, ) from litellm.caching.caching import DualCache from litellm.proxy._types import UserAPIKeyAuth @@ -29,6 +32,13 @@ URL_ENCODED_KEY = "Bearer%20sk-Ab3dEf6Gh7Ij8Kl9Mn0Pq2Rs3Tu4Vw5X" AWS_KEYS = [f"AKIAIOSFODNN7EXAMPL{suffix}" for suffix in "FEDCBA"] +@pytest.fixture(autouse=True) +def _isolate_masked_entity_count(): + token = _masked_entity_count.set(None) + yield + _masked_entity_count.reset(token) + + def _guardrail() -> _ENTERPRISE_SecretDetection: return _ENTERPRISE_SecretDetection(guardrail_name="hide-secrets", event_hook="pre_call", default_on=True) @@ -58,6 +68,561 @@ def test_scan_message_preserves_quoted_benign_identifiers(): assert guardrail.redact_text(content) == content +@pytest.mark.parametrize( + "content,secret", + [ + ("REDIS_PASSWORD=aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + ("SESSION_SECRET=Kp7Nq2Wz9Bt4Xr6Vm1Ls", "Kp7Nq2Wz9Bt4Xr6Vm1Ls"), + ('{"db_password": "Tq8Zm2XpLv9KdNbRcYw3"}', "Tq8Zm2XpLv9KdNbRcYw3"), + ("api_secret: Zx4Kp9Lm2Qr7Ns3Vt", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("password = hunter2brahms9x", "hunter2brahms9x"), + ("client_secret=Hq7Zm3XkLp9Wd2Nb", "Hq7Zm3XkLp9Wd2Nb"), + ('apiKey: "aB3dE6gH9jK2mN5p"', "aB3dE6gH9jK2mN5p"), + ('{"clientSecret": "Kp7Nq2Wz9Bt4Xr6Vm1Ls"}', "Kp7Nq2Wz9Bt4Xr6Vm1Ls"), + ('dbPassword = "Zx4Kp9Lm2Qr7Ns3Vt"', "Zx4Kp9Lm2Qr7Ns3Vt"), + ("MY_APP_DB_PASSWORD=Kp7Nq2Wz9Bt4Xr6Vm1Ls", "Kp7Nq2Wz9Bt4Xr6Vm1Ls"), + ("x_api_key: 8f3Kd9Lm2Qr7Ns3Vt", "8f3Kd9Lm2Qr7Ns3Vt"), + ("password: Zm9vYmFyYmF6+abc/def123=", "Zm9vYmFyYmF6+abc/def123="), + ("REDIS_PASSWORD=correcthorsebattery", "correcthorsebattery"), + ('SECRET_KEY = "django-insecure-9v2xk4qw8z"', "django-insecure-9v2xk4qw8z"), + ( + "aws_secret_access_key = wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", + "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", + ), + ("password=aB3dE6gH9jK2", "aB3dE6gH9jK2"), + ("api_key: hunter2!brahms", "hunter2!brahms"), + ('db_password: "p@ssw0rd!2026"', "p@ssw0rd!2026"), + ( + 'url: "postgresql://user:s3cr3t@db-host:5432/app"', + "postgresql://user:s3cr3t@db-host:5432/app", + ), + ( + 'db_password: "postgresql://user:s3cr3t@db-host:5432/app"', + "postgresql://user:s3cr3t@db-host:5432/app", + ), + ( + 'signing_secret_url: "https://example.com/cb?sig=Zx4Kp9Lm2Qr7Ns3Vt"', + "https://example.com/cb?sig=Zx4Kp9Lm2Qr7Ns3Vt", + ), + ( + 'redis_secret_url: "redis://:Zx4Kp9Lm2Qr7Ns3Vt@cache-host:6379/0"', + "Zx4Kp9Lm2Qr7Ns3Vt", + ), + ("password=2026-09-08T17:38:40Zbrahms", "2026-09-08T17:38:40Zbrahms"), + ( + '{"password": "YOUR_API_KEY_HERE", "client_secret": "correcthorsebattery"}', + "correcthorsebattery", + ), + ("docker run -e REDIS_PASSWORD=aB3dE6gH9jK2mN5p \\\n -e REDIS_PORT=6379 redis", "aB3dE6gH9jK2mN5p"), + ("DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt && echo done", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("password = Zx4Kp9Lm2Qr7Ns3Vt # rotate me", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("my db password: Zx4Kp9Lm2Qr7Ns3Vt.", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("export DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt DB_HOST=db.internal", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt; systemctl restart app", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt | tee creds.txt", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt > setup.log", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("docker run -e DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt --name app postgres", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("password=correcthorsebattery please", "correcthorsebattery"), + ("DB_PASSWORD = Zx4Kp9Lm2Qr7Ns3Vt; systemctl restart app", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD = Zx4Kp9Lm2Qr7Ns3Vt \\", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD = Zx4Kp9Lm2Qr7Ns3Vt DB_HOST=db.internal", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD = Zx4Kp9Lm2Qr7Ns3Vt --db-host=db.internal", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("DB_PASSWORD = Zx4Kp9Lm2Qr7Ns3Vt DEBUG=", "Zx4Kp9Lm2Qr7Ns3Vt"), + ], + ids=[ + "env-password", + "env-secret", + "json-field", + "yaml-field", + "bare-assignment", + "client-secret", + "camel-case-key", + "camel-case-secret", + "camel-case-password", + "namespaced-env", + "underscored-header", + "base64-padding", + "digit-free-value", + "django-secret-key", + "slashed-aws-secret", + "shortest-accepted-value", + "punctuation-bearing-password", + "symbol-heavy-password", + "connection-string-under-a-url-key", + "connection-string-under-a-credential-key", + "signed-url-under-a-credential-key", + "password-only-url-under-a-credential-key", + "timestamp-prefixed-password", + "credential-after-a-rejected-placeholder", + "docker-flag-with-a-line-continuation", + "shell-command-after-the-value", + "inline-comment-after-the-value", + "sentence-ending-in-the-value", + "second-assignment-after-the-value", + "semicolon-after-the-value", + "pipe-after-the-value", + "redirect-after-the-value", + "docker-flag-after-the-value", + "prose-after-a-shell-assignment", + "spaced-assignment-then-a-shell-command", + "spaced-assignment-then-a-line-continuation", + "spaced-assignment-then-a-second-assignment", + "spaced-assignment-then-a-dashed-flag", + "spaced-assignment-then-an-empty-assignment", + ], +) +def test_scan_message_redacts_credentials_assigned_to_credential_keys(content, secret): + guardrail = _guardrail() + + assert secret not in guardrail.redact_text(content) + + +def test_scan_message_redacts_only_the_first_token_of_a_shell_assignment(): + guardrail = _guardrail() + content = "docker run -e REDIS_PASSWORD=aB3dE6gH9jK2mN5p \\\n -e REDIS_PORT=6379 redis && echo done" + + assert ( + guardrail.redact_text(content) + == "docker run -e REDIS_PASSWORD=[REDACTED] \\\n -e REDIS_PORT=6379 redis && echo done" + ) + + +@pytest.mark.parametrize("operator", [";", "&&", "|"]) +def test_scan_message_keeps_a_shell_operator_glued_to_the_value(operator): + guardrail = _guardrail() + + assert ( + guardrail.redact_text(f"DB_PASSWORD=Zx4Kp9Lm2Qr7Ns3Vt{operator} systemctl restart app") + == f"DB_PASSWORD=[REDACTED]{operator} systemctl restart app" + ) + + +def test_scan_message_closes_a_yaml_block_at_the_next_unindented_line(): + guardrail = _guardrail() + content = "api_key: >\n aB3dE6gH9jK2mN5p\nSteps\n Rotate-Before-Friday please" + + assert guardrail.redact_text(content) == "api_key: >\n [REDACTED]\nSteps\n Rotate-Before-Friday please" + + +def test_scan_message_redacts_every_credential_on_one_line(): + guardrail = _guardrail() + content = '{"db_password": "Tq8Zm2XpLv9KdNbRcYw3", "client_secret": "correcthorsebattery"}' + + assert guardrail.redact_text(content) == '{"db_password": "[REDACTED]", "client_secret": "[REDACTED]"}' + + +@pytest.mark.parametrize( + "content", + [ + "The user forgot their password and asked for a reset link", + "Rotate the client secret every 90 days", + "The secret: keep it quiet", + "My password: correct horse battery staple", + "secretary: Maria Gonzalez", + "password_reset_email: Please click the link below to reset", + 'config = {"api_key": "YOUR_API_KEY_HERE"}', + "api_key: ", + '{"max_tokens": 4096, "model": "gpt-4o-mini"}', + 'def get_api_key():\n return os.environ["OPENAI_API_KEY"]', + ' valid_token = UserAPIKeyAuth(user_id="u1")', + 'password = get_password(user, "prod")', + "monkey=aB3dE6gH9jK2mN5p", + "idempotency_key: req_2026090712000000", + 'cache_key = "u1_user_api_key_user_id"', + "the key: 2026-09-07T12:00:00Z", + "api_key: os.environ/E2B_API_KEY", + "langfuse_secret: os.environ/LANGFUSE_PROJECT1_SECRET", + "api_key = OPENAI_API_KEY", + "password = pwd12345678", + "model_key: gpt-4o-mini-2024-07-18", + "openrouter/anthropic/claude-3-5-sonnet-20240620", + '{"content-type": "application/json"}', + "passwordless_login: enabled-for-all-users", + 'password: "I forgot mine, can you reset it"', + "secret_sauce: tomatoes-basil-garlic-oregano", + "user_secret_question: what-was-your-first-pet", + "password_reset_url: example.com/reset-password/flow", + "private_key_path: keys/prod/server-cert.pem", + "litellm.completion(model=model, api_key=openai_api_key)", + "params['aws_secret_access_key'] = aws_secret_access_key", + 'api_key = "OPENAI_API_KEY"', + "model_list:\n - litellm_params:\n api_key: 'PERPLEXITY_API_KEY'", + 'config = build(_provider("ve_missing", api_key_env="VE_MISSING_KEY"))', + "api_key = get_api_key_from_env()", + "api_key = get_secret_str(MISTRAL_OCR_API_KEY_ENV_VAR)", + "secret_manager = MagicMock(spec=BaseSecretManager)", + "api_key = self.resolve_server_api_key(", + "api_key = sys.argv[1]", + "password = credentials[environment]", + 'api_key_created_at: "2026-09-08T17:38:40Z"', + 'api_key_expires_at: "2026-09-08T17:38:40.123456+05:30"', + 'password_reset_url: "https://example.com/reset-password/flow"', + 'secret_docs_url: "https://example.com/reset-password/flow#step-2"', + '{"api_key_created_at": "2026-09-08T17:38:40Z", "password_reset_url": "https://example.com/reset/flow"}', + "secret_sauce: tomatoes-basil-garlic-oregano.", + "secret_docs_url: https://example.com/docs/keys, then rotate", + "api_key_created_at: 2026-09-08T17:38:40Z; api_key_env: OPENAI_API_KEY!", + "api_key: $OPENAI_API_KEY", + 'api_key: "${OPENAI_API_KEY}"', + "private_key_path: /keys/prod/server-cert.pem", + "password_hint: your usual one followed by Ticket-LIT7049-Suffix", + "Translate this recipe note into French:\nsecret_sauce: Worcestershire sauce", + "api_key = Massachusetts (the state, not a key)", + "secret_sauce:Worcestershire sauce", + "password: correctHorseBattery != anotherValue", + ], + ids=[ + "prose-password", + "prose-secret", + "colon-prose-secret", + "colon-prose-password", + "secretary", + "sentence-after-keyword", + "uppercase-placeholder", + "templated-placeholder", + "max-tokens", + "code-paste", + "constructor-call", + "indirect-reference", + "word-ending-in-key", + "idempotency-key", + "cache-key", + "timestamp-after-key", + "env-reference", + "env-reference-nested", + "env-variable-name", + "below-minimum-length", + "model-name", + "namespaced-model-name", + "media-type", + "hyphenated-english", + "quoted-sentence-under-a-credential-key", + "hyphenated-phrase", + "hyphenated-question", + "url-under-credential-key", + "path-under-credential-key", + "snake-case-argument", + "snake-case-assignment", + "quoted-env-variable-name", + "quoted-env-name-in-a-config", + "quoted-env-name-in-a-code-paste", + "bare-call", + "call-with-an-argument", + "keyword-argument-call", + "unclosed-call", + "positional-subscript", + "keyed-subscript", + "timestamp-under-a-credential-key", + "offset-timestamp-under-a-credential-key", + "url-under-a-credential-key", + "fragment-url-under-a-credential-key", + "metadata-object-under-credential-keys", + "hyphenated-english-ending-a-sentence", + "url-followed-by-a-clause", + "timestamp-and-env-name-with-trailing-punctuation", + "shell-variable-reference", + "quoted-braced-shell-variable-reference", + "absolute-path-under-a-credential-key", + "sentence-holding-a-later-mixed-case-token", + "capitalized-word-starting-a-phrase", + "capitalized-word-before-a-parenthetical", + "yaml-scalar-without-a-space-after-the-colon", + "comparison-operator-after-the-value", + ], +) +def test_scan_message_keeps_benign_values(content): + guardrail = _guardrail() + + assert guardrail.scan_message_for_secrets(content) == [] + assert guardrail.redact_text(content) == content + + +@pytest.mark.parametrize( + "value,redacted", + [("aB3dE6gH9jK2", True), ("aB3dE6gH9jK", False)], + ids=["at-minimum-length", "below-minimum-length"], +) +def test_credential_keyword_detector_honours_its_minimum_length(value, redacted): + guardrail = _guardrail() + + assert (value not in guardrail.redact_text(f"password={value}")) is redacted + + +@pytest.mark.parametrize( + "value,redacted", + [("aB3dE6gH9jK2", True), ("aB3dE6gH9jK", False)], + ids=["at-default-minimum-length", "below-default-minimum-length"], +) +def test_credential_keyword_detector_defaults_its_minimum_length(value, redacted): + guardrail = _ENTERPRISE_SecretDetection( + guardrail_name="hide-secrets", + event_hook="pre_call", + default_on=True, + detect_secrets_config={ + "plugins_used": [ + {key: setting for key, setting in plugin.items() if key != "minimum_length"} + for plugin in _default_detect_secrets_config["plugins_used"] + ] + }, + ) + + assert (value not in guardrail.redact_text(f"password={value}")) is redacted + + +def test_credential_keyword_detector_honours_keyword_exclude(): + guardrail = _ENTERPRISE_SecretDetection( + guardrail_name="hide-secrets", + event_hook="pre_call", + default_on=True, + detect_secrets_config={ + "plugins_used": [ + {**plugin, "keyword_exclude": "fixture_"} if plugin["name"] == "CredentialKeywordDetector" else plugin + for plugin in _default_detect_secrets_config["plugins_used"] + ] + }, + ) + content = "fixture_password=aB3dE6gH9jK2mN5p\npassword=Kp7Nq2Wz9Bt4Xr6Vm1Ls" + + assert guardrail.redact_text(content) == "fixture_password=aB3dE6gH9jK2mN5p\npassword=[REDACTED]" + + +@pytest.mark.parametrize("minimum_length", ["12", 0, -1, 1.5], ids=["string", "zero", "negative", "float"]) +def test_credential_keyword_detector_rejects_an_unusable_minimum_length(minimum_length): + guardrail = _ENTERPRISE_SecretDetection( + guardrail_name="hide-secrets", + event_hook="pre_call", + default_on=True, + detect_secrets_config={ + "plugins_used": [ + {**plugin, "minimum_length": minimum_length} + if plugin["name"] == "CredentialKeywordDetector" + else plugin + for plugin in _default_detect_secrets_config["plugins_used"] + ] + }, + ) + + with pytest.raises(ValueError, match="minimum_length"): + guardrail.scan_message_for_secrets("password=aB3dE6gH9jK2mN5p") + + +@pytest.mark.parametrize( + "content", + [ + "[db\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + "[\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + "[note] have a look\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + "]\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + "[]\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + ], + ids=["unclosed", "bare-bracket", "bracketed-prose", "stray-close", "empty-header"], +) +def test_scan_message_reads_a_config_with_a_broken_section_header(content): + guardrail = _guardrail() + + assert "Zx4Kp9Lm2Qr7Ns3Vt" not in guardrail.redact_text(content) + + +@pytest.mark.parametrize( + "content", + [ + "=orphan\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + " indented before any key\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + "greeting = %(name)s\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + "token = a\x00b\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n", + ], + ids=["empty-key", "leading-continuation", "interpolation", "nul-byte"], +) +def test_scan_message_reads_lines_that_a_stock_ini_parser_rejects(content): + guardrail = _guardrail() + + assert "Zx4Kp9Lm2Qr7Ns3Vt" not in guardrail.redact_text(content) + + +def test_scan_message_reads_a_config_that_repeats_a_section(): + guardrail = _guardrail() + content = "[db]\nhost = localhost\n[db]\npassword = Zx4Kp9Lm2Qr7Ns3Vt\n" + + assert "Zx4Kp9Lm2Qr7Ns3Vt" not in guardrail.redact_text(content) + + +def test_scan_message_keeps_every_value_when_a_config_repeats_a_key(): + guardrail = _guardrail() + content = ( + "model_list:\n" + " - model_name: gpt-4o\n litellm_params:\n api_key: aB3dE6gH9jK2mN5p\n" + " - model_name: claude\n litellm_params:\n api_key: Kp7Nq2Wz9Bt4Xr6Vm1Ls\n" + ) + + redacted = guardrail.redact_text(content) + + assert "aB3dE6gH9jK2mN5p" not in redacted + assert "Kp7Nq2Wz9Bt4Xr6Vm1Ls" not in redacted + + +@pytest.mark.parametrize( + "content,secret", + [ + ( + f"api_key: {OPENAI_KEY}\nREDIS_PASSWORD=aB3dE6gH9jK2mN5p", + "aB3dE6gH9jK2mN5p", + ), + ( + f"OPENAI_API_KEY={OPENAI_KEY}\nDB_PASSWORD=Kp7Nq2Wz9Bt4Xr6Vm1Ls", + "Kp7Nq2Wz9Bt4Xr6Vm1Ls", + ), + ( + f"api_key: {OPENAI_KEY}\npassword =\n Zx4Kp9Lm2Qr7Ns3Vt", + "Zx4Kp9Lm2Qr7Ns3Vt", + ), + ( + "Here is my config, can you review it?\nREDIS_PASSWORD=aB3dE6gH9jK2mN5p", + "aB3dE6gH9jK2mN5p", + ), + ( + "REDIS_PASSWORD=aB3dE6gH9jK2mN5p\nCan you tell me what is wrong with it?", + "aB3dE6gH9jK2mN5p", + ), + ( + "Hi team\nplease rotate this before Friday\ndb_password=Zx4Kp9Lm2Qr7Ns3Vt\nthanks!", + "Zx4Kp9Lm2Qr7Ns3Vt", + ), + ( + "model_list:\n - model_name: gpt-4o\n litellm_params:\n api_key: aB3dE6gH9jK2mN5p\n", + "aB3dE6gH9jK2mN5p", + ), + ("api_key: >\n aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + ("api_key: |-\n aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + ("secret= \\\n aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + ("password =\n# rotate me\n Zx4Kp9Lm2Qr7Ns3Vt", "Zx4Kp9Lm2Qr7Ns3Vt"), + ("api_key =\n; rotate me\n aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + (" # pasted from the vault\napi_key=aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + (" [db]\napi_key=aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + (" pasted with a leading indent\napi_key=aB3dE6gH9jK2mN5p", "aB3dE6gH9jK2mN5p"), + ], + ids=[ + "flat-assignment", + "env-file", + "continuation-line", + "prose-before", + "prose-after", + "prose-both-sides", + "indented-config", + "yaml-folded-block", + "yaml-literal-block", + "backslash-continuation", + "comment-inside-a-value", + "semicolon-comment-inside-a-value", + "indented-comment-above", + "indented-section-header-above", + "indented-prose-above", + ], +) +def test_scan_message_still_sees_assignments_sharing_a_message_with_a_vendor_key(content, secret): + guardrail = _guardrail() + + redacted = guardrail.redact_text(content) + assert secret not in redacted + assert OPENAI_KEY not in redacted + + +def test_environment_reference_filter_only_drops_the_whole_value(): + guardrail = _guardrail() + + for reference in ("os.environ/OPENAI_API_KEY", "os.environ/e2b_api_key"): + assert guardrail.redact_text(f"password={reference}") == f"password={reference}" + assert guardrail.redact_text("password=notos.environ/OPENAI_API_KEY") == ("password=[REDACTED]") + + +def test_environment_variable_names_are_dropped_only_for_the_keyword_plugin(): + guardrail = _guardrail() + + assert guardrail.redact_text("password=REDIS_PASSWORD") == "password=REDIS_PASSWORD" + assert guardrail.scan_message_for_secrets('k = "ABCD1234_EFGH5678_IJKLMN"') == [ + {"type": "Base64 High Entropy String", "value": "ABCD1234_EFGH5678_IJKLMN"} + ] + + +def test_masked_entity_count_keeps_the_vendor_type_beside_the_entropy_type(): + guardrail = _guardrail() + _masked_entity_count.set({}) + + guardrail.redact_text('k = "ghp_abcdefghijklmnopqrstuvwxyzABCDEF1234"') + + assert _masked_entity_count.get() == { + "Base64 High Entropy String": 1, + "GitHub Token": 1, + } + + +@pytest.mark.parametrize( + "content", + [ + f"api_key: '{OPENAI_KEY}'\n" + + "a: &a [" + + ", ".join(['"x"'] * 9) + + "]\n" + + "".join(f"{chr(98 + i)}: &{chr(98 + i)} [" + ", ".join([f"*{chr(97 + i)}"] * 9) + "]\n" for i in range(7)), + f"api_key: '{OPENAI_KEY}'\ndeep: " + "[" * 400 + "]" * 400, + f"api_key: '{OPENAI_KEY}'\nbroken: [unclosed", + ], + ids=["anchor-expansion", "deep-nesting", "unparseable"], +) +def test_scan_message_contains_hostile_config_text(content, monkeypatch, tmp_path): + guardrail = _guardrail() + monkeypatch.setenv("TMPDIR", str(tmp_path)) + monkeypatch.setattr(tempfile, "tempdir", None) + + started = time.perf_counter() + found = guardrail.scan_message_for_secrets(content) + + assert time.perf_counter() - started < 10.0 + assert OPENAI_KEY in [secret["value"] for secret in found] + assert list(tmp_path.iterdir()) == [] + + +@pytest.mark.parametrize( + "content", + [ + f"api_key = '{OPENAI_KEY}'\nbase = abcdefghijkl\npassword = x\n %(base)sZZZZQQQQ\n", + "base = abcdefghijkl\npassword = x\n %(base)sZZZZQQQQ\n", + f"api_key = '{OPENAI_KEY}'\nbase = Kp7Nq2Wz9Bt4\npassword = x\n" + " %(base)s-primary\nnote = Kp7Nq2Wz9Bt4-primary is the hostname\n", + 'base = "abcdefghijkl"\npassword = "%(base)sZZZZQQQQ"\n', + ], + ids=[ + "vendor-key-present", + "no-vendor-key", + "value-echoed-elsewhere", + "quoted-interpolation", + ], +) +def test_scan_message_never_reports_a_value_the_message_does_not_hold(content): + guardrail = _guardrail() + + for secret in guardrail.scan_message_for_secrets(content): + assert secret["value"] in content + + +def test_scan_message_leaves_unrelated_text_alone_when_a_value_is_echoed(): + guardrail = _guardrail() + content = ( + f"api_key = '{OPENAI_KEY}'\nbase = Kp7Nq2Wz9Bt4\npassword = x\n" + " %(base)s-primary\nnote = Kp7Nq2Wz9Bt4-primary is the hostname\n" + ) + + assert "note = Kp7Nq2Wz9Bt4-primary is the hostname" in guardrail.redact_text(content) + + +def test_masked_entity_count_counts_each_secret_once(): + guardrail = _guardrail() + _masked_entity_count.set({}) + + guardrail.redact_text(f"first {OPENAI_KEY} second {OPENAI_KEY}") + + assert _masked_entity_count.get() == {"Strict OpenAI API Key": 1} + + def test_scan_message_redacts_every_openai_key_occurrence(): guardrail = _guardrail() content = f"first {OPENAI_KEY}, second {OPENAI_KEY}" @@ -81,9 +646,7 @@ def test_scan_message_requires_ascii_digits_for_openai_like_values(): def test_scan_message_redacts_openai_key_after_separator(): guardrail = _guardrail() - assert guardrail.redact_text(f"openai_{OPENAI_KEY} key-{OPENAI_KEY}") == ( - "openai_[REDACTED] key-[REDACTED]" - ) + assert guardrail.redact_text(f"openai_{OPENAI_KEY} key-{OPENAI_KEY}") == ("openai_[REDACTED] key-[REDACTED]") assert guardrail.redact_text(URL_ENCODED_KEY) == "Bearer%20[REDACTED]" @@ -102,6 +665,31 @@ def test_scan_message_stays_linear_on_repeated_sk_separators(): assert time.perf_counter() - started < 2.0 +@pytest.mark.parametrize( + "content", + [ + f"api_key: '{OPENAI_KEY}'\npassword=" + "a-" * 10_000 + "!", + f"api_key: '{OPENAI_KEY}'\npassword:" + '"' * 20_000, + f"api_key: '{OPENAI_KEY}'\n" + "api_key:" * 10_000, + f"api_key: '{OPENAI_KEY}'\nsecret=" + "aB3dE6gH9jK2mN5p " * 2_000, + f"api_key: '{OPENAI_KEY}'\n" + "\n".join(f"password{i}=aB3dE6gH9jK2mN5p{i}" for i in range(3_000)), + ], + ids=[ + "value-run", + "quote-run", + "keyword-run", + "value-repeat", + "assignment-flood", + ], +) +def test_scan_message_stays_linear_on_adversarial_credential_lines(content): + guardrail = _guardrail() + + started = time.perf_counter() + guardrail.redact_text(content) + assert time.perf_counter() - started < 10.0 + + def test_scan_message_redacts_whole_stripe_live_key(): guardrail = _guardrail() @@ -119,8 +707,8 @@ def test_scan_message_replaces_longest_overlapping_match_first(): guardrail = _guardrail() content = f'token = "{OPENAI_KEY}/extra"' - detected = guardrail.scan_message_for_secrets(content) - assert [secret["value"] for secret in detected] == [f"{OPENAI_KEY}/extra", OPENAI_KEY] + values = [secret["value"] for secret in guardrail.scan_message_for_secrets(content)] + assert values == [f"{OPENAI_KEY}/extra", OPENAI_KEY] assert guardrail.redact_text(content) == 'token = "[REDACTED]"' From dab7f6a86acadb790e66be12dbd71e6706f60d2c Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 17:14:40 -0700 Subject: [PATCH 135/157] feat(proxy): expose complexity routing headers (#40792) (cherry picked from commit c817faec7aaf001baec138ff748fb0583a6d0c4c) Co-authored-by: Tin Co-authored-by: Claude Code --- litellm/proxy/common_request_processing.py | 13 +- litellm/router.py | 24 ++- .../add_retry_fallback_headers.py | 67 +++++- .../proxy/test_common_request_processing.py | 77 +++++++ .../test_add_retry_fallback_headers.py | 121 +++++++++++ tests/test_litellm/test_router.py | 199 +++++++++++++++++- 6 files changed, 487 insertions(+), 14 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index e3a2b892721..9f58aaf24f1 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -2437,6 +2437,7 @@ class ProxyBaseLLMRequestProcessing: if self._is_streaming_request( data=self.data, is_streaming_request=is_streaming_request ) or self._is_streaming_response(response): # use generate_responses to stream responses + selected_data_generator: AsyncGenerator[str, None] | None = None # Call response headers hook for streaming success stream_callback_headers: Final = await proxy_logging_obj.post_call_response_headers_hook( data=self.data, @@ -2561,14 +2562,9 @@ class ProxyBaseLLMRequestProcessing: None if _should_return_raw_model_name(self.data) else requested_model_from_client ), ) - return await create_response( - generator=wrap_sse_stream_with_keepalive_pings( - stream=selected_data_generator, - ping_interval_seconds=litellm.anthropic_sse_ping_interval_seconds, - ), - media_type="text/event-stream", - headers=custom_headers, - request=request, + selected_data_generator = wrap_sse_stream_with_keepalive_pings( + stream=selected_data_generator, + ping_interval_seconds=litellm.anthropic_sse_ping_interval_seconds, ) # Non-streaming response - fall through to normal response handling elif select_data_generator: @@ -2595,6 +2591,7 @@ class ProxyBaseLLMRequestProcessing: user_api_key_dict=user_api_key_dict, ) ) + if selected_data_generator is not None: return await create_response( generator=selected_data_generator, media_type="text/event-stream", diff --git a/litellm/router.py b/litellm/router.py index 597bcfaa20f..8865543badd 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -124,9 +124,11 @@ from litellm.router_utils.add_retry_fallback_headers import ( add_retry_headers_to_response, apply_quality_router_decision_headers, apply_remaining_usage_headers, + complexity_router_decision_headers, ensure_response_additional_headers, get_hidden_params_dict, prepare_response_for_header_attachment, + replace_complexity_router_headers, response_in_flight_token_count, ) from litellm.router_utils.auto_router_model_naming import ( @@ -555,6 +557,7 @@ class FallbackAwareAnthropicMessagesStream: def __init__(self, async_generator: AsyncGenerator[bytes, None], source_iterator: object) -> None: self._async_generator = async_generator self._source_iterator = source_iterator + self.fallback_headers_adopted = False self._hidden_params = dict( # mutable-ok: mutated in place by merge_fallback_hidden_params getattr(source_iterator, "_hidden_params", None) or {} ) @@ -565,6 +568,7 @@ class FallbackAwareAnthropicMessagesStream: def adopt_fallback_source(self, fallback_response: object) -> None: self._source_iterator = fallback_response + self.fallback_headers_adopted = True def __aiter__(self) -> "FallbackAwareAnthropicMessagesStream": return self @@ -593,7 +597,9 @@ class FallbackAwareAnthropicMessagesStream: self._hidden_params = { # mutable-ok: matches _hidden_params' existing dict[str, object] shape **self._hidden_params, **fallback_hidden_params, - "additional_headers": {**existing_headers, **fallback_headers}, # mutable-ok: same shape + "additional_headers": dict( # mutable-ok: hidden params expect a writable header bag + replace_complexity_router_headers(existing_headers, fallback_headers) + ), } @@ -3128,6 +3134,8 @@ class Router: async generator. """ + fallback_headers_adopted: bool = False + def __init__(self, async_generator: AsyncGenerator): import time from datetime import datetime @@ -3179,6 +3187,12 @@ class Router: # api_base, additional_headers) keep flowing. self._hidden_params = dict(getattr(source_iterator, "_hidden_params", None) or {}) + def adopt_fallback_headers(self, fallback_response: object) -> tuple[dict[str, object], dict[str, object]]: + prepared: Final = Router._prepare_fallback_hidden_params(fallback_response) + self._hidden_params = {**prepared[0], "additional_headers": prepared[1]} # mutable-ok: stream metadata + self.fallback_headers_adopted = True + return prepared + def __aiter__(self): return self @@ -3269,8 +3283,8 @@ class Router: include_fallback_errors=initial_kwargs.get("include_fallback_errors", False) is True, ) + prepared_fallback_hidden_params = wrapper.adopt_fallback_headers(fallback_response) if hasattr(fallback_response, "__aiter__"): - prepared_fallback_hidden_params = Router._prepare_fallback_hidden_params(fallback_response) async for fallback_item in fallback_response: Router._apply_fallback_hidden_params_to_item(fallback_item, prepared_fallback_hidden_params) if partial_usage is not None: @@ -3305,7 +3319,8 @@ class Router: exc, ) - return FallbackResponsesStreamWrapper(stream_with_fallbacks()) + wrapper: Final = FallbackResponsesStreamWrapper(stream_with_fallbacks()) + return wrapper def _completion_streaming_iterator( self, @@ -11108,7 +11123,7 @@ class Router: self, response: object, model_group: str | None = None, - request_kwargs: dict | None = None, + request_kwargs: dict[str, object] | None = None, ) -> Any: """ Add the most accurate rate limit headers for a given model response. @@ -11124,6 +11139,7 @@ class Router: additional_headers: Final = ensure_response_additional_headers(response) additional_headers["x-litellm-model-group"] = model_group apply_quality_router_decision_headers(additional_headers, request_kwargs) + additional_headers.update(complexity_router_decision_headers(request_kwargs)) if model_group is not None: remaining_usage: Final = await self.get_remaining_model_group_usage(model_group) diff --git a/litellm/router_utils/add_retry_fallback_headers.py b/litellm/router_utils/add_retry_fallback_headers.py index 3251ea457cf..bc88feef7d2 100644 --- a/litellm/router_utils/add_retry_fallback_headers.py +++ b/litellm/router_utils/add_retry_fallback_headers.py @@ -1,7 +1,10 @@ import json +import math +from collections.abc import Mapping +from types import MappingProxyType from typing import Any, Final, Protocol, TypedDict, cast -from pydantic import BaseModel +from pydantic import BaseModel, TypeAdapter, ValidationError class FallbackErrorInfo(TypedDict): @@ -15,6 +18,68 @@ class _HiddenParamsHost(Protocol): _hidden_params: dict[str, object] +_EMPTY_OBJECT_MAPPING: Final[Mapping[str, object]] = MappingProxyType({}) +_ROUTING_HEADER_MAPPING: Final = TypeAdapter(Mapping[str, object]) +_COMPLEXITY_ROUTER_HEADER_PREFIX: Final = "x-litellm-complexity-router-" + + +def _routing_header_mapping(value: object) -> Mapping[str, object]: + try: + mapping: Final[Mapping[str, object]] = _ROUTING_HEADER_MAPPING.validate_python(value, strict=True) + return mapping + except ValidationError: + return _EMPTY_OBJECT_MAPPING + + +def _header_string(value: object) -> str | None: + if not isinstance(value, str): + return None + normalized: Final = value.strip() + return normalized if normalized and all(" " <= character <= "~" for character in normalized) else None + + +def complexity_router_decision_headers(request_kwargs: object) -> Mapping[str, str]: + data: Final = _routing_header_mapping(request_kwargs) + metadata_key: Final = "litellm_metadata" if "litellm_metadata" in data else "metadata" + decision: Final = _routing_header_mapping(_routing_header_mapping(data.get(metadata_key)).get("routing_decision")) + if decision.get("router_type") != "complexity": + return MappingProxyType({}) + score: Final = decision.get("score") + values: Final = ( + ("tier", decision.get("tier")), + ("cause", decision.get("cause")), + ( + "score", + str(score) + if isinstance(score, (int, float)) and not isinstance(score, bool) and math.isfinite(score) + else None, + ), + ( + "reasoning-effort", + _routing_header_mapping(decision.get("tier_litellm_params")).get("reasoning_effort"), + ), + ) + return MappingProxyType( + { + f"{_COMPLEXITY_ROUTER_HEADER_PREFIX}{key}": header_value + for key, value in values + if (header_value := _header_string(value)) is not None + } + ) + + +def replace_complexity_router_headers( + existing_headers: Mapping[str, object], new_headers: Mapping[str, object] +) -> Mapping[str, object]: + return MappingProxyType( + { + key: value + for key, value in (*existing_headers.items(), *new_headers.items()) + if key in new_headers or not key.startswith(_COMPLEXITY_ROUTER_HEADER_PREFIX) + } + ) + + class HiddenParamsAsyncIteratorWrapper: """ Wraps a bare async generator/iterator (e.g. a provider's raw SSE diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 50b26577e5c..efbb5eedad4 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -8259,6 +8259,83 @@ class TestStreamingResponseHeadersFollowFallback: assert result.headers["x-callback-header"] == "kept" +class _MessagesFallbackStream: + def __init__(self) -> None: + self.fallback_headers_adopted = False + self._hidden_params: dict[str, object] = { + "additional_headers": { + "x-litellm-complexity-router-tier": "REASONING", + "x-litellm-complexity-router-reasoning-effort": "xhigh", + } + } + self._chunks = iter( + ( + b'event: content_block_delta\ndata: {"type":"content_block_delta","delta":{"type":"text_delta","text":"OK"}}\n\n', + ) + ) + + def __aiter__(self) -> "_MessagesFallbackStream": + return self + + async def __anext__(self) -> bytes: + self._hidden_params = { + "model_id": "fallback-deployment", + "additional_headers": {"x-fallback-only": "yes"}, + } + self.fallback_headers_adopted = True + try: + return next(self._chunks) + except StopIteration: + raise StopAsyncIteration from None + + async def aclose(self) -> None: + return None + + +@pytest.mark.asyncio +async def test_messages_http_headers_refresh_after_lazy_fallback(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.caching.caching import DualCache + + stream = _MessagesFallbackStream() + logging_obj = MagicMock() + logging_obj.litellm_call_id = "messages-fallback-headers" + logging_obj._defer_async_logging = False + logging_obj._on_deferred_stream_complete = None + logging_obj.cost_breakdown = None + logging_obj.litellm_params = {} + processor = ProxyBaseLLMRequestProcessing( + data={"model": "auto-router", "stream": True, "litellm_logging_obj": logging_obj} + ) + proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache()) + monkeypatch.setattr(litellm, "callbacks", []) + + async def call() -> _MessagesFallbackStream: + return stream + + async def fake_route_request(**_kwargs: object) -> object: + return call() + + monkeypatch.setattr(litellm.proxy.common_request_processing, "route_request", fake_route_request) + response = await processor.base_process_llm_request( + request=Request(scope={"type": "http", "headers": []}), + fastapi_response=Response(), + user_api_key_dict=ProxyUserAPIKeyAuth(api_key="sk-test"), + route_type="anthropic_messages", + proxy_logging_obj=proxy_logging_obj, + general_settings={}, + proxy_config=MagicMock(spec=ProxyConfig), + is_streaming_request=True, + skip_pre_call_logic=True, + ) + + assert isinstance(response, StreamingResponse) + assert stream.fallback_headers_adopted is True + assert response.headers["x-litellm-model-id"] == "fallback-deployment" + assert response.headers["x-fallback-only"] == "yes" + assert "x-litellm-complexity-router-tier" not in response.headers + assert "x-litellm-complexity-router-reasoning-effort" not in response.headers + + class TestPassthroughHeadersAcceptImmutableMappings: """LIT-6767: the streaming branch now hands the passthrough helpers an immutable mapping.""" diff --git a/tests/test_litellm/router_utils/test_add_retry_fallback_headers.py b/tests/test_litellm/router_utils/test_add_retry_fallback_headers.py index 3a0deeb13d8..eb7490d76f8 100644 --- a/tests/test_litellm/router_utils/test_add_retry_fallback_headers.py +++ b/tests/test_litellm/router_utils/test_add_retry_fallback_headers.py @@ -1,12 +1,17 @@ import json +from collections.abc import Mapping +from typing import Literal +import pytest from pydantic import BaseModel from litellm.router_utils.add_retry_fallback_headers import ( add_fallback_headers_to_response, add_retry_headers_to_response, + complexity_router_decision_headers, get_fallback_errors_from_headers, get_hidden_params_dict, + replace_complexity_router_headers, ) @@ -15,6 +20,122 @@ class StreamingWrapper: self._hidden_params = {"additional_headers": {"x-existing": "keep"}} +@pytest.mark.parametrize("metadata_key", ["metadata", "litellm_metadata"]) +def test_complexity_router_decision_headers_exposes_only_bounded_fields( + metadata_key: Literal["metadata", "litellm_metadata"], +) -> None: + headers = complexity_router_decision_headers( + { + metadata_key: { + "routing_decision": { + "router_type": "complexity", + "tier": " REASONING ", + "cause": "heuristic_scorer", + "score": 0.75, + "tier_litellm_params": {"reasoning_effort": "xhigh", "api_key": "secret"}, + "signals": ["private prompt"], + "matched_keyword": "private prompt", + } + } + } + ) + + assert dict(headers) == { + "x-litellm-complexity-router-tier": "REASONING", + "x-litellm-complexity-router-cause": "heuristic_scorer", + "x-litellm-complexity-router-score": "0.75", + "x-litellm-complexity-router-reasoning-effort": "xhigh", + } + + +@pytest.mark.parametrize( + "decision, expected", + [ + ( + {"router_type": "complexity", "tier": "SIMPLE", "cause": "heuristic_scorer", "score": 0}, + { + "x-litellm-complexity-router-tier": "SIMPLE", + "x-litellm-complexity-router-cause": "heuristic_scorer", + "x-litellm-complexity-router-score": "0", + }, + ), + ( + {"router_type": "complexity", "tier": "COMPLEX", "cause": "llm_classifier"}, + { + "x-litellm-complexity-router-tier": "COMPLEX", + "x-litellm-complexity-router-cause": "llm_classifier", + }, + ), + ( + {"router_type": "complexity", "tier": "REASONING", "cause": "literal_keyword_match"}, + { + "x-litellm-complexity-router-tier": "REASONING", + "x-litellm-complexity-router-cause": "literal_keyword_match", + }, + ), + ({"router_type": "quality", "tier": "premium", "cause": "quality_tier"}, {}), + ({"router_type": "complexity", "score": True}, {}), + ({"router_type": "complexity", "score": float("nan")}, {}), + ({"router_type": "complexity", "score": float("inf")}, {}), + ({"router_type": "complexity", "tier": "研究", "cause": "bad\r\nX-Injected: true"}, {}), + ({"router_type": "complexity", "tier_litellm_params": {"reasoning_effort": 1}}, {}), + ({"router_type": "complexity", "tier_litellm_params": "invalid"}, {}), + ([], {}), + (None, {}), + ], +) +def test_complexity_router_decision_headers_omits_absent_or_invalid_fields( + decision: object, + expected: Mapping[str, str], +) -> None: + assert dict(complexity_router_decision_headers({"metadata": {"routing_decision": decision}})) == expected + + +@pytest.mark.parametrize( + "litellm_decision, metadata_decision, expected", + [ + ( + {"router_type": "complexity", "tier": "SIMPLE", "cause": "heuristic_scorer"}, + {"router_type": "complexity", "tier": "REASONING", "tier_litellm_params": {"reasoning_effort": "xhigh"}}, + {"x-litellm-complexity-router-tier": "SIMPLE", "x-litellm-complexity-router-cause": "heuristic_scorer"}, + ), + ( + {"router_type": "quality", "tier": "premium"}, + {"router_type": "complexity", "tier": "FORGED", "cause": "heuristic_scorer"}, + {}, + ), + ( + {}, + {"router_type": "complexity", "tier": "FORGED", "cause": "heuristic_scorer"}, + {}, + ), + ], +) +def test_complexity_router_decision_headers_never_falls_back_from_internal_metadata( + litellm_decision: Mapping[str, object], + metadata_decision: Mapping[str, object], + expected: Mapping[str, str], +) -> None: + headers = complexity_router_decision_headers( + { + "litellm_metadata": {"routing_decision": litellm_decision}, + "metadata": {"routing_decision": metadata_decision}, + } + ) + assert dict(headers) == expected + + +def test_replace_complexity_router_headers_drops_stale_values() -> None: + assert replace_complexity_router_headers( + { + "x-existing": "keep", + "x-litellm-complexity-router-tier": "REASONING", + "x-litellm-complexity-router-reasoning-effort": "xhigh", + }, + {"x-litellm-complexity-router-tier": "SIMPLE"}, + ) == {"x-existing": "keep", "x-litellm-complexity-router-tier": "SIMPLE"} + + def test_add_fallback_headers_to_streaming_wrapper(): response = StreamingWrapper() diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index 2a97e92396a..f5e9b2091a0 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -8,7 +8,7 @@ import threading from datetime import datetime from collections.abc import Awaitable, Callable, Mapping from types import SimpleNamespace -from typing import Final +from typing import Final, Literal from unittest.mock import AsyncMock, MagicMock, patch import httpx @@ -2622,6 +2622,78 @@ def test_adopt_fallback_response_headers_keeps_identity_when_fallback_has_none() assert wrapper.fallback_headers_adopted is True +@pytest.mark.asyncio +@pytest.mark.parametrize("response_kind", ["object", "dict", "async-generator"]) +async def test_set_response_headers_exposes_complexity_decision_on_every_response_shape( + response_kind: Literal["object", "dict", "async-generator"], +) -> None: + class HeaderResponse: + def __init__(self) -> None: + self._hidden_params: dict[str, object] = {} + + response: object + if response_kind == "object": + response = HeaderResponse() + elif response_kind == "dict": + response = {} + else: + response = _AsyncList() + + router = Router(model_list=[]) + result = await router.set_response_headers( + response=response, + request_kwargs={ + "metadata": { + "routing_decision": { + "router_type": "complexity", + "tier": "SIMPLE", + "cause": "heuristic_scorer", + "score": 0.25, + "tier_litellm_params": {"reasoning_effort": "low"}, + } + } + }, + ) + hidden_params = result["_hidden_params"] if isinstance(result, dict) else result._hidden_params + additional_headers = hidden_params["additional_headers"] + + assert additional_headers == { + "x-litellm-model-group": None, + "x-litellm-complexity-router-tier": "SIMPLE", + "x-litellm-complexity-router-cause": "heuristic_scorer", + "x-litellm-complexity-router-score": "0.25", + "x-litellm-complexity-router-reasoning-effort": "low", + } + + +@pytest.mark.asyncio +async def test_set_response_headers_is_the_only_complexity_header_source_for_proxy_headers() -> None: + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing + + router = Router(model_list=[]) + response = await router.set_response_headers(response={}, request_kwargs={}) + additional_headers = response["_hidden_params"]["additional_headers"] + proxy_headers = ProxyBaseLLMRequestProcessing.get_custom_headers( + user_api_key_dict=UserAPIKeyAuth(), + request_data={ + "metadata": { + "routing_decision": { + "router_type": "complexity", + "tier": "REASONING", + "cause": "heuristic_scorer", + "tier_litellm_params": {"reasoning_effort": "xhigh"}, + } + } + }, + **additional_headers, + ) + + assert not { + key for key in proxy_headers if key.startswith("x-litellm-complexity-router-") + } + + @pytest.mark.asyncio async def test_acompletion_streaming_iterator_adopts_fallback_response_headers(): """LIT-6767: after a successful pre-first-chunk fallback, the wrapper must @@ -3503,6 +3575,26 @@ def _make_router_with_fallback(primary="gpt-4", secondary="gpt-3.5-turbo"): ) +class _InjectedFallbackRouter(Router): + def __init__(self, fallback_response: object) -> None: + super().__init__(model_list=[]) + self._fallback_response: Final = fallback_response + + async def async_function_with_fallbacks_common_utils( + self, + e: Exception, + disable_fallbacks: bool | None, + fallbacks: list | None, + context_window_fallbacks: list | None, + content_policy_fallbacks: list | None, + model_group: str | None, + args: tuple[object, ...], + kwargs: dict[str, object], + include_fallback_errors: bool = False, + ) -> object: + return self._fallback_response + + @pytest.mark.asyncio async def test_aresponses_streaming_iterator_fallback(): """Catches MidStreamFallbackError, re-enters the fallback chain via @@ -3562,6 +3654,63 @@ async def test_aresponses_streaming_iterator_fallback(): assert call_kwargs["disable_fallbacks"] is False +@pytest.mark.asyncio +@pytest.mark.parametrize( + "fallback_headers", + [ + {"x-fallback-only": "yes"}, + { + "x-fallback-only": "yes", + "x-litellm-complexity-router-tier": "SIMPLE", + }, + ], + ids=["plain-fallback", "complexity-tier-fallback"], +) +async def test_aresponses_streaming_iterator_replaces_complexity_headers_before_fallback_output( + fallback_headers: dict[str, str], +) -> None: + primary_headers: Final = { + "x-litellm-complexity-router-tier": "REASONING", + "x-litellm-complexity-router-reasoning-effort": "xhigh", + } + source: Final = _make_responses_iterator( + error=MidStreamFallbackError( + message="primary failed before output", + model="gpt-4", + llm_provider="openai", + is_pre_first_chunk=True, + generated_content="", + ), + hidden_params={"additional_headers": primary_headers}, + ) + fallback_output: Final = MagicMock(type="response.output_text.delta") + fallback: Final = _AsyncList([fallback_output]) + fallback._hidden_params = { + "model_id": "fallback-deployment", + "additional_headers": fallback_headers, + } + router: Final = _InjectedFallbackRouter(fallback) + + wrapped: Final = await router._aresponses_streaming_iterator( + response=source, + initial_kwargs={ + "model": "gpt-4", + "stream": True, + "input": "Hello", + "original_generic_function": litellm.aresponses, + }, + ) + assert wrapped._hidden_params["additional_headers"] == primary_headers + first_output: Final = await wrapped.__anext__() + + assert first_output is fallback_output + assert wrapped.fallback_headers_adopted is True + assert wrapped._hidden_params == { + "model_id": "fallback-deployment", + "additional_headers": fallback_headers, + } + + @pytest.mark.asyncio async def test_aresponses_streaming_iterator_writes_litellm_metadata_on_fallback(): """Regression: model_group must land under "litellm_metadata" (the key @@ -12579,6 +12728,54 @@ async def test_anthropic_messages_fallback_merges_fallback_hidden_params(): assert headers["x-fallback-only"] == "yes" +@pytest.mark.asyncio +@pytest.mark.parametrize( + "fallback_headers", + [ + {"x-fallback-only": "yes"}, + { + "x-fallback-only": "yes", + "x-litellm-complexity-router-tier": "SIMPLE", + }, + ], + ids=["plain-fallback", "complexity-tier-fallback"], +) +async def test_anthropic_messages_fallback_replaces_complexity_headers_before_output( + fallback_headers: dict[str, str], +) -> None: + primary_headers: Final = { + "x-litellm-complexity-router-tier": "REASONING", + "x-litellm-complexity-router-reasoning-effort": "xhigh", + } + source: Final = _AnthropicMessagesFallbackByteStream( + [_anthropic_messages_overloaded_error_chunk()], + hidden_params={"additional_headers": primary_headers}, + ) + fallback_output: Final = _anthropic_messages_content_chunk("fallback answer") + fallback: Final = _AnthropicMessagesFallbackByteStream( + [fallback_output], + hidden_params={ + "model_id": "fallback-deployment", + "additional_headers": fallback_headers, + }, + ) + router: Final = _InjectedFallbackRouter(fallback) + + wrapped: Final = await router._aanthropic_messages_streaming_iterator( + response=source, + initial_kwargs={"model": "primary"}, + ) + assert wrapped._hidden_params["additional_headers"] == primary_headers + first_output: Final = await wrapped.__anext__() + + assert first_output == fallback_output + assert wrapped.fallback_headers_adopted is True + assert wrapped._hidden_params == { + "model_id": "fallback-deployment", + "additional_headers": fallback_headers, + } + + @pytest.mark.asyncio async def test_aanthropic_messages_with_streaming_fallbacks_deep_copies_nested_metadata(): """Bugbot regression: a shallow .copy() of kwargs still shares the From b6b5e27d7b82a2c7df634f417ab75cd50fcb0b8e Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 17:23:31 -0700 Subject: [PATCH 136/157] docs(pr-template): add an Assumptions section Action item from the v1.100.0 OOM RCA. LIT-6780 recorded "running without --detailed_debug was not tried" as unverified; the fix PR closing it stated "Only happens with --detailed_debug on" as fact without running that test, and the untested half is where the customer-facing OOM lived. Nothing in the template asked for the hedge, so it disappeared between the ticket and review. The section asks for each untested claim plus what breaks if it is wrong, and for any hedge on the linked ticket to be carried forward or explicitly closed out, so reviewers and coding agents have something concrete to attack. Co-Authored-By: Claude Code --- .github/pull_request_template.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index e85a397cbd2..f3d0155b1cd 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -129,6 +129,23 @@ If you're seeing a delay in your PR being merged, ping the LiteLLM Team on [Slac human reader Leave this section empty if there are none --> +## Assumptions + + + ## QA runbook -## Assumptions +## Assumptions Made +## Assumptions Made + + + ## TLDR @@ -129,23 +146,6 @@ If you're seeing a delay in your PR being merged, ping the LiteLLM Team on [Slac human reader Leave this section empty if there are none --> -## Assumptions Made - - - ## QA runbook -## Assumptions Made - - - ## TLDR @@ -144,7 +127,24 @@ If you're seeing a delay in your PR being merged, ping the LiteLLM Team on [Slac - Low: anything else worth noting: naming, cleanup, an edge case nobody hits Nest bullets as deep as helps: hierarchy beats one long line when it makes things clearer to a human reader - Leave this section empty if there are none --> + Leave this section empty if there are none + + Also list, under an "### Assumptions Made" subheading, one bullet per claim this PR relies on + that you did not actually test, each written as the claim followed by what breaks if it turns + out to be wrong. Write it for a reviewer who wants to attack the weakest one, not to reassure + them. Anything that narrows the scope of a bug belongs here unless you ran the test that proves + it: "only reproduces with X on", "no user-observable behavior difference", "this path is + debug-only", "no caller passes that shape". A claim you did verify is not an assumption; put the + proof in Screenshots / Proof of Fix instead. If the linked issue or ticket recorded something as + untested or unverified, carry it into this subsection or say here which run closed it out. Do + not drop it silently + Example: + ### Assumptions Made + - The alert path is debug-only, so the growth cannot be hit at default log level. If a second + caller stringifies the same structure without a level gate, this ships the bug to every + deployment. Not tested: no run with the debug flag off + Include this subheading even when the rest of Caveats is empty; write "None" under it only if + the change rests on nothing untested --> ## QA runbook From 55c5babbc56e404db5fb6b6ea21d1cba29877c45 Mon Sep 17 00:00:00 2001 From: Kerry Lu Date: Fri, 11 Sep 2026 17:33:53 -0700 Subject: [PATCH 140/157] docs(pr-template): fold assumptions guidance into Caveats instructions Co-Authored-By: Claude Code --- .github/pull_request_template.md | 21 +++------------------ 1 file changed, 3 insertions(+), 18 deletions(-) diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 479aac49c05..1a2c81d1f92 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -127,24 +127,9 @@ If you're seeing a delay in your PR being merged, ping the LiteLLM Team on [Slac - Low: anything else worth noting: naming, cleanup, an edge case nobody hits Nest bullets as deep as helps: hierarchy beats one long line when it makes things clearer to a human reader - Leave this section empty if there are none - - Also list, under an "### Assumptions Made" subheading, one bullet per claim this PR relies on - that you did not actually test, each written as the claim followed by what breaks if it turns - out to be wrong. Write it for a reviewer who wants to attack the weakest one, not to reassure - them. Anything that narrows the scope of a bug belongs here unless you ran the test that proves - it: "only reproduces with X on", "no user-observable behavior difference", "this path is - debug-only", "no caller passes that shape". A claim you did verify is not an assumption; put the - proof in Screenshots / Proof of Fix instead. If the linked issue or ticket recorded something as - untested or unverified, carry it into this subsection or say here which run closed it out. Do - not drop it silently - Example: - ### Assumptions Made - - The alert path is debug-only, so the growth cannot be hit at default log level. If a second - caller stringifies the same structure without a level gate, this ships the bug to every - deployment. Not tested: no run with the debug flag off - Include this subheading even when the rest of Caveats is empty; write "None" under it only if - the change rests on nothing untested --> + If you assumed something instead of testing it, e.g. "only reproduces with X on" or "no + user-observable behavior difference", list it here too with what breaks if it is wrong + Leave this section empty if there are none --> ## QA runbook From 6cffb31e5cda6805a6c62ae900a90852ecf308da Mon Sep 17 00:00:00 2001 From: mateo Date: Sat, 12 Sep 2026 00:36:20 +0000 Subject: [PATCH 141/157] feat(model_prices): add DeepSeek V4.1 Flash on Fireworks Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 34 +++++++++++++++++++ model_prices_and_context_window.json | 34 +++++++++++++++++++ .../test_fireworks_serverless_model_costs.py | 11 ++++-- 3 files changed, 77 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index aadd9bd3028..7fa09951eae 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -57695,6 +57695,23 @@ "supports_tool_choice": true, "supports_vision": false }, + "fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash": { + "cache_read_input_token_cost": 7e-09, + "input_cost_per_token": 2.2e-07, + "litellm_provider": "fireworks_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 6.6e-07, + "source": "https://fireworks.ai/models/deepseek-ai/deepseek-v4p1-flash", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, "input_cost_per_token": 2.2e-07, @@ -57745,6 +57762,23 @@ "supports_tool_choice": true, "supports_vision": false }, + "fireworks_ai/deepseek-v4p1-flash": { + "cache_read_input_token_cost": 7e-09, + "input_cost_per_token": 2.2e-07, + "litellm_provider": "fireworks_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 6.6e-07, + "source": "https://fireworks.ai/models/deepseek-ai/deepseek-v4p1-flash", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "fireworks_ai/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, "input_cost_per_token": 2.2e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index aadd9bd3028..7fa09951eae 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -57695,6 +57695,23 @@ "supports_tool_choice": true, "supports_vision": false }, + "fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash": { + "cache_read_input_token_cost": 7e-09, + "input_cost_per_token": 2.2e-07, + "litellm_provider": "fireworks_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 6.6e-07, + "source": "https://fireworks.ai/models/deepseek-ai/deepseek-v4p1-flash", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, "input_cost_per_token": 2.2e-07, @@ -57745,6 +57762,23 @@ "supports_tool_choice": true, "supports_vision": false }, + "fireworks_ai/deepseek-v4p1-flash": { + "cache_read_input_token_cost": 7e-09, + "input_cost_per_token": 2.2e-07, + "litellm_provider": "fireworks_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 6.6e-07, + "source": "https://fireworks.ai/models/deepseek-ai/deepseek-v4p1-flash", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "fireworks_ai/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, "input_cost_per_token": 2.2e-07, diff --git a/tests/test_litellm/test_fireworks_serverless_model_costs.py b/tests/test_litellm/test_fireworks_serverless_model_costs.py index 701938f5677..b8d36b996df 100644 --- a/tests/test_litellm/test_fireworks_serverless_model_costs.py +++ b/tests/test_litellm/test_fireworks_serverless_model_costs.py @@ -62,11 +62,18 @@ TWIN_PINNED_PRICES = { "cache_read_input_token_cost": 7e-09, "output_cost_per_token": 6.6e-07, }, + "deepseek-v4p1-flash": { + "input_cost_per_token": 2.2e-07, + "cache_read_input_token_cost": 7e-09, + "output_cost_per_token": 6.6e-07, + "supports_vision": True, + "max_output_tokens": 393216, + }, } -def test_deepseek_v4_flash_0731_twins_pin_published_pricing(model_data): - """Both 0731 entries carry the price published at docs.fireworks.ai/serverless/pricing.""" +def test_deepseek_v4_flash_twins_pin_published_pricing(model_data): + """Both entries of each Flash twin pair carry the price published at docs.fireworks.ai/serverless/pricing.""" for bare_suffix, expected in TWIN_PINNED_PRICES.items(): for key in ( f"fireworks_ai/{bare_suffix}", From bf146e2caced6520eee71381b4ede17ba3b0d272 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 17:37:35 -0700 Subject: [PATCH 142/157] fix(otel v2): name Langfuse traces from the langfuse_trace_name header or metadata.trace_name (#40793) * fix(otel v2): name Langfuse traces from the langfuse_trace_name header or metadata.trace_name Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(otel v2): type the named-request helper in the Langfuse logger tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/otel/langfuse_logger.py | 19 ++++- litellm/integrations/otel/logger.py | 7 +- litellm/integrations/otel/mappers/langfuse.py | 2 + litellm/integrations/otel/model/metadata.py | 24 ++++++ litellm/integrations/otel/model/payloads.py | 3 + .../integrations/otel/test_langfuse_logger.py | 76 ++++++++++++++++++- .../otel/test_otel_v2_sources_of_truth.py | 32 ++++++++ .../otel/test_otel_v2_vendor_mappers.py | 5 ++ 8 files changed, 160 insertions(+), 8 deletions(-) diff --git a/litellm/integrations/otel/langfuse_logger.py b/litellm/integrations/otel/langfuse_logger.py index ed47533e700..f8fd417392f 100644 --- a/litellm/integrations/otel/langfuse_logger.py +++ b/litellm/integrations/otel/langfuse_logger.py @@ -3,7 +3,12 @@ from typing import TYPE_CHECKING, Final from litellm._logging import verbose_logger from litellm.integrations.otel.logger import OpenTelemetryV2 -from litellm.integrations.otel.mappers.langfuse import LANGFUSE_OBSERVATION_INPUT, LANGFUSE_OBSERVATION_OUTPUT +from litellm.integrations.otel.mappers.langfuse import ( + LANGFUSE_OBSERVATION_INPUT, + LANGFUSE_OBSERVATION_OUTPUT, + LANGFUSE_TRACE_NAME, +) +from litellm.integrations.otel.model.metadata import caller_trace_name from litellm.integrations.otel.model.request_io import request_input, response_output, stream_output from litellm.integrations.otel.plumbing.context import request_root_span @@ -13,6 +18,18 @@ if TYPE_CHECKING: class LangfuseOpenTelemetryV2(OpenTelemetryV2): + """Names the trace from the request. Langfuse reads ``langfuse.trace.name`` off the root observation, + and the proxy's root span is still recording when the LLM call starts.""" + + def log_pre_api_call(self, model: str, messages: object, kwargs: Mapping[str, object]) -> None: + root: Final = request_root_span() + name: Final = caller_trace_name(kwargs) + if root is not None and root.is_recording() and name is not None: + root.set_attribute(LANGFUSE_TRACE_NAME, name) + super().log_pre_api_call(model, messages, kwargs) + + +class LangfuseContentOpenTelemetryV2(LangfuseOpenTelemetryV2): """Stamps the request's input and output on the root observation while it is still recording. Langfuse shows a trace's input and output from its root observation. The proxy's root span ends diff --git a/litellm/integrations/otel/logger.py b/litellm/integrations/otel/logger.py index 630aa313dc9..9ac748b231c 100644 --- a/litellm/integrations/otel/logger.py +++ b/litellm/integrations/otel/logger.py @@ -554,6 +554,7 @@ class OpenTelemetryV2(CustomLogger): capture_content=self.config.capture_span_content, time_to_first_chunk_seconds=call.time_to_first_chunk_seconds, request_route=request_root_http_route(), + trace_name=call.trace_name, ) end_time_ns: Final = to_ns(end_time) if carrier is not None and carrier.span is not None: @@ -984,8 +985,8 @@ def build_otel_v2_logger( def _logger_class(config: OpenTelemetryV2Config) -> type[OpenTelemetryV2]: - if "langfuse" not in config.mapper_names or not config.capture_span_content: + if "langfuse" not in config.mapper_names: return OpenTelemetryV2 - from litellm.integrations.otel.langfuse_logger import LangfuseOpenTelemetryV2 + from litellm.integrations.otel.langfuse_logger import LangfuseContentOpenTelemetryV2, LangfuseOpenTelemetryV2 - return LangfuseOpenTelemetryV2 + return LangfuseContentOpenTelemetryV2 if config.capture_span_content else LangfuseOpenTelemetryV2 diff --git a/litellm/integrations/otel/mappers/langfuse.py b/litellm/integrations/otel/mappers/langfuse.py index 01063d85355..98ff0f155a1 100644 --- a/litellm/integrations/otel/mappers/langfuse.py +++ b/litellm/integrations/otel/mappers/langfuse.py @@ -28,6 +28,7 @@ from litellm.integrations.otel.model.payloads import ( LANGFUSE_OBSERVATION_INPUT: Final = "langfuse.observation.input" LANGFUSE_OBSERVATION_OUTPUT: Final = "langfuse.observation.output" +LANGFUSE_TRACE_NAME: Final = "langfuse.trace.name" class LangfuseMapper: @@ -36,6 +37,7 @@ class LangfuseMapper: "langfuse.observation.model.name": lambda d: d.request_model or None, "langfuse.observation.metadata.provider": lambda d: d.provider or None, "langfuse.observation.id": lambda d: d.identity.call_id or None, + LANGFUSE_TRACE_NAME: lambda d: d.trace_name or None, "langfuse.trace.metadata.team_id": lambda d: d.identity.team_id or None, "langfuse.trace.metadata.team_alias": lambda d: d.identity.team_alias or None, } diff --git a/litellm/integrations/otel/model/metadata.py b/litellm/integrations/otel/model/metadata.py index ee116aca46b..cc81b689708 100644 --- a/litellm/integrations/otel/model/metadata.py +++ b/litellm/integrations/otel/model/metadata.py @@ -48,6 +48,8 @@ from litellm.integrations.otel.model.utils import as_str, to_seconds if TYPE_CHECKING: from litellm.types.utils import StandardLoggingPayload +LANGFUSE_TRACE_NAME_HEADER: Final = "langfuse_trace_name" + @dataclass(frozen=True) class RequestIdentity: @@ -215,6 +217,7 @@ class LLMCallEvent: # needs to be reasonable for a span that never gets closed (a leak). provisional_span_name: str time_to_first_chunk_seconds: float | None + trace_name: str | None @classmethod def from_dict(cls, kwargs: Mapping[str, Any]) -> LLMCallEvent: @@ -231,9 +234,30 @@ class LLMCallEvent: upstream_started=kwargs.get("api_call_start_time") is not None, provisional_span_name=f"{operation.value} {model}".strip(), time_to_first_chunk_seconds=time_to_first_chunk_seconds(kwargs), + trace_name=caller_trace_name(kwargs), ) +def caller_trace_name(kwargs: Mapping[str, object]) -> str | None: + request: Final = _as_str_mapping(kwargs.get("litellm_params")) + if request is None: + return None + proxy_request: Final = _as_str_mapping(request.get("proxy_server_request")) + headers: Final = _as_str_mapping(proxy_request.get("headers")) if proxy_request is not None else None + from_header: Final = as_str(headers.get(LANGFUSE_TRACE_NAME_HEADER)) if headers is not None else None + if from_header: + return from_header + return next( + ( + name + for key in ("metadata", "litellm_metadata") + if (metadata := _as_str_mapping(request.get(key))) is not None + and (name := as_str(metadata.get("trace_name"))) + ), + None, + ) + + def time_to_first_chunk_seconds(kwargs: Mapping[str, Any]) -> float | None: """Seconds from the upstream request being issued (``api_call_start_time``) to the first streamed chunk (``completion_start_time``); ``None`` for diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index d0959a6c2e9..c11c4a7a27d 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -387,6 +387,7 @@ class LLMCallSpanData: output_type: GenAIOutputType | None = None call_type: str | None = None request_route: str | None = None + trace_name: str | None = None @classmethod def from_standard_logging_payload( @@ -395,6 +396,7 @@ class LLMCallSpanData: capture_content: bool = False, time_to_first_chunk_seconds: float | None = None, request_route: str | None = None, + trace_name: str | None = None, ) -> LLMCallSpanData: params: Final = cast(Mapping[str, object], payload.get("model_parameters") or {}) # The single parse of the request's metadata — the request-vs-provider @@ -436,6 +438,7 @@ class LLMCallSpanData: output_type=resolve_output_type(call_type), call_type=call_type or None, request_route=request_route or context.identity.request_route, + trace_name=trace_name, ) diff --git a/tests/test_litellm/integrations/otel/test_langfuse_logger.py b/tests/test_litellm/integrations/otel/test_langfuse_logger.py index 8f93a9a564f..8db84b090a0 100644 --- a/tests/test_litellm/integrations/otel/test_langfuse_logger.py +++ b/tests/test_litellm/integrations/otel/test_langfuse_logger.py @@ -1,9 +1,9 @@ -"""Tests for ``LangfuseOpenTelemetryV2``: the root observation's input and output are stamped from the -request-task hooks, while the root span is still recording, so Langfuse can show them on the trace.""" +"""Tests for the Langfuse OTel v2 loggers: the trace name and the root observation's input and output are +stamped from the request task while the root span is still recording, so Langfuse can show them on the trace.""" import asyncio import json -from collections.abc import AsyncIterator, Sequence +from collections.abc import AsyncIterator, Mapping, Sequence from typing import Final import pytest @@ -14,7 +14,7 @@ from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanE import litellm # noqa: E402 from litellm.caching.dual_cache import DualCache # noqa: E402 -from litellm.integrations.otel.logger import build_otel_v2_logger # noqa: E402 +from litellm.integrations.otel.logger import OpenTelemetryV2, build_otel_v2_logger # noqa: E402 from litellm.integrations.otel.model.config import OpenTelemetryV2Config, is_otel_v2_enabled # noqa: E402 from litellm.integrations.otel.model.spans import LITELLM_PROXY_REQUEST_SPAN_NAME, SpanRole # noqa: E402 from litellm.integrations.otel.plumbing import context as otel_context # noqa: E402 @@ -41,6 +41,7 @@ from litellm.types.utils import ( # noqa: E402 INPUT_ATTR: Final = "langfuse.observation.input" OUTPUT_ATTR: Final = "langfuse.observation.output" +TRACE_NAME_ATTR: Final = "langfuse.trace.name" CHAT_DATA: Final = {"model": "gpt-5.4-mini", "messages": [{"role": "user", "content": "ping"}]} @@ -306,6 +307,73 @@ def test_unrenderable_output_never_raises_into_the_request(): assert INPUT_ATTR not in attrs and OUTPUT_ATTR not in attrs +def _run_named_request( + logger: OpenTelemetryV2, exporter: InMemorySpanExporter, litellm_params: Mapping[str, object] +) -> tuple[Mapping[str, object], Mapping[str, object]]: + response: Final = ModelResponse(choices=[Choices(message=Message(role="assistant", content="pong"))]) + root: Final = _start_root(logger) + logger.log_pre_api_call( + model="gpt-5.4-mini", messages=[], kwargs={"litellm_call_id": "call_1", "litellm_params": litellm_params} + ) + root.end() + payload: Final = { + "call_type": "acompletion", + "custom_llm_provider": "openai", + "model": "gpt-5.4-mini", + "messages": CHAT_DATA["messages"], + "response": response.model_dump(), + "status": "success", + "litellm_call_id": "call_1", + "metadata": {}, + "hidden_params": {}, + } + asyncio.run( + logger.async_log_success_event( + {"standard_logging_object": payload, "litellm_params": litellm_params}, response, None, None + ) + ) + generation: Final = next( + span for span in exporter.get_finished_spans() if span.name != LITELLM_PROXY_REQUEST_SPAN_NAME + ) + return _root_attrs(exporter), dict(generation.attributes or {}) + + +@pytest.mark.parametrize("capture", ["span_only", "no_content"]) +def test_langfuse_trace_name_header_names_the_root_and_the_generation_over_body_metadata(capture): + logger, exporter = _logger(capture=capture) + + root_attrs, generation_attrs = _run_named_request( + logger, + exporter, + { + "metadata": {"trace_name": "from-body"}, + "proxy_server_request": {"headers": {"langfuse_trace_name": "from-header"}}, + }, + ) + + assert root_attrs[TRACE_NAME_ATTR] == "from-header" + assert generation_attrs[TRACE_NAME_ATTR] == "from-header" + + +def test_body_metadata_trace_name_names_the_root_and_the_generation(): + logger, exporter = _logger() + + root_attrs, generation_attrs = _run_named_request( + logger, exporter, {"metadata": {"trace_name": "from-body"}, "proxy_server_request": {"headers": {}}} + ) + + assert root_attrs[TRACE_NAME_ATTR] == "from-body" + assert generation_attrs[TRACE_NAME_ATTR] == "from-body" + + +def test_unnamed_request_leaves_the_trace_name_off_both_spans(): + logger, exporter = _logger() + + root_attrs, generation_attrs = _run_named_request(logger, exporter, {"proxy_server_request": {"headers": {}}}) + + assert TRACE_NAME_ATTR not in root_attrs and TRACE_NAME_ATTR not in generation_attrs + + @pytest.mark.parametrize( ("capture", "mappers"), [("no_content", ("genai", "langfuse")), ("span_only", ("genai",))], diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py b/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py index addadf8e598..8baf9310538 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py @@ -28,6 +28,7 @@ from litellm.integrations.otel import ( ) from litellm.integrations.otel.mappers.genai import GenAIMapper from litellm.integrations.otel.model import spans as spans_mod +from litellm.integrations.otel.model.metadata import LLMCallEvent, caller_trace_name from litellm.integrations.otel.model.payloads import ( LLMCallSpanData, RequestIdentity, @@ -722,6 +723,37 @@ def test_request_identity_falls_back_to_legacy_team_keys(): assert ident.team_alias == "legacy" +@pytest.mark.parametrize( + ("request_data", "expected"), + [ + ({"proxy_server_request": {"headers": {"langfuse_trace_name": "from-header"}}}, "from-header"), + ({"metadata": {"trace_name": "from-body"}}, "from-body"), + ({"litellm_metadata": {"trace_name": "from-anthropic-body"}}, "from-anthropic-body"), + ( + { + "proxy_server_request": {"headers": {"langfuse_trace_name": "from-header"}}, + "metadata": {"trace_name": "from-body"}, + }, + "from-header", + ), + ({"proxy_server_request": {"headers": {"langfuse_trace_name": ""}}, "metadata": {"trace_name": "body"}}, "body"), + ({"proxy_server_request": {"headers": {}}, "metadata": {"user_api_key_team_id": "t1"}}, None), + ({}, None), + ], + ids=["header", "body", "anthropic-body", "header-beats-body", "blank-header-falls-through", "neither", "empty"], +) +def test_caller_trace_name_prefers_the_langfuse_header_over_body_metadata(request_data, expected): + assert caller_trace_name({"litellm_params": request_data}) == expected + assert LLMCallEvent.from_dict({"litellm_params": request_data}).trace_name == expected + + +def test_llm_span_data_carries_the_caller_trace_name(): + data: Final = LLMCallSpanData.from_standard_logging_payload(_sample_payload(), trace_name="nightly-eval") + + assert data.trace_name == "nightly-eval" + assert LLMCallSpanData.from_standard_logging_payload(_sample_payload()).trace_name is None + + def test_llm_span_carries_proxy_request_route(): """The LLM span records the proxy route the request arrived on, so it can be filtered by endpoint (``/v1/responses`` vs ``/v1/chat/completions``) without diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_vendor_mappers.py b/tests/test_litellm/integrations/otel/test_otel_v2_vendor_mappers.py index 94cb79f53b8..bcdda93383a 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_vendor_mappers.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_vendor_mappers.py @@ -134,6 +134,11 @@ def test_langfuse_mapper_observation_attrs(): assert attrs["langfuse.trace.metadata.team_id"] == "t1" +def test_langfuse_mapper_names_the_trace_from_the_caller(): + assert LangfuseMapper().map(_llm_call(trace_name="nightly-eval"))["langfuse.trace.name"] == "nightly-eval" + assert "langfuse.trace.name" not in LangfuseMapper().map(_llm_call(trace_name=None)) + + def test_langfuse_mapper_skips_when_no_messages(): data = _llm_call(messages_in=(), choices_out=()) attrs = LangfuseMapper().map(data) From 4a2edf17022d2dee9e0c377ddafecfa83322d623 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 17:47:35 -0700 Subject: [PATCH 143/157] feat(ui): link the Organization cell on the Teams page (#40749) The Teams table showed a team's organization as plain text, so getting from a team to the org that owns it meant copying the alias and searching the Organizations page by hand. It now uses the same link helper the key tables use, so the cell points at the org detail page. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 --- .../components/TeamsPage/TeamsTable.test.tsx | 24 +++++++++++++++++++ .../components/TeamsPage/teamTableColumns.tsx | 5 ++-- 2 files changed, 27 insertions(+), 2 deletions(-) diff --git a/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.test.tsx b/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.test.tsx index d3b9a55a7d5..95958994775 100644 --- a/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.test.tsx +++ b/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.test.tsx @@ -34,6 +34,10 @@ vi.mock("@/app/(dashboard)/hooks/teams/useTeams", () => ({ teamsTableKeys: { all: ["teamsTable"] }, })); +vi.mock("next/navigation", () => ({ + useRouter: () => ({ push: vi.fn() }), +})); + vi.mock("@/app/(dashboard)/hooks/organizations/useOrganizations", () => ({ useOrganizations: vi.fn().mockReturnValue({ data: [{ organization_id: "org-1", organization_alias: "Test Organization" }], @@ -306,10 +310,30 @@ describe("column rendering details", () => { }); }); + it("points the organization cell at the org's detail page, aliased or not", async () => { + mockUseTeamsTable.mockReturnValue( + teamsResult([ + { ...mockTeam, team_id: "a", organization_id: "org-1" }, + { ...mockTeam, team_id: "b", team_alias: "Orphan Team", organization_id: "org-unknown" }, + ]), + ); + renderTable(); + + expect(await screen.findByRole("link", { name: "Test Organization" })).toHaveAttribute( + "href", + "/ui/organizations?org=org-1", + ); + expect(screen.getByRole("link", { name: "org-unknown" })).toHaveAttribute( + "href", + "/ui/organizations?org=org-unknown", + ); + }); + it("renders an em dash for a team with no organization", () => { mockUseTeamsTable.mockReturnValue(teamsResult([{ ...mockTeam, organization_id: null as unknown as string }])); renderTable(); expect(screen.getByText("—")).toBeInTheDocument(); + expect(screen.queryByRole("link", { name: /organization/i })).not.toBeInTheDocument(); }); it("falls back to keys.length when keys_count is absent", () => { diff --git a/ui/litellm-dashboard/src/components/TeamsPage/teamTableColumns.tsx b/ui/litellm-dashboard/src/components/TeamsPage/teamTableColumns.tsx index da4948dd892..84369a58307 100644 --- a/ui/litellm-dashboard/src/components/TeamsPage/teamTableColumns.tsx +++ b/ui/litellm-dashboard/src/components/TeamsPage/teamTableColumns.tsx @@ -15,6 +15,7 @@ import { } from "@/components/ui/dropdown-menu"; import { Skeleton } from "@/components/ui/skeleton"; import { cn } from "@/lib/cva.config"; +import { orgDetailHref } from "@/utils/entityLinks"; import { copyToClipboard, formatNumberWithCommas } from "@/utils/dataUtils"; import { Team } from "../key_team_helpers/key_list"; @@ -183,8 +184,8 @@ export const getTeamTableColumns = ({ const displayValue = org?.organization_alias || orgId; const width = info.cell.column.getSize(); return ( - - {displayValue} + + ); }, From 1be930664f5273a77fbbd4576f35d6468b029dec Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 17:48:44 -0700 Subject: [PATCH 144/157] feat(ui): link the User ID and Team ID cells on the Memory page (#40752) Both columns rendered as dead pills, so tracing a memory row back to its owner meant copying an id into another page's search box. IdCell grows an href prop that turns the pill into a client-routed link, and the Memory columns pass the shared entityLinks helpers so the proxy admin and dashboard sentinels stay unlinked. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 --- .../memory/_components/MemoryTable.test.tsx | 19 +++++++++++ .../memory/_components/MemoryTableColumns.tsx | 11 ++++-- .../shared/table_cells/id_cell.test.tsx | 23 +++++++++++++ .../components/shared/table_cells/id_cell.tsx | 34 +++++++++++++++++-- 4 files changed, 83 insertions(+), 4 deletions(-) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTable.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTable.test.tsx index 984b8135466..5785d0b37a1 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTable.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTable.test.tsx @@ -8,6 +8,8 @@ import { MemoryRow } from "@/components/networking"; import { MemoryTable } from "./MemoryTable"; +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + const makeMemory = (overrides: Partial = {}): MemoryRow => ({ memory_id: "mem-1", key: "user:profile", @@ -36,6 +38,23 @@ const baseProps = { }; describe("MemoryTable", () => { + it("links the User ID and Team ID cells to their detail pages", () => { + render(); + + expect(screen.getByRole("link", { name: "user-42" })).toHaveAttribute("href", "/ui/users?user=user-42"); + expect(screen.getByRole("link", { name: "team-7" })).toHaveAttribute("href", "/ui/teams?team=team-7"); + }); + + it("leaves the proxy admin and dashboard sentinels unlinked", () => { + const sentinelRow = makeMemory({ user_id: "default_user_id", team_id: "litellm-dashboard" }); + render(); + + expect(screen.getByText("default_user_id")).toBeInTheDocument(); + expect(screen.getByText("litellm-dashboard")).toBeInTheDocument(); + expect(screen.queryByRole("link", { name: "default_user_id" })).not.toBeInTheDocument(); + expect(screen.queryByRole("link", { name: "litellm-dashboard" })).not.toBeInTheDocument(); + }); + it("renders every column header", () => { render(); for (const header of ["ID", "Name", "Preview", "User ID", "Team ID", "Updated"]) { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTableColumns.tsx b/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTableColumns.tsx index 6b2a6b08704..62b8ec624fa 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTableColumns.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/memory/_components/MemoryTableColumns.tsx @@ -14,6 +14,7 @@ import { DropdownMenuTrigger, } from "@/components/ui/dropdown-menu"; import { cn } from "@/lib/cva.config"; +import { teamDetailHref, userDetailHref } from "@/utils/entityLinks"; interface MemoryRowActionsProps { row: MemoryRow; @@ -109,7 +110,10 @@ export const getMemoryTableColumns = ({ header: "User ID", size: 160, enableSorting: false, - cell: ({ row }) => , + cell: ({ row }) => { + const userId = row.original.user_id; + return ; + }, }, { id: "team_id", @@ -118,7 +122,10 @@ export const getMemoryTableColumns = ({ header: "Team ID", size: 160, enableSorting: false, - cell: ({ row }) => , + cell: ({ row }) => { + const teamId = row.original.team_id; + return ; + }, }, { id: "updated_at", diff --git a/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.test.tsx b/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.test.tsx index c715210fdde..c6720ecc7b1 100644 --- a/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.test.tsx +++ b/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.test.tsx @@ -4,6 +4,10 @@ import { describe, expect, it, vi } from "vitest"; import { IdCell } from "./id_cell"; +const { routerPushMock } = vi.hoisted(() => ({ routerPushMock: vi.fn() })); + +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: routerPushMock }) })); + const { copyToClipboardMock } = vi.hoisted(() => ({ copyToClipboardMock: vi.fn() })); vi.mock("@/utils/dataUtils", async (importOriginal) => ({ @@ -83,4 +87,23 @@ describe("IdCell", () => { render(); expect(screen.getByTestId("key-id-cell")).toHaveTextContent("k-1"); }); + + it("renders the id as a link and routes client side when href is set", async () => { + const user = userEvent.setup(); + render(); + + const link = screen.getByRole("link", { name: "user-42" }); + expect(link).toHaveAttribute("href", "/ui/users?user=user-42"); + expect(link).toHaveClass("cursor-pointer"); + + await user.click(link); + expect(routerPushMock).toHaveBeenCalledWith("/ui/users?user=user-42"); + }); + + it("stays plain text when href is undefined", () => { + render(); + + expect(screen.getByText("default_user_id").tagName).toBe("SPAN"); + expect(screen.queryByRole("link")).not.toBeInTheDocument(); + }); }); diff --git a/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.tsx b/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.tsx index 33c7f835e64..47d0d750223 100644 --- a/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.tsx +++ b/ui/litellm-dashboard/src/components/shared/table_cells/id_cell.tsx @@ -3,6 +3,7 @@ import { Copy } from "lucide-react"; import * as React from "react"; +import { useEntityLinkClick } from "@/components/shared/EntityLink"; import { cn } from "@/lib/cva.config"; import { copyToClipboard } from "@/utils/dataUtils"; @@ -13,6 +14,7 @@ export type IdCellVariant = "pill" | "plain"; interface IdCellProps { value: string | null | undefined; variant?: IdCellVariant; + href?: string; onClick?: (value: string) => void; copyable?: boolean; copyLabel?: string; @@ -38,6 +40,7 @@ const VARIANT_CLASS: Record export function IdCell({ value, variant = "pill", + href, onClick, copyable = false, copyLabel = "Copy ID", @@ -52,16 +55,17 @@ export function IdCell({ return {fallback}; } + const linked = !!href && !disabled; const clickable = !!onClick && !disabled; const classes = cn( VARIANT_CLASS[variant].base, - clickable && VARIANT_CLASS[variant].clickable, + (linked || clickable) && VARIANT_CLASS[variant].clickable, truncate && "block max-w-[15ch] truncate", disabled && "opacity-50", className, ); - const idElement = clickable ? ( + const unlinkedElement = clickable ? ( @@ -71,6 +75,14 @@ export function IdCell({ ); + const idElement = linked ? ( + + {value} + + ) : ( + unlinkedElement + ); + const withTooltip = ; if (!copyable) { @@ -94,3 +106,21 @@ export function IdCell({ ); } + +interface IdLinkProps extends React.ComponentPropsWithoutRef<"a"> { + href: string; + dataTestId?: string; +} + +const IdLink = React.forwardRef(function IdLink( + { href, dataTestId, children, ...props }, + ref, +) { + const handleClick = useEntityLinkClick(href); + + return ( + + {children} + + ); +}); From 06964e5603ecce13b0e4aabcbcab5d1382c4c6e5 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 17:48:57 -0700 Subject: [PATCH 145/157] feat(ui): link the User ID, Created By and Deleted By cells on Deleted Keys (#40750) * feat(ui): link the User ID, Created By and Deleted By cells on Deleted Keys All three columns rendered as plain text, so auditing a deleted key meant copying an id into the Users page search box. Route them through IdentityCell with userDetailHref, which keeps the proxy admin placeholder unlinked. User Email and Team Alias stay as they are: the deleted key table has no column for either, so the API never populates them. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 * test(ui): mount a router mock for the Deleted Keys page test The page test renders the table, and the newly linked cells call useRouter, which throws without an App Router mounted. Matches how the other 35 test files in the suite stub next/navigation. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 --- .../DeletedKeysPage/DeletedKeysPage.test.tsx | 2 ++ .../DeletedKeysTable.test.tsx | 18 +++++++++++++++++ .../DeletedKeysTableColumns.tsx | 20 +++++++++++++++---- 3 files changed, 36 insertions(+), 4 deletions(-) diff --git a/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysPage.test.tsx b/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysPage.test.tsx index ee2f42ad85b..31cd407a5e6 100644 --- a/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysPage.test.tsx +++ b/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysPage.test.tsx @@ -5,6 +5,8 @@ import { renderWithProviders } from "../../../tests/test-utils"; import DeletedKeysPage from "./DeletedKeysPage"; import { useDeletedKeys, DeletedKeyResponse } from "@/app/(dashboard)/hooks/keys/useKeys"; +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + vi.mock("@/app/(dashboard)/hooks/keys/useKeys", () => ({ useDeletedKeys: vi.fn(), })); diff --git a/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTable.test.tsx b/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTable.test.tsx index 7e30ef2c135..dd8007be264 100644 --- a/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTable.test.tsx +++ b/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTable.test.tsx @@ -5,6 +5,8 @@ import { renderWithProviders } from "../../../../tests/test-utils"; import { DeletedKeysTable } from "./DeletedKeysTable"; import { DeletedKeyResponse } from "@/app/(dashboard)/hooks/keys/useKeys"; +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + const makeDeletedKey = (overrides: Partial = {}): DeletedKeyResponse => ({ token: "sk-1234567890abcdef", @@ -86,3 +88,19 @@ it("should show the empty state when there are no deleted keys", () => { expect(screen.getByText("No deleted keys found")).toBeInTheDocument(); }); + +it("links the owner, creator and deleter cells to their user detail pages", () => { + renderWithProviders(); + + expect(screen.getByRole("link", { name: "user-1" })).toHaveAttribute("href", "/ui/users?user=user-1"); + expect(screen.getByRole("link", { name: "creator-1" })).toHaveAttribute("href", "/ui/users?user=creator-1"); + expect(screen.getByRole("link", { name: "deleter-1" })).toHaveAttribute("href", "/ui/users?user=deleter-1"); +}); + +it("leaves the default_user_id placeholder unlinked", () => { + const placeholderKey = makeDeletedKey({ user_id: "default_user_id", created_by: "default_user_id" }); + renderWithProviders(); + + expect(screen.getAllByText("default_user_id")).toHaveLength(2); + expect(screen.queryByRole("link", { name: "default_user_id" })).not.toBeInTheDocument(); +}); diff --git a/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTableColumns.tsx b/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTableColumns.tsx index aa7d6380cc3..32185f867b4 100644 --- a/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTableColumns.tsx +++ b/ui/litellm-dashboard/src/components/DeletedKeysPage/DeletedKeysTable/DeletedKeysTableColumns.tsx @@ -3,8 +3,9 @@ import { ColumnDef } from "@tanstack/react-table"; import { DataTableSortHeader } from "@/components/shared/DataTable"; -import { DateCell, IdCell, MoneyCell } from "@/components/shared/table_cells"; +import { DateCell, IdCell, IdentityCell, MoneyCell } from "@/components/shared/table_cells"; import { DeletedKeyResponse } from "@/app/(dashboard)/hooks/keys/useKeys"; +import { userDetailHref } from "@/utils/entityLinks"; function TruncatedTextCell({ value }: { value: string | null | undefined }) { if (!value) { @@ -17,6 +18,17 @@ function TruncatedTextCell({ value }: { value: string | null | undefined }) { ); } +function UserLinkCell({ userId }: { userId: string | null | undefined }) { + if (!userId) { + return -; + } + return ( + + + + ); +} + export const getDeletedKeysTableColumns = (): ColumnDef[] => [ { id: "token", @@ -89,7 +101,7 @@ export const getDeletedKeysTableColumns = (): ColumnDef[] => header: "User ID", size: 120, enableSorting: false, - cell: ({ row }) => , + cell: ({ row }) => , }, { id: "created_at", @@ -107,7 +119,7 @@ export const getDeletedKeysTableColumns = (): ColumnDef[] => header: "Created By", size: 120, enableSorting: false, - cell: ({ row }) => , + cell: ({ row }) => , }, { id: "deleted_at", @@ -125,6 +137,6 @@ export const getDeletedKeysTableColumns = (): ColumnDef[] => header: "Deleted By", size: 120, enableSorting: false, - cell: ({ row }) => , + cell: ({ row }) => , }, ]; From c53f72c764f9c1ae9015df667e350b130ebb8a9f Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 17:49:01 -0700 Subject: [PATCH 146/157] feat(ui): link the Created By cell on the Prompts page (#40753) The column was plain muted text, so finding out who owns a prompt meant copying the id into the Users page search box. Route it through IdentityCell with userDetailHref, which keeps the proxy admin placeholder unlinked. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 --- .../prompts/_components/PromptTable.test.tsx | 11 +++++++++++ .../prompts/_components/PromptTableColumns.tsx | 12 ++++++++++-- 2 files changed, 21 insertions(+), 2 deletions(-) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTable.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTable.test.tsx index edbb897fb2f..009abf93d1f 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTable.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTable.test.tsx @@ -10,6 +10,8 @@ vi.mock("@/components/networking", () => ({ modelHubCall: vi.fn().mockResolvedValue({ data: [] }), })); +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + const mockPrompts: PromptSpec[] = [ { prompt_id: "prompt-newer", @@ -53,6 +55,15 @@ describe("PromptTable", () => { } }); + it("links the Created By cell to the creator's detail page, leaving the placeholder unlinked", () => { + const prompts = [mockPrompts[0], { ...mockPrompts[1], created_by: "default_user_id" }]; + render(); + + expect(screen.getByRole("link", { name: "user-1" })).toHaveAttribute("href", "/ui/users?user=user-1"); + expect(screen.getByText("default_user_id")).toBeInTheDocument(); + expect(screen.queryByRole("link", { name: "default_user_id" })).not.toBeInTheDocument(); + }); + it("should display the empty state when data is empty", () => { render(); expect(screen.getByText("No prompts yet")).toBeInTheDocument(); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTableColumns.tsx b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTableColumns.tsx index f927a6d1486..db0392bbd6e 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTableColumns.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/PromptTableColumns.tsx @@ -17,6 +17,7 @@ import { } from "@/components/ui/dropdown-menu"; import { cn } from "@/lib/cva.config"; import { copyToClipboard } from "@/utils/dataUtils"; +import { userDetailHref } from "@/utils/entityLinks"; import { extractModel, getProviderFromModelHub, ModelGroupInfo } from "./prompt_utils"; @@ -191,9 +192,16 @@ export const getPromptTableColumns = ({ enableSorting: false, cell: ({ row }) => { const createdBy = row.original.created_by; + if (!createdBy) { + return -; + } return ( - - {createdBy || "-"} + + ); }, From b27d2cce77cd1d0ba9662ccb5b2fe9640471f2c6 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Fri, 11 Sep 2026 17:49:07 -0700 Subject: [PATCH 147/157] feat(ui): link the Organization and Deleted By cells on Deleted Teams (#40751) * feat(ui): link the Organization and Deleted By cells on Deleted Teams Both columns rendered as plain text, so tracing a deleted team back to its org or to whoever removed it meant copying an id into another page's search box. Route them through IdentityCell with orgDetailHref and userDetailHref. Team ID stays unlinked because the team itself is gone. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 * test(ui): mount a router mock for the Deleted Teams page test The page test renders the table, and the newly linked cells call useRouter, which throws without an App Router mounted. Matches how the other 35 test files in the suite stub next/navigation. Claude-Session: https://claude.ai/code/session_01NfwfQhamRNnSqgXMUjf3h4 --- .../DeletedTeamsPage.test.tsx | 2 ++ .../DeletedTeamsTable.test.tsx | 20 +++++++++++++ .../DeletedTeamsTableColumns.tsx | 30 ++++++++++++------- 3 files changed, 41 insertions(+), 11 deletions(-) diff --git a/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsPage.test.tsx b/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsPage.test.tsx index 952d8764463..d2d25f5e47e 100644 --- a/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsPage.test.tsx +++ b/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsPage.test.tsx @@ -5,6 +5,8 @@ import { renderWithProviders } from "../../../tests/test-utils"; import DeletedTeamsPage from "./DeletedTeamsPage"; import { useDeletedTeams, DeletedTeam } from "@/app/(dashboard)/hooks/teams/useTeams"; +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + vi.mock("@/app/(dashboard)/hooks/teams/useTeams", () => ({ useDeletedTeams: vi.fn(), })); diff --git a/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTable.test.tsx b/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTable.test.tsx index e166f6b0d1b..837fb583f30 100644 --- a/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTable.test.tsx +++ b/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTable.test.tsx @@ -4,6 +4,8 @@ import { renderWithProviders } from "../../../../tests/test-utils"; import { DeletedTeamsTable } from "./DeletedTeamsTable"; import { DeletedTeam } from "@/app/(dashboard)/hooks/teams/useTeams"; +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + const makeDeletedTeam = (overrides: Partial = {}): DeletedTeam => ({ team_id: "team-1", team_alias: "Test Team", @@ -81,3 +83,21 @@ it("renders the shared pagination footer with the server row count", () => { expect(screen.getByTestId("pagination-prev")).toBeEnabled(); expect(screen.getByTestId("pagination-next")).toBeDisabled(); }); + +it("links the organization and deleted by cells, leaving the deleted team id unlinked", () => { + renderWithProviders( + , + ); + + expect(screen.getByRole("link", { name: "org-1" })).toHaveAttribute("href", "/ui/organizations?org=org-1"); + expect(screen.getByRole("link", { name: "user-1" })).toHaveAttribute("href", "/ui/users?user=user-1"); + expect(screen.queryByRole("link", { name: "team-1" })).not.toBeInTheDocument(); +}); + +it("leaves the default_user_id placeholder unlinked in the deleted by cell", () => { + const team = makeDeletedTeam({ deleted_by: "default_user_id", organization_id: null }); + renderWithProviders(); + + expect(screen.getByText("default_user_id")).toBeInTheDocument(); + expect(screen.queryByRole("link", { name: "default_user_id" })).not.toBeInTheDocument(); +}); diff --git a/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTableColumns.tsx b/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTableColumns.tsx index e36077fd2c3..172f0417027 100644 --- a/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTableColumns.tsx +++ b/ui/litellm-dashboard/src/components/DeletedTeamsPage/DeletedTeamsTable/DeletedTeamsTableColumns.tsx @@ -3,8 +3,20 @@ import { ColumnDef } from "@tanstack/react-table"; import { DataTableSortHeader } from "@/components/shared/DataTable"; -import { DateCell, IdCell, ModelsCell, MoneyCell } from "@/components/shared/table_cells"; +import { DateCell, IdCell, IdentityCell, ModelsCell, MoneyCell } from "@/components/shared/table_cells"; import { DeletedTeam } from "@/app/(dashboard)/hooks/teams/useTeams"; +import { orgDetailHref, userDetailHref } from "@/utils/entityLinks"; + +function EntityCell({ value, href }: { value: string | null | undefined; href: string | undefined }) { + if (!value) { + return -; + } + return ( + + + + ); +} export const getDeletedTeamsTableColumns = (): ColumnDef[] => [ { @@ -78,7 +90,10 @@ export const getDeletedTeamsTableColumns = (): ColumnDef[] => [ header: "Organization", size: 150, enableSorting: false, - cell: ({ row }) => , + cell: ({ row }) => { + const orgId = row.original.organization_id; + return ; + }, }, { id: "deleted_at", @@ -97,15 +112,8 @@ export const getDeletedTeamsTableColumns = (): ColumnDef[] => [ size: 120, enableSorting: false, cell: ({ row }) => { - const value = row.original.deleted_by; - if (!value) { - return -; - } - return ( - - {value} - - ); + const deletedBy = row.original.deleted_by; + return ; }, }, ]; From 115535c3add1f79bf8b5056663c1456a0be5e1cf Mon Sep 17 00:00:00 2001 From: mateo Date: Sat, 12 Sep 2026 00:49:26 +0000 Subject: [PATCH 148/157] test(model_prices): cover Fireworks DeepSeek V4.1 costs Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_fireworks_serverless_model_costs.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tests/test_litellm/test_fireworks_serverless_model_costs.py b/tests/test_litellm/test_fireworks_serverless_model_costs.py index b8d36b996df..e3404e477b7 100644 --- a/tests/test_litellm/test_fireworks_serverless_model_costs.py +++ b/tests/test_litellm/test_fireworks_serverless_model_costs.py @@ -14,6 +14,8 @@ import os import pytest +from litellm import completion_cost +from litellm.types.utils import Choices, Message, ModelResponse, Usage from litellm.utils import get_model_info @@ -56,6 +58,20 @@ def test_bare_fireworks_ids_resolve_through_prefixed_entries(): assert info["max_output_tokens"] == expected["max_output_tokens"] +def test_deepseek_v4p1_flash_twin_costs(): + for model in ( + "fireworks_ai/deepseek-v4p1-flash", + "fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash", + ): + response = ModelResponse( + model=model, + choices=[Choices(index=0, message=Message(role="assistant", content="ok"))], + usage=Usage(prompt_tokens=1000, completion_tokens=1000, total_tokens=2000), + ) + cost = completion_cost(completion_response=response, model=model) + assert cost == pytest.approx(8.8e-04) + + TWIN_PINNED_PRICES = { "deepseek-v4-flash-0731": { "input_cost_per_token": 2.2e-07, From 6c07876dcfe9ce74401526c3a29657848ffb4afe Mon Sep 17 00:00:00 2001 From: mateo Date: Sat, 12 Sep 2026 01:00:11 +0000 Subject: [PATCH 149/157] test(model_prices): use local cost map for Fireworks cost coverage Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/test_litellm/test_fireworks_serverless_model_costs.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_litellm/test_fireworks_serverless_model_costs.py b/tests/test_litellm/test_fireworks_serverless_model_costs.py index e3404e477b7..1303f46e8fa 100644 --- a/tests/test_litellm/test_fireworks_serverless_model_costs.py +++ b/tests/test_litellm/test_fireworks_serverless_model_costs.py @@ -58,7 +58,7 @@ def test_bare_fireworks_ids_resolve_through_prefixed_entries(): assert info["max_output_tokens"] == expected["max_output_tokens"] -def test_deepseek_v4p1_flash_twin_costs(): +def test_deepseek_v4p1_flash_twin_costs(local_model_cost_map): for model in ( "fireworks_ai/deepseek-v4p1-flash", "fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash", From 105dc7710959e63264c99f0f68c581781e254494 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 18:14:21 -0700 Subject: [PATCH 150/157] fix(realtime): probe Azure's GA realtime upstream in health checks when no protocol is pinned --- litellm/realtime_api/main.py | 8 ++--- tests/test_litellm/realtime_api/test_main.py | 36 +++++++++++++++----- 2 files changed, 31 insertions(+), 13 deletions(-) diff --git a/litellm/realtime_api/main.py b/litellm/realtime_api/main.py index d42e7e18b75..44c47af57f4 100644 --- a/litellm/realtime_api/main.py +++ b/litellm/realtime_api/main.py @@ -586,9 +586,7 @@ def _azure_realtime_health_protocol( configured: Final = configured_raw if isinstance(configured_raw, str) else None if configured is not None: return configured, query_params - if query_params is not None: - return "GA", query_params - return "beta", None + return "GA", query_params def _realtime_health_check_auth_headers( @@ -621,8 +619,8 @@ async def _realtime_health_check( api_key: str - api key custom_llm_provider: str - custom llm provider realtime_protocol: Optional[str] - protocol version ("GA"/"v1" for GA path, "beta" for beta path); - None resolves it for Azure from model_params/env, with transcription-only models probing GA - plus intent=transcription the way real calls do + None resolves it for Azure from model_params/env and otherwise probes GA, the upstream a client + without the OpenAI-Beta header is bridged to, with transcription-only models adding intent=transcription Returns: bool - True if connection is successful, False otherwise diff --git a/tests/test_litellm/realtime_api/test_main.py b/tests/test_litellm/realtime_api/test_main.py index 8a9abe819e5..0827bbcdc38 100644 --- a/tests/test_litellm/realtime_api/test_main.py +++ b/tests/test_litellm/realtime_api/test_main.py @@ -266,7 +266,8 @@ def test_transcription_only_detection_rejects_speech_model(local_model_cost_map) @pytest.mark.asyncio -async def test_azure_health_check_keeps_beta_path_for_speech_model(): +async def test_azure_health_check_probes_the_ga_upstream_for_an_unconfigured_speech_model(monkeypatch): + monkeypatch.delenv("LITELLM_AZURE_REALTIME_PROTOCOL", raising=False) connect = _CapturingConnect() with patch("websockets.connect", connect): assert await realtime_main._realtime_health_check( @@ -276,14 +277,18 @@ async def test_azure_health_check_keeps_beta_path_for_speech_model(): api_base="https://my-endpoint.openai.azure.com", api_version="2024-10-01-preview", ) - assert connect.url == ( - "wss://my-endpoint.openai.azure.com/openai/realtime" - "?api-version=2024-10-01-preview&deployment=gpt-4o-realtime-preview" - ) + assert connect.url == "wss://my-endpoint.openai.azure.com/openai/v1/realtime?model=gpt-4o-realtime-preview" + + +_AZURE_BETA_HEALTH_URL: Final = ( + "wss://my-endpoint.openai.azure.com/openai/realtime" + "?api-version=2024-10-01-preview&deployment=gpt-4o-realtime-preview" +) @pytest.mark.asyncio -async def test_azure_health_check_honors_deployment_realtime_protocol(): +async def test_azure_health_check_honors_deployment_realtime_protocol(monkeypatch): + monkeypatch.delenv("LITELLM_AZURE_REALTIME_PROTOCOL", raising=False) connect = _CapturingConnect() with patch("websockets.connect", connect): assert await realtime_main._realtime_health_check( @@ -292,9 +297,24 @@ async def test_azure_health_check_honors_deployment_realtime_protocol(): api_key="fake-key", api_base="https://my-endpoint.openai.azure.com", api_version="2024-10-01-preview", - model_params={"realtime_protocol": "GA"}, + model_params={"realtime_protocol": "beta"}, ) - assert connect.url == "wss://my-endpoint.openai.azure.com/openai/v1/realtime?model=gpt-4o-realtime-preview" + assert connect.url == _AZURE_BETA_HEALTH_URL + + +@pytest.mark.asyncio +async def test_azure_health_check_honors_env_realtime_protocol(monkeypatch): + monkeypatch.setenv("LITELLM_AZURE_REALTIME_PROTOCOL", "beta") + connect = _CapturingConnect() + with patch("websockets.connect", connect): + assert await realtime_main._realtime_health_check( + model="gpt-4o-realtime-preview", + custom_llm_provider="azure", + api_key="fake-key", + api_base="https://my-endpoint.openai.azure.com", + api_version="2024-10-01-preview", + ) + assert connect.url == _AZURE_BETA_HEALTH_URL class _ConnectThatStopsAfterCapturingTheUrl: From f84f986b4ee571c8b3db18a560e007801379a933 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 11 Sep 2026 18:16:33 -0700 Subject: [PATCH 151/157] fix(guardrails): keep post_call guardrail info on streamed chat completions (#40806) * fix(guardrails): sync logging_obj guardrail info on every record so post_call entries survive streamed chat completions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(guardrails): hoist regression test imports to module scope Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/custom_guardrail.py | 1 + .../integrations/test_custom_guardrail.py | 61 ++++++++++++++++++- 2 files changed, 61 insertions(+), 1 deletion(-) diff --git a/litellm/integrations/custom_guardrail.py b/litellm/integrations/custom_guardrail.py index bb54767edef..8a976a966a6 100644 --- a/litellm/integrations/custom_guardrail.py +++ b/litellm/integrations/custom_guardrail.py @@ -1217,6 +1217,7 @@ class CustomGuardrail(CustomLogger): _, metadata_bucket = get_or_create_metadata_bucket(request_data) _append_guardrail_info(metadata_bucket) + _sync_guardrail_info_to_logging_obj(request_data, request_data.get("litellm_logging_obj")) _guardrail_self_recorded.set(True) diff --git a/tests/test_litellm/integrations/test_custom_guardrail.py b/tests/test_litellm/integrations/test_custom_guardrail.py index 1644d78ae37..bb4822eae57 100644 --- a/tests/test_litellm/integrations/test_custom_guardrail.py +++ b/tests/test_litellm/integrations/test_custom_guardrail.py @@ -1,4 +1,5 @@ import asyncio +import datetime as dt from typing import TYPE_CHECKING, ClassVar, Final, Literal, Optional from unittest.mock import AsyncMock @@ -9,9 +10,16 @@ from litellm.integrations.custom_guardrail import ( CustomGuardrail, log_guardrail_information, ) +from litellm.litellm_core_utils.litellm_logging import Logging from litellm.proxy._types import CallTypes, UserAPIKeyAuth from litellm.types.guardrails import GuardrailEventHooks, Mode -from litellm.types.utils import GenericGuardrailAPIInputs, GuardrailTracingDetail +from litellm.types.utils import ( + Choices, + GenericGuardrailAPIInputs, + GuardrailTracingDetail, + Message, + ModelResponse, +) if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj @@ -2345,6 +2353,57 @@ class TestUndecoratedApplyGuardrailIsLogged: assert _Labelled.seen_label == "docs-style" + @pytest.mark.asyncio + async def test_post_call_recorded_outside_decorator_reaches_standard_logging_object(self): + """LIT-7608 regression: the auto-wrapped pre_call apply_guardrail copies the request bucket + into logging_obj.litellm_params["metadata"]. A post_call entry recorded later without the + decorator (the Bedrock streaming hook) must not be shadowed by that stale copy.""" + messages: Final = [{"role": "user", "content": "hello there"}] + litellm_metadata: Final[dict] = {"user_api_key_user_id": "u1"} + logging_obj: Final = Logging( + model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", + messages=messages, + stream=True, + call_type=CallTypes.acompletion.value, + start_time=dt.datetime.now(), + litellm_call_id="call-1", + function_id="fn-1", + ) + logging_obj.update_environment_variables( + litellm_params={"litellm_metadata": litellm_metadata}, + optional_params={}, + model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", + custom_llm_provider="bedrock", + ) + request_data: Final = { + "model": "bedrock-haiku", + "messages": messages, + "litellm_metadata": litellm_metadata, + "litellm_logging_obj": logging_obj, + } + guardrail: Final = _UndecoratedGuardrail(guardrail_name="bedrock-pre", event_hook=GuardrailEventHooks.pre_call) + + await guardrail.apply_guardrail( + inputs=GenericGuardrailAPIInputs(texts=["hello there"]), + request_data=request_data, + input_type="request", + logging_obj=logging_obj, + ) + guardrail.add_standard_logging_guardrail_information_to_request_data( + guardrail_json_response={"action": "NONE"}, + request_data=request_data, + guardrail_status="success", + event_type=GuardrailEventHooks.post_call, + ) + await logging_obj.async_success_handler( + result=ModelResponse(choices=[Choices(message=Message(role="assistant", content="general kenobi"))]), + start_time=dt.datetime.now(), + end_time=dt.datetime.now(), + ) + + entries: Final = logging_obj.model_call_details["standard_logging_object"]["guardrail_information"] + assert [e["guardrail_mode"] for e in entries] == ["pre_call", "post_call"] + class _ApplyOnlyObserver(CustomGuardrail): """Overrides only apply_guardrail, like panw_prisma_airs; inherits async_logging_hook.""" From cf97b757a4356b02baef6b88b5b216ed01dcb515 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Fri, 11 Sep 2026 18:37:08 -0700 Subject: [PATCH 152/157] fix(ui): preserve cleared shared select values (#40795) Preserve explicit null when shared selectors clear and adapt affected forms, validation, and request payloads. Clear stale dependent relationships and retain required-selection checks. Document project detachment, user model-budget clearing, and routing-compression clearing as deferred follow-ups. --- .../add_agent_form.integration.test.tsx | 33 ++++++++++++++- .../agents/_components/add_agent_form.tsx | 16 ++++---- .../_components/ShadowEvalStartForm.tsx | 17 ++++---- .../_components/pricing_calculator/index.tsx | 4 +- .../_components/pricing_calculator/types.ts | 4 +- .../_components/TeamGuardrailsTab.tsx | 5 ++- .../components/AllModelsTable.tsx | 4 +- .../panels/AddModelPanel.tsx | 6 ++- ...I.test.tsx => ChatUI.integration.test.tsx} | 16 ++++++++ .../playground/components/chat_ui/ChatUI.tsx | 30 +++++++++----- .../components/chat_ui/EndpointSelector.tsx | 4 +- .../components/chat_ui/SessionManagement.tsx | 2 +- .../compareUI/components/ModelSelector.tsx | 3 +- ...x => add_policy_form.integration.test.tsx} | 0 .../policies/_components/add_policy_form.tsx | 14 +++---- .../_components/ai_suggestion_modal.tsx | 6 +-- .../_components/pipeline_flow_builder.tsx | 2 +- .../_components/template_parameter_modal.tsx | 6 +-- .../ProjectModals/EditProjectModal.tsx | 6 +-- .../ProjectModals/ProjectBaseForm.tsx | 6 +-- .../ProjectModals/projectFormSchema.ts | 10 +++-- .../prompt_editor_view/ModelConfigCard.tsx | 4 +- .../prompt_editor_view/PromptEditorHeader.tsx | 4 +- .../_components/prompt_editor_view/types.ts | 2 +- .../prompt_editor_view/utils.test.ts | 16 ++++++++ .../_components/prompt_editor_view/utils.ts | 4 +- .../DefaultUserSettingsForm.tsx | 11 +++-- .../default-user-settings/mapper.test.ts | 8 ++-- .../default-user-settings/mapper.ts | 11 +++-- .../default-user-settings/schema.ts | 12 ++++-- .../_components/view_users/UsersTable.tsx | 4 +- .../src/components/CreateUserButton.tsx | 2 +- .../MCPSemanticFilterSettings.tsx | 2 +- .../MCPSemanticFilterTestPanel.tsx | 4 +- .../semanticFilterTestUtils.ts | 6 +-- .../toolSearchForm.test.ts | 1 + .../MCPToolSearchSettings/toolSearchForm.ts | 6 +-- .../Fallbacks/FallbackGroupConfig.tsx | 10 ++--- ui/litellm-dashboard/src/components/Teams.tsx | 4 +- .../src/components/TeamsPage/TeamsTable.tsx | 2 +- .../VirtualKeysPage/VirtualKeysTable.tsx | 4 +- ....tsx => AddModelForm.integration.test.tsx} | 2 +- .../src/components/add_model/AddModelForm.tsx | 15 ++++--- .../add_model/ClassificationMethodConfig.tsx | 3 +- .../add_model/ComplexityRouterConfig.tsx | 2 +- .../add_model/CompressionControls.tsx | 8 ++-- ... RouterConfigBuilder.integration.test.tsx} | 2 +- .../add_model/RouterConfigBuilder.test.ts | 10 +++++ .../add_model/RouterConfigBuilder.tsx | 23 +++++++---- .../add_model/SemanticKeywordMatching.tsx | 4 +- .../add_model/add_auto_router_tab.tsx | 19 +++++---- .../add_model/litellm_model_name.tsx | 10 ++--- .../add_model/provider_specific_fields.tsx | 3 +- .../common_components/ModelAliasManager.tsx | 14 +++++-- .../common_components/ModelSelector.tsx | 16 ++++---- .../OrganizationDropdown.tsx | 4 +- .../common_components/ProjectDropdown.tsx | 4 +- .../common_components/UserDropdown.tsx | 4 +- .../common_components/team_dropdown.tsx | 8 ++-- ...=> user_search_modal.integration.test.tsx} | 18 +++++--- .../common_components/user_search_modal.tsx | 28 +++++++++---- .../edit_auto_router_modal.tsx | 8 ++-- .../BudgetFallbacksEditor.tsx | 4 +- .../key_team_helpers/ModelMaxBudgetEditor.tsx | 4 +- .../components/model_add/CredentialModal.tsx | 8 ++-- .../model_add/credential_form_helpers.ts | 6 +-- .../components/organisms/createKeyPayload.ts | 2 + .../create_key_button.integration.test.tsx | 32 ++++++++++++++- .../organisms/create_key_button.tsx | 24 +++++------ .../src/components/provider_info_helpers.tsx | 2 +- ...aginatedSearchSelect.integration.test.tsx} | 41 ++++++++++++++----- .../shared/PaginatedSearchSelect.tsx | 10 ++--- ....tsx => SearchSelect.integration.test.tsx} | 32 ++++++++++++--- .../src/components/shared/SearchSelect.tsx | 10 ++--- .../src/components/team/TeamInfo.tsx | 7 +++- ...tsx => key_edit_view.integration.test.tsx} | 32 +++++++++++++-- .../components/templates/key_edit_view.tsx | 15 ++++--- .../view_logs/RequestLogsFilters.tsx | 10 ++--- 78 files changed, 492 insertions(+), 263 deletions(-) rename ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/{ChatUI.test.tsx => ChatUI.integration.test.tsx} (95%) rename ui/litellm-dashboard/src/app/(dashboard)/policies/_components/{add_policy_form.test.tsx => add_policy_form.integration.test.tsx} (100%) rename ui/litellm-dashboard/src/components/add_model/{AddModelForm.test.tsx => AddModelForm.integration.test.tsx} (99%) rename ui/litellm-dashboard/src/components/add_model/{RouterConfigBuilder.test.tsx => RouterConfigBuilder.integration.test.tsx} (99%) create mode 100644 ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.ts rename ui/litellm-dashboard/src/components/common_components/{user_search_modal.test.tsx => user_search_modal.integration.test.tsx} (94%) rename ui/litellm-dashboard/src/components/shared/{PaginatedSearchSelect.test.tsx => PaginatedSearchSelect.integration.test.tsx} (89%) rename ui/litellm-dashboard/src/components/shared/{SearchSelect.test.tsx => SearchSelect.integration.test.tsx} (76%) rename ui/litellm-dashboard/src/components/templates/{key_edit_view.test.tsx => key_edit_view.integration.test.tsx} (98%) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx index dad8599e967..ebc97891744 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx @@ -1,11 +1,11 @@ import React from "react"; -import { render, screen, waitFor, within } from "@testing-library/react"; +import { screen, waitFor, within } from "@testing-library/react"; import userEvent, { PointerEventsCheckLevel } from "@testing-library/user-event"; import { describe, it, expect, vi, beforeEach } from "vitest"; import AddAgentForm from "./add_agent_form"; import * as networking from "@/components/networking"; import type { AgentCreateInfo } from "@/components/networking"; -import { chooseSelectOption } from "../../../../../tests/test-utils"; +import { chooseSelectOption, renderWithProviders as render } from "../../../../../tests/test-utils"; vi.mock("@/components/networking", () => ({ createAgentCall: vi.fn(), @@ -351,4 +351,33 @@ describe("AddAgentForm submit payload", () => { expect(await screen.findByText("Agent Created!")).toBeInTheDocument(); expect(within(screen.getByText("Agent Created!").parentElement!).getByText("created-agent")).toBeInTheDocument(); }); + it("blocks creation after clearing the existing key and assigns the reselected key", async () => { + vi.mocked(networking.keyListCall).mockResolvedValue({ + keys: [{ token: "key-maple", key_alias: "Maple key" }], + }); + const user = userEvent.setup(); + renderForm(); + await user.type(await screen.findByLabelText("Agent Name"), "key-selection-agent"); + await user.type(screen.getByLabelText("Display Name"), "Key selection"); + await user.type(screen.getByPlaceholderText("Describe what this agent does..."), "d"); + for (let step = 0; step < 3; step++) { + await user.click(screen.getByRole("button", { name: /^Next/ })); + } + await user.click(screen.getByRole("radio", { name: "Assign an existing key" })); + const keySelector = await screen.findByPlaceholderText("Search by key name…"); + await chooseSelectOption(user, keySelector, "Maple key"); + await user.click(screen.getByRole("button", { name: "Clear" })); + await user.click(screen.getByRole("button", { name: /Create Agent/ })); + expect(networking.createAgentCall).not.toHaveBeenCalled(); + expect(networking.keyUpdateCall).not.toHaveBeenCalled(); + await chooseSelectOption(user, keySelector, "Maple key"); + await user.click(screen.getByRole("button", { name: /Create Agent/ })); + await waitFor(() => + expect(networking.keyUpdateCall).toHaveBeenCalledWith("tok", { + key: "key-maple", + agent_id: "agent-1", + }), + ); + expect(networking.createAgentCall).toHaveBeenCalledTimes(1); + }); }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx index 108bae977e1..e71fed40209 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx @@ -338,6 +338,11 @@ const AddAgentForm: React.FC = ({ visible, onClose, accessTok return; } + if (keyAssignOption === "existing_key" && !selectedExistingKey) { + toast.error("Please select an existing key to assign"); + return; + } + setIsSubmitting(true); try { const isValid = await form.trigger(); @@ -406,12 +411,7 @@ const AddAgentForm: React.FC = ({ visible, onClose, accessTok selectedTeamId, ); setCreatedKeyValue(keyResponse.key || null); - } else if (keyAssignOption === "existing_key") { - if (!selectedExistingKey) { - toast.error("Please select an existing key to assign"); - setIsSubmitting(false); - return; - } + } else if (keyAssignOption === "existing_key" && selectedExistingKey) { await keyUpdateCall(accessToken, { key: selectedExistingKey, agent_id: agentId, @@ -963,8 +963,8 @@ const AddAgentForm: React.FC = ({ visible, onClose, accessTok setSelectedExistingKey(value || null)} + value={selectedExistingKey} + onValueChange={setSelectedExistingKey} options={existingKeys.map((k) => ({ label: k.key_alias || k.token?.slice(0, 12) + "…", value: k.token, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx index d85d26a21a8..06e332d2cfc 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx @@ -170,8 +170,8 @@ interface StartFormValidityInputs { models: string[]; routerNames: string[]; direction: ShadowEvalDirection; - baselineModel: string; - judgeModel: string; + baselineModel: string | null; + judgeModel: string | null; percentage: string; maxBudget: string; } @@ -181,13 +181,13 @@ const startFormValidity = (inputs: StartFormValidityInputs) => { const percentageValid = parsedPct >= 0.1 && parsedPct <= 100; const parsedMaxBudget = Number.parseFloat(inputs.maxBudget); const maxBudgetValid = parsedMaxBudget >= 0.01 && parsedMaxBudget <= 10000; - const baselinePicked = inputs.direction === "forward" || inputs.baselineModel !== ""; + const baselinePicked = inputs.direction === "forward" || Boolean(inputs.baselineModel); const targetsPicked = inputs.apiKeyIds.length + inputs.teamIds.length + inputs.userIds.length > 0; const routerCountValid = inputs.routerNames.length >= 1 && inputs.routerNames.length <= MAX_ROUTERS; const routersMatchDirection = inputs.direction === "forward" || inputs.routerNames.length === 1; const routersValid = routerCountValid && routersMatchDirection; const scopeValid = routersValid && (inputs.direction === "reverse" || inputs.models.length <= MAX_MODELS); - const modelsPicked = scopeValid && inputs.judgeModel !== "" && baselinePicked; + const modelsPicked = scopeValid && Boolean(inputs.judgeModel) && baselinePicked; const filled = targetsPicked && modelsPicked; const boundsValid = percentageValid && maxBudgetValid; const valid = Boolean(inputs.accessToken) && filled && boundsValid; @@ -201,7 +201,7 @@ interface StartBodyInputs { models: string[]; routerNames: string[]; direction: ShadowEvalDirection; - baselineModel: string; + baselineModel: string | null; shadowPercentage: number; durationDays: number; maxBudget: number; @@ -215,7 +215,7 @@ const buildStartBody = (inputs: StartBodyInputs) => ({ models: inputs.direction === "forward" ? inputs.models : [], router_names: inputs.routerNames, direction: inputs.direction, - ...(inputs.direction === "reverse" ? { baseline_model: inputs.baselineModel } : {}), + ...(inputs.direction === "reverse" ? { baseline_model: inputs.baselineModel ?? undefined } : {}), shadow_percentage: inputs.shadowPercentage, duration_days: inputs.durationDays, max_budget: inputs.maxBudget, @@ -230,10 +230,10 @@ export const StartForm: React.FC = () => { const [models, setModels] = useState([]); const [routerNames, setRouterNames] = useState([]); const [direction, setDirection] = useState("forward"); - const [baselineModel, setBaselineModel] = useState(""); + const [baselineModel, setBaselineModel] = useState(null); const [percentage, setPercentage] = useState("10"); const [durationDays, setDurationDays] = useState("7"); - const [judgeModel, setJudgeModel] = useState(""); + const [judgeModel, setJudgeModel] = useState(null); const [maxBudget, setMaxBudget] = useState("10"); const { data: autoRouters } = useAutoRouters(); const configuredGroups = usePlainModelGroups(); @@ -286,6 +286,7 @@ export const StartForm: React.FC = () => { }; const { parsedPct, parsedMaxBudget, percentageValid, maxBudgetValid, valid } = startFormValidity(validityInputs); const handleStart = () => { + if (!valid || !judgeModel) return; const bodyInputs: StartBodyInputs = { apiKeyIds, teamIds, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/index.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/index.tsx index f3bd74260ad..b0612412059 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/index.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/index.tsx @@ -15,7 +15,7 @@ const generateId = () => `entry-${Date.now()}-${Math.random().toString(36).subst const createDefaultEntry = (): ModelEntry => ({ id: generateId(), - model: "", + model: null, input_tokens: 1000, output_tokens: 500, num_requests_per_day: undefined, @@ -28,7 +28,7 @@ const PricingCalculator: React.FC = ({ accessToken, mode const { debouncedFetchForEntry, removeEntry, getMultiModelResult } = useMultiCostEstimate(accessToken); const handleEntryChange = useCallback( - (id: string, field: keyof ModelEntry, value: string | number | undefined) => { + (id: string, field: keyof ModelEntry, value: string | number | null | undefined) => { setEntries((prev) => { const updated = prev.map((entry) => (entry.id === id ? { ...entry, [field]: value } : entry)); const changedEntry = updated.find((e) => e.id === id); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/types.ts b/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/types.ts index 250857a74f5..859d2f3e8d9 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/types.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-tracking/_components/pricing_calculator/types.ts @@ -4,7 +4,7 @@ export interface PricingCalculatorProps { } export interface PricingFormValues { - model: string; + model: string | null; input_tokens: number; output_tokens: number; num_requests_per_day?: number; @@ -13,7 +13,7 @@ export interface PricingFormValues { export interface ModelEntry { id: string; - model: string; + model: string | null; input_tokens: number; output_tokens: number; num_requests_per_day?: number; diff --git a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/TeamGuardrailsTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/TeamGuardrailsTab.tsx index 1de4e697f64..d45cfc3fe7d 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/TeamGuardrailsTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/TeamGuardrailsTab.tsx @@ -48,7 +48,10 @@ const GUARDRAIL_MODES = [ ] as const; const submitGuardrailSchema = z.object({ - team_id: z.string().min(1, "Select a team"), + team_id: z + .string() + .nullable() + .pipe(z.string({ error: "Select a team" }).min(1, "Select a team")), guardrail_name: z.string().min(1, "Enter a guardrail name"), mode: z.string().min(1, "Select a mode"), api_base: z.string().min(1, "Enter the API base URL").refine(isValidUrl, "Must be a valid URL"), diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx index 5c7dbb18428..8482d0832c3 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx @@ -278,7 +278,7 @@ export function AllModelsTable({ options={modelGroupOptions} value={(get(MODEL_NAME_COLUMN_ID) as string) ?? ALL_MODEL_GROUPS_VALUE} onValueChange={(value) => - set(MODEL_NAME_COLUMN_ID, value === ALL_MODEL_GROUPS_VALUE ? undefined : value) + set(MODEL_NAME_COLUMN_ID, value === ALL_MODEL_GROUPS_VALUE ? undefined : value ?? undefined) } placeholder="Filter by Public Model Name" emptyText="No models found" @@ -289,7 +289,7 @@ export function AllModelsTable({ options={accessGroupOptions} value={(get(ACCESS_GROUPS_COLUMN_ID) as string) ?? ALL_MODEL_GROUPS_VALUE} onValueChange={(value) => - set(ACCESS_GROUPS_COLUMN_ID, value === ALL_MODEL_GROUPS_VALUE ? undefined : value) + set(ACCESS_GROUPS_COLUMN_ID, value === ALL_MODEL_GROUPS_VALUE ? undefined : value ?? undefined) } placeholder="Filter by Model Access Group" emptyText="No model access groups found" diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/panels/AddModelPanel.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/panels/AddModelPanel.tsx index 1443c065d9b..1cdce04d07c 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/panels/AddModelPanel.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/panels/AddModelPanel.tsx @@ -26,7 +26,7 @@ export default function AddModelPanel() { const { data: modelCostMapData } = useModelCostMap(); const { data: credentialsResponse } = useCredentials(); const { data: teams } = useTeams(); - const [selectedProvider, setSelectedProvider] = useState(Providers.Anthropic); + const [selectedProvider, setSelectedProvider] = useState(Providers.Anthropic); const [providerModels, setProviderModels] = useState([]); const [showAdvancedSettings, setShowAdvancedSettings] = useState(false); @@ -57,7 +57,9 @@ export default function AddModelPanel() { selectedProvider={selectedProvider} setSelectedProvider={setSelectedProvider} providerModels={providerModels} - setProviderModelsFn={(provider) => setProviderModels(getProviderModels(provider, modelCostMapData))} + setProviderModelsFn={(provider) => + setProviderModels(provider === null ? [] : getProviderModels(provider, modelCostMapData)) + } getPlaceholder={getPlaceholder} showAdvancedSettings={showAdvancedSettings} setShowAdvancedSettings={setShowAdvancedSettings} diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.integration.test.tsx similarity index 95% rename from ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.test.tsx rename to ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.integration.test.tsx index bf6b092a2b0..984996351df 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.integration.test.tsx @@ -153,6 +153,7 @@ describe("ChatUI", () => { }); it("should allow the user to select a model", async () => { + const user = userEvent.setup(); render( { await waitFor(() => { expect(screen.getAllByText("Model 1").length).toBeGreaterThan(0); }); + await user.click(screen.getByRole("option", { name: "Model 1Mode: chat" })); + expect(screen.getByPlaceholderText("Select a Model")).toHaveValue("Model 1"); + + await user.click(screen.getAllByRole("button", { name: "Clear" })[0]); + const input = screen.getByPlaceholderText("Describe the image you want to generate..."); + fireEvent.change(input, { target: { value: "Contract endpoint check" } }); + expect(screen.getByRole("button", { name: "Send message" })).toBeDisabled(); + fireEvent.keyDown(input, { key: "Enter", code: "Enter" }); + expect(input).toHaveValue("Contract endpoint check"); + expect(makeOpenAIChatCompletionRequest).not.toHaveBeenCalled(); + expect(sessionStorage.getItem("endpointType")).toBeNull(); + + await selectComboboxOption("Select an endpoint", "/v1/chat/completions"); + await selectComboboxOption("Select a Model", "Model 1"); + expect(screen.getByRole("button", { name: "Send message" })).toBeEnabled(); }); it("shows only endpoint-compatible models when chat endpoint is selected", async () => { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.tsx index eca9e2323ae..ed8679cfdc1 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/ChatUI.tsx @@ -196,17 +196,17 @@ const ChatUI: React.FC = ({ () => sessionStorage.getItem("customProxyBaseUrl") || "", ); const [inputMessage, setInputMessage] = useState(""); - const [selectedModel, setSelectedModel] = useState(simplified ? fixedModel : undefined); + const [selectedModel, setSelectedModel] = useState(simplified ? fixedModel : null); const [showCustomModelInput, setShowCustomModelInput] = useState(false); const [modelInfo, setModelInfo] = useState([]); const [isLoadingModels, setIsLoadingModels] = useState(false); const [modelLoadError, setModelLoadError] = useState(false); const [agentInfo, setAgentInfo] = useState([]); - const [selectedAgent, setSelectedAgent] = useState(undefined); + const [selectedAgent, setSelectedAgent] = useState(null); const debouncedSetSelectedModel = useDebouncedCallback((value: string) => setSelectedModel(value), { wait: CUSTOM_MODEL_DEBOUNCE_WAIT_MS, }); - const [endpointType, setEndpointType] = useState( + const [endpointType, setEndpointType] = useState( () => sessionStorage.getItem("endpointType") || EndpointType.CHAT, ); const [isLoading, setIsLoading] = useState(false); @@ -327,7 +327,7 @@ const ChatUI: React.FC = ({ }; useEffect(() => { - if (isGetCodeModalVisible) { + if (isGetCodeModalVisible && endpointType !== null) { const code = generateCodeSnippet({ apiKeySource, accessToken, @@ -342,7 +342,7 @@ const ChatUI: React.FC = ({ mcpServers, mcpServerToolRestrictions, endpointType, - selectedModel, + selectedModel: selectedModel ?? undefined, selectedSdk, selectedVoice, proxySettings, @@ -376,7 +376,8 @@ const ChatUI: React.FC = ({ } catch { // Storage full or unavailable — non-critical, skip persisting. } - sessionStorage.setItem("endpointType", endpointType); + if (endpointType === null) sessionStorage.removeItem("endpointType"); + else sessionStorage.setItem("endpointType", endpointType); sessionStorage.setItem("selectedTags", JSON.stringify(selectedTags)); sessionStorage.setItem("selectedVectorStores", JSON.stringify(selectedVectorStores)); sessionStorage.setItem("selectedGuardrails", JSON.stringify(selectedGuardrails)); @@ -493,7 +494,7 @@ const ChatUI: React.FC = ({ setAgentInfo(agents); // Clear selection if current agent not in list if (selectedAgent && !agents.some((a) => a.agent_name === selectedAgent)) { - setSelectedAgent(undefined); + setSelectedAgent(null); } } catch (error) { console.error("Error fetching agents:", error); @@ -616,10 +617,11 @@ const ChatUI: React.FC = ({ setUploadedAudio(file); }; - const handleEndpointChange = (value: string) => { + const handleEndpointChange = (value: string | null) => { setEndpointType(value); - setSelectedModel(undefined); - setSelectedAgent(undefined); + setGeneratedCode(""); + setSelectedModel(null); + setSelectedAgent(null); setShowCustomModelInput(false); setSelectedMCPDirectTool(undefined); if (value === EndpointType.MCP) { @@ -710,6 +712,11 @@ const ChatUI: React.FC = ({ }; const handleSendMessage = async () => { + if (endpointType === null) { + toast.fromError("Please select an endpoint before sending a request"); + return; + } + if (inputMessage.trim() === "" && endpointType !== EndpointType.TRANSCRIPTION && endpointType !== EndpointType.MCP) return; @@ -1152,7 +1159,7 @@ const ChatUI: React.FC = ({ toast.success("Chat history cleared."); }; - const onModelChange = (value: string) => { + const onModelChange = (value: string | null) => { setSelectedModel(value); setShowCustomModelInput(value === "custom"); @@ -1210,6 +1217,7 @@ const ChatUI: React.FC = ({ : "Describe the image you want to generate..."; const sendDisabled = + endpointType === null || isLoading || (endpointType === EndpointType.MCP ? !(selectedMCPServers.length === 1 && selectedMCPServers[0] !== "__all__" && selectedMCPDirectTool) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/EndpointSelector.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/EndpointSelector.tsx index 644bab5b5a4..6b286c403a8 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/EndpointSelector.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/EndpointSelector.tsx @@ -3,8 +3,8 @@ import React from "react"; import { ENDPOINT_OPTIONS } from "./chatConstants"; interface EndpointSelectorProps { - endpointType: string; // Accept string to avoid type conflicts - onEndpointChange: (value: string) => void; + endpointType: string | null; + onEndpointChange: (value: string | null) => void; className?: string; } diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/SessionManagement.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/SessionManagement.tsx index 35c8e839159..4dbf9168897 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/SessionManagement.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/chat_ui/SessionManagement.tsx @@ -7,7 +7,7 @@ import { Switch } from "@/components/ui/switch"; import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip"; interface SessionManagementProps { - endpointType: string; + endpointType: string | null; responsesSessionId: string | null; useApiSessionManagement: boolean; onToggleSessionManagement: (useApi: boolean) => void; diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/compareUI/components/ModelSelector.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/compareUI/components/ModelSelector.tsx index 5659875cf82..8ecc0e0c8bb 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/playground/components/compareUI/components/ModelSelector.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/components/compareUI/components/ModelSelector.tsx @@ -22,7 +22,8 @@ export function ModelSelector({ value, onChange, models, loading, disabled }: Mo const selectValue = isAddingCustom ? "__custom__" : value || undefined; - const handleSelectChange = (selected: string) => { + const handleSelectChange = (selected: string | null) => { + if (selected === null) return; if (selected === "__custom__") { setIsAddingCustom(true); if (value && !options.includes(value)) { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.integration.test.tsx similarity index 100% rename from ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.test.tsx rename to ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.integration.test.tsx diff --git a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.tsx b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.tsx index 2830b0fda12..d14165dd849 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/add_policy_form.tsx @@ -44,10 +44,10 @@ const policyShape = { .min(1, "Please enter a policy name") .regex(/^[a-zA-Z0-9_-]+$/, "Policy name can only contain letters, numbers, hyphens, and underscores"), description: z.string(), - inherit: z.string(), + inherit: z.string().nullable(), guardrails_add: z.array(z.string()), guardrails_remove: z.array(z.string()), - model_condition: z.string(), + model_condition: z.string().nullable(), }; const policySchema = z.object(policyShape); @@ -57,19 +57,19 @@ type PolicyFormValues = z.infer; const EMPTY_VALUES: PolicyFormValues = { policy_name: "", description: "", - inherit: "", + inherit: null, guardrails_add: [], guardrails_remove: [], - model_condition: "", + model_condition: null, }; const toFormValues = (policy: Policy): PolicyFormValues => ({ policy_name: policy.policy_name, description: policy.description ?? "", - inherit: policy.inherit ?? "", + inherit: policy.inherit ?? null, guardrails_add: policy.guardrails_add || [], guardrails_remove: policy.guardrails_remove || [], - model_condition: policy.condition?.model ?? "", + model_condition: policy.condition?.model ?? null, }); const buildPolicyRequest = (values: PolicyFormValues): PolicyCreateRequest | PolicyUpdateRequest => ({ @@ -529,7 +529,7 @@ const AddPolicyForm: React.FC = ({ {...control} id={id} ref={ref} - value={value} + value={value ?? ""} onChange={onChange} placeholder="Leave empty to apply to all models (e.g., gpt-4.* or bedrock/claude-.*)" /> diff --git a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/ai_suggestion_modal.tsx b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/ai_suggestion_modal.tsx index 32e777949d0..c4f76b07a4e 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/ai_suggestion_modal.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/ai_suggestion_modal.tsx @@ -68,7 +68,7 @@ const AiSuggestionModal: React.FC = ({ const [suggestions, setSuggestions] = useState(null); const [explanation, setExplanation] = useState(null); const [selectedIds, setSelectedIds] = useState>(new Set()); - const [selectedModel, setSelectedModel] = useState(undefined); + const [selectedModel, setSelectedModel] = useState(null); const [availableModels, setAvailableModels] = useState([]); const [isLoadingModels, setIsLoadingModels] = useState(false); // Test panel state @@ -114,7 +114,7 @@ const AiSuggestionModal: React.FC = ({ setSuggestions(null); setExplanation(null); setSelectedIds(new Set()); - setSelectedModel(undefined); + setSelectedModel(null); setShowTestPanel(false); setTestInputText(""); setIsTestLoading(false); @@ -837,7 +837,7 @@ const AiSuggestionModal: React.FC = ({ ({ label: m, value: m }))} value={selectedModel} - onValueChange={(value) => setSelectedModel(value || undefined)} + onValueChange={setSelectedModel} placeholder={isLoadingModels ? "Loading models..." : "Select a model to analyze your requirements"} emptyText="No models found" disabled={isLoadingModels} diff --git a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/pipeline_flow_builder.tsx b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/pipeline_flow_builder.tsx index 8651be39f0d..da77b348028 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/pipeline_flow_builder.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/pipeline_flow_builder.tsx @@ -343,7 +343,7 @@ const StepCard: React.FC = ({ onChange({ guardrail: value })} + onValueChange={(value) => onChange({ guardrail: value ?? undefined })} placeholder="Select a guardrail" emptyText="No guardrails found" /> diff --git a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/template_parameter_modal.tsx b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/template_parameter_modal.tsx index 410d4ea8467..b90379b149d 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/template_parameter_modal.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/policies/_components/template_parameter_modal.tsx @@ -46,7 +46,7 @@ const TemplateParameterModal: React.FC = ({ }) => { const [parameterValues, setParameterValues] = useState>({}); const [competitorMode, setCompetitorMode] = useState<"ai" | "manual">("ai"); - const [selectedModel, setSelectedModel] = useState(undefined); + const [selectedModel, setSelectedModel] = useState(null); const [availableModels, setAvailableModels] = useState([]); const [isLoadingModels, setIsLoadingModels] = useState(false); const [competitorTags, setCompetitorTags] = useState([]); @@ -72,7 +72,7 @@ const TemplateParameterModal: React.FC = ({ }); setParameterValues(initial); setCompetitorMode("ai"); - setSelectedModel(undefined); + setSelectedModel(null); setCompetitorTags([]); setVariationsMap({}); setIsGenerating(false); @@ -297,7 +297,7 @@ const TemplateParameterModal: React.FC = ({ ({ label: m, value: m }))} value={selectedModel} - onValueChange={(value) => setSelectedModel(value || undefined)} + onValueChange={setSelectedModel} placeholder={isLoadingModels ? "Loading models..." : "Select a model to generate names"} emptyText="No models found" disabled={isLoadingModels} diff --git a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/EditProjectModal.tsx b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/EditProjectModal.tsx index 5c1a443d6c5..77f28b05ea5 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/EditProjectModal.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/EditProjectModal.tsx @@ -10,7 +10,7 @@ import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner"; import { ProjectResponse } from "@/app/(dashboard)/hooks/projects/useProjects"; import { useUpdateProject, ProjectUpdateParams } from "@/app/(dashboard)/hooks/projects/useUpdateProject"; import { ProjectBaseForm } from "./ProjectBaseForm"; -import { projectFormSchema, type ProjectFormValues } from "./projectFormSchema"; +import { projectFormSchema, type ProjectFormValues, type ProjectSubmitValues } from "./projectFormSchema"; import { buildProjectUpdateParams } from "./projectFormUtils"; import { Dialog, DialogContent, DialogHeader, DialogTitle } from "@/components/ui/dialog"; @@ -58,7 +58,7 @@ export const toFormValues = (project: ProjectResponse): ProjectFormValues => { return { project_alias: project.project_alias ?? "", - team_id: project.team_id ?? "", + team_id: project.team_id ?? null, description: project.description ?? "", models: project.models ?? [], max_budget: project.litellm_budget_table?.max_budget ?? undefined, @@ -81,7 +81,7 @@ function EditProjectForm({ project, onClose, onSuccess }: Omit { - const submitted: ProjectFormValues = advancedEverOpened + const submitted: ProjectSubmitValues = advancedEverOpened ? values : { ...values, guardrails: undefined, modelLimits: undefined, metadata: undefined }; diff --git a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/ProjectBaseForm.tsx b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/ProjectBaseForm.tsx index fed1a88fe98..1ad8a1e4953 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/ProjectBaseForm.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/ProjectBaseForm.tsx @@ -5,7 +5,7 @@ import { useFieldArray, useWatch, type UseFormReturn } from "react-hook-form"; import { ChevronDown, CircleAlert, Minus, Plus } from "lucide-react"; import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized"; -import { ALL_TEAM_MODELS, type ProjectFormValues } from "./projectFormSchema"; +import { ALL_TEAM_MODELS, type ProjectFormValues, type ProjectSubmitValues } from "./projectFormSchema"; import { useTeams } from "@/app/(dashboard)/hooks/teams/useTeams"; import { Team } from "@/components/key_team_helpers/key_list"; import { fetchTeamModels } from "@/components/organisms/create_key_button"; @@ -32,7 +32,7 @@ const toOptionalNumber = (raw: string): number | undefined => { }; interface ProjectBaseFormProps { - form: UseFormReturn; + form: UseFormReturn; advancedOpen: boolean; onAdvancedOpenChange: (open: boolean) => void; } @@ -94,7 +94,7 @@ export function ProjectBaseForm({ form, advancedOpen, onAdvancedOpenChange }: Pr } }, [selectedTeam, accessToken, userId, userRole]); - const handleTeamChange = (teamId: string) => { + const handleTeamChange = (teamId: string | null) => { const team = teams?.find((t) => t.team_id === teamId) ?? null; setSelectedTeam(team); form.setValue("models", []); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/projectFormSchema.ts b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/projectFormSchema.ts index 6bfaa831bba..d4c85d89616 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/projectFormSchema.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectModals/projectFormSchema.ts @@ -16,7 +16,10 @@ const modelLimitSchema = z.object({ export const projectFormSchema = z .object({ project_alias: z.string().min(1, "Please enter a project name"), - team_id: z.string().min(1, "Please select a team"), + team_id: z + .string() + .nullable() + .pipe(z.string({ error: "Please select a team" }).min(1, "Please select a team")), description: z.string().optional(), models: z.array(z.string()), max_budget: z.number().optional(), @@ -48,11 +51,12 @@ export const projectFormSchema = z }); }); -export type ProjectFormValues = z.output; +export type ProjectFormValues = z.input; +export type ProjectSubmitValues = z.output; export const emptyProjectFormValues: ProjectFormValues = { project_alias: "", - team_id: "", + team_id: null, description: undefined, models: [], max_budget: undefined, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/ModelConfigCard.tsx b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/ModelConfigCard.tsx index a14e45de99a..e9b526c9703 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/ModelConfigCard.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/ModelConfigCard.tsx @@ -6,11 +6,11 @@ import { SettingsIcon } from "lucide-react"; import ModelSelector from "@/components/common_components/ModelSelector"; interface ModelConfigCardProps { - model: string; + model: string | null; temperature?: number; maxTokens?: number; accessToken: string | null; - onModelChange: (model: string) => void; + onModelChange: (model: string | null) => void; onTemperatureChange: (temp: number) => void; onMaxTokensChange: (tokens: number) => void; } diff --git a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/PromptEditorHeader.tsx b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/PromptEditorHeader.tsx index 04cac01365a..0f001979c28 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/PromptEditorHeader.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/PromptEditorHeader.tsx @@ -21,7 +21,7 @@ interface PromptEditorHeaderProps { editMode?: boolean; onShowHistory?: () => void; version?: string | null; - promptModel?: string; + promptModel?: string | null; promptVariables?: Record; accessToken: string | null; proxySettings?: { @@ -85,7 +85,7 @@ const PromptEditorHeader: React.FC = ({
{ expect(result).toContain("output:"); expect(result).toContain("format: text"); expect(result).toContain("User: Hello world"); + const cleared = convertToDotPrompt({ ...prompt, model: null }); + expect(cleared).toBe(result.replace("model: gpt-4\n", "")); }); it("should include config parameters when set", () => { @@ -203,6 +205,20 @@ describe("convertToDotPrompt", () => { }); describe("parseExistingPrompt", () => { + it("should keep saved prompts with missing or blank models unassigned", () => { + for (const modelLine of ["", "model: \n"]) { + const prompt = parseExistingPrompt({ + prompt_spec: { + prompt_id: "unassigned-prompt", + litellm_params: { dotprompt_content: `---\n${modelLine}temperature: 0\n---\nUser: Keep this message` }, + }, + }); + + expect(prompt.model).toBeNull(); + expect(convertToDotPrompt(prompt)).not.toMatch(/^model:/m); + } + }); + it("should parse basic dotprompt content", () => { const apiResponse = { prompt_spec: { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/utils.ts b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/utils.ts index ef327d93495..3d768e930a9 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/utils.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/prompts/_components/prompt_editor_view/utils.ts @@ -23,7 +23,7 @@ export const extractVariables = (prompt: PromptType): string[] => { export const convertToDotPrompt = (prompt: PromptType): string => { const variables = extractVariables(prompt); - let result = `---\nmodel: ${prompt.model}\n`; + let result = prompt.model ? `---\nmodel: ${prompt.model}\n` : "---\n"; // Add temperature if set if (prompt.config.temperature !== undefined) { @@ -237,7 +237,7 @@ export const parseExistingPrompt = (apiResponse: any): PromptType => { return { name: baseName, - model: parsedFrontmatter.model || "gpt-4o", + model: parsedFrontmatter.model || null, config: parsedFrontmatter.config, tools: parsedFrontmatter.tools, developerMessage: parsedBody.developerMessage, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/DefaultUserSettingsForm.tsx b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/DefaultUserSettingsForm.tsx index 269f3b0af39..0239844851f 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/DefaultUserSettingsForm.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/DefaultUserSettingsForm.tsx @@ -20,7 +20,12 @@ import { useZodForm } from "@/lib/forms/useZodForm"; import { fetchClient } from "@/lib/http/api"; import { buildBody, settingsToForm, type DefaultInternalUserParams, type InternalUserSettings } from "./mapper"; -import { defaultUserSettingsSchema, EMPTY_TEAM_ROW, type DefaultUserSettingsFormValues } from "./schema"; +import { + defaultUserSettingsSchema, + EMPTY_TEAM_ROW, + type DefaultUserSettingsFormValues, + type DefaultUserSettingsSubmitValues, +} from "./schema"; const NO_RESET = "never"; @@ -63,7 +68,7 @@ interface RoleOption { description: string; } -type SettingsControl = Control; +type SettingsControl = Control; const TeamPickerField = ({ control, index }: { control: SettingsControl; index: number }) => { const [search, setSearch] = React.useState(""); @@ -233,7 +238,7 @@ const SettingsForm = ({ initialValues, roleOptions, updateSettings, onCancel, on const { isDirty } = form.formState; const mutation = useMutation({ - mutationFn: (values: DefaultUserSettingsFormValues) => updateSettings(buildBody(values)), + mutationFn: (values: DefaultUserSettingsSubmitValues) => updateSettings(buildBody(values)), onSuccess: (_result, values) => { toast.success("Default user settings updated successfully"); queryClient.invalidateQueries({ queryKey: SETTINGS_QUERY_KEY }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.test.ts b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.test.ts index e8b350332c8..f4a7f001480 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.test.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.test.ts @@ -1,7 +1,7 @@ import { describe, expect, it } from "vitest"; import { buildBody, settingsToForm } from "./mapper"; -import type { DefaultUserSettingsFormValues } from "./schema"; +import type { DefaultUserSettingsSubmitValues } from "./schema"; const CONFIGURED_SETTINGS = { user_role: "internal_user", @@ -59,13 +59,13 @@ describe("settingsToForm", () => { it("degrades an unrecognisable team entry to a blank row instead of throwing", () => { expect(settingsToForm({ teams: [{ max_budget_in_team: 5 }, 7] }).teams).toStrictEqual([ - { team_id: "", max_budget_in_team: "", user_role: "user" }, - { team_id: "", max_budget_in_team: "", user_role: "user" }, + { team_id: null, max_budget_in_team: "", user_role: "user" }, + { team_id: null, max_budget_in_team: "", user_role: "user" }, ]); }); }); -const formValues = (overrides: Partial = {}): DefaultUserSettingsFormValues => ({ +const formValues = (overrides: Partial = {}): DefaultUserSettingsSubmitValues => ({ user_role: "internal_user", max_budget: "100", budget_duration: "30d", diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.ts b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.ts index 50365081afb..ac528542ae2 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/mapper.ts @@ -2,7 +2,12 @@ import { z } from "zod/v4"; import type { components } from "@/lib/http/schema"; -import { EMPTY_TEAM_ROW, type DefaultTeamRowValues, type DefaultUserSettingsFormValues } from "./schema"; +import { + EMPTY_TEAM_ROW, + type DefaultTeamRowValues, + type DefaultUserSettingsFormValues, + type DefaultUserSettingsSubmitValues, +} from "./schema"; export type InternalUserSettings = components["schemas"]["InternalUserSettingsResponse"]; export type DefaultInternalUserParams = components["schemas"]["DefaultInternalUserParams"]; @@ -60,13 +65,13 @@ const textOrNull = (raw: string): string | null => (raw.trim() === "" ? null : r const listOrNull = (items: readonly T[]): T[] | null => (items.length === 0 ? null : [...items]); -const toTeamBody = (team: DefaultTeamRowValues): DefaultTeamBody => ({ +const toTeamBody = (team: DefaultUserSettingsSubmitValues["teams"][number]): DefaultTeamBody => ({ team_id: team.team_id, max_budget_in_team: numberOrNull(team.max_budget_in_team), user_role: team.user_role, }); -export const buildBody = (values: DefaultUserSettingsFormValues): DefaultInternalUserParams => ({ +export const buildBody = (values: DefaultUserSettingsSubmitValues): DefaultInternalUserParams => ({ user_role: asDefaultUserRole(values.user_role), max_budget: numberOrNull(values.max_budget), budget_duration: textOrNull(values.budget_duration), diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/schema.ts b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/schema.ts index 7309e3745da..cd66c3addc8 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/schema.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/default-user-settings/schema.ts @@ -10,14 +10,17 @@ const amountOrEmpty = z ); const defaultTeamRowSchema = z.object({ - team_id: z.string().min(1, "Select a team"), + team_id: z + .string() + .nullable() + .pipe(z.string({ error: "Select a team" }).min(1, "Select a team")), max_budget_in_team: amountOrEmpty, user_role: z.enum(["user", "admin"]), }); -export type DefaultTeamRowValues = z.output; +export type DefaultTeamRowValues = z.input; -export const EMPTY_TEAM_ROW: DefaultTeamRowValues = { team_id: "", max_budget_in_team: "", user_role: "user" }; +export const EMPTY_TEAM_ROW: DefaultTeamRowValues = { team_id: null, max_budget_in_team: "", user_role: "user" }; const defaultUserSettingsShape = { user_role: z.string(), @@ -41,4 +44,5 @@ export const defaultUserSettingsSchema = z.object(defaultUserSettingsShape).supe ); }); -export type DefaultUserSettingsFormValues = z.output; +export type DefaultUserSettingsFormValues = z.input; +export type DefaultUserSettingsSubmitValues = z.output; diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/UsersTable.tsx b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/UsersTable.tsx index 875df74652c..8fa76a64af6 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/UsersTable.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/UsersTable.tsx @@ -192,7 +192,7 @@ export function UsersTable({ set("user_role", value)} + onValueChange={(value) => set("user_role", value ?? undefined)} placeholder="Select a role…" emptyText="No roles found" /> @@ -201,7 +201,7 @@ export function UsersTable({ set("team", value)} + onValueChange={(value) => set("team", value ?? undefined)} placeholder="Select a team…" emptyText="No teams found" /> diff --git a/ui/litellm-dashboard/src/components/CreateUserButton.tsx b/ui/litellm-dashboard/src/components/CreateUserButton.tsx index 0f7c356b8cc..83a72e50f5c 100644 --- a/ui/litellm-dashboard/src/components/CreateUserButton.tsx +++ b/ui/litellm-dashboard/src/components/CreateUserButton.tsx @@ -59,7 +59,7 @@ interface UISettings { interface CreateUserFormValues { user_email?: string; user_role: string; - team_id?: string; + team_id?: string | null; organization_ids?: string[]; metadata?: string; send_invite_email: boolean; diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterSettings.tsx b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterSettings.tsx index 390684d3739..f78790dea88 100644 --- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterSettings.tsx +++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterSettings.tsx @@ -115,7 +115,7 @@ export default function MCPSemanticFilterSettings({ accessToken }: MCPSemanticFi // Test section state const [testQuery, setTestQuery] = useState(""); - const [testModel, setTestModel] = useState("gpt-4o"); + const [testModel, setTestModel] = useState("gpt-4o"); const [testResult, setTestResult] = useState(null); const [testError, setTestError] = useState(null); const [isTesting, setIsTesting] = useState(false); diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterTestPanel.tsx b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterTestPanel.tsx index 5dc661391ac..82c699d9983 100644 --- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterTestPanel.tsx +++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterTestPanel.tsx @@ -11,8 +11,8 @@ interface MCPSemanticFilterTestPanelProps { accessToken: string | null; testQuery: string; setTestQuery: (value: string) => void; - testModel: string; - setTestModel: (value: string) => void; + testModel: string | null; + setTestModel: (value: string | null) => void; isTesting: boolean; onTest: () => void; filterEnabled: boolean; diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/semanticFilterTestUtils.ts b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/semanticFilterTestUtils.ts index d3e585e8052..c41b081da88 100644 --- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/semanticFilterTestUtils.ts +++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPSemanticFilterSettings/semanticFilterTestUtils.ts @@ -32,7 +32,7 @@ export const runSemanticFilterTest = async ({ setTestError, }: { accessToken: string; - testModel: string; + testModel: string | null; testQuery: string; setIsTesting: (value: boolean) => void; setTestResult: (result: TestResult | null) => void; @@ -68,12 +68,12 @@ export const runSemanticFilterTest = async ({ } }; -export const getCurlCommand = (testModel: string, testQuery: string) => +export const getCurlCommand = (testModel: string | null, testQuery: string) => `curl --location 'http://localhost:4000/v1/responses' \\ --header 'Content-Type: application/json' \\ --header 'Authorization: Bearer sk-1234' \\ --data '{ - "model": "${testModel}", + "model": "${testModel ?? "YOUR_MODEL"}", "input": [ { "role": "user", diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.test.ts b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.test.ts index 2f9032537c1..6d19c17c577 100644 --- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.test.ts +++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.test.ts @@ -51,6 +51,7 @@ describe("parseCoreTools", () => { describe("formToPayload", () => { it("sends null for a cleared embedding model so the proxy returns to keyword matching", () => { expect(formToPayload({ ...DEFAULT_FORM_VALUES, embedding_model: " " })).toEqual(KEYWORD_PAYLOAD); + expect(formToPayload({ ...DEFAULT_FORM_VALUES, embedding_model: null })).toEqual(KEYWORD_PAYLOAD); }); it("clamps top_k into the range the proxy accepts and lists core tools", () => { diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.ts b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.ts index f10bfd249d7..d3535fe91a3 100644 --- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.ts +++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/MCPToolSearchSettings/toolSearchForm.ts @@ -1,7 +1,7 @@ import type { MCPToolSearchSettings } from "@/app/(dashboard)/hooks/mcpToolSearchSettings/useMCPToolSearchSettings"; export interface ToolSearchFormValues { - embedding_model: string; + embedding_model: string | null; top_k: number; similarity_threshold: number; core_tools_text: string; @@ -11,7 +11,7 @@ export const TOP_K_MIN = 1; export const TOP_K_MAX = 100; export const DEFAULT_FORM_VALUES: ToolSearchFormValues = { - embedding_model: "", + embedding_model: null, top_k: 5, similarity_threshold: 0, core_tools_text: "", @@ -42,7 +42,7 @@ export const storedValuesToForm = (values: Record): ToolSearchF }); export const formToPayload = (form: ToolSearchFormValues): MCPToolSearchSettings => ({ - embedding_model: form.embedding_model.trim() === "" ? null : form.embedding_model.trim(), + embedding_model: form.embedding_model?.trim() || null, top_k: clampTopK(form.top_k), similarity_threshold: form.similarity_threshold, core_tools: parseCoreTools(form.core_tools_text), diff --git a/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/FallbackGroupConfig.tsx b/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/FallbackGroupConfig.tsx index a82e0fdfa88..3897a9d9fa2 100644 --- a/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/FallbackGroupConfig.tsx +++ b/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/FallbackGroupConfig.tsx @@ -32,12 +32,8 @@ export function FallbackGroupConfig({ // Filter available options for fallbacks (exclude primary only, allow already selected to be shown for deselection) const availableFallbackOptions = availableModels.filter((m) => m !== group.primaryModel); - const handlePrimaryChange = (value: string) => { - let newFallbacks = [...group.fallbackModels]; - // Remove from fallbacks if it was there - if (newFallbacks.includes(value)) { - newFallbacks = newFallbacks.filter((m) => m !== value); - } + const handlePrimaryChange = (value: string | null) => { + const newFallbacks = group.fallbackModels.filter((model) => model !== value); onChange({ ...group, primaryModel: value, @@ -76,7 +72,7 @@ export function FallbackGroupConfig({ ({ label: m, value: m }))} - value={group.primaryModel ?? ""} + value={group.primaryModel} onValueChange={handlePrimaryChange} placeholder="Select primary model" emptyText="No models found" diff --git a/ui/litellm-dashboard/src/components/Teams.tsx b/ui/litellm-dashboard/src/components/Teams.tsx index d2d18139893..f0d21cc5350 100644 --- a/ui/litellm-dashboard/src/components/Teams.tsx +++ b/ui/litellm-dashboard/src/components/Teams.tsx @@ -328,11 +328,11 @@ const Teams: React.FC = ({ accessToken, userID, userRole, premiumUser }; const selectCreateTeamOrganization = ( - next: string, + next: string | null, currentOrganizationId: string | null, onChange: (organizationId: string | null) => void, ) => { - const nextOrganizationId = next === "" ? null : next; + const nextOrganizationId = next; if (nextOrganizationId === currentOrganizationId) return; onChange(nextOrganizationId); form.setValue("models", []); diff --git a/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.tsx b/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.tsx index 3fb19e522f3..bde1c7724c3 100644 --- a/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.tsx +++ b/ui/litellm-dashboard/src/components/TeamsPage/TeamsTable.tsx @@ -203,7 +203,7 @@ export function TeamsTable({ userRole, userID, onSelectTeam, onEditTeam, onDelet set("org_id", value)} + onValueChange={(value) => set("org_id", value ?? undefined)} placeholder="Select an organization…" emptyText="No organizations found" /> diff --git a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx index f28ae27b6bc..d0367cb4d5b 100644 --- a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx +++ b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx @@ -311,7 +311,7 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) { set("team_id", value)} + onValueChange={(value) => set("team_id", value ?? undefined)} placeholder="Select a team…" emptyText="No teams found" /> @@ -320,7 +320,7 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) { set("org_id", value)} + onValueChange={(value) => set("org_id", value ?? undefined)} placeholder="Select an organization…" emptyText="No organizations found" /> diff --git a/ui/litellm-dashboard/src/components/add_model/AddModelForm.test.tsx b/ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx similarity index 99% rename from ui/litellm-dashboard/src/components/add_model/AddModelForm.test.tsx rename to ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx index 3b879da2d68..536bc231008 100644 --- a/ui/litellm-dashboard/src/components/add_model/AddModelForm.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx @@ -164,7 +164,7 @@ const createTestProps = (userRole = "proxy_admin", userId = "user-1", isTeamAdmi handleOk: vi.fn().mockResolvedValue(true), setSelectedProvider: vi.fn(), setProviderModelsFn: vi.fn(), - getPlaceholder: vi.fn((provider: Providers) => `Enter ${provider} model name`), + getPlaceholder: vi.fn((provider: string) => `Enter ${provider} model name`), setShowAdvancedSettings: vi.fn(), selectedProvider: Providers.OpenAI, providerModels: ["gpt-4", "gpt-3.5-turbo"], diff --git a/ui/litellm-dashboard/src/components/add_model/AddModelForm.tsx b/ui/litellm-dashboard/src/components/add_model/AddModelForm.tsx index 92951b68fd5..30b4911da3e 100644 --- a/ui/litellm-dashboard/src/components/add_model/AddModelForm.tsx +++ b/ui/litellm-dashboard/src/components/add_model/AddModelForm.tsx @@ -25,7 +25,6 @@ import { } from "../common_components/MountedFormField"; import type { Team } from "../key_team_helpers/key_list"; import { type CredentialItem, type ProviderCreateInfo, modelAvailableCall } from "../networking"; -import { Providers } from "../provider_info_helpers"; import { ProviderLogo } from "../molecules/models/ProviderLogo"; import AccessGroupTagsCombobox from "./AccessGroupTagsCombobox"; import AdvancedSettings from "./advanced_settings"; @@ -42,11 +41,11 @@ interface AddModelFormProps { registry: MountRegistry; mountedValues: () => MountedFormValues; handleOk: () => Promise; - selectedProvider: Providers; - setSelectedProvider: (provider: Providers) => void; + selectedProvider: string | null; + setSelectedProvider: (provider: string | null) => void; providerModels: string[]; - setProviderModelsFn: (provider: Providers) => void; - getPlaceholder: (provider: Providers) => string; + setProviderModelsFn: (provider: string | null) => void; + getPlaceholder: (provider: string) => string; showAdvancedSettings: boolean; setShowAdvancedSettings: (show: boolean) => void; teams: Team[] | null; @@ -140,7 +139,7 @@ const AddModelForm: React.FC = ({ [credentials], ); - const applyProviderSelection = (provider: Providers) => { + const applyProviderSelection = (provider: string | null) => { setSelectedProvider(provider); setProviderModelsFn(provider); form.setValue("model", []); @@ -227,10 +226,10 @@ const AddModelForm: React.FC = ({ options={providerOptions} emptyText={providerMetadataErrorText ?? "No providers found"} placeholder={isProviderMetadataLoading ? "Loading providers..." : "Select a provider"} - value={(control.value as string | undefined) ?? ""} + value={typeof control.value === "string" ? control.value : null} onValueChange={(value) => { control.onChange(value); - applyProviderSelection(value as Providers); + applyProviderSelection(value); }} /> )} diff --git a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx index 594c6d022ce..93ccf561387 100644 --- a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx @@ -328,7 +328,8 @@ const ClassificationMethodConfig: React.FC = ({ onChange(nextValue); }; - const handleClassifierModelChange = (model: string) => { + const handleClassifierModelChange = (model: string | null) => { + if (model === null) return; if (model === value.classifier_llm_config?.model) return; const { reasoning_effort: _reasoningEffort, ...classifierLlmConfig } = value.classifier_llm_config ?? { model: "", diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx index 97d6739aa22..febcde269f7 100644 --- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx @@ -632,7 +632,7 @@ const ComplexityRouterConfig: React.FC = ({ // Clearing the select drops the key entirely rather than storing "", so an emptied pin reads as // "track the tiers" everywhere downstream instead of as a blank model name. - const handleDefaultModelChange = (model: string | undefined) => { + const handleDefaultModelChange = (model: string | null | undefined) => { onChange({ ...value, default_model: model || undefined }); }; diff --git a/ui/litellm-dashboard/src/components/add_model/CompressionControls.tsx b/ui/litellm-dashboard/src/components/add_model/CompressionControls.tsx index c1817918f60..e67583febf3 100644 --- a/ui/litellm-dashboard/src/components/add_model/CompressionControls.tsx +++ b/ui/litellm-dashboard/src/components/add_model/CompressionControls.tsx @@ -42,8 +42,8 @@ const CompressionControls: React.FC = ({ value, onChan
onRoutingChange(value === "" ? undefined : value)} + value={routing} + onValueChange={(value) => onRoutingChange(value ?? undefined)} placeholder="Inherit from the request's own compression guardrails" emptyText="No compression guardrails found" aria-label="Routing decision compression" @@ -74,8 +74,8 @@ const CompressionControls: React.FC = ({ value, onChan
onModelChange(value === "" ? undefined : value)} + value={model} + onValueChange={(value) => onModelChange(value ?? undefined)} placeholder="None (no compression)" emptyText="No compression guardrails found" aria-label="Model call compression" diff --git a/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.tsx b/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.integration.test.tsx similarity index 99% rename from ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.tsx rename to ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.integration.test.tsx index ecd5abfa451..495fd207266 100644 --- a/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.integration.test.tsx @@ -47,7 +47,7 @@ describe("RouterConfigBuilder", () => { expect(onChange).toHaveBeenCalledWith({ routes: [ expect.objectContaining({ - name: "", + name: null, utterances: [], description: "", score_threshold: 0.5, diff --git a/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.ts b/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.ts new file mode 100644 index 00000000000..d681c911c46 --- /dev/null +++ b/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.test.ts @@ -0,0 +1,10 @@ +import { describe, expect, it } from "vitest"; +import { serializeRouterConfig } from "./RouterConfigBuilder"; + +describe("serializeRouterConfig", () => { + it("rejects a cleared route model and preserves selected route settings", () => { + expect(() => serializeRouterConfig({ routes: [{ name: null }] })).toThrow("Please select a model for every route"); + const config = { routes: [{ name: "model-silver", utterances: [], description: "", score_threshold: 0 }] }; + expect(JSON.parse(serializeRouterConfig(config))).toEqual(config); + }); +}); diff --git a/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.tsx b/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.tsx index 6dfb9f32091..9713d18872b 100644 --- a/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.tsx +++ b/ui/litellm-dashboard/src/components/add_model/RouterConfigBuilder.tsx @@ -15,7 +15,7 @@ import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/comp interface Route { id: string; - model: string; + model: string | null; utterances: string[]; description: string; score_threshold: number; @@ -23,21 +23,28 @@ interface Route { interface SavedRoute { id?: string; - name?: string; - model?: string; + name?: string | null; + model?: string | null; utterances?: string[]; description?: string; score_threshold?: number; } -interface RouterConfig { +export interface RouterConfig { routes?: SavedRoute[]; } +export function serializeRouterConfig(config: RouterConfig | null): string { + if (config?.routes?.some((route) => !(route.name ?? route.model))) { + throw new Error("Please select a model for every route"); + } + return JSON.stringify(config); +} + interface RouterConfigBuilderProps { modelInfo: ModelGroup[]; - value?: RouterConfig; - onChange?: (config: any) => void; + value?: RouterConfig | null; + onChange?: (config: RouterConfig) => void; } interface UtteranceInputProps { @@ -136,7 +143,7 @@ const RouterConfigBuilder: React.FC = ({ modelInfo, va routeIds.push(id); return { id, - model: route.name || route.model || "", + model: route.name || route.model || null, utterances: route.utterances || [], description: route.description || "", score_threshold: route.score_threshold ?? 0.5, @@ -165,7 +172,7 @@ const RouterConfigBuilder: React.FC = ({ modelInfo, va const newRouteId = `route-${Date.now()}`; const updatedRoutes = [ ...routes, - { id: newRouteId, model: "", utterances: [], description: "", score_threshold: 0.5 }, + { id: newRouteId, model: null, utterances: [], description: "", score_threshold: 0.5 }, ]; setRoutes(updatedRoutes); updateConfig(updatedRoutes); diff --git a/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx b/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx index 5684d32a319..0a5a9a5bfb6 100644 --- a/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx +++ b/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx @@ -61,7 +61,9 @@ const SemanticKeywordMatching: React.FC = ({ { + if (model !== null) onEmbeddingModelChange(model); + }} placeholder="Select an embedding model" emptyText="No embedding models found" aria-label="Embedding model" diff --git a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx index d7c4a671317..632f4427a82 100644 --- a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx +++ b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx @@ -150,7 +150,10 @@ export const getSubmitBlockedReason = ( const autoRouterSchema = (requiresTeamScope: boolean) => z.object({ auto_router_name: z.string().min(1, "Auto router name is required"), - team_id: requiresTeamScope ? z.string().min(1, "Please select a team to continue") : z.string(), + team_id: z + .string() + .nullable() + .refine((teamId) => !requiresTeamScope || Boolean(teamId), "Please select a team to continue"), model_access_group: z.array(z.string()).optional(), }); @@ -158,12 +161,12 @@ type AddAutoRouterFormValues = z.infer>; const EMPTY_FORM_VALUES: AddAutoRouterFormValues = { auto_router_name: "", - team_id: "", + team_id: null, model_access_group: undefined, }; -const teamScopePayload = (requiresTeamScope: boolean, teamId: string): { team_id?: string } => - requiresTeamScope ? { team_id: teamId } : {}; +const teamScopePayload = (requiresTeamScope: boolean, teamId: string | null): { team_id?: string } => + requiresTeamScope && teamId ? { team_id: teamId } : {}; const BlockedReasonTooltip: React.FC<{ reason: string | null; children: React.ReactElement }> = ({ reason, @@ -458,7 +461,7 @@ const AddAutoRouterTab: React.FC = ({ const serverVerdict = await validateAutoRouterConfig( accessToken, complexityRouterConfigPayload as unknown as Record, - requiresTeamScope ? form.getValues("team_id") : undefined, + requiresTeamScope ? form.getValues("team_id") ?? undefined : undefined, ); const dryRunError = dryRunRejection(serverVerdict); if (dryRunError) { @@ -630,9 +633,7 @@ const AddAutoRouterTab: React.FC = ({ "Select the team this auto router belongs to. Only keys for this team will be able to call it.", )} > - {({ id, value, onChange }) => ( - onChange(next ?? "")} /> - )} + {({ id, value, onChange }) => } )} @@ -776,7 +777,7 @@ const AddAutoRouterTab: React.FC = ({ config={buildComplexityRouterConfig(complexityRouterConfigParams)} defaultModel={resolveComplexityDefaultModel(complexityRouterConfig, complexityRouterConfig.default_model)} routerName={watchedName} - teamId={requiresTeamScope ? watchedTeamId : undefined} + teamId={requiresTeamScope ? watchedTeamId ?? undefined : undefined} /> )} diff --git a/ui/litellm-dashboard/src/components/add_model/litellm_model_name.tsx b/ui/litellm-dashboard/src/components/add_model/litellm_model_name.tsx index bfa2df8f54d..8cf4b077f39 100644 --- a/ui/litellm-dashboard/src/components/add_model/litellm_model_name.tsx +++ b/ui/litellm-dashboard/src/components/add_model/litellm_model_name.tsx @@ -8,9 +8,9 @@ import { MountedFormField, type MountedFormValues } from "../common_components/M import { Providers } from "../provider_info_helpers"; interface LiteLLMModelNameFieldProps { - selectedProvider: Providers; + selectedProvider: string | null; providerModels: string[]; - getPlaceholder: (provider: Providers) => string; + getPlaceholder: (provider: string) => string; } const LiteLLMModelNameField: React.FC = ({ @@ -123,7 +123,7 @@ const LiteLLMModelNameField: React.FC = ({ id={control.id} value={(control.value as string | undefined) ?? ""} onBlur={control.onBlur} - placeholder={getPlaceholder(selectedProvider)} + placeholder={selectedProvider === null ? "Select a provider first" : getPlaceholder(selectedProvider)} onChange={(event) => { control.onChange(event); if (selectedProvider === Providers.Azure) { @@ -147,7 +147,7 @@ const LiteLLMModelNameField: React.FC = ({ value: "custom", }, { - label: `All ${selectedProvider} Models (Wildcard)`, + label: `All ${selectedProvider ?? "provider"} Models (Wildcard)`, value: "all-wildcard", }, ...providerModels.map((model) => ({ @@ -163,7 +163,7 @@ const LiteLLMModelNameField: React.FC = ({ value={(control.value as string | undefined) ?? ""} onChange={control.onChange} onBlur={control.onBlur} - placeholder={getPlaceholder(selectedProvider)} + placeholder={selectedProvider === null ? "Select a provider first" : getPlaceholder(selectedProvider)} /> ) } diff --git a/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx b/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx index e1a47e9792d..4add7b918f7 100644 --- a/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx +++ b/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx @@ -18,7 +18,7 @@ import { provider_map, Providers } from "../provider_info_helpers"; import { labelWithHint } from "@/components/shared/form/LabelWithHint"; interface ProviderSpecificFieldsProps { - selectedProvider: Providers; + selectedProvider: string | null; } const readTextFile = (file: File, onLoaded: (contents: string) => void) => { @@ -168,6 +168,7 @@ const ProviderSpecificFields: React.FC = ({ selecte }, [cacheEntries]); const allFields = React.useMemo(() => { + if (selectedProvider === null) return []; // First try to resolve from the in-memory cache. We support both the // enum/display-name form and the raw provider slug (e.g. "petals"). const cachedFields = diff --git a/ui/litellm-dashboard/src/components/common_components/ModelAliasManager.tsx b/ui/litellm-dashboard/src/components/common_components/ModelAliasManager.tsx index c6e24d3b470..384fa44cfde 100644 --- a/ui/litellm-dashboard/src/components/common_components/ModelAliasManager.tsx +++ b/ui/litellm-dashboard/src/components/common_components/ModelAliasManager.tsx @@ -27,8 +27,13 @@ const ModelAliasManager: React.FC = ({ showExampleConfig = true, }) => { const [aliases, setAliases] = useState([]); - const [newAlias, setNewAlias] = useState({ aliasName: "", targetModel: "" }); - const [editingAlias, setEditingAlias] = useState(null); + const [newAlias, setNewAlias] = useState<{ aliasName: string; targetModel: string | null }>({ + aliasName: "", + targetModel: null, + }); + const [editingAlias, setEditingAlias] = useState< + (Omit & { targetModel: string | null }) | null + >(null); const aliasNameId = useId(); useEffect(() => { @@ -61,7 +66,7 @@ const ModelAliasManager: React.FC = ({ const updatedAliases = [...aliases, newAliasObj]; setAliases(updatedAliases); - setNewAlias({ aliasName: "", targetModel: "" }); + setNewAlias({ aliasName: "", targetModel: null }); // Convert array back to object format and notify parent const aliasObject: { [key: string]: string } = {}; @@ -94,7 +99,8 @@ const ModelAliasManager: React.FC = ({ return; } - const updatedAliases = aliases.map((alias) => (alias.id === editingAlias.id ? editingAlias : alias)); + const savedAlias: AliasItem = { ...editingAlias, targetModel: editingAlias.targetModel }; + const updatedAliases = aliases.map((alias) => (alias.id === savedAlias.id ? savedAlias : alias)); setAliases(updatedAliases); setEditingAlias(null); diff --git a/ui/litellm-dashboard/src/components/common_components/ModelSelector.tsx b/ui/litellm-dashboard/src/components/common_components/ModelSelector.tsx index 6a6807de5c0..15b6b1644b6 100644 --- a/ui/litellm-dashboard/src/components/common_components/ModelSelector.tsx +++ b/ui/litellm-dashboard/src/components/common_components/ModelSelector.tsx @@ -9,9 +9,9 @@ const MODEL_SELECT_DEBOUNCE_MS = 500; interface ModelSelectorProps { accessToken: string; - value?: string; + value?: string | null; placeholder?: string; - onChange?: (value: string) => void; + onChange?: (value: string | null) => void; disabled?: boolean; style?: React.CSSProperties; className?: string; @@ -30,12 +30,12 @@ const ModelSelector: React.FC = ({ showLabel = true, labelText = "Select Model", }) => { - const [selectedModel, setSelectedModel] = useState(value); + const [selectedModel, setSelectedModel] = useState(value ?? null); const [showCustomModelInput, setShowCustomModelInput] = useState(false); const [modelInfo, setModelInfo] = useState([]); useEffect(() => { - setSelectedModel(value); + setSelectedModel(value ?? null); }, [value]); useEffect(() => { @@ -56,13 +56,13 @@ const ModelSelector: React.FC = ({ loadModels(); }, [accessToken]); - const onModelChange = (value: string) => { + const onModelChange = (value: string | null) => { if (value === "custom") { setShowCustomModelInput(true); - setSelectedModel(undefined); + setSelectedModel(null); } else { setShowCustomModelInput(false); - setSelectedModel(value); + setSelectedModel(value ?? null); if (onChange) { onChange(value); } @@ -71,7 +71,7 @@ const ModelSelector: React.FC = ({ const debouncedSelect = useDebouncedCallback( (value: string) => { - setSelectedModel(value); + setSelectedModel(value ?? null); onChange?.(value); }, { wait: MODEL_SELECT_DEBOUNCE_MS }, diff --git a/ui/litellm-dashboard/src/components/common_components/OrganizationDropdown.tsx b/ui/litellm-dashboard/src/components/common_components/OrganizationDropdown.tsx index 663028b2d92..f2f1504fed5 100644 --- a/ui/litellm-dashboard/src/components/common_components/OrganizationDropdown.tsx +++ b/ui/litellm-dashboard/src/components/common_components/OrganizationDropdown.tsx @@ -4,7 +4,7 @@ import { Organization } from "../networking"; interface OrganizationDropdownProps { organizations?: Organization[] | null; - value?: string; + value?: string | null; onChange?: (value: string | null) => void; disabled?: boolean; loading?: boolean; @@ -32,7 +32,7 @@ const OrganizationDropdown: React.FC = ({ sublabel: org.organization_id, }))} value={value} - onValueChange={(organizationId) => onChange?.(organizationId || null)} + onValueChange={(organizationId) => onChange?.(organizationId)} placeholder={placeholder} emptyText={loading ? "Loading organizations…" : "No organizations found"} disabled={disabled} diff --git a/ui/litellm-dashboard/src/components/common_components/ProjectDropdown.tsx b/ui/litellm-dashboard/src/components/common_components/ProjectDropdown.tsx index 0b5a84f9589..f27e1a683dd 100644 --- a/ui/litellm-dashboard/src/components/common_components/ProjectDropdown.tsx +++ b/ui/litellm-dashboard/src/components/common_components/ProjectDropdown.tsx @@ -4,8 +4,8 @@ import { ProjectResponse } from "@/app/(dashboard)/hooks/projects/useProjects"; interface ProjectDropdownProps { projects?: ProjectResponse[] | null; - value?: string; - onChange?: (value: string) => void; + value?: string | null; + onChange?: (value: string | null) => void; disabled?: boolean; loading?: boolean; /** When set, only show projects belonging to this team */ diff --git a/ui/litellm-dashboard/src/components/common_components/UserDropdown.tsx b/ui/litellm-dashboard/src/components/common_components/UserDropdown.tsx index eef1a8233bc..c80f654bbd1 100644 --- a/ui/litellm-dashboard/src/components/common_components/UserDropdown.tsx +++ b/ui/litellm-dashboard/src/components/common_components/UserDropdown.tsx @@ -47,8 +47,8 @@ const UserDropdown: React.FC = ({ value, onChange, disabled,
onChange(next === "" ? null : next)} + value={value} + onValueChange={onChange} onSearchChange={setSearch} onLoadMore={fetchNextPage} hasNextPage={hasNextPage} diff --git a/ui/litellm-dashboard/src/components/common_components/team_dropdown.tsx b/ui/litellm-dashboard/src/components/common_components/team_dropdown.tsx index 7d385c2a3f7..e10a8d1e562 100644 --- a/ui/litellm-dashboard/src/components/common_components/team_dropdown.tsx +++ b/ui/litellm-dashboard/src/components/common_components/team_dropdown.tsx @@ -4,7 +4,7 @@ import { useInfiniteTeams } from "@/app/(dashboard)/hooks/teams/useTeams"; import { Team } from "../key_team_helpers/key_list"; interface TeamDropdownProps { - value?: string; + value?: string | null; onChange?: (value: string | null) => void; /** Callback with the full Team object (or null on clear). */ onTeamSelect?: (team: Team | null) => void; @@ -46,8 +46,8 @@ const TeamDropdown: React.FC = ({ return result; }, [data]); - const handleChange = (teamId: string) => { - onChange?.(teamId || null); + const handleChange = (teamId: string | null) => { + onChange?.(teamId); if (onTeamSelect) { onTeamSelect(teamId ? teams.find((t) => t.team_id === teamId) ?? null : null); } @@ -61,7 +61,7 @@ const TeamDropdown: React.FC = ({ value: team.team_id, sublabel: team.team_id, }))} - value={value || undefined} + value={value} onValueChange={handleChange} onSearchChange={setSearch} onLoadMore={fetchNextPage} diff --git a/ui/litellm-dashboard/src/components/common_components/user_search_modal.test.tsx b/ui/litellm-dashboard/src/components/common_components/user_search_modal.integration.test.tsx similarity index 94% rename from ui/litellm-dashboard/src/components/common_components/user_search_modal.test.tsx rename to ui/litellm-dashboard/src/components/common_components/user_search_modal.integration.test.tsx index 7ca2b530235..5769d8c2f88 100644 --- a/ui/litellm-dashboard/src/components/common_components/user_search_modal.test.tsx +++ b/ui/litellm-dashboard/src/components/common_components/user_search_modal.integration.test.tsx @@ -1,4 +1,5 @@ -import { act, fireEvent, render, screen, waitFor, within } from "@testing-library/react"; +import { act, fireEvent, screen, waitFor, within } from "@testing-library/react"; +import { renderWithProviders as render } from "../../../tests/test-utils"; import userEvent, { PointerEventsCheckLevel } from "@testing-library/user-event"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import UserSearchModal from "./user_search_modal"; @@ -101,15 +102,18 @@ describe("UserSearchModal submit payload", () => { await user.click(await screen.findByRole("option", { name: "picked@example.com" })); }; - it("submits every registered field, with the untouched identity fields undefined", async () => { + it("should block submission without a selected user and after clearing the paired identity", async () => { const { user, onSubmit } = setup(); + expect(save()).toBeDisabled(); + await searchByEmail(user, "pick"); + await user.click(screen.getAllByRole("button", { name: "Clear" })[0]); + expect(screen.getByLabelText("Email")).toHaveValue(""); + expect(screen.getByLabelText("User ID")).toHaveValue(""); + expect(save()).toBeDisabled(); await user.click(save()); - await waitFor(() => expect(onSubmit).toHaveBeenCalledTimes(1)); - const values = onSubmit.mock.calls[0][0]; - expect(Object.keys(values).sort()).toEqual(["role", "user_email", "user_id"]); - expect(values).toStrictEqual({ user_email: undefined, user_id: undefined, role: "user" }); + expect(onSubmit).not.toHaveBeenCalled(); }); it("carries the picked user's email and id into the payload", async () => { @@ -130,6 +134,7 @@ describe("UserSearchModal submit payload", () => { const { onSubmit } = setup(); const user = userEvent.setup({ pointerEventsCheck: PointerEventsCheckLevel.Never }); + await searchByEmail(user, "pick"); await user.click(screen.getByLabelText("Member Role")); await user.click(await screen.findByRole("option", { name: /^admin/ })); await user.click(save()); @@ -184,6 +189,7 @@ describe("UserSearchModal submit payload", () => { it("does not submit on Enter in any field, while the button still does", async () => { const { user, onSubmit } = setup(); + await searchByEmail(user, "pick"); await user.click(getEmailSearchInput()); await user.keyboard("{Enter}"); await user.click(screen.getByLabelText("User ID")); diff --git a/ui/litellm-dashboard/src/components/common_components/user_search_modal.tsx b/ui/litellm-dashboard/src/components/common_components/user_search_modal.tsx index 20c45cd0d2e..04ac412397f 100644 --- a/ui/litellm-dashboard/src/components/common_components/user_search_modal.tsx +++ b/ui/litellm-dashboard/src/components/common_components/user_search_modal.tsx @@ -31,8 +31,8 @@ interface Role { } interface FormValues { - user_email: string | undefined; - user_id: string | undefined; + user_email: string | null | undefined; + user_id: string | null | undefined; role: string; } @@ -66,6 +66,8 @@ const UserSearchModal: React.FC = ({ }) => { const emptyValues: FormValues = { user_email: undefined, user_id: undefined, role: defaultRole }; const form = useForm({ defaultValues: emptyValues }); + const selectedUserId = form.watch("user_id"); + const selectedUserEmail = form.watch("user_email"); const [userOptions, setUserOptions] = useState([]); const [loading, setLoading] = useState(false); const [selectedField, setSelectedField] = useState<"user_email" | "user_id">("user_email"); @@ -143,19 +145,29 @@ const UserSearchModal: React.FC = ({ const renderUserSearch = ( fieldName: "user_email" | "user_id", placeholder: string, - controlProps: { id: string; value: string | undefined; onChange: (value: string | undefined) => void }, + controlProps: { + id: string; + value: string | null | undefined; + onChange: (value: string | null | undefined) => void; + }, testId?: string, ) => { const items = selectedField === fieldName ? userOptions : []; + const handleValueChange = (value: string | null) => { + if (value === null) { + form.setValue("user_email", null); + form.setValue("user_id", null); + return; + } + controlProps.onChange(value); + handleSelect(items.find((option) => option.value === value) ?? null); + }; return (
{ - controlProps.onChange(value === "" ? undefined : value); - handleSelect(items.find((option) => option.value === value) ?? null); - }} + onValueChange={handleValueChange} onSearchChange={(query: string) => handleSearch(query, fieldName)} autoHighlight="always" isLoading={loading} @@ -226,7 +238,7 @@ const UserSearchModal: React.FC = ({
- diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx index 8a62e86e842..86f18ee9b12 100644 --- a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx +++ b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx @@ -13,7 +13,7 @@ import AccessGroupTagsCombobox from "../add_model/AccessGroupTagsCombobox"; import ModelChoiceCombobox, { type ModelChoice } from "../add_model/ModelChoiceCombobox"; import { modelAvailableCall, modelPatchUpdateCall, validateAutoRouterConfig } from "../networking"; import { fetchAvailableModels, ModelGroup } from "@/components/llm_calls/fetch_models"; -import RouterConfigBuilder from "../add_model/RouterConfigBuilder"; +import RouterConfigBuilder, { type RouterConfig, serializeRouterConfig } from "../add_model/RouterConfigBuilder"; import { hydrateTierModelParams } from "../add_model/complexity_router_tiers"; import { type ActiveTierSet, @@ -452,7 +452,7 @@ const EditAutoRouterModal: React.FC = ({ const [modelInfo, setModelInfo] = useState([]); const [showValidationErrors, setShowValidationErrors] = useState(false); const [editingTiers, setEditingTiers] = useState(false); - const [routerConfig, setRouterConfig] = useState(null); + const [routerConfig, setRouterConfig] = useState(null); const [customTechnicalKeywords, setCustomTechnicalKeywords] = useState([]); const [keywordTierRules, setKeywordTierRules] = useState([]); const [escalationKeywords, setEscalationKeywords] = useState([]); @@ -706,7 +706,7 @@ const EditAutoRouterModal: React.FC = ({ // Prepare the updated litellm_params const updatedLitellmParams = { ...modelData.litellm_params, - auto_router_config: JSON.stringify(routerConfig), + auto_router_config: serializeRouterConfig(routerConfig), auto_router_default_model: values.auto_router_default_model, auto_router_embedding_model: values.auto_router_embedding_model || undefined, }; @@ -745,7 +745,7 @@ const EditAutoRouterModal: React.FC = ({ })(); } catch (error) { console.error("Error updating auto router:", error); - toast.fromError("Failed to update auto router configuration"); + toast.fromError(error); } finally { setLoading(false); } diff --git a/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx b/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx index c0c12356c6f..dbc270fa62e 100644 --- a/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx +++ b/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx @@ -96,10 +96,10 @@ export function BudgetFallbacksEditor({ value, onChange, availableModels }: Budg ({ label: m, value: m }))} - value={entry.primaryModel ?? ""} + value={entry.primaryModel} onValueChange={(v) => { const newFallbacks = entry.fallbackModels.filter((m) => m !== v); - updateEntry(entry.id, { primaryModel: v === "" ? null : v, fallbackModels: newFallbacks }); + updateEntry(entry.id, { primaryModel: v, fallbackModels: newFallbacks }); }} placeholder="Select model" emptyText="No models found" diff --git a/ui/litellm-dashboard/src/components/key_team_helpers/ModelMaxBudgetEditor.tsx b/ui/litellm-dashboard/src/components/key_team_helpers/ModelMaxBudgetEditor.tsx index ef5a3e15b52..0fab5555343 100644 --- a/ui/litellm-dashboard/src/components/key_team_helpers/ModelMaxBudgetEditor.tsx +++ b/ui/litellm-dashboard/src/components/key_team_helpers/ModelMaxBudgetEditor.tsx @@ -153,8 +153,8 @@ export function ModelMaxBudgetEditor({ ({ label: model, value: model }))} - value={entry.model ?? ""} - onValueChange={(model) => updateEntry(entry.id, { model: model === "" ? null : model })} + value={entry.model} + onValueChange={(model) => updateEntry(entry.id, { model })} placeholder="Select model" emptyText="No models found" disabled={!premiumUser} diff --git a/ui/litellm-dashboard/src/components/model_add/CredentialModal.tsx b/ui/litellm-dashboard/src/components/model_add/CredentialModal.tsx index 8108cc1c20e..fdc02b4fc41 100644 --- a/ui/litellm-dashboard/src/components/model_add/CredentialModal.tsx +++ b/ui/litellm-dashboard/src/components/model_add/CredentialModal.tsx @@ -42,7 +42,7 @@ export default function CredentialModal({ existingCredential = null, }: CredentialModalProps) { const isEdit = mode === "edit"; - const [selectedProvider, setSelectedProvider] = useState( + const [selectedProvider, setSelectedProvider] = useState( (existingCredential?.credential_info.custom_llm_provider as Providers) ?? Providers.OpenAI, ); @@ -110,7 +110,7 @@ export default function CredentialModal({ {(control) => ( { control.onChange(value); - resetCredentialFormOnProviderChange(formAdapter, value as Providers, setSelectedProvider); + resetCredentialFormOnProviderChange(formAdapter, value, setSelectedProvider); }} /> )} diff --git a/ui/litellm-dashboard/src/components/model_add/credential_form_helpers.ts b/ui/litellm-dashboard/src/components/model_add/credential_form_helpers.ts index 7190c539c81..e7b1e861811 100644 --- a/ui/litellm-dashboard/src/components/model_add/credential_form_helpers.ts +++ b/ui/litellm-dashboard/src/components/model_add/credential_form_helpers.ts @@ -1,5 +1,3 @@ -import { Providers } from "../provider_info_helpers"; - interface CredentialFormAdapter { getFieldValue: (field: string) => unknown; resetFields: () => void; @@ -25,8 +23,8 @@ interface CredentialFormAdapter { */ export function resetCredentialFormOnProviderChange( form: CredentialFormAdapter, - newProvider: Providers, - setSelectedProvider: (p: Providers) => void, + newProvider: string | null, + setSelectedProvider: (p: string | null) => void, ): void { const preservedName = form.getFieldValue("credential_name"); form.resetFields(); diff --git a/ui/litellm-dashboard/src/components/organisms/createKeyPayload.ts b/ui/litellm-dashboard/src/components/organisms/createKeyPayload.ts index 37a973d5def..311e9357825 100644 --- a/ui/litellm-dashboard/src/components/organisms/createKeyPayload.ts +++ b/ui/litellm-dashboard/src/components/organisms/createKeyPayload.ts @@ -202,6 +202,8 @@ export const buildKeyCreatePayload = (input: KeyCreateInput): KeyPayloadResult = endpoint: input.keyOwner === "service_account" ? "service_account" : "standard", payload: { ...withoutKeys(values, dropped), + ...(values.organization_id === null && { organization_id: undefined }), + ...(values.project_id === null && { project_id: undefined }), ...(input.keyOwner === "you" && { user_id: input.userID }), ...(input.keyOwner === "agent" && { agent_id: input.selectedAgentId }), ...(input.autoRotationEnabled && { auto_rotate: true, rotation_interval: input.rotationInterval }), diff --git a/ui/litellm-dashboard/src/components/organisms/create_key_button.integration.test.tsx b/ui/litellm-dashboard/src/components/organisms/create_key_button.integration.test.tsx index 2bf39bf4cd7..0d5d9f5ec8d 100644 --- a/ui/litellm-dashboard/src/components/organisms/create_key_button.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/organisms/create_key_button.integration.test.tsx @@ -16,7 +16,7 @@ const state = vi.hoisted(() => ({ can: {} as Record, uiSettings: {} as Record, tags: {} as Record, - teams: [] as { team_id: string; team_alias: string; models: string[] }[], + teams: [] as { team_id: string; team_alias: string; models: string[]; organization_id?: string }[], organizations: [] as { organization_id: string; organization_alias: string }[], accessGroups: [] as { access_group_id: string; access_group_name: string }[], projects: [] as { project_id: string; project_alias: string; team_id?: string; models?: string[] }[], @@ -806,6 +806,36 @@ describe("CreateKey", () => { expect((await createdPayload()).organization_id).toBe("org-1"); }); + it("discards the old project and team when the organization changes", async () => { + state.uiSettings = { enable_projects_ui: true }; + state.organizations = [ + { organization_id: "scope-silver", organization_alias: "Silver" }, + { organization_id: "scope-copper", organization_alias: "Copper" }, + ]; + state.teams = [{ team_id: "group-maple", team_alias: "Maple", organization_id: "scope-silver", models: [] }]; + state.projects = [{ project_id: "project-orbit", project_alias: "Orbit", team_id: "group-maple", models: [] }]; + await openModal({ teams: state.teams as Team[] }); + await nameTheKey(); + await userEvent.click(await screen.findByLabelText("Organization")); + await userEvent.click(await screen.findByRole("option", { name: /Silver/ })); + await userEvent.click(await screen.findByLabelText("Project")); + await userEvent.click(await screen.findByRole("option", { name: /Orbit/ })); + await waitFor(() => expect(screen.getByLabelText("Team")).toHaveValue("Maple")); + expect(screen.getByLabelText("Team")).toBeDisabled(); + + await userEvent.click(screen.getByLabelText("Organization")); + await userEvent.click(await screen.findByRole("option", { name: /Copper/ })); + expect(screen.getByLabelText("Project")).toHaveValue(""); + expect(screen.getByLabelText("Team")).toHaveValue(""); + expect(screen.getByLabelText("Team")).toBeEnabled(); + await submit(); + + const payload = JSON.parse(JSON.stringify(await createdPayload())); + expect(payload.organization_id).toBe("scope-copper"); + expect(payload.team_id).toBeNull(); + expect(payload).not.toHaveProperty("project_id"); + }); + it("drops organization_id when the chosen organization is cleared again", async () => { state.organizations = [{ organization_id: "org-1", organization_alias: "Engineering" }]; await openModal(); diff --git a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx index 831cb0cf6a6..82580fd4667 100644 --- a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx +++ b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx @@ -592,35 +592,35 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp }; const changeOrganization = (write: FieldWrite) => (orgId: string | null) => { - write(orgId ?? undefined); + write(orgId); setSelectedOrganizationId(orgId); // Clear team and project when org changes setSelectedCreateKeyTeam(null); setSelectedProjectId(null); - form.setValue("team_id", undefined); - form.setValue("project_id", undefined); + form.setValue("team_id", null); + form.setValue("project_id", null); }; const selectTeam = (team: Team | null) => { setSelectedCreateKeyTeam(team); setSelectedProjectId(null); - form.setValue("project_id", undefined); + form.setValue("project_id", null); // Auto-populate org from team for non-admin users if (team?.organization_id) { setSelectedOrganizationId(team.organization_id); form.setValue("organization_id", team.organization_id); } else if (!team) { setSelectedOrganizationId(null); - form.setValue("organization_id", undefined); + form.setValue("organization_id", null); } }; - const changeProject = (write: FieldWrite) => (projectId: string) => { + const changeProject = (write: FieldWrite) => (projectId: string | null) => { write(projectId); if (!projectId) { setSelectedProjectId(null); setSelectedCreateKeyTeam(null); - form.setValue("team_id", undefined); + form.setValue("team_id", null); return; } setSelectedProjectId(projectId); @@ -756,8 +756,8 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp inputId="create-key-agent" placeholder="Select an agent" emptyText="No agents found" - value={selectedAgentId ?? undefined} - onValueChange={(value) => setSelectedAgentId(value === "" ? null : value)} + value={selectedAgentId} + onValueChange={setSelectedAgentId} options={agentsList.map((a) => ({ label: a.agent_name || a.agent_id, value: a.agent_id, @@ -783,7 +783,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp {(control) => ( = ({ team, teams, data, addKey, autoOp {(control) => ( = ({ team, teams, data, addKey, autoOp {(control) => ( { return providerPlaceholderMap[resolvedProvider] ?? "gpt-3.5-turbo"; }; -export const getProviderModels = (provider: Providers, modelMap: any): Array => { +export const getProviderModels = (provider: string, modelMap: any): Array => { let providerKey = provider; let custom_llm_provider = provider_map[providerKey]; diff --git a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.test.tsx b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx similarity index 89% rename from ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.test.tsx rename to ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx index 04c8d3e9012..2b948ca8420 100644 --- a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.test.tsx +++ b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx @@ -1,4 +1,4 @@ -import { fireEvent, render, screen, waitFor } from "@testing-library/react"; +import { fireEvent, renderWithProviders as render, screen, waitFor } from "../../../tests/test-utils"; import userEvent from "@testing-library/user-event"; import { useState } from "react"; import { describe, expect, it, vi } from "vitest"; @@ -51,7 +51,7 @@ describe("PaginatedSearchSelect", () => { const onSearchChange = vi.fn(); function Controlled() { - const [value, setValue] = useState(""); + const [value, setValue] = useState(null); return ( { expect(onSearchChange).not.toHaveBeenCalled(); }); - it("still reports a cleared input so the unfiltered page comes back", async () => { + it("should keep a cleared selection empty after a late page arrives and reset the query", async () => { const user = userEvent.setup(); const onSearchChange = vi.fn(); - renderSelect({ onSearchChange, value: "alias-alpha" }); - - await user.click(document.querySelector('[data-slot="combobox-clear"]') as HTMLElement); - - await waitFor(() => expect(onSearchChange).toHaveBeenCalledWith("")); + const onValueChange = vi.fn(); + function Controlled({ options }: { options: SearchSelectOption[] }) { + const [value, setValue] = useState("alias-alpha"); + return ( + { + setValue(next); + onValueChange(next); + }} + /> + ); + } + const { rerender } = render(); + await user.click(screen.getByRole("button", { name: "Clear" })); + expect(onValueChange).toHaveBeenLastCalledWith(null); + rerender( ({ ...option }))} />); + await waitFor(() => expect(onSearchChange).toHaveBeenLastCalledWith("")); + expect(screen.getByRole("combobox")).toHaveValue(""); + expect(screen.queryByRole("button", { name: "Clear" })).not.toBeInTheDocument(); + await chooseSelectOption(user, screen.getByRole("combobox"), "alias-beta"); + expect(onValueChange).toHaveBeenLastCalledWith("alias-beta"); }); it("requests the next page once the list is scrolled near the bottom", async () => { @@ -147,7 +166,7 @@ describe("PaginatedSearchSelect", () => { function ServerBacked() { const [search, setSearch] = useState(""); - const [value, setValue] = useState("alias-alpha"); + const [value, setValue] = useState("alias-alpha"); const freshlyBuiltOptions = OPTIONS.filter((option) => option.label.includes(search)).map((option) => ({ ...option, })); @@ -236,7 +255,7 @@ describe("PaginatedSearchSelect", () => { function Refetching() { const [options, setOptions] = useState([{ label: "Beta Team", value: "team-2" }]); - const [value, setValue] = useState(""); + const [value, setValue] = useState(null); return ( <> { function ServerBacked() { const [search, setSearch] = useState(""); - const [value, setValue] = useState(""); + const [value, setValue] = useState(null); return ( option.label.includes(search))} diff --git a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.tsx b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.tsx index 4b7ef7401c7..1314afdf2bf 100644 --- a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.tsx +++ b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.tsx @@ -17,8 +17,8 @@ import { usePaginatedCombobox } from "./usePaginatedCombobox"; interface PaginatedSearchSelectProps { options: SearchSelectOption[]; - value?: string; - onValueChange: (value: string) => void; + value?: string | null; + onValueChange: (value: string | null) => void; onSearchChange: (query: string) => void; onLoadMore?: () => void; hasNextPage?: boolean; @@ -86,7 +86,7 @@ export function PaginatedSearchSelect({ }; const selected = useMemo(() => { - if (value === undefined || value === "") return null; + if (value == null || value === "") return null; return ( options.find((option) => option.value === value) ?? (pickedOption?.value === value ? pickedOption : { label: value, value }) @@ -118,7 +118,7 @@ export function PaginatedSearchSelect({ inputValue={typedQuery ?? selected?.label ?? ""} onValueChange={(item: SearchSelectOption | null) => { setPickedOption(item); - onValueChange(item?.value ?? ""); + onValueChange(item?.value ?? null); }} onInputValueChange={(next, eventDetails) => handleTypedInput(next, eventDetails.reason)} onOpenChange={(nextOpen, eventDetails) => handleOpenChange(nextOpen, eventDetails.reason)} @@ -139,7 +139,7 @@ export function PaginatedSearchSelect({ onKeyDown={snapshotWholeSelection} onPaste={snapshotWholeSelection} placeholder={placeholder} - showClear={value !== undefined && value !== ""} + showClear={value != null && value !== ""} className={`w-full ${className ?? ""}`} /> diff --git a/ui/litellm-dashboard/src/components/shared/SearchSelect.test.tsx b/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx similarity index 76% rename from ui/litellm-dashboard/src/components/shared/SearchSelect.test.tsx rename to ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx index 62a510268dd..c981010dff9 100644 --- a/ui/litellm-dashboard/src/components/shared/SearchSelect.test.tsx +++ b/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx @@ -1,4 +1,5 @@ -import { fireEvent, render, screen } from "@testing-library/react"; +import { fireEvent, renderWithProviders as render, screen } from "../../../tests/test-utils"; +import { useState } from "react"; import userEvent from "@testing-library/user-event"; import { describe, expect, it, vi } from "vitest"; @@ -36,11 +37,30 @@ describe("SearchSelect", () => { expect(screen.getByRole("combobox")).toHaveValue("Growth"); }); - it("shows a clear control only when a value is selected", () => { - const { rerender } = render(); - expect(document.querySelector('[data-slot="combobox-clear"]')).toBeNull(); - rerender(); - expect(document.querySelector('[data-slot="combobox-clear"]')).not.toBeNull(); + it("should clear to null and allow selecting again through the real control", async () => { + const onValueChange = vi.fn(); + const user = userEvent.setup(); + function Controlled() { + const [value, setValue] = useState(null); + return ( + { + setValue(next); + onValueChange(next); + }} + /> + ); + } + render(); + expect(screen.queryByRole("button", { name: "Clear" })).not.toBeInTheDocument(); + await chooseSelectOption(user, screen.getByRole("combobox"), "Growth"); + await user.click(screen.getByRole("button", { name: "Clear" })); + expect(onValueChange).toHaveBeenLastCalledWith(null); + expect(screen.getByRole("combobox")).toHaveValue(""); + await chooseSelectOption(user, screen.getByRole("combobox"), "Data Team"); + expect(onValueChange).toHaveBeenLastCalledWith("team-3"); }); it("filters the options client-side as you type", async () => { diff --git a/ui/litellm-dashboard/src/components/shared/SearchSelect.tsx b/ui/litellm-dashboard/src/components/shared/SearchSelect.tsx index 53a5371d72d..21d38e9458c 100644 --- a/ui/litellm-dashboard/src/components/shared/SearchSelect.tsx +++ b/ui/litellm-dashboard/src/components/shared/SearchSelect.tsx @@ -21,8 +21,8 @@ export interface SearchSelectOption { interface SearchSelectProps { options: SearchSelectOption[]; - value?: string; - onValueChange: (value: string) => void; + value?: string | null; + onValueChange: (value: string | null) => void; placeholder?: string; emptyText?: string; disabled?: boolean; @@ -51,9 +51,7 @@ export function SearchSelect({ "aria-label": ariaLabel, }: SearchSelectProps) { const selected = - value === undefined || value === "" - ? null - : options.find((option) => option.value === value) ?? { label: value, value }; + value == null || value === "" ? null : options.find((option) => option.value === value) ?? { label: value, value }; const items = selected !== null && !options.some((option) => option.value === selected.value) ? [selected, ...options] : options; @@ -61,7 +59,7 @@ export function SearchSelect({ onValueChange(item?.value ?? "")} + onValueChange={(item: SearchSelectOption | null) => onValueChange(item?.value ?? null)} isItemEqualToValue={(a: SearchSelectOption, b: SearchSelectOption) => a.value === b.value} itemToStringLabel={(item: SearchSelectOption) => item.label} filter={matchesQuery} diff --git a/ui/litellm-dashboard/src/components/team/TeamInfo.tsx b/ui/litellm-dashboard/src/components/team/TeamInfo.tsx index 988e2e082aa..8e3a17c2622 100644 --- a/ui/litellm-dashboard/src/components/team/TeamInfo.tsx +++ b/ui/litellm-dashboard/src/components/team/TeamInfo.tsx @@ -333,7 +333,10 @@ const teamUpdateFieldsSchema = z.object({ modelLimits: z .array( z.object({ - model: z.string().min(1, "Missing model"), + model: z + .string() + .nullable() + .refine((model) => Boolean(model), "Missing model"), tpm: z.number().nullish(), rpm: z.number().nullish(), }), @@ -1879,7 +1882,7 @@ const TeamInfoView: React.FC = ({ onChange(next === "" ? null : next)} + onValueChange={onChange} options={userOrganizations.map((org) => ({ value: org.organization_id ?? "", label: org.organization_alias || org.organization_id || "", diff --git a/ui/litellm-dashboard/src/components/templates/key_edit_view.test.tsx b/ui/litellm-dashboard/src/components/templates/key_edit_view.integration.test.tsx similarity index 98% rename from ui/litellm-dashboard/src/components/templates/key_edit_view.test.tsx rename to ui/litellm-dashboard/src/components/templates/key_edit_view.integration.test.tsx index b2c2a381b42..7c2226f0369 100644 --- a/ui/litellm-dashboard/src/components/templates/key_edit_view.test.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_edit_view.integration.test.tsx @@ -1483,11 +1483,11 @@ describe("KeyEditView", () => { }); }); - it("submits organization_id as null after the organization is cleared", async () => { + it("clears the organization and its dependent team in the update payload", async () => { const onSubmit = vi.fn().mockResolvedValue(undefined); renderWithProviders( {}} onSubmit={onSubmit} accessToken="" @@ -1504,9 +1504,33 @@ describe("KeyEditView", () => { await userEvent.click(screen.getByRole("button", { name: /save changes/i })); await waitFor(() => { - expect(onSubmit).toHaveBeenCalledWith(expect.objectContaining({ organization_id: null })); + expect(onSubmit).toHaveBeenCalledWith(expect.objectContaining({ organization_id: null, team_id: null })); }); - expect(JSON.parse(JSON.stringify(onSubmit.mock.calls[0][0]))).toHaveProperty("organization_id", null); + expect(JSON.parse(JSON.stringify(onSubmit.mock.calls[0][0]))).toMatchObject({ + organization_id: null, + team_id: null, + }); + }); + + it("keeps project key relationships locked and omits unsupported project updates", async () => { + const onSubmit = vi.fn().mockResolvedValue(undefined); + renderWithProviders( + {}} + onSubmit={onSubmit} + accessToken="" + userID="" + userRole="Admin" + premiumUser={false} + />, + ); + expect(await screen.findByRole("combobox", { name: "Organization" })).toBeDisabled(); + expect(screen.getByRole("combobox", { name: "Team ID" })).toBeDisabled(); + await userEvent.click(screen.getByRole("button", { name: /save changes/i })); + await waitFor(() => expect(onSubmit).toHaveBeenCalledTimes(1)); + expect(onSubmit.mock.calls[0][0]).toMatchObject({ organization_id: "org-1", team_id: "group-maple" }); + expect(onSubmit.mock.calls[0][0]).not.toHaveProperty("project_id"); }); }); diff --git a/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx b/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx index e464dfe5008..f0d63f9671d 100644 --- a/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx @@ -305,7 +305,7 @@ export function KeyEditView({ const handleOrganizationChange = (setField: (value: string | null) => void, orgId: string | null) => { setField(orgId); setSelectedOrganizationId(orgId); - form.setValue("team_id", undefined); + form.setValue("team_id", null); }; const handleTeamChange = (setField: (value: string | null) => void, teamId: string | null) => { @@ -316,7 +316,7 @@ export function KeyEditView({ form.setValue("organization_id", selectedTeam.organization_id); } else if (!teamId) { setSelectedOrganizationId(null); - form.setValue("organization_id", undefined); + form.setValue("organization_id", null); } }; @@ -769,14 +769,15 @@ export function KeyEditView({ "Organization", "The organization this key belongs to. Selecting an organization filters the available teams.", )} + description={hasProject ? "Organization is locked because this key belongs to a project" : undefined} > {({ value, onChange, id }) => ( handleOrganizationChange(onChange, orgId)} /> )} @@ -786,15 +787,13 @@ export function KeyEditView({ control={form.control} name="team_id" label="Team ID" - description={ - enableProjectsUI && hasProject ? "Team is locked because this key belongs to a project" : undefined - } + description={hasProject ? "Team is locked because this key belongs to a project" : undefined} > {({ value, onChange, id }) => (