mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
refactor(rust): rename litellm-core to litellm-inference (#44802)
* refactor(rust): rename litellm-core to litellm-inference Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(rust): expose inference base API for format crates Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(rust): move shared inference test helpers behind test-support Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
10444df3a0
commit
1a9b533e71
105 changed files with 310 additions and 276 deletions
97
litellm-rust/Cargo.lock
generated
97
litellm-rust/Cargo.lock
generated
|
|
@ -3697,50 +3697,6 @@ dependencies = [
|
|||
"thiserror 2.0.19",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-core"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"base64 0.22.1",
|
||||
"bytes",
|
||||
"futures-util",
|
||||
"litellm-auth",
|
||||
"litellm-auth-aws",
|
||||
"litellm-auth-gcp",
|
||||
"litellm-cache",
|
||||
"litellm-cache-memory",
|
||||
"litellm-cache-response",
|
||||
"litellm-core-utils",
|
||||
"litellm-framer",
|
||||
"litellm-host",
|
||||
"litellm-host-native",
|
||||
"litellm-http",
|
||||
"litellm-llms",
|
||||
"litellm-llms-types",
|
||||
"litellm-secrets",
|
||||
"litellm-tracing",
|
||||
"mime_guess",
|
||||
"moka",
|
||||
"rand 0.8.7",
|
||||
"reqwest 0.12.28",
|
||||
"rstest",
|
||||
"rstest_reuse",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"sha2 0.10.9",
|
||||
"strum",
|
||||
"subtle",
|
||||
"thiserror 2.0.19",
|
||||
"time",
|
||||
"tokio",
|
||||
"tokio-tungstenite",
|
||||
"tokio-util",
|
||||
"tracing",
|
||||
"url",
|
||||
"veil",
|
||||
"wiremock",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-core-utils"
|
||||
version = "0.1.0"
|
||||
|
|
@ -3821,12 +3777,12 @@ dependencies = [
|
|||
"http-body-util",
|
||||
"litellm-auth-types",
|
||||
"litellm-config",
|
||||
"litellm-core",
|
||||
"litellm-gateway-auth",
|
||||
"litellm-gateway-inference",
|
||||
"litellm-gateway-mcp",
|
||||
"litellm-gateway-ui",
|
||||
"litellm-http",
|
||||
"litellm-inference",
|
||||
"litellm-llms",
|
||||
"litellm-secrets",
|
||||
"litellm-tracing",
|
||||
|
|
@ -3875,11 +3831,11 @@ dependencies = [
|
|||
"litellm-auth",
|
||||
"litellm-cache-memory",
|
||||
"litellm-cache-response",
|
||||
"litellm-core",
|
||||
"litellm-gateway-auth",
|
||||
"litellm-host",
|
||||
"litellm-host-http",
|
||||
"litellm-http",
|
||||
"litellm-inference",
|
||||
"litellm-llms",
|
||||
"litellm-llms-types",
|
||||
"litellm-router",
|
||||
|
|
@ -4038,6 +3994,51 @@ dependencies = [
|
|||
"webpki-roots",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-inference"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"base64 0.22.1",
|
||||
"bytes",
|
||||
"futures-util",
|
||||
"litellm-auth",
|
||||
"litellm-auth-aws",
|
||||
"litellm-auth-gcp",
|
||||
"litellm-cache",
|
||||
"litellm-cache-memory",
|
||||
"litellm-cache-response",
|
||||
"litellm-core-utils",
|
||||
"litellm-framer",
|
||||
"litellm-host",
|
||||
"litellm-host-native",
|
||||
"litellm-http",
|
||||
"litellm-inference",
|
||||
"litellm-llms",
|
||||
"litellm-llms-types",
|
||||
"litellm-secrets",
|
||||
"litellm-tracing",
|
||||
"mime_guess",
|
||||
"moka",
|
||||
"rand 0.8.7",
|
||||
"reqwest 0.12.28",
|
||||
"rstest",
|
||||
"rstest_reuse",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"sha2 0.10.9",
|
||||
"strum",
|
||||
"subtle",
|
||||
"thiserror 2.0.19",
|
||||
"time",
|
||||
"tokio",
|
||||
"tokio-tungstenite",
|
||||
"tokio-util",
|
||||
"tracing",
|
||||
"url",
|
||||
"veil",
|
||||
"wiremock",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-llms"
|
||||
version = "0.1.0"
|
||||
|
|
@ -4116,11 +4117,11 @@ dependencies = [
|
|||
"litellm-cache-redis",
|
||||
"litellm-cache-response",
|
||||
"litellm-callbacks-legacy-python",
|
||||
"litellm-core",
|
||||
"litellm-core-utils",
|
||||
"litellm-host",
|
||||
"litellm-host-python",
|
||||
"litellm-http",
|
||||
"litellm-inference",
|
||||
"litellm-llms",
|
||||
"litellm-llms-types",
|
||||
"litellm-secrets",
|
||||
|
|
@ -4171,7 +4172,7 @@ name = "litellm-router"
|
|||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"litellm-config",
|
||||
"litellm-core",
|
||||
"litellm-inference",
|
||||
"rstest",
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ litellm-traces = { path = "crates/traces" }
|
|||
litellm-traces-cache = { path = "crates/traces-cache" }
|
||||
litellm-traces-clickhouse = { path = "crates/traces-clickhouse" }
|
||||
litellm-storage-clickhouse = { path = "crates/storage-clickhouse" }
|
||||
litellm-core = { path = "crates/core" }
|
||||
litellm-inference = { path = "crates/inference" }
|
||||
litellm-gateway-mcp = { path = "crates/gateway-mcp" }
|
||||
litellm-gateway = { path = "crates/gateway" }
|
||||
litellm-gateway-inference = { path = "crates/gateway-inference" }
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ base64.workspace = true
|
|||
bytes.workspace = true
|
||||
litellm-auth.workspace = true
|
||||
litellm-gateway-auth.workspace = true
|
||||
litellm-core.workspace = true
|
||||
litellm-inference.workspace = true
|
||||
litellm-host-http.workspace = true
|
||||
litellm-host.workspace = true
|
||||
litellm-http.workspace = true
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ use std::{path::Path, sync::Arc};
|
|||
|
||||
use axum::{Json, extract::State, response::IntoResponse};
|
||||
use base64::{Engine, engine::general_purpose::STANDARD};
|
||||
use litellm_core::audio_transcription::types::AudioTranscriptionRequest;
|
||||
use litellm_inference::audio_transcription::types::AudioTranscriptionRequest;
|
||||
use serde_json::{Value, json};
|
||||
|
||||
use crate::{
|
||||
|
|
|
|||
|
|
@ -72,26 +72,26 @@ fn duration(seconds: f64) -> Result<Duration, Error> {
|
|||
#[derive(Clone, Default)]
|
||||
pub(crate) struct CacheHeaders(std::sync::Arc<std::sync::OnceLock<String>>);
|
||||
|
||||
impl litellm_host::interceptors::Interceptors<litellm_core::RouteError> for CacheHeaders {
|
||||
impl litellm_host::interceptors::Interceptors<litellm_inference::RouteError> for CacheHeaders {
|
||||
async fn before_provider_request(
|
||||
&self,
|
||||
wire: litellm_host::interceptors::WireRequest,
|
||||
_: litellm_host::interceptors::RequestContext,
|
||||
) -> Result<litellm_host::interceptors::WireRequest, litellm_core::RouteError> {
|
||||
) -> Result<litellm_host::interceptors::WireRequest, litellm_inference::RouteError> {
|
||||
Ok(wire)
|
||||
}
|
||||
|
||||
async fn after_provider_response(
|
||||
&self,
|
||||
_: litellm_host::interceptors::RawResponse,
|
||||
) -> Result<(), litellm_core::RouteError> {
|
||||
) -> Result<(), litellm_inference::RouteError> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn result_ready(
|
||||
&self,
|
||||
facts: litellm_host::interceptors::ExecutionFacts,
|
||||
) -> Result<(), litellm_core::RouteError> {
|
||||
) -> Result<(), litellm_inference::RouteError> {
|
||||
if let litellm_host::interceptors::ResultSource::Cache { key } = facts.source {
|
||||
let _ = self.0.set(key);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ use axum::{
|
|||
extract::{Path, State},
|
||||
response::{IntoResponse, Response},
|
||||
};
|
||||
use litellm_core::chat_completions::types::ChatCompletionsCall;
|
||||
use litellm_inference::chat_completions::types::ChatCompletionsCall;
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
use crate::{Error, Gateway, JsonObject, request};
|
||||
|
|
|
|||
|
|
@ -3,8 +3,8 @@ use axum::{
|
|||
Json,
|
||||
response::{IntoResponse, Response},
|
||||
};
|
||||
use litellm_core::RouteError;
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_inference::RouteError;
|
||||
use litellm_llms::base_llm::ocr::error::Error as OcrError;
|
||||
use serde_json::{Map, Value, json};
|
||||
|
||||
|
|
|
|||
|
|
@ -15,11 +15,11 @@ mod responses;
|
|||
use std::sync::Arc;
|
||||
|
||||
use axum::{Router, routing::post};
|
||||
use litellm_core::{
|
||||
use litellm_http::{ClientVariant, HttpClientConfig, media::UrlPolicy};
|
||||
use litellm_inference::{
|
||||
audio_transcription::AudioTranscriptionRoute, chat_completions::ChatCompletionsRoute,
|
||||
messages::MessagesRoute, ocr::OcrRoute, resources::CoreResources, responses::ResponsesRoute,
|
||||
};
|
||||
use litellm_http::{ClientVariant, HttpClientConfig, media::UrlPolicy};
|
||||
use litellm_llms::base_llm::ocr::{handler::OcrClient, settings::OcrSettings};
|
||||
use litellm_secrets::source::SecretSource;
|
||||
|
||||
|
|
|
|||
|
|
@ -10,8 +10,8 @@ use axum::{
|
|||
http::HeaderMap,
|
||||
response::{IntoResponse, Response},
|
||||
};
|
||||
use litellm_core::messages::{MessagesCall, messages_body, route::Messages};
|
||||
use litellm_host_http::Sse;
|
||||
use litellm_inference::messages::{MessagesCall, messages_body, route::Messages};
|
||||
use litellm_llms_types::headers::{ProviderSpecificHeader, ProviderSpecificHeaders};
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ use std::sync::Arc;
|
|||
|
||||
use axum::{Json, extract::State, http::HeaderMap, response::IntoResponse};
|
||||
use litellm_auth::SecretValue;
|
||||
use litellm_core::ocr::types::{LiteLLMOcrRequest, OcrConnectionInputs, OcrDocumentInput};
|
||||
use litellm_inference::ocr::types::{LiteLLMOcrRequest, OcrConnectionInputs, OcrDocumentInput};
|
||||
use litellm_llms::base_llm::ocr::transformation::decode_request_value;
|
||||
use litellm_llms_types::formats::ocr::OcrDocument;
|
||||
use serde_json::Value;
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
use std::sync::Arc;
|
||||
|
||||
use axum::{Json, body::Bytes, extract::State, response::Response};
|
||||
use litellm_core::responses::{route::Responses, types::ResponsesCall};
|
||||
use litellm_gateway_auth::AuthenticatedRequest;
|
||||
use litellm_host_http::Sse;
|
||||
use litellm_inference::responses::{route::Responses, types::ResponsesCall};
|
||||
use serde_json::json;
|
||||
|
||||
use crate::{Error, Gateway, JsonObject, request};
|
||||
|
|
|
|||
|
|
@ -6,13 +6,13 @@ use axum::{
|
|||
body::{Body, Bytes},
|
||||
http::Request,
|
||||
};
|
||||
use litellm_core::{
|
||||
chat_completions::{ChatCompletionsRoute, types::ChatCompletionsRequest},
|
||||
resources::CoreResources,
|
||||
};
|
||||
use litellm_http::{
|
||||
ClientVariant, HttpClientPool, HttpSettings, Resolution, media::PublicDnsResolver,
|
||||
};
|
||||
use litellm_inference::{
|
||||
chat_completions::{ChatCompletionsRoute, types::ChatCompletionsRequest},
|
||||
resources::CoreResources,
|
||||
};
|
||||
use rstest::rstest;
|
||||
use serde_json::{Value, json};
|
||||
use tower::ServiceExt;
|
||||
|
|
|
|||
|
|
@ -10,9 +10,9 @@ use axum::{
|
|||
response::Response,
|
||||
};
|
||||
use futures_util::future::BoxFuture;
|
||||
use litellm_core::resources::CoreResources;
|
||||
use litellm_gateway_inference::{Deployment, Gateway, router};
|
||||
use litellm_http::{HttpClientPool, HttpSettings, Resolution, media::PublicDnsResolver};
|
||||
use litellm_inference::resources::CoreResources;
|
||||
use litellm_secrets::{SecretValue, source::SecretSource};
|
||||
use serde_json::Value;
|
||||
use tower::ServiceExt;
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ repository.workspace = true
|
|||
axum.workspace = true
|
||||
envy = "0.4.2"
|
||||
http-body-util = "0.1"
|
||||
litellm-core.workspace = true
|
||||
litellm-inference.workspace = true
|
||||
litellm-gateway-inference.workspace = true
|
||||
litellm-gateway-mcp.workspace = true
|
||||
tokio-util = "0.7"
|
||||
|
|
|
|||
|
|
@ -15,13 +15,13 @@ use axum::{
|
|||
use http_body_util::BodyExt;
|
||||
|
||||
use litellm_config::Config;
|
||||
use litellm_core::resources::CoreResources;
|
||||
use litellm_gateway_auth::Auth;
|
||||
use litellm_gateway_inference::{Gateway, ModelRouter};
|
||||
use litellm_http::{
|
||||
ClientVariant, HttpClientPool, HttpSettings, HttpSettingsLayer, Resolution, SslVerify,
|
||||
media::PublicDnsResolver,
|
||||
};
|
||||
use litellm_inference::resources::CoreResources;
|
||||
use litellm_secrets::source::EnvironmentSecrets;
|
||||
use litellm_tracing::ByteChunk;
|
||||
use uuid::Uuid;
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
litellm-core owns route orchestration. Messages and HTTP Responses return `litellm_host::call::CallOutput`, containing either a completed response or a stream head and chunks. OCR and currently non-streaming Chat Completions return their completed response directly
|
||||
litellm-inference owns route orchestration. Messages and HTTP Responses return `litellm_host::call::CallOutput`, containing either a completed response or a stream head and chunks. OCR and currently non-streaming Chat Completions return their completed response directly
|
||||
|
||||
Hosts assemble route objects from shared `CoreResources`, HTTP settings, and secret sources. Each route owns its provider client and authentication dependencies. Gateway routes live for the gateway lifetime; Python assembles routes per call from its settings snapshot
|
||||
|
||||
|
|
@ -10,7 +10,7 @@ Responses WebSocket sessions remain separate from the HTTP call driver because a
|
|||
|
||||
## Crate layering
|
||||
|
||||
For Messages, Responses, Chat Completions, OCR, and other API formats, `core/src/<format>/` owns orchestration. Shared API data contracts belong in `litellm-llms-types`, adapter contracts and shared transformation machinery in `llms/src/base_llm/<format>/`, and provider policy in `llms/src/<provider>/<format>/`. A repeated format directory name does not imply interchangeable responsibilities. Select concrete adapters here, then invoke their contracts instead of applying one provider's policy to every call. Route types describe call envelopes and execution state, not duplicate public payload schemas
|
||||
For Messages, Responses, Chat Completions, OCR, and other API formats, `inference/src/<format>/` owns orchestration. Shared API data contracts belong in `litellm-llms-types`, adapter contracts and shared transformation machinery in `llms/src/base_llm/<format>/`, and provider policy in `llms/src/<provider>/<format>/`. A repeated format directory name does not imply interchangeable responsibilities. Select concrete adapters here, then invoke their contracts instead of applying one provider's policy to every call. Route types describe call envelopes and execution state, not duplicate public payload schemas
|
||||
|
||||
Crates separate API data, transformations, transport, and orchestration. Python package names identify counterparts, not ownership. Dependencies only point down:
|
||||
|
||||
|
|
@ -18,9 +18,9 @@ Crates separate API data, transformations, transport, and orchestration. Python
|
|||
- `litellm-core-utils` mirrors `litellm/litellm_core_utils/`: pure helpers (provider resolution, prompt factory, call arguments, settings lookup and layer merge), no network I/O
|
||||
- `litellm-http` is Rust-only and route-neutral: settings resolution, the pooled `reqwest` clients, TLS, proxies, the SSRF-safe media fetcher, request and header helpers, and transport errors. Python's `litellm/llms/custom_httpx/` is split by responsibility instead of mirrored: its transport half lives here, its OCR handler in `litellm-llms`
|
||||
- `litellm-llms` mirrors `litellm/llms/`: `base_llm/<api>/transformation.rs`, `<provider>/<api>/transformation.rs`, and `base_llm/ocr/handler.rs` (the OCR request handler)
|
||||
- `litellm-core` mirrors the route packages (`litellm/ocr/`, `litellm/messages/`, ...): entrypoints, route request types, provider dispatch, the route machine, and hooks
|
||||
- `litellm-inference` mirrors the route packages (`litellm/ocr/`, `litellm/messages/`, ...): entrypoints, route request types, provider dispatch, the route machine, and hooks
|
||||
|
||||
A route module owns the call entrypoint, route request types (`*Request<'a>`), credential fallback, provider dispatch, and the handler glue that runs a provider config. Provider code never imports from core; when it needs the caller's hooks mid-call it goes through `litellm_llms::base_llm::ocr::handler::CallHooks`, the provider-level hooks OCR implements over its host until it folds into `litellm_host::interceptors::Interceptors`. Import every item from its canonical path. Never re-export another crate's items or give an item a second public path; the only re-export allowed is a private submodule surfacing its item at its module root (`mod error; pub use error::Error;`). Handlers belong in core or llms, never in a host crate
|
||||
A route module owns the call entrypoint, route request types (`*Request<'a>`), credential fallback, provider dispatch, and the handler glue that runs a provider config. Provider code never imports from inference; when it needs the caller's hooks mid-call it goes through `litellm_llms::base_llm::ocr::handler::CallHooks`, the provider-level hooks OCR implements over its host until it folds into `litellm_host::interceptors::Interceptors`. Import every item from its canonical path. Never re-export another crate's items or give an item a second public path; the only re-export allowed is a private submodule surfacing its item at its module root (`mod error; pub use error::Error;`). Handlers belong in core or llms, never in a host crate
|
||||
|
||||
## Error placement
|
||||
|
||||
|
|
@ -1,10 +1,13 @@
|
|||
[package]
|
||||
name = "litellm-core"
|
||||
name = "litellm-inference"
|
||||
version = "0.1.0"
|
||||
edition.workspace = true
|
||||
license.workspace = true
|
||||
repository.workspace = true
|
||||
|
||||
[features]
|
||||
test-support = []
|
||||
|
||||
[dependencies]
|
||||
litellm-cache.workspace = true
|
||||
litellm-cache-response.workspace = true
|
||||
|
|
@ -40,6 +43,7 @@ url.workspace = true
|
|||
veil.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
litellm-inference = { workspace = true, features = ["test-support"] }
|
||||
litellm-cache-memory.workspace = true
|
||||
litellm-http = { workspace = true, features = ["test-support"] }
|
||||
litellm-auth-gcp.workspace = true
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
use litellm_http::request::string_headers;
|
||||
use litellm_http::request::with_default_headers;
|
||||
use litellm_llms::{
|
||||
|
|
@ -13,7 +14,7 @@ use super::Error;
|
|||
use crate::audio_transcription::types::{
|
||||
AudioTranscriptionRequest, ProviderAudioTranscriptionRequest,
|
||||
};
|
||||
use crate::provider::{LlmProviders, resolve_llm_provider};
|
||||
use crate::provider::resolve_llm_provider;
|
||||
|
||||
fn provider_config(provider: LlmProviders) -> Option<&'static dyn BaseAudioTranscriptionConfig> {
|
||||
match provider {
|
||||
|
|
@ -221,13 +221,13 @@ where
|
|||
Ok(cache.finish(output, &source).await)
|
||||
}
|
||||
|
||||
pub(crate) struct CallCache<P> {
|
||||
pub struct CallCache<P> {
|
||||
session: Option<CacheSession>,
|
||||
protocol: PhantomData<P>,
|
||||
}
|
||||
|
||||
impl<P: StreamCachable> CallCache<P> {
|
||||
pub(crate) fn from_wire(
|
||||
pub fn from_wire(
|
||||
cache: Option<&ScopedCache>,
|
||||
policy: CachePolicy,
|
||||
identity: &ProviderIdentity,
|
||||
|
|
@ -254,7 +254,7 @@ impl<P: StreamCachable> CallCache<P> {
|
|||
}
|
||||
}
|
||||
|
||||
pub(crate) async fn lookup(&self) -> Option<(OutputOf<P>, ResultSource)>
|
||||
pub async fn lookup(&self) -> Option<(OutputOf<P>, ResultSource)>
|
||||
where
|
||||
P::Response: DeserializeOwned,
|
||||
{
|
||||
|
|
@ -271,7 +271,7 @@ impl<P: StreamCachable> CallCache<P> {
|
|||
))
|
||||
}
|
||||
|
||||
pub(crate) async fn finish(self, output: OutputOf<P>, source: &ResultSource) -> OutputOf<P>
|
||||
pub async fn finish(self, output: OutputOf<P>, source: &ResultSource) -> OutputOf<P>
|
||||
where
|
||||
P::Response: Serialize,
|
||||
{
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
use litellm_http::request::string_headers as shared_string_headers;
|
||||
use litellm_llms::{
|
||||
anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG,
|
||||
|
|
@ -8,7 +9,6 @@ use litellm_llms::{
|
|||
use serde_json::{Map, Value};
|
||||
|
||||
use super::Error;
|
||||
use crate::provider::LlmProviders;
|
||||
|
||||
const HEADER_CONTEXT: &str = "chat completions";
|
||||
|
||||
|
|
@ -7,7 +7,7 @@ use litellm_host::{
|
|||
|
||||
use crate::{CallOptions, RouteError};
|
||||
|
||||
pub(crate) struct CallContext<'a, I> {
|
||||
pub struct CallContext<'a, I> {
|
||||
pub interceptors: &'a I,
|
||||
pub observers: Option<ObservationSender>,
|
||||
pub cache: CachePolicy,
|
||||
|
|
@ -41,17 +41,17 @@ impl Drop for Completion {
|
|||
}
|
||||
}
|
||||
|
||||
pub(crate) fn provider(model: &str, provider: &str) {
|
||||
pub fn provider(model: &str, provider: &str) {
|
||||
let span = Span::current();
|
||||
span.record("resolved_model", model);
|
||||
span.record("provider", provider);
|
||||
}
|
||||
|
||||
pub(crate) async fn unary<R, E>(execute: impl Future<Output = Result<R, E>>) -> Result<R, E> {
|
||||
pub async fn unary<R, E>(execute: impl Future<Output = Result<R, E>>) -> Result<R, E> {
|
||||
operation("litellm.route", execute).await
|
||||
}
|
||||
|
||||
pub(crate) async fn operation<R, E>(
|
||||
pub async fn operation<R, E>(
|
||||
name: &str,
|
||||
execute: impl Future<Output = Result<R, E>>,
|
||||
) -> Result<R, E> {
|
||||
|
|
@ -61,7 +61,7 @@ pub(crate) async fn operation<R, E>(
|
|||
result
|
||||
}
|
||||
|
||||
pub(crate) async fn call<R, H, C, E>(
|
||||
pub async fn call<R, H, C, E>(
|
||||
execute: impl Future<Output = Result<CallOutput<R, H, C, E>, E>>,
|
||||
) -> Result<CallOutput<R, H, C, E>, E>
|
||||
where
|
||||
|
|
@ -1,5 +1,5 @@
|
|||
mod context;
|
||||
mod diagnostic;
|
||||
pub mod context;
|
||||
pub mod diagnostic;
|
||||
|
||||
pub mod audio_transcription;
|
||||
pub mod caching;
|
||||
|
|
@ -8,10 +8,12 @@ pub mod constants;
|
|||
pub mod error;
|
||||
pub mod messages;
|
||||
pub mod ocr;
|
||||
mod outbound;
|
||||
mod provider;
|
||||
pub mod outbound;
|
||||
pub mod provider;
|
||||
pub mod resources;
|
||||
pub mod responses;
|
||||
#[cfg(feature = "test-support")]
|
||||
pub mod test_support;
|
||||
|
||||
pub use error::RouteError;
|
||||
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
use litellm_http::request::string_headers as shared_string_headers;
|
||||
pub(super) use litellm_http::request::truncate_error_body;
|
||||
use litellm_llms::{
|
||||
|
|
@ -9,7 +10,6 @@ use litellm_llms::{
|
|||
use serde_json::{Map, Value};
|
||||
|
||||
use super::Error;
|
||||
use crate::provider::LlmProviders;
|
||||
|
||||
const HEADER_CONTEXT: &str = "messages";
|
||||
|
||||
|
|
@ -68,7 +68,7 @@ mod tests {
|
|||
|
||||
use super::{MessagesProvider, messages_provider, string_headers, truncate_error_body};
|
||||
use crate::messages::Error;
|
||||
use crate::provider::LlmProviders;
|
||||
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
|
||||
#[rstest]
|
||||
#[case::anthropic("anthropic", MessagesProvider::Anthropic)]
|
||||
|
|
@ -1,4 +1,5 @@
|
|||
use litellm_auth::{InputSource, SecretValue, Sourced};
|
||||
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
use litellm_llms::base_llm::ocr::{
|
||||
handler::OcrClient,
|
||||
transformation::{OcrConnection, OcrCredentialInputs, PreparedOcrRequest},
|
||||
|
|
@ -6,7 +7,6 @@ use litellm_llms::base_llm::ocr::{
|
|||
use litellm_secrets::source::Secrets;
|
||||
|
||||
use crate::ocr::types::{LiteLLMOcrRequest, ResolvedOcrRequest};
|
||||
use crate::provider::LlmProviders;
|
||||
|
||||
pub(crate) fn prepare_request(
|
||||
request: ResolvedOcrRequest,
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
use crate::provider::LlmProviders;
|
||||
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
use litellm_core_utils::get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider};
|
||||
use litellm_llms::{
|
||||
aws_textract::ocr::{
|
||||
|
|
@ -10,7 +10,7 @@ use serde_json::Value;
|
|||
skip_all,
|
||||
fields(status)
|
||||
)]
|
||||
pub(crate) async fn send(
|
||||
pub async fn send(
|
||||
request: OutboundRequest,
|
||||
client: &litellm_http::Client,
|
||||
) -> Result<reqwest::Response, reqwest::Error> {
|
||||
|
|
@ -21,7 +21,7 @@ pub(crate) async fn send(
|
|||
|
||||
/// Header credentials are already in `headers`; SigV4 is applied here, over the
|
||||
/// bytes that are sent.
|
||||
pub(crate) fn outbound_request(
|
||||
pub fn outbound_request(
|
||||
authenticated: Authenticated,
|
||||
url: String,
|
||||
body: &Value,
|
||||
|
|
@ -1,15 +1,16 @@
|
|||
pub use litellm_core_utils::get_llm_provider_logic::LlmProviders;
|
||||
use litellm_core_utils::get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider};
|
||||
use litellm_core_utils::get_llm_provider_logic::{
|
||||
CustomLlmProvider, LlmProviders, get_custom_llm_provider,
|
||||
};
|
||||
|
||||
use crate::error::RouteError as Error;
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct ResolvedProvider<'a> {
|
||||
pub(crate) model: &'a str,
|
||||
pub(crate) provider: LlmProviders,
|
||||
pub struct ResolvedProvider<'a> {
|
||||
pub model: &'a str,
|
||||
pub provider: LlmProviders,
|
||||
}
|
||||
|
||||
pub(crate) fn resolve_llm_provider<'a>(
|
||||
pub fn resolve_llm_provider<'a>(
|
||||
model: &'a str,
|
||||
custom_llm_provider: Option<&'a str>,
|
||||
route: &'static str,
|
||||
87
litellm-rust/crates/inference/src/test_support.rs
Normal file
87
litellm-rust/crates/inference/src/test_support.rs
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use futures_util::future::BoxFuture;
|
||||
use litellm_http::{
|
||||
ClientVariant, HttpClientConfig, HttpClientPool, HttpSettings, Resolution,
|
||||
media::PublicDnsResolver,
|
||||
};
|
||||
use litellm_secrets::{SecretValue, source::SecretSource};
|
||||
|
||||
use crate::resources::CoreResources;
|
||||
|
||||
pub fn http_pool() -> HttpClientPool {
|
||||
HttpClientPool::new(Arc::new(PublicDnsResolver))
|
||||
}
|
||||
|
||||
pub fn resources() -> CoreResources {
|
||||
CoreResources::new(Arc::new(http_pool()))
|
||||
}
|
||||
|
||||
pub fn no_secrets() -> Arc<dyn SecretSource> {
|
||||
Arc::new(RecordingSecrets::empty())
|
||||
}
|
||||
|
||||
pub fn provider_http(resources: &CoreResources, config: &HttpClientConfig) -> litellm_http::Client {
|
||||
resources
|
||||
.pool
|
||||
.client(config, ClientVariant::Provider)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
pub fn http_config() -> HttpClientConfig {
|
||||
Resolution::from(&HttpSettings::default()).config
|
||||
}
|
||||
|
||||
/// A secret source that answers from a fixed table and records every name it was asked for.
|
||||
pub struct RecordingSecrets {
|
||||
values: Vec<(String, String)>,
|
||||
fails: bool,
|
||||
requested: Mutex<Vec<String>>,
|
||||
}
|
||||
|
||||
impl RecordingSecrets {
|
||||
pub fn new<'a>(values: impl IntoIterator<Item = (&'a str, &'a str)>) -> Self {
|
||||
Self {
|
||||
values: values
|
||||
.into_iter()
|
||||
.map(|(name, value)| (name.to_string(), value.to_string()))
|
||||
.collect(),
|
||||
fails: false,
|
||||
requested: Mutex::new(Vec::new()),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn empty() -> Self {
|
||||
Self::new([])
|
||||
}
|
||||
|
||||
pub fn failing() -> Self {
|
||||
Self {
|
||||
fails: true,
|
||||
..Self::empty()
|
||||
}
|
||||
}
|
||||
|
||||
pub fn requested(&self) -> Vec<String> {
|
||||
self.requested.lock().unwrap().clone()
|
||||
}
|
||||
}
|
||||
|
||||
impl SecretSource for RecordingSecrets {
|
||||
fn get_secret_str<'a>(
|
||||
&'a self,
|
||||
name: &'a str,
|
||||
) -> BoxFuture<'a, Result<Option<SecretValue>, litellm_secrets::Error>> {
|
||||
Box::pin(async move {
|
||||
self.requested.lock().unwrap().push(name.to_string());
|
||||
if self.fails {
|
||||
return Err(litellm_secrets::Error::ManagedSecretMissing);
|
||||
}
|
||||
Ok(self
|
||||
.values
|
||||
.iter()
|
||||
.find(|(key, _)| key == name)
|
||||
.map(|(_, value)| SecretValue::new(value.clone())))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
use litellm_core::audio_transcription::{Error, types::AudioTranscriptionRequest};
|
||||
use litellm_inference::audio_transcription::{Error, types::AudioTranscriptionRequest};
|
||||
use rstest::{fixture, rstest};
|
||||
use serde_json::{Map, Value, json};
|
||||
use wiremock::ResponseTemplate;
|
||||
|
|
@ -15,10 +15,6 @@ use litellm_cache_response::{
|
|||
CacheOptions, CachePolicy, CacheScope, ResponseCache, ResponseCacheConfig,
|
||||
ResponseCacheService, ResponseEnvelope,
|
||||
};
|
||||
use litellm_core::{
|
||||
RouteError,
|
||||
caching::{Cachable, CacheRequest, StreamCachable, execute_streaming, execute_unary},
|
||||
};
|
||||
use litellm_host::{
|
||||
call::{CallOutput, OutputOf},
|
||||
interceptors::{
|
||||
|
|
@ -29,6 +25,10 @@ use litellm_host::{
|
|||
observation::observation_channel,
|
||||
protocol::Protocol,
|
||||
};
|
||||
use litellm_inference::{
|
||||
RouteError,
|
||||
caching::{Cachable, CacheRequest, StreamCachable, execute_streaming, execute_unary},
|
||||
};
|
||||
use rstest::{fixture, rstest};
|
||||
use serde_json::{Value, json};
|
||||
|
||||
|
|
@ -410,7 +410,7 @@ async fn an_invalid_cached_envelope_is_replaced_by_a_provider_result(#[case] poi
|
|||
async fn responses_refetches_instead_of_deserializing_another_api_response(
|
||||
#[case] poisoned: Value,
|
||||
) {
|
||||
use litellm_core::responses::route::Responses;
|
||||
use litellm_inference::responses::route::Responses;
|
||||
use litellm_llms_types::formats::responses::ResponsesApiResponse;
|
||||
|
||||
let cache: Arc<dyn ResponseCacheService> = Arc::new(InvalidEntryCache(
|
||||
|
|
@ -460,7 +460,7 @@ async fn messages_cache_identity_includes_provider_native_parameters(
|
|||
#[case] original: Value,
|
||||
#[case] changed: Value,
|
||||
) {
|
||||
use litellm_core::messages::route::Messages;
|
||||
use litellm_inference::messages::route::Messages;
|
||||
use litellm_llms_types::formats::messages::MessagesResponse;
|
||||
|
||||
let calls = AtomicUsize::new(0);
|
||||
|
|
@ -842,7 +842,7 @@ async fn responses_cache_only_reuses_completed_responses(
|
|||
#[case] status: &str,
|
||||
#[case] expected_calls: usize,
|
||||
) {
|
||||
use litellm_core::responses::route::Responses;
|
||||
use litellm_inference::responses::route::Responses;
|
||||
use litellm_llms_types::formats::responses::ResponsesApiResponse;
|
||||
|
||||
let calls = AtomicUsize::new(0);
|
||||
|
|
@ -883,7 +883,7 @@ async fn the_same_route_entrypoint_reports_facts_with_or_without_caching(
|
|||
traces: support::TraceCapture,
|
||||
) {
|
||||
use litellm_cache_response::ScopedCache;
|
||||
use litellm_core::chat_completions::types::ChatCompletionsRequest;
|
||||
use litellm_inference::chat_completions::types::ChatCompletionsRequest;
|
||||
use wiremock::{Mock, MockServer, ResponseTemplate, matchers::method};
|
||||
|
||||
let upstream = MockServer::start().await;
|
||||
|
|
@ -1038,7 +1038,7 @@ async fn cache_identity_follows_resolved_configuration_and_request_callbacks(
|
|||
#[case] change: &str,
|
||||
) {
|
||||
use litellm_cache_response::ScopedCache;
|
||||
use litellm_core::{
|
||||
use litellm_inference::{
|
||||
chat_completions::{ChatCompletionsRoute, types::ChatCompletionsRequest},
|
||||
messages::MessagesCall,
|
||||
responses::types::ResponsesCall,
|
||||
|
|
@ -1178,7 +1178,7 @@ async fn cache_identity_follows_resolved_configuration_and_request_callbacks(
|
|||
#[tokio::test]
|
||||
async fn signed_requests_bypass_response_caching(cache: Arc<dyn ResponseCacheService>) {
|
||||
use litellm_cache_response::ScopedCache;
|
||||
use litellm_core::chat_completions::types::ChatCompletionsRequest;
|
||||
use litellm_inference::chat_completions::types::ChatCompletionsRequest;
|
||||
use wiremock::{Mock, MockServer, ResponseTemplate, matchers::method};
|
||||
|
||||
let upstream = MockServer::start().await;
|
||||
|
|
@ -5,8 +5,8 @@ use litellm_host::{
|
|||
};
|
||||
use std::time::Duration;
|
||||
|
||||
use litellm_core::chat_completions::{Error, types::ChatCompletionsRequest};
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_inference::chat_completions::{Error, types::ChatCompletionsRequest};
|
||||
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
|
||||
use rstest::{fixture, rstest};
|
||||
use serde_json::{Map, Value, json};
|
||||
|
|
@ -257,8 +257,8 @@ async fn direct_and_hosted_calls_share_hooks_and_lifecycle(
|
|||
request: ChatCompletionsRequest<'static>,
|
||||
#[case] hosted: bool,
|
||||
) {
|
||||
use litellm_core::chat_completions::route::ChatCompletions;
|
||||
use litellm_host::{call::HostedCompletion, lifecycle::CallEvent};
|
||||
use litellm_inference::chat_completions::route::ChatCompletions;
|
||||
|
||||
let upstream = upstream([anthropic_response(ANTHROPIC_MESSAGE)]).await;
|
||||
let base = upstream.uri();
|
||||
|
|
@ -1,11 +1,12 @@
|
|||
use litellm_host::lifecycle::ExecutionEvent;
|
||||
use std::sync::Mutex;
|
||||
|
||||
use litellm_core::messages::{MessagesCallResponse, route::Messages};
|
||||
use litellm_host::{
|
||||
interceptors::{ExecutionFacts, RequestContext, ResultSource, WireRequest},
|
||||
lifecycle::CallEvent,
|
||||
};
|
||||
use litellm_inference::messages::{MessagesCallResponse, route::Messages};
|
||||
use litellm_inference::test_support::{RecordingSecrets, no_secrets};
|
||||
use litellm_llms::base_llm::messages::context::MessagesModelCapabilities as AnthropicModelCapabilities;
|
||||
use rstest::rstest;
|
||||
|
||||
|
|
@ -3,11 +3,12 @@ use std::{
|
|||
time::Duration,
|
||||
};
|
||||
|
||||
use litellm_core::messages::{
|
||||
use litellm_http::{HttpSettings, Resolution};
|
||||
use litellm_inference::messages::{
|
||||
Error, MessagesCall, MessagesShaping,
|
||||
route::{Messages, MessagesMachine, MessagesOutput},
|
||||
};
|
||||
use litellm_http::{HttpSettings, Resolution};
|
||||
use litellm_inference::test_support::RecordingSecrets;
|
||||
use litellm_llms_types::formats::messages::{MessagesRequest, MessagesResponse};
|
||||
use litellm_secrets::source::SecretSource;
|
||||
use rstest::fixture;
|
||||
|
|
@ -1,9 +1,12 @@
|
|||
use litellm_core::messages::{MessagesCallResponse, messages_body};
|
||||
use litellm_host::{
|
||||
interceptors::{ExecutionFacts, ResultSource},
|
||||
lifecycle::ExecutionEvent,
|
||||
};
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_inference::messages::{MessagesCallResponse, messages_body};
|
||||
use litellm_inference::test_support::{
|
||||
RecordingSecrets, http_config, no_secrets, provider_http, resources,
|
||||
};
|
||||
use rstest::rstest;
|
||||
|
||||
use super::*;
|
||||
|
|
@ -268,8 +271,8 @@ async fn the_facade_sends_through_the_injected_http_pool_configuration(call: Mes
|
|||
..HttpSettings::default()
|
||||
};
|
||||
|
||||
let resources = support::resources();
|
||||
let response = litellm_core::messages::MessagesRoute::new(
|
||||
let resources = resources();
|
||||
let response = litellm_inference::messages::MessagesRoute::new(
|
||||
provider_http(&resources, &Resolution::from(&settings).config),
|
||||
resources.auth,
|
||||
no_secrets(),
|
||||
|
|
@ -348,7 +351,7 @@ async fn route_uses_injected_dependencies_and_optional_cache(
|
|||
) {
|
||||
use litellm_cache_memory::InMemoryCache;
|
||||
use litellm_cache_response::{CacheScope, ResponseCache, ScopedCache};
|
||||
use litellm_core::messages::MessagesRoute;
|
||||
use litellm_inference::messages::MessagesRoute;
|
||||
|
||||
let upstream = upstream([message_response(), message_response()]).await;
|
||||
let resources = resources();
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
use litellm_inference::test_support::RecordingSecrets;
|
||||
use rstest::rstest;
|
||||
|
||||
use super::*;
|
||||
|
|
@ -5,10 +5,11 @@ use std::{
|
|||
|
||||
use bytes::Bytes;
|
||||
use futures_util::{StreamExt, TryStreamExt};
|
||||
use litellm_core::messages::{
|
||||
use litellm_inference::messages::{
|
||||
MessagesCallResponse,
|
||||
route::{Messages, MessagesStreamHead},
|
||||
};
|
||||
use litellm_inference::test_support::{RecordingSecrets, no_secrets};
|
||||
use litellm_tracing::{Logger, Metadata, Record, Sink};
|
||||
use rstest::rstest;
|
||||
use tokio::{
|
||||
|
|
@ -35,7 +36,7 @@ struct TraceSink(mpsc::Sender<(String, Value)>);
|
|||
|
||||
impl Sink for TraceSink {
|
||||
fn enabled(&self, metadata: &Metadata<'_>) -> bool {
|
||||
metadata.target().starts_with("litellm_core::messages")
|
||||
metadata.target().starts_with("litellm_inference::messages")
|
||||
}
|
||||
|
||||
fn emit(&self, record: &Record) {
|
||||
|
|
@ -1,6 +1,7 @@
|
|||
use base64::Engine;
|
||||
use litellm_core::ocr::types::OcrDocumentInput;
|
||||
use litellm_host::interceptors::WireRequest;
|
||||
use litellm_inference::ocr::types::OcrDocumentInput;
|
||||
use litellm_inference::test_support::{http_config, no_secrets, resources};
|
||||
use rstest::rstest;
|
||||
use wiremock::{Mock, matchers::any};
|
||||
|
||||
|
|
@ -5,14 +5,14 @@ use std::sync::{
|
|||
atomic::{AtomicUsize, Ordering},
|
||||
};
|
||||
|
||||
use litellm_core::ocr::{
|
||||
route::{Ocr, OcrCall, OcrOp},
|
||||
types::OcrDocumentInput,
|
||||
};
|
||||
use litellm_host::{
|
||||
interceptors::{RequestContext, WireRequest},
|
||||
lifecycle::CallEvent,
|
||||
};
|
||||
use litellm_inference::ocr::{
|
||||
route::{Ocr, OcrCall, OcrOp},
|
||||
types::OcrDocumentInput,
|
||||
};
|
||||
use rstest::rstest;
|
||||
|
||||
use super::*;
|
||||
|
|
@ -6,16 +6,16 @@ use std::{
|
|||
time::Duration,
|
||||
};
|
||||
|
||||
use litellm_core::ocr::{
|
||||
route::{OcrCall, OcrMachine, OcrOp},
|
||||
types::OcrDocumentInput,
|
||||
};
|
||||
use litellm_host::{
|
||||
interceptors::{Interceptors, WireRequest},
|
||||
machine::{HostFailure, Machine, MachineStep},
|
||||
protocol::{HostRequest, InterceptRequest},
|
||||
};
|
||||
use litellm_host_native::services::HostCallHandler;
|
||||
use litellm_inference::ocr::{
|
||||
route::{OcrCall, OcrMachine, OcrOp},
|
||||
types::OcrDocumentInput,
|
||||
};
|
||||
use litellm_llms::base_llm::ocr::transformation::OcrTransportConfig;
|
||||
use rstest::rstest;
|
||||
use tokio::{io::AsyncReadExt, net::TcpListener, sync::Notify};
|
||||
|
|
@ -1,14 +1,15 @@
|
|||
use litellm_core::ocr::{
|
||||
use litellm_host::{
|
||||
interceptors::{RequestContext, WireRequest},
|
||||
lifecycle::CallEvent,
|
||||
};
|
||||
use litellm_inference::ocr::{
|
||||
OcrRoute,
|
||||
document::prepare_document,
|
||||
route::{Ocr, OcrCall, OcrOp},
|
||||
types::{LiteLLMOcrRequest, OcrDocumentInput},
|
||||
wire::{OcrWireRequest, decode_request},
|
||||
};
|
||||
use litellm_host::{
|
||||
interceptors::{RequestContext, WireRequest},
|
||||
lifecycle::CallEvent,
|
||||
};
|
||||
use litellm_inference::test_support::{http_config, no_secrets, resources};
|
||||
use litellm_llms::base_llm::ocr::{error::Error, settings::OcrSettings};
|
||||
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument};
|
||||
use serde_json::{Map, Value, json};
|
||||
|
|
@ -1,6 +1,7 @@
|
|||
use std::sync::Arc;
|
||||
|
||||
use litellm_http::{HttpSettings, Resolution};
|
||||
use litellm_inference::test_support::{RecordingSecrets, http_config, no_secrets, resources};
|
||||
use litellm_llms::{
|
||||
base_llm::ocr::transformation::{BaseOcrConfig, OCR_RESPONSE_MAX_BYTES},
|
||||
mistral::ocr::transformation::MistralOcrConfig,
|
||||
|
|
@ -1,5 +1,5 @@
|
|||
use litellm_auth::{InputSource, Sourced};
|
||||
use litellm_core::ocr::arguments::is_supported_request;
|
||||
use litellm_inference::ocr::arguments::is_supported_request;
|
||||
use litellm_llms::base_llm::ocr::settings::OcrSettings;
|
||||
use rstest::rstest;
|
||||
|
||||
|
|
@ -9,15 +9,16 @@ use litellm_auth::AuthServices;
|
|||
use litellm_auth_gcp::{
|
||||
CredentialSource, VertexAuth, VertexAuthFuture, VertexProviderLoader, VertexTokenSource,
|
||||
};
|
||||
use litellm_core::{
|
||||
use litellm_http::{HttpSettings, Resolution};
|
||||
use litellm_inference::test_support::{RecordingSecrets, http_pool};
|
||||
use litellm_inference::{
|
||||
ocr::wire::{OcrWireRequest, decode_request},
|
||||
resources::CoreResources,
|
||||
};
|
||||
use litellm_http::{HttpSettings, Resolution};
|
||||
use litellm_llms::base_llm::ocr::settings::OcrSettings;
|
||||
use rstest::{fixture, rstest};
|
||||
use serde_json::json;
|
||||
use support::{ReceivedRequest, RecordingSecrets, http_pool, json_response, upstream};
|
||||
use support::{ReceivedRequest, json_response, upstream};
|
||||
|
||||
struct TokenSource(String);
|
||||
|
||||
|
|
@ -5,11 +5,12 @@ use litellm_host::{
|
|||
use std::sync::Arc;
|
||||
|
||||
use futures_util::TryStreamExt;
|
||||
use litellm_core::responses::{
|
||||
use litellm_host::{call::HostedCompletion, lifecycle::CallEvent};
|
||||
use litellm_inference::responses::{
|
||||
route::Responses,
|
||||
types::{ResponsesCall, ResponsesOutput},
|
||||
};
|
||||
use litellm_host::{call::HostedCompletion, lifecycle::CallEvent};
|
||||
use litellm_inference::test_support::{RecordingSecrets, no_secrets};
|
||||
use rstest::{fixture, rstest};
|
||||
use serde_json::json;
|
||||
use wiremock::ResponseTemplate;
|
||||
|
|
@ -417,7 +418,7 @@ async fn websocket_operations_trace_outcomes_without_capturing_frames_or_credent
|
|||
traces: TraceCapture,
|
||||
) {
|
||||
use futures_util::{SinkExt, StreamExt};
|
||||
use litellm_core::responses::websocket::ResponsesWebSocketConnection;
|
||||
use litellm_inference::responses::websocket::ResponsesWebSocketConnection;
|
||||
|
||||
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
|
||||
let address = listener.local_addr().unwrap();
|
||||
|
|
@ -8,70 +8,50 @@ use std::{
|
|||
sync::{Arc, Mutex},
|
||||
};
|
||||
|
||||
use futures_util::future::BoxFuture;
|
||||
use litellm_http::{
|
||||
ClientVariant, HttpClientConfig, HttpClientPool, HttpSettings, Resolution,
|
||||
media::PublicDnsResolver,
|
||||
};
|
||||
use litellm_secrets::{SecretValue, source::SecretSource};
|
||||
use litellm_http::HttpClientConfig;
|
||||
use litellm_inference::test_support::{http_config, no_secrets, provider_http, resources};
|
||||
use litellm_secrets::source::SecretSource;
|
||||
use serde_json::Value;
|
||||
use wiremock::{Mock, MockServer, Request, ResponseTemplate, matchers::any};
|
||||
|
||||
/// A port nothing listens on, for calls that must fail before any request is sent.
|
||||
pub const UNREACHABLE_BASE: &str = "http://127.0.0.1:1";
|
||||
|
||||
pub fn http_pool() -> HttpClientPool {
|
||||
HttpClientPool::new(Arc::new(PublicDnsResolver))
|
||||
}
|
||||
|
||||
pub fn resources() -> litellm_core::resources::CoreResources {
|
||||
litellm_core::resources::CoreResources::new(Arc::new(http_pool()))
|
||||
}
|
||||
|
||||
pub fn no_secrets() -> Arc<dyn SecretSource> {
|
||||
Arc::new(RecordingSecrets::empty())
|
||||
}
|
||||
|
||||
pub fn provider_http(
|
||||
resources: &litellm_core::resources::CoreResources,
|
||||
config: &HttpClientConfig,
|
||||
) -> litellm_http::Client {
|
||||
resources
|
||||
.pool
|
||||
.client(config, ClientVariant::Provider)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
pub fn messages_route(secrets: Arc<dyn SecretSource>) -> litellm_core::messages::MessagesRoute {
|
||||
pub fn messages_route(
|
||||
secrets: Arc<dyn SecretSource>,
|
||||
) -> litellm_inference::messages::MessagesRoute {
|
||||
let resources = resources();
|
||||
litellm_core::messages::MessagesRoute::new(
|
||||
litellm_inference::messages::MessagesRoute::new(
|
||||
provider_http(&resources, &http_config()),
|
||||
resources.auth,
|
||||
secrets,
|
||||
)
|
||||
}
|
||||
|
||||
pub fn chat_completions_route() -> litellm_core::chat_completions::ChatCompletionsRoute {
|
||||
pub fn chat_completions_route() -> litellm_inference::chat_completions::ChatCompletionsRoute {
|
||||
let resources = resources();
|
||||
litellm_core::chat_completions::ChatCompletionsRoute::new(
|
||||
litellm_inference::chat_completions::ChatCompletionsRoute::new(
|
||||
provider_http(&resources, &http_config()),
|
||||
resources.auth,
|
||||
no_secrets(),
|
||||
)
|
||||
}
|
||||
|
||||
pub fn responses_route(secrets: Arc<dyn SecretSource>) -> litellm_core::responses::ResponsesRoute {
|
||||
pub fn responses_route(
|
||||
secrets: Arc<dyn SecretSource>,
|
||||
) -> litellm_inference::responses::ResponsesRoute {
|
||||
let resources = resources();
|
||||
litellm_core::responses::ResponsesRoute::new(
|
||||
litellm_inference::responses::ResponsesRoute::new(
|
||||
provider_http(&resources, &http_config()),
|
||||
resources.auth,
|
||||
secrets,
|
||||
)
|
||||
}
|
||||
|
||||
pub fn audio_transcription_route() -> litellm_core::audio_transcription::AudioTranscriptionRoute {
|
||||
pub fn audio_transcription_route() -> litellm_inference::audio_transcription::AudioTranscriptionRoute
|
||||
{
|
||||
let resources = resources();
|
||||
litellm_core::audio_transcription::AudioTranscriptionRoute::new(
|
||||
litellm_inference::audio_transcription::AudioTranscriptionRoute::new(
|
||||
provider_http(&resources, &http_config()),
|
||||
resources.auth,
|
||||
no_secrets(),
|
||||
|
|
@ -79,13 +59,13 @@ pub fn audio_transcription_route() -> litellm_core::audio_transcription::AudioTr
|
|||
}
|
||||
|
||||
pub fn build_ocr_route(
|
||||
resources: &litellm_core::resources::CoreResources,
|
||||
resources: &litellm_inference::resources::CoreResources,
|
||||
config: &HttpClientConfig,
|
||||
url_policy: litellm_http::media::UrlPolicy,
|
||||
settings: litellm_llms::base_llm::ocr::settings::OcrSettings,
|
||||
secrets: Arc<dyn SecretSource>,
|
||||
) -> litellm_core::ocr::OcrRoute {
|
||||
litellm_core::ocr::OcrRoute::new(
|
||||
) -> litellm_inference::ocr::OcrRoute {
|
||||
litellm_inference::ocr::OcrRoute::new(
|
||||
litellm_llms::base_llm::ocr::handler::OcrClient::new(
|
||||
&resources.pool,
|
||||
config,
|
||||
|
|
@ -98,10 +78,6 @@ pub fn build_ocr_route(
|
|||
)
|
||||
}
|
||||
|
||||
pub fn http_config() -> HttpClientConfig {
|
||||
Resolution::from(&HttpSettings::default()).config
|
||||
}
|
||||
|
||||
/// Starts an upstream that answers its n-th request with the n-th response and 404s after.
|
||||
pub async fn upstream(responses: impl IntoIterator<Item = ResponseTemplate>) -> MockServer {
|
||||
let server = MockServer::start().await;
|
||||
|
|
@ -189,60 +165,6 @@ impl ReceivedRequest for Request {
|
|||
}
|
||||
}
|
||||
|
||||
/// A secret source that answers from a fixed table and records every name it was asked for.
|
||||
pub struct RecordingSecrets {
|
||||
values: Vec<(String, String)>,
|
||||
fails: bool,
|
||||
requested: Mutex<Vec<String>>,
|
||||
}
|
||||
|
||||
impl RecordingSecrets {
|
||||
pub fn new<'a>(values: impl IntoIterator<Item = (&'a str, &'a str)>) -> Self {
|
||||
Self {
|
||||
values: values
|
||||
.into_iter()
|
||||
.map(|(name, value)| (name.to_string(), value.to_string()))
|
||||
.collect(),
|
||||
fails: false,
|
||||
requested: Mutex::new(Vec::new()),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn empty() -> Self {
|
||||
Self::new([])
|
||||
}
|
||||
|
||||
pub fn failing() -> Self {
|
||||
Self {
|
||||
fails: true,
|
||||
..Self::empty()
|
||||
}
|
||||
}
|
||||
|
||||
pub fn requested(&self) -> Vec<String> {
|
||||
self.requested.lock().unwrap().clone()
|
||||
}
|
||||
}
|
||||
|
||||
impl SecretSource for RecordingSecrets {
|
||||
fn get_secret_str<'a>(
|
||||
&'a self,
|
||||
name: &'a str,
|
||||
) -> BoxFuture<'a, Result<Option<SecretValue>, litellm_secrets::Error>> {
|
||||
Box::pin(async move {
|
||||
self.requested.lock().unwrap().push(name.to_string());
|
||||
if self.fails {
|
||||
return Err(litellm_secrets::Error::ManagedSecretMissing);
|
||||
}
|
||||
Ok(self
|
||||
.values
|
||||
.iter()
|
||||
.find(|(key, _)| key == name)
|
||||
.map(|(_, value)| SecretValue::new(value.clone())))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
pub struct RecordingCall<P: litellm_host::protocol::Protocol> {
|
||||
pub request: Mutex<Option<P::Request>>,
|
||||
pub events: Arc<CallEvents>,
|
||||
|
|
@ -396,7 +318,7 @@ impl TraceCapture {
|
|||
|
||||
impl litellm_tracing::Sink for TraceCapture {
|
||||
fn enabled(&self, metadata: &litellm_tracing::Metadata<'_>) -> bool {
|
||||
metadata.target().starts_with("litellm_core")
|
||||
metadata.target().starts_with("litellm_inference")
|
||||
}
|
||||
|
||||
fn emit(&self, record: &litellm_tracing::Record) {
|
||||
|
|
@ -16,7 +16,7 @@ Shared OCR document and response contracts live in `litellm-llms-types::formats:
|
|||
|
||||
For Mistral, `async_transform_ocr_request` uses the base default in both languages. `resolve_headers` and `build_ocr_url` implement the respective environment and URL operations, and `normalize_response` implements the typed part of response transformation. Existing auth key/header handling and top-level response-extra preservation differ between languages; layout refactors must preserve those behaviors and verify them with the existing tests
|
||||
|
||||
For non-OCR pairs, order corresponding methods as parameter support/mapping, environment validation, URL construction, request transformation, and response transformation, followed by Rust-only runtime hooks. Auth resolution remains split between configs and route preparation in litellm-core. Chat `supported_openai_param_mappings` describes accepted OpenAI/provider name pairs, unlike Python's `get_supported_openai_params` name list. Audio `map_transcription_params` remains a Rust filtering helper
|
||||
For non-OCR pairs, order corresponding methods as parameter support/mapping, environment validation, URL construction, request transformation, and response transformation, followed by Rust-only runtime hooks. Auth resolution remains split between configs and route preparation in litellm-inference. Chat `supported_openai_param_mappings` describes accepted OpenAI/provider name pairs, unlike Python's `get_supported_openai_params` name list. Audio `map_transcription_params` remains a Rust filtering helper
|
||||
|
||||
Azure Messages maps to `llms/azure_ai/anthropic/messages_transformation.py`; Bedrock Converse maps to `llms/bedrock/chat/converse_transformation.py`. `AnthropicConfig`, `AmazonConverseConfig`, and the non-OCR base traits are partial ports. `OpenAiResponsesApiConfig` implements WebSocket transformations and a direct HTTP Responses path. Its HTTP path does not implement Python model-specific parameter rewriting or Responses-to-Chat emulation. Preserve their acceptance gates, passthrough behavior, and host fallback contracts when aligning layout
|
||||
|
||||
|
|
|
|||
|
|
@ -82,11 +82,11 @@ GIL handling to `litellm-host-python`.
|
|||
## Bridge Shape
|
||||
|
||||
- Prefer one stable method per top-level LiteLLM route, for example
|
||||
`messages(...)`, calling the matching `litellm-core` entrypoint.
|
||||
`messages(...)`, calling the matching `litellm-inference` entrypoint.
|
||||
- Do not add one exported PyO3 function per provider helper unless there is a
|
||||
measured reason.
|
||||
- Provider dispatch belongs in the `litellm-core` route module (e.g.
|
||||
`litellm_core::messages`), not in this PyO3 crate.
|
||||
- Provider dispatch belongs in the `litellm-inference` route module (e.g.
|
||||
`litellm_inference::messages`), not in this PyO3 crate.
|
||||
- Python owns rollout state and fallback. Rust should return errors; Python
|
||||
decides whether to raise or fall back. For a rust-only provider/route (no
|
||||
Python reference), the Python side is a thin dispatch that calls Rust and
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ serde.workspace = true
|
|||
litellm-auth.workspace = true
|
||||
litellm-auth-aws.workspace = true
|
||||
litellm-callbacks-legacy-python.workspace = true
|
||||
litellm-core.workspace = true
|
||||
litellm-inference.workspace = true
|
||||
litellm-core-utils.workspace = true
|
||||
litellm-http.workspace = true
|
||||
litellm-llms.workspace = true
|
||||
|
|
|
|||
|
|
@ -4,11 +4,11 @@ use litellm_cache::Error;
|
|||
use litellm_cache_response::{
|
||||
ResponseCacheConfig, ResponseCacheRequest, ResponseCacheService, ResponseEnvelope,
|
||||
};
|
||||
use litellm_core::caching::CachedOutput;
|
||||
use litellm_host::{
|
||||
machine::{HostServices, MachineFault},
|
||||
protocol::{Protocol, Reply},
|
||||
};
|
||||
use litellm_inference::caching::CachedOutput;
|
||||
use serde_json::Value;
|
||||
|
||||
const STREAM_EVENTS_KEY: &str = "litellm_cached_anthropic_sse_events";
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
use litellm_core::RouteError;
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_inference::RouteError;
|
||||
use pyo3::{
|
||||
exceptions::{PyRuntimeError, PyValueError},
|
||||
prelude::*,
|
||||
|
|
|
|||
|
|
@ -80,13 +80,13 @@ fn decode_ssl_verify(field: &Field<'_>) -> Result<Option<SslVerify>, ProjectionE
|
|||
Err(field.invalid("a Boolean, Boolean string, CA path, or None"))
|
||||
}
|
||||
|
||||
static RESOURCES: LazyLock<litellm_core::resources::CoreResources> = LazyLock::new(|| {
|
||||
litellm_core::resources::CoreResources::new(Arc::new(HttpClientPool::new(Arc::new(
|
||||
static RESOURCES: LazyLock<litellm_inference::resources::CoreResources> = LazyLock::new(|| {
|
||||
litellm_inference::resources::CoreResources::new(Arc::new(HttpClientPool::new(Arc::new(
|
||||
PublicDnsResolver,
|
||||
))))
|
||||
});
|
||||
|
||||
pub(crate) fn resources() -> &'static litellm_core::resources::CoreResources {
|
||||
pub(crate) fn resources() -> &'static litellm_inference::resources::CoreResources {
|
||||
&RESOURCES
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1,8 +1,8 @@
|
|||
use crate::execution::{run_async, run_sync};
|
||||
use litellm_core::audio_transcription::{
|
||||
use litellm_host_python::from_py_argument;
|
||||
use litellm_inference::audio_transcription::{
|
||||
AudioTranscriptionRoute, Error, types::AudioTranscriptionRequest,
|
||||
};
|
||||
use litellm_host_python::from_py_argument;
|
||||
use pyo3::{prelude::*, types::PyDict};
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,9 @@ mod host;
|
|||
use pyo3::types::{PyDict, PyTuple};
|
||||
|
||||
use crate::execution::{run_async, run_sync};
|
||||
use litellm_core::chat_completions::{ChatCompletionsRoute, Error, types::ChatCompletionsRequest};
|
||||
use litellm_inference::chat_completions::{
|
||||
ChatCompletionsRoute, Error, types::ChatCompletionsRequest,
|
||||
};
|
||||
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
|
||||
use pyo3::prelude::*;
|
||||
use serde_json::{Map, Value};
|
||||
|
|
|
|||
|
|
@ -1,8 +1,10 @@
|
|||
use std::convert::Infallible;
|
||||
|
||||
use super::super::inference::InferenceHost;
|
||||
use litellm_core::chat_completions::{Error, route::ChatCompletions, types::ChatCompletionsCall};
|
||||
use litellm_host_python::{InvokeError, PythonBinding, PythonHostCalls, PythonOwned};
|
||||
use litellm_inference::chat_completions::{
|
||||
Error, route::ChatCompletions, types::ChatCompletionsCall,
|
||||
};
|
||||
use pyo3::{
|
||||
gc::{PyTraverseError, PyVisit},
|
||||
prelude::*,
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
use litellm_core::RouteError;
|
||||
use litellm_core_utils::get_llm_provider_logic::get_custom_llm_provider;
|
||||
use litellm_host_python::{from_py, lookup, to_py};
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_inference::RouteError;
|
||||
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
|
||||
use serde::Serialize;
|
||||
use serde_json::{Map, Value};
|
||||
|
|
|
|||
|
|
@ -2,12 +2,12 @@ use crate::cache::{CacheCall, Cached, PythonCache, Selection};
|
|||
use litellm_host_python::{PythonHostCalls, PythonOwned};
|
||||
|
||||
use bytes::Bytes;
|
||||
use litellm_core::messages::{
|
||||
use litellm_host_python::{InvokeError, PythonBinding, from_py, lookup, to_py};
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_inference::messages::{
|
||||
Error, MessagesCall, MessagesShaping, messages_body,
|
||||
route::{Messages, MessagesStreamHead},
|
||||
};
|
||||
use litellm_host_python::{InvokeError, PythonBinding, from_py, lookup, to_py};
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
use litellm_llms_types::headers::ProviderSpecificHeaders;
|
||||
use pyo3::{
|
||||
exceptions::{PyException, PyValueError},
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ fn run_messages(
|
|||
py,
|
||||
arguments,
|
||||
move |py, arguments, request| {
|
||||
let route = litellm_core::messages::MessagesRoute::new(
|
||||
let route = litellm_inference::messages::MessagesRoute::new(
|
||||
crate::http::provider_client(py, arguments, asynchronous)?
|
||||
.map_err(crate::http::client_error)?,
|
||||
crate::http::resources().auth.clone(),
|
||||
|
|
@ -48,7 +48,7 @@ fn run_messages(
|
|||
.execute(
|
||||
call,
|
||||
&interceptors,
|
||||
litellm_core::CallOptions {
|
||||
litellm_inference::CallOptions {
|
||||
cache: Some(options.policy),
|
||||
observers,
|
||||
},
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
use std::path::PathBuf;
|
||||
|
||||
use litellm_core::ocr::types::OcrDocumentInput;
|
||||
use litellm_host_python::{PythonFileReader, py_bytes};
|
||||
use litellm_inference::ocr::types::OcrDocumentInput;
|
||||
use pyo3::{
|
||||
exceptions::PyValueError,
|
||||
prelude::*,
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
use litellm_auth::ResolvedCredential;
|
||||
use litellm_core::ocr::route::{Ocr, OcrCall, OcrOp};
|
||||
use litellm_host_python::{InvokeError, PythonBinding, missing_state, to_py};
|
||||
use litellm_host_python::{PythonHostCalls, PythonOwned};
|
||||
use litellm_inference::ocr::route::{Ocr, OcrCall, OcrOp};
|
||||
use litellm_llms::base_llm::ocr::error::Error;
|
||||
use litellm_llms_types::formats::ocr::LiteLLMOcrResponse;
|
||||
use pyo3::{
|
||||
|
|
|
|||
|
|
@ -5,9 +5,9 @@ mod project;
|
|||
|
||||
use host::OcrPythonHost;
|
||||
use litellm_callbacks_legacy_python::LoggingOperation;
|
||||
use litellm_core::ocr::provider_config;
|
||||
use litellm_core_utils::settings::ProcessEnvironment;
|
||||
use litellm_host_python::to_py;
|
||||
use litellm_inference::ocr::provider_config;
|
||||
use litellm_llms::base_llm::ocr::settings::OcrSettings;
|
||||
use pyo3::{
|
||||
prelude::*,
|
||||
|
|
@ -58,7 +58,7 @@ fn run_ocr(
|
|||
crate::secrets::source(py)?,
|
||||
)
|
||||
.map_err(http::client_error)?;
|
||||
let route = litellm_core::ocr::OcrRoute::new(client);
|
||||
let route = litellm_inference::ocr::OcrRoute::new(client);
|
||||
Ok(route.machine(request, None))
|
||||
},
|
||||
OcrPythonHost::new(request.unbind()),
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
use litellm_auth::SecretValue;
|
||||
use litellm_core::ocr::{
|
||||
use litellm_host_python::from_py;
|
||||
use litellm_inference::ocr::{
|
||||
types::{LiteLLMOcrRequest, OcrDocumentInput},
|
||||
wire::{OcrWireRequest, consumed_optional_params, decode_document, decode_request_input},
|
||||
};
|
||||
use litellm_host_python::from_py;
|
||||
use litellm_llms::base_llm::ocr::error::Error;
|
||||
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
|
||||
use serde_json::{Map, Value};
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
mod host;
|
||||
|
||||
use litellm_core::responses::websocket::ResponsesWebSocketConnection as RustResponsesWebSocketConnection;
|
||||
use litellm_inference::responses::websocket::ResponsesWebSocketConnection as RustResponsesWebSocketConnection;
|
||||
use pyo3::{
|
||||
prelude::*,
|
||||
types::{PyDict, PyTuple},
|
||||
|
|
@ -81,7 +81,7 @@ fn run_public(
|
|||
py,
|
||||
arguments,
|
||||
move |py, arguments, request| {
|
||||
let route = litellm_core::responses::ResponsesRoute::new(
|
||||
let route = litellm_inference::responses::ResponsesRoute::new(
|
||||
crate::http::provider_client(py, arguments, asynchronous)?
|
||||
.map_err(crate::http::client_error)?,
|
||||
crate::http::resources().auth.clone(),
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue