refactor(rust): rename litellm-core to litellm-inference (#44802)

* refactor(rust): rename litellm-core to litellm-inference

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(rust): expose inference base API for format crates

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(rust): move shared inference test helpers behind test-support

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: Yujong Lee <yujong@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-06 07:17:24 -07:00 • committed by GitHub
parent 10444df3a0
commit 1a9b533e71
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
105 changed files with 310 additions and 276 deletions

View file

@ -3697,50 +3697,6 @@ dependencies = [
"thiserror 2.0.19",
]
[[package]]
name = "litellm-core"
version = "0.1.0"
dependencies = [
"base64 0.22.1",
"bytes",
"futures-util",
"litellm-auth",
"litellm-auth-aws",
"litellm-auth-gcp",
"litellm-cache",
"litellm-cache-memory",
"litellm-cache-response",
"litellm-core-utils",
"litellm-framer",
"litellm-host",
"litellm-host-native",
"litellm-http",
"litellm-llms",
"litellm-llms-types",
"litellm-secrets",
"litellm-tracing",
"mime_guess",
"moka",
"rand 0.8.7",
"reqwest 0.12.28",
"rstest",
"rstest_reuse",
"serde",
"serde_json",
"sha2 0.10.9",
"strum",
"subtle",
"thiserror 2.0.19",
"time",
"tokio",
"tokio-tungstenite",
"tokio-util",
"tracing",
"url",
"veil",
"wiremock",
]
[[package]]
name = "litellm-core-utils"
version = "0.1.0"
@ -3821,12 +3777,12 @@ dependencies = [
"http-body-util",
"litellm-auth-types",
"litellm-config",
"litellm-core",
"litellm-gateway-auth",
"litellm-gateway-inference",
"litellm-gateway-mcp",
"litellm-gateway-ui",
"litellm-http",
"litellm-inference",
"litellm-llms",
"litellm-secrets",
"litellm-tracing",
@ -3875,11 +3831,11 @@ dependencies = [
"litellm-auth",
"litellm-cache-memory",
"litellm-cache-response",
"litellm-core",
"litellm-gateway-auth",
"litellm-host",
"litellm-host-http",
"litellm-http",
"litellm-inference",
"litellm-llms",
"litellm-llms-types",
"litellm-router",
@ -4038,6 +3994,51 @@ dependencies = [
"webpki-roots",
]
[[package]]
name = "litellm-inference"
version = "0.1.0"
dependencies = [
"base64 0.22.1",
"bytes",
"futures-util",
"litellm-auth",
"litellm-auth-aws",
"litellm-auth-gcp",
"litellm-cache",
"litellm-cache-memory",
"litellm-cache-response",
"litellm-core-utils",
"litellm-framer",
"litellm-host",
"litellm-host-native",
"litellm-http",
"litellm-inference",
"litellm-llms",
"litellm-llms-types",
"litellm-secrets",
"litellm-tracing",
"mime_guess",
"moka",
"rand 0.8.7",
"reqwest 0.12.28",
"rstest",
"rstest_reuse",
"serde",
"serde_json",
"sha2 0.10.9",
"strum",
"subtle",
"thiserror 2.0.19",
"time",
"tokio",
"tokio-tungstenite",
"tokio-util",
"tracing",
"url",
"veil",
"wiremock",
]
[[package]]
name = "litellm-llms"
version = "0.1.0"
@ -4116,11 +4117,11 @@ dependencies = [
"litellm-cache-redis",
"litellm-cache-response",
"litellm-callbacks-legacy-python",
"litellm-core",
"litellm-core-utils",
"litellm-host",
"litellm-host-python",
"litellm-http",
"litellm-inference",
"litellm-llms",
"litellm-llms-types",
"litellm-secrets",
@ -4171,7 +4172,7 @@ name = "litellm-router"
version = "0.1.0"
dependencies = [
"litellm-config",
"litellm-core",
"litellm-inference",
"rstest",
]

View file

@ -16,7 +16,7 @@ litellm-traces = { path = "crates/traces" }
litellm-traces-cache = { path = "crates/traces-cache" }
litellm-traces-clickhouse = { path = "crates/traces-clickhouse" }
litellm-storage-clickhouse = { path = "crates/storage-clickhouse" }
litellm-core = { path = "crates/core" }
litellm-inference = { path = "crates/inference" }
litellm-gateway-mcp = { path = "crates/gateway-mcp" }
litellm-gateway = { path = "crates/gateway" }
litellm-gateway-inference = { path = "crates/gateway-inference" }

View file

@ -12,7 +12,7 @@ base64.workspace = true
bytes.workspace = true
litellm-auth.workspace = true
litellm-gateway-auth.workspace = true
litellm-core.workspace = true
litellm-inference.workspace = true
litellm-host-http.workspace = true
litellm-host.workspace = true
litellm-http.workspace = true

View file

@ -3,7 +3,7 @@ use std::{path::Path, sync::Arc};
use axum::{Json, extract::State, response::IntoResponse};
use base64::{Engine, engine::general_purpose::STANDARD};
use litellm_core::audio_transcription::types::AudioTranscriptionRequest;
use litellm_inference::audio_transcription::types::AudioTranscriptionRequest;
use serde_json::{Value, json};
use crate::{

View file

@ -72,26 +72,26 @@ fn duration(seconds: f64) -> Result<Duration, Error> {
#[derive(Clone, Default)]
pub(crate) struct CacheHeaders(std::sync::Arc<std::sync::OnceLock<String>>);
impl litellm_host::interceptors::Interceptors<litellm_core::RouteError> for CacheHeaders {
impl litellm_host::interceptors::Interceptors<litellm_inference::RouteError> for CacheHeaders {
async fn before_provider_request(
&self,
wire: litellm_host::interceptors::WireRequest,
_: litellm_host::interceptors::RequestContext,
) -> Result<litellm_host::interceptors::WireRequest, litellm_core::RouteError> {
) -> Result<litellm_host::interceptors::WireRequest, litellm_inference::RouteError> {
Ok(wire)
}
async fn after_provider_response(
&self,
_: litellm_host::interceptors::RawResponse,
) -> Result<(), litellm_core::RouteError> {
) -> Result<(), litellm_inference::RouteError> {
Ok(())
}
async fn result_ready(
&self,
facts: litellm_host::interceptors::ExecutionFacts,
) -> Result<(), litellm_core::RouteError> {
) -> Result<(), litellm_inference::RouteError> {
if let litellm_host::interceptors::ResultSource::Cache { key } = facts.source {
let _ = self.0.set(key);
}

View file

@ -6,7 +6,7 @@ use axum::{
extract::{Path, State},
response::{IntoResponse, Response},
};
use litellm_core::chat_completions::types::ChatCompletionsCall;
use litellm_inference::chat_completions::types::ChatCompletionsCall;
use serde_json::{Map, Value};
use crate::{Error, Gateway, JsonObject, request};

View file

@ -3,8 +3,8 @@ use axum::{
Json,
response::{IntoResponse, Response},
};
use litellm_core::RouteError;
use litellm_http::transport::Error as TransportError;
use litellm_inference::RouteError;
use litellm_llms::base_llm::ocr::error::Error as OcrError;
use serde_json::{Map, Value, json};

View file

@ -15,11 +15,11 @@ mod responses;
use std::sync::Arc;
use axum::{Router, routing::post};
use litellm_core::{
use litellm_http::{ClientVariant, HttpClientConfig, media::UrlPolicy};
use litellm_inference::{
audio_transcription::AudioTranscriptionRoute, chat_completions::ChatCompletionsRoute,
messages::MessagesRoute, ocr::OcrRoute, resources::CoreResources, responses::ResponsesRoute,
};
use litellm_http::{ClientVariant, HttpClientConfig, media::UrlPolicy};
use litellm_llms::base_llm::ocr::{handler::OcrClient, settings::OcrSettings};
use litellm_secrets::source::SecretSource;

View file

@ -10,8 +10,8 @@ use axum::{
http::HeaderMap,
response::{IntoResponse, Response},
};
use litellm_core::messages::{MessagesCall, messages_body, route::Messages};
use litellm_host_http::Sse;
use litellm_inference::messages::{MessagesCall, messages_body, route::Messages};
use litellm_llms_types::headers::{ProviderSpecificHeader, ProviderSpecificHeaders};
use serde_json::{Map, Value};

View file

@ -3,7 +3,7 @@ use std::sync::Arc;
use axum::{Json, extract::State, http::HeaderMap, response::IntoResponse};
use litellm_auth::SecretValue;
use litellm_core::ocr::types::{LiteLLMOcrRequest, OcrConnectionInputs, OcrDocumentInput};
use litellm_inference::ocr::types::{LiteLLMOcrRequest, OcrConnectionInputs, OcrDocumentInput};
use litellm_llms::base_llm::ocr::transformation::decode_request_value;
use litellm_llms_types::formats::ocr::OcrDocument;
use serde_json::Value;

View file

@ -1,9 +1,9 @@
use std::sync::Arc;
use axum::{Json, body::Bytes, extract::State, response::Response};
use litellm_core::responses::{route::Responses, types::ResponsesCall};
use litellm_gateway_auth::AuthenticatedRequest;
use litellm_host_http::Sse;
use litellm_inference::responses::{route::Responses, types::ResponsesCall};
use serde_json::json;
use crate::{Error, Gateway, JsonObject, request};

View file

@ -6,13 +6,13 @@ use axum::{
body::{Body, Bytes},
http::Request,
};
use litellm_core::{
chat_completions::{ChatCompletionsRoute, types::ChatCompletionsRequest},
resources::CoreResources,
};
use litellm_http::{
ClientVariant, HttpClientPool, HttpSettings, Resolution, media::PublicDnsResolver,
};
use litellm_inference::{
chat_completions::{ChatCompletionsRoute, types::ChatCompletionsRequest},
resources::CoreResources,
};
use rstest::rstest;
use serde_json::{Value, json};
use tower::ServiceExt;

View file

@ -10,9 +10,9 @@ use axum::{
response::Response,
};
use futures_util::future::BoxFuture;
use litellm_core::resources::CoreResources;
use litellm_gateway_inference::{Deployment, Gateway, router};
use litellm_http::{HttpClientPool, HttpSettings, Resolution, media::PublicDnsResolver};
use litellm_inference::resources::CoreResources;
use litellm_secrets::{SecretValue, source::SecretSource};
use serde_json::Value;
use tower::ServiceExt;

View file

@ -9,7 +9,7 @@ repository.workspace = true
axum.workspace = true
envy = "0.4.2"
http-body-util = "0.1"
litellm-core.workspace = true
litellm-inference.workspace = true
litellm-gateway-inference.workspace = true
litellm-gateway-mcp.workspace = true
tokio-util = "0.7"

View file

@ -15,13 +15,13 @@ use axum::{
use http_body_util::BodyExt;
use litellm_config::Config;
use litellm_core::resources::CoreResources;
use litellm_gateway_auth::Auth;
use litellm_gateway_inference::{Gateway, ModelRouter};
use litellm_http::{
ClientVariant, HttpClientPool, HttpSettings, HttpSettingsLayer, Resolution, SslVerify,
media::PublicDnsResolver,
};
use litellm_inference::resources::CoreResources;
use litellm_secrets::source::EnvironmentSecrets;
use litellm_tracing::ByteChunk;
use uuid::Uuid;

View file

@ -1,4 +1,4 @@
litellm-core owns route orchestration. Messages and HTTP Responses return `litellm_host::call::CallOutput`, containing either a completed response or a stream head and chunks. OCR and currently non-streaming Chat Completions return their completed response directly
litellm-inference owns route orchestration. Messages and HTTP Responses return `litellm_host::call::CallOutput`, containing either a completed response or a stream head and chunks. OCR and currently non-streaming Chat Completions return their completed response directly
Hosts assemble route objects from shared `CoreResources`, HTTP settings, and secret sources. Each route owns its provider client and authentication dependencies. Gateway routes live for the gateway lifetime; Python assembles routes per call from its settings snapshot
@ -10,7 +10,7 @@ Responses WebSocket sessions remain separate from the HTTP call driver because a
## Crate layering
For Messages, Responses, Chat Completions, OCR, and other API formats, `core/src/<format>/` owns orchestration. Shared API data contracts belong in `litellm-llms-types`, adapter contracts and shared transformation machinery in `llms/src/base_llm/<format>/`, and provider policy in `llms/src/<provider>/<format>/`. A repeated format directory name does not imply interchangeable responsibilities. Select concrete adapters here, then invoke their contracts instead of applying one provider's policy to every call. Route types describe call envelopes and execution state, not duplicate public payload schemas
For Messages, Responses, Chat Completions, OCR, and other API formats, `inference/src/<format>/` owns orchestration. Shared API data contracts belong in `litellm-llms-types`, adapter contracts and shared transformation machinery in `llms/src/base_llm/<format>/`, and provider policy in `llms/src/<provider>/<format>/`. A repeated format directory name does not imply interchangeable responsibilities. Select concrete adapters here, then invoke their contracts instead of applying one provider's policy to every call. Route types describe call envelopes and execution state, not duplicate public payload schemas
Crates separate API data, transformations, transport, and orchestration. Python package names identify counterparts, not ownership. Dependencies only point down:
@ -18,9 +18,9 @@ Crates separate API data, transformations, transport, and orchestration. Python
- `litellm-core-utils` mirrors `litellm/litellm_core_utils/`: pure helpers (provider resolution, prompt factory, call arguments, settings lookup and layer merge), no network I/O
- `litellm-http` is Rust-only and route-neutral: settings resolution, the pooled `reqwest` clients, TLS, proxies, the SSRF-safe media fetcher, request and header helpers, and transport errors. Python's `litellm/llms/custom_httpx/` is split by responsibility instead of mirrored: its transport half lives here, its OCR handler in `litellm-llms`
- `litellm-llms` mirrors `litellm/llms/`: `base_llm/<api>/transformation.rs`, `<provider>/<api>/transformation.rs`, and `base_llm/ocr/handler.rs` (the OCR request handler)
- `litellm-core` mirrors the route packages (`litellm/ocr/`, `litellm/messages/`, ...): entrypoints, route request types, provider dispatch, the route machine, and hooks
- `litellm-inference` mirrors the route packages (`litellm/ocr/`, `litellm/messages/`, ...): entrypoints, route request types, provider dispatch, the route machine, and hooks
A route module owns the call entrypoint, route request types (`*Request<'a>`), credential fallback, provider dispatch, and the handler glue that runs a provider config. Provider code never imports from core; when it needs the caller's hooks mid-call it goes through `litellm_llms::base_llm::ocr::handler::CallHooks`, the provider-level hooks OCR implements over its host until it folds into `litellm_host::interceptors::Interceptors`. Import every item from its canonical path. Never re-export another crate's items or give an item a second public path; the only re-export allowed is a private submodule surfacing its item at its module root (`mod error; pub use error::Error;`). Handlers belong in core or llms, never in a host crate
A route module owns the call entrypoint, route request types (`*Request<'a>`), credential fallback, provider dispatch, and the handler glue that runs a provider config. Provider code never imports from inference; when it needs the caller's hooks mid-call it goes through `litellm_llms::base_llm::ocr::handler::CallHooks`, the provider-level hooks OCR implements over its host until it folds into `litellm_host::interceptors::Interceptors`. Import every item from its canonical path. Never re-export another crate's items or give an item a second public path; the only re-export allowed is a private submodule surfacing its item at its module root (`mod error; pub use error::Error;`). Handlers belong in core or llms, never in a host crate
## Error placement

View file

@ -1,10 +1,13 @@
[package]
name = "litellm-core"
name = "litellm-inference"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[features]
test-support = []
[dependencies]
litellm-cache.workspace = true
litellm-cache-response.workspace = true
@ -40,6 +43,7 @@ url.workspace = true
veil.workspace = true
[dev-dependencies]
litellm-inference = { workspace = true, features = ["test-support"] }
litellm-cache-memory.workspace = true
litellm-http = { workspace = true, features = ["test-support"] }
litellm-auth-gcp.workspace = true

View file

@ -1,3 +1,4 @@
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
use litellm_http::request::string_headers;
use litellm_http::request::with_default_headers;
use litellm_llms::{
@ -13,7 +14,7 @@ use super::Error;
use crate::audio_transcription::types::{
AudioTranscriptionRequest, ProviderAudioTranscriptionRequest,
};
use crate::provider::{LlmProviders, resolve_llm_provider};
use crate::provider::resolve_llm_provider;
fn provider_config(provider: LlmProviders) -> Option<&'static dyn BaseAudioTranscriptionConfig> {
match provider {

View file

@ -221,13 +221,13 @@ where
Ok(cache.finish(output, &source).await)
}
pub(crate) struct CallCache<P> {
pub struct CallCache<P> {
session: Option<CacheSession>,
protocol: PhantomData<P>,
}
impl<P: StreamCachable> CallCache<P> {
pub(crate) fn from_wire(
pub fn from_wire(
cache: Option<&ScopedCache>,
policy: CachePolicy,
identity: &ProviderIdentity,
@ -254,7 +254,7 @@ impl<P: StreamCachable> CallCache<P> {
}
}
pub(crate) async fn lookup(&self) -> Option<(OutputOf<P>, ResultSource)>
pub async fn lookup(&self) -> Option<(OutputOf<P>, ResultSource)>
where
P::Response: DeserializeOwned,
{
@ -271,7 +271,7 @@ impl<P: StreamCachable> CallCache<P> {
))
}
pub(crate) async fn finish(self, output: OutputOf<P>, source: &ResultSource) -> OutputOf<P>
pub async fn finish(self, output: OutputOf<P>, source: &ResultSource) -> OutputOf<P>
where
P::Response: Serialize,
{

View file

@ -1,3 +1,4 @@
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
use litellm_http::request::string_headers as shared_string_headers;
use litellm_llms::{
anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG,
@ -8,7 +9,6 @@ use litellm_llms::{
use serde_json::{Map, Value};
use super::Error;
use crate::provider::LlmProviders;
const HEADER_CONTEXT: &str = "chat completions";

View file

@ -7,7 +7,7 @@ use litellm_host::{
use crate::{CallOptions, RouteError};
pub(crate) struct CallContext<'a, I> {
pub struct CallContext<'a, I> {
pub interceptors: &'a I,
pub observers: Option<ObservationSender>,
pub cache: CachePolicy,

View file

@ -41,17 +41,17 @@ impl Drop for Completion {
}
}
pub(crate) fn provider(model: &str, provider: &str) {
pub fn provider(model: &str, provider: &str) {
let span = Span::current();
span.record("resolved_model", model);
span.record("provider", provider);
}
pub(crate) async fn unary<R, E>(execute: impl Future<Output = Result<R, E>>) -> Result<R, E> {
pub async fn unary<R, E>(execute: impl Future<Output = Result<R, E>>) -> Result<R, E> {
operation("litellm.route", execute).await
}
pub(crate) async fn operation<R, E>(
pub async fn operation<R, E>(
name: &str,
execute: impl Future<Output = Result<R, E>>,
) -> Result<R, E> {
@ -61,7 +61,7 @@ pub(crate) async fn operation<R, E>(
result
}
pub(crate) async fn call<R, H, C, E>(
pub async fn call<R, H, C, E>(
execute: impl Future<Output = Result<CallOutput<R, H, C, E>, E>>,
) -> Result<CallOutput<R, H, C, E>, E>
where

View file

@ -1,5 +1,5 @@
mod context;
mod diagnostic;
pub mod context;
pub mod diagnostic;
pub mod audio_transcription;
pub mod caching;
@ -8,10 +8,12 @@ pub mod constants;
pub mod error;
pub mod messages;
pub mod ocr;
mod outbound;
mod provider;
pub mod outbound;
pub mod provider;
pub mod resources;
pub mod responses;
#[cfg(feature = "test-support")]
pub mod test_support;
pub use error::RouteError;

View file

@ -1,3 +1,4 @@
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
use litellm_http::request::string_headers as shared_string_headers;
pub(super) use litellm_http::request::truncate_error_body;
use litellm_llms::{
@ -9,7 +10,6 @@ use litellm_llms::{
use serde_json::{Map, Value};
use super::Error;
use crate::provider::LlmProviders;
const HEADER_CONTEXT: &str = "messages";
@ -68,7 +68,7 @@ mod tests {
use super::{MessagesProvider, messages_provider, string_headers, truncate_error_body};
use crate::messages::Error;
use crate::provider::LlmProviders;
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
#[rstest]
#[case::anthropic("anthropic", MessagesProvider::Anthropic)]

View file

@ -1,4 +1,5 @@
use litellm_auth::{InputSource, SecretValue, Sourced};
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
use litellm_llms::base_llm::ocr::{
handler::OcrClient,
transformation::{OcrConnection, OcrCredentialInputs, PreparedOcrRequest},
@ -6,7 +7,6 @@ use litellm_llms::base_llm::ocr::{
use litellm_secrets::source::Secrets;
use crate::ocr::types::{LiteLLMOcrRequest, ResolvedOcrRequest};
use crate::provider::LlmProviders;
pub(crate) fn prepare_request(
request: ResolvedOcrRequest,

View file

@ -1,4 +1,4 @@
use crate::provider::LlmProviders;
use litellm_core_utils::get_llm_provider_logic::LlmProviders;
use litellm_core_utils::get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider};
use litellm_llms::{
aws_textract::ocr::{

View file

@ -10,7 +10,7 @@ use serde_json::Value;
skip_all,
fields(status)
)]
pub(crate) async fn send(
pub async fn send(
request: OutboundRequest,
client: &litellm_http::Client,
) -> Result<reqwest::Response, reqwest::Error> {
@ -21,7 +21,7 @@ pub(crate) async fn send(
/// Header credentials are already in `headers`; SigV4 is applied here, over the
/// bytes that are sent.
pub(crate) fn outbound_request(
pub fn outbound_request(
authenticated: Authenticated,
url: String,
body: &Value,

View file

@ -1,15 +1,16 @@
pub use litellm_core_utils::get_llm_provider_logic::LlmProviders;
use litellm_core_utils::get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider};
use litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, LlmProviders, get_custom_llm_provider,
};
use crate::error::RouteError as Error;
#[derive(Debug)]
pub(crate) struct ResolvedProvider<'a> {
pub(crate) model: &'a str,
pub(crate) provider: LlmProviders,
pub struct ResolvedProvider<'a> {
pub model: &'a str,
pub provider: LlmProviders,
}
pub(crate) fn resolve_llm_provider<'a>(
pub fn resolve_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
route: &'static str,

View file

@ -0,0 +1,87 @@
use std::sync::{Arc, Mutex};
use futures_util::future::BoxFuture;
use litellm_http::{
ClientVariant, HttpClientConfig, HttpClientPool, HttpSettings, Resolution,
media::PublicDnsResolver,
};
use litellm_secrets::{SecretValue, source::SecretSource};
use crate::resources::CoreResources;
pub fn http_pool() -> HttpClientPool {
HttpClientPool::new(Arc::new(PublicDnsResolver))
}
pub fn resources() -> CoreResources {
CoreResources::new(Arc::new(http_pool()))
}
pub fn no_secrets() -> Arc<dyn SecretSource> {
Arc::new(RecordingSecrets::empty())
}
pub fn provider_http(resources: &CoreResources, config: &HttpClientConfig) -> litellm_http::Client {
resources
.pool
.client(config, ClientVariant::Provider)
.unwrap()
}
pub fn http_config() -> HttpClientConfig {
Resolution::from(&HttpSettings::default()).config
}
/// A secret source that answers from a fixed table and records every name it was asked for.
pub struct RecordingSecrets {
values: Vec<(String, String)>,
fails: bool,
requested: Mutex<Vec<String>>,
}
impl RecordingSecrets {
pub fn new<'a>(values: impl IntoIterator<Item = (&'a str, &'a str)>) -> Self {
Self {
values: values
.into_iter()
.map(|(name, value)| (name.to_string(), value.to_string()))
.collect(),
fails: false,
requested: Mutex::new(Vec::new()),
}
}
pub fn empty() -> Self {
Self::new([])
}
pub fn failing() -> Self {
Self {
fails: true,
..Self::empty()
}
}
pub fn requested(&self) -> Vec<String> {
self.requested.lock().unwrap().clone()
}
}
impl SecretSource for RecordingSecrets {
fn get_secret_str<'a>(
&'a self,
name: &'a str,
) -> BoxFuture<'a, Result<Option<SecretValue>, litellm_secrets::Error>> {
Box::pin(async move {
self.requested.lock().unwrap().push(name.to_string());
if self.fails {
return Err(litellm_secrets::Error::ManagedSecretMissing);
}
Ok(self
.values
.iter()
.find(|(key, _)| key == name)
.map(|(_, value)| SecretValue::new(value.clone())))
})
}
}

View file

@ -1,4 +1,4 @@
use litellm_core::audio_transcription::{Error, types::AudioTranscriptionRequest};
use litellm_inference::audio_transcription::{Error, types::AudioTranscriptionRequest};
use rstest::{fixture, rstest};
use serde_json::{Map, Value, json};
use wiremock::ResponseTemplate;

View file

@ -15,10 +15,6 @@ use litellm_cache_response::{
CacheOptions, CachePolicy, CacheScope, ResponseCache, ResponseCacheConfig,
ResponseCacheService, ResponseEnvelope,
};
use litellm_core::{
RouteError,
caching::{Cachable, CacheRequest, StreamCachable, execute_streaming, execute_unary},
};
use litellm_host::{
call::{CallOutput, OutputOf},
interceptors::{
@ -29,6 +25,10 @@ use litellm_host::{
observation::observation_channel,
protocol::Protocol,
};
use litellm_inference::{
RouteError,
caching::{Cachable, CacheRequest, StreamCachable, execute_streaming, execute_unary},
};
use rstest::{fixture, rstest};
use serde_json::{Value, json};
@ -410,7 +410,7 @@ async fn an_invalid_cached_envelope_is_replaced_by_a_provider_result(#[case] poi
async fn responses_refetches_instead_of_deserializing_another_api_response(
#[case] poisoned: Value,
) {
use litellm_core::responses::route::Responses;
use litellm_inference::responses::route::Responses;
use litellm_llms_types::formats::responses::ResponsesApiResponse;
let cache: Arc<dyn ResponseCacheService> = Arc::new(InvalidEntryCache(
@ -460,7 +460,7 @@ async fn messages_cache_identity_includes_provider_native_parameters(
#[case] original: Value,
#[case] changed: Value,
) {
use litellm_core::messages::route::Messages;
use litellm_inference::messages::route::Messages;
use litellm_llms_types::formats::messages::MessagesResponse;
let calls = AtomicUsize::new(0);
@ -842,7 +842,7 @@ async fn responses_cache_only_reuses_completed_responses(
#[case] status: &str,
#[case] expected_calls: usize,
) {
use litellm_core::responses::route::Responses;
use litellm_inference::responses::route::Responses;
use litellm_llms_types::formats::responses::ResponsesApiResponse;
let calls = AtomicUsize::new(0);
@ -883,7 +883,7 @@ async fn the_same_route_entrypoint_reports_facts_with_or_without_caching(
traces: support::TraceCapture,
) {
use litellm_cache_response::ScopedCache;
use litellm_core::chat_completions::types::ChatCompletionsRequest;
use litellm_inference::chat_completions::types::ChatCompletionsRequest;
use wiremock::{Mock, MockServer, ResponseTemplate, matchers::method};
let upstream = MockServer::start().await;
@ -1038,7 +1038,7 @@ async fn cache_identity_follows_resolved_configuration_and_request_callbacks(
#[case] change: &str,
) {
use litellm_cache_response::ScopedCache;
use litellm_core::{
use litellm_inference::{
chat_completions::{ChatCompletionsRoute, types::ChatCompletionsRequest},
messages::MessagesCall,
responses::types::ResponsesCall,
@ -1178,7 +1178,7 @@ async fn cache_identity_follows_resolved_configuration_and_request_callbacks(
#[tokio::test]
async fn signed_requests_bypass_response_caching(cache: Arc<dyn ResponseCacheService>) {
use litellm_cache_response::ScopedCache;
use litellm_core::chat_completions::types::ChatCompletionsRequest;
use litellm_inference::chat_completions::types::ChatCompletionsRequest;
use wiremock::{Mock, MockServer, ResponseTemplate, matchers::method};
let upstream = MockServer::start().await;

View file

@ -5,8 +5,8 @@ use litellm_host::{
};
use std::time::Duration;
use litellm_core::chat_completions::{Error, types::ChatCompletionsRequest};
use litellm_http::transport::Error as TransportError;
use litellm_inference::chat_completions::{Error, types::ChatCompletionsRequest};
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
use rstest::{fixture, rstest};
use serde_json::{Map, Value, json};
@ -257,8 +257,8 @@ async fn direct_and_hosted_calls_share_hooks_and_lifecycle(
request: ChatCompletionsRequest<'static>,
#[case] hosted: bool,
) {
use litellm_core::chat_completions::route::ChatCompletions;
use litellm_host::{call::HostedCompletion, lifecycle::CallEvent};
use litellm_inference::chat_completions::route::ChatCompletions;
let upstream = upstream([anthropic_response(ANTHROPIC_MESSAGE)]).await;
let base = upstream.uri();

View file

@ -1,11 +1,12 @@
use litellm_host::lifecycle::ExecutionEvent;
use std::sync::Mutex;
use litellm_core::messages::{MessagesCallResponse, route::Messages};
use litellm_host::{
interceptors::{ExecutionFacts, RequestContext, ResultSource, WireRequest},
lifecycle::CallEvent,
};
use litellm_inference::messages::{MessagesCallResponse, route::Messages};
use litellm_inference::test_support::{RecordingSecrets, no_secrets};
use litellm_llms::base_llm::messages::context::MessagesModelCapabilities as AnthropicModelCapabilities;
use rstest::rstest;

View file

@ -3,11 +3,12 @@ use std::{
time::Duration,
};
use litellm_core::messages::{
use litellm_http::{HttpSettings, Resolution};
use litellm_inference::messages::{
Error, MessagesCall, MessagesShaping,
route::{Messages, MessagesMachine, MessagesOutput},
};
use litellm_http::{HttpSettings, Resolution};
use litellm_inference::test_support::RecordingSecrets;
use litellm_llms_types::formats::messages::{MessagesRequest, MessagesResponse};
use litellm_secrets::source::SecretSource;
use rstest::fixture;

View file

@ -1,9 +1,12 @@
use litellm_core::messages::{MessagesCallResponse, messages_body};
use litellm_host::{
interceptors::{ExecutionFacts, ResultSource},
lifecycle::ExecutionEvent,
};
use litellm_http::transport::Error as TransportError;
use litellm_inference::messages::{MessagesCallResponse, messages_body};
use litellm_inference::test_support::{
RecordingSecrets, http_config, no_secrets, provider_http, resources,
};
use rstest::rstest;
use super::*;
@ -268,8 +271,8 @@ async fn the_facade_sends_through_the_injected_http_pool_configuration(call: Mes
..HttpSettings::default()
};
let resources = support::resources();
let response = litellm_core::messages::MessagesRoute::new(
let resources = resources();
let response = litellm_inference::messages::MessagesRoute::new(
provider_http(&resources, &Resolution::from(&settings).config),
resources.auth,
no_secrets(),
@ -348,7 +351,7 @@ async fn route_uses_injected_dependencies_and_optional_cache(
) {
use litellm_cache_memory::InMemoryCache;
use litellm_cache_response::{CacheScope, ResponseCache, ScopedCache};
use litellm_core::messages::MessagesRoute;
use litellm_inference::messages::MessagesRoute;
let upstream = upstream([message_response(), message_response()]).await;
let resources = resources();

View file

@ -1,3 +1,4 @@
use litellm_inference::test_support::RecordingSecrets;
use rstest::rstest;
use super::*;

View file

@ -5,10 +5,11 @@ use std::{
use bytes::Bytes;
use futures_util::{StreamExt, TryStreamExt};
use litellm_core::messages::{
use litellm_inference::messages::{
MessagesCallResponse,
route::{Messages, MessagesStreamHead},
};
use litellm_inference::test_support::{RecordingSecrets, no_secrets};
use litellm_tracing::{Logger, Metadata, Record, Sink};
use rstest::rstest;
use tokio::{
@ -35,7 +36,7 @@ struct TraceSink(mpsc::Sender<(String, Value)>);
impl Sink for TraceSink {
fn enabled(&self, metadata: &Metadata<'_>) -> bool {
metadata.target().starts_with("litellm_core::messages")
metadata.target().starts_with("litellm_inference::messages")
}
fn emit(&self, record: &Record) {

View file

@ -1,6 +1,7 @@
use base64::Engine;
use litellm_core::ocr::types::OcrDocumentInput;
use litellm_host::interceptors::WireRequest;
use litellm_inference::ocr::types::OcrDocumentInput;
use litellm_inference::test_support::{http_config, no_secrets, resources};
use rstest::rstest;
use wiremock::{Mock, matchers::any};

View file

@ -5,14 +5,14 @@ use std::sync::{
atomic::{AtomicUsize, Ordering},
};
use litellm_core::ocr::{
route::{Ocr, OcrCall, OcrOp},
types::OcrDocumentInput,
};
use litellm_host::{
interceptors::{RequestContext, WireRequest},
lifecycle::CallEvent,
};
use litellm_inference::ocr::{
route::{Ocr, OcrCall, OcrOp},
types::OcrDocumentInput,
};
use rstest::rstest;
use super::*;

View file

@ -6,16 +6,16 @@ use std::{
time::Duration,
};
use litellm_core::ocr::{
route::{OcrCall, OcrMachine, OcrOp},
types::OcrDocumentInput,
};
use litellm_host::{
interceptors::{Interceptors, WireRequest},
machine::{HostFailure, Machine, MachineStep},
protocol::{HostRequest, InterceptRequest},
};
use litellm_host_native::services::HostCallHandler;
use litellm_inference::ocr::{
route::{OcrCall, OcrMachine, OcrOp},
types::OcrDocumentInput,
};
use litellm_llms::base_llm::ocr::transformation::OcrTransportConfig;
use rstest::rstest;
use tokio::{io::AsyncReadExt, net::TcpListener, sync::Notify};

View file

@ -1,14 +1,15 @@
use litellm_core::ocr::{
use litellm_host::{
interceptors::{RequestContext, WireRequest},
lifecycle::CallEvent,
};
use litellm_inference::ocr::{
OcrRoute,
document::prepare_document,
route::{Ocr, OcrCall, OcrOp},
types::{LiteLLMOcrRequest, OcrDocumentInput},
wire::{OcrWireRequest, decode_request},
};
use litellm_host::{
interceptors::{RequestContext, WireRequest},
lifecycle::CallEvent,
};
use litellm_inference::test_support::{http_config, no_secrets, resources};
use litellm_llms::base_llm::ocr::{error::Error, settings::OcrSettings};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument};
use serde_json::{Map, Value, json};

View file

@ -1,6 +1,7 @@
use std::sync::Arc;
use litellm_http::{HttpSettings, Resolution};
use litellm_inference::test_support::{RecordingSecrets, http_config, no_secrets, resources};
use litellm_llms::{
base_llm::ocr::transformation::{BaseOcrConfig, OCR_RESPONSE_MAX_BYTES},
mistral::ocr::transformation::MistralOcrConfig,

View file

@ -1,5 +1,5 @@
use litellm_auth::{InputSource, Sourced};
use litellm_core::ocr::arguments::is_supported_request;
use litellm_inference::ocr::arguments::is_supported_request;
use litellm_llms::base_llm::ocr::settings::OcrSettings;
use rstest::rstest;

View file

@ -9,15 +9,16 @@ use litellm_auth::AuthServices;
use litellm_auth_gcp::{
CredentialSource, VertexAuth, VertexAuthFuture, VertexProviderLoader, VertexTokenSource,
};
use litellm_core::{
use litellm_http::{HttpSettings, Resolution};
use litellm_inference::test_support::{RecordingSecrets, http_pool};
use litellm_inference::{
ocr::wire::{OcrWireRequest, decode_request},
resources::CoreResources,
};
use litellm_http::{HttpSettings, Resolution};
use litellm_llms::base_llm::ocr::settings::OcrSettings;
use rstest::{fixture, rstest};
use serde_json::json;
use support::{ReceivedRequest, RecordingSecrets, http_pool, json_response, upstream};
use support::{ReceivedRequest, json_response, upstream};
struct TokenSource(String);

View file

@ -5,11 +5,12 @@ use litellm_host::{
use std::sync::Arc;
use futures_util::TryStreamExt;
use litellm_core::responses::{
use litellm_host::{call::HostedCompletion, lifecycle::CallEvent};
use litellm_inference::responses::{
route::Responses,
types::{ResponsesCall, ResponsesOutput},
};
use litellm_host::{call::HostedCompletion, lifecycle::CallEvent};
use litellm_inference::test_support::{RecordingSecrets, no_secrets};
use rstest::{fixture, rstest};
use serde_json::json;
use wiremock::ResponseTemplate;
@ -417,7 +418,7 @@ async fn websocket_operations_trace_outcomes_without_capturing_frames_or_credent
traces: TraceCapture,
) {
use futures_util::{SinkExt, StreamExt};
use litellm_core::responses::websocket::ResponsesWebSocketConnection;
use litellm_inference::responses::websocket::ResponsesWebSocketConnection;
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
let address = listener.local_addr().unwrap();

View file

@ -8,70 +8,50 @@ use std::{
sync::{Arc, Mutex},
};
use futures_util::future::BoxFuture;
use litellm_http::{
ClientVariant, HttpClientConfig, HttpClientPool, HttpSettings, Resolution,
media::PublicDnsResolver,
};
use litellm_secrets::{SecretValue, source::SecretSource};
use litellm_http::HttpClientConfig;
use litellm_inference::test_support::{http_config, no_secrets, provider_http, resources};
use litellm_secrets::source::SecretSource;
use serde_json::Value;
use wiremock::{Mock, MockServer, Request, ResponseTemplate, matchers::any};
/// A port nothing listens on, for calls that must fail before any request is sent.
pub const UNREACHABLE_BASE: &str = "http://127.0.0.1:1";
pub fn http_pool() -> HttpClientPool {
HttpClientPool::new(Arc::new(PublicDnsResolver))
}
pub fn resources() -> litellm_core::resources::CoreResources {
litellm_core::resources::CoreResources::new(Arc::new(http_pool()))
}
pub fn no_secrets() -> Arc<dyn SecretSource> {
Arc::new(RecordingSecrets::empty())
}
pub fn provider_http(
resources: &litellm_core::resources::CoreResources,
config: &HttpClientConfig,
) -> litellm_http::Client {
resources
.pool
.client(config, ClientVariant::Provider)
.unwrap()
}
pub fn messages_route(secrets: Arc<dyn SecretSource>) -> litellm_core::messages::MessagesRoute {
pub fn messages_route(
secrets: Arc<dyn SecretSource>,
) -> litellm_inference::messages::MessagesRoute {
let resources = resources();
litellm_core::messages::MessagesRoute::new(
litellm_inference::messages::MessagesRoute::new(
provider_http(&resources, &http_config()),
resources.auth,
secrets,
)
}
pub fn chat_completions_route() -> litellm_core::chat_completions::ChatCompletionsRoute {
pub fn chat_completions_route() -> litellm_inference::chat_completions::ChatCompletionsRoute {
let resources = resources();
litellm_core::chat_completions::ChatCompletionsRoute::new(
litellm_inference::chat_completions::ChatCompletionsRoute::new(
provider_http(&resources, &http_config()),
resources.auth,
no_secrets(),
)
}
pub fn responses_route(secrets: Arc<dyn SecretSource>) -> litellm_core::responses::ResponsesRoute {
pub fn responses_route(
secrets: Arc<dyn SecretSource>,
) -> litellm_inference::responses::ResponsesRoute {
let resources = resources();
litellm_core::responses::ResponsesRoute::new(
litellm_inference::responses::ResponsesRoute::new(
provider_http(&resources, &http_config()),
resources.auth,
secrets,
)
}
pub fn audio_transcription_route() -> litellm_core::audio_transcription::AudioTranscriptionRoute {
pub fn audio_transcription_route() -> litellm_inference::audio_transcription::AudioTranscriptionRoute
{
let resources = resources();
litellm_core::audio_transcription::AudioTranscriptionRoute::new(
litellm_inference::audio_transcription::AudioTranscriptionRoute::new(
provider_http(&resources, &http_config()),
resources.auth,
no_secrets(),
@ -79,13 +59,13 @@ pub fn audio_transcription_route() -> litellm_core::audio_transcription::AudioTr
}
pub fn build_ocr_route(
resources: &litellm_core::resources::CoreResources,
resources: &litellm_inference::resources::CoreResources,
config: &HttpClientConfig,
url_policy: litellm_http::media::UrlPolicy,
settings: litellm_llms::base_llm::ocr::settings::OcrSettings,
secrets: Arc<dyn SecretSource>,
) -> litellm_core::ocr::OcrRoute {
litellm_core::ocr::OcrRoute::new(
) -> litellm_inference::ocr::OcrRoute {
litellm_inference::ocr::OcrRoute::new(
litellm_llms::base_llm::ocr::handler::OcrClient::new(
&resources.pool,
config,
@ -98,10 +78,6 @@ pub fn build_ocr_route(
)
}
pub fn http_config() -> HttpClientConfig {
Resolution::from(&HttpSettings::default()).config
}
/// Starts an upstream that answers its n-th request with the n-th response and 404s after.
pub async fn upstream(responses: impl IntoIterator<Item = ResponseTemplate>) -> MockServer {
let server = MockServer::start().await;
@ -189,60 +165,6 @@ impl ReceivedRequest for Request {
}
}
/// A secret source that answers from a fixed table and records every name it was asked for.
pub struct RecordingSecrets {
values: Vec<(String, String)>,
fails: bool,
requested: Mutex<Vec<String>>,
}
impl RecordingSecrets {
pub fn new<'a>(values: impl IntoIterator<Item = (&'a str, &'a str)>) -> Self {
Self {
values: values
.into_iter()
.map(|(name, value)| (name.to_string(), value.to_string()))
.collect(),
fails: false,
requested: Mutex::new(Vec::new()),
}
}
pub fn empty() -> Self {
Self::new([])
}
pub fn failing() -> Self {
Self {
fails: true,
..Self::empty()
}
}
pub fn requested(&self) -> Vec<String> {
self.requested.lock().unwrap().clone()
}
}
impl SecretSource for RecordingSecrets {
fn get_secret_str<'a>(
&'a self,
name: &'a str,
) -> BoxFuture<'a, Result<Option<SecretValue>, litellm_secrets::Error>> {
Box::pin(async move {
self.requested.lock().unwrap().push(name.to_string());
if self.fails {
return Err(litellm_secrets::Error::ManagedSecretMissing);
}
Ok(self
.values
.iter()
.find(|(key, _)| key == name)
.map(|(_, value)| SecretValue::new(value.clone())))
})
}
}
pub struct RecordingCall<P: litellm_host::protocol::Protocol> {
pub request: Mutex<Option<P::Request>>,
pub events: Arc<CallEvents>,
@ -396,7 +318,7 @@ impl TraceCapture {
impl litellm_tracing::Sink for TraceCapture {
fn enabled(&self, metadata: &litellm_tracing::Metadata<'_>) -> bool {
metadata.target().starts_with("litellm_core")
metadata.target().starts_with("litellm_inference")
}
fn emit(&self, record: &litellm_tracing::Record) {

View file

@ -16,7 +16,7 @@ Shared OCR document and response contracts live in `litellm-llms-types::formats:
For Mistral, `async_transform_ocr_request` uses the base default in both languages. `resolve_headers` and `build_ocr_url` implement the respective environment and URL operations, and `normalize_response` implements the typed part of response transformation. Existing auth key/header handling and top-level response-extra preservation differ between languages; layout refactors must preserve those behaviors and verify them with the existing tests
For non-OCR pairs, order corresponding methods as parameter support/mapping, environment validation, URL construction, request transformation, and response transformation, followed by Rust-only runtime hooks. Auth resolution remains split between configs and route preparation in litellm-core. Chat `supported_openai_param_mappings` describes accepted OpenAI/provider name pairs, unlike Python's `get_supported_openai_params` name list. Audio `map_transcription_params` remains a Rust filtering helper
For non-OCR pairs, order corresponding methods as parameter support/mapping, environment validation, URL construction, request transformation, and response transformation, followed by Rust-only runtime hooks. Auth resolution remains split between configs and route preparation in litellm-inference. Chat `supported_openai_param_mappings` describes accepted OpenAI/provider name pairs, unlike Python's `get_supported_openai_params` name list. Audio `map_transcription_params` remains a Rust filtering helper
Azure Messages maps to `llms/azure_ai/anthropic/messages_transformation.py`; Bedrock Converse maps to `llms/bedrock/chat/converse_transformation.py`. `AnthropicConfig`, `AmazonConverseConfig`, and the non-OCR base traits are partial ports. `OpenAiResponsesApiConfig` implements WebSocket transformations and a direct HTTP Responses path. Its HTTP path does not implement Python model-specific parameter rewriting or Responses-to-Chat emulation. Preserve their acceptance gates, passthrough behavior, and host fallback contracts when aligning layout

View file

@ -82,11 +82,11 @@ GIL handling to `litellm-host-python`.
## Bridge Shape
- Prefer one stable method per top-level LiteLLM route, for example
`messages(...)`, calling the matching `litellm-core` entrypoint.
`messages(...)`, calling the matching `litellm-inference` entrypoint.
- Do not add one exported PyO3 function per provider helper unless there is a
measured reason.
- Provider dispatch belongs in the `litellm-core` route module (e.g.
`litellm_core::messages`), not in this PyO3 crate.
- Provider dispatch belongs in the `litellm-inference` route module (e.g.
`litellm_inference::messages`), not in this PyO3 crate.
- Python owns rollout state and fallback. Rust should return errors; Python
decides whether to raise or fall back. For a rust-only provider/route (no
Python reference), the Python side is a thin dispatch that calls Rust and

View file

@ -36,7 +36,7 @@ serde.workspace = true
litellm-auth.workspace = true
litellm-auth-aws.workspace = true
litellm-callbacks-legacy-python.workspace = true
litellm-core.workspace = true
litellm-inference.workspace = true
litellm-core-utils.workspace = true
litellm-http.workspace = true
litellm-llms.workspace = true

View file

@ -4,11 +4,11 @@ use litellm_cache::Error;
use litellm_cache_response::{
ResponseCacheConfig, ResponseCacheRequest, ResponseCacheService, ResponseEnvelope,
};
use litellm_core::caching::CachedOutput;
use litellm_host::{
machine::{HostServices, MachineFault},
protocol::{Protocol, Reply},
};
use litellm_inference::caching::CachedOutput;
use serde_json::Value;
const STREAM_EVENTS_KEY: &str = "litellm_cached_anthropic_sse_events";

View file

@ -1,5 +1,5 @@
use litellm_core::RouteError;
use litellm_http::transport::Error as TransportError;
use litellm_inference::RouteError;
use pyo3::{
exceptions::{PyRuntimeError, PyValueError},
prelude::*,

View file

@ -80,13 +80,13 @@ fn decode_ssl_verify(field: &Field<'_>) -> Result<Option<SslVerify>, ProjectionE
Err(field.invalid("a Boolean, Boolean string, CA path, or None"))
}
static RESOURCES: LazyLock<litellm_core::resources::CoreResources> = LazyLock::new(|| {
litellm_core::resources::CoreResources::new(Arc::new(HttpClientPool::new(Arc::new(
static RESOURCES: LazyLock<litellm_inference::resources::CoreResources> = LazyLock::new(|| {
litellm_inference::resources::CoreResources::new(Arc::new(HttpClientPool::new(Arc::new(
PublicDnsResolver,
))))
});
pub(crate) fn resources() -> &'static litellm_core::resources::CoreResources {
pub(crate) fn resources() -> &'static litellm_inference::resources::CoreResources {
&RESOURCES
}

View file

@ -1,8 +1,8 @@
use crate::execution::{run_async, run_sync};
use litellm_core::audio_transcription::{
use litellm_host_python::from_py_argument;
use litellm_inference::audio_transcription::{
AudioTranscriptionRoute, Error, types::AudioTranscriptionRequest,
};
use litellm_host_python::from_py_argument;
use pyo3::{prelude::*, types::PyDict};
use serde_json::{Map, Value};

View file

@ -3,7 +3,9 @@ mod host;
use pyo3::types::{PyDict, PyTuple};
use crate::execution::{run_async, run_sync};
use litellm_core::chat_completions::{ChatCompletionsRoute, Error, types::ChatCompletionsRequest};
use litellm_inference::chat_completions::{
ChatCompletionsRoute, Error, types::ChatCompletionsRequest,
};
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
use pyo3::prelude::*;
use serde_json::{Map, Value};

View file

@ -1,8 +1,10 @@
use std::convert::Infallible;
use super::super::inference::InferenceHost;
use litellm_core::chat_completions::{Error, route::ChatCompletions, types::ChatCompletionsCall};
use litellm_host_python::{InvokeError, PythonBinding, PythonHostCalls, PythonOwned};
use litellm_inference::chat_completions::{
Error, route::ChatCompletions, types::ChatCompletionsCall,
};
use pyo3::{
gc::{PyTraverseError, PyVisit},
prelude::*,

View file

@ -1,7 +1,7 @@
use litellm_core::RouteError;
use litellm_core_utils::get_llm_provider_logic::get_custom_llm_provider;
use litellm_host_python::{from_py, lookup, to_py};
use litellm_http::transport::Error as TransportError;
use litellm_inference::RouteError;
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
use serde::Serialize;
use serde_json::{Map, Value};

View file

@ -2,12 +2,12 @@ use crate::cache::{CacheCall, Cached, PythonCache, Selection};
use litellm_host_python::{PythonHostCalls, PythonOwned};
use bytes::Bytes;
use litellm_core::messages::{
use litellm_host_python::{InvokeError, PythonBinding, from_py, lookup, to_py};
use litellm_http::transport::Error as TransportError;
use litellm_inference::messages::{
Error, MessagesCall, MessagesShaping, messages_body,
route::{Messages, MessagesStreamHead},
};
use litellm_host_python::{InvokeError, PythonBinding, from_py, lookup, to_py};
use litellm_http::transport::Error as TransportError;
use litellm_llms_types::headers::ProviderSpecificHeaders;
use pyo3::{
exceptions::{PyException, PyValueError},

View file

@ -26,7 +26,7 @@ fn run_messages(
py,
arguments,
move |py, arguments, request| {
let route = litellm_core::messages::MessagesRoute::new(
let route = litellm_inference::messages::MessagesRoute::new(
crate::http::provider_client(py, arguments, asynchronous)?
.map_err(crate::http::client_error)?,
crate::http::resources().auth.clone(),
@ -48,7 +48,7 @@ fn run_messages(
.execute(
call,
&interceptors,
litellm_core::CallOptions {
litellm_inference::CallOptions {
cache: Some(options.policy),
observers,
},

View file

@ -1,7 +1,7 @@
use std::path::PathBuf;
use litellm_core::ocr::types::OcrDocumentInput;
use litellm_host_python::{PythonFileReader, py_bytes};
use litellm_inference::ocr::types::OcrDocumentInput;
use pyo3::{
exceptions::PyValueError,
prelude::*,

View file

@ -1,7 +1,7 @@
use litellm_auth::ResolvedCredential;
use litellm_core::ocr::route::{Ocr, OcrCall, OcrOp};
use litellm_host_python::{InvokeError, PythonBinding, missing_state, to_py};
use litellm_host_python::{PythonHostCalls, PythonOwned};
use litellm_inference::ocr::route::{Ocr, OcrCall, OcrOp};
use litellm_llms::base_llm::ocr::error::Error;
use litellm_llms_types::formats::ocr::LiteLLMOcrResponse;
use pyo3::{

View file

@ -5,9 +5,9 @@ mod project;
use host::OcrPythonHost;
use litellm_callbacks_legacy_python::LoggingOperation;
use litellm_core::ocr::provider_config;
use litellm_core_utils::settings::ProcessEnvironment;
use litellm_host_python::to_py;
use litellm_inference::ocr::provider_config;
use litellm_llms::base_llm::ocr::settings::OcrSettings;
use pyo3::{
prelude::*,
@ -58,7 +58,7 @@ fn run_ocr(
crate::secrets::source(py)?,
)
.map_err(http::client_error)?;
let route = litellm_core::ocr::OcrRoute::new(client);
let route = litellm_inference::ocr::OcrRoute::new(client);
Ok(route.machine(request, None))
},
OcrPythonHost::new(request.unbind()),

View file

@ -1,9 +1,9 @@
use litellm_auth::SecretValue;
use litellm_core::ocr::{
use litellm_host_python::from_py;
use litellm_inference::ocr::{
types::{LiteLLMOcrRequest, OcrDocumentInput},
wire::{OcrWireRequest, consumed_optional_params, decode_document, decode_request_input},
};
use litellm_host_python::from_py;
use litellm_llms::base_llm::ocr::error::Error;
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
use serde_json::{Map, Value};

View file

@ -1,6 +1,6 @@
mod host;
use litellm_core::responses::websocket::ResponsesWebSocketConnection as RustResponsesWebSocketConnection;
use litellm_inference::responses::websocket::ResponsesWebSocketConnection as RustResponsesWebSocketConnection;
use pyo3::{
prelude::*,
types::{PyDict, PyTuple},
@ -81,7 +81,7 @@ fn run_public(
py,
arguments,
move |py, arguments, request| {
let route = litellm_core::responses::ResponsesRoute::new(
let route = litellm_inference::responses::ResponsesRoute::new(
crate::http::provider_client(py, arguments, asynchronous)?
.map_err(crate::http::client_error)?,
crate::http::resources().auth.clone(),

Some files were not shown because too many files have changed in this diff Show more