pub struct LoadBalancedWorker { /* private fields */ }Expand description
A worker pool that distributes inference requests across multiple backends.
Wraps any number of Arc<dyn ModelWorker> instances and routes requests
using a configurable LoadBalanceStrategy. All strategies are
thread-safe; the pool is safe to share across Tokio tasks via Arc.
§Example
use std::sync::Arc;
let pool = LoadBalancedWorker::round_robin(vec![
Arc::new(AnthropicWorker::new("claude-sonnet-4-6")?) as Arc<dyn tokio_prompt_orchestrator::ModelWorker>,
Arc::new(OpenAiWorker::new("gpt-4o")?) as Arc<dyn tokio_prompt_orchestrator::ModelWorker>,
]);Implementations§
Source§impl LoadBalancedWorker
impl LoadBalancedWorker
Sourcepub fn round_robin(workers: Vec<Arc<dyn ModelWorker>>) -> Self
pub fn round_robin(workers: Vec<Arc<dyn ModelWorker>>) -> Self
Sourcepub fn replicate(worker: Arc<dyn ModelWorker>, n: usize) -> Self
pub fn replicate(worker: Arc<dyn ModelWorker>, n: usize) -> Self
Create a round-robin pool by replicating a single worker n times.
All pool slots share the same underlying Arc; this is useful for
concurrency control rather than load distribution across different backends.
§Panics
Panics if n is zero.
Sourcepub fn least_loaded(workers: Vec<Arc<dyn ModelWorker>>) -> Self
pub fn least_loaded(workers: Vec<Arc<dyn ModelWorker>>) -> Self
Sourcepub fn with_names(self, names: Vec<String>) -> Self
pub fn with_names(self, names: Vec<String>) -> Self
Attach human-readable names to each worker slot for metrics labels.
If names is shorter than the pool, unnamed workers default to "unknown".
Trait Implementations§
Source§impl ModelWorker for LoadBalancedWorker
impl ModelWorker for LoadBalancedWorker
Source§fn infer<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<Vec<String>, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
fn infer<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<Vec<String>, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
Perform inference on the given prompt. Read more
Source§fn infer_stream<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<TokenStream, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
fn infer_stream<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<TokenStream, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
Stream inference tokens as they arrive from the provider. Read more
Auto Trait Implementations§
impl !Freeze for LoadBalancedWorker
impl !RefUnwindSafe for LoadBalancedWorker
impl !UnwindSafe for LoadBalancedWorker
impl Send for LoadBalancedWorker
impl Sync for LoadBalancedWorker
impl Unpin for LoadBalancedWorker
impl UnsafeUnpin for LoadBalancedWorker
Blanket Implementations§
Source§impl<T> BorrowMut<T> for Twhere
T: ?Sized,
impl<T> BorrowMut<T> for Twhere
T: ?Sized,
Source§fn borrow_mut(&mut self) -> &mut T
fn borrow_mut(&mut self) -> &mut T
Mutably borrows from an owned value. Read more
impl<ST, DT> CastableFrom<ST, Initialized, Initialized> for DT
impl<ST, DT> CastableFrom<ST, Uninit, Uninit> for DT
§impl<T> FutureExt for T
impl<T> FutureExt for T
§fn with_context(self, otel_cx: Context) -> WithContext<Self>
fn with_context(self, otel_cx: Context) -> WithContext<Self>
§fn with_current_context(self) -> WithContext<Self>
fn with_current_context(self) -> WithContext<Self>
§impl<T> Instrument for T
impl<T> Instrument for T
§fn instrument(self, span: Span) -> Instrumented<Self>
fn instrument(self, span: Span) -> Instrumented<Self>
§fn in_current_span(self) -> Instrumented<Self>
fn in_current_span(self) -> Instrumented<Self>
Source§impl<T> IntoRequest<T> for T
impl<T> IntoRequest<T> for T
Source§fn into_request(self) -> Request<T>
fn into_request(self) -> Request<T>
Wrap the input message
T in a tonic::Request