pub struct VllmWorker { /* private fields */ }Expand description
vLLM inference server worker
Connects to a vLLM server instance. Server URL can be set via VLLM_URL environment variable or defaults to http://localhost:8000
§Example
use tokio_prompt_orchestrator::VllmWorker;
use std::sync::Arc;
let worker = Arc::new(
VllmWorker::new()
.with_url("http://localhost:8000")
.with_max_tokens(1024)
);Implementations§
Source§impl VllmWorker
impl VllmWorker
Sourcepub fn new() -> Self
pub fn new() -> Self
Create a new VllmWorker pointing at a vLLM inference server.
Reads the server URL from the VLLM_URL environment variable. If the
variable is not set the worker falls back to http://localhost:8000.
Default settings: 512 max tokens, temperature 0.7, top_p 0.95, 60 s
timeout.
§Environment Variables
| Variable | Required | Description |
|---|---|---|
VLLM_URL | No | vLLM server base URL (default: http://localhost:8000) |
§Errors
This constructor never returns an error. Network errors are deferred
until ModelWorker::infer is called.
§Examples
use tokio_prompt_orchestrator::VllmWorker;
use std::sync::Arc;
// Optionally set VLLM_URL=http://gpu-host:8000 in the environment.
let worker = Arc::new(
VllmWorker::new()
.with_max_tokens(1024)
.with_temperature(0.5),
);Sourcepub fn with_max_tokens(self, max_tokens: u32) -> Self
pub fn with_max_tokens(self, max_tokens: u32) -> Self
Set maximum tokens to generate
Sourcepub fn with_temperature(self, temperature: f32) -> Self
pub fn with_temperature(self, temperature: f32) -> Self
Set temperature (0.0 – 2.0).
Values outside [0.0, 2.0] are clamped and a WARN-level log line is
emitted. Most vLLM-compatible servers reject values outside this range.
Sourcepub fn with_top_p(self, top_p: f32) -> Self
pub fn with_top_p(self, top_p: f32) -> Self
Set top_p sampling parameter
Sourcepub fn with_timeout(self, timeout: Duration) -> Self
pub fn with_timeout(self, timeout: Duration) -> Self
Set request timeout
Trait Implementations§
Source§impl Default for VllmWorker
impl Default for VllmWorker
Source§impl ModelWorker for VllmWorker
impl ModelWorker for VllmWorker
Source§fn infer<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<Vec<String>, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
fn infer<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<Vec<String>, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
Source§fn infer_stream<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<TokenStream, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
fn infer_stream<'life0, 'life1, 'async_trait>(
&'life0 self,
prompt: &'life1 str,
) -> Pin<Box<dyn Future<Output = Result<TokenStream, OrchestratorError>> + Send + 'async_trait>>where
Self: 'async_trait,
'life0: 'async_trait,
'life1: 'async_trait,
Auto Trait Implementations§
impl !RefUnwindSafe for VllmWorker
impl !UnwindSafe for VllmWorker
impl Freeze for VllmWorker
impl Send for VllmWorker
impl Sync for VllmWorker
impl Unpin for VllmWorker
impl UnsafeUnpin for VllmWorker
Blanket Implementations§
Source§impl<T> BorrowMut<T> for Twhere
T: ?Sized,
impl<T> BorrowMut<T> for Twhere
T: ?Sized,
Source§fn borrow_mut(&mut self) -> &mut T
fn borrow_mut(&mut self) -> &mut T
impl<ST, DT> CastableFrom<ST, Initialized, Initialized> for DT
impl<ST, DT> CastableFrom<ST, Uninit, Uninit> for DT
§impl<T> FutureExt for T
impl<T> FutureExt for T
§fn with_context(self, otel_cx: Context) -> WithContext<Self>
fn with_context(self, otel_cx: Context) -> WithContext<Self>
§fn with_current_context(self) -> WithContext<Self>
fn with_current_context(self) -> WithContext<Self>
§impl<T> Instrument for T
impl<T> Instrument for T
§fn instrument(self, span: Span) -> Instrumented<Self>
fn instrument(self, span: Span) -> Instrumented<Self>
§fn in_current_span(self) -> Instrumented<Self>
fn in_current_span(self) -> Instrumented<Self>
Source§impl<T> IntoRequest<T> for T
impl<T> IntoRequest<T> for T
Source§fn into_request(self) -> Request<T>
fn into_request(self) -> Request<T>
T in a tonic::Request