Skip to main content

LlmProvider

Trait LlmProvider 

Source
pub trait LlmProvider: Send + Sync {
    // Required method
    fn complete<'life0, 'life1, 'life2, 'async_trait>(
        &'life0 self,
        prompt: &'life1 str,
        model: &'life2 str,
    ) -> Pin<Box<dyn Future<Output = Result<String, AgentRuntimeError>> + Send + 'async_trait>>
       where Self: 'async_trait,
             'life0: 'async_trait,
             'life1: 'async_trait,
             'life2: 'async_trait;

    // Provided methods
    fn complete_with_options<'life0, 'life1, 'life2, 'async_trait>(
        &'life0 self,
        prompt: &'life1 str,
        options: CompletionOptions<'life2>,
    ) -> Pin<Box<dyn Future<Output = Result<String, AgentRuntimeError>> + Send + 'async_trait>>
       where Self: 'async_trait,
             'life0: 'async_trait,
             'life1: 'async_trait,
             'life2: 'async_trait { ... }
    fn stream_complete<'life0, 'life1, 'life2, 'async_trait>(
        &'life0 self,
        prompt: &'life1 str,
        model: &'life2 str,
    ) -> Pin<Box<dyn Future<Output = Result<Receiver<Result<String, AgentRuntimeError>>, AgentRuntimeError>> + Send + 'async_trait>>
       where Self: 'async_trait,
             'life0: 'async_trait,
             'life1: 'async_trait,
             'life2: 'async_trait { ... }
}
Expand description

Abstraction over an LLM inference endpoint.

Implement this trait to integrate any model API with AgentRuntime. Built-in implementations are provided for Anthropic and OpenAI when the corresponding feature flags are enabled.

Required Methods§

Source

fn complete<'life0, 'life1, 'life2, 'async_trait>( &'life0 self, prompt: &'life1 str, model: &'life2 str, ) -> Pin<Box<dyn Future<Output = Result<String, AgentRuntimeError>> + Send + 'async_trait>>
where Self: 'async_trait, 'life0: 'async_trait, 'life1: 'async_trait, 'life2: 'async_trait,

Send a prompt to the model and return the completion text.

§Arguments
  • prompt — the full prompt / context string
  • model — model identifier (e.g. "claude-sonnet-4-6", "gpt-4o")

Provided Methods§

Source

fn complete_with_options<'life0, 'life1, 'life2, 'async_trait>( &'life0 self, prompt: &'life1 str, options: CompletionOptions<'life2>, ) -> Pin<Box<dyn Future<Output = Result<String, AgentRuntimeError>> + Send + 'async_trait>>
where Self: 'async_trait, 'life0: 'async_trait, 'life1: 'async_trait, 'life2: 'async_trait,

Send a prompt with additional per-request options.

The default implementation ignores options.max_tokens, options.temperature, and options.timeout and delegates to complete(prompt, options.model). Override this method to honour those fields in your provider implementation.

Source

fn stream_complete<'life0, 'life1, 'life2, 'async_trait>( &'life0 self, prompt: &'life1 str, model: &'life2 str, ) -> Pin<Box<dyn Future<Output = Result<Receiver<Result<String, AgentRuntimeError>>, AgentRuntimeError>> + Send + 'async_trait>>
where Self: 'async_trait, 'life0: 'async_trait, 'life1: 'async_trait, 'life2: 'async_trait,

Stream the completion token-by-token.

Returns a Receiver that yields string chunks as they arrive. The channel closes when the stream is complete or an error occurs.

§Default implementation

The default wraps complete into a single-chunk stream using a channel with a buffer of 64 slots. Custom providers that support true token streaming should override this method; the 64-slot buffer is sized so that fast producers do not block waiting for a slow consumer to drain the first chunk.

§Note for implementors

If you override this method, choose a channel capacity that balances memory use against throughput for your expected token rate. A capacity of 1 will cause the producer to block after each token; a capacity of 0 is unbounded and may exhaust memory on a slow consumer.

Implementors§