Chat & Streaming
Basic Chat
Section titled “Basic Chat”Send a message and get a response:
Basic chat completion with a single user message
import asyncioimport osfrom liter_llm import create_clientfrom liter_llm._internal_bindings import ChatCompletionRequest
async def main() -> None: client = create_client(api_key=os.environ["API_KEY"]) req = ChatCompletionRequest.from_json("{\"messages\":[{\"content\":\"Say hello\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"temperature\":0}") result = await client.chat(req) print(result.choices) print(result.choices[0].message.content) print(result.choices[0].finish_reason) print((result.usage.total_tokens if result.usage else None)) print(result.model)
asyncio.run(main())Basic chat completion with a single user message
import { createClient } from "@xberg-io/liter-llm";async function main() { const client = createClient("your-api-key"); const result = await client.chat({ messages: [{ content: "Say hello", role: "user" }], model: "gpt-4", temperature: 0 }); console.log(result.choices); console.log(result.choices?.[0]?.message?.content); console.log(result.choices?.[0]?.finishReason); console.log(result.usage?.totalTokens); console.log(result.model);}
void main();Basic chat completion with a single user message
use liter_llm::BatchClient;use liter_llm::FileClient;use liter_llm::LlmClient;use liter_llm::ResponseClient;
#[tokio::main]async fn main() { let req_json: serde_json::Value = serde_json::from_str(r#"{"messages":[{"content":"Say hello","role":"user"}],"model":"gpt-4","temperature":0}"#).unwrap(); let req = serde_json::from_value(req_json).unwrap(); let client = liter_llm::create_client(std::env::var("API_KEY").expect("API_KEY must be set"), None, None, None, None).unwrap(); let result = client.chat(req).await.expect("call failed"); println!("{:?}", result.choices); println!("{:?}", result.choices[0].message.content); println!("{:?}", result.choices[0].finish_reason); println!("{:?}", result.usage.as_ref().unwrap().total_tokens); println!("{:?}", result.model);}Basic chat completion with a single user message
package main
import ( "fmt" pkg "github.com/xberg-io/liter-llm/packages/go/v2")
func ptr[T any](value T) *T { return &value }func main() { req := pkg.ChatCompletionRequest{ Model: `gpt-4`, Temperature: ptr(float64(0)), } client, clientErr := pkg.CreateClient("your-api-key", nil, nil, nil, nil) if clientErr != nil { panic(clientErr) } defer client.Free() result, err := client.Chat(req) if err != nil { panic(err) } fmt.Printf("%+v\n", result.Choices) fmt.Printf("%+v\n", result.Choices[0].Message.Content) fmt.Printf("%+v\n", result.Choices[0].FinishReason) fmt.Printf("%+v\n", result.Usage.TotalTokens) fmt.Printf("%+v\n", result.Model)}Basic chat completion with a single user message
import io.xberg.literllm.*;
public final class Example { public static void main(String[] args) throws Exception { var reqJson = "{\"messages\":[{\"content\":\"Say hello\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"temperature\":0}"; var req = JsonUtil.fromJson(reqJson, ChatCompletionRequest.class); var apiKey = System.getenv("API_KEY"); if (apiKey == null || apiKey.isEmpty()) throw new IllegalStateException("API_KEY must be set"); try (var client = LiterLlm.createClient(apiKey, null, null, null, null)) { var result = client.chat(req); System.out.println(result.choices()); System.out.println(result.choices().get(0).message().content()); System.out.println(result.choices().get(0).finishReason()); System.out.println(result.usage().totalTokens()); System.out.println(result.model()); } }}Basic chat completion with a single user message
using System;using System.Text.Json;using LiterLlm;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var apiKey = Environment.GetEnvironmentVariable("API_KEY") ?? throw new InvalidOperationException("API_KEY must be set"); using var client = LiterLlmConverter.CreateClient(apiKey, null, null, null, null);var result = await client.ChatAsync(new ChatCompletionRequest { Messages = new List<Message>() { JsonSerializer.Deserialize<Message>("{\"content\":\"Say hello\",\"role\":\"user\"}", ConfigOptions)! }, Model = "gpt-4", Temperature = 0 });Console.WriteLine(result.Choices);Console.WriteLine(result.Choices[0].Message.Content);Console.WriteLine(result.Choices[0].FinishReason);Console.WriteLine(result.Usage!.TotalTokens);Console.WriteLine(result.Model);Basic chat completion with a single user message
require "liter_llm"result = LiterLlm.chat(LiterLlm::ChatCompletionRequest.new(messages: [{ 'content' => 'Say hello', 'role' => 'user' }], model: 'gpt-4', temperature: 0))puts result.choices.inspectputs result.choices[0].message.content.inspectputs result.choices[0].finish_reason.inspectputs result.usage.total_tokens.inspectputs result.model.inspectBasic chat completion with a single user message
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Liter\Llm\LiterLlm;use Liter\Llm\ChatCompletionRequest;$req = \Liter\Llm\ChatCompletionRequest::from_json(json_encode(["messages" => [["content" => "Say hello", "role" => "user"]], "model" => "gpt-4", "temperature" => 0]));$result = LiterLlm::chat($req);var_dump($result->getChoices());var_dump($result->getChoices()[0]->getMessage()->getContent());var_dump($result->getChoices()[0]->finishReason);var_dump($result->getUsage()?->totalTokens);var_dump($result->model);Basic chat completion with a single user message
api_key = System.fetch_env!("API_KEY"){:ok, client} = LiterLlm.create_client(api_key)result = LiterLlm.defaultclient_chat_async(client, "{\"messages\":[{\"content\":\"Say hello\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"temperature\":0}")IO.inspect(result.choices)IO.inspect(Enum.at(result.choices, 0).message.content)IO.inspect(Enum.at(result.choices, 0).finish_reason)IO.inspect(result.usage.total_tokens)IO.inspect(result.model)Basic chat completion with a single user message
import { WasmChatCompletionRequest, createClient } from "@xberg-io/liter-llm-wasm";async function main() { const req: WasmChatCompletionRequest = (() => { const _u0 = WasmChatCompletionRequest.default(); _u0.messages = [{ content: "Say hello", role: "user" }]; _u0.model = "gpt-4"; _u0.temperature = 0; return _u0; })(); const client = createClient("your-api-key"); try { const result = await client.chat(req); console.log(result.choices); console.log(result.choices[0].message.content); console.log(result.choices[0].finishReason); console.log(result.usage?.totalTokens); console.log(result.model); } finally { client.free(); }}
void main();Provider Routing
Section titled “Provider Routing”Liter-llm uses a provider/model prefix convention. The prefix determines which API endpoint, auth header, and parameter mappings to use:
openai/gpt-4o -> OpenAIanthropic/claude-sonnet-4-20250514 -> Anthropicgroq/llama3-70b -> Groqgoogle/gemini-2.0-flash -> Google AImistral/mistral-large -> Mistralbedrock/anthropic.claude-v2 -> AWS BedrockSwitch providers by changing the model string – no other code changes needed.
Message Roles
Section titled “Message Roles”| Role | Purpose |
|---|---|
system |
Sets the assistant’s behavior. Sent once at the start. |
user |
User input – questions, instructions, data. |
assistant |
Previous assistant responses for multi-turn context. |
tool |
Results from tool calls. |
developer |
Developer-level instructions (some providers). |
Multi-Turn Conversations
Section titled “Multi-Turn Conversations”Append the assistant’s response and the next user message, then call chat again:
Multi-turn conversation with system, user, assistant, and follow-up user messages
import asyncioimport osfrom liter_llm import create_clientfrom liter_llm._internal_bindings import ChatCompletionRequest
async def main() -> None: client = create_client(api_key=os.environ["API_KEY"]) req = ChatCompletionRequest.from_json("{\"messages\":[{\"content\":\"You are a helpful assistant.\",\"role\":\"system\"},{\"content\":\"What is 2 + 2?\",\"role\":\"user\"},{\"content\":\"2 + 2 equals 4.\",\"role\":\"assistant\"},{\"content\":\"And what is 4 + 4?\",\"role\":\"user\"}],\"model\":\"gpt-4\"}") result = await client.chat(req) print(result.choices) print(result.choices[0].message.content) print(result.choices[0].finish_reason)
asyncio.run(main())Multi-turn conversation with system, user, assistant, and follow-up user messages
import { createClient } from "@xberg-io/liter-llm";async function main() { const client = createClient("your-api-key"); const result = await client.chat({ messages: [{ content: "You are a helpful assistant.", role: "system" }, { content: "What is 2 + 2?", role: "user" }, { content: "2 + 2 equals 4.", role: "assistant" }, { content: "And what is 4 + 4?", role: "user" }], model: "gpt-4" }); console.log(result.choices); console.log(result.choices?.[0]?.message?.content); console.log(result.choices?.[0]?.finishReason);}
void main();Multi-turn conversation with system, user, assistant, and follow-up user messages
use liter_llm::BatchClient;use liter_llm::FileClient;use liter_llm::LlmClient;use liter_llm::ResponseClient;
#[tokio::main]async fn main() { let req_json: serde_json::Value = serde_json::from_str(r#"{"messages":[{"content":"You are a helpful assistant.","role":"system"},{"content":"What is 2 + 2?","role":"user"},{"content":"2 + 2 equals 4.","role":"assistant"},{"content":"And what is 4 + 4?","role":"user"}],"model":"gpt-4"}"#).unwrap(); let req = serde_json::from_value(req_json).unwrap(); let client = liter_llm::create_client(std::env::var("API_KEY").expect("API_KEY must be set"), None, None, None, None).unwrap(); let result = client.chat(req).await.expect("call failed"); println!("{:?}", result.choices); println!("{:?}", result.choices[0].message.content); println!("{:?}", result.choices[0].finish_reason);}Multi-turn conversation with system, user, assistant, and follow-up user messages
package main
import ( "fmt" pkg "github.com/xberg-io/liter-llm/packages/go/v2")
func main() { req := pkg.ChatCompletionRequest{ Model: `gpt-4`, } client, clientErr := pkg.CreateClient("your-api-key", nil, nil, nil, nil) if clientErr != nil { panic(clientErr) } defer client.Free() result, err := client.Chat(req) if err != nil { panic(err) } fmt.Printf("%+v\n", result.Choices) fmt.Printf("%+v\n", result.Choices[0].Message.Content) fmt.Printf("%+v\n", result.Choices[0].FinishReason)}Multi-turn conversation with system, user, assistant, and follow-up user messages
import io.xberg.literllm.*;
public final class Example { public static void main(String[] args) throws Exception { var reqJson = "{\"messages\":[{\"content\":\"You are a helpful assistant.\",\"role\":\"system\"},{\"content\":\"What is 2 + 2?\",\"role\":\"user\"},{\"content\":\"2 + 2 equals 4.\",\"role\":\"assistant\"},{\"content\":\"And what is 4 + 4?\",\"role\":\"user\"}],\"model\":\"gpt-4\"}"; var req = JsonUtil.fromJson(reqJson, ChatCompletionRequest.class); var apiKey = System.getenv("API_KEY"); if (apiKey == null || apiKey.isEmpty()) throw new IllegalStateException("API_KEY must be set"); try (var client = LiterLlm.createClient(apiKey, null, null, null, null)) { var result = client.chat(req); System.out.println(result.choices()); System.out.println(result.choices().get(0).message().content()); System.out.println(result.choices().get(0).finishReason()); } }}Multi-turn conversation with system, user, assistant, and follow-up user messages
using System;using System.Text.Json;using LiterLlm;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var apiKey = Environment.GetEnvironmentVariable("API_KEY") ?? throw new InvalidOperationException("API_KEY must be set"); using var client = LiterLlmConverter.CreateClient(apiKey, null, null, null, null);var result = await client.ChatAsync(new ChatCompletionRequest { Messages = new List<Message>() { JsonSerializer.Deserialize<Message>("{\"content\":\"You are a helpful assistant.\",\"role\":\"system\"}", ConfigOptions)!, JsonSerializer.Deserialize<Message>("{\"content\":\"What is 2 + 2?\",\"role\":\"user\"}", ConfigOptions)!, JsonSerializer.Deserialize<Message>("{\"content\":\"2 + 2 equals 4.\",\"role\":\"assistant\"}", ConfigOptions)!, JsonSerializer.Deserialize<Message>("{\"content\":\"And what is 4 + 4?\",\"role\":\"user\"}", ConfigOptions)! }, Model = "gpt-4" });Console.WriteLine(result.Choices);Console.WriteLine(result.Choices[0].Message.Content);Console.WriteLine(result.Choices[0].FinishReason);Multi-turn conversation with system, user, assistant, and follow-up user messages
require "liter_llm"result = LiterLlm.chat(LiterLlm::ChatCompletionRequest.new(messages: [{ 'content' => 'You are a helpful assistant.', 'role' => 'system' }, { 'content' => 'What is 2 + 2?', 'role' => 'user' }, { 'content' => '2 + 2 equals 4.', 'role' => 'assistant' }, { 'content' => 'And what is 4 + 4?', 'role' => 'user' }], model: 'gpt-4'))puts result.choices.inspectputs result.choices[0].message.content.inspectputs result.choices[0].finish_reason.inspectMulti-turn conversation with system, user, assistant, and follow-up user messages
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Liter\Llm\LiterLlm;use Liter\Llm\ChatCompletionRequest;$req = \Liter\Llm\ChatCompletionRequest::from_json(json_encode(["messages" => [["content" => "You are a helpful assistant.", "role" => "system"], ["content" => "What is 2 + 2?", "role" => "user"], ["content" => "2 + 2 equals 4.", "role" => "assistant"], ["content" => "And what is 4 + 4?", "role" => "user"]], "model" => "gpt-4"]));$result = LiterLlm::chat($req);var_dump($result->getChoices());var_dump($result->getChoices()[0]->getMessage()->getContent());var_dump($result->getChoices()[0]->finishReason);Multi-turn conversation with system, user, assistant, and follow-up user messages
api_key = System.fetch_env!("API_KEY"){:ok, client} = LiterLlm.create_client(api_key)result = LiterLlm.defaultclient_chat_async(client, "{\"messages\":[{\"content\":\"You are a helpful assistant.\",\"role\":\"system\"},{\"content\":\"What is 2 + 2?\",\"role\":\"user\"},{\"content\":\"2 + 2 equals 4.\",\"role\":\"assistant\"},{\"content\":\"And what is 4 + 4?\",\"role\":\"user\"}],\"model\":\"gpt-4\"}")IO.inspect(result.choices)IO.inspect(Enum.at(result.choices, 0).message.content)IO.inspect(Enum.at(result.choices, 0).finish_reason)Multi-turn conversation with system, user, assistant, and follow-up user messages
import { WasmChatCompletionRequest, createClient } from "@xberg-io/liter-llm-wasm";async function main() { const req: WasmChatCompletionRequest = (() => { const _u0 = WasmChatCompletionRequest.default(); _u0.messages = [{ content: "You are a helpful assistant.", role: "system" }, { content: "What is 2 + 2?", role: "user" }, { content: "2 + 2 equals 4.", role: "assistant" }, { content: "And what is 4 + 4?", role: "user" }]; _u0.model = "gpt-4"; return _u0; })(); const client = createClient("your-api-key"); try { const result = await client.chat(req); console.log(result.choices); console.log(result.choices[0].message.content); console.log(result.choices[0].finishReason); } finally { client.free(); }}
void main();Streaming
Section titled “Streaming”Stream tokens as they arrive instead of waiting for the full response:
Streaming chat completion that produces content across multiple SSE chunks
import asyncioimport osfrom liter_llm import create_clientfrom liter_llm._internal_bindings import ChatCompletionRequest
async def main() -> None: client = create_client(api_key=os.environ["API_KEY"]) req = ChatCompletionRequest.from_json("{\"messages\":[{\"content\":\"Count to 3\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"stream\":true}") result = client.chat_stream(req) chunks = [] async for chunk in result: chunks.append(chunk) print(result)
asyncio.run(main())Streaming chat completion that produces content across multiple SSE chunks
import { createClient } from "@xberg-io/liter-llm";async function main() { const client = createClient("your-api-key"); const result = await client.chatStream({ messages: [{ content: "Count to 3", role: "user" }], model: "gpt-4", stream: true }); for await (const resultChunk of result) { console.log(resultChunk); }}
void main();Streaming chat completion that produces content across multiple SSE chunks
use liter_llm::BatchClient;use liter_llm::FileClient;use liter_llm::LlmClient;use liter_llm::ResponseClient;
#[tokio::main]async fn main() { let req_json: serde_json::Value = serde_json::from_str(r#"{"messages":[{"content":"Count to 3","role":"user"}],"model":"gpt-4","stream":true}"#).unwrap(); let req = serde_json::from_value(req_json).unwrap(); let client = liter_llm::create_client(std::env::var("API_KEY").expect("API_KEY must be set"), None, None, None, None).unwrap(); let stream = client.chat_stream(req).await.expect("call failed"); let chunks: Vec<_> = tokio_stream::StreamExt::collect::<Vec<_>>(stream).await .into_iter() .map(|r| r.expect("stream item failed")) .collect(); println!("{:?}", chunks);}Streaming chat completion that produces content across multiple SSE chunks
package main
import ( "context" "fmt" pkg "github.com/xberg-io/liter-llm/packages/go/v2")
func ptr[T any](value T) *T { return &value }func main() { req := pkg.ChatCompletionRequest{ Model: `gpt-4`, Stream: ptr(true), } client, clientErr := pkg.CreateClient("your-api-key", nil, nil, nil, nil) if clientErr != nil { panic(clientErr) } defer client.Free() result, err := client.ChatStream(context.Background(), req) if err != nil { panic(err) } for resultChunk := range result.Chan() { fmt.Printf("%+v\n", resultChunk) } if err := result.Err(); err != nil { panic(err) }}Streaming chat completion that produces content across multiple SSE chunks
import io.xberg.literllm.*;
public final class Example { public static void main(String[] args) throws Exception { var reqJson = "{\"messages\":[{\"content\":\"Count to 3\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"stream\":true}"; var req = JsonUtil.fromJson(reqJson, ChatCompletionRequest.class); var apiKey = System.getenv("API_KEY"); if (apiKey == null || apiKey.isEmpty()) throw new IllegalStateException("API_KEY must be set"); try (var client = LiterLlm.createClient(apiKey, null, null, null, null)) { var result = client.chatStream(req); try (result) { result.forEach(resultChunk -> { System.out.println(resultChunk); }); } } }}Streaming chat completion that produces content across multiple SSE chunks
using System;using System.Text.Json;using LiterLlm;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var apiKey = Environment.GetEnvironmentVariable("API_KEY") ?? throw new InvalidOperationException("API_KEY must be set"); using var client = LiterLlmConverter.CreateClient(apiKey, null, null, null, null);await foreach (var chunk in client.ChatStreamAsync(new ChatCompletionRequest { Messages = new List<Message>() { JsonSerializer.Deserialize<Message>("{\"content\":\"Count to 3\",\"role\":\"user\"}", ConfigOptions)! }, Model = "gpt-4", Stream = true })) { Console.WriteLine(chunk); }Streaming chat completion that produces content across multiple SSE chunks
require "liter_llm"result = LiterLlm.chat_stream(LiterLlm::ChatCompletionRequest.new(messages: [{ 'content' => 'Count to 3', 'role' => 'user' }], model: 'gpt-4', stream: true)).to_aputs result.inspectStreaming chat completion that produces content across multiple SSE chunks
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Liter\Llm\LiterLlm;use Liter\Llm\ChatCompletionRequest;$req = \Liter\Llm\ChatCompletionRequest::from_json(json_encode(["messages" => [["content" => "Count to 3", "role" => "user"]], "model" => "gpt-4", "stream" => true]));$chunks = iterator_to_array(LiterLlm::chatStream($req));Streaming chat completion that produces content across multiple SSE chunks
api_key = System.fetch_env!("API_KEY"){:ok, client} = LiterLlm.create_client(api_key)result = LiterLlm.chat_stream(client, "{\"messages\":[{\"content\":\"Count to 3\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"stream\":true}") |> Enum.to_list()IO.inspect(result)Streaming chat completion that produces content across multiple SSE chunks
import { WasmChatCompletionChunk, WasmChatCompletionRequest, createClient } from "@xberg-io/liter-llm-wasm";async function main() { const req: WasmChatCompletionRequest = (() => { const _u0 = WasmChatCompletionRequest.default(); _u0.messages = [{ content: "Count to 3", role: "user" }]; _u0.model = "gpt-4"; _u0.stream = true; return _u0; })(); const client = createClient("your-api-key"); try { const result = await client.chatStream(req); try { while (true) { const resultChunk: WasmChatCompletionChunk | null = await result.next(); if (resultChunk === null) break; try { console.log(resultChunk); } finally { resultChunk.free(); } } } finally { result.free(); } } finally { client.free(); }}
void main();Each chunk contains choices[].delta.content with incremental text. The final chunk includes finish_reason: "stop".
Collecting the Full Response
Section titled “Collecting the Full Response”Accumulate deltas to get both real-time output and the complete text:
Streaming chat completion that includes a usage summary in the final chunk
import asyncioimport osfrom liter_llm import create_clientfrom liter_llm._internal_bindings import ChatCompletionRequest
async def main() -> None: client = create_client(api_key=os.environ["API_KEY"]) req = ChatCompletionRequest.from_json("{\"messages\":[{\"content\":\"Say hi\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"stream\":true,\"stream_options\":{\"include_usage\":true}}") result = client.chat_stream(req) chunks = [] async for chunk in result: chunks.append(chunk) for result_chunk in chunks: if result_chunk.usage is not None: print((result_chunk.usage.total_tokens if result_chunk.usage else None))
asyncio.run(main())Streaming chat completion that includes a usage summary in the final chunk
import { createClient } from "@xberg-io/liter-llm";async function main() { const client = createClient("your-api-key"); const result = await client.chatStream({ messages: [{ content: "Say hi", role: "user" }], model: "gpt-4", stream: true, streamOptions: { includeUsage: true } }); for await (const resultChunk of result) { console.log(resultChunk.usage?.totalTokens); }}
void main();Streaming chat completion that includes a usage summary in the final chunk
use liter_llm::BatchClient;use liter_llm::FileClient;use liter_llm::LlmClient;use liter_llm::ResponseClient;
#[tokio::main]async fn main() { let req_json: serde_json::Value = serde_json::from_str(r#"{"messages":[{"content":"Say hi","role":"user"}],"model":"gpt-4","stream":true,"stream_options":{"include_usage":true}}"#).unwrap(); let req = serde_json::from_value(req_json).unwrap(); let client = liter_llm::create_client(std::env::var("API_KEY").expect("API_KEY must be set"), None, None, None, None).unwrap(); let stream = client.chat_stream(req).await.expect("call failed"); let chunks: Vec<_> = tokio_stream::StreamExt::collect::<Vec<_>>(stream).await .into_iter() .map(|r| r.expect("stream item failed")) .collect(); println!("{:?}", chunks);}Streaming chat completion that includes a usage summary in the final chunk
package main
import ( "context" "fmt" pkg "github.com/xberg-io/liter-llm/packages/go/v2")
func ptr[T any](value T) *T { return &value }func main() { req := pkg.ChatCompletionRequest{ Model: `gpt-4`, Stream: ptr(true), StreamOptions: &pkg.StreamOptions{ IncludeUsage: ptr(true), }, } client, clientErr := pkg.CreateClient("your-api-key", nil, nil, nil, nil) if clientErr != nil { panic(clientErr) } defer client.Free() result, err := client.ChatStream(context.Background(), req) if err != nil { panic(err) } for resultChunk := range result.Chan() { if resultChunk.Usage != nil { fmt.Printf("%+v\n", resultChunk.Usage.TotalTokens) } } if err := result.Err(); err != nil { panic(err) }}Streaming chat completion that includes a usage summary in the final chunk
import io.xberg.literllm.*;
public final class Example { public static void main(String[] args) throws Exception { var reqJson = "{\"messages\":[{\"content\":\"Say hi\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"stream\":true,\"stream_options\":{\"include_usage\":true}}"; var req = JsonUtil.fromJson(reqJson, ChatCompletionRequest.class); var apiKey = System.getenv("API_KEY"); if (apiKey == null || apiKey.isEmpty()) throw new IllegalStateException("API_KEY must be set"); try (var client = LiterLlm.createClient(apiKey, null, null, null, null)) { var result = client.chatStream(req); try (result) { result.forEach(resultChunk -> { if (resultChunk.usage() != null) { System.out.println(resultChunk.usage().totalTokens()); } }); } } }}Streaming chat completion that includes a usage summary in the final chunk
using System;using System.Text.Json;using LiterLlm;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var apiKey = Environment.GetEnvironmentVariable("API_KEY") ?? throw new InvalidOperationException("API_KEY must be set"); using var client = LiterLlmConverter.CreateClient(apiKey, null, null, null, null);await foreach (var chunk in client.ChatStreamAsync(new ChatCompletionRequest { Messages = new List<Message>() { JsonSerializer.Deserialize<Message>("{\"content\":\"Say hi\",\"role\":\"user\"}", ConfigOptions)! }, Model = "gpt-4", Stream = true, StreamOptions = new StreamOptions { IncludeUsage = true } })) { Console.WriteLine(chunk); }Streaming chat completion that includes a usage summary in the final chunk
require "liter_llm"result = LiterLlm.chat_stream(LiterLlm::ChatCompletionRequest.new(messages: [{ 'content' => 'Say hi', 'role' => 'user' }], model: 'gpt-4', stream: true, stream_options: { 'include_usage' => true })).to_aputs result.usage.total_tokens.inspectStreaming chat completion that includes a usage summary in the final chunk
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Liter\Llm\LiterLlm;use Liter\Llm\ChatCompletionRequest;use Liter\Llm\Usage;$req = \Liter\Llm\ChatCompletionRequest::from_json(json_encode(["messages" => [["content" => "Say hi", "role" => "user"]], "model" => "gpt-4", "stream" => true, "streamOptions" => ["includeUsage" => true]]));$chunks = iterator_to_array(LiterLlm::chatStream($req));Streaming chat completion that includes a usage summary in the final chunk
api_key = System.fetch_env!("API_KEY"){:ok, client} = LiterLlm.create_client(api_key)result = LiterLlm.chat_stream(client, "{\"messages\":[{\"content\":\"Say hi\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"stream\":true,\"stream_options\":{\"include_usage\":true}}") |> Enum.to_list()IO.inspect(result.usage.total_tokens)Streaming chat completion that includes a usage summary in the final chunk
import { WasmChatCompletionChunk, WasmChatCompletionRequest, WasmStreamOptions, createClient } from "@xberg-io/liter-llm-wasm";async function main() { const req: WasmChatCompletionRequest = (() => { const _u0 = WasmChatCompletionRequest.default(); _u0.messages = [{ content: "Say hi", role: "user" }]; _u0.model = "gpt-4"; _u0.stream = true; _u0.streamOptions = (() => { const _u1 = WasmStreamOptions.default(); _u1.includeUsage = true; return _u1; })(); return _u0; })(); const client = createClient("your-api-key"); try { const result = await client.chatStream(req); try { while (true) { const resultChunk: WasmChatCompletionChunk | null = await result.next(); if (resultChunk === null) break; try { const resultChunkShow0Resource0 = resultChunk.usage; try { console.log(resultChunkShow0Resource0?.totalTokens); } finally { resultChunkShow0Resource0?.free(); } } finally { resultChunk.free(); } } } finally { result.free(); } } finally { client.free(); }}
void main();Tool Calling
Section titled “Tool Calling”Define tools as JSON schema functions. The model can request tool calls, which you execute and return results for:
Chat request with a tool definition; assistant responds with a tool call
import asyncioimport osfrom liter_llm import create_clientfrom liter_llm._internal_bindings import ChatCompletionRequest
async def main() -> None: client = create_client(api_key=os.environ["API_KEY"]) req = ChatCompletionRequest.from_json("{\"messages\":[{\"content\":\"What is the weather in San Francisco?\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"tool_choice\":\"auto\",\"tools\":[{\"function\":{\"description\":\"Get the current weather for a given location\",\"name\":\"get_weather\",\"parameters\":{\"properties\":{\"location\":{\"description\":\"The city and state, e.g. San Francisco, CA\",\"type\":\"string\"},\"unit\":{\"description\":\"The temperature unit to use\",\"enum\":[\"celsius\",\"fahrenheit\"],\"type\":\"string\"}},\"required\":[\"location\"],\"type\":\"object\"}},\"type\":\"function\"}]}") result = await client.chat(req) print(result.choices) print(result.choices[0].message.tool_calls) print((result.choices[0].message.tool_calls[0].function.name if result.choices[0].message.tool_calls else None)) print(result.choices[0].finish_reason)
asyncio.run(main())Chat request with a tool definition; assistant responds with a tool call
import { ToolType, createClient } from "@xberg-io/liter-llm";async function main() { const client = createClient("your-api-key"); const result = await client.chat({ messages: [{ content: "What is the weather in San Francisco?", role: "user" }], model: "gpt-4", toolChoice: "auto", tools: [{ function: { description: "Get the current weather for a given location", name: "get_weather", parameters: { properties: { location: { description: "The city and state, e.g. San Francisco, CA", type: "string" }, unit: { description: "The temperature unit to use", enum: ["celsius", "fahrenheit"], type: "string" } }, required: ["location"], type: "object" } }, toolType: ToolType.Function }] }); console.log(result.choices); console.log(result.choices?.[0]?.message?.toolCalls); console.log(result.choices?.[0]?.message?.toolCalls?.[0]?.function.name); console.log(result.choices?.[0]?.finishReason);}
void main();Chat request with a tool definition; assistant responds with a tool call
use liter_llm::BatchClient;use liter_llm::FileClient;use liter_llm::LlmClient;use liter_llm::ResponseClient;
#[tokio::main]async fn main() { let req_json: serde_json::Value = serde_json::from_str(r#"{"messages":[{"content":"What is the weather in San Francisco?","role":"user"}],"model":"gpt-4","tool_choice":"auto","tools":[{"function":{"description":"Get the current weather for a given location","name":"get_weather","parameters":{"properties":{"location":{"description":"The city and state, e.g. San Francisco, CA","type":"string"},"unit":{"description":"The temperature unit to use","enum":["celsius","fahrenheit"],"type":"string"}},"required":["location"],"type":"object"}},"type":"function"}]}"#).unwrap(); let req = serde_json::from_value(req_json).unwrap(); let client = liter_llm::create_client(std::env::var("API_KEY").expect("API_KEY must be set"), None, None, None, None).unwrap(); let result = client.chat(req).await.expect("call failed"); println!("{:?}", result.choices); println!("{:?}", result.choices[0].message.tool_calls); println!("{:?}", result.choices[0].message.tool_calls.as_ref().unwrap()[0].function.name); println!("{:?}", result.choices[0].finish_reason);}Chat request with a tool definition; assistant responds with a tool call
package main
import ( "fmt" pkg "github.com/xberg-io/liter-llm/packages/go/v2")
func ptr[T any](value T) *T { return &value }func main() { req := pkg.ChatCompletionRequest{ Model: `gpt-4`, ToolChoice: ptr(pkg.ToolChoice{Mode: ptr(pkg.ToolChoiceModeAuto)}), } client, clientErr := pkg.CreateClient("your-api-key", nil, nil, nil, nil) if clientErr != nil { panic(clientErr) } defer client.Free() result, err := client.Chat(req) if err != nil { panic(err) } fmt.Printf("%+v\n", result.Choices) fmt.Printf("%+v\n", result.Choices[0].Message.ToolCalls) fmt.Printf("%+v\n", result.Choices[0].Message.ToolCalls[0].Function.Name) fmt.Printf("%+v\n", result.Choices[0].FinishReason)}Chat request with a tool definition; assistant responds with a tool call
import io.xberg.literllm.*;
public final class Example { public static void main(String[] args) throws Exception { var reqJson = "{\"messages\":[{\"content\":\"What is the weather in San Francisco?\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"tool_choice\":\"auto\",\"tools\":[{\"function\":{\"description\":\"Get the current weather for a given location\",\"name\":\"get_weather\",\"parameters\":{\"properties\":{\"location\":{\"description\":\"The city and state, e.g. San Francisco, CA\",\"type\":\"string\"},\"unit\":{\"description\":\"The temperature unit to use\",\"enum\":[\"celsius\",\"fahrenheit\"],\"type\":\"string\"}},\"required\":[\"location\"],\"type\":\"object\"}},\"type\":\"function\"}]}"; var req = JsonUtil.fromJson(reqJson, ChatCompletionRequest.class); var apiKey = System.getenv("API_KEY"); if (apiKey == null || apiKey.isEmpty()) throw new IllegalStateException("API_KEY must be set"); try (var client = LiterLlm.createClient(apiKey, null, null, null, null)) { var result = client.chat(req); System.out.println(result.choices()); System.out.println(result.choices().get(0).message().toolCalls()); System.out.println(result.choices().get(0).message().toolCalls().get(0).function().name()); System.out.println(result.choices().get(0).finishReason()); } }}Chat request with a tool definition; assistant responds with a tool call
using System;using System.Text.Json;using LiterLlm;
var ConfigOptions = new JsonSerializerOptions { PropertyNameCaseInsensitive = true };var apiKey = Environment.GetEnvironmentVariable("API_KEY") ?? throw new InvalidOperationException("API_KEY must be set"); using var client = LiterLlmConverter.CreateClient(apiKey, null, null, null, null);var result = await client.ChatAsync(new ChatCompletionRequest { Messages = new List<Message>() { JsonSerializer.Deserialize<Message>("{\"content\":\"What is the weather in San Francisco?\",\"role\":\"user\"}", ConfigOptions)! }, Model = "gpt-4", ToolChoice = JsonSerializer.Deserialize<ToolChoice>("\"auto\"", ConfigOptions)!, Tools = new List<ChatCompletionTool>() { JsonSerializer.Deserialize<ChatCompletionTool>("{\"function\":{\"description\":\"Get the current weather for a given location\",\"name\":\"get_weather\",\"parameters\":{\"properties\":{\"location\":{\"description\":\"The city and state, e.g. San Francisco, CA\",\"type\":\"string\"},\"unit\":{\"description\":\"The temperature unit to use\",\"enum\":[\"celsius\",\"fahrenheit\"],\"type\":\"string\"}},\"required\":[\"location\"],\"type\":\"object\"}},\"type\":\"function\"}", ConfigOptions)! } });Console.WriteLine(result.Choices);Console.WriteLine(result.Choices[0].Message.ToolCalls);Console.WriteLine(result.Choices[0].Message.ToolCalls![0].Function.Name);Console.WriteLine(result.Choices[0].FinishReason);Chat request with a tool definition; assistant responds with a tool call
require "liter_llm"result = LiterLlm.chat(LiterLlm::ChatCompletionRequest.new(messages: [{ 'content' => 'What is the weather in San Francisco?', 'role' => 'user' }], model: 'gpt-4', tool_choice: 'auto', tools: [{ 'function' => { 'description' => 'Get the current weather for a given location', 'name' => 'get_weather', 'parameters' => { 'properties' => { 'location' => { 'description' => 'The city and state, e.g. San Francisco, CA', 'type' => 'string' }, 'unit' => { 'description' => 'The temperature unit to use', 'enum' => ['celsius', 'fahrenheit'], 'type' => 'string' } }, 'required' => ['location'], 'type' => 'object' } }, 'type' => 'function' }]))puts result.choices.inspectputs result.choices[0].message.tool_calls.inspectputs result.choices[0].message.tool_calls[0].function.name.inspectputs result.choices[0].finish_reason.inspectChat request with a tool definition; assistant responds with a tool call
<?php
declare(strict_types=1);
require_once __DIR__ . '/vendor/autoload.php';
use Liter\Llm\LiterLlm;use Liter\Llm\ChatCompletionRequest;use Liter\Llm\Choice;$req = \Liter\Llm\ChatCompletionRequest::from_json(json_encode(["messages" => [["content" => "What is the weather in San Francisco?", "role" => "user"]], "model" => "gpt-4", "toolChoice" => "auto", "tools" => [["function" => ["description" => "Get the current weather for a given location", "name" => "get_weather", "parameters" => ["properties" => ["location" => ["description" => "The city and state, e.g. San Francisco, CA", "type" => "string"], "unit" => ["description" => "The temperature unit to use", "enum" => ["celsius", "fahrenheit"], "type" => "string"]], "required" => ["location"], "type" => "object"]], "toolType" => "function"]]]));$result = LiterLlm::chat($req);var_dump($result->getChoices());var_dump($result->getChoices()[0]->getMessage()->getToolCalls());var_dump($result->getChoices()[0]->getMessage()->getToolCalls()[0]->getFunction()->name);var_dump($result->getChoices()[0]->finishReason);Chat request with a tool definition; assistant responds with a tool call
api_key = System.fetch_env!("API_KEY"){:ok, client} = LiterLlm.create_client(api_key)result = LiterLlm.defaultclient_chat_async(client, "{\"messages\":[{\"content\":\"What is the weather in San Francisco?\",\"role\":\"user\"}],\"model\":\"gpt-4\",\"tool_choice\":\"auto\",\"tools\":[{\"function\":{\"description\":\"Get the current weather for a given location\",\"name\":\"get_weather\",\"parameters\":{\"properties\":{\"location\":{\"description\":\"The city and state, e.g. San Francisco, CA\",\"type\":\"string\"},\"unit\":{\"description\":\"The temperature unit to use\",\"enum\":[\"celsius\",\"fahrenheit\"],\"type\":\"string\"}},\"required\":[\"location\"],\"type\":\"object\"}},\"type\":\"function\"}]}")IO.inspect(result.choices)IO.inspect(Enum.at(result.choices, 0).message.tool_calls)IO.inspect(Enum.at(Enum.at(result.choices, 0).message.tool_calls, 0).function.name)IO.inspect(Enum.at(result.choices, 0).finish_reason)Chat request with a tool definition; assistant responds with a tool call
import { WasmChatCompletionRequest, WasmChatCompletionTool, WasmFunctionDefinition, WasmToolType, createClient } from "@xberg-io/liter-llm-wasm";async function main() { const req: WasmChatCompletionRequest = (() => { const _u0 = WasmChatCompletionRequest.default(); _u0.messages = [{ content: "What is the weather in San Francisco?", role: "user" }]; _u0.model = "gpt-4"; _u0.toolChoice = "auto"; _u0.tools = [(() => { const _u1 = WasmChatCompletionTool.default(); _u1.function = (() => { const _u2 = WasmFunctionDefinition.default(); _u2.description = "Get the current weather for a given location"; _u2.name = "get_weather"; _u2.parameters = { properties: { location: { description: "The city and state, e.g. San Francisco, CA", type: "string" }, unit: { description: "The temperature unit to use", enum: ["celsius", "fahrenheit"], type: "string" } }, required: ["location"], type: "object" }; return _u2; })(); _u1.toolType = WasmToolType.Function; return _u1; })()]; return _u0; })(); const client = createClient("your-api-key"); try { const result = await client.chat(req); console.log(result.choices); console.log(result.choices[0].message.toolCalls); console.log(result.choices[0].message.toolCalls?.[0]?.function.name); console.log(result.choices[0].finishReason); } finally { client.free(); }}
void main();Chat Parameters
Section titled “Chat Parameters”All chat parameters work with both chat and chat_stream:
| Parameter | Type | Description |
|---|---|---|
model |
string | Provider/model identifier (e.g. "openai/gpt-4o") |
messages |
array | Conversation messages |
temperature |
float | Sampling temperature (0.0-2.0) |
max_tokens |
int | Maximum tokens to generate |
top_p |
float | Nucleus sampling threshold |
n |
int | Number of completions to generate |
stop |
string/array | Stop sequences |
tools |
array | Tool/function definitions |
tool_choice |
string/object | Tool selection strategy |
response_format |
object | Force JSON output ({"type": "json_object"}) |
seed |
int | Deterministic sampling seed |
presence_penalty |
float | Penalize new topics (-2.0 to 2.0) |
frequency_penalty |
float | Penalize repetition (-2.0 to 2.0) |
reasoning_effort |
string | Reasoning budget for o-series and extended-thinking models. |
extra_body |
object | Provider-specific fields passed through verbatim. |
Reasoning Effort
Section titled “Reasoning Effort”OpenAI o-series models and Anthropic extended-thinking models accept a reasoning_effort parameter that controls how much compute the model spends on internal reasoning before producing the final response.
response = client.chat({ "model": "openai/o3-mini", "messages": [{"role": "user", "content": "Prove the Pythagorean theorem."}], "reasoning_effort": "high",})const response = await client.chat({ model: "openai/o3-mini", messages: [{ role: "user", content: "Prove the Pythagorean theorem." }], reasoningEffort: "high",});use liter_llm::types::ReasoningEffort;
let req = ChatCompletionRequest { model: "openai/o3-mini".into(), messages: vec![/* ... */], reasoning_effort: Some(ReasoningEffort::High), ..Default::default()};var req llm.ChatCompletionRequestif err := json.Unmarshal([]byte(`{ "model": "openai/o3-mini", "messages": [{"role": "user", "content": "Think step by step."}], "reasoning_effort": "high"}`), &req); err != nil { return err}
resp, err := client.Chat(req)Accepted values for OpenAI o-series: "low", "medium", "high". Anthropic extended thinking uses a budget_tokens integer instead, which maps to reasoning_effort when the binding converts the field.
Structured Outputs (JSON Schema)
Section titled “Structured Outputs (JSON Schema)”Pass a JSON Schema to response_format to constrain the model output to a specific structure. Use "type": "json_schema" instead of "type": "json_object" for schema-validated output.
schema = { "type": "object", "properties": { "name": {"type": "string"}, "age": {"type": "integer"}, }, "required": ["name", "age"], "additionalProperties": False,}
response = client.chat({ "model": "openai/gpt-4o", "messages": [{"role": "user", "content": "Extract: Alice is 30 years old."}], "response_format": { "type": "json_schema", "json_schema": { "name": "person", "strict": True, "schema": schema, }, },})const response = await client.chat({ model: "openai/gpt-4o", messages: [{ role: "user", content: "Extract: Alice is 30 years old." }], responseFormat: { type: "json_schema", jsonSchema: { name: "person", strict: true, schema: { type: "object", properties: { name: { type: "string" }, age: { type: "integer" }, }, required: ["name", "age"], additionalProperties: false, }, }, },});use serde_json::json;
let req = ChatCompletionRequest { model: "openai/gpt-4o".into(), messages: vec![/* ... */], response_format: Some(json!({ "type": "json_schema", "json_schema": { "name": "person", "strict": true, "schema": { "type": "object", "properties": { "name": { "type": "string" }, "age": { "type": "integer" } }, "required": ["name", "age"], "additionalProperties": false } } })), ..Default::default()};Structured output availability depends on provider support. OpenAI gpt-4o and later support json_schema. Providers that do not support it fall back to json_object or return EndpointNotSupported.
Extra_body
Section titled “Extra_body”Pass provider-specific parameters that liter-llm does not model natively via extra_body. Fields in extra_body are merged into the top-level request JSON before it is sent to the provider.
response = client.chat({ "model": "openai/gpt-4o", "messages": [{"role": "user", "content": "Hello"}], "extra_body": { "store": True, # OpenAI conversation store "metadata": {"user": "alice"}, },})const response = await client.chat({ model: "openai/gpt-4o", messages: [{ role: "user", content: "Hello" }], extraBody: { store: true, metadata: { user: "alice" }, },});use serde_json::json;
let req = ChatCompletionRequest { model: "openai/gpt-4o".into(), messages: vec![/* ... */], extra_body: Some(json!({ "store": true, "metadata": { "user": "alice" } })), ..Default::default()};extra_body fields take lower precedence than named fields. If a named field and an extra_body key conflict, the named field wins.
Audio Content Parts
Section titled “Audio Content Parts”Send audio inline in a user message using the input_audio content part type. The audio must be base64-encoded.
import base64
with open("audio.wav", "rb") as f: audio_b64 = base64.b64encode(f.read()).decode()
response = client.chat({ "model": "openai/gpt-4o-audio-preview", "messages": [{ "role": "user", "content": [ { "type": "input_audio", "input_audio": { "data": audio_b64, "format": "wav", }, }, {"type": "text", "text": "Transcribe and summarize this audio."}, ], }],})import { readFileSync } from "fs";
const audioB64 = readFileSync("audio.wav").toString("base64");
const response = await client.chat({ model: "openai/gpt-4o-audio-preview", messages: [{ role: "user", content: [ { type: "input_audio", inputAudio: { data: audioB64, format: "wav" }, }, { type: "text", text: "Transcribe and summarize this audio." }, ], }],});use base64::{Engine, engine::general_purpose::STANDARD};use liter_llm::types::{AudioContent, ContentPart};
let audio_bytes = std::fs::read("audio.wav")?;let audio_b64 = STANDARD.encode(&audio_bytes);
let content = vec![ ContentPart::InputAudio { input_audio: AudioContent { data: audio_b64, format: "wav".into(), }, }, ContentPart::Text { text: "Transcribe and summarize this audio.".into() },];Supported formats depend on the provider. OpenAI gpt-4o-audio-preview accepts wav, mp3, ogg, flac, m4a.
AWS EventStream Streaming
Section titled “AWS EventStream Streaming”When routing to Bedrock providers, responses arrive in AWS EventStream framing rather than SSE. Liter-llm handles the framing transparently. chat_stream works the same way regardless of provider.
// EventStream framing is transparent to the caller.let stream = client.chat_stream(ChatCompletionRequest { model: "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0".into(), messages: vec![/* ... */], ..Default::default()}).await?;
// Consume exactly like any other stream.pin_mut!(stream);while let Some(chunk) = stream.next().await { let chunk = chunk?; if let Some(content) = chunk.choices[0].delta.content.as_deref() { print!("{content}"); }}# EventStream framing is transparent to the caller.for chunk in client.chat_stream({ "model": "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0", "messages": [{"role": "user", "content": "Hello"}],}): print(chunk["choices"][0]["delta"].get("content", ""), end="", flush=True)