About trymirai/uzu
trymirai/uzu is an open-source project on GitHub, mainly written in Rust. A high-performance inference engine for AI models It currently holds 2,146 stars and 116 forks with 22 open issues, and was last pushed on 2026-10-10 (repository created 2025-06-23).
Project Overview
Git Homed tracks it on the Today's Trending board.
GitHub Repository Details
README
uzu
A high-performance inference engine for AI models. It allows you to deploy AI directly in your app with zero latency, full data privacy, and no inference costs. Key features:
- Simple, high-level API
- Unified model configurations, making it easy to add support for new models
- Traceable computations to ensure correctness against the source-of-truth implementation
- Utilizes unified memory on Apple devices
- Broad model support
Quick Start
Rust
Add the dependency:
[dependencies]
uzu = { git = "https://github.com/trymirai/uzu", branch = "main", package = "uzu" }
Run the code below:
use std::io::{self, Write};
use uzu::{
engine::{Engine, EngineConfig},
types::session::chat::{ChatConfig, ChatMessage, ChatReplyConfig},
};
[tokio::main]
async fn main() -> Result<(), Box> {
let engine_config = EngineConfig::default();
let engine = Engine::new(engine_config).await?;
let model = engine.model("alibaba:qwen3.5:0.8b:mirai:mirai-m:4".to_string()).await?.ok_or("Model not found")?;
let downloader = engine.download(&model).await?;
while let Some(update) = downloader.next().await {
print!("\r\u{001B}[2KDownload progress: {:.2}%", update.progress() * 100.0);
io::stdout().flush()?;
}
println!();
let session = engine.chat(model, ChatConfig::default()).await?;
let messages = vec![
ChatMessage::system().with_text("You are a helpful assistant".to_string()),
ChatMessage::user().with_text("Tell me a short, funny story about a robot".to_string()),
];
let replies = session.reply(messages, ChatReplyConfig::default()).await?;
if let Some(reply) = replies.last() {
println!("Reasoning: {}", reply.message.reasoning().unwrap_or_default());
println!("Text: {}", reply.message.text().unwrap_or_default());
}
Ok(())
}
Python
Add the dependency:
uv add uzu==0.6.3
Run the code below:
import asyncio
from uzu import ChatConfig, ChatMessage, ChatReplyConfig, Engine, EngineConfig
async def main() -> None:
engine_config = EngineConfig.create()
engine = await Engine.create(engine_config)
model = await engine.model("alibaba:qwen3.5:0.8b:mirai:mirai-m:4")
if model is None:
return
async for update in (await engine.download(model)).iterator():
print(f"\rDownload progress: {update.progress:.2%}", end="", flush=True)
print()
session = await engine.chat(model, ChatConfig.create())
messages = [
ChatMessage.system().with_text("You are a helpful assistant"),
ChatMessage.user().with_text("Tell me a short, funny story about a robot"),
]
replies = await session.reply(messages, ChatReplyConfig.create())
if not replies:
return
message = replies[-1].message
print(f"Reasoning: {message.reasoning}")
print(f"Text: {message.text}")
if __name__ == "__main__":
asyncio.run(main())
Swift
Add the dependency:
dependencies: [
.package(url: "https://github.com/trymirai/uzu.git", from: "0.6.3")
]
Run the code below:
import Foundation
import Uzu
public func runQuickStart() async throws {
let engineConfig = EngineConfig.create()
let engine = try await Engine.create(config: engineConfig)
guard let model = try await engine.model(identifier: "alibaba:qwen3.5:0.8b:mirai:mirai-m:4") else {
return
}
for try await update in try await engine.download(model: model).iterator() {
print(String(format: "\r\u{001B}[2KDownload progress: %.2f%%", update.progress() * 100), terminator: "")
fflush(stdout)
}
print()
let session = try await engine.chat(model: model, config: .create())
let messages = [
ChatMessage.system().withText(text: "You are a helpful assistant"),
ChatMessage.user().withText(text: "Tell me a short, funny story about a robot")
]
let reply = try await session.reply(input: messages, config: .create())
guard let message = reply.last?.message else {
return
}
print("Reasoning: \(message.reasoning() ?? "empty")")
print("Text: \(message.text() ?? "empty")")
}
TypeScript
Add the dependency:
pnpm add @trymirai/[email protected]
Run the code below:
import { ChatConfig, ChatMessage, ChatReplyConfig, Engine, EngineConfig } from '@trymirai/uzu';
async function main() {
let engineConfig = EngineConfig.create();
let engine = await Engine.create(engineConfig);
let model = await engine.model('alibaba:qwen3.5:0.8b:mirai:mirai-m:4');
if (!model) {
throw new Error('Model not found');
}
for await (const update of await engine.download(model)) {
process.stdout.write(\rDownload progress: ${(update.progress * 100).toFixed(2)}%);
}
console.log();
let session = await engine.chat(model, ChatConfig.create());
let messages = [
ChatMessage.system().withText('You are a helpful assistant'),
ChatMessage.user().withText('Tell me a short, funny story about a robot')
];
let reply = await session.reply(messages, ChatReplyConfig.create());
let message = reply[0]?.message;
if (message) {
console.log('Reasoning: ', message.reasoning);
console.log('Text: ', message.text);
}
}
main().catch((error) => {
console.error(error);
});
Everything from model downloading to inference configuration is handled automatically. Refer to the documentation for details on how to customize each step of the process.
Examples
You can run any example via cargo tools example \<rust | python | swift | typescript\> \<chat | chat-cloud | chat-shared-instance | chat-structured-output | quick-start | tool-calls\>:
Chat
In this example, we will download a model and get a reply to a specific list of messages:
Rust
use std::io::{self, Write};
use uzu::{
engine::{Engine, EngineConfig},
session::chat::ChatSessionStreamChunk,
types::session::chat::{ChatConfig, ChatMessage, ChatReplyConfig},
};
[tokio::main]
async fn main() -> Result<(), Box> {
let engine_config = EngineConfig::default();
let engine = Engine::new(engine_config).await?;
let model = engine.model("alibaba:qwen3.5:0.8b:mirai:mirai-m:4".to_string()).await?.ok_or("Model not found")?;
let downloader = engine.download(&model).await?;
while let Some(update) = downloader.next().await {
print!("\r\u{001B}[2KDownload progress: {:.2}%", update.progress() * 100.0);
io::stdout().flush()?;
}
println!();
let messages = vec![
ChatMessage::system().with_text("You are a helpful assistant".to_string()),
ChatMessage::user().with_text("Tell me a short, funny story about a robot".to_string()),
];
let session = engine.chat(model, ChatConfig::default()).await?;
let stream = session.reply_with_stream(messages, ChatReplyConfig::default()).await;
let mut last_message: Option = None;
while let Some(chunk) = stream.next().await {
match chunk {
ChatSessionStreamChunk::ToolResults {
..
} => {},
ChatSessionStreamChunk::Replies {
replies,
} => {
if let Some(reply) = replies.first() {
last_message = Some(reply.message.clone());
println!("Generated tokens: {}", reply.stats.tokens_count_output.unwrap_or_default());
}
},
ChatSessionStreamChunk::Error {
error,
} => {
println!("Error: {error}");
},
}
}
if let Some(message) = last_message {
println!("Reasoning: {}", message.reasoning().unwrap_or_default());
println!("Text: {}", message.text().unwrap_or_default());
}
Ok(())
}
Python
import asyncio
from uzu import (
ChatConfig,
ChatMessage,
ChatReplyConfig,
ChatSessionStreamChunk,
Engine,
EngineConfig,
)
async def main() -> None:
engine_config = EngineConfig.create()
engine = await Engine.create(engine_config)
model = await engine.model("alibaba:qwen3.5:0.8b:mirai:mirai-m:4")
if model is None:
raise RuntimeError("Model not found")
async for update in (await engine.download(model)).iterator():
print(f"\rDownload progress: {update.progress:.2%}", end="", flush=True)
print()
messages = [
ChatMessage.system().with_text("You are a helpful assistant"),
ChatMessage.user().with_text("Tell me a short, funny story about a robot"),
]
session = await engine.chat(model, ChatConfig.create())
stream = await session.reply_with_stream(messages, ChatReplyConfig.create())
message: ChatMessage | None = None
async for chunk in stream.iterator():
if isinstance(chunk, ChatSessionStreamChunk.Replies):
replies = chunk.replies
if replies:
reply = replies[0]
message = reply.message
print(f"Generated tokens: {reply.stats.tokens_count_output}")
elif isinstance(chunk, ChatSessionStreamChunk.Error):
print(f"Error: {chunk.error}")
if message is not None:
print(f"Reasoning: {message.reasoning}")
print(f"Text: {message.text}")
if __name__ == "__main__":
asyncio.run(main())
Swift
import Foundation
import Uzu
public func runChat() async throws {
let engineConfig = EngineConfig.create()
let engine = try await Engine.create(config: engineConfig)
guard let model = try await engine.model(identifier: "alibaba:qwen3.5:0.8b:mirai:mirai-m:4") else {
return
}
for try await update in try await engine.download(model: model).iterator() {
print(String(format: "\r\u{001B}[2KDownload progress: %.2f%%", update.progress() * 100), terminator: "")
fflush(stdout)
}
print()
let messages = [
ChatMessage.system().withText(text: "You are a helpful assistant"),
ChatMessage.user().withText(text: "Tell me a short, funny story about a robot")
]
let session = try await engine.chat(model: model, config: .create())
let stream = await session.replyWithStream(input: messages, config: .create())
var message: ChatMessage? = nil
for try await update in stream.iterator() {
switch update {
case .replies(let replies):
let reply = replies.last
message = reply?.message
print("Generated tokens: \(reply?.stats.tokensCountOutput ?? 0)")
case .error(let error):
print("Error: \(error)")
case .toolResults:
break
}
}
print("Reasoning: \(message?.reasoning() ?? "empty")")
print("Text: \(message?.text() ?? "empty")")
}
TypeScript
import {
ChatConfig,
ChatMessage,
ChatReplyConfig,
ChatSessionStreamChunkError,
ChatSessionStreamChunkReplies,
Engine,
EngineConfig
} from '@trymirai/uzu';
async function main() {
let engineConfig = EngineConfig.create();
let engine = await Engine.create(engineConfig);
let model = await engine.model('alibaba:qwen3.5:0.8b:mirai:mirai-m:4');
if (!model) {
throw new Error('Model not found');
}
for await (const update of await engine.download(model)) {
process.stdout.write(\rDownload progress: ${(update.progress * 100).toFixed(2)}%);
}
console.log();
let messages = [
ChatMessage.system().withText('You are a helpful assistant'),
ChatMessage.user().withText('Tell me a short, funny story about a robot')
];
let session = await engine.chat(model, ChatConfig.create());
let stream = await session.replyWithStream(messages, ChatReplyConfig.create());
let message: ChatMessage | undefined;
for await (const chunk of stream) {
if (chunk instanceof ChatSessionStreamChunkReplies) {
message = chunk.replies[0]?.message;
console.log('Generated tokens: ', chunk.replies[0]?.stats.tokensCountOutput);
} else if (chunk instanceof ChatSessionStreamChunkError) {
console.error('Error: ', chunk.error);
}
}
console.log('Reasoning: ', message?.reasoning);
console.log('Text: ', message?.text);
}
main().catch((error) => {
console.error(error);
});
Once loaded, the same ChatSession can be reused for multiple requests until you drop it. Each model may consume a significant amount of RAM, so it's important to keep only one session loaded at a time. For iOS apps, we recommend adding the Increased Memory Capability entitlement to ensure your app can allocate the required memory.
Chat with the cloud model
In this example, we will get a reply to a specific list of messages from a cloud model:
Rust
use uzu::{
engine::{Engine, EngineConfig},
types::{
basic::ReasoningEffort,
session::chat::{ChatConfig, ChatMessage, ChatReplyConfig},
},
};
[tokio::main]
async fn main() -> Result<(), Box> {
let engine_config = EngineConfig::default().with_openai_api_key("OPENAI_API_KEY".to_string());
let engine = Engine::new(engine_config).await?;
let model = engine.model("gpt-5".to_string()).await?.ok_or("Model not found")?;
let messages = vec![
ChatMessage::system().with_reasoning_effort(ReasoningEffort::Low),
ChatMessage::user().with_text("How LLMs work".to_string()),
];
let session = engine.chat(model, ChatConfig::default()).await?;
let replies = session.reply(messages, ChatReplyConfig::default()).await?;
if let Some(reply) = replies.first() {
println!("Reasoning: {}", reply.message.reasoning().unwrap_or_default());
println!("Text: {}", reply.message.text().unwrap_or_default());
}
Ok(())
}
Python
import asyncio
from uzu import ChatConfig, ChatMessage, ChatReplyConfig, Engine, EngineConfig, ReasoningEffort
async def main() -> None:
engine_config = EngineConfig.create().with_openai_api_key("OPENAI_API_KEY")
engine = await Engine.create(engine_config)
model = await engine.model("gpt-5")
if model is None:
raise RuntimeError("Model not found")
messages = [
ChatMessage.system().with_reasoning_effort(ReasoningEffort.Low),
ChatMessage.user().with_text("How LLMs work"),
]
session = await engine.chat(model, ChatConfig.create())
replies = await session.reply(messages, ChatReplyConfig.create())
if replies:
message = replies[0].message
print(f"Reasoning: {message.reasoning}")
print(f"Text: {message.text}")
if __name__ == "__main__":
asyncio.run(main())
Swift
import Uzu
public func runChatCloud() async throws {
let engineConfig = EngineConfig.create().withOpenaiApiKey(openaiApiKey: "OPENAI_API_KEY")
let engine = try await Engine.create(config: engineConfig)
guard let model = try await engine.model(identifier: "gpt-5") else {
return
}
let messages = [
ChatMessage.system().withReasoningEffort(reasoningEffort: .low),
ChatMessage.user().withText(text: "How LLMs work")
]
let session = try await engine.chat(model: model, config: .create())
let reply = try await session.reply(input: messages, config: .create())
guard let message = reply.last?.message else {
return
}
print("Reasoning: \(message.reasoning() ?? "empty")")
print("Text: \(message.text() ?? "empty")")
}
TypeScript
import { ChatConfig, ChatMessage, ChatReplyConfig, Engine, EngineConfig, ReasoningEffort } from '@trymirai/uzu';
async function main() {
let engineConfig = EngineConfig.create().withOpenaiApiKey('OPENAI_API_KEY');
let engine = await Engine.create(engineConfig);
let model = await engine.model('gpt-5');
if (!model) {
throw new Error('Model not found');
}
let messages = [
ChatMessage.system().withReasoningEffort("Low" as ReasoningEffort),
ChatMessage.user().withText('How LLMs work')
];
let session = await engine.chat(model, ChatConfig.create());
let reply = await session.reply(messages, ChatReplyConfig.create());
let message = reply[0]?.message;
if (message) {
console.log('Reasoning: ', message.reasoning);
console.log('Text: ', message.text);
}
}
main().catch((error) => {
console.error(error);
});
Chat with shared instance
This example shows how to reuse chat instance without reloading model into memory:
Rust
use std::io::{self, Write};
use uzu::{
engine::{Engine, EngineConfig},
types::session::chat::{ChatConfig, ChatMessage, ChatReplyConfig},
};
[tokio::main]
async fn main() -> Result<(), Box> {
let engine_config = EngineConfig::default();
let engine = Engine::new(engine_config).await?;
let model = engine.model("alibaba:qwen3.5:0.8b:mirai:mirai-m:4".to_string()).await?.ok_or("Model not found")?;
let downloader = engine.download(&model).await?;
while let Some(update) = downloader.next().await {
print!("\r\u{001B}[2KDownload progress: {:.2}%", update.progress() * 100.0);
io::stdout().flush()?;
}
println!();
// The chat_instance owns the loaded model and can be shared between sessions
let chat_instance = engine.chat_instance(model, ChatConfig::default()).await?;
let first_session = engine.chat_with_instance(&chat_instance).await?;
let replies = first_session
.reply(
vec![ChatMessage::user().with_text("Tell me a short, funny story about a robot".to_string())],
ChatReplyConfig::default(),
)
.await?;
if let Some(reply) = replies.last() {
println!("First session reasoning: {}", reply.message.reasoning().unwrap_or_default());
println!("First session text: {}", reply.message.text().unwrap_or_default());
}
// The second session reuses the already-loaded weights instead of loading the model again
let second_session = engine.chat_with_instance(&chat_instance).await?;
let replies = second_session
.reply(
vec![ChatMessage::user().with_text("What is the capital of France?".to_string())],
ChatReplyConfig::default(),
)
.await?;
if let Some(reply) = replies.last() {
println!("\nSecond session reasoning: {}", reply.message.reasoning().unwrap_or_default());
println!("Second session text: {}", reply.message.text().unwrap_or_default());
}
Ok(())
}
Python
import asyncio
from uzu import ChatConfig, ChatMessage, ChatReplyConfig, Engine, EngineConfig
async def main() -> None:
engine_config = EngineConfig.create()
engine = await Engine.create(engine_config)
model = await engine.model("alibaba:qwen3.5:0.8b:mirai:mirai-m:4")
if model is None:
raise RuntimeError("Model not found")
async for update in (await engine.download(model)).iterator():
print(f"\rDownload progress: {update.progress:.2%}", end="", flush=True)
print()
# The chat_instance owns the loaded model and can be shared between sessions.
chat_instance = await engine.chat_instance(model, ChatConfig.create())
first_session = await engine.chat_with_instance(chat_instance)
replies = await first_session.reply(
[ChatMessage.user().with_text("Tell me a short, funny story about a robot")],
ChatReplyConfig.create(),
)
if replies:
message = replies[-1].message
print(f"First session reasoning: {message.reasoning}")
print(f"First session text: {message.text}")
# The second session reuses the already-loaded weights instead of loading the model again.
second_session = await engine.chat_with_instance(chat_instance)
replies = await second_session.reply(
[ChatMessage.user().with_text("What is the capital of France?")],
ChatReplyConfig.create(),
)
if replies:
message = replies[-1].message
print(f"\nSecond session reasoning: {message.reasoning}")
print(f"Second session text: {message.text}")
if __name__ == "__main__":
asyncio.run(main())
Swift
```swift import Foundation import Uzu
public func runChatSharedInstance() async throws { let engineConfig = EngineConfig.create() let engine = try await Engine.create(config: engineConfig)
guard let model = try await engine.model(identifier: "alibaba:qwen3.5:0.8b:mirai:mirai-m:4") else { return } for try await update in try await engine.download(model: model).iterator() { print(String(format: "\r\u{001B}[2KDownload progress: %.2f%%", update.progress() * 100), terminator: "") fflush(stdout) } print()
// The chatInstance owns the loaded model and can be shared between sessions. let chatInstance = try await engine.chatInstance(model: model, config: .create())
let firstSession = try await engine.chatWithInstance(instance: chatInstance) le