Compare commits

..

20 Commits

Author SHA1 Message Date
Slavi Pantaleev
f4c698ad33 Release 1.10.0 2025-12-06 07:01:39 +02:00
Slavi Pantaleev
bd39001417 Update dependencies and matrix-sdk (0.14.0 -> 0.16.0) 2025-12-06 07:00:24 +02:00
Slavi Pantaleev
2692d0322e Release 1.9.0 2025-11-30 12:29:58 +02:00
Slavi Pantaleev
8eb70f0f2c Upgrade async-openai from our own etkecc fork to upstream's 0.31.1
Switches `async-openai` from our own etkecc fork (0.28.1-patched) to the
official crates.io version 0.31.1.

We adapt to async-openai's types reorganization and making use of crate
features to only enable what we need.
2025-11-30 11:48:56 +02:00
Slavi Pantaleev
5c0a7be7a2 Add changelog entry for v1.8.3 2025-11-28 14:42:48 +02:00
Slavi Pantaleev
0a8f9fc3e5 Release 1.8.3 2025-11-28 14:40:38 +02:00
Slavi Pantaleev
1ac3b2e060 Update dependencies 2025-11-28 14:23:07 +02:00
Slavi Pantaleev
a3ef9fd1bf Upgrade services 2025-11-28 14:19:59 +02:00
Slavi Pantaleev
df507eb201 Make use of the BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED environment variable for configuring config.user.encryption.recovery_reset_allowed 2025-11-28 14:18:35 +02:00
Slavi Pantaleev
4dcd9eff40 Make use of the BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY environment variable for configuring config.persistence.session_encryption_key
Seems like we already had a constant defined, but weren't making use of it.
2025-11-28 14:17:23 +02:00
Slavi Pantaleev
ea760ce755 Release 1.8.2 2025-11-20 06:00:52 +02:00
Slavi Pantaleev
1528df6a55 Upgrade Rust (1.90.0 -> 1.91.1) 2025-11-20 05:51:45 +02:00
Slavi Pantaleev
b0fa024297 Update services 2025-11-20 05:50:36 +02:00
Slavi Pantaleev
3ec203128a Update dependencies 2025-11-20 05:49:15 +02:00
Slavi Pantaleev
da97361e1b Bump default OpenAI text-generation model (gpt-5 -> gpt-5.1) 2025-11-20 05:25:20 +02:00
Slavi Pantaleev
b430fe0189 Update sample OpenAI config (for gpt-5) misleading users into using max_response_tokens & remove openai-o1.yml sample config
No need to have both sample configs now.

Fixes https://github.com/etkecc/baibot/issues/57
2025-11-20 05:25:20 +02:00
Slavi Pantaleev
f03126a9e1 Upgrade Rust (1.89.0 -> 1.90.0) 2025-10-26 09:00:01 +02:00
Slavi Pantaleev
7d46b926c1 Update services 2025-10-26 08:30:42 +02:00
Slavi Pantaleev
6f3c048195 Release 1.8.1 2025-09-12 16:54:11 +03:00
Slavi Pantaleev
b47cf598b5 Update dependencies 2025-09-12 16:53:35 +03:00
18 changed files with 856 additions and 1266 deletions

View File

@@ -1,3 +1,27 @@
# (2025-12-06) Version 1.10.0
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.11.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.16.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.16.0).
# (2025-11-30) Version 1.9.0
- (**Internal Improvement**) Upgrade [async-openai](https://crates.io/crates/async-openai) from our own etkecc fork (0.28.1-patched) to the official upstream version 0.31.1. This upgrade required some code adaptations to the new module structure, etc. While tested, regressions are possible.
# (2025-11-28) Version 1.8.3
- (**Improvement**) Add support for the `BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY` environment variable for configuring `persistence.session_encryption_key`
- (**Improvement**) Add support for the `BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED` environment variable for configuring `user.encryption.recovery_reset_allowed`
- (**Internal Improvement**) Dependency updates.
# (2025-11-20) Version 1.8.2
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.89.0 -> 1.91.1).
# (2025-09-12) Version 1.8.1
- (**Internal Improvement**) Dependency updates.
# (2025-09-08) Version 1.8.0 # (2025-09-08) Version 1.8.0
- (**Internal Improvement**) Upgrade [mxlink](https://crates.io/crates/mxlink) (1.9.0 -> 1.10.0) and [matrix-sdk](https://crates.io/crates/matrix-sdk) (0.13.0 -> 0.14.0) - (**Internal Improvement**) Upgrade [mxlink](https://crates.io/crates/mxlink) (1.9.0 -> 1.10.0) and [matrix-sdk](https://crates.io/crates/matrix-sdk) (0.13.0 -> 0.14.0)

1827
Cargo.lock generated

File diff suppressed because it is too large Load Diff

View File

@@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later"
readme = "README.md" readme = "README.md"
keywords = ["matrix", "chat", "bot", "AI", "LLM"] keywords = ["matrix", "chat", "bot", "AI", "LLM"]
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"] include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
version = "1.8.0" version = "1.10.0"
edition = "2024" edition = "2024"
[lib] [lib]
@@ -17,23 +17,23 @@ path = "src/lib.rs"
[dependencies] [dependencies]
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" } anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
anyhow = "1.0.*" anyhow = "1.0.*"
async-openai = { git = "https://github.com/etkecc/async-openai", branch = "async-openai-v0.28.1-patched" } async-openai = { version = "0.31.1", features = ["audio", "chat-completion", "image"] }
base64 = "0.22.*" base64 = "0.22.*"
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] } chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it. # We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
# We add the `native-tls` feature, because of https://github.com/etkecc/rust-mxlink/issues/1 # We add the `native-tls` feature, because of https://github.com/etkecc/rust-mxlink/issues/1
matrix-sdk = { version = "0.14.0", default-features = false, features = ["native-tls"] } matrix-sdk = { version = "0.16.0", default-features = false, features = ["native-tls"] }
mxidwc = "1.0.*" mxidwc = "1.0.*"
mxlink = ">=1.10.0" mxlink = ">=1.11.0"
etke_openai_api_rust = "0.1.*" etke_openai_api_rust = "0.1.*"
quick_cache = "0.6.*" quick_cache = "0.6.*"
regex = "1.11.*" regex = "1.12.*"
serde = { version = "1.0.*", features = ["derive"], default-features = false } serde = { version = "1.0.*", features = ["derive"], default-features = false }
serde_json = "1.0.*" serde_json = "1.0.*"
serde_yaml = "0.9.*" serde_yaml = "0.9.*"
tempfile = "3.21.*" tempfile = "3.23.*"
tiktoken-rs = { version = "0.7.*", default-features = false } tiktoken-rs = { version = "0.9.*", default-features = false }
tokio = { version = "1.47.*", features = ["rt", "rt-multi-thread", "macros"] } tokio = { version = "1.48.*", features = ["rt", "rt-multi-thread", "macros"] }
tracing = "0.1.*" tracing = "0.1.*"
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] } tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
url = "2.5.*" url = "2.5.*"

View File

@@ -4,7 +4,7 @@
# # # #
####################################### #######################################
FROM docker.io/rust:1.89.0-slim-trixie AS build FROM docker.io/rust:1.91.1-slim-trixie AS build
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev

View File

@@ -4,7 +4,7 @@
# # # #
####################################### #######################################
FROM docker.io/rust:1.89.0-slim-trixie AS build FROM docker.io/rust:1.91.1-slim-trixie AS build
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev

View File

@@ -125,10 +125,7 @@ For services which are not fully compatible with the OpenAI API, consider using
- create a room-local agent: `!bai agent create-room-local openai my-openai-agent` - create a room-local agent: `!bai agent create-room-local openai my-openai-agent`
- create a global agent: `!bai agent create-global openai my-openai-agent` - create a global agent: `!bai agent create-global openai my-openai-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which: 💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/openai.yml).
- in the general case looks [like this](./sample-provider-configs/openai.yml)
- for the [o1](https://platform.openai.com/docs/models/o1) models needs to look [like this](./sample-provider-configs/openai-o1.yml)
### OpenAI Compatible ### OpenAI Compatible

View File

@@ -1,24 +0,0 @@
base_url: https://api.openai.com/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: o1-mini
# o1 models do not support a system prompt
prompt: null
temperature: 1.0
# o1 models do not support max_response_tokens.
# They use `max_completion_tokens` as an alternative
max_response_tokens: null
max_completion_tokens: 16384
max_context_tokens: 128000
speech_to_text:
model_id: whisper-1
text_to_speech:
model_id: tts-1-hd
voice: onyx
speed: 1.0
response_format: opus
image_generation:
model_id: gpt-image-1
style: null
size: null
quality: null

View File

@@ -1,11 +1,14 @@
base_url: https://api.openai.com/v1 base_url: https://api.openai.com/v1
api_key: YOUR_API_KEY_HERE api_key: YOUR_API_KEY_HERE
text_generation: text_generation:
model_id: gpt-5 model_id: gpt-5.1
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}." prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
temperature: 1.0 temperature: 1.0
max_response_tokens: 16384 # Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`.
max_context_tokens: 128000 # If you're dealing with a non-reasoning model, specify `max_response_tokens` and unset `max_completion_tokens`.
max_response_tokens: null
max_completion_tokens: 128000
max_context_tokens: 400000
speech_to_text: speech_to_text:
model_id: whisper-1 model_id: whisper-1
text_to_speech: text_to_speech:

View File

@@ -76,11 +76,12 @@ agents:
# base_url: https://api.openai.com/v1 # base_url: https://api.openai.com/v1
# api_key: "" # api_key: ""
# text_generation: # text_generation:
# model_id: gpt-5 # model_id: gpt-5.1
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}." # prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
# temperature: 1.0 # temperature: 1.0
# max_response_tokens: ~
# # Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`. # # Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`.
# # If you're dealing with a non-reasoning model, specify `max_response_tokens` and unset `max_completion_tokens`.
# max_response_tokens: null
# max_completion_tokens: 128000 # max_completion_tokens: 128000
# max_context_tokens: 400000 # max_context_tokens: 400000
# speech_to_text: # speech_to_text:

View File

@@ -1,6 +1,6 @@
services: services:
postgres: postgres:
image: docker.io/postgres:17.6-alpine image: docker.io/postgres:18.1-alpine
user: ${UID}:${GID} user: ${UID}:${GID}
restart: unless-stopped restart: unless-stopped
environment: environment:
@@ -8,12 +8,13 @@ services:
POSTGRES_PASSWORD: synapse-password POSTGRES_PASSWORD: synapse-password
POSTGRES_DB: homeserver POSTGRES_DB: homeserver
POSTGRES_INITDB_ARGS: --lc-collate C --lc-ctype C --encoding UTF8 POSTGRES_INITDB_ARGS: --lc-collate C --lc-ctype C --encoding UTF8
PGDATA: /data
volumes: volumes:
- ./postgres:/var/lib/postgresql/data - ./postgres:/data
- /etc/passwd:/etc/passwd:ro - /etc/passwd:/etc/passwd:ro
synapse: synapse:
image: ghcr.io/element-hq/synapse:v1.137.0 image: ghcr.io/element-hq/synapse:v1.143.0
user: "${UID}:${GID}" user: "${UID}:${GID}"
restart: unless-stopped restart: unless-stopped
entrypoint: python entrypoint: python
@@ -26,7 +27,7 @@ services:
- ./synapse/media-store:/media-store - ./synapse/media-store:/media-store
element-web: element-web:
image: ghcr.io/element-hq/element-web:v1.11.110 image: ghcr.io/element-hq/element-web:v1.12.4
user: "${UID}:${GID}" user: "${UID}:${GID}"
restart: unless-stopped restart: unless-stopped
environment: environment:

View File

@@ -1,6 +1,6 @@
services: services:
ollama: ollama:
image: docker.io/ollama/ollama:0.11.10 image: docker.io/ollama/ollama:0.13.0
restart: unless-stopped restart: unless-stopped
ports: ports:
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434" - "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"

View File

@@ -56,12 +56,8 @@ impl ImageSource {
} }
} }
impl From<ImageSource> for async_openai::types::ImageInput { impl From<ImageSource> for async_openai::types::images::ImageInput {
fn from(value: ImageSource) -> Self { fn from(value: ImageSource) -> Self {
async_openai::types::ImageInput::from_vec_u8( async_openai::types::images::ImageInput::from_vec_u8(value.filename, value.bytes)
value.filename,
value.bytes,
value.mime_type.to_string(),
)
} }
} }

View File

@@ -80,7 +80,7 @@ impl Default for TextGenerationConfig {
} }
fn default_text_model_id() -> String { fn default_text_model_id() -> String {
"gpt-5".to_owned() "gpt-5.1".to_owned()
} }
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]
@@ -104,16 +104,16 @@ fn default_speech_to_text_model_id() -> String {
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextToSpeechConfig { pub struct TextToSpeechConfig {
#[serde(default = "default_text_to_speech_model_id")] #[serde(default = "default_text_to_speech_model_id")]
pub model_id: async_openai::types::SpeechModel, pub model_id: async_openai::types::audio::SpeechModel,
#[serde(default = "default_text_to_speech_voice")] #[serde(default = "default_text_to_speech_voice")]
pub voice: async_openai::types::Voice, pub voice: async_openai::types::audio::Voice,
#[serde(default = "default_text_to_speech_speed")] #[serde(default = "default_text_to_speech_speed")]
pub speed: f32, pub speed: f32,
#[serde(default = "default_text_to_speech_response_format")] #[serde(default = "default_text_to_speech_response_format")]
pub response_format: async_openai::types::SpeechResponseFormat, pub response_format: async_openai::types::audio::SpeechResponseFormat,
} }
impl Default for TextToSpeechConfig { impl Default for TextToSpeechConfig {
@@ -127,22 +127,22 @@ impl Default for TextToSpeechConfig {
} }
} }
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel { fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel {
async_openai::types::SpeechModel::Tts1Hd async_openai::types::audio::SpeechModel::Tts1Hd
} }
fn default_text_to_speech_voice() -> async_openai::types::Voice { fn default_text_to_speech_voice() -> async_openai::types::audio::Voice {
async_openai::types::Voice::Onyx async_openai::types::audio::Voice::Onyx
} }
fn default_text_to_speech_speed() -> f32 { fn default_text_to_speech_speed() -> f32 {
1.0 1.0
} }
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat { fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat {
// The API defaults to mp3, but we prefer Opus because it's smaller. // The API defaults to mp3, but we prefer Opus because it's smaller.
// Our clients should all have support for it. // Our clients should all have support for it.
async_openai::types::SpeechResponseFormat::Opus async_openai::types::audio::SpeechResponseFormat::Opus
} }
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]
@@ -150,13 +150,13 @@ pub struct ImageGenerationConfig {
pub model_id: String, pub model_id: String,
#[serde(default = "default_image_style")] #[serde(default = "default_image_style")]
pub style: Option<async_openai::types::ImageStyle>, pub style: Option<async_openai::types::images::ImageStyle>,
#[serde(default = "default_image_size")] #[serde(default = "default_image_size")]
pub size: Option<async_openai::types::ImageSize>, pub size: Option<async_openai::types::images::ImageSize>,
#[serde(default = "default_image_quality")] #[serde(default = "default_image_quality")]
pub quality: Option<async_openai::types::ImageQuality>, pub quality: Option<async_openai::types::images::ImageQuality>,
} }
impl Default for ImageGenerationConfig { impl Default for ImageGenerationConfig {
@@ -173,23 +173,25 @@ impl Default for ImageGenerationConfig {
impl ImageGenerationConfig { impl ImageGenerationConfig {
pub fn model_id_as_openai_image_model( pub fn model_id_as_openai_image_model(
&self, &self,
) -> Result<async_openai::types::ImageModel, String> { ) -> Result<async_openai::types::images::ImageModel, String> {
match self.model_id.as_str() { match self.model_id.as_str() {
"dall-e-2" => Ok(async_openai::types::ImageModel::DallE2), "dall-e-2" => Ok(async_openai::types::images::ImageModel::DallE2),
"dall-e-3" => Ok(async_openai::types::ImageModel::DallE3), "dall-e-3" => Ok(async_openai::types::images::ImageModel::DallE3),
other => Ok(async_openai::types::ImageModel::Other(other.to_owned())), "gpt-image-1" => Ok(async_openai::types::images::ImageModel::GptImage1),
"gpt-image-1-mini" => Ok(async_openai::types::images::ImageModel::GptImage1Mini),
other => Ok(async_openai::types::images::ImageModel::Other(other.to_owned())),
} }
} }
} }
fn default_image_style() -> Option<async_openai::types::ImageStyle> { fn default_image_style() -> Option<async_openai::types::images::ImageStyle> {
None None
} }
fn default_image_size() -> Option<async_openai::types::ImageSize> { fn default_image_size() -> Option<async_openai::types::images::ImageSize> {
None None
} }
fn default_image_quality() -> Option<async_openai::types::ImageQuality> { fn default_image_quality() -> Option<async_openai::types::images::ImageQuality> {
None None
} }

View File

@@ -4,9 +4,12 @@ use async_openai::{
Client as OpenAIClient, Client as OpenAIClient,
config::OpenAIConfig, config::OpenAIConfig,
types::{ types::{
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageEditRequestArgs, audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs},
CreateImageRequestArgs, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs, chat::{ChatCompletionRequestMessage, CreateChatCompletionRequestArgs},
DallE2ImageSize, Image, ImageModel, ImageResponseFormat, images::{
CreateImageEditRequestArgs, CreateImageRequestArgs,
Image, ImageInput, ImageModel, ImageResponseFormat,
},
}, },
}; };
@@ -38,8 +41,6 @@ use crate::{
use super::config::Config; use super::config::Config;
use super::OPENAI_IMAGE_MODEL_GPT_IMAGE_1;
#[derive(Debug, Clone)] #[derive(Debug, Clone)]
pub struct Controller { pub struct Controller {
config: Config, config: Config,
@@ -209,11 +210,7 @@ impl ControllerTrait for Controller {
let request = CreateTranscriptionRequestArgs::default() let request = CreateTranscriptionRequestArgs::default()
.model(&speech_to_text_config.model_id) .model(&speech_to_text_config.model_id)
.file(async_openai::types::AudioInput::from_vec_u8( .file(AudioInput::from_vec_u8(filename, media))
filename,
media,
mime_type.to_string(),
))
.language(language.clone()) .language(language.clone())
.build()?; .build()?;
@@ -223,7 +220,7 @@ impl ControllerTrait for Controller {
"Sending OpenAI speech-to-text API request" "Sending OpenAI speech-to-text API request"
); );
let response = self.client.audio().transcribe(request).await?; let response = self.client.audio().transcription().create(request).await?;
tracing::trace!( tracing::trace!(
?response, ?response,
@@ -255,11 +252,12 @@ impl ControllerTrait for Controller {
let model = if params.cheaper_model_switching_allowed { let model = if params.cheaper_model_switching_allowed {
// Switch to a cheaper model // Switch to a cheaper model
match original_model { match original_model {
async_openai::types::ImageModel::DallE2 => async_openai::types::ImageModel::DallE2, ImageModel::DallE2 => ImageModel::DallE2,
async_openai::types::ImageModel::DallE3 => async_openai::types::ImageModel::DallE2, ImageModel::DallE3 => ImageModel::DallE2,
async_openai::types::ImageModel::Other(_) => { ImageModel::Other(_) => {
async_openai::types::ImageModel::DallE2 ImageModel::DallE2
} }
_ => original_model.clone(),
} }
} else { } else {
original_model original_model
@@ -269,11 +267,24 @@ impl ControllerTrait for Controller {
// Switch to a cheaper quality // Switch to a cheaper quality
match &image_generation_config.quality { match &image_generation_config.quality {
Some(quality) => match quality { Some(quality) => match quality {
async_openai::types::ImageQuality::Standard => { async_openai::types::images::ImageQuality::Standard => {
Some(async_openai::types::ImageQuality::Standard) Some(async_openai::types::images::ImageQuality::Standard)
} }
async_openai::types::ImageQuality::HD => { async_openai::types::images::ImageQuality::HD => {
Some(async_openai::types::ImageQuality::Standard) Some(async_openai::types::images::ImageQuality::Standard)
}
// New quality levels - keep as-is or downgrade to Standard
async_openai::types::images::ImageQuality::High => {
Some(async_openai::types::images::ImageQuality::Standard)
}
async_openai::types::images::ImageQuality::Medium => {
Some(async_openai::types::images::ImageQuality::Medium)
}
async_openai::types::images::ImageQuality::Low => {
Some(async_openai::types::images::ImageQuality::Low)
}
async_openai::types::images::ImageQuality::Auto => {
Some(async_openai::types::images::ImageQuality::Auto)
} }
}, },
None => None, None => None,
@@ -284,18 +295,17 @@ impl ControllerTrait for Controller {
let size = params let size = params
.size_override .size_override
.map(|s| convert_string_to_enum::<async_openai::types::ImageSize>(&s).unwrap()) .map(|s| convert_string_to_enum::<async_openai::types::images::ImageSize>(&s).unwrap())
.or(image_generation_config.size); .or(image_generation_config.size);
let response_format = match model.clone() { let response_format = match model.clone() {
ImageModel::DallE2 => Some(ImageResponseFormat::B64Json), ImageModel::DallE2 => Some(ImageResponseFormat::B64Json),
ImageModel::DallE3 => Some(ImageResponseFormat::B64Json), ImageModel::DallE3 => Some(ImageResponseFormat::B64Json),
ImageModel::Other(model_str) => match model_str.as_str() { // gpt-image-1 only outputs base64 and we don't need to specify the response format.
// gpt-image-1 only outputs base64 and we don't need to specify the response format. // In fact, specifying the response format results in an error.
// In fact, specifying the response format results in an error. ImageModel::GptImage1 => None,
OPENAI_IMAGE_MODEL_GPT_IMAGE_1 => None, ImageModel::GptImage1Mini => None,
_ => Some(ImageResponseFormat::B64Json), ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
},
}; };
let mut request_builder = CreateImageRequestArgs::default(); let mut request_builder = CreateImageRequestArgs::default();
@@ -329,15 +339,15 @@ impl ControllerTrait for Controller {
"Sending OpenAI image generation API request" "Sending OpenAI image generation API request"
); );
let response = self.client.images().create(request).await?; let response = self.client.images().generate(request).await?;
if let Some(image) = response.data.into_iter().next() { if let Some(image) = response.data.into_iter().next() {
match image.deref() { match image.deref() {
async_openai::types::Image::B64Json { Image::B64Json {
b64_json, b64_json,
revised_prompt, revised_prompt,
} => { } => {
let bytes = base64_decode(b64_json)?; let bytes = base64_decode(b64_json.as_ref())?;
return Ok(ImageGenerationResult { return Ok(ImageGenerationResult {
bytes, bytes,
@@ -374,15 +384,15 @@ impl ControllerTrait for Controller {
return Err(anyhow::anyhow!("No image sources provided")); return Err(anyhow::anyhow!("No image sources provided"));
} }
let mut image_inputs = Vec::new(); let mut image_inputs: Vec<ImageInput> = Vec::new();
for image in images { for image in images {
image_inputs.push(image.into()); image_inputs.push(image.into());
} }
let dalle2_size = match image_generation_config.size { let dalle2_size = match image_generation_config.size {
Some(async_openai::types::ImageSize::S256x256) => Some(DallE2ImageSize::S256x256), Some(async_openai::types::images::ImageSize::S256x256) => Some(async_openai::types::images::ImageSize::S256x256),
Some(async_openai::types::ImageSize::S512x512) => Some(DallE2ImageSize::S512x512), Some(async_openai::types::images::ImageSize::S512x512) => Some(async_openai::types::images::ImageSize::S512x512),
Some(async_openai::types::ImageSize::S1024x1024) => Some(DallE2ImageSize::S1024x1024), Some(async_openai::types::images::ImageSize::S1024x1024) => Some(async_openai::types::images::ImageSize::S1024x1024),
_ => None, _ => None,
}; };
@@ -391,16 +401,17 @@ impl ControllerTrait for Controller {
.map_err(|err| anyhow::anyhow!(err))?; .map_err(|err| anyhow::anyhow!(err))?;
let response_format = match model.clone() { let response_format = match model.clone() {
async_openai::types::ImageModel::DallE2 => { ImageModel::DallE2 => {
Some(async_openai::types::ImageResponseFormat::B64Json) Some(ImageResponseFormat::B64Json)
} }
async_openai::types::ImageModel::DallE3 => { ImageModel::DallE3 => {
Some(async_openai::types::ImageResponseFormat::B64Json) Some(ImageResponseFormat::B64Json)
} }
async_openai::types::ImageModel::Other(model_str) => match model_str.as_str() { // gpt-image-1 only outputs base64 and we don't need to specify the response format.
OPENAI_IMAGE_MODEL_GPT_IMAGE_1 => None, // In fact, specifying the response format results in an error.
_ => Some(async_openai::types::ImageResponseFormat::B64Json), ImageModel::GptImage1 => None,
}, ImageModel::GptImage1Mini => None,
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
}; };
let mut request_builder = CreateImageEditRequestArgs::default(); let mut request_builder = CreateImageEditRequestArgs::default();
@@ -429,12 +440,12 @@ impl ControllerTrait for Controller {
"Sending OpenAI image edit API request" "Sending OpenAI image edit API request"
); );
let response = self.client.images().create_edit(request).await?; let response = self.client.images().edit(request).await?;
if let Some(image_data) = response.data.into_iter().next() { if let Some(image_data) = response.data.into_iter().next() {
match image_data.deref() { match image_data.deref() {
Image::B64Json { b64_json, .. } => { Image::B64Json { b64_json, .. } => {
let bytes = base64_decode(b64_json)?; let bytes = base64_decode(b64_json.as_ref())?;
return Ok(ImageEditResult { return Ok(ImageEditResult {
bytes, bytes,
mime_type: mxlink::mime::IMAGE_PNG, mime_type: mxlink::mime::IMAGE_PNG,
@@ -471,7 +482,7 @@ impl ControllerTrait for Controller {
let voice = if let Some(voice_string) = params.voice_override { let voice = if let Some(voice_string) = params.voice_override {
// This is a hacky way to construct a Voice enum from the string we have. // This is a hacky way to construct a Voice enum from the string we have.
let voice: serde_json::Result<async_openai::types::Voice> = let voice: serde_json::Result<async_openai::types::audio::Voice> =
serde_json::from_str(&format!("\"{}\"", voice_string)); serde_json::from_str(&format!("\"{}\"", voice_string));
match voice { match voice {
Ok(voice) => voice, Ok(voice) => voice,
@@ -511,7 +522,7 @@ impl ControllerTrait for Controller {
"Sending OpenAI text-to-speech API request" "Sending OpenAI text-to-speech API request"
); );
let result = self.client.audio().speech(request).await?; let result = self.client.audio().speech().create(request).await?;
Ok(TextToSpeechResult { Ok(TextToSpeechResult {
bytes: result.bytes.into(), bytes: result.bytes.into(),
@@ -570,15 +581,15 @@ impl ControllerTrait for Controller {
} }
fn response_format_to_mime_type( fn response_format_to_mime_type(
response_format: &async_openai::types::SpeechResponseFormat, response_format: &async_openai::types::audio::SpeechResponseFormat,
) -> Option<mxlink::mime::Mime> { ) -> Option<mxlink::mime::Mime> {
let content_type = match response_format { let content_type = match response_format {
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(), async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(), async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(), async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(), async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(), async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(), async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
}; };
match content_type.parse() { match content_type.parse() {

View File

@@ -1,8 +1,12 @@
use async_openai::types::{ use async_openai::types::{
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage, chat::{
ChatCompletionRequestMessageContentPartImage, ChatCompletionRequestSystemMessageArgs, ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
ChatCompletionRequestUserMessageArgs, ChatCompletionRequestUserMessageContent, ChatCompletionRequestMessageContentPartImage,
ChatCompletionRequestUserMessageContentPart, ImageUrlArgs, ChatCompletionRequestSystemMessageArgs,
ChatCompletionRequestUserMessageArgs, ChatCompletionRequestUserMessageContent,
ChatCompletionRequestUserMessageContentPart,
ImageUrlArgs,
},
}; };
use crate::conversation::llm::{ use crate::conversation::llm::{

View File

@@ -161,11 +161,11 @@ impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
type Error = String; type Error = String;
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> { fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?; let model_id = convert_string_to_enum::<async_openai::types::audio::SpeechModel>(&self.model_id)?;
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?; let voice = convert_string_to_enum::<async_openai::types::audio::Voice>(&self.voice)?;
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>( let response_format = convert_string_to_enum::<async_openai::types::audio::SpeechResponseFormat>(
&self.response_format, &self.response_format,
)?; )?;
@@ -224,7 +224,7 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> { fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> {
let size = if let Some(size) = &self.size { let size = if let Some(size) = &self.size {
Some(convert_string_to_enum::<async_openai::types::ImageSize>( Some(convert_string_to_enum::<async_openai::types::images::ImageSize>(
size, size,
)?) )?)
} else { } else {
@@ -232,7 +232,7 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
}; };
let style = if let Some(style) = &self.style { let style = if let Some(style) = &self.style {
Some(convert_string_to_enum::<async_openai::types::ImageStyle>( Some(convert_string_to_enum::<async_openai::types::images::ImageStyle>(
style, style,
)?) )?)
} else { } else {
@@ -240,7 +240,7 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
}; };
let quality = if let Some(quality) = &self.quality { let quality = if let Some(quality) = &self.quality {
Some(convert_string_to_enum::<async_openai::types::ImageQuality>( Some(convert_string_to_enum::<async_openai::types::images::ImageQuality>(
quality, quality,
)?) )?)
} else { } else {

View File

@@ -33,6 +33,9 @@ pub fn load() -> anyhow::Result<Config> {
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE => { cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE => {
config.user.encryption.recovery_passphrase = Some(value); config.user.encryption.recovery_passphrase = Some(value);
} }
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED => {
config.user.encryption.recovery_reset_allowed = value.parse::<bool>()?;
}
cfg_env::BAIBOT_USER_NAME => config.user.name = value, cfg_env::BAIBOT_USER_NAME => config.user.name = value,
cfg_env::BAIBOT_COMMAND_PREFIX => config.command_prefix = value, cfg_env::BAIBOT_COMMAND_PREFIX => config.command_prefix = value,
cfg_env::BAIBOT_ROOM_POST_JOIN_SELF_INTRODUCTION_ENABLED => { cfg_env::BAIBOT_ROOM_POST_JOIN_SELF_INTRODUCTION_ENABLED => {
@@ -51,6 +54,9 @@ pub fn load() -> anyhow::Result<Config> {
cfg_env::BAIBOT_PERSISTENCE_DATA_DIR_PATH => { cfg_env::BAIBOT_PERSISTENCE_DATA_DIR_PATH => {
config.persistence.data_dir_path = Some(value); config.persistence.data_dir_path = Some(value);
} }
cfg_env::BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY => {
config.persistence.session_encryption_key = Some(value);
}
cfg_env::BAIBOT_PERSISTENCE_CONFIG_ENCRYPTION_KEY => { cfg_env::BAIBOT_PERSISTENCE_CONFIG_ENCRYPTION_KEY => {
config.persistence.config_encryption_key = Some(value); config.persistence.config_encryption_key = Some(value);
} }

View File

@@ -8,6 +8,8 @@ pub const BAIBOT_USER_PASSWORD: &str = "BAIBOT_USER_PASSWORD";
pub const BAIBOT_USER_NAME: &str = "BAIBOT_USER_NAME"; pub const BAIBOT_USER_NAME: &str = "BAIBOT_USER_NAME";
pub const BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE: &str = pub const BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE: &str =
"BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE"; "BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE";
pub const BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED: &str =
"BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED";
pub const BAIBOT_COMMAND_PREFIX: &str = "BAIBOT_COMMAND_PREFIX"; pub const BAIBOT_COMMAND_PREFIX: &str = "BAIBOT_COMMAND_PREFIX";