Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2692d0322e | ||
|
|
8eb70f0f2c | ||
|
|
5c0a7be7a2 |
12
CHANGELOG.md
12
CHANGELOG.md
@@ -1,3 +1,15 @@
|
|||||||
|
# (2025-11-30) Version 1.9.0
|
||||||
|
|
||||||
|
- (**Internal Improvement**) Upgrade [async-openai](https://crates.io/crates/async-openai) from our own etkecc fork (0.28.1-patched) to the official upstream version 0.31.1. This upgrade required some code adaptations to the new module structure, etc. While tested, regressions are possible.
|
||||||
|
|
||||||
|
# (2025-11-28) Version 1.8.3
|
||||||
|
|
||||||
|
- (**Improvement**) Add support for the `BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY` environment variable for configuring `persistence.session_encryption_key`
|
||||||
|
|
||||||
|
- (**Improvement**) Add support for the `BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED` environment variable for configuring `user.encryption.recovery_reset_allowed`
|
||||||
|
|
||||||
|
- (**Internal Improvement**) Dependency updates.
|
||||||
|
|
||||||
# (2025-11-20) Version 1.8.2
|
# (2025-11-20) Version 1.8.2
|
||||||
|
|
||||||
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.89.0 -> 1.91.1).
|
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.89.0 -> 1.91.1).
|
||||||
|
|||||||
14
Cargo.lock
generated
14
Cargo.lock
generated
@@ -191,8 +191,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "async-openai"
|
name = "async-openai"
|
||||||
version = "0.28.1"
|
version = "0.31.1"
|
||||||
source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1cc4c602409022b854d89332fae01a6c69a4122fdd0ec66a071b9b5c3a87750b"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-openai-macros",
|
"async-openai-macros",
|
||||||
"backoff",
|
"backoff",
|
||||||
@@ -201,23 +202,26 @@ dependencies = [
|
|||||||
"derive_builder 0.20.2",
|
"derive_builder 0.20.2",
|
||||||
"eventsource-stream",
|
"eventsource-stream",
|
||||||
"futures",
|
"futures",
|
||||||
"rand 0.8.5",
|
"rand 0.9.2",
|
||||||
"reqwest 0.12.24",
|
"reqwest 0.12.24",
|
||||||
"reqwest-eventsource 0.6.0",
|
"reqwest-eventsource 0.6.0",
|
||||||
"secrecy",
|
"secrecy",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
|
"serde_urlencoded",
|
||||||
"thiserror 2.0.17",
|
"thiserror 2.0.17",
|
||||||
"tokio",
|
"tokio",
|
||||||
"tokio-stream",
|
"tokio-stream",
|
||||||
"tokio-util",
|
"tokio-util",
|
||||||
"tracing",
|
"tracing",
|
||||||
|
"url",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "async-openai-macros"
|
name = "async-openai-macros"
|
||||||
version = "0.1.0"
|
version = "0.1.0"
|
||||||
source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0289cba6d5143bfe8251d57b4a8cac036adf158525a76533a7082ba65ec76398"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
@@ -296,7 +300,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "baibot"
|
name = "baibot"
|
||||||
version = "1.8.3"
|
version = "1.9.0"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anthropic",
|
"anthropic",
|
||||||
"anyhow",
|
"anyhow",
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later"
|
|||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
||||||
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
||||||
version = "1.8.3"
|
version = "1.9.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
|
||||||
[lib]
|
[lib]
|
||||||
@@ -17,7 +17,7 @@ path = "src/lib.rs"
|
|||||||
[dependencies]
|
[dependencies]
|
||||||
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
||||||
anyhow = "1.0.*"
|
anyhow = "1.0.*"
|
||||||
async-openai = { git = "https://github.com/etkecc/async-openai", branch = "async-openai-v0.28.1-patched" }
|
async-openai = { version = "0.31.1", features = ["audio", "chat-completion", "image"] }
|
||||||
base64 = "0.22.*"
|
base64 = "0.22.*"
|
||||||
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
||||||
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
||||||
|
|||||||
@@ -56,12 +56,8 @@ impl ImageSource {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl From<ImageSource> for async_openai::types::ImageInput {
|
impl From<ImageSource> for async_openai::types::images::ImageInput {
|
||||||
fn from(value: ImageSource) -> Self {
|
fn from(value: ImageSource) -> Self {
|
||||||
async_openai::types::ImageInput::from_vec_u8(
|
async_openai::types::images::ImageInput::from_vec_u8(value.filename, value.bytes)
|
||||||
value.filename,
|
|
||||||
value.bytes,
|
|
||||||
value.mime_type.to_string(),
|
|
||||||
)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -104,16 +104,16 @@ fn default_speech_to_text_model_id() -> String {
|
|||||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||||
pub struct TextToSpeechConfig {
|
pub struct TextToSpeechConfig {
|
||||||
#[serde(default = "default_text_to_speech_model_id")]
|
#[serde(default = "default_text_to_speech_model_id")]
|
||||||
pub model_id: async_openai::types::SpeechModel,
|
pub model_id: async_openai::types::audio::SpeechModel,
|
||||||
|
|
||||||
#[serde(default = "default_text_to_speech_voice")]
|
#[serde(default = "default_text_to_speech_voice")]
|
||||||
pub voice: async_openai::types::Voice,
|
pub voice: async_openai::types::audio::Voice,
|
||||||
|
|
||||||
#[serde(default = "default_text_to_speech_speed")]
|
#[serde(default = "default_text_to_speech_speed")]
|
||||||
pub speed: f32,
|
pub speed: f32,
|
||||||
|
|
||||||
#[serde(default = "default_text_to_speech_response_format")]
|
#[serde(default = "default_text_to_speech_response_format")]
|
||||||
pub response_format: async_openai::types::SpeechResponseFormat,
|
pub response_format: async_openai::types::audio::SpeechResponseFormat,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for TextToSpeechConfig {
|
impl Default for TextToSpeechConfig {
|
||||||
@@ -127,22 +127,22 @@ impl Default for TextToSpeechConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel {
|
fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel {
|
||||||
async_openai::types::SpeechModel::Tts1Hd
|
async_openai::types::audio::SpeechModel::Tts1Hd
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_text_to_speech_voice() -> async_openai::types::Voice {
|
fn default_text_to_speech_voice() -> async_openai::types::audio::Voice {
|
||||||
async_openai::types::Voice::Onyx
|
async_openai::types::audio::Voice::Onyx
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_text_to_speech_speed() -> f32 {
|
fn default_text_to_speech_speed() -> f32 {
|
||||||
1.0
|
1.0
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat {
|
fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat {
|
||||||
// The API defaults to mp3, but we prefer Opus because it's smaller.
|
// The API defaults to mp3, but we prefer Opus because it's smaller.
|
||||||
// Our clients should all have support for it.
|
// Our clients should all have support for it.
|
||||||
async_openai::types::SpeechResponseFormat::Opus
|
async_openai::types::audio::SpeechResponseFormat::Opus
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||||
@@ -150,13 +150,13 @@ pub struct ImageGenerationConfig {
|
|||||||
pub model_id: String,
|
pub model_id: String,
|
||||||
|
|
||||||
#[serde(default = "default_image_style")]
|
#[serde(default = "default_image_style")]
|
||||||
pub style: Option<async_openai::types::ImageStyle>,
|
pub style: Option<async_openai::types::images::ImageStyle>,
|
||||||
|
|
||||||
#[serde(default = "default_image_size")]
|
#[serde(default = "default_image_size")]
|
||||||
pub size: Option<async_openai::types::ImageSize>,
|
pub size: Option<async_openai::types::images::ImageSize>,
|
||||||
|
|
||||||
#[serde(default = "default_image_quality")]
|
#[serde(default = "default_image_quality")]
|
||||||
pub quality: Option<async_openai::types::ImageQuality>,
|
pub quality: Option<async_openai::types::images::ImageQuality>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for ImageGenerationConfig {
|
impl Default for ImageGenerationConfig {
|
||||||
@@ -173,23 +173,25 @@ impl Default for ImageGenerationConfig {
|
|||||||
impl ImageGenerationConfig {
|
impl ImageGenerationConfig {
|
||||||
pub fn model_id_as_openai_image_model(
|
pub fn model_id_as_openai_image_model(
|
||||||
&self,
|
&self,
|
||||||
) -> Result<async_openai::types::ImageModel, String> {
|
) -> Result<async_openai::types::images::ImageModel, String> {
|
||||||
match self.model_id.as_str() {
|
match self.model_id.as_str() {
|
||||||
"dall-e-2" => Ok(async_openai::types::ImageModel::DallE2),
|
"dall-e-2" => Ok(async_openai::types::images::ImageModel::DallE2),
|
||||||
"dall-e-3" => Ok(async_openai::types::ImageModel::DallE3),
|
"dall-e-3" => Ok(async_openai::types::images::ImageModel::DallE3),
|
||||||
other => Ok(async_openai::types::ImageModel::Other(other.to_owned())),
|
"gpt-image-1" => Ok(async_openai::types::images::ImageModel::GptImage1),
|
||||||
|
"gpt-image-1-mini" => Ok(async_openai::types::images::ImageModel::GptImage1Mini),
|
||||||
|
other => Ok(async_openai::types::images::ImageModel::Other(other.to_owned())),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_image_style() -> Option<async_openai::types::ImageStyle> {
|
fn default_image_style() -> Option<async_openai::types::images::ImageStyle> {
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_image_size() -> Option<async_openai::types::ImageSize> {
|
fn default_image_size() -> Option<async_openai::types::images::ImageSize> {
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
|
|
||||||
fn default_image_quality() -> Option<async_openai::types::ImageQuality> {
|
fn default_image_quality() -> Option<async_openai::types::images::ImageQuality> {
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,9 +4,12 @@ use async_openai::{
|
|||||||
Client as OpenAIClient,
|
Client as OpenAIClient,
|
||||||
config::OpenAIConfig,
|
config::OpenAIConfig,
|
||||||
types::{
|
types::{
|
||||||
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageEditRequestArgs,
|
audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs},
|
||||||
CreateImageRequestArgs, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs,
|
chat::{ChatCompletionRequestMessage, CreateChatCompletionRequestArgs},
|
||||||
DallE2ImageSize, Image, ImageModel, ImageResponseFormat,
|
images::{
|
||||||
|
CreateImageEditRequestArgs, CreateImageRequestArgs,
|
||||||
|
Image, ImageInput, ImageModel, ImageResponseFormat,
|
||||||
|
},
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -38,8 +41,6 @@ use crate::{
|
|||||||
|
|
||||||
use super::config::Config;
|
use super::config::Config;
|
||||||
|
|
||||||
use super::OPENAI_IMAGE_MODEL_GPT_IMAGE_1;
|
|
||||||
|
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct Controller {
|
pub struct Controller {
|
||||||
config: Config,
|
config: Config,
|
||||||
@@ -209,11 +210,7 @@ impl ControllerTrait for Controller {
|
|||||||
|
|
||||||
let request = CreateTranscriptionRequestArgs::default()
|
let request = CreateTranscriptionRequestArgs::default()
|
||||||
.model(&speech_to_text_config.model_id)
|
.model(&speech_to_text_config.model_id)
|
||||||
.file(async_openai::types::AudioInput::from_vec_u8(
|
.file(AudioInput::from_vec_u8(filename, media))
|
||||||
filename,
|
|
||||||
media,
|
|
||||||
mime_type.to_string(),
|
|
||||||
))
|
|
||||||
.language(language.clone())
|
.language(language.clone())
|
||||||
.build()?;
|
.build()?;
|
||||||
|
|
||||||
@@ -223,7 +220,7 @@ impl ControllerTrait for Controller {
|
|||||||
"Sending OpenAI speech-to-text API request"
|
"Sending OpenAI speech-to-text API request"
|
||||||
);
|
);
|
||||||
|
|
||||||
let response = self.client.audio().transcribe(request).await?;
|
let response = self.client.audio().transcription().create(request).await?;
|
||||||
|
|
||||||
tracing::trace!(
|
tracing::trace!(
|
||||||
?response,
|
?response,
|
||||||
@@ -255,11 +252,12 @@ impl ControllerTrait for Controller {
|
|||||||
let model = if params.cheaper_model_switching_allowed {
|
let model = if params.cheaper_model_switching_allowed {
|
||||||
// Switch to a cheaper model
|
// Switch to a cheaper model
|
||||||
match original_model {
|
match original_model {
|
||||||
async_openai::types::ImageModel::DallE2 => async_openai::types::ImageModel::DallE2,
|
ImageModel::DallE2 => ImageModel::DallE2,
|
||||||
async_openai::types::ImageModel::DallE3 => async_openai::types::ImageModel::DallE2,
|
ImageModel::DallE3 => ImageModel::DallE2,
|
||||||
async_openai::types::ImageModel::Other(_) => {
|
ImageModel::Other(_) => {
|
||||||
async_openai::types::ImageModel::DallE2
|
ImageModel::DallE2
|
||||||
}
|
}
|
||||||
|
_ => original_model.clone(),
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
original_model
|
original_model
|
||||||
@@ -269,11 +267,24 @@ impl ControllerTrait for Controller {
|
|||||||
// Switch to a cheaper quality
|
// Switch to a cheaper quality
|
||||||
match &image_generation_config.quality {
|
match &image_generation_config.quality {
|
||||||
Some(quality) => match quality {
|
Some(quality) => match quality {
|
||||||
async_openai::types::ImageQuality::Standard => {
|
async_openai::types::images::ImageQuality::Standard => {
|
||||||
Some(async_openai::types::ImageQuality::Standard)
|
Some(async_openai::types::images::ImageQuality::Standard)
|
||||||
}
|
}
|
||||||
async_openai::types::ImageQuality::HD => {
|
async_openai::types::images::ImageQuality::HD => {
|
||||||
Some(async_openai::types::ImageQuality::Standard)
|
Some(async_openai::types::images::ImageQuality::Standard)
|
||||||
|
}
|
||||||
|
// New quality levels - keep as-is or downgrade to Standard
|
||||||
|
async_openai::types::images::ImageQuality::High => {
|
||||||
|
Some(async_openai::types::images::ImageQuality::Standard)
|
||||||
|
}
|
||||||
|
async_openai::types::images::ImageQuality::Medium => {
|
||||||
|
Some(async_openai::types::images::ImageQuality::Medium)
|
||||||
|
}
|
||||||
|
async_openai::types::images::ImageQuality::Low => {
|
||||||
|
Some(async_openai::types::images::ImageQuality::Low)
|
||||||
|
}
|
||||||
|
async_openai::types::images::ImageQuality::Auto => {
|
||||||
|
Some(async_openai::types::images::ImageQuality::Auto)
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
None => None,
|
None => None,
|
||||||
@@ -284,18 +295,17 @@ impl ControllerTrait for Controller {
|
|||||||
|
|
||||||
let size = params
|
let size = params
|
||||||
.size_override
|
.size_override
|
||||||
.map(|s| convert_string_to_enum::<async_openai::types::ImageSize>(&s).unwrap())
|
.map(|s| convert_string_to_enum::<async_openai::types::images::ImageSize>(&s).unwrap())
|
||||||
.or(image_generation_config.size);
|
.or(image_generation_config.size);
|
||||||
|
|
||||||
let response_format = match model.clone() {
|
let response_format = match model.clone() {
|
||||||
ImageModel::DallE2 => Some(ImageResponseFormat::B64Json),
|
ImageModel::DallE2 => Some(ImageResponseFormat::B64Json),
|
||||||
ImageModel::DallE3 => Some(ImageResponseFormat::B64Json),
|
ImageModel::DallE3 => Some(ImageResponseFormat::B64Json),
|
||||||
ImageModel::Other(model_str) => match model_str.as_str() {
|
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
||||||
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
// In fact, specifying the response format results in an error.
|
||||||
// In fact, specifying the response format results in an error.
|
ImageModel::GptImage1 => None,
|
||||||
OPENAI_IMAGE_MODEL_GPT_IMAGE_1 => None,
|
ImageModel::GptImage1Mini => None,
|
||||||
_ => Some(ImageResponseFormat::B64Json),
|
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
|
||||||
},
|
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut request_builder = CreateImageRequestArgs::default();
|
let mut request_builder = CreateImageRequestArgs::default();
|
||||||
@@ -329,15 +339,15 @@ impl ControllerTrait for Controller {
|
|||||||
"Sending OpenAI image generation API request"
|
"Sending OpenAI image generation API request"
|
||||||
);
|
);
|
||||||
|
|
||||||
let response = self.client.images().create(request).await?;
|
let response = self.client.images().generate(request).await?;
|
||||||
|
|
||||||
if let Some(image) = response.data.into_iter().next() {
|
if let Some(image) = response.data.into_iter().next() {
|
||||||
match image.deref() {
|
match image.deref() {
|
||||||
async_openai::types::Image::B64Json {
|
Image::B64Json {
|
||||||
b64_json,
|
b64_json,
|
||||||
revised_prompt,
|
revised_prompt,
|
||||||
} => {
|
} => {
|
||||||
let bytes = base64_decode(b64_json)?;
|
let bytes = base64_decode(b64_json.as_ref())?;
|
||||||
|
|
||||||
return Ok(ImageGenerationResult {
|
return Ok(ImageGenerationResult {
|
||||||
bytes,
|
bytes,
|
||||||
@@ -374,15 +384,15 @@ impl ControllerTrait for Controller {
|
|||||||
return Err(anyhow::anyhow!("No image sources provided"));
|
return Err(anyhow::anyhow!("No image sources provided"));
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut image_inputs = Vec::new();
|
let mut image_inputs: Vec<ImageInput> = Vec::new();
|
||||||
for image in images {
|
for image in images {
|
||||||
image_inputs.push(image.into());
|
image_inputs.push(image.into());
|
||||||
}
|
}
|
||||||
|
|
||||||
let dalle2_size = match image_generation_config.size {
|
let dalle2_size = match image_generation_config.size {
|
||||||
Some(async_openai::types::ImageSize::S256x256) => Some(DallE2ImageSize::S256x256),
|
Some(async_openai::types::images::ImageSize::S256x256) => Some(async_openai::types::images::ImageSize::S256x256),
|
||||||
Some(async_openai::types::ImageSize::S512x512) => Some(DallE2ImageSize::S512x512),
|
Some(async_openai::types::images::ImageSize::S512x512) => Some(async_openai::types::images::ImageSize::S512x512),
|
||||||
Some(async_openai::types::ImageSize::S1024x1024) => Some(DallE2ImageSize::S1024x1024),
|
Some(async_openai::types::images::ImageSize::S1024x1024) => Some(async_openai::types::images::ImageSize::S1024x1024),
|
||||||
_ => None,
|
_ => None,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -391,16 +401,17 @@ impl ControllerTrait for Controller {
|
|||||||
.map_err(|err| anyhow::anyhow!(err))?;
|
.map_err(|err| anyhow::anyhow!(err))?;
|
||||||
|
|
||||||
let response_format = match model.clone() {
|
let response_format = match model.clone() {
|
||||||
async_openai::types::ImageModel::DallE2 => {
|
ImageModel::DallE2 => {
|
||||||
Some(async_openai::types::ImageResponseFormat::B64Json)
|
Some(ImageResponseFormat::B64Json)
|
||||||
}
|
}
|
||||||
async_openai::types::ImageModel::DallE3 => {
|
ImageModel::DallE3 => {
|
||||||
Some(async_openai::types::ImageResponseFormat::B64Json)
|
Some(ImageResponseFormat::B64Json)
|
||||||
}
|
}
|
||||||
async_openai::types::ImageModel::Other(model_str) => match model_str.as_str() {
|
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
||||||
OPENAI_IMAGE_MODEL_GPT_IMAGE_1 => None,
|
// In fact, specifying the response format results in an error.
|
||||||
_ => Some(async_openai::types::ImageResponseFormat::B64Json),
|
ImageModel::GptImage1 => None,
|
||||||
},
|
ImageModel::GptImage1Mini => None,
|
||||||
|
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut request_builder = CreateImageEditRequestArgs::default();
|
let mut request_builder = CreateImageEditRequestArgs::default();
|
||||||
@@ -429,12 +440,12 @@ impl ControllerTrait for Controller {
|
|||||||
"Sending OpenAI image edit API request"
|
"Sending OpenAI image edit API request"
|
||||||
);
|
);
|
||||||
|
|
||||||
let response = self.client.images().create_edit(request).await?;
|
let response = self.client.images().edit(request).await?;
|
||||||
|
|
||||||
if let Some(image_data) = response.data.into_iter().next() {
|
if let Some(image_data) = response.data.into_iter().next() {
|
||||||
match image_data.deref() {
|
match image_data.deref() {
|
||||||
Image::B64Json { b64_json, .. } => {
|
Image::B64Json { b64_json, .. } => {
|
||||||
let bytes = base64_decode(b64_json)?;
|
let bytes = base64_decode(b64_json.as_ref())?;
|
||||||
return Ok(ImageEditResult {
|
return Ok(ImageEditResult {
|
||||||
bytes,
|
bytes,
|
||||||
mime_type: mxlink::mime::IMAGE_PNG,
|
mime_type: mxlink::mime::IMAGE_PNG,
|
||||||
@@ -471,7 +482,7 @@ impl ControllerTrait for Controller {
|
|||||||
|
|
||||||
let voice = if let Some(voice_string) = params.voice_override {
|
let voice = if let Some(voice_string) = params.voice_override {
|
||||||
// This is a hacky way to construct a Voice enum from the string we have.
|
// This is a hacky way to construct a Voice enum from the string we have.
|
||||||
let voice: serde_json::Result<async_openai::types::Voice> =
|
let voice: serde_json::Result<async_openai::types::audio::Voice> =
|
||||||
serde_json::from_str(&format!("\"{}\"", voice_string));
|
serde_json::from_str(&format!("\"{}\"", voice_string));
|
||||||
match voice {
|
match voice {
|
||||||
Ok(voice) => voice,
|
Ok(voice) => voice,
|
||||||
@@ -511,7 +522,7 @@ impl ControllerTrait for Controller {
|
|||||||
"Sending OpenAI text-to-speech API request"
|
"Sending OpenAI text-to-speech API request"
|
||||||
);
|
);
|
||||||
|
|
||||||
let result = self.client.audio().speech(request).await?;
|
let result = self.client.audio().speech().create(request).await?;
|
||||||
|
|
||||||
Ok(TextToSpeechResult {
|
Ok(TextToSpeechResult {
|
||||||
bytes: result.bytes.into(),
|
bytes: result.bytes.into(),
|
||||||
@@ -570,15 +581,15 @@ impl ControllerTrait for Controller {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn response_format_to_mime_type(
|
fn response_format_to_mime_type(
|
||||||
response_format: &async_openai::types::SpeechResponseFormat,
|
response_format: &async_openai::types::audio::SpeechResponseFormat,
|
||||||
) -> Option<mxlink::mime::Mime> {
|
) -> Option<mxlink::mime::Mime> {
|
||||||
let content_type = match response_format {
|
let content_type = match response_format {
|
||||||
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||||
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||||
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||||
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||||
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||||
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||||
};
|
};
|
||||||
|
|
||||||
match content_type.parse() {
|
match content_type.parse() {
|
||||||
|
|||||||
@@ -1,8 +1,12 @@
|
|||||||
use async_openai::types::{
|
use async_openai::types::{
|
||||||
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
|
chat::{
|
||||||
ChatCompletionRequestMessageContentPartImage, ChatCompletionRequestSystemMessageArgs,
|
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
|
||||||
ChatCompletionRequestUserMessageArgs, ChatCompletionRequestUserMessageContent,
|
ChatCompletionRequestMessageContentPartImage,
|
||||||
ChatCompletionRequestUserMessageContentPart, ImageUrlArgs,
|
ChatCompletionRequestSystemMessageArgs,
|
||||||
|
ChatCompletionRequestUserMessageArgs, ChatCompletionRequestUserMessageContent,
|
||||||
|
ChatCompletionRequestUserMessageContentPart,
|
||||||
|
ImageUrlArgs,
|
||||||
|
},
|
||||||
};
|
};
|
||||||
|
|
||||||
use crate::conversation::llm::{
|
use crate::conversation::llm::{
|
||||||
|
|||||||
@@ -161,11 +161,11 @@ impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
|
|||||||
type Error = String;
|
type Error = String;
|
||||||
|
|
||||||
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
|
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
|
||||||
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?;
|
let model_id = convert_string_to_enum::<async_openai::types::audio::SpeechModel>(&self.model_id)?;
|
||||||
|
|
||||||
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?;
|
let voice = convert_string_to_enum::<async_openai::types::audio::Voice>(&self.voice)?;
|
||||||
|
|
||||||
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>(
|
let response_format = convert_string_to_enum::<async_openai::types::audio::SpeechResponseFormat>(
|
||||||
&self.response_format,
|
&self.response_format,
|
||||||
)?;
|
)?;
|
||||||
|
|
||||||
@@ -224,7 +224,7 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
|
|||||||
|
|
||||||
fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> {
|
fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> {
|
||||||
let size = if let Some(size) = &self.size {
|
let size = if let Some(size) = &self.size {
|
||||||
Some(convert_string_to_enum::<async_openai::types::ImageSize>(
|
Some(convert_string_to_enum::<async_openai::types::images::ImageSize>(
|
||||||
size,
|
size,
|
||||||
)?)
|
)?)
|
||||||
} else {
|
} else {
|
||||||
@@ -232,7 +232,7 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
|
|||||||
};
|
};
|
||||||
|
|
||||||
let style = if let Some(style) = &self.style {
|
let style = if let Some(style) = &self.style {
|
||||||
Some(convert_string_to_enum::<async_openai::types::ImageStyle>(
|
Some(convert_string_to_enum::<async_openai::types::images::ImageStyle>(
|
||||||
style,
|
style,
|
||||||
)?)
|
)?)
|
||||||
} else {
|
} else {
|
||||||
@@ -240,7 +240,7 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
|
|||||||
};
|
};
|
||||||
|
|
||||||
let quality = if let Some(quality) = &self.quality {
|
let quality = if let Some(quality) = &self.quality {
|
||||||
Some(convert_string_to_enum::<async_openai::types::ImageQuality>(
|
Some(convert_string_to_enum::<async_openai::types::images::ImageQuality>(
|
||||||
quality,
|
quality,
|
||||||
)?)
|
)?)
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
Reference in New Issue
Block a user