From f6700ed601d783d67364b137aa00c160d47279e9 Mon Sep 17 00:00:00 2001 From: Slavi Pantaleev Date: Sat, 8 Nov 2025 13:21:14 +0200 Subject: [PATCH] Switch back to upstream async-openai Once async-openai v0.31.0 gets released as a final version, we'll be able top pull it from crates.io. This does not quite compile yet, because of https://github.com/64bit/async-openai/issues/465 --- Cargo.lock | 8 ++--- Cargo.toml | 2 +- src/agent/provider/entity/image.rs | 6 +--- src/agent/provider/openai/config.rs | 18 +++++----- src/agent/provider/openai/controller.rs | 42 +++++++++++++--------- src/agent/provider/openai_compat/config.rs | 6 ++-- 6 files changed, 44 insertions(+), 38 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 2209c9b..04a4e06 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -200,8 +200,8 @@ dependencies = [ [[package]] name = "async-openai" -version = "0.28.1" -source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8" +version = "0.31.0-alpha.2" +source = "git+https://github.com/64bit/async-openai?branch=main#08cae25ff0aa84552ccb32960e640cca139398fb" dependencies = [ "async-openai-macros", "backoff", @@ -210,7 +210,7 @@ dependencies = [ "derive_builder 0.20.2", "eventsource-stream", "futures", - "rand 0.8.5", + "rand 0.9.2", "reqwest 0.12.23", "reqwest-eventsource 0.6.0", "secrecy", @@ -226,7 +226,7 @@ dependencies = [ [[package]] name = "async-openai-macros" version = "0.1.0" -source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8" +source = "git+https://github.com/64bit/async-openai?branch=main#08cae25ff0aa84552ccb32960e640cca139398fb" dependencies = [ "proc-macro2", "quote", diff --git a/Cargo.toml b/Cargo.toml index e0a647b..7f0d678 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -17,7 +17,7 @@ path = "src/lib.rs" [dependencies] anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" } anyhow = "1.0.*" -async-openai = { git = "https://github.com/etkecc/async-openai", branch = "async-openai-v0.28.1-patched" } +async-openai = { git = "https://github.com/64bit/async-openai", branch = "main" } base64 = "0.22.*" chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] } # We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it. diff --git a/src/agent/provider/entity/image.rs b/src/agent/provider/entity/image.rs index df95a7f..1c46298 100644 --- a/src/agent/provider/entity/image.rs +++ b/src/agent/provider/entity/image.rs @@ -58,10 +58,6 @@ impl ImageSource { impl From for async_openai::types::ImageInput { fn from(value: ImageSource) -> Self { - async_openai::types::ImageInput::from_vec_u8( - value.filename, - value.bytes, - value.mime_type.to_string(), - ) + async_openai::types::ImageInput::from_vec_u8(value.filename, value.bytes) } } diff --git a/src/agent/provider/openai/config.rs b/src/agent/provider/openai/config.rs index d0735ee..20dd330 100644 --- a/src/agent/provider/openai/config.rs +++ b/src/agent/provider/openai/config.rs @@ -104,16 +104,16 @@ fn default_speech_to_text_model_id() -> String { #[derive(Debug, Clone, Serialize, Deserialize)] pub struct TextToSpeechConfig { #[serde(default = "default_text_to_speech_model_id")] - pub model_id: async_openai::types::SpeechModel, + pub model_id: async_openai::types::audio::SpeechModel, #[serde(default = "default_text_to_speech_voice")] - pub voice: async_openai::types::Voice, + pub voice: async_openai::types::audio::Voice, #[serde(default = "default_text_to_speech_speed")] pub speed: f32, #[serde(default = "default_text_to_speech_response_format")] - pub response_format: async_openai::types::SpeechResponseFormat, + pub response_format: async_openai::types::audio::SpeechResponseFormat, } impl Default for TextToSpeechConfig { @@ -127,22 +127,22 @@ impl Default for TextToSpeechConfig { } } -fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel { - async_openai::types::SpeechModel::Tts1Hd +fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel { + async_openai::types::audio::SpeechModel::Tts1Hd } -fn default_text_to_speech_voice() -> async_openai::types::Voice { - async_openai::types::Voice::Onyx +fn default_text_to_speech_voice() -> async_openai::types::audio::Voice { + async_openai::types::audio::Voice::Onyx } fn default_text_to_speech_speed() -> f32 { 1.0 } -fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat { +fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat { // The API defaults to mp3, but we prefer Opus because it's smaller. // Our clients should all have support for it. - async_openai::types::SpeechResponseFormat::Opus + async_openai::types::audio::SpeechResponseFormat::Opus } #[derive(Debug, Clone, Serialize, Deserialize)] diff --git a/src/agent/provider/openai/controller.rs b/src/agent/provider/openai/controller.rs index bd6ee86..5511c37 100644 --- a/src/agent/provider/openai/controller.rs +++ b/src/agent/provider/openai/controller.rs @@ -5,8 +5,9 @@ use async_openai::{ config::OpenAIConfig, types::{ ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageEditRequestArgs, - CreateImageRequestArgs, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs, + CreateImageRequestArgs, DallE2ImageSize, Image, ImageModel, ImageResponseFormat, + audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs}, }, }; @@ -209,11 +210,7 @@ impl ControllerTrait for Controller { let request = CreateTranscriptionRequestArgs::default() .model(&speech_to_text_config.model_id) - .file(async_openai::types::AudioInput::from_vec_u8( - filename, - media, - mime_type.to_string(), - )) + .file(AudioInput::from_vec_u8(filename, media)) .language(language.clone()) .build()?; @@ -223,7 +220,7 @@ impl ControllerTrait for Controller { "Sending OpenAI speech-to-text API request" ); - let response = self.client.audio().transcribe(request).await?; + let response = self.client.audio().transcription().create(request).await?; tracing::trace!( ?response, @@ -275,6 +272,19 @@ impl ControllerTrait for Controller { async_openai::types::ImageQuality::HD => { Some(async_openai::types::ImageQuality::Standard) } + // New quality levels - keep as-is or downgrade to Standard + async_openai::types::ImageQuality::High => { + Some(async_openai::types::ImageQuality::Standard) + } + async_openai::types::ImageQuality::Medium => { + Some(async_openai::types::ImageQuality::Medium) + } + async_openai::types::ImageQuality::Low => { + Some(async_openai::types::ImageQuality::Low) + } + async_openai::types::ImageQuality::Auto => { + Some(async_openai::types::ImageQuality::Auto) + } }, None => None, } @@ -471,7 +481,7 @@ impl ControllerTrait for Controller { let voice = if let Some(voice_string) = params.voice_override { // This is a hacky way to construct a Voice enum from the string we have. - let voice: serde_json::Result = + let voice: serde_json::Result = serde_json::from_str(&format!("\"{}\"", voice_string)); match voice { Ok(voice) => voice, @@ -511,7 +521,7 @@ impl ControllerTrait for Controller { "Sending OpenAI text-to-speech API request" ); - let result = self.client.audio().speech(request).await?; + let result = self.client.audio().speech().create(request).await?; Ok(TextToSpeechResult { bytes: result.bytes.into(), @@ -570,15 +580,15 @@ impl ControllerTrait for Controller { } fn response_format_to_mime_type( - response_format: &async_openai::types::SpeechResponseFormat, + response_format: &async_openai::types::audio::SpeechResponseFormat, ) -> Option { let content_type = match response_format { - async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(), - async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(), - async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(), - async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(), - async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(), - async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(), + async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(), + async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(), + async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(), + async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(), + async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(), + async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(), }; match content_type.parse() { diff --git a/src/agent/provider/openai_compat/config.rs b/src/agent/provider/openai_compat/config.rs index e4ed370..e671033 100644 --- a/src/agent/provider/openai_compat/config.rs +++ b/src/agent/provider/openai_compat/config.rs @@ -161,11 +161,11 @@ impl TryInto for TextToSpeechConfig { type Error = String; fn try_into(self) -> Result { - let model_id = convert_string_to_enum::(&self.model_id)?; + let model_id = convert_string_to_enum::(&self.model_id)?; - let voice = convert_string_to_enum::(&self.voice)?; + let voice = convert_string_to_enum::(&self.voice)?; - let response_format = convert_string_to_enum::( + let response_format = convert_string_to_enum::( &self.response_format, )?;