Compare commits

...

1 Commits

Author SHA1 Message Date
Slavi Pantaleev
f6700ed601 Switch back to upstream async-openai
Once async-openai v0.31.0 gets released as a final version,
we'll be able top pull it from crates.io.

This does not quite compile yet, because of https://github.com/64bit/async-openai/issues/465
2025-11-08 13:33:46 +02:00
6 changed files with 44 additions and 38 deletions

8
Cargo.lock generated
View File

@@ -200,8 +200,8 @@ dependencies = [
[[package]] [[package]]
name = "async-openai" name = "async-openai"
version = "0.28.1" version = "0.31.0-alpha.2"
source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8" source = "git+https://github.com/64bit/async-openai?branch=main#08cae25ff0aa84552ccb32960e640cca139398fb"
dependencies = [ dependencies = [
"async-openai-macros", "async-openai-macros",
"backoff", "backoff",
@@ -210,7 +210,7 @@ dependencies = [
"derive_builder 0.20.2", "derive_builder 0.20.2",
"eventsource-stream", "eventsource-stream",
"futures", "futures",
"rand 0.8.5", "rand 0.9.2",
"reqwest 0.12.23", "reqwest 0.12.23",
"reqwest-eventsource 0.6.0", "reqwest-eventsource 0.6.0",
"secrecy", "secrecy",
@@ -226,7 +226,7 @@ dependencies = [
[[package]] [[package]]
name = "async-openai-macros" name = "async-openai-macros"
version = "0.1.0" version = "0.1.0"
source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8" source = "git+https://github.com/64bit/async-openai?branch=main#08cae25ff0aa84552ccb32960e640cca139398fb"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",

View File

@@ -17,7 +17,7 @@ path = "src/lib.rs"
[dependencies] [dependencies]
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" } anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
anyhow = "1.0.*" anyhow = "1.0.*"
async-openai = { git = "https://github.com/etkecc/async-openai", branch = "async-openai-v0.28.1-patched" } async-openai = { git = "https://github.com/64bit/async-openai", branch = "main" }
base64 = "0.22.*" base64 = "0.22.*"
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] } chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it. # We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.

View File

@@ -58,10 +58,6 @@ impl ImageSource {
impl From<ImageSource> for async_openai::types::ImageInput { impl From<ImageSource> for async_openai::types::ImageInput {
fn from(value: ImageSource) -> Self { fn from(value: ImageSource) -> Self {
async_openai::types::ImageInput::from_vec_u8( async_openai::types::ImageInput::from_vec_u8(value.filename, value.bytes)
value.filename,
value.bytes,
value.mime_type.to_string(),
)
} }
} }

View File

@@ -104,16 +104,16 @@ fn default_speech_to_text_model_id() -> String {
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextToSpeechConfig { pub struct TextToSpeechConfig {
#[serde(default = "default_text_to_speech_model_id")] #[serde(default = "default_text_to_speech_model_id")]
pub model_id: async_openai::types::SpeechModel, pub model_id: async_openai::types::audio::SpeechModel,
#[serde(default = "default_text_to_speech_voice")] #[serde(default = "default_text_to_speech_voice")]
pub voice: async_openai::types::Voice, pub voice: async_openai::types::audio::Voice,
#[serde(default = "default_text_to_speech_speed")] #[serde(default = "default_text_to_speech_speed")]
pub speed: f32, pub speed: f32,
#[serde(default = "default_text_to_speech_response_format")] #[serde(default = "default_text_to_speech_response_format")]
pub response_format: async_openai::types::SpeechResponseFormat, pub response_format: async_openai::types::audio::SpeechResponseFormat,
} }
impl Default for TextToSpeechConfig { impl Default for TextToSpeechConfig {
@@ -127,22 +127,22 @@ impl Default for TextToSpeechConfig {
} }
} }
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel { fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel {
async_openai::types::SpeechModel::Tts1Hd async_openai::types::audio::SpeechModel::Tts1Hd
} }
fn default_text_to_speech_voice() -> async_openai::types::Voice { fn default_text_to_speech_voice() -> async_openai::types::audio::Voice {
async_openai::types::Voice::Onyx async_openai::types::audio::Voice::Onyx
} }
fn default_text_to_speech_speed() -> f32 { fn default_text_to_speech_speed() -> f32 {
1.0 1.0
} }
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat { fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat {
// The API defaults to mp3, but we prefer Opus because it's smaller. // The API defaults to mp3, but we prefer Opus because it's smaller.
// Our clients should all have support for it. // Our clients should all have support for it.
async_openai::types::SpeechResponseFormat::Opus async_openai::types::audio::SpeechResponseFormat::Opus
} }
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]

View File

@@ -5,8 +5,9 @@ use async_openai::{
config::OpenAIConfig, config::OpenAIConfig,
types::{ types::{
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageEditRequestArgs, ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageEditRequestArgs,
CreateImageRequestArgs, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs, CreateImageRequestArgs,
DallE2ImageSize, Image, ImageModel, ImageResponseFormat, DallE2ImageSize, Image, ImageModel, ImageResponseFormat,
audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs},
}, },
}; };
@@ -209,11 +210,7 @@ impl ControllerTrait for Controller {
let request = CreateTranscriptionRequestArgs::default() let request = CreateTranscriptionRequestArgs::default()
.model(&speech_to_text_config.model_id) .model(&speech_to_text_config.model_id)
.file(async_openai::types::AudioInput::from_vec_u8( .file(AudioInput::from_vec_u8(filename, media))
filename,
media,
mime_type.to_string(),
))
.language(language.clone()) .language(language.clone())
.build()?; .build()?;
@@ -223,7 +220,7 @@ impl ControllerTrait for Controller {
"Sending OpenAI speech-to-text API request" "Sending OpenAI speech-to-text API request"
); );
let response = self.client.audio().transcribe(request).await?; let response = self.client.audio().transcription().create(request).await?;
tracing::trace!( tracing::trace!(
?response, ?response,
@@ -275,6 +272,19 @@ impl ControllerTrait for Controller {
async_openai::types::ImageQuality::HD => { async_openai::types::ImageQuality::HD => {
Some(async_openai::types::ImageQuality::Standard) Some(async_openai::types::ImageQuality::Standard)
} }
// New quality levels - keep as-is or downgrade to Standard
async_openai::types::ImageQuality::High => {
Some(async_openai::types::ImageQuality::Standard)
}
async_openai::types::ImageQuality::Medium => {
Some(async_openai::types::ImageQuality::Medium)
}
async_openai::types::ImageQuality::Low => {
Some(async_openai::types::ImageQuality::Low)
}
async_openai::types::ImageQuality::Auto => {
Some(async_openai::types::ImageQuality::Auto)
}
}, },
None => None, None => None,
} }
@@ -471,7 +481,7 @@ impl ControllerTrait for Controller {
let voice = if let Some(voice_string) = params.voice_override { let voice = if let Some(voice_string) = params.voice_override {
// This is a hacky way to construct a Voice enum from the string we have. // This is a hacky way to construct a Voice enum from the string we have.
let voice: serde_json::Result<async_openai::types::Voice> = let voice: serde_json::Result<async_openai::types::audio::Voice> =
serde_json::from_str(&format!("\"{}\"", voice_string)); serde_json::from_str(&format!("\"{}\"", voice_string));
match voice { match voice {
Ok(voice) => voice, Ok(voice) => voice,
@@ -511,7 +521,7 @@ impl ControllerTrait for Controller {
"Sending OpenAI text-to-speech API request" "Sending OpenAI text-to-speech API request"
); );
let result = self.client.audio().speech(request).await?; let result = self.client.audio().speech().create(request).await?;
Ok(TextToSpeechResult { Ok(TextToSpeechResult {
bytes: result.bytes.into(), bytes: result.bytes.into(),
@@ -570,15 +580,15 @@ impl ControllerTrait for Controller {
} }
fn response_format_to_mime_type( fn response_format_to_mime_type(
response_format: &async_openai::types::SpeechResponseFormat, response_format: &async_openai::types::audio::SpeechResponseFormat,
) -> Option<mxlink::mime::Mime> { ) -> Option<mxlink::mime::Mime> {
let content_type = match response_format { let content_type = match response_format {
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(), async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(), async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(), async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(), async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(), async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(), async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
}; };
match content_type.parse() { match content_type.parse() {

View File

@@ -161,11 +161,11 @@ impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
type Error = String; type Error = String;
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> { fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?; let model_id = convert_string_to_enum::<async_openai::types::audio::SpeechModel>(&self.model_id)?;
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?; let voice = convert_string_to_enum::<async_openai::types::audio::Voice>(&self.voice)?;
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>( let response_format = convert_string_to_enum::<async_openai::types::audio::SpeechResponseFormat>(
&self.response_format, &self.response_format,
)?; )?;