Compare commits
1 Commits
v1.14.1
...
back-to-up
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f6700ed601 |
8
Cargo.lock
generated
8
Cargo.lock
generated
@@ -200,8 +200,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "async-openai"
|
||||
version = "0.28.1"
|
||||
source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8"
|
||||
version = "0.31.0-alpha.2"
|
||||
source = "git+https://github.com/64bit/async-openai?branch=main#08cae25ff0aa84552ccb32960e640cca139398fb"
|
||||
dependencies = [
|
||||
"async-openai-macros",
|
||||
"backoff",
|
||||
@@ -210,7 +210,7 @@ dependencies = [
|
||||
"derive_builder 0.20.2",
|
||||
"eventsource-stream",
|
||||
"futures",
|
||||
"rand 0.8.5",
|
||||
"rand 0.9.2",
|
||||
"reqwest 0.12.23",
|
||||
"reqwest-eventsource 0.6.0",
|
||||
"secrecy",
|
||||
@@ -226,7 +226,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "async-openai-macros"
|
||||
version = "0.1.0"
|
||||
source = "git+https://github.com/etkecc/async-openai?branch=async-openai-v0.28.1-patched#856953c2d4485342df625fd0525363362075e8a8"
|
||||
source = "git+https://github.com/64bit/async-openai?branch=main#08cae25ff0aa84552ccb32960e640cca139398fb"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
|
||||
@@ -17,7 +17,7 @@ path = "src/lib.rs"
|
||||
[dependencies]
|
||||
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
||||
anyhow = "1.0.*"
|
||||
async-openai = { git = "https://github.com/etkecc/async-openai", branch = "async-openai-v0.28.1-patched" }
|
||||
async-openai = { git = "https://github.com/64bit/async-openai", branch = "main" }
|
||||
base64 = "0.22.*"
|
||||
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
||||
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
||||
|
||||
@@ -58,10 +58,6 @@ impl ImageSource {
|
||||
|
||||
impl From<ImageSource> for async_openai::types::ImageInput {
|
||||
fn from(value: ImageSource) -> Self {
|
||||
async_openai::types::ImageInput::from_vec_u8(
|
||||
value.filename,
|
||||
value.bytes,
|
||||
value.mime_type.to_string(),
|
||||
)
|
||||
async_openai::types::ImageInput::from_vec_u8(value.filename, value.bytes)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -104,16 +104,16 @@ fn default_speech_to_text_model_id() -> String {
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextToSpeechConfig {
|
||||
#[serde(default = "default_text_to_speech_model_id")]
|
||||
pub model_id: async_openai::types::SpeechModel,
|
||||
pub model_id: async_openai::types::audio::SpeechModel,
|
||||
|
||||
#[serde(default = "default_text_to_speech_voice")]
|
||||
pub voice: async_openai::types::Voice,
|
||||
pub voice: async_openai::types::audio::Voice,
|
||||
|
||||
#[serde(default = "default_text_to_speech_speed")]
|
||||
pub speed: f32,
|
||||
|
||||
#[serde(default = "default_text_to_speech_response_format")]
|
||||
pub response_format: async_openai::types::SpeechResponseFormat,
|
||||
pub response_format: async_openai::types::audio::SpeechResponseFormat,
|
||||
}
|
||||
|
||||
impl Default for TextToSpeechConfig {
|
||||
@@ -127,22 +127,22 @@ impl Default for TextToSpeechConfig {
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel {
|
||||
async_openai::types::SpeechModel::Tts1Hd
|
||||
fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel {
|
||||
async_openai::types::audio::SpeechModel::Tts1Hd
|
||||
}
|
||||
|
||||
fn default_text_to_speech_voice() -> async_openai::types::Voice {
|
||||
async_openai::types::Voice::Onyx
|
||||
fn default_text_to_speech_voice() -> async_openai::types::audio::Voice {
|
||||
async_openai::types::audio::Voice::Onyx
|
||||
}
|
||||
|
||||
fn default_text_to_speech_speed() -> f32 {
|
||||
1.0
|
||||
}
|
||||
|
||||
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat {
|
||||
fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat {
|
||||
// The API defaults to mp3, but we prefer Opus because it's smaller.
|
||||
// Our clients should all have support for it.
|
||||
async_openai::types::SpeechResponseFormat::Opus
|
||||
async_openai::types::audio::SpeechResponseFormat::Opus
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
|
||||
@@ -5,8 +5,9 @@ use async_openai::{
|
||||
config::OpenAIConfig,
|
||||
types::{
|
||||
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageEditRequestArgs,
|
||||
CreateImageRequestArgs, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs,
|
||||
CreateImageRequestArgs,
|
||||
DallE2ImageSize, Image, ImageModel, ImageResponseFormat,
|
||||
audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -209,11 +210,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let request = CreateTranscriptionRequestArgs::default()
|
||||
.model(&speech_to_text_config.model_id)
|
||||
.file(async_openai::types::AudioInput::from_vec_u8(
|
||||
filename,
|
||||
media,
|
||||
mime_type.to_string(),
|
||||
))
|
||||
.file(AudioInput::from_vec_u8(filename, media))
|
||||
.language(language.clone())
|
||||
.build()?;
|
||||
|
||||
@@ -223,7 +220,7 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI speech-to-text API request"
|
||||
);
|
||||
|
||||
let response = self.client.audio().transcribe(request).await?;
|
||||
let response = self.client.audio().transcription().create(request).await?;
|
||||
|
||||
tracing::trace!(
|
||||
?response,
|
||||
@@ -275,6 +272,19 @@ impl ControllerTrait for Controller {
|
||||
async_openai::types::ImageQuality::HD => {
|
||||
Some(async_openai::types::ImageQuality::Standard)
|
||||
}
|
||||
// New quality levels - keep as-is or downgrade to Standard
|
||||
async_openai::types::ImageQuality::High => {
|
||||
Some(async_openai::types::ImageQuality::Standard)
|
||||
}
|
||||
async_openai::types::ImageQuality::Medium => {
|
||||
Some(async_openai::types::ImageQuality::Medium)
|
||||
}
|
||||
async_openai::types::ImageQuality::Low => {
|
||||
Some(async_openai::types::ImageQuality::Low)
|
||||
}
|
||||
async_openai::types::ImageQuality::Auto => {
|
||||
Some(async_openai::types::ImageQuality::Auto)
|
||||
}
|
||||
},
|
||||
None => None,
|
||||
}
|
||||
@@ -471,7 +481,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let voice = if let Some(voice_string) = params.voice_override {
|
||||
// This is a hacky way to construct a Voice enum from the string we have.
|
||||
let voice: serde_json::Result<async_openai::types::Voice> =
|
||||
let voice: serde_json::Result<async_openai::types::audio::Voice> =
|
||||
serde_json::from_str(&format!("\"{}\"", voice_string));
|
||||
match voice {
|
||||
Ok(voice) => voice,
|
||||
@@ -511,7 +521,7 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI text-to-speech API request"
|
||||
);
|
||||
|
||||
let result = self.client.audio().speech(request).await?;
|
||||
let result = self.client.audio().speech().create(request).await?;
|
||||
|
||||
Ok(TextToSpeechResult {
|
||||
bytes: result.bytes.into(),
|
||||
@@ -570,15 +580,15 @@ impl ControllerTrait for Controller {
|
||||
}
|
||||
|
||||
fn response_format_to_mime_type(
|
||||
response_format: &async_openai::types::SpeechResponseFormat,
|
||||
response_format: &async_openai::types::audio::SpeechResponseFormat,
|
||||
) -> Option<mxlink::mime::Mime> {
|
||||
let content_type = match response_format {
|
||||
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||
};
|
||||
|
||||
match content_type.parse() {
|
||||
|
||||
@@ -161,11 +161,11 @@ impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
|
||||
type Error = String;
|
||||
|
||||
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
|
||||
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?;
|
||||
let model_id = convert_string_to_enum::<async_openai::types::audio::SpeechModel>(&self.model_id)?;
|
||||
|
||||
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?;
|
||||
let voice = convert_string_to_enum::<async_openai::types::audio::Voice>(&self.voice)?;
|
||||
|
||||
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>(
|
||||
let response_format = convert_string_to_enum::<async_openai::types::audio::SpeechResponseFormat>(
|
||||
&self.response_format,
|
||||
)?;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user