Implement Venice.ai provider
This commit is contained in:
@@ -109,6 +109,9 @@ fn create_controller_from_provider_and_json_value_config(
|
||||
AgentProvider::TogetherAI => {
|
||||
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
|
||||
}
|
||||
AgentProvider::Venice => {
|
||||
provider::venice::create_controller_from_yaml_value_config(agent_id, config)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -150,5 +153,9 @@ pub fn default_config_for_provider(provider: &AgentProvider) -> serde_yaml_ng::V
|
||||
let config = super::provider::togetherai::default_config();
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::Venice => {
|
||||
let config = super::provider::venice::default_config();
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -61,6 +61,7 @@ pub enum ControllerType {
|
||||
OpenAI(Box<super::openai::Controller>),
|
||||
OpenAICompat(Box<super::openai_compat::Controller>),
|
||||
Anthropic(Box<super::anthropic::Controller>),
|
||||
Venice(Box<super::venice::Controller>),
|
||||
}
|
||||
|
||||
impl ControllerTrait for ControllerType {
|
||||
@@ -69,6 +70,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.supports_purpose(purpose),
|
||||
ControllerType::OpenAICompat(controller) => controller.supports_purpose(purpose),
|
||||
ControllerType::Anthropic(controller) => controller.supports_purpose(purpose),
|
||||
ControllerType::Venice(controller) => controller.supports_purpose(purpose),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -77,6 +79,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_generation_model_id(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_generation_model_id(),
|
||||
ControllerType::Anthropic(controller) => controller.text_generation_model_id(),
|
||||
ControllerType::Venice(controller) => controller.text_generation_model_id(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -85,6 +88,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_generation_prompt(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_generation_prompt(),
|
||||
ControllerType::Anthropic(controller) => controller.text_generation_prompt(),
|
||||
ControllerType::Venice(controller) => controller.text_generation_prompt(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -93,6 +97,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_to_speech_voice(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_to_speech_voice(),
|
||||
ControllerType::Anthropic(controller) => controller.text_to_speech_voice(),
|
||||
ControllerType::Venice(controller) => controller.text_to_speech_voice(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -101,6 +106,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_to_speech_speed(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_to_speech_speed(),
|
||||
ControllerType::Anthropic(controller) => controller.text_to_speech_speed(),
|
||||
ControllerType::Venice(controller) => controller.text_to_speech_speed(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -109,6 +115,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_generation_temperature(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_generation_temperature(),
|
||||
ControllerType::Anthropic(controller) => controller.text_generation_temperature(),
|
||||
ControllerType::Venice(controller) => controller.text_generation_temperature(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -117,6 +124,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.ping().await,
|
||||
ControllerType::OpenAICompat(controller) => controller.ping().await,
|
||||
ControllerType::Anthropic(controller) => controller.ping().await,
|
||||
ControllerType::Venice(controller) => controller.ping().await,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -135,6 +143,9 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.generate_text(conversation, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.generate_text(conversation, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -154,6 +165,9 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.speech_to_text(mime_type, media, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.speech_to_text(mime_type, media, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -170,6 +184,9 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.generate_image(prompt, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.generate_image(prompt, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -189,6 +206,9 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -203,6 +223,7 @@ impl ControllerTrait for ControllerType {
|
||||
controller.text_to_speech(text, params).await
|
||||
}
|
||||
ControllerType::Anthropic(controller) => controller.text_to_speech(text, params).await,
|
||||
ControllerType::Venice(controller) => controller.text_to_speech(text, params).await,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ pub enum AgentProvider {
|
||||
OpenAICompat,
|
||||
OpenRouter,
|
||||
TogetherAI,
|
||||
Venice,
|
||||
}
|
||||
|
||||
impl AgentProvider {
|
||||
@@ -25,6 +26,7 @@ impl AgentProvider {
|
||||
&Self::OpenAICompat,
|
||||
&Self::OpenRouter,
|
||||
&Self::TogetherAI,
|
||||
&Self::Venice,
|
||||
]
|
||||
}
|
||||
|
||||
@@ -39,6 +41,7 @@ impl AgentProvider {
|
||||
Self::OpenAICompat => "openai-compatible",
|
||||
Self::OpenRouter => "openrouter",
|
||||
Self::TogetherAI => "together-ai",
|
||||
Self::Venice => "venice",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -53,6 +56,7 @@ impl AgentProvider {
|
||||
"openai-compatible" => Ok(Self::OpenAICompat),
|
||||
"openrouter" => Ok(Self::OpenRouter),
|
||||
"together-ai" => Ok(Self::TogetherAI),
|
||||
"venice" => Ok(Self::Venice),
|
||||
_ => Err("Unexpected string value"),
|
||||
}
|
||||
}
|
||||
@@ -181,6 +185,25 @@ impl AgentProvider {
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::Venice => AgentProviderInfo {
|
||||
id: Self::Venice.to_static_str(),
|
||||
name: "Venice",
|
||||
description: "Venice AI runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses. It serves frontier proprietary and open-source models with text-generation (including vision), speech-to-text, text-to-speech, native image generation and editing, and native web search.",
|
||||
homepage_url: Some("https://venice.ai"),
|
||||
wiki_url: None,
|
||||
sign_up_url: Some("https://venice.ai"),
|
||||
models_list_url: Some("https://api.venice.ai/api/v1/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::ImageGeneration,
|
||||
AgentPurpose::TextGeneration,
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: true,
|
||||
// Venice does native web search via `venice_parameters`, NOT baibot's built-in
|
||||
// tools mechanism (the OpenAI web_search/code_interpreter block), so this is false.
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ pub mod openai;
|
||||
pub mod openai_compat;
|
||||
pub(super) mod openrouter;
|
||||
pub(super) mod togetherai;
|
||||
pub mod venice;
|
||||
|
||||
fn default_temperature() -> f32 {
|
||||
1.0
|
||||
|
||||
156
src/agent/provider/venice/audio.rs
Normal file
156
src/agent/provider/venice/audio.rs
Normal file
@@ -0,0 +1,156 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{TextToSpeechParams, TextToSpeechResult};
|
||||
use crate::agent::provider::{SpeechToTextParams, SpeechToTextResult};
|
||||
use crate::strings;
|
||||
|
||||
use super::config::Config;
|
||||
use super::wire::{SpeechRequest, TranscriptionResponse};
|
||||
|
||||
pub async fn speech_to_text(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
mime_type: &mxlink::mime::Mime,
|
||||
media: Vec<u8>,
|
||||
params: SpeechToTextParams,
|
||||
) -> anyhow::Result<SpeechToTextResult> {
|
||||
let Some(speech_to_text_config) = &config.speech_to_text else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::SpeechToText
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
// Unlike the openai_compat path (which writes the audio to a temp file because its library
|
||||
// can't take bytes), reqwest's multipart takes the bytes directly.
|
||||
let part = reqwest::multipart::Part::bytes(media)
|
||||
.file_name("audio")
|
||||
.mime_str(mime_type.as_ref())?;
|
||||
|
||||
let mut form = reqwest::multipart::Form::new()
|
||||
.part("file", part)
|
||||
.text("model", speech_to_text_config.model_id.clone())
|
||||
.text("response_format", "json");
|
||||
|
||||
if let Some(language) = ¶ms.language_override {
|
||||
form = form.text("language", language.clone());
|
||||
}
|
||||
|
||||
let url = format!(
|
||||
"{}/audio/transcriptions",
|
||||
config.base_url.trim_end_matches('/')
|
||||
);
|
||||
|
||||
tracing::trace!(
|
||||
model_id = speech_to_text_config.model_id,
|
||||
language = ?params.language_override,
|
||||
"Sending Venice audio transcription API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.multipart(form)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice audio transcription request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice audio transcription request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
let response: TranscriptionResponse = response.json().await?;
|
||||
|
||||
Ok(SpeechToTextResult {
|
||||
text: response.text,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn text_to_speech(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
input: &str,
|
||||
params: TextToSpeechParams,
|
||||
) -> anyhow::Result<TextToSpeechResult> {
|
||||
let Some(text_to_speech_config) = &config.text_to_speech else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::TextToSpeech
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
// Per-call overrides win over the configured defaults.
|
||||
let voice = params
|
||||
.voice_override
|
||||
.or_else(|| text_to_speech_config.voice.clone());
|
||||
let speed = params.speed_override.or(text_to_speech_config.speed);
|
||||
|
||||
let response_format = text_to_speech_config.response_format.clone();
|
||||
let mime_type = response_format_to_mime_type(response_format.as_deref());
|
||||
|
||||
let request = SpeechRequest {
|
||||
model: text_to_speech_config.model_id.clone(),
|
||||
input: input.to_owned(),
|
||||
voice,
|
||||
speed,
|
||||
response_format,
|
||||
prompt: text_to_speech_config.prompt.clone(),
|
||||
temperature: text_to_speech_config.temperature,
|
||||
top_p: text_to_speech_config.top_p,
|
||||
};
|
||||
|
||||
let url = format!("{}/audio/speech", config.base_url.trim_end_matches('/'));
|
||||
|
||||
tracing::trace!(
|
||||
model_id = text_to_speech_config.model_id,
|
||||
voice = ?request.voice,
|
||||
"Sending Venice text-to-speech API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice text-to-speech request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice text-to-speech request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
// The speech endpoint answers with raw binary audio; read the body directly.
|
||||
let bytes = response.bytes().await?.to_vec();
|
||||
|
||||
Ok(TextToSpeechResult { bytes, mime_type })
|
||||
}
|
||||
|
||||
/// Map a Venice TTS `response_format` to its MIME type. Defaults to `audio/mpeg` (the
|
||||
/// IANA-registered MP3 type, RFC 3003) when the format is unset, matching Venice's own `mp3`
|
||||
/// default. This deliberately uses `audio/mpeg` rather than the `audio/mp3` alias the openai
|
||||
/// provider emits; baibot's downstream audio-filename mapping treats both as `.mp3`.
|
||||
fn response_format_to_mime_type(response_format: Option<&str>) -> mxlink::mime::Mime {
|
||||
let raw = match response_format.unwrap_or("mp3") {
|
||||
"mp3" => "audio/mpeg",
|
||||
"opus" => "audio/ogg",
|
||||
"aac" => "audio/aac",
|
||||
"flac" => "audio/flac",
|
||||
"wav" => "audio/wav",
|
||||
"pcm" => "audio/L8",
|
||||
_ => "audio/mpeg",
|
||||
};
|
||||
|
||||
raw.parse()
|
||||
.unwrap_or(mxlink::mime::APPLICATION_OCTET_STREAM)
|
||||
}
|
||||
122
src/agent/provider/venice/chat.rs
Normal file
122
src/agent/provider/venice/chat.rs
Normal file
@@ -0,0 +1,122 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
use super::config::Config;
|
||||
use super::utils::convert_llm_messages_to_venice;
|
||||
use super::wire::{ChatCompletionRequest, ChatCompletionResponse};
|
||||
|
||||
pub async fn generate_text(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
conversation: LLMConversation,
|
||||
params: TextGenerationParams,
|
||||
) -> anyhow::Result<TextGenerationResult> {
|
||||
let Some(text_generation_config) = &config.text_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::TextGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
let prompt_text = params.prompt_variables.format(
|
||||
params
|
||||
.prompt_override
|
||||
.unwrap_or(text_generation_config.prompt.clone().unwrap_or_default())
|
||||
.trim(),
|
||||
);
|
||||
|
||||
let prompt_message = if prompt_text.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
|
||||
let mut conversation_messages = conversation.messages;
|
||||
|
||||
if params.context_management_enabled {
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
text_generation_config.max_context_tokens,
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(prompt_message) = prompt_message {
|
||||
conversation_messages.insert(0, prompt_message);
|
||||
}
|
||||
|
||||
let messages = convert_llm_messages_to_venice(conversation_messages);
|
||||
|
||||
let temperature = params
|
||||
.temperature_override
|
||||
.unwrap_or(text_generation_config.temperature);
|
||||
|
||||
let request = ChatCompletionRequest {
|
||||
model: text_generation_config.model_id.clone(),
|
||||
messages,
|
||||
temperature: Some(temperature),
|
||||
// Web search rides entirely inside `venice_parameters`; there is no `tools` array here.
|
||||
// `max_tokens` is deprecated on Venice in favor of `max_completion_tokens`.
|
||||
max_completion_tokens: text_generation_config.max_response_tokens,
|
||||
venice_parameters: text_generation_config.venice_parameters.clone(),
|
||||
};
|
||||
|
||||
let url = format!(
|
||||
"{}/chat/completions",
|
||||
config.base_url.trim_end_matches('/')
|
||||
);
|
||||
|
||||
tracing::trace!(
|
||||
model = text_generation_config.model_id,
|
||||
messages_count = request.messages.len(),
|
||||
"Sending Venice chat completion API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Log the body server-side for debugging (Venice explains a rejected strict body there),
|
||||
// but keep it OUT of the returned error: that error surfaces in the Matrix room, and the
|
||||
// body can carry account / rate-limit details that shouldn't reach room members.
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice chat completion request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice chat completion request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
let response: ChatCompletionResponse = response.json().await?;
|
||||
|
||||
let Some(choice) = response.choices.into_iter().next() else {
|
||||
return Err(anyhow::anyhow!(
|
||||
"No choices were returned from the Venice chat completion API"
|
||||
));
|
||||
};
|
||||
|
||||
let Some(content) = choice.message.content else {
|
||||
return Err(anyhow::anyhow!(
|
||||
"No message content was returned from the Venice chat completion API"
|
||||
));
|
||||
};
|
||||
|
||||
Ok(TextGenerationResult { text: content })
|
||||
}
|
||||
355
src/agent/provider/venice/config.rs
Normal file
355
src/agent/provider/venice/config.rs
Normal file
@@ -0,0 +1,355 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::agent::{default_prompt, provider::ConfigTrait};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct Config {
|
||||
pub base_url: String,
|
||||
|
||||
pub api_key: String,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub text_generation: Option<TextGenerationConfig>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub speech_to_text: Option<SpeechToTextConfig>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub text_to_speech: Option<TextToSpeechConfig>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub image_generation: Option<ImageGenerationConfig>,
|
||||
}
|
||||
|
||||
impl Default for Config {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
base_url: "https://api.venice.ai/api/v1".to_owned(),
|
||||
api_key: "YOUR_API_KEY_HERE".to_owned(),
|
||||
text_generation: Some(TextGenerationConfig::default()),
|
||||
speech_to_text: Some(SpeechToTextConfig::default()),
|
||||
text_to_speech: Some(TextToSpeechConfig::default()),
|
||||
image_generation: Some(ImageGenerationConfig::default()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ConfigTrait for Config {
|
||||
fn validate(&self) -> Result<(), String> {
|
||||
if self.base_url.is_empty() {
|
||||
return Err("The base URL must not be empty.".to_owned());
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextGenerationConfig {
|
||||
#[serde(default = "default_text_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default)]
|
||||
pub prompt: Option<String>,
|
||||
|
||||
#[serde(default = "super::super::default_temperature")]
|
||||
pub temperature: f32,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_response_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_context_tokens: u32,
|
||||
|
||||
/// Venice-specific request knobs, serialized 1:1 into the `venice_parameters` bag on the
|
||||
/// wire. Any unset field is omitted, so Venice applies its own server-side default.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub venice_parameters: Option<VeniceParameters>,
|
||||
}
|
||||
|
||||
impl Default for TextGenerationConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_text_model_id(),
|
||||
prompt: Some(default_prompt().to_owned()),
|
||||
temperature: super::super::default_temperature(),
|
||||
// Reserved output budget: sent as the response cap AND subtracted from the context
|
||||
// window when trimming history. Mirrors the openai_compat sibling's default.
|
||||
max_response_tokens: Some(4096),
|
||||
// Matches Venice's own `availableContextTokens` (131072) and the non-OpenAI sibling
|
||||
// providers (ollama/localai/mistral all default to 128_000).
|
||||
max_context_tokens: 128_000,
|
||||
// A usable starting point, not an everything-set dump: only these three are sent;
|
||||
// every other knob stays None so Venice applies its own default (omitting != false).
|
||||
venice_parameters: Some(VeniceParameters {
|
||||
enable_web_search: Some(WebSearchMode::Auto),
|
||||
strip_thinking_response: Some(true),
|
||||
enable_e2ee: Some(false),
|
||||
..Default::default()
|
||||
}),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_model_id() -> String {
|
||||
"kimi-k2-5".to_owned()
|
||||
}
|
||||
|
||||
/// The full `venice_parameters` knob set, mirroring Venice's `ChatCompletionRequest`
|
||||
/// schema field-for-field. Every field is optional with `skip_serializing_if`, so the
|
||||
/// request never carries a knob the user didn't set (the body is `additionalProperties: false`,
|
||||
/// and an unset knob simply omits rather than sending `null`).
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||
pub struct VeniceParameters {
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_search: Option<WebSearchMode>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_citations: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_scraping: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub include_venice_system_prompt: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub include_search_results_in_stream: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub return_search_results_as_documents: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_x_search: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_e2ee: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub character_slug: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub strip_thinking_response: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub disable_thinking: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum WebSearchMode {
|
||||
Auto,
|
||||
On,
|
||||
Off,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct SpeechToTextConfig {
|
||||
#[serde(default = "default_speech_to_text_model_id")]
|
||||
pub model_id: String,
|
||||
}
|
||||
|
||||
impl Default for SpeechToTextConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_speech_to_text_model_id(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_speech_to_text_model_id() -> String {
|
||||
"nvidia/parakeet-tdt-0.6b-v3".to_owned()
|
||||
}
|
||||
|
||||
/// `/audio/speech` (`CreateSpeechRequestSchema`) request knobs. Only `model_id` is required on
|
||||
/// the wire; everything else is optional with `skip_serializing_if` so an unset knob is omitted
|
||||
/// rather than sent as `null` (the body is `additionalProperties: false`). `voice` is a free
|
||||
/// `Option<String>`, not a closed enum: Venice's voice set spans dozens of model-specific names
|
||||
/// plus arbitrary cloned-voice handles (`vv_<id>`), so an enum would reject valid handles.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextToSpeechConfig {
|
||||
#[serde(default = "default_text_to_speech_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(
|
||||
default = "default_text_to_speech_voice",
|
||||
skip_serializing_if = "Option::is_none"
|
||||
)]
|
||||
pub voice: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub speed: Option<f32>,
|
||||
|
||||
#[serde(
|
||||
default = "default_text_to_speech_response_format",
|
||||
skip_serializing_if = "Option::is_none"
|
||||
)]
|
||||
pub response_format: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub prompt: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub temperature: Option<f32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub top_p: Option<f32>,
|
||||
}
|
||||
|
||||
impl Default for TextToSpeechConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_text_to_speech_model_id(),
|
||||
voice: default_text_to_speech_voice(),
|
||||
speed: None,
|
||||
response_format: default_text_to_speech_response_format(),
|
||||
prompt: None,
|
||||
temperature: None,
|
||||
top_p: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_to_speech_model_id() -> String {
|
||||
"tts-kokoro".to_owned()
|
||||
}
|
||||
|
||||
fn default_text_to_speech_voice() -> Option<String> {
|
||||
Some("af_sky".to_owned())
|
||||
}
|
||||
|
||||
fn default_text_to_speech_response_format() -> Option<String> {
|
||||
Some("mp3".to_owned())
|
||||
}
|
||||
|
||||
/// `/image/generate` (`GenerateImageRequest`) request knobs, mirroring Venice's schema
|
||||
/// field-for-field. Only `model_id` is required; every other knob is optional with
|
||||
/// `skip_serializing_if` so unset knobs are omitted (the body is `additionalProperties: false`).
|
||||
/// The full knob set is deliberate: the native `/image/generate` endpoint is the flagship's
|
||||
/// reason to exist over the knob-dropping OpenAI-compat path, so the knobs ARE the feature.
|
||||
/// The deprecated `inpaint` knob is intentionally absent.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct ImageGenerationConfig {
|
||||
#[serde(default = "default_image_generation_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub negative_prompt: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub cfg_scale: Option<f32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub steps: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub style_preset: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub seed: Option<i64>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub hide_watermark: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub format: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub width: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub height: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub quality: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub lora_strength: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub embed_exif_metadata: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_search: Option<bool>,
|
||||
|
||||
/// Image-edit settings, nested here because baibot has a single `ImageGeneration` purpose
|
||||
/// and edit shares its config gate. The gen and edit model sets are disjoint, so edit
|
||||
/// carries its own model field.
|
||||
#[serde(default)]
|
||||
pub edit: ImageEditSettings,
|
||||
}
|
||||
|
||||
impl Default for ImageGenerationConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_image_generation_model_id(),
|
||||
negative_prompt: None,
|
||||
cfg_scale: None,
|
||||
steps: None,
|
||||
style_preset: None,
|
||||
seed: None,
|
||||
safe_mode: None,
|
||||
hide_watermark: None,
|
||||
format: None,
|
||||
width: None,
|
||||
height: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
quality: None,
|
||||
lora_strength: None,
|
||||
embed_exif_metadata: None,
|
||||
enable_web_search: None,
|
||||
edit: ImageEditSettings::default(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_image_generation_model_id() -> String {
|
||||
"chroma".to_owned()
|
||||
}
|
||||
|
||||
/// `/image/edit` (`EditImageRequest`) request knobs, mirroring Venice's schema. The source image
|
||||
/// and prompt are supplied per-call (not config), so only the model and the output-shaping knobs
|
||||
/// live here. Each knob is optional with `skip_serializing_if`.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct ImageEditSettings {
|
||||
#[serde(default = "default_image_edit_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub output_format: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
}
|
||||
|
||||
impl Default for ImageEditSettings {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_image_edit_model_id(),
|
||||
output_format: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
safe_mode: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_image_edit_model_id() -> String {
|
||||
"firered-image-edit".to_owned()
|
||||
}
|
||||
146
src/agent/provider/venice/controller.rs
Normal file
146
src/agent/provider/venice/controller.rs
Normal file
@@ -0,0 +1,146 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
|
||||
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
|
||||
};
|
||||
use crate::agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use super::config::Config;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Controller {
|
||||
config: Config,
|
||||
http: reqwest::Client,
|
||||
}
|
||||
|
||||
impl Controller {
|
||||
pub fn new(config: Config) -> Self {
|
||||
// Image generation and text-to-speech can run long, so give the client a generous timeout
|
||||
// instead of reqwest's default (none). `build` only fails on TLS/system init; fall back to
|
||||
// the infallible `Client::new()` so this constructor stays infallible.
|
||||
let http = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(120))
|
||||
.build()
|
||||
.unwrap_or_else(|_| reqwest::Client::new());
|
||||
|
||||
Self { config, http }
|
||||
}
|
||||
}
|
||||
|
||||
impl ControllerTrait for Controller {
|
||||
async fn ping(&self) -> anyhow::Result<PingResult> {
|
||||
if !self.supports_purpose(AgentPurpose::TextGeneration) {
|
||||
return Ok(PingResult::Inconclusive);
|
||||
}
|
||||
|
||||
// Mirror the openai/openai_compat ping: a real "Hello!" round-trip exercises the strict
|
||||
// /chat/completions body and auth, so a successful ping proves text generation works.
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
let conversation = LLMConversation { messages };
|
||||
|
||||
self.generate_text(conversation, TextGenerationParams::default())
|
||||
.await?;
|
||||
|
||||
Ok(PingResult::Successful)
|
||||
}
|
||||
|
||||
async fn generate_text(
|
||||
&self,
|
||||
conversation: LLMConversation,
|
||||
params: TextGenerationParams,
|
||||
) -> anyhow::Result<TextGenerationResult> {
|
||||
super::chat::generate_text(&self.config, &self.http, conversation, params).await
|
||||
}
|
||||
|
||||
async fn speech_to_text(
|
||||
&self,
|
||||
mime_type: &mxlink::mime::Mime,
|
||||
media: Vec<u8>,
|
||||
params: SpeechToTextParams,
|
||||
) -> anyhow::Result<SpeechToTextResult> {
|
||||
super::audio::speech_to_text(&self.config, &self.http, mime_type, media, params).await
|
||||
}
|
||||
|
||||
async fn generate_image(
|
||||
&self,
|
||||
prompt: &str,
|
||||
params: ImageGenerationParams,
|
||||
) -> anyhow::Result<ImageGenerationResult> {
|
||||
super::images::generate_image(&self.config, &self.http, prompt, params).await
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
super::images::create_image_edit(&self.config, &self.http, prompt, images, params).await
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
input: &str,
|
||||
params: TextToSpeechParams,
|
||||
) -> anyhow::Result<TextToSpeechResult> {
|
||||
super::audio::text_to_speech(&self.config, &self.http, input, params).await
|
||||
}
|
||||
|
||||
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
|
||||
match purpose {
|
||||
AgentPurpose::TextGeneration => self.config.text_generation.is_some(),
|
||||
AgentPurpose::SpeechToText => self.config.speech_to_text.is_some(),
|
||||
AgentPurpose::TextToSpeech => self.config.text_to_speech.is_some(),
|
||||
AgentPurpose::ImageGeneration => self.config.image_generation.is_some(),
|
||||
AgentPurpose::CatchAll => true,
|
||||
}
|
||||
}
|
||||
|
||||
fn text_generation_model_id(&self) -> Option<String> {
|
||||
self.config
|
||||
.text_generation
|
||||
.as_ref()
|
||||
.map(|config| config.model_id.to_owned())
|
||||
}
|
||||
|
||||
fn text_generation_prompt(&self) -> Option<String> {
|
||||
self.config
|
||||
.text_generation
|
||||
.as_ref()
|
||||
.and_then(|config| config.prompt.clone())
|
||||
}
|
||||
|
||||
fn text_generation_temperature(&self) -> Option<f32> {
|
||||
self.config
|
||||
.text_generation
|
||||
.as_ref()
|
||||
.map(|config| config.temperature)
|
||||
}
|
||||
|
||||
fn text_to_speech_voice(&self) -> Option<String> {
|
||||
self.config
|
||||
.text_to_speech
|
||||
.as_ref()
|
||||
.and_then(|config| config.voice.clone())
|
||||
}
|
||||
|
||||
fn text_to_speech_speed(&self) -> Option<f32> {
|
||||
self.config
|
||||
.text_to_speech
|
||||
.as_ref()
|
||||
.and_then(|config| config.speed)
|
||||
}
|
||||
}
|
||||
192
src/agent/provider/venice/images.rs
Normal file
192
src/agent/provider/venice/images.rs
Normal file
@@ -0,0 +1,192 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{ImageEditResult, ImageGenerationResult, ImageSource};
|
||||
use crate::agent::provider::{ImageEditParams, ImageGenerationParams};
|
||||
use crate::strings;
|
||||
use crate::utils::base64::{base64_decode, base64_encode};
|
||||
|
||||
use super::config::Config;
|
||||
use super::wire::{EditImageRequest, GenerateImageRequest, GenerateImageResponse};
|
||||
|
||||
/// Generate an image via Venice's native `/image/generate` endpoint.
|
||||
///
|
||||
/// This is the base64-in-JSON path: we pin `return_binary: false` so Venice answers with a JSON
|
||||
/// envelope (`GenerateImageResponse`) carrying the image as a base64 string, which we decode. The
|
||||
/// sibling `create_image_edit` is the *other* response shape (raw binary); the two must not be
|
||||
/// crossed. `params` is advisory only; the Venice config drives the request.
|
||||
pub async fn generate_image(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
prompt: &str,
|
||||
_params: ImageGenerationParams,
|
||||
) -> anyhow::Result<ImageGenerationResult> {
|
||||
let Some(image_generation_config) = &config.image_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::ImageGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
let request = GenerateImageRequest {
|
||||
model: image_generation_config.model_id.clone(),
|
||||
prompt: prompt.to_owned(),
|
||||
// Pinned: baibot wants exactly one image, returned as base64-in-JSON so `GenerateImageResponse`
|
||||
// can decode it. Flipping `return_binary` would make Venice answer with raw binary and break
|
||||
// the JSON decode below, so neither knob is configurable.
|
||||
return_binary: false,
|
||||
variants: 1,
|
||||
negative_prompt: image_generation_config.negative_prompt.clone(),
|
||||
cfg_scale: image_generation_config.cfg_scale,
|
||||
steps: image_generation_config.steps,
|
||||
style_preset: image_generation_config.style_preset.clone(),
|
||||
seed: image_generation_config.seed,
|
||||
safe_mode: image_generation_config.safe_mode,
|
||||
hide_watermark: image_generation_config.hide_watermark,
|
||||
format: image_generation_config.format.clone(),
|
||||
width: image_generation_config.width,
|
||||
height: image_generation_config.height,
|
||||
aspect_ratio: image_generation_config.aspect_ratio.clone(),
|
||||
resolution: image_generation_config.resolution.clone(),
|
||||
quality: image_generation_config.quality.clone(),
|
||||
lora_strength: image_generation_config.lora_strength,
|
||||
embed_exif_metadata: image_generation_config.embed_exif_metadata,
|
||||
enable_web_search: image_generation_config.enable_web_search,
|
||||
};
|
||||
|
||||
let url = format!("{}/image/generate", config.base_url.trim_end_matches('/'));
|
||||
|
||||
// The prompt is user content; keep it out of logs (mirrors the STT/TTS paths).
|
||||
tracing::trace!(
|
||||
model_id = image_generation_config.model_id,
|
||||
"Sending Venice image generation API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice image generation request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice image generation request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
let response: GenerateImageResponse = response.json().await?;
|
||||
|
||||
tracing::trace!(request_id = ?response.id, "Venice image generation succeeded");
|
||||
|
||||
let Some(image_base64) = response.images.into_iter().next() else {
|
||||
return Err(anyhow::anyhow!(
|
||||
"The Venice image generation API returned no images"
|
||||
));
|
||||
};
|
||||
|
||||
// Swallow the decode error's detail (it can echo input bytes/offsets); the returned error
|
||||
// reaches the Matrix room, so it stays generic while the real cause goes to the server log.
|
||||
let bytes = base64_decode(&image_base64).map_err(|decode_err| {
|
||||
tracing::warn!(%decode_err, "Venice image generation returned undecodable base64");
|
||||
anyhow::anyhow!("Venice image generation returned invalid base64 image data")
|
||||
})?;
|
||||
|
||||
Ok(ImageGenerationResult {
|
||||
bytes,
|
||||
mime_type: image_format_to_mime_type(image_generation_config.format.as_deref()),
|
||||
revised_prompt: None,
|
||||
})
|
||||
}
|
||||
|
||||
/// Edit an image via Venice's native `/image/edit` endpoint.
|
||||
///
|
||||
/// This is the raw-binary path: the request is JSON carrying the source image as a base64 string
|
||||
/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64, no multipart), and the
|
||||
/// response body IS the edited image bytes (no JSON envelope). `params` is advisory only.
|
||||
pub async fn create_image_edit(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
let Some(image_generation_config) = &config.image_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::ImageGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
let edit_config = &image_generation_config.edit;
|
||||
|
||||
let Some(source) = images.into_iter().next() else {
|
||||
return Err(anyhow::anyhow!("No image sources provided"));
|
||||
};
|
||||
|
||||
let request = EditImageRequest {
|
||||
model: edit_config.model_id.clone(),
|
||||
prompt: prompt.to_owned(),
|
||||
image: base64_encode(&source.bytes),
|
||||
output_format: edit_config.output_format.clone(),
|
||||
aspect_ratio: edit_config.aspect_ratio.clone(),
|
||||
resolution: edit_config.resolution.clone(),
|
||||
safe_mode: edit_config.safe_mode,
|
||||
};
|
||||
|
||||
let url = format!("{}/image/edit", config.base_url.trim_end_matches('/'));
|
||||
|
||||
// The prompt is user content; keep it out of logs (mirrors the STT/TTS paths).
|
||||
tracing::trace!(
|
||||
model_id = edit_config.model_id,
|
||||
"Sending Venice image edit API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice image edit request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice image edit request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
// The edit endpoint answers with raw binary image bytes, so read the body directly instead of
|
||||
// parsing JSON. The actual format comes from the response Content-Type header; fall back to the
|
||||
// configured `output_format` when the header is missing or unparseable.
|
||||
let mime_type = response
|
||||
.headers()
|
||||
.get(reqwest::header::CONTENT_TYPE)
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<mxlink::mime::Mime>().ok())
|
||||
.unwrap_or_else(|| image_format_to_mime_type(edit_config.output_format.as_deref()));
|
||||
|
||||
let bytes = response.bytes().await?.to_vec();
|
||||
|
||||
Ok(ImageEditResult { bytes, mime_type })
|
||||
}
|
||||
|
||||
/// Map a Venice image `format`/`output_format` value (`jpeg`/`png`/`webp`) to its MIME type.
|
||||
/// Venice defaults to `webp` when the format is unset, so an absent value maps to `image/webp`.
|
||||
fn image_format_to_mime_type(format: Option<&str>) -> mxlink::mime::Mime {
|
||||
match format.unwrap_or("webp") {
|
||||
"jpeg" | "jpg" => mxlink::mime::IMAGE_JPEG,
|
||||
"png" => mxlink::mime::IMAGE_PNG,
|
||||
// No mxlink::mime constant for webp; parse it, falling back to PNG on any surprise value.
|
||||
_ => "image/webp"
|
||||
.parse()
|
||||
.unwrap_or(mxlink::mime::IMAGE_PNG),
|
||||
}
|
||||
}
|
||||
47
src/agent/provider/venice/mod.rs
Normal file
47
src/agent/provider/venice/mod.rs
Normal file
@@ -0,0 +1,47 @@
|
||||
mod audio;
|
||||
mod chat;
|
||||
mod config;
|
||||
mod controller;
|
||||
mod images;
|
||||
mod utils;
|
||||
mod wire;
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests;
|
||||
|
||||
pub use config::Config;
|
||||
pub use controller::Controller;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> AgentInstantiationResult<ControllerType> {
|
||||
let config = match &config {
|
||||
serde_yaml_ng::Value::Mapping(_) => {
|
||||
let config: Config =
|
||||
serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
|
||||
config
|
||||
.validate()
|
||||
.map_err(AgentInstantiationError::ConfigFailsValidation)?;
|
||||
|
||||
config
|
||||
}
|
||||
_ => {
|
||||
return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping(
|
||||
agent_id.to_owned(),
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
Ok(ControllerType::Venice(Box::new(Controller::new(config))))
|
||||
}
|
||||
|
||||
pub fn default_config() -> Config {
|
||||
Config::default()
|
||||
}
|
||||
269
src/agent/provider/venice/tests.rs
Normal file
269
src/agent/provider/venice/tests.rs
Normal file
@@ -0,0 +1,269 @@
|
||||
use mxlink::matrix_sdk::ruma::OwnedMxcUri;
|
||||
use mxlink::matrix_sdk::ruma::events::room::message::{
|
||||
FileMessageEventContent, ImageMessageEventContent,
|
||||
};
|
||||
use mxlink::mime;
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, FileDetails, ImageDetails, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
use super::config::{Config, VeniceParameters, WebSearchMode};
|
||||
use super::controller::Controller;
|
||||
use super::utils::convert_llm_messages_to_venice;
|
||||
use super::wire::{
|
||||
ContentPart, EditImageRequest, GenerateImageRequest, MessageContent, SpeechRequest,
|
||||
};
|
||||
|
||||
#[test]
|
||||
fn config_round_trips_with_venice_parameters() {
|
||||
let yaml = r#"
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: test-key
|
||||
text_generation:
|
||||
model_id: kimi-k2-5
|
||||
temperature: 0.7
|
||||
max_response_tokens: 1024
|
||||
max_context_tokens: 65536
|
||||
venice_parameters:
|
||||
enable_web_search: "auto"
|
||||
enable_web_citations: true
|
||||
speech_to_text:
|
||||
model_id: nvidia/parakeet-tdt-0.6b-v3
|
||||
"#;
|
||||
|
||||
let config: Config = serde_yaml_ng::from_str(yaml).expect("config should deserialize");
|
||||
|
||||
let tg = config.text_generation.expect("text_generation present");
|
||||
let vp = tg.venice_parameters.expect("venice_parameters present");
|
||||
|
||||
assert!(matches!(vp.enable_web_search, Some(WebSearchMode::Auto)));
|
||||
assert_eq!(vp.enable_web_citations, Some(true));
|
||||
assert_eq!(vp.character_slug, None);
|
||||
|
||||
// The bag must serialize the enum to the exact wire string, and an unset knob must be ABSENT
|
||||
// (not `null`) so the strict `additionalProperties: false` body is honored.
|
||||
let json = serde_json::to_string(&vp).expect("serialize venice_parameters");
|
||||
assert!(
|
||||
json.contains("\"enable_web_search\":\"auto\""),
|
||||
"web search should be the literal \"auto\": {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("character_slug"),
|
||||
"an unset knob must be omitted entirely: {json}"
|
||||
);
|
||||
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_image_to_data_uri_and_skips_files() {
|
||||
let messages = vec![
|
||||
LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
timestamp: chrono::Utc::now(),
|
||||
content: LLMMessageContent::Text("describe this".to_owned()),
|
||||
},
|
||||
LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
timestamp: chrono::Utc::now(),
|
||||
content: LLMMessageContent::Image(ImageDetails::new(
|
||||
ImageMessageEventContent::plain(
|
||||
"pic.png".to_owned(),
|
||||
OwnedMxcUri::from("mxc://example.com/abc"),
|
||||
),
|
||||
mime::IMAGE_PNG,
|
||||
vec![1, 2, 3],
|
||||
)),
|
||||
},
|
||||
LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
timestamp: chrono::Utc::now(),
|
||||
content: LLMMessageContent::File(FileDetails::new(
|
||||
FileMessageEventContent::plain(
|
||||
"doc.pdf".to_owned(),
|
||||
OwnedMxcUri::from("mxc://example.com/def"),
|
||||
),
|
||||
mime::APPLICATION_PDF,
|
||||
vec![4, 5, 6],
|
||||
)),
|
||||
},
|
||||
];
|
||||
|
||||
let converted = convert_llm_messages_to_venice(messages);
|
||||
|
||||
// Text and image survive; the file is warn-skipped.
|
||||
assert_eq!(converted.len(), 2);
|
||||
|
||||
match &converted[0].content {
|
||||
MessageContent::Text(text) => assert_eq!(text, "describe this"),
|
||||
other => panic!("expected bare text, got {other:?}"),
|
||||
}
|
||||
|
||||
match &converted[1].content {
|
||||
MessageContent::Parts(parts) => match &parts[0] {
|
||||
ContentPart::ImageUrl { image_url } => assert!(
|
||||
image_url.url.starts_with("data:image/png;base64,"),
|
||||
"image should be inlined as a data URI: {}",
|
||||
image_url.url
|
||||
),
|
||||
},
|
||||
other => panic!("expected image parts, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn supports_purpose_truth_table() {
|
||||
let config: Config = serde_yaml_ng::from_str(
|
||||
r#"
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: test-key
|
||||
text_generation:
|
||||
model_id: kimi-k2-5
|
||||
speech_to_text:
|
||||
model_id: nvidia/parakeet-tdt-0.6b-v3
|
||||
"#,
|
||||
)
|
||||
.expect("config should deserialize");
|
||||
|
||||
let controller = Controller::new(config);
|
||||
|
||||
assert!(controller.supports_purpose(AgentPurpose::TextGeneration));
|
||||
assert!(controller.supports_purpose(AgentPurpose::SpeechToText));
|
||||
assert!(controller.supports_purpose(AgentPurpose::CatchAll));
|
||||
assert!(!controller.supports_purpose(AgentPurpose::TextToSpeech));
|
||||
assert!(!controller.supports_purpose(AgentPurpose::ImageGeneration));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn supports_purpose_true_when_image_and_tts_blocks_present() {
|
||||
let config: Config = serde_yaml_ng::from_str(
|
||||
r#"
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: test-key
|
||||
text_to_speech:
|
||||
model_id: tts-kokoro
|
||||
image_generation:
|
||||
model_id: chroma
|
||||
"#,
|
||||
)
|
||||
.expect("config should deserialize");
|
||||
|
||||
let controller = Controller::new(config);
|
||||
|
||||
assert!(controller.supports_purpose(AgentPurpose::TextToSpeech));
|
||||
assert!(controller.supports_purpose(AgentPurpose::ImageGeneration));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn speech_request_serializes_voice_and_omits_unset() {
|
||||
let request = SpeechRequest {
|
||||
model: "tts-kokoro".to_owned(),
|
||||
input: "hello".to_owned(),
|
||||
voice: Some("af_sky".to_owned()),
|
||||
speed: None,
|
||||
response_format: Some("mp3".to_owned()),
|
||||
prompt: None,
|
||||
temperature: None,
|
||||
top_p: None,
|
||||
};
|
||||
|
||||
let json = serde_json::to_string(&request).expect("serialize SpeechRequest");
|
||||
|
||||
assert!(
|
||||
json.contains("\"voice\":\"af_sky\""),
|
||||
"voice should be present: {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("temperature"),
|
||||
"an unset knob must be omitted (not null): {json}"
|
||||
);
|
||||
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn generate_image_request_pins_flags_and_omits_unset() {
|
||||
let request = GenerateImageRequest {
|
||||
model: "chroma".to_owned(),
|
||||
prompt: "a cat".to_owned(),
|
||||
return_binary: false,
|
||||
variants: 1,
|
||||
negative_prompt: None,
|
||||
cfg_scale: None,
|
||||
steps: None,
|
||||
style_preset: None,
|
||||
seed: None,
|
||||
safe_mode: None,
|
||||
hide_watermark: None,
|
||||
format: None,
|
||||
width: None,
|
||||
height: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
quality: None,
|
||||
lora_strength: None,
|
||||
embed_exif_metadata: None,
|
||||
enable_web_search: None,
|
||||
};
|
||||
|
||||
let json = serde_json::to_string(&request).expect("serialize GenerateImageRequest");
|
||||
|
||||
assert!(json.contains("\"model\":\"chroma\""), "{json}");
|
||||
assert!(
|
||||
json.contains("\"return_binary\":false"),
|
||||
"return_binary must be pinned false: {json}"
|
||||
);
|
||||
assert!(
|
||||
json.contains("\"variants\":1"),
|
||||
"variants must be pinned 1: {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("cfg_scale"),
|
||||
"an unset knob must be omitted: {json}"
|
||||
);
|
||||
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edit_image_request_carries_model_and_base64_image() {
|
||||
let request = EditImageRequest {
|
||||
model: "firered-image-edit".to_owned(),
|
||||
prompt: "make it a sunrise".to_owned(),
|
||||
image: "aGVsbG8=".to_owned(),
|
||||
output_format: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
safe_mode: None,
|
||||
};
|
||||
|
||||
let json = serde_json::to_string(&request).expect("serialize EditImageRequest");
|
||||
|
||||
assert!(
|
||||
json.contains("\"model\":\"firered-image-edit\""),
|
||||
"{json}"
|
||||
);
|
||||
assert!(
|
||||
json.contains("\"image\":\"aGVsbG8=\""),
|
||||
"the base64 image string must be present: {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("output_format"),
|
||||
"an unset knob must be omitted: {json}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn web_search_mode_off_deserializes_from_bare_yaml_off() {
|
||||
// `off` is a YAML-1.1 boolean but a plain string under serde_yaml_ng's YAML-1.2 core schema,
|
||||
// so it deserializes straight into the lowercase `WebSearchMode::Off`. This pins that the
|
||||
// sample config and docs can use the bare, unquoted `off` without it parsing as a boolean.
|
||||
let params: VeniceParameters =
|
||||
serde_yaml_ng::from_str("enable_web_search: off").expect("bare `off` should deserialize");
|
||||
|
||||
assert!(matches!(params.enable_web_search, Some(WebSearchMode::Off)));
|
||||
}
|
||||
55
src/agent/provider/venice/utils.rs
Normal file
55
src/agent/provider/venice/utils.rs
Normal file
@@ -0,0 +1,55 @@
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
use crate::utils::base64::base64_encode;
|
||||
|
||||
use super::wire::{ChatMessage, ContentPart, ImageUrl, MessageContent};
|
||||
|
||||
pub fn convert_llm_messages_to_venice(messages: Vec<LLMMessage>) -> Vec<ChatMessage> {
|
||||
let mut venice_messages: Vec<ChatMessage> = Vec::with_capacity(messages.len());
|
||||
|
||||
for message in messages {
|
||||
if let Some(venice_message) = convert_llm_message_to_venice(message) {
|
||||
venice_messages.push(venice_message);
|
||||
}
|
||||
}
|
||||
|
||||
venice_messages
|
||||
}
|
||||
|
||||
fn convert_llm_message_to_venice(message: LLMMessage) -> Option<ChatMessage> {
|
||||
let role = match message.author {
|
||||
LLMAuthor::Prompt => "system",
|
||||
LLMAuthor::Assistant => "assistant",
|
||||
LLMAuthor::User => "user",
|
||||
};
|
||||
|
||||
match message.content {
|
||||
LLMMessageContent::Text(text) => Some(ChatMessage {
|
||||
role: role.to_owned(),
|
||||
content: MessageContent::Text(text),
|
||||
}),
|
||||
LLMMessageContent::Image(image_details) => {
|
||||
// Inline the image as a base64 data URI, the same shape the OpenAI vision content
|
||||
// part uses. This is the gap the openai_compat provider can't fill (it drops images).
|
||||
let data_uri = format!(
|
||||
"data:{};base64,{}",
|
||||
image_details.mime,
|
||||
base64_encode(&image_details.data)
|
||||
);
|
||||
|
||||
Some(ChatMessage {
|
||||
role: role.to_owned(),
|
||||
content: MessageContent::Parts(vec![ContentPart::ImageUrl {
|
||||
image_url: ImageUrl { url: data_uri },
|
||||
}]),
|
||||
})
|
||||
}
|
||||
LLMMessageContent::File(_file_details) => {
|
||||
tracing::warn!(
|
||||
"The Venice provider does not support file content. This file message will be skipped."
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
212
src/agent/provider/venice/wire.rs
Normal file
212
src/agent/provider/venice/wire.rs
Normal file
@@ -0,0 +1,212 @@
|
||||
//! Serde structs modeling Venice's `/chat/completions`, `/audio/transcriptions`,
|
||||
//! `/audio/speech`, `/image/generate`, and `/image/edit` wire shapes. Request types are
|
||||
//! `Serialize`-only (we build them, Venice never sends them back); response types are
|
||||
//! `Deserialize`-only. Keeping the split means the untagged request content enum is never on a
|
||||
//! deserialize path, so a surprise response shape can't fail to match it.
|
||||
//!
|
||||
//! Field names match Venice's schema 1:1 (so the config's `model_id` becomes `model` here). Every
|
||||
//! request body is `additionalProperties: false`, so optional knobs carry `skip_serializing_if`
|
||||
//! to omit rather than send `null`. `/audio/speech` and `/image/edit` return raw binary (no
|
||||
//! response struct); only `/image/generate` returns JSON (`GenerateImageResponse`).
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::config::VeniceParameters;
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ChatCompletionRequest {
|
||||
pub model: String,
|
||||
|
||||
pub messages: Vec<ChatMessage>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub temperature: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub max_completion_tokens: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub venice_parameters: Option<VeniceParameters>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ChatMessage {
|
||||
pub role: String,
|
||||
|
||||
pub content: MessageContent,
|
||||
}
|
||||
|
||||
/// A message body is either a bare string or a list of content parts. Venice accepts both; we
|
||||
/// send the parts form only when a message carries an image (baibot keeps text and images in
|
||||
/// separate messages, so a parts list only ever holds images in v1).
|
||||
#[derive(Debug, Serialize)]
|
||||
#[serde(untagged)]
|
||||
pub enum MessageContent {
|
||||
Text(String),
|
||||
Parts(Vec<ContentPart>),
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
#[serde(tag = "type", rename_all = "snake_case")]
|
||||
pub enum ContentPart {
|
||||
ImageUrl { image_url: ImageUrl },
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ImageUrl {
|
||||
/// A `data:<mime>;base64,<data>` URI for inline images.
|
||||
pub url: String,
|
||||
}
|
||||
|
||||
/// Standard OpenAI-shaped chat completion response. We only read `choices[0].message.content`;
|
||||
/// when web search is on, Venice inlines citations as `^n^` superscripts in that content and we
|
||||
/// pass it through untouched.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ChatCompletionResponse {
|
||||
pub choices: Vec<ChatChoice>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ChatChoice {
|
||||
pub message: ResponseMessage,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ResponseMessage {
|
||||
#[serde(default)]
|
||||
pub content: Option<String>,
|
||||
}
|
||||
|
||||
/// `/audio/transcriptions` response. We read `text`; the optional `duration`/`timestamps` the
|
||||
/// API can return are not used in v1.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct TranscriptionResponse {
|
||||
pub text: String,
|
||||
}
|
||||
|
||||
/// `/audio/speech` (`CreateSpeechRequestSchema`) request. `input` and `model` are always sent;
|
||||
/// the rest are omitted when unset. The response is raw binary audio, so there is no response
|
||||
/// struct.
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct SpeechRequest {
|
||||
pub model: String,
|
||||
|
||||
pub input: String,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub voice: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub speed: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub response_format: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub prompt: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub temperature: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub top_p: Option<f32>,
|
||||
}
|
||||
|
||||
/// `/image/generate` (`GenerateImageRequest`) request. `return_binary` is pinned `false` and
|
||||
/// `variants` to `1` by the builder: baibot wants exactly one image returned as base64-in-JSON,
|
||||
/// which `GenerateImageResponse` then decodes. Flipping `return_binary` would make Venice answer
|
||||
/// with raw binary and break that JSON decode, so it is not configurable.
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct GenerateImageRequest {
|
||||
pub model: String,
|
||||
|
||||
pub prompt: String,
|
||||
|
||||
pub return_binary: bool,
|
||||
|
||||
pub variants: u32,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub negative_prompt: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cfg_scale: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub steps: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub style_preset: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub seed: Option<i64>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub hide_watermark: Option<bool>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub format: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub width: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub height: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub quality: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub lora_strength: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub embed_exif_metadata: Option<bool>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_search: Option<bool>,
|
||||
}
|
||||
|
||||
/// `/image/generate` response when `return_binary` is false: a JSON envelope carrying the images
|
||||
/// as base64 strings. We read `images[0]`; `request`/`timing` and other fields are ignored. `id`
|
||||
/// is telemetry only (logged, never used for correctness), so it is optional: a response that
|
||||
/// carries usable `images` must not fail to deserialize just because the telemetry field drifted.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct GenerateImageResponse {
|
||||
#[serde(default)]
|
||||
pub id: Option<String>,
|
||||
pub images: Vec<String>,
|
||||
}
|
||||
|
||||
/// `/image/edit` (`EditImageRequest`) request. The source `image` is a base64-encoded string
|
||||
/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64-in-JSON, no multipart).
|
||||
/// The response is raw binary, so there is no response struct.
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct EditImageRequest {
|
||||
pub model: String,
|
||||
|
||||
pub prompt: String,
|
||||
|
||||
/// Base64-encoded source image bytes.
|
||||
pub image: String,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub output_format: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
}
|
||||
Reference in New Issue
Block a user