diff --git a/CHANGELOG.md b/CHANGELOG.md index d4c9ab0..d6dcd7d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,8 @@ +# (2026-06-21) Version 1.22.0 + +- (**Feature**) Add a native [Venice](https://venice.ai) provider with [đŸ–Œī¸ image-generation](./docs/features.md#ī¸-image-creation) (incl. editing), [đŸ’Ŧ text-generation](./docs/features.md#-text-generation) (incl. vision), [đŸ—Ŗī¸ text-to-speech](./docs/features.md#ī¸-text-to-speech), [đŸĻģ speech-to-text](./docs/features.md#-speech-to-text), and Venice's native web search via the full `venice_parameters` knob set. Unlike the [OpenAI-compatible](./docs/providers.md#openai-compatible) path (which drops images and can't reach Venice's audio or native image endpoints), it talks to Venice's API directly, using the knob-rich native `/image/generate` and `/image/edit` endpoints. See the [Venice provider docs](./docs/providers.md#venice). + + # (2026-06-05) Version 1.21.1 - (**Security**) Update the [anthropic](https://github.com/etkecc/anthropic-rs) dependency to use [reqwest](https://crates.io/crates/reqwest) 0.12 / [rustls](https://crates.io/crates/rustls) 0.23, replacing the vulnerable `rustls-webpki` 0.101 line with 0.103.13. This resolves [`GHSA-82j2-j2ch-gfr8`](https://github.com/advisories/GHSA-82j2-j2ch-gfr8) (high — denial of service via panic on a malformed CRL), [`GHSA-xgp8-3hg3-c2mh`](https://github.com/advisories/GHSA-xgp8-3hg3-c2mh) and [`GHSA-965h-392x-2mh5`](https://github.com/advisories/GHSA-965h-392x-2mh5) (name-constraint validation issues). diff --git a/Cargo.lock b/Cargo.lock index 490c524..b245857 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -315,7 +315,7 @@ dependencies = [ [[package]] name = "baibot" -version = "1.21.1" +version = "1.22.0" dependencies = [ "anthropic", "anyhow", @@ -329,6 +329,7 @@ dependencies = [ "mxlink", "quick_cache", "regex", + "reqwest 0.12.28", "serde", "serde_json", "serde_yaml_ng", @@ -1590,6 +1591,7 @@ dependencies = [ "tokio", "tokio-rustls", "tower-service", + "webpki-roots 1.0.7", ] [[package]] @@ -3074,6 +3076,7 @@ dependencies = [ "hyper-util", "js-sys", "log", + "mime_guess", "percent-encoding", "pin-project-lite", "quinn", @@ -3095,6 +3098,7 @@ dependencies = [ "wasm-bindgen-futures", "wasm-streams 0.4.2", "web-sys", + "webpki-roots 1.0.7", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index d494876..0de5d8a 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later" readme = "README.md" keywords = ["matrix", "chat", "bot", "AI", "LLM"] include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"] -version = "1.21.1" +version = "1.22.0" edition = "2024" [lib] @@ -28,6 +28,10 @@ mxlink = ">=1.15.0" etke_openai_api_rust = "0.1.*" quick_cache = "0.6.*" regex = "1.12.*" +# Direct dep for the native `venice` provider's HTTP client. Pinned to 0.12 (the version +# async-openai 0.41 already resolves) with rustls only and default-features off, so we ride +# the existing reqwest+rustls copy instead of pulling a second TLS stack (native-tls/openssl). +reqwest = { version = "0.12.*", default-features = false, features = ["json", "multipart", "rustls-tls"] } serde = { version = "1.0.*", features = ["derive"], default-features = false } serde_json = "1.0.*" serde_yaml_ng = "0.10.*" diff --git a/README.md b/README.md index 9ae618d..f3f5374 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use ## 🌟 Features -- 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai) and [â˜ī¸ many more](./docs/providers.md#ī¸-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**: +- 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai), [Venice](./docs/providers.md#venice) and [â˜ī¸ many more](./docs/providers.md#ī¸-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**: - Supports **different use purposes** (depending on the [â˜ī¸ provider](./docs/providers.md) & model): diff --git a/docs/configuration/text-to-speech.md b/docs/configuration/text-to-speech.md index b794d5a..89476e6 100644 --- a/docs/configuration/text-to-speech.md +++ b/docs/configuration/text-to-speech.md @@ -56,7 +56,7 @@ Example: `!bai config room text-to-speech set-speed-override 1.5` (this can also ### đŸ‘Ģ Voice override -The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (usually `onyx` when using [OpenAI](../providers.md#openai)). +The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (e.g. `onyx` when using [OpenAI](../providers.md#openai), or `af_sky` when using [Venice](../providers.md#venice)). Possible values (e.g. `onyx`) depend on the model you're using. For example, for [OpenAI](../providers.md#openai)'s Whisper model, [these voices](https://platform.openai.com/docs/guides/text-to-speech/voice-options) are available. diff --git a/docs/providers.md b/docs/providers.md index 7850d85..17a8d4a 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -19,6 +19,7 @@ The list of supported providers is below. - [OpenAI Compatible](#openai-compatible) - [OpenRouter](#openrouter) - [Together AI](#together-ai) + - [Venice](#venice) ### How to choose a provider @@ -171,3 +172,82 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo - create a global agent: `!bai agent create-global together-ai my-together-ai-agent` 💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/together-ai.yml). + + +### Venice + +[Venice AI](https://venice.ai) runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones. + +- 🆔 Identifier: `venice` +- 🔗 Links: [🏠 Home page](https://venice.ai), [👤 Sign up](https://venice.ai), [📋 Models list](https://api.venice.ai/api/v1/models) +- 🌟 Capabilities: [đŸ–Œī¸ image-generation](./features.md#ī¸-image-creation) (incl. editing, via the native knob-rich `/image/generate` and `/image/edit` endpoints), [đŸ’Ŧ text-generation](./features.md#-text-generation) (incl. vision; native web search via the `venice_parameters` config), [đŸ—Ŗī¸ text-to-speech](./features.md#ī¸-text-to-speech), [đŸĻģ speech-to-text](./features.md#-speech-to-text) +- 🗲 Quick start: + - create a room-local agent: `!bai agent create-room-local venice my-venice-agent` + - create a global agent: `!bai agent create-global venice my-venice-agent` + +💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/venice.yml). + +Unlike the [OpenAI Compatible](#openai-compatible) provider (which can talk to Venice but drops images and can't reach its audio or native image endpoints), this is a first-class Venice integration that exposes Venice's full parameter set. Image generation uses the native `/image/generate` endpoint rather than the OpenAI-compatible `/images/generations` shim, so every Venice-specific knob below is available. + +#### Configuration reference + +Every parameter below is optional unless marked otherwise. Omitting a knob lets Venice apply its own server-side default; this is **not** the same as setting it to `false`, which actively sends `false`. + +**`text_generation.venice_parameters`** — Venice-specific request knobs sent in the `venice_parameters` bag (alongside the standard `model_id`, `prompt`, `temperature`, `max_response_tokens`, and `max_context_tokens` fields). Set any of them to override Venice's behavior. The `Default` column shows the value baibot's sample config ships; a `—` means the knob is left unset, so Venice's own default applies. + +| Knob | What it does | Default | +|------|--------------|---------| +| `enable_web_search` | Web search mode: `auto` (model decides), `on` (always), or `off`. | `auto` | +| `enable_web_citations` | Append source citations to web-search answers. | — | +| `enable_web_scraping` | Allow the model to scrape page contents during web search. | — | +| `enable_x_search` | Include X (Twitter) in web search. | — | +| `include_search_results_in_stream` | Stream search results back as they arrive. | — | +| `return_search_results_as_documents` | Return search results as structured documents. | — | +| `include_venice_system_prompt` | Prepend Venice's own system prompt alongside yours. | — | +| `character_slug` | Use a public Venice character by its slug. | — | +| `strip_thinking_response` | Strip `` blocks from reasoning models so the user sees only the answer. | `true` | +| `disable_thinking` | Disable the model's reasoning step entirely. | — | +| `enable_e2ee` | Run in end-to-end-encrypted mode rather than the default TEE-only mode. | `false` | + +**`text_to_speech`**: + +| Knob | What it does | Default | +|------|--------------|---------| +| `model_id` | The Venice TTS model (e.g. `tts-kokoro`, `tts-qwen3-1-7b`, `tts-xai-v1`). | `tts-kokoro` | +| `voice` | The voice to synthesize with. Model-specific (Kokoro: `af_*`/`am_*`/`bf_*`/`bm_*`); a cloned-voice handle (`vv_`) also works. | `af_sky` | +| `response_format` | Audio format: `mp3`, `opus`, `aac`, `flac`, `wav`, or `pcm`. | `mp3` | +| `speed` | Playback speed, `0.25`–`4.0`. | `1.0` | +| `prompt` | A style prompt steering emotion/delivery. Only Qwen 3 TTS honors it. | — | +| `temperature` | Sampling temperature, `0.0`–`2.0`. Only Qwen 3 / Orpheus / Chatterbox HD honor it. | — | +| `top_p` | Nucleus sampling, `0.0`–`1.0`. Only Qwen 3 TTS honors it. | — | + +**`image_generation`**: + +| Knob | What it does | Default | +|------|--------------|---------| +| `model_id` | The image-generation model. | `chroma` | +| `negative_prompt` | A description of what should **not** appear in the image. | — | +| `cfg_scale` | CFG scale, `0`–`20`. Higher values adhere more closely to the prompt. | — | +| `steps` | Number of inference steps. Model-specific; some models ignore it. | — | +| `style_preset` | A named style to apply (e.g. `3D Model`). | — | +| `seed` | Random seed, `-999999999`–`999999999`. Fix it for reproducible results. | random | +| `safe_mode` | Blur images classified as adult content. | `true` | +| `hide_watermark` | Hide the Venice watermark (may be ignored for some content). | `false` | +| `format` | Output format: `jpeg`, `png`, or `webp`. | `webp` | +| `width` / `height` | Image dimensions in pixels, each `1`–`1280`. | `1024` | +| `aspect_ratio` | Aspect ratio for models that support it (e.g. `1:1`, `16:9`). Alternative to `width`/`height`. | — | +| `resolution` | Resolution tier for models that support it (`1K`, `2K`, `4K`). | — | +| `quality` | Output quality for supported models: `low`, `medium`, `high`. Higher can cost more. | — | +| `lora_strength` | Lora strength, `0`–`100`. Only applies if the model uses additional Loras. | — | +| `embed_exif_metadata` | Embed the generation prompt into the image's EXIF metadata. | `false` | +| `enable_web_search` | Let the model pull the latest info from the web. Model-specific; costs extra credits. | — | + +**`image_generation.edit`** — image editing reuses the `image_generation` block; only the model and a few output knobs differ: + +| Knob | What it does | Default | +|------|--------------|---------| +| `model_id` | The image-edit model. | `firered-image-edit` | +| `output_format` | Output format: `jpeg`, `png`, or `webp`. When omitted, Venice infers it (PNG at 1K, JPEG at 2K/4K). | inferred | +| `aspect_ratio` | Aspect ratio of the result: `auto`, `1:1`, `3:2`, `16:9`, `21:9`, `9:16`, `2:3`, `3:4`, `4:5` (model-specific). | — | +| `resolution` | Resolution tier: `1K`, `2K`, `4K` (model-specific). | `1K` | +| `safe_mode` | Blur images classified as adult content. | `true` | diff --git a/docs/sample-provider-configs/venice.yml b/docs/sample-provider-configs/venice.yml new file mode 100644 index 0000000..2bf8e68 --- /dev/null +++ b/docs/sample-provider-configs/venice.yml @@ -0,0 +1,97 @@ +base_url: https://api.venice.ai/api/v1 +api_key: YOUR_API_KEY_HERE +text_generation: + model_id: kimi-k2-5 + prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}." + temperature: 1.0 + max_response_tokens: 4096 + max_context_tokens: 128000 + # Venice-specific request parameters. Only the keys present below are sent to Venice; omit a + # key to fall back to Venice's own default. Omitting a knob is NOT the same as setting it to + # `false` — `false` actively sends `false`. + venice_parameters: + # Web search: "auto" (model decides), "on" (always), or "off". + enable_web_search: "auto" + # Strip blocks from reasoning models so the user sees only the answer. + strip_thinking_response: true + # Run in TEE-only mode instead of end-to-end encryption (works across all models). + enable_e2ee: false + # Other available knobs — uncomment to override Venice's default: + # enable_web_citations: true + # enable_web_scraping: true + # include_venice_system_prompt: false + # include_search_results_in_stream: true + # return_search_results_as_documents: true + # enable_x_search: true + # disable_thinking: true + # character_slug: public-character-id +speech_to_text: + model_id: nvidia/parakeet-tdt-0.6b-v3 +text_to_speech: + # The Venice TTS model. Others include tts-qwen3-1-7b, tts-xai-v1, + # tts-elevenlabs-turbo-v2-5, tts-minimax-speech-02-hd. See the models list endpoint. + model_id: tts-kokoro + # The voice to synthesize with. Voices are model-specific: Kokoro uses af_*/am_*/bf_*/bm_* + # (e.g. af_sky, am_adam), other models have their own sets. You can also pass a cloned-voice + # handle (vv_) created via Venice's voice-cloning API. An incompatible voice returns an error. + voice: af_sky + # Output audio format: mp3, opus, aac, flac, wav, or pcm. mp3 is the broadest Matrix-client fit. + response_format: mp3 + # Other available knobs — uncomment to override Venice's default: + # Playback speed, 0.25–4.0 (1.0 is normal). + # speed: 1.0 + # A style prompt steering emotion/delivery (e.g. "Excited and energetic."). Only Qwen 3 TTS uses it. + # prompt: "Calm and warm." + # Sampling temperature, 0.0–2.0 (higher = more varied). Only Qwen 3 / Orpheus / Chatterbox HD use it. + # temperature: 0.9 + # Nucleus sampling, 0.0–1.0. Only Qwen 3 TTS uses it. + # top_p: 1.0 +image_generation: + # The image-generation model. See the models list endpoint for the full set. + model_id: chroma + # The image-edit model, used when editing an existing image rather than generating a new one. + # Editing shares this same image_generation config block; only the model differs. + edit: + model_id: firered-image-edit + # Other edit knobs — uncomment to override Venice's default: + # Output format: jpeg, png, or webp. When omitted, Venice infers it (PNG at 1K, JPEG at 2K/4K). + # output_format: png + # Aspect ratio of the result: auto, 1:1, 3:2, 16:9, 21:9, 9:16, 2:3, 3:4, 4:5 (model-specific). + # aspect_ratio: auto + # Resolution tier: 1K, 2K, 4K (model-specific). Defaults to 1K. + # resolution: 1K + # Blur images classified as adult content. Defaults to true. + # safe_mode: true + # Other generation knobs — uncomment to override Venice's default. Omitting a knob is NOT the same + # as setting it: an omitted knob lets Venice apply its own default, a set value is sent verbatim. + # A description of what should NOT appear in the image. + # negative_prompt: "blurry, watermark, text" + # CFG scale, 0–20. Higher values make the image adhere more closely to the prompt. + # cfg_scale: 7.5 + # Number of inference steps. Model-specific; some models ignore it. + # steps: 8 + # A named style to apply (e.g. "3D Model"). See Venice's image-styles reference. + # style_preset: "3D Model" + # Random seed, -999999999–999999999. Fix it for reproducible results; omit for a random seed. + # seed: 123456789 + # Blur images classified as adult content. Defaults to true. + # safe_mode: true + # Hide the Venice watermark. Venice may ignore this for certain generated content. Defaults to false. + # hide_watermark: false + # Output format: jpeg, png, or webp. webp is smallest; png is highest-quality. Defaults to webp. + # format: webp + # Image dimensions in pixels, each 1–1280. Default 1024×1024. + # width: 1024 + # height: 1024 + # Aspect ratio (used by certain models, e.g. Nano Banana): "1:1", "16:9". An alternative to width/height. + # aspect_ratio: "1:1" + # Resolution tier (used by certain models): "1K", "2K", "4K". + # resolution: "1K" + # Output quality for supported models (e.g. GPT Image 2): low, medium, high. Higher can cost more. + # quality: high + # Lora strength, 0–100. Only applies if the model uses additional Loras. + # lora_strength: 50 + # Embed the generation prompt into the image's EXIF metadata. Defaults to false. + # embed_exif_metadata: false + # Let the model pull the latest info from the web for the image. Model-specific; costs extra credits. + # enable_web_search: false diff --git a/src/agent/instantiation.rs b/src/agent/instantiation.rs index ed46560..9e8e9be 100644 --- a/src/agent/instantiation.rs +++ b/src/agent/instantiation.rs @@ -109,6 +109,9 @@ fn create_controller_from_provider_and_json_value_config( AgentProvider::TogetherAI => { provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config) } + AgentProvider::Venice => { + provider::venice::create_controller_from_yaml_value_config(agent_id, config) + } } } @@ -150,5 +153,9 @@ pub fn default_config_for_provider(provider: &AgentProvider) -> serde_yaml_ng::V let config = super::provider::togetherai::default_config(); serde_yaml_ng::to_value(config).expect("Failed to serialize config") } + AgentProvider::Venice => { + let config = super::provider::venice::default_config(); + serde_yaml_ng::to_value(config).expect("Failed to serialize config") + } } } diff --git a/src/agent/provider/controller.rs b/src/agent/provider/controller.rs index 7b85a76..4e2df11 100644 --- a/src/agent/provider/controller.rs +++ b/src/agent/provider/controller.rs @@ -61,6 +61,7 @@ pub enum ControllerType { OpenAI(Box), OpenAICompat(Box), Anthropic(Box), + Venice(Box), } impl ControllerTrait for ControllerType { @@ -69,6 +70,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.supports_purpose(purpose), ControllerType::OpenAICompat(controller) => controller.supports_purpose(purpose), ControllerType::Anthropic(controller) => controller.supports_purpose(purpose), + ControllerType::Venice(controller) => controller.supports_purpose(purpose), } } @@ -77,6 +79,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.text_generation_model_id(), ControllerType::OpenAICompat(controller) => controller.text_generation_model_id(), ControllerType::Anthropic(controller) => controller.text_generation_model_id(), + ControllerType::Venice(controller) => controller.text_generation_model_id(), } } @@ -85,6 +88,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.text_generation_prompt(), ControllerType::OpenAICompat(controller) => controller.text_generation_prompt(), ControllerType::Anthropic(controller) => controller.text_generation_prompt(), + ControllerType::Venice(controller) => controller.text_generation_prompt(), } } @@ -93,6 +97,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.text_to_speech_voice(), ControllerType::OpenAICompat(controller) => controller.text_to_speech_voice(), ControllerType::Anthropic(controller) => controller.text_to_speech_voice(), + ControllerType::Venice(controller) => controller.text_to_speech_voice(), } } @@ -101,6 +106,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.text_to_speech_speed(), ControllerType::OpenAICompat(controller) => controller.text_to_speech_speed(), ControllerType::Anthropic(controller) => controller.text_to_speech_speed(), + ControllerType::Venice(controller) => controller.text_to_speech_speed(), } } @@ -109,6 +115,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.text_generation_temperature(), ControllerType::OpenAICompat(controller) => controller.text_generation_temperature(), ControllerType::Anthropic(controller) => controller.text_generation_temperature(), + ControllerType::Venice(controller) => controller.text_generation_temperature(), } } @@ -117,6 +124,7 @@ impl ControllerTrait for ControllerType { ControllerType::OpenAI(controller) => controller.ping().await, ControllerType::OpenAICompat(controller) => controller.ping().await, ControllerType::Anthropic(controller) => controller.ping().await, + ControllerType::Venice(controller) => controller.ping().await, } } @@ -135,6 +143,9 @@ impl ControllerTrait for ControllerType { ControllerType::Anthropic(controller) => { controller.generate_text(conversation, params).await } + ControllerType::Venice(controller) => { + controller.generate_text(conversation, params).await + } } } @@ -154,6 +165,9 @@ impl ControllerTrait for ControllerType { ControllerType::Anthropic(controller) => { controller.speech_to_text(mime_type, media, params).await } + ControllerType::Venice(controller) => { + controller.speech_to_text(mime_type, media, params).await + } } } @@ -170,6 +184,9 @@ impl ControllerTrait for ControllerType { ControllerType::Anthropic(controller) => { controller.generate_image(prompt, params).await } + ControllerType::Venice(controller) => { + controller.generate_image(prompt, params).await + } } } @@ -189,6 +206,9 @@ impl ControllerTrait for ControllerType { ControllerType::Anthropic(controller) => { controller.create_image_edit(prompt, images, params).await } + ControllerType::Venice(controller) => { + controller.create_image_edit(prompt, images, params).await + } } } @@ -203,6 +223,7 @@ impl ControllerTrait for ControllerType { controller.text_to_speech(text, params).await } ControllerType::Anthropic(controller) => controller.text_to_speech(text, params).await, + ControllerType::Venice(controller) => controller.text_to_speech(text, params).await, } } } diff --git a/src/agent/provider/entity/agent_provider.rs b/src/agent/provider/entity/agent_provider.rs index 5533531..82deb22 100644 --- a/src/agent/provider/entity/agent_provider.rs +++ b/src/agent/provider/entity/agent_provider.rs @@ -11,6 +11,7 @@ pub enum AgentProvider { OpenAICompat, OpenRouter, TogetherAI, + Venice, } impl AgentProvider { @@ -25,6 +26,7 @@ impl AgentProvider { &Self::OpenAICompat, &Self::OpenRouter, &Self::TogetherAI, + &Self::Venice, ] } @@ -39,6 +41,7 @@ impl AgentProvider { Self::OpenAICompat => "openai-compatible", Self::OpenRouter => "openrouter", Self::TogetherAI => "together-ai", + Self::Venice => "venice", } } @@ -53,6 +56,7 @@ impl AgentProvider { "openai-compatible" => Ok(Self::OpenAICompat), "openrouter" => Ok(Self::OpenRouter), "together-ai" => Ok(Self::TogetherAI), + "venice" => Ok(Self::Venice), _ => Err("Unexpected string value"), } } @@ -181,6 +185,25 @@ impl AgentProvider { text_generation_supports_vision: false, text_generation_supports_tools: false, }, + Self::Venice => AgentProviderInfo { + id: Self::Venice.to_static_str(), + name: "Venice", + description: "Venice AI runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses. It serves frontier proprietary and open-source models with text-generation (including vision), speech-to-text, text-to-speech, native image generation and editing, and native web search.", + homepage_url: Some("https://venice.ai"), + wiki_url: None, + sign_up_url: Some("https://venice.ai"), + models_list_url: Some("https://api.venice.ai/api/v1/models"), + supported_purposes: vec![ + AgentPurpose::ImageGeneration, + AgentPurpose::TextGeneration, + AgentPurpose::TextToSpeech, + AgentPurpose::SpeechToText, + ], + text_generation_supports_vision: true, + // Venice does native web search via `venice_parameters`, NOT baibot's built-in + // tools mechanism (the OpenAI web_search/code_interpreter block), so this is false. + text_generation_supports_tools: false, + }, } } } diff --git a/src/agent/provider/mod.rs b/src/agent/provider/mod.rs index 0daca2d..853bfa3 100644 --- a/src/agent/provider/mod.rs +++ b/src/agent/provider/mod.rs @@ -10,6 +10,7 @@ pub mod openai; pub mod openai_compat; pub(super) mod openrouter; pub(super) mod togetherai; +pub mod venice; fn default_temperature() -> f32 { 1.0 diff --git a/src/agent/provider/venice/audio.rs b/src/agent/provider/venice/audio.rs new file mode 100644 index 0000000..7578a6d --- /dev/null +++ b/src/agent/provider/venice/audio.rs @@ -0,0 +1,156 @@ +use crate::agent::AgentPurpose; +use crate::agent::provider::entity::{TextToSpeechParams, TextToSpeechResult}; +use crate::agent::provider::{SpeechToTextParams, SpeechToTextResult}; +use crate::strings; + +use super::config::Config; +use super::wire::{SpeechRequest, TranscriptionResponse}; + +pub async fn speech_to_text( + config: &Config, + http: &reqwest::Client, + mime_type: &mxlink::mime::Mime, + media: Vec, + params: SpeechToTextParams, +) -> anyhow::Result { + let Some(speech_to_text_config) = &config.speech_to_text else { + return Err(anyhow::anyhow!( + strings::agent::no_configuration_for_purpose_so_cannot_be_used( + &AgentPurpose::SpeechToText + ), + )); + }; + + // Unlike the openai_compat path (which writes the audio to a temp file because its library + // can't take bytes), reqwest's multipart takes the bytes directly. + let part = reqwest::multipart::Part::bytes(media) + .file_name("audio") + .mime_str(mime_type.as_ref())?; + + let mut form = reqwest::multipart::Form::new() + .part("file", part) + .text("model", speech_to_text_config.model_id.clone()) + .text("response_format", "json"); + + if let Some(language) = ¶ms.language_override { + form = form.text("language", language.clone()); + } + + let url = format!( + "{}/audio/transcriptions", + config.base_url.trim_end_matches('/') + ); + + tracing::trace!( + model_id = speech_to_text_config.model_id, + language = ?params.language_override, + "Sending Venice audio transcription API request" + ); + + let response = http + .post(&url) + .bearer_auth(&config.api_key) + .multipart(form) + .send() + .await?; + + let status = response.status(); + if !status.is_success() { + // Body to the server log only, not into the returned error (which reaches the Matrix room). + let body = response.text().await.unwrap_or_default(); + tracing::warn!(%status, body, "Venice audio transcription request failed"); + return Err(anyhow::anyhow!( + "Venice audio transcription request failed with status {status}" + )); + } + + let response: TranscriptionResponse = response.json().await?; + + Ok(SpeechToTextResult { + text: response.text, + }) +} + +pub async fn text_to_speech( + config: &Config, + http: &reqwest::Client, + input: &str, + params: TextToSpeechParams, +) -> anyhow::Result { + let Some(text_to_speech_config) = &config.text_to_speech else { + return Err(anyhow::anyhow!( + strings::agent::no_configuration_for_purpose_so_cannot_be_used( + &AgentPurpose::TextToSpeech + ), + )); + }; + + // Per-call overrides win over the configured defaults. + let voice = params + .voice_override + .or_else(|| text_to_speech_config.voice.clone()); + let speed = params.speed_override.or(text_to_speech_config.speed); + + let response_format = text_to_speech_config.response_format.clone(); + let mime_type = response_format_to_mime_type(response_format.as_deref()); + + let request = SpeechRequest { + model: text_to_speech_config.model_id.clone(), + input: input.to_owned(), + voice, + speed, + response_format, + prompt: text_to_speech_config.prompt.clone(), + temperature: text_to_speech_config.temperature, + top_p: text_to_speech_config.top_p, + }; + + let url = format!("{}/audio/speech", config.base_url.trim_end_matches('/')); + + tracing::trace!( + model_id = text_to_speech_config.model_id, + voice = ?request.voice, + "Sending Venice text-to-speech API request" + ); + + let response = http + .post(&url) + .bearer_auth(&config.api_key) + .json(&request) + .send() + .await?; + + let status = response.status(); + if !status.is_success() { + // Body to the server log only, not into the returned error (which reaches the Matrix room). + let body = response.text().await.unwrap_or_default(); + tracing::warn!(%status, body, "Venice text-to-speech request failed"); + return Err(anyhow::anyhow!( + "Venice text-to-speech request failed with status {status}" + )); + } + + // The speech endpoint answers with raw binary audio; read the body directly. + let bytes = response.bytes().await?.to_vec(); + + Ok(TextToSpeechResult { bytes, mime_type }) +} + +/// Map a Venice TTS `response_format` to its MIME type. Defaults to `audio/mpeg` (the +/// IANA-registered MP3 type, RFC 3003) when the format is unset, matching Venice's own `mp3` +/// default. This deliberately uses `audio/mpeg` rather than the `audio/mp3` alias the openai +/// provider emits; baibot's downstream audio-filename mapping treats both as `.mp3`. +fn response_format_to_mime_type(response_format: Option<&str>) -> mxlink::mime::Mime { + let raw = match response_format.unwrap_or("mp3") { + "mp3" => "audio/mpeg", + "opus" => "audio/ogg", + "aac" => "audio/aac", + "flac" => "audio/flac", + "wav" => "audio/wav", + "pcm" => "audio/L8", + _ => "audio/mpeg", + }; + + raw.parse() + .unwrap_or(mxlink::mime::APPLICATION_OCTET_STREAM) +} diff --git a/src/agent/provider/venice/chat.rs b/src/agent/provider/venice/chat.rs new file mode 100644 index 0000000..03e3f23 --- /dev/null +++ b/src/agent/provider/venice/chat.rs @@ -0,0 +1,122 @@ +use crate::agent::AgentPurpose; +use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult}; +use crate::conversation::llm::{ + Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage, + MessageContent as LLMMessageContent, shorten_messages_list_to_context_size, +}; +use crate::strings; + +use super::config::Config; +use super::utils::convert_llm_messages_to_venice; +use super::wire::{ChatCompletionRequest, ChatCompletionResponse}; + +pub async fn generate_text( + config: &Config, + http: &reqwest::Client, + conversation: LLMConversation, + params: TextGenerationParams, +) -> anyhow::Result { + let Some(text_generation_config) = &config.text_generation else { + return Err(anyhow::anyhow!( + strings::agent::no_configuration_for_purpose_so_cannot_be_used( + &AgentPurpose::TextGeneration + ), + )); + }; + + let prompt_text = params.prompt_variables.format( + params + .prompt_override + .unwrap_or(text_generation_config.prompt.clone().unwrap_or_default()) + .trim(), + ); + + let prompt_message = if prompt_text.is_empty() { + None + } else { + Some(LLMMessage { + author: LLMAuthor::Prompt, + sender_id: None, + content: LLMMessageContent::Text(prompt_text), + timestamp: chrono::Utc::now(), + }) + }; + + let mut conversation_messages = conversation.messages; + + if params.context_management_enabled { + conversation_messages = shorten_messages_list_to_context_size( + &text_generation_config.model_id, + &prompt_message, + conversation_messages, + text_generation_config.max_response_tokens, + text_generation_config.max_context_tokens, + ); + } + + if let Some(prompt_message) = prompt_message { + conversation_messages.insert(0, prompt_message); + } + + let messages = convert_llm_messages_to_venice(conversation_messages); + + let temperature = params + .temperature_override + .unwrap_or(text_generation_config.temperature); + + let request = ChatCompletionRequest { + model: text_generation_config.model_id.clone(), + messages, + temperature: Some(temperature), + // Web search rides entirely inside `venice_parameters`; there is no `tools` array here. + // `max_tokens` is deprecated on Venice in favor of `max_completion_tokens`. + max_completion_tokens: text_generation_config.max_response_tokens, + venice_parameters: text_generation_config.venice_parameters.clone(), + }; + + let url = format!( + "{}/chat/completions", + config.base_url.trim_end_matches('/') + ); + + tracing::trace!( + model = text_generation_config.model_id, + messages_count = request.messages.len(), + "Sending Venice chat completion API request" + ); + + let response = http + .post(&url) + .bearer_auth(&config.api_key) + .json(&request) + .send() + .await?; + + let status = response.status(); + if !status.is_success() { + // Log the body server-side for debugging (Venice explains a rejected strict body there), + // but keep it OUT of the returned error: that error surfaces in the Matrix room, and the + // body can carry account / rate-limit details that shouldn't reach room members. + let body = response.text().await.unwrap_or_default(); + tracing::warn!(%status, body, "Venice chat completion request failed"); + return Err(anyhow::anyhow!( + "Venice chat completion request failed with status {status}" + )); + } + + let response: ChatCompletionResponse = response.json().await?; + + let Some(choice) = response.choices.into_iter().next() else { + return Err(anyhow::anyhow!( + "No choices were returned from the Venice chat completion API" + )); + }; + + let Some(content) = choice.message.content else { + return Err(anyhow::anyhow!( + "No message content was returned from the Venice chat completion API" + )); + }; + + Ok(TextGenerationResult { text: content }) +} diff --git a/src/agent/provider/venice/config.rs b/src/agent/provider/venice/config.rs new file mode 100644 index 0000000..8a3a83b --- /dev/null +++ b/src/agent/provider/venice/config.rs @@ -0,0 +1,355 @@ +use serde::{Deserialize, Serialize}; + +use crate::agent::{default_prompt, provider::ConfigTrait}; + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Config { + pub base_url: String, + + pub api_key: String, + + #[serde(skip_serializing_if = "Option::is_none")] + pub text_generation: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub speech_to_text: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub text_to_speech: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub image_generation: Option, +} + +impl Default for Config { + fn default() -> Self { + Self { + base_url: "https://api.venice.ai/api/v1".to_owned(), + api_key: "YOUR_API_KEY_HERE".to_owned(), + text_generation: Some(TextGenerationConfig::default()), + speech_to_text: Some(SpeechToTextConfig::default()), + text_to_speech: Some(TextToSpeechConfig::default()), + image_generation: Some(ImageGenerationConfig::default()), + } + } +} + +impl ConfigTrait for Config { + fn validate(&self) -> Result<(), String> { + if self.base_url.is_empty() { + return Err("The base URL must not be empty.".to_owned()); + } + + Ok(()) + } +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct TextGenerationConfig { + #[serde(default = "default_text_model_id")] + pub model_id: String, + + #[serde(default)] + pub prompt: Option, + + #[serde(default = "super::super::default_temperature")] + pub temperature: f32, + + #[serde(default)] + pub max_response_tokens: Option, + + #[serde(default)] + pub max_context_tokens: u32, + + /// Venice-specific request knobs, serialized 1:1 into the `venice_parameters` bag on the + /// wire. Any unset field is omitted, so Venice applies its own server-side default. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub venice_parameters: Option, +} + +impl Default for TextGenerationConfig { + fn default() -> Self { + Self { + model_id: default_text_model_id(), + prompt: Some(default_prompt().to_owned()), + temperature: super::super::default_temperature(), + // Reserved output budget: sent as the response cap AND subtracted from the context + // window when trimming history. Mirrors the openai_compat sibling's default. + max_response_tokens: Some(4096), + // Matches Venice's own `availableContextTokens` (131072) and the non-OpenAI sibling + // providers (ollama/localai/mistral all default to 128_000). + max_context_tokens: 128_000, + // A usable starting point, not an everything-set dump: only these three are sent; + // every other knob stays None so Venice applies its own default (omitting != false). + venice_parameters: Some(VeniceParameters { + enable_web_search: Some(WebSearchMode::Auto), + strip_thinking_response: Some(true), + enable_e2ee: Some(false), + ..Default::default() + }), + } + } +} + +fn default_text_model_id() -> String { + "kimi-k2-5".to_owned() +} + +/// The full `venice_parameters` knob set, mirroring Venice's `ChatCompletionRequest` +/// schema field-for-field. Every field is optional with `skip_serializing_if`, so the +/// request never carries a knob the user didn't set (the body is `additionalProperties: false`, +/// and an unset knob simply omits rather than sending `null`). +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +pub struct VeniceParameters { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub enable_web_search: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub enable_web_citations: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub enable_web_scraping: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub include_venice_system_prompt: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub include_search_results_in_stream: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub return_search_results_as_documents: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub enable_x_search: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub enable_e2ee: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub character_slug: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub strip_thinking_response: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub disable_thinking: Option, +} + +#[derive(Debug, Clone, Copy, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum WebSearchMode { + Auto, + On, + Off, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SpeechToTextConfig { + #[serde(default = "default_speech_to_text_model_id")] + pub model_id: String, +} + +impl Default for SpeechToTextConfig { + fn default() -> Self { + Self { + model_id: default_speech_to_text_model_id(), + } + } +} + +fn default_speech_to_text_model_id() -> String { + "nvidia/parakeet-tdt-0.6b-v3".to_owned() +} + +/// `/audio/speech` (`CreateSpeechRequestSchema`) request knobs. Only `model_id` is required on +/// the wire; everything else is optional with `skip_serializing_if` so an unset knob is omitted +/// rather than sent as `null` (the body is `additionalProperties: false`). `voice` is a free +/// `Option`, not a closed enum: Venice's voice set spans dozens of model-specific names +/// plus arbitrary cloned-voice handles (`vv_`), so an enum would reject valid handles. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct TextToSpeechConfig { + #[serde(default = "default_text_to_speech_model_id")] + pub model_id: String, + + #[serde( + default = "default_text_to_speech_voice", + skip_serializing_if = "Option::is_none" + )] + pub voice: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub speed: Option, + + #[serde( + default = "default_text_to_speech_response_format", + skip_serializing_if = "Option::is_none" + )] + pub response_format: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub prompt: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub temperature: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub top_p: Option, +} + +impl Default for TextToSpeechConfig { + fn default() -> Self { + Self { + model_id: default_text_to_speech_model_id(), + voice: default_text_to_speech_voice(), + speed: None, + response_format: default_text_to_speech_response_format(), + prompt: None, + temperature: None, + top_p: None, + } + } +} + +fn default_text_to_speech_model_id() -> String { + "tts-kokoro".to_owned() +} + +fn default_text_to_speech_voice() -> Option { + Some("af_sky".to_owned()) +} + +fn default_text_to_speech_response_format() -> Option { + Some("mp3".to_owned()) +} + +/// `/image/generate` (`GenerateImageRequest`) request knobs, mirroring Venice's schema +/// field-for-field. Only `model_id` is required; every other knob is optional with +/// `skip_serializing_if` so unset knobs are omitted (the body is `additionalProperties: false`). +/// The full knob set is deliberate: the native `/image/generate` endpoint is the flagship's +/// reason to exist over the knob-dropping OpenAI-compat path, so the knobs ARE the feature. +/// The deprecated `inpaint` knob is intentionally absent. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ImageGenerationConfig { + #[serde(default = "default_image_generation_model_id")] + pub model_id: String, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub negative_prompt: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub cfg_scale: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub steps: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub style_preset: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub seed: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub safe_mode: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub hide_watermark: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub format: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub width: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub height: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub aspect_ratio: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub resolution: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub quality: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub lora_strength: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub embed_exif_metadata: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub enable_web_search: Option, + + /// Image-edit settings, nested here because baibot has a single `ImageGeneration` purpose + /// and edit shares its config gate. The gen and edit model sets are disjoint, so edit + /// carries its own model field. + #[serde(default)] + pub edit: ImageEditSettings, +} + +impl Default for ImageGenerationConfig { + fn default() -> Self { + Self { + model_id: default_image_generation_model_id(), + negative_prompt: None, + cfg_scale: None, + steps: None, + style_preset: None, + seed: None, + safe_mode: None, + hide_watermark: None, + format: None, + width: None, + height: None, + aspect_ratio: None, + resolution: None, + quality: None, + lora_strength: None, + embed_exif_metadata: None, + enable_web_search: None, + edit: ImageEditSettings::default(), + } + } +} + +fn default_image_generation_model_id() -> String { + "chroma".to_owned() +} + +/// `/image/edit` (`EditImageRequest`) request knobs, mirroring Venice's schema. The source image +/// and prompt are supplied per-call (not config), so only the model and the output-shaping knobs +/// live here. Each knob is optional with `skip_serializing_if`. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ImageEditSettings { + #[serde(default = "default_image_edit_model_id")] + pub model_id: String, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output_format: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub aspect_ratio: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub resolution: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub safe_mode: Option, +} + +impl Default for ImageEditSettings { + fn default() -> Self { + Self { + model_id: default_image_edit_model_id(), + output_format: None, + aspect_ratio: None, + resolution: None, + safe_mode: None, + } + } +} + +fn default_image_edit_model_id() -> String { + "firered-image-edit".to_owned() +} diff --git a/src/agent/provider/venice/controller.rs b/src/agent/provider/venice/controller.rs new file mode 100644 index 0000000..816ae95 --- /dev/null +++ b/src/agent/provider/venice/controller.rs @@ -0,0 +1,146 @@ +use crate::agent::AgentPurpose; +use crate::agent::provider::entity::{ + ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams, + TextGenerationResult, TextToSpeechParams, TextToSpeechResult, +}; +use crate::agent::provider::{ + ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult, +}; +use crate::conversation::llm::{ + Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage, + MessageContent as LLMMessageContent, +}; + +use super::super::ControllerTrait; +use super::config::Config; + +#[derive(Debug, Clone)] +pub struct Controller { + config: Config, + http: reqwest::Client, +} + +impl Controller { + pub fn new(config: Config) -> Self { + // Image generation and text-to-speech can run long, so give the client a generous timeout + // instead of reqwest's default (none). `build` only fails on TLS/system init; fall back to + // the infallible `Client::new()` so this constructor stays infallible. + let http = reqwest::Client::builder() + .timeout(std::time::Duration::from_secs(120)) + .build() + .unwrap_or_else(|_| reqwest::Client::new()); + + Self { config, http } + } +} + +impl ControllerTrait for Controller { + async fn ping(&self) -> anyhow::Result { + if !self.supports_purpose(AgentPurpose::TextGeneration) { + return Ok(PingResult::Inconclusive); + } + + // Mirror the openai/openai_compat ping: a real "Hello!" round-trip exercises the strict + // /chat/completions body and auth, so a successful ping proves text generation works. + let messages = vec![LLMMessage { + author: LLMAuthor::User, + sender_id: None, + content: LLMMessageContent::Text("Hello!".to_string()), + timestamp: chrono::Utc::now(), + }]; + + let conversation = LLMConversation { messages }; + + self.generate_text(conversation, TextGenerationParams::default()) + .await?; + + Ok(PingResult::Successful) + } + + async fn generate_text( + &self, + conversation: LLMConversation, + params: TextGenerationParams, + ) -> anyhow::Result { + super::chat::generate_text(&self.config, &self.http, conversation, params).await + } + + async fn speech_to_text( + &self, + mime_type: &mxlink::mime::Mime, + media: Vec, + params: SpeechToTextParams, + ) -> anyhow::Result { + super::audio::speech_to_text(&self.config, &self.http, mime_type, media, params).await + } + + async fn generate_image( + &self, + prompt: &str, + params: ImageGenerationParams, + ) -> anyhow::Result { + super::images::generate_image(&self.config, &self.http, prompt, params).await + } + + async fn create_image_edit( + &self, + prompt: &str, + images: Vec, + params: ImageEditParams, + ) -> anyhow::Result { + super::images::create_image_edit(&self.config, &self.http, prompt, images, params).await + } + + async fn text_to_speech( + &self, + input: &str, + params: TextToSpeechParams, + ) -> anyhow::Result { + super::audio::text_to_speech(&self.config, &self.http, input, params).await + } + + fn supports_purpose(&self, purpose: AgentPurpose) -> bool { + match purpose { + AgentPurpose::TextGeneration => self.config.text_generation.is_some(), + AgentPurpose::SpeechToText => self.config.speech_to_text.is_some(), + AgentPurpose::TextToSpeech => self.config.text_to_speech.is_some(), + AgentPurpose::ImageGeneration => self.config.image_generation.is_some(), + AgentPurpose::CatchAll => true, + } + } + + fn text_generation_model_id(&self) -> Option { + self.config + .text_generation + .as_ref() + .map(|config| config.model_id.to_owned()) + } + + fn text_generation_prompt(&self) -> Option { + self.config + .text_generation + .as_ref() + .and_then(|config| config.prompt.clone()) + } + + fn text_generation_temperature(&self) -> Option { + self.config + .text_generation + .as_ref() + .map(|config| config.temperature) + } + + fn text_to_speech_voice(&self) -> Option { + self.config + .text_to_speech + .as_ref() + .and_then(|config| config.voice.clone()) + } + + fn text_to_speech_speed(&self) -> Option { + self.config + .text_to_speech + .as_ref() + .and_then(|config| config.speed) + } +} diff --git a/src/agent/provider/venice/images.rs b/src/agent/provider/venice/images.rs new file mode 100644 index 0000000..53416de --- /dev/null +++ b/src/agent/provider/venice/images.rs @@ -0,0 +1,192 @@ +use crate::agent::AgentPurpose; +use crate::agent::provider::entity::{ImageEditResult, ImageGenerationResult, ImageSource}; +use crate::agent::provider::{ImageEditParams, ImageGenerationParams}; +use crate::strings; +use crate::utils::base64::{base64_decode, base64_encode}; + +use super::config::Config; +use super::wire::{EditImageRequest, GenerateImageRequest, GenerateImageResponse}; + +/// Generate an image via Venice's native `/image/generate` endpoint. +/// +/// This is the base64-in-JSON path: we pin `return_binary: false` so Venice answers with a JSON +/// envelope (`GenerateImageResponse`) carrying the image as a base64 string, which we decode. The +/// sibling `create_image_edit` is the *other* response shape (raw binary); the two must not be +/// crossed. `params` is advisory only; the Venice config drives the request. +pub async fn generate_image( + config: &Config, + http: &reqwest::Client, + prompt: &str, + _params: ImageGenerationParams, +) -> anyhow::Result { + let Some(image_generation_config) = &config.image_generation else { + return Err(anyhow::anyhow!( + strings::agent::no_configuration_for_purpose_so_cannot_be_used( + &AgentPurpose::ImageGeneration + ), + )); + }; + + let request = GenerateImageRequest { + model: image_generation_config.model_id.clone(), + prompt: prompt.to_owned(), + // Pinned: baibot wants exactly one image, returned as base64-in-JSON so `GenerateImageResponse` + // can decode it. Flipping `return_binary` would make Venice answer with raw binary and break + // the JSON decode below, so neither knob is configurable. + return_binary: false, + variants: 1, + negative_prompt: image_generation_config.negative_prompt.clone(), + cfg_scale: image_generation_config.cfg_scale, + steps: image_generation_config.steps, + style_preset: image_generation_config.style_preset.clone(), + seed: image_generation_config.seed, + safe_mode: image_generation_config.safe_mode, + hide_watermark: image_generation_config.hide_watermark, + format: image_generation_config.format.clone(), + width: image_generation_config.width, + height: image_generation_config.height, + aspect_ratio: image_generation_config.aspect_ratio.clone(), + resolution: image_generation_config.resolution.clone(), + quality: image_generation_config.quality.clone(), + lora_strength: image_generation_config.lora_strength, + embed_exif_metadata: image_generation_config.embed_exif_metadata, + enable_web_search: image_generation_config.enable_web_search, + }; + + let url = format!("{}/image/generate", config.base_url.trim_end_matches('/')); + + // The prompt is user content; keep it out of logs (mirrors the STT/TTS paths). + tracing::trace!( + model_id = image_generation_config.model_id, + "Sending Venice image generation API request" + ); + + let response = http + .post(&url) + .bearer_auth(&config.api_key) + .json(&request) + .send() + .await?; + + let status = response.status(); + if !status.is_success() { + // Body to the server log only, not into the returned error (which reaches the Matrix room). + let body = response.text().await.unwrap_or_default(); + tracing::warn!(%status, body, "Venice image generation request failed"); + return Err(anyhow::anyhow!( + "Venice image generation request failed with status {status}" + )); + } + + let response: GenerateImageResponse = response.json().await?; + + tracing::trace!(request_id = ?response.id, "Venice image generation succeeded"); + + let Some(image_base64) = response.images.into_iter().next() else { + return Err(anyhow::anyhow!( + "The Venice image generation API returned no images" + )); + }; + + // Swallow the decode error's detail (it can echo input bytes/offsets); the returned error + // reaches the Matrix room, so it stays generic while the real cause goes to the server log. + let bytes = base64_decode(&image_base64).map_err(|decode_err| { + tracing::warn!(%decode_err, "Venice image generation returned undecodable base64"); + anyhow::anyhow!("Venice image generation returned invalid base64 image data") + })?; + + Ok(ImageGenerationResult { + bytes, + mime_type: image_format_to_mime_type(image_generation_config.format.as_deref()), + revised_prompt: None, + }) +} + +/// Edit an image via Venice's native `/image/edit` endpoint. +/// +/// This is the raw-binary path: the request is JSON carrying the source image as a base64 string +/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64, no multipart), and the +/// response body IS the edited image bytes (no JSON envelope). `params` is advisory only. +pub async fn create_image_edit( + config: &Config, + http: &reqwest::Client, + prompt: &str, + images: Vec, + _params: ImageEditParams, +) -> anyhow::Result { + let Some(image_generation_config) = &config.image_generation else { + return Err(anyhow::anyhow!( + strings::agent::no_configuration_for_purpose_so_cannot_be_used( + &AgentPurpose::ImageGeneration + ), + )); + }; + + let edit_config = &image_generation_config.edit; + + let Some(source) = images.into_iter().next() else { + return Err(anyhow::anyhow!("No image sources provided")); + }; + + let request = EditImageRequest { + model: edit_config.model_id.clone(), + prompt: prompt.to_owned(), + image: base64_encode(&source.bytes), + output_format: edit_config.output_format.clone(), + aspect_ratio: edit_config.aspect_ratio.clone(), + resolution: edit_config.resolution.clone(), + safe_mode: edit_config.safe_mode, + }; + + let url = format!("{}/image/edit", config.base_url.trim_end_matches('/')); + + // The prompt is user content; keep it out of logs (mirrors the STT/TTS paths). + tracing::trace!( + model_id = edit_config.model_id, + "Sending Venice image edit API request" + ); + + let response = http + .post(&url) + .bearer_auth(&config.api_key) + .json(&request) + .send() + .await?; + + let status = response.status(); + if !status.is_success() { + // Body to the server log only, not into the returned error (which reaches the Matrix room). + let body = response.text().await.unwrap_or_default(); + tracing::warn!(%status, body, "Venice image edit request failed"); + return Err(anyhow::anyhow!( + "Venice image edit request failed with status {status}" + )); + } + + // The edit endpoint answers with raw binary image bytes, so read the body directly instead of + // parsing JSON. The actual format comes from the response Content-Type header; fall back to the + // configured `output_format` when the header is missing or unparseable. + let mime_type = response + .headers() + .get(reqwest::header::CONTENT_TYPE) + .and_then(|value| value.to_str().ok()) + .and_then(|value| value.parse::().ok()) + .unwrap_or_else(|| image_format_to_mime_type(edit_config.output_format.as_deref())); + + let bytes = response.bytes().await?.to_vec(); + + Ok(ImageEditResult { bytes, mime_type }) +} + +/// Map a Venice image `format`/`output_format` value (`jpeg`/`png`/`webp`) to its MIME type. +/// Venice defaults to `webp` when the format is unset, so an absent value maps to `image/webp`. +fn image_format_to_mime_type(format: Option<&str>) -> mxlink::mime::Mime { + match format.unwrap_or("webp") { + "jpeg" | "jpg" => mxlink::mime::IMAGE_JPEG, + "png" => mxlink::mime::IMAGE_PNG, + // No mxlink::mime constant for webp; parse it, falling back to PNG on any surprise value. + _ => "image/webp" + .parse() + .unwrap_or(mxlink::mime::IMAGE_PNG), + } +} diff --git a/src/agent/provider/venice/mod.rs b/src/agent/provider/venice/mod.rs new file mode 100644 index 0000000..bd95c17 --- /dev/null +++ b/src/agent/provider/venice/mod.rs @@ -0,0 +1,47 @@ +mod audio; +mod chat; +mod config; +mod controller; +mod images; +mod utils; +mod wire; + +#[cfg(test)] +mod tests; + +pub use config::Config; +pub use controller::Controller; + +use super::super::AgentInstantiationError; +use super::super::AgentInstantiationResult; +use super::ConfigTrait; +use super::controller::ControllerType; + +pub fn create_controller_from_yaml_value_config( + agent_id: &str, + config: serde_yaml_ng::Value, +) -> AgentInstantiationResult { + let config = match &config { + serde_yaml_ng::Value::Mapping(_) => { + let config: Config = + serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?; + + config + .validate() + .map_err(AgentInstantiationError::ConfigFailsValidation)?; + + config + } + _ => { + return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping( + agent_id.to_owned(), + )); + } + }; + + Ok(ControllerType::Venice(Box::new(Controller::new(config)))) +} + +pub fn default_config() -> Config { + Config::default() +} diff --git a/src/agent/provider/venice/tests.rs b/src/agent/provider/venice/tests.rs new file mode 100644 index 0000000..0cc1b42 --- /dev/null +++ b/src/agent/provider/venice/tests.rs @@ -0,0 +1,269 @@ +use mxlink::matrix_sdk::ruma::OwnedMxcUri; +use mxlink::matrix_sdk::ruma::events::room::message::{ + FileMessageEventContent, ImageMessageEventContent, +}; +use mxlink::mime; + +use super::super::ControllerTrait; +use crate::agent::AgentPurpose; +use crate::conversation::llm::{ + Author as LLMAuthor, FileDetails, ImageDetails, Message as LLMMessage, + MessageContent as LLMMessageContent, +}; + +use super::config::{Config, VeniceParameters, WebSearchMode}; +use super::controller::Controller; +use super::utils::convert_llm_messages_to_venice; +use super::wire::{ + ContentPart, EditImageRequest, GenerateImageRequest, MessageContent, SpeechRequest, +}; + +#[test] +fn config_round_trips_with_venice_parameters() { + let yaml = r#" +base_url: https://api.venice.ai/api/v1 +api_key: test-key +text_generation: + model_id: kimi-k2-5 + temperature: 0.7 + max_response_tokens: 1024 + max_context_tokens: 65536 + venice_parameters: + enable_web_search: "auto" + enable_web_citations: true +speech_to_text: + model_id: nvidia/parakeet-tdt-0.6b-v3 +"#; + + let config: Config = serde_yaml_ng::from_str(yaml).expect("config should deserialize"); + + let tg = config.text_generation.expect("text_generation present"); + let vp = tg.venice_parameters.expect("venice_parameters present"); + + assert!(matches!(vp.enable_web_search, Some(WebSearchMode::Auto))); + assert_eq!(vp.enable_web_citations, Some(true)); + assert_eq!(vp.character_slug, None); + + // The bag must serialize the enum to the exact wire string, and an unset knob must be ABSENT + // (not `null`) so the strict `additionalProperties: false` body is honored. + let json = serde_json::to_string(&vp).expect("serialize venice_parameters"); + assert!( + json.contains("\"enable_web_search\":\"auto\""), + "web search should be the literal \"auto\": {json}" + ); + assert!( + !json.contains("character_slug"), + "an unset knob must be omitted entirely: {json}" + ); + assert!(!json.contains("null"), "no nulls belong in the body: {json}"); +} + +#[test] +fn converts_image_to_data_uri_and_skips_files() { + let messages = vec![ + LLMMessage { + author: LLMAuthor::User, + sender_id: None, + timestamp: chrono::Utc::now(), + content: LLMMessageContent::Text("describe this".to_owned()), + }, + LLMMessage { + author: LLMAuthor::User, + sender_id: None, + timestamp: chrono::Utc::now(), + content: LLMMessageContent::Image(ImageDetails::new( + ImageMessageEventContent::plain( + "pic.png".to_owned(), + OwnedMxcUri::from("mxc://example.com/abc"), + ), + mime::IMAGE_PNG, + vec![1, 2, 3], + )), + }, + LLMMessage { + author: LLMAuthor::User, + sender_id: None, + timestamp: chrono::Utc::now(), + content: LLMMessageContent::File(FileDetails::new( + FileMessageEventContent::plain( + "doc.pdf".to_owned(), + OwnedMxcUri::from("mxc://example.com/def"), + ), + mime::APPLICATION_PDF, + vec![4, 5, 6], + )), + }, + ]; + + let converted = convert_llm_messages_to_venice(messages); + + // Text and image survive; the file is warn-skipped. + assert_eq!(converted.len(), 2); + + match &converted[0].content { + MessageContent::Text(text) => assert_eq!(text, "describe this"), + other => panic!("expected bare text, got {other:?}"), + } + + match &converted[1].content { + MessageContent::Parts(parts) => match &parts[0] { + ContentPart::ImageUrl { image_url } => assert!( + image_url.url.starts_with("data:image/png;base64,"), + "image should be inlined as a data URI: {}", + image_url.url + ), + }, + other => panic!("expected image parts, got {other:?}"), + } +} + +#[test] +fn supports_purpose_truth_table() { + let config: Config = serde_yaml_ng::from_str( + r#" +base_url: https://api.venice.ai/api/v1 +api_key: test-key +text_generation: + model_id: kimi-k2-5 +speech_to_text: + model_id: nvidia/parakeet-tdt-0.6b-v3 +"#, + ) + .expect("config should deserialize"); + + let controller = Controller::new(config); + + assert!(controller.supports_purpose(AgentPurpose::TextGeneration)); + assert!(controller.supports_purpose(AgentPurpose::SpeechToText)); + assert!(controller.supports_purpose(AgentPurpose::CatchAll)); + assert!(!controller.supports_purpose(AgentPurpose::TextToSpeech)); + assert!(!controller.supports_purpose(AgentPurpose::ImageGeneration)); +} + +#[test] +fn supports_purpose_true_when_image_and_tts_blocks_present() { + let config: Config = serde_yaml_ng::from_str( + r#" +base_url: https://api.venice.ai/api/v1 +api_key: test-key +text_to_speech: + model_id: tts-kokoro +image_generation: + model_id: chroma +"#, + ) + .expect("config should deserialize"); + + let controller = Controller::new(config); + + assert!(controller.supports_purpose(AgentPurpose::TextToSpeech)); + assert!(controller.supports_purpose(AgentPurpose::ImageGeneration)); +} + +#[test] +fn speech_request_serializes_voice_and_omits_unset() { + let request = SpeechRequest { + model: "tts-kokoro".to_owned(), + input: "hello".to_owned(), + voice: Some("af_sky".to_owned()), + speed: None, + response_format: Some("mp3".to_owned()), + prompt: None, + temperature: None, + top_p: None, + }; + + let json = serde_json::to_string(&request).expect("serialize SpeechRequest"); + + assert!( + json.contains("\"voice\":\"af_sky\""), + "voice should be present: {json}" + ); + assert!( + !json.contains("temperature"), + "an unset knob must be omitted (not null): {json}" + ); + assert!(!json.contains("null"), "no nulls belong in the body: {json}"); +} + +#[test] +fn generate_image_request_pins_flags_and_omits_unset() { + let request = GenerateImageRequest { + model: "chroma".to_owned(), + prompt: "a cat".to_owned(), + return_binary: false, + variants: 1, + negative_prompt: None, + cfg_scale: None, + steps: None, + style_preset: None, + seed: None, + safe_mode: None, + hide_watermark: None, + format: None, + width: None, + height: None, + aspect_ratio: None, + resolution: None, + quality: None, + lora_strength: None, + embed_exif_metadata: None, + enable_web_search: None, + }; + + let json = serde_json::to_string(&request).expect("serialize GenerateImageRequest"); + + assert!(json.contains("\"model\":\"chroma\""), "{json}"); + assert!( + json.contains("\"return_binary\":false"), + "return_binary must be pinned false: {json}" + ); + assert!( + json.contains("\"variants\":1"), + "variants must be pinned 1: {json}" + ); + assert!( + !json.contains("cfg_scale"), + "an unset knob must be omitted: {json}" + ); + assert!(!json.contains("null"), "no nulls belong in the body: {json}"); +} + +#[test] +fn edit_image_request_carries_model_and_base64_image() { + let request = EditImageRequest { + model: "firered-image-edit".to_owned(), + prompt: "make it a sunrise".to_owned(), + image: "aGVsbG8=".to_owned(), + output_format: None, + aspect_ratio: None, + resolution: None, + safe_mode: None, + }; + + let json = serde_json::to_string(&request).expect("serialize EditImageRequest"); + + assert!( + json.contains("\"model\":\"firered-image-edit\""), + "{json}" + ); + assert!( + json.contains("\"image\":\"aGVsbG8=\""), + "the base64 image string must be present: {json}" + ); + assert!( + !json.contains("output_format"), + "an unset knob must be omitted: {json}" + ); +} + +#[test] +fn web_search_mode_off_deserializes_from_bare_yaml_off() { + // `off` is a YAML-1.1 boolean but a plain string under serde_yaml_ng's YAML-1.2 core schema, + // so it deserializes straight into the lowercase `WebSearchMode::Off`. This pins that the + // sample config and docs can use the bare, unquoted `off` without it parsing as a boolean. + let params: VeniceParameters = + serde_yaml_ng::from_str("enable_web_search: off").expect("bare `off` should deserialize"); + + assert!(matches!(params.enable_web_search, Some(WebSearchMode::Off))); +} diff --git a/src/agent/provider/venice/utils.rs b/src/agent/provider/venice/utils.rs new file mode 100644 index 0000000..b4af102 --- /dev/null +++ b/src/agent/provider/venice/utils.rs @@ -0,0 +1,55 @@ +use crate::conversation::llm::{ + Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent, +}; +use crate::utils::base64::base64_encode; + +use super::wire::{ChatMessage, ContentPart, ImageUrl, MessageContent}; + +pub fn convert_llm_messages_to_venice(messages: Vec) -> Vec { + let mut venice_messages: Vec = Vec::with_capacity(messages.len()); + + for message in messages { + if let Some(venice_message) = convert_llm_message_to_venice(message) { + venice_messages.push(venice_message); + } + } + + venice_messages +} + +fn convert_llm_message_to_venice(message: LLMMessage) -> Option { + let role = match message.author { + LLMAuthor::Prompt => "system", + LLMAuthor::Assistant => "assistant", + LLMAuthor::User => "user", + }; + + match message.content { + LLMMessageContent::Text(text) => Some(ChatMessage { + role: role.to_owned(), + content: MessageContent::Text(text), + }), + LLMMessageContent::Image(image_details) => { + // Inline the image as a base64 data URI, the same shape the OpenAI vision content + // part uses. This is the gap the openai_compat provider can't fill (it drops images). + let data_uri = format!( + "data:{};base64,{}", + image_details.mime, + base64_encode(&image_details.data) + ); + + Some(ChatMessage { + role: role.to_owned(), + content: MessageContent::Parts(vec![ContentPart::ImageUrl { + image_url: ImageUrl { url: data_uri }, + }]), + }) + } + LLMMessageContent::File(_file_details) => { + tracing::warn!( + "The Venice provider does not support file content. This file message will be skipped." + ); + None + } + } +} diff --git a/src/agent/provider/venice/wire.rs b/src/agent/provider/venice/wire.rs new file mode 100644 index 0000000..af594bd --- /dev/null +++ b/src/agent/provider/venice/wire.rs @@ -0,0 +1,212 @@ +//! Serde structs modeling Venice's `/chat/completions`, `/audio/transcriptions`, +//! `/audio/speech`, `/image/generate`, and `/image/edit` wire shapes. Request types are +//! `Serialize`-only (we build them, Venice never sends them back); response types are +//! `Deserialize`-only. Keeping the split means the untagged request content enum is never on a +//! deserialize path, so a surprise response shape can't fail to match it. +//! +//! Field names match Venice's schema 1:1 (so the config's `model_id` becomes `model` here). Every +//! request body is `additionalProperties: false`, so optional knobs carry `skip_serializing_if` +//! to omit rather than send `null`. `/audio/speech` and `/image/edit` return raw binary (no +//! response struct); only `/image/generate` returns JSON (`GenerateImageResponse`). + +use serde::{Deserialize, Serialize}; + +use super::config::VeniceParameters; + +#[derive(Debug, Serialize)] +pub struct ChatCompletionRequest { + pub model: String, + + pub messages: Vec, + + #[serde(skip_serializing_if = "Option::is_none")] + pub temperature: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub max_completion_tokens: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub venice_parameters: Option, +} + +#[derive(Debug, Serialize)] +pub struct ChatMessage { + pub role: String, + + pub content: MessageContent, +} + +/// A message body is either a bare string or a list of content parts. Venice accepts both; we +/// send the parts form only when a message carries an image (baibot keeps text and images in +/// separate messages, so a parts list only ever holds images in v1). +#[derive(Debug, Serialize)] +#[serde(untagged)] +pub enum MessageContent { + Text(String), + Parts(Vec), +} + +#[derive(Debug, Serialize)] +#[serde(tag = "type", rename_all = "snake_case")] +pub enum ContentPart { + ImageUrl { image_url: ImageUrl }, +} + +#[derive(Debug, Serialize)] +pub struct ImageUrl { + /// A `data:;base64,` URI for inline images. + pub url: String, +} + +/// Standard OpenAI-shaped chat completion response. We only read `choices[0].message.content`; +/// when web search is on, Venice inlines citations as `^n^` superscripts in that content and we +/// pass it through untouched. +#[derive(Debug, Deserialize)] +pub struct ChatCompletionResponse { + pub choices: Vec, +} + +#[derive(Debug, Deserialize)] +pub struct ChatChoice { + pub message: ResponseMessage, +} + +#[derive(Debug, Deserialize)] +pub struct ResponseMessage { + #[serde(default)] + pub content: Option, +} + +/// `/audio/transcriptions` response. We read `text`; the optional `duration`/`timestamps` the +/// API can return are not used in v1. +#[derive(Debug, Deserialize)] +pub struct TranscriptionResponse { + pub text: String, +} + +/// `/audio/speech` (`CreateSpeechRequestSchema`) request. `input` and `model` are always sent; +/// the rest are omitted when unset. The response is raw binary audio, so there is no response +/// struct. +#[derive(Debug, Serialize)] +pub struct SpeechRequest { + pub model: String, + + pub input: String, + + #[serde(skip_serializing_if = "Option::is_none")] + pub voice: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub speed: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub response_format: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub prompt: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub temperature: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub top_p: Option, +} + +/// `/image/generate` (`GenerateImageRequest`) request. `return_binary` is pinned `false` and +/// `variants` to `1` by the builder: baibot wants exactly one image returned as base64-in-JSON, +/// which `GenerateImageResponse` then decodes. Flipping `return_binary` would make Venice answer +/// with raw binary and break that JSON decode, so it is not configurable. +#[derive(Debug, Serialize)] +pub struct GenerateImageRequest { + pub model: String, + + pub prompt: String, + + pub return_binary: bool, + + pub variants: u32, + + #[serde(skip_serializing_if = "Option::is_none")] + pub negative_prompt: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub cfg_scale: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub steps: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub style_preset: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub seed: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub safe_mode: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub hide_watermark: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub format: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub width: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub height: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub aspect_ratio: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub resolution: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub quality: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub lora_strength: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub embed_exif_metadata: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub enable_web_search: Option, +} + +/// `/image/generate` response when `return_binary` is false: a JSON envelope carrying the images +/// as base64 strings. We read `images[0]`; `request`/`timing` and other fields are ignored. `id` +/// is telemetry only (logged, never used for correctness), so it is optional: a response that +/// carries usable `images` must not fail to deserialize just because the telemetry field drifted. +#[derive(Debug, Deserialize)] +pub struct GenerateImageResponse { + #[serde(default)] + pub id: Option, + pub images: Vec, +} + +/// `/image/edit` (`EditImageRequest`) request. The source `image` is a base64-encoded string +/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64-in-JSON, no multipart). +/// The response is raw binary, so there is no response struct. +#[derive(Debug, Serialize)] +pub struct EditImageRequest { + pub model: String, + + pub prompt: String, + + /// Base64-encoded source image bytes. + pub image: String, + + #[serde(skip_serializing_if = "Option::is_none")] + pub output_format: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub aspect_ratio: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub resolution: Option, + + #[serde(skip_serializing_if = "Option::is_none")] + pub safe_mode: Option, +}