From 5de497482dbade329cd137c44b567d9d2660871b Mon Sep 17 00:00:00 2001 From: Aine Date: Tue, 23 Jun 2026 01:30:37 +0100 Subject: [PATCH] =?UTF-8?q?Venice:=20add=20`=F0=9F=92=AD=20Reasoning`=20sp?= =?UTF-8?q?oiler=20block=20(html=20
)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- CHANGELOG.md | 2 +- docs/providers.md | 2 +- docs/sample-provider-configs/venice.yml | 5 ++-- src/agent/provider/controller.rs | 4 +-- src/agent/provider/venice/chat.rs | 20 ++++++++++---- src/agent/provider/venice/images.rs | 4 +-- src/agent/provider/venice/tests.rs | 35 ++++++++++++++++--------- src/agent/provider/venice/utils.rs | 4 ++- 8 files changed, 47 insertions(+), 29 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 534a0c3..10f2b6d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,7 +4,7 @@ - (**Feature**) Add prompt caching to the Venice provider, on by default (`prompt_cache_retention: 24h`). baibot derives the cache key from the system prompt and the conversation start time (both fixed for the life of a conversation), so a long, stable system prompt stays cached across the day instead of being reprocessed and re-billed on every turn. See [Text Generation / Prompt Override](./docs/configuration/text-generation.md#️-prompt-override). -- (**Feature**) Wire up the rest of Venice's sampling and reasoning controls: top-level `top_p`, `frequency_penalty`, `presence_penalty`, `repetition_penalty`, and `reasoning_effort`; `verbosity` in the `venice_parameters` bag; and a `show_reasoning` toggle that appends the model's reasoning to the reply (off by default). See the [Venice configuration reference](./docs/providers.md#venice). +- (**Feature**) Wire up the rest of Venice's sampling and reasoning controls: top-level `top_p`, `frequency_penalty`, `presence_penalty`, `repetition_penalty`, and `reasoning_effort`; `verbosity` in the `venice_parameters` bag; and a `show_reasoning` toggle that appends the model's reasoning to the reply as a collapsible, folded-by-default `💭 Reasoning` block (off by default). See the [Venice configuration reference](./docs/providers.md#venice). - (**Feature**) Render Venice web-search citations as readable `[n]` references with a `Sources:` list of links, instead of leaving Venice's raw `^n^` superscripts in the reply. diff --git a/docs/providers.md b/docs/providers.md index 0e6d160..4332150 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -203,7 +203,7 @@ Every parameter below is optional unless marked otherwise. Omitting a knob lets | `repetition_penalty` | Penalize repetition. Values above `1.0` discourage repeats. | — | | `reasoning_effort` | Reasoning budget for models that support it: `low`, `medium`, `high`. | — | | `prompt_cache_retention` | How long Venice keeps the prompt prefix cached: `default`, `extended`, or `24h`. `24h` is the lever that makes a long, stable system prompt cheap across a day of conversations. | `24h` | -| `show_reasoning` | Append the model's reasoning (its `reasoning_content`) below the answer. Reads a field separate from the answer text, so it works regardless of `strip_thinking_response`. | `false` | +| `show_reasoning` | Append the model's reasoning (its `reasoning_content`) below the answer, as a collapsible `💭 Reasoning` block that stays folded until clicked. Reads a field separate from the answer text, so it works regardless of `strip_thinking_response`. | `false` | **`text_generation.venice_parameters`** — Venice-specific request knobs sent in the `venice_parameters` bag. Set any of them to override Venice's behavior. The `Default` column shows the value baibot's sample config ships; a `—` means the knob is left unset, so Venice's own default applies. diff --git a/docs/sample-provider-configs/venice.yml b/docs/sample-provider-configs/venice.yml index 7762dcc..c1835d5 100644 --- a/docs/sample-provider-configs/venice.yml +++ b/docs/sample-provider-configs/venice.yml @@ -20,8 +20,9 @@ text_generation: # repetition_penalty: 1.0 # Reasoning budget for models that support it: low, medium, high. # reasoning_effort: medium - # Append the model's reasoning below the answer. Reads a field separate from the answer text, so - # it works alongside strip_thinking_response (which only strips blocks from the answer). + # Append the model's reasoning below the answer as a collapsible "💭 Reasoning" block (folded by + # default). Reads a field separate from the answer text, so it works alongside + # strip_thinking_response (which only strips blocks from the answer). # show_reasoning: true # Venice-specific request parameters. Only the keys present below are sent to Venice; omit a # key to fall back to Venice's own default. Omitting a knob is NOT the same as setting it to diff --git a/src/agent/provider/controller.rs b/src/agent/provider/controller.rs index 4e2df11..cfe9d2d 100644 --- a/src/agent/provider/controller.rs +++ b/src/agent/provider/controller.rs @@ -184,9 +184,7 @@ impl ControllerTrait for ControllerType { ControllerType::Anthropic(controller) => { controller.generate_image(prompt, params).await } - ControllerType::Venice(controller) => { - controller.generate_image(prompt, params).await - } + ControllerType::Venice(controller) => controller.generate_image(prompt, params).await, } } diff --git a/src/agent/provider/venice/chat.rs b/src/agent/provider/venice/chat.rs index 2c74644..89e2709 100644 --- a/src/agent/provider/venice/chat.rs +++ b/src/agent/provider/venice/chat.rs @@ -115,10 +115,7 @@ pub async fn generate_text( venice_parameters, }; - let url = format!( - "{}/chat/completions", - config.base_url.trim_end_matches('/') - ); + let url = format!("{}/chat/completions", config.base_url.trim_end_matches('/')); tracing::trace!( model = text_generation_config.model_id, @@ -201,6 +198,13 @@ pub(super) fn derive_prompt_cache_key(prompt_text: &str, conversation_start_time /// `strip_thinking_response`, which only strips inline `` blocks from `content`), so reading /// it here is independent of that knob. Default-off matches today's behavior: thinking never reaches /// a room that did not ask for it. +/// +/// The thinking renders as a Matrix-native collapsible `
` block: folded by default, one +/// click to expand, so it stays out of the way of the answer instead of dumping a wall of reasoning +/// inline. This survives the send path: the reply goes through markdown (`send_text_markdown`), +/// whose pulldown-cmark pass writes raw HTML verbatim rather than escaping it, and ruma's HTML +/// sanitizer allow-lists `
`/``. Clients that do not render `
` degrade to +/// showing the summary and reasoning inline, so nothing is lost there either. pub(super) fn append_reasoning( text: String, reasoning_content: Option, @@ -212,7 +216,13 @@ pub(super) fn append_reasoning( match reasoning_content { Some(reasoning) if !reasoning.trim().is_empty() => { - format!("{text}\n\n---\n\n*Reasoning:*\n\n{reasoning}") + // The blank lines around the trimmed reasoning keep it a separate markdown block from + // the surrounding `
`/`
` HTML blocks, so the reasoning itself still + // renders as markdown (lists, code, emphasis) inside the collapsible. + let reasoning = reasoning.trim(); + format!( + "{text}\n\n
💭 Reasoning\n\n{reasoning}\n\n
" + ) } _ => text, } diff --git a/src/agent/provider/venice/images.rs b/src/agent/provider/venice/images.rs index 53416de..af2e721 100644 --- a/src/agent/provider/venice/images.rs +++ b/src/agent/provider/venice/images.rs @@ -185,8 +185,6 @@ fn image_format_to_mime_type(format: Option<&str>) -> mxlink::mime::Mime { "jpeg" | "jpg" => mxlink::mime::IMAGE_JPEG, "png" => mxlink::mime::IMAGE_PNG, // No mxlink::mime constant for webp; parse it, falling back to PNG on any surprise value. - _ => "image/webp" - .parse() - .unwrap_or(mxlink::mime::IMAGE_PNG), + _ => "image/webp".parse().unwrap_or(mxlink::mime::IMAGE_PNG), } } diff --git a/src/agent/provider/venice/tests.rs b/src/agent/provider/venice/tests.rs index 3395845..d951932 100644 --- a/src/agent/provider/venice/tests.rs +++ b/src/agent/provider/venice/tests.rs @@ -57,7 +57,10 @@ speech_to_text: !json.contains("character_slug"), "an unset knob must be omitted entirely: {json}" ); - assert!(!json.contains("null"), "no nulls belong in the body: {json}"); + assert!( + !json.contains("null"), + "no nulls belong in the body: {json}" + ); } #[test] @@ -97,8 +100,7 @@ fn converts_text_image_and_file_to_content_parts() { }, ]; - let converted = - convert_llm_messages_to_venice(messages).expect("conversion should succeed"); + let converted = convert_llm_messages_to_venice(messages).expect("conversion should succeed"); // Text, image, AND file all survive now: the file is no longer warn-skipped. assert_eq!(converted.len(), 3); @@ -202,7 +204,10 @@ fn speech_request_serializes_voice_and_omits_unset() { !json.contains("temperature"), "an unset knob must be omitted (not null): {json}" ); - assert!(!json.contains("null"), "no nulls belong in the body: {json}"); + assert!( + !json.contains("null"), + "no nulls belong in the body: {json}" + ); } #[test] @@ -245,7 +250,10 @@ fn generate_image_request_pins_flags_and_omits_unset() { !json.contains("cfg_scale"), "an unset knob must be omitted: {json}" ); - assert!(!json.contains("null"), "no nulls belong in the body: {json}"); + assert!( + !json.contains("null"), + "no nulls belong in the body: {json}" + ); } #[test] @@ -262,10 +270,7 @@ fn edit_image_request_carries_model_and_base64_image() { let json = serde_json::to_string(&request).expect("serialize EditImageRequest"); - assert!( - json.contains("\"model\":\"firered-image-edit\""), - "{json}" - ); + assert!(json.contains("\"model\":\"firered-image-edit\""), "{json}"); assert!( json.contains("\"image\":\"aGVsbG8=\""), "the base64 image string must be present: {json}" @@ -504,10 +509,14 @@ fn reasoning_is_appended_only_when_show_reasoning_is_set() { let off = append_reasoning(base.clone(), Some("secret thinking".to_owned()), false); assert_eq!(off, "the answer"); - // On: thinking is appended below the answer. - let on = append_reasoning(base.clone(), Some("visible thinking".to_owned()), true); - assert!(on.contains("the answer")); - assert!(on.contains("visible thinking")); + // On: thinking is appended below the answer in a collapsible
block (folded by + // default, expandable in clients that support it). + let on = append_reasoning(base.clone(), Some(" visible thinking ".to_owned()), true); + assert!(on.starts_with("the answer")); + assert!(on.contains("
💭 Reasoning")); + assert!(on.contains("
")); + // The reasoning sits as its own markdown block (blank lines around it) and is trimmed. + assert!(on.contains("\n\nvisible thinking\n\n")); // On but empty or missing reasoning: nothing is appended. assert_eq!( diff --git a/src/agent/provider/venice/utils.rs b/src/agent/provider/venice/utils.rs index 245ee87..e8ce428 100644 --- a/src/agent/provider/venice/utils.rs +++ b/src/agent/provider/venice/utils.rs @@ -10,7 +10,9 @@ use super::wire::{ChatMessage, ContentPart, FilePart, ImageUrl, MessageContent}; /// API; the 413 status branch in `chat.rs` is the backstop if a file slips past this guard. const MAX_FILE_BYTES: usize = 25 * 1024 * 1024; -pub fn convert_llm_messages_to_venice(messages: Vec) -> anyhow::Result> { +pub fn convert_llm_messages_to_venice( + messages: Vec, +) -> anyhow::Result> { let mut venice_messages: Vec = Vec::with_capacity(messages.len()); for message in messages {