Venice: add 💭 Reasoning spoiler block (html <details>)

This commit is contained in:
Aine
2026-06-23 01:30:37 +01:00
parent 9e5f6de965
commit 5de497482d
8 changed files with 47 additions and 29 deletions

View File

@@ -4,7 +4,7 @@
- (**Feature**) Add prompt caching to the Venice provider, on by default (`prompt_cache_retention: 24h`). baibot derives the cache key from the system prompt and the conversation start time (both fixed for the life of a conversation), so a long, stable system prompt stays cached across the day instead of being reprocessed and re-billed on every turn. See [Text Generation / Prompt Override](./docs/configuration/text-generation.md#️-prompt-override).
- (**Feature**) Wire up the rest of Venice's sampling and reasoning controls: top-level `top_p`, `frequency_penalty`, `presence_penalty`, `repetition_penalty`, and `reasoning_effort`; `verbosity` in the `venice_parameters` bag; and a `show_reasoning` toggle that appends the model's reasoning to the reply (off by default). See the [Venice configuration reference](./docs/providers.md#venice).
- (**Feature**) Wire up the rest of Venice's sampling and reasoning controls: top-level `top_p`, `frequency_penalty`, `presence_penalty`, `repetition_penalty`, and `reasoning_effort`; `verbosity` in the `venice_parameters` bag; and a `show_reasoning` toggle that appends the model's reasoning to the reply as a collapsible, folded-by-default `💭 Reasoning` block (off by default). See the [Venice configuration reference](./docs/providers.md#venice).
- (**Feature**) Render Venice web-search citations as readable `[n]` references with a `Sources:` list of links, instead of leaving Venice's raw `^n^` superscripts in the reply.

View File

@@ -203,7 +203,7 @@ Every parameter below is optional unless marked otherwise. Omitting a knob lets
| `repetition_penalty` | Penalize repetition. Values above `1.0` discourage repeats. | — |
| `reasoning_effort` | Reasoning budget for models that support it: `low`, `medium`, `high`. | — |
| `prompt_cache_retention` | How long Venice keeps the prompt prefix cached: `default`, `extended`, or `24h`. `24h` is the lever that makes a long, stable system prompt cheap across a day of conversations. | `24h` |
| `show_reasoning` | Append the model's reasoning (its `reasoning_content`) below the answer. Reads a field separate from the answer text, so it works regardless of `strip_thinking_response`. | `false` |
| `show_reasoning` | Append the model's reasoning (its `reasoning_content`) below the answer, as a collapsible `💭 Reasoning` block that stays folded until clicked. Reads a field separate from the answer text, so it works regardless of `strip_thinking_response`. | `false` |
**`text_generation.venice_parameters`** — Venice-specific request knobs sent in the `venice_parameters` bag. Set any of them to override Venice's behavior. The `Default` column shows the value baibot's sample config ships; a `—` means the knob is left unset, so Venice's own default applies.

View File

@@ -20,8 +20,9 @@ text_generation:
# repetition_penalty: 1.0
# Reasoning budget for models that support it: low, medium, high.
# reasoning_effort: medium
# Append the model's reasoning below the answer. Reads a field separate from the answer text, so
# it works alongside strip_thinking_response (which only strips <think> blocks from the answer).
# Append the model's reasoning below the answer as a collapsible "💭 Reasoning" block (folded by
# default). Reads a field separate from the answer text, so it works alongside
# strip_thinking_response (which only strips <think> blocks from the answer).
# show_reasoning: true
# Venice-specific request parameters. Only the keys present below are sent to Venice; omit a
# key to fall back to Venice's own default. Omitting a knob is NOT the same as setting it to

View File

@@ -184,9 +184,7 @@ impl ControllerTrait for ControllerType {
ControllerType::Anthropic(controller) => {
controller.generate_image(prompt, params).await
}
ControllerType::Venice(controller) => {
controller.generate_image(prompt, params).await
}
ControllerType::Venice(controller) => controller.generate_image(prompt, params).await,
}
}

View File

@@ -115,10 +115,7 @@ pub async fn generate_text(
venice_parameters,
};
let url = format!(
"{}/chat/completions",
config.base_url.trim_end_matches('/')
);
let url = format!("{}/chat/completions", config.base_url.trim_end_matches('/'));
tracing::trace!(
model = text_generation_config.model_id,
@@ -201,6 +198,13 @@ pub(super) fn derive_prompt_cache_key(prompt_text: &str, conversation_start_time
/// `strip_thinking_response`, which only strips inline `<think>` blocks from `content`), so reading
/// it here is independent of that knob. Default-off matches today's behavior: thinking never reaches
/// a room that did not ask for it.
///
/// The thinking renders as a Matrix-native collapsible `<details>` block: folded by default, one
/// click to expand, so it stays out of the way of the answer instead of dumping a wall of reasoning
/// inline. This survives the send path: the reply goes through markdown (`send_text_markdown`),
/// whose pulldown-cmark pass writes raw HTML verbatim rather than escaping it, and ruma's HTML
/// sanitizer allow-lists `<details>`/`<summary>`. Clients that do not render `<details>` degrade to
/// showing the summary and reasoning inline, so nothing is lost there either.
pub(super) fn append_reasoning(
text: String,
reasoning_content: Option<String>,
@@ -212,7 +216,13 @@ pub(super) fn append_reasoning(
match reasoning_content {
Some(reasoning) if !reasoning.trim().is_empty() => {
format!("{text}\n\n---\n\n*Reasoning:*\n\n{reasoning}")
// The blank lines around the trimmed reasoning keep it a separate markdown block from
// the surrounding `<details>`/`</details>` HTML blocks, so the reasoning itself still
// renders as markdown (lists, code, emphasis) inside the collapsible.
let reasoning = reasoning.trim();
format!(
"{text}\n\n<details><summary>💭 Reasoning</summary>\n\n{reasoning}\n\n</details>"
)
}
_ => text,
}

View File

@@ -185,8 +185,6 @@ fn image_format_to_mime_type(format: Option<&str>) -> mxlink::mime::Mime {
"jpeg" | "jpg" => mxlink::mime::IMAGE_JPEG,
"png" => mxlink::mime::IMAGE_PNG,
// No mxlink::mime constant for webp; parse it, falling back to PNG on any surprise value.
_ => "image/webp"
.parse()
.unwrap_or(mxlink::mime::IMAGE_PNG),
_ => "image/webp".parse().unwrap_or(mxlink::mime::IMAGE_PNG),
}
}

View File

@@ -57,7 +57,10 @@ speech_to_text:
!json.contains("character_slug"),
"an unset knob must be omitted entirely: {json}"
);
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
assert!(
!json.contains("null"),
"no nulls belong in the body: {json}"
);
}
#[test]
@@ -97,8 +100,7 @@ fn converts_text_image_and_file_to_content_parts() {
},
];
let converted =
convert_llm_messages_to_venice(messages).expect("conversion should succeed");
let converted = convert_llm_messages_to_venice(messages).expect("conversion should succeed");
// Text, image, AND file all survive now: the file is no longer warn-skipped.
assert_eq!(converted.len(), 3);
@@ -202,7 +204,10 @@ fn speech_request_serializes_voice_and_omits_unset() {
!json.contains("temperature"),
"an unset knob must be omitted (not null): {json}"
);
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
assert!(
!json.contains("null"),
"no nulls belong in the body: {json}"
);
}
#[test]
@@ -245,7 +250,10 @@ fn generate_image_request_pins_flags_and_omits_unset() {
!json.contains("cfg_scale"),
"an unset knob must be omitted: {json}"
);
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
assert!(
!json.contains("null"),
"no nulls belong in the body: {json}"
);
}
#[test]
@@ -262,10 +270,7 @@ fn edit_image_request_carries_model_and_base64_image() {
let json = serde_json::to_string(&request).expect("serialize EditImageRequest");
assert!(
json.contains("\"model\":\"firered-image-edit\""),
"{json}"
);
assert!(json.contains("\"model\":\"firered-image-edit\""), "{json}");
assert!(
json.contains("\"image\":\"aGVsbG8=\""),
"the base64 image string must be present: {json}"
@@ -504,10 +509,14 @@ fn reasoning_is_appended_only_when_show_reasoning_is_set() {
let off = append_reasoning(base.clone(), Some("secret thinking".to_owned()), false);
assert_eq!(off, "the answer");
// On: thinking is appended below the answer.
let on = append_reasoning(base.clone(), Some("visible thinking".to_owned()), true);
assert!(on.contains("the answer"));
assert!(on.contains("visible thinking"));
// On: thinking is appended below the answer in a collapsible <details> block (folded by
// default, expandable in clients that support it).
let on = append_reasoning(base.clone(), Some(" visible thinking ".to_owned()), true);
assert!(on.starts_with("the answer"));
assert!(on.contains("<details><summary>💭 Reasoning</summary>"));
assert!(on.contains("</details>"));
// The reasoning sits as its own markdown block (blank lines around it) and is trimmed.
assert!(on.contains("\n\nvisible thinking\n\n"));
// On but empty or missing reasoning: nothing is appended.
assert_eq!(

View File

@@ -10,7 +10,9 @@ use super::wire::{ChatMessage, ContentPart, FilePart, ImageUrl, MessageContent};
/// API; the 413 status branch in `chat.rs` is the backstop if a file slips past this guard.
const MAX_FILE_BYTES: usize = 25 * 1024 * 1024;
pub fn convert_llm_messages_to_venice(messages: Vec<LLMMessage>) -> anyhow::Result<Vec<ChatMessage>> {
pub fn convert_llm_messages_to_venice(
messages: Vec<LLMMessage>,
) -> anyhow::Result<Vec<ChatMessage>> {
let mut venice_messages: Vec<ChatMessage> = Vec::with_capacity(messages.len());
for message in messages {