diff --git a/CHANGELOG.md b/CHANGELOG.md index 23edc7d..1fa58b6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,10 @@ +# (2026-06-26) Version 1.24.0 + +- (**Feature**) Add an opt-in 💭 **thinking notice** for text generation. When enabled, a slow response (for example, from a reasoning model that runs for minutes) posts a "thinking…" placeholder after a short delay, refreshes it periodically with varying flavor text, and then edits that same message into the final answer, so a long wait no longer looks like a stuck bot. The notice is **disabled by default** and configurable per-room or globally via `text-generation set-thinking-notice-enabled true`. Fast responses (under the delay threshold) never show a placeholder. See the [text-generation configuration docs](./docs/configuration/text-generation.md#-thinking-notice). + +- (**Bugfix**) The [Venice](https://venice.ai) unsupported-field auto-recovery (added in 1.23.1) now actually remembers rejections across messages. The cache lived on the provider's controller, which is rebuilt on every message for room-local and global agents, so each turn started with an empty cache, re-sent the unsupported field, and logged the same `400 Bad Request` warning again. The cache is now process-global (keyed per Venice deployment), so a field a model rejects is dropped proactively on every later request instead of being re-discovered each turn. + + # (2026-06-24) Version 1.23.1 - (**Bugfix**) The [Venice](https://venice.ai) provider now auto-recovers when a model rejects an optional knob it does not support. Venice's request body is strict (`additionalProperties: false`), so a model that lacks `prompt_cache_retention`, `reasoning_effort`, or `prompt_cache_key` rejected the whole request with a `400 Bad Request` — breaking agent creation and every reply. baibot now drops the unsupported field and retries, remembering the rejection per model so later requests skip it without a wasted round-trip. Only these meaning-preserving fields are dropped; sampling knobs that change the output (`temperature`, `top_p`, the penalties) are never silently removed and still surface as an error. diff --git a/Cargo.lock b/Cargo.lock index 84d7d34..b1d9e1c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -315,7 +315,7 @@ dependencies = [ [[package]] name = "baibot" -version = "1.23.1" +version = "1.24.0" dependencies = [ "anthropic", "anyhow", diff --git a/Cargo.toml b/Cargo.toml index b1f29f8..e110c7e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later" readme = "README.md" keywords = ["matrix", "chat", "bot", "AI", "LLM"] include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"] -version = "1.23.1" +version = "1.24.0" edition = "2024" [lib] diff --git a/docs/configuration/text-generation.md b/docs/configuration/text-generation.md index 283fb36..0a2f0f3 100644 --- a/docs/configuration/text-generation.md +++ b/docs/configuration/text-generation.md @@ -57,6 +57,15 @@ This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_langua This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)). +### 💭 Thinking Notice + +The bot can post a 💭 **"thinking…" notice** while text generation is running, useful for slow models (for example, reasoning models that may run for minutes) where the response would otherwise look stuck. + +When enabled, a placeholder message appears only after a short delay (so fast responses get no notice), updates periodically with varying status text, and is then edited in place to become the final answer. + +This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-thinking-notice-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)). + + ### 👤 Sender Context Mode In multi-user rooms, it may be useful for the model to know which participant sent each message in the conversation context. diff --git a/src/agent/provider/entity/text_generation/prompt_variables.rs b/src/agent/provider/entity/text_generation/prompt_variables.rs index d25b557..772c0f0 100644 --- a/src/agent/provider/entity/text_generation/prompt_variables.rs +++ b/src/agent/provider/entity/text_generation/prompt_variables.rs @@ -1,6 +1,7 @@ use chrono::{DateTime, Utc}; use std::collections::HashMap; +#[derive(Clone)] pub struct TextGenerationPromptVariables { map: HashMap, } diff --git a/src/agent/provider/venice/controller.rs b/src/agent/provider/venice/controller.rs index e17f1f0..94daec5 100644 --- a/src/agent/provider/venice/controller.rs +++ b/src/agent/provider/venice/controller.rs @@ -35,10 +35,15 @@ impl Controller { .build() .unwrap_or_else(|_| reqwest::Client::new()); + // Process-global per-deployment cache rather than a fresh one: dynamic agents rebuild their + // controller on every message, so an instance-owned cache would never retain a learned + // rejection and the same unsupported field would 400 (and warn) on every turn. + let unsupported_fields = UnsupportedFieldsCache::shared_for(&config.base_url); + Self { config, http, - unsupported_fields: UnsupportedFieldsCache::default(), + unsupported_fields, } } } diff --git a/src/agent/provider/venice/recovery.rs b/src/agent/provider/venice/recovery.rs index ab52f3b..84bfc45 100644 --- a/src/agent/provider/venice/recovery.rs +++ b/src/agent/provider/venice/recovery.rs @@ -45,7 +45,38 @@ pub(super) struct UnsupportedFieldsCache { inner: Arc>>>, } +/// Process-global registry of per-deployment caches, keyed by Venice base URL. A `Controller` is +/// rebuilt from its config on every message for dynamic (global/room-local) agents (see +/// `agent::manager::available_room_agents_by_room_config_context`), so a cache stored on the +/// controller would be reset each message and the "process-lived" intent above would never hold. +/// The registry lets a rebuilt controller re-attach to the same cache. Keyed by base URL because +/// "which fields a model rejects" is a property of the deployment+model, not of the agent instance: +/// agents on the same Venice deployment share learnings, while a distinct deployment keeps its own, +/// so a model name that supports a field on one deployment is never proactively stripped against +/// another. +fn registry() -> &'static RwLock> { + static REGISTRY: OnceLock>> = OnceLock::new(); + REGISTRY.get_or_init(|| RwLock::new(HashMap::new())) +} + impl UnsupportedFieldsCache { + /// The process-global cache for `base_url`, created on first use and shared by every controller + /// for that deployment thereafter. This is what makes a learned rejection survive the + /// per-message controller rebuild. A poisoned registry lock degrades to an isolated cache: + /// correctness holds, only cross-controller sharing is lost for that call. + pub(super) fn shared_for(base_url: &str) -> Self { + if let Ok(map) = registry().read() + && let Some(cache) = map.get(base_url) + { + return cache.clone(); + } + + match registry().write() { + Ok(mut map) => map.entry(base_url.to_owned()).or_default().clone(), + Err(_) => Self::default(), + } + } + /// Fields already known unsupported for `model_id`. Returns an empty set on an unknown model or /// a poisoned lock, so a cache failure degrades to "strip nothing proactively" rather than /// breaking the request path. @@ -231,6 +262,33 @@ mod tests { assert!(cache.known_for("kimi-k2-5").is_empty()); } + #[test] + fn shared_for_survives_controller_rebuild_and_isolates_deployments() { + // Distinct, test-only base URLs so the process-global registry can't collide with another + // test running in parallel. + let url_a = "https://shared-for-test-a.invalid/api/v1"; + let url_b = "https://shared-for-test-b.invalid/api/v1"; + + // A controller rebuilt for the same deployment (a fresh `shared_for` call, as happens per + // message for dynamic agents) re-attaches to the SAME cache, so an earlier learning holds. + let first = UnsupportedFieldsCache::shared_for(url_a); + first.record("model-x", "reasoning_effort"); + + let rebuilt = UnsupportedFieldsCache::shared_for(url_a); + assert!( + rebuilt.known_for("model-x").contains("reasoning_effort"), + "a rebuilt controller for the same deployment must see the earlier rejection" + ); + + // A different deployment keeps its own learnings: a model name that rejects a field on one + // Venice must not silence/strip it on another. + let other_deployment = UnsupportedFieldsCache::shared_for(url_b); + assert!( + other_deployment.known_for("model-x").is_empty(), + "rejections must not leak across deployments" + ); + } + #[test] fn extracts_a_human_message_from_error_envelopes() { assert_eq!( diff --git a/src/bot/messaging.rs b/src/bot/messaging.rs index ed83ef0..623254f 100644 --- a/src/bot/messaging.rs +++ b/src/bot/messaging.rs @@ -1,8 +1,11 @@ use mxlink::matrix_sdk::{ Room, + room::edit::EditedContent, ruma::{ - OwnedEventId, api::client::receipt::create_receipt::v3::ReceiptType, - events::room::message::OriginalSyncRoomMessageEvent, + EventId, OwnedEventId, api::client::receipt::create_receipt::v3::ReceiptType, + events::room::message::{ + OriginalSyncRoomMessageEvent, RoomMessageEventContentWithoutRelation, + }, }, }; @@ -52,6 +55,83 @@ impl Messaging { } } + /// Like `send_text_markdown_no_fail`, but logs send failures at `warn` instead of `error`. + /// For cosmetic, best-effort sends (the thinking-notice placeholder) where a failure is + /// acceptable-impact: the real response still ships, so this must NOT page the team. + pub async fn send_text_markdown_no_fail_quietly( + &self, + room: &Room, + message: String, + response_type: MessageResponseType, + ) -> Option + { + let result = self + .bot + .matrix_link() + .messaging() + .send_text_markdown(room, message, response_type) + .await; + + match result { + Ok(result) => Some(result), + Err(err) => { + tracing::warn!( + room_id = format!("{:?}", room.room_id()), + ?err, + "Failed to send thinking-notice placeholder to room", + ); + None + } + } + } + + /// Edits an existing message's text in place (`m.replace`), best-effort. + /// + /// Built and sent directly via matrix-sdk's `make_edit_event` + `room.send`, + /// deliberately bypassing the mxlink messaging layer: that layer overwrites + /// `relates_to` from a `MessageResponseType`, which would clobber the + /// `m.replace` relation and turn the edit into a brand-new message. The edit + /// stays in the original's thread by inheritance, so no `response_type` is needed. + /// Failures are warned and swallowed so a flickered notice never breaks the real response. + pub async fn edit_text_markdown_no_fail( + &self, + room: &Room, + event_id: &EventId, + markdown: String, + ) -> Option + { + let new_content = RoomMessageEventContentWithoutRelation::text_markdown(markdown); + + let edit_content = match room + .make_edit_event(event_id, EditedContent::RoomMessage(new_content)) + .await + { + Ok(edit_content) => edit_content, + Err(err) => { + tracing::warn!( + room_id = format!("{:?}", room.room_id()), + ?event_id, + ?err, + "Failed to build edit event", + ); + return None; + } + }; + + match room.send(edit_content).await { + Ok(result) => Some(result.response), + Err(err) => { + tracing::warn!( + room_id = format!("{:?}", room.room_id()), + ?event_id, + ?err, + "Failed to send edit to room", + ); + None + } + } + } + pub async fn send_notice_markdown_no_fail( &self, room: &Room, diff --git a/src/controller/cfg/controller_type.rs b/src/controller/cfg/controller_type.rs index f7c108d..fad9ba6 100644 --- a/src/controller/cfg/controller_type.rs +++ b/src/controller/cfg/controller_type.rs @@ -38,6 +38,9 @@ pub enum ConfigTextGenerationSettingRelatedControllerType { GetContextManagementEnabled, SetContextManagementEnabled(Option), + GetThinkingNoticeEnabled, + SetThinkingNoticeEnabled(Option), + GetPrefixRequirementType, SetPrefixRequirementType(Option), diff --git a/src/controller/cfg/determination/text_generation/mod.rs b/src/controller/cfg/determination/text_generation/mod.rs index 15a31ec..bd95a0b 100644 --- a/src/controller/cfg/determination/text_generation/mod.rs +++ b/src/controller/cfg/determination/text_generation/mod.rs @@ -55,6 +55,44 @@ pub(super) fn determine( ); } + if let Some(remaining_text) = text.strip_prefix("thinking-notice-enabled") { + let remaining_text = remaining_text.trim(); + + if !remaining_text.is_empty() { + return Err(ControllerType::Error( + strings::cfg::configuration_getter_used_with_extra_text( + "thinking-notice-enabled", + remaining_text, + ) + .to_owned(), + )); + } + + return Ok(ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled); + } + + if let Some(value_string) = text.strip_prefix("set-thinking-notice-enabled") { + let value_string = value_string.trim().to_owned(); + let value_opt = if value_string.is_empty() { + None + } else { + let value_string_lowercase = value_string.to_lowercase(); + Some(match value_string_lowercase.as_str() { + "true" => true, + "false" => false, + _ => { + return Err(ControllerType::Error( + strings::cfg::configuration_value_unrecognized(&value_string).to_owned(), + )); + } + }) + }; + + return Ok( + ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value_opt), + ); + } + if let Some(remaining_text) = text.strip_prefix("prefix-requirement-type") { let remaining_text = remaining_text.trim(); diff --git a/src/controller/cfg/dispatching/text_generation.rs b/src/controller/cfg/dispatching/text_generation.rs index 505a336..720ac59 100644 --- a/src/controller/cfg/dispatching/text_generation.rs +++ b/src/controller/cfg/dispatching/text_generation.rs @@ -43,6 +43,27 @@ pub(super) async fn dispatch( } } + ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled => { + let value = &room_settings.text_generation.thinking_notice_enabled; + setting_get::(bot, message_context, value).await + } + ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value) => { + let value = value.to_owned(); + + let setter_callback = Box::new(move |room_settings: &mut RoomSettings| { + room_settings.text_generation.thinking_notice_enabled = value; + }); + + match config_type { + SettingsStorageSource::Room => { + room_setting_set::(bot, message_context, &value, setter_callback).await + } + SettingsStorageSource::Global => { + global_setting_set::(bot, message_context, &value, setter_callback).await + } + } + } + ConfigTextGenerationSettingRelatedControllerType::GetPrefixRequirementType => { let value = &room_settings.text_generation.prefix_requirement_type; setting_get::(bot, message_context, value).await diff --git a/src/controller/cfg/help.rs b/src/controller/cfg/help.rs index 4ed5118..4c95e2d 100644 --- a/src/controller/cfg/help.rs +++ b/src/controller/cfg/help.rs @@ -234,6 +234,44 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St )); message.push_str("\n\n"); + // Thinking Notice + + message.push_str(&format!( + "#### {}", + strings::help::cfg::text_generation_thinking_notice_heading() + )); + message.push_str("\n\n"); + message.push_str(&strings::help::cfg::text_generation_thinking_notice_intro()); + message.push('\n'); + message.push_str( + &strings::help::cfg::the_following_configuration_values_are_recognized(vec![true, false]), + ); + message.push_str("\n\n"); + message.push_str(&format!( + "- {}", + &strings::help::cfg::current_setting_show( + command_prefix, + "text-generation thinking-notice-enabled" + ) + )); + message.push('\n'); + message.push_str(&format!( + "- {}", + &strings::help::cfg::current_setting_set( + command_prefix, + "text-generation set-thinking-notice-enabled VALUE" + ) + )); + message.push('\n'); + message.push_str(&format!( + "- {}", + &strings::help::cfg::current_setting_unset( + command_prefix, + "text-generation set-thinking-notice-enabled" + ) + )); + message.push_str("\n\n"); + // Sender Context message.push_str(&format!( diff --git a/src/controller/cfg/status.rs b/src/controller/cfg/status.rs index 669d3e5..e287328 100644 --- a/src/controller/cfg/status.rs +++ b/src/controller/cfg/status.rs @@ -359,6 +359,33 @@ async fn generate_text_generation_section( ), ); + // Thinking Notice + + let effective_thinking_notice = room_config_context.text_generation_thinking_notice_enabled(); + let room_config_thinking_notice = room_config_context + .room_config + .settings + .text_generation + .thinking_notice_enabled; + let global_config_thinking_notice = room_config_context + .global_config + .fallback_room_settings + .text_generation + .thinking_notice_enabled; + + let thinking_notice_set_where = if room_config_thinking_notice.is_some() { + strings::cfg::status_badge_set_in_room_config() + } else if global_config_thinking_notice.is_some() { + strings::cfg::status_badge_set_in_global_config() + } else { + strings::cfg::status_badge_using_hardcoded_default() + }; + + message.push_str(&strings::cfg::status_text_generation_entry_thinking_notice( + effective_thinking_notice, + thinking_notice_set_where, + )); + // Sender Context let effective_sender_context = room_config_context.text_generation_sender_context_mode(); diff --git a/src/controller/chat_completion/mod.rs b/src/controller/chat_completion/mod.rs index 692786e..1d99c39 100644 --- a/src/controller/chat_completion/mod.rs +++ b/src/controller/chat_completion/mod.rs @@ -30,6 +30,13 @@ use crate::{ entity::MessageContext, }; +/// How long text generation must run before the first "thinking…" placeholder appears. +/// Fast responses (under this threshold) never get a placeholder. +const THINKING_NOTICE_FIRST_DELAY: std::time::Duration = std::time::Duration::from_secs(3); + +/// How often the "thinking…" placeholder is refreshed once it has appeared. +const THINKING_NOTICE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(10); + #[derive(Debug, PartialEq)] pub enum ChatCompletionControllerType { // Invoked via a command prefix (e.g. `!bai Hello!`) @@ -520,6 +527,16 @@ async fn handle_stage_text_generation( conversation.start_time(), ); + // Cloned only when the thinking-notice is enabled; the original is moved into `params` below. + let notice_prompt_variables = if message_context + .room_config_context() + .text_generation_thinking_notice_enabled() + { + Some(prompt_variables.clone()) + } else { + None + }; + let params = TextGenerationParams { context_management_enabled: message_context .room_config_context() @@ -536,10 +553,73 @@ async fn handle_stage_text_generation( prompt_variables, }; - let result = controller - .generate_text(conversation, params) - .instrument(span) - .await; + // When the thinking-notice is enabled, race generation against a timer that posts and then + // periodically edits a "thinking…" placeholder. `biased;` makes generation win a tie, and the + // loop exits the instant generation resolves, so there is no detached task and no late edit can + // ever clobber the real answer. `placeholder` is the event we must finalize in every exit path. + let (result, placeholder) = if let Some(notice_prompt_variables) = notice_prompt_variables { + let generation = controller.generate_text(conversation, params).instrument(span); + tokio::pin!(generation); + + let mut placeholder: Option = None; + // Seed the flavor sequence per-generation so different turns don't all open on the same + // line; the monotonic increment then guarantees consecutive notices differ. + let mut notice_sequence: usize = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|since| since.subsec_nanos() as usize) + .unwrap_or(0); + let mut tick = tokio::time::interval_at( + tokio::time::Instant::now() + THINKING_NOTICE_FIRST_DELAY, + THINKING_NOTICE_INTERVAL, + ); + // If an edit runs long, hold ~INTERVAL spacing rather than bursting the missed ticks. + tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + + let result = loop { + tokio::select! { + biased; + generation_result = &mut generation => break generation_result, + _ = tick.tick() => { + let message = notice_prompt_variables.format( + strings::thinking::pick_message(start_time.elapsed(), notice_sequence), + ); + notice_sequence = notice_sequence.wrapping_add(1); + + match &placeholder { + None => { + placeholder = bot + .messaging() + .send_text_markdown_no_fail_quietly( + message_context.room(), + message, + response_type.clone(), + ) + .await + .map(|response| response.event_id); + } + Some(event_id) => { + bot.messaging() + .edit_text_markdown_no_fail( + message_context.room(), + event_id, + message, + ) + .await; + } + } + } + } + }; + + (result, placeholder) + } else { + let result = controller + .generate_text(conversation, params) + .instrument(span) + .await; + + (result, None) + }; let duration = std::time::Instant::now().duration_since(start_time); @@ -560,17 +640,20 @@ async fn handle_stage_text_generation( err, ); - bot.messaging() - .send_error_markdown_no_fail( - message_context.room(), - &strings::agent::error_while_serving_purpose( - agent.identifier(), - &AgentPurpose::TextGeneration, - &err, - ), - response_type, - ) - .await; + let error_text = strings::agent::error_while_serving_purpose( + agent.identifier(), + &AgentPurpose::TextGeneration, + &err, + ); + + finalize_thinking_notice_with_error( + bot, + message_context, + placeholder.as_ref(), + &error_text, + response_type, + ) + .await; return None; } @@ -583,26 +666,70 @@ async fn handle_stage_text_generation( "Agent returned empty text", ); - bot.messaging() - .send_error_markdown_no_fail( - message_context.room(), - &strings::agent::empty_response_returned(agent.identifier()), - response_type, - ) - .await; + let empty_text = strings::agent::empty_response_returned(agent.identifier()); + + finalize_thinking_notice_with_error( + bot, + message_context, + placeholder.as_ref(), + &empty_text, + response_type, + ) + .await; return None; } - let send_message_response = bot - .messaging() - .send_text_markdown_no_fail(message_context.room(), text.clone(), response_type) - .await?; + // Finalize the answer into a single message. With a placeholder, edit it in place (so the + // "thinking…" message becomes the answer); the TTS payload then points at that same event. + // If the edit fails, fall back to a fresh send so the real answer is never lost. + let event_id = match &placeholder { + Some(event_id) + if bot + .messaging() + .edit_text_markdown_no_fail(message_context.room(), event_id, text.clone()) + .await + .is_some() => + { + event_id.clone() + } + _ => { + bot.messaging() + .send_text_markdown_no_fail(message_context.room(), text.clone(), response_type) + .await? + .event_id + } + }; - Some(TextToSpeechEligiblePayload { - text, - event_id: send_message_response.event_id, - }) + Some(TextToSpeechEligiblePayload { text, event_id }) +} + +/// Finalizes a thinking-notice placeholder (if one was posted) with error/notice text, so a +/// failed or empty generation never leaves an orphaned "thinking…" message behind. With no +/// placeholder, this is the original behavior: a fresh error notice. +async fn finalize_thinking_notice_with_error( + bot: &Bot, + message_context: &MessageContext, + placeholder: Option<&OwnedEventId>, + text: &str, + response_type: MessageResponseType, +) { + match placeholder { + Some(event_id) => { + bot.messaging() + .edit_text_markdown_no_fail( + message_context.room(), + event_id, + crate::utils::status::create_error_message_text(text), + ) + .await; + } + None => { + bot.messaging() + .send_error_markdown_no_fail(message_context.room(), text, response_type) + .await; + } + } } async fn handle_stage_speech_to_text_actual_transcribing( diff --git a/src/entity/room_config_context.rs b/src/entity/room_config_context.rs index 112117d..62eda03 100644 --- a/src/entity/room_config_context.rs +++ b/src/entity/room_config_context.rs @@ -136,6 +136,20 @@ impl RoomConfigContext { .unwrap_or(false) } + pub fn text_generation_thinking_notice_enabled(&self) -> bool { + self.room_config + .settings + .text_generation + .thinking_notice_enabled + .or({ + self.global_config + .fallback_room_settings + .text_generation + .thinking_notice_enabled + }) + .unwrap_or(false) + } + pub fn text_generation_sender_context_mode(&self) -> TextGenerationSenderContextMode { self.room_config .settings diff --git a/src/entity/roomconfig/entity/text_generation.rs b/src/entity/roomconfig/entity/text_generation.rs index cc93c01..7558780 100644 --- a/src/entity/roomconfig/entity/text_generation.rs +++ b/src/entity/roomconfig/entity/text_generation.rs @@ -14,6 +14,10 @@ pub struct RoomSettingsTextGeneration { /// When enabled, the bot will automatically tokenize messages and try to shorten the message context intelligently. pub context_management_enabled: Option, + /// Controls whether a "thinking…" notice is posted while text generation runs longer than a threshold. + /// When enabled, a placeholder message appears for slow responses and is edited in place (with elapsed-tiered flavor text) until it becomes the final answer. + pub thinking_notice_enabled: Option, + /// Controls how each message in the conversation context is annotated with sender metadata. pub sender_context_mode: Option, diff --git a/src/strings/cfg.rs b/src/strings/cfg.rs index bcefd88..1499375 100644 --- a/src/strings/cfg.rs +++ b/src/strings/cfg.rs @@ -249,6 +249,10 @@ pub fn status_text_generation_entry_context_management(value: bool, set_where: & format!("- ♻️ Context management: `{}` ({})\n", value, set_where) } +pub fn status_text_generation_entry_thinking_notice(value: bool, set_where: &str) -> String { + format!("- 💭 Thinking notice: `{}` ({})\n", value, set_where) +} + pub fn status_text_generation_entry_sender_context( value: impl std::fmt::Display, set_where: &str, diff --git a/src/strings/help/cfg.rs b/src/strings/help/cfg.rs index db5a432..4d1cd0a 100644 --- a/src/strings/help/cfg.rs +++ b/src/strings/help/cfg.rs @@ -132,6 +132,18 @@ pub fn text_generation_context_management_intro() -> String { ) } +pub fn text_generation_thinking_notice_heading() -> &'static str { + "💭 Thinking Notice" +} + +pub fn text_generation_thinking_notice_intro() -> String { + format!( + "{}\n{}", + "Controls whether the bot posts a **\"thinking…\" notice** while text generation takes a while, so slow responses (for example, from reasoning models that run for minutes) don't look stuck.", + "When enabled, a placeholder message appears only after a short delay, updates periodically with varying status text, and is then edited in place to become the final answer. Disabled by default.", + ) +} + pub fn text_generation_sender_context_heading() -> &'static str { "👤 Sender Context Mode" } diff --git a/src/strings/mod.rs b/src/strings/mod.rs index 5717255..abaa341 100644 --- a/src/strings/mod.rs +++ b/src/strings/mod.rs @@ -11,6 +11,7 @@ pub mod provider; pub mod room_config; pub mod speech_to_text; pub mod text_to_speech; +pub mod thinking; pub mod usage; pub const PROGRESS_INDICATOR_EMOJI: &str = "⏳"; diff --git a/src/strings/thinking.rs b/src/strings/thinking.rs new file mode 100644 index 0000000..9be6649 --- /dev/null +++ b/src/strings/thinking.rs @@ -0,0 +1,134 @@ +use std::time::Duration; + +/// After this much elapsed generation time, the notice escalates to the "medium" pool. +const TIER_MEDIUM_AFTER: Duration = Duration::from_secs(30); + +/// After this much elapsed generation time, the notice escalates to the "deep" pool. +const TIER_DEEP_AFTER: Duration = Duration::from_secs(90); + +/// Early flavor: the response is just taking a moment longer than instant. +const MESSAGES_LIGHT: &[&str] = &[ + "*{{ baibot_name }} is thinking…*", + "*{{ baibot_name }} is mulling that over…*", + "*Hmm, let me think about that…*", + "*{{ baibot_name }} is gathering some thoughts…*", + "*One moment, {{ baibot_name }} is working on it…*", + "*{{ baibot_name }} is putting the pieces together…*", + "*Give {{ baibot_name }} a second here…*", + "*Let me think this one through…*", + "*{{ baibot_name }} is warming up the gears…*", + "*{{ baibot_name }} is on it…*", +]; + +/// Mid flavor: this is a real question and the model is genuinely working. +const MESSAGES_MEDIUM: &[&str] = &[ + "*Huh, good one. {{ baibot_name }} is really thinking now…*", + "*Still working on it, {{ baibot_name }} wants to get this right…*", + "*This one needs a bit more thought…*", + "*{{ baibot_name }} is digging into this…*", + "*Hang tight, {{ baibot_name }} is turning it over…*", + "*Not a quick one, this. {{ baibot_name }} is still at it…*", + "*{{ baibot_name }} is chewing on this properly now…*", + "*This deserves some real thought, bear with {{ baibot_name }}…*", + "*{{ baibot_name }} is working through the details…*", + "*Won't be long, {{ baibot_name }} is closing in on it…*", +]; + +/// Deep flavor: a long-running generation (e.g. a reasoning model going for minutes). +const MESSAGES_DEEP: &[&str] = &[ + "*Okay, this is a hard one. {{ baibot_name }} is really deep in thought…*", + "*{{ baibot_name }} is in the weeds on this one, thanks for your patience…*", + "*Still here, still thinking. {{ baibot_name }} doesn't want to rush it…*", + "*A proper puzzle, this. {{ baibot_name }} is taking the time to do it justice…*", + "*{{ baibot_name }} is really wrestling with this one…*", + "*A meaty question. {{ baibot_name }} is still turning it over…*", + "*{{ baibot_name }} hasn't forgotten you, just thinking hard…*", + "*Almost there, {{ baibot_name }} is pulling it all together…*", + "*{{ baibot_name }} is going the extra mile on this one…*", + "*Deep thoughts in progress. {{ baibot_name }} appreciates your patience…*", +]; + +/// Returns the flavor pool matching how long generation has been running. +pub fn messages_for_elapsed(elapsed: Duration) -> &'static [&'static str] { + if elapsed >= TIER_DEEP_AFTER { + MESSAGES_DEEP + } else if elapsed >= TIER_MEDIUM_AFTER { + MESSAGES_MEDIUM + } else { + MESSAGES_LIGHT + } +} + +/// Picks one (still-untemplated) flavor message from the tier active at `elapsed`, +/// indexed by a monotonic `sequence` the caller increments once per notice. +/// +/// Using a monotonic counter (rather than a clock-derived value) guarantees two +/// things the gesture-novelty goal needs: consecutive notices never repeat a line +/// (the index advances by one each tick, so it differs whenever the tier has more +/// than one message), and the pool is fully walked before any line recurs. The +/// caller seeds `sequence` with a per-generation value so different turns don't all +/// open on the same line. A clock-derived index can't promise this: `interval_at` +/// ticks on a near-fixed schedule, so its sub-second component clusters and would +/// re-pick the same line. +pub fn pick_message(elapsed: Duration, sequence: usize) -> &'static str { + let pool = messages_for_elapsed(elapsed); + pool[sequence % pool.len()] +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn messages_for_elapsed_returns_the_right_tier() { + // Boundaries: light below 30s, medium [30s, 90s), deep at/after 90s. + // The pools have distinct content, so value comparison identifies the tier. + assert_eq!(messages_for_elapsed(Duration::from_secs(0)), MESSAGES_LIGHT); + assert_eq!(messages_for_elapsed(Duration::from_secs(29)), MESSAGES_LIGHT); + assert_eq!(messages_for_elapsed(Duration::from_secs(30)), MESSAGES_MEDIUM); + assert_eq!(messages_for_elapsed(Duration::from_secs(89)), MESSAGES_MEDIUM); + assert_eq!(messages_for_elapsed(Duration::from_secs(90)), MESSAGES_DEEP); + assert_eq!(messages_for_elapsed(Duration::from_secs(600)), MESSAGES_DEEP); + } + + #[test] + fn every_tier_is_non_empty() { + for pool in [MESSAGES_LIGHT, MESSAGES_MEDIUM, MESSAGES_DEEP] { + assert!(!pool.is_empty()); + for message in pool { + assert!(!message.trim().is_empty()); + } + } + } + + #[test] + fn every_tier_uses_the_bot_name_template() { + // Not every line names the bot (some are first-person flavor), but each + // tier exercises the template var so substitution is wired through. + for pool in [MESSAGES_LIGHT, MESSAGES_MEDIUM, MESSAGES_DEEP] { + assert!(pool.iter().any(|m| m.contains("{{ baibot_name }}"))); + } + } + + #[test] + fn pick_message_stays_within_the_active_tier() { + let deep = messages_for_elapsed(Duration::from_secs(120)); + for sequence in 0..50 { + assert!(deep.contains(&pick_message(Duration::from_secs(120), sequence))); + } + } + + #[test] + fn pick_message_never_repeats_on_consecutive_sequences() { + // The gesture-novelty guarantee: a monotonic sequence must not pick the same line twice + // in a row (and walks the whole tier before any line recurs). + let elapsed = Duration::from_secs(0); + let pool_len = messages_for_elapsed(elapsed).len(); + for sequence in 0..(pool_len * 3) { + assert_ne!( + pick_message(elapsed, sequence), + pick_message(elapsed, sequence + 1), + ); + } + } +}