Add thinking notice + fix Venice auto-recover caching (#193)

Opt-in 💭 thinking-notice for slow text generation, plus a fix making Venice unsupported-field auto-recovery survive the per-message controller rebuild. Prepares 1.24.0.

Co-authored-by: Aine <aine@etke.cc>
This commit is contained in:
Aine
2026-06-26 05:07:50 +01:00
committed by GitHub
parent 025accdeb0
commit 318a8fd91d
20 changed files with 618 additions and 35 deletions

View File

@@ -38,6 +38,9 @@ pub enum ConfigTextGenerationSettingRelatedControllerType {
GetContextManagementEnabled,
SetContextManagementEnabled(Option<bool>),
GetThinkingNoticeEnabled,
SetThinkingNoticeEnabled(Option<bool>),
GetPrefixRequirementType,
SetPrefixRequirementType(Option<TextGenerationPrefixRequirementType>),

View File

@@ -55,6 +55,44 @@ pub(super) fn determine(
);
}
if let Some(remaining_text) = text.strip_prefix("thinking-notice-enabled") {
let remaining_text = remaining_text.trim();
if !remaining_text.is_empty() {
return Err(ControllerType::Error(
strings::cfg::configuration_getter_used_with_extra_text(
"thinking-notice-enabled",
remaining_text,
)
.to_owned(),
));
}
return Ok(ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled);
}
if let Some(value_string) = text.strip_prefix("set-thinking-notice-enabled") {
let value_string = value_string.trim().to_owned();
let value_opt = if value_string.is_empty() {
None
} else {
let value_string_lowercase = value_string.to_lowercase();
Some(match value_string_lowercase.as_str() {
"true" => true,
"false" => false,
_ => {
return Err(ControllerType::Error(
strings::cfg::configuration_value_unrecognized(&value_string).to_owned(),
));
}
})
};
return Ok(
ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value_opt),
);
}
if let Some(remaining_text) = text.strip_prefix("prefix-requirement-type") {
let remaining_text = remaining_text.trim();

View File

@@ -43,6 +43,27 @@ pub(super) async fn dispatch(
}
}
ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled => {
let value = &room_settings.text_generation.thinking_notice_enabled;
setting_get::<bool>(bot, message_context, value).await
}
ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value) => {
let value = value.to_owned();
let setter_callback = Box::new(move |room_settings: &mut RoomSettings| {
room_settings.text_generation.thinking_notice_enabled = value;
});
match config_type {
SettingsStorageSource::Room => {
room_setting_set::<bool>(bot, message_context, &value, setter_callback).await
}
SettingsStorageSource::Global => {
global_setting_set::<bool>(bot, message_context, &value, setter_callback).await
}
}
}
ConfigTextGenerationSettingRelatedControllerType::GetPrefixRequirementType => {
let value = &room_settings.text_generation.prefix_requirement_type;
setting_get::<TextGenerationPrefixRequirementType>(bot, message_context, value).await

View File

@@ -234,6 +234,44 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
));
message.push_str("\n\n");
// Thinking Notice
message.push_str(&format!(
"#### {}",
strings::help::cfg::text_generation_thinking_notice_heading()
));
message.push_str("\n\n");
message.push_str(&strings::help::cfg::text_generation_thinking_notice_intro());
message.push('\n');
message.push_str(
&strings::help::cfg::the_following_configuration_values_are_recognized(vec![true, false]),
);
message.push_str("\n\n");
message.push_str(&format!(
"- {}",
&strings::help::cfg::current_setting_show(
command_prefix,
"text-generation thinking-notice-enabled"
)
));
message.push('\n');
message.push_str(&format!(
"- {}",
&strings::help::cfg::current_setting_set(
command_prefix,
"text-generation set-thinking-notice-enabled VALUE"
)
));
message.push('\n');
message.push_str(&format!(
"- {}",
&strings::help::cfg::current_setting_unset(
command_prefix,
"text-generation set-thinking-notice-enabled"
)
));
message.push_str("\n\n");
// Sender Context
message.push_str(&format!(

View File

@@ -359,6 +359,33 @@ async fn generate_text_generation_section(
),
);
// Thinking Notice
let effective_thinking_notice = room_config_context.text_generation_thinking_notice_enabled();
let room_config_thinking_notice = room_config_context
.room_config
.settings
.text_generation
.thinking_notice_enabled;
let global_config_thinking_notice = room_config_context
.global_config
.fallback_room_settings
.text_generation
.thinking_notice_enabled;
let thinking_notice_set_where = if room_config_thinking_notice.is_some() {
strings::cfg::status_badge_set_in_room_config()
} else if global_config_thinking_notice.is_some() {
strings::cfg::status_badge_set_in_global_config()
} else {
strings::cfg::status_badge_using_hardcoded_default()
};
message.push_str(&strings::cfg::status_text_generation_entry_thinking_notice(
effective_thinking_notice,
thinking_notice_set_where,
));
// Sender Context
let effective_sender_context = room_config_context.text_generation_sender_context_mode();

View File

@@ -30,6 +30,13 @@ use crate::{
entity::MessageContext,
};
/// How long text generation must run before the first "thinking…" placeholder appears.
/// Fast responses (under this threshold) never get a placeholder.
const THINKING_NOTICE_FIRST_DELAY: std::time::Duration = std::time::Duration::from_secs(3);
/// How often the "thinking…" placeholder is refreshed once it has appeared.
const THINKING_NOTICE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(10);
#[derive(Debug, PartialEq)]
pub enum ChatCompletionControllerType {
// Invoked via a command prefix (e.g. `!bai Hello!`)
@@ -520,6 +527,16 @@ async fn handle_stage_text_generation(
conversation.start_time(),
);
// Cloned only when the thinking-notice is enabled; the original is moved into `params` below.
let notice_prompt_variables = if message_context
.room_config_context()
.text_generation_thinking_notice_enabled()
{
Some(prompt_variables.clone())
} else {
None
};
let params = TextGenerationParams {
context_management_enabled: message_context
.room_config_context()
@@ -536,10 +553,73 @@ async fn handle_stage_text_generation(
prompt_variables,
};
let result = controller
.generate_text(conversation, params)
.instrument(span)
.await;
// When the thinking-notice is enabled, race generation against a timer that posts and then
// periodically edits a "thinking…" placeholder. `biased;` makes generation win a tie, and the
// loop exits the instant generation resolves, so there is no detached task and no late edit can
// ever clobber the real answer. `placeholder` is the event we must finalize in every exit path.
let (result, placeholder) = if let Some(notice_prompt_variables) = notice_prompt_variables {
let generation = controller.generate_text(conversation, params).instrument(span);
tokio::pin!(generation);
let mut placeholder: Option<OwnedEventId> = None;
// Seed the flavor sequence per-generation so different turns don't all open on the same
// line; the monotonic increment then guarantees consecutive notices differ.
let mut notice_sequence: usize = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|since| since.subsec_nanos() as usize)
.unwrap_or(0);
let mut tick = tokio::time::interval_at(
tokio::time::Instant::now() + THINKING_NOTICE_FIRST_DELAY,
THINKING_NOTICE_INTERVAL,
);
// If an edit runs long, hold ~INTERVAL spacing rather than bursting the missed ticks.
tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
let result = loop {
tokio::select! {
biased;
generation_result = &mut generation => break generation_result,
_ = tick.tick() => {
let message = notice_prompt_variables.format(
strings::thinking::pick_message(start_time.elapsed(), notice_sequence),
);
notice_sequence = notice_sequence.wrapping_add(1);
match &placeholder {
None => {
placeholder = bot
.messaging()
.send_text_markdown_no_fail_quietly(
message_context.room(),
message,
response_type.clone(),
)
.await
.map(|response| response.event_id);
}
Some(event_id) => {
bot.messaging()
.edit_text_markdown_no_fail(
message_context.room(),
event_id,
message,
)
.await;
}
}
}
}
};
(result, placeholder)
} else {
let result = controller
.generate_text(conversation, params)
.instrument(span)
.await;
(result, None)
};
let duration = std::time::Instant::now().duration_since(start_time);
@@ -560,17 +640,20 @@ async fn handle_stage_text_generation(
err,
);
bot.messaging()
.send_error_markdown_no_fail(
message_context.room(),
&strings::agent::error_while_serving_purpose(
agent.identifier(),
&AgentPurpose::TextGeneration,
&err,
),
response_type,
)
.await;
let error_text = strings::agent::error_while_serving_purpose(
agent.identifier(),
&AgentPurpose::TextGeneration,
&err,
);
finalize_thinking_notice_with_error(
bot,
message_context,
placeholder.as_ref(),
&error_text,
response_type,
)
.await;
return None;
}
@@ -583,26 +666,70 @@ async fn handle_stage_text_generation(
"Agent returned empty text",
);
bot.messaging()
.send_error_markdown_no_fail(
message_context.room(),
&strings::agent::empty_response_returned(agent.identifier()),
response_type,
)
.await;
let empty_text = strings::agent::empty_response_returned(agent.identifier());
finalize_thinking_notice_with_error(
bot,
message_context,
placeholder.as_ref(),
&empty_text,
response_type,
)
.await;
return None;
}
let send_message_response = bot
.messaging()
.send_text_markdown_no_fail(message_context.room(), text.clone(), response_type)
.await?;
// Finalize the answer into a single message. With a placeholder, edit it in place (so the
// "thinking…" message becomes the answer); the TTS payload then points at that same event.
// If the edit fails, fall back to a fresh send so the real answer is never lost.
let event_id = match &placeholder {
Some(event_id)
if bot
.messaging()
.edit_text_markdown_no_fail(message_context.room(), event_id, text.clone())
.await
.is_some() =>
{
event_id.clone()
}
_ => {
bot.messaging()
.send_text_markdown_no_fail(message_context.room(), text.clone(), response_type)
.await?
.event_id
}
};
Some(TextToSpeechEligiblePayload {
text,
event_id: send_message_response.event_id,
})
Some(TextToSpeechEligiblePayload { text, event_id })
}
/// Finalizes a thinking-notice placeholder (if one was posted) with error/notice text, so a
/// failed or empty generation never leaves an orphaned "thinking…" message behind. With no
/// placeholder, this is the original behavior: a fresh error notice.
async fn finalize_thinking_notice_with_error(
bot: &Bot,
message_context: &MessageContext,
placeholder: Option<&OwnedEventId>,
text: &str,
response_type: MessageResponseType,
) {
match placeholder {
Some(event_id) => {
bot.messaging()
.edit_text_markdown_no_fail(
message_context.room(),
event_id,
crate::utils::status::create_error_message_text(text),
)
.await;
}
None => {
bot.messaging()
.send_error_markdown_no_fail(message_context.room(), text, response_type)
.await;
}
}
}
async fn handle_stage_speech_to_text_actual_transcribing(