Compare commits
14 Commits
add-venice
...
provider-n
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9fc3998f8e | ||
|
|
0be08c0292 | ||
|
|
8904edb2a0 | ||
|
|
e18c083c4f | ||
|
|
1948f58473 | ||
|
|
c8a90049ad | ||
|
|
4f7778e62f | ||
|
|
355b1b299c | ||
|
|
318a8fd91d | ||
|
|
025accdeb0 | ||
|
|
edf5bc9fdb | ||
|
|
982ebc6657 | ||
|
|
4a75be742f | ||
|
|
aae3c4c08f |
28
.github/workflows/ci.yml
vendored
28
.github/workflows/ci.yml
vendored
@@ -14,13 +14,31 @@ concurrency:
|
||||
group: ci-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
jobs:
|
||||
test-and-clippy:
|
||||
name: Unit testing and linting
|
||||
prek:
|
||||
name: Lint, format & test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: dtolnay/rust-toolchain@1.93.0
|
||||
|
||||
# Toolchain version + components come from rust-toolchain.toml. rustflags is
|
||||
# cleared so plain builds don't fail on warnings; the clippy hook still does.
|
||||
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||
with:
|
||||
rustflags: ''
|
||||
|
||||
- name: Install SQLite3
|
||||
run: sudo apt-get update && sudo apt-get install -y libsqlite3-dev
|
||||
- run: cargo test --all-features
|
||||
- run: cargo clippy
|
||||
|
||||
# just drives the prek recipes; mise provides the pinned prek (mise.toml).
|
||||
- uses: taiki-e/install-action@v2
|
||||
with:
|
||||
tool: just
|
||||
- uses: jdx/mise-action@v4
|
||||
|
||||
# Run the same prek hooks devs run locally; .pre-commit-config.yaml is the
|
||||
# source of truth. Tests are a separate step for visible timing.
|
||||
- name: Lint & format (prek hooks, excluding tests)
|
||||
run: just prek-run-on-all --skip test-unit
|
||||
|
||||
- name: Unit tests
|
||||
run: just test
|
||||
|
||||
16
CHANGELOG.md
16
CHANGELOG.md
@@ -1,4 +1,18 @@
|
||||
# (2026-06-23) Version 1.23.1
|
||||
# (2026-06-28) Version 1.25.0
|
||||
|
||||
- (**Feature**) [♻️ Context management](./docs/configuration/text-generation.md#️-context-management) now works with every provider, not only [OpenAI](./docs/providers.md#openai). Token counting previously went through [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs), which is accurate only for OpenAI models and silently mis-counted everything else (worst of all for non-English text). OpenAI agents keep using tiktoken-rs; every other provider, including the recommended [Venice](./docs/providers.md#venice), now uses a provider-neutral approximation that needs no per-model tokenizer (ASCII counted at about four characters per token, other scripts such as Cyrillic and CJK at about two), landing within roughly 10-20% of the real count. See the [context management docs](./docs/configuration/text-generation.md#️-context-management).
|
||||
|
||||
- (**Improvement**) Context management now trims a conversation on whole-turn boundaries for every provider, so an assistant reply is never kept without the user message it answered. This also adjusts how the OpenAI provider trims: a dangling assistant reply at the oldest edge of the kept history is now dropped along with its missing prompt, rather than left in place.
|
||||
|
||||
|
||||
# (2026-06-26) Version 1.24.0
|
||||
|
||||
- (**Feature**) Add an opt-in 💭 **thinking notice** for text generation. When enabled, a slow response (for example, from a reasoning model that runs for minutes) posts a "thinking…" placeholder after a short delay, refreshes it periodically with varying flavor text, and then edits that same message into the final answer, so a long wait no longer looks like a stuck bot. The notice is **disabled by default** and configurable per-room or globally via `text-generation set-thinking-notice-enabled true`. Fast responses (under the delay threshold) never show a placeholder. See the [text-generation configuration docs](./docs/configuration/text-generation.md#-thinking-notice).
|
||||
|
||||
- (**Bugfix**) The [Venice](https://venice.ai) unsupported-field auto-recovery (added in 1.23.1) now actually remembers rejections across messages. The cache lived on the provider's controller, which is rebuilt on every message for room-local and global agents, so each turn started with an empty cache, re-sent the unsupported field, and logged the same `400 Bad Request` warning again. The cache is now process-global (keyed per Venice deployment), so a field a model rejects is dropped proactively on every later request instead of being re-discovered each turn.
|
||||
|
||||
|
||||
# (2026-06-24) Version 1.23.1
|
||||
|
||||
- (**Bugfix**) The [Venice](https://venice.ai) provider now auto-recovers when a model rejects an optional knob it does not support. Venice's request body is strict (`additionalProperties: false`), so a model that lacks `prompt_cache_retention`, `reasoning_effort`, or `prompt_cache_key` rejected the whole request with a `400 Bad Request` — breaking agent creation and every reply. baibot now drops the unsupported field and retries, remembering the rejection per model so later requests skip it without a wasted round-trip. Only these meaning-preserving fields are dropped; sampling knobs that change the output (`temperature`, `top_p`, the penalties) are never silently removed and still surface as an error.
|
||||
|
||||
|
||||
34
Cargo.lock
generated
34
Cargo.lock
generated
@@ -95,9 +95,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.102"
|
||||
version = "1.0.103"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c"
|
||||
checksum = "2a4385e2e34eb35d6b3efe798b9eb88096925d87726c0798709bf56d9ed84af3"
|
||||
|
||||
[[package]]
|
||||
name = "anymap2"
|
||||
@@ -315,7 +315,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "baibot"
|
||||
version = "1.23.1"
|
||||
version = "1.25.0"
|
||||
dependencies = [
|
||||
"anthropic",
|
||||
"anyhow",
|
||||
@@ -327,7 +327,7 @@ dependencies = [
|
||||
"mime_guess",
|
||||
"mxidwc",
|
||||
"mxlink",
|
||||
"quick_cache",
|
||||
"quick_cache 0.7.0",
|
||||
"regex",
|
||||
"reqwest 0.13.4",
|
||||
"serde",
|
||||
@@ -1233,6 +1233,12 @@ version = "0.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2"
|
||||
|
||||
[[package]]
|
||||
name = "foldhash"
|
||||
version = "0.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "77ce24cb58228fbb8aa041425bb1050850ac19177686ea6e0f41a70416f56fdb"
|
||||
|
||||
[[package]]
|
||||
name = "form_urlencoded"
|
||||
version = "1.2.2"
|
||||
@@ -1457,7 +1463,7 @@ version = "0.15.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1"
|
||||
dependencies = [
|
||||
"foldhash",
|
||||
"foldhash 0.1.5",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2492,7 +2498,7 @@ dependencies = [
|
||||
"hex",
|
||||
"matrix-sdk",
|
||||
"mime",
|
||||
"quick_cache",
|
||||
"quick_cache 0.6.24",
|
||||
"rand 0.10.1",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -2848,9 +2854,9 @@ checksum = "007d8adb5ddab6f8e3f491ac63566a7d5002cc7ed73901f72057943fa71ae1ae"
|
||||
|
||||
[[package]]
|
||||
name = "quick_cache"
|
||||
version = "0.6.23"
|
||||
version = "0.6.24"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3a3db184a8b66cfe87f0263a1de147a6b554c864d1767c6f7fa4eb0e5497b565"
|
||||
checksum = "b9c6658afe513a3b484e3abfdaa0d03ef3c0bbf017542c178dd55f94eb3051f9"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"equivalent",
|
||||
@@ -2858,6 +2864,18 @@ dependencies = [
|
||||
"parking_lot",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quick_cache"
|
||||
version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "403c1a912fec895cafb223201e368234842acb9220aaf08ab042ae89ba5f135c"
|
||||
dependencies = [
|
||||
"equivalent",
|
||||
"foldhash 0.2.0",
|
||||
"hashbrown 0.17.1",
|
||||
"parking_lot",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quinn"
|
||||
version = "0.11.9"
|
||||
|
||||
@@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later"
|
||||
readme = "README.md"
|
||||
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
||||
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
||||
version = "1.23.1"
|
||||
version = "1.25.0"
|
||||
edition = "2024"
|
||||
|
||||
[lib]
|
||||
@@ -26,7 +26,7 @@ mime_guess = "2.0.*"
|
||||
mxidwc = "1.0.*"
|
||||
mxlink = ">=1.15.0"
|
||||
etke_openai_api_rust = "0.1.*"
|
||||
quick_cache = "0.6.*"
|
||||
quick_cache = "0.7.*"
|
||||
regex = "1.12.*"
|
||||
# HTTP client for the native `venice` provider. rustls only (no extra TLS stack), matching the
|
||||
# reqwest copy async-openai/matrix-sdk/mxlink already use.
|
||||
|
||||
@@ -30,7 +30,7 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
|
||||
|
||||
- 🔒 Supports [encryption](./docs/features.md#-encryption) for Matrix communication and Account-Data-stored configuration
|
||||
|
||||
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) handling on some models (automatically adjusting the message history length, etc.)
|
||||
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) for every [provider](./docs/providers.md) (automatically trimming older messages on whole-turn boundaries once a conversation outgrows the context window)
|
||||
|
||||
- 🛠️ Allows **customizing much of the bot's [configuration](./docs/configuration/README.md)** at runtime (using commands sent via chat)
|
||||
|
||||
|
||||
@@ -50,13 +50,22 @@ Example: `!bai config room text-generation set-auto-usage only_for_voice` (this
|
||||
|
||||
### ♻️ Context Management
|
||||
|
||||
The bot also supports ♻️ **context management**, which automatically adjusts the message history length, etc.
|
||||
The bot also supports ♻️ **context management**, which automatically trims the oldest messages once a conversation grows past the context window. It drops whole turns at a time, so a reply is never separated from the message it answered.
|
||||
|
||||
This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](../providers.md#openai) models.
|
||||
Counting tokens precisely needs the model's own tokenizer. For [OpenAI](../providers.md#openai) models, the bot counts them with the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library. For every other provider, including the recommended [Venice](../providers.md#venice), the bot falls back to a provider-neutral **approximation** that needs no per-model tokenizer: it counts ASCII text at about four characters per token and other scripts (Cyrillic, CJK, and so on) at about two. Treat it as rough, within roughly 10-20% of the real count for typical text, which is plenty for keeping a long conversation inside the context window.
|
||||
|
||||
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
||||
|
||||
|
||||
### 💭 Thinking Notice
|
||||
|
||||
The bot can post a 💭 **"thinking…" notice** while text generation is running, useful for slow models (for example, reasoning models that may run for minutes) where the response would otherwise look stuck.
|
||||
|
||||
When enabled, a placeholder message appears only after a short delay (so fast responses get no notice), updates periodically with varying status text, and is then edited in place to become the final answer.
|
||||
|
||||
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-thinking-notice-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
||||
|
||||
|
||||
### 👤 Sender Context Mode
|
||||
|
||||
In multi-user rooms, it may be useful for the model to know which participant sent each message in the conversation context.
|
||||
|
||||
@@ -24,7 +24,7 @@ The list of supported providers is below.
|
||||
|
||||
### How to choose a provider
|
||||
|
||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation) (incl. vision, incl. [🛠️ tools](./features.md#️-built-in-tools-openai-only)), [🖌️ image-generation](./features.md#️image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
||||
If you're not sure which provider to start with, **we recommend [Venice](#venice)**: it's the most capable provider baibot supports (covering [💬 text-generation](./features.md#-text-generation) with vision, file inputs, prompt caching, and native web search, plus [🖌️ image-generation](./features.md#️-image-creation) incl. editing, [🦻 speech-to-text](./features.md#-speech-to-text), and [🗣️ text-to-speech](./features.md#️-text-to-speech)) and the only one that runs inference with no logging and no training on your data. If you'd rather start with the most widely-used option, [OpenAI](#openai) is a solid, well-supported choice too.
|
||||
|
||||
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
|
||||
|
||||
@@ -176,10 +176,10 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
||||
|
||||
### Venice
|
||||
|
||||
[Venice AI](https://venice.ai) runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
||||
[Venice AI](https://venice.ai/chat?ref=kpXDe6) _(ref link with a $10 bonus for you)_ runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
||||
|
||||
- 🆔 Identifier: `venice`
|
||||
- 🔗 Links: [🏠 Home page](https://venice.ai), [👤 Sign up](https://venice.ai), [📋 Models list](https://api.venice.ai/api/v1/models)
|
||||
- 🔗 Links: [🏠 Home page](https://venice.ai/chat?ref=kpXDe6), [👤 Sign up](https://venice.ai/chat?ref=kpXDe6), [📋 Models list](https://docs.venice.ai/models/overview)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation) (incl. editing, via the native knob-rich `/image/generate` and `/image/edit` endpoints), [💬 text-generation](./features.md#-text-generation) (incl. vision, file inputs like PDF and DOCX, and prompt caching; native web search via the `venice_parameters` config), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local venice my-venice-agent`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
element-web:
|
||||
image: ghcr.io/element-hq/element-web:v1.12.21
|
||||
image: ghcr.io/element-hq/element-web:v1.12.22
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
ollama:
|
||||
image: docker.io/ollama/ollama:0.30.10
|
||||
image: docker.io/ollama/ollama:0.30.11
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
|
||||
|
||||
@@ -15,7 +15,7 @@ use crate::agent::provider::{
|
||||
};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
@@ -130,7 +130,7 @@ impl ControllerTrait for Controller {
|
||||
tracing::trace!("Shortening messages list to context size");
|
||||
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Approximate,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
Some(text_generation_config.max_response_tokens),
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
use chrono::{DateTime, Utc};
|
||||
use std::collections::HashMap;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct TextGenerationPromptVariables {
|
||||
map: HashMap<String, String>,
|
||||
}
|
||||
|
||||
@@ -24,7 +24,7 @@ use crate::{
|
||||
},
|
||||
conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
},
|
||||
utils::base64::base64_decode,
|
||||
};
|
||||
@@ -117,7 +117,7 @@ impl ControllerTrait for Controller {
|
||||
tracing::trace!("Shortening messages list to context size");
|
||||
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Tiktoken(&text_generation_config.model_id),
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
|
||||
@@ -15,7 +15,7 @@ use crate::{
|
||||
},
|
||||
conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
},
|
||||
};
|
||||
use crate::{
|
||||
@@ -114,7 +114,7 @@ impl ControllerTrait for Controller {
|
||||
tracing::trace!("Shortening messages list to context size");
|
||||
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Approximate,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
|
||||
@@ -8,7 +8,7 @@ use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
@@ -64,7 +64,7 @@ pub async fn generate_text(
|
||||
|
||||
if params.context_management_enabled {
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Approximate,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
|
||||
@@ -35,10 +35,15 @@ impl Controller {
|
||||
.build()
|
||||
.unwrap_or_else(|_| reqwest::Client::new());
|
||||
|
||||
// Process-global per-deployment cache rather than a fresh one: dynamic agents rebuild their
|
||||
// controller on every message, so an instance-owned cache would never retain a learned
|
||||
// rejection and the same unsupported field would 400 (and warn) on every turn.
|
||||
let unsupported_fields = UnsupportedFieldsCache::shared_for(&config.base_url);
|
||||
|
||||
Self {
|
||||
config,
|
||||
http,
|
||||
unsupported_fields: UnsupportedFieldsCache::default(),
|
||||
unsupported_fields,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +45,38 @@ pub(super) struct UnsupportedFieldsCache {
|
||||
inner: Arc<RwLock<HashMap<String, HashSet<String>>>>,
|
||||
}
|
||||
|
||||
/// Process-global registry of per-deployment caches, keyed by Venice base URL. A `Controller` is
|
||||
/// rebuilt from its config on every message for dynamic (global/room-local) agents (see
|
||||
/// `agent::manager::available_room_agents_by_room_config_context`), so a cache stored on the
|
||||
/// controller would be reset each message and the "process-lived" intent above would never hold.
|
||||
/// The registry lets a rebuilt controller re-attach to the same cache. Keyed by base URL because
|
||||
/// "which fields a model rejects" is a property of the deployment+model, not of the agent instance:
|
||||
/// agents on the same Venice deployment share learnings, while a distinct deployment keeps its own,
|
||||
/// so a model name that supports a field on one deployment is never proactively stripped against
|
||||
/// another.
|
||||
fn registry() -> &'static RwLock<HashMap<String, UnsupportedFieldsCache>> {
|
||||
static REGISTRY: OnceLock<RwLock<HashMap<String, UnsupportedFieldsCache>>> = OnceLock::new();
|
||||
REGISTRY.get_or_init(|| RwLock::new(HashMap::new()))
|
||||
}
|
||||
|
||||
impl UnsupportedFieldsCache {
|
||||
/// The process-global cache for `base_url`, created on first use and shared by every controller
|
||||
/// for that deployment thereafter. This is what makes a learned rejection survive the
|
||||
/// per-message controller rebuild. A poisoned registry lock degrades to an isolated cache:
|
||||
/// correctness holds, only cross-controller sharing is lost for that call.
|
||||
pub(super) fn shared_for(base_url: &str) -> Self {
|
||||
if let Ok(map) = registry().read()
|
||||
&& let Some(cache) = map.get(base_url)
|
||||
{
|
||||
return cache.clone();
|
||||
}
|
||||
|
||||
match registry().write() {
|
||||
Ok(mut map) => map.entry(base_url.to_owned()).or_default().clone(),
|
||||
Err(_) => Self::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Fields already known unsupported for `model_id`. Returns an empty set on an unknown model or
|
||||
/// a poisoned lock, so a cache failure degrades to "strip nothing proactively" rather than
|
||||
/// breaking the request path.
|
||||
@@ -231,6 +262,33 @@ mod tests {
|
||||
assert!(cache.known_for("kimi-k2-5").is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shared_for_survives_controller_rebuild_and_isolates_deployments() {
|
||||
// Distinct, test-only base URLs so the process-global registry can't collide with another
|
||||
// test running in parallel.
|
||||
let url_a = "https://shared-for-test-a.invalid/api/v1";
|
||||
let url_b = "https://shared-for-test-b.invalid/api/v1";
|
||||
|
||||
// A controller rebuilt for the same deployment (a fresh `shared_for` call, as happens per
|
||||
// message for dynamic agents) re-attaches to the SAME cache, so an earlier learning holds.
|
||||
let first = UnsupportedFieldsCache::shared_for(url_a);
|
||||
first.record("model-x", "reasoning_effort");
|
||||
|
||||
let rebuilt = UnsupportedFieldsCache::shared_for(url_a);
|
||||
assert!(
|
||||
rebuilt.known_for("model-x").contains("reasoning_effort"),
|
||||
"a rebuilt controller for the same deployment must see the earlier rejection"
|
||||
);
|
||||
|
||||
// A different deployment keeps its own learnings: a model name that rejects a field on one
|
||||
// Venice must not silence/strip it on another.
|
||||
let other_deployment = UnsupportedFieldsCache::shared_for(url_b);
|
||||
assert!(
|
||||
other_deployment.known_for("model-x").is_empty(),
|
||||
"rejections must not leak across deployments"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extracts_a_human_message_from_error_envelopes() {
|
||||
assert_eq!(
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
use mxlink::matrix_sdk::{
|
||||
Room,
|
||||
room::edit::EditedContent,
|
||||
ruma::{
|
||||
OwnedEventId, api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::OriginalSyncRoomMessageEvent,
|
||||
EventId, OwnedEventId,
|
||||
api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::{
|
||||
OriginalSyncRoomMessageEvent, RoomMessageEventContentWithoutRelation,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -52,6 +56,83 @@ impl Messaging {
|
||||
}
|
||||
}
|
||||
|
||||
/// Like `send_text_markdown_no_fail`, but logs send failures at `warn` instead of `error`.
|
||||
/// For cosmetic, best-effort sends (the thinking-notice placeholder) where a failure is
|
||||
/// acceptable-impact: the real response still ships, so this must NOT page the team.
|
||||
pub async fn send_text_markdown_no_fail_quietly(
|
||||
&self,
|
||||
room: &Room,
|
||||
message: String,
|
||||
response_type: MessageResponseType,
|
||||
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
|
||||
{
|
||||
let result = self
|
||||
.bot
|
||||
.matrix_link()
|
||||
.messaging()
|
||||
.send_text_markdown(room, message, response_type)
|
||||
.await;
|
||||
|
||||
match result {
|
||||
Ok(result) => Some(result),
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
room_id = format!("{:?}", room.room_id()),
|
||||
?err,
|
||||
"Failed to send thinking-notice placeholder to room",
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Edits an existing message's text in place (`m.replace`), best-effort.
|
||||
///
|
||||
/// Built and sent directly via matrix-sdk's `make_edit_event` + `room.send`,
|
||||
/// deliberately bypassing the mxlink messaging layer: that layer overwrites
|
||||
/// `relates_to` from a `MessageResponseType`, which would clobber the
|
||||
/// `m.replace` relation and turn the edit into a brand-new message. The edit
|
||||
/// stays in the original's thread by inheritance, so no `response_type` is needed.
|
||||
/// Failures are warned and swallowed so a flickered notice never breaks the real response.
|
||||
pub async fn edit_text_markdown_no_fail(
|
||||
&self,
|
||||
room: &Room,
|
||||
event_id: &EventId,
|
||||
markdown: String,
|
||||
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
|
||||
{
|
||||
let new_content = RoomMessageEventContentWithoutRelation::text_markdown(markdown);
|
||||
|
||||
let edit_content = match room
|
||||
.make_edit_event(event_id, EditedContent::RoomMessage(new_content))
|
||||
.await
|
||||
{
|
||||
Ok(edit_content) => edit_content,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
room_id = format!("{:?}", room.room_id()),
|
||||
?event_id,
|
||||
?err,
|
||||
"Failed to build edit event",
|
||||
);
|
||||
return None;
|
||||
}
|
||||
};
|
||||
|
||||
match room.send(edit_content).await {
|
||||
Ok(result) => Some(result.response),
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
room_id = format!("{:?}", room.room_id()),
|
||||
?event_id,
|
||||
?err,
|
||||
"Failed to send edit to room",
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn send_notice_markdown_no_fail(
|
||||
&self,
|
||||
room: &Room,
|
||||
|
||||
@@ -38,6 +38,9 @@ pub enum ConfigTextGenerationSettingRelatedControllerType {
|
||||
GetContextManagementEnabled,
|
||||
SetContextManagementEnabled(Option<bool>),
|
||||
|
||||
GetThinkingNoticeEnabled,
|
||||
SetThinkingNoticeEnabled(Option<bool>),
|
||||
|
||||
GetPrefixRequirementType,
|
||||
SetPrefixRequirementType(Option<TextGenerationPrefixRequirementType>),
|
||||
|
||||
|
||||
@@ -55,6 +55,44 @@ pub(super) fn determine(
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(remaining_text) = text.strip_prefix("thinking-notice-enabled") {
|
||||
let remaining_text = remaining_text.trim();
|
||||
|
||||
if !remaining_text.is_empty() {
|
||||
return Err(ControllerType::Error(
|
||||
strings::cfg::configuration_getter_used_with_extra_text(
|
||||
"thinking-notice-enabled",
|
||||
remaining_text,
|
||||
)
|
||||
.to_owned(),
|
||||
));
|
||||
}
|
||||
|
||||
return Ok(ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled);
|
||||
}
|
||||
|
||||
if let Some(value_string) = text.strip_prefix("set-thinking-notice-enabled") {
|
||||
let value_string = value_string.trim().to_owned();
|
||||
let value_opt = if value_string.is_empty() {
|
||||
None
|
||||
} else {
|
||||
let value_string_lowercase = value_string.to_lowercase();
|
||||
Some(match value_string_lowercase.as_str() {
|
||||
"true" => true,
|
||||
"false" => false,
|
||||
_ => {
|
||||
return Err(ControllerType::Error(
|
||||
strings::cfg::configuration_value_unrecognized(&value_string).to_owned(),
|
||||
));
|
||||
}
|
||||
})
|
||||
};
|
||||
|
||||
return Ok(
|
||||
ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value_opt),
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(remaining_text) = text.strip_prefix("prefix-requirement-type") {
|
||||
let remaining_text = remaining_text.trim();
|
||||
|
||||
|
||||
@@ -43,6 +43,27 @@ pub(super) async fn dispatch(
|
||||
}
|
||||
}
|
||||
|
||||
ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled => {
|
||||
let value = &room_settings.text_generation.thinking_notice_enabled;
|
||||
setting_get::<bool>(bot, message_context, value).await
|
||||
}
|
||||
ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value) => {
|
||||
let value = value.to_owned();
|
||||
|
||||
let setter_callback = Box::new(move |room_settings: &mut RoomSettings| {
|
||||
room_settings.text_generation.thinking_notice_enabled = value;
|
||||
});
|
||||
|
||||
match config_type {
|
||||
SettingsStorageSource::Room => {
|
||||
room_setting_set::<bool>(bot, message_context, &value, setter_callback).await
|
||||
}
|
||||
SettingsStorageSource::Global => {
|
||||
global_setting_set::<bool>(bot, message_context, &value, setter_callback).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ConfigTextGenerationSettingRelatedControllerType::GetPrefixRequirementType => {
|
||||
let value = &room_settings.text_generation.prefix_requirement_type;
|
||||
setting_get::<TextGenerationPrefixRequirementType>(bot, message_context, value).await
|
||||
|
||||
@@ -234,6 +234,44 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Thinking Notice
|
||||
|
||||
message.push_str(&format!(
|
||||
"#### {}",
|
||||
strings::help::cfg::text_generation_thinking_notice_heading()
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&strings::help::cfg::text_generation_thinking_notice_intro());
|
||||
message.push('\n');
|
||||
message.push_str(
|
||||
&strings::help::cfg::the_following_configuration_values_are_recognized(vec![true, false]),
|
||||
);
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation thinking-notice-enabled"
|
||||
)
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-thinking-notice-enabled VALUE"
|
||||
)
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-thinking-notice-enabled"
|
||||
)
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Sender Context
|
||||
|
||||
message.push_str(&format!(
|
||||
|
||||
@@ -359,6 +359,33 @@ async fn generate_text_generation_section(
|
||||
),
|
||||
);
|
||||
|
||||
// Thinking Notice
|
||||
|
||||
let effective_thinking_notice = room_config_context.text_generation_thinking_notice_enabled();
|
||||
let room_config_thinking_notice = room_config_context
|
||||
.room_config
|
||||
.settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled;
|
||||
let global_config_thinking_notice = room_config_context
|
||||
.global_config
|
||||
.fallback_room_settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled;
|
||||
|
||||
let thinking_notice_set_where = if room_config_thinking_notice.is_some() {
|
||||
strings::cfg::status_badge_set_in_room_config()
|
||||
} else if global_config_thinking_notice.is_some() {
|
||||
strings::cfg::status_badge_set_in_global_config()
|
||||
} else {
|
||||
strings::cfg::status_badge_using_hardcoded_default()
|
||||
};
|
||||
|
||||
message.push_str(&strings::cfg::status_text_generation_entry_thinking_notice(
|
||||
effective_thinking_notice,
|
||||
thinking_notice_set_where,
|
||||
));
|
||||
|
||||
// Sender Context
|
||||
|
||||
let effective_sender_context = room_config_context.text_generation_sender_context_mode();
|
||||
|
||||
@@ -30,6 +30,13 @@ use crate::{
|
||||
entity::MessageContext,
|
||||
};
|
||||
|
||||
/// How long text generation must run before the first "thinking…" placeholder appears.
|
||||
/// Fast responses (under this threshold) never get a placeholder.
|
||||
const THINKING_NOTICE_FIRST_DELAY: std::time::Duration = std::time::Duration::from_secs(3);
|
||||
|
||||
/// How often the "thinking…" placeholder is refreshed once it has appeared.
|
||||
const THINKING_NOTICE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(10);
|
||||
|
||||
#[derive(Debug, PartialEq)]
|
||||
pub enum ChatCompletionControllerType {
|
||||
// Invoked via a command prefix (e.g. `!bai Hello!`)
|
||||
@@ -520,6 +527,16 @@ async fn handle_stage_text_generation(
|
||||
conversation.start_time(),
|
||||
);
|
||||
|
||||
// Cloned only when the thinking-notice is enabled; the original is moved into `params` below.
|
||||
let notice_prompt_variables = if message_context
|
||||
.room_config_context()
|
||||
.text_generation_thinking_notice_enabled()
|
||||
{
|
||||
Some(prompt_variables.clone())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let params = TextGenerationParams {
|
||||
context_management_enabled: message_context
|
||||
.room_config_context()
|
||||
@@ -536,10 +553,75 @@ async fn handle_stage_text_generation(
|
||||
prompt_variables,
|
||||
};
|
||||
|
||||
let result = controller
|
||||
.generate_text(conversation, params)
|
||||
.instrument(span)
|
||||
.await;
|
||||
// When the thinking-notice is enabled, race generation against a timer that posts and then
|
||||
// periodically edits a "thinking…" placeholder. `biased;` makes generation win a tie, and the
|
||||
// loop exits the instant generation resolves, so there is no detached task and no late edit can
|
||||
// ever clobber the real answer. `placeholder` is the event we must finalize in every exit path.
|
||||
let (result, placeholder) = if let Some(notice_prompt_variables) = notice_prompt_variables {
|
||||
let generation = controller
|
||||
.generate_text(conversation, params)
|
||||
.instrument(span);
|
||||
tokio::pin!(generation);
|
||||
|
||||
let mut placeholder: Option<OwnedEventId> = None;
|
||||
// Seed the flavor sequence per-generation so different turns don't all open on the same
|
||||
// line; the monotonic increment then guarantees consecutive notices differ.
|
||||
let mut notice_sequence: usize = std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|since| since.subsec_nanos() as usize)
|
||||
.unwrap_or(0);
|
||||
let mut tick = tokio::time::interval_at(
|
||||
tokio::time::Instant::now() + THINKING_NOTICE_FIRST_DELAY,
|
||||
THINKING_NOTICE_INTERVAL,
|
||||
);
|
||||
// If an edit runs long, hold ~INTERVAL spacing rather than bursting the missed ticks.
|
||||
tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
|
||||
|
||||
let result = loop {
|
||||
tokio::select! {
|
||||
biased;
|
||||
generation_result = &mut generation => break generation_result,
|
||||
_ = tick.tick() => {
|
||||
let message = notice_prompt_variables.format(
|
||||
strings::thinking::pick_message(start_time.elapsed(), notice_sequence),
|
||||
);
|
||||
notice_sequence = notice_sequence.wrapping_add(1);
|
||||
|
||||
match &placeholder {
|
||||
None => {
|
||||
placeholder = bot
|
||||
.messaging()
|
||||
.send_text_markdown_no_fail_quietly(
|
||||
message_context.room(),
|
||||
message,
|
||||
response_type.clone(),
|
||||
)
|
||||
.await
|
||||
.map(|response| response.event_id);
|
||||
}
|
||||
Some(event_id) => {
|
||||
bot.messaging()
|
||||
.edit_text_markdown_no_fail(
|
||||
message_context.room(),
|
||||
event_id,
|
||||
message,
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
(result, placeholder)
|
||||
} else {
|
||||
let result = controller
|
||||
.generate_text(conversation, params)
|
||||
.instrument(span)
|
||||
.await;
|
||||
|
||||
(result, None)
|
||||
};
|
||||
|
||||
let duration = std::time::Instant::now().duration_since(start_time);
|
||||
|
||||
@@ -560,17 +642,20 @@ async fn handle_stage_text_generation(
|
||||
err,
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::TextGeneration,
|
||||
&err,
|
||||
),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
let error_text = strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::TextGeneration,
|
||||
&err,
|
||||
);
|
||||
|
||||
finalize_thinking_notice_with_error(
|
||||
bot,
|
||||
message_context,
|
||||
placeholder.as_ref(),
|
||||
&error_text,
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return None;
|
||||
}
|
||||
@@ -583,26 +668,70 @@ async fn handle_stage_text_generation(
|
||||
"Agent returned empty text",
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::empty_response_returned(agent.identifier()),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
let empty_text = strings::agent::empty_response_returned(agent.identifier());
|
||||
|
||||
finalize_thinking_notice_with_error(
|
||||
bot,
|
||||
message_context,
|
||||
placeholder.as_ref(),
|
||||
&empty_text,
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return None;
|
||||
}
|
||||
|
||||
let send_message_response = bot
|
||||
.messaging()
|
||||
.send_text_markdown_no_fail(message_context.room(), text.clone(), response_type)
|
||||
.await?;
|
||||
// Finalize the answer into a single message. With a placeholder, edit it in place (so the
|
||||
// "thinking…" message becomes the answer); the TTS payload then points at that same event.
|
||||
// If the edit fails, fall back to a fresh send so the real answer is never lost.
|
||||
let event_id = match &placeholder {
|
||||
Some(event_id)
|
||||
if bot
|
||||
.messaging()
|
||||
.edit_text_markdown_no_fail(message_context.room(), event_id, text.clone())
|
||||
.await
|
||||
.is_some() =>
|
||||
{
|
||||
event_id.clone()
|
||||
}
|
||||
_ => {
|
||||
bot.messaging()
|
||||
.send_text_markdown_no_fail(message_context.room(), text.clone(), response_type)
|
||||
.await?
|
||||
.event_id
|
||||
}
|
||||
};
|
||||
|
||||
Some(TextToSpeechEligiblePayload {
|
||||
text,
|
||||
event_id: send_message_response.event_id,
|
||||
})
|
||||
Some(TextToSpeechEligiblePayload { text, event_id })
|
||||
}
|
||||
|
||||
/// Finalizes a thinking-notice placeholder (if one was posted) with error/notice text, so a
|
||||
/// failed or empty generation never leaves an orphaned "thinking…" message behind. With no
|
||||
/// placeholder, this is the original behavior: a fresh error notice.
|
||||
async fn finalize_thinking_notice_with_error(
|
||||
bot: &Bot,
|
||||
message_context: &MessageContext,
|
||||
placeholder: Option<&OwnedEventId>,
|
||||
text: &str,
|
||||
response_type: MessageResponseType,
|
||||
) {
|
||||
match placeholder {
|
||||
Some(event_id) => {
|
||||
bot.messaging()
|
||||
.edit_text_markdown_no_fail(
|
||||
message_context.room(),
|
||||
event_id,
|
||||
crate::utils::status::create_error_message_text(text),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
None => {
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(message_context.room(), text, response_type)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn handle_stage_speech_to_text_actual_transcribing(
|
||||
|
||||
@@ -6,5 +6,5 @@ mod utils;
|
||||
mod tests;
|
||||
|
||||
pub use entity::*;
|
||||
pub use tokenization::shorten_messages_list_to_context_size;
|
||||
pub use tokenization::{TokenEstimate, shorten_messages_list_to_context_size};
|
||||
pub use utils::*;
|
||||
|
||||
@@ -4,6 +4,22 @@ use tiktoken_rs::tokenizer;
|
||||
|
||||
use super::{Author, Message, MessageContent};
|
||||
|
||||
/// How to count the tokens in a conversation when trimming it to fit the context window.
|
||||
pub enum TokenEstimate<'a> {
|
||||
/// Count via the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library.
|
||||
/// Accurate for OpenAI models; every other model falls back to the gpt-4
|
||||
/// tokenizer, which misreads it (badly so for non-English text). Use this only
|
||||
/// for the OpenAI provider.
|
||||
Tiktoken(&'a str),
|
||||
|
||||
/// Provider-neutral approximation that needs no per-model tokenizer. Expect it
|
||||
/// to land within roughly 10-20% of the real count for typical text, leaning
|
||||
/// slightly high: over-counting trims a little extra history, while
|
||||
/// under-counting would overflow the model's real context window. Use this for
|
||||
/// every non-OpenAI provider.
|
||||
Approximate,
|
||||
}
|
||||
|
||||
fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
||||
let tokenizer = tokenizer::get_tokenizer(model)
|
||||
.or_else(|| tokenizer::get_tokenizer("gpt-4"))
|
||||
@@ -13,21 +29,27 @@ fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
||||
}
|
||||
|
||||
pub fn shorten_messages_list_to_context_size(
|
||||
model: &str,
|
||||
estimate: TokenEstimate<'_>,
|
||||
prompt_message: &Option<Message>,
|
||||
mut messages: Vec<Message>,
|
||||
max_response_tokens: Option<u32>,
|
||||
max_context_tokens: u32,
|
||||
) -> Vec<Message> {
|
||||
// Loading the tokenization data is an expensive process, so
|
||||
// se construct the BPE instance once and then use it for all messages.
|
||||
let bpe = get_bpe_for_model(model);
|
||||
// Loading the tiktoken data is expensive, so we resolve the counter once up
|
||||
// front and reuse it for every message.
|
||||
let tiktoken = match estimate {
|
||||
TokenEstimate::Tiktoken(model) => Some((get_bpe_for_model(model), model)),
|
||||
TokenEstimate::Approximate => None,
|
||||
};
|
||||
let count = |message: &Message| match tiktoken {
|
||||
Some((bpe, model)) => tiktoken_token_size_for_message(bpe, model, message),
|
||||
None => approximate_token_size_for_message(message),
|
||||
};
|
||||
|
||||
// We want to retain the prompt in all cases, so we always count it first.
|
||||
// We also always reserve enough tokens for the maximum response we expect.
|
||||
let mut current_context_length: u32 = if let Some(prompt_message) = prompt_message {
|
||||
calculate_token_size_for_message(bpe, model, prompt_message)
|
||||
+ max_response_tokens.unwrap_or(0)
|
||||
count(prompt_message) + max_response_tokens.unwrap_or(0)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
@@ -37,7 +59,7 @@ pub fn shorten_messages_list_to_context_size(
|
||||
let mut messages_to_keep: Vec<Message> = Vec::new();
|
||||
|
||||
for message in messages {
|
||||
let tokens_for_message = calculate_token_size_for_message(bpe, model, &message);
|
||||
let tokens_for_message = count(&message);
|
||||
|
||||
if current_context_length + tokens_for_message > max_context_tokens {
|
||||
break;
|
||||
@@ -48,14 +70,26 @@ pub fn shorten_messages_list_to_context_size(
|
||||
messages_to_keep.push(message);
|
||||
}
|
||||
|
||||
// Cut on a turn boundary: the loop may stop right after an assistant reply
|
||||
// whose triggering user message did not fit, which would leave the kept window
|
||||
// starting on an orphaned reply. `messages_to_keep` is newest-first here, so
|
||||
// the oldest kept messages are at the end; drop any trailing assistant messages
|
||||
// until the window begins at the start of a turn (a user message).
|
||||
while matches!(
|
||||
messages_to_keep.last().map(|message| &message.author),
|
||||
Some(Author::Assistant)
|
||||
) {
|
||||
messages_to_keep.pop();
|
||||
}
|
||||
|
||||
messages_to_keep.reverse();
|
||||
|
||||
messages_to_keep
|
||||
}
|
||||
|
||||
/// Calculate the token size of a message for a given model, with a preloaded CoreBPE object.
|
||||
/// Related to `calculate_token_size_for_model_message`.
|
||||
fn calculate_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Message) -> u32 {
|
||||
/// Token size of a message via tiktoken, for a preloaded CoreBPE object.
|
||||
/// Accurate only for OpenAI models (see [`TokenEstimate::Tiktoken`]).
|
||||
fn tiktoken_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Message) -> u32 {
|
||||
let (tokens_per_message, tokens_per_name) = if model.starts_with("gpt-3.5") {
|
||||
(
|
||||
4, // every message follows <im_start>{role/name}\n{content}<im_end>\n
|
||||
@@ -80,6 +114,52 @@ fn calculate_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Messag
|
||||
(text_length + role_length + tokens_per_message + tokens_per_name) as u32
|
||||
}
|
||||
|
||||
/// ASCII text averages about four characters per token.
|
||||
const ASCII_TOKENS_PER_CHAR: f32 = 0.25;
|
||||
|
||||
/// Non-ASCII scripts (Cyrillic, CJK, and others) pack more information per
|
||||
/// character: real tokenizers land around two characters per token for them, so
|
||||
/// each counts as half a token. CJK runs a touch denser than that, so its estimate
|
||||
/// can read slightly low, still within the tolerance this approximation targets.
|
||||
const WIDE_TOKENS_PER_CHAR: f32 = 0.5;
|
||||
|
||||
/// Structural per-message overhead (role marker plus message framing), mirroring
|
||||
/// the small constant the tiktoken path adds.
|
||||
const APPROX_TOKENS_PER_MESSAGE: u32 = 4;
|
||||
|
||||
/// Provider-neutral, tokenizer-free token size of a message
|
||||
/// (see [`TokenEstimate::Approximate`]).
|
||||
fn approximate_token_size_for_message(message: &Message) -> u32 {
|
||||
let text_tokens = match &message.content {
|
||||
MessageContent::Text(text) => approximate_token_size_for_text(text),
|
||||
// Images and files are not counted as text, matching the tiktoken path.
|
||||
MessageContent::Image(..) | MessageContent::File(..) => 0,
|
||||
};
|
||||
|
||||
text_tokens + APPROX_TOKENS_PER_MESSAGE
|
||||
}
|
||||
|
||||
/// Rough token estimate for a piece of text, with no tokenizer.
|
||||
///
|
||||
/// ASCII characters count as a quarter-token each (~4 chars/token); characters
|
||||
/// outside ASCII count as half a token each (~2 chars/token), matching how real
|
||||
/// tokenizers treat Cyrillic and CJK. Weighting non-ASCII up keeps the estimate
|
||||
/// from badly under-counting non-English text, the case the tiktoken fallback gets
|
||||
/// most wrong.
|
||||
fn approximate_token_size_for_text(text: &str) -> u32 {
|
||||
let mut estimate = 0.0_f32;
|
||||
|
||||
for character in text.chars() {
|
||||
estimate += if character.is_ascii() {
|
||||
ASCII_TOKENS_PER_CHAR
|
||||
} else {
|
||||
WIDE_TOKENS_PER_CHAR
|
||||
};
|
||||
}
|
||||
|
||||
estimate.ceil() as u32
|
||||
}
|
||||
|
||||
pub mod test {
|
||||
#[test]
|
||||
fn message_size_counting_works() {
|
||||
@@ -94,7 +174,7 @@ pub mod test {
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let tokens = super::calculate_token_size_for_message(bpe, model, &message);
|
||||
let tokens = super::tiktoken_token_size_for_message(bpe, model, &message);
|
||||
|
||||
assert_eq!(8, tokens);
|
||||
}
|
||||
@@ -118,7 +198,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
prompt_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &prompt)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &prompt)
|
||||
);
|
||||
|
||||
let mut conversation_messages = Vec::new();
|
||||
@@ -133,7 +213,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
first_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &first)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &first)
|
||||
);
|
||||
|
||||
conversation_messages.push(first);
|
||||
@@ -148,7 +228,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
second_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &second)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &second)
|
||||
);
|
||||
|
||||
conversation_messages.push(second);
|
||||
@@ -165,7 +245,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
third_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &third)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &third)
|
||||
);
|
||||
|
||||
conversation_messages.push(third.clone());
|
||||
@@ -182,7 +262,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
forth_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &forth)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &forth)
|
||||
);
|
||||
|
||||
conversation_messages.push(forth.clone());
|
||||
@@ -190,7 +270,7 @@ pub mod test {
|
||||
assert_eq!(4, conversation_messages.len());
|
||||
|
||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||
model,
|
||||
super::TokenEstimate::Tiktoken(model),
|
||||
&Some(prompt),
|
||||
conversation_messages,
|
||||
max_response_tokens,
|
||||
@@ -229,7 +309,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
prompt_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &prompt)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &prompt)
|
||||
);
|
||||
|
||||
let mut conversation_messages = Vec::new();
|
||||
@@ -244,7 +324,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
first_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &first)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &first)
|
||||
);
|
||||
|
||||
conversation_messages.push(first);
|
||||
@@ -259,7 +339,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
second_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &second)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &second)
|
||||
);
|
||||
|
||||
conversation_messages.push(second);
|
||||
@@ -276,7 +356,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
third_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &third)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &third)
|
||||
);
|
||||
|
||||
conversation_messages.push(third.clone());
|
||||
@@ -293,7 +373,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
forth_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &forth)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &forth)
|
||||
);
|
||||
|
||||
conversation_messages.push(forth.clone());
|
||||
@@ -301,7 +381,7 @@ pub mod test {
|
||||
assert_eq!(4, conversation_messages.len());
|
||||
|
||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||
model,
|
||||
super::TokenEstimate::Tiktoken(model),
|
||||
&Some(prompt),
|
||||
conversation_messages,
|
||||
max_response_tokens,
|
||||
@@ -320,4 +400,126 @@ pub mod test {
|
||||
forth.content
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn approximate_counting_weights_ascii_and_wide_scripts() {
|
||||
// 12 ASCII characters at ~4 chars/token = 3 text tokens.
|
||||
assert_eq!(3, super::approximate_token_size_for_text("Hello there!"));
|
||||
|
||||
// 5 CJK characters at ~0.5 token/char = 3 text tokens. The ASCII rate would
|
||||
// have under-counted these to 2, the failure mode this path avoids.
|
||||
assert_eq!(3, super::approximate_token_size_for_text("こんにちは"));
|
||||
|
||||
let message = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("Hello there!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
// 3 text tokens plus the per-message overhead (4).
|
||||
assert_eq!(7, super::approximate_token_size_for_message(&message));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn approximate_shortening_trims_to_budget() {
|
||||
let prompt = super::Message {
|
||||
author: super::Author::Prompt,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("You are a bot!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let older = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("This is the older message.".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let newer = super::Message {
|
||||
// A user message, so it is a valid window start: keeping a lone
|
||||
// assistant reply would be an orphan and get trimmed (see
|
||||
// `shortening_cuts_on_a_turn_boundary`).
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("This is the newer message.".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
// Budget room for the prompt and only the newest message.
|
||||
let max_context_tokens = super::approximate_token_size_for_message(&prompt)
|
||||
+ super::approximate_token_size_for_message(&newer);
|
||||
|
||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||
super::TokenEstimate::Approximate,
|
||||
&Some(prompt),
|
||||
vec![older, newer.clone()],
|
||||
None,
|
||||
max_context_tokens,
|
||||
);
|
||||
|
||||
assert_eq!(1, new_conversation_messages.len());
|
||||
assert_eq!(
|
||||
new_conversation_messages.first().unwrap().content,
|
||||
newer.content
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shortening_cuts_on_a_turn_boundary() {
|
||||
// A four-message conversation of two full turns. All four messages are the
|
||||
// same length, so they cost the same number of tokens.
|
||||
let prompt = super::Message {
|
||||
author: super::Author::Prompt,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("system".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let user_one = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("user msg 1".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let asst_one = super::Message {
|
||||
author: super::Author::Assistant,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("asst msg 1".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let user_two = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("user msg 2".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let asst_two = super::Message {
|
||||
author: super::Author::Assistant,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("asst msg 2".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let per_message = super::approximate_token_size_for_message(&user_one);
|
||||
// Budget fits the prompt plus three messages. By raw token budget the loop
|
||||
// would keep asst_two, user_two, and asst_one, but asst_one's own user
|
||||
// message (user_one) does not fit, so it must be dropped too rather than
|
||||
// left as an orphaned reply.
|
||||
let max_context_tokens =
|
||||
super::approximate_token_size_for_message(&prompt) + (per_message * 3);
|
||||
|
||||
let kept = super::shorten_messages_list_to_context_size(
|
||||
super::TokenEstimate::Approximate,
|
||||
&Some(prompt),
|
||||
vec![user_one, asst_one, user_two.clone(), asst_two.clone()],
|
||||
None,
|
||||
max_context_tokens,
|
||||
);
|
||||
|
||||
// Only the last whole turn survives; the orphaned asst_one is dropped.
|
||||
assert_eq!(2, kept.len());
|
||||
assert_eq!(kept.first().unwrap().content, user_two.content);
|
||||
assert_eq!(kept.last().unwrap().content, asst_two.content);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -136,6 +136,20 @@ impl RoomConfigContext {
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
pub fn text_generation_thinking_notice_enabled(&self) -> bool {
|
||||
self.room_config
|
||||
.settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled
|
||||
.or({
|
||||
self.global_config
|
||||
.fallback_room_settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled
|
||||
})
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
pub fn text_generation_sender_context_mode(&self) -> TextGenerationSenderContextMode {
|
||||
self.room_config
|
||||
.settings
|
||||
|
||||
@@ -14,6 +14,10 @@ pub struct RoomSettingsTextGeneration {
|
||||
/// When enabled, the bot will automatically tokenize messages and try to shorten the message context intelligently.
|
||||
pub context_management_enabled: Option<bool>,
|
||||
|
||||
/// Controls whether a "thinking…" notice is posted while text generation runs longer than a threshold.
|
||||
/// When enabled, a placeholder message appears for slow responses and is edited in place (with elapsed-tiered flavor text) until it becomes the final answer.
|
||||
pub thinking_notice_enabled: Option<bool>,
|
||||
|
||||
/// Controls how each message in the conversation context is annotated with sender metadata.
|
||||
pub sender_context_mode: Option<TextGenerationSenderContextMode>,
|
||||
|
||||
|
||||
@@ -249,6 +249,10 @@ pub fn status_text_generation_entry_context_management(value: bool, set_where: &
|
||||
format!("- ♻️ Context management: `{}` ({})\n", value, set_where)
|
||||
}
|
||||
|
||||
pub fn status_text_generation_entry_thinking_notice(value: bool, set_where: &str) -> String {
|
||||
format!("- 💭 Thinking notice: `{}` ({})\n", value, set_where)
|
||||
}
|
||||
|
||||
pub fn status_text_generation_entry_sender_context(
|
||||
value: impl std::fmt::Display,
|
||||
set_where: &str,
|
||||
|
||||
@@ -128,7 +128,19 @@ pub fn text_generation_context_management_intro() -> String {
|
||||
format!(
|
||||
"{}\n{}",
|
||||
"Controls the bot's ability to **intelligently drop old messages from the conversation context** when it gets too large.",
|
||||
"This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](./providers.md#openai) models.",
|
||||
"Counting tokens precisely needs the model's own tokenizer. For [OpenAI](./providers.md#openai) models the bot uses the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library; for every other provider (including the recommended [Venice](./providers.md#venice)) it falls back to a provider-neutral **approximation** (ASCII counted at ~4 characters per token, other scripts at ~2), within roughly 10-20% of the real count for typical text.",
|
||||
)
|
||||
}
|
||||
|
||||
pub fn text_generation_thinking_notice_heading() -> &'static str {
|
||||
"💭 Thinking Notice"
|
||||
}
|
||||
|
||||
pub fn text_generation_thinking_notice_intro() -> String {
|
||||
format!(
|
||||
"{}\n{}",
|
||||
"Controls whether the bot posts a **\"thinking…\" notice** while text generation takes a while, so slow responses (for example, from reasoning models that run for minutes) don't look stuck.",
|
||||
"When enabled, a placeholder message appears only after a short delay, updates periodically with varying status text, and is then edited in place to become the final answer. Disabled by default.",
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ pub mod provider;
|
||||
pub mod room_config;
|
||||
pub mod speech_to_text;
|
||||
pub mod text_to_speech;
|
||||
pub mod thinking;
|
||||
pub mod usage;
|
||||
|
||||
pub const PROGRESS_INDICATOR_EMOJI: &str = "⏳";
|
||||
|
||||
146
src/strings/thinking.rs
Normal file
146
src/strings/thinking.rs
Normal file
@@ -0,0 +1,146 @@
|
||||
use std::time::Duration;
|
||||
|
||||
/// After this much elapsed generation time, the notice escalates to the "medium" pool.
|
||||
const TIER_MEDIUM_AFTER: Duration = Duration::from_secs(30);
|
||||
|
||||
/// After this much elapsed generation time, the notice escalates to the "deep" pool.
|
||||
const TIER_DEEP_AFTER: Duration = Duration::from_secs(90);
|
||||
|
||||
/// Early flavor: the response is just taking a moment longer than instant.
|
||||
const MESSAGES_LIGHT: &[&str] = &[
|
||||
"*{{ baibot_name }} is thinking…*",
|
||||
"*{{ baibot_name }} is mulling that over…*",
|
||||
"*Hmm, let me think about that…*",
|
||||
"*{{ baibot_name }} is gathering some thoughts…*",
|
||||
"*One moment, {{ baibot_name }} is working on it…*",
|
||||
"*{{ baibot_name }} is putting the pieces together…*",
|
||||
"*Give {{ baibot_name }} a second here…*",
|
||||
"*Let me think this one through…*",
|
||||
"*{{ baibot_name }} is warming up the gears…*",
|
||||
"*{{ baibot_name }} is on it…*",
|
||||
];
|
||||
|
||||
/// Mid flavor: this is a real question and the model is genuinely working.
|
||||
const MESSAGES_MEDIUM: &[&str] = &[
|
||||
"*Huh, good one. {{ baibot_name }} is really thinking now…*",
|
||||
"*Still working on it, {{ baibot_name }} wants to get this right…*",
|
||||
"*This one needs a bit more thought…*",
|
||||
"*{{ baibot_name }} is digging into this…*",
|
||||
"*Hang tight, {{ baibot_name }} is turning it over…*",
|
||||
"*Not a quick one, this. {{ baibot_name }} is still at it…*",
|
||||
"*{{ baibot_name }} is chewing on this properly now…*",
|
||||
"*This deserves some real thought, bear with {{ baibot_name }}…*",
|
||||
"*{{ baibot_name }} is working through the details…*",
|
||||
"*Won't be long, {{ baibot_name }} is closing in on it…*",
|
||||
];
|
||||
|
||||
/// Deep flavor: a long-running generation (e.g. a reasoning model going for minutes).
|
||||
const MESSAGES_DEEP: &[&str] = &[
|
||||
"*Okay, this is a hard one. {{ baibot_name }} is really deep in thought…*",
|
||||
"*{{ baibot_name }} is in the weeds on this one, thanks for your patience…*",
|
||||
"*Still here, still thinking. {{ baibot_name }} doesn't want to rush it…*",
|
||||
"*A proper puzzle, this. {{ baibot_name }} is taking the time to do it justice…*",
|
||||
"*{{ baibot_name }} is really wrestling with this one…*",
|
||||
"*A meaty question. {{ baibot_name }} is still turning it over…*",
|
||||
"*{{ baibot_name }} hasn't forgotten you, just thinking hard…*",
|
||||
"*Almost there, {{ baibot_name }} is pulling it all together…*",
|
||||
"*{{ baibot_name }} is going the extra mile on this one…*",
|
||||
"*Deep thoughts in progress. {{ baibot_name }} appreciates your patience…*",
|
||||
];
|
||||
|
||||
/// Returns the flavor pool matching how long generation has been running.
|
||||
pub fn messages_for_elapsed(elapsed: Duration) -> &'static [&'static str] {
|
||||
if elapsed >= TIER_DEEP_AFTER {
|
||||
MESSAGES_DEEP
|
||||
} else if elapsed >= TIER_MEDIUM_AFTER {
|
||||
MESSAGES_MEDIUM
|
||||
} else {
|
||||
MESSAGES_LIGHT
|
||||
}
|
||||
}
|
||||
|
||||
/// Picks one (still-untemplated) flavor message from the tier active at `elapsed`,
|
||||
/// indexed by a monotonic `sequence` the caller increments once per notice.
|
||||
///
|
||||
/// Using a monotonic counter (rather than a clock-derived value) guarantees two
|
||||
/// things the gesture-novelty goal needs: consecutive notices never repeat a line
|
||||
/// (the index advances by one each tick, so it differs whenever the tier has more
|
||||
/// than one message), and the pool is fully walked before any line recurs. The
|
||||
/// caller seeds `sequence` with a per-generation value so different turns don't all
|
||||
/// open on the same line. A clock-derived index can't promise this: `interval_at`
|
||||
/// ticks on a near-fixed schedule, so its sub-second component clusters and would
|
||||
/// re-pick the same line.
|
||||
pub fn pick_message(elapsed: Duration, sequence: usize) -> &'static str {
|
||||
let pool = messages_for_elapsed(elapsed);
|
||||
pool[sequence % pool.len()]
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn messages_for_elapsed_returns_the_right_tier() {
|
||||
// Boundaries: light below 30s, medium [30s, 90s), deep at/after 90s.
|
||||
// The pools have distinct content, so value comparison identifies the tier.
|
||||
assert_eq!(messages_for_elapsed(Duration::from_secs(0)), MESSAGES_LIGHT);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(29)),
|
||||
MESSAGES_LIGHT
|
||||
);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(30)),
|
||||
MESSAGES_MEDIUM
|
||||
);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(89)),
|
||||
MESSAGES_MEDIUM
|
||||
);
|
||||
assert_eq!(messages_for_elapsed(Duration::from_secs(90)), MESSAGES_DEEP);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(600)),
|
||||
MESSAGES_DEEP
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_tier_is_non_empty() {
|
||||
for pool in [MESSAGES_LIGHT, MESSAGES_MEDIUM, MESSAGES_DEEP] {
|
||||
assert!(!pool.is_empty());
|
||||
for message in pool {
|
||||
assert!(!message.trim().is_empty());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_tier_uses_the_bot_name_template() {
|
||||
// Not every line names the bot (some are first-person flavor), but each
|
||||
// tier exercises the template var so substitution is wired through.
|
||||
for pool in [MESSAGES_LIGHT, MESSAGES_MEDIUM, MESSAGES_DEEP] {
|
||||
assert!(pool.iter().any(|m| m.contains("{{ baibot_name }}")));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pick_message_stays_within_the_active_tier() {
|
||||
let deep = messages_for_elapsed(Duration::from_secs(120));
|
||||
for sequence in 0..50 {
|
||||
assert!(deep.contains(&pick_message(Duration::from_secs(120), sequence)));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pick_message_never_repeats_on_consecutive_sequences() {
|
||||
// The gesture-novelty guarantee: a monotonic sequence must not pick the same line twice
|
||||
// in a row (and walks the whole tier before any line recurs).
|
||||
let elapsed = Duration::from_secs(0);
|
||||
let pool_len = messages_for_elapsed(elapsed).len();
|
||||
for sequence in 0..(pool_len * 3) {
|
||||
assert_ne!(
|
||||
pick_message(elapsed, sequence),
|
||||
pick_message(elapsed, sequence + 1),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user