Compare commits
95 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
072fe3ec9f | ||
|
|
0d7ae70d1d | ||
|
|
f912162c4d | ||
|
|
264f71d885 | ||
|
|
2861ea6141 | ||
|
|
2014e25f77 | ||
|
|
b5c5fdb357 | ||
|
|
5ee8d18439 | ||
|
|
194a47bfc6 | ||
|
|
4f3a753f52 | ||
|
|
1a7b194588 | ||
|
|
616607433f | ||
|
|
f1043e03de | ||
|
|
6f3132edfd | ||
|
|
4f86a321e2 | ||
|
|
9d876a58a8 | ||
|
|
dfb2288fa3 | ||
|
|
24d99627ae | ||
|
|
f7c88fe532 | ||
|
|
f25d50c7f2 | ||
|
|
f22ac1636f | ||
|
|
43904e5032 | ||
|
|
4e29d2200f | ||
|
|
30fe2a8a8c | ||
|
|
e59e88f12e | ||
|
|
f27d8cfdcf | ||
|
|
909c5215d0 | ||
|
|
ca1147c10a | ||
|
|
4a4e6c1090 | ||
|
|
ea3af62afc | ||
|
|
ee7ffc8970 | ||
|
|
45f5b7a1b0 | ||
|
|
6d0396270e | ||
|
|
d52a5ea0ac | ||
|
|
30985baf0c | ||
|
|
077a86e5c9 | ||
|
|
9b3d5400ff | ||
|
|
721abe8c75 | ||
|
|
dae5fec4c3 | ||
|
|
8b3644f3db | ||
|
|
570a4be83c | ||
|
|
24266a4886 | ||
|
|
be94441ddb | ||
|
|
a4280eb8c0 | ||
|
|
9ad9a04040 | ||
|
|
57ca095df0 | ||
|
|
4b5cc24513 | ||
|
|
53daa06151 | ||
|
|
77103e2523 | ||
|
|
0a755d287e | ||
|
|
fe3296594b | ||
|
|
1e7f557334 | ||
|
|
be09ccf5da | ||
|
|
fa3646774a | ||
|
|
1774bf29eb | ||
|
|
2da4e6793a | ||
|
|
166c1f7659 | ||
|
|
0f0868b795 | ||
|
|
65d24e4681 | ||
|
|
ec83b38a1a | ||
|
|
ea13dd42d7 | ||
|
|
b405859db7 | ||
|
|
23a09cdf60 | ||
|
|
dd13deff33 | ||
|
|
0c109d4f92 | ||
|
|
e346951395 | ||
|
|
eee3c05e86 | ||
|
|
4bf44665a5 | ||
|
|
eae0851328 | ||
|
|
5212d6643e | ||
|
|
b4a070f536 | ||
|
|
de2b24f91b | ||
|
|
fbe956c0b7 | ||
|
|
a64b7da7ae | ||
|
|
ac5e776345 | ||
|
|
bea27df2e0 | ||
|
|
b7c1b6f48f | ||
|
|
8a5a38f241 | ||
|
|
5306d3bb8c | ||
|
|
75631982e7 | ||
|
|
6058bf733b | ||
|
|
a445933c0b | ||
|
|
0be08c0292 | ||
|
|
8904edb2a0 | ||
|
|
e18c083c4f | ||
|
|
1948f58473 | ||
|
|
c8a90049ad | ||
|
|
4f7778e62f | ||
|
|
355b1b299c | ||
|
|
318a8fd91d | ||
|
|
025accdeb0 | ||
|
|
edf5bc9fdb | ||
|
|
982ebc6657 | ||
|
|
4a75be742f | ||
|
|
aae3c4c08f |
88
.github/workflows/ci.yml
vendored
88
.github/workflows/ci.yml
vendored
@@ -14,13 +14,91 @@ concurrency:
|
||||
group: ci-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
jobs:
|
||||
test-and-clippy:
|
||||
name: Unit testing and linting
|
||||
prek:
|
||||
name: Lint, format & test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: dtolnay/rust-toolchain@1.93.0
|
||||
|
||||
# Toolchain version + components come from rust-toolchain.toml. rustflags is
|
||||
# cleared so plain builds don't fail on warnings; the clippy hook still does.
|
||||
- uses: actions-rust-lang/setup-rust-toolchain@v2
|
||||
with:
|
||||
rustflags: ''
|
||||
|
||||
- name: Install SQLite3
|
||||
run: sudo apt-get update && sudo apt-get install -y libsqlite3-dev
|
||||
- run: cargo test --all-features
|
||||
- run: cargo clippy
|
||||
|
||||
# just drives the prek recipes; mise provides the pinned prek (mise.toml).
|
||||
- uses: taiki-e/install-action@v2
|
||||
with:
|
||||
tool: just
|
||||
- uses: jdx/mise-action@v4
|
||||
|
||||
# Run the same prek hooks devs run locally; .pre-commit-config.yaml is the
|
||||
# source of truth. Tests are a separate step for visible timing.
|
||||
- name: Lint & format (prek hooks, excluding tests)
|
||||
run: just prek-run-on-all --skip test-unit
|
||||
|
||||
- name: Unit tests
|
||||
run: just test
|
||||
|
||||
# The prek job never builds a container image, so a bump of the Dockerfile's
|
||||
# base image reaches main unvalidated and fails later, in Publish, after
|
||||
# ghcr.io/etkecc/baibot:latest has already been attempted. These two jobs close
|
||||
# that gap: decide whether a Dockerfile changed, and if so build the image the
|
||||
# way Publish does - but without pushing anything.
|
||||
#
|
||||
# The build is gated rather than unconditional because it is a full Rust
|
||||
# release build; running it on every push would turn a ~1 minute pipeline into
|
||||
# a ~10 minute one for changes that cannot affect the image.
|
||||
docker-gate:
|
||||
name: Decide whether the image needs building
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
build: ${{ steps.decide.outputs.build }}
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Look for Dockerfile changes against main
|
||||
id: decide
|
||||
run: |
|
||||
if [ "${{ github.event_name }}" = 'workflow_dispatch' ]; then
|
||||
echo 'Forced via workflow_dispatch.'
|
||||
echo 'build=true' >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Publish builds and pushes from main, so a main-side build here would
|
||||
# be redundant. This gate exists for branches, before they merge.
|
||||
if [ "${{ github.ref_name }}" = 'main' ]; then
|
||||
echo 'On main; Publish covers this.'
|
||||
echo 'build=false' >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
git fetch --no-tags origin main
|
||||
if git diff --name-only origin/main HEAD -- Dockerfile | grep -q .; then
|
||||
echo 'A Dockerfile changed; the image will be built.'
|
||||
echo 'build=true' >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo 'No Dockerfile changed.'
|
||||
echo 'build=false' >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
docker-build:
|
||||
name: Build the container image (without publishing it)
|
||||
needs: docker-gate
|
||||
if: needs.docker-gate.outputs.build == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
# No build cache on purpose: a bump of the base image is exactly the case
|
||||
# where a cold build is the honest test.
|
||||
- name: Build
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
push: false
|
||||
|
||||
@@ -5,6 +5,10 @@ repos:
|
||||
- id: trailing-whitespace
|
||||
- id: end-of-file-fixer
|
||||
- id: check-yaml
|
||||
# This is the stock Synapse sample homeserver.yaml (mostly commented-out
|
||||
# docs). prek's stricter YAML parser (serde-saphyr, since v0.4.6) rejects
|
||||
# its long runs of consecutive comment lines in several places.
|
||||
exclude: '^etc/services/synapse/config/homeserver\.yaml$'
|
||||
- id: check-merge-conflict
|
||||
- id: check-added-large-files
|
||||
args: ['--maxkb=1024']
|
||||
|
||||
21
CHANGELOG.md
21
CHANGELOG.md
@@ -1,3 +1,24 @@
|
||||
# (2026-06-29) Version 1.25.0
|
||||
|
||||
- (**Feature**) [♻️ Context management](./docs/configuration/text-generation.md#️-context-management) now works with every provider, not only [OpenAI](./docs/providers.md#openai). Token counting previously went through [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs), which is accurate only for OpenAI models and silently mis-counted everything else (worst of all for non-English text). OpenAI agents keep using tiktoken-rs; every other provider, including the recommended [Venice](./docs/providers.md#venice), now uses a provider-neutral approximation that needs no per-model tokenizer (ASCII counted at about four characters per token, other scripts such as Cyrillic and CJK at about two), landing within roughly 10-20% of the real count. See the [context management docs](./docs/configuration/text-generation.md#️-context-management).
|
||||
|
||||
- (**Improvement**) Context management now trims a conversation on whole-turn boundaries for every provider, so an assistant reply is never kept without the user message it answered. This also adjusts how the OpenAI provider trims: a dangling assistant reply at the oldest edge of the kept history is now dropped along with its missing prompt, rather than left in place.
|
||||
|
||||
|
||||
# (2026-06-26) Version 1.24.0
|
||||
|
||||
- (**Feature**) Add an opt-in 💭 **thinking notice** for text generation. When enabled, a slow response (for example, from a reasoning model that runs for minutes) posts a "thinking…" placeholder after a short delay, refreshes it periodically with varying flavor text, and then edits that same message into the final answer, so a long wait no longer looks like a stuck bot. The notice is **disabled by default** and configurable per-room or globally via `text-generation set-thinking-notice-enabled true`. Fast responses (under the delay threshold) never show a placeholder. See the [text-generation configuration docs](./docs/configuration/text-generation.md#-thinking-notice).
|
||||
|
||||
- (**Bugfix**) The [Venice](https://venice.ai) unsupported-field auto-recovery (added in 1.23.1) now actually remembers rejections across messages. The cache lived on the provider's controller, which is rebuilt on every message for room-local and global agents, so each turn started with an empty cache, re-sent the unsupported field, and logged the same `400 Bad Request` warning again. The cache is now process-global (keyed per Venice deployment), so a field a model rejects is dropped proactively on every later request instead of being re-discovered each turn.
|
||||
|
||||
|
||||
# (2026-06-24) Version 1.23.1
|
||||
|
||||
- (**Bugfix**) The [Venice](https://venice.ai) provider now auto-recovers when a model rejects an optional knob it does not support. Venice's request body is strict (`additionalProperties: false`), so a model that lacks `prompt_cache_retention`, `reasoning_effort`, or `prompt_cache_key` rejected the whole request with a `400 Bad Request` — breaking agent creation and every reply. baibot now drops the unsupported field and retries, remembering the rejection per model so later requests skip it without a wasted round-trip. Only these meaning-preserving fields are dropped; sampling knobs that change the output (`temperature`, `top_p`, the penalties) are never silently removed and still surface as an error.
|
||||
|
||||
- (**Improvement**) When Venice rejects a request with a `400 Bad Request`, baibot now surfaces Venice's actual error message (e.g. `Extra inputs are not permitted, field: 'prompt_cache_retention'`) instead of a generic "configuration does not result in a working agent". This makes agent-creation failures self-explanatory. Other error statuses keep their bodies redacted, since those can carry account or rate-limit details.
|
||||
|
||||
|
||||
# (2026-06-23) Version 1.23.0
|
||||
|
||||
- (**Feature**) The [Venice](https://venice.ai) provider now accepts file inputs (PDF, DOCX, and other documents, up to 25MB), the same way it already handled images. This makes Venice the second provider after OpenAI to accept files; the others (Anthropic and the OpenAI-compatible providers) skip them. See the [text-generation feature docs](./docs/features.md#-text-generation).
|
||||
|
||||
105
Cargo.lock
generated
105
Cargo.lock
generated
@@ -95,9 +95,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.102"
|
||||
version = "1.0.104"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c"
|
||||
checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470"
|
||||
|
||||
[[package]]
|
||||
name = "anymap2"
|
||||
@@ -184,9 +184,9 @@ checksum = "4288f83726785267c6f2ef073a3d83dc3f9b81464e9f99898240cced85fce35a"
|
||||
|
||||
[[package]]
|
||||
name = "async-openai"
|
||||
version = "0.41.1"
|
||||
version = "0.41.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3007014661d5b98168b7b6f1014147bce8b1362a194783543eeb9f6117a20be9"
|
||||
checksum = "d72db2750faea2ca5edbf6d0c50277a89dc8f75f5e6ddd695ef30f75e335019b"
|
||||
dependencies = [
|
||||
"async-openai-macros",
|
||||
"base64 0.22.1",
|
||||
@@ -196,7 +196,7 @@ dependencies = [
|
||||
"futures",
|
||||
"getrandom 0.3.4",
|
||||
"rand 0.9.4",
|
||||
"reqwest 0.13.4",
|
||||
"reqwest 0.13.5",
|
||||
"secrecy",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -315,21 +315,21 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "baibot"
|
||||
version = "1.23.0"
|
||||
version = "1.25.0"
|
||||
dependencies = [
|
||||
"anthropic",
|
||||
"anyhow",
|
||||
"async-openai",
|
||||
"base64 0.22.1",
|
||||
"base64 0.23.1",
|
||||
"chrono",
|
||||
"etke_openai_api_rust",
|
||||
"matrix-sdk",
|
||||
"mime_guess",
|
||||
"mxidwc",
|
||||
"mxlink",
|
||||
"quick_cache",
|
||||
"quick_cache 0.7.0",
|
||||
"regex",
|
||||
"reqwest 0.13.4",
|
||||
"reqwest 0.13.5",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"serde_yaml_ng",
|
||||
@@ -353,6 +353,12 @@ version = "0.22.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
|
||||
|
||||
[[package]]
|
||||
name = "base64"
|
||||
version = "0.23.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5"
|
||||
|
||||
[[package]]
|
||||
name = "base64ct"
|
||||
version = "1.8.3"
|
||||
@@ -1233,6 +1239,12 @@ version = "0.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2"
|
||||
|
||||
[[package]]
|
||||
name = "foldhash"
|
||||
version = "0.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "77ce24cb58228fbb8aa041425bb1050850ac19177686ea6e0f41a70416f56fdb"
|
||||
|
||||
[[package]]
|
||||
name = "form_urlencoded"
|
||||
version = "1.2.2"
|
||||
@@ -1457,7 +1469,7 @@ version = "0.15.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1"
|
||||
dependencies = [
|
||||
"foldhash",
|
||||
"foldhash 0.1.5",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2182,7 +2194,7 @@ dependencies = [
|
||||
"oauth2-reqwest",
|
||||
"percent-encoding",
|
||||
"pin-project-lite",
|
||||
"reqwest 0.13.4",
|
||||
"reqwest 0.13.5",
|
||||
"ruma",
|
||||
"rustls",
|
||||
"rustls-native-certs 0.8.3",
|
||||
@@ -2474,9 +2486,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "mxidwc"
|
||||
version = "1.0.2"
|
||||
version = "1.0.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5e253f96a03d24d1c0006c5661b1e20a13dc3861f509d76c33b6b44349d4ff3b"
|
||||
checksum = "45b5d51fcf414d2aa6bffc6cd9b037e62732734a944c5da4ace6b9895ec37b93"
|
||||
dependencies = [
|
||||
"regex",
|
||||
]
|
||||
@@ -2492,7 +2504,7 @@ dependencies = [
|
||||
"hex",
|
||||
"matrix-sdk",
|
||||
"mime",
|
||||
"quick_cache",
|
||||
"quick_cache 0.6.24",
|
||||
"rand 0.10.1",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -2577,7 +2589,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "234fb5c965bbce983ee5de636a7a51d6a3223da8067ea02f9ab2d2d78ac08be2"
|
||||
dependencies = [
|
||||
"oauth2",
|
||||
"reqwest 0.13.4",
|
||||
"reqwest 0.13.5",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2848,9 +2860,9 @@ checksum = "007d8adb5ddab6f8e3f491ac63566a7d5002cc7ed73901f72057943fa71ae1ae"
|
||||
|
||||
[[package]]
|
||||
name = "quick_cache"
|
||||
version = "0.6.23"
|
||||
version = "0.6.24"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3a3db184a8b66cfe87f0263a1de147a6b554c864d1767c6f7fa4eb0e5497b565"
|
||||
checksum = "b9c6658afe513a3b484e3abfdaa0d03ef3c0bbf017542c178dd55f94eb3051f9"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"equivalent",
|
||||
@@ -2858,6 +2870,18 @@ dependencies = [
|
||||
"parking_lot",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quick_cache"
|
||||
version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "403c1a912fec895cafb223201e368234842acb9220aaf08ab042ae89ba5f135c"
|
||||
dependencies = [
|
||||
"equivalent",
|
||||
"foldhash 0.2.0",
|
||||
"hashbrown 0.17.1",
|
||||
"parking_lot",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quinn"
|
||||
version = "0.11.9"
|
||||
@@ -3046,9 +3070,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "regex"
|
||||
version = "1.12.4"
|
||||
version = "1.13.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba"
|
||||
checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d"
|
||||
dependencies = [
|
||||
"aho-corasick",
|
||||
"memchr",
|
||||
@@ -3058,9 +3082,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "regex-automata"
|
||||
version = "0.4.14"
|
||||
version = "0.4.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f"
|
||||
checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad"
|
||||
dependencies = [
|
||||
"aho-corasick",
|
||||
"memchr",
|
||||
@@ -3116,11 +3140,11 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "reqwest"
|
||||
version = "0.13.4"
|
||||
version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3"
|
||||
checksum = "16a1cfa75cc186dd73d5818e510e042e40927bccc9c236b061cea97e1eb08029"
|
||||
dependencies = [
|
||||
"base64 0.22.1",
|
||||
"base64 0.23.1",
|
||||
"bytes",
|
||||
"futures-core",
|
||||
"futures-util",
|
||||
@@ -3593,9 +3617,9 @@ checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd"
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.228"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e"
|
||||
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||
dependencies = [
|
||||
"serde_core",
|
||||
"serde_derive",
|
||||
@@ -3624,22 +3648,22 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_core"
|
||||
version = "1.0.228"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad"
|
||||
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.228"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79"
|
||||
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.117",
|
||||
"syn 3.0.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -3657,9 +3681,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.150"
|
||||
version = "1.0.151"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
|
||||
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
@@ -3881,6 +3905,17 @@ dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "3.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f2fac314a64dc9a36e61a9eb4261a5e9bbfbc922b27e518af97bc32b926cf967"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "sync_wrapper"
|
||||
version = "1.0.2"
|
||||
@@ -4046,9 +4081,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20"
|
||||
|
||||
[[package]]
|
||||
name = "tokio"
|
||||
version = "1.52.3"
|
||||
version = "1.53.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe"
|
||||
checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"libc",
|
||||
|
||||
10
Cargo.toml
10
Cargo.toml
@@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later"
|
||||
readme = "README.md"
|
||||
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
||||
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
||||
version = "1.23.0"
|
||||
version = "1.25.0"
|
||||
edition = "2024"
|
||||
|
||||
[lib]
|
||||
@@ -18,7 +18,7 @@ path = "src/lib.rs"
|
||||
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
||||
anyhow = "1.0.*"
|
||||
async-openai = { version = "0.41.0", features = ["audio", "chat-completion", "image", "responses"] }
|
||||
base64 = "0.22.*"
|
||||
base64 = "0.23.*"
|
||||
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
||||
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
||||
matrix-sdk = { version = "0.18.0", default-features = false }
|
||||
@@ -26,8 +26,8 @@ mime_guess = "2.0.*"
|
||||
mxidwc = "1.0.*"
|
||||
mxlink = ">=1.15.0"
|
||||
etke_openai_api_rust = "0.1.*"
|
||||
quick_cache = "0.6.*"
|
||||
regex = "1.12.*"
|
||||
quick_cache = "0.7.*"
|
||||
regex = "1.13.*"
|
||||
# HTTP client for the native `venice` provider. rustls only (no extra TLS stack), matching the
|
||||
# reqwest copy async-openai/matrix-sdk/mxlink already use.
|
||||
reqwest = { version = "0.13.*", default-features = false, features = ["json", "multipart", "rustls"] }
|
||||
@@ -36,7 +36,7 @@ serde_json = "1.0.*"
|
||||
serde_yaml_ng = "0.10.*"
|
||||
tempfile = "3.27.*"
|
||||
tiktoken-rs = { version = "0.12.*", default-features = false }
|
||||
tokio = { version = "1.52.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||
tokio = { version = "1.53.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||
tracing = "0.1.*"
|
||||
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
|
||||
url = "2.5.*"
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/rust:1.96.0-slim-trixie AS build
|
||||
FROM docker.io/rust:1.98.0-slim-trixie AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||
|
||||
|
||||
@@ -1,35 +0,0 @@
|
||||
#######################################
|
||||
# #
|
||||
# Stage 1: building #
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/rust:1.96.0-slim-trixie AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY . /app
|
||||
|
||||
RUN cargo build --release
|
||||
|
||||
#######################################
|
||||
# #
|
||||
# Stage 2: packaging #
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/debian:trixie-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y ca-certificates sqlite3 && \
|
||||
apt-get clean && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=build /app/target/release/baibot .
|
||||
|
||||
ENTRYPOINT ["/bin/sh", "-c"]
|
||||
|
||||
CMD ["/app/baibot"]
|
||||
@@ -30,7 +30,7 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
|
||||
|
||||
- 🔒 Supports [encryption](./docs/features.md#-encryption) for Matrix communication and Account-Data-stored configuration
|
||||
|
||||
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) handling on some models (automatically adjusting the message history length, etc.)
|
||||
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) for every [provider](./docs/providers.md) (automatically trimming older messages on whole-turn boundaries once a conversation outgrows the context window)
|
||||
|
||||
- 🛠️ Allows **customizing much of the bot's [configuration](./docs/configuration/README.md)** at runtime (using commands sent via chat)
|
||||
|
||||
|
||||
@@ -50,13 +50,22 @@ Example: `!bai config room text-generation set-auto-usage only_for_voice` (this
|
||||
|
||||
### ♻️ Context Management
|
||||
|
||||
The bot also supports ♻️ **context management**, which automatically adjusts the message history length, etc.
|
||||
The bot also supports ♻️ **context management**, which automatically trims the oldest messages once a conversation grows past the context window. It drops whole turns at a time, so a reply is never separated from the message it answered.
|
||||
|
||||
This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](../providers.md#openai) models.
|
||||
Counting tokens precisely needs the model's own tokenizer. For [OpenAI](../providers.md#openai) models, the bot counts them with the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library. For every other provider, including the recommended [Venice](../providers.md#venice), the bot falls back to a provider-neutral **approximation** that needs no per-model tokenizer: it counts ASCII text at about four characters per token and other scripts (Cyrillic, CJK, and so on) at about two. Treat it as rough, within roughly 10-20% of the real count for typical text, which is plenty for keeping a long conversation inside the context window.
|
||||
|
||||
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
||||
|
||||
|
||||
### 💭 Thinking Notice
|
||||
|
||||
The bot can post a 💭 **"thinking…" notice** while text generation is running, useful for slow models (for example, reasoning models that may run for minutes) where the response would otherwise look stuck.
|
||||
|
||||
When enabled, a placeholder message appears only after a short delay (so fast responses get no notice), updates periodically with varying status text, and is then edited in place to become the final answer.
|
||||
|
||||
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-thinking-notice-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
||||
|
||||
|
||||
### 👤 Sender Context Mode
|
||||
|
||||
In multi-user rooms, it may be useful for the model to know which participant sent each message in the conversation context.
|
||||
|
||||
@@ -57,6 +57,34 @@ CONTAINER_IMAGE_NAME=ghcr.io/etkecc/baibot:v1.0.0
|
||||
$CONTAINER_IMAGE_NAME
|
||||
```
|
||||
|
||||
Alternatively, you can use [Docker Compose](https://docs.docker.com/compose/) with a `compose.yml` file like this:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
baibot:
|
||||
container_name: baibot
|
||||
# Adjust the version tag to point to the latest available tagged version.
|
||||
# If building your own container image name, adjust to something like `localhost/baibot:latest`.
|
||||
image: ghcr.io/etkecc/baibot:v1.0.0
|
||||
# Set `UID` and `GID` in a `.env` file next to `compose.yml` (e.g. `UID=1000`, `GID=1000`)
|
||||
# or export them in your shell (`export UID GID="$(id -g)"`).
|
||||
# These should match the user that owns the data directory.
|
||||
user: "${UID:-1000}:${GID:-1000}"
|
||||
environment:
|
||||
# Other settings can also be set via environment variables.
|
||||
# See the 🛠️ Configuration documentation (docs/configuration/README.md) for details.
|
||||
BAIBOT_PERSISTENCE_DATA_DIR_PATH: /data
|
||||
volumes:
|
||||
- /path/to/config.yml:/app/config.yml:ro
|
||||
- /path/to/data:/data
|
||||
cap_drop:
|
||||
- ALL
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp:rw,noexec,nosuid,size=1024m
|
||||
restart: unless-stopped
|
||||
```
|
||||
|
||||
💡 If you've defined the `persistence.data_dir_path` setting in the `config.yml` file, you can skip the `BAIBOT_PERSISTENCE_DATA_DIR_PATH` environment variable.
|
||||
|
||||
|
||||
|
||||
@@ -24,7 +24,7 @@ The list of supported providers is below.
|
||||
|
||||
### How to choose a provider
|
||||
|
||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation) (incl. vision, incl. [🛠️ tools](./features.md#️-built-in-tools-openai-only)), [🖌️ image-generation](./features.md#️image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
||||
If you're not sure which provider to start with, **we recommend [Venice](#venice)**: it's the most capable provider baibot supports (covering [💬 text-generation](./features.md#-text-generation) with vision, file inputs, prompt caching, and native web search, plus [🖌️ image-generation](./features.md#️-image-creation) incl. editing, [🦻 speech-to-text](./features.md#-speech-to-text), and [🗣️ text-to-speech](./features.md#️-text-to-speech)) and the only one that runs inference with no logging and no training on your data. If you'd rather start with the most widely-used option, [OpenAI](#openai) is a solid, well-supported choice too.
|
||||
|
||||
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
|
||||
|
||||
@@ -176,10 +176,10 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
||||
|
||||
### Venice
|
||||
|
||||
[Venice AI](https://venice.ai) runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
||||
[Venice AI](https://venice.ai/chat?ref=kpXDe6) _(ref link with a $10 bonus for you)_ runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
||||
|
||||
- 🆔 Identifier: `venice`
|
||||
- 🔗 Links: [🏠 Home page](https://venice.ai), [👤 Sign up](https://venice.ai), [📋 Models list](https://api.venice.ai/api/v1/models)
|
||||
- 🔗 Links: [🏠 Home page](https://venice.ai/chat?ref=kpXDe6), [👤 Sign up](https://venice.ai/chat?ref=kpXDe6), [📋 Models list](https://docs.venice.ai/models/overview)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation) (incl. editing, via the native knob-rich `/image/generate` and `/image/edit` endpoints), [💬 text-generation](./features.md#-text-generation) (incl. vision, file inputs like PDF and DOCX, and prompt caching; native web search via the `venice_parameters` config), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local venice my-venice-agent`
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
services:
|
||||
continuwuity:
|
||||
image: forgejo.ellis.link/continuwuation/continuwuity:v0.5.10
|
||||
image: forgejo.ellis.link/continuwuation/continuwuity:v26.8.1
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
cap_drop:
|
||||
- ALL
|
||||
read_only: true
|
||||
environment:
|
||||
CONDUWUIT_CONFIG: /etc/continuwuity/continuwuity.toml
|
||||
CONDUWUIT_DATABASE_PATH: /var/lib/continuwuity
|
||||
CONTINUWUITY_CONFIG: /etc/continuwuity/continuwuity.toml
|
||||
CONTINUWUITY_DATABASE_PATH: /var/lib/continuwuity
|
||||
ports:
|
||||
- "${SERVICE_CONTINUWUITY_BIND_PORT_CLIENT_API}:6167"
|
||||
volumes:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
element-web:
|
||||
image: ghcr.io/element-hq/element-web:v1.12.21
|
||||
image: ghcr.io/element-hq/element-web:v1.12.27
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
ollama:
|
||||
image: docker.io/ollama/ollama:0.30.10
|
||||
image: docker.io/ollama/ollama:0.33.3
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
postgres:
|
||||
image: docker.io/postgres:18.4-alpine
|
||||
image: docker.io/postgres:18.6-alpine
|
||||
user: ${UID}:${GID}
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
@@ -14,7 +14,7 @@ services:
|
||||
- /etc/passwd:/etc/passwd:ro
|
||||
|
||||
synapse:
|
||||
image: ghcr.io/element-hq/synapse:v1.155.0
|
||||
image: ghcr.io/element-hq/synapse:v1.160.0
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
entrypoint: python
|
||||
|
||||
24
justfile
24
justfile
@@ -271,7 +271,29 @@ prek-run-on-all *args: _ensure_mise_tools_installed
|
||||
|
||||
# Installs the git pre-commit hook (runs prek automatically before each commit)
|
||||
prek-install-git-pre-commit-hook: _ensure_mise_tools_installed
|
||||
@just --justfile {{ justfile() }} mise exec -- prek install
|
||||
#!/usr/bin/env sh
|
||||
set -eu
|
||||
just --justfile {{ justfile() }} mise exec -- prek install
|
||||
# The installed git hooks run later under Git, outside this just/mise environment,
|
||||
# so they need to be told how to find their tooling:
|
||||
#
|
||||
# - MISE_DATA_DIR / MISE_TRUSTED_CONFIG_PATHS make mise resolve against this project's
|
||||
# own data directory. Without them mise falls back to the global one and silently
|
||||
# installs a second copy of the tool there.
|
||||
# - prek bakes the full path of the currently installed version into the hook
|
||||
# (var/mise/installs/prek/<version>/...), which stops working as soon as the pinned
|
||||
# version changes or old versions are pruned. Pointing at mise's shim instead makes
|
||||
# the hook resolve whatever mise.toml pins, at the time it runs.
|
||||
#
|
||||
# Which hook files prek installs depends on `default_install_hook_types` in
|
||||
# .pre-commit-config.yaml, so patch every hook file that prek generated.
|
||||
for hook in "{{ justfile_directory() }}"/.git/hooks/*; do
|
||||
[ -f "$hook" ] || continue
|
||||
grep -q 'generated by prek' "$hook" || continue
|
||||
grep -q '^export MISE_DATA_DIR=' "$hook" || sed -i '2iexport MISE_DATA_DIR="{{ mise_data_dir }}"' "$hook"
|
||||
grep -q '^export MISE_TRUSTED_CONFIG_PATHS=' "$hook" || sed -i '3iexport MISE_TRUSTED_CONFIG_PATHS="{{ mise_trusted_config_paths }}"' "$hook"
|
||||
sed -i 's#^PREK=".*"$#PREK="{{ mise_data_dir }}/shims/prek"#' "$hook"
|
||||
done
|
||||
|
||||
# Internal - ensures var/mise directory exists
|
||||
_ensure_mise_data_directory:
|
||||
|
||||
@@ -1,6 +1,2 @@
|
||||
[tools]
|
||||
prek = "0.4.5"
|
||||
|
||||
[settings]
|
||||
# Disable automatic trust prompts - we trust this config
|
||||
yes = true
|
||||
prek = "0.5.2"
|
||||
|
||||
@@ -5,5 +5,62 @@
|
||||
],
|
||||
"labels": [
|
||||
"dependencies"
|
||||
],
|
||||
"packageRules": [
|
||||
{
|
||||
"description": "Cargo dependencies, the Rust toolchain pin, prek (via mise) and the workflows' own actions merge by pushing to main, without a pull request. ci.yml runs on `push: [\"**\"]`, so the Renovate branch itself is compiled, linted with `clippy -D warnings` and unit-tested first; a failure leaves the branch red and Renovate raises a pull request instead of merging.",
|
||||
"matchManagers": [
|
||||
"cargo",
|
||||
"rust-toolchain",
|
||||
"mise",
|
||||
"github-actions"
|
||||
],
|
||||
"matchUpdateTypes": [
|
||||
"minor",
|
||||
"patch",
|
||||
"digest"
|
||||
],
|
||||
"automerge": true,
|
||||
"automergeType": "branch",
|
||||
"platformAutomerge": false
|
||||
},
|
||||
{
|
||||
"description": "The release image's base (Dockerfile). Gated by ci.yml's docker-build job, which builds the image exactly as Publish does but with `push: false`, and which only runs when the Dockerfile actually changed.",
|
||||
"matchManagers": [
|
||||
"dockerfile"
|
||||
],
|
||||
"matchUpdateTypes": [
|
||||
"minor",
|
||||
"patch",
|
||||
"digest"
|
||||
],
|
||||
"automerge": true,
|
||||
"automergeType": "branch",
|
||||
"platformAutomerge": false
|
||||
},
|
||||
{
|
||||
"description": "Local development service images under etc/services/** - the homeservers and LLM backends that `just services-start` brings up. They are never part of a shipped artifact and CI does not run them, so the justification here is blast radius rather than validation: the worst case is a broken local development stack, fixed by pinning back.",
|
||||
"matchManagers": [
|
||||
"docker-compose"
|
||||
],
|
||||
"matchFileNames": [
|
||||
"etc/services/**"
|
||||
],
|
||||
"matchUpdateTypes": [
|
||||
"minor",
|
||||
"patch",
|
||||
"digest"
|
||||
],
|
||||
"automerge": true,
|
||||
"automergeType": "branch",
|
||||
"platformAutomerge": false
|
||||
},
|
||||
{
|
||||
"description": "Major updates always get a pull request and a human. This is deliberately the last rule so that it overrides the automerge rules above for every manager.",
|
||||
"matchUpdateTypes": [
|
||||
"major"
|
||||
],
|
||||
"automerge": false
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
[toolchain]
|
||||
channel = "1.96.0"
|
||||
channel = "1.98.1"
|
||||
components = ["rustfmt", "clippy"]
|
||||
profile = "default"
|
||||
|
||||
@@ -15,7 +15,7 @@ use crate::agent::provider::{
|
||||
};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
@@ -130,7 +130,7 @@ impl ControllerTrait for Controller {
|
||||
tracing::trace!("Shortening messages list to context size");
|
||||
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Approximate,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
Some(text_generation_config.max_response_tokens),
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
use chrono::{DateTime, Utc};
|
||||
use std::collections::HashMap;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct TextGenerationPromptVariables {
|
||||
map: HashMap<String, String>,
|
||||
}
|
||||
|
||||
@@ -24,7 +24,7 @@ use crate::{
|
||||
},
|
||||
conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
},
|
||||
utils::base64::base64_decode,
|
||||
};
|
||||
@@ -117,7 +117,7 @@ impl ControllerTrait for Controller {
|
||||
tracing::trace!("Shortening messages list to context size");
|
||||
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Tiktoken(&text_generation_config.model_id),
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
|
||||
@@ -15,7 +15,7 @@ use crate::{
|
||||
},
|
||||
conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
},
|
||||
};
|
||||
use crate::{
|
||||
@@ -114,7 +114,7 @@ impl ControllerTrait for Controller {
|
||||
tracing::trace!("Shortening messages list to context size");
|
||||
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Approximate,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
|
||||
@@ -8,7 +8,7 @@ use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
@@ -19,6 +19,7 @@ use super::wire::{ChatCompletionRequest, ChatCompletionResponse, WebSearchCitati
|
||||
pub async fn generate_text(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
unsupported: &super::recovery::UnsupportedFieldsCache,
|
||||
conversation: LLMConversation,
|
||||
params: TextGenerationParams,
|
||||
) -> anyhow::Result<TextGenerationResult> {
|
||||
@@ -63,7 +64,7 @@ pub async fn generate_text(
|
||||
|
||||
if params.context_management_enabled {
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
TokenEstimate::Approximate,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
@@ -98,7 +99,7 @@ pub async fn generate_text(
|
||||
vp
|
||||
});
|
||||
|
||||
let request = ChatCompletionRequest {
|
||||
let mut request = ChatCompletionRequest {
|
||||
model: text_generation_config.model_id.clone(),
|
||||
messages,
|
||||
temperature: Some(temperature),
|
||||
@@ -117,40 +118,82 @@ pub async fn generate_text(
|
||||
|
||||
let url = format!("{}/chat/completions", config.base_url.trim_end_matches('/'));
|
||||
|
||||
tracing::trace!(
|
||||
model = text_generation_config.model_id,
|
||||
messages_count = request.messages.len(),
|
||||
"Sending Venice chat completion API request"
|
||||
);
|
||||
let model_id = text_generation_config.model_id.clone();
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
// Proactively drop fields this model has already rejected earlier in this process, so a known
|
||||
// mismatch costs zero wasted round-trips after the first discovery. Venice's body is
|
||||
// `additionalProperties: false`, so sending a known-unsupported field would 400 again.
|
||||
for field in unsupported.known_for(&model_id) {
|
||||
super::recovery::strip_droppable_field(&mut request, &field);
|
||||
}
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Log the body server-side for debugging (Venice explains a rejected strict body there),
|
||||
// but keep it OUT of the returned error: that error surfaces in the Matrix room, and the
|
||||
// body can carry account / rate-limit details that shouldn't reach room members.
|
||||
// Send with bounded auto-recovery. When Venice 400s because a model does not support an optional
|
||||
// knob, it names the field (`field: '...'`); if that field is one we may safely drop, we strip
|
||||
// it, remember the rejection for this model, and retry. The loop is bounded: each retry clears a
|
||||
// distinct droppable field (strip returns false once it is gone), so after at most
|
||||
// `DROPPABLE_FIELDS.len()` retries the request either succeeds or surfaces the error.
|
||||
let response = loop {
|
||||
tracing::trace!(
|
||||
model = model_id,
|
||||
messages_count = request.messages.len(),
|
||||
"Sending Venice chat completion API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if status.is_success() {
|
||||
break response;
|
||||
}
|
||||
|
||||
// Always log the body server-side: Venice explains a rejected strict body there.
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice chat completion request failed");
|
||||
|
||||
// Recover from a strict-body 400 over an unsupported optional knob: strip the named field
|
||||
// and retry. Only fields in `DROPPABLE_FIELDS` are eligible, so a meaning-bearing knob (a
|
||||
// sampling parameter) is never silently dropped; that case falls through to surface below.
|
||||
if status == reqwest::StatusCode::BAD_REQUEST
|
||||
&& let Some(field) = super::recovery::parse_rejected_field(&body)
|
||||
&& super::recovery::strip_droppable_field(&mut request, &field)
|
||||
{
|
||||
unsupported.record(&model_id, &field);
|
||||
tracing::info!(
|
||||
model = model_id,
|
||||
field,
|
||||
"Venice rejected an unsupported field; dropping it and retrying"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
// A 413 almost always means an attached file pushed the request past Venice's size limit.
|
||||
// Surface a clear, actionable message rather than the opaque status; the raw body still
|
||||
// stays out of the room for the reason above.
|
||||
if status == reqwest::StatusCode::PAYLOAD_TOO_LARGE {
|
||||
return Err(anyhow::anyhow!(
|
||||
"The request was too large for Venice, most likely an attached file over the 25MB limit."
|
||||
));
|
||||
}
|
||||
|
||||
// A 400 is a complaint about the request baibot built, so the body is safe and useful to
|
||||
// surface: it tells the operator (e.g. at agent-create time) exactly which field or value
|
||||
// Venice rejected, instead of an opaque status. Other statuses keep the body OUT of the
|
||||
// returned error, since it can carry account / rate-limit details that shouldn't reach the
|
||||
// room.
|
||||
if status == reqwest::StatusCode::BAD_REQUEST {
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice rejected the request (400 Bad Request): {}",
|
||||
super::recovery::extract_error_message(&body)
|
||||
));
|
||||
}
|
||||
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice chat completion request failed with status {status}"
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
let response: ChatCompletionResponse = response.json().await?;
|
||||
|
||||
|
||||
@@ -13,11 +13,16 @@ use crate::conversation::llm::{
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use super::config::Config;
|
||||
use super::recovery::UnsupportedFieldsCache;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Controller {
|
||||
config: Config,
|
||||
http: reqwest::Client,
|
||||
// Per-model record of chat fields this Venice deployment has rejected as unsupported, learned at
|
||||
// runtime. `Arc`-backed inside, so the `Clone` derive shares one cache across all clones of an
|
||||
// agent's controller.
|
||||
unsupported_fields: UnsupportedFieldsCache,
|
||||
}
|
||||
|
||||
impl Controller {
|
||||
@@ -30,7 +35,16 @@ impl Controller {
|
||||
.build()
|
||||
.unwrap_or_else(|_| reqwest::Client::new());
|
||||
|
||||
Self { config, http }
|
||||
// Process-global per-deployment cache rather than a fresh one: dynamic agents rebuild their
|
||||
// controller on every message, so an instance-owned cache would never retain a learned
|
||||
// rejection and the same unsupported field would 400 (and warn) on every turn.
|
||||
let unsupported_fields = UnsupportedFieldsCache::shared_for(&config.base_url);
|
||||
|
||||
Self {
|
||||
config,
|
||||
http,
|
||||
unsupported_fields,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -62,7 +76,14 @@ impl ControllerTrait for Controller {
|
||||
conversation: LLMConversation,
|
||||
params: TextGenerationParams,
|
||||
) -> anyhow::Result<TextGenerationResult> {
|
||||
super::chat::generate_text(&self.config, &self.http, conversation, params).await
|
||||
super::chat::generate_text(
|
||||
&self.config,
|
||||
&self.http,
|
||||
&self.unsupported_fields,
|
||||
conversation,
|
||||
params,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn speech_to_text(
|
||||
|
||||
@@ -3,6 +3,7 @@ mod chat;
|
||||
mod config;
|
||||
mod controller;
|
||||
mod images;
|
||||
mod recovery;
|
||||
mod utils;
|
||||
mod wire;
|
||||
|
||||
|
||||
312
src/agent/provider/venice/recovery.rs
Normal file
312
src/agent/provider/venice/recovery.rs
Normal file
@@ -0,0 +1,312 @@
|
||||
//! Auto-recovery for Venice's strict request bodies.
|
||||
//!
|
||||
//! Venice's `/chat/completions` body is `additionalProperties: false`, so a model that does not
|
||||
//! support an optional knob rejects the whole request with a 400 instead of ignoring the field.
|
||||
//! Some knobs are documented as model-specific ("for supported models") and are pure
|
||||
//! optimization/tuning hints: dropping them changes nothing about the answer, only loses the
|
||||
//! optimization. When such a field is the reason for a 400, we strip it and retry, then remember
|
||||
//! the rejection per model so later requests skip the field (and the wasted round-trip) entirely.
|
||||
//!
|
||||
//! Universal sampling knobs (`temperature`, `top_p`, the penalties, `max_completion_tokens`) are
|
||||
//! deliberately NOT recoverable here: dropping one silently changes the model's output, so a model
|
||||
//! that rejects one is a real configuration problem the operator must see, not paper over.
|
||||
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::sync::{Arc, OnceLock, RwLock};
|
||||
|
||||
use regex::Regex;
|
||||
|
||||
use super::wire::ChatCompletionRequest;
|
||||
|
||||
/// Top-level chat-completion fields baibot may drop to recover from a 400. Each is documented by
|
||||
/// Venice as model-specific or as a routing hint, and is meaning-preserving to omit (Venice falls
|
||||
/// back to its server-side default):
|
||||
/// - `prompt_cache_retention` — "extends retention ... for supported models" (cache TTL only)
|
||||
/// - `reasoning_effort` — "control reasoning effort level for supported models"
|
||||
/// - `prompt_cache_key` — cache-routing hint; dropping it only forfeits a cache-hit optimization
|
||||
///
|
||||
/// Recovery operates on TOP-LEVEL request fields only. Sub-fields inside the `venice_parameters`
|
||||
/// bag (`disable_thinking`, `enable_e2ee`, `character_slug`, …) are intentionally absent: a model
|
||||
/// that rejects one surfaces a clear 400 to the operator rather than being auto-stripped. Adding a
|
||||
/// new bag field does not extend recovery to it; only a name listed here is droppable.
|
||||
pub(super) const DROPPABLE_FIELDS: &[&str] = &[
|
||||
"prompt_cache_retention",
|
||||
"prompt_cache_key",
|
||||
"reasoning_effort",
|
||||
];
|
||||
|
||||
/// Per-model record of fields a Venice model has rejected as unsupported, learned at runtime from
|
||||
/// 400 responses. The `Arc` is shared across `Controller` clones, so a rejection learned once is
|
||||
/// seen by every clone of the same agent. The cache is process-lived only: a restart re-learns on
|
||||
/// the first request to each model, which costs one extra round-trip and nothing else, so there is
|
||||
/// no persistence to keep in sync with config changes.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub(super) struct UnsupportedFieldsCache {
|
||||
inner: Arc<RwLock<HashMap<String, HashSet<String>>>>,
|
||||
}
|
||||
|
||||
/// Process-global registry of per-deployment caches, keyed by Venice base URL. A `Controller` is
|
||||
/// rebuilt from its config on every message for dynamic (global/room-local) agents (see
|
||||
/// `agent::manager::available_room_agents_by_room_config_context`), so a cache stored on the
|
||||
/// controller would be reset each message and the "process-lived" intent above would never hold.
|
||||
/// The registry lets a rebuilt controller re-attach to the same cache. Keyed by base URL because
|
||||
/// "which fields a model rejects" is a property of the deployment+model, not of the agent instance:
|
||||
/// agents on the same Venice deployment share learnings, while a distinct deployment keeps its own,
|
||||
/// so a model name that supports a field on one deployment is never proactively stripped against
|
||||
/// another.
|
||||
fn registry() -> &'static RwLock<HashMap<String, UnsupportedFieldsCache>> {
|
||||
static REGISTRY: OnceLock<RwLock<HashMap<String, UnsupportedFieldsCache>>> = OnceLock::new();
|
||||
REGISTRY.get_or_init(|| RwLock::new(HashMap::new()))
|
||||
}
|
||||
|
||||
impl UnsupportedFieldsCache {
|
||||
/// The process-global cache for `base_url`, created on first use and shared by every controller
|
||||
/// for that deployment thereafter. This is what makes a learned rejection survive the
|
||||
/// per-message controller rebuild. A poisoned registry lock degrades to an isolated cache:
|
||||
/// correctness holds, only cross-controller sharing is lost for that call.
|
||||
pub(super) fn shared_for(base_url: &str) -> Self {
|
||||
if let Ok(map) = registry().read()
|
||||
&& let Some(cache) = map.get(base_url)
|
||||
{
|
||||
return cache.clone();
|
||||
}
|
||||
|
||||
match registry().write() {
|
||||
Ok(mut map) => map.entry(base_url.to_owned()).or_default().clone(),
|
||||
Err(_) => Self::default(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Fields already known unsupported for `model_id`. Returns an empty set on an unknown model or
|
||||
/// a poisoned lock, so a cache failure degrades to "strip nothing proactively" rather than
|
||||
/// breaking the request path.
|
||||
pub(super) fn known_for(&self, model_id: &str) -> HashSet<String> {
|
||||
self.inner
|
||||
.read()
|
||||
.ok()
|
||||
.and_then(|map| map.get(model_id).cloned())
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Records that `model_id` rejected `field`. A poisoned lock is ignored: failing to memoize only
|
||||
/// means the next request re-discovers the rejection, never a wrong result.
|
||||
pub(super) fn record(&self, model_id: &str, field: &str) {
|
||||
if let Ok(mut map) = self.inner.write() {
|
||||
map.entry(model_id.to_owned())
|
||||
.or_default()
|
||||
.insert(field.to_owned());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses the offending field name out of a Venice 400 body. A field the schema does not allow is
|
||||
/// reported as `... field: 'prompt_cache_retention', value: '...'`, so this matches the `field: '..'`
|
||||
/// marker wherever it sits in the message. Returns `None` when the body carries no such marker (a
|
||||
/// different 400 class, e.g. a missing required field), so the caller surfaces that error instead.
|
||||
///
|
||||
/// Returns only the FIRST `field: '..'` match by design. If Venice ever names several rejected
|
||||
/// fields in one body, the retry loop strips this one, retries, and rediscovers the next on the
|
||||
/// following 400 — bounded and correct. Do not switch to `captures_iter` to "batch" them without
|
||||
/// re-checking the loop's per-field termination bound in `chat.rs`.
|
||||
pub(super) fn parse_rejected_field(body: &str) -> Option<String> {
|
||||
static RE: OnceLock<Regex> = OnceLock::new();
|
||||
let re =
|
||||
RE.get_or_init(|| Regex::new(r"field: '([^']+)'").expect("rejected-field regex is valid"));
|
||||
re.captures(body)
|
||||
.map(|caps| caps[1].to_owned())
|
||||
.filter(|field| !field.is_empty())
|
||||
}
|
||||
|
||||
/// Clears `field` from the request when it is one baibot may safely drop and it is currently set.
|
||||
/// Returns `true` only when a value was actually removed, which is what bounds the retry loop: once
|
||||
/// a field is `None`, a repeat rejection for the same name returns `false` and the caller stops
|
||||
/// instead of retrying forever. A field outside [`DROPPABLE_FIELDS`] always returns `false`, so a
|
||||
/// meaning-bearing knob is never silently dropped.
|
||||
pub(super) fn strip_droppable_field(request: &mut ChatCompletionRequest, field: &str) -> bool {
|
||||
if !DROPPABLE_FIELDS.contains(&field) {
|
||||
return false;
|
||||
}
|
||||
match field {
|
||||
"prompt_cache_retention" => request.prompt_cache_retention.take().is_some(),
|
||||
"prompt_cache_key" => request.prompt_cache_key.take().is_some(),
|
||||
"reasoning_effort" => request.reasoning_effort.take().is_some(),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Pulls a human-readable message out of a Venice error body for surfacing in the room. Venice's
|
||||
/// usual envelope is `{"error": "..."}`; some OpenAI-compatible paths nest `{"error": {"message":
|
||||
/// "..."}}`. Falls back to the trimmed raw body (length-capped so a large body cannot flood the
|
||||
/// room) and finally to a fixed string for an empty body, so the caller always has something to
|
||||
/// show.
|
||||
pub(super) fn extract_error_message(body: &str) -> String {
|
||||
let trimmed = body.trim();
|
||||
if trimmed.is_empty() {
|
||||
return "no response body".to_owned();
|
||||
}
|
||||
|
||||
if let Ok(value) = serde_json::from_str::<serde_json::Value>(trimmed) {
|
||||
if let Some(msg) = value.get("error").and_then(|e| e.as_str()) {
|
||||
return msg.to_owned();
|
||||
}
|
||||
if let Some(msg) = value
|
||||
.get("error")
|
||||
.and_then(|e| e.get("message"))
|
||||
.and_then(|m| m.as_str())
|
||||
{
|
||||
return msg.to_owned();
|
||||
}
|
||||
}
|
||||
|
||||
const MAX: usize = 500;
|
||||
if trimmed.chars().count() > MAX {
|
||||
trimmed.chars().take(MAX).collect::<String>() + "…"
|
||||
} else {
|
||||
trimmed.to_owned()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn full_request() -> ChatCompletionRequest {
|
||||
ChatCompletionRequest {
|
||||
model: "venice-uncensored".to_owned(),
|
||||
messages: vec![],
|
||||
temperature: Some(0.7),
|
||||
max_completion_tokens: Some(1024),
|
||||
top_p: None,
|
||||
frequency_penalty: None,
|
||||
presence_penalty: None,
|
||||
repetition_penalty: None,
|
||||
reasoning_effort: Some("high".to_owned()),
|
||||
prompt_cache_key: Some("cafef00d".to_owned()),
|
||||
prompt_cache_retention: Some("24h".to_owned()),
|
||||
venice_parameters: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_the_rejected_field_from_a_real_venice_body() {
|
||||
let body = r#"{"error":"Extra inputs are not permitted, field: 'prompt_cache_retention', value: 'default'","request_id":"qM_DmKSXKF07wRxmQJ-hc"}"#;
|
||||
assert_eq!(
|
||||
parse_rejected_field(body).as_deref(),
|
||||
Some("prompt_cache_retention")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn returns_no_field_when_the_body_has_no_field_marker() {
|
||||
// A different 400 class (e.g. a genuinely malformed request) carries no `field: '..'`
|
||||
// marker, so there is nothing to strip and the caller must surface the error instead.
|
||||
assert_eq!(parse_rejected_field(r#"{"error":"Invalid request"}"#), None);
|
||||
assert_eq!(parse_rejected_field(""), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_a_droppable_field_once_then_reports_no_progress() {
|
||||
let mut request = full_request();
|
||||
|
||||
// First strip clears the field and reports progress, so the caller retries.
|
||||
assert!(strip_droppable_field(
|
||||
&mut request,
|
||||
"prompt_cache_retention"
|
||||
));
|
||||
assert!(request.prompt_cache_retention.is_none());
|
||||
|
||||
// A repeat rejection for the same (now absent) field reports no progress: this is what
|
||||
// stops the retry loop instead of spinning forever.
|
||||
assert!(!strip_droppable_field(
|
||||
&mut request,
|
||||
"prompt_cache_retention"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn refuses_to_strip_a_meaning_bearing_field() {
|
||||
let mut request = full_request();
|
||||
|
||||
// `temperature` is universal and changes the output; a rejection for it must surface, never
|
||||
// be silently dropped. The whole droppable set is the only thing strip will touch.
|
||||
assert!(!strip_droppable_field(&mut request, "temperature"));
|
||||
assert_eq!(request.temperature, Some(0.7));
|
||||
|
||||
assert!(!strip_droppable_field(
|
||||
&mut request,
|
||||
"max_completion_tokens"
|
||||
));
|
||||
assert_eq!(request.max_completion_tokens, Some(1024));
|
||||
|
||||
for field in DROPPABLE_FIELDS {
|
||||
assert!(
|
||||
strip_droppable_field(&mut full_request(), field),
|
||||
"every advertised droppable field must actually be strippable: {field}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cache_records_per_model_and_isolates_models() {
|
||||
let cache = UnsupportedFieldsCache::default();
|
||||
assert!(cache.known_for("venice-uncensored").is_empty());
|
||||
|
||||
cache.record("venice-uncensored", "prompt_cache_retention");
|
||||
cache.record("venice-uncensored", "reasoning_effort");
|
||||
|
||||
let known = cache.known_for("venice-uncensored");
|
||||
assert!(known.contains("prompt_cache_retention"));
|
||||
assert!(known.contains("reasoning_effort"));
|
||||
|
||||
// A rejection learned for one model must not leak to another.
|
||||
assert!(cache.known_for("kimi-k2-5").is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shared_for_survives_controller_rebuild_and_isolates_deployments() {
|
||||
// Distinct, test-only base URLs so the process-global registry can't collide with another
|
||||
// test running in parallel.
|
||||
let url_a = "https://shared-for-test-a.invalid/api/v1";
|
||||
let url_b = "https://shared-for-test-b.invalid/api/v1";
|
||||
|
||||
// A controller rebuilt for the same deployment (a fresh `shared_for` call, as happens per
|
||||
// message for dynamic agents) re-attaches to the SAME cache, so an earlier learning holds.
|
||||
let first = UnsupportedFieldsCache::shared_for(url_a);
|
||||
first.record("model-x", "reasoning_effort");
|
||||
|
||||
let rebuilt = UnsupportedFieldsCache::shared_for(url_a);
|
||||
assert!(
|
||||
rebuilt.known_for("model-x").contains("reasoning_effort"),
|
||||
"a rebuilt controller for the same deployment must see the earlier rejection"
|
||||
);
|
||||
|
||||
// A different deployment keeps its own learnings: a model name that rejects a field on one
|
||||
// Venice must not silence/strip it on another.
|
||||
let other_deployment = UnsupportedFieldsCache::shared_for(url_b);
|
||||
assert!(
|
||||
other_deployment.known_for("model-x").is_empty(),
|
||||
"rejections must not leak across deployments"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extracts_a_human_message_from_error_envelopes() {
|
||||
assert_eq!(
|
||||
extract_error_message(r#"{"error":"Extra inputs are not permitted","request_id":"x"}"#),
|
||||
"Extra inputs are not permitted"
|
||||
);
|
||||
|
||||
// OpenAI-style nested envelope.
|
||||
assert_eq!(
|
||||
extract_error_message(r#"{"error":{"message":"context length exceeded"}}"#),
|
||||
"context length exceeded"
|
||||
);
|
||||
|
||||
// Unknown shape falls back to the raw body; empty falls back to a fixed string.
|
||||
assert_eq!(
|
||||
extract_error_message("plain text failure"),
|
||||
"plain text failure"
|
||||
);
|
||||
assert_eq!(extract_error_message(" "), "no response body");
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,12 @@
|
||||
use mxlink::matrix_sdk::{
|
||||
Room,
|
||||
room::edit::EditedContent,
|
||||
ruma::{
|
||||
OwnedEventId, api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::OriginalSyncRoomMessageEvent,
|
||||
EventId, OwnedEventId,
|
||||
api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::{
|
||||
OriginalSyncRoomMessageEvent, RoomMessageEventContentWithoutRelation,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -52,6 +56,83 @@ impl Messaging {
|
||||
}
|
||||
}
|
||||
|
||||
/// Like `send_text_markdown_no_fail`, but logs send failures at `warn` instead of `error`.
|
||||
/// For cosmetic, best-effort sends (the thinking-notice placeholder) where a failure is
|
||||
/// acceptable-impact: the real response still ships, so this must NOT page the team.
|
||||
pub async fn send_text_markdown_no_fail_quietly(
|
||||
&self,
|
||||
room: &Room,
|
||||
message: String,
|
||||
response_type: MessageResponseType,
|
||||
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
|
||||
{
|
||||
let result = self
|
||||
.bot
|
||||
.matrix_link()
|
||||
.messaging()
|
||||
.send_text_markdown(room, message, response_type)
|
||||
.await;
|
||||
|
||||
match result {
|
||||
Ok(result) => Some(result),
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
room_id = format!("{:?}", room.room_id()),
|
||||
?err,
|
||||
"Failed to send thinking-notice placeholder to room",
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Edits an existing message's text in place (`m.replace`), best-effort.
|
||||
///
|
||||
/// Built and sent directly via matrix-sdk's `make_edit_event` + `room.send`,
|
||||
/// deliberately bypassing the mxlink messaging layer: that layer overwrites
|
||||
/// `relates_to` from a `MessageResponseType`, which would clobber the
|
||||
/// `m.replace` relation and turn the edit into a brand-new message. The edit
|
||||
/// stays in the original's thread by inheritance, so no `response_type` is needed.
|
||||
/// Failures are warned and swallowed so a flickered notice never breaks the real response.
|
||||
pub async fn edit_text_markdown_no_fail(
|
||||
&self,
|
||||
room: &Room,
|
||||
event_id: &EventId,
|
||||
markdown: String,
|
||||
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
|
||||
{
|
||||
let new_content = RoomMessageEventContentWithoutRelation::text_markdown(markdown);
|
||||
|
||||
let edit_content = match room
|
||||
.make_edit_event(event_id, EditedContent::RoomMessage(new_content))
|
||||
.await
|
||||
{
|
||||
Ok(edit_content) => edit_content,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
room_id = format!("{:?}", room.room_id()),
|
||||
?event_id,
|
||||
?err,
|
||||
"Failed to build edit event",
|
||||
);
|
||||
return None;
|
||||
}
|
||||
};
|
||||
|
||||
match room.send(edit_content).await {
|
||||
Ok(result) => Some(result.response),
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
room_id = format!("{:?}", room.room_id()),
|
||||
?event_id,
|
||||
?err,
|
||||
"Failed to send edit to room",
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn send_notice_markdown_no_fail(
|
||||
&self,
|
||||
room: &Room,
|
||||
|
||||
@@ -38,6 +38,9 @@ pub enum ConfigTextGenerationSettingRelatedControllerType {
|
||||
GetContextManagementEnabled,
|
||||
SetContextManagementEnabled(Option<bool>),
|
||||
|
||||
GetThinkingNoticeEnabled,
|
||||
SetThinkingNoticeEnabled(Option<bool>),
|
||||
|
||||
GetPrefixRequirementType,
|
||||
SetPrefixRequirementType(Option<TextGenerationPrefixRequirementType>),
|
||||
|
||||
|
||||
@@ -55,6 +55,44 @@ pub(super) fn determine(
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(remaining_text) = text.strip_prefix("thinking-notice-enabled") {
|
||||
let remaining_text = remaining_text.trim();
|
||||
|
||||
if !remaining_text.is_empty() {
|
||||
return Err(ControllerType::Error(
|
||||
strings::cfg::configuration_getter_used_with_extra_text(
|
||||
"thinking-notice-enabled",
|
||||
remaining_text,
|
||||
)
|
||||
.to_owned(),
|
||||
));
|
||||
}
|
||||
|
||||
return Ok(ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled);
|
||||
}
|
||||
|
||||
if let Some(value_string) = text.strip_prefix("set-thinking-notice-enabled") {
|
||||
let value_string = value_string.trim().to_owned();
|
||||
let value_opt = if value_string.is_empty() {
|
||||
None
|
||||
} else {
|
||||
let value_string_lowercase = value_string.to_lowercase();
|
||||
Some(match value_string_lowercase.as_str() {
|
||||
"true" => true,
|
||||
"false" => false,
|
||||
_ => {
|
||||
return Err(ControllerType::Error(
|
||||
strings::cfg::configuration_value_unrecognized(&value_string).to_owned(),
|
||||
));
|
||||
}
|
||||
})
|
||||
};
|
||||
|
||||
return Ok(
|
||||
ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value_opt),
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(remaining_text) = text.strip_prefix("prefix-requirement-type") {
|
||||
let remaining_text = remaining_text.trim();
|
||||
|
||||
|
||||
@@ -43,6 +43,27 @@ pub(super) async fn dispatch(
|
||||
}
|
||||
}
|
||||
|
||||
ConfigTextGenerationSettingRelatedControllerType::GetThinkingNoticeEnabled => {
|
||||
let value = &room_settings.text_generation.thinking_notice_enabled;
|
||||
setting_get::<bool>(bot, message_context, value).await
|
||||
}
|
||||
ConfigTextGenerationSettingRelatedControllerType::SetThinkingNoticeEnabled(value) => {
|
||||
let value = value.to_owned();
|
||||
|
||||
let setter_callback = Box::new(move |room_settings: &mut RoomSettings| {
|
||||
room_settings.text_generation.thinking_notice_enabled = value;
|
||||
});
|
||||
|
||||
match config_type {
|
||||
SettingsStorageSource::Room => {
|
||||
room_setting_set::<bool>(bot, message_context, &value, setter_callback).await
|
||||
}
|
||||
SettingsStorageSource::Global => {
|
||||
global_setting_set::<bool>(bot, message_context, &value, setter_callback).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ConfigTextGenerationSettingRelatedControllerType::GetPrefixRequirementType => {
|
||||
let value = &room_settings.text_generation.prefix_requirement_type;
|
||||
setting_get::<TextGenerationPrefixRequirementType>(bot, message_context, value).await
|
||||
|
||||
@@ -136,7 +136,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation prefix-requirement-type"
|
||||
)
|
||||
@@ -144,7 +144,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-prefix-requirement-type VALUE"
|
||||
)
|
||||
@@ -152,7 +152,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-prefix-requirement-type"
|
||||
)
|
||||
@@ -176,12 +176,12 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(command_prefix, "text-generation auto-usage")
|
||||
strings::help::cfg::current_setting_show(command_prefix, "text-generation auto-usage")
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-auto-usage VALUE"
|
||||
)
|
||||
@@ -189,10 +189,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-auto-usage"
|
||||
)
|
||||
strings::help::cfg::current_setting_unset(command_prefix, "text-generation set-auto-usage")
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
@@ -211,7 +208,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation context-management-enabled"
|
||||
)
|
||||
@@ -219,7 +216,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-context-management-enabled VALUE"
|
||||
)
|
||||
@@ -227,13 +224,51 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-context-management-enabled"
|
||||
)
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Thinking Notice
|
||||
|
||||
message.push_str(&format!(
|
||||
"#### {}",
|
||||
strings::help::cfg::text_generation_thinking_notice_heading()
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&strings::help::cfg::text_generation_thinking_notice_intro());
|
||||
message.push('\n');
|
||||
message.push_str(
|
||||
&strings::help::cfg::the_following_configuration_values_are_recognized(vec![true, false]),
|
||||
);
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation thinking-notice-enabled"
|
||||
)
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-thinking-notice-enabled VALUE"
|
||||
)
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-thinking-notice-enabled"
|
||||
)
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Sender Context
|
||||
|
||||
message.push_str(&format!(
|
||||
@@ -251,7 +286,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation sender-context-mode"
|
||||
)
|
||||
@@ -259,7 +294,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-sender-context-mode VALUE"
|
||||
)
|
||||
@@ -267,7 +302,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-sender-context-mode"
|
||||
)
|
||||
@@ -285,15 +320,12 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation prompt-override"
|
||||
)
|
||||
strings::help::cfg::current_setting_show(command_prefix, "text-generation prompt-override")
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-prompt-override VALUE"
|
||||
)
|
||||
@@ -301,7 +333,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-prompt-override"
|
||||
)
|
||||
@@ -319,7 +351,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-generation temperature-override"
|
||||
)
|
||||
@@ -327,7 +359,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-generation set-temperature-override VALUE"
|
||||
)
|
||||
@@ -335,7 +367,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-generation set-temperature-override"
|
||||
)
|
||||
@@ -372,12 +404,12 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(command_prefix, "speech-to-text flow-type")
|
||||
strings::help::cfg::current_setting_show(command_prefix, "speech-to-text flow-type")
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"speech-to-text set-flow-type VALUE"
|
||||
)
|
||||
@@ -385,7 +417,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-flow-type")
|
||||
strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-flow-type")
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
@@ -406,7 +438,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"speech-to-text msg-type-for-non-threaded-only-transcribed-messages"
|
||||
)
|
||||
@@ -414,7 +446,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages VALUE"
|
||||
)
|
||||
@@ -422,7 +454,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages"
|
||||
)
|
||||
@@ -440,12 +472,12 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(command_prefix, "speech-to-text language")
|
||||
strings::help::cfg::current_setting_show(command_prefix, "speech-to-text language")
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"speech-to-text set-language VALUE"
|
||||
)
|
||||
@@ -453,7 +485,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-language")
|
||||
strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-language")
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
@@ -488,7 +520,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-to-speech bot-msgs-flow-type"
|
||||
)
|
||||
@@ -496,7 +528,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-to-speech set-bot-msgs-flow-type VALUE"
|
||||
)
|
||||
@@ -504,7 +536,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-to-speech set-bot-msgs-flow-type"
|
||||
)
|
||||
@@ -528,7 +560,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"text-to-speech user-msgs-flow-type"
|
||||
)
|
||||
@@ -536,7 +568,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-to-speech set-user-msgs-flow-type VALUE"
|
||||
)
|
||||
@@ -544,7 +576,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-to-speech set-user-msgs-flow-type"
|
||||
)
|
||||
@@ -562,12 +594,12 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(command_prefix, "text-to-speech speed-override")
|
||||
strings::help::cfg::current_setting_show(command_prefix, "text-to-speech speed-override")
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-to-speech set-speed-override VALUE"
|
||||
)
|
||||
@@ -575,7 +607,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-to-speech set-speed-override"
|
||||
)
|
||||
@@ -593,12 +625,12 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(command_prefix, "text-to-speech voice-override")
|
||||
strings::help::cfg::current_setting_show(command_prefix, "text-to-speech voice-override")
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"text-to-speech set-voice-override VALUE"
|
||||
)
|
||||
@@ -606,7 +638,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"text-to-speech set-voice-override"
|
||||
)
|
||||
|
||||
@@ -359,6 +359,33 @@ async fn generate_text_generation_section(
|
||||
),
|
||||
);
|
||||
|
||||
// Thinking Notice
|
||||
|
||||
let effective_thinking_notice = room_config_context.text_generation_thinking_notice_enabled();
|
||||
let room_config_thinking_notice = room_config_context
|
||||
.room_config
|
||||
.settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled;
|
||||
let global_config_thinking_notice = room_config_context
|
||||
.global_config
|
||||
.fallback_room_settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled;
|
||||
|
||||
let thinking_notice_set_where = if room_config_thinking_notice.is_some() {
|
||||
strings::cfg::status_badge_set_in_room_config()
|
||||
} else if global_config_thinking_notice.is_some() {
|
||||
strings::cfg::status_badge_set_in_global_config()
|
||||
} else {
|
||||
strings::cfg::status_badge_using_hardcoded_default()
|
||||
};
|
||||
|
||||
message.push_str(&strings::cfg::status_text_generation_entry_thinking_notice(
|
||||
effective_thinking_notice,
|
||||
thinking_notice_set_where,
|
||||
));
|
||||
|
||||
// Sender Context
|
||||
|
||||
let effective_sender_context = room_config_context.text_generation_sender_context_mode();
|
||||
|
||||
@@ -30,6 +30,13 @@ use crate::{
|
||||
entity::MessageContext,
|
||||
};
|
||||
|
||||
/// How long text generation must run before the first "thinking…" placeholder appears.
|
||||
/// Fast responses (under this threshold) never get a placeholder.
|
||||
const THINKING_NOTICE_FIRST_DELAY: std::time::Duration = std::time::Duration::from_secs(3);
|
||||
|
||||
/// How often the "thinking…" placeholder is refreshed once it has appeared.
|
||||
const THINKING_NOTICE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(10);
|
||||
|
||||
#[derive(Debug, PartialEq)]
|
||||
pub enum ChatCompletionControllerType {
|
||||
// Invoked via a command prefix (e.g. `!bai Hello!`)
|
||||
@@ -140,9 +147,7 @@ pub async fn handle(
|
||||
// Let's proceed below where we potentially handle text-generation.
|
||||
}
|
||||
|
||||
let text_to_speech_stage_params: Option<TextToSpeechParams>;
|
||||
|
||||
if message_context
|
||||
let text_to_speech_stage_params: Option<TextToSpeechParams> = if message_context
|
||||
.room_config_context()
|
||||
.should_auto_text_generate(original_message_is_audio)
|
||||
{
|
||||
@@ -203,7 +208,7 @@ pub async fn handle(
|
||||
return Ok(());
|
||||
};
|
||||
|
||||
text_to_speech_stage_params = match message_context
|
||||
match message_context
|
||||
.room_config_context()
|
||||
.text_to_speech_bot_messages_flow_type()
|
||||
{
|
||||
@@ -236,7 +241,7 @@ pub async fn handle(
|
||||
text_to_speech_eligible_payload,
|
||||
response_type,
|
||||
)),
|
||||
};
|
||||
}
|
||||
} else {
|
||||
tracing::debug!("Not generating text due to auto-usage configuration");
|
||||
|
||||
@@ -255,7 +260,7 @@ pub async fn handle(
|
||||
event_id: message_context.event_id().clone(),
|
||||
};
|
||||
|
||||
text_to_speech_stage_params = match message_context
|
||||
match message_context
|
||||
.room_config_context()
|
||||
.text_to_speech_user_messages_flow_type()
|
||||
{
|
||||
@@ -268,8 +273,8 @@ pub async fn handle(
|
||||
text_to_speech_eligible_payload,
|
||||
response_type,
|
||||
)),
|
||||
};
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// We're potentially dealing with some text in text_to_speech_eligible_payload - either coming directly from the user or generated by an agent.
|
||||
|
||||
@@ -520,6 +525,16 @@ async fn handle_stage_text_generation(
|
||||
conversation.start_time(),
|
||||
);
|
||||
|
||||
// Cloned only when the thinking-notice is enabled; the original is moved into `params` below.
|
||||
let notice_prompt_variables = if message_context
|
||||
.room_config_context()
|
||||
.text_generation_thinking_notice_enabled()
|
||||
{
|
||||
Some(prompt_variables.clone())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let params = TextGenerationParams {
|
||||
context_management_enabled: message_context
|
||||
.room_config_context()
|
||||
@@ -536,10 +551,75 @@ async fn handle_stage_text_generation(
|
||||
prompt_variables,
|
||||
};
|
||||
|
||||
let result = controller
|
||||
.generate_text(conversation, params)
|
||||
.instrument(span)
|
||||
.await;
|
||||
// When the thinking-notice is enabled, race generation against a timer that posts and then
|
||||
// periodically edits a "thinking…" placeholder. `biased;` makes generation win a tie, and the
|
||||
// loop exits the instant generation resolves, so there is no detached task and no late edit can
|
||||
// ever clobber the real answer. `placeholder` is the event we must finalize in every exit path.
|
||||
let (result, placeholder) = if let Some(notice_prompt_variables) = notice_prompt_variables {
|
||||
let generation = controller
|
||||
.generate_text(conversation, params)
|
||||
.instrument(span);
|
||||
tokio::pin!(generation);
|
||||
|
||||
let mut placeholder: Option<OwnedEventId> = None;
|
||||
// Seed the flavor sequence per-generation so different turns don't all open on the same
|
||||
// line; the monotonic increment then guarantees consecutive notices differ.
|
||||
let mut notice_sequence: usize = std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|since| since.subsec_nanos() as usize)
|
||||
.unwrap_or(0);
|
||||
let mut tick = tokio::time::interval_at(
|
||||
tokio::time::Instant::now() + THINKING_NOTICE_FIRST_DELAY,
|
||||
THINKING_NOTICE_INTERVAL,
|
||||
);
|
||||
// If an edit runs long, hold ~INTERVAL spacing rather than bursting the missed ticks.
|
||||
tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
|
||||
|
||||
let result = loop {
|
||||
tokio::select! {
|
||||
biased;
|
||||
generation_result = &mut generation => break generation_result,
|
||||
_ = tick.tick() => {
|
||||
let message = notice_prompt_variables.format(
|
||||
strings::thinking::pick_message(start_time.elapsed(), notice_sequence),
|
||||
);
|
||||
notice_sequence = notice_sequence.wrapping_add(1);
|
||||
|
||||
match &placeholder {
|
||||
None => {
|
||||
placeholder = bot
|
||||
.messaging()
|
||||
.send_text_markdown_no_fail_quietly(
|
||||
message_context.room(),
|
||||
message,
|
||||
response_type.clone(),
|
||||
)
|
||||
.await
|
||||
.map(|response| response.event_id);
|
||||
}
|
||||
Some(event_id) => {
|
||||
bot.messaging()
|
||||
.edit_text_markdown_no_fail(
|
||||
message_context.room(),
|
||||
event_id,
|
||||
message,
|
||||
)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
(result, placeholder)
|
||||
} else {
|
||||
let result = controller
|
||||
.generate_text(conversation, params)
|
||||
.instrument(span)
|
||||
.await;
|
||||
|
||||
(result, None)
|
||||
};
|
||||
|
||||
let duration = std::time::Instant::now().duration_since(start_time);
|
||||
|
||||
@@ -560,17 +640,20 @@ async fn handle_stage_text_generation(
|
||||
err,
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::TextGeneration,
|
||||
&err,
|
||||
),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
let error_text = strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::TextGeneration,
|
||||
&err,
|
||||
);
|
||||
|
||||
finalize_thinking_notice_with_error(
|
||||
bot,
|
||||
message_context,
|
||||
placeholder.as_ref(),
|
||||
&error_text,
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return None;
|
||||
}
|
||||
@@ -583,26 +666,70 @@ async fn handle_stage_text_generation(
|
||||
"Agent returned empty text",
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::empty_response_returned(agent.identifier()),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
let empty_text = strings::agent::empty_response_returned(agent.identifier());
|
||||
|
||||
finalize_thinking_notice_with_error(
|
||||
bot,
|
||||
message_context,
|
||||
placeholder.as_ref(),
|
||||
&empty_text,
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return None;
|
||||
}
|
||||
|
||||
let send_message_response = bot
|
||||
.messaging()
|
||||
.send_text_markdown_no_fail(message_context.room(), text.clone(), response_type)
|
||||
.await?;
|
||||
// Finalize the answer into a single message. With a placeholder, edit it in place (so the
|
||||
// "thinking…" message becomes the answer); the TTS payload then points at that same event.
|
||||
// If the edit fails, fall back to a fresh send so the real answer is never lost.
|
||||
let event_id = match &placeholder {
|
||||
Some(event_id)
|
||||
if bot
|
||||
.messaging()
|
||||
.edit_text_markdown_no_fail(message_context.room(), event_id, text.clone())
|
||||
.await
|
||||
.is_some() =>
|
||||
{
|
||||
event_id.clone()
|
||||
}
|
||||
_ => {
|
||||
bot.messaging()
|
||||
.send_text_markdown_no_fail(message_context.room(), text.clone(), response_type)
|
||||
.await?
|
||||
.event_id
|
||||
}
|
||||
};
|
||||
|
||||
Some(TextToSpeechEligiblePayload {
|
||||
text,
|
||||
event_id: send_message_response.event_id,
|
||||
})
|
||||
Some(TextToSpeechEligiblePayload { text, event_id })
|
||||
}
|
||||
|
||||
/// Finalizes a thinking-notice placeholder (if one was posted) with error/notice text, so a
|
||||
/// failed or empty generation never leaves an orphaned "thinking…" message behind. With no
|
||||
/// placeholder, this is the original behavior: a fresh error notice.
|
||||
async fn finalize_thinking_notice_with_error(
|
||||
bot: &Bot,
|
||||
message_context: &MessageContext,
|
||||
placeholder: Option<&OwnedEventId>,
|
||||
text: &str,
|
||||
response_type: MessageResponseType,
|
||||
) {
|
||||
match placeholder {
|
||||
Some(event_id) => {
|
||||
bot.messaging()
|
||||
.edit_text_markdown_no_fail(
|
||||
message_context.room(),
|
||||
event_id,
|
||||
crate::utils::status::create_error_message_text(text),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
None => {
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(message_context.room(), text, response_type)
|
||||
.await;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn handle_stage_speech_to_text_actual_transcribing(
|
||||
|
||||
@@ -6,5 +6,5 @@ mod utils;
|
||||
mod tests;
|
||||
|
||||
pub use entity::*;
|
||||
pub use tokenization::shorten_messages_list_to_context_size;
|
||||
pub use tokenization::{TokenEstimate, shorten_messages_list_to_context_size};
|
||||
pub use utils::*;
|
||||
|
||||
@@ -4,6 +4,22 @@ use tiktoken_rs::tokenizer;
|
||||
|
||||
use super::{Author, Message, MessageContent};
|
||||
|
||||
/// How to count the tokens in a conversation when trimming it to fit the context window.
|
||||
pub enum TokenEstimate<'a> {
|
||||
/// Count via the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library.
|
||||
/// Accurate for OpenAI models; every other model falls back to the gpt-4
|
||||
/// tokenizer, which misreads it (badly so for non-English text). Use this only
|
||||
/// for the OpenAI provider.
|
||||
Tiktoken(&'a str),
|
||||
|
||||
/// Provider-neutral approximation that needs no per-model tokenizer. Expect it
|
||||
/// to land within roughly 10-20% of the real count for typical text, leaning
|
||||
/// slightly high: over-counting trims a little extra history, while
|
||||
/// under-counting would overflow the model's real context window. Use this for
|
||||
/// every non-OpenAI provider.
|
||||
Approximate,
|
||||
}
|
||||
|
||||
fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
||||
let tokenizer = tokenizer::get_tokenizer(model)
|
||||
.or_else(|| tokenizer::get_tokenizer("gpt-4"))
|
||||
@@ -13,21 +29,27 @@ fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
||||
}
|
||||
|
||||
pub fn shorten_messages_list_to_context_size(
|
||||
model: &str,
|
||||
estimate: TokenEstimate<'_>,
|
||||
prompt_message: &Option<Message>,
|
||||
mut messages: Vec<Message>,
|
||||
max_response_tokens: Option<u32>,
|
||||
max_context_tokens: u32,
|
||||
) -> Vec<Message> {
|
||||
// Loading the tokenization data is an expensive process, so
|
||||
// se construct the BPE instance once and then use it for all messages.
|
||||
let bpe = get_bpe_for_model(model);
|
||||
// Loading the tiktoken data is expensive, so we resolve the counter once up
|
||||
// front and reuse it for every message.
|
||||
let tiktoken = match estimate {
|
||||
TokenEstimate::Tiktoken(model) => Some((get_bpe_for_model(model), model)),
|
||||
TokenEstimate::Approximate => None,
|
||||
};
|
||||
let count = |message: &Message| match tiktoken {
|
||||
Some((bpe, model)) => tiktoken_token_size_for_message(bpe, model, message),
|
||||
None => approximate_token_size_for_message(message),
|
||||
};
|
||||
|
||||
// We want to retain the prompt in all cases, so we always count it first.
|
||||
// We also always reserve enough tokens for the maximum response we expect.
|
||||
let mut current_context_length: u32 = if let Some(prompt_message) = prompt_message {
|
||||
calculate_token_size_for_message(bpe, model, prompt_message)
|
||||
+ max_response_tokens.unwrap_or(0)
|
||||
count(prompt_message) + max_response_tokens.unwrap_or(0)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
@@ -37,7 +59,7 @@ pub fn shorten_messages_list_to_context_size(
|
||||
let mut messages_to_keep: Vec<Message> = Vec::new();
|
||||
|
||||
for message in messages {
|
||||
let tokens_for_message = calculate_token_size_for_message(bpe, model, &message);
|
||||
let tokens_for_message = count(&message);
|
||||
|
||||
if current_context_length + tokens_for_message > max_context_tokens {
|
||||
break;
|
||||
@@ -48,14 +70,26 @@ pub fn shorten_messages_list_to_context_size(
|
||||
messages_to_keep.push(message);
|
||||
}
|
||||
|
||||
// Cut on a turn boundary: the loop may stop right after an assistant reply
|
||||
// whose triggering user message did not fit, which would leave the kept window
|
||||
// starting on an orphaned reply. `messages_to_keep` is newest-first here, so
|
||||
// the oldest kept messages are at the end; drop any trailing assistant messages
|
||||
// until the window begins at the start of a turn (a user message).
|
||||
while matches!(
|
||||
messages_to_keep.last().map(|message| &message.author),
|
||||
Some(Author::Assistant)
|
||||
) {
|
||||
messages_to_keep.pop();
|
||||
}
|
||||
|
||||
messages_to_keep.reverse();
|
||||
|
||||
messages_to_keep
|
||||
}
|
||||
|
||||
/// Calculate the token size of a message for a given model, with a preloaded CoreBPE object.
|
||||
/// Related to `calculate_token_size_for_model_message`.
|
||||
fn calculate_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Message) -> u32 {
|
||||
/// Token size of a message via tiktoken, for a preloaded CoreBPE object.
|
||||
/// Accurate only for OpenAI models (see [`TokenEstimate::Tiktoken`]).
|
||||
fn tiktoken_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Message) -> u32 {
|
||||
let (tokens_per_message, tokens_per_name) = if model.starts_with("gpt-3.5") {
|
||||
(
|
||||
4, // every message follows <im_start>{role/name}\n{content}<im_end>\n
|
||||
@@ -80,6 +114,52 @@ fn calculate_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Messag
|
||||
(text_length + role_length + tokens_per_message + tokens_per_name) as u32
|
||||
}
|
||||
|
||||
/// ASCII text averages about four characters per token.
|
||||
const ASCII_TOKENS_PER_CHAR: f32 = 0.25;
|
||||
|
||||
/// Non-ASCII scripts (Cyrillic, CJK, and others) pack more information per
|
||||
/// character: real tokenizers land around two characters per token for them, so
|
||||
/// each counts as half a token. CJK runs a touch denser than that, so its estimate
|
||||
/// can read slightly low, still within the tolerance this approximation targets.
|
||||
const WIDE_TOKENS_PER_CHAR: f32 = 0.5;
|
||||
|
||||
/// Structural per-message overhead (role marker plus message framing), mirroring
|
||||
/// the small constant the tiktoken path adds.
|
||||
const APPROX_TOKENS_PER_MESSAGE: u32 = 4;
|
||||
|
||||
/// Provider-neutral, tokenizer-free token size of a message
|
||||
/// (see [`TokenEstimate::Approximate`]).
|
||||
fn approximate_token_size_for_message(message: &Message) -> u32 {
|
||||
let text_tokens = match &message.content {
|
||||
MessageContent::Text(text) => approximate_token_size_for_text(text),
|
||||
// Images and files are not counted as text, matching the tiktoken path.
|
||||
MessageContent::Image(..) | MessageContent::File(..) => 0,
|
||||
};
|
||||
|
||||
text_tokens + APPROX_TOKENS_PER_MESSAGE
|
||||
}
|
||||
|
||||
/// Rough token estimate for a piece of text, with no tokenizer.
|
||||
///
|
||||
/// ASCII characters count as a quarter-token each (~4 chars/token); characters
|
||||
/// outside ASCII count as half a token each (~2 chars/token), matching how real
|
||||
/// tokenizers treat Cyrillic and CJK. Weighting non-ASCII up keeps the estimate
|
||||
/// from badly under-counting non-English text, the case the tiktoken fallback gets
|
||||
/// most wrong.
|
||||
fn approximate_token_size_for_text(text: &str) -> u32 {
|
||||
let mut estimate = 0.0_f32;
|
||||
|
||||
for character in text.chars() {
|
||||
estimate += if character.is_ascii() {
|
||||
ASCII_TOKENS_PER_CHAR
|
||||
} else {
|
||||
WIDE_TOKENS_PER_CHAR
|
||||
};
|
||||
}
|
||||
|
||||
estimate.ceil() as u32
|
||||
}
|
||||
|
||||
pub mod test {
|
||||
#[test]
|
||||
fn message_size_counting_works() {
|
||||
@@ -94,7 +174,7 @@ pub mod test {
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let tokens = super::calculate_token_size_for_message(bpe, model, &message);
|
||||
let tokens = super::tiktoken_token_size_for_message(bpe, model, &message);
|
||||
|
||||
assert_eq!(8, tokens);
|
||||
}
|
||||
@@ -118,7 +198,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
prompt_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &prompt)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &prompt)
|
||||
);
|
||||
|
||||
let mut conversation_messages = Vec::new();
|
||||
@@ -133,7 +213,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
first_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &first)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &first)
|
||||
);
|
||||
|
||||
conversation_messages.push(first);
|
||||
@@ -148,7 +228,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
second_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &second)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &second)
|
||||
);
|
||||
|
||||
conversation_messages.push(second);
|
||||
@@ -165,7 +245,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
third_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &third)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &third)
|
||||
);
|
||||
|
||||
conversation_messages.push(third.clone());
|
||||
@@ -182,7 +262,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
forth_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &forth)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &forth)
|
||||
);
|
||||
|
||||
conversation_messages.push(forth.clone());
|
||||
@@ -190,7 +270,7 @@ pub mod test {
|
||||
assert_eq!(4, conversation_messages.len());
|
||||
|
||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||
model,
|
||||
super::TokenEstimate::Tiktoken(model),
|
||||
&Some(prompt),
|
||||
conversation_messages,
|
||||
max_response_tokens,
|
||||
@@ -229,7 +309,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
prompt_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &prompt)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &prompt)
|
||||
);
|
||||
|
||||
let mut conversation_messages = Vec::new();
|
||||
@@ -244,7 +324,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
first_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &first)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &first)
|
||||
);
|
||||
|
||||
conversation_messages.push(first);
|
||||
@@ -259,7 +339,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
second_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &second)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &second)
|
||||
);
|
||||
|
||||
conversation_messages.push(second);
|
||||
@@ -276,7 +356,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
third_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &third)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &third)
|
||||
);
|
||||
|
||||
conversation_messages.push(third.clone());
|
||||
@@ -293,7 +373,7 @@ pub mod test {
|
||||
|
||||
assert_eq!(
|
||||
forth_length,
|
||||
super::calculate_token_size_for_message(bpe, model, &forth)
|
||||
super::tiktoken_token_size_for_message(bpe, model, &forth)
|
||||
);
|
||||
|
||||
conversation_messages.push(forth.clone());
|
||||
@@ -301,7 +381,7 @@ pub mod test {
|
||||
assert_eq!(4, conversation_messages.len());
|
||||
|
||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||
model,
|
||||
super::TokenEstimate::Tiktoken(model),
|
||||
&Some(prompt),
|
||||
conversation_messages,
|
||||
max_response_tokens,
|
||||
@@ -320,4 +400,126 @@ pub mod test {
|
||||
forth.content
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn approximate_counting_weights_ascii_and_wide_scripts() {
|
||||
// 12 ASCII characters at ~4 chars/token = 3 text tokens.
|
||||
assert_eq!(3, super::approximate_token_size_for_text("Hello there!"));
|
||||
|
||||
// 5 CJK characters at ~0.5 token/char = 3 text tokens. The ASCII rate would
|
||||
// have under-counted these to 2, the failure mode this path avoids.
|
||||
assert_eq!(3, super::approximate_token_size_for_text("こんにちは"));
|
||||
|
||||
let message = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("Hello there!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
// 3 text tokens plus the per-message overhead (4).
|
||||
assert_eq!(7, super::approximate_token_size_for_message(&message));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn approximate_shortening_trims_to_budget() {
|
||||
let prompt = super::Message {
|
||||
author: super::Author::Prompt,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("You are a bot!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let older = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("This is the older message.".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let newer = super::Message {
|
||||
// A user message, so it is a valid window start: keeping a lone
|
||||
// assistant reply would be an orphan and get trimmed (see
|
||||
// `shortening_cuts_on_a_turn_boundary`).
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("This is the newer message.".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
// Budget room for the prompt and only the newest message.
|
||||
let max_context_tokens = super::approximate_token_size_for_message(&prompt)
|
||||
+ super::approximate_token_size_for_message(&newer);
|
||||
|
||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||
super::TokenEstimate::Approximate,
|
||||
&Some(prompt),
|
||||
vec![older, newer.clone()],
|
||||
None,
|
||||
max_context_tokens,
|
||||
);
|
||||
|
||||
assert_eq!(1, new_conversation_messages.len());
|
||||
assert_eq!(
|
||||
new_conversation_messages.first().unwrap().content,
|
||||
newer.content
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shortening_cuts_on_a_turn_boundary() {
|
||||
// A four-message conversation of two full turns. All four messages are the
|
||||
// same length, so they cost the same number of tokens.
|
||||
let prompt = super::Message {
|
||||
author: super::Author::Prompt,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("system".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let user_one = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("user msg 1".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let asst_one = super::Message {
|
||||
author: super::Author::Assistant,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("asst msg 1".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let user_two = super::Message {
|
||||
author: super::Author::User,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("user msg 2".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
let asst_two = super::Message {
|
||||
author: super::Author::Assistant,
|
||||
sender_id: None,
|
||||
content: super::MessageContent::Text("asst msg 2".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
};
|
||||
|
||||
let per_message = super::approximate_token_size_for_message(&user_one);
|
||||
// Budget fits the prompt plus three messages. By raw token budget the loop
|
||||
// would keep asst_two, user_two, and asst_one, but asst_one's own user
|
||||
// message (user_one) does not fit, so it must be dropped too rather than
|
||||
// left as an orphaned reply.
|
||||
let max_context_tokens =
|
||||
super::approximate_token_size_for_message(&prompt) + (per_message * 3);
|
||||
|
||||
let kept = super::shorten_messages_list_to_context_size(
|
||||
super::TokenEstimate::Approximate,
|
||||
&Some(prompt),
|
||||
vec![user_one, asst_one, user_two.clone(), asst_two.clone()],
|
||||
None,
|
||||
max_context_tokens,
|
||||
);
|
||||
|
||||
// Only the last whole turn survives; the orphaned asst_one is dropped.
|
||||
assert_eq!(2, kept.len());
|
||||
assert_eq!(kept.first().unwrap().content, user_two.content);
|
||||
assert_eq!(kept.last().unwrap().content, asst_two.content);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -136,6 +136,20 @@ impl RoomConfigContext {
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
pub fn text_generation_thinking_notice_enabled(&self) -> bool {
|
||||
self.room_config
|
||||
.settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled
|
||||
.or({
|
||||
self.global_config
|
||||
.fallback_room_settings
|
||||
.text_generation
|
||||
.thinking_notice_enabled
|
||||
})
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
pub fn text_generation_sender_context_mode(&self) -> TextGenerationSenderContextMode {
|
||||
self.room_config
|
||||
.settings
|
||||
|
||||
@@ -14,6 +14,10 @@ pub struct RoomSettingsTextGeneration {
|
||||
/// When enabled, the bot will automatically tokenize messages and try to shorten the message context intelligently.
|
||||
pub context_management_enabled: Option<bool>,
|
||||
|
||||
/// Controls whether a "thinking…" notice is posted while text generation runs longer than a threshold.
|
||||
/// When enabled, a placeholder message appears for slow responses and is edited in place (with elapsed-tiered flavor text) until it becomes the final answer.
|
||||
pub thinking_notice_enabled: Option<bool>,
|
||||
|
||||
/// Controls how each message in the conversation context is annotated with sender metadata.
|
||||
pub sender_context_mode: Option<TextGenerationSenderContextMode>,
|
||||
|
||||
|
||||
@@ -101,13 +101,13 @@ pub fn post_creation_helpful_commands(
|
||||
for purpose in supported_purposes {
|
||||
message.push_str(&format!(
|
||||
"\n- {}",
|
||||
&set_as_purpose_handler_in_room(agent_identifier, purpose, command_prefix,)
|
||||
set_as_purpose_handler_in_room(agent_identifier, purpose, command_prefix,)
|
||||
));
|
||||
|
||||
if !is_room_local {
|
||||
message.push_str(&format!(
|
||||
"\n- {}",
|
||||
&set_as_purpose_handler_globally(agent_identifier, purpose, command_prefix,)
|
||||
set_as_purpose_handler_globally(agent_identifier, purpose, command_prefix,)
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -249,6 +249,10 @@ pub fn status_text_generation_entry_context_management(value: bool, set_where: &
|
||||
format!("- ♻️ Context management: `{}` ({})\n", value, set_where)
|
||||
}
|
||||
|
||||
pub fn status_text_generation_entry_thinking_notice(value: bool, set_where: &str) -> String {
|
||||
format!("- 💭 Thinking notice: `{}` ({})\n", value, set_where)
|
||||
}
|
||||
|
||||
pub fn status_text_generation_entry_sender_context(
|
||||
value: impl std::fmt::Display,
|
||||
set_where: &str,
|
||||
|
||||
@@ -128,7 +128,19 @@ pub fn text_generation_context_management_intro() -> String {
|
||||
format!(
|
||||
"{}\n{}",
|
||||
"Controls the bot's ability to **intelligently drop old messages from the conversation context** when it gets too large.",
|
||||
"This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](./providers.md#openai) models.",
|
||||
"Counting tokens precisely needs the model's own tokenizer. For [OpenAI](./providers.md#openai) models the bot uses the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library; for every other provider (including the recommended [Venice](./providers.md#venice)) it falls back to a provider-neutral **approximation** (ASCII counted at ~4 characters per token, other scripts at ~2), within roughly 10-20% of the real count for typical text.",
|
||||
)
|
||||
}
|
||||
|
||||
pub fn text_generation_thinking_notice_heading() -> &'static str {
|
||||
"💭 Thinking Notice"
|
||||
}
|
||||
|
||||
pub fn text_generation_thinking_notice_intro() -> String {
|
||||
format!(
|
||||
"{}\n{}",
|
||||
"Controls whether the bot posts a **\"thinking…\" notice** while text generation takes a while, so slow responses (for example, from reasoning models that run for minutes) don't look stuck.",
|
||||
"When enabled, a placeholder message appears only after a short delay, updates periodically with varying status text, and is then edited in place to become the final answer. Disabled by default.",
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ pub mod provider;
|
||||
pub mod room_config;
|
||||
pub mod speech_to_text;
|
||||
pub mod text_to_speech;
|
||||
pub mod thinking;
|
||||
pub mod usage;
|
||||
|
||||
pub const PROGRESS_INDICATOR_EMOJI: &str = "⏳";
|
||||
|
||||
146
src/strings/thinking.rs
Normal file
146
src/strings/thinking.rs
Normal file
@@ -0,0 +1,146 @@
|
||||
use std::time::Duration;
|
||||
|
||||
/// After this much elapsed generation time, the notice escalates to the "medium" pool.
|
||||
const TIER_MEDIUM_AFTER: Duration = Duration::from_secs(30);
|
||||
|
||||
/// After this much elapsed generation time, the notice escalates to the "deep" pool.
|
||||
const TIER_DEEP_AFTER: Duration = Duration::from_secs(90);
|
||||
|
||||
/// Early flavor: the response is just taking a moment longer than instant.
|
||||
const MESSAGES_LIGHT: &[&str] = &[
|
||||
"*{{ baibot_name }} is thinking…*",
|
||||
"*{{ baibot_name }} is mulling that over…*",
|
||||
"*Hmm, let me think about that…*",
|
||||
"*{{ baibot_name }} is gathering some thoughts…*",
|
||||
"*One moment, {{ baibot_name }} is working on it…*",
|
||||
"*{{ baibot_name }} is putting the pieces together…*",
|
||||
"*Give {{ baibot_name }} a second here…*",
|
||||
"*Let me think this one through…*",
|
||||
"*{{ baibot_name }} is warming up the gears…*",
|
||||
"*{{ baibot_name }} is on it…*",
|
||||
];
|
||||
|
||||
/// Mid flavor: this is a real question and the model is genuinely working.
|
||||
const MESSAGES_MEDIUM: &[&str] = &[
|
||||
"*Huh, good one. {{ baibot_name }} is really thinking now…*",
|
||||
"*Still working on it, {{ baibot_name }} wants to get this right…*",
|
||||
"*This one needs a bit more thought…*",
|
||||
"*{{ baibot_name }} is digging into this…*",
|
||||
"*Hang tight, {{ baibot_name }} is turning it over…*",
|
||||
"*Not a quick one, this. {{ baibot_name }} is still at it…*",
|
||||
"*{{ baibot_name }} is chewing on this properly now…*",
|
||||
"*This deserves some real thought, bear with {{ baibot_name }}…*",
|
||||
"*{{ baibot_name }} is working through the details…*",
|
||||
"*Won't be long, {{ baibot_name }} is closing in on it…*",
|
||||
];
|
||||
|
||||
/// Deep flavor: a long-running generation (e.g. a reasoning model going for minutes).
|
||||
const MESSAGES_DEEP: &[&str] = &[
|
||||
"*Okay, this is a hard one. {{ baibot_name }} is really deep in thought…*",
|
||||
"*{{ baibot_name }} is in the weeds on this one, thanks for your patience…*",
|
||||
"*Still here, still thinking. {{ baibot_name }} doesn't want to rush it…*",
|
||||
"*A proper puzzle, this. {{ baibot_name }} is taking the time to do it justice…*",
|
||||
"*{{ baibot_name }} is really wrestling with this one…*",
|
||||
"*A meaty question. {{ baibot_name }} is still turning it over…*",
|
||||
"*{{ baibot_name }} hasn't forgotten you, just thinking hard…*",
|
||||
"*Almost there, {{ baibot_name }} is pulling it all together…*",
|
||||
"*{{ baibot_name }} is going the extra mile on this one…*",
|
||||
"*Deep thoughts in progress. {{ baibot_name }} appreciates your patience…*",
|
||||
];
|
||||
|
||||
/// Returns the flavor pool matching how long generation has been running.
|
||||
pub fn messages_for_elapsed(elapsed: Duration) -> &'static [&'static str] {
|
||||
if elapsed >= TIER_DEEP_AFTER {
|
||||
MESSAGES_DEEP
|
||||
} else if elapsed >= TIER_MEDIUM_AFTER {
|
||||
MESSAGES_MEDIUM
|
||||
} else {
|
||||
MESSAGES_LIGHT
|
||||
}
|
||||
}
|
||||
|
||||
/// Picks one (still-untemplated) flavor message from the tier active at `elapsed`,
|
||||
/// indexed by a monotonic `sequence` the caller increments once per notice.
|
||||
///
|
||||
/// Using a monotonic counter (rather than a clock-derived value) guarantees two
|
||||
/// things the gesture-novelty goal needs: consecutive notices never repeat a line
|
||||
/// (the index advances by one each tick, so it differs whenever the tier has more
|
||||
/// than one message), and the pool is fully walked before any line recurs. The
|
||||
/// caller seeds `sequence` with a per-generation value so different turns don't all
|
||||
/// open on the same line. A clock-derived index can't promise this: `interval_at`
|
||||
/// ticks on a near-fixed schedule, so its sub-second component clusters and would
|
||||
/// re-pick the same line.
|
||||
pub fn pick_message(elapsed: Duration, sequence: usize) -> &'static str {
|
||||
let pool = messages_for_elapsed(elapsed);
|
||||
pool[sequence % pool.len()]
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn messages_for_elapsed_returns_the_right_tier() {
|
||||
// Boundaries: light below 30s, medium [30s, 90s), deep at/after 90s.
|
||||
// The pools have distinct content, so value comparison identifies the tier.
|
||||
assert_eq!(messages_for_elapsed(Duration::from_secs(0)), MESSAGES_LIGHT);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(29)),
|
||||
MESSAGES_LIGHT
|
||||
);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(30)),
|
||||
MESSAGES_MEDIUM
|
||||
);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(89)),
|
||||
MESSAGES_MEDIUM
|
||||
);
|
||||
assert_eq!(messages_for_elapsed(Duration::from_secs(90)), MESSAGES_DEEP);
|
||||
assert_eq!(
|
||||
messages_for_elapsed(Duration::from_secs(600)),
|
||||
MESSAGES_DEEP
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_tier_is_non_empty() {
|
||||
for pool in [MESSAGES_LIGHT, MESSAGES_MEDIUM, MESSAGES_DEEP] {
|
||||
assert!(!pool.is_empty());
|
||||
for message in pool {
|
||||
assert!(!message.trim().is_empty());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_tier_uses_the_bot_name_template() {
|
||||
// Not every line names the bot (some are first-person flavor), but each
|
||||
// tier exercises the template var so substitution is wired through.
|
||||
for pool in [MESSAGES_LIGHT, MESSAGES_MEDIUM, MESSAGES_DEEP] {
|
||||
assert!(pool.iter().any(|m| m.contains("{{ baibot_name }}")));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pick_message_stays_within_the_active_tier() {
|
||||
let deep = messages_for_elapsed(Duration::from_secs(120));
|
||||
for sequence in 0..50 {
|
||||
assert!(deep.contains(&pick_message(Duration::from_secs(120), sequence)));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pick_message_never_repeats_on_consecutive_sequences() {
|
||||
// The gesture-novelty guarantee: a monotonic sequence must not pick the same line twice
|
||||
// in a row (and walks the whole tier before any line recurs).
|
||||
let elapsed = Duration::from_secs(0);
|
||||
let pool_len = messages_for_elapsed(elapsed).len();
|
||||
for sequence in 0..(pool_len * 3) {
|
||||
assert_ne!(
|
||||
pick_message(elapsed, sequence),
|
||||
pick_message(elapsed, sequence + 1),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user