Compare commits
82 Commits
provider-n
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
072fe3ec9f | ||
|
|
0d7ae70d1d | ||
|
|
f912162c4d | ||
|
|
264f71d885 | ||
|
|
2861ea6141 | ||
|
|
2014e25f77 | ||
|
|
b5c5fdb357 | ||
|
|
5ee8d18439 | ||
|
|
194a47bfc6 | ||
|
|
4f3a753f52 | ||
|
|
1a7b194588 | ||
|
|
616607433f | ||
|
|
f1043e03de | ||
|
|
6f3132edfd | ||
|
|
4f86a321e2 | ||
|
|
9d876a58a8 | ||
|
|
dfb2288fa3 | ||
|
|
24d99627ae | ||
|
|
f7c88fe532 | ||
|
|
f25d50c7f2 | ||
|
|
f22ac1636f | ||
|
|
43904e5032 | ||
|
|
4e29d2200f | ||
|
|
30fe2a8a8c | ||
|
|
e59e88f12e | ||
|
|
f27d8cfdcf | ||
|
|
909c5215d0 | ||
|
|
ca1147c10a | ||
|
|
4a4e6c1090 | ||
|
|
ea3af62afc | ||
|
|
ee7ffc8970 | ||
|
|
45f5b7a1b0 | ||
|
|
6d0396270e | ||
|
|
d52a5ea0ac | ||
|
|
30985baf0c | ||
|
|
077a86e5c9 | ||
|
|
9b3d5400ff | ||
|
|
721abe8c75 | ||
|
|
dae5fec4c3 | ||
|
|
8b3644f3db | ||
|
|
570a4be83c | ||
|
|
24266a4886 | ||
|
|
be94441ddb | ||
|
|
a4280eb8c0 | ||
|
|
9ad9a04040 | ||
|
|
57ca095df0 | ||
|
|
4b5cc24513 | ||
|
|
53daa06151 | ||
|
|
77103e2523 | ||
|
|
0a755d287e | ||
|
|
fe3296594b | ||
|
|
1e7f557334 | ||
|
|
be09ccf5da | ||
|
|
fa3646774a | ||
|
|
1774bf29eb | ||
|
|
2da4e6793a | ||
|
|
166c1f7659 | ||
|
|
0f0868b795 | ||
|
|
65d24e4681 | ||
|
|
ec83b38a1a | ||
|
|
ea13dd42d7 | ||
|
|
b405859db7 | ||
|
|
23a09cdf60 | ||
|
|
dd13deff33 | ||
|
|
0c109d4f92 | ||
|
|
e346951395 | ||
|
|
eee3c05e86 | ||
|
|
4bf44665a5 | ||
|
|
eae0851328 | ||
|
|
5212d6643e | ||
|
|
b4a070f536 | ||
|
|
de2b24f91b | ||
|
|
fbe956c0b7 | ||
|
|
a64b7da7ae | ||
|
|
ac5e776345 | ||
|
|
bea27df2e0 | ||
|
|
b7c1b6f48f | ||
|
|
8a5a38f241 | ||
|
|
5306d3bb8c | ||
|
|
75631982e7 | ||
|
|
6058bf733b | ||
|
|
a445933c0b |
62
.github/workflows/ci.yml
vendored
62
.github/workflows/ci.yml
vendored
@@ -22,7 +22,7 @@ jobs:
|
|||||||
|
|
||||||
# Toolchain version + components come from rust-toolchain.toml. rustflags is
|
# Toolchain version + components come from rust-toolchain.toml. rustflags is
|
||||||
# cleared so plain builds don't fail on warnings; the clippy hook still does.
|
# cleared so plain builds don't fail on warnings; the clippy hook still does.
|
||||||
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
- uses: actions-rust-lang/setup-rust-toolchain@v2
|
||||||
with:
|
with:
|
||||||
rustflags: ''
|
rustflags: ''
|
||||||
|
|
||||||
@@ -42,3 +42,63 @@ jobs:
|
|||||||
|
|
||||||
- name: Unit tests
|
- name: Unit tests
|
||||||
run: just test
|
run: just test
|
||||||
|
|
||||||
|
# The prek job never builds a container image, so a bump of the Dockerfile's
|
||||||
|
# base image reaches main unvalidated and fails later, in Publish, after
|
||||||
|
# ghcr.io/etkecc/baibot:latest has already been attempted. These two jobs close
|
||||||
|
# that gap: decide whether a Dockerfile changed, and if so build the image the
|
||||||
|
# way Publish does - but without pushing anything.
|
||||||
|
#
|
||||||
|
# The build is gated rather than unconditional because it is a full Rust
|
||||||
|
# release build; running it on every push would turn a ~1 minute pipeline into
|
||||||
|
# a ~10 minute one for changes that cannot affect the image.
|
||||||
|
docker-gate:
|
||||||
|
name: Decide whether the image needs building
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
outputs:
|
||||||
|
build: ${{ steps.decide.outputs.build }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
|
- name: Look for Dockerfile changes against main
|
||||||
|
id: decide
|
||||||
|
run: |
|
||||||
|
if [ "${{ github.event_name }}" = 'workflow_dispatch' ]; then
|
||||||
|
echo 'Forced via workflow_dispatch.'
|
||||||
|
echo 'build=true' >> "$GITHUB_OUTPUT"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Publish builds and pushes from main, so a main-side build here would
|
||||||
|
# be redundant. This gate exists for branches, before they merge.
|
||||||
|
if [ "${{ github.ref_name }}" = 'main' ]; then
|
||||||
|
echo 'On main; Publish covers this.'
|
||||||
|
echo 'build=false' >> "$GITHUB_OUTPUT"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
git fetch --no-tags origin main
|
||||||
|
if git diff --name-only origin/main HEAD -- Dockerfile | grep -q .; then
|
||||||
|
echo 'A Dockerfile changed; the image will be built.'
|
||||||
|
echo 'build=true' >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo 'No Dockerfile changed.'
|
||||||
|
echo 'build=false' >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
|
docker-build:
|
||||||
|
name: Build the container image (without publishing it)
|
||||||
|
needs: docker-gate
|
||||||
|
if: needs.docker-gate.outputs.build == 'true'
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
|
# No build cache on purpose: a bump of the base image is exactly the case
|
||||||
|
# where a cold build is the honest test.
|
||||||
|
- name: Build
|
||||||
|
uses: docker/build-push-action@v7
|
||||||
|
with:
|
||||||
|
push: false
|
||||||
|
|||||||
@@ -5,6 +5,10 @@ repos:
|
|||||||
- id: trailing-whitespace
|
- id: trailing-whitespace
|
||||||
- id: end-of-file-fixer
|
- id: end-of-file-fixer
|
||||||
- id: check-yaml
|
- id: check-yaml
|
||||||
|
# This is the stock Synapse sample homeserver.yaml (mostly commented-out
|
||||||
|
# docs). prek's stricter YAML parser (serde-saphyr, since v0.4.6) rejects
|
||||||
|
# its long runs of consecutive comment lines in several places.
|
||||||
|
exclude: '^etc/services/synapse/config/homeserver\.yaml$'
|
||||||
- id: check-merge-conflict
|
- id: check-merge-conflict
|
||||||
- id: check-added-large-files
|
- id: check-added-large-files
|
||||||
args: ['--maxkb=1024']
|
args: ['--maxkb=1024']
|
||||||
|
|||||||
@@ -1,3 +1,10 @@
|
|||||||
|
# (2026-06-29) Version 1.25.0
|
||||||
|
|
||||||
|
- (**Feature**) [♻️ Context management](./docs/configuration/text-generation.md#️-context-management) now works with every provider, not only [OpenAI](./docs/providers.md#openai). Token counting previously went through [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs), which is accurate only for OpenAI models and silently mis-counted everything else (worst of all for non-English text). OpenAI agents keep using tiktoken-rs; every other provider, including the recommended [Venice](./docs/providers.md#venice), now uses a provider-neutral approximation that needs no per-model tokenizer (ASCII counted at about four characters per token, other scripts such as Cyrillic and CJK at about two), landing within roughly 10-20% of the real count. See the [context management docs](./docs/configuration/text-generation.md#️-context-management).
|
||||||
|
|
||||||
|
- (**Improvement**) Context management now trims a conversation on whole-turn boundaries for every provider, so an assistant reply is never kept without the user message it answered. This also adjusts how the OpenAI provider trims: a dangling assistant reply at the oldest edge of the kept history is now dropped along with its missing prompt, rather than left in place.
|
||||||
|
|
||||||
|
|
||||||
# (2026-06-26) Version 1.24.0
|
# (2026-06-26) Version 1.24.0
|
||||||
|
|
||||||
- (**Feature**) Add an opt-in 💭 **thinking notice** for text generation. When enabled, a slow response (for example, from a reasoning model that runs for minutes) posts a "thinking…" placeholder after a short delay, refreshes it periodically with varying flavor text, and then edits that same message into the final answer, so a long wait no longer looks like a stuck bot. The notice is **disabled by default** and configurable per-room or globally via `text-generation set-thinking-notice-enabled true`. Fast responses (under the delay threshold) never show a placeholder. See the [text-generation configuration docs](./docs/configuration/text-generation.md#-thinking-notice).
|
- (**Feature**) Add an opt-in 💭 **thinking notice** for text generation. When enabled, a slow response (for example, from a reasoning model that runs for minutes) posts a "thinking…" placeholder after a short delay, refreshes it periodically with varying flavor text, and then edits that same message into the final answer, so a long wait no longer looks like a stuck bot. The notice is **disabled by default** and configurable per-room or globally via `text-generation set-thinking-notice-enabled true`. Fast responses (under the delay threshold) never show a placeholder. See the [text-generation configuration docs](./docs/configuration/text-generation.md#-thinking-notice).
|
||||||
|
|||||||
77
Cargo.lock
generated
77
Cargo.lock
generated
@@ -95,9 +95,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "anyhow"
|
name = "anyhow"
|
||||||
version = "1.0.103"
|
version = "1.0.104"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "2a4385e2e34eb35d6b3efe798b9eb88096925d87726c0798709bf56d9ed84af3"
|
checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "anymap2"
|
name = "anymap2"
|
||||||
@@ -184,9 +184,9 @@ checksum = "4288f83726785267c6f2ef073a3d83dc3f9b81464e9f99898240cced85fce35a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "async-openai"
|
name = "async-openai"
|
||||||
version = "0.41.1"
|
version = "0.41.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "3007014661d5b98168b7b6f1014147bce8b1362a194783543eeb9f6117a20be9"
|
checksum = "d72db2750faea2ca5edbf6d0c50277a89dc8f75f5e6ddd695ef30f75e335019b"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-openai-macros",
|
"async-openai-macros",
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
@@ -196,7 +196,7 @@ dependencies = [
|
|||||||
"futures",
|
"futures",
|
||||||
"getrandom 0.3.4",
|
"getrandom 0.3.4",
|
||||||
"rand 0.9.4",
|
"rand 0.9.4",
|
||||||
"reqwest 0.13.4",
|
"reqwest 0.13.5",
|
||||||
"secrecy",
|
"secrecy",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
@@ -315,12 +315,12 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "baibot"
|
name = "baibot"
|
||||||
version = "1.24.0"
|
version = "1.25.0"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anthropic",
|
"anthropic",
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"async-openai",
|
"async-openai",
|
||||||
"base64 0.22.1",
|
"base64 0.23.1",
|
||||||
"chrono",
|
"chrono",
|
||||||
"etke_openai_api_rust",
|
"etke_openai_api_rust",
|
||||||
"matrix-sdk",
|
"matrix-sdk",
|
||||||
@@ -329,7 +329,7 @@ dependencies = [
|
|||||||
"mxlink",
|
"mxlink",
|
||||||
"quick_cache 0.7.0",
|
"quick_cache 0.7.0",
|
||||||
"regex",
|
"regex",
|
||||||
"reqwest 0.13.4",
|
"reqwest 0.13.5",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
"serde_yaml_ng",
|
"serde_yaml_ng",
|
||||||
@@ -353,6 +353,12 @@ version = "0.22.1"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
|
checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "base64"
|
||||||
|
version = "0.23.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "base64ct"
|
name = "base64ct"
|
||||||
version = "1.8.3"
|
version = "1.8.3"
|
||||||
@@ -2188,7 +2194,7 @@ dependencies = [
|
|||||||
"oauth2-reqwest",
|
"oauth2-reqwest",
|
||||||
"percent-encoding",
|
"percent-encoding",
|
||||||
"pin-project-lite",
|
"pin-project-lite",
|
||||||
"reqwest 0.13.4",
|
"reqwest 0.13.5",
|
||||||
"ruma",
|
"ruma",
|
||||||
"rustls",
|
"rustls",
|
||||||
"rustls-native-certs 0.8.3",
|
"rustls-native-certs 0.8.3",
|
||||||
@@ -2480,9 +2486,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "mxidwc"
|
name = "mxidwc"
|
||||||
version = "1.0.2"
|
version = "1.0.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "5e253f96a03d24d1c0006c5661b1e20a13dc3861f509d76c33b6b44349d4ff3b"
|
checksum = "45b5d51fcf414d2aa6bffc6cd9b037e62732734a944c5da4ace6b9895ec37b93"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"regex",
|
"regex",
|
||||||
]
|
]
|
||||||
@@ -2583,7 +2589,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "234fb5c965bbce983ee5de636a7a51d6a3223da8067ea02f9ab2d2d78ac08be2"
|
checksum = "234fb5c965bbce983ee5de636a7a51d6a3223da8067ea02f9ab2d2d78ac08be2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"oauth2",
|
"oauth2",
|
||||||
"reqwest 0.13.4",
|
"reqwest 0.13.5",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -3064,9 +3070,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "regex"
|
name = "regex"
|
||||||
version = "1.12.4"
|
version = "1.13.1"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba"
|
checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"aho-corasick",
|
"aho-corasick",
|
||||||
"memchr",
|
"memchr",
|
||||||
@@ -3076,9 +3082,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "regex-automata"
|
name = "regex-automata"
|
||||||
version = "0.4.14"
|
version = "0.4.16"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f"
|
checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"aho-corasick",
|
"aho-corasick",
|
||||||
"memchr",
|
"memchr",
|
||||||
@@ -3134,11 +3140,11 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "reqwest"
|
name = "reqwest"
|
||||||
version = "0.13.4"
|
version = "0.13.5"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3"
|
checksum = "16a1cfa75cc186dd73d5818e510e042e40927bccc9c236b061cea97e1eb08029"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.23.1",
|
||||||
"bytes",
|
"bytes",
|
||||||
"futures-core",
|
"futures-core",
|
||||||
"futures-util",
|
"futures-util",
|
||||||
@@ -3611,9 +3617,9 @@ checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "serde"
|
name = "serde"
|
||||||
version = "1.0.228"
|
version = "1.0.229"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e"
|
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"serde_core",
|
"serde_core",
|
||||||
"serde_derive",
|
"serde_derive",
|
||||||
@@ -3642,22 +3648,22 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "serde_core"
|
name = "serde_core"
|
||||||
version = "1.0.228"
|
version = "1.0.229"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad"
|
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"serde_derive",
|
"serde_derive",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "serde_derive"
|
name = "serde_derive"
|
||||||
version = "1.0.228"
|
version = "1.0.229"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79"
|
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
"syn 2.0.117",
|
"syn 3.0.0",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -3675,9 +3681,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "serde_json"
|
name = "serde_json"
|
||||||
version = "1.0.150"
|
version = "1.0.151"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
|
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"itoa",
|
"itoa",
|
||||||
"memchr",
|
"memchr",
|
||||||
@@ -3899,6 +3905,17 @@ dependencies = [
|
|||||||
"unicode-ident",
|
"unicode-ident",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "3.0.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f2fac314a64dc9a36e61a9eb4261a5e9bbfbc922b27e518af97bc32b926cf967"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "sync_wrapper"
|
name = "sync_wrapper"
|
||||||
version = "1.0.2"
|
version = "1.0.2"
|
||||||
@@ -4064,9 +4081,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "tokio"
|
name = "tokio"
|
||||||
version = "1.52.3"
|
version = "1.53.1"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe"
|
checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"bytes",
|
"bytes",
|
||||||
"libc",
|
"libc",
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later"
|
|||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
||||||
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
||||||
version = "1.24.0"
|
version = "1.25.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
|
|
||||||
[lib]
|
[lib]
|
||||||
@@ -18,7 +18,7 @@ path = "src/lib.rs"
|
|||||||
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
||||||
anyhow = "1.0.*"
|
anyhow = "1.0.*"
|
||||||
async-openai = { version = "0.41.0", features = ["audio", "chat-completion", "image", "responses"] }
|
async-openai = { version = "0.41.0", features = ["audio", "chat-completion", "image", "responses"] }
|
||||||
base64 = "0.22.*"
|
base64 = "0.23.*"
|
||||||
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
||||||
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
||||||
matrix-sdk = { version = "0.18.0", default-features = false }
|
matrix-sdk = { version = "0.18.0", default-features = false }
|
||||||
@@ -27,7 +27,7 @@ mxidwc = "1.0.*"
|
|||||||
mxlink = ">=1.15.0"
|
mxlink = ">=1.15.0"
|
||||||
etke_openai_api_rust = "0.1.*"
|
etke_openai_api_rust = "0.1.*"
|
||||||
quick_cache = "0.7.*"
|
quick_cache = "0.7.*"
|
||||||
regex = "1.12.*"
|
regex = "1.13.*"
|
||||||
# HTTP client for the native `venice` provider. rustls only (no extra TLS stack), matching the
|
# HTTP client for the native `venice` provider. rustls only (no extra TLS stack), matching the
|
||||||
# reqwest copy async-openai/matrix-sdk/mxlink already use.
|
# reqwest copy async-openai/matrix-sdk/mxlink already use.
|
||||||
reqwest = { version = "0.13.*", default-features = false, features = ["json", "multipart", "rustls"] }
|
reqwest = { version = "0.13.*", default-features = false, features = ["json", "multipart", "rustls"] }
|
||||||
@@ -36,7 +36,7 @@ serde_json = "1.0.*"
|
|||||||
serde_yaml_ng = "0.10.*"
|
serde_yaml_ng = "0.10.*"
|
||||||
tempfile = "3.27.*"
|
tempfile = "3.27.*"
|
||||||
tiktoken-rs = { version = "0.12.*", default-features = false }
|
tiktoken-rs = { version = "0.12.*", default-features = false }
|
||||||
tokio = { version = "1.52.*", features = ["rt", "rt-multi-thread", "macros"] }
|
tokio = { version = "1.53.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||||
tracing = "0.1.*"
|
tracing = "0.1.*"
|
||||||
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
|
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
|
||||||
url = "2.5.*"
|
url = "2.5.*"
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
# #
|
# #
|
||||||
#######################################
|
#######################################
|
||||||
|
|
||||||
FROM docker.io/rust:1.96.0-slim-trixie AS build
|
FROM docker.io/rust:1.98.0-slim-trixie AS build
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||||
|
|
||||||
|
|||||||
@@ -1,35 +0,0 @@
|
|||||||
#######################################
|
|
||||||
# #
|
|
||||||
# Stage 1: building #
|
|
||||||
# #
|
|
||||||
#######################################
|
|
||||||
|
|
||||||
FROM docker.io/rust:1.96.0-slim-trixie AS build
|
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
|
||||||
|
|
||||||
WORKDIR /app
|
|
||||||
|
|
||||||
COPY . /app
|
|
||||||
|
|
||||||
RUN cargo build --release
|
|
||||||
|
|
||||||
#######################################
|
|
||||||
# #
|
|
||||||
# Stage 2: packaging #
|
|
||||||
# #
|
|
||||||
#######################################
|
|
||||||
|
|
||||||
FROM docker.io/debian:trixie-slim
|
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y ca-certificates sqlite3 && \
|
|
||||||
apt-get clean && \
|
|
||||||
rm -rf /var/lib/apt/lists/*
|
|
||||||
|
|
||||||
WORKDIR /app
|
|
||||||
|
|
||||||
COPY --from=build /app/target/release/baibot .
|
|
||||||
|
|
||||||
ENTRYPOINT ["/bin/sh", "-c"]
|
|
||||||
|
|
||||||
CMD ["/app/baibot"]
|
|
||||||
@@ -30,7 +30,7 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
|
|||||||
|
|
||||||
- 🔒 Supports [encryption](./docs/features.md#-encryption) for Matrix communication and Account-Data-stored configuration
|
- 🔒 Supports [encryption](./docs/features.md#-encryption) for Matrix communication and Account-Data-stored configuration
|
||||||
|
|
||||||
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) handling on some models (automatically adjusting the message history length, etc.)
|
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) for every [provider](./docs/providers.md) (automatically trimming older messages on whole-turn boundaries once a conversation outgrows the context window)
|
||||||
|
|
||||||
- 🛠️ Allows **customizing much of the bot's [configuration](./docs/configuration/README.md)** at runtime (using commands sent via chat)
|
- 🛠️ Allows **customizing much of the bot's [configuration](./docs/configuration/README.md)** at runtime (using commands sent via chat)
|
||||||
|
|
||||||
|
|||||||
@@ -50,9 +50,9 @@ Example: `!bai config room text-generation set-auto-usage only_for_voice` (this
|
|||||||
|
|
||||||
### ♻️ Context Management
|
### ♻️ Context Management
|
||||||
|
|
||||||
The bot also supports ♻️ **context management**, which automatically adjusts the message history length, etc.
|
The bot also supports ♻️ **context management**, which automatically trims the oldest messages once a conversation grows past the context window. It drops whole turns at a time, so a reply is never separated from the message it answered.
|
||||||
|
|
||||||
This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](../providers.md#openai) models.
|
Counting tokens precisely needs the model's own tokenizer. For [OpenAI](../providers.md#openai) models, the bot counts them with the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library. For every other provider, including the recommended [Venice](../providers.md#venice), the bot falls back to a provider-neutral **approximation** that needs no per-model tokenizer: it counts ASCII text at about four characters per token and other scripts (Cyrillic, CJK, and so on) at about two. Treat it as rough, within roughly 10-20% of the real count for typical text, which is plenty for keeping a long conversation inside the context window.
|
||||||
|
|
||||||
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
||||||
|
|
||||||
|
|||||||
@@ -57,6 +57,34 @@ CONTAINER_IMAGE_NAME=ghcr.io/etkecc/baibot:v1.0.0
|
|||||||
$CONTAINER_IMAGE_NAME
|
$CONTAINER_IMAGE_NAME
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Alternatively, you can use [Docker Compose](https://docs.docker.com/compose/) with a `compose.yml` file like this:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
baibot:
|
||||||
|
container_name: baibot
|
||||||
|
# Adjust the version tag to point to the latest available tagged version.
|
||||||
|
# If building your own container image name, adjust to something like `localhost/baibot:latest`.
|
||||||
|
image: ghcr.io/etkecc/baibot:v1.0.0
|
||||||
|
# Set `UID` and `GID` in a `.env` file next to `compose.yml` (e.g. `UID=1000`, `GID=1000`)
|
||||||
|
# or export them in your shell (`export UID GID="$(id -g)"`).
|
||||||
|
# These should match the user that owns the data directory.
|
||||||
|
user: "${UID:-1000}:${GID:-1000}"
|
||||||
|
environment:
|
||||||
|
# Other settings can also be set via environment variables.
|
||||||
|
# See the 🛠️ Configuration documentation (docs/configuration/README.md) for details.
|
||||||
|
BAIBOT_PERSISTENCE_DATA_DIR_PATH: /data
|
||||||
|
volumes:
|
||||||
|
- /path/to/config.yml:/app/config.yml:ro
|
||||||
|
- /path/to/data:/data
|
||||||
|
cap_drop:
|
||||||
|
- ALL
|
||||||
|
read_only: true
|
||||||
|
tmpfs:
|
||||||
|
- /tmp:rw,noexec,nosuid,size=1024m
|
||||||
|
restart: unless-stopped
|
||||||
|
```
|
||||||
|
|
||||||
💡 If you've defined the `persistence.data_dir_path` setting in the `config.yml` file, you can skip the `BAIBOT_PERSISTENCE_DATA_DIR_PATH` environment variable.
|
💡 If you've defined the `persistence.data_dir_path` setting in the `config.yml` file, you can skip the `BAIBOT_PERSISTENCE_DATA_DIR_PATH` environment variable.
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ The list of supported providers is below.
|
|||||||
|
|
||||||
### How to choose a provider
|
### How to choose a provider
|
||||||
|
|
||||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation) (incl. vision, incl. [🛠️ tools](./features.md#️-built-in-tools-openai-only)), [🖌️ image-generation](./features.md#️image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
If you're not sure which provider to start with, **we recommend [Venice](#venice)**: it's the most capable provider baibot supports (covering [💬 text-generation](./features.md#-text-generation) with vision, file inputs, prompt caching, and native web search, plus [🖌️ image-generation](./features.md#️-image-creation) incl. editing, [🦻 speech-to-text](./features.md#-speech-to-text), and [🗣️ text-to-speech](./features.md#️-text-to-speech)) and the only one that runs inference with no logging and no training on your data. If you'd rather start with the most widely-used option, [OpenAI](#openai) is a solid, well-supported choice too.
|
||||||
|
|
||||||
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
|
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
|
||||||
|
|
||||||
@@ -176,7 +176,7 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
|||||||
|
|
||||||
### Venice
|
### Venice
|
||||||
|
|
||||||
[Venice AI](https://venice.ai/chat?ref=kpXDe6) _(ref link with $10 bonus for you)_ runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
[Venice AI](https://venice.ai/chat?ref=kpXDe6) _(ref link with a $10 bonus for you)_ runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
||||||
|
|
||||||
- 🆔 Identifier: `venice`
|
- 🆔 Identifier: `venice`
|
||||||
- 🔗 Links: [🏠 Home page](https://venice.ai/chat?ref=kpXDe6), [👤 Sign up](https://venice.ai/chat?ref=kpXDe6), [📋 Models list](https://docs.venice.ai/models/overview)
|
- 🔗 Links: [🏠 Home page](https://venice.ai/chat?ref=kpXDe6), [👤 Sign up](https://venice.ai/chat?ref=kpXDe6), [📋 Models list](https://docs.venice.ai/models/overview)
|
||||||
|
|||||||
@@ -1,14 +1,14 @@
|
|||||||
services:
|
services:
|
||||||
continuwuity:
|
continuwuity:
|
||||||
image: forgejo.ellis.link/continuwuation/continuwuity:v0.5.10
|
image: forgejo.ellis.link/continuwuation/continuwuity:v26.8.1
|
||||||
user: "${UID}:${GID}"
|
user: "${UID}:${GID}"
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
cap_drop:
|
cap_drop:
|
||||||
- ALL
|
- ALL
|
||||||
read_only: true
|
read_only: true
|
||||||
environment:
|
environment:
|
||||||
CONDUWUIT_CONFIG: /etc/continuwuity/continuwuity.toml
|
CONTINUWUITY_CONFIG: /etc/continuwuity/continuwuity.toml
|
||||||
CONDUWUIT_DATABASE_PATH: /var/lib/continuwuity
|
CONTINUWUITY_DATABASE_PATH: /var/lib/continuwuity
|
||||||
ports:
|
ports:
|
||||||
- "${SERVICE_CONTINUWUITY_BIND_PORT_CLIENT_API}:6167"
|
- "${SERVICE_CONTINUWUITY_BIND_PORT_CLIENT_API}:6167"
|
||||||
volumes:
|
volumes:
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
services:
|
services:
|
||||||
element-web:
|
element-web:
|
||||||
image: ghcr.io/element-hq/element-web:v1.12.22
|
image: ghcr.io/element-hq/element-web:v1.12.27
|
||||||
user: "${UID}:${GID}"
|
user: "${UID}:${GID}"
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
environment:
|
environment:
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
services:
|
services:
|
||||||
ollama:
|
ollama:
|
||||||
image: docker.io/ollama/ollama:0.30.11
|
image: docker.io/ollama/ollama:0.33.3
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
ports:
|
ports:
|
||||||
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
|
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
services:
|
services:
|
||||||
postgres:
|
postgres:
|
||||||
image: docker.io/postgres:18.4-alpine
|
image: docker.io/postgres:18.6-alpine
|
||||||
user: ${UID}:${GID}
|
user: ${UID}:${GID}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
environment:
|
environment:
|
||||||
@@ -14,7 +14,7 @@ services:
|
|||||||
- /etc/passwd:/etc/passwd:ro
|
- /etc/passwd:/etc/passwd:ro
|
||||||
|
|
||||||
synapse:
|
synapse:
|
||||||
image: ghcr.io/element-hq/synapse:v1.155.0
|
image: ghcr.io/element-hq/synapse:v1.160.0
|
||||||
user: "${UID}:${GID}"
|
user: "${UID}:${GID}"
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
entrypoint: python
|
entrypoint: python
|
||||||
|
|||||||
24
justfile
24
justfile
@@ -271,7 +271,29 @@ prek-run-on-all *args: _ensure_mise_tools_installed
|
|||||||
|
|
||||||
# Installs the git pre-commit hook (runs prek automatically before each commit)
|
# Installs the git pre-commit hook (runs prek automatically before each commit)
|
||||||
prek-install-git-pre-commit-hook: _ensure_mise_tools_installed
|
prek-install-git-pre-commit-hook: _ensure_mise_tools_installed
|
||||||
@just --justfile {{ justfile() }} mise exec -- prek install
|
#!/usr/bin/env sh
|
||||||
|
set -eu
|
||||||
|
just --justfile {{ justfile() }} mise exec -- prek install
|
||||||
|
# The installed git hooks run later under Git, outside this just/mise environment,
|
||||||
|
# so they need to be told how to find their tooling:
|
||||||
|
#
|
||||||
|
# - MISE_DATA_DIR / MISE_TRUSTED_CONFIG_PATHS make mise resolve against this project's
|
||||||
|
# own data directory. Without them mise falls back to the global one and silently
|
||||||
|
# installs a second copy of the tool there.
|
||||||
|
# - prek bakes the full path of the currently installed version into the hook
|
||||||
|
# (var/mise/installs/prek/<version>/...), which stops working as soon as the pinned
|
||||||
|
# version changes or old versions are pruned. Pointing at mise's shim instead makes
|
||||||
|
# the hook resolve whatever mise.toml pins, at the time it runs.
|
||||||
|
#
|
||||||
|
# Which hook files prek installs depends on `default_install_hook_types` in
|
||||||
|
# .pre-commit-config.yaml, so patch every hook file that prek generated.
|
||||||
|
for hook in "{{ justfile_directory() }}"/.git/hooks/*; do
|
||||||
|
[ -f "$hook" ] || continue
|
||||||
|
grep -q 'generated by prek' "$hook" || continue
|
||||||
|
grep -q '^export MISE_DATA_DIR=' "$hook" || sed -i '2iexport MISE_DATA_DIR="{{ mise_data_dir }}"' "$hook"
|
||||||
|
grep -q '^export MISE_TRUSTED_CONFIG_PATHS=' "$hook" || sed -i '3iexport MISE_TRUSTED_CONFIG_PATHS="{{ mise_trusted_config_paths }}"' "$hook"
|
||||||
|
sed -i 's#^PREK=".*"$#PREK="{{ mise_data_dir }}/shims/prek"#' "$hook"
|
||||||
|
done
|
||||||
|
|
||||||
# Internal - ensures var/mise directory exists
|
# Internal - ensures var/mise directory exists
|
||||||
_ensure_mise_data_directory:
|
_ensure_mise_data_directory:
|
||||||
|
|||||||
@@ -1,6 +1,2 @@
|
|||||||
[tools]
|
[tools]
|
||||||
prek = "0.4.5"
|
prek = "0.5.2"
|
||||||
|
|
||||||
[settings]
|
|
||||||
# Disable automatic trust prompts - we trust this config
|
|
||||||
yes = true
|
|
||||||
|
|||||||
@@ -5,5 +5,62 @@
|
|||||||
],
|
],
|
||||||
"labels": [
|
"labels": [
|
||||||
"dependencies"
|
"dependencies"
|
||||||
|
],
|
||||||
|
"packageRules": [
|
||||||
|
{
|
||||||
|
"description": "Cargo dependencies, the Rust toolchain pin, prek (via mise) and the workflows' own actions merge by pushing to main, without a pull request. ci.yml runs on `push: [\"**\"]`, so the Renovate branch itself is compiled, linted with `clippy -D warnings` and unit-tested first; a failure leaves the branch red and Renovate raises a pull request instead of merging.",
|
||||||
|
"matchManagers": [
|
||||||
|
"cargo",
|
||||||
|
"rust-toolchain",
|
||||||
|
"mise",
|
||||||
|
"github-actions"
|
||||||
|
],
|
||||||
|
"matchUpdateTypes": [
|
||||||
|
"minor",
|
||||||
|
"patch",
|
||||||
|
"digest"
|
||||||
|
],
|
||||||
|
"automerge": true,
|
||||||
|
"automergeType": "branch",
|
||||||
|
"platformAutomerge": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"description": "The release image's base (Dockerfile). Gated by ci.yml's docker-build job, which builds the image exactly as Publish does but with `push: false`, and which only runs when the Dockerfile actually changed.",
|
||||||
|
"matchManagers": [
|
||||||
|
"dockerfile"
|
||||||
|
],
|
||||||
|
"matchUpdateTypes": [
|
||||||
|
"minor",
|
||||||
|
"patch",
|
||||||
|
"digest"
|
||||||
|
],
|
||||||
|
"automerge": true,
|
||||||
|
"automergeType": "branch",
|
||||||
|
"platformAutomerge": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"description": "Local development service images under etc/services/** - the homeservers and LLM backends that `just services-start` brings up. They are never part of a shipped artifact and CI does not run them, so the justification here is blast radius rather than validation: the worst case is a broken local development stack, fixed by pinning back.",
|
||||||
|
"matchManagers": [
|
||||||
|
"docker-compose"
|
||||||
|
],
|
||||||
|
"matchFileNames": [
|
||||||
|
"etc/services/**"
|
||||||
|
],
|
||||||
|
"matchUpdateTypes": [
|
||||||
|
"minor",
|
||||||
|
"patch",
|
||||||
|
"digest"
|
||||||
|
],
|
||||||
|
"automerge": true,
|
||||||
|
"automergeType": "branch",
|
||||||
|
"platformAutomerge": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"description": "Major updates always get a pull request and a human. This is deliberately the last rule so that it overrides the automerge rules above for every manager.",
|
||||||
|
"matchUpdateTypes": [
|
||||||
|
"major"
|
||||||
|
],
|
||||||
|
"automerge": false
|
||||||
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
[toolchain]
|
[toolchain]
|
||||||
channel = "1.96.0"
|
channel = "1.98.1"
|
||||||
components = ["rustfmt", "clippy"]
|
components = ["rustfmt", "clippy"]
|
||||||
profile = "default"
|
profile = "default"
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ use crate::agent::provider::{
|
|||||||
};
|
};
|
||||||
use crate::conversation::llm::{
|
use crate::conversation::llm::{
|
||||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||||
};
|
};
|
||||||
use crate::strings;
|
use crate::strings;
|
||||||
|
|
||||||
@@ -130,7 +130,7 @@ impl ControllerTrait for Controller {
|
|||||||
tracing::trace!("Shortening messages list to context size");
|
tracing::trace!("Shortening messages list to context size");
|
||||||
|
|
||||||
conversation_messages = shorten_messages_list_to_context_size(
|
conversation_messages = shorten_messages_list_to_context_size(
|
||||||
&text_generation_config.model_id,
|
TokenEstimate::Approximate,
|
||||||
&prompt_message,
|
&prompt_message,
|
||||||
conversation_messages,
|
conversation_messages,
|
||||||
Some(text_generation_config.max_response_tokens),
|
Some(text_generation_config.max_response_tokens),
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ use crate::{
|
|||||||
},
|
},
|
||||||
conversation::llm::{
|
conversation::llm::{
|
||||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||||
},
|
},
|
||||||
utils::base64::base64_decode,
|
utils::base64::base64_decode,
|
||||||
};
|
};
|
||||||
@@ -117,7 +117,7 @@ impl ControllerTrait for Controller {
|
|||||||
tracing::trace!("Shortening messages list to context size");
|
tracing::trace!("Shortening messages list to context size");
|
||||||
|
|
||||||
conversation_messages = shorten_messages_list_to_context_size(
|
conversation_messages = shorten_messages_list_to_context_size(
|
||||||
&text_generation_config.model_id,
|
TokenEstimate::Tiktoken(&text_generation_config.model_id),
|
||||||
&prompt_message,
|
&prompt_message,
|
||||||
conversation_messages,
|
conversation_messages,
|
||||||
text_generation_config.max_response_tokens,
|
text_generation_config.max_response_tokens,
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ use crate::{
|
|||||||
},
|
},
|
||||||
conversation::llm::{
|
conversation::llm::{
|
||||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
use crate::{
|
use crate::{
|
||||||
@@ -114,7 +114,7 @@ impl ControllerTrait for Controller {
|
|||||||
tracing::trace!("Shortening messages list to context size");
|
tracing::trace!("Shortening messages list to context size");
|
||||||
|
|
||||||
conversation_messages = shorten_messages_list_to_context_size(
|
conversation_messages = shorten_messages_list_to_context_size(
|
||||||
&text_generation_config.model_id,
|
TokenEstimate::Approximate,
|
||||||
&prompt_message,
|
&prompt_message,
|
||||||
conversation_messages,
|
conversation_messages,
|
||||||
text_generation_config.max_response_tokens,
|
text_generation_config.max_response_tokens,
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ use crate::agent::AgentPurpose;
|
|||||||
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
|
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
|
||||||
use crate::conversation::llm::{
|
use crate::conversation::llm::{
|
||||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
MessageContent as LLMMessageContent, TokenEstimate, shorten_messages_list_to_context_size,
|
||||||
};
|
};
|
||||||
use crate::strings;
|
use crate::strings;
|
||||||
|
|
||||||
@@ -64,7 +64,7 @@ pub async fn generate_text(
|
|||||||
|
|
||||||
if params.context_management_enabled {
|
if params.context_management_enabled {
|
||||||
conversation_messages = shorten_messages_list_to_context_size(
|
conversation_messages = shorten_messages_list_to_context_size(
|
||||||
&text_generation_config.model_id,
|
TokenEstimate::Approximate,
|
||||||
&prompt_message,
|
&prompt_message,
|
||||||
conversation_messages,
|
conversation_messages,
|
||||||
text_generation_config.max_response_tokens,
|
text_generation_config.max_response_tokens,
|
||||||
|
|||||||
@@ -136,7 +136,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation prefix-requirement-type"
|
"text-generation prefix-requirement-type"
|
||||||
)
|
)
|
||||||
@@ -144,7 +144,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-prefix-requirement-type VALUE"
|
"text-generation set-prefix-requirement-type VALUE"
|
||||||
)
|
)
|
||||||
@@ -152,7 +152,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-prefix-requirement-type"
|
"text-generation set-prefix-requirement-type"
|
||||||
)
|
)
|
||||||
@@ -176,12 +176,12 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(command_prefix, "text-generation auto-usage")
|
strings::help::cfg::current_setting_show(command_prefix, "text-generation auto-usage")
|
||||||
));
|
));
|
||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-auto-usage VALUE"
|
"text-generation set-auto-usage VALUE"
|
||||||
)
|
)
|
||||||
@@ -189,10 +189,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(command_prefix, "text-generation set-auto-usage")
|
||||||
command_prefix,
|
|
||||||
"text-generation set-auto-usage"
|
|
||||||
)
|
|
||||||
));
|
));
|
||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
|
|
||||||
@@ -211,7 +208,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation context-management-enabled"
|
"text-generation context-management-enabled"
|
||||||
)
|
)
|
||||||
@@ -219,7 +216,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-context-management-enabled VALUE"
|
"text-generation set-context-management-enabled VALUE"
|
||||||
)
|
)
|
||||||
@@ -227,7 +224,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-context-management-enabled"
|
"text-generation set-context-management-enabled"
|
||||||
)
|
)
|
||||||
@@ -249,7 +246,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation thinking-notice-enabled"
|
"text-generation thinking-notice-enabled"
|
||||||
)
|
)
|
||||||
@@ -257,7 +254,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-thinking-notice-enabled VALUE"
|
"text-generation set-thinking-notice-enabled VALUE"
|
||||||
)
|
)
|
||||||
@@ -265,7 +262,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-thinking-notice-enabled"
|
"text-generation set-thinking-notice-enabled"
|
||||||
)
|
)
|
||||||
@@ -289,7 +286,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation sender-context-mode"
|
"text-generation sender-context-mode"
|
||||||
)
|
)
|
||||||
@@ -297,7 +294,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-sender-context-mode VALUE"
|
"text-generation set-sender-context-mode VALUE"
|
||||||
)
|
)
|
||||||
@@ -305,7 +302,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-sender-context-mode"
|
"text-generation set-sender-context-mode"
|
||||||
)
|
)
|
||||||
@@ -323,15 +320,12 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(command_prefix, "text-generation prompt-override")
|
||||||
command_prefix,
|
|
||||||
"text-generation prompt-override"
|
|
||||||
)
|
|
||||||
));
|
));
|
||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-prompt-override VALUE"
|
"text-generation set-prompt-override VALUE"
|
||||||
)
|
)
|
||||||
@@ -339,7 +333,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-prompt-override"
|
"text-generation set-prompt-override"
|
||||||
)
|
)
|
||||||
@@ -357,7 +351,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation temperature-override"
|
"text-generation temperature-override"
|
||||||
)
|
)
|
||||||
@@ -365,7 +359,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-temperature-override VALUE"
|
"text-generation set-temperature-override VALUE"
|
||||||
)
|
)
|
||||||
@@ -373,7 +367,7 @@ fn build_section_text_generation(command_prefix: &str, bot_username: &str) -> St
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-generation set-temperature-override"
|
"text-generation set-temperature-override"
|
||||||
)
|
)
|
||||||
@@ -410,12 +404,12 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(command_prefix, "speech-to-text flow-type")
|
strings::help::cfg::current_setting_show(command_prefix, "speech-to-text flow-type")
|
||||||
));
|
));
|
||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"speech-to-text set-flow-type VALUE"
|
"speech-to-text set-flow-type VALUE"
|
||||||
)
|
)
|
||||||
@@ -423,7 +417,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-flow-type")
|
strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-flow-type")
|
||||||
));
|
));
|
||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
|
|
||||||
@@ -444,7 +438,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"speech-to-text msg-type-for-non-threaded-only-transcribed-messages"
|
"speech-to-text msg-type-for-non-threaded-only-transcribed-messages"
|
||||||
)
|
)
|
||||||
@@ -452,7 +446,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages VALUE"
|
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages VALUE"
|
||||||
)
|
)
|
||||||
@@ -460,7 +454,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages"
|
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages"
|
||||||
)
|
)
|
||||||
@@ -478,12 +472,12 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(command_prefix, "speech-to-text language")
|
strings::help::cfg::current_setting_show(command_prefix, "speech-to-text language")
|
||||||
));
|
));
|
||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"speech-to-text set-language VALUE"
|
"speech-to-text set-language VALUE"
|
||||||
)
|
)
|
||||||
@@ -491,7 +485,7 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-language")
|
strings::help::cfg::current_setting_unset(command_prefix, "speech-to-text set-language")
|
||||||
));
|
));
|
||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
|
|
||||||
@@ -526,7 +520,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech bot-msgs-flow-type"
|
"text-to-speech bot-msgs-flow-type"
|
||||||
)
|
)
|
||||||
@@ -534,7 +528,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-bot-msgs-flow-type VALUE"
|
"text-to-speech set-bot-msgs-flow-type VALUE"
|
||||||
)
|
)
|
||||||
@@ -542,7 +536,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-bot-msgs-flow-type"
|
"text-to-speech set-bot-msgs-flow-type"
|
||||||
)
|
)
|
||||||
@@ -566,7 +560,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(
|
strings::help::cfg::current_setting_show(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech user-msgs-flow-type"
|
"text-to-speech user-msgs-flow-type"
|
||||||
)
|
)
|
||||||
@@ -574,7 +568,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-user-msgs-flow-type VALUE"
|
"text-to-speech set-user-msgs-flow-type VALUE"
|
||||||
)
|
)
|
||||||
@@ -582,7 +576,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-user-msgs-flow-type"
|
"text-to-speech set-user-msgs-flow-type"
|
||||||
)
|
)
|
||||||
@@ -600,12 +594,12 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(command_prefix, "text-to-speech speed-override")
|
strings::help::cfg::current_setting_show(command_prefix, "text-to-speech speed-override")
|
||||||
));
|
));
|
||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-speed-override VALUE"
|
"text-to-speech set-speed-override VALUE"
|
||||||
)
|
)
|
||||||
@@ -613,7 +607,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-speed-override"
|
"text-to-speech set-speed-override"
|
||||||
)
|
)
|
||||||
@@ -631,12 +625,12 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push_str("\n\n");
|
message.push_str("\n\n");
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_show(command_prefix, "text-to-speech voice-override")
|
strings::help::cfg::current_setting_show(command_prefix, "text-to-speech voice-override")
|
||||||
));
|
));
|
||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_set(
|
strings::help::cfg::current_setting_set(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-voice-override VALUE"
|
"text-to-speech set-voice-override VALUE"
|
||||||
)
|
)
|
||||||
@@ -644,7 +638,7 @@ fn build_section_text_to_speech(command_prefix: &str) -> String {
|
|||||||
message.push('\n');
|
message.push('\n');
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"- {}",
|
"- {}",
|
||||||
&strings::help::cfg::current_setting_unset(
|
strings::help::cfg::current_setting_unset(
|
||||||
command_prefix,
|
command_prefix,
|
||||||
"text-to-speech set-voice-override"
|
"text-to-speech set-voice-override"
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -147,9 +147,7 @@ pub async fn handle(
|
|||||||
// Let's proceed below where we potentially handle text-generation.
|
// Let's proceed below where we potentially handle text-generation.
|
||||||
}
|
}
|
||||||
|
|
||||||
let text_to_speech_stage_params: Option<TextToSpeechParams>;
|
let text_to_speech_stage_params: Option<TextToSpeechParams> = if message_context
|
||||||
|
|
||||||
if message_context
|
|
||||||
.room_config_context()
|
.room_config_context()
|
||||||
.should_auto_text_generate(original_message_is_audio)
|
.should_auto_text_generate(original_message_is_audio)
|
||||||
{
|
{
|
||||||
@@ -210,7 +208,7 @@ pub async fn handle(
|
|||||||
return Ok(());
|
return Ok(());
|
||||||
};
|
};
|
||||||
|
|
||||||
text_to_speech_stage_params = match message_context
|
match message_context
|
||||||
.room_config_context()
|
.room_config_context()
|
||||||
.text_to_speech_bot_messages_flow_type()
|
.text_to_speech_bot_messages_flow_type()
|
||||||
{
|
{
|
||||||
@@ -243,7 +241,7 @@ pub async fn handle(
|
|||||||
text_to_speech_eligible_payload,
|
text_to_speech_eligible_payload,
|
||||||
response_type,
|
response_type,
|
||||||
)),
|
)),
|
||||||
};
|
}
|
||||||
} else {
|
} else {
|
||||||
tracing::debug!("Not generating text due to auto-usage configuration");
|
tracing::debug!("Not generating text due to auto-usage configuration");
|
||||||
|
|
||||||
@@ -262,7 +260,7 @@ pub async fn handle(
|
|||||||
event_id: message_context.event_id().clone(),
|
event_id: message_context.event_id().clone(),
|
||||||
};
|
};
|
||||||
|
|
||||||
text_to_speech_stage_params = match message_context
|
match message_context
|
||||||
.room_config_context()
|
.room_config_context()
|
||||||
.text_to_speech_user_messages_flow_type()
|
.text_to_speech_user_messages_flow_type()
|
||||||
{
|
{
|
||||||
@@ -275,8 +273,8 @@ pub async fn handle(
|
|||||||
text_to_speech_eligible_payload,
|
text_to_speech_eligible_payload,
|
||||||
response_type,
|
response_type,
|
||||||
)),
|
)),
|
||||||
};
|
}
|
||||||
}
|
};
|
||||||
|
|
||||||
// We're potentially dealing with some text in text_to_speech_eligible_payload - either coming directly from the user or generated by an agent.
|
// We're potentially dealing with some text in text_to_speech_eligible_payload - either coming directly from the user or generated by an agent.
|
||||||
|
|
||||||
|
|||||||
@@ -6,5 +6,5 @@ mod utils;
|
|||||||
mod tests;
|
mod tests;
|
||||||
|
|
||||||
pub use entity::*;
|
pub use entity::*;
|
||||||
pub use tokenization::shorten_messages_list_to_context_size;
|
pub use tokenization::{TokenEstimate, shorten_messages_list_to_context_size};
|
||||||
pub use utils::*;
|
pub use utils::*;
|
||||||
|
|||||||
@@ -4,6 +4,22 @@ use tiktoken_rs::tokenizer;
|
|||||||
|
|
||||||
use super::{Author, Message, MessageContent};
|
use super::{Author, Message, MessageContent};
|
||||||
|
|
||||||
|
/// How to count the tokens in a conversation when trimming it to fit the context window.
|
||||||
|
pub enum TokenEstimate<'a> {
|
||||||
|
/// Count via the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library.
|
||||||
|
/// Accurate for OpenAI models; every other model falls back to the gpt-4
|
||||||
|
/// tokenizer, which misreads it (badly so for non-English text). Use this only
|
||||||
|
/// for the OpenAI provider.
|
||||||
|
Tiktoken(&'a str),
|
||||||
|
|
||||||
|
/// Provider-neutral approximation that needs no per-model tokenizer. Expect it
|
||||||
|
/// to land within roughly 10-20% of the real count for typical text, leaning
|
||||||
|
/// slightly high: over-counting trims a little extra history, while
|
||||||
|
/// under-counting would overflow the model's real context window. Use this for
|
||||||
|
/// every non-OpenAI provider.
|
||||||
|
Approximate,
|
||||||
|
}
|
||||||
|
|
||||||
fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
||||||
let tokenizer = tokenizer::get_tokenizer(model)
|
let tokenizer = tokenizer::get_tokenizer(model)
|
||||||
.or_else(|| tokenizer::get_tokenizer("gpt-4"))
|
.or_else(|| tokenizer::get_tokenizer("gpt-4"))
|
||||||
@@ -13,21 +29,27 @@ fn get_bpe_for_model(model: &str) -> &'static CoreBPE {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn shorten_messages_list_to_context_size(
|
pub fn shorten_messages_list_to_context_size(
|
||||||
model: &str,
|
estimate: TokenEstimate<'_>,
|
||||||
prompt_message: &Option<Message>,
|
prompt_message: &Option<Message>,
|
||||||
mut messages: Vec<Message>,
|
mut messages: Vec<Message>,
|
||||||
max_response_tokens: Option<u32>,
|
max_response_tokens: Option<u32>,
|
||||||
max_context_tokens: u32,
|
max_context_tokens: u32,
|
||||||
) -> Vec<Message> {
|
) -> Vec<Message> {
|
||||||
// Loading the tokenization data is an expensive process, so
|
// Loading the tiktoken data is expensive, so we resolve the counter once up
|
||||||
// se construct the BPE instance once and then use it for all messages.
|
// front and reuse it for every message.
|
||||||
let bpe = get_bpe_for_model(model);
|
let tiktoken = match estimate {
|
||||||
|
TokenEstimate::Tiktoken(model) => Some((get_bpe_for_model(model), model)),
|
||||||
|
TokenEstimate::Approximate => None,
|
||||||
|
};
|
||||||
|
let count = |message: &Message| match tiktoken {
|
||||||
|
Some((bpe, model)) => tiktoken_token_size_for_message(bpe, model, message),
|
||||||
|
None => approximate_token_size_for_message(message),
|
||||||
|
};
|
||||||
|
|
||||||
// We want to retain the prompt in all cases, so we always count it first.
|
// We want to retain the prompt in all cases, so we always count it first.
|
||||||
// We also always reserve enough tokens for the maximum response we expect.
|
// We also always reserve enough tokens for the maximum response we expect.
|
||||||
let mut current_context_length: u32 = if let Some(prompt_message) = prompt_message {
|
let mut current_context_length: u32 = if let Some(prompt_message) = prompt_message {
|
||||||
calculate_token_size_for_message(bpe, model, prompt_message)
|
count(prompt_message) + max_response_tokens.unwrap_or(0)
|
||||||
+ max_response_tokens.unwrap_or(0)
|
|
||||||
} else {
|
} else {
|
||||||
0
|
0
|
||||||
};
|
};
|
||||||
@@ -37,7 +59,7 @@ pub fn shorten_messages_list_to_context_size(
|
|||||||
let mut messages_to_keep: Vec<Message> = Vec::new();
|
let mut messages_to_keep: Vec<Message> = Vec::new();
|
||||||
|
|
||||||
for message in messages {
|
for message in messages {
|
||||||
let tokens_for_message = calculate_token_size_for_message(bpe, model, &message);
|
let tokens_for_message = count(&message);
|
||||||
|
|
||||||
if current_context_length + tokens_for_message > max_context_tokens {
|
if current_context_length + tokens_for_message > max_context_tokens {
|
||||||
break;
|
break;
|
||||||
@@ -48,14 +70,26 @@ pub fn shorten_messages_list_to_context_size(
|
|||||||
messages_to_keep.push(message);
|
messages_to_keep.push(message);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Cut on a turn boundary: the loop may stop right after an assistant reply
|
||||||
|
// whose triggering user message did not fit, which would leave the kept window
|
||||||
|
// starting on an orphaned reply. `messages_to_keep` is newest-first here, so
|
||||||
|
// the oldest kept messages are at the end; drop any trailing assistant messages
|
||||||
|
// until the window begins at the start of a turn (a user message).
|
||||||
|
while matches!(
|
||||||
|
messages_to_keep.last().map(|message| &message.author),
|
||||||
|
Some(Author::Assistant)
|
||||||
|
) {
|
||||||
|
messages_to_keep.pop();
|
||||||
|
}
|
||||||
|
|
||||||
messages_to_keep.reverse();
|
messages_to_keep.reverse();
|
||||||
|
|
||||||
messages_to_keep
|
messages_to_keep
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Calculate the token size of a message for a given model, with a preloaded CoreBPE object.
|
/// Token size of a message via tiktoken, for a preloaded CoreBPE object.
|
||||||
/// Related to `calculate_token_size_for_model_message`.
|
/// Accurate only for OpenAI models (see [`TokenEstimate::Tiktoken`]).
|
||||||
fn calculate_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Message) -> u32 {
|
fn tiktoken_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Message) -> u32 {
|
||||||
let (tokens_per_message, tokens_per_name) = if model.starts_with("gpt-3.5") {
|
let (tokens_per_message, tokens_per_name) = if model.starts_with("gpt-3.5") {
|
||||||
(
|
(
|
||||||
4, // every message follows <im_start>{role/name}\n{content}<im_end>\n
|
4, // every message follows <im_start>{role/name}\n{content}<im_end>\n
|
||||||
@@ -80,6 +114,52 @@ fn calculate_token_size_for_message(bpe: &CoreBPE, model: &str, message: &Messag
|
|||||||
(text_length + role_length + tokens_per_message + tokens_per_name) as u32
|
(text_length + role_length + tokens_per_message + tokens_per_name) as u32
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// ASCII text averages about four characters per token.
|
||||||
|
const ASCII_TOKENS_PER_CHAR: f32 = 0.25;
|
||||||
|
|
||||||
|
/// Non-ASCII scripts (Cyrillic, CJK, and others) pack more information per
|
||||||
|
/// character: real tokenizers land around two characters per token for them, so
|
||||||
|
/// each counts as half a token. CJK runs a touch denser than that, so its estimate
|
||||||
|
/// can read slightly low, still within the tolerance this approximation targets.
|
||||||
|
const WIDE_TOKENS_PER_CHAR: f32 = 0.5;
|
||||||
|
|
||||||
|
/// Structural per-message overhead (role marker plus message framing), mirroring
|
||||||
|
/// the small constant the tiktoken path adds.
|
||||||
|
const APPROX_TOKENS_PER_MESSAGE: u32 = 4;
|
||||||
|
|
||||||
|
/// Provider-neutral, tokenizer-free token size of a message
|
||||||
|
/// (see [`TokenEstimate::Approximate`]).
|
||||||
|
fn approximate_token_size_for_message(message: &Message) -> u32 {
|
||||||
|
let text_tokens = match &message.content {
|
||||||
|
MessageContent::Text(text) => approximate_token_size_for_text(text),
|
||||||
|
// Images and files are not counted as text, matching the tiktoken path.
|
||||||
|
MessageContent::Image(..) | MessageContent::File(..) => 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
text_tokens + APPROX_TOKENS_PER_MESSAGE
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rough token estimate for a piece of text, with no tokenizer.
|
||||||
|
///
|
||||||
|
/// ASCII characters count as a quarter-token each (~4 chars/token); characters
|
||||||
|
/// outside ASCII count as half a token each (~2 chars/token), matching how real
|
||||||
|
/// tokenizers treat Cyrillic and CJK. Weighting non-ASCII up keeps the estimate
|
||||||
|
/// from badly under-counting non-English text, the case the tiktoken fallback gets
|
||||||
|
/// most wrong.
|
||||||
|
fn approximate_token_size_for_text(text: &str) -> u32 {
|
||||||
|
let mut estimate = 0.0_f32;
|
||||||
|
|
||||||
|
for character in text.chars() {
|
||||||
|
estimate += if character.is_ascii() {
|
||||||
|
ASCII_TOKENS_PER_CHAR
|
||||||
|
} else {
|
||||||
|
WIDE_TOKENS_PER_CHAR
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
estimate.ceil() as u32
|
||||||
|
}
|
||||||
|
|
||||||
pub mod test {
|
pub mod test {
|
||||||
#[test]
|
#[test]
|
||||||
fn message_size_counting_works() {
|
fn message_size_counting_works() {
|
||||||
@@ -94,7 +174,7 @@ pub mod test {
|
|||||||
timestamp: chrono::Utc::now(),
|
timestamp: chrono::Utc::now(),
|
||||||
};
|
};
|
||||||
|
|
||||||
let tokens = super::calculate_token_size_for_message(bpe, model, &message);
|
let tokens = super::tiktoken_token_size_for_message(bpe, model, &message);
|
||||||
|
|
||||||
assert_eq!(8, tokens);
|
assert_eq!(8, tokens);
|
||||||
}
|
}
|
||||||
@@ -118,7 +198,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
prompt_length,
|
prompt_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &prompt)
|
super::tiktoken_token_size_for_message(bpe, model, &prompt)
|
||||||
);
|
);
|
||||||
|
|
||||||
let mut conversation_messages = Vec::new();
|
let mut conversation_messages = Vec::new();
|
||||||
@@ -133,7 +213,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
first_length,
|
first_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &first)
|
super::tiktoken_token_size_for_message(bpe, model, &first)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(first);
|
conversation_messages.push(first);
|
||||||
@@ -148,7 +228,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
second_length,
|
second_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &second)
|
super::tiktoken_token_size_for_message(bpe, model, &second)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(second);
|
conversation_messages.push(second);
|
||||||
@@ -165,7 +245,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
third_length,
|
third_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &third)
|
super::tiktoken_token_size_for_message(bpe, model, &third)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(third.clone());
|
conversation_messages.push(third.clone());
|
||||||
@@ -182,7 +262,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
forth_length,
|
forth_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &forth)
|
super::tiktoken_token_size_for_message(bpe, model, &forth)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(forth.clone());
|
conversation_messages.push(forth.clone());
|
||||||
@@ -190,7 +270,7 @@ pub mod test {
|
|||||||
assert_eq!(4, conversation_messages.len());
|
assert_eq!(4, conversation_messages.len());
|
||||||
|
|
||||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||||
model,
|
super::TokenEstimate::Tiktoken(model),
|
||||||
&Some(prompt),
|
&Some(prompt),
|
||||||
conversation_messages,
|
conversation_messages,
|
||||||
max_response_tokens,
|
max_response_tokens,
|
||||||
@@ -229,7 +309,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
prompt_length,
|
prompt_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &prompt)
|
super::tiktoken_token_size_for_message(bpe, model, &prompt)
|
||||||
);
|
);
|
||||||
|
|
||||||
let mut conversation_messages = Vec::new();
|
let mut conversation_messages = Vec::new();
|
||||||
@@ -244,7 +324,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
first_length,
|
first_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &first)
|
super::tiktoken_token_size_for_message(bpe, model, &first)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(first);
|
conversation_messages.push(first);
|
||||||
@@ -259,7 +339,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
second_length,
|
second_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &second)
|
super::tiktoken_token_size_for_message(bpe, model, &second)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(second);
|
conversation_messages.push(second);
|
||||||
@@ -276,7 +356,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
third_length,
|
third_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &third)
|
super::tiktoken_token_size_for_message(bpe, model, &third)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(third.clone());
|
conversation_messages.push(third.clone());
|
||||||
@@ -293,7 +373,7 @@ pub mod test {
|
|||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
forth_length,
|
forth_length,
|
||||||
super::calculate_token_size_for_message(bpe, model, &forth)
|
super::tiktoken_token_size_for_message(bpe, model, &forth)
|
||||||
);
|
);
|
||||||
|
|
||||||
conversation_messages.push(forth.clone());
|
conversation_messages.push(forth.clone());
|
||||||
@@ -301,7 +381,7 @@ pub mod test {
|
|||||||
assert_eq!(4, conversation_messages.len());
|
assert_eq!(4, conversation_messages.len());
|
||||||
|
|
||||||
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||||
model,
|
super::TokenEstimate::Tiktoken(model),
|
||||||
&Some(prompt),
|
&Some(prompt),
|
||||||
conversation_messages,
|
conversation_messages,
|
||||||
max_response_tokens,
|
max_response_tokens,
|
||||||
@@ -320,4 +400,126 @@ pub mod test {
|
|||||||
forth.content
|
forth.content
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn approximate_counting_weights_ascii_and_wide_scripts() {
|
||||||
|
// 12 ASCII characters at ~4 chars/token = 3 text tokens.
|
||||||
|
assert_eq!(3, super::approximate_token_size_for_text("Hello there!"));
|
||||||
|
|
||||||
|
// 5 CJK characters at ~0.5 token/char = 3 text tokens. The ASCII rate would
|
||||||
|
// have under-counted these to 2, the failure mode this path avoids.
|
||||||
|
assert_eq!(3, super::approximate_token_size_for_text("こんにちは"));
|
||||||
|
|
||||||
|
let message = super::Message {
|
||||||
|
author: super::Author::User,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("Hello there!".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
|
||||||
|
// 3 text tokens plus the per-message overhead (4).
|
||||||
|
assert_eq!(7, super::approximate_token_size_for_message(&message));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn approximate_shortening_trims_to_budget() {
|
||||||
|
let prompt = super::Message {
|
||||||
|
author: super::Author::Prompt,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("You are a bot!".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let older = super::Message {
|
||||||
|
author: super::Author::User,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("This is the older message.".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
let newer = super::Message {
|
||||||
|
// A user message, so it is a valid window start: keeping a lone
|
||||||
|
// assistant reply would be an orphan and get trimmed (see
|
||||||
|
// `shortening_cuts_on_a_turn_boundary`).
|
||||||
|
author: super::Author::User,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("This is the newer message.".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
|
||||||
|
// Budget room for the prompt and only the newest message.
|
||||||
|
let max_context_tokens = super::approximate_token_size_for_message(&prompt)
|
||||||
|
+ super::approximate_token_size_for_message(&newer);
|
||||||
|
|
||||||
|
let new_conversation_messages = super::shorten_messages_list_to_context_size(
|
||||||
|
super::TokenEstimate::Approximate,
|
||||||
|
&Some(prompt),
|
||||||
|
vec![older, newer.clone()],
|
||||||
|
None,
|
||||||
|
max_context_tokens,
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_eq!(1, new_conversation_messages.len());
|
||||||
|
assert_eq!(
|
||||||
|
new_conversation_messages.first().unwrap().content,
|
||||||
|
newer.content
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn shortening_cuts_on_a_turn_boundary() {
|
||||||
|
// A four-message conversation of two full turns. All four messages are the
|
||||||
|
// same length, so they cost the same number of tokens.
|
||||||
|
let prompt = super::Message {
|
||||||
|
author: super::Author::Prompt,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("system".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let user_one = super::Message {
|
||||||
|
author: super::Author::User,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("user msg 1".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
let asst_one = super::Message {
|
||||||
|
author: super::Author::Assistant,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("asst msg 1".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
let user_two = super::Message {
|
||||||
|
author: super::Author::User,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("user msg 2".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
let asst_two = super::Message {
|
||||||
|
author: super::Author::Assistant,
|
||||||
|
sender_id: None,
|
||||||
|
content: super::MessageContent::Text("asst msg 2".to_string()),
|
||||||
|
timestamp: chrono::Utc::now(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let per_message = super::approximate_token_size_for_message(&user_one);
|
||||||
|
// Budget fits the prompt plus three messages. By raw token budget the loop
|
||||||
|
// would keep asst_two, user_two, and asst_one, but asst_one's own user
|
||||||
|
// message (user_one) does not fit, so it must be dropped too rather than
|
||||||
|
// left as an orphaned reply.
|
||||||
|
let max_context_tokens =
|
||||||
|
super::approximate_token_size_for_message(&prompt) + (per_message * 3);
|
||||||
|
|
||||||
|
let kept = super::shorten_messages_list_to_context_size(
|
||||||
|
super::TokenEstimate::Approximate,
|
||||||
|
&Some(prompt),
|
||||||
|
vec![user_one, asst_one, user_two.clone(), asst_two.clone()],
|
||||||
|
None,
|
||||||
|
max_context_tokens,
|
||||||
|
);
|
||||||
|
|
||||||
|
// Only the last whole turn survives; the orphaned asst_one is dropped.
|
||||||
|
assert_eq!(2, kept.len());
|
||||||
|
assert_eq!(kept.first().unwrap().content, user_two.content);
|
||||||
|
assert_eq!(kept.last().unwrap().content, asst_two.content);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -101,13 +101,13 @@ pub fn post_creation_helpful_commands(
|
|||||||
for purpose in supported_purposes {
|
for purpose in supported_purposes {
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"\n- {}",
|
"\n- {}",
|
||||||
&set_as_purpose_handler_in_room(agent_identifier, purpose, command_prefix,)
|
set_as_purpose_handler_in_room(agent_identifier, purpose, command_prefix,)
|
||||||
));
|
));
|
||||||
|
|
||||||
if !is_room_local {
|
if !is_room_local {
|
||||||
message.push_str(&format!(
|
message.push_str(&format!(
|
||||||
"\n- {}",
|
"\n- {}",
|
||||||
&set_as_purpose_handler_globally(agent_identifier, purpose, command_prefix,)
|
set_as_purpose_handler_globally(agent_identifier, purpose, command_prefix,)
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -128,7 +128,7 @@ pub fn text_generation_context_management_intro() -> String {
|
|||||||
format!(
|
format!(
|
||||||
"{}\n{}",
|
"{}\n{}",
|
||||||
"Controls the bot's ability to **intelligently drop old messages from the conversation context** when it gets too large.",
|
"Controls the bot's ability to **intelligently drop old messages from the conversation context** when it gets too large.",
|
||||||
"This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](./providers.md#openai) models.",
|
"Counting tokens precisely needs the model's own tokenizer. For [OpenAI](./providers.md#openai) models the bot uses the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library; for every other provider (including the recommended [Venice](./providers.md#venice)) it falls back to a provider-neutral **approximation** (ASCII counted at ~4 characters per token, other scripts at ~2), within roughly 10-20% of the real count for typical text.",
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user