Compare commits
309 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
de980f0165 | ||
|
|
cc3888a4cf | ||
|
|
eb3285f104 | ||
|
|
0a03dde523 | ||
|
|
a5b575da68 | ||
|
|
fe7afec920 | ||
|
|
b6641523da | ||
|
|
c579977e59 | ||
|
|
0a5f37c3eb | ||
|
|
58246e4eb3 | ||
|
|
9b767a420e | ||
|
|
54c27311c1 | ||
|
|
3d15f11f76 | ||
|
|
ead70d71ff | ||
|
|
49bb52a091 | ||
|
|
1fc2f0f65a | ||
|
|
f7cb38b620 | ||
|
|
04e7667d11 | ||
|
|
78cbfda481 | ||
|
|
36fd6a4dda | ||
|
|
1308196419 | ||
|
|
b3546ebaf1 | ||
|
|
2499e6baca | ||
|
|
871b9d2f3b | ||
|
|
4a36cf9446 | ||
|
|
a2180452c9 | ||
|
|
0cb0fc18ce | ||
|
|
78bddac716 | ||
|
|
f7209221f6 | ||
|
|
5f2054e07f | ||
|
|
8301c5a29f | ||
|
|
b325b6b1a5 | ||
|
|
0600721663 | ||
|
|
dc17cf7e96 | ||
|
|
d6a8f6ba0b | ||
|
|
c4c3d71195 | ||
|
|
84ae29d034 | ||
|
|
447f43df7b | ||
|
|
dabac790ad | ||
|
|
fa01f012a1 | ||
|
|
20cb33bc66 | ||
|
|
2791bdb08b | ||
|
|
5dd505202a | ||
|
|
d61078002c | ||
|
|
cf5b346558 | ||
|
|
ebbb6658e1 | ||
|
|
369dc0c1ba | ||
|
|
3185f44a93 | ||
|
|
aa8ccde0ed | ||
|
|
d5df0d7416 | ||
|
|
140ca9ed68 | ||
|
|
89d77d52b6 | ||
|
|
5092700275 | ||
|
|
10c365124e | ||
|
|
08bdf4f7a2 | ||
|
|
af557a7e45 | ||
|
|
fa7eb11b1d | ||
|
|
ff1e128f0e | ||
|
|
c9da927c66 | ||
|
|
7b16c5c1c3 | ||
|
|
f809bf8d7c | ||
|
|
44aea427fc | ||
|
|
8f45a57ced | ||
|
|
6a9100309c | ||
|
|
1953878e6b | ||
|
|
35761a5bf7 | ||
|
|
c7fa0cc0ee | ||
|
|
87fcf9d019 | ||
|
|
4b52bc906c | ||
|
|
517a6c5e33 | ||
|
|
636c8a35eb | ||
|
|
a50de600da | ||
|
|
e978d3cb2f | ||
|
|
2b1bdbd3d2 | ||
|
|
7d183b91d1 | ||
|
|
420f380417 | ||
|
|
6f541e2361 | ||
|
|
0737c2761e | ||
|
|
45852a0d53 | ||
|
|
0fcf8d8703 | ||
|
|
41905b006a | ||
|
|
ee3d27701b | ||
|
|
1b5c2fbdc1 | ||
|
|
a865a26093 | ||
|
|
1563e83672 | ||
|
|
cf114b37b3 | ||
|
|
b8ef2b978b | ||
|
|
7e37aee1b9 | ||
|
|
53836e556a | ||
|
|
89052bbfd9 | ||
|
|
07eb12d406 | ||
|
|
3272cd6fb2 | ||
|
|
7ab97a39ca | ||
|
|
11c5a9942e | ||
|
|
251dec454c | ||
|
|
1031dbf672 | ||
|
|
581f00b9fb | ||
|
|
e57778d2bd | ||
|
|
2d659964a7 | ||
|
|
661e7263fb | ||
|
|
1705c16762 | ||
|
|
2455117e41 | ||
|
|
3e9c110afc | ||
|
|
d9b5524c97 | ||
|
|
9b169a7d28 | ||
|
|
cb29419d75 | ||
|
|
51ca8c9948 | ||
|
|
12b938d2d1 | ||
|
|
f3d1b32ad7 | ||
|
|
3d3bd3c9f9 | ||
|
|
a25845e89e | ||
|
|
479f54b93d | ||
|
|
90ab6807ac | ||
|
|
5022f79bf5 | ||
|
|
2ba3b5a437 | ||
|
|
2819011b9d | ||
|
|
ede9065f77 | ||
|
|
5e61f5b3a3 | ||
|
|
8633e82f62 | ||
|
|
6aa3d70d57 | ||
|
|
75e29ee3fb | ||
|
|
eab7978b9f | ||
|
|
765e7c17b2 | ||
|
|
748d2b7fd4 | ||
|
|
527759dd02 | ||
|
|
3290255bad | ||
|
|
9b987395b3 | ||
|
|
4852d1fe92 | ||
|
|
8bd313f0d4 | ||
|
|
711e1099d6 | ||
|
|
91c8dd8f7d | ||
|
|
afc5572d6a | ||
|
|
73e13dcf2f | ||
|
|
2bebd109b1 | ||
|
|
7f7c58be1f | ||
|
|
47e5a464a0 | ||
|
|
85f751e514 | ||
|
|
95acad3558 | ||
|
|
304056c59a | ||
|
|
bedc0335f1 | ||
|
|
a8be8c3c1e | ||
|
|
826fa728a9 | ||
|
|
5aef8e8b2f | ||
|
|
f70f20181e | ||
|
|
fcdd4f39ee | ||
|
|
891adfec49 | ||
|
|
35ab79844b | ||
|
|
bbc122fbb1 | ||
|
|
2413c8b88b | ||
|
|
10c3c64469 | ||
|
|
b3307b404b | ||
|
|
7a0d1e830d | ||
|
|
b3bd241823 | ||
|
|
de3d8b054f | ||
|
|
0a55e276a2 | ||
|
|
1f2c65d2e6 | ||
|
|
3b5e4745f2 | ||
|
|
407bb022d9 | ||
|
|
faf92cac09 | ||
|
|
a82e9a1d1f | ||
|
|
8f87f05a08 | ||
|
|
61d18b2e13 | ||
|
|
d831c08306 | ||
|
|
c70387b0c3 | ||
|
|
38516f2e17 | ||
|
|
5481b5a763 | ||
|
|
26bc437678 | ||
|
|
ec93f1ee2a | ||
|
|
b920b6e556 | ||
|
|
7136d34843 | ||
|
|
e0b4a40dd8 | ||
|
|
691aeeb1c7 | ||
|
|
257ffae9e7 | ||
|
|
f7bf3d7b60 | ||
|
|
3a88b0d656 | ||
|
|
08c689a889 | ||
|
|
ae8e878817 | ||
|
|
edbd72ece6 | ||
|
|
22906aa2d3 | ||
|
|
99bde53ef6 | ||
|
|
b3fd8e548f | ||
|
|
062fbbb8ef | ||
|
|
2801c78ad9 | ||
|
|
f4c698ad33 | ||
|
|
bd39001417 | ||
|
|
2692d0322e | ||
|
|
8eb70f0f2c | ||
|
|
5c0a7be7a2 | ||
|
|
0a8f9fc3e5 | ||
|
|
1ac3b2e060 | ||
|
|
a3ef9fd1bf | ||
|
|
df507eb201 | ||
|
|
4dcd9eff40 | ||
|
|
ea760ce755 | ||
|
|
1528df6a55 | ||
|
|
b0fa024297 | ||
|
|
3ec203128a | ||
|
|
da97361e1b | ||
|
|
b430fe0189 | ||
|
|
f03126a9e1 | ||
|
|
7d46b926c1 | ||
|
|
6f3c048195 | ||
|
|
b47cf598b5 | ||
|
|
265ad7e1cb | ||
|
|
a159f67e45 | ||
|
|
624b9de35b | ||
|
|
941bf7ca42 | ||
|
|
ef0f1671da | ||
|
|
b43f61f5ff | ||
|
|
1967d2b34c | ||
|
|
bb3734ad24 | ||
|
|
eb6db34177 | ||
|
|
6e845caa2e | ||
|
|
1004966785 | ||
|
|
7ae1864c2e | ||
|
|
68a2fb161f | ||
|
|
3a3eb58d7b | ||
|
|
74d988e650 | ||
|
|
ed8bedcd7e | ||
|
|
2842632969 | ||
|
|
10a5bd2abb | ||
|
|
5308b75f52 | ||
|
|
dad61e1270 | ||
|
|
91986a129c | ||
|
|
264f683d6a | ||
|
|
62f0f4fa0d | ||
|
|
69627abd74 | ||
|
|
d2660be33c | ||
|
|
ce81fe69bd | ||
|
|
1162636b88 | ||
|
|
8c90e13a79 | ||
|
|
274b614d25 | ||
|
|
7bd46821dc | ||
|
|
a84135ff32 | ||
|
|
231528a0d8 | ||
|
|
d8e47b0578 | ||
|
|
96c1542f4a | ||
|
|
2f9c3dfce0 | ||
|
|
de958208b2 | ||
|
|
ac4f2080ce | ||
|
|
3ffa50b7b9 | ||
|
|
8f86289373 | ||
|
|
e0dcc39a72 | ||
|
|
c94376109c | ||
|
|
256ed05662 | ||
|
|
8222681e27 | ||
|
|
f304b93c68 | ||
|
|
889d8a1d04 | ||
|
|
6082bfaf56 | ||
|
|
1d629e0859 | ||
|
|
49471c1df0 | ||
|
|
8f956d2329 | ||
|
|
ba4aa35987 | ||
|
|
06d699a17d | ||
|
|
77d41fb7eb | ||
|
|
aaf283dde3 | ||
|
|
17eafa86af | ||
|
|
4704934b06 | ||
|
|
6719538530 | ||
|
|
59e2746578 | ||
|
|
47d8edea70 | ||
|
|
692d61b239 | ||
|
|
05902f4c17 | ||
|
|
7e66068b16 | ||
|
|
406141cd7d | ||
|
|
c051da2f4a | ||
|
|
1ff7e8cf79 | ||
|
|
b3bca98e84 | ||
|
|
c07b712318 | ||
|
|
6741483056 | ||
|
|
06b2b6d776 | ||
|
|
a1bd292752 | ||
|
|
e4e1fe0e7b | ||
|
|
45a2d96029 | ||
|
|
ec1879d212 | ||
|
|
5e6a600895 | ||
|
|
3db924b124 | ||
|
|
ff7a5ef7af | ||
|
|
cd7d9137e8 | ||
|
|
0d509b2d0e | ||
|
|
3c47d40781 | ||
|
|
78893247e7 | ||
|
|
4847bd8ba8 | ||
|
|
c8abf0e316 | ||
|
|
39a184e5d0 | ||
|
|
9d166e35ba | ||
|
|
4a5966401c | ||
|
|
d92dfba2bf | ||
|
|
8538d6b2b8 | ||
|
|
23f763ba72 | ||
|
|
a9e4ab1bdb | ||
|
|
d9a045a5e4 | ||
|
|
393be9be5a | ||
|
|
a7b016a3d3 | ||
|
|
85e66406dc | ||
|
|
db9422740c | ||
|
|
90fbad5b64 | ||
|
|
b40226826f | ||
|
|
b89f0db71a | ||
|
|
36fdb46633 | ||
|
|
04ce8db1fc | ||
|
|
9908512968 | ||
|
|
eae6472c7a | ||
|
|
e6aa956423 | ||
|
|
7a38216192 | ||
|
|
d522d268e2 | ||
|
|
72120c5dc2 | ||
|
|
a2c35238c2 | ||
|
|
97f5cbb00b |
26
.github/workflows/ci.yml
vendored
Normal file
@@ -0,0 +1,26 @@
|
||||
name: CI
|
||||
on:
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
branches: [ "main" ]
|
||||
push:
|
||||
branches:
|
||||
- "**"
|
||||
tags: [ "v*" ]
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read
|
||||
concurrency:
|
||||
group: ci-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
jobs:
|
||||
test-and-clippy:
|
||||
name: Unit testing and linting
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: dtolnay/rust-toolchain@1.93.0
|
||||
- name: Install SQLite3
|
||||
run: sudo apt-get update && sudo apt-get install -y libsqlite3-dev
|
||||
- run: cargo test --all-features
|
||||
- run: cargo clippy
|
||||
124
.github/workflows/publish.yml
vendored
Normal file
@@ -0,0 +1,124 @@
|
||||
name: Publish
|
||||
on:
|
||||
workflow_run:
|
||||
workflows: [ "CI" ]
|
||||
types: [ "completed" ]
|
||||
permissions:
|
||||
contents: read
|
||||
concurrency:
|
||||
group: publish-${{ github.event.workflow_run.id || github.ref }}
|
||||
cancel-in-progress: false
|
||||
jobs:
|
||||
docker-clean-metadata:
|
||||
if: |
|
||||
github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.event == 'push' &&
|
||||
(
|
||||
github.event.workflow_run.head_branch == 'main' ||
|
||||
startsWith(github.event.workflow_run.head_branch || '', 'v')
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
json: ${{ steps.meta.outputs.json }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
- name: Extract metadata (tags, labels) for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v6
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/${{ github.repository }}
|
||||
tags: |
|
||||
type=raw,value=latest,enable=${{ github.event.workflow_run.head_branch == 'main' }}
|
||||
type=semver,pattern={{raw}},value=${{ github.event.workflow_run.head_branch }},enable=${{ startsWith(github.event.workflow_run.head_branch || '', 'v') }}
|
||||
|
||||
docker-build:
|
||||
if: |
|
||||
github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.event == 'push' &&
|
||||
(
|
||||
github.event.workflow_run.head_branch == 'main' ||
|
||||
startsWith(github.event.workflow_run.head_branch || '', 'v')
|
||||
)
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
attestations: write
|
||||
id-token: write
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- os: self-hosted
|
||||
arch: amd64
|
||||
- os: ubuntu-24.04-arm
|
||||
arch: arm64
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
- name: Log in to the GitHub Container registry
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Extract metadata (tags, labels) for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v6
|
||||
with:
|
||||
tags: |
|
||||
type=raw,value=latest,enable=${{ github.event.workflow_run.head_branch == 'main' }}
|
||||
type=semver,pattern={{raw}},value=${{ github.event.workflow_run.head_branch }},enable=${{ startsWith(github.event.workflow_run.head_branch || '', 'v') }}
|
||||
flavor: |
|
||||
latest=auto
|
||||
suffix=-${{ matrix.arch }},onlatest=true
|
||||
images: |
|
||||
ghcr.io/${{ github.repository }}
|
||||
|
||||
- name: Build and push Docker images
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
|
||||
docker-manifest:
|
||||
if: |
|
||||
github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.event == 'push' &&
|
||||
(
|
||||
github.event.workflow_run.head_branch == 'main' ||
|
||||
startsWith(github.event.workflow_run.head_branch || '', 'v')
|
||||
)
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
needs:
|
||||
- docker-build
|
||||
- docker-clean-metadata
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
image: ${{ fromJson(needs.docker-clean-metadata.outputs.json).tags }}
|
||||
|
||||
steps:
|
||||
- name: Log in to the GitHub Container registry
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Create and push manifest
|
||||
run: |
|
||||
docker buildx imagetools create -t ${{ matrix.image }} ${{ matrix.image }}-amd64 ${{ matrix.image }}-arm64
|
||||
56
.github/workflows/workflow.yml
vendored
@@ -1,56 +0,0 @@
|
||||
name: CI (main and tags)
|
||||
on:
|
||||
push:
|
||||
branches: [ "main" ]
|
||||
tags: [ "v*" ]
|
||||
schedule:
|
||||
- cron: '0 0 * * 1'
|
||||
permissions:
|
||||
checks: write
|
||||
contents: write
|
||||
packages: write
|
||||
pull-requests: read
|
||||
jobs:
|
||||
test-and-clippy:
|
||||
name: Unit testing and linting
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: dtolnay/rust-toolchain@stable
|
||||
- run: cargo test --all-features
|
||||
- run: cargo clippy
|
||||
|
||||
build-publish:
|
||||
name: Build and Publish
|
||||
runs-on: self-hosted
|
||||
steps:
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
with:
|
||||
platforms: arm64
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v1
|
||||
- name: Login to ghcr.io
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Extract metadata (tags, labels) for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/${{ github.repository }}
|
||||
registry.etke.cc/${{ github.repository }}
|
||||
tags: |
|
||||
type=raw,value=latest,enable=${{ github.ref_name == 'main' }}
|
||||
type=semver,pattern={{raw}}
|
||||
- name: Build and push
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
platforms: linux/amd64,linux/arm64
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
file: Dockerfile.ci
|
||||
36
.pre-commit-config.yaml
Normal file
@@ -0,0 +1,36 @@
|
||||
repos:
|
||||
# Fast built-in hooks (Rust-native, no dependencies)
|
||||
- repo: builtin
|
||||
hooks:
|
||||
- id: trailing-whitespace
|
||||
- id: end-of-file-fixer
|
||||
- id: check-yaml
|
||||
- id: check-merge-conflict
|
||||
- id: check-added-large-files
|
||||
args: ['--maxkb=1024']
|
||||
|
||||
# Local hooks that run project-specific tools
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: cargo-fmt-check
|
||||
name: Cargo Format Check
|
||||
entry: cargo fmt --all -- --check
|
||||
language: system
|
||||
files: '\.rs$'
|
||||
pass_filenames: false
|
||||
|
||||
- id: cargo-clippy
|
||||
name: Cargo Clippy
|
||||
entry: cargo clippy -- -D warnings
|
||||
language: system
|
||||
files: '\.rs$'
|
||||
pass_filenames: false
|
||||
priority: 100
|
||||
|
||||
- id: test-unit
|
||||
name: Unit Tests
|
||||
entry: just test
|
||||
language: system
|
||||
files: '\.rs$'
|
||||
pass_filenames: false
|
||||
priority: 100
|
||||
307
CHANGELOG.md
@@ -1,3 +1,310 @@
|
||||
# (2026-06-21) Version 1.22.0
|
||||
|
||||
- (**Feature**) Add a native [Venice](https://venice.ai) provider with [🖌️ image-generation](./docs/features.md#️-image-creation) (incl. editing), [💬 text-generation](./docs/features.md#-text-generation) (incl. vision), [🗣️ text-to-speech](./docs/features.md#️-text-to-speech), [🦻 speech-to-text](./docs/features.md#-speech-to-text), and Venice's native web search via the full `venice_parameters` knob set. Unlike the [OpenAI-compatible](./docs/providers.md#openai-compatible) path (which drops images and can't reach Venice's audio or native image endpoints), it talks to Venice's API directly, using the knob-rich native `/image/generate` and `/image/edit` endpoints. See the [Venice provider docs](./docs/providers.md#venice).
|
||||
|
||||
|
||||
# (2026-06-05) Version 1.21.1
|
||||
|
||||
- (**Security**) Update the [anthropic](https://github.com/etkecc/anthropic-rs) dependency to use [reqwest](https://crates.io/crates/reqwest) 0.12 / [rustls](https://crates.io/crates/rustls) 0.23, replacing the vulnerable `rustls-webpki` 0.101 line with 0.103.13. This resolves [`GHSA-82j2-j2ch-gfr8`](https://github.com/advisories/GHSA-82j2-j2ch-gfr8) (high — denial of service via panic on a malformed CRL), [`GHSA-xgp8-3hg3-c2mh`](https://github.com/advisories/GHSA-xgp8-3hg3-c2mh) and [`GHSA-965h-392x-2mh5`](https://github.com/advisories/GHSA-965h-392x-2mh5) (name-constraint validation issues).
|
||||
|
||||
|
||||
# (2026-06-05) Version 1.21.0
|
||||
|
||||
- (**Improvement**) Default to OpenAI's `gpt-image-2` model for image generation (in newly-created OpenAI agents and the sample provider configs).
|
||||
|
||||
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) from 0.40 to 0.41, which [resynchronizes with the upstream OpenAI API spec](https://github.com/64bit/async-openai/issues/557) after it had drifted out of sync — a mismatch that was already causing some breakage (hopefully now resolved). Adapts to the newly-added `gpt-image-2` image model and an `ImageSize` type change.
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-06-02) Version 1.20.0
|
||||
|
||||
- (**Internal Improvement**) Update [matrix-sdk](https://crates.io/crates/matrix-sdk) from 0.17 to 0.18 and [mxlink](https://crates.io/crates/mxlink) to 1.15.0.
|
||||
|
||||
- (**Internal Improvement**) Update [tiktoken-rs](https://crates.io/crates/tiktoken-rs) to 0.12, backporting OpenAI [tiktoken](https://github.com/openai/tiktoken) 0.13.0 for better alignment with upstream tokenization behavior.
|
||||
|
||||
- (**Internal Improvement**) Bump the pinned Rust toolchain from 1.95.0 to 1.96.0 (in `rust-toolchain.toml` and the Docker build images).
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-05-27) Version 1.19.3
|
||||
|
||||
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) to 0.40.2, pulling in several upstream fixes (streaming HTTP error surfacing, default `ResponseTextParam.format` deserialization, etc.).
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-05-21) Version 1.19.2
|
||||
|
||||
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) to 0.40.0.
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-05-09) Version 1.19.1
|
||||
|
||||
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) to 0.38.0.
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-05-09) Version 1.19.0
|
||||
|
||||
- (**Internal Improvement**) Update [matrix-sdk](https://crates.io/crates/matrix-sdk) from 0.16 to 0.17 and [mxlink](https://crates.io/crates/mxlink) to 1.14.0. matrix-sdk 0.17 dropped its `native-tls` feature and now uses [rustls](https://github.com/rustls/rustls) exclusively as its TLS backend.
|
||||
|
||||
- (**Internal Improvement**) Bump the pinned Rust toolchain from 1.93.0 to 1.95.0 (in `rust-toolchain.toml` and the Docker build images).
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-04-11) Version 1.18.0
|
||||
|
||||
- (**Bugfix**) Fix the bot not sending a welcome message when joining a room on homeservers (like [Continuwuity](https://continuwuity.org/)) that place the join membership event in the sync response's `state` block rather than the `timeline` block, via [mxlink](https://crates.io/crates/mxlink) 1.13.1
|
||||
|
||||
- (**Improvement**) Update [tiktoken-rs](https://crates.io/crates/tiktoken-rs) to 0.11, adding tokenization support for newer GPT models (gpt-5.x, codex, etc.) and fixing context sizes for o1-mini/chatgpt-4o/gpt-4.5
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
|
||||
# (2026-03-25) Version 1.17.0
|
||||
|
||||
- (**Feature**) Add `text-generation sender-context-mode` for attaching sender metadata to conversation messages. See the [💬 Text Generation](./docs/configuration/text-generation.md#-sender-context-mode) documentation for details. Thanks to [kschwank](https://github.com/kschwank) for the contribution in [#104](https://github.com/etkecc/baibot/pull/104)!
|
||||
|
||||
|
||||
# (2026-03-24) Version 1.16.1
|
||||
|
||||
- (**Bugfix**) Fix compatibility with [async-openai](https://crates.io/crates/async-openai) 0.34.0 by populating the new `phase` field required for OpenAI Responses API message inputs. baibot does not currently distinguish between assistant `commentary` and `final_answer` turns, so using `None` preserves the previous behavior while remaining compatible with the updated crate.
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-03-20) Version 1.16.0
|
||||
|
||||
- (**Feature**) Add support for file attachments (`m.file` Matrix messages) in conversations. Files like PDFs, text documents, spreadsheets, code files, etc. are now downloaded and forwarded to the LLM alongside the conversation context, similar to how images (`m.image`) are already handled. See the [💬 Text Generation](./docs/features.md#-text-generation) documentation for details and known limitations.
|
||||
|
||||
- (**Improvement**) Use the [mime_guess](https://crates.io/crates/mime_guess) crate for MIME type detection from file extensions, replacing a hand-maintained mapping. This covers hundreds of file extensions out of the box.
|
||||
|
||||
|
||||
|
||||
# (2026-03-07) Version 1.15.0
|
||||
|
||||
- (**Feature**) Add support for authentication via access tokens (for [Matrix Authentication Service](https://github.com/element-hq/matrix-authentication-service)/OIDC-enabled homeservers) as an alternative to password authentication. See [🔐 Authentication](./docs/configuration/authentication.md) for setup details. Thanks to [Taylor Southwick](https://github.com/twsouthwick) for the contribution in [#83](https://github.com/etkecc/baibot/pull/83)!
|
||||
|
||||
- (**Internal Improvement**) Pin the Rust toolchain to `1.93.0` in both CI and local development to avoid `matrix-sdk` build failures on newer stable toolchains.
|
||||
|
||||
- (**Internal Improvement**) Documentation updates.
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
|
||||
# (2026-02-18) Version 1.14.3
|
||||
|
||||
- (**Internal Improvement**) Add [Renovate](https://docs.renovatebot.com/) configuration for automated dependency updates
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
|
||||
# (2026-02-18) Version 1.14.2
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
- (**Internal Improvement**) Reorganize the development environment to support [Continuwuity](https://continuwuity.org/) as a homeserver choice (in addition to [Synapse](https://github.com/element-hq/synapse)). Continuwuity is now the default for its lighter footprint (no external database required). See [development docs](./docs/development.md) for details.
|
||||
|
||||
|
||||
# (2026-02-10) Version 1.14.1
|
||||
|
||||
- (**Security**) Dependency updates to fix security vulnerabilities ([time](https://crates.io/crates/time) stack exhaustion DoS, [bytes](https://crates.io/crates/bytes) integer overflow), via [mxlink](https://crates.io/crates/mxlink) 1.12.0
|
||||
|
||||
- (**Internal Improvement**) Switch from deprecated [serde_yaml](https://crates.io/crates/serde_yaml) to its maintained fork [serde_yaml_ng](https://crates.io/crates/serde_yaml_ng)
|
||||
|
||||
- (**Internal Improvement**) Add [prek](https://github.com/nicholasgasior/prek) pre-commit hooks via [mise](https://mise.jdx.dev/) for automated code quality checks (formatting, clippy, tests)
|
||||
|
||||
- (**Internal Improvement**) Fix clippy warnings and formatting issues
|
||||
|
||||
|
||||
# (2026-02-04) Version 1.14.0
|
||||
|
||||
- (**Feature**) The `openai` provider now uses OpenAI's [Responses API](https://platform.openai.com/docs/api-reference/responses) (instead of the older Chat Completions API), adding support for [🛠️ built-in tools](./docs/features.md#️-built-in-tools-openai-only) (`web_search` and `code_interpreter`). These tools are **disabled by default** and can be enabled via the `text_generation.tools` configuration (see the [sample configuration](https://github.com/etkecc/baibot/blob/c70387b0c38d8d0f30bba2179a2a21a3710dbeaf/docs/sample-provider-configs/openai.yml#L12-L15)). To enable tools on an existing agent, you need to [update the agent](./docs/agents.md#updating-agents) to re-create it with the `text_generation.tools` section added and enable the tools you need. Thanks to [Layla Manley](https://github.com/yeslayla) for the contribution in [#62](https://github.com/etkecc/baibot/pull/62)!
|
||||
|
||||
- (**Bugfix**) Fix sticker generation for newer GPT image models (`gpt-image-1`, `gpt-image-1-mini`, `gpt-image-1.5`) which don't support the previously hardcoded `256x256` size (minimum is `1024x1024`)
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
|
||||
# (2026-01-23) Version 1.13.0
|
||||
|
||||
- (**Improvement**) Extend auto-switching to support cheaper models (`gpt-image-1-mini`) for `gpt-image-1` and `gpt-image-1.5` when generating stickers ([e0b4a40](https://github.com/etkecc/baibot/commit/e0b4a40))
|
||||
|
||||
- (**Internal Improvement**) Upgrade Rust compiler (1.92.0 -> 1.93.0) ([691aeeb](https://github.com/etkecc/baibot/commit/691aeeb))
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
|
||||
# (2025-12-21) Version 1.12.0
|
||||
|
||||
- (**Improvement**) Upgrade [async-openai](https://crates.io/crates/async-openai) (0.31.1 -> 0.32.2) and add support for OpenAI's `gpt-image-1.5` model ([08c689a](https://github.com/etkecc/baibot/commit/08c689a), [f7bf3d7](https://github.com/etkecc/baibot/commit/f7bf3d7))
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
|
||||
# (2025-12-15) Version 1.11.0
|
||||
|
||||
- (**Feature**) Add support for custom avatars via file path and for keeping the already-set avatar (for those who wish to manage it by themselves via other means). See the [sample config](./etc/app/config.yml.dist) for details. ([062fbbb](https://github.com/etkecc/baibot/commit/062fbbb8ef9ad600db483a431c5c782402191023))
|
||||
|
||||
- (**Internal Improvement**) Dependency updates ([99bde53](https://github.com/etkecc/baibot/commit/99bde53ef648a5a9086a96778fde4a9dbc1ede58))
|
||||
|
||||
- (**Internal Improvement**) Documentation updates ([b3fd8e5](https://github.com/etkecc/baibot/commit/b3fd8e548f83fe46398ced4760d7e2bb7588c24d))
|
||||
|
||||
- (**Internal Improvement**) Upgrade Rust compiler (1.91.1 -> 1.92.0) ([22906aa](https://github.com/etkecc/baibot/commit/22906aa2d3cae51815fad2560a545eaa69c247b6))
|
||||
|
||||
|
||||
# (2025-12-06) Version 1.10.0
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.11.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.16.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.16.0).
|
||||
|
||||
# (2025-11-30) Version 1.9.0
|
||||
|
||||
- (**Internal Improvement**) Upgrade [async-openai](https://crates.io/crates/async-openai) from our own etkecc fork (0.28.1-patched) to the official upstream version 0.31.1. This upgrade required some code adaptations to the new module structure, etc. While tested, regressions are possible.
|
||||
|
||||
# (2025-11-28) Version 1.8.3
|
||||
|
||||
- (**Improvement**) Add support for the `BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY` environment variable for configuring `persistence.session_encryption_key`
|
||||
|
||||
- (**Improvement**) Add support for the `BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED` environment variable for configuring `user.encryption.recovery_reset_allowed`
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
# (2025-11-20) Version 1.8.2
|
||||
|
||||
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.89.0 -> 1.91.1).
|
||||
|
||||
# (2025-09-12) Version 1.8.1
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
# (2025-09-08) Version 1.8.0
|
||||
|
||||
- (**Internal Improvement**) Upgrade [mxlink](https://crates.io/crates/mxlink) (1.9.0 -> 1.10.0) and [matrix-sdk](https://crates.io/crates/matrix-sdk) (0.13.0 -> 0.14.0)
|
||||
|
||||
- (**Internal Improvement**) Upgrade [Rust](https://www.rust-lang.org/) (1.88.0 -> 1.89.0)
|
||||
|
||||
- (**Internal Improvement**) Upgrade Debian base for container images (12/bookworm -> 13/trixie)
|
||||
|
||||
# (2025-07-11) Version 1.7.6
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.9.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.13.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.13.0), which contains fixes for some security vulnerabilities)
|
||||
|
||||
# (2025-06-10) Version 1.7.5
|
||||
|
||||
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.86 -> 1.86).
|
||||
|
||||
# (2025-06-10) Version 1.7.4
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
# (2025-06-10) Version 1.7.3
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.8.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.12.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.12.0), which contains fixes for important security vulnerabilities)
|
||||
|
||||
# (2025-05-11) Version 1.7.2
|
||||
|
||||
- (**Bugfix**) Allow `image_generation.size` configuration value for OpenAI to be `null` to allow the model to choose the size automatically and default to that
|
||||
|
||||
# (2025-05-11) Version 1.7.1
|
||||
|
||||
- (**Bugfix**) Fix lack of documentation for the new [image-editing](./docs/features.md#-image-editing) feature in the `!bai usage` command's output
|
||||
|
||||
# (2025-05-10) Version 1.7.0
|
||||
|
||||
- (**Feature**) Add vision support to the OpenAI and Anthropic providers. You can now mix text and images in your conversations - fixes [issue #5](https://github.com/etkecc/baibot/issues/5)
|
||||
|
||||
- (**Feature**) Add [image-editing](./docs/features.md#-image-editing) support to the OpenAI provider
|
||||
|
||||
- (**Improvement**) Add compatibility with OpenAI's `gpt-image-1` model - fixes [issue #40](https://github.com/etkecc/baibot/issues/40)
|
||||
|
||||
- (**Change**) Rework [image-creation](./docs/features.md#-image-creation) to avoid command conflicts with [image-editing](./docs/features.md#-image-editing). The image-creation command syntax is now `!bai image create <prompt>` (previously: `!bai image <prompt>`).
|
||||
|
||||
- (**Internal Improvement**) Dependency and compiler updates
|
||||
|
||||
> [!WARNING]
|
||||
> Unlike other releases, this release is not published to [crates.io](https://crates.io), because it relies on multiple library forks (`async-openai` and `anthropic-rs`) sourced from Github.
|
||||
|
||||
|
||||
# (2025-04-12) Version 1.6.0
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.7.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.11.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.11.0))
|
||||
|
||||
|
||||
# (2025-03-31) Version 1.5.1
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
# (2025-02-27) Version 1.5.0
|
||||
|
||||
- (**Feature**) Add support for sending Speech-to-Text replies for [Transcribe-only mode](./docs/features.md#transcribe-only-mode) as regular text messages instead of notices and doing it so by default ([a1bd292752](https://github.com/etkecc/baibot/commit/a1bd292752bdd37a196788c73d00b5619e843a78)) - improvement for [issue #14](https://github.com/etkecc/baibot/issues/14). See [🦻 Speech-to-Text / 🪄 Message Type for non-threaded only-transcribed messages](./docs/configuration/speech-to-text.md#-message-type-for-non-threaded-only-transcribed-messages) for details.
|
||||
|
||||
- (**Feature**) Add config setting controlling if a self-introduction message is posted after joining a room ([c051da2f4a](https://github.com/etkecc/baibot/commit/c051da2f4a161de0974ebb917f7a52d01f5a001f)) - fixes [issue #32](https://github.com/etkecc/baibot/issues/32). You may wish to add a `room.post_join_self_introduction_enabled` property to your configuration. See the [sample config](./etc/app/config.yml.dist) for details. If unspecified, it defaults to `true` anyway which preserves the old behavior.
|
||||
|
||||
- (**Feature**) Add support for configuring `max_completion_tokens` for OpenAI ([47d8edea70](https://github.com/etkecc/baibot/commit/47d8edea705a44aa25a9bfaec4888c0f9ea8700e))
|
||||
|
||||
- (**Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.6.1 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.10.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.10.0))
|
||||
|
||||
- (**Improvement**) Populate image/audio attachment `body` with a filename, not with text to avoid incorrect rendering in Element Web, etc. ([ec1879d212](https://github.com/etkecc/baibot/commit/ec1879d212fa8d6e5f8590486e94c72abfcb75a5))
|
||||
|
||||
- (**Improvement**) Replace Anthropic library ([anthropic-rs](https://crates.io/crates/anthropic-rs) -> [anthropic](https://crates.io/crates/anthropic)) and switch default recommended model (`claude-3-5-sonnet-20240620` -> `claude-3-7-sonnet-20250219`) ([692d61b239](https://github.com/etkecc/baibot/commit/692d61b2398f073b81d32d4cbe8145ab3929e48c)) - fixes [issue #22](https://github.com/etkecc/baibot/issues/22)
|
||||
|
||||
- (**Internal Improvement**) Switch to native building of `arm64` container images to decrease total build times from ~40 minutes to ~8 minutes ([6719538530b](https://github.com/etkecc/baibot/commit/6719538530bf76b3ff2d24077b2a7fa868276b79))
|
||||
|
||||
- (**Internal Improvement**) Various other internal changes, including upgrading [Rust from 1.82 to 1.85 and switching to Rust edition 2024](https://blog.rust-lang.org/2025/02/20/Rust-1.85.0.html)
|
||||
|
||||
|
||||
# (2024-12-12) Version 1.4.1
|
||||
|
||||
- (**Bugfix**) Fix detection for whether the bot is the last member in a room, to avoid incorrectly leaving multi-user rooms that have had at least one person `leave` ([3c47d40781](https://github.com/etkecc/baibot/commit/3c47d407819aa9c0121117a411858238724f06da))
|
||||
|
||||
|
||||
# (2024-11-19) Version 1.4.0
|
||||
|
||||
- (**Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.4.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.8.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.8.0)). Once you run this version at least once and your matrix-sdk datastore gets upgraded to the new schema, **you will not be able to downgrade to older baibot versions** (based on the older matrix-sdk), unless you start with an empty datastore.
|
||||
|
||||
- (**Bugfix**) Add missing typing notices sending functionality while generating images ([9d166e35ba](https://github.com/etkecc/baibot/commit/9d166e35ba6fc0daaf69318870e92436f3302056))
|
||||
|
||||
- (**Feature**) Support for [Matrix authenticated media](https://matrix.org/docs/spec-guides/authed-media-servers/), thanks to upgrading [mxlink](https://crates.io/crates/mxlink) / [matrix-sdk](https://crates.io/crates/matrix-sdk) - fixes [issue #12](https://github.com/etkecc/baibot/issues/12)
|
||||
|
||||
|
||||
# (2024-11-12) Version 1.3.2
|
||||
|
||||
Dependency updates.
|
||||
|
||||
|
||||
# (2024-10-03) Version 1.3.1
|
||||
|
||||
- (**Improvement**) Improves fallback user mentions support for old clients (like Element iOS) which use the bot's display name (not its full Matrix User ID). ([d9a045a5e4](https://github.com/etkecc/baibot/commit/d9a045a5e41d2b99694f92ec9e90f47529546d89))
|
||||
|
||||
|
||||
# (2024-10-03) Version 1.3.0
|
||||
|
||||
**TLDR**: you can now use OpenAI's [o1](https://platform.openai.com/docs/models/o1) models, benefit from [prompt caching](https://platform.openai.com/docs/guides/prompt-caching) and mention the bot again from old clients lacking proper [user mentions support](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions) (like Element iOS).
|
||||
|
||||
- (**Feature**) Introduces a new `baibot_conversation_start_time_utc` [prompt variable](./docs/configuration/text-generation.md#️-prompt-override) which is not a moving target (like the `baibot_now_utc` variable) and allows [prompt caching](https://platform.openai.com/docs/guides/prompt-caching) to work. All default/sample configs have been adjusted to make use of this new variable, but users need to adjust your existing dynamically-created agents to start using it. ([85e66406dc](https://github.com/etkecc/baibot/commit/85e66406dc6f430741c7819f420e2df4ae6e8d3b))
|
||||
|
||||
- (**Improvement**) Allows for the `max_response_tokens` configuration value for the [OpenAI provider](./docs/providers.md#openai) to be set to `null` to allow [o1](https://platform.openai.com/docs/models/o1) models (which do not support `max_response_tokens`) to be used. See the new o1 sample config [here](./docs/sample-provider-configs/openai-o1.yml). ([db9422740c](https://github.com/etkecc/baibot/commit/db9422740ceca32956d9628b6326b8be206344e2))
|
||||
|
||||
- (**Improvement**) Switches the sample configs for the [OpenAI provider](./docs/providers.md#openai) to point to the `gpt-4o` model, which since 2024-10-02 is the same as the `gpt-4o-2024-08-06` model. We previously explicitly pointed the bot to the `gpt-4o-2024-08-06` model, because it was much better (longer context window). Now that `gpt-4o` points to the same powerful model, we don't need to pin its version anymore. Existing users may wish to adjust their configuration to match. ([90fbad5b64](https://github.com/etkecc/baibot/commit/90fbad5b643cd06c23179f055a309ec6a7cba161))
|
||||
|
||||
- (**Bugfix**) Restores fallback user mentions support (via regular text, not via the [user mentions spec](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions)) to allow certain old clients (like Element iOS) to be able to mention the bot again. Support for this was intentionally removed recently (in [v1.2.0](#2024-10-01-version-120)), but it turned out to be too early to do this. ([b40226826f](https://github.com/etkecc/baibot/commit/b40226826fe914d0d5d265230ebc5bac8058b6f7))
|
||||
|
||||
|
||||
# (2024-10-01) Version 1.2.0
|
||||
|
||||
- (**Feature**) Adds support for [on-demand involvement](./docs/features.md#on-demand-involvement) of the bot (via mention) in arbitrary threads and reply chains ([9908512968](https://github.com/etkecc/baibot/commit/990851296828168c2106eb3f4668833e9e5a7463)) - fixes [issue #15](https://github.com/etkecc/baibot/issues/15)
|
||||
|
||||
- (**Improvement**) Simplifies [Transcribe-only mode](./docs/features.md#transcribe-only-mode) reply format (removing `> 🦻` prefixing) to allow easier forwarding, etc. ([e6aa956423](https://github.com/etkecc/baibot/commit/e6aa95642376ee7d87932d0e66dcfedf261b188b)) - fixes [issue #14](https://github.com/etkecc/baibot/issues/14)
|
||||
|
||||
- (**Bugfix**) Fixes speech-to-text replies rendering incorrectly in certain clients, due to them confusing our old reply format with [fallback for rich replies](https://spec.matrix.org/v1.11/client-server-api/#fallbacks-for-rich-replies) ([e6aa956423](https://github.com/etkecc/baibot/commit/e6aa95642376ee7d87932d0e66dcfedf261b188b)) - fixes [issue #17](https://github.com/etkecc/baibot/issues/17)
|
||||
|
||||
|
||||
# (2024-09-22) Version 1.1.1
|
||||
|
||||
- (**Bugfix**) Fix thread messages being lost due to lack of pagination support ([d4ddd29660](https://github.com/etkecc/baibot/commit/d4ddd29660d9f51d248119dd6032e68ab29e7d35)) - fixes [issue #13](https://github.com/etkecc/baibot/issues/13)
|
||||
|
||||
3748
Cargo.lock
generated
27
Cargo.toml
@@ -7,32 +7,37 @@ license = "AGPL-3.0-or-later"
|
||||
readme = "README.md"
|
||||
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
||||
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
||||
version = "1.1.1"
|
||||
edition = "2021"
|
||||
version = "1.22.0"
|
||||
edition = "2024"
|
||||
|
||||
[lib]
|
||||
name = "baibot"
|
||||
path = "src/lib.rs"
|
||||
|
||||
[dependencies]
|
||||
anthropic-rs = "0.1.*"
|
||||
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
||||
anyhow = "1.0.*"
|
||||
async-openai = "0.24.*"
|
||||
async-openai = { version = "0.41.0", features = ["audio", "chat-completion", "image", "responses"] }
|
||||
base64 = "0.22.*"
|
||||
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
||||
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
||||
matrix-sdk = { version = "0.7.1", default-features = false }
|
||||
matrix-sdk = { version = "0.18.0", default-features = false }
|
||||
mime_guess = "2.0.*"
|
||||
mxidwc = "1.0.*"
|
||||
mxlink = ">=1.3.0"
|
||||
mxlink = ">=1.15.0"
|
||||
etke_openai_api_rust = "0.1.*"
|
||||
quick_cache = "0.6.*"
|
||||
regex = "1.10.*"
|
||||
regex = "1.12.*"
|
||||
# Direct dep for the native `venice` provider's HTTP client. Pinned to 0.12 (the version
|
||||
# async-openai 0.41 already resolves) with rustls only and default-features off, so we ride
|
||||
# the existing reqwest+rustls copy instead of pulling a second TLS stack (native-tls/openssl).
|
||||
reqwest = { version = "0.12.*", default-features = false, features = ["json", "multipart", "rustls-tls"] }
|
||||
serde = { version = "1.0.*", features = ["derive"], default-features = false }
|
||||
serde_json = "1.0.*"
|
||||
serde_yaml = "0.9.*"
|
||||
tempfile = "3.12.*"
|
||||
tiktoken-rs = { version = "0.5.*", features = ["async-openai"] }
|
||||
tokio = { version = "1.40.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||
serde_yaml_ng = "0.10.*"
|
||||
tempfile = "3.27.*"
|
||||
tiktoken-rs = { version = "0.12.*", default-features = false }
|
||||
tokio = { version = "1.52.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||
tracing = "0.1.*"
|
||||
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
|
||||
url = "2.5.*"
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/rust:1.81.0-slim-bookworm AS build
|
||||
FROM docker.io/rust:1.96.0-slim-trixie AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||
|
||||
@@ -39,7 +39,7 @@ RUN --mount=type=cache,target=/target,sharing=locked \
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/debian:bookworm-slim
|
||||
FROM docker.io/debian:trixie-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y ca-certificates sqlite3 && \
|
||||
apt-get clean && \
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/rust:1.81.0-slim-bookworm AS build
|
||||
FROM docker.io/rust:1.96.0-slim-trixie AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||
|
||||
@@ -20,7 +20,7 @@ RUN cargo build --release
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/debian:bookworm-slim
|
||||
FROM docker.io/debian:trixie-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y ca-certificates sqlite3 && \
|
||||
apt-get clean && \
|
||||
|
||||
@@ -13,14 +13,14 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
|
||||
|
||||
## 🌟 Features
|
||||
|
||||
- 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai) and [☁️ many more](./docs/providers.md#️-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**:
|
||||
- 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai), [Venice](./docs/providers.md#venice) and [☁️ many more](./docs/providers.md#️-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**:
|
||||
|
||||
- Supports **different use purposes** (depending on the [☁️ provider](./docs/providers.md) & model):
|
||||
|
||||
- [💬 text-generation](./docs/features.md#-text-generation): communicating with you via text
|
||||
- [💬 text-generation](./docs/features.md#-text-generation): communicating with you via text (though certain models may "see" images as well). The [OpenAI provider](./docs/providers.md#openai) also supports [🛠️ built-in tools](./docs/features.md#️-built-in-tools-openai-only) (web search, code interpreter)
|
||||
- [🦻 speech-to-text](./docs/features.md#-speech-to-text): turning your voice messages into text
|
||||
- [🗣️ text-to-speech](./docs/features.md#%EF%B8%8F-text-to-speech): turning bot or users text messages into voice messages
|
||||
- [🖌️ image-generation](./docs/features.md#%EF%B8%8F-image-generation): generating images based on instructions
|
||||
- [🖌️ image-generation](./docs/features.md#image-generation): creating and editing images based on instructions
|
||||
|
||||
- 🪄 Supports [seamless voice interaction](./docs/features.md#seamless-voice-interaction) (turning user voice messages into text, answering in text, then turning that text back into voice)
|
||||
|
||||
@@ -41,7 +41,7 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
|
||||
|
||||

|
||||
|
||||
You can find more screenshots on the the [🌟 Features](./docs/features.md) and other [📚 Documentation](./docs/README.md) pages, as well as in the [docs/screenshots](./docs/screenshots) directory.
|
||||
You can find more screenshots on the [🌟 Features](./docs/features.md) and other [📚 Documentation](./docs/README.md) pages, as well as in the [docs/screenshots](./docs/screenshots) directory.
|
||||
|
||||
|
||||
## 🚀 Getting Started
|
||||
|
||||
@@ -16,6 +16,7 @@ Users:
|
||||
|
||||
- ✅ can **invite the bot to rooms**
|
||||
- ✅ can **use all the bot's [features](./features.md)** ([💬 Text Generation](./features.md#-text-generation), [🦻 Speech-to-Text](./features.md#-speech-to-text), etc.) by sending room messages
|
||||
- ✅ can **mention the bot** in threads and reply chains to provoke it to respond to non-user messages (see [🌟 Features / 💬 Text Generation / On-demand involvement](./features.md#on-demand-involvement))
|
||||
- ✅ can **change the bot's configuration in a room** (e.g. `!bai config room ...` commands)
|
||||
- ❌ cannot **change the bot's global configuration** (e.g. `!bai config global ...` commands)
|
||||
- ❌ cannot **create new [🤖 Agents](./agents.md)** (neither in rooms, nor globally). See [💼 Room-local agent managers](#-room-local-agent-managers) for controlling which users can create agents.
|
||||
@@ -42,7 +43,8 @@ Administrators cannot be changed without adjusting the bot's configuration on th
|
||||
|
||||
Room-local agent managers are users privileged to **create their own [agents](./agents.md)** (see `!bai agent`) in rooms.
|
||||
|
||||
**⚠️ WARNING**: Letting regular users create agents which contact arbitrary network services **may be a security issue**.
|
||||
> [!WARNING]
|
||||
> Letting regular users create agents which contact arbitrary network services **may be a security issue**.
|
||||
|
||||
The following commands are available:
|
||||
- **Show** the currently allowed users: `!bai access room-local-agent-managers`
|
||||
|
||||
@@ -35,7 +35,7 @@ Depending on where the agent is defined (within a room, globally, or [statically
|
||||
|
||||
When creating an agent, you will be given some sample [YAML](https://en.wikipedia.org/wiki/YAML) configuration which you can use to customize the agent's behavior.
|
||||
|
||||
This configuration varies depending on the [☁️ provider](./providers.md) used and the capabilities of the agent. Based on the configuration keys you pass, certain features will be enabled or disabled. For example, if you skip the `image_generation` key for an [OpenAI](./providers.md#openai) agent, it won't be able to generate images (see [🖌️ Image Generation](./features.md#-image-generation)).
|
||||
This configuration varies depending on the [☁️ provider](./providers.md) used and the capabilities of the agent. Based on the configuration keys you pass, certain features will be enabled or disabled. For example, if you skip the `image_generation` key for an [OpenAI](./providers.md#openai) agent, it won't be able to generate images (see [🖌️ Image Creation](./features.md#-image-creation), [🎨 Image Editing](./features.md#-image-editing), [🫵 Sticker Creation](./features.md#-sticker-creation)).
|
||||
|
||||
After making your modifications to the sample YAML, you submit it back to the bot and the new agent will be created.
|
||||
|
||||
|
||||
@@ -12,12 +12,17 @@ This file is created from the template found in [etc/app/config.yml.dist](../../
|
||||
|
||||
Certain keys can be left unset, in which case [📝 hardcoded defaults](../../src/entity/cfg/defaults.rs) would be used.
|
||||
|
||||
Each configuration key found in the YAML configuration can be overridden by setting an environment variable (dots should be replaced with `_`). Example:
|
||||
Some configuration keys found in the YAML configuration can be overridden by setting an environment variable (dots should be replaced with `_`). Example:
|
||||
|
||||
- to override `command_prefix`, set an environment variable `BAIBOT_COMMAND_PREFIX`
|
||||
- to override `homeserver.server_name`, set an environment variable `BAIBOT_HOMESERVER_SERVER_NAME`
|
||||
|
||||
The static configuration contains an `initial_global_config` key, which is used to populate the bot's global configuration (stored as [dynamic configuration](#dynamic-configuration)) the first time the bot starts. Modifying this subsequently will not have any effect. After initial global configuration creation, it's expected to be managed dynamically via chat commands.
|
||||
You can see the list of supported environment variables in the [🦀 src/entity/cfg/env.rs](../../src/entity/cfg/env.rs) file.
|
||||
|
||||
> [!WARNING]
|
||||
> The static configuration contains an `initial_global_config` key, which is used to populate the bot's global configuration (stored as [dynamic configuration](#dynamic-configuration)) the first time the bot starts. Modifying this subsequently will not have any effect. After initial global configuration creation, it's expected to be managed dynamically via chat commands.
|
||||
|
||||
For Matrix-account authentication setup, see [🔐 Authentication](./authentication.md).
|
||||
|
||||
|
||||
### Dynamic configuration
|
||||
@@ -40,7 +45,7 @@ You can adjust the following settings per room and/or globally:
|
||||
- [💬 Text Generation](text-generation.md)
|
||||
- [🦻 Speech-to-Text](speech-to-text.md)
|
||||
- [🗣️ Text-to-Speech](text-to-speech.md)
|
||||
- [🖌️ Image Generation](image-generation.md)
|
||||
- [🖌️ Image Creation](image-generation.md)
|
||||
- [🤝 Handlers](handlers.md)
|
||||
|
||||
Refer to the bot's help messages (as a response to a `!bai config` help command) for the most up-to-date information on what Room Settings can be configured.
|
||||
|
||||
23
docs/configuration/authentication.md
Normal file
@@ -0,0 +1,23 @@
|
||||
## 🔐 Authentication
|
||||
|
||||
baibot supports 2 authentication modes for the Matrix account (`user.*` keys in config).
|
||||
|
||||
Set **exactly one** mode. If both are set (or neither is set), startup validation fails.
|
||||
|
||||
### Password authentication
|
||||
|
||||
- Config key: `user.password`
|
||||
- Environment variable: `BAIBOT_USER_PASSWORD`
|
||||
|
||||
### Access token authentication
|
||||
|
||||
- Config keys: `user.access_token` + `user.device_id`
|
||||
- Environment variables: `BAIBOT_USER_ACCESS_TOKEN` + `BAIBOT_USER_DEVICE_ID`
|
||||
|
||||
Access-token authentication is useful for OIDC-enabled homeservers (e.g. those using [Matrix Authentication Service](https://github.com/element-hq/matrix-authentication-service)).
|
||||
|
||||
Example token-generation command:
|
||||
|
||||
```sh
|
||||
mas-cli manage issue-compatibility-token <username> [device_id]
|
||||
```
|
||||
@@ -8,10 +8,10 @@ You can also use **different models within the same room** (e.g. [💬 text-gene
|
||||
|
||||
The bot supports the following use-purposes:
|
||||
|
||||
- [💬 text-generation](../features.md#-text-generation): communicating with you via text
|
||||
- [💬 text-generation](../features.md#-text-generation): communicating with you via text (though certain models may also process images and files)
|
||||
- [🦻 speech-to-text](../features.md#-speech-to-text): turning your voice messages into text
|
||||
- [🗣️ text-to-speech](../features.md#️-text-to-speech): turning bot or users text messages into voice messages
|
||||
- [🖌️ image-generation](../features.md#-image-generation): generating images based on instructions
|
||||
- [🖌️ image-generation](../features.md#image-generation): generating images based on instructions
|
||||
|
||||
In a given room, each different purpose can be served by a different [provider](../providers.md) and model. This combination of provider and model configuration is called an [🤖 agent](../agents.md). Each purpose can be served by a different **handler** agent.
|
||||
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
|
||||
## 🖌️ Image Generation
|
||||
## Image Generation
|
||||
|
||||
The Image Generation feature is not configurable at this moment.
|
||||
The Image Creation and Image Editing features are not configurable at this moment.
|
||||
|
||||
You may also wish to see:
|
||||
|
||||
- [🌟 Features / 🖌️ Image Generation](../features.md#-image-generation) for a higher-level introduction to the Image Generation features
|
||||
- [📖 Usage / 🖌️ Image Generation](../usage.md#-image-generation) section for more details on how to use the bot for Image Generation in a room
|
||||
- [🌟 Features / Image Generation / 🖌️ Image Creation](../features.md#-image-creation) for a higher-level introduction to the Image Creation features
|
||||
- [🌟 Features / Image Generation / 🎨 Image Editing](../features.md#-image-editing) for a higher-level introduction to the Image Editing features
|
||||
- [📖 Usage / Image Generation / 🖌️ Creating Images](../usage.md#-creating-images) section for more details on how to use the bot for Image Creation in a room
|
||||
- [📖 Usage / Image Generation / 🎨 Editing images](../usage.md#-editing-images) section for more details on how to use the bot for Image Editing in a room
|
||||
|
||||
@@ -23,6 +23,19 @@ The following configuration values are recognized:
|
||||
Example: `!bai config room speech-to-text set-flow-type ignore` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
|
||||
|
||||
|
||||
### 🪄 Message Type for non-threaded only-transcribed messages
|
||||
|
||||
Controls how the transcribed text of voice messages is sent to the chat when Flow Type = `only_transcribe`.
|
||||
|
||||
The following configuration values are recognized:
|
||||
|
||||
- (default) `text`: the transcribed text is sent as a regular message. This is more convenient if you'd like to forward the transcribed message to other rooms.
|
||||
|
||||
- `notice`: the transcribed text is sent as a notice message. This provides better compatibility with other bots in the room, as they are less likely to interact with messages of type notice.
|
||||
|
||||
Example: `!bai config room speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages notice` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
|
||||
|
||||
|
||||
### 🔤 Language
|
||||
|
||||
Lets you specify the language of the input voice messages, to avoid using auto-detection.
|
||||
|
||||
@@ -13,7 +13,7 @@ You may also wish to see:
|
||||
|
||||
In Direct Message rooms with the bot (1:1 rooms), it most usually makes sense for the bot to respond to **all** of your messages, as shown on this [🖼️ screenshot](../screenshots/text-generation.webp).
|
||||
|
||||
In group rooms (with multiple users), it may be more appropriate for the bot to only respond to messages that are **prefixed** with the command prefix (e.g. `!bai`), so that other chat exchange in the room will not trigger it. Such a setup is shown on this [🖼️ screenshot](../screenshots/text-generation-prefix-requirement.webp).
|
||||
In group rooms (with multiple users), it may be more appropriate for the bot to only respond to messages that are **prefixed** with the command prefix (e.g. `!bai`) or which are [mentioning](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions) the bot (e.g. `@baibot`), so that other chat exchange in the room will not trigger it. Such a setup is shown on the [🖼️ On-demand involvement in the room](../screenshots/text-generation-prefix-requirement.webp) screenshot.
|
||||
|
||||
There are exceptions to these rules, and you can configure the bot to respond only to prefixed messages in a 1:1 room, or to respond to all messages even in a multi-user group room.
|
||||
|
||||
@@ -27,7 +27,10 @@ By default, the bot is **auto-configured (upon joining a new room)** to use the
|
||||
|
||||
Example: `!bai config room text-generation set-prefix-requirement-type command_prefix` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
|
||||
|
||||
Regardless of this configuration, **the bot will also respond to messages which directly [mention](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions) the bot** (e.g. `@baibot`), even if they are not prefixed. An example of this can be seen on this [🖼️ screenshot](../screenshots/text-generation-prefix-requirement.webp).
|
||||
Regardless of this configuration, **the bot will also respond to messages by allowed [👥 Users](../access.md#-users) which directly [mention](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions) the bot** (e.g. `@baibot`), even if they are not prefixed. An example of this can be seen on these screenshots:
|
||||
|
||||
- [🖼️ On-demand involvement in a thread](../screenshots/text-generation-on-demand-thread-involvement.webp)
|
||||
- [🖼️ On-demand involvement in a reply chain](../screenshots/text-generation-on-demand-reply-involvement.webp)
|
||||
|
||||
|
||||
### 🪄 Auto Usage
|
||||
@@ -54,6 +57,25 @@ This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_langua
|
||||
This setting is **disabled by default**, but can be enabled via `!bai config room text-generation set-context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
|
||||
|
||||
|
||||
### 👤 Sender Context Mode
|
||||
|
||||
In multi-user rooms, it may be useful for the model to know which participant sent each message in the conversation context.
|
||||
|
||||
To support this, the bot has a `text-generation sender-context-mode` setting, which can be set to:
|
||||
|
||||
- (default) `disabled`: do not attach sender metadata to messages before sending them to the model
|
||||
|
||||
- `matrix_user_id`: prefix text messages with the sender's Matrix user ID, for example: `[sender=@alice:example.com] Hello bot`
|
||||
|
||||
- `matrix_user_id_and_timestamp`: prefix text messages with the sender's Matrix user ID and the message timestamp, for example: `[sender=@alice:example.com sent_at=2026-03-23T14:30:00Z] Hello bot`
|
||||
|
||||
This sender metadata is attached to conversation messages before they are sent to the model provider. It applies to user and assistant text messages, but not to system prompts or non-text content.
|
||||
|
||||
⚠️ Enabling this sends Matrix user IDs, and optionally timestamps, to the model provider.
|
||||
|
||||
Example: `!bai config room text-generation set-sender-context-mode matrix_user_id` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
|
||||
|
||||
|
||||
### ⌨️ Prompt Override
|
||||
|
||||
You can override the [system prompt](https://huggingface.co/docs/transformers/en/tasks/prompting) configured at the [🤖 agent](../agents.md) level.
|
||||
@@ -74,11 +96,14 @@ Prompts may contain the following **placeholder variables** which will be replac
|
||||
|---------------------------|-------------|---------|
|
||||
| `{{ baibot_name }}` | Name of the bot as configured in the `user.name` field in the [Static configuration](./README.md#static-configuration) | `Baibot` |
|
||||
| `{{ baibot_model_id }}` | Text-Generation model ID as configured in the [🤖 agent](../agents.md)'s configuration | `gpt-4o` |
|
||||
| `{{ baibot_now_utc }}` | Current date and time in UTC | `2024-09-20 (Friday), 14:26:42 UTC (local timezone/time: unknown)` |
|
||||
| `{{ baibot_now_utc }}` | Current date and time in UTC (⚠️ usage may break prompt caching - see below) | `2024-09-20 (Friday), 14:26:42 UTC` |
|
||||
| `{{ baibot_conversation_start_time_utc }}` | The date and time in UTC that the conversation started | `2024-09-20 (Friday), 14:26:42 UTC` |
|
||||
|
||||
💡 `{{ baibot_now_utc }}` changes as time goes on, which prevents [prompt caching](https://platform.openai.com/docs/guides/prompt-caching) from working. It's better to use `{{ baibot_conversation_start_time_utc }}` in prompts, as its value doesn't change yet still orients the bot to the current date/time.
|
||||
|
||||
Here's a prompt that combines some of the above variables:
|
||||
|
||||
> You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
> You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
|
||||
|
||||
### 🌡️ Temperature Override
|
||||
|
||||
@@ -56,7 +56,7 @@ Example: `!bai config room text-to-speech set-speed-override 1.5` (this can also
|
||||
|
||||
### 👫 Voice override
|
||||
|
||||
The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (usually `onyx` when using [OpenAI](../providers.md#openai)).
|
||||
The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (e.g. `onyx` when using [OpenAI](../providers.md#openai), or `af_sky` when using [Venice](../providers.md#venice)).
|
||||
|
||||
Possible values (e.g. `onyx`) depend on the model you're using. For example, for [OpenAI](../providers.md#openai)'s Whisper model, [these voices](https://platform.openai.com/docs/guides/text-to-speech/voice-options) are available.
|
||||
|
||||
|
||||
@@ -18,6 +18,27 @@ For local development, we run all dependency services in [🐋 Docker](https://w
|
||||
- (Optional) an API key for some Large Language Model [☁️ provider](./providers.md) (e.g. [OpenAI](./providers.md#openai)), though we recommend using [LocalAI](#localai) or [Ollama](#ollama) for local development
|
||||
|
||||
|
||||
### Choosing a homeserver
|
||||
|
||||
The development environment supports two homeserver implementations:
|
||||
|
||||
- **[Continuwuity](https://continuwuity.org/)** (default) — lightweight, no external database required. Good for most development needs.
|
||||
- **[Synapse](https://github.com/element-hq/synapse)** — the reference implementation, bundled with Postgres. Use this if you need Synapse-specific behavior.
|
||||
|
||||
To choose a homeserver (optional — defaults to Continuwuity if skipped):
|
||||
|
||||
```sh
|
||||
just homeserver-init continuwuity # or: just homeserver-init synapse
|
||||
```
|
||||
|
||||
The choice is stored in `var/homeserver` and affects all subsequent commands.
|
||||
|
||||
> **Note:** If you switch homeservers after initial setup, you will need to:
|
||||
> - Delete `var/app/local/` and/or `var/app/container/` (app config and data)
|
||||
> - Delete `var/services/element-web/` (to regenerate its config)
|
||||
> - Re-run the prepare and user registration steps
|
||||
|
||||
|
||||
### Getting started guide
|
||||
|
||||
Developing [locally](#running-locally) is possible, but requires a [Rust](https://www.rust-lang.org/) toolchain.
|
||||
@@ -28,11 +49,12 @@ In any case, you will need [🐋 Docker](https://www.docker.com/) as [dependency
|
||||
|
||||
#### Running locally
|
||||
|
||||
1. Start the core dependency services (Postgres, Synapse, Element Web): `just services-start`
|
||||
2. (Only the first time around) Prepare initial app configuration in `var/app/local/config.yml`: `just app-local-prepare`
|
||||
3. (Only the first time around) [Prepare your configuration file](#prepare-your-configuration-file)
|
||||
4. (Only the first time around) Prepare initial default Matrix user accounts (`admin` and `baibot`): `just users-prepare`
|
||||
5. (Optional) Start additional services depending on which [agent provider you've chosen](#choosing-an-agent-provider):
|
||||
1. (Optional) Choose a homeserver: `just homeserver-init continuwuity` (or `synapse`). Default is `continuwuity`.
|
||||
2. Start the homeserver and Element Web: `just services-start`
|
||||
3. (Only the first time around) Prepare initial app configuration in `var/app/local/config.yml`: `just app-local-prepare`
|
||||
4. (Only the first time around) [Prepare your configuration file](#prepare-your-configuration-file)
|
||||
5. (Only the first time around) Prepare initial default Matrix user accounts (`admin` and `baibot`): `just users-prepare`
|
||||
6. (Optional) Start additional services depending on which [agent provider you've chosen](#choosing-an-agent-provider):
|
||||
- for [LocalAI](#localai):
|
||||
- Start services: `just localai-start`
|
||||
- Wait a while for LocalAI to start up. It has a lot of models to download. Monitor progress using `just localai-tail-logs`
|
||||
@@ -40,12 +62,12 @@ In any case, you will need [🐋 Docker](https://www.docker.com/) as [dependency
|
||||
- for [Ollama](#ollama):
|
||||
- Start services: `just ollama-start`
|
||||
- (Only the first time around) Pull the model configured in `agents.static_definitions` in the configuration file: `just ollama-pull-model gemma2:2b`
|
||||
6. Start the bot: `just run-locally`
|
||||
7. Go to http://element.127.0.0.1.nip.io:42025/ and login with `admin` / `admin`
|
||||
8. Create a new room and invite `@baibot:synapse.127.0.0.1.nip.io`
|
||||
9. When done, stop the bot (`Ctrl` + `C`)
|
||||
10. Stop the core dependency services: `just services-stop`
|
||||
11. (Optional) Stop additional services:
|
||||
7. Start the bot: `just run-locally`
|
||||
8. Go to http://element.127.0.0.1.nip.io:42025/ and login with `admin` / `admin`
|
||||
9. Create a new room and invite `@baibot:continuwuity.127.0.0.1.nip.io` (or `@baibot:synapse.127.0.0.1.nip.io` if using Synapse)
|
||||
10. When done, stop the bot (`Ctrl` + `C`)
|
||||
11. Stop the services: `just services-stop`
|
||||
12. (Optional) Stop additional services:
|
||||
- for [LocalAI](#localai): `just localai-stop`
|
||||
- for [Ollama](#ollama): `just ollama-stop`
|
||||
|
||||
@@ -54,11 +76,12 @@ In any case, you will need [🐋 Docker](https://www.docker.com/) as [dependency
|
||||
|
||||
You can avoid having a [Rust](https://www.rust-lang.org/) toolchain installed locally and build/run this in a container.
|
||||
|
||||
1. Start the core dependency services (Postgres, Synapse, Element Web): `just services-start`
|
||||
2. (Only the first time around) Prepare initial app configuration in `var/app/container/config.yml`: `just app-container-prepare`
|
||||
3. (Only the first time around) [Prepare your configuration file](#prepare-your-configuration-file)
|
||||
4. (Only the first time around) Prepare initial default Matrix user accounts (`admin` and `baibot`): `just users-prepare`
|
||||
5. (Optional) Start additional services depending on which [agent provider you've chosen](#choosing-an-agent-provider):
|
||||
1. (Optional) Choose a homeserver: `just homeserver-init continuwuity` (or `synapse`). Default is `continuwuity`.
|
||||
2. Start the homeserver and Element Web: `just services-start`
|
||||
3. (Only the first time around) Prepare initial app configuration in `var/app/container/config.yml`: `just app-container-prepare`
|
||||
4. (Only the first time around) [Prepare your configuration file](#prepare-your-configuration-file)
|
||||
5. (Only the first time around) Prepare initial default Matrix user accounts (`admin` and `baibot`): `just users-prepare`
|
||||
6. (Optional) Start additional services depending on which [agent provider you've chosen](#choosing-an-agent-provider):
|
||||
- for [LocalAI](#localai):
|
||||
- Start services: `just localai-start`
|
||||
- Wait a while for LocalAI to start up. It has a lot of models to download. Monitor progress using `just localai-tail-logs`
|
||||
@@ -66,12 +89,12 @@ You can avoid having a [Rust](https://www.rust-lang.org/) toolchain installed lo
|
||||
- for [Ollama](#ollama):
|
||||
- Start services: `just ollama-start`
|
||||
- (Only the first time around) Pull the model configured in `agents.static_definitions` in the configuration file: `just ollama-pull-model gemma2:2b`
|
||||
6. Start the bot: `just run-in-container`
|
||||
7. Go to http://element.127.0.0.1.nip.io:42025/ and login with `admin` / `admin`
|
||||
8. Create a new room and invite `@baibot:synapse.127.0.0.1.nip.io`
|
||||
9. When done, stop the bot (`Ctrl` + `C`)
|
||||
10. Stop the dependency services: `just services-stop`
|
||||
11. (Optional) Stop additional services:
|
||||
7. Start the bot: `just run-in-container`
|
||||
8. Go to http://element.127.0.0.1.nip.io:42025/ and login with `admin` / `admin`
|
||||
9. Create a new room and invite `@baibot:continuwuity.127.0.0.1.nip.io` (or `@baibot:synapse.127.0.0.1.nip.io` if using Synapse)
|
||||
10. When done, stop the bot (`Ctrl` + `C`)
|
||||
11. Stop the services: `just services-stop`
|
||||
12. (Optional) Stop additional services:
|
||||
- for [LocalAI](#localai): `just localai-stop`
|
||||
- for [Ollama](#ollama): `just ollama-stop`
|
||||
|
||||
@@ -93,7 +116,7 @@ For getting started most quickly (and locally), we recommend using [LocalAI](#lo
|
||||
|
||||
**Ollama is most lightweight** (~2GB for the container image + ~1.6GB for the model), but supports only [💬 text-generation](./features.md#-text-generation).
|
||||
|
||||
**LocalAI requires 4x more disk space** (~6GB for the container image + ~12GB for the models), but supports [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text) and [🖼️ image-generation](./features.md#️-image-generation).
|
||||
**LocalAI requires 4x more disk space** (~6GB for the container image + ~12GB for the models), but supports [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text) and [🖼️ image-generation](./features.md#️-image-creation).
|
||||
|
||||
**OpenAI supports all of these capabilities** as well and does not require powerful hardware or lots of disk space. However, it requires signup and an API key.
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@ You can also use **different models within the same room** (e.g. [💬 text-gene
|
||||
|
||||
The bot supports the following use-purposes:
|
||||
|
||||
- [💬 text-generation](#-text-generation): communicating with you via text
|
||||
- [💬 text-generation](#-text-generation): communicating with you via text (though certain models may also process images and files)
|
||||
- [🦻 speech-to-text](#-speech-to-text): turning your voice messages into text
|
||||
- [🗣️ text-to-speech](#%EF%B8%8F-text-to-speech): turning bot or users text messages into voice messages
|
||||
- [🖌️ image-generation](#%EF%B8%8F-image-generation): generating images based on instructions
|
||||
@@ -22,12 +22,18 @@ For more information about configuring handlers, see the [🤝 Handlers / Config
|
||||
|
||||
### 💬 Text Generation
|
||||
|
||||
Text Generation is the bot's ability to **respond to users' text messages with text**.
|
||||
Text Generation is the bot's ability to **respond to users' messages with text**.
|
||||
|
||||

|
||||
|
||||
Some models also support vision and document understanding, so you may be able to mix text, images, and files (PDFs, text documents, etc.) in the same conversation. Note that certain providers may not support all file types or may have issues with specific files (e.g. scanned/image-based PDFs). If a file is rejected by the provider, the conversation thread may become unusable — start a new thread to work around this.
|
||||
|
||||
In multi-user (group) rooms, to avoid disturbing the normal conversation between people, the bot is auto-configured to only respond to messages starting with the command prefix (`!bai`) or direct mentions via the [💬 Text Generation / 🗟 Prefix Requirement Type](./configuration/text-generation.md#-prefix-requirement-type) setting.
|
||||
|
||||
Normally, the bot only responds to allowed [👥 Users](./access.md#-users). In certain cases, it's useful for an allowed user to provoke the bot to respond even in foreign threads or reply chains. You can learn more about this feature in the [On-demand involvement](./features.md#on-demand-involvement) section below.
|
||||
|
||||
If needed, the bot can also attach sender metadata to conversation messages before sending them to the model, which can help the model distinguish between participants in multi-user rooms. See [🛠️ Configuration / 💬 Text Generation / 👤 Sender Context Mode](./configuration/text-generation.md#-sender-context-mode).
|
||||
|
||||
A few other features (like [🗣️ Text-to-Speech](#️-text-to-speech) and [🦻 Speech-to-Text](#-speech-to-text)) combine well with Text Generation, so you **don't necessarily need to communicate with the bot via text** (with [Seamless voice interaction](#seamless-voice-interaction), you can communicate only with voice).
|
||||
|
||||
You may also wish to see:
|
||||
@@ -36,6 +42,39 @@ You may also wish to see:
|
||||
- [📖 Usage / 💬 Text Generation](./usage.md#-text-generation) section for more details on how to use the bot for Text Generation in a room
|
||||
|
||||
|
||||
#### 🛠️ Built-in Tools (OpenAI only)
|
||||
|
||||
|
||||
|
||||
The [OpenAI provider](./providers.md#openai) supports built-in tools that extend the model's capabilities:
|
||||
|
||||
- [🔍 Web Search](https://platform.openai.com/docs/guides/tools-web-search) (`web_search`): allows the model to search the web for up-to-date information. [🖼️ Screenshot](./screenshots/text-generation-tools-web-search.webp)
|
||||
|
||||
- [💻 Code Interpreter](https://platform.openai.com/docs/guides/tools-code-interpreter) (`code_interpreter`): allows the model to write and execute Python code in a sandbox
|
||||
|
||||
These tools are **disabled by default** and need to be explicitly enabled in the agent's `text_generation.tools` configuration. See the [OpenAI sample configuration](https://github.com/etkecc/baibot/blob/c70387b0c38d8d0f30bba2179a2a21a3710dbeaf/docs/sample-provider-configs/openai.yml#L12-L15) for reference.
|
||||
|
||||
To enable tools on an existing dynamically-created agent, you need to [update the agent](./agents.md#updating-agents) to re-create it with the `text_generation.tools` section added and enable the tools you need
|
||||
|
||||
💡 **Note**: These tools run on OpenAI's infrastructure and may incur additional costs. Web search results include citations that are incorporated into the response.
|
||||
|
||||
|
||||
#### On-demand involvement
|
||||
|
||||
In the following 2 cases, it's useful to involve the bot in conversations on-demand:
|
||||
|
||||
1. In multi-user rooms (with the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting set to "required")
|
||||
2. In rooms with foreign users (users that are not authorized bot [👥 users](./access.md#-users))
|
||||
|
||||
In these instances, an allowed [👥 user](./access.md#-users) can also provoke the bot to respond to **any** thread or reply chain by [mentioning](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions) the bot (e.g. `@baibot Hello!`). The following screenshots demonstrate this behavior:
|
||||
|
||||
- [🖼️ On-demand involvement in the room](./screenshots/text-generation-prefix-requirement.webp)
|
||||
- [🖼️ On-demand involvement in a thread](./screenshots/text-generation-on-demand-thread-involvement.webp) (the Alice user in this example is not an allowed user, yet her messages are still considered as part of the conversation context)
|
||||
- [🖼️ On-demand involvement in a reply chain](./screenshots/text-generation-on-demand-reply-involvement.webp) (the Alice user in this example is not an allowed user, yet her messages are still considered as part of the conversation context)
|
||||
|
||||
💡 **NOTE**: Normally, the bot **only considers messages from allowed [👥 Users](./access.md#-users)** and ignores all other messages when responding. However, **when the bot is explicitly invoked (via mention)** in a thread or reply chain, **it will consider all messages** in the thread and reply chain (even those from foreign users) as part of the conversation context.
|
||||
|
||||
|
||||
### 🗣️ Text-to-Speech
|
||||
|
||||
Text-to-Speech is the bot's ability to **turn text messages into voice messages**.
|
||||
@@ -118,27 +157,45 @@ To operate in this mode, you can:
|
||||
|
||||
- adjust the [🦻 Speech-to-Text / 🪄 Flow Type](./configuration/speech-to-text.md#-flow-type) setting to make the bot only transcribe (without doing [💬 Text Generation](#-text-generation)): `!bai config room speech-to-text set-flow-type only_transcribe`
|
||||
|
||||
- optionally adjust [🦻 Speech-to-Text / 🪄 Message Type for non-threaded only-transcribed messages](./configuration/speech-to-text.md#-message-type-for-non-threaded-only-transcribed-messages), if you'd like to bot to send messages of type `notice` (for better compatibility with other bots in the room) instead of sending regular `text` messages (default)
|
||||
|
||||
### 🖌️ Image Generation
|
||||
|
||||
Image generation is the bot's ability to **generate images** based on text prompts.
|
||||
### Image Generation
|
||||
|
||||
See a [🖼️ Screenshot of the Image Generation feature](./screenshots/image-generation.webp).
|
||||
#### 🖌️ Image Creation
|
||||
|
||||
Image creation is the bot's ability to **create images** based on text prompts.
|
||||
|
||||
See a [🖼️ Screenshot of the Image Creation feature](./screenshots/image-creation.webp).
|
||||
|
||||
You may also wish to see:
|
||||
|
||||
- [🛠️ Configuration / 🖌️ Image Generation](./configuration/image-generation.md) for configuration options related to Image Generation
|
||||
- [📖 Usage / 🖌️ Image Generation](./usage.md#-image-generation) section for more details on how to use the bot for Image Generation in a room
|
||||
- [🫵 Sticker Generation](#-sticker-generation) - a special case of Image Generation
|
||||
- [📖 Usage / Image Generation / 🖌️ Creating Images](./usage.md#-creating-images) section for more details on how to use the bot for Image Creation in a room
|
||||
- [🖌️ Image Editing](#️-image-editing) - another image generation feature
|
||||
- [🫵 Sticker Creation](#-sticker-creation) - a special case of Image Creation
|
||||
|
||||
|
||||
### 🫵 Sticker Generation
|
||||
#### 🎨 Image Editing
|
||||
|
||||
Sticker generation is the bot's ability to **generate sticker** images based on text prompts. It's a special case of [🖌️ Image Generation](#️-image-generation).
|
||||
Image editing is the bot's ability to **edit images** based on a prompt and one or more existing images.
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Generation feature](./screenshots/sticker-generation.webp).
|
||||
See a [🖼️ Screenshot of the Image Editing feature (manipulating a single image)](./screenshots/image-editing-single-image.webp) and a [🖼️ Screenshot of the Image Editing feature (manipulating multiple images)](./screenshots/image-editing-multiple-images.webp).
|
||||
|
||||
See [📖 Usage / 🖌️ Image Generation / Generating Stickers](./usage.md#generating-stickers) for details.
|
||||
You may also wish to see:
|
||||
|
||||
- [🛠️ Configuration / 🖌️ Image Generation](./configuration/image-generation.md) for configuration options related to Image Generation
|
||||
- [📖 Usage / Image Generation / 🎨 Editing images](./usage.md#-editing-images) section for more details on how to use the bot for Image Editing in a room
|
||||
- [🖌️ Image Creation](#️-image-creation) - another image generation feature
|
||||
|
||||
|
||||
#### 🫵 Sticker Creation
|
||||
|
||||
Sticker generation is the bot's ability to **generate sticker** images based on text prompts. It's a special case of [🖌️ Image Creation](#️-image-creation).
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Creation feature](./screenshots/sticker-generation.webp).
|
||||
|
||||
See [📖 Usage / Image Generation / 🫵 Creating Stickers](./usage.md#-creating-stickers) for details.
|
||||
|
||||
|
||||
### 🔒 Encryption
|
||||
|
||||
@@ -53,6 +53,7 @@ CONTAINER_IMAGE_NAME=ghcr.io/etkecc/baibot:v1.0.0
|
||||
--env BAIBOT_PERSISTENCE_DATA_DIR_PATH=/data \
|
||||
--mount type=bind,src=/path/to/config.yml,dst=/app/config.yml,ro \
|
||||
--mount type=bind,src=/path/to/data,dst=/data \
|
||||
--tmpfs=/tmp:rw,noexec,nosuid,size=1024m \
|
||||
$CONTAINER_IMAGE_NAME
|
||||
```
|
||||
|
||||
|
||||
@@ -19,11 +19,12 @@ The list of supported providers is below.
|
||||
- [OpenAI Compatible](#openai-compatible)
|
||||
- [OpenRouter](#openrouter)
|
||||
- [Together AI](#together-ai)
|
||||
- [Venice](#venice)
|
||||
|
||||
|
||||
### How to choose a provider
|
||||
|
||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation), [🖌️ image-generation](./features.md#️-image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation) (incl. vision, incl. [🛠️ tools](./features.md#️-built-in-tools-openai-only)), [🖌️ image-generation](./features.md#️image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
||||
|
||||
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
|
||||
|
||||
@@ -47,7 +48,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `anthropic`
|
||||
- 🔗 Links: [🏠 Home page](https://www.anthropic.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/Anthropic), [👤 Sign up](https://console.anthropic.com/), [📋 Models list](https://docs.anthropic.com/en/docs/about-claude/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (incl. vision, no tools)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local anthropic my-anthropic-agent`
|
||||
- create a global agent: `!bai agent create-global anthropic my-anthropic-agent`
|
||||
@@ -61,7 +62,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `groq`
|
||||
- 🔗 Links: [🏠 Home page](https://groq.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/Groq), [👤 Sign up](https://console.groq.com/login), [📋 Models list](https://console.groq.com/docs/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision, no tools), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local groq my-groq-agent`
|
||||
- create a global agent: `!bai agent create-global groq my-groq-agent`
|
||||
@@ -75,7 +76,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `localai`
|
||||
- 🔗 Links: [🏠 Home page](https://localai.io/), [📋 Models list](https://localai.io/gallery.html)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision, no tools), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local localai my-localai-agent`
|
||||
- create a global agent: `!bai agent create-global localai my-localai-agent`
|
||||
@@ -89,7 +90,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `mistral`
|
||||
- 🔗 Links: [🏠 Home page](https://mistral.ai/), [🌐 Wiki](https://en.wikipedia.org/wiki/Mistral_AI), [👤 Sign up](https://auth.mistral.ai/ui/registration), [📋 Models list](https://docs.mistral.ai/getting-started/models/)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision, no tools)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local mistral my-mistral-agent`
|
||||
- create a global agent: `!bai agent create-global mistral my-mistral-agent`
|
||||
@@ -103,7 +104,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `ollama`
|
||||
- 🔗 Links: [🏠 Home page](https://ollama.com/), [📋 Models list](https://ollama.com/library)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision, no tools)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local ollama my-ollama-agent`
|
||||
- create a global agent: `!bai agent create-global ollama my-ollama-agent`
|
||||
@@ -120,7 +121,7 @@ For services which are not fully compatible with the OpenAI API, consider using
|
||||
|
||||
- 🆔 Identifier: `openai`
|
||||
- 🔗 Links: [🏠 Home page](https://openai.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/OpenAI), [👤 Sign up](https://platform.openai.com/signup), [📋 Models list](https://platform.openai.com/docs/models)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-generation), [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation), [💬 text-generation](./features.md#-text-generation) (incl. vision, incl. [🛠️ tools](./features.md#️-built-in-tools-openai-only)), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local openai my-openai-agent`
|
||||
- create a global agent: `!bai agent create-global openai my-openai-agent`
|
||||
@@ -137,7 +138,7 @@ Some of these popular services already have **shortcut** providers (leading to t
|
||||
This provider is just as featureful as the [OpenAI](#openai) provider, but is more compatible with services which do not fully adhere to the [OpenAI API spec](https://github.com/openai/openai-openapi/).
|
||||
|
||||
- 🆔 Identifier: `openai-compatible`
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-generation), [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation), [💬 text-generation](./features.md#-text-generation) (no vision, no tools), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local openai-compatible my-openai-compatible-agent`
|
||||
- create a global agent: `!bai agent create-global openai-compatible my-openai-compatible-agent`
|
||||
@@ -151,7 +152,7 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
||||
|
||||
- 🆔 Identifier: `openrouter`
|
||||
- 🔗 Links: [🏠 Home page](https://openrouter.ai/), [👤 Sign up](https://openrouter.ai/), [📋 Models list](https://openrouter.ai/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision, no tools)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local openrouter my-openrouter-agent`
|
||||
- create a global agent: `!bai agent create-global openrouter my-openrouter-agent`
|
||||
@@ -165,9 +166,88 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
||||
|
||||
- 🆔 Identifier: `together-ai`
|
||||
- 🔗 Links: [🏠 Home page](https://www.together.ai/), [👤 Sign up](https://api.together.ai/signup), [📋 Models list](https://api.together.xyz/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision, no tools)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local together-ai my-together-ai-agent`
|
||||
- create a global agent: `!bai agent create-global together-ai my-together-ai-agent`
|
||||
|
||||
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/together-ai.yml).
|
||||
|
||||
|
||||
### Venice
|
||||
|
||||
[Venice AI](https://venice.ai) runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
|
||||
|
||||
- 🆔 Identifier: `venice`
|
||||
- 🔗 Links: [🏠 Home page](https://venice.ai), [👤 Sign up](https://venice.ai), [📋 Models list](https://api.venice.ai/api/v1/models)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation) (incl. editing, via the native knob-rich `/image/generate` and `/image/edit` endpoints), [💬 text-generation](./features.md#-text-generation) (incl. vision; native web search via the `venice_parameters` config), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local venice my-venice-agent`
|
||||
- create a global agent: `!bai agent create-global venice my-venice-agent`
|
||||
|
||||
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/venice.yml).
|
||||
|
||||
Unlike the [OpenAI Compatible](#openai-compatible) provider (which can talk to Venice but drops images and can't reach its audio or native image endpoints), this is a first-class Venice integration that exposes Venice's full parameter set. Image generation uses the native `/image/generate` endpoint rather than the OpenAI-compatible `/images/generations` shim, so every Venice-specific knob below is available.
|
||||
|
||||
#### Configuration reference
|
||||
|
||||
Every parameter below is optional unless marked otherwise. Omitting a knob lets Venice apply its own server-side default; this is **not** the same as setting it to `false`, which actively sends `false`.
|
||||
|
||||
**`text_generation.venice_parameters`** — Venice-specific request knobs sent in the `venice_parameters` bag (alongside the standard `model_id`, `prompt`, `temperature`, `max_response_tokens`, and `max_context_tokens` fields). Set any of them to override Venice's behavior. The `Default` column shows the value baibot's sample config ships; a `—` means the knob is left unset, so Venice's own default applies.
|
||||
|
||||
| Knob | What it does | Default |
|
||||
|------|--------------|---------|
|
||||
| `enable_web_search` | Web search mode: `auto` (model decides), `on` (always), or `off`. | `auto` |
|
||||
| `enable_web_citations` | Append source citations to web-search answers. | — |
|
||||
| `enable_web_scraping` | Allow the model to scrape page contents during web search. | — |
|
||||
| `enable_x_search` | Include X (Twitter) in web search. | — |
|
||||
| `include_search_results_in_stream` | Stream search results back as they arrive. | — |
|
||||
| `return_search_results_as_documents` | Return search results as structured documents. | — |
|
||||
| `include_venice_system_prompt` | Prepend Venice's own system prompt alongside yours. | — |
|
||||
| `character_slug` | Use a public Venice character by its slug. | — |
|
||||
| `strip_thinking_response` | Strip `<think></think>` blocks from reasoning models so the user sees only the answer. | `true` |
|
||||
| `disable_thinking` | Disable the model's reasoning step entirely. | — |
|
||||
| `enable_e2ee` | Run in end-to-end-encrypted mode rather than the default TEE-only mode. | `false` |
|
||||
|
||||
**`text_to_speech`**:
|
||||
|
||||
| Knob | What it does | Default |
|
||||
|------|--------------|---------|
|
||||
| `model_id` | The Venice TTS model (e.g. `tts-kokoro`, `tts-qwen3-1-7b`, `tts-xai-v1`). | `tts-kokoro` |
|
||||
| `voice` | The voice to synthesize with. Model-specific (Kokoro: `af_*`/`am_*`/`bf_*`/`bm_*`); a cloned-voice handle (`vv_<id>`) also works. | `af_sky` |
|
||||
| `response_format` | Audio format: `mp3`, `opus`, `aac`, `flac`, `wav`, or `pcm`. | `mp3` |
|
||||
| `speed` | Playback speed, `0.25`–`4.0`. | `1.0` |
|
||||
| `prompt` | A style prompt steering emotion/delivery. Only Qwen 3 TTS honors it. | — |
|
||||
| `temperature` | Sampling temperature, `0.0`–`2.0`. Only Qwen 3 / Orpheus / Chatterbox HD honor it. | — |
|
||||
| `top_p` | Nucleus sampling, `0.0`–`1.0`. Only Qwen 3 TTS honors it. | — |
|
||||
|
||||
**`image_generation`**:
|
||||
|
||||
| Knob | What it does | Default |
|
||||
|------|--------------|---------|
|
||||
| `model_id` | The image-generation model. | `chroma` |
|
||||
| `negative_prompt` | A description of what should **not** appear in the image. | — |
|
||||
| `cfg_scale` | CFG scale, `0`–`20`. Higher values adhere more closely to the prompt. | — |
|
||||
| `steps` | Number of inference steps. Model-specific; some models ignore it. | — |
|
||||
| `style_preset` | A named style to apply (e.g. `3D Model`). | — |
|
||||
| `seed` | Random seed, `-999999999`–`999999999`. Fix it for reproducible results. | random |
|
||||
| `safe_mode` | Blur images classified as adult content. | `true` |
|
||||
| `hide_watermark` | Hide the Venice watermark (may be ignored for some content). | `false` |
|
||||
| `format` | Output format: `jpeg`, `png`, or `webp`. | `webp` |
|
||||
| `width` / `height` | Image dimensions in pixels, each `1`–`1280`. | `1024` |
|
||||
| `aspect_ratio` | Aspect ratio for models that support it (e.g. `1:1`, `16:9`). Alternative to `width`/`height`. | — |
|
||||
| `resolution` | Resolution tier for models that support it (`1K`, `2K`, `4K`). | — |
|
||||
| `quality` | Output quality for supported models: `low`, `medium`, `high`. Higher can cost more. | — |
|
||||
| `lora_strength` | Lora strength, `0`–`100`. Only applies if the model uses additional Loras. | — |
|
||||
| `embed_exif_metadata` | Embed the generation prompt into the image's EXIF metadata. | `false` |
|
||||
| `enable_web_search` | Let the model pull the latest info from the web. Model-specific; costs extra credits. | — |
|
||||
|
||||
**`image_generation.edit`** — image editing reuses the `image_generation` block; only the model and a few output knobs differ:
|
||||
|
||||
| Knob | What it does | Default |
|
||||
|------|--------------|---------|
|
||||
| `model_id` | The image-edit model. | `firered-image-edit` |
|
||||
| `output_format` | Output format: `jpeg`, `png`, or `webp`. When omitted, Venice infers it (PNG at 1K, JPEG at 2K/4K). | inferred |
|
||||
| `aspect_ratio` | Aspect ratio of the result: `auto`, `1:1`, `3:2`, `16:9`, `21:9`, `9:16`, `2:3`, `3:4`, `4:5` (model-specific). | — |
|
||||
| `resolution` | Resolution tier: `1K`, `2K`, `4K` (model-specific). | `1K` |
|
||||
| `safe_mode` | Blur images classified as adult content. | `true` |
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
base_url: https://api.anthropic.com/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: claude-3-5-sonnet-20240620
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
model_id: claude-3-7-sonnet-20250219
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 8192
|
||||
max_context_tokens: 204800
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: https://api.groq.com/openai/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: llama3-70b-8192
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 4096
|
||||
max_context_tokens: 131072
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: http://my-localai-self-hosted-service:8080/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: gpt-4
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 4096
|
||||
max_context_tokens: 128000
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: https://api.mistral.ai/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: mistral-large-latest
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 4096
|
||||
max_context_tokens: 128000
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: http://my-ollama-self-hosted-service:11434/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: gemma2:2b
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 4096
|
||||
max_context_tokens: 128000
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: ''
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: some-model
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 4096
|
||||
max_context_tokens: 128000
|
||||
|
||||
@@ -1,11 +1,18 @@
|
||||
base_url: https://api.openai.com/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: gpt-4o-2024-08-06
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
model_id: gpt-5.4
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 16384
|
||||
max_context_tokens: 128000
|
||||
# Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`.
|
||||
# If you're dealing with a non-reasoning model, specify `max_response_tokens` and unset `max_completion_tokens`.
|
||||
max_response_tokens: null
|
||||
max_completion_tokens: 128000
|
||||
max_context_tokens: 400000
|
||||
# Built-in tools
|
||||
tools:
|
||||
web_search: false
|
||||
code_interpreter: false
|
||||
speech_to_text:
|
||||
model_id: whisper-1
|
||||
text_to_speech:
|
||||
@@ -14,7 +21,7 @@ text_to_speech:
|
||||
speed: 1.0
|
||||
response_format: opus
|
||||
image_generation:
|
||||
model_id: dall-e-3
|
||||
style: vivid
|
||||
size: 1024x1024
|
||||
quality: standard
|
||||
model_id: gpt-image-2
|
||||
style: null
|
||||
size: null
|
||||
quality: null
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: https://openrouter.ai/api/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: mattshumer/reflection-70b:free
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 2048
|
||||
max_context_tokens: 8192
|
||||
|
||||
@@ -2,7 +2,7 @@ base_url: https://api.together.xyz/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 2048
|
||||
max_context_tokens: 8192
|
||||
|
||||
97
docs/sample-provider-configs/venice.yml
Normal file
@@ -0,0 +1,97 @@
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: kimi-k2-5
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 4096
|
||||
max_context_tokens: 128000
|
||||
# Venice-specific request parameters. Only the keys present below are sent to Venice; omit a
|
||||
# key to fall back to Venice's own default. Omitting a knob is NOT the same as setting it to
|
||||
# `false` — `false` actively sends `false`.
|
||||
venice_parameters:
|
||||
# Web search: "auto" (model decides), "on" (always), or "off".
|
||||
enable_web_search: "auto"
|
||||
# Strip <think></think> blocks from reasoning models so the user sees only the answer.
|
||||
strip_thinking_response: true
|
||||
# Run in TEE-only mode instead of end-to-end encryption (works across all models).
|
||||
enable_e2ee: false
|
||||
# Other available knobs — uncomment to override Venice's default:
|
||||
# enable_web_citations: true
|
||||
# enable_web_scraping: true
|
||||
# include_venice_system_prompt: false
|
||||
# include_search_results_in_stream: true
|
||||
# return_search_results_as_documents: true
|
||||
# enable_x_search: true
|
||||
# disable_thinking: true
|
||||
# character_slug: public-character-id
|
||||
speech_to_text:
|
||||
model_id: nvidia/parakeet-tdt-0.6b-v3
|
||||
text_to_speech:
|
||||
# The Venice TTS model. Others include tts-qwen3-1-7b, tts-xai-v1,
|
||||
# tts-elevenlabs-turbo-v2-5, tts-minimax-speech-02-hd. See the models list endpoint.
|
||||
model_id: tts-kokoro
|
||||
# The voice to synthesize with. Voices are model-specific: Kokoro uses af_*/am_*/bf_*/bm_*
|
||||
# (e.g. af_sky, am_adam), other models have their own sets. You can also pass a cloned-voice
|
||||
# handle (vv_<id>) created via Venice's voice-cloning API. An incompatible voice returns an error.
|
||||
voice: af_sky
|
||||
# Output audio format: mp3, opus, aac, flac, wav, or pcm. mp3 is the broadest Matrix-client fit.
|
||||
response_format: mp3
|
||||
# Other available knobs — uncomment to override Venice's default:
|
||||
# Playback speed, 0.25–4.0 (1.0 is normal).
|
||||
# speed: 1.0
|
||||
# A style prompt steering emotion/delivery (e.g. "Excited and energetic."). Only Qwen 3 TTS uses it.
|
||||
# prompt: "Calm and warm."
|
||||
# Sampling temperature, 0.0–2.0 (higher = more varied). Only Qwen 3 / Orpheus / Chatterbox HD use it.
|
||||
# temperature: 0.9
|
||||
# Nucleus sampling, 0.0–1.0. Only Qwen 3 TTS uses it.
|
||||
# top_p: 1.0
|
||||
image_generation:
|
||||
# The image-generation model. See the models list endpoint for the full set.
|
||||
model_id: chroma
|
||||
# The image-edit model, used when editing an existing image rather than generating a new one.
|
||||
# Editing shares this same image_generation config block; only the model differs.
|
||||
edit:
|
||||
model_id: firered-image-edit
|
||||
# Other edit knobs — uncomment to override Venice's default:
|
||||
# Output format: jpeg, png, or webp. When omitted, Venice infers it (PNG at 1K, JPEG at 2K/4K).
|
||||
# output_format: png
|
||||
# Aspect ratio of the result: auto, 1:1, 3:2, 16:9, 21:9, 9:16, 2:3, 3:4, 4:5 (model-specific).
|
||||
# aspect_ratio: auto
|
||||
# Resolution tier: 1K, 2K, 4K (model-specific). Defaults to 1K.
|
||||
# resolution: 1K
|
||||
# Blur images classified as adult content. Defaults to true.
|
||||
# safe_mode: true
|
||||
# Other generation knobs — uncomment to override Venice's default. Omitting a knob is NOT the same
|
||||
# as setting it: an omitted knob lets Venice apply its own default, a set value is sent verbatim.
|
||||
# A description of what should NOT appear in the image.
|
||||
# negative_prompt: "blurry, watermark, text"
|
||||
# CFG scale, 0–20. Higher values make the image adhere more closely to the prompt.
|
||||
# cfg_scale: 7.5
|
||||
# Number of inference steps. Model-specific; some models ignore it.
|
||||
# steps: 8
|
||||
# A named style to apply (e.g. "3D Model"). See Venice's image-styles reference.
|
||||
# style_preset: "3D Model"
|
||||
# Random seed, -999999999–999999999. Fix it for reproducible results; omit for a random seed.
|
||||
# seed: 123456789
|
||||
# Blur images classified as adult content. Defaults to true.
|
||||
# safe_mode: true
|
||||
# Hide the Venice watermark. Venice may ignore this for certain generated content. Defaults to false.
|
||||
# hide_watermark: false
|
||||
# Output format: jpeg, png, or webp. webp is smallest; png is highest-quality. Defaults to webp.
|
||||
# format: webp
|
||||
# Image dimensions in pixels, each 1–1280. Default 1024×1024.
|
||||
# width: 1024
|
||||
# height: 1024
|
||||
# Aspect ratio (used by certain models, e.g. Nano Banana): "1:1", "16:9". An alternative to width/height.
|
||||
# aspect_ratio: "1:1"
|
||||
# Resolution tier (used by certain models): "1K", "2K", "4K".
|
||||
# resolution: "1K"
|
||||
# Output quality for supported models (e.g. GPT Image 2): low, medium, high. Higher can cost more.
|
||||
# quality: high
|
||||
# Lora strength, 0–100. Only applies if the model uses additional Loras.
|
||||
# lora_strength: 50
|
||||
# Embed the generation prompt into the image's EXIF metadata. Defaults to false.
|
||||
# embed_exif_metadata: false
|
||||
# Let the model pull the latest info from the web for the image. Model-specific; costs extra credits.
|
||||
# enable_web_search: false
|
||||
BIN
docs/screenshots/image-creation.webp
Normal file
|
After Width: | Height: | Size: 298 KiB |
BIN
docs/screenshots/image-editing-multiple-images.webp
Normal file
|
After Width: | Height: | Size: 339 KiB |
BIN
docs/screenshots/image-editing-single-image.webp
Normal file
|
After Width: | Height: | Size: 285 KiB |
|
Before Width: | Height: | Size: 684 KiB |
|
Before Width: | Height: | Size: 13 KiB After Width: | Height: | Size: 22 KiB |
|
After Width: | Height: | Size: 92 KiB |
|
After Width: | Height: | Size: 60 KiB |
BIN
docs/screenshots/text-generation-tools-web-search.webp
Normal file
|
After Width: | Height: | Size: 66 KiB |
@@ -11,10 +11,13 @@ This is related to the [💬 Text Generation](./features.md#-text-generation) fe
|
||||
|
||||
If there's a text-generation handler agent configured, the bot **may** respond to messages sent in the room.
|
||||
|
||||
🖼️ See screenshots of:
|
||||
Some models also support vision and document understanding, so you may be able to mix text, images, and files (PDFs, text documents, etc.) in the same conversation.
|
||||
|
||||
- the [default Text Generation flow](./screenshots/text-generation.webp) for 1:1 rooms
|
||||
- the [Text Generation flow in multi-user rooms](./screenshots/text-generation-prefix-requirement.webp) (where the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting is auto-configured to "required")
|
||||
See screenshots of:
|
||||
|
||||
- 🖼️ [the default Text Generation flow](./screenshots/text-generation.webp) in 1:1 rooms
|
||||
- 🖼️ [the Text Generation flow in multi-user rooms](./screenshots/text-generation-prefix-requirement.webp) (where the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting is auto-configured to "required")
|
||||
- the [on-demand involvement](./features.md#on-demand-involvement) feature
|
||||
|
||||
Whether the bot responds depends on:
|
||||
|
||||
@@ -24,9 +27,9 @@ Whether the bot responds depends on:
|
||||
|
||||
- (🎨 agent capabilities) whether the configured `text-generation` (or `catch-all`) handler agent actually supports text-generation. The provider may lack support for this feature or it may be disabled in the [🤖 agents](./agents.md) configuration
|
||||
|
||||
- (the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting) whether a prefix (e.g. `!bai`) is required in front of messages sent to the room. For multi-user rooms, this setting defaults to "required"
|
||||
- (the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting) whether a prefix (e.g. `!bai`) or user mention (e.g. `@baibot`) is required for messages sent to the room. For multi-user rooms, this setting defaults to "required". See [🌟 Features / 💬 Text Generation / On-demand involvement](./features.md#on-demand-involvement) for details.
|
||||
|
||||
Room messages start a threaded conversation where you can continue back-and-forth communication with the bot.
|
||||
Room messages start a threaded conversation where you can continue back-and-forth communication with the bot. Using [on-demand involvement](./features.md#on-demand-involvement), you can can also mention the bot to provoke it to get involved in any conversation thread or reply chain.
|
||||
|
||||
Unless you've enabled the [♻️ Context Management](./features.md#️-context-management) feature, all messages will be sent to the agent's API each time. If the context management feature is enabled, older messages may be dropped.
|
||||
|
||||
@@ -63,34 +66,48 @@ The speech-to-text feature triggers automatically by default, but can be adjuste
|
||||
If all your messages are in the same language, you can improve accuracy & latency by configuring the language (see [🦻 Speech-to-Text / 🔤 Language](./configuration/speech-to-text.md#-language)).
|
||||
|
||||
|
||||
### 🖌️ Image Generation
|
||||
|
||||
This is related to the [🖌️ Image Generation](./features.md#️-image-generation) feature.
|
||||
### Image Generation
|
||||
|
||||
This feature is not configurable at the moment. The configuration (size, quality, style) specified at the [🤖 agent](./agents.md) level will be used.
|
||||
|
||||
Capabilities depend on the [☁️ provider](./providers.md) and model used.
|
||||
|
||||
#### Generating images
|
||||
|
||||
Simply send a command like `!bai image A beautiful sunset over the ocean` and the bot will start a threaded conversation and post an image based on your prompt.
|
||||
#### 🖌️ Creating images
|
||||
|
||||
See a [🖼️ Screenshot of the Image Generation feature](./screenshots/image-generation.webp).
|
||||
Simply send a command like `!bai image create A beautiful sunset over the ocean` and the bot will start a threaded conversation and post an image based on your prompt.
|
||||
|
||||
You can then, respond in the same message thread with:
|
||||
See a [🖼️ Screenshot of the Image Creation feature](./screenshots/image-creation.webp).
|
||||
|
||||
You can then respond in the same message thread with:
|
||||
|
||||
- more messages, to add more criteria to your prompt.
|
||||
- a message saying `again`, to generate one more image with the current prompt.
|
||||
|
||||
|
||||
#### Generating stickers
|
||||
#### 🎨 Editing images
|
||||
|
||||
A variation of [generating images](#generating-images) is to generate "sticker images".
|
||||
Simply send a command like `!bai image edit Turn the following image into an anime-style drawing` and the bot will start a threaded conversation asking for more details.
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Generation feature](./screenshots/sticker-generation.webp).
|
||||
See a [🖼️ Screenshot of the Image Editing feature (manipulating a single image)](./screenshots/image-editing-single-image.webp) and a [🖼️ Screenshot of the Image Editing feature (manipulating multiple images)](./screenshots/image-editing-multiple-images.webp).
|
||||
|
||||
To generate a sticker, send a command like `!bai sticker A huge ramen bowl with lots of chashu and a mountain of beansprouts on top`.
|
||||
You can then respond in the same message thread with:
|
||||
|
||||
The difference from [generating images](#generating-images) is that the bot will:
|
||||
- more messages, to add more criteria to your prompt.
|
||||
- one or more images, to provide the images that the bot will operate on.
|
||||
- a message saying `go`, to start the image generation process.
|
||||
- a message saying `again`, to prompt the bot to generate one more image edit with the current prompt.
|
||||
|
||||
|
||||
#### 🫵 Creating stickers
|
||||
|
||||
A variation of [creating images](#creating-images) is creating "sticker images".
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Creation feature](./screenshots/sticker-generation.webp).
|
||||
|
||||
To create a sticker, send a command like `!bai sticker A huge ramen bowl with lots of chashu and a mountain of beansprouts on top`.
|
||||
|
||||
The difference from [creating images](#creating-images) is that the bot will:
|
||||
|
||||
- generate a smaller-resolution image (currently hardcoded to `256x256`) - smaller/quicker, but still good enough for a sticker
|
||||
- potentially switch to a different (cheaper or otherwise more suitable) model, if available
|
||||
|
||||
@@ -1,16 +1,31 @@
|
||||
homeserver:
|
||||
# The canonical homeserver domain name
|
||||
server_name: synapse.127.0.0.1.nip.io
|
||||
url: http://synapse.127.0.0.1.nip.io:42020
|
||||
server_name: __HOMESERVER_SERVER_NAME__
|
||||
url: __HOMESERVER_URL__
|
||||
|
||||
user:
|
||||
mxid_localpart: baibot
|
||||
|
||||
# Authentication: set EITHER password OR access_token + device_id.
|
||||
#
|
||||
# Password-based login (traditional homeservers):
|
||||
password: baibot
|
||||
|
||||
# Access token login (for Matrix Authentication Service/OIDC-enabled homeservers):
|
||||
# Generate a token via: mas-cli manage issue-compatibility-token <username> [device_id]
|
||||
# access_token: null
|
||||
# device_id: null
|
||||
|
||||
# The name the bot uses as a display name and when it refers to itself.
|
||||
# Leave empty to use the default (baibot).
|
||||
name: baibot
|
||||
|
||||
# An optional path to an image file to be used as a custom avatar image.
|
||||
# - null or empty string: use the default avatar
|
||||
# - "keep": don't touch the avatar, keep whatever is already set
|
||||
# - any other value: path to a custom avatar image file
|
||||
avatar: null
|
||||
|
||||
encryption:
|
||||
# An optional passphrase to use for backing up and recovering the bot's encryption keys.
|
||||
# You can use any string here.
|
||||
@@ -32,10 +47,14 @@ user:
|
||||
# Command prefix. Leave empty to use the default (!bai).
|
||||
command_prefix: "!bai"
|
||||
|
||||
room:
|
||||
# Whether the bot should send an introduction message after joining a room.
|
||||
post_join_self_introduction_enabled: true
|
||||
|
||||
access:
|
||||
# Space-separated list of MXID patterns which specify who is an admin.
|
||||
admin_patterns:
|
||||
- "@admin:synapse.127.0.0.1.nip.io"
|
||||
- "@admin:__HOMESERVER_SERVER_NAME__"
|
||||
|
||||
persistence:
|
||||
# This is unset here, because we expect the configuration to come from an environment variable (BAIBOT_PERSISTENCE_DATA_DIR_PATH).
|
||||
@@ -72,11 +91,18 @@ agents:
|
||||
# base_url: https://api.openai.com/v1
|
||||
# api_key: ""
|
||||
# text_generation:
|
||||
# model_id: gpt-4o-2024-08-06
|
||||
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
# model_id: gpt-5.4
|
||||
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
# temperature: 1.0
|
||||
# max_response_tokens: 16384
|
||||
# max_context_tokens: 128000
|
||||
# # Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`.
|
||||
# # If you're dealing with a non-reasoning model, specify `max_response_tokens` and unset `max_completion_tokens`.
|
||||
# max_response_tokens: null
|
||||
# max_completion_tokens: 128000
|
||||
# max_context_tokens: 400000
|
||||
# # Built-in tools
|
||||
# tools:
|
||||
# web_search: false
|
||||
# code_interpreter: false
|
||||
# speech_to_text:
|
||||
# model_id: whisper-1
|
||||
# text_to_speech:
|
||||
@@ -85,10 +111,10 @@ agents:
|
||||
# speed: 1.0
|
||||
# response_format: opus
|
||||
# image_generation:
|
||||
# model_id: dall-e-3
|
||||
# style: vivid
|
||||
# size: 1024x1024
|
||||
# quality: standard
|
||||
# model_id: gpt-image-2
|
||||
# style: null
|
||||
# size: null
|
||||
# quality: null
|
||||
#
|
||||
# - id: localai
|
||||
# provider: localai
|
||||
@@ -97,7 +123,7 @@ agents:
|
||||
# api_key: null
|
||||
# text_generation:
|
||||
# model_id: gpt-4
|
||||
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
# temperature: 1.0
|
||||
# max_response_tokens: 16384
|
||||
# max_context_tokens: 128000
|
||||
@@ -122,7 +148,7 @@ agents:
|
||||
# api_key: null
|
||||
# text_generation:
|
||||
# model_id: "gemma2:2b"
|
||||
# prompt: "You are an assistant based on the gemma2:2b model. Be brief in your responses."
|
||||
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
# temperature: 1.0
|
||||
# max_response_tokens: 4096
|
||||
# max_context_tokens: 128000
|
||||
@@ -140,7 +166,7 @@ initial_global_config:
|
||||
# Space-separated list of MXID patterns which specify who can use the bot.
|
||||
# By default, we let anyone on the homeserver use the bot.
|
||||
user_patterns:
|
||||
- "@*:synapse.127.0.0.1.nip.io"
|
||||
- "@*:__HOMESERVER_SERVER_NAME__"
|
||||
|
||||
# Controls logging.
|
||||
#
|
||||
|
||||
23
etc/services/continuwuity/compose.yml
Normal file
@@ -0,0 +1,23 @@
|
||||
services:
|
||||
continuwuity:
|
||||
image: forgejo.ellis.link/continuwuation/continuwuity:v0.5.10
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
cap_drop:
|
||||
- ALL
|
||||
read_only: true
|
||||
environment:
|
||||
CONDUWUIT_CONFIG: /etc/continuwuity/continuwuity.toml
|
||||
CONDUWUIT_DATABASE_PATH: /var/lib/continuwuity
|
||||
ports:
|
||||
- "${SERVICE_CONTINUWUITY_BIND_PORT_CLIENT_API}:6167"
|
||||
volumes:
|
||||
- ../../etc/services/continuwuity/config:/etc/continuwuity:ro
|
||||
- ./continuwuity/data:/var/lib/continuwuity
|
||||
tmpfs:
|
||||
- /tmp:rw,noexec,nosuid,size=500m
|
||||
|
||||
networks:
|
||||
default:
|
||||
name: ${NETWORK_NAME}
|
||||
external: true
|
||||
19
etc/services/continuwuity/config/continuwuity.toml
Normal file
@@ -0,0 +1,19 @@
|
||||
[global]
|
||||
server_name = "continuwuity.127.0.0.1.nip.io"
|
||||
|
||||
address = "0.0.0.0"
|
||||
port = 6167
|
||||
|
||||
database_path = "/var/lib/continuwuity"
|
||||
|
||||
allow_registration = true
|
||||
yes_i_am_very_very_sure_i_want_an_open_registration_server_prone_to_abuse = true
|
||||
|
||||
new_user_displayname_suffix = ""
|
||||
|
||||
max_request_size = 20_000_000
|
||||
|
||||
allow_federation = false
|
||||
trusted_servers = ["matrix.org"]
|
||||
|
||||
log = "info,state_res=warn,rocket=off,_=off,sled=off"
|
||||
48
etc/services/continuwuity/register-user.sh
Executable file
@@ -0,0 +1,48 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
if [ $# -ne 3 ]; then
|
||||
echo "Usage: $0 <env-file> <username> <password>"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
ENV_FILE="$1"
|
||||
USERNAME="$2"
|
||||
PASSWORD="$3"
|
||||
|
||||
SERVER="http://$(grep '^SERVICE_CONTINUWUITY_BIND_PORT_CLIENT_API=' "${ENV_FILE}" | cut -d= -f2)"
|
||||
REGISTER_URL="${SERVER}/_matrix/client/v3/register"
|
||||
|
||||
echo "Registering user '${USERNAME}' on ${SERVER}..."
|
||||
|
||||
SESSION_RESPONSE=$(curl -s -X POST "${REGISTER_URL}" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d "{\"username\": \"${USERNAME}\", \"password\": \"${PASSWORD}\"}")
|
||||
|
||||
SESSION_ID=$(echo "${SESSION_RESPONSE}" | grep -o '"session":"[^"]*"' | head -1 | cut -d'"' -f4)
|
||||
if [ -z "${SESSION_ID}" ]; then
|
||||
echo "Error: Could not get session ID. Response: ${SESSION_RESPONSE}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Determine the required auth flow from the server response.
|
||||
# The first user requires m.login.registration_token (bootstrap token from logs).
|
||||
# Subsequent users use m.login.dummy (open registration).
|
||||
if echo "${SESSION_RESPONSE}" | grep -q 'm.login.registration_token'; then
|
||||
CONTAINER_ID=$(docker ps -q --filter name=baibot-continuwuity-continuwuity)
|
||||
REG_TOKEN=$(docker logs "${CONTAINER_ID}" 2>&1 | sed 's/\x1b\[[0-9;]*m//g' | grep 'using the registration token' | grep -oP 'registration token \K[A-Za-z0-9]+' | head -1)
|
||||
AUTH_BODY="{\"type\": \"m.login.registration_token\", \"token\": \"${REG_TOKEN}\", \"session\": \"${SESSION_ID}\"}"
|
||||
else
|
||||
AUTH_BODY="{\"type\": \"m.login.dummy\", \"session\": \"${SESSION_ID}\"}"
|
||||
fi
|
||||
|
||||
RESULT=$(curl -s -X POST "${REGISTER_URL}" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d "{\"username\": \"${USERNAME}\", \"password\": \"${PASSWORD}\", \"auth\": ${AUTH_BODY}}")
|
||||
|
||||
if echo "${RESULT}" | grep -q '"user_id"'; then
|
||||
echo "Successfully registered user: $(echo "${RESULT}" | grep -o '"user_id":"[^"]*"' | cut -d'"' -f4)"
|
||||
else
|
||||
echo "Registration failed. Response: ${RESULT}"
|
||||
exit 1
|
||||
fi
|
||||
@@ -1,60 +0,0 @@
|
||||
# This is a custom nginx configuration file that we use in the container (instead of the default one),
|
||||
# because it allows us to run nginx with a non-root user.
|
||||
#
|
||||
# For this to work, the default vhost file (`/etc/nginx/conf.d/default.conf`) also needs to be removed.
|
||||
# (mounting `/dev/null` over `/etc/nginx/conf.d/default.conf` works well)
|
||||
#
|
||||
# The following changes have been done compared to a default nginx configuration file:
|
||||
# - default server port is changed (80 -> 8080), so that a non-root user can bind it
|
||||
# - various temp paths are changed to `/tmp`, so that a non-root user can write to them
|
||||
# - the `user` directive was removed, as we don't want nginx to switch users
|
||||
|
||||
worker_processes 1;
|
||||
|
||||
error_log /var/log/nginx/error.log warn;
|
||||
pid /tmp/nginx.pid;
|
||||
|
||||
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
|
||||
http {
|
||||
client_body_temp_path /tmp/client_body_temp;
|
||||
proxy_temp_path /tmp/proxy_temp;
|
||||
fastcgi_temp_path /tmp/fastcgi_temp;
|
||||
uwsgi_temp_path /tmp/uwsgi_temp;
|
||||
scgi_temp_path /tmp/scgi_temp;
|
||||
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
|
||||
log_format main '$remote_addr - $remote_user [$time_local] "$request" '
|
||||
'$status $body_bytes_sent "$http_referer" '
|
||||
'"$http_user_agent" "$http_x_forwarded_for"';
|
||||
|
||||
access_log /var/log/nginx/access.log main;
|
||||
|
||||
sendfile on;
|
||||
#tcp_nopush on;
|
||||
|
||||
keepalive_timeout 65;
|
||||
|
||||
#gzip on;
|
||||
|
||||
server {
|
||||
listen 8080;
|
||||
server_name localhost;
|
||||
|
||||
location / {
|
||||
root /usr/share/nginx/html;
|
||||
index index.html index.htm;
|
||||
}
|
||||
|
||||
error_page 500 502 503 504 /50x.html;
|
||||
location = /50x.html {
|
||||
root /usr/share/nginx/html;
|
||||
}
|
||||
}
|
||||
}
|
||||
21
etc/services/element-web/compose.yml
Normal file
@@ -0,0 +1,21 @@
|
||||
services:
|
||||
element-web:
|
||||
image: ghcr.io/element-hq/element-web:v1.12.21
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
ELEMENT_WEB_PORT: 8080
|
||||
ports:
|
||||
- "${SERVICE_ELEMENT_WEB_BIND_PORT_HTTP}:8080"
|
||||
volumes:
|
||||
- ./element-web/config.json:/app/config.json:ro
|
||||
tmpfs:
|
||||
- /var/cache/nginx:rw,mode=777
|
||||
- /var/run:rw,mode=777
|
||||
- /tmp/element-web-config:rw,mode=777
|
||||
- /etc/nginx/conf.d:rw,mode=777
|
||||
|
||||
networks:
|
||||
default:
|
||||
name: ${NETWORK_NAME}
|
||||
external: true
|
||||
@@ -1,9 +1,9 @@
|
||||
{
|
||||
"default_hs_url": "http://synapse.127.0.0.1.nip.io:42020",
|
||||
"default_hs_url": "__HOMESERVER_CLIENT_URL__",
|
||||
"default_is_url": "https://vector.im",
|
||||
"integrations_ui_url": "https://scalar.vector.im/",
|
||||
"integrations_rest_url": "https://scalar.vector.im/api",
|
||||
"bug_report_endpoint_url": "https://riot.im/bugreports/submit",
|
||||
"bug_report_endpoint_url": "https://element.io/bugreports/submit",
|
||||
"enableLabs": true,
|
||||
"roomDirectory": {
|
||||
"servers": [
|
||||
@@ -3,6 +3,8 @@ SERVICE_SYNAPSE_BIND_PORT_FEDERATION_API=127.0.0.1:42028
|
||||
|
||||
SERVICE_ELEMENT_WEB_BIND_PORT_HTTP=127.0.0.1:42025
|
||||
|
||||
SERVICE_CONTINUWUITY_BIND_PORT_CLIENT_API=127.0.0.1:42030
|
||||
|
||||
SERVICE_OLLAMA_BIND_PORT_HTTP=127.0.0.1:42026
|
||||
|
||||
# See https://localai.io/basics/container/#all-in-one-images for the list of available images
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
ollama:
|
||||
image: docker.io/ollama/ollama:0.3.9
|
||||
image: docker.io/ollama/ollama:0.30.10
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
postgres:
|
||||
image: docker.io/postgres:16.3-alpine
|
||||
image: docker.io/postgres:18.4-alpine
|
||||
user: ${UID}:${GID}
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
@@ -8,12 +8,13 @@ services:
|
||||
POSTGRES_PASSWORD: synapse-password
|
||||
POSTGRES_DB: homeserver
|
||||
POSTGRES_INITDB_ARGS: --lc-collate C --lc-ctype C --encoding UTF8
|
||||
PGDATA: /data
|
||||
volumes:
|
||||
- ./postgres:/var/lib/postgresql/data
|
||||
- ./postgres:/data
|
||||
- /etc/passwd:/etc/passwd:ro
|
||||
|
||||
synapse:
|
||||
image: ghcr.io/element-hq/synapse:v1.114.0
|
||||
image: ghcr.io/element-hq/synapse:v1.155.0
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
entrypoint: python
|
||||
@@ -22,19 +23,9 @@ services:
|
||||
- "${SERVICE_SYNAPSE_BIND_PORT_CLIENT_API}:8008"
|
||||
- "${SERVICE_SYNAPSE_BIND_PORT_FEDERATION_API}:8008"
|
||||
volumes:
|
||||
- ../../etc/services/core/synapse/config:/config:ro
|
||||
- ../../etc/services/synapse/config:/config:ro
|
||||
- ./synapse/media-store:/media-store
|
||||
|
||||
element-web:
|
||||
image: docker.io/vectorim/element-web:v1.11.77
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${SERVICE_ELEMENT_WEB_BIND_PORT_HTTP}:8080"
|
||||
volumes:
|
||||
- ../../etc/services/core/element-web/nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
- ../../etc/services/core/element-web/config.json:/app/config.json:ro
|
||||
|
||||
networks:
|
||||
default:
|
||||
name: ${NETWORK_NAME}
|
||||
@@ -579,7 +579,7 @@ rc_login:
|
||||
#
|
||||
#federation_rr_transactions_per_room_per_second: 50
|
||||
|
||||
|
||||
enable_authenticated_media: true
|
||||
|
||||
# Directory where uploaded images and attachments are stored.
|
||||
#
|
||||
216
justfile
@@ -2,10 +2,33 @@ project_name := "baibot"
|
||||
container_image_name := "localhost/baibot"
|
||||
project_container_network := "baibot"
|
||||
|
||||
admin_username := "admin"
|
||||
admin_password := "admin"
|
||||
bot_username := "baibot"
|
||||
bot_password := "baibot"
|
||||
|
||||
homeserver := `cat var/homeserver 2>/dev/null || echo continuwuity`
|
||||
|
||||
mise_data_dir := env("MISE_DATA_DIR", justfile_directory() / "var/mise")
|
||||
mise_trusted_config_paths := justfile_directory() / "mise.toml"
|
||||
|
||||
# Show help by default
|
||||
default:
|
||||
@just --list --justfile {{ justfile() }}
|
||||
|
||||
# Selects which homeserver implementation to use (continuwuity or synapse)
|
||||
homeserver-init value:
|
||||
#!/bin/sh
|
||||
mkdir -p {{ justfile_directory() }}/var
|
||||
echo {{ value }} > {{ justfile_directory() }}/var/homeserver
|
||||
echo ""
|
||||
echo "⚠️ If you had already prepared your app configuration (var/app/local/config.yml or var/app/container/config.yml),"
|
||||
echo " you will need to update it manually or delete it and re-run the prepare step."
|
||||
echo " You should also delete var/app/local/data and/or var/app/container/data,"
|
||||
echo " as old application state is not compatible across homeserver implementations."
|
||||
echo ""
|
||||
echo "⚠️ If Element Web was already prepared, delete var/services/element-web/ to regenerate its config."
|
||||
|
||||
# Builds and runs a development binary
|
||||
run-locally *extra_args: app-local-prepare
|
||||
RUST_BACKTRACE=1 \
|
||||
@@ -32,6 +55,10 @@ run-in-container *extra_args: app-container-prepare build-container-image-debug
|
||||
test *extra_args:
|
||||
RUST_BACKTRACE=1 cargo test {{ extra_args }}
|
||||
|
||||
# Formats the code
|
||||
fmt:
|
||||
RUST_BACKTRACE=1 cargo fmt --all
|
||||
|
||||
# Builds a debug binary (target/debug/*)
|
||||
build-debug *extra_args:
|
||||
RUST_BACKTRACE=1 cargo build {{ extra_args }}
|
||||
@@ -61,9 +88,13 @@ docker-compose services_type *extra_args:
|
||||
-p {{ project_name }}-{{ services_type }} \
|
||||
{{ extra_args }}
|
||||
|
||||
# Runs a docker-compose command against the core services
|
||||
docker-compose-core *extra_args:
|
||||
just docker-compose core {{ extra_args }}
|
||||
# Runs a docker-compose command against the synapse services
|
||||
docker-compose-synapse *extra_args:
|
||||
just docker-compose synapse {{ extra_args }}
|
||||
|
||||
# Runs a docker-compose command against the element-web services
|
||||
docker-compose-element-web *extra_args:
|
||||
just docker-compose element-web {{ extra_args }}
|
||||
|
||||
# Runs a docker-compose command against the localai services
|
||||
docker-compose-localai *extra_args:
|
||||
@@ -73,17 +104,52 @@ docker-compose-localai *extra_args:
|
||||
docker-compose-ollama *extra_args:
|
||||
just docker-compose ollama {{ extra_args }}
|
||||
|
||||
# Runs all core dependency components (in the background)
|
||||
services-start: services-prepare (docker-compose-core "up" "-d")
|
||||
# Runs a docker-compose command against the continuwuity services
|
||||
docker-compose-continuwuity *extra_args:
|
||||
just docker-compose continuwuity {{ extra_args }}
|
||||
|
||||
# Stops all core dependency components
|
||||
services-stop: (docker-compose-core "down")
|
||||
# Runs the homeserver and Element Web (in the background)
|
||||
services-start: services-prepare
|
||||
just -f {{ justfile_directory() }}/justfile {{ homeserver }}-start
|
||||
just -f {{ justfile_directory() }}/justfile element-web-start
|
||||
|
||||
# Tails the logs for all running core services
|
||||
services-tail-logs: (docker-compose-core "logs" "-f")
|
||||
# Stops Element Web and the homeserver
|
||||
services-stop:
|
||||
just -f {{ justfile_directory() }}/justfile element-web-stop
|
||||
just -f {{ justfile_directory() }}/justfile {{ homeserver }}-stop
|
||||
|
||||
# Prepares the core services for running
|
||||
services-prepare: _prepare-var-services-env _prepare-var-services-postgres _prepare-var-services-synapse _prepare-container-network
|
||||
# Tails the logs for the homeserver and Element Web
|
||||
services-tail-logs:
|
||||
just -f {{ justfile_directory() }}/justfile {{ homeserver }}-tail-logs
|
||||
|
||||
# Prepares the homeserver and Element Web for running
|
||||
services-prepare:
|
||||
just -f {{ justfile_directory() }}/justfile {{ homeserver }}-prepare
|
||||
just -f {{ justfile_directory() }}/justfile element-web-prepare
|
||||
|
||||
# Runs Synapse (in the background)
|
||||
synapse-start: synapse-prepare (docker-compose-synapse "up" "-d")
|
||||
|
||||
# Stops Synapse
|
||||
synapse-stop: (docker-compose-synapse "down")
|
||||
|
||||
# Tails the logs for Synapse
|
||||
synapse-tail-logs: (docker-compose-synapse "logs" "-f")
|
||||
|
||||
# Prepares Synapse for running
|
||||
synapse-prepare: _prepare-var-services-env _prepare-var-services-postgres _prepare-var-services-synapse _prepare-container-network
|
||||
|
||||
# Runs Element Web (in the background)
|
||||
element-web-start: element-web-prepare (docker-compose-element-web "up" "-d")
|
||||
|
||||
# Stops Element Web
|
||||
element-web-stop: (docker-compose-element-web "down")
|
||||
|
||||
# Tails the logs for Element Web
|
||||
element-web-tail-logs: (docker-compose-element-web "logs" "-f")
|
||||
|
||||
# Prepares Element Web for running
|
||||
element-web-prepare: _prepare-var-services-env _prepare-var-services-element-web _prepare-container-network
|
||||
|
||||
# Runs LocalAI (in the background)
|
||||
localai-start: localai-prepare (docker-compose-localai "up" "-d")
|
||||
@@ -109,6 +175,27 @@ ollama-tail-logs: (docker-compose-ollama "logs" "-f")
|
||||
# Prepares Ollama for running
|
||||
ollama-prepare: _prepare-var-services-env _prepare-var-services-ollama _prepare-container-network
|
||||
|
||||
# Runs Continuwuity (in the background)
|
||||
continuwuity-start: continuwuity-prepare (docker-compose-continuwuity "up" "-d")
|
||||
|
||||
# Stops Continuwuity
|
||||
continuwuity-stop: (docker-compose-continuwuity "down")
|
||||
|
||||
# Tails the logs for Continuwuity
|
||||
continuwuity-tail-logs: (docker-compose-continuwuity "logs" "-f")
|
||||
|
||||
# Prepares Continuwuity for running
|
||||
continuwuity-prepare: _prepare-var-services-env _prepare-var-services-continuwuity _prepare-container-network
|
||||
|
||||
# Registers a user on Continuwuity via the Matrix Client-Server API
|
||||
continuwuity-register-user username password:
|
||||
{{ justfile_directory() }}/etc/services/continuwuity/register-user.sh {{ justfile_directory() }}/var/services/env {{ username }} {{ password }}
|
||||
|
||||
# Prepares the Continuwuity user accounts
|
||||
continuwuity-users-prepare: continuwuity-prepare
|
||||
just -f {{ justfile_directory() }}/justfile continuwuity-register-user "{{ admin_username }}" "{{ admin_password }}"
|
||||
just -f {{ justfile_directory() }}/justfile continuwuity-register-user "{{ bot_username }}" "{{ bot_password }}"
|
||||
|
||||
# Pulls an Ollama model
|
||||
ollama-pull-model model_id:
|
||||
just -f {{ justfile_directory() }}/justfile docker-compose-ollama \
|
||||
@@ -122,16 +209,20 @@ app-local-prepare: _prepare-var-app-local-config_yml _prepare-var-app-local-data
|
||||
app-container-prepare: _prepare-var-app-container-config_yml _prepare-var-app-container-data
|
||||
|
||||
# Prepares the user accounts
|
||||
users-prepare: services-prepare
|
||||
just -f {{ justfile_directory() }}/justfile synapse-register-admin-user "admin" "admin"
|
||||
just -f {{ justfile_directory() }}/justfile synapse-register-regular-user "baibot" "baibot"
|
||||
users-prepare:
|
||||
just -f {{ justfile_directory() }}/justfile {{ homeserver }}-users-prepare
|
||||
|
||||
# Prepares the Synapse user accounts
|
||||
synapse-users-prepare: synapse-prepare
|
||||
just -f {{ justfile_directory() }}/justfile synapse-register-admin-user "{{ admin_username }}" "{{ admin_password }}"
|
||||
just -f {{ justfile_directory() }}/justfile synapse-register-regular-user "{{ bot_username }}" "{{ bot_password }}"
|
||||
|
||||
# Starts a Postgres CLI (psql)
|
||||
postgres-cli: services-prepare (docker-compose-core "exec" "postgres" "/bin/sh" "-c" "'PGUSER=synapse PGPASSWORD=synapse-password PGDATABASE=homeserver psql -h postgres'")
|
||||
postgres-cli: synapse-prepare (docker-compose-synapse "exec" "postgres" "/bin/sh" "-c" "'PGUSER=synapse PGPASSWORD=synapse-password PGDATABASE=homeserver psql -h postgres'")
|
||||
|
||||
# Creates an administrator user
|
||||
synapse-register-admin-user username password: services-prepare
|
||||
just -f {{ justfile_directory() }}/justfile docker-compose-core \
|
||||
# Creates an administrator user on Synapse
|
||||
synapse-register-admin-user username password: synapse-prepare
|
||||
just -f {{ justfile_directory() }}/justfile docker-compose-synapse \
|
||||
exec synapse \
|
||||
register_new_matrix_user \
|
||||
--admin \
|
||||
@@ -140,9 +231,9 @@ synapse-register-admin-user username password: services-prepare
|
||||
-c /config/homeserver.yaml \
|
||||
http://localhost:8008
|
||||
|
||||
# Create a regular user
|
||||
synapse-register-regular-user username password: services-prepare
|
||||
just -f {{ justfile_directory() }}/justfile docker-compose-core \
|
||||
# Creates a regular user on Synapse
|
||||
synapse-register-regular-user username password: synapse-prepare
|
||||
just -f {{ justfile_directory() }}/justfile docker-compose-synapse \
|
||||
exec synapse \
|
||||
register_new_matrix_user \
|
||||
--no-admin \
|
||||
@@ -155,6 +246,44 @@ synapse-register-regular-user username password: services-prepare
|
||||
clippy *extra_args:
|
||||
cargo clippy {{ extra_args }}
|
||||
|
||||
# Checks that the code compiles without building
|
||||
check:
|
||||
cargo check
|
||||
|
||||
# Invokes mise with the project-local data directory
|
||||
mise *args: _ensure_mise_data_directory
|
||||
#!/bin/sh
|
||||
export MISE_DATA_DIR="{{ mise_data_dir }}"
|
||||
export MISE_TRUSTED_CONFIG_PATHS="{{ mise_trusted_config_paths }}"
|
||||
mise {{ args }}
|
||||
|
||||
# Runs prek (pre-commit hooks manager) with the given arguments
|
||||
prek *args: _ensure_mise_tools_installed
|
||||
@just --justfile {{ justfile() }} mise exec -- prek {{ args }}
|
||||
|
||||
# Runs pre-commit hooks on staged files
|
||||
prek-run-on-staged *args: _ensure_mise_tools_installed
|
||||
@just --justfile {{ justfile() }} mise exec -- prek run {{ args }}
|
||||
|
||||
# Runs pre-commit hooks on all files
|
||||
prek-run-on-all *args: _ensure_mise_tools_installed
|
||||
@just --justfile {{ justfile() }} mise exec -- prek run --all-files {{ args }}
|
||||
|
||||
# Installs the git pre-commit hook (runs prek automatically before each commit)
|
||||
prek-install-git-pre-commit-hook: _ensure_mise_tools_installed
|
||||
@just --justfile {{ justfile() }} mise exec -- prek install
|
||||
|
||||
# Internal - ensures var/mise directory exists
|
||||
_ensure_mise_data_directory:
|
||||
#!/bin/sh
|
||||
if [ ! -d "{{ mise_data_dir }}" ]; then
|
||||
mkdir -p "{{ mise_data_dir }}"
|
||||
fi
|
||||
|
||||
# Internal - ensures mise tools are installed
|
||||
_ensure_mise_tools_installed: _ensure_mise_data_directory
|
||||
@just --justfile {{ justfile() }} mise install --quiet
|
||||
|
||||
_prepare-var-services-env:
|
||||
#!/bin/sh
|
||||
cd {{ justfile_directory() }};
|
||||
@@ -184,6 +313,22 @@ _prepare-var-services-synapse:
|
||||
mkdir -p var/services/synapse/media-store
|
||||
fi
|
||||
|
||||
_prepare-var-services-element-web:
|
||||
#!/bin/sh
|
||||
cd {{ justfile_directory() }};
|
||||
|
||||
if [ ! -f var/services/element-web/config.json ]; then
|
||||
mkdir -p var/services/element-web
|
||||
cp {{ justfile_directory() }}/etc/services/element-web/config.json.dist var/services/element-web/config.json
|
||||
|
||||
homeserver="{{ homeserver }}"
|
||||
if [ "$homeserver" = "continuwuity" ]; then
|
||||
sed --in-place 's|__HOMESERVER_CLIENT_URL__|http://continuwuity.127.0.0.1.nip.io:42030|g' var/services/element-web/config.json
|
||||
elif [ "$homeserver" = "synapse" ]; then
|
||||
sed --in-place 's|__HOMESERVER_CLIENT_URL__|http://synapse.127.0.0.1.nip.io:42020|g' var/services/element-web/config.json
|
||||
fi
|
||||
fi
|
||||
|
||||
_prepare-var-services-ollama:
|
||||
#!/bin/sh
|
||||
cd {{ justfile_directory() }};
|
||||
@@ -192,6 +337,14 @@ _prepare-var-services-ollama:
|
||||
mkdir -p var/services/ollama
|
||||
fi
|
||||
|
||||
_prepare-var-services-continuwuity:
|
||||
#!/bin/sh
|
||||
cd {{ justfile_directory() }};
|
||||
|
||||
if [ ! -f var/services/continuwuity ]; then
|
||||
mkdir -p var/services/continuwuity/data
|
||||
fi
|
||||
|
||||
_prepare-var-services-localai:
|
||||
#!/bin/sh
|
||||
cd {{ justfile_directory() }};
|
||||
@@ -215,6 +368,15 @@ _prepare-var-app-local-config_yml:
|
||||
if [ ! -f var/app/local/config.yml ]; then
|
||||
mkdir -p var/app/local
|
||||
cp {{ justfile_directory() }}/etc/app/config.yml.dist var/app/local/config.yml
|
||||
|
||||
homeserver="{{ homeserver }}"
|
||||
if [ "$homeserver" = "continuwuity" ]; then
|
||||
sed --in-place 's/__HOMESERVER_SERVER_NAME__/continuwuity.127.0.0.1.nip.io/g' var/app/local/config.yml
|
||||
sed --in-place 's|__HOMESERVER_URL__|http://continuwuity.127.0.0.1.nip.io:42030|g' var/app/local/config.yml
|
||||
elif [ "$homeserver" = "synapse" ]; then
|
||||
sed --in-place 's/__HOMESERVER_SERVER_NAME__/synapse.127.0.0.1.nip.io/g' var/app/local/config.yml
|
||||
sed --in-place 's|__HOMESERVER_URL__|http://synapse.127.0.0.1.nip.io:42020|g' var/app/local/config.yml
|
||||
fi
|
||||
fi
|
||||
|
||||
_prepare-var-app-local-data:
|
||||
@@ -232,7 +394,18 @@ _prepare-var-app-container-config_yml:
|
||||
if [ ! -f var/app/container/config.yml ]; then
|
||||
mkdir -p var/app/container
|
||||
cp {{ justfile_directory() }}/etc/app/config.yml.dist var/app/container/config.yml
|
||||
|
||||
homeserver="{{ homeserver }}"
|
||||
if [ "$homeserver" = "continuwuity" ]; then
|
||||
sed --in-place 's/__HOMESERVER_SERVER_NAME__/continuwuity.127.0.0.1.nip.io/g' var/app/container/config.yml
|
||||
sed --in-place 's|__HOMESERVER_URL__|http://continuwuity.127.0.0.1.nip.io:42030|g' var/app/container/config.yml
|
||||
sed --in-place 's/continuwuity.127.0.0.1.nip.io:42030/continuwuity:6167/g' var/app/container/config.yml
|
||||
elif [ "$homeserver" = "synapse" ]; then
|
||||
sed --in-place 's/__HOMESERVER_SERVER_NAME__/synapse.127.0.0.1.nip.io/g' var/app/container/config.yml
|
||||
sed --in-place 's|__HOMESERVER_URL__|http://synapse.127.0.0.1.nip.io:42020|g' var/app/container/config.yml
|
||||
sed --in-place 's/synapse.127.0.0.1.nip.io:42020/synapse:8008/g' var/app/container/config.yml
|
||||
fi
|
||||
|
||||
sed --in-place 's/127.0.0.1:42026/ollama:11434/g' var/app/container/config.yml
|
||||
sed --in-place 's/127.0.0.1:42027/localai:8080/g' var/app/container/config.yml
|
||||
fi
|
||||
@@ -244,4 +417,3 @@ _prepare-var-app-container-data:
|
||||
if [ ! -f var/app/container/data ]; then
|
||||
mkdir -p var/app/container/data
|
||||
fi
|
||||
|
||||
|
||||
6
mise.toml
Normal file
@@ -0,0 +1,6 @@
|
||||
[tools]
|
||||
prek = "0.4.5"
|
||||
|
||||
[settings]
|
||||
# Disable automatic trust prompts - we trust this config
|
||||
yes = true
|
||||
9
renovate.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
|
||||
"extends": [
|
||||
"config:recommended"
|
||||
],
|
||||
"labels": [
|
||||
"dependencies"
|
||||
]
|
||||
}
|
||||
4
rust-toolchain.toml
Normal file
@@ -0,0 +1,4 @@
|
||||
[toolchain]
|
||||
channel = "1.96.0"
|
||||
components = ["rustfmt", "clippy"]
|
||||
profile = "default"
|
||||
@@ -33,11 +33,11 @@ pub struct AgentDefinition {
|
||||
)]
|
||||
pub provider: AgentProvider,
|
||||
|
||||
pub config: serde_yaml::Value,
|
||||
pub config: serde_yaml_ng::Value,
|
||||
}
|
||||
|
||||
impl AgentDefinition {
|
||||
pub fn new(id: String, provider: AgentProvider, config: serde_yaml::Value) -> Self {
|
||||
pub fn new(id: String, provider: AgentProvider, config: serde_yaml_ng::Value) -> Self {
|
||||
Self {
|
||||
id,
|
||||
provider,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use super::{
|
||||
provider::{self, ControllerType},
|
||||
AgentDefinition, AgentProvider, PublicIdentifier,
|
||||
provider::{self, ControllerType},
|
||||
};
|
||||
|
||||
// Dead-code is allowed. We do not use these enum struct payloads directly,
|
||||
@@ -15,7 +15,7 @@ pub enum Error {
|
||||
// Contains the error from the constructor function
|
||||
ConstructionFailed(anyhow::Error),
|
||||
// Contains the error from the YAML deserialization function
|
||||
Yaml(serde_yaml::Error),
|
||||
Yaml(serde_yaml_ng::Error),
|
||||
}
|
||||
|
||||
pub type Result<T> = std::result::Result<T, Error>;
|
||||
@@ -69,7 +69,7 @@ pub(super) fn create(
|
||||
pub fn create_from_provider_and_yaml_value_config(
|
||||
provider: &AgentProvider,
|
||||
identifier: &PublicIdentifier,
|
||||
config: serde_yaml::Value,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> Result<AgentInstance> {
|
||||
let definition = AgentDefinition::new(identifier.prefixless(), provider.to_owned(), config);
|
||||
|
||||
@@ -79,7 +79,7 @@ pub fn create_from_provider_and_yaml_value_config(
|
||||
fn create_controller_from_provider_and_json_value_config(
|
||||
agent_id: &str,
|
||||
provider: &AgentProvider,
|
||||
config: serde_yaml::Value,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> Result<ControllerType> {
|
||||
match provider {
|
||||
AgentProvider::Anthropic => {
|
||||
@@ -109,46 +109,53 @@ fn create_controller_from_provider_and_json_value_config(
|
||||
AgentProvider::TogetherAI => {
|
||||
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
|
||||
}
|
||||
AgentProvider::Venice => {
|
||||
provider::venice::create_controller_from_yaml_value_config(agent_id, config)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn default_config_for_provider(provider: &AgentProvider) -> serde_yaml::Value {
|
||||
pub fn default_config_for_provider(provider: &AgentProvider) -> serde_yaml_ng::Value {
|
||||
match provider {
|
||||
AgentProvider::Anthropic => {
|
||||
let config = super::provider::anthropic::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::Groq => {
|
||||
let config = super::provider::groq::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::LocalAI => {
|
||||
let config = super::provider::localai::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::Mistral => {
|
||||
let config = super::provider::mistral::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::Ollama => {
|
||||
let config = super::provider::ollama::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::OpenAI => {
|
||||
let config = super::provider::openai::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::OpenAICompat => {
|
||||
let config = super::provider::openai_compat::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::OpenRouter => {
|
||||
let config = super::provider::openrouter::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::TogetherAI => {
|
||||
let config = super::provider::togetherai::default_config();
|
||||
serde_yaml::to_value(config).expect("Failed to serialize config")
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
AgentProvider::Venice => {
|
||||
let config = super::provider::venice::default_config();
|
||||
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use super::instantiation;
|
||||
use super::instantiation::AgentInstance;
|
||||
use super::AgentDefinition;
|
||||
use super::PublicIdentifier;
|
||||
use super::instantiation;
|
||||
use super::instantiation::AgentInstance;
|
||||
use crate::entity::RoomConfigContext;
|
||||
|
||||
#[derive(Debug)]
|
||||
|
||||
@@ -11,15 +11,15 @@ pub use manager::Manager;
|
||||
|
||||
pub use definition::AgentDefinition;
|
||||
|
||||
pub use instantiation::create_from_provider_and_yaml_value_config;
|
||||
pub use instantiation::default_config_for_provider;
|
||||
pub use instantiation::AgentInstance;
|
||||
pub use instantiation::Error as AgentInstantiationError;
|
||||
pub use instantiation::Result as AgentInstantiationResult;
|
||||
pub use instantiation::create_from_provider_and_yaml_value_config;
|
||||
pub use instantiation::default_config_for_provider;
|
||||
|
||||
pub use provider::{AgentProvider, AgentProviderInfo, ControllerTrait};
|
||||
pub use purpose::AgentPurpose;
|
||||
|
||||
pub(super) fn default_prompt() -> &'static str {
|
||||
"You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time now is: {{ baibot_now_utc }}."
|
||||
"You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
}
|
||||
|
||||
@@ -1,7 +1,5 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use anthropic_rs::models::claude::ClaudeModel;
|
||||
|
||||
use crate::agent::{default_prompt, provider::ConfigTrait};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -28,6 +26,9 @@ impl ConfigTrait for Config {
|
||||
if self.base_url.is_empty() {
|
||||
return Err("The base URL must not be empty.".to_owned());
|
||||
}
|
||||
if !self.base_url.ends_with("/v1") {
|
||||
return Err("The base URL must end with '/v1'.".to_owned());
|
||||
}
|
||||
if self.api_key.is_empty() {
|
||||
return Err("The API key must not be empty.".to_owned());
|
||||
}
|
||||
@@ -67,5 +68,5 @@ impl Default for TextGenerationConfig {
|
||||
}
|
||||
|
||||
fn default_text_model_id() -> String {
|
||||
ClaudeModel::Claude35Sonnet.as_str().to_owned()
|
||||
"claude-3-7-sonnet-20250219".to_owned()
|
||||
}
|
||||
|
||||
@@ -1,30 +1,28 @@
|
||||
use std::fmt::Debug;
|
||||
use std::str::FromStr;
|
||||
use std::sync::Arc;
|
||||
|
||||
use anthropic_rs::completion::message::ContentType;
|
||||
use anthropic_rs::{
|
||||
client::Client as AnthropicClient, config::Config as AnthropicConfig,
|
||||
models::claude::ClaudeModel,
|
||||
};
|
||||
use anthropic::client::{Client, ClientBuilder};
|
||||
use anthropic::types::ContentBlock;
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::agent::provider::entity::{
|
||||
ImageGenerationResult, PingResult, TextGenerationParams, TextGenerationResult,
|
||||
TextToSpeechParams, TextToSpeechResult,
|
||||
};
|
||||
use crate::agent::provider::{ImageGenerationParams, SpeechToTextParams, SpeechToTextResult};
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
|
||||
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
|
||||
};
|
||||
use crate::agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
};
|
||||
use crate::conversation::llm::{
|
||||
shorten_messages_list_to_context_size, Author as LLMAuthor, Conversation as LLMConversation,
|
||||
Message as LLMMessage,
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
use super::config::Config;
|
||||
|
||||
struct ControllerInner {
|
||||
client: AnthropicClient,
|
||||
client: Client,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
@@ -43,18 +41,20 @@ impl Debug for Controller {
|
||||
|
||||
impl Controller {
|
||||
pub fn new(config: Config) -> anyhow::Result<Self> {
|
||||
let anthropic_config =
|
||||
AnthropicConfig::new(config.api_key.clone()).with_base_url(config.base_url.clone());
|
||||
|
||||
let client = match AnthropicClient::new(anthropic_config) {
|
||||
Ok(client) => client,
|
||||
Err(err) => {
|
||||
return Err(anyhow::anyhow!(
|
||||
"Failed to create Anthropic client: {}",
|
||||
err.to_string()
|
||||
));
|
||||
// The previous library that we used expected a base URL that ends with "/v1"
|
||||
// (e.g. "https://api.anthropic.com/v1"), while the new one doesn't.
|
||||
//
|
||||
// To keep backward compatibility, we don't ask people to change their configuration
|
||||
// and rather adapt by removing the "/v1" from the base URL.
|
||||
if !config.base_url.ends_with("/v1") {
|
||||
return Err(anyhow::anyhow!("base_url must end with '/v1'"));
|
||||
}
|
||||
};
|
||||
|
||||
let base_url = &config.base_url[..config.base_url.len() - 3];
|
||||
let client = ClientBuilder::default()
|
||||
.api_base(base_url.to_string())
|
||||
.api_key(config.api_key.clone())
|
||||
.build()?;
|
||||
|
||||
Ok(Self {
|
||||
config,
|
||||
@@ -71,7 +71,9 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
message_text: "Hello!".to_string(),
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
let conversation = LLMConversation { messages };
|
||||
@@ -107,7 +109,9 @@ impl ControllerTrait for Controller {
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
message_text: prompt_text,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
|
||||
@@ -129,7 +133,7 @@ impl ControllerTrait for Controller {
|
||||
&text_generation_config.model_id,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
Some(text_generation_config.max_response_tokens),
|
||||
text_generation_config.max_context_tokens,
|
||||
);
|
||||
|
||||
@@ -140,29 +144,19 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let mut request = super::utils::create_anthropic_message_request(conversation_messages);
|
||||
|
||||
let model = match ClaudeModel::from_str(&text_generation_config.model_id) {
|
||||
Ok(model) => model,
|
||||
Err(err) => {
|
||||
tracing::debug!(?err, "Failed to parse model ID");
|
||||
|
||||
return Err(anyhow::anyhow!(
|
||||
"Failed to parse model ID: {}",
|
||||
&text_generation_config.model_id
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
let temperature = params
|
||||
.temperature_override
|
||||
.unwrap_or(text_generation_config.temperature);
|
||||
|
||||
if let Some(prompt_message) = prompt_message {
|
||||
request.system = Some(prompt_message.message_text);
|
||||
if let Some(prompt_message) = prompt_message
|
||||
&& let LLMMessageContent::Text(text) = &prompt_message.content
|
||||
{
|
||||
request.system = text.clone();
|
||||
}
|
||||
|
||||
request.model = model;
|
||||
request.temperature = Some(temperature);
|
||||
request.max_tokens = text_generation_config.max_response_tokens;
|
||||
request.model = text_generation_config.model_id.clone();
|
||||
request.temperature = Some(temperature as f64);
|
||||
request.max_tokens = text_generation_config.max_response_tokens as usize;
|
||||
|
||||
if let Ok(request_as_json) = serde_json::to_string(&request) {
|
||||
tracing::trace!(
|
||||
@@ -173,19 +167,20 @@ impl ControllerTrait for Controller {
|
||||
);
|
||||
}
|
||||
|
||||
let response = self.inner.client.create_message(request).await?;
|
||||
let response = self.inner.client.messages(request).await?;
|
||||
|
||||
tracing::trace!(?response, "Got response from Anthropic create message API");
|
||||
|
||||
// response.content usually contains a single element, but we support handling multiple to account for all possibilities
|
||||
let mut text_parts = vec![];
|
||||
for content in response.content {
|
||||
let content_type = content.content_type;
|
||||
|
||||
match content_type {
|
||||
ContentType::Text => {
|
||||
text_parts.push(content.text);
|
||||
} // There are no other content types to handle yet, but there may be in the future
|
||||
match content {
|
||||
ContentBlock::Text { text } => {
|
||||
text_parts.push(text);
|
||||
}
|
||||
ContentBlock::Image { .. } => {
|
||||
text_parts.push("The model responded with an image".to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -217,6 +212,15 @@ impl ControllerTrait for Controller {
|
||||
Err(anyhow::anyhow!("Image generation not supported"))
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
_prompt: &str,
|
||||
_images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
Err(anyhow::anyhow!("Image editing is not supported"))
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
_input: &str,
|
||||
|
||||
@@ -7,17 +7,17 @@ pub use controller::Controller;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::controller::ControllerType;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
config: serde_yaml::Value,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> AgentInstantiationResult<ControllerType> {
|
||||
let config = match &config {
|
||||
serde_yaml::Value::Mapping(_) => {
|
||||
serde_yaml_ng::Value::Mapping(_) => {
|
||||
let config: Config =
|
||||
serde_yaml::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
|
||||
config
|
||||
.validate()
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
use anthropic_rs::completion::message::{Content, ContentType, Message, MessageRequest, Role};
|
||||
use anthropic::types::{
|
||||
ContentBlock, ImageSource, Message, MessagesRequest, MessagesRequestBuilder, Role,
|
||||
};
|
||||
|
||||
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) -> MessageRequest {
|
||||
pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) -> MessagesRequest {
|
||||
let mut messages = vec![];
|
||||
|
||||
for message in llm_messages {
|
||||
@@ -14,19 +18,33 @@ pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) ->
|
||||
}
|
||||
};
|
||||
|
||||
let content = vec![Content {
|
||||
content_type: ContentType::Text,
|
||||
text: message.message_text,
|
||||
}];
|
||||
let content = match &message.content {
|
||||
LLMMessageContent::Text(text) => vec![ContentBlock::Text { text: text.clone() }],
|
||||
LLMMessageContent::Image(image_details) => {
|
||||
vec![ContentBlock::Image {
|
||||
source: ImageSource::Base64 {
|
||||
media_type: image_details.mime.to_string(),
|
||||
data: crate::utils::base64::base64_encode(&image_details.data),
|
||||
},
|
||||
}]
|
||||
}
|
||||
LLMMessageContent::File(file_details) => {
|
||||
tracing::warn!(
|
||||
"The Anthropic provider's library does not support file/document content. This file message ({}) will be skipped.",
|
||||
file_details.filename(),
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
let message = Message { role, content };
|
||||
|
||||
messages.push(message);
|
||||
}
|
||||
|
||||
MessageRequest {
|
||||
stream: false,
|
||||
messages,
|
||||
..Default::default()
|
||||
}
|
||||
MessagesRequestBuilder::default()
|
||||
.messages(messages)
|
||||
.stream(false)
|
||||
.build()
|
||||
.expect("Failed to build messages request")
|
||||
}
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
use crate::{agent::AgentPurpose, conversation::llm::Conversation};
|
||||
|
||||
use super::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
entity::{
|
||||
ImageGenerationResult, PingResult, TextGenerationParams, TextGenerationResult,
|
||||
TextToSpeechParams, TextToSpeechResult,
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
|
||||
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
|
||||
},
|
||||
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
};
|
||||
|
||||
pub trait ControllerTrait {
|
||||
@@ -42,6 +42,13 @@ pub trait ControllerTrait {
|
||||
params: ImageGenerationParams,
|
||||
) -> impl std::future::Future<Output = anyhow::Result<ImageGenerationResult>> + Send;
|
||||
|
||||
fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
params: ImageEditParams,
|
||||
) -> impl std::future::Future<Output = anyhow::Result<ImageEditResult>> + Send;
|
||||
|
||||
fn text_to_speech(
|
||||
&self,
|
||||
text: &str,
|
||||
@@ -54,6 +61,7 @@ pub enum ControllerType {
|
||||
OpenAI(Box<super::openai::Controller>),
|
||||
OpenAICompat(Box<super::openai_compat::Controller>),
|
||||
Anthropic(Box<super::anthropic::Controller>),
|
||||
Venice(Box<super::venice::Controller>),
|
||||
}
|
||||
|
||||
impl ControllerTrait for ControllerType {
|
||||
@@ -62,6 +70,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.supports_purpose(purpose),
|
||||
ControllerType::OpenAICompat(controller) => controller.supports_purpose(purpose),
|
||||
ControllerType::Anthropic(controller) => controller.supports_purpose(purpose),
|
||||
ControllerType::Venice(controller) => controller.supports_purpose(purpose),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -70,6 +79,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_generation_model_id(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_generation_model_id(),
|
||||
ControllerType::Anthropic(controller) => controller.text_generation_model_id(),
|
||||
ControllerType::Venice(controller) => controller.text_generation_model_id(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -78,6 +88,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_generation_prompt(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_generation_prompt(),
|
||||
ControllerType::Anthropic(controller) => controller.text_generation_prompt(),
|
||||
ControllerType::Venice(controller) => controller.text_generation_prompt(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -86,6 +97,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_to_speech_voice(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_to_speech_voice(),
|
||||
ControllerType::Anthropic(controller) => controller.text_to_speech_voice(),
|
||||
ControllerType::Venice(controller) => controller.text_to_speech_voice(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -94,6 +106,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_to_speech_speed(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_to_speech_speed(),
|
||||
ControllerType::Anthropic(controller) => controller.text_to_speech_speed(),
|
||||
ControllerType::Venice(controller) => controller.text_to_speech_speed(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -102,6 +115,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.text_generation_temperature(),
|
||||
ControllerType::OpenAICompat(controller) => controller.text_generation_temperature(),
|
||||
ControllerType::Anthropic(controller) => controller.text_generation_temperature(),
|
||||
ControllerType::Venice(controller) => controller.text_generation_temperature(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -110,6 +124,7 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::OpenAI(controller) => controller.ping().await,
|
||||
ControllerType::OpenAICompat(controller) => controller.ping().await,
|
||||
ControllerType::Anthropic(controller) => controller.ping().await,
|
||||
ControllerType::Venice(controller) => controller.ping().await,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -128,6 +143,9 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.generate_text(conversation, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.generate_text(conversation, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -147,6 +165,9 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.speech_to_text(mime_type, media, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.speech_to_text(mime_type, media, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -163,6 +184,31 @@ impl ControllerTrait for ControllerType {
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.generate_image(prompt, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.generate_image(prompt, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
match &self {
|
||||
ControllerType::OpenAI(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
ControllerType::OpenAICompat(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
ControllerType::Venice(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -177,6 +223,7 @@ impl ControllerTrait for ControllerType {
|
||||
controller.text_to_speech(text, params).await
|
||||
}
|
||||
ControllerType::Anthropic(controller) => controller.text_to_speech(text, params).await,
|
||||
ControllerType::Venice(controller) => controller.text_to_speech(text, params).await,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ pub enum AgentProvider {
|
||||
OpenAICompat,
|
||||
OpenRouter,
|
||||
TogetherAI,
|
||||
Venice,
|
||||
}
|
||||
|
||||
impl AgentProvider {
|
||||
@@ -25,6 +26,7 @@ impl AgentProvider {
|
||||
&Self::OpenAICompat,
|
||||
&Self::OpenRouter,
|
||||
&Self::TogetherAI,
|
||||
&Self::Venice,
|
||||
]
|
||||
}
|
||||
|
||||
@@ -39,6 +41,7 @@ impl AgentProvider {
|
||||
Self::OpenAICompat => "openai-compatible",
|
||||
Self::OpenRouter => "openrouter",
|
||||
Self::TogetherAI => "together-ai",
|
||||
Self::Venice => "venice",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -53,6 +56,7 @@ impl AgentProvider {
|
||||
"openai-compatible" => Ok(Self::OpenAICompat),
|
||||
"openrouter" => Ok(Self::OpenRouter),
|
||||
"together-ai" => Ok(Self::TogetherAI),
|
||||
"venice" => Ok(Self::Venice),
|
||||
_ => Err("Unexpected string value"),
|
||||
}
|
||||
}
|
||||
@@ -67,9 +71,9 @@ impl AgentProvider {
|
||||
wiki_url: Some("https://en.wikipedia.org/wiki/Anthropic"),
|
||||
sign_up_url: Some("https://console.anthropic.com/"),
|
||||
models_list_url: Some("https://docs.anthropic.com/en/docs/about-claude/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: true,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::Groq => AgentProviderInfo {
|
||||
id: Self::Groq.to_static_str(),
|
||||
@@ -79,15 +83,14 @@ impl AgentProvider {
|
||||
wiki_url: Some("https://en.wikipedia.org/wiki/Groq"),
|
||||
sign_up_url: Some("https://console.groq.com/login"),
|
||||
models_list_url: Some("https://console.groq.com/docs/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration, AgentPurpose::SpeechToText],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::LocalAI => AgentProviderInfo {
|
||||
id: Self::LocalAI.to_static_str(),
|
||||
name: "LocalAI",
|
||||
description: "LocalAI is the free, Open Source OpenAI alternative. LocalAI act as a drop-in replacement REST API that’s compatible with OpenAI API specifications for local inferencing. It allows you to run LLMs, generate images, audio (and not only) locally or on-prem with consumer grade hardware, supporting multiple model families and architectures.",
|
||||
description: "LocalAI is the free, Open Source OpenAI alternative. LocalAI act as a drop-in replacement REST API that's compatible with OpenAI API specifications for local inferencing. It allows you to run LLMs, generate images, audio (and not only) locally or on-prem with consumer grade hardware, supporting multiple model families and architectures.",
|
||||
homepage_url: Some("https://localai.io/"),
|
||||
wiki_url: None,
|
||||
sign_up_url: None,
|
||||
@@ -97,6 +100,8 @@ impl AgentProvider {
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::Mistral => AgentProviderInfo {
|
||||
id: Self::Mistral.to_static_str(),
|
||||
@@ -106,9 +111,9 @@ impl AgentProvider {
|
||||
wiki_url: Some("https://en.wikipedia.org/wiki/Mistral_AI"),
|
||||
sign_up_url: Some("https://auth.mistral.ai/ui/registration"),
|
||||
models_list_url: Some("https://docs.mistral.ai/getting-started/models/"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::Ollama => AgentProviderInfo {
|
||||
id: Self::Ollama.to_static_str(),
|
||||
@@ -118,9 +123,9 @@ impl AgentProvider {
|
||||
wiki_url: None,
|
||||
sign_up_url: None,
|
||||
models_list_url: Some("https://ollama.com/library"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::OpenAI => AgentProviderInfo {
|
||||
id: Self::OpenAI.to_static_str(),
|
||||
@@ -136,6 +141,8 @@ impl AgentProvider {
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: true,
|
||||
text_generation_supports_tools: true,
|
||||
},
|
||||
Self::OpenAICompat => AgentProviderInfo {
|
||||
id: Self::OpenAICompat.to_static_str(),
|
||||
@@ -151,6 +158,8 @@ impl AgentProvider {
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::OpenRouter => AgentProviderInfo {
|
||||
id: Self::OpenRouter.to_static_str(),
|
||||
@@ -160,9 +169,9 @@ impl AgentProvider {
|
||||
wiki_url: None,
|
||||
sign_up_url: Some("https://openrouter.ai/"),
|
||||
models_list_url: Some("https://openrouter.ai/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::TogetherAI => AgentProviderInfo {
|
||||
id: Self::TogetherAI.to_static_str(),
|
||||
@@ -172,9 +181,28 @@ impl AgentProvider {
|
||||
wiki_url: None,
|
||||
sign_up_url: Some("https://api.together.ai/signup"),
|
||||
models_list_url: Some("https://api.together.xyz/models"),
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
Self::Venice => AgentProviderInfo {
|
||||
id: Self::Venice.to_static_str(),
|
||||
name: "Venice",
|
||||
description: "Venice AI runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses. It serves frontier proprietary and open-source models with text-generation (including vision), speech-to-text, text-to-speech, native image generation and editing, and native web search.",
|
||||
homepage_url: Some("https://venice.ai"),
|
||||
wiki_url: None,
|
||||
sign_up_url: Some("https://venice.ai"),
|
||||
models_list_url: Some("https://api.venice.ai/api/v1/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::ImageGeneration,
|
||||
AgentPurpose::TextGeneration,
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: true,
|
||||
// Venice does native web search via `venice_parameters`, NOT baibot's built-in
|
||||
// tools mechanism (the OpenAI web_search/code_interpreter block), so this is false.
|
||||
text_generation_supports_tools: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -195,4 +223,6 @@ pub struct AgentProviderInfo {
|
||||
pub sign_up_url: Option<&'static str>,
|
||||
pub models_list_url: Option<&'static str>,
|
||||
pub supported_purposes: Vec<AgentPurpose>,
|
||||
pub text_generation_supports_vision: bool,
|
||||
pub text_generation_supports_tools: bool,
|
||||
}
|
||||
|
||||
63
src/agent/provider/entity/image.rs
Normal file
@@ -0,0 +1,63 @@
|
||||
use mxlink::mime;
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ImageGenerationParams {
|
||||
pub smallest_size_possible: bool,
|
||||
|
||||
pub cheaper_model_switching_allowed: bool,
|
||||
|
||||
pub cheaper_quality_switching_allowed: bool,
|
||||
}
|
||||
|
||||
impl ImageGenerationParams {
|
||||
pub fn with_smallest_size_possible(mut self, value: bool) -> Self {
|
||||
self.smallest_size_possible = value;
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_cheaper_model_switching_allowed(mut self, value: bool) -> Self {
|
||||
self.cheaper_model_switching_allowed = value;
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_cheaper_quality_switching_allowed(mut self, value: bool) -> Self {
|
||||
self.cheaper_quality_switching_allowed = value;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
pub struct ImageGenerationResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mime::Mime,
|
||||
pub revised_prompt: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ImageEditParams {}
|
||||
|
||||
pub struct ImageEditResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mime::Mime,
|
||||
}
|
||||
|
||||
pub struct ImageSource {
|
||||
pub filename: String,
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mime::Mime,
|
||||
}
|
||||
|
||||
impl ImageSource {
|
||||
pub fn new(filename: String, bytes: Vec<u8>, mime_type: mime::Mime) -> Self {
|
||||
Self {
|
||||
filename,
|
||||
bytes,
|
||||
mime_type,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<ImageSource> for async_openai::types::images::ImageInput {
|
||||
fn from(value: ImageSource) -> Self {
|
||||
async_openai::types::images::ImageInput::from_vec_u8(value.filename, value.bytes)
|
||||
}
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
#[derive(Default)]
|
||||
pub struct ImageGenerationParams {
|
||||
pub size_override: Option<String>,
|
||||
|
||||
pub cheaper_model_switching_allowed: bool,
|
||||
|
||||
pub cheaper_quality_switching_allowed: bool,
|
||||
}
|
||||
|
||||
impl ImageGenerationParams {
|
||||
pub fn with_size_override(mut self, value: Option<String>) -> Self {
|
||||
self.size_override = value;
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_cheaper_model_switching_allowed(mut self, value: bool) -> Self {
|
||||
self.cheaper_model_switching_allowed = value;
|
||||
self
|
||||
}
|
||||
|
||||
pub fn with_cheaper_quality_switching_allowed(mut self, value: bool) -> Self {
|
||||
self.cheaper_quality_switching_allowed = value;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
pub struct ImageGenerationResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mxlink::mime::Mime,
|
||||
pub revised_prompt: Option<String>,
|
||||
}
|
||||
@@ -1,12 +1,14 @@
|
||||
mod agent_provider;
|
||||
mod image_generation;
|
||||
mod image;
|
||||
mod ping;
|
||||
mod speech_to_text;
|
||||
mod text_generation;
|
||||
mod text_to_speech;
|
||||
|
||||
pub use agent_provider::{AgentProvider, AgentProviderInfo};
|
||||
pub use image_generation::{ImageGenerationParams, ImageGenerationResult};
|
||||
pub use image::{
|
||||
ImageEditParams, ImageEditResult, ImageGenerationParams, ImageGenerationResult, ImageSource,
|
||||
};
|
||||
pub use ping::PingResult;
|
||||
pub use speech_to_text::{SpeechToTextParams, SpeechToTextResult};
|
||||
pub use text_generation::{
|
||||
|
||||
@@ -7,17 +7,33 @@ pub struct TextGenerationPromptVariables {
|
||||
|
||||
impl Default for TextGenerationPromptVariables {
|
||||
fn default() -> Self {
|
||||
Self::new("unnamed", "unknown-model", Utc::now())
|
||||
let now = Utc::now();
|
||||
Self::new("unnamed", "unknown-model", now, Some(now))
|
||||
}
|
||||
}
|
||||
|
||||
impl TextGenerationPromptVariables {
|
||||
pub fn new(bot_name: &str, model_id: &str, utc_time: DateTime<Utc>) -> Self {
|
||||
pub fn new(
|
||||
bot_name: &str,
|
||||
model_id: &str,
|
||||
now_time: DateTime<Utc>,
|
||||
conversation_start_time: Option<DateTime<Utc>>,
|
||||
) -> Self {
|
||||
let mut map = HashMap::new();
|
||||
|
||||
map.insert("baibot_name".to_string(), bot_name.to_string());
|
||||
map.insert("baibot_model_id".to_string(), model_id.to_string());
|
||||
map.insert("baibot_now_utc".to_string(), format_utc_time(utc_time));
|
||||
map.insert("baibot_now_utc".to_string(), format_utc_time(now_time));
|
||||
|
||||
let baibot_conversation_start_time_utc = match conversation_start_time {
|
||||
Some(conversation_start_time) => format_utc_time(conversation_start_time),
|
||||
None => "unknown".to_string(),
|
||||
};
|
||||
|
||||
map.insert(
|
||||
"baibot_conversation_start_time_utc".to_string(),
|
||||
baibot_conversation_start_time_utc,
|
||||
);
|
||||
|
||||
Self { map }
|
||||
}
|
||||
@@ -52,7 +68,18 @@ mod tests {
|
||||
.with_nanosecond(250000000)
|
||||
.unwrap();
|
||||
|
||||
let variables = TextGenerationPromptVariables::new("baibot", "gpt-4o", now_utc);
|
||||
let conversation_start_time_utc = Utc
|
||||
.with_ymd_and_hms(2024, 9, 19, 18, 34, 15)
|
||||
.unwrap()
|
||||
.with_nanosecond(250000000)
|
||||
.unwrap();
|
||||
|
||||
let variables = TextGenerationPromptVariables::new(
|
||||
"baibot",
|
||||
"gpt-4o",
|
||||
now_utc,
|
||||
Some(conversation_start_time_utc),
|
||||
);
|
||||
|
||||
assert_eq!(
|
||||
variables.map.get("baibot_name"),
|
||||
@@ -66,9 +93,13 @@ mod tests {
|
||||
variables.map.get("baibot_now_utc"),
|
||||
Some(&format_utc_time(now_utc))
|
||||
);
|
||||
assert_eq!(
|
||||
variables.map.get("baibot_conversation_start_time_utc"),
|
||||
Some(&format_utc_time(conversation_start_time_utc))
|
||||
);
|
||||
|
||||
let prompt = "Hello, I'm {{ baibot_name }} using {{ baibot_model_id }}. The date/time now is {{ baibot_now_utc }}.";
|
||||
let expected = "Hello, I'm baibot using gpt-4o. The date/time now is 2024-09-20 (Friday), 18:34:15 UTC.";
|
||||
let prompt = "Hello, I'm {{ baibot_name }} using {{ baibot_model_id }}. The date/time now is {{ baibot_now_utc }} and this conversation started at {{ baibot_conversation_start_time_utc }}.";
|
||||
let expected = "Hello, I'm baibot using gpt-4o. The date/time now is 2024-09-20 (Friday), 18:34:15 UTC and this conversation started at 2024-09-19 (Thursday), 18:34:15 UTC.";
|
||||
|
||||
assert_eq!(variables.format(prompt), expected);
|
||||
}
|
||||
|
||||
@@ -15,7 +15,7 @@ pub fn default_config() -> Config {
|
||||
if let Some(ref mut config) = config.text_generation.as_mut() {
|
||||
config.model_id = "llama3-70b-8192".to_owned();
|
||||
config.max_context_tokens = 131_072;
|
||||
config.max_response_tokens = 4096;
|
||||
config.max_response_tokens = Some(4096);
|
||||
}
|
||||
|
||||
if let Some(ref mut config) = config.speech_to_text.as_mut() {
|
||||
|
||||
@@ -13,7 +13,7 @@ pub fn default_config() -> Config {
|
||||
if let Some(ref mut config) = config.text_generation.as_mut() {
|
||||
config.model_id = "gpt-4".to_owned();
|
||||
config.max_context_tokens = 128_000;
|
||||
config.max_response_tokens = 4096;
|
||||
config.max_response_tokens = Some(4096);
|
||||
}
|
||||
|
||||
if let Some(ref mut config) = config.text_to_speech.as_mut() {
|
||||
|
||||
@@ -10,6 +10,7 @@ pub mod openai;
|
||||
pub mod openai_compat;
|
||||
pub(super) mod openrouter;
|
||||
pub(super) mod togetherai;
|
||||
pub mod venice;
|
||||
|
||||
fn default_temperature() -> f32 {
|
||||
1.0
|
||||
@@ -20,6 +21,7 @@ pub use controller::{ControllerTrait, ControllerType};
|
||||
pub use config::ConfigTrait;
|
||||
|
||||
pub use entity::{
|
||||
AgentProvider, AgentProviderInfo, ImageGenerationParams, PingResult, SpeechToTextParams,
|
||||
SpeechToTextResult, TextGenerationParams, TextGenerationPromptVariables, TextToSpeechParams,
|
||||
AgentProvider, AgentProviderInfo, ImageEditParams, ImageGenerationParams, ImageSource,
|
||||
PingResult, SpeechToTextParams, SpeechToTextResult, TextGenerationParams,
|
||||
TextGenerationPromptVariables, TextToSpeechParams,
|
||||
};
|
||||
|
||||
@@ -17,7 +17,7 @@ pub fn default_config() -> Config {
|
||||
if let Some(ref mut config) = config.text_generation.as_mut() {
|
||||
config.model_id = "gemma2:2b".to_owned();
|
||||
config.max_context_tokens = 128_000;
|
||||
config.max_response_tokens = 4096;
|
||||
config.max_response_tokens = Some(4096);
|
||||
}
|
||||
|
||||
config
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::OPENAI_IMAGE_MODEL_GPT_IMAGE_2;
|
||||
use crate::agent::{default_prompt, provider::ConfigTrait};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -56,10 +57,16 @@ pub struct TextGenerationConfig {
|
||||
pub temperature: f32,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_response_tokens: u32,
|
||||
pub max_response_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_completion_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_context_tokens: u32,
|
||||
|
||||
#[serde(default)]
|
||||
pub tools: ToolsConfig,
|
||||
}
|
||||
|
||||
impl Default for TextGenerationConfig {
|
||||
@@ -68,14 +75,25 @@ impl Default for TextGenerationConfig {
|
||||
model_id: default_text_model_id(),
|
||||
prompt: Some(default_prompt().to_owned()),
|
||||
temperature: super::super::default_temperature(),
|
||||
max_response_tokens: 16_384,
|
||||
max_context_tokens: 128_000,
|
||||
max_response_tokens: None,
|
||||
max_completion_tokens: Some(128_000),
|
||||
max_context_tokens: 400_000,
|
||||
tools: ToolsConfig::default(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_model_id() -> String {
|
||||
"gpt-4o-2024-08-06".to_owned()
|
||||
"gpt-5.4".to_owned()
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||
pub struct ToolsConfig {
|
||||
#[serde(default)]
|
||||
pub web_search: bool,
|
||||
|
||||
#[serde(default)]
|
||||
pub code_interpreter: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -99,16 +117,16 @@ fn default_speech_to_text_model_id() -> String {
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextToSpeechConfig {
|
||||
#[serde(default = "default_text_to_speech_model_id")]
|
||||
pub model_id: async_openai::types::SpeechModel,
|
||||
pub model_id: async_openai::types::audio::SpeechModel,
|
||||
|
||||
#[serde(default = "default_text_to_speech_voice")]
|
||||
pub voice: async_openai::types::Voice,
|
||||
pub voice: async_openai::types::audio::Voice,
|
||||
|
||||
#[serde(default = "default_text_to_speech_speed")]
|
||||
pub speed: f32,
|
||||
|
||||
#[serde(default = "default_text_to_speech_response_format")]
|
||||
pub response_format: async_openai::types::SpeechResponseFormat,
|
||||
pub response_format: async_openai::types::audio::SpeechResponseFormat,
|
||||
}
|
||||
|
||||
impl Default for TextToSpeechConfig {
|
||||
@@ -122,22 +140,22 @@ impl Default for TextToSpeechConfig {
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel {
|
||||
async_openai::types::SpeechModel::Tts1Hd
|
||||
fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel {
|
||||
async_openai::types::audio::SpeechModel::Tts1Hd
|
||||
}
|
||||
|
||||
fn default_text_to_speech_voice() -> async_openai::types::Voice {
|
||||
async_openai::types::Voice::Onyx
|
||||
fn default_text_to_speech_voice() -> async_openai::types::audio::Voice {
|
||||
async_openai::types::audio::Voice::Onyx
|
||||
}
|
||||
|
||||
fn default_text_to_speech_speed() -> f32 {
|
||||
1.0
|
||||
}
|
||||
|
||||
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat {
|
||||
fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat {
|
||||
// The API defaults to mp3, but we prefer Opus because it's smaller.
|
||||
// Our clients should all have support for it.
|
||||
async_openai::types::SpeechResponseFormat::Opus
|
||||
async_openai::types::audio::SpeechResponseFormat::Opus
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -145,19 +163,19 @@ pub struct ImageGenerationConfig {
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default = "default_image_style")]
|
||||
pub style: async_openai::types::ImageStyle,
|
||||
pub style: Option<async_openai::types::images::ImageStyle>,
|
||||
|
||||
#[serde(default = "default_image_size")]
|
||||
pub size: async_openai::types::ImageSize,
|
||||
pub size: Option<async_openai::types::images::ImageSize>,
|
||||
|
||||
#[serde(default = "default_image_quality")]
|
||||
pub quality: async_openai::types::ImageQuality,
|
||||
pub quality: Option<async_openai::types::images::ImageQuality>,
|
||||
}
|
||||
|
||||
impl Default for ImageGenerationConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: "dall-e-3".to_owned(),
|
||||
model_id: OPENAI_IMAGE_MODEL_GPT_IMAGE_2.to_owned(),
|
||||
style: default_image_style(),
|
||||
size: default_image_size(),
|
||||
quality: default_image_quality(),
|
||||
@@ -168,23 +186,29 @@ impl Default for ImageGenerationConfig {
|
||||
impl ImageGenerationConfig {
|
||||
pub fn model_id_as_openai_image_model(
|
||||
&self,
|
||||
) -> Result<async_openai::types::ImageModel, String> {
|
||||
) -> Result<async_openai::types::images::ImageModel, String> {
|
||||
match self.model_id.as_str() {
|
||||
"dall-e-2" => Ok(async_openai::types::ImageModel::DallE2),
|
||||
"dall-e-3" => Ok(async_openai::types::ImageModel::DallE3),
|
||||
other => Ok(async_openai::types::ImageModel::Other(other.to_owned())),
|
||||
"dall-e-2" => Ok(async_openai::types::images::ImageModel::DallE2),
|
||||
"dall-e-3" => Ok(async_openai::types::images::ImageModel::DallE3),
|
||||
"gpt-image-1" => Ok(async_openai::types::images::ImageModel::GptImage1),
|
||||
"gpt-image-1.5" => Ok(async_openai::types::images::ImageModel::GptImage1dot5),
|
||||
"gpt-image-1-mini" => Ok(async_openai::types::images::ImageModel::GptImage1Mini),
|
||||
"gpt-image-2" => Ok(async_openai::types::images::ImageModel::GptImage2),
|
||||
other => Ok(async_openai::types::images::ImageModel::Other(
|
||||
other.to_owned(),
|
||||
)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_image_style() -> async_openai::types::ImageStyle {
|
||||
async_openai::types::ImageStyle::Vivid
|
||||
fn default_image_style() -> Option<async_openai::types::images::ImageStyle> {
|
||||
None
|
||||
}
|
||||
|
||||
fn default_image_size() -> async_openai::types::ImageSize {
|
||||
async_openai::types::ImageSize::S1024x1024
|
||||
fn default_image_size() -> Option<async_openai::types::images::ImageSize> {
|
||||
None
|
||||
}
|
||||
|
||||
fn default_image_quality() -> async_openai::types::ImageQuality {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
fn default_image_quality() -> Option<async_openai::types::images::ImageQuality> {
|
||||
None
|
||||
}
|
||||
|
||||
@@ -1,37 +1,42 @@
|
||||
use std::ops::Deref;
|
||||
|
||||
use async_openai::{
|
||||
Client as OpenAIClient,
|
||||
config::OpenAIConfig,
|
||||
types::{
|
||||
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageRequestArgs,
|
||||
CreateSpeechRequestArgs, CreateTranscriptionRequestArgs,
|
||||
audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs},
|
||||
images::{
|
||||
CreateImageEditRequestArgs, CreateImageRequestArgs, Image, ImageInput, ImageModel,
|
||||
ImageResponseFormat,
|
||||
},
|
||||
responses::{
|
||||
CodeInterpreterContainerAuto, CodeInterpreterTool, CodeInterpreterToolContainer,
|
||||
CreateResponseArgs, OutputItem, OutputMessageContent, Tool, WebSearchTool,
|
||||
},
|
||||
},
|
||||
Client as OpenAIClient,
|
||||
};
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::{
|
||||
agent::{
|
||||
provider::{
|
||||
entity::{ImageGenerationResult, PingResult, TextToSpeechParams, TextToSpeechResult},
|
||||
openai::utils::convert_string_to_enum,
|
||||
agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
entity::{TextGenerationParams, TextGenerationResult},
|
||||
},
|
||||
AgentPurpose,
|
||||
conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
},
|
||||
strings,
|
||||
utils::base64::base64_decode,
|
||||
};
|
||||
use crate::{
|
||||
agent::{
|
||||
provider::{
|
||||
entity::{TextGenerationParams, TextGenerationResult},
|
||||
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
AgentPurpose,
|
||||
provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextToSpeechParams,
|
||||
TextToSpeechResult,
|
||||
},
|
||||
utils::base64_decode,
|
||||
},
|
||||
conversation::llm::{
|
||||
shorten_messages_list_to_context_size, Author as LLMAuthor,
|
||||
Conversation as LLMConversation, Message as LLMMessage,
|
||||
},
|
||||
strings,
|
||||
};
|
||||
|
||||
use super::config::Config;
|
||||
@@ -62,7 +67,9 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
message_text: "Hello!".to_string(),
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
let conversation = LLMConversation { messages };
|
||||
@@ -98,7 +105,9 @@ impl ControllerTrait for Controller {
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
message_text: prompt_text,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
|
||||
@@ -122,54 +131,76 @@ impl ControllerTrait for Controller {
|
||||
conversation_messages.insert(0, prompt_message);
|
||||
}
|
||||
|
||||
let openai_conversation_messages: Vec<ChatCompletionRequestMessage> =
|
||||
super::utils::convert_llm_messages_to_openai_messages(conversation_messages);
|
||||
let input =
|
||||
super::utils::convert_llm_messages_to_openai_response_input(conversation_messages);
|
||||
|
||||
let messages_count = openai_conversation_messages.len();
|
||||
let messages_count = match &input {
|
||||
async_openai::types::responses::InputParam::Items(items) => items.len(),
|
||||
_ => 1,
|
||||
};
|
||||
|
||||
let temperature = params
|
||||
.temperature_override
|
||||
.unwrap_or(text_generation_config.temperature);
|
||||
|
||||
let request = CreateChatCompletionRequestArgs::default()
|
||||
.max_tokens(text_generation_config.max_response_tokens)
|
||||
let mut request_builder = CreateResponseArgs::default();
|
||||
|
||||
request_builder
|
||||
.model(&text_generation_config.model_id)
|
||||
.temperature(temperature)
|
||||
.messages(openai_conversation_messages)
|
||||
.build()?;
|
||||
.input(input);
|
||||
|
||||
let mut tools = Vec::new();
|
||||
if text_generation_config.tools.web_search {
|
||||
tools.push(Tool::WebSearch(WebSearchTool::default()));
|
||||
}
|
||||
if text_generation_config.tools.code_interpreter {
|
||||
tools.push(Tool::CodeInterpreter(CodeInterpreterTool {
|
||||
container: CodeInterpreterToolContainer::Auto(
|
||||
CodeInterpreterContainerAuto::default(),
|
||||
),
|
||||
}));
|
||||
}
|
||||
|
||||
if !tools.is_empty() {
|
||||
request_builder.tools(tools);
|
||||
}
|
||||
|
||||
if let Some(max_response_tokens) = text_generation_config.max_response_tokens {
|
||||
request_builder.max_output_tokens(max_response_tokens);
|
||||
} else if let Some(max_completion_tokens) = text_generation_config.max_completion_tokens {
|
||||
request_builder.max_output_tokens(max_completion_tokens);
|
||||
}
|
||||
|
||||
let request = request_builder.build()?;
|
||||
|
||||
if let Ok(request_as_json) = serde_json::to_string(&request) {
|
||||
tracing::trace!(
|
||||
model = format!("{:?}", request.model),
|
||||
?messages_count,
|
||||
request = request_as_json,
|
||||
"Sending OpenAI chat completion API request"
|
||||
"Sending OpenAI response API request"
|
||||
);
|
||||
}
|
||||
|
||||
let response = self.client.chat().create(request).await?;
|
||||
let response = self.client.responses().create(request).await?;
|
||||
|
||||
tracing::trace!(
|
||||
?response,
|
||||
"Got response from the OpenAI chat completion API"
|
||||
);
|
||||
tracing::trace!(?response, "Got response from the OpenAI response API");
|
||||
|
||||
// We only request 1 result, so there should only be 1 choice.
|
||||
if let Some(choice) = response.choices.into_iter().next() {
|
||||
match choice.message.content {
|
||||
Some(text) => {
|
||||
return Ok(TextGenerationResult { text });
|
||||
for item in response.output {
|
||||
if let OutputItem::Message(message) = item {
|
||||
for content in message.content {
|
||||
if let OutputMessageContent::OutputText(text_content) = content {
|
||||
return Ok(TextGenerationResult {
|
||||
text: text_content.text,
|
||||
});
|
||||
}
|
||||
None => {
|
||||
return Err(anyhow::anyhow!(
|
||||
"No content was found in the response choice from the OpenAI chat completion API"
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(anyhow::anyhow!(
|
||||
"No response messages choices were returned from the OpenAI chat completion API"
|
||||
"No response messages choices were returned from the OpenAI response API"
|
||||
))
|
||||
}
|
||||
|
||||
@@ -193,12 +224,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let request = CreateTranscriptionRequestArgs::default()
|
||||
.model(&speech_to_text_config.model_id)
|
||||
.file(async_openai::types::AudioInput {
|
||||
source: async_openai::types::InputSource::VecU8 {
|
||||
filename,
|
||||
vec: media,
|
||||
},
|
||||
})
|
||||
.file(AudioInput::from_vec_u8(filename, media))
|
||||
.language(language.clone())
|
||||
.build()?;
|
||||
|
||||
@@ -208,7 +234,7 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI speech-to-text API request"
|
||||
);
|
||||
|
||||
let response = self.client.audio().transcribe(request).await?;
|
||||
let response = self.client.audio().transcription().create(request).await?;
|
||||
|
||||
tracing::trace!(
|
||||
?response,
|
||||
@@ -240,11 +266,13 @@ impl ControllerTrait for Controller {
|
||||
let model = if params.cheaper_model_switching_allowed {
|
||||
// Switch to a cheaper model
|
||||
match original_model {
|
||||
async_openai::types::ImageModel::DallE2 => async_openai::types::ImageModel::DallE2,
|
||||
async_openai::types::ImageModel::DallE3 => async_openai::types::ImageModel::DallE2,
|
||||
async_openai::types::ImageModel::Other(_) => {
|
||||
async_openai::types::ImageModel::DallE2
|
||||
}
|
||||
ImageModel::DallE2 => ImageModel::DallE2,
|
||||
ImageModel::DallE3 => ImageModel::DallE2,
|
||||
ImageModel::GptImage1 => ImageModel::GptImage1Mini,
|
||||
ImageModel::GptImage1dot5 => ImageModel::GptImage1Mini,
|
||||
ImageModel::GptImage1Mini => ImageModel::GptImage1Mini,
|
||||
ImageModel::GptImage2 => ImageModel::GptImage1Mini,
|
||||
ImageModel::Other(_) => ImageModel::DallE2,
|
||||
}
|
||||
} else {
|
||||
original_model
|
||||
@@ -253,33 +281,72 @@ impl ControllerTrait for Controller {
|
||||
let quality = if params.cheaper_quality_switching_allowed {
|
||||
// Switch to a cheaper quality
|
||||
match &image_generation_config.quality {
|
||||
async_openai::types::ImageQuality::Standard => {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
Some(quality) => match quality {
|
||||
async_openai::types::images::ImageQuality::Standard => {
|
||||
Some(async_openai::types::images::ImageQuality::Standard)
|
||||
}
|
||||
async_openai::types::ImageQuality::HD => {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
async_openai::types::images::ImageQuality::HD => {
|
||||
Some(async_openai::types::images::ImageQuality::Standard)
|
||||
}
|
||||
// New quality levels - keep as-is or downgrade to Standard
|
||||
async_openai::types::images::ImageQuality::High => {
|
||||
Some(async_openai::types::images::ImageQuality::Standard)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::Medium => {
|
||||
Some(async_openai::types::images::ImageQuality::Medium)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::Low => {
|
||||
Some(async_openai::types::images::ImageQuality::Low)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::Auto => {
|
||||
Some(async_openai::types::images::ImageQuality::Auto)
|
||||
}
|
||||
},
|
||||
None => None,
|
||||
}
|
||||
} else {
|
||||
image_generation_config.quality.clone()
|
||||
};
|
||||
|
||||
let size = params
|
||||
.size_override
|
||||
.map(|s| {
|
||||
convert_string_to_enum::<async_openai::types::ImageSize>(&s)
|
||||
.unwrap_or(image_generation_config.size)
|
||||
})
|
||||
.unwrap_or(image_generation_config.size);
|
||||
let size = if params.smallest_size_possible {
|
||||
Some(get_sticker_size(&model))
|
||||
} else {
|
||||
image_generation_config.size.clone()
|
||||
};
|
||||
|
||||
let request = CreateImageRequestArgs::default()
|
||||
.model(model)
|
||||
.prompt(prompt.to_owned())
|
||||
.response_format(async_openai::types::ImageResponseFormat::B64Json)
|
||||
.size(size)
|
||||
.style(image_generation_config.style.clone())
|
||||
.quality(quality)
|
||||
.build()?;
|
||||
let response_format = match model.clone() {
|
||||
ImageModel::DallE2 => Some(ImageResponseFormat::B64Json),
|
||||
ImageModel::DallE3 => Some(ImageResponseFormat::B64Json),
|
||||
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
||||
// In fact, specifying the response format results in an error.
|
||||
ImageModel::GptImage1 => None,
|
||||
ImageModel::GptImage1Mini => None,
|
||||
ImageModel::GptImage1dot5 => None,
|
||||
ImageModel::GptImage2 => None,
|
||||
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
|
||||
};
|
||||
|
||||
let mut request_builder = CreateImageRequestArgs::default();
|
||||
|
||||
request_builder.model(model).prompt(prompt.to_owned());
|
||||
|
||||
if let Some(response_format) = response_format {
|
||||
request_builder.response_format(response_format);
|
||||
}
|
||||
|
||||
if let Some(style) = &image_generation_config.style {
|
||||
request_builder.style(style.clone());
|
||||
}
|
||||
|
||||
if let Some(quality) = quality {
|
||||
request_builder.quality(quality.clone());
|
||||
}
|
||||
|
||||
if let Some(size) = size {
|
||||
request_builder.size(size);
|
||||
}
|
||||
|
||||
let request = request_builder.build()?;
|
||||
|
||||
tracing::trace!(
|
||||
?prompt,
|
||||
@@ -290,15 +357,15 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI image generation API request"
|
||||
);
|
||||
|
||||
let response = self.client.images().create(request).await?;
|
||||
let response = self.client.images().generate(request).await?;
|
||||
|
||||
if let Some(image) = response.data.into_iter().next() {
|
||||
match image.deref() {
|
||||
async_openai::types::Image::B64Json {
|
||||
Image::B64Json {
|
||||
b64_json,
|
||||
revised_prompt,
|
||||
} => {
|
||||
let bytes = base64_decode(b64_json)?;
|
||||
let bytes = base64_decode(b64_json.as_ref())?;
|
||||
|
||||
return Ok(ImageGenerationResult {
|
||||
bytes,
|
||||
@@ -317,6 +384,109 @@ impl ControllerTrait for Controller {
|
||||
))
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
let Some(image_generation_config) = &self.config.image_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::ImageGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
if images.is_empty() {
|
||||
return Err(anyhow::anyhow!("No image sources provided"));
|
||||
}
|
||||
|
||||
let mut image_inputs: Vec<ImageInput> = Vec::new();
|
||||
for image in images {
|
||||
image_inputs.push(image.into());
|
||||
}
|
||||
|
||||
let dalle2_size = match image_generation_config.size {
|
||||
Some(async_openai::types::images::ImageSize::S256x256) => {
|
||||
Some(async_openai::types::images::ImageSize::S256x256)
|
||||
}
|
||||
Some(async_openai::types::images::ImageSize::S512x512) => {
|
||||
Some(async_openai::types::images::ImageSize::S512x512)
|
||||
}
|
||||
Some(async_openai::types::images::ImageSize::S1024x1024) => {
|
||||
Some(async_openai::types::images::ImageSize::S1024x1024)
|
||||
}
|
||||
_ => None,
|
||||
};
|
||||
|
||||
let model = image_generation_config
|
||||
.model_id_as_openai_image_model()
|
||||
.map_err(|err| anyhow::anyhow!(err))?;
|
||||
|
||||
let response_format = match model.clone() {
|
||||
ImageModel::DallE2 => Some(ImageResponseFormat::B64Json),
|
||||
ImageModel::DallE3 => Some(ImageResponseFormat::B64Json),
|
||||
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
||||
// In fact, specifying the response format results in an error.
|
||||
ImageModel::GptImage1 => None,
|
||||
ImageModel::GptImage1Mini => None,
|
||||
ImageModel::GptImage1dot5 => None,
|
||||
ImageModel::GptImage2 => None,
|
||||
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
|
||||
};
|
||||
|
||||
let mut request_builder = CreateImageEditRequestArgs::default();
|
||||
|
||||
request_builder
|
||||
.image(image_inputs)
|
||||
.prompt(prompt.to_owned())
|
||||
.model(model);
|
||||
|
||||
if let Some(size) = dalle2_size {
|
||||
request_builder.size(size);
|
||||
}
|
||||
|
||||
if let Some(response_format) = response_format {
|
||||
request_builder.response_format(response_format);
|
||||
}
|
||||
|
||||
let request = request_builder
|
||||
.build()
|
||||
.map_err(|e| anyhow::anyhow!("Failed to build CreateImageEditRequest: {}", e))?;
|
||||
|
||||
tracing::trace!(
|
||||
model = format!("{:?}", request.model),
|
||||
size = format!("{:?}", request.size),
|
||||
response_format = format!("{:?}", request.response_format),
|
||||
"Sending OpenAI image edit API request"
|
||||
);
|
||||
|
||||
let response = self.client.images().edit(request).await?;
|
||||
|
||||
if let Some(image_data) = response.data.into_iter().next() {
|
||||
match image_data.deref() {
|
||||
Image::B64Json { b64_json, .. } => {
|
||||
let bytes = base64_decode(b64_json.as_ref())?;
|
||||
return Ok(ImageEditResult {
|
||||
bytes,
|
||||
mime_type: mxlink::mime::IMAGE_PNG,
|
||||
});
|
||||
}
|
||||
Image::Url { url, .. } => {
|
||||
tracing::warn!(?url, "Received URL instead of B64Json for image edit");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Unexpected image type (URL) when B64Json was requested"
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(anyhow::anyhow!(
|
||||
"The OpenAI image edit API returned no images"
|
||||
))
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
input: &str,
|
||||
@@ -334,7 +504,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let voice = if let Some(voice_string) = params.voice_override {
|
||||
// This is a hacky way to construct a Voice enum from the string we have.
|
||||
let voice: serde_json::Result<async_openai::types::Voice> =
|
||||
let voice: serde_json::Result<async_openai::types::audio::Voice> =
|
||||
serde_json::from_str(&format!("\"{}\"", voice_string));
|
||||
match voice {
|
||||
Ok(voice) => voice,
|
||||
@@ -374,7 +544,7 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI text-to-speech API request"
|
||||
);
|
||||
|
||||
let result = self.client.audio().speech(request).await?;
|
||||
let result = self.client.audio().speech().create(request).await?;
|
||||
|
||||
Ok(TextToSpeechResult {
|
||||
bytes: result.bytes.into(),
|
||||
@@ -433,15 +603,15 @@ impl ControllerTrait for Controller {
|
||||
}
|
||||
|
||||
fn response_format_to_mime_type(
|
||||
response_format: &async_openai::types::SpeechResponseFormat,
|
||||
response_format: &async_openai::types::audio::SpeechResponseFormat,
|
||||
) -> Option<mxlink::mime::Mime> {
|
||||
let content_type = match response_format {
|
||||
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||
};
|
||||
|
||||
match content_type.parse() {
|
||||
@@ -469,3 +639,18 @@ fn audio_mime_type_to_file_name(mime_type: &mxlink::mime::Mime) -> Option<String
|
||||
|
||||
Some(format!("audio.{}", file_extension))
|
||||
}
|
||||
|
||||
/// Returns the smallest supported size for stickers based on what the image model supports.
|
||||
fn get_sticker_size(model: &ImageModel) -> async_openai::types::images::ImageSize {
|
||||
use async_openai::types::images::ImageSize;
|
||||
|
||||
match model {
|
||||
ImageModel::DallE2 => ImageSize::S256x256,
|
||||
ImageModel::DallE3 => ImageSize::S1024x1024,
|
||||
ImageModel::GptImage1 => ImageSize::S1024x1024,
|
||||
ImageModel::GptImage1Mini => ImageSize::S1024x1024,
|
||||
ImageModel::GptImage1dot5 => ImageSize::S1024x1024,
|
||||
ImageModel::GptImage2 => ImageSize::S1024x1024,
|
||||
ImageModel::Other(_) => ImageSize::S1024x1024,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,17 +13,19 @@ pub(super) use config::TextToSpeechConfig;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::controller::ControllerType;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub const OPENAI_IMAGE_MODEL_GPT_IMAGE_2: &str = "gpt-image-2";
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
config: serde_yaml::Value,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> AgentInstantiationResult<ControllerType> {
|
||||
let config = match &config {
|
||||
serde_yaml::Value::Mapping(_) => {
|
||||
serde_yaml_ng::Value::Mapping(_) => {
|
||||
let config: Config =
|
||||
serde_yaml::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
|
||||
config
|
||||
.validate()
|
||||
|
||||
@@ -1,55 +1,64 @@
|
||||
use async_openai::types::{
|
||||
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
|
||||
ChatCompletionRequestSystemMessageArgs, ChatCompletionRequestUserMessageArgs,
|
||||
use async_openai::types::responses::{
|
||||
EasyInputContent, EasyInputMessage, ImageDetail, InputContent, InputFileArgs,
|
||||
InputImageContent, InputItem, InputParam, MessageType, Role,
|
||||
};
|
||||
|
||||
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
use crate::utils::base64::base64_encode;
|
||||
|
||||
pub fn convert_llm_messages_to_openai_messages(
|
||||
pub fn convert_llm_messages_to_openai_response_input(
|
||||
conversation_messages: Vec<LLMMessage>,
|
||||
) -> Vec<ChatCompletionRequestMessage> {
|
||||
let mut openai_conversation_messages: Vec<ChatCompletionRequestMessage> =
|
||||
Vec::with_capacity(conversation_messages.len());
|
||||
) -> InputParam {
|
||||
let mut items = Vec::with_capacity(conversation_messages.len());
|
||||
|
||||
for message in conversation_messages {
|
||||
openai_conversation_messages.push(convert_llm_message_to_openai_message(message));
|
||||
let role = match message.author {
|
||||
LLMAuthor::Prompt => Role::System,
|
||||
LLMAuthor::Assistant => Role::Assistant,
|
||||
LLMAuthor::User => Role::User,
|
||||
};
|
||||
|
||||
let content = match message.content {
|
||||
LLMMessageContent::Text(text) => EasyInputContent::Text(text),
|
||||
LLMMessageContent::Image(image_details) => {
|
||||
let image_url = format!(
|
||||
"data:{};base64,{}",
|
||||
image_details.mime,
|
||||
base64_encode(&image_details.data)
|
||||
);
|
||||
|
||||
EasyInputContent::ContentList(vec![InputContent::InputImage(InputImageContent {
|
||||
image_url: Some(image_url),
|
||||
detail: ImageDetail::Auto,
|
||||
file_id: None,
|
||||
})])
|
||||
}
|
||||
LLMMessageContent::File(file_details) => {
|
||||
let file_data = format!(
|
||||
"data:{};base64,{}",
|
||||
file_details.mime,
|
||||
base64_encode(&file_details.data)
|
||||
);
|
||||
|
||||
openai_conversation_messages
|
||||
}
|
||||
|
||||
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> ChatCompletionRequestMessage {
|
||||
match llm_message.author {
|
||||
LLMAuthor::Prompt => ChatCompletionRequestSystemMessageArgs::default()
|
||||
.content(llm_message.message_text)
|
||||
let file_content = InputFileArgs::default()
|
||||
.file_data(file_data)
|
||||
.filename(file_details.filename())
|
||||
.build()
|
||||
.expect("Failed building OpenAI system message")
|
||||
.into(),
|
||||
LLMAuthor::Assistant => ChatCompletionRequestAssistantMessageArgs::default()
|
||||
.content(llm_message.message_text)
|
||||
.build()
|
||||
.expect("Failed building OpenAI assistant message")
|
||||
.into(),
|
||||
LLMAuthor::User => ChatCompletionRequestUserMessageArgs::default()
|
||||
.content(llm_message.message_text)
|
||||
.build()
|
||||
.expect("Failed building OpenAI user message")
|
||||
.into(),
|
||||
}
|
||||
}
|
||||
.expect("Failed to build InputFileContent");
|
||||
|
||||
pub(super) fn convert_string_to_enum<T>(value: &str) -> Result<T, String>
|
||||
where
|
||||
T: serde::de::DeserializeOwned,
|
||||
{
|
||||
// This is a hacky way to construct an enum from the string we have.
|
||||
let enum_result: serde_json::Result<T> = serde_json::from_str(&format!("\"{}\"", value));
|
||||
match enum_result {
|
||||
Ok(enum_result) => Ok(enum_result),
|
||||
Err(err) => {
|
||||
tracing::debug!(?err, "Failed to parse into enum");
|
||||
EasyInputContent::ContentList(vec![InputContent::InputFile(file_content)])
|
||||
}
|
||||
};
|
||||
|
||||
Err(format!("The value ({}) is not supported.", value))
|
||||
}
|
||||
items.push(InputItem::EasyMessage(EasyInputMessage {
|
||||
r#type: MessageType::Message,
|
||||
role,
|
||||
content,
|
||||
phase: None,
|
||||
}));
|
||||
}
|
||||
|
||||
InputParam::Items(items)
|
||||
}
|
||||
|
||||
@@ -66,7 +66,7 @@ pub struct TextGenerationConfig {
|
||||
pub temperature: f32,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_response_tokens: u32,
|
||||
pub max_response_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_context_tokens: u32,
|
||||
@@ -78,7 +78,7 @@ impl Default for TextGenerationConfig {
|
||||
model_id: default_text_model_id(),
|
||||
prompt: Some(default_prompt().to_owned()),
|
||||
temperature: super::super::default_temperature(),
|
||||
max_response_tokens: 4096,
|
||||
max_response_tokens: Some(4096),
|
||||
max_context_tokens: 128_000,
|
||||
}
|
||||
}
|
||||
@@ -93,7 +93,9 @@ impl TryInto<OpenAITextGenerationConfig> for TextGenerationConfig {
|
||||
prompt: self.prompt,
|
||||
temperature: self.temperature,
|
||||
max_response_tokens: self.max_response_tokens,
|
||||
max_completion_tokens: None,
|
||||
max_context_tokens: self.max_context_tokens,
|
||||
tools: Default::default(),
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -160,13 +162,14 @@ impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
|
||||
type Error = String;
|
||||
|
||||
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
|
||||
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?;
|
||||
let model_id =
|
||||
convert_string_to_enum::<async_openai::types::audio::SpeechModel>(&self.model_id)?;
|
||||
|
||||
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?;
|
||||
let voice = convert_string_to_enum::<async_openai::types::audio::Voice>(&self.voice)?;
|
||||
|
||||
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>(
|
||||
&self.response_format,
|
||||
)?;
|
||||
let response_format = convert_string_to_enum::<
|
||||
async_openai::types::audio::SpeechResponseFormat,
|
||||
>(&self.response_format)?;
|
||||
|
||||
Ok(OpenAITextToSpeechConfig {
|
||||
model_id,
|
||||
@@ -223,21 +226,27 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
|
||||
|
||||
fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> {
|
||||
let size = if let Some(size) = &self.size {
|
||||
convert_string_to_enum::<async_openai::types::ImageSize>(size)?
|
||||
Some(convert_string_to_enum::<
|
||||
async_openai::types::images::ImageSize,
|
||||
>(size)?)
|
||||
} else {
|
||||
async_openai::types::ImageSize::S1024x1024
|
||||
None
|
||||
};
|
||||
|
||||
let style = if let Some(style) = &self.style {
|
||||
convert_string_to_enum::<async_openai::types::ImageStyle>(style)?
|
||||
Some(convert_string_to_enum::<
|
||||
async_openai::types::images::ImageStyle,
|
||||
>(style)?)
|
||||
} else {
|
||||
async_openai::types::ImageStyle::Vivid
|
||||
None
|
||||
};
|
||||
|
||||
let quality = if let Some(quality) = &self.quality {
|
||||
convert_string_to_enum::<async_openai::types::ImageQuality>(quality)?
|
||||
Some(convert_string_to_enum::<
|
||||
async_openai::types::images::ImageQuality,
|
||||
>(quality)?)
|
||||
} else {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
None
|
||||
};
|
||||
|
||||
Ok(OpenAIImageGenerationConfig {
|
||||
|
||||
@@ -3,24 +3,28 @@ use etke_openai_api_rust::chat::{ChatApi, ChatBody};
|
||||
use etke_openai_api_rust::images::{ImagesApi, ImagesBody};
|
||||
use etke_openai_api_rust::{Auth, Message, OpenAI};
|
||||
|
||||
const SMALLEST_IMAGE_SIZE: &str = "256x256";
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::agent::utils::base64_decode;
|
||||
use crate::utils::base64::base64_decode;
|
||||
use crate::{
|
||||
agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, ImageSource, SpeechToTextParams,
|
||||
SpeechToTextResult,
|
||||
entity::{TextGenerationParams, TextGenerationResult},
|
||||
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
},
|
||||
conversation::llm::{
|
||||
shorten_messages_list_to_context_size, Author as LLMAuthor,
|
||||
Conversation as LLMConversation, Message as LLMMessage,
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
},
|
||||
};
|
||||
use crate::{
|
||||
agent::{
|
||||
provider::entity::{
|
||||
ImageGenerationResult, PingResult, TextToSpeechParams, TextToSpeechResult,
|
||||
},
|
||||
AgentPurpose,
|
||||
provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, PingResult, TextToSpeechParams,
|
||||
TextToSpeechResult,
|
||||
},
|
||||
},
|
||||
strings,
|
||||
};
|
||||
@@ -60,7 +64,9 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
message_text: "Hello!".to_string(),
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
let conversation = LLMConversation { messages };
|
||||
@@ -96,7 +102,9 @@ impl ControllerTrait for Controller {
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
message_text: prompt_text,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
|
||||
@@ -131,12 +139,15 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let max_tokens = text_generation_config
|
||||
.max_response_tokens
|
||||
.map(|max_response_tokens| {
|
||||
max_response_tokens
|
||||
.try_into()
|
||||
.expect("Failed converting max_response_tokens from u32 to i32");
|
||||
.expect("Failed converting max_response_tokens from u32 to i32")
|
||||
});
|
||||
|
||||
let request = ChatBody {
|
||||
model: text_generation_config.model_id.clone(),
|
||||
max_tokens: Some(max_tokens),
|
||||
max_tokens,
|
||||
temperature: Some(temperature),
|
||||
top_p: None,
|
||||
n: Some(1),
|
||||
@@ -296,9 +307,11 @@ impl ControllerTrait for Controller {
|
||||
// when they span multiple lines.
|
||||
let prompt = prompt.replace("\n", " ");
|
||||
|
||||
let size: Option<String> = params
|
||||
.size_override
|
||||
.or_else(|| image_generation_config.size.clone());
|
||||
let size: Option<String> = if params.smallest_size_possible {
|
||||
Some(SMALLEST_IMAGE_SIZE.to_owned())
|
||||
} else {
|
||||
image_generation_config.size.clone()
|
||||
};
|
||||
|
||||
let request = ImagesBody {
|
||||
model: Some(image_generation_config.model_id.to_owned()),
|
||||
@@ -361,6 +374,17 @@ impl ControllerTrait for Controller {
|
||||
))
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
_prompt: &str,
|
||||
_images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
Err(anyhow::anyhow!(
|
||||
"The OpenAI image edit API is not supported by the OpenAI-compat provider"
|
||||
))
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
input: &str,
|
||||
|
||||
@@ -21,17 +21,17 @@ pub use controller::Controller;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::controller::ControllerType;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
config: serde_yaml::Value,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> AgentInstantiationResult<ControllerType> {
|
||||
let config = match &config {
|
||||
serde_yaml::Value::Mapping(_) => {
|
||||
serde_yaml_ng::Value::Mapping(_) => {
|
||||
let config: Config =
|
||||
serde_yaml::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
|
||||
config
|
||||
.validate()
|
||||
@@ -56,7 +56,7 @@ pub fn default_config() -> Config {
|
||||
|
||||
if let Some(text_generation) = &mut config.text_generation {
|
||||
text_generation.model_id = "some-model".to_string();
|
||||
text_generation.max_response_tokens = 4096;
|
||||
text_generation.max_response_tokens = Some(4096);
|
||||
text_generation.max_context_tokens = 128_000;
|
||||
}
|
||||
|
||||
|
||||
@@ -2,7 +2,9 @@ use etke_openai_api_rust::{Message, Role};
|
||||
|
||||
use crate::agent::provider::openai::Config as OpenAIConfig;
|
||||
|
||||
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
pub fn convert_llm_messages_to_openai_messages(
|
||||
conversation_messages: Vec<LLMMessage>,
|
||||
@@ -11,22 +13,39 @@ pub fn convert_llm_messages_to_openai_messages(
|
||||
Vec::with_capacity(conversation_messages.len());
|
||||
|
||||
for message in conversation_messages {
|
||||
openai_conversation_messages.push(convert_llm_message_to_openai_message(message));
|
||||
let openai_message = convert_llm_message_to_openai_message(message);
|
||||
if let Some(openai_message) = openai_message {
|
||||
openai_conversation_messages.push(openai_message);
|
||||
}
|
||||
}
|
||||
|
||||
openai_conversation_messages
|
||||
}
|
||||
|
||||
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> Message {
|
||||
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> Option<Message> {
|
||||
let role = match llm_message.author {
|
||||
LLMAuthor::Prompt => Role::System,
|
||||
LLMAuthor::Assistant => Role::Assistant,
|
||||
LLMAuthor::User => Role::User,
|
||||
};
|
||||
|
||||
Message {
|
||||
match &llm_message.content {
|
||||
LLMMessageContent::Text(text) => Some(Message {
|
||||
role,
|
||||
content: llm_message.message_text,
|
||||
content: text.clone(),
|
||||
}),
|
||||
LLMMessageContent::Image(_image_details) => {
|
||||
tracing::warn!(
|
||||
"The OpenAI-compat provider's library does not support image content. This image message will be skipped."
|
||||
);
|
||||
None
|
||||
}
|
||||
LLMMessageContent::File(_file_details) => {
|
||||
tracing::warn!(
|
||||
"The OpenAI-compat provider's library does not support file content. This file message will be skipped."
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ pub fn default_config() -> Config {
|
||||
if let Some(ref mut config) = config.text_generation.as_mut() {
|
||||
config.model_id = "mattshumer/reflection-70b:free".to_owned();
|
||||
config.max_context_tokens = 8192;
|
||||
config.max_response_tokens = 2048;
|
||||
config.max_response_tokens = Some(2048);
|
||||
}
|
||||
|
||||
config
|
||||
|
||||
@@ -14,7 +14,7 @@ pub fn default_config() -> Config {
|
||||
if let Some(ref mut config) = config.text_generation.as_mut() {
|
||||
config.model_id = "meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo".to_owned();
|
||||
config.max_context_tokens = 8192;
|
||||
config.max_response_tokens = 2048;
|
||||
config.max_response_tokens = Some(2048);
|
||||
}
|
||||
|
||||
config
|
||||
|
||||
156
src/agent/provider/venice/audio.rs
Normal file
@@ -0,0 +1,156 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{TextToSpeechParams, TextToSpeechResult};
|
||||
use crate::agent::provider::{SpeechToTextParams, SpeechToTextResult};
|
||||
use crate::strings;
|
||||
|
||||
use super::config::Config;
|
||||
use super::wire::{SpeechRequest, TranscriptionResponse};
|
||||
|
||||
pub async fn speech_to_text(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
mime_type: &mxlink::mime::Mime,
|
||||
media: Vec<u8>,
|
||||
params: SpeechToTextParams,
|
||||
) -> anyhow::Result<SpeechToTextResult> {
|
||||
let Some(speech_to_text_config) = &config.speech_to_text else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::SpeechToText
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
// Unlike the openai_compat path (which writes the audio to a temp file because its library
|
||||
// can't take bytes), reqwest's multipart takes the bytes directly.
|
||||
let part = reqwest::multipart::Part::bytes(media)
|
||||
.file_name("audio")
|
||||
.mime_str(mime_type.as_ref())?;
|
||||
|
||||
let mut form = reqwest::multipart::Form::new()
|
||||
.part("file", part)
|
||||
.text("model", speech_to_text_config.model_id.clone())
|
||||
.text("response_format", "json");
|
||||
|
||||
if let Some(language) = ¶ms.language_override {
|
||||
form = form.text("language", language.clone());
|
||||
}
|
||||
|
||||
let url = format!(
|
||||
"{}/audio/transcriptions",
|
||||
config.base_url.trim_end_matches('/')
|
||||
);
|
||||
|
||||
tracing::trace!(
|
||||
model_id = speech_to_text_config.model_id,
|
||||
language = ?params.language_override,
|
||||
"Sending Venice audio transcription API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.multipart(form)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice audio transcription request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice audio transcription request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
let response: TranscriptionResponse = response.json().await?;
|
||||
|
||||
Ok(SpeechToTextResult {
|
||||
text: response.text,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn text_to_speech(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
input: &str,
|
||||
params: TextToSpeechParams,
|
||||
) -> anyhow::Result<TextToSpeechResult> {
|
||||
let Some(text_to_speech_config) = &config.text_to_speech else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::TextToSpeech
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
// Per-call overrides win over the configured defaults.
|
||||
let voice = params
|
||||
.voice_override
|
||||
.or_else(|| text_to_speech_config.voice.clone());
|
||||
let speed = params.speed_override.or(text_to_speech_config.speed);
|
||||
|
||||
let response_format = text_to_speech_config.response_format.clone();
|
||||
let mime_type = response_format_to_mime_type(response_format.as_deref());
|
||||
|
||||
let request = SpeechRequest {
|
||||
model: text_to_speech_config.model_id.clone(),
|
||||
input: input.to_owned(),
|
||||
voice,
|
||||
speed,
|
||||
response_format,
|
||||
prompt: text_to_speech_config.prompt.clone(),
|
||||
temperature: text_to_speech_config.temperature,
|
||||
top_p: text_to_speech_config.top_p,
|
||||
};
|
||||
|
||||
let url = format!("{}/audio/speech", config.base_url.trim_end_matches('/'));
|
||||
|
||||
tracing::trace!(
|
||||
model_id = text_to_speech_config.model_id,
|
||||
voice = ?request.voice,
|
||||
"Sending Venice text-to-speech API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice text-to-speech request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice text-to-speech request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
// The speech endpoint answers with raw binary audio; read the body directly.
|
||||
let bytes = response.bytes().await?.to_vec();
|
||||
|
||||
Ok(TextToSpeechResult { bytes, mime_type })
|
||||
}
|
||||
|
||||
/// Map a Venice TTS `response_format` to its MIME type. Defaults to `audio/mpeg` (the
|
||||
/// IANA-registered MP3 type, RFC 3003) when the format is unset, matching Venice's own `mp3`
|
||||
/// default. This deliberately uses `audio/mpeg` rather than the `audio/mp3` alias the openai
|
||||
/// provider emits; baibot's downstream audio-filename mapping treats both as `.mp3`.
|
||||
fn response_format_to_mime_type(response_format: Option<&str>) -> mxlink::mime::Mime {
|
||||
let raw = match response_format.unwrap_or("mp3") {
|
||||
"mp3" => "audio/mpeg",
|
||||
"opus" => "audio/ogg",
|
||||
"aac" => "audio/aac",
|
||||
"flac" => "audio/flac",
|
||||
"wav" => "audio/wav",
|
||||
"pcm" => "audio/L8",
|
||||
_ => "audio/mpeg",
|
||||
};
|
||||
|
||||
raw.parse()
|
||||
.unwrap_or(mxlink::mime::APPLICATION_OCTET_STREAM)
|
||||
}
|
||||
122
src/agent/provider/venice/chat.rs
Normal file
@@ -0,0 +1,122 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
use super::config::Config;
|
||||
use super::utils::convert_llm_messages_to_venice;
|
||||
use super::wire::{ChatCompletionRequest, ChatCompletionResponse};
|
||||
|
||||
pub async fn generate_text(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
conversation: LLMConversation,
|
||||
params: TextGenerationParams,
|
||||
) -> anyhow::Result<TextGenerationResult> {
|
||||
let Some(text_generation_config) = &config.text_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::TextGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
let prompt_text = params.prompt_variables.format(
|
||||
params
|
||||
.prompt_override
|
||||
.unwrap_or(text_generation_config.prompt.clone().unwrap_or_default())
|
||||
.trim(),
|
||||
);
|
||||
|
||||
let prompt_message = if prompt_text.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
|
||||
let mut conversation_messages = conversation.messages;
|
||||
|
||||
if params.context_management_enabled {
|
||||
conversation_messages = shorten_messages_list_to_context_size(
|
||||
&text_generation_config.model_id,
|
||||
&prompt_message,
|
||||
conversation_messages,
|
||||
text_generation_config.max_response_tokens,
|
||||
text_generation_config.max_context_tokens,
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(prompt_message) = prompt_message {
|
||||
conversation_messages.insert(0, prompt_message);
|
||||
}
|
||||
|
||||
let messages = convert_llm_messages_to_venice(conversation_messages);
|
||||
|
||||
let temperature = params
|
||||
.temperature_override
|
||||
.unwrap_or(text_generation_config.temperature);
|
||||
|
||||
let request = ChatCompletionRequest {
|
||||
model: text_generation_config.model_id.clone(),
|
||||
messages,
|
||||
temperature: Some(temperature),
|
||||
// Web search rides entirely inside `venice_parameters`; there is no `tools` array here.
|
||||
// `max_tokens` is deprecated on Venice in favor of `max_completion_tokens`.
|
||||
max_completion_tokens: text_generation_config.max_response_tokens,
|
||||
venice_parameters: text_generation_config.venice_parameters.clone(),
|
||||
};
|
||||
|
||||
let url = format!(
|
||||
"{}/chat/completions",
|
||||
config.base_url.trim_end_matches('/')
|
||||
);
|
||||
|
||||
tracing::trace!(
|
||||
model = text_generation_config.model_id,
|
||||
messages_count = request.messages.len(),
|
||||
"Sending Venice chat completion API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Log the body server-side for debugging (Venice explains a rejected strict body there),
|
||||
// but keep it OUT of the returned error: that error surfaces in the Matrix room, and the
|
||||
// body can carry account / rate-limit details that shouldn't reach room members.
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice chat completion request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice chat completion request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
let response: ChatCompletionResponse = response.json().await?;
|
||||
|
||||
let Some(choice) = response.choices.into_iter().next() else {
|
||||
return Err(anyhow::anyhow!(
|
||||
"No choices were returned from the Venice chat completion API"
|
||||
));
|
||||
};
|
||||
|
||||
let Some(content) = choice.message.content else {
|
||||
return Err(anyhow::anyhow!(
|
||||
"No message content was returned from the Venice chat completion API"
|
||||
));
|
||||
};
|
||||
|
||||
Ok(TextGenerationResult { text: content })
|
||||
}
|
||||
355
src/agent/provider/venice/config.rs
Normal file
@@ -0,0 +1,355 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::agent::{default_prompt, provider::ConfigTrait};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct Config {
|
||||
pub base_url: String,
|
||||
|
||||
pub api_key: String,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub text_generation: Option<TextGenerationConfig>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub speech_to_text: Option<SpeechToTextConfig>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub text_to_speech: Option<TextToSpeechConfig>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub image_generation: Option<ImageGenerationConfig>,
|
||||
}
|
||||
|
||||
impl Default for Config {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
base_url: "https://api.venice.ai/api/v1".to_owned(),
|
||||
api_key: "YOUR_API_KEY_HERE".to_owned(),
|
||||
text_generation: Some(TextGenerationConfig::default()),
|
||||
speech_to_text: Some(SpeechToTextConfig::default()),
|
||||
text_to_speech: Some(TextToSpeechConfig::default()),
|
||||
image_generation: Some(ImageGenerationConfig::default()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ConfigTrait for Config {
|
||||
fn validate(&self) -> Result<(), String> {
|
||||
if self.base_url.is_empty() {
|
||||
return Err("The base URL must not be empty.".to_owned());
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextGenerationConfig {
|
||||
#[serde(default = "default_text_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default)]
|
||||
pub prompt: Option<String>,
|
||||
|
||||
#[serde(default = "super::super::default_temperature")]
|
||||
pub temperature: f32,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_response_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_context_tokens: u32,
|
||||
|
||||
/// Venice-specific request knobs, serialized 1:1 into the `venice_parameters` bag on the
|
||||
/// wire. Any unset field is omitted, so Venice applies its own server-side default.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub venice_parameters: Option<VeniceParameters>,
|
||||
}
|
||||
|
||||
impl Default for TextGenerationConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_text_model_id(),
|
||||
prompt: Some(default_prompt().to_owned()),
|
||||
temperature: super::super::default_temperature(),
|
||||
// Reserved output budget: sent as the response cap AND subtracted from the context
|
||||
// window when trimming history. Mirrors the openai_compat sibling's default.
|
||||
max_response_tokens: Some(4096),
|
||||
// Matches Venice's own `availableContextTokens` (131072) and the non-OpenAI sibling
|
||||
// providers (ollama/localai/mistral all default to 128_000).
|
||||
max_context_tokens: 128_000,
|
||||
// A usable starting point, not an everything-set dump: only these three are sent;
|
||||
// every other knob stays None so Venice applies its own default (omitting != false).
|
||||
venice_parameters: Some(VeniceParameters {
|
||||
enable_web_search: Some(WebSearchMode::Auto),
|
||||
strip_thinking_response: Some(true),
|
||||
enable_e2ee: Some(false),
|
||||
..Default::default()
|
||||
}),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_model_id() -> String {
|
||||
"kimi-k2-5".to_owned()
|
||||
}
|
||||
|
||||
/// The full `venice_parameters` knob set, mirroring Venice's `ChatCompletionRequest`
|
||||
/// schema field-for-field. Every field is optional with `skip_serializing_if`, so the
|
||||
/// request never carries a knob the user didn't set (the body is `additionalProperties: false`,
|
||||
/// and an unset knob simply omits rather than sending `null`).
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||
pub struct VeniceParameters {
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_search: Option<WebSearchMode>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_citations: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_scraping: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub include_venice_system_prompt: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub include_search_results_in_stream: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub return_search_results_as_documents: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_x_search: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_e2ee: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub character_slug: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub strip_thinking_response: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub disable_thinking: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum WebSearchMode {
|
||||
Auto,
|
||||
On,
|
||||
Off,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct SpeechToTextConfig {
|
||||
#[serde(default = "default_speech_to_text_model_id")]
|
||||
pub model_id: String,
|
||||
}
|
||||
|
||||
impl Default for SpeechToTextConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_speech_to_text_model_id(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_speech_to_text_model_id() -> String {
|
||||
"nvidia/parakeet-tdt-0.6b-v3".to_owned()
|
||||
}
|
||||
|
||||
/// `/audio/speech` (`CreateSpeechRequestSchema`) request knobs. Only `model_id` is required on
|
||||
/// the wire; everything else is optional with `skip_serializing_if` so an unset knob is omitted
|
||||
/// rather than sent as `null` (the body is `additionalProperties: false`). `voice` is a free
|
||||
/// `Option<String>`, not a closed enum: Venice's voice set spans dozens of model-specific names
|
||||
/// plus arbitrary cloned-voice handles (`vv_<id>`), so an enum would reject valid handles.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextToSpeechConfig {
|
||||
#[serde(default = "default_text_to_speech_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(
|
||||
default = "default_text_to_speech_voice",
|
||||
skip_serializing_if = "Option::is_none"
|
||||
)]
|
||||
pub voice: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub speed: Option<f32>,
|
||||
|
||||
#[serde(
|
||||
default = "default_text_to_speech_response_format",
|
||||
skip_serializing_if = "Option::is_none"
|
||||
)]
|
||||
pub response_format: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub prompt: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub temperature: Option<f32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub top_p: Option<f32>,
|
||||
}
|
||||
|
||||
impl Default for TextToSpeechConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_text_to_speech_model_id(),
|
||||
voice: default_text_to_speech_voice(),
|
||||
speed: None,
|
||||
response_format: default_text_to_speech_response_format(),
|
||||
prompt: None,
|
||||
temperature: None,
|
||||
top_p: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_to_speech_model_id() -> String {
|
||||
"tts-kokoro".to_owned()
|
||||
}
|
||||
|
||||
fn default_text_to_speech_voice() -> Option<String> {
|
||||
Some("af_sky".to_owned())
|
||||
}
|
||||
|
||||
fn default_text_to_speech_response_format() -> Option<String> {
|
||||
Some("mp3".to_owned())
|
||||
}
|
||||
|
||||
/// `/image/generate` (`GenerateImageRequest`) request knobs, mirroring Venice's schema
|
||||
/// field-for-field. Only `model_id` is required; every other knob is optional with
|
||||
/// `skip_serializing_if` so unset knobs are omitted (the body is `additionalProperties: false`).
|
||||
/// The full knob set is deliberate: the native `/image/generate` endpoint is the flagship's
|
||||
/// reason to exist over the knob-dropping OpenAI-compat path, so the knobs ARE the feature.
|
||||
/// The deprecated `inpaint` knob is intentionally absent.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct ImageGenerationConfig {
|
||||
#[serde(default = "default_image_generation_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub negative_prompt: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub cfg_scale: Option<f32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub steps: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub style_preset: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub seed: Option<i64>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub hide_watermark: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub format: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub width: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub height: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub quality: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub lora_strength: Option<u32>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub embed_exif_metadata: Option<bool>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_search: Option<bool>,
|
||||
|
||||
/// Image-edit settings, nested here because baibot has a single `ImageGeneration` purpose
|
||||
/// and edit shares its config gate. The gen and edit model sets are disjoint, so edit
|
||||
/// carries its own model field.
|
||||
#[serde(default)]
|
||||
pub edit: ImageEditSettings,
|
||||
}
|
||||
|
||||
impl Default for ImageGenerationConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_image_generation_model_id(),
|
||||
negative_prompt: None,
|
||||
cfg_scale: None,
|
||||
steps: None,
|
||||
style_preset: None,
|
||||
seed: None,
|
||||
safe_mode: None,
|
||||
hide_watermark: None,
|
||||
format: None,
|
||||
width: None,
|
||||
height: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
quality: None,
|
||||
lora_strength: None,
|
||||
embed_exif_metadata: None,
|
||||
enable_web_search: None,
|
||||
edit: ImageEditSettings::default(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_image_generation_model_id() -> String {
|
||||
"chroma".to_owned()
|
||||
}
|
||||
|
||||
/// `/image/edit` (`EditImageRequest`) request knobs, mirroring Venice's schema. The source image
|
||||
/// and prompt are supplied per-call (not config), so only the model and the output-shaping knobs
|
||||
/// live here. Each knob is optional with `skip_serializing_if`.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct ImageEditSettings {
|
||||
#[serde(default = "default_image_edit_model_id")]
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub output_format: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
}
|
||||
|
||||
impl Default for ImageEditSettings {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: default_image_edit_model_id(),
|
||||
output_format: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
safe_mode: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_image_edit_model_id() -> String {
|
||||
"firered-image-edit".to_owned()
|
||||
}
|
||||
146
src/agent/provider/venice/controller.rs
Normal file
@@ -0,0 +1,146 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
|
||||
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
|
||||
};
|
||||
use crate::agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use super::config::Config;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Controller {
|
||||
config: Config,
|
||||
http: reqwest::Client,
|
||||
}
|
||||
|
||||
impl Controller {
|
||||
pub fn new(config: Config) -> Self {
|
||||
// Image generation and text-to-speech can run long, so give the client a generous timeout
|
||||
// instead of reqwest's default (none). `build` only fails on TLS/system init; fall back to
|
||||
// the infallible `Client::new()` so this constructor stays infallible.
|
||||
let http = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(120))
|
||||
.build()
|
||||
.unwrap_or_else(|_| reqwest::Client::new());
|
||||
|
||||
Self { config, http }
|
||||
}
|
||||
}
|
||||
|
||||
impl ControllerTrait for Controller {
|
||||
async fn ping(&self) -> anyhow::Result<PingResult> {
|
||||
if !self.supports_purpose(AgentPurpose::TextGeneration) {
|
||||
return Ok(PingResult::Inconclusive);
|
||||
}
|
||||
|
||||
// Mirror the openai/openai_compat ping: a real "Hello!" round-trip exercises the strict
|
||||
// /chat/completions body and auth, so a successful ping proves text generation works.
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
let conversation = LLMConversation { messages };
|
||||
|
||||
self.generate_text(conversation, TextGenerationParams::default())
|
||||
.await?;
|
||||
|
||||
Ok(PingResult::Successful)
|
||||
}
|
||||
|
||||
async fn generate_text(
|
||||
&self,
|
||||
conversation: LLMConversation,
|
||||
params: TextGenerationParams,
|
||||
) -> anyhow::Result<TextGenerationResult> {
|
||||
super::chat::generate_text(&self.config, &self.http, conversation, params).await
|
||||
}
|
||||
|
||||
async fn speech_to_text(
|
||||
&self,
|
||||
mime_type: &mxlink::mime::Mime,
|
||||
media: Vec<u8>,
|
||||
params: SpeechToTextParams,
|
||||
) -> anyhow::Result<SpeechToTextResult> {
|
||||
super::audio::speech_to_text(&self.config, &self.http, mime_type, media, params).await
|
||||
}
|
||||
|
||||
async fn generate_image(
|
||||
&self,
|
||||
prompt: &str,
|
||||
params: ImageGenerationParams,
|
||||
) -> anyhow::Result<ImageGenerationResult> {
|
||||
super::images::generate_image(&self.config, &self.http, prompt, params).await
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
super::images::create_image_edit(&self.config, &self.http, prompt, images, params).await
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
input: &str,
|
||||
params: TextToSpeechParams,
|
||||
) -> anyhow::Result<TextToSpeechResult> {
|
||||
super::audio::text_to_speech(&self.config, &self.http, input, params).await
|
||||
}
|
||||
|
||||
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
|
||||
match purpose {
|
||||
AgentPurpose::TextGeneration => self.config.text_generation.is_some(),
|
||||
AgentPurpose::SpeechToText => self.config.speech_to_text.is_some(),
|
||||
AgentPurpose::TextToSpeech => self.config.text_to_speech.is_some(),
|
||||
AgentPurpose::ImageGeneration => self.config.image_generation.is_some(),
|
||||
AgentPurpose::CatchAll => true,
|
||||
}
|
||||
}
|
||||
|
||||
fn text_generation_model_id(&self) -> Option<String> {
|
||||
self.config
|
||||
.text_generation
|
||||
.as_ref()
|
||||
.map(|config| config.model_id.to_owned())
|
||||
}
|
||||
|
||||
fn text_generation_prompt(&self) -> Option<String> {
|
||||
self.config
|
||||
.text_generation
|
||||
.as_ref()
|
||||
.and_then(|config| config.prompt.clone())
|
||||
}
|
||||
|
||||
fn text_generation_temperature(&self) -> Option<f32> {
|
||||
self.config
|
||||
.text_generation
|
||||
.as_ref()
|
||||
.map(|config| config.temperature)
|
||||
}
|
||||
|
||||
fn text_to_speech_voice(&self) -> Option<String> {
|
||||
self.config
|
||||
.text_to_speech
|
||||
.as_ref()
|
||||
.and_then(|config| config.voice.clone())
|
||||
}
|
||||
|
||||
fn text_to_speech_speed(&self) -> Option<f32> {
|
||||
self.config
|
||||
.text_to_speech
|
||||
.as_ref()
|
||||
.and_then(|config| config.speed)
|
||||
}
|
||||
}
|
||||
192
src/agent/provider/venice/images.rs
Normal file
@@ -0,0 +1,192 @@
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{ImageEditResult, ImageGenerationResult, ImageSource};
|
||||
use crate::agent::provider::{ImageEditParams, ImageGenerationParams};
|
||||
use crate::strings;
|
||||
use crate::utils::base64::{base64_decode, base64_encode};
|
||||
|
||||
use super::config::Config;
|
||||
use super::wire::{EditImageRequest, GenerateImageRequest, GenerateImageResponse};
|
||||
|
||||
/// Generate an image via Venice's native `/image/generate` endpoint.
|
||||
///
|
||||
/// This is the base64-in-JSON path: we pin `return_binary: false` so Venice answers with a JSON
|
||||
/// envelope (`GenerateImageResponse`) carrying the image as a base64 string, which we decode. The
|
||||
/// sibling `create_image_edit` is the *other* response shape (raw binary); the two must not be
|
||||
/// crossed. `params` is advisory only; the Venice config drives the request.
|
||||
pub async fn generate_image(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
prompt: &str,
|
||||
_params: ImageGenerationParams,
|
||||
) -> anyhow::Result<ImageGenerationResult> {
|
||||
let Some(image_generation_config) = &config.image_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::ImageGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
let request = GenerateImageRequest {
|
||||
model: image_generation_config.model_id.clone(),
|
||||
prompt: prompt.to_owned(),
|
||||
// Pinned: baibot wants exactly one image, returned as base64-in-JSON so `GenerateImageResponse`
|
||||
// can decode it. Flipping `return_binary` would make Venice answer with raw binary and break
|
||||
// the JSON decode below, so neither knob is configurable.
|
||||
return_binary: false,
|
||||
variants: 1,
|
||||
negative_prompt: image_generation_config.negative_prompt.clone(),
|
||||
cfg_scale: image_generation_config.cfg_scale,
|
||||
steps: image_generation_config.steps,
|
||||
style_preset: image_generation_config.style_preset.clone(),
|
||||
seed: image_generation_config.seed,
|
||||
safe_mode: image_generation_config.safe_mode,
|
||||
hide_watermark: image_generation_config.hide_watermark,
|
||||
format: image_generation_config.format.clone(),
|
||||
width: image_generation_config.width,
|
||||
height: image_generation_config.height,
|
||||
aspect_ratio: image_generation_config.aspect_ratio.clone(),
|
||||
resolution: image_generation_config.resolution.clone(),
|
||||
quality: image_generation_config.quality.clone(),
|
||||
lora_strength: image_generation_config.lora_strength,
|
||||
embed_exif_metadata: image_generation_config.embed_exif_metadata,
|
||||
enable_web_search: image_generation_config.enable_web_search,
|
||||
};
|
||||
|
||||
let url = format!("{}/image/generate", config.base_url.trim_end_matches('/'));
|
||||
|
||||
// The prompt is user content; keep it out of logs (mirrors the STT/TTS paths).
|
||||
tracing::trace!(
|
||||
model_id = image_generation_config.model_id,
|
||||
"Sending Venice image generation API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice image generation request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice image generation request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
let response: GenerateImageResponse = response.json().await?;
|
||||
|
||||
tracing::trace!(request_id = ?response.id, "Venice image generation succeeded");
|
||||
|
||||
let Some(image_base64) = response.images.into_iter().next() else {
|
||||
return Err(anyhow::anyhow!(
|
||||
"The Venice image generation API returned no images"
|
||||
));
|
||||
};
|
||||
|
||||
// Swallow the decode error's detail (it can echo input bytes/offsets); the returned error
|
||||
// reaches the Matrix room, so it stays generic while the real cause goes to the server log.
|
||||
let bytes = base64_decode(&image_base64).map_err(|decode_err| {
|
||||
tracing::warn!(%decode_err, "Venice image generation returned undecodable base64");
|
||||
anyhow::anyhow!("Venice image generation returned invalid base64 image data")
|
||||
})?;
|
||||
|
||||
Ok(ImageGenerationResult {
|
||||
bytes,
|
||||
mime_type: image_format_to_mime_type(image_generation_config.format.as_deref()),
|
||||
revised_prompt: None,
|
||||
})
|
||||
}
|
||||
|
||||
/// Edit an image via Venice's native `/image/edit` endpoint.
|
||||
///
|
||||
/// This is the raw-binary path: the request is JSON carrying the source image as a base64 string
|
||||
/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64, no multipart), and the
|
||||
/// response body IS the edited image bytes (no JSON envelope). `params` is advisory only.
|
||||
pub async fn create_image_edit(
|
||||
config: &Config,
|
||||
http: &reqwest::Client,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
let Some(image_generation_config) = &config.image_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::ImageGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
let edit_config = &image_generation_config.edit;
|
||||
|
||||
let Some(source) = images.into_iter().next() else {
|
||||
return Err(anyhow::anyhow!("No image sources provided"));
|
||||
};
|
||||
|
||||
let request = EditImageRequest {
|
||||
model: edit_config.model_id.clone(),
|
||||
prompt: prompt.to_owned(),
|
||||
image: base64_encode(&source.bytes),
|
||||
output_format: edit_config.output_format.clone(),
|
||||
aspect_ratio: edit_config.aspect_ratio.clone(),
|
||||
resolution: edit_config.resolution.clone(),
|
||||
safe_mode: edit_config.safe_mode,
|
||||
};
|
||||
|
||||
let url = format!("{}/image/edit", config.base_url.trim_end_matches('/'));
|
||||
|
||||
// The prompt is user content; keep it out of logs (mirrors the STT/TTS paths).
|
||||
tracing::trace!(
|
||||
model_id = edit_config.model_id,
|
||||
"Sending Venice image edit API request"
|
||||
);
|
||||
|
||||
let response = http
|
||||
.post(&url)
|
||||
.bearer_auth(&config.api_key)
|
||||
.json(&request)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let status = response.status();
|
||||
if !status.is_success() {
|
||||
// Body to the server log only, not into the returned error (which reaches the Matrix room).
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
tracing::warn!(%status, body, "Venice image edit request failed");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Venice image edit request failed with status {status}"
|
||||
));
|
||||
}
|
||||
|
||||
// The edit endpoint answers with raw binary image bytes, so read the body directly instead of
|
||||
// parsing JSON. The actual format comes from the response Content-Type header; fall back to the
|
||||
// configured `output_format` when the header is missing or unparseable.
|
||||
let mime_type = response
|
||||
.headers()
|
||||
.get(reqwest::header::CONTENT_TYPE)
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<mxlink::mime::Mime>().ok())
|
||||
.unwrap_or_else(|| image_format_to_mime_type(edit_config.output_format.as_deref()));
|
||||
|
||||
let bytes = response.bytes().await?.to_vec();
|
||||
|
||||
Ok(ImageEditResult { bytes, mime_type })
|
||||
}
|
||||
|
||||
/// Map a Venice image `format`/`output_format` value (`jpeg`/`png`/`webp`) to its MIME type.
|
||||
/// Venice defaults to `webp` when the format is unset, so an absent value maps to `image/webp`.
|
||||
fn image_format_to_mime_type(format: Option<&str>) -> mxlink::mime::Mime {
|
||||
match format.unwrap_or("webp") {
|
||||
"jpeg" | "jpg" => mxlink::mime::IMAGE_JPEG,
|
||||
"png" => mxlink::mime::IMAGE_PNG,
|
||||
// No mxlink::mime constant for webp; parse it, falling back to PNG on any surprise value.
|
||||
_ => "image/webp"
|
||||
.parse()
|
||||
.unwrap_or(mxlink::mime::IMAGE_PNG),
|
||||
}
|
||||
}
|
||||
47
src/agent/provider/venice/mod.rs
Normal file
@@ -0,0 +1,47 @@
|
||||
mod audio;
|
||||
mod chat;
|
||||
mod config;
|
||||
mod controller;
|
||||
mod images;
|
||||
mod utils;
|
||||
mod wire;
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests;
|
||||
|
||||
pub use config::Config;
|
||||
pub use controller::Controller;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
config: serde_yaml_ng::Value,
|
||||
) -> AgentInstantiationResult<ControllerType> {
|
||||
let config = match &config {
|
||||
serde_yaml_ng::Value::Mapping(_) => {
|
||||
let config: Config =
|
||||
serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?;
|
||||
|
||||
config
|
||||
.validate()
|
||||
.map_err(AgentInstantiationError::ConfigFailsValidation)?;
|
||||
|
||||
config
|
||||
}
|
||||
_ => {
|
||||
return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping(
|
||||
agent_id.to_owned(),
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
Ok(ControllerType::Venice(Box::new(Controller::new(config))))
|
||||
}
|
||||
|
||||
pub fn default_config() -> Config {
|
||||
Config::default()
|
||||
}
|
||||
269
src/agent/provider/venice/tests.rs
Normal file
@@ -0,0 +1,269 @@
|
||||
use mxlink::matrix_sdk::ruma::OwnedMxcUri;
|
||||
use mxlink::matrix_sdk::ruma::events::room::message::{
|
||||
FileMessageEventContent, ImageMessageEventContent,
|
||||
};
|
||||
use mxlink::mime;
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, FileDetails, ImageDetails, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
use super::config::{Config, VeniceParameters, WebSearchMode};
|
||||
use super::controller::Controller;
|
||||
use super::utils::convert_llm_messages_to_venice;
|
||||
use super::wire::{
|
||||
ContentPart, EditImageRequest, GenerateImageRequest, MessageContent, SpeechRequest,
|
||||
};
|
||||
|
||||
#[test]
|
||||
fn config_round_trips_with_venice_parameters() {
|
||||
let yaml = r#"
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: test-key
|
||||
text_generation:
|
||||
model_id: kimi-k2-5
|
||||
temperature: 0.7
|
||||
max_response_tokens: 1024
|
||||
max_context_tokens: 65536
|
||||
venice_parameters:
|
||||
enable_web_search: "auto"
|
||||
enable_web_citations: true
|
||||
speech_to_text:
|
||||
model_id: nvidia/parakeet-tdt-0.6b-v3
|
||||
"#;
|
||||
|
||||
let config: Config = serde_yaml_ng::from_str(yaml).expect("config should deserialize");
|
||||
|
||||
let tg = config.text_generation.expect("text_generation present");
|
||||
let vp = tg.venice_parameters.expect("venice_parameters present");
|
||||
|
||||
assert!(matches!(vp.enable_web_search, Some(WebSearchMode::Auto)));
|
||||
assert_eq!(vp.enable_web_citations, Some(true));
|
||||
assert_eq!(vp.character_slug, None);
|
||||
|
||||
// The bag must serialize the enum to the exact wire string, and an unset knob must be ABSENT
|
||||
// (not `null`) so the strict `additionalProperties: false` body is honored.
|
||||
let json = serde_json::to_string(&vp).expect("serialize venice_parameters");
|
||||
assert!(
|
||||
json.contains("\"enable_web_search\":\"auto\""),
|
||||
"web search should be the literal \"auto\": {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("character_slug"),
|
||||
"an unset knob must be omitted entirely: {json}"
|
||||
);
|
||||
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_image_to_data_uri_and_skips_files() {
|
||||
let messages = vec![
|
||||
LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
timestamp: chrono::Utc::now(),
|
||||
content: LLMMessageContent::Text("describe this".to_owned()),
|
||||
},
|
||||
LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
timestamp: chrono::Utc::now(),
|
||||
content: LLMMessageContent::Image(ImageDetails::new(
|
||||
ImageMessageEventContent::plain(
|
||||
"pic.png".to_owned(),
|
||||
OwnedMxcUri::from("mxc://example.com/abc"),
|
||||
),
|
||||
mime::IMAGE_PNG,
|
||||
vec![1, 2, 3],
|
||||
)),
|
||||
},
|
||||
LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
sender_id: None,
|
||||
timestamp: chrono::Utc::now(),
|
||||
content: LLMMessageContent::File(FileDetails::new(
|
||||
FileMessageEventContent::plain(
|
||||
"doc.pdf".to_owned(),
|
||||
OwnedMxcUri::from("mxc://example.com/def"),
|
||||
),
|
||||
mime::APPLICATION_PDF,
|
||||
vec![4, 5, 6],
|
||||
)),
|
||||
},
|
||||
];
|
||||
|
||||
let converted = convert_llm_messages_to_venice(messages);
|
||||
|
||||
// Text and image survive; the file is warn-skipped.
|
||||
assert_eq!(converted.len(), 2);
|
||||
|
||||
match &converted[0].content {
|
||||
MessageContent::Text(text) => assert_eq!(text, "describe this"),
|
||||
other => panic!("expected bare text, got {other:?}"),
|
||||
}
|
||||
|
||||
match &converted[1].content {
|
||||
MessageContent::Parts(parts) => match &parts[0] {
|
||||
ContentPart::ImageUrl { image_url } => assert!(
|
||||
image_url.url.starts_with("data:image/png;base64,"),
|
||||
"image should be inlined as a data URI: {}",
|
||||
image_url.url
|
||||
),
|
||||
},
|
||||
other => panic!("expected image parts, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn supports_purpose_truth_table() {
|
||||
let config: Config = serde_yaml_ng::from_str(
|
||||
r#"
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: test-key
|
||||
text_generation:
|
||||
model_id: kimi-k2-5
|
||||
speech_to_text:
|
||||
model_id: nvidia/parakeet-tdt-0.6b-v3
|
||||
"#,
|
||||
)
|
||||
.expect("config should deserialize");
|
||||
|
||||
let controller = Controller::new(config);
|
||||
|
||||
assert!(controller.supports_purpose(AgentPurpose::TextGeneration));
|
||||
assert!(controller.supports_purpose(AgentPurpose::SpeechToText));
|
||||
assert!(controller.supports_purpose(AgentPurpose::CatchAll));
|
||||
assert!(!controller.supports_purpose(AgentPurpose::TextToSpeech));
|
||||
assert!(!controller.supports_purpose(AgentPurpose::ImageGeneration));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn supports_purpose_true_when_image_and_tts_blocks_present() {
|
||||
let config: Config = serde_yaml_ng::from_str(
|
||||
r#"
|
||||
base_url: https://api.venice.ai/api/v1
|
||||
api_key: test-key
|
||||
text_to_speech:
|
||||
model_id: tts-kokoro
|
||||
image_generation:
|
||||
model_id: chroma
|
||||
"#,
|
||||
)
|
||||
.expect("config should deserialize");
|
||||
|
||||
let controller = Controller::new(config);
|
||||
|
||||
assert!(controller.supports_purpose(AgentPurpose::TextToSpeech));
|
||||
assert!(controller.supports_purpose(AgentPurpose::ImageGeneration));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn speech_request_serializes_voice_and_omits_unset() {
|
||||
let request = SpeechRequest {
|
||||
model: "tts-kokoro".to_owned(),
|
||||
input: "hello".to_owned(),
|
||||
voice: Some("af_sky".to_owned()),
|
||||
speed: None,
|
||||
response_format: Some("mp3".to_owned()),
|
||||
prompt: None,
|
||||
temperature: None,
|
||||
top_p: None,
|
||||
};
|
||||
|
||||
let json = serde_json::to_string(&request).expect("serialize SpeechRequest");
|
||||
|
||||
assert!(
|
||||
json.contains("\"voice\":\"af_sky\""),
|
||||
"voice should be present: {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("temperature"),
|
||||
"an unset knob must be omitted (not null): {json}"
|
||||
);
|
||||
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn generate_image_request_pins_flags_and_omits_unset() {
|
||||
let request = GenerateImageRequest {
|
||||
model: "chroma".to_owned(),
|
||||
prompt: "a cat".to_owned(),
|
||||
return_binary: false,
|
||||
variants: 1,
|
||||
negative_prompt: None,
|
||||
cfg_scale: None,
|
||||
steps: None,
|
||||
style_preset: None,
|
||||
seed: None,
|
||||
safe_mode: None,
|
||||
hide_watermark: None,
|
||||
format: None,
|
||||
width: None,
|
||||
height: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
quality: None,
|
||||
lora_strength: None,
|
||||
embed_exif_metadata: None,
|
||||
enable_web_search: None,
|
||||
};
|
||||
|
||||
let json = serde_json::to_string(&request).expect("serialize GenerateImageRequest");
|
||||
|
||||
assert!(json.contains("\"model\":\"chroma\""), "{json}");
|
||||
assert!(
|
||||
json.contains("\"return_binary\":false"),
|
||||
"return_binary must be pinned false: {json}"
|
||||
);
|
||||
assert!(
|
||||
json.contains("\"variants\":1"),
|
||||
"variants must be pinned 1: {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("cfg_scale"),
|
||||
"an unset knob must be omitted: {json}"
|
||||
);
|
||||
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edit_image_request_carries_model_and_base64_image() {
|
||||
let request = EditImageRequest {
|
||||
model: "firered-image-edit".to_owned(),
|
||||
prompt: "make it a sunrise".to_owned(),
|
||||
image: "aGVsbG8=".to_owned(),
|
||||
output_format: None,
|
||||
aspect_ratio: None,
|
||||
resolution: None,
|
||||
safe_mode: None,
|
||||
};
|
||||
|
||||
let json = serde_json::to_string(&request).expect("serialize EditImageRequest");
|
||||
|
||||
assert!(
|
||||
json.contains("\"model\":\"firered-image-edit\""),
|
||||
"{json}"
|
||||
);
|
||||
assert!(
|
||||
json.contains("\"image\":\"aGVsbG8=\""),
|
||||
"the base64 image string must be present: {json}"
|
||||
);
|
||||
assert!(
|
||||
!json.contains("output_format"),
|
||||
"an unset knob must be omitted: {json}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn web_search_mode_off_deserializes_from_bare_yaml_off() {
|
||||
// `off` is a YAML-1.1 boolean but a plain string under serde_yaml_ng's YAML-1.2 core schema,
|
||||
// so it deserializes straight into the lowercase `WebSearchMode::Off`. This pins that the
|
||||
// sample config and docs can use the bare, unquoted `off` without it parsing as a boolean.
|
||||
let params: VeniceParameters =
|
||||
serde_yaml_ng::from_str("enable_web_search: off").expect("bare `off` should deserialize");
|
||||
|
||||
assert!(matches!(params.enable_web_search, Some(WebSearchMode::Off)));
|
||||
}
|
||||
55
src/agent/provider/venice/utils.rs
Normal file
@@ -0,0 +1,55 @@
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
use crate::utils::base64::base64_encode;
|
||||
|
||||
use super::wire::{ChatMessage, ContentPart, ImageUrl, MessageContent};
|
||||
|
||||
pub fn convert_llm_messages_to_venice(messages: Vec<LLMMessage>) -> Vec<ChatMessage> {
|
||||
let mut venice_messages: Vec<ChatMessage> = Vec::with_capacity(messages.len());
|
||||
|
||||
for message in messages {
|
||||
if let Some(venice_message) = convert_llm_message_to_venice(message) {
|
||||
venice_messages.push(venice_message);
|
||||
}
|
||||
}
|
||||
|
||||
venice_messages
|
||||
}
|
||||
|
||||
fn convert_llm_message_to_venice(message: LLMMessage) -> Option<ChatMessage> {
|
||||
let role = match message.author {
|
||||
LLMAuthor::Prompt => "system",
|
||||
LLMAuthor::Assistant => "assistant",
|
||||
LLMAuthor::User => "user",
|
||||
};
|
||||
|
||||
match message.content {
|
||||
LLMMessageContent::Text(text) => Some(ChatMessage {
|
||||
role: role.to_owned(),
|
||||
content: MessageContent::Text(text),
|
||||
}),
|
||||
LLMMessageContent::Image(image_details) => {
|
||||
// Inline the image as a base64 data URI, the same shape the OpenAI vision content
|
||||
// part uses. This is the gap the openai_compat provider can't fill (it drops images).
|
||||
let data_uri = format!(
|
||||
"data:{};base64,{}",
|
||||
image_details.mime,
|
||||
base64_encode(&image_details.data)
|
||||
);
|
||||
|
||||
Some(ChatMessage {
|
||||
role: role.to_owned(),
|
||||
content: MessageContent::Parts(vec![ContentPart::ImageUrl {
|
||||
image_url: ImageUrl { url: data_uri },
|
||||
}]),
|
||||
})
|
||||
}
|
||||
LLMMessageContent::File(_file_details) => {
|
||||
tracing::warn!(
|
||||
"The Venice provider does not support file content. This file message will be skipped."
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
212
src/agent/provider/venice/wire.rs
Normal file
@@ -0,0 +1,212 @@
|
||||
//! Serde structs modeling Venice's `/chat/completions`, `/audio/transcriptions`,
|
||||
//! `/audio/speech`, `/image/generate`, and `/image/edit` wire shapes. Request types are
|
||||
//! `Serialize`-only (we build them, Venice never sends them back); response types are
|
||||
//! `Deserialize`-only. Keeping the split means the untagged request content enum is never on a
|
||||
//! deserialize path, so a surprise response shape can't fail to match it.
|
||||
//!
|
||||
//! Field names match Venice's schema 1:1 (so the config's `model_id` becomes `model` here). Every
|
||||
//! request body is `additionalProperties: false`, so optional knobs carry `skip_serializing_if`
|
||||
//! to omit rather than send `null`. `/audio/speech` and `/image/edit` return raw binary (no
|
||||
//! response struct); only `/image/generate` returns JSON (`GenerateImageResponse`).
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::config::VeniceParameters;
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ChatCompletionRequest {
|
||||
pub model: String,
|
||||
|
||||
pub messages: Vec<ChatMessage>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub temperature: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub max_completion_tokens: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub venice_parameters: Option<VeniceParameters>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ChatMessage {
|
||||
pub role: String,
|
||||
|
||||
pub content: MessageContent,
|
||||
}
|
||||
|
||||
/// A message body is either a bare string or a list of content parts. Venice accepts both; we
|
||||
/// send the parts form only when a message carries an image (baibot keeps text and images in
|
||||
/// separate messages, so a parts list only ever holds images in v1).
|
||||
#[derive(Debug, Serialize)]
|
||||
#[serde(untagged)]
|
||||
pub enum MessageContent {
|
||||
Text(String),
|
||||
Parts(Vec<ContentPart>),
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
#[serde(tag = "type", rename_all = "snake_case")]
|
||||
pub enum ContentPart {
|
||||
ImageUrl { image_url: ImageUrl },
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ImageUrl {
|
||||
/// A `data:<mime>;base64,<data>` URI for inline images.
|
||||
pub url: String,
|
||||
}
|
||||
|
||||
/// Standard OpenAI-shaped chat completion response. We only read `choices[0].message.content`;
|
||||
/// when web search is on, Venice inlines citations as `^n^` superscripts in that content and we
|
||||
/// pass it through untouched.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ChatCompletionResponse {
|
||||
pub choices: Vec<ChatChoice>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ChatChoice {
|
||||
pub message: ResponseMessage,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ResponseMessage {
|
||||
#[serde(default)]
|
||||
pub content: Option<String>,
|
||||
}
|
||||
|
||||
/// `/audio/transcriptions` response. We read `text`; the optional `duration`/`timestamps` the
|
||||
/// API can return are not used in v1.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct TranscriptionResponse {
|
||||
pub text: String,
|
||||
}
|
||||
|
||||
/// `/audio/speech` (`CreateSpeechRequestSchema`) request. `input` and `model` are always sent;
|
||||
/// the rest are omitted when unset. The response is raw binary audio, so there is no response
|
||||
/// struct.
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct SpeechRequest {
|
||||
pub model: String,
|
||||
|
||||
pub input: String,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub voice: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub speed: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub response_format: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub prompt: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub temperature: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub top_p: Option<f32>,
|
||||
}
|
||||
|
||||
/// `/image/generate` (`GenerateImageRequest`) request. `return_binary` is pinned `false` and
|
||||
/// `variants` to `1` by the builder: baibot wants exactly one image returned as base64-in-JSON,
|
||||
/// which `GenerateImageResponse` then decodes. Flipping `return_binary` would make Venice answer
|
||||
/// with raw binary and break that JSON decode, so it is not configurable.
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct GenerateImageRequest {
|
||||
pub model: String,
|
||||
|
||||
pub prompt: String,
|
||||
|
||||
pub return_binary: bool,
|
||||
|
||||
pub variants: u32,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub negative_prompt: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cfg_scale: Option<f32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub steps: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub style_preset: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub seed: Option<i64>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub hide_watermark: Option<bool>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub format: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub width: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub height: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub quality: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub lora_strength: Option<u32>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub embed_exif_metadata: Option<bool>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub enable_web_search: Option<bool>,
|
||||
}
|
||||
|
||||
/// `/image/generate` response when `return_binary` is false: a JSON envelope carrying the images
|
||||
/// as base64 strings. We read `images[0]`; `request`/`timing` and other fields are ignored. `id`
|
||||
/// is telemetry only (logged, never used for correctness), so it is optional: a response that
|
||||
/// carries usable `images` must not fail to deserialize just because the telemetry field drifted.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct GenerateImageResponse {
|
||||
#[serde(default)]
|
||||
pub id: Option<String>,
|
||||
pub images: Vec<String>,
|
||||
}
|
||||
|
||||
/// `/image/edit` (`EditImageRequest`) request. The source `image` is a base64-encoded string
|
||||
/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64-in-JSON, no multipart).
|
||||
/// The response is raw binary, so there is no response struct.
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct EditImageRequest {
|
||||
pub model: String,
|
||||
|
||||
pub prompt: String,
|
||||
|
||||
/// Base64-encoded source image bytes.
|
||||
pub image: String,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub output_format: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub aspect_ratio: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub resolution: Option<String>,
|
||||
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub safe_mode: Option<bool>,
|
||||
}
|
||||
@@ -1,5 +1,3 @@
|
||||
use base64::{engine::general_purpose::STANDARD, Engine as _};
|
||||
|
||||
use crate::{
|
||||
agent::{
|
||||
AgentInstance, AgentPurpose, ControllerTrait, Manager as AgentManager, PublicIdentifier,
|
||||
@@ -140,7 +138,3 @@ async fn get_global_agent_id_for_purpose(
|
||||
.handler
|
||||
.get_by_purpose_with_catch_all_fallback(purpose)
|
||||
}
|
||||
|
||||
pub(crate) fn base64_decode(base64_string: &str) -> Result<Vec<u8>, base64::DecodeError> {
|
||||
STANDARD.decode(base64_string)
|
||||
}
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
use std::fs;
|
||||
use std::sync::Arc;
|
||||
use std::{future::Future, pin::Pin};
|
||||
|
||||
use mxlink::matrix_sdk::media::{MediaFormat, MediaRequest};
|
||||
use mxlink::matrix_sdk::ruma::{
|
||||
events::room::MediaSource, MilliSecondsSinceUnixEpoch, OwnedUserId,
|
||||
};
|
||||
use mxlink::matrix_sdk::Room;
|
||||
use mxlink::matrix_sdk::media::{MediaFormat, MediaRequestParameters};
|
||||
use mxlink::matrix_sdk::ruma::api::client::profile::{AvatarUrl, DisplayName};
|
||||
use mxlink::matrix_sdk::ruma::{
|
||||
MilliSecondsSinceUnixEpoch, OwnedUserId, events::room::MediaSource,
|
||||
};
|
||||
|
||||
use mxlink::{
|
||||
InitConfig, LoginConfig, LoginCredentials, LoginEncryption, MatrixLink, PersistenceConfig,
|
||||
@@ -17,12 +19,13 @@ use mxlink::helpers::account_data_config::{
|
||||
RoomConfigManager as AccountDataRoomConfigManager,
|
||||
};
|
||||
use mxlink::helpers::encryption::Manager as EncryptionManager;
|
||||
use mxlink::mime::Mime;
|
||||
|
||||
use crate::agent::Manager as AgentManager;
|
||||
use crate::entity::catch_up_marker::{
|
||||
CatchUpMarker, CatchUpMarkerManager, DelayedCatchUpMarkerManager,
|
||||
};
|
||||
use crate::entity::cfg::Config;
|
||||
use crate::entity::cfg::{Avatar, Config, ConfigUserAuth};
|
||||
use crate::entity::globalconfig::{GlobalConfig, GlobalConfigurationManager};
|
||||
use crate::entity::roomconfig::{RoomConfig, RoomConfigurationManager};
|
||||
|
||||
@@ -140,6 +143,10 @@ impl Bot {
|
||||
&self.inner.config.command_prefix
|
||||
}
|
||||
|
||||
pub(crate) fn post_join_self_introduction_enabled(&self) -> bool {
|
||||
self.inner.config.room.post_join_self_introduction_enabled
|
||||
}
|
||||
|
||||
pub(crate) fn homeserver_name(&self) -> &str {
|
||||
&self.inner.config.homeserver.server_name
|
||||
}
|
||||
@@ -172,6 +179,24 @@ impl Bot {
|
||||
self.matrix_link().user_id()
|
||||
}
|
||||
|
||||
pub(crate) async fn user_display_name_in_room(&self, room: &Room) -> Option<String> {
|
||||
let bot_display_name = self
|
||||
.room_display_name_fetcher()
|
||||
.own_display_name_in_room(room)
|
||||
.await;
|
||||
|
||||
match bot_display_name {
|
||||
Ok(value) => value,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
?err,
|
||||
"Failed to fetch bot display name. Proceeding without it"
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn reacting(&self) -> super::reacting::Reacting {
|
||||
super::reacting::Reacting::new(self.clone())
|
||||
}
|
||||
@@ -263,24 +288,27 @@ impl Bot {
|
||||
async fn do_prepare_profile(&self) -> anyhow::Result<()> {
|
||||
tracing::debug!("Preparing profile..");
|
||||
|
||||
let desired_display_name = self.inner.config.user.name.clone();
|
||||
|
||||
let account = self.inner.matrix_link.client().account();
|
||||
let media = self.inner.matrix_link.client().media();
|
||||
|
||||
let desired_display_name = self.inner.config.user.name.clone();
|
||||
|
||||
let profile = account
|
||||
.get_profile()
|
||||
.fetch_user_profile()
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed fetching profile: {:?}", e))?;
|
||||
|
||||
let should_update_display_name = match &profile.displayname {
|
||||
let current_display_name = profile.get_static::<DisplayName>()?;
|
||||
let current_avatar_url = profile.get_static::<AvatarUrl>()?;
|
||||
|
||||
let should_update_display_name = match ¤t_display_name {
|
||||
Some(displayname) => displayname != &desired_display_name,
|
||||
None => true,
|
||||
};
|
||||
|
||||
if should_update_display_name {
|
||||
tracing::info!(
|
||||
?profile.displayname,
|
||||
?current_display_name,
|
||||
?desired_display_name,
|
||||
"Updating display name.."
|
||||
);
|
||||
@@ -290,9 +318,36 @@ impl Bot {
|
||||
}
|
||||
}
|
||||
|
||||
let should_update_avatar = match &profile.avatar_url {
|
||||
let desired_avatar: Option<(Vec<u8>, Mime)> = match &self.inner.config.user.avatar {
|
||||
Avatar::Keep => {
|
||||
tracing::info!("Avatar configured to keep current, skipping avatar management");
|
||||
None
|
||||
}
|
||||
Avatar::Default => {
|
||||
tracing::info!("Avatar configured to use default");
|
||||
Some((
|
||||
LOGO_BYTES.to_vec(),
|
||||
LOGO_MIME_TYPE
|
||||
.parse()
|
||||
.expect("Failed parsing mime type for logo"),
|
||||
))
|
||||
}
|
||||
Avatar::Custom(avatar_path) => {
|
||||
tracing::info!(?avatar_path, "Avatar configured to use custom path");
|
||||
let bytes = fs::read(avatar_path).map_err(|e| {
|
||||
anyhow::anyhow!("Failed reading avatar from {:?}: {:?}", avatar_path, e)
|
||||
})?;
|
||||
let mime = mime_guess::from_path(avatar_path).first_or_octet_stream();
|
||||
tracing::debug!(?mime, bytes_len = bytes.len(), "Loaded custom avatar");
|
||||
Some((bytes, mime))
|
||||
}
|
||||
};
|
||||
|
||||
if let Some((desired_bytes, mime_type)) = desired_avatar {
|
||||
let should_update_avatar = match ¤t_avatar_url {
|
||||
Some(avatar_url) => {
|
||||
let request = MediaRequest {
|
||||
tracing::debug!(?avatar_url, "Fetching current avatar to compare");
|
||||
let request = MediaRequestParameters {
|
||||
source: MediaSource::Plain(avatar_url.to_owned()),
|
||||
format: MediaFormat::File,
|
||||
};
|
||||
@@ -302,22 +357,33 @@ impl Bot {
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed fetching existing avatar: {:?}", e))?;
|
||||
|
||||
content.as_slice() != LOGO_BYTES
|
||||
let needs_update = content.as_slice() != desired_bytes;
|
||||
|
||||
tracing::debug!(
|
||||
current_bytes_len = content.len(),
|
||||
desired_bytes_len = desired_bytes.len(),
|
||||
?needs_update,
|
||||
"Compared current and desired avatar"
|
||||
);
|
||||
|
||||
needs_update
|
||||
}
|
||||
None => {
|
||||
tracing::debug!("No current avatar set, will upload");
|
||||
true
|
||||
}
|
||||
None => true,
|
||||
};
|
||||
|
||||
if should_update_avatar {
|
||||
tracing::info!("Updating avatar..");
|
||||
|
||||
let mime_type = LOGO_MIME_TYPE
|
||||
.parse()
|
||||
.expect("Failed parsing mime type for logo");
|
||||
|
||||
account
|
||||
.upload_avatar(&mime_type, LOGO_BYTES.to_vec())
|
||||
.upload_avatar(&mime_type, desired_bytes)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed uploading avatar: {:?}", e))?;
|
||||
tracing::info!("Avatar updated successfully");
|
||||
} else {
|
||||
tracing::debug!("Avatar already up to date, skipping upload");
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
@@ -329,10 +395,22 @@ async fn create_matrix_link(config: &Config) -> anyhow::Result<MatrixLink> {
|
||||
let session_encryption_key = config.persistence.session_encryption_key()?;
|
||||
let db_dir_path: std::path::PathBuf = config.persistence.db_dir_path()?;
|
||||
|
||||
let login_creds = LoginCredentials::UserPassword(
|
||||
config.user.mxid_localpart.to_owned(),
|
||||
config.user.password.to_owned(),
|
||||
);
|
||||
let user_auth = config.user.auth_config(&config.homeserver.server_name)?;
|
||||
|
||||
let login_creds = match user_auth {
|
||||
ConfigUserAuth::UserPassword { username, password } => {
|
||||
LoginCredentials::UserPassword(username, password)
|
||||
}
|
||||
ConfigUserAuth::AccessToken {
|
||||
user_id,
|
||||
device_id,
|
||||
access_token,
|
||||
} => LoginCredentials::AccessToken {
|
||||
user_id,
|
||||
device_id,
|
||||
access_token,
|
||||
},
|
||||
};
|
||||
|
||||
let login_encryption = LoginEncryption::new(
|
||||
config.user.encryption.recovery_passphrase.clone(),
|
||||
|
||||
@@ -5,7 +5,7 @@ use anyhow::anyhow;
|
||||
|
||||
use crate::agent::AgentPurpose;
|
||||
|
||||
pub use crate::entity::cfg::{defaults as cfg_defaults, env as cfg_env, Config};
|
||||
pub use crate::entity::cfg::{Avatar, Config, defaults as cfg_defaults, env as cfg_env};
|
||||
|
||||
pub fn load() -> anyhow::Result<Config> {
|
||||
let config_file_path = env::var(cfg_env::BAIBOT_CONFIG_FILE_PATH)
|
||||
@@ -21,7 +21,7 @@ pub fn load() -> anyhow::Result<Config> {
|
||||
}
|
||||
|
||||
let config_str = std::fs::read_to_string(config_file_path)?;
|
||||
let mut config: Config = serde_yaml::from_str(&config_str)?;
|
||||
let mut config: Config = serde_yaml_ng::from_str(&config_str)?;
|
||||
|
||||
// Allow environment variables to override some configuration keys
|
||||
for (key, value) in env::vars() {
|
||||
@@ -29,12 +29,29 @@ pub fn load() -> anyhow::Result<Config> {
|
||||
cfg_env::BAIBOT_HOMESERVER_SERVER_NAME => config.homeserver.server_name = value,
|
||||
cfg_env::BAIBOT_HOMESERVER_URL => config.homeserver.url = value,
|
||||
cfg_env::BAIBOT_USER_MXID_LOCALPART => config.user.mxid_localpart = value,
|
||||
cfg_env::BAIBOT_USER_PASSWORD => config.user.password = value,
|
||||
cfg_env::BAIBOT_USER_PASSWORD => {
|
||||
config.user.password = optional_non_empty(value);
|
||||
}
|
||||
cfg_env::BAIBOT_USER_ACCESS_TOKEN => {
|
||||
config.user.access_token = optional_non_empty(value);
|
||||
}
|
||||
cfg_env::BAIBOT_USER_DEVICE_ID => {
|
||||
config.user.device_id = optional_non_empty(value);
|
||||
}
|
||||
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE => {
|
||||
config.user.encryption.recovery_passphrase = Some(value);
|
||||
}
|
||||
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED => {
|
||||
config.user.encryption.recovery_reset_allowed = value.parse::<bool>()?;
|
||||
}
|
||||
cfg_env::BAIBOT_USER_NAME => config.user.name = value,
|
||||
cfg_env::BAIBOT_USER_AVATAR => {
|
||||
config.user.avatar = Avatar::from_string(value);
|
||||
}
|
||||
cfg_env::BAIBOT_COMMAND_PREFIX => config.command_prefix = value,
|
||||
cfg_env::BAIBOT_ROOM_POST_JOIN_SELF_INTRODUCTION_ENABLED => {
|
||||
config.room.post_join_self_introduction_enabled = value.parse::<bool>()?;
|
||||
}
|
||||
cfg_env::BAIBOT_LOGGING => {
|
||||
config.logging = value;
|
||||
}
|
||||
@@ -48,6 +65,9 @@ pub fn load() -> anyhow::Result<Config> {
|
||||
cfg_env::BAIBOT_PERSISTENCE_DATA_DIR_PATH => {
|
||||
config.persistence.data_dir_path = Some(value);
|
||||
}
|
||||
cfg_env::BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY => {
|
||||
config.persistence.session_encryption_key = Some(value);
|
||||
}
|
||||
cfg_env::BAIBOT_PERSISTENCE_CONFIG_ENCRYPTION_KEY => {
|
||||
config.persistence.config_encryption_key = Some(value);
|
||||
}
|
||||
@@ -108,3 +128,7 @@ pub fn load() -> anyhow::Result<Config> {
|
||||
|
||||
Ok(config)
|
||||
}
|
||||
|
||||
fn optional_non_empty(value: String) -> Option<String> {
|
||||
if value.is_empty() { None } else { Some(value) }
|
||||
}
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use mxlink::matrix_sdk::{
|
||||
ruma::{
|
||||
api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::OriginalSyncRoomMessageEvent, OwnedEventId,
|
||||
},
|
||||
Room,
|
||||
ruma::{
|
||||
OwnedEventId, api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::OriginalSyncRoomMessageEvent,
|
||||
},
|
||||
};
|
||||
|
||||
use mxlink::{CallbackError, MessageResponseType};
|
||||
@@ -11,7 +11,7 @@ use mxlink::{CallbackError, MessageResponseType};
|
||||
use tracing::Instrument;
|
||||
|
||||
use crate::{
|
||||
conversation::matrix::determine_thread_context_for_room_event,
|
||||
conversation::matrix::determine_interaction_context_for_room_event,
|
||||
entity::{MessageContext, MessagePayload, RoomConfigContext, TriggerEventInfo},
|
||||
};
|
||||
|
||||
@@ -239,8 +239,11 @@ impl Messaging {
|
||||
}
|
||||
};
|
||||
|
||||
let thread_context = determine_thread_context_for_room_event(
|
||||
let bot_display_name = self.bot.user_display_name_in_room(&room).await;
|
||||
|
||||
let interaction_context = determine_interaction_context_for_room_event(
|
||||
self.bot.user_id(),
|
||||
&bot_display_name,
|
||||
&room,
|
||||
&event,
|
||||
&payload,
|
||||
@@ -248,16 +251,18 @@ impl Messaging {
|
||||
)
|
||||
.await;
|
||||
|
||||
let thread_context = match thread_context {
|
||||
let interaction_context = match interaction_context {
|
||||
Ok(value) => value,
|
||||
Err(err) => {
|
||||
tracing::error!(?err, "Failed to determine thread context for event");
|
||||
tracing::error!(?err, "Failed to determine interaction context for event");
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
|
||||
let Some(thread_context) = thread_context else {
|
||||
tracing::debug!("Ignoring message with unknown thread context (likely not a threaded message or a top-level message)");
|
||||
let Some(interaction_context) = interaction_context else {
|
||||
tracing::debug!(
|
||||
"Ignoring message with unknown interaction context (likely not a message for us)"
|
||||
);
|
||||
return Ok(());
|
||||
};
|
||||
|
||||
@@ -276,33 +281,14 @@ impl Messaging {
|
||||
room_config_context,
|
||||
self.bot.admin_pattern_regexes().clone(),
|
||||
trigger_event_info,
|
||||
thread_context.info.clone(),
|
||||
);
|
||||
interaction_context.thread_info.clone(),
|
||||
)
|
||||
.with_bot_display_name(bot_display_name);
|
||||
|
||||
let bot_display_name = self
|
||||
.bot
|
||||
.room_display_name_fetcher()
|
||||
.own_display_name_in_room(message_context.room())
|
||||
.await;
|
||||
|
||||
let bot_display_name = match bot_display_name {
|
||||
Ok(value) => value,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
?err,
|
||||
"Failed to fetch bot display name. Proceeding without it"
|
||||
);
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// The first event in the thread determines which handler processes the current event.
|
||||
let controller_type = crate::controller::determine_controller(
|
||||
self.bot.command_prefix(),
|
||||
&thread_context.first_message,
|
||||
&interaction_context.trigger,
|
||||
&message_context,
|
||||
self.bot.user_id(),
|
||||
&bot_display_name,
|
||||
);
|
||||
|
||||
tracing::info!(?controller_type, "Determined controller");
|
||||
@@ -310,7 +296,7 @@ impl Messaging {
|
||||
let _ = room
|
||||
.send_single_receipt(
|
||||
ReceiptType::Read,
|
||||
thread_context.info.clone().into(),
|
||||
interaction_context.thread_info.clone().into(),
|
||||
event.event_id.clone(),
|
||||
)
|
||||
.await;
|
||||
|
||||