Compare commits

..

82 Commits

Author SHA1 Message Date
Aine
de980f0165 Implement Venice.ai provider 2026-06-21 07:18:37 +03:00
renovate[bot]
cc3888a4cf Update forgejo.ellis.link/continuwuation/continuwuity Docker tag to v0.5.10 2026-06-20 21:49:16 +03:00
renovate[bot]
eb3285f104 Update actions/checkout action to v7 2026-06-19 11:02:19 +03:00
renovate[bot]
0a03dde523 Update Rust crate async-openai to v0.41.1 2026-06-19 11:02:07 +03:00
renovate[bot]
a5b575da68 Update docker.io/ollama/ollama Docker tag to v0.30.10 2026-06-18 09:55:57 +03:00
renovate[bot]
fe7afec920 Update ghcr.io/element-hq/synapse Docker tag to v1.155.0 2026-06-17 06:24:58 +03:00
renovate[bot]
b6641523da Update docker.io/ollama/ollama Docker tag to v0.30.9 2026-06-17 06:19:12 +03:00
renovate[bot]
c579977e59 Update dependency prek to v0.4.5 2026-06-15 16:13:50 +03:00
renovate[bot]
0a5f37c3eb Update docker.io/ollama/ollama Docker tag to v0.30.8 2026-06-14 07:09:07 +03:00
renovate[bot]
58246e4eb3 Update ghcr.io/element-hq/element-web Docker tag to v1.12.21 2026-06-09 22:04:39 +03:00
renovate[bot]
9b767a420e Update Rust crate regex to v1.12.4 2026-06-09 22:04:26 +03:00
renovate[bot]
54c27311c1 Update docker.io/ollama/ollama Docker tag to v0.30.7 2026-06-09 07:50:10 +03:00
renovate[bot]
3d15f11f76 Update docker.io/ollama/ollama Docker tag to v0.30.6 2026-06-05 12:33:56 +03:00
Slavi Pantaleev
ead70d71ff Prepare 1.21.1 release
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 10:07:16 +03:00
Slavi Pantaleev
49bb52a091 Update anthropic-rs fork to drop vulnerable rustls-webpki 0.101
The anthropic git dependency now uses reqwest 0.12 / rustls 0.23, pulling
rustls-webpki 0.103.13 instead of the 0.101.7 that was dragged in via the
old reqwest 0.11. This clears three RUSTSEC/Dependabot advisories:

- GHSA-82j2-j2ch-gfr8 (high): DoS via panic on malformed CRL BIT STRING
- GHSA-xgp8-3hg3-c2mh (low): name constraints accepted for wildcard names
- GHSA-965h-392x-2mh5 (low): name constraints for URI names incorrectly accepted

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 10:06:27 +03:00
Slavi Pantaleev
1fc2f0f65a Prepare 1.21.0 release
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 09:44:25 +03:00
Slavi Pantaleev
f7cb38b620 Fix clippy warnings in tests
- Avoid unwrap_or() on a statically-Some value by using the raw token count
- Use an array literal instead of vec! for the non-allocated test cases

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 09:41:00 +03:00
Slavi Pantaleev
04e7667d11 Default to gpt-image-2 for OpenAI image generation
Make gpt-image-2 the default image-generation model and add it to the
recognized model-id mapping, updating the sample provider configs to match.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 09:39:45 +03:00
Slavi Pantaleev
78cbfda481 Update Rust crate async-openai to 0.41.0
Adapt to upstream API changes:
- Handle the new ImageModel::GptImage2 variant in image-model match arms
- ImageSize dropped Copy (gained an Other(String) variant), so clone it

Supersedes #170.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 09:37:46 +03:00
renovate[bot]
36fd6a4dda Update Rust crate quick_cache to v0.6.23 2026-06-05 09:33:16 +03:00
renovate[bot]
1308196419 Update docker.io/ollama/ollama Docker tag to v0.30.5 2026-06-05 09:33:07 +03:00
renovate[bot]
b3546ebaf1 Update ghcr.io/element-hq/synapse Docker tag to v1.154.0 2026-06-04 18:50:24 +03:00
renovate[bot]
2499e6baca Update Rust crate chrono to v0.4.45 2026-06-04 18:50:01 +03:00
renovate[bot]
871b9d2f3b Update dependency prek to v0.4.4 2026-06-04 14:48:21 +03:00
renovate[bot]
4a36cf9446 Update docker.io/ollama/ollama Docker tag to v0.30.4 2026-06-04 07:13:27 +03:00
renovate[bot]
a2180452c9 Update docker.io/ollama/ollama Docker tag to v0.30.2 2026-06-03 07:39:57 +03:00
Slavi Pantaleev
0cb0fc18ce Prepare 1.20.0 release
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 23:05:29 +03:00
renovate[bot]
78bddac716 Update Rust crate tiktoken-rs to 0.12.* (#162)
Co-authored-by: renovate[bot] <29139614+renovate[bot]@users.noreply.github.com>
2026-06-02 23:03:20 +03:00
Slavi Pantaleev
f7209221f6 Update matrix-sdk to v0.18.0 (and mxlink to v1.15.0)
matrix-sdk 0.18.0 introduces no breaking changes that affect baibot, but
it does require a matching mxlink release: mxlink 1.14.0 pins matrix-sdk
0.17.0, so without bumping mxlink the two matrix-sdk versions conflict.
mxlink 1.15.0 (released alongside this) tracks matrix-sdk 0.18.0, so we
bump the floor to >=1.15.0.

Verified locally: cargo build and the full test suite pass.

Supersedes Renovate PR #161.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 22:47:13 +03:00
renovate[bot]
5f2054e07f Update Rust crate async-openai to v0.40.3 2026-06-02 12:06:50 +03:00
renovate[bot]
8301c5a29f Update docker.io/ollama/ollama Docker tag to v0.30.0 2026-06-02 12:06:30 +03:00
Slavi Pantaleev
b325b6b1a5 Upgrade Rust (1.95.0 -> 1.96.0)
Bumps the base image in Dockerfile/Dockerfile.ci (via Renovate #158) and
keeps rust-toolchain.toml in sync, since rustup honors the toolchain file
inside the container build and would otherwise keep using 1.95.0.

Verified locally under 1.96.0: cargo check, clippy -D warnings, fmt check,
and cargo test --all-features all pass.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-05-29 09:02:05 +03:00
renovate[bot]
0600721663 Update ghcr.io/element-hq/element-web Docker tag to v1.12.20 2026-05-27 20:50:26 +03:00
renovate[bot]
dc17cf7e96 Update ghcr.io/element-hq/element-web Docker tag to v1.12.19 2026-05-27 13:50:35 +03:00
Slavi Pantaleev
d6a8f6ba0b Prepare 1.19.3 release
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-27 10:09:11 +03:00
renovate[bot]
c4c3d71195 Update dependency prek to v0.4.3 2026-05-27 09:56:15 +03:00
renovate[bot]
84ae29d034 Update dependency prek to v0.4.2 2026-05-26 15:14:23 +03:00
renovate[bot]
447f43df7b Update Rust crate async-openai to v0.40.2 2026-05-22 23:11:28 +03:00
renovate[bot]
dabac790ad Update Rust crate async-openai to v0.40.1 2026-05-22 10:19:16 +03:00
renovate[bot]
fa01f012a1 Update Rust crate serde_json to v1.0.150 2026-05-22 10:19:05 +03:00
Slavi Pantaleev
20cb33bc66 Prepare 1.19.2 release
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-21 12:23:20 +03:00
Slavi Pantaleev
2791bdb08b Merge pull request #150 from etkecc/chore/update-dependencies
Update dependencies
2026-05-21 09:39:11 +03:00
Slavi Pantaleev
5dd505202a Update dependencies
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-21 09:33:28 +03:00
Slavi Pantaleev
d61078002c Merge pull request #149 from etkecc/renovate/async-openai-0.x
Update Rust crate async-openai to 0.40.0
2026-05-21 09:26:46 +03:00
renovate[bot]
cf5b346558 Update Rust crate async-openai to 0.40.0 2026-05-21 05:55:20 +00:00
renovate[bot]
ebbb6658e1 Update dependency prek to v0.4.1 2026-05-20 09:13:57 +03:00
renovate[bot]
369dc0c1ba Update ghcr.io/element-hq/synapse Docker tag to v1.153.0 2026-05-19 21:46:36 +03:00
renovate[bot]
3185f44a93 Update Rust crate quick_cache to v0.6.22 2026-05-18 07:32:20 +03:00
renovate[bot]
aa8ccde0ed Update docker.io/postgres Docker tag to v18.4 2026-05-15 12:48:23 +03:00
renovate[bot]
d5df0d7416 Update docker.io/ollama/ollama Docker tag to v0.24.0 2026-05-15 12:48:07 +03:00
renovate[bot]
140ca9ed68 Update dependency prek to v0.4.0 2026-05-14 16:40:26 +03:00
renovate[bot]
89d77d52b6 Update Rust crate async-openai to v0.38.2 2026-05-14 07:32:06 +03:00
renovate[bot]
5092700275 Update docker.io/ollama/ollama Docker tag to v0.23.4 2026-05-14 07:31:22 +03:00
renovate[bot]
10c365124e Update docker.io/ollama/ollama Docker tag to v0.23.3 2026-05-13 07:28:16 +03:00
renovate[bot]
08bdf4f7a2 Update ghcr.io/element-hq/element-web Docker tag to v1.12.18 2026-05-12 20:38:24 +03:00
Slavi Pantaleev
af557a7e45 Fix multi-arch manifest publish: use buildx imagetools
docker/build-push-action now wraps single-platform images in an OCI
image index (to carry provenance attestations), so the per-arch
`*-amd64`/`*-arm64` tags are manifest lists. `docker manifest create`
refuses manifest-list sources ("X is a manifest list"). Switch to
`docker buildx imagetools create`, which flattens index sources
correctly.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-12 09:22:50 +03:00
renovate[bot]
fa7eb11b1d Update Rust crate async-openai to v0.38.1 2026-05-12 02:39:22 +03:00
Slavi Pantaleev
ff1e128f0e Fix versioned image publish under workflow_run trigger
e978d3c switched the publish trigger from `push` to `workflow_run` and
correctly migrated the `type=raw,value=latest` rule to read the upstream
head_branch from `github.event.workflow_run.*` — but left the
`type=semver,pattern={{raw}}` rule unchanged. That rule still reads
`github.ref`, which under workflow_run dispatch is always
`refs/heads/main` (the default branch where the workflow file lives),
not the triggering tag ref. As a result, no semver tag was extracted,
metadata-action produced no tags, and `buildx` failed with
"tag is needed when pushing to registry". The `latest` tag kept
publishing because its rule was migrated; versioned tags (v1.19.0,
v1.19.1) silently stopped publishing.

Pass the upstream head_branch to the semver rule explicitly via `value`,
gated by `enable` so it only fires for v* tags. Mirrors the migration
the raw rule already received.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-09 17:28:54 +03:00
Slavi Pantaleev
c9da927c66 Prepare 1.19.1 release
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-09 16:40:24 +03:00
Slavi Pantaleev
7b16c5c1c3 Update async-openai to v0.38.0
Closes #137. The 0.36 -> 0.38 bump (skipping 0.37) introduces Tower-based
middleware support and a fix to the ReasoningItem `type` field, but
neither affects baibot's call sites — no code adaptation was needed.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-09 16:40:10 +03:00
Slavi Pantaleev
f809bf8d7c Prepare 1.19.0 release
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-09 16:34:49 +03:00
Slavi Pantaleev
44aea427fc Update matrix-sdk to v0.17.0 and Rust toolchain to 1.95.0
matrix-sdk 0.17.0 brings a few breaking API changes that we adapt to:

- Drop the `native-tls` feature on matrix-sdk; it was removed upstream
  in 2026-04 (matrix-sdk now hardwires rustls). The previous workaround
  comment referencing rust-mxlink issue #1 is no longer relevant.
- `Relation::Reply` is now a tuple variant wrapping a `Reply` struct;
  pattern matches updated to bind through `reply.in_reply_to.event_id`.

Also bumps the Rust toolchain from 1.93.0 to 1.95.0 (closes #85, which
proposed the Dockerfile bump in isolation). rustc 1.94+ trips a
query-depth overflow when computing async layouts in the matrix-sdk
timeline future graph; matrix-rust-sdk PR #6489 raises the limit, but
\`recursion_limit\` is per-crate, so we repeat \`#![recursion_limit = "256"]\`
on this crate root.

Bumps the mxlink floor to 1.14.0, which carries the matching adapter.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-09 16:29:41 +03:00
renovate[bot]
8f45a57ced Update docker.io/ollama/ollama Docker tag to v0.23.2 2026-05-09 16:06:25 +03:00
renovate[bot]
6a9100309c Update ghcr.io/element-hq/synapse Docker tag to v1.152.1 2026-05-09 16:06:17 +03:00
renovate[bot]
1953878e6b Update Rust crate tokio to v1.52.3 2026-05-09 08:12:03 +03:00
renovate[bot]
35761a5bf7 Update ghcr.io/element-hq/element-web Docker tag to v1.12.17 2026-05-09 08:11:38 +03:00
renovate[bot]
c7fa0cc0ee Update forgejo.ellis.link/continuwuation/continuwuity Docker tag to v0.5.9 2026-05-09 08:11:25 +03:00
renovate[bot]
87fcf9d019 Update dependency prek to v0.3.13 2026-05-09 08:11:04 +03:00
renovate[bot]
4b52bc906c Update forgejo.ellis.link/continuwuation/continuwuity Docker tag to v0.5.8 2026-04-25 22:16:54 +03:00
renovate[bot]
517a6c5e33 Update docker.io/ollama/ollama Docker tag to v0.21.2 2026-04-24 08:17:02 +03:00
renovate[bot]
636c8a35eb Update Rust crate async-openai to 0.36.0 2026-04-24 06:54:26 +03:00
renovate[bot]
a50de600da Update docker.io/ollama/ollama Docker tag to v0.21.1 2026-04-22 07:15:30 +03:00
Slavi Pantaleev
e978d3cb2f Publish only after successful CI
Run the publish workflow from the workflow_run event so Docker publishing happens only after the CI workflow completes successfully for push events on main or v* tags.

Check out the exact SHA validated by CI and derive Docker metadata from the upstream CI ref, so publishing follows the tested revision instead of the default branch tip.
2026-04-20 22:09:37 +03:00
Slavi Pantaleev
2b1bdbd3d2 Split CI and publish workflows
This supersedes 7d183b9 ("Run CI for pull requests"), which mixed validation and publishing in one workflow and regressed docker-manifest by dropping the package-write permission it needs to publish the manifest.

Split the workflows so CI handles pull requests, branch pushes, tags, and manual runs, while publishing stays focused on Docker delivery with the manifest permission fixed explicitly at the job level.
2026-04-20 22:08:07 +03:00
Slavi Pantaleev
7d183b91d1 Run CI for pull requests 2026-04-20 21:40:03 +03:00
renovate[bot]
420f380417 Update Rust crate async-openai to 0.35.0 (#125)
Co-authored-by: renovate[bot] <29139614+renovate[bot]@users.noreply.github.com>
2026-04-20 21:39:09 +03:00
renovate[bot]
6f541e2361 Update forgejo.ellis.link/continuwuation/continuwuity Docker tag to v0.5.7 2026-04-17 21:52:46 +03:00
renovate[bot]
0737c2761e Update docker.io/ollama/ollama Docker tag to v0.21.0 2026-04-17 10:41:04 +03:00
renovate[bot]
45852a0d53 Update Rust crate tokio to v1.52.1 2026-04-17 10:38:22 +03:00
renovate[bot]
0fcf8d8703 Update Rust crate tokio to 1.52.* 2026-04-15 09:43:04 +03:00
renovate[bot]
41905b006a Update docker.io/ollama/ollama Docker tag to v0.20.7 2026-04-14 07:33:22 +03:00
renovate[bot]
ee3d27701b Update docker.io/ollama/ollama Docker tag to v0.20.6 2026-04-13 08:58:29 +03:00
39 changed files with 2496 additions and 945 deletions

26
.github/workflows/ci.yml vendored Normal file
View File

@@ -0,0 +1,26 @@
name: CI
on:
workflow_dispatch:
pull_request:
branches: [ "main" ]
push:
branches:
- "**"
tags: [ "v*" ]
permissions:
contents: read
pull-requests: read
concurrency:
group: ci-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
test-and-clippy:
name: Unit testing and linting
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: dtolnay/rust-toolchain@1.93.0
- name: Install SQLite3
run: sudo apt-get update && sudo apt-get install -y libsqlite3-dev
- run: cargo test --all-features
- run: cargo clippy

View File

@@ -1,33 +1,31 @@
name: CI (main and tags) name: Publish
on: on:
push: workflow_run:
branches: [ "main" ] workflows: [ "CI" ]
tags: [ "v*" ] types: [ "completed" ]
permissions: permissions:
checks: write contents: read
contents: write
packages: write
pull-requests: read
concurrency: concurrency:
group: ${{ github.workflow }}-${{ github.ref }} group: publish-${{ github.event.workflow_run.id || github.ref }}
cancel-in-progress: false cancel-in-progress: false
jobs: jobs:
test-and-clippy:
name: Unit testing and linting
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v6
- uses: dtolnay/rust-toolchain@1.93.0
- name: Install SQLite3
run: sudo apt-get update && sudo apt-get install -y libsqlite3-dev
- run: cargo test --all-features
- run: cargo clippy
docker-clean-metadata: docker-clean-metadata:
if: |
github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.event == 'push' &&
(
github.event.workflow_run.head_branch == 'main' ||
startsWith(github.event.workflow_run.head_branch || '', 'v')
)
runs-on: ubuntu-latest runs-on: ubuntu-latest
outputs: outputs:
json: ${{ steps.meta.outputs.json }} json: ${{ steps.meta.outputs.json }}
steps: steps:
- name: Checkout
uses: actions/checkout@v7
with:
ref: ${{ github.event.workflow_run.head_sha }}
fetch-depth: 0
- name: Extract metadata (tags, labels) for Docker - name: Extract metadata (tags, labels) for Docker
id: meta id: meta
uses: docker/metadata-action@v6 uses: docker/metadata-action@v6
@@ -35,10 +33,17 @@ jobs:
images: | images: |
ghcr.io/${{ github.repository }} ghcr.io/${{ github.repository }}
tags: | tags: |
type=raw,value=latest,enable=${{ github.ref_name == 'main' }} type=raw,value=latest,enable=${{ github.event.workflow_run.head_branch == 'main' }}
type=semver,pattern={{raw}} type=semver,pattern={{raw}},value=${{ github.event.workflow_run.head_branch }},enable=${{ startsWith(github.event.workflow_run.head_branch || '', 'v') }}
docker-build: docker-build:
if: |
github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.event == 'push' &&
(
github.event.workflow_run.head_branch == 'main' ||
startsWith(github.event.workflow_run.head_branch || '', 'v')
)
permissions: permissions:
contents: read contents: read
packages: write packages: write
@@ -56,7 +61,10 @@ jobs:
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@v6 uses: actions/checkout@v7
with:
ref: ${{ github.event.workflow_run.head_sha }}
fetch-depth: 0
- name: Log in to the GitHub Container registry - name: Log in to the GitHub Container registry
uses: docker/login-action@v4 uses: docker/login-action@v4
with: with:
@@ -68,8 +76,8 @@ jobs:
uses: docker/metadata-action@v6 uses: docker/metadata-action@v6
with: with:
tags: | tags: |
type=raw,value=latest,enable=${{ github.ref_name == 'main' }} type=raw,value=latest,enable=${{ github.event.workflow_run.head_branch == 'main' }}
type=semver,pattern={{raw}} type=semver,pattern={{raw}},value=${{ github.event.workflow_run.head_branch }},enable=${{ startsWith(github.event.workflow_run.head_branch || '', 'v') }}
flavor: | flavor: |
latest=auto latest=auto
suffix=-${{ matrix.arch }},onlatest=true suffix=-${{ matrix.arch }},onlatest=true
@@ -84,6 +92,16 @@ jobs:
labels: ${{ steps.meta.outputs.labels }} labels: ${{ steps.meta.outputs.labels }}
docker-manifest: docker-manifest:
if: |
github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.event == 'push' &&
(
github.event.workflow_run.head_branch == 'main' ||
startsWith(github.event.workflow_run.head_branch || '', 'v')
)
permissions:
contents: read
packages: write
needs: needs:
- docker-build - docker-build
- docker-clean-metadata - docker-clean-metadata
@@ -103,5 +121,4 @@ jobs:
- name: Create and push manifest - name: Create and push manifest
run: | run: |
docker manifest create ${{ matrix.image }} ${{ matrix.image }}-amd64 ${{ matrix.image }}-arm64 docker buildx imagetools create -t ${{ matrix.image }} ${{ matrix.image }}-amd64 ${{ matrix.image }}-arm64
docker manifest push ${{ matrix.image }}

View File

@@ -1,3 +1,63 @@
# (2026-06-21) Version 1.22.0
- (**Feature**) Add a native [Venice](https://venice.ai) provider with [🖌️ image-generation](./docs/features.md#️-image-creation) (incl. editing), [💬 text-generation](./docs/features.md#-text-generation) (incl. vision), [🗣️ text-to-speech](./docs/features.md#️-text-to-speech), [🦻 speech-to-text](./docs/features.md#-speech-to-text), and Venice's native web search via the full `venice_parameters` knob set. Unlike the [OpenAI-compatible](./docs/providers.md#openai-compatible) path (which drops images and can't reach Venice's audio or native image endpoints), it talks to Venice's API directly, using the knob-rich native `/image/generate` and `/image/edit` endpoints. See the [Venice provider docs](./docs/providers.md#venice).
# (2026-06-05) Version 1.21.1
- (**Security**) Update the [anthropic](https://github.com/etkecc/anthropic-rs) dependency to use [reqwest](https://crates.io/crates/reqwest) 0.12 / [rustls](https://crates.io/crates/rustls) 0.23, replacing the vulnerable `rustls-webpki` 0.101 line with 0.103.13. This resolves [`GHSA-82j2-j2ch-gfr8`](https://github.com/advisories/GHSA-82j2-j2ch-gfr8) (high — denial of service via panic on a malformed CRL), [`GHSA-xgp8-3hg3-c2mh`](https://github.com/advisories/GHSA-xgp8-3hg3-c2mh) and [`GHSA-965h-392x-2mh5`](https://github.com/advisories/GHSA-965h-392x-2mh5) (name-constraint validation issues).
# (2026-06-05) Version 1.21.0
- (**Improvement**) Default to OpenAI's `gpt-image-2` model for image generation (in newly-created OpenAI agents and the sample provider configs).
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) from 0.40 to 0.41, which [resynchronizes with the upstream OpenAI API spec](https://github.com/64bit/async-openai/issues/557) after it had drifted out of sync — a mismatch that was already causing some breakage (hopefully now resolved). Adapts to the newly-added `gpt-image-2` image model and an `ImageSize` type change.
- (**Internal Improvement**) Dependency updates.
# (2026-06-02) Version 1.20.0
- (**Internal Improvement**) Update [matrix-sdk](https://crates.io/crates/matrix-sdk) from 0.17 to 0.18 and [mxlink](https://crates.io/crates/mxlink) to 1.15.0.
- (**Internal Improvement**) Update [tiktoken-rs](https://crates.io/crates/tiktoken-rs) to 0.12, backporting OpenAI [tiktoken](https://github.com/openai/tiktoken) 0.13.0 for better alignment with upstream tokenization behavior.
- (**Internal Improvement**) Bump the pinned Rust toolchain from 1.95.0 to 1.96.0 (in `rust-toolchain.toml` and the Docker build images).
- (**Internal Improvement**) Dependency updates.
# (2026-05-27) Version 1.19.3
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) to 0.40.2, pulling in several upstream fixes (streaming HTTP error surfacing, default `ResponseTextParam.format` deserialization, etc.).
- (**Internal Improvement**) Dependency updates.
# (2026-05-21) Version 1.19.2
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) to 0.40.0.
- (**Internal Improvement**) Dependency updates.
# (2026-05-09) Version 1.19.1
- (**Internal Improvement**) Update [async-openai](https://crates.io/crates/async-openai) to 0.38.0.
- (**Internal Improvement**) Dependency updates.
# (2026-05-09) Version 1.19.0
- (**Internal Improvement**) Update [matrix-sdk](https://crates.io/crates/matrix-sdk) from 0.16 to 0.17 and [mxlink](https://crates.io/crates/mxlink) to 1.14.0. matrix-sdk 0.17 dropped its `native-tls` feature and now uses [rustls](https://github.com/rustls/rustls) exclusively as its TLS backend.
- (**Internal Improvement**) Bump the pinned Rust toolchain from 1.93.0 to 1.95.0 (in `rust-toolchain.toml` and the Docker build images).
- (**Internal Improvement**) Dependency updates.
# (2026-04-11) Version 1.18.0 # (2026-04-11) Version 1.18.0
- (**Bugfix**) Fix the bot not sending a welcome message when joining a room on homeservers (like [Continuwuity](https://continuwuity.org/)) that place the join membership event in the sync response's `state` block rather than the `timeline` block, via [mxlink](https://crates.io/crates/mxlink) 1.13.1 - (**Bugfix**) Fix the bot not sending a welcome message when joining a room on homeservers (like [Continuwuity](https://continuwuity.org/)) that place the join membership event in the sync response's `state` block rather than the `timeline` block, via [mxlink](https://crates.io/crates/mxlink) 1.13.1

1421
Cargo.lock generated

File diff suppressed because it is too large Load Diff

View File

@@ -7,7 +7,7 @@ license = "AGPL-3.0-or-later"
readme = "README.md" readme = "README.md"
keywords = ["matrix", "chat", "bot", "AI", "LLM"] keywords = ["matrix", "chat", "bot", "AI", "LLM"]
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"] include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
version = "1.18.0" version = "1.22.0"
edition = "2024" edition = "2024"
[lib] [lib]
@@ -17,24 +17,27 @@ path = "src/lib.rs"
[dependencies] [dependencies]
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" } anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
anyhow = "1.0.*" anyhow = "1.0.*"
async-openai = { version = "0.34.0", features = ["audio", "chat-completion", "image", "responses"] } async-openai = { version = "0.41.0", features = ["audio", "chat-completion", "image", "responses"] }
base64 = "0.22.*" base64 = "0.22.*"
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] } chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it. # We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
# We add the `native-tls` feature, because of https://github.com/etkecc/rust-mxlink/issues/1 matrix-sdk = { version = "0.18.0", default-features = false }
matrix-sdk = { version = "0.16.0", default-features = false, features = ["native-tls"] }
mime_guess = "2.0.*" mime_guess = "2.0.*"
mxidwc = "1.0.*" mxidwc = "1.0.*"
mxlink = ">=1.13.0" mxlink = ">=1.15.0"
etke_openai_api_rust = "0.1.*" etke_openai_api_rust = "0.1.*"
quick_cache = "0.6.*" quick_cache = "0.6.*"
regex = "1.12.*" regex = "1.12.*"
# Direct dep for the native `venice` provider's HTTP client. Pinned to 0.12 (the version
# async-openai 0.41 already resolves) with rustls only and default-features off, so we ride
# the existing reqwest+rustls copy instead of pulling a second TLS stack (native-tls/openssl).
reqwest = { version = "0.12.*", default-features = false, features = ["json", "multipart", "rustls-tls"] }
serde = { version = "1.0.*", features = ["derive"], default-features = false } serde = { version = "1.0.*", features = ["derive"], default-features = false }
serde_json = "1.0.*" serde_json = "1.0.*"
serde_yaml_ng = "0.10.*" serde_yaml_ng = "0.10.*"
tempfile = "3.27.*" tempfile = "3.27.*"
tiktoken-rs = { version = "0.11.*", default-features = false } tiktoken-rs = { version = "0.12.*", default-features = false }
tokio = { version = "1.51.*", features = ["rt", "rt-multi-thread", "macros"] } tokio = { version = "1.52.*", features = ["rt", "rt-multi-thread", "macros"] }
tracing = "0.1.*" tracing = "0.1.*"
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] } tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
url = "2.5.*" url = "2.5.*"

View File

@@ -4,7 +4,7 @@
# # # #
####################################### #######################################
FROM docker.io/rust:1.93.1-slim-trixie AS build FROM docker.io/rust:1.96.0-slim-trixie AS build
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev

View File

@@ -4,7 +4,7 @@
# # # #
####################################### #######################################
FROM docker.io/rust:1.93.1-slim-trixie AS build FROM docker.io/rust:1.96.0-slim-trixie AS build
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev

View File

@@ -13,7 +13,7 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
## 🌟 Features ## 🌟 Features
- 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai) and [☁️ many more](./docs/providers.md#️-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**: - 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai), [Venice](./docs/providers.md#venice) and [☁️ many more](./docs/providers.md#️-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**:
- Supports **different use purposes** (depending on the [☁️ provider](./docs/providers.md) & model): - Supports **different use purposes** (depending on the [☁️ provider](./docs/providers.md) & model):

View File

@@ -56,7 +56,7 @@ Example: `!bai config room text-to-speech set-speed-override 1.5` (this can also
### 👫 Voice override ### 👫 Voice override
The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (usually `onyx` when using [OpenAI](../providers.md#openai)). The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (e.g. `onyx` when using [OpenAI](../providers.md#openai), or `af_sky` when using [Venice](../providers.md#venice)).
Possible values (e.g. `onyx`) depend on the model you're using. For example, for [OpenAI](../providers.md#openai)'s Whisper model, [these voices](https://platform.openai.com/docs/guides/text-to-speech/voice-options) are available. Possible values (e.g. `onyx`) depend on the model you're using. For example, for [OpenAI](../providers.md#openai)'s Whisper model, [these voices](https://platform.openai.com/docs/guides/text-to-speech/voice-options) are available.

View File

@@ -19,6 +19,7 @@ The list of supported providers is below.
- [OpenAI Compatible](#openai-compatible) - [OpenAI Compatible](#openai-compatible)
- [OpenRouter](#openrouter) - [OpenRouter](#openrouter)
- [Together AI](#together-ai) - [Together AI](#together-ai)
- [Venice](#venice)
### How to choose a provider ### How to choose a provider
@@ -171,3 +172,82 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
- create a global agent: `!bai agent create-global together-ai my-together-ai-agent` - create a global agent: `!bai agent create-global together-ai my-together-ai-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/together-ai.yml). 💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/together-ai.yml).
### Venice
[Venice AI](https://venice.ai) runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses, so your conversations don't linger anywhere. It serves both frontier proprietary models and the latest open-source ones.
- 🆔 Identifier: `venice`
- 🔗 Links: [🏠 Home page](https://venice.ai), [👤 Sign up](https://venice.ai), [📋 Models list](https://api.venice.ai/api/v1/models)
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation) (incl. editing, via the native knob-rich `/image/generate` and `/image/edit` endpoints), [💬 text-generation](./features.md#-text-generation) (incl. vision; native web search via the `venice_parameters` config), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local venice my-venice-agent`
- create a global agent: `!bai agent create-global venice my-venice-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/venice.yml).
Unlike the [OpenAI Compatible](#openai-compatible) provider (which can talk to Venice but drops images and can't reach its audio or native image endpoints), this is a first-class Venice integration that exposes Venice's full parameter set. Image generation uses the native `/image/generate` endpoint rather than the OpenAI-compatible `/images/generations` shim, so every Venice-specific knob below is available.
#### Configuration reference
Every parameter below is optional unless marked otherwise. Omitting a knob lets Venice apply its own server-side default; this is **not** the same as setting it to `false`, which actively sends `false`.
**`text_generation.venice_parameters`** — Venice-specific request knobs sent in the `venice_parameters` bag (alongside the standard `model_id`, `prompt`, `temperature`, `max_response_tokens`, and `max_context_tokens` fields). Set any of them to override Venice's behavior. The `Default` column shows the value baibot's sample config ships; a `—` means the knob is left unset, so Venice's own default applies.
| Knob | What it does | Default |
|------|--------------|---------|
| `enable_web_search` | Web search mode: `auto` (model decides), `on` (always), or `off`. | `auto` |
| `enable_web_citations` | Append source citations to web-search answers. | — |
| `enable_web_scraping` | Allow the model to scrape page contents during web search. | — |
| `enable_x_search` | Include X (Twitter) in web search. | — |
| `include_search_results_in_stream` | Stream search results back as they arrive. | — |
| `return_search_results_as_documents` | Return search results as structured documents. | — |
| `include_venice_system_prompt` | Prepend Venice's own system prompt alongside yours. | — |
| `character_slug` | Use a public Venice character by its slug. | — |
| `strip_thinking_response` | Strip `<think></think>` blocks from reasoning models so the user sees only the answer. | `true` |
| `disable_thinking` | Disable the model's reasoning step entirely. | — |
| `enable_e2ee` | Run in end-to-end-encrypted mode rather than the default TEE-only mode. | `false` |
**`text_to_speech`**:
| Knob | What it does | Default |
|------|--------------|---------|
| `model_id` | The Venice TTS model (e.g. `tts-kokoro`, `tts-qwen3-1-7b`, `tts-xai-v1`). | `tts-kokoro` |
| `voice` | The voice to synthesize with. Model-specific (Kokoro: `af_*`/`am_*`/`bf_*`/`bm_*`); a cloned-voice handle (`vv_<id>`) also works. | `af_sky` |
| `response_format` | Audio format: `mp3`, `opus`, `aac`, `flac`, `wav`, or `pcm`. | `mp3` |
| `speed` | Playback speed, `0.25`–`4.0`. | `1.0` |
| `prompt` | A style prompt steering emotion/delivery. Only Qwen 3 TTS honors it. | — |
| `temperature` | Sampling temperature, `0.0`–`2.0`. Only Qwen 3 / Orpheus / Chatterbox HD honor it. | — |
| `top_p` | Nucleus sampling, `0.0`–`1.0`. Only Qwen 3 TTS honors it. | — |
**`image_generation`**:
| Knob | What it does | Default |
|------|--------------|---------|
| `model_id` | The image-generation model. | `chroma` |
| `negative_prompt` | A description of what should **not** appear in the image. | — |
| `cfg_scale` | CFG scale, `0`–`20`. Higher values adhere more closely to the prompt. | — |
| `steps` | Number of inference steps. Model-specific; some models ignore it. | — |
| `style_preset` | A named style to apply (e.g. `3D Model`). | — |
| `seed` | Random seed, `-999999999`–`999999999`. Fix it for reproducible results. | random |
| `safe_mode` | Blur images classified as adult content. | `true` |
| `hide_watermark` | Hide the Venice watermark (may be ignored for some content). | `false` |
| `format` | Output format: `jpeg`, `png`, or `webp`. | `webp` |
| `width` / `height` | Image dimensions in pixels, each `1`–`1280`. | `1024` |
| `aspect_ratio` | Aspect ratio for models that support it (e.g. `1:1`, `16:9`). Alternative to `width`/`height`. | — |
| `resolution` | Resolution tier for models that support it (`1K`, `2K`, `4K`). | — |
| `quality` | Output quality for supported models: `low`, `medium`, `high`. Higher can cost more. | — |
| `lora_strength` | Lora strength, `0`–`100`. Only applies if the model uses additional Loras. | — |
| `embed_exif_metadata` | Embed the generation prompt into the image's EXIF metadata. | `false` |
| `enable_web_search` | Let the model pull the latest info from the web. Model-specific; costs extra credits. | — |
**`image_generation.edit`** — image editing reuses the `image_generation` block; only the model and a few output knobs differ:
| Knob | What it does | Default |
|------|--------------|---------|
| `model_id` | The image-edit model. | `firered-image-edit` |
| `output_format` | Output format: `jpeg`, `png`, or `webp`. When omitted, Venice infers it (PNG at 1K, JPEG at 2K/4K). | inferred |
| `aspect_ratio` | Aspect ratio of the result: `auto`, `1:1`, `3:2`, `16:9`, `21:9`, `9:16`, `2:3`, `3:4`, `4:5` (model-specific). | — |
| `resolution` | Resolution tier: `1K`, `2K`, `4K` (model-specific). | `1K` |
| `safe_mode` | Blur images classified as adult content. | `true` |

View File

@@ -21,7 +21,7 @@ text_to_speech:
speed: 1.0 speed: 1.0
response_format: opus response_format: opus
image_generation: image_generation:
model_id: gpt-image-1.5 model_id: gpt-image-2
style: null style: null
size: null size: null
quality: null quality: null

View File

@@ -0,0 +1,97 @@
base_url: https://api.venice.ai/api/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: kimi-k2-5
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
temperature: 1.0
max_response_tokens: 4096
max_context_tokens: 128000
# Venice-specific request parameters. Only the keys present below are sent to Venice; omit a
# key to fall back to Venice's own default. Omitting a knob is NOT the same as setting it to
# `false` — `false` actively sends `false`.
venice_parameters:
# Web search: "auto" (model decides), "on" (always), or "off".
enable_web_search: "auto"
# Strip <think></think> blocks from reasoning models so the user sees only the answer.
strip_thinking_response: true
# Run in TEE-only mode instead of end-to-end encryption (works across all models).
enable_e2ee: false
# Other available knobs — uncomment to override Venice's default:
# enable_web_citations: true
# enable_web_scraping: true
# include_venice_system_prompt: false
# include_search_results_in_stream: true
# return_search_results_as_documents: true
# enable_x_search: true
# disable_thinking: true
# character_slug: public-character-id
speech_to_text:
model_id: nvidia/parakeet-tdt-0.6b-v3
text_to_speech:
# The Venice TTS model. Others include tts-qwen3-1-7b, tts-xai-v1,
# tts-elevenlabs-turbo-v2-5, tts-minimax-speech-02-hd. See the models list endpoint.
model_id: tts-kokoro
# The voice to synthesize with. Voices are model-specific: Kokoro uses af_*/am_*/bf_*/bm_*
# (e.g. af_sky, am_adam), other models have their own sets. You can also pass a cloned-voice
# handle (vv_<id>) created via Venice's voice-cloning API. An incompatible voice returns an error.
voice: af_sky
# Output audio format: mp3, opus, aac, flac, wav, or pcm. mp3 is the broadest Matrix-client fit.
response_format: mp3
# Other available knobs — uncomment to override Venice's default:
# Playback speed, 0.25–4.0 (1.0 is normal).
# speed: 1.0
# A style prompt steering emotion/delivery (e.g. "Excited and energetic."). Only Qwen 3 TTS uses it.
# prompt: "Calm and warm."
# Sampling temperature, 0.0–2.0 (higher = more varied). Only Qwen 3 / Orpheus / Chatterbox HD use it.
# temperature: 0.9
# Nucleus sampling, 0.0–1.0. Only Qwen 3 TTS uses it.
# top_p: 1.0
image_generation:
# The image-generation model. See the models list endpoint for the full set.
model_id: chroma
# The image-edit model, used when editing an existing image rather than generating a new one.
# Editing shares this same image_generation config block; only the model differs.
edit:
model_id: firered-image-edit
# Other edit knobs — uncomment to override Venice's default:
# Output format: jpeg, png, or webp. When omitted, Venice infers it (PNG at 1K, JPEG at 2K/4K).
# output_format: png
# Aspect ratio of the result: auto, 1:1, 3:2, 16:9, 21:9, 9:16, 2:3, 3:4, 4:5 (model-specific).
# aspect_ratio: auto
# Resolution tier: 1K, 2K, 4K (model-specific). Defaults to 1K.
# resolution: 1K
# Blur images classified as adult content. Defaults to true.
# safe_mode: true
# Other generation knobs — uncomment to override Venice's default. Omitting a knob is NOT the same
# as setting it: an omitted knob lets Venice apply its own default, a set value is sent verbatim.
# A description of what should NOT appear in the image.
# negative_prompt: "blurry, watermark, text"
# CFG scale, 0–20. Higher values make the image adhere more closely to the prompt.
# cfg_scale: 7.5
# Number of inference steps. Model-specific; some models ignore it.
# steps: 8
# A named style to apply (e.g. "3D Model"). See Venice's image-styles reference.
# style_preset: "3D Model"
# Random seed, -999999999–999999999. Fix it for reproducible results; omit for a random seed.
# seed: 123456789
# Blur images classified as adult content. Defaults to true.
# safe_mode: true
# Hide the Venice watermark. Venice may ignore this for certain generated content. Defaults to false.
# hide_watermark: false
# Output format: jpeg, png, or webp. webp is smallest; png is highest-quality. Defaults to webp.
# format: webp
# Image dimensions in pixels, each 1–1280. Default 1024×1024.
# width: 1024
# height: 1024
# Aspect ratio (used by certain models, e.g. Nano Banana): "1:1", "16:9". An alternative to width/height.
# aspect_ratio: "1:1"
# Resolution tier (used by certain models): "1K", "2K", "4K".
# resolution: "1K"
# Output quality for supported models (e.g. GPT Image 2): low, medium, high. Higher can cost more.
# quality: high
# Lora strength, 0–100. Only applies if the model uses additional Loras.
# lora_strength: 50
# Embed the generation prompt into the image's EXIF metadata. Defaults to false.
# embed_exif_metadata: false
# Let the model pull the latest info from the web for the image. Model-specific; costs extra credits.
# enable_web_search: false

View File

@@ -111,7 +111,7 @@ agents:
# speed: 1.0 # speed: 1.0
# response_format: opus # response_format: opus
# image_generation: # image_generation:
# model_id: gpt-image-1.5 # model_id: gpt-image-2
# style: null # style: null
# size: null # size: null
# quality: null # quality: null

View File

@@ -1,6 +1,6 @@
services: services:
continuwuity: continuwuity:
image: forgejo.ellis.link/continuwuation/continuwuity:v0.5.6 image: forgejo.ellis.link/continuwuation/continuwuity:v0.5.10
user: "${UID}:${GID}" user: "${UID}:${GID}"
restart: unless-stopped restart: unless-stopped
cap_drop: cap_drop:

View File

@@ -1,6 +1,6 @@
services: services:
element-web: element-web:
image: ghcr.io/element-hq/element-web:v1.12.15 image: ghcr.io/element-hq/element-web:v1.12.21
user: "${UID}:${GID}" user: "${UID}:${GID}"
restart: unless-stopped restart: unless-stopped
environment: environment:

View File

@@ -1,6 +1,6 @@
services: services:
ollama: ollama:
image: docker.io/ollama/ollama:0.20.5 image: docker.io/ollama/ollama:0.30.10
restart: unless-stopped restart: unless-stopped
ports: ports:
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434" - "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"

View File

@@ -1,6 +1,6 @@
services: services:
postgres: postgres:
image: docker.io/postgres:18.3-alpine image: docker.io/postgres:18.4-alpine
user: ${UID}:${GID} user: ${UID}:${GID}
restart: unless-stopped restart: unless-stopped
environment: environment:
@@ -14,7 +14,7 @@ services:
- /etc/passwd:/etc/passwd:ro - /etc/passwd:/etc/passwd:ro
synapse: synapse:
image: ghcr.io/element-hq/synapse:v1.151.0 image: ghcr.io/element-hq/synapse:v1.155.0
user: "${UID}:${GID}" user: "${UID}:${GID}"
restart: unless-stopped restart: unless-stopped
entrypoint: python entrypoint: python

View File

@@ -1,5 +1,5 @@
[tools] [tools]
prek = "0.3.2" prek = "0.4.5"
[settings] [settings]
# Disable automatic trust prompts - we trust this config # Disable automatic trust prompts - we trust this config

View File

@@ -1,4 +1,4 @@
[toolchain] [toolchain]
channel = "1.93.0" channel = "1.96.0"
components = ["rustfmt", "clippy"] components = ["rustfmt", "clippy"]
profile = "default" profile = "default"

View File

@@ -109,6 +109,9 @@ fn create_controller_from_provider_and_json_value_config(
AgentProvider::TogetherAI => { AgentProvider::TogetherAI => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config) provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
} }
AgentProvider::Venice => {
provider::venice::create_controller_from_yaml_value_config(agent_id, config)
}
} }
} }
@@ -150,5 +153,9 @@ pub fn default_config_for_provider(provider: &AgentProvider) -> serde_yaml_ng::V
let config = super::provider::togetherai::default_config(); let config = super::provider::togetherai::default_config();
serde_yaml_ng::to_value(config).expect("Failed to serialize config") serde_yaml_ng::to_value(config).expect("Failed to serialize config")
} }
AgentProvider::Venice => {
let config = super::provider::venice::default_config();
serde_yaml_ng::to_value(config).expect("Failed to serialize config")
}
} }
} }

View File

@@ -61,6 +61,7 @@ pub enum ControllerType {
OpenAI(Box<super::openai::Controller>), OpenAI(Box<super::openai::Controller>),
OpenAICompat(Box<super::openai_compat::Controller>), OpenAICompat(Box<super::openai_compat::Controller>),
Anthropic(Box<super::anthropic::Controller>), Anthropic(Box<super::anthropic::Controller>),
Venice(Box<super::venice::Controller>),
} }
impl ControllerTrait for ControllerType { impl ControllerTrait for ControllerType {
@@ -69,6 +70,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.supports_purpose(purpose), ControllerType::OpenAI(controller) => controller.supports_purpose(purpose),
ControllerType::OpenAICompat(controller) => controller.supports_purpose(purpose), ControllerType::OpenAICompat(controller) => controller.supports_purpose(purpose),
ControllerType::Anthropic(controller) => controller.supports_purpose(purpose), ControllerType::Anthropic(controller) => controller.supports_purpose(purpose),
ControllerType::Venice(controller) => controller.supports_purpose(purpose),
} }
} }
@@ -77,6 +79,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.text_generation_model_id(), ControllerType::OpenAI(controller) => controller.text_generation_model_id(),
ControllerType::OpenAICompat(controller) => controller.text_generation_model_id(), ControllerType::OpenAICompat(controller) => controller.text_generation_model_id(),
ControllerType::Anthropic(controller) => controller.text_generation_model_id(), ControllerType::Anthropic(controller) => controller.text_generation_model_id(),
ControllerType::Venice(controller) => controller.text_generation_model_id(),
} }
} }
@@ -85,6 +88,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.text_generation_prompt(), ControllerType::OpenAI(controller) => controller.text_generation_prompt(),
ControllerType::OpenAICompat(controller) => controller.text_generation_prompt(), ControllerType::OpenAICompat(controller) => controller.text_generation_prompt(),
ControllerType::Anthropic(controller) => controller.text_generation_prompt(), ControllerType::Anthropic(controller) => controller.text_generation_prompt(),
ControllerType::Venice(controller) => controller.text_generation_prompt(),
} }
} }
@@ -93,6 +97,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.text_to_speech_voice(), ControllerType::OpenAI(controller) => controller.text_to_speech_voice(),
ControllerType::OpenAICompat(controller) => controller.text_to_speech_voice(), ControllerType::OpenAICompat(controller) => controller.text_to_speech_voice(),
ControllerType::Anthropic(controller) => controller.text_to_speech_voice(), ControllerType::Anthropic(controller) => controller.text_to_speech_voice(),
ControllerType::Venice(controller) => controller.text_to_speech_voice(),
} }
} }
@@ -101,6 +106,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.text_to_speech_speed(), ControllerType::OpenAI(controller) => controller.text_to_speech_speed(),
ControllerType::OpenAICompat(controller) => controller.text_to_speech_speed(), ControllerType::OpenAICompat(controller) => controller.text_to_speech_speed(),
ControllerType::Anthropic(controller) => controller.text_to_speech_speed(), ControllerType::Anthropic(controller) => controller.text_to_speech_speed(),
ControllerType::Venice(controller) => controller.text_to_speech_speed(),
} }
} }
@@ -109,6 +115,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.text_generation_temperature(), ControllerType::OpenAI(controller) => controller.text_generation_temperature(),
ControllerType::OpenAICompat(controller) => controller.text_generation_temperature(), ControllerType::OpenAICompat(controller) => controller.text_generation_temperature(),
ControllerType::Anthropic(controller) => controller.text_generation_temperature(), ControllerType::Anthropic(controller) => controller.text_generation_temperature(),
ControllerType::Venice(controller) => controller.text_generation_temperature(),
} }
} }
@@ -117,6 +124,7 @@ impl ControllerTrait for ControllerType {
ControllerType::OpenAI(controller) => controller.ping().await, ControllerType::OpenAI(controller) => controller.ping().await,
ControllerType::OpenAICompat(controller) => controller.ping().await, ControllerType::OpenAICompat(controller) => controller.ping().await,
ControllerType::Anthropic(controller) => controller.ping().await, ControllerType::Anthropic(controller) => controller.ping().await,
ControllerType::Venice(controller) => controller.ping().await,
} }
} }
@@ -135,6 +143,9 @@ impl ControllerTrait for ControllerType {
ControllerType::Anthropic(controller) => { ControllerType::Anthropic(controller) => {
controller.generate_text(conversation, params).await controller.generate_text(conversation, params).await
} }
ControllerType::Venice(controller) => {
controller.generate_text(conversation, params).await
}
} }
} }
@@ -154,6 +165,9 @@ impl ControllerTrait for ControllerType {
ControllerType::Anthropic(controller) => { ControllerType::Anthropic(controller) => {
controller.speech_to_text(mime_type, media, params).await controller.speech_to_text(mime_type, media, params).await
} }
ControllerType::Venice(controller) => {
controller.speech_to_text(mime_type, media, params).await
}
} }
} }
@@ -170,6 +184,9 @@ impl ControllerTrait for ControllerType {
ControllerType::Anthropic(controller) => { ControllerType::Anthropic(controller) => {
controller.generate_image(prompt, params).await controller.generate_image(prompt, params).await
} }
ControllerType::Venice(controller) => {
controller.generate_image(prompt, params).await
}
} }
} }
@@ -189,6 +206,9 @@ impl ControllerTrait for ControllerType {
ControllerType::Anthropic(controller) => { ControllerType::Anthropic(controller) => {
controller.create_image_edit(prompt, images, params).await controller.create_image_edit(prompt, images, params).await
} }
ControllerType::Venice(controller) => {
controller.create_image_edit(prompt, images, params).await
}
} }
} }
@@ -203,6 +223,7 @@ impl ControllerTrait for ControllerType {
controller.text_to_speech(text, params).await controller.text_to_speech(text, params).await
} }
ControllerType::Anthropic(controller) => controller.text_to_speech(text, params).await, ControllerType::Anthropic(controller) => controller.text_to_speech(text, params).await,
ControllerType::Venice(controller) => controller.text_to_speech(text, params).await,
} }
} }
} }

View File

@@ -11,6 +11,7 @@ pub enum AgentProvider {
OpenAICompat, OpenAICompat,
OpenRouter, OpenRouter,
TogetherAI, TogetherAI,
Venice,
} }
impl AgentProvider { impl AgentProvider {
@@ -25,6 +26,7 @@ impl AgentProvider {
&Self::OpenAICompat, &Self::OpenAICompat,
&Self::OpenRouter, &Self::OpenRouter,
&Self::TogetherAI, &Self::TogetherAI,
&Self::Venice,
] ]
} }
@@ -39,6 +41,7 @@ impl AgentProvider {
Self::OpenAICompat => "openai-compatible", Self::OpenAICompat => "openai-compatible",
Self::OpenRouter => "openrouter", Self::OpenRouter => "openrouter",
Self::TogetherAI => "together-ai", Self::TogetherAI => "together-ai",
Self::Venice => "venice",
} }
} }
@@ -53,6 +56,7 @@ impl AgentProvider {
"openai-compatible" => Ok(Self::OpenAICompat), "openai-compatible" => Ok(Self::OpenAICompat),
"openrouter" => Ok(Self::OpenRouter), "openrouter" => Ok(Self::OpenRouter),
"together-ai" => Ok(Self::TogetherAI), "together-ai" => Ok(Self::TogetherAI),
"venice" => Ok(Self::Venice),
_ => Err("Unexpected string value"), _ => Err("Unexpected string value"),
} }
} }
@@ -181,6 +185,25 @@ impl AgentProvider {
text_generation_supports_vision: false, text_generation_supports_vision: false,
text_generation_supports_tools: false, text_generation_supports_tools: false,
}, },
Self::Venice => AgentProviderInfo {
id: Self::Venice.to_static_str(),
name: "Venice",
description: "Venice AI runs inference on Venice-controlled GPUs or zero-data-retention partner infrastructure and stores no prompts or responses. It serves frontier proprietary and open-source models with text-generation (including vision), speech-to-text, text-to-speech, native image generation and editing, and native web search.",
homepage_url: Some("https://venice.ai"),
wiki_url: None,
sign_up_url: Some("https://venice.ai"),
models_list_url: Some("https://api.venice.ai/api/v1/models"),
supported_purposes: vec![
AgentPurpose::ImageGeneration,
AgentPurpose::TextGeneration,
AgentPurpose::TextToSpeech,
AgentPurpose::SpeechToText,
],
text_generation_supports_vision: true,
// Venice does native web search via `venice_parameters`, NOT baibot's built-in
// tools mechanism (the OpenAI web_search/code_interpreter block), so this is false.
text_generation_supports_tools: false,
},
} }
} }
} }

View File

@@ -10,6 +10,7 @@ pub mod openai;
pub mod openai_compat; pub mod openai_compat;
pub(super) mod openrouter; pub(super) mod openrouter;
pub(super) mod togetherai; pub(super) mod togetherai;
pub mod venice;
fn default_temperature() -> f32 { fn default_temperature() -> f32 {
1.0 1.0

View File

@@ -1,6 +1,6 @@
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use super::OPENAI_IMAGE_MODEL_GPT_IMAGE_1_DOT_5; use super::OPENAI_IMAGE_MODEL_GPT_IMAGE_2;
use crate::agent::{default_prompt, provider::ConfigTrait}; use crate::agent::{default_prompt, provider::ConfigTrait};
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]
@@ -175,7 +175,7 @@ pub struct ImageGenerationConfig {
impl Default for ImageGenerationConfig { impl Default for ImageGenerationConfig {
fn default() -> Self { fn default() -> Self {
Self { Self {
model_id: OPENAI_IMAGE_MODEL_GPT_IMAGE_1_DOT_5.to_owned(), model_id: OPENAI_IMAGE_MODEL_GPT_IMAGE_2.to_owned(),
style: default_image_style(), style: default_image_style(),
size: default_image_size(), size: default_image_size(),
quality: default_image_quality(), quality: default_image_quality(),
@@ -193,6 +193,7 @@ impl ImageGenerationConfig {
"gpt-image-1" => Ok(async_openai::types::images::ImageModel::GptImage1), "gpt-image-1" => Ok(async_openai::types::images::ImageModel::GptImage1),
"gpt-image-1.5" => Ok(async_openai::types::images::ImageModel::GptImage1dot5), "gpt-image-1.5" => Ok(async_openai::types::images::ImageModel::GptImage1dot5),
"gpt-image-1-mini" => Ok(async_openai::types::images::ImageModel::GptImage1Mini), "gpt-image-1-mini" => Ok(async_openai::types::images::ImageModel::GptImage1Mini),
"gpt-image-2" => Ok(async_openai::types::images::ImageModel::GptImage2),
other => Ok(async_openai::types::images::ImageModel::Other( other => Ok(async_openai::types::images::ImageModel::Other(
other.to_owned(), other.to_owned(),
)), )),

View File

@@ -271,6 +271,7 @@ impl ControllerTrait for Controller {
ImageModel::GptImage1 => ImageModel::GptImage1Mini, ImageModel::GptImage1 => ImageModel::GptImage1Mini,
ImageModel::GptImage1dot5 => ImageModel::GptImage1Mini, ImageModel::GptImage1dot5 => ImageModel::GptImage1Mini,
ImageModel::GptImage1Mini => ImageModel::GptImage1Mini, ImageModel::GptImage1Mini => ImageModel::GptImage1Mini,
ImageModel::GptImage2 => ImageModel::GptImage1Mini,
ImageModel::Other(_) => ImageModel::DallE2, ImageModel::Other(_) => ImageModel::DallE2,
} }
} else { } else {
@@ -310,7 +311,7 @@ impl ControllerTrait for Controller {
let size = if params.smallest_size_possible { let size = if params.smallest_size_possible {
Some(get_sticker_size(&model)) Some(get_sticker_size(&model))
} else { } else {
image_generation_config.size image_generation_config.size.clone()
}; };
let response_format = match model.clone() { let response_format = match model.clone() {
@@ -321,6 +322,7 @@ impl ControllerTrait for Controller {
ImageModel::GptImage1 => None, ImageModel::GptImage1 => None,
ImageModel::GptImage1Mini => None, ImageModel::GptImage1Mini => None,
ImageModel::GptImage1dot5 => None, ImageModel::GptImage1dot5 => None,
ImageModel::GptImage2 => None,
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json), ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
}; };
@@ -430,6 +432,7 @@ impl ControllerTrait for Controller {
ImageModel::GptImage1 => None, ImageModel::GptImage1 => None,
ImageModel::GptImage1Mini => None, ImageModel::GptImage1Mini => None,
ImageModel::GptImage1dot5 => None, ImageModel::GptImage1dot5 => None,
ImageModel::GptImage2 => None,
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json), ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
}; };
@@ -647,6 +650,7 @@ fn get_sticker_size(model: &ImageModel) -> async_openai::types::images::ImageSiz
ImageModel::GptImage1 => ImageSize::S1024x1024, ImageModel::GptImage1 => ImageSize::S1024x1024,
ImageModel::GptImage1Mini => ImageSize::S1024x1024, ImageModel::GptImage1Mini => ImageSize::S1024x1024,
ImageModel::GptImage1dot5 => ImageSize::S1024x1024, ImageModel::GptImage1dot5 => ImageSize::S1024x1024,
ImageModel::GptImage2 => ImageSize::S1024x1024,
ImageModel::Other(_) => ImageSize::S1024x1024, ImageModel::Other(_) => ImageSize::S1024x1024,
} }
} }

View File

@@ -16,7 +16,7 @@ use super::super::AgentInstantiationResult;
use super::ConfigTrait; use super::ConfigTrait;
use super::controller::ControllerType; use super::controller::ControllerType;
pub const OPENAI_IMAGE_MODEL_GPT_IMAGE_1_DOT_5: &str = "gpt-image-1.5"; pub const OPENAI_IMAGE_MODEL_GPT_IMAGE_2: &str = "gpt-image-2";
pub fn create_controller_from_yaml_value_config( pub fn create_controller_from_yaml_value_config(
agent_id: &str, agent_id: &str,

View File

@@ -0,0 +1,156 @@
use crate::agent::AgentPurpose;
use crate::agent::provider::entity::{TextToSpeechParams, TextToSpeechResult};
use crate::agent::provider::{SpeechToTextParams, SpeechToTextResult};
use crate::strings;
use super::config::Config;
use super::wire::{SpeechRequest, TranscriptionResponse};
pub async fn speech_to_text(
config: &Config,
http: &reqwest::Client,
mime_type: &mxlink::mime::Mime,
media: Vec<u8>,
params: SpeechToTextParams,
) -> anyhow::Result<SpeechToTextResult> {
let Some(speech_to_text_config) = &config.speech_to_text else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::SpeechToText
),
));
};
// Unlike the openai_compat path (which writes the audio to a temp file because its library
// can't take bytes), reqwest's multipart takes the bytes directly.
let part = reqwest::multipart::Part::bytes(media)
.file_name("audio")
.mime_str(mime_type.as_ref())?;
let mut form = reqwest::multipart::Form::new()
.part("file", part)
.text("model", speech_to_text_config.model_id.clone())
.text("response_format", "json");
if let Some(language) = &params.language_override {
form = form.text("language", language.clone());
}
let url = format!(
"{}/audio/transcriptions",
config.base_url.trim_end_matches('/')
);
tracing::trace!(
model_id = speech_to_text_config.model_id,
language = ?params.language_override,
"Sending Venice audio transcription API request"
);
let response = http
.post(&url)
.bearer_auth(&config.api_key)
.multipart(form)
.send()
.await?;
let status = response.status();
if !status.is_success() {
// Body to the server log only, not into the returned error (which reaches the Matrix room).
let body = response.text().await.unwrap_or_default();
tracing::warn!(%status, body, "Venice audio transcription request failed");
return Err(anyhow::anyhow!(
"Venice audio transcription request failed with status {status}"
));
}
let response: TranscriptionResponse = response.json().await?;
Ok(SpeechToTextResult {
text: response.text,
})
}
pub async fn text_to_speech(
config: &Config,
http: &reqwest::Client,
input: &str,
params: TextToSpeechParams,
) -> anyhow::Result<TextToSpeechResult> {
let Some(text_to_speech_config) = &config.text_to_speech else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextToSpeech
),
));
};
// Per-call overrides win over the configured defaults.
let voice = params
.voice_override
.or_else(|| text_to_speech_config.voice.clone());
let speed = params.speed_override.or(text_to_speech_config.speed);
let response_format = text_to_speech_config.response_format.clone();
let mime_type = response_format_to_mime_type(response_format.as_deref());
let request = SpeechRequest {
model: text_to_speech_config.model_id.clone(),
input: input.to_owned(),
voice,
speed,
response_format,
prompt: text_to_speech_config.prompt.clone(),
temperature: text_to_speech_config.temperature,
top_p: text_to_speech_config.top_p,
};
let url = format!("{}/audio/speech", config.base_url.trim_end_matches('/'));
tracing::trace!(
model_id = text_to_speech_config.model_id,
voice = ?request.voice,
"Sending Venice text-to-speech API request"
);
let response = http
.post(&url)
.bearer_auth(&config.api_key)
.json(&request)
.send()
.await?;
let status = response.status();
if !status.is_success() {
// Body to the server log only, not into the returned error (which reaches the Matrix room).
let body = response.text().await.unwrap_or_default();
tracing::warn!(%status, body, "Venice text-to-speech request failed");
return Err(anyhow::anyhow!(
"Venice text-to-speech request failed with status {status}"
));
}
// The speech endpoint answers with raw binary audio; read the body directly.
let bytes = response.bytes().await?.to_vec();
Ok(TextToSpeechResult { bytes, mime_type })
}
/// Map a Venice TTS `response_format` to its MIME type. Defaults to `audio/mpeg` (the
/// IANA-registered MP3 type, RFC 3003) when the format is unset, matching Venice's own `mp3`
/// default. This deliberately uses `audio/mpeg` rather than the `audio/mp3` alias the openai
/// provider emits; baibot's downstream audio-filename mapping treats both as `.mp3`.
fn response_format_to_mime_type(response_format: Option<&str>) -> mxlink::mime::Mime {
let raw = match response_format.unwrap_or("mp3") {
"mp3" => "audio/mpeg",
"opus" => "audio/ogg",
"aac" => "audio/aac",
"flac" => "audio/flac",
"wav" => "audio/wav",
"pcm" => "audio/L8",
_ => "audio/mpeg",
};
raw.parse()
.unwrap_or(mxlink::mime::APPLICATION_OCTET_STREAM)
}

View File

@@ -0,0 +1,122 @@
use crate::agent::AgentPurpose;
use crate::agent::provider::entity::{TextGenerationParams, TextGenerationResult};
use crate::conversation::llm::{
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
};
use crate::strings;
use super::config::Config;
use super::utils::convert_llm_messages_to_venice;
use super::wire::{ChatCompletionRequest, ChatCompletionResponse};
pub async fn generate_text(
config: &Config,
http: &reqwest::Client,
conversation: LLMConversation,
params: TextGenerationParams,
) -> anyhow::Result<TextGenerationResult> {
let Some(text_generation_config) = &config.text_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextGeneration
),
));
};
let prompt_text = params.prompt_variables.format(
params
.prompt_override
.unwrap_or(text_generation_config.prompt.clone().unwrap_or_default())
.trim(),
);
let prompt_message = if prompt_text.is_empty() {
None
} else {
Some(LLMMessage {
author: LLMAuthor::Prompt,
sender_id: None,
content: LLMMessageContent::Text(prompt_text),
timestamp: chrono::Utc::now(),
})
};
let mut conversation_messages = conversation.messages;
if params.context_management_enabled {
conversation_messages = shorten_messages_list_to_context_size(
&text_generation_config.model_id,
&prompt_message,
conversation_messages,
text_generation_config.max_response_tokens,
text_generation_config.max_context_tokens,
);
}
if let Some(prompt_message) = prompt_message {
conversation_messages.insert(0, prompt_message);
}
let messages = convert_llm_messages_to_venice(conversation_messages);
let temperature = params
.temperature_override
.unwrap_or(text_generation_config.temperature);
let request = ChatCompletionRequest {
model: text_generation_config.model_id.clone(),
messages,
temperature: Some(temperature),
// Web search rides entirely inside `venice_parameters`; there is no `tools` array here.
// `max_tokens` is deprecated on Venice in favor of `max_completion_tokens`.
max_completion_tokens: text_generation_config.max_response_tokens,
venice_parameters: text_generation_config.venice_parameters.clone(),
};
let url = format!(
"{}/chat/completions",
config.base_url.trim_end_matches('/')
);
tracing::trace!(
model = text_generation_config.model_id,
messages_count = request.messages.len(),
"Sending Venice chat completion API request"
);
let response = http
.post(&url)
.bearer_auth(&config.api_key)
.json(&request)
.send()
.await?;
let status = response.status();
if !status.is_success() {
// Log the body server-side for debugging (Venice explains a rejected strict body there),
// but keep it OUT of the returned error: that error surfaces in the Matrix room, and the
// body can carry account / rate-limit details that shouldn't reach room members.
let body = response.text().await.unwrap_or_default();
tracing::warn!(%status, body, "Venice chat completion request failed");
return Err(anyhow::anyhow!(
"Venice chat completion request failed with status {status}"
));
}
let response: ChatCompletionResponse = response.json().await?;
let Some(choice) = response.choices.into_iter().next() else {
return Err(anyhow::anyhow!(
"No choices were returned from the Venice chat completion API"
));
};
let Some(content) = choice.message.content else {
return Err(anyhow::anyhow!(
"No message content was returned from the Venice chat completion API"
));
};
Ok(TextGenerationResult { text: content })
}

View File

@@ -0,0 +1,355 @@
use serde::{Deserialize, Serialize};
use crate::agent::{default_prompt, provider::ConfigTrait};
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Config {
pub base_url: String,
pub api_key: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub text_generation: Option<TextGenerationConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub speech_to_text: Option<SpeechToTextConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub text_to_speech: Option<TextToSpeechConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub image_generation: Option<ImageGenerationConfig>,
}
impl Default for Config {
fn default() -> Self {
Self {
base_url: "https://api.venice.ai/api/v1".to_owned(),
api_key: "YOUR_API_KEY_HERE".to_owned(),
text_generation: Some(TextGenerationConfig::default()),
speech_to_text: Some(SpeechToTextConfig::default()),
text_to_speech: Some(TextToSpeechConfig::default()),
image_generation: Some(ImageGenerationConfig::default()),
}
}
}
impl ConfigTrait for Config {
fn validate(&self) -> Result<(), String> {
if self.base_url.is_empty() {
return Err("The base URL must not be empty.".to_owned());
}
Ok(())
}
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextGenerationConfig {
#[serde(default = "default_text_model_id")]
pub model_id: String,
#[serde(default)]
pub prompt: Option<String>,
#[serde(default = "super::super::default_temperature")]
pub temperature: f32,
#[serde(default)]
pub max_response_tokens: Option<u32>,
#[serde(default)]
pub max_context_tokens: u32,
/// Venice-specific request knobs, serialized 1:1 into the `venice_parameters` bag on the
/// wire. Any unset field is omitted, so Venice applies its own server-side default.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub venice_parameters: Option<VeniceParameters>,
}
impl Default for TextGenerationConfig {
fn default() -> Self {
Self {
model_id: default_text_model_id(),
prompt: Some(default_prompt().to_owned()),
temperature: super::super::default_temperature(),
// Reserved output budget: sent as the response cap AND subtracted from the context
// window when trimming history. Mirrors the openai_compat sibling's default.
max_response_tokens: Some(4096),
// Matches Venice's own `availableContextTokens` (131072) and the non-OpenAI sibling
// providers (ollama/localai/mistral all default to 128_000).
max_context_tokens: 128_000,
// A usable starting point, not an everything-set dump: only these three are sent;
// every other knob stays None so Venice applies its own default (omitting != false).
venice_parameters: Some(VeniceParameters {
enable_web_search: Some(WebSearchMode::Auto),
strip_thinking_response: Some(true),
enable_e2ee: Some(false),
..Default::default()
}),
}
}
}
fn default_text_model_id() -> String {
"kimi-k2-5".to_owned()
}
/// The full `venice_parameters` knob set, mirroring Venice's `ChatCompletionRequest`
/// schema field-for-field. Every field is optional with `skip_serializing_if`, so the
/// request never carries a knob the user didn't set (the body is `additionalProperties: false`,
/// and an unset knob simply omits rather than sending `null`).
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
pub struct VeniceParameters {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub enable_web_search: Option<WebSearchMode>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub enable_web_citations: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub enable_web_scraping: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub include_venice_system_prompt: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub include_search_results_in_stream: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub return_search_results_as_documents: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub enable_x_search: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub enable_e2ee: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub character_slug: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub strip_thinking_response: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub disable_thinking: Option<bool>,
}
#[derive(Debug, Clone, Copy, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum WebSearchMode {
Auto,
On,
Off,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct SpeechToTextConfig {
#[serde(default = "default_speech_to_text_model_id")]
pub model_id: String,
}
impl Default for SpeechToTextConfig {
fn default() -> Self {
Self {
model_id: default_speech_to_text_model_id(),
}
}
}
fn default_speech_to_text_model_id() -> String {
"nvidia/parakeet-tdt-0.6b-v3".to_owned()
}
/// `/audio/speech` (`CreateSpeechRequestSchema`) request knobs. Only `model_id` is required on
/// the wire; everything else is optional with `skip_serializing_if` so an unset knob is omitted
/// rather than sent as `null` (the body is `additionalProperties: false`). `voice` is a free
/// `Option<String>`, not a closed enum: Venice's voice set spans dozens of model-specific names
/// plus arbitrary cloned-voice handles (`vv_<id>`), so an enum would reject valid handles.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextToSpeechConfig {
#[serde(default = "default_text_to_speech_model_id")]
pub model_id: String,
#[serde(
default = "default_text_to_speech_voice",
skip_serializing_if = "Option::is_none"
)]
pub voice: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub speed: Option<f32>,
#[serde(
default = "default_text_to_speech_response_format",
skip_serializing_if = "Option::is_none"
)]
pub response_format: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub prompt: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub temperature: Option<f32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub top_p: Option<f32>,
}
impl Default for TextToSpeechConfig {
fn default() -> Self {
Self {
model_id: default_text_to_speech_model_id(),
voice: default_text_to_speech_voice(),
speed: None,
response_format: default_text_to_speech_response_format(),
prompt: None,
temperature: None,
top_p: None,
}
}
}
fn default_text_to_speech_model_id() -> String {
"tts-kokoro".to_owned()
}
fn default_text_to_speech_voice() -> Option<String> {
Some("af_sky".to_owned())
}
fn default_text_to_speech_response_format() -> Option<String> {
Some("mp3".to_owned())
}
/// `/image/generate` (`GenerateImageRequest`) request knobs, mirroring Venice's schema
/// field-for-field. Only `model_id` is required; every other knob is optional with
/// `skip_serializing_if` so unset knobs are omitted (the body is `additionalProperties: false`).
/// The full knob set is deliberate: the native `/image/generate` endpoint is the flagship's
/// reason to exist over the knob-dropping OpenAI-compat path, so the knobs ARE the feature.
/// The deprecated `inpaint` knob is intentionally absent.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ImageGenerationConfig {
#[serde(default = "default_image_generation_model_id")]
pub model_id: String,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub negative_prompt: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub cfg_scale: Option<f32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub steps: Option<u32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub style_preset: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub seed: Option<i64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub safe_mode: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub hide_watermark: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub format: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub width: Option<u32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub height: Option<u32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub aspect_ratio: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub resolution: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub quality: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub lora_strength: Option<u32>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub embed_exif_metadata: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub enable_web_search: Option<bool>,
/// Image-edit settings, nested here because baibot has a single `ImageGeneration` purpose
/// and edit shares its config gate. The gen and edit model sets are disjoint, so edit
/// carries its own model field.
#[serde(default)]
pub edit: ImageEditSettings,
}
impl Default for ImageGenerationConfig {
fn default() -> Self {
Self {
model_id: default_image_generation_model_id(),
negative_prompt: None,
cfg_scale: None,
steps: None,
style_preset: None,
seed: None,
safe_mode: None,
hide_watermark: None,
format: None,
width: None,
height: None,
aspect_ratio: None,
resolution: None,
quality: None,
lora_strength: None,
embed_exif_metadata: None,
enable_web_search: None,
edit: ImageEditSettings::default(),
}
}
}
fn default_image_generation_model_id() -> String {
"chroma".to_owned()
}
/// `/image/edit` (`EditImageRequest`) request knobs, mirroring Venice's schema. The source image
/// and prompt are supplied per-call (not config), so only the model and the output-shaping knobs
/// live here. Each knob is optional with `skip_serializing_if`.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ImageEditSettings {
#[serde(default = "default_image_edit_model_id")]
pub model_id: String,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub output_format: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub aspect_ratio: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub resolution: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub safe_mode: Option<bool>,
}
impl Default for ImageEditSettings {
fn default() -> Self {
Self {
model_id: default_image_edit_model_id(),
output_format: None,
aspect_ratio: None,
resolution: None,
safe_mode: None,
}
}
}
fn default_image_edit_model_id() -> String {
"firered-image-edit".to_owned()
}

View File

@@ -0,0 +1,146 @@
use crate::agent::AgentPurpose;
use crate::agent::provider::entity::{
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
};
use crate::agent::provider::{
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
};
use crate::conversation::llm::{
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
MessageContent as LLMMessageContent,
};
use super::super::ControllerTrait;
use super::config::Config;
#[derive(Debug, Clone)]
pub struct Controller {
config: Config,
http: reqwest::Client,
}
impl Controller {
pub fn new(config: Config) -> Self {
// Image generation and text-to-speech can run long, so give the client a generous timeout
// instead of reqwest's default (none). `build` only fails on TLS/system init; fall back to
// the infallible `Client::new()` so this constructor stays infallible.
let http = reqwest::Client::builder()
.timeout(std::time::Duration::from_secs(120))
.build()
.unwrap_or_else(|_| reqwest::Client::new());
Self { config, http }
}
}
impl ControllerTrait for Controller {
async fn ping(&self) -> anyhow::Result<PingResult> {
if !self.supports_purpose(AgentPurpose::TextGeneration) {
return Ok(PingResult::Inconclusive);
}
// Mirror the openai/openai_compat ping: a real "Hello!" round-trip exercises the strict
// /chat/completions body and auth, so a successful ping proves text generation works.
let messages = vec![LLMMessage {
author: LLMAuthor::User,
sender_id: None,
content: LLMMessageContent::Text("Hello!".to_string()),
timestamp: chrono::Utc::now(),
}];
let conversation = LLMConversation { messages };
self.generate_text(conversation, TextGenerationParams::default())
.await?;
Ok(PingResult::Successful)
}
async fn generate_text(
&self,
conversation: LLMConversation,
params: TextGenerationParams,
) -> anyhow::Result<TextGenerationResult> {
super::chat::generate_text(&self.config, &self.http, conversation, params).await
}
async fn speech_to_text(
&self,
mime_type: &mxlink::mime::Mime,
media: Vec<u8>,
params: SpeechToTextParams,
) -> anyhow::Result<SpeechToTextResult> {
super::audio::speech_to_text(&self.config, &self.http, mime_type, media, params).await
}
async fn generate_image(
&self,
prompt: &str,
params: ImageGenerationParams,
) -> anyhow::Result<ImageGenerationResult> {
super::images::generate_image(&self.config, &self.http, prompt, params).await
}
async fn create_image_edit(
&self,
prompt: &str,
images: Vec<ImageSource>,
params: ImageEditParams,
) -> anyhow::Result<ImageEditResult> {
super::images::create_image_edit(&self.config, &self.http, prompt, images, params).await
}
async fn text_to_speech(
&self,
input: &str,
params: TextToSpeechParams,
) -> anyhow::Result<TextToSpeechResult> {
super::audio::text_to_speech(&self.config, &self.http, input, params).await
}
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
match purpose {
AgentPurpose::TextGeneration => self.config.text_generation.is_some(),
AgentPurpose::SpeechToText => self.config.speech_to_text.is_some(),
AgentPurpose::TextToSpeech => self.config.text_to_speech.is_some(),
AgentPurpose::ImageGeneration => self.config.image_generation.is_some(),
AgentPurpose::CatchAll => true,
}
}
fn text_generation_model_id(&self) -> Option<String> {
self.config
.text_generation
.as_ref()
.map(|config| config.model_id.to_owned())
}
fn text_generation_prompt(&self) -> Option<String> {
self.config
.text_generation
.as_ref()
.and_then(|config| config.prompt.clone())
}
fn text_generation_temperature(&self) -> Option<f32> {
self.config
.text_generation
.as_ref()
.map(|config| config.temperature)
}
fn text_to_speech_voice(&self) -> Option<String> {
self.config
.text_to_speech
.as_ref()
.and_then(|config| config.voice.clone())
}
fn text_to_speech_speed(&self) -> Option<f32> {
self.config
.text_to_speech
.as_ref()
.and_then(|config| config.speed)
}
}

View File

@@ -0,0 +1,192 @@
use crate::agent::AgentPurpose;
use crate::agent::provider::entity::{ImageEditResult, ImageGenerationResult, ImageSource};
use crate::agent::provider::{ImageEditParams, ImageGenerationParams};
use crate::strings;
use crate::utils::base64::{base64_decode, base64_encode};
use super::config::Config;
use super::wire::{EditImageRequest, GenerateImageRequest, GenerateImageResponse};
/// Generate an image via Venice's native `/image/generate` endpoint.
///
/// This is the base64-in-JSON path: we pin `return_binary: false` so Venice answers with a JSON
/// envelope (`GenerateImageResponse`) carrying the image as a base64 string, which we decode. The
/// sibling `create_image_edit` is the *other* response shape (raw binary); the two must not be
/// crossed. `params` is advisory only; the Venice config drives the request.
pub async fn generate_image(
config: &Config,
http: &reqwest::Client,
prompt: &str,
_params: ImageGenerationParams,
) -> anyhow::Result<ImageGenerationResult> {
let Some(image_generation_config) = &config.image_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::ImageGeneration
),
));
};
let request = GenerateImageRequest {
model: image_generation_config.model_id.clone(),
prompt: prompt.to_owned(),
// Pinned: baibot wants exactly one image, returned as base64-in-JSON so `GenerateImageResponse`
// can decode it. Flipping `return_binary` would make Venice answer with raw binary and break
// the JSON decode below, so neither knob is configurable.
return_binary: false,
variants: 1,
negative_prompt: image_generation_config.negative_prompt.clone(),
cfg_scale: image_generation_config.cfg_scale,
steps: image_generation_config.steps,
style_preset: image_generation_config.style_preset.clone(),
seed: image_generation_config.seed,
safe_mode: image_generation_config.safe_mode,
hide_watermark: image_generation_config.hide_watermark,
format: image_generation_config.format.clone(),
width: image_generation_config.width,
height: image_generation_config.height,
aspect_ratio: image_generation_config.aspect_ratio.clone(),
resolution: image_generation_config.resolution.clone(),
quality: image_generation_config.quality.clone(),
lora_strength: image_generation_config.lora_strength,
embed_exif_metadata: image_generation_config.embed_exif_metadata,
enable_web_search: image_generation_config.enable_web_search,
};
let url = format!("{}/image/generate", config.base_url.trim_end_matches('/'));
// The prompt is user content; keep it out of logs (mirrors the STT/TTS paths).
tracing::trace!(
model_id = image_generation_config.model_id,
"Sending Venice image generation API request"
);
let response = http
.post(&url)
.bearer_auth(&config.api_key)
.json(&request)
.send()
.await?;
let status = response.status();
if !status.is_success() {
// Body to the server log only, not into the returned error (which reaches the Matrix room).
let body = response.text().await.unwrap_or_default();
tracing::warn!(%status, body, "Venice image generation request failed");
return Err(anyhow::anyhow!(
"Venice image generation request failed with status {status}"
));
}
let response: GenerateImageResponse = response.json().await?;
tracing::trace!(request_id = ?response.id, "Venice image generation succeeded");
let Some(image_base64) = response.images.into_iter().next() else {
return Err(anyhow::anyhow!(
"The Venice image generation API returned no images"
));
};
// Swallow the decode error's detail (it can echo input bytes/offsets); the returned error
// reaches the Matrix room, so it stays generic while the real cause goes to the server log.
let bytes = base64_decode(&image_base64).map_err(|decode_err| {
tracing::warn!(%decode_err, "Venice image generation returned undecodable base64");
anyhow::anyhow!("Venice image generation returned invalid base64 image data")
})?;
Ok(ImageGenerationResult {
bytes,
mime_type: image_format_to_mime_type(image_generation_config.format.as_deref()),
revised_prompt: None,
})
}
/// Edit an image via Venice's native `/image/edit` endpoint.
///
/// This is the raw-binary path: the request is JSON carrying the source image as a base64 string
/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64, no multipart), and the
/// response body IS the edited image bytes (no JSON envelope). `params` is advisory only.
pub async fn create_image_edit(
config: &Config,
http: &reqwest::Client,
prompt: &str,
images: Vec<ImageSource>,
_params: ImageEditParams,
) -> anyhow::Result<ImageEditResult> {
let Some(image_generation_config) = &config.image_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::ImageGeneration
),
));
};
let edit_config = &image_generation_config.edit;
let Some(source) = images.into_iter().next() else {
return Err(anyhow::anyhow!("No image sources provided"));
};
let request = EditImageRequest {
model: edit_config.model_id.clone(),
prompt: prompt.to_owned(),
image: base64_encode(&source.bytes),
output_format: edit_config.output_format.clone(),
aspect_ratio: edit_config.aspect_ratio.clone(),
resolution: edit_config.resolution.clone(),
safe_mode: edit_config.safe_mode,
};
let url = format!("{}/image/edit", config.base_url.trim_end_matches('/'));
// The prompt is user content; keep it out of logs (mirrors the STT/TTS paths).
tracing::trace!(
model_id = edit_config.model_id,
"Sending Venice image edit API request"
);
let response = http
.post(&url)
.bearer_auth(&config.api_key)
.json(&request)
.send()
.await?;
let status = response.status();
if !status.is_success() {
// Body to the server log only, not into the returned error (which reaches the Matrix room).
let body = response.text().await.unwrap_or_default();
tracing::warn!(%status, body, "Venice image edit request failed");
return Err(anyhow::anyhow!(
"Venice image edit request failed with status {status}"
));
}
// The edit endpoint answers with raw binary image bytes, so read the body directly instead of
// parsing JSON. The actual format comes from the response Content-Type header; fall back to the
// configured `output_format` when the header is missing or unparseable.
let mime_type = response
.headers()
.get(reqwest::header::CONTENT_TYPE)
.and_then(|value| value.to_str().ok())
.and_then(|value| value.parse::<mxlink::mime::Mime>().ok())
.unwrap_or_else(|| image_format_to_mime_type(edit_config.output_format.as_deref()));
let bytes = response.bytes().await?.to_vec();
Ok(ImageEditResult { bytes, mime_type })
}
/// Map a Venice image `format`/`output_format` value (`jpeg`/`png`/`webp`) to its MIME type.
/// Venice defaults to `webp` when the format is unset, so an absent value maps to `image/webp`.
fn image_format_to_mime_type(format: Option<&str>) -> mxlink::mime::Mime {
match format.unwrap_or("webp") {
"jpeg" | "jpg" => mxlink::mime::IMAGE_JPEG,
"png" => mxlink::mime::IMAGE_PNG,
// No mxlink::mime constant for webp; parse it, falling back to PNG on any surprise value.
_ => "image/webp"
.parse()
.unwrap_or(mxlink::mime::IMAGE_PNG),
}
}

View File

@@ -0,0 +1,47 @@
mod audio;
mod chat;
mod config;
mod controller;
mod images;
mod utils;
mod wire;
#[cfg(test)]
mod tests;
pub use config::Config;
pub use controller::Controller;
use super::super::AgentInstantiationError;
use super::super::AgentInstantiationResult;
use super::ConfigTrait;
use super::controller::ControllerType;
pub fn create_controller_from_yaml_value_config(
agent_id: &str,
config: serde_yaml_ng::Value,
) -> AgentInstantiationResult<ControllerType> {
let config = match &config {
serde_yaml_ng::Value::Mapping(_) => {
let config: Config =
serde_yaml_ng::from_value(config).map_err(AgentInstantiationError::Yaml)?;
config
.validate()
.map_err(AgentInstantiationError::ConfigFailsValidation)?;
config
}
_ => {
return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping(
agent_id.to_owned(),
));
}
};
Ok(ControllerType::Venice(Box::new(Controller::new(config))))
}
pub fn default_config() -> Config {
Config::default()
}

View File

@@ -0,0 +1,269 @@
use mxlink::matrix_sdk::ruma::OwnedMxcUri;
use mxlink::matrix_sdk::ruma::events::room::message::{
FileMessageEventContent, ImageMessageEventContent,
};
use mxlink::mime;
use super::super::ControllerTrait;
use crate::agent::AgentPurpose;
use crate::conversation::llm::{
Author as LLMAuthor, FileDetails, ImageDetails, Message as LLMMessage,
MessageContent as LLMMessageContent,
};
use super::config::{Config, VeniceParameters, WebSearchMode};
use super::controller::Controller;
use super::utils::convert_llm_messages_to_venice;
use super::wire::{
ContentPart, EditImageRequest, GenerateImageRequest, MessageContent, SpeechRequest,
};
#[test]
fn config_round_trips_with_venice_parameters() {
let yaml = r#"
base_url: https://api.venice.ai/api/v1
api_key: test-key
text_generation:
model_id: kimi-k2-5
temperature: 0.7
max_response_tokens: 1024
max_context_tokens: 65536
venice_parameters:
enable_web_search: "auto"
enable_web_citations: true
speech_to_text:
model_id: nvidia/parakeet-tdt-0.6b-v3
"#;
let config: Config = serde_yaml_ng::from_str(yaml).expect("config should deserialize");
let tg = config.text_generation.expect("text_generation present");
let vp = tg.venice_parameters.expect("venice_parameters present");
assert!(matches!(vp.enable_web_search, Some(WebSearchMode::Auto)));
assert_eq!(vp.enable_web_citations, Some(true));
assert_eq!(vp.character_slug, None);
// The bag must serialize the enum to the exact wire string, and an unset knob must be ABSENT
// (not `null`) so the strict `additionalProperties: false` body is honored.
let json = serde_json::to_string(&vp).expect("serialize venice_parameters");
assert!(
json.contains("\"enable_web_search\":\"auto\""),
"web search should be the literal \"auto\": {json}"
);
assert!(
!json.contains("character_slug"),
"an unset knob must be omitted entirely: {json}"
);
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
}
#[test]
fn converts_image_to_data_uri_and_skips_files() {
let messages = vec![
LLMMessage {
author: LLMAuthor::User,
sender_id: None,
timestamp: chrono::Utc::now(),
content: LLMMessageContent::Text("describe this".to_owned()),
},
LLMMessage {
author: LLMAuthor::User,
sender_id: None,
timestamp: chrono::Utc::now(),
content: LLMMessageContent::Image(ImageDetails::new(
ImageMessageEventContent::plain(
"pic.png".to_owned(),
OwnedMxcUri::from("mxc://example.com/abc"),
),
mime::IMAGE_PNG,
vec![1, 2, 3],
)),
},
LLMMessage {
author: LLMAuthor::User,
sender_id: None,
timestamp: chrono::Utc::now(),
content: LLMMessageContent::File(FileDetails::new(
FileMessageEventContent::plain(
"doc.pdf".to_owned(),
OwnedMxcUri::from("mxc://example.com/def"),
),
mime::APPLICATION_PDF,
vec![4, 5, 6],
)),
},
];
let converted = convert_llm_messages_to_venice(messages);
// Text and image survive; the file is warn-skipped.
assert_eq!(converted.len(), 2);
match &converted[0].content {
MessageContent::Text(text) => assert_eq!(text, "describe this"),
other => panic!("expected bare text, got {other:?}"),
}
match &converted[1].content {
MessageContent::Parts(parts) => match &parts[0] {
ContentPart::ImageUrl { image_url } => assert!(
image_url.url.starts_with("data:image/png;base64,"),
"image should be inlined as a data URI: {}",
image_url.url
),
},
other => panic!("expected image parts, got {other:?}"),
}
}
#[test]
fn supports_purpose_truth_table() {
let config: Config = serde_yaml_ng::from_str(
r#"
base_url: https://api.venice.ai/api/v1
api_key: test-key
text_generation:
model_id: kimi-k2-5
speech_to_text:
model_id: nvidia/parakeet-tdt-0.6b-v3
"#,
)
.expect("config should deserialize");
let controller = Controller::new(config);
assert!(controller.supports_purpose(AgentPurpose::TextGeneration));
assert!(controller.supports_purpose(AgentPurpose::SpeechToText));
assert!(controller.supports_purpose(AgentPurpose::CatchAll));
assert!(!controller.supports_purpose(AgentPurpose::TextToSpeech));
assert!(!controller.supports_purpose(AgentPurpose::ImageGeneration));
}
#[test]
fn supports_purpose_true_when_image_and_tts_blocks_present() {
let config: Config = serde_yaml_ng::from_str(
r#"
base_url: https://api.venice.ai/api/v1
api_key: test-key
text_to_speech:
model_id: tts-kokoro
image_generation:
model_id: chroma
"#,
)
.expect("config should deserialize");
let controller = Controller::new(config);
assert!(controller.supports_purpose(AgentPurpose::TextToSpeech));
assert!(controller.supports_purpose(AgentPurpose::ImageGeneration));
}
#[test]
fn speech_request_serializes_voice_and_omits_unset() {
let request = SpeechRequest {
model: "tts-kokoro".to_owned(),
input: "hello".to_owned(),
voice: Some("af_sky".to_owned()),
speed: None,
response_format: Some("mp3".to_owned()),
prompt: None,
temperature: None,
top_p: None,
};
let json = serde_json::to_string(&request).expect("serialize SpeechRequest");
assert!(
json.contains("\"voice\":\"af_sky\""),
"voice should be present: {json}"
);
assert!(
!json.contains("temperature"),
"an unset knob must be omitted (not null): {json}"
);
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
}
#[test]
fn generate_image_request_pins_flags_and_omits_unset() {
let request = GenerateImageRequest {
model: "chroma".to_owned(),
prompt: "a cat".to_owned(),
return_binary: false,
variants: 1,
negative_prompt: None,
cfg_scale: None,
steps: None,
style_preset: None,
seed: None,
safe_mode: None,
hide_watermark: None,
format: None,
width: None,
height: None,
aspect_ratio: None,
resolution: None,
quality: None,
lora_strength: None,
embed_exif_metadata: None,
enable_web_search: None,
};
let json = serde_json::to_string(&request).expect("serialize GenerateImageRequest");
assert!(json.contains("\"model\":\"chroma\""), "{json}");
assert!(
json.contains("\"return_binary\":false"),
"return_binary must be pinned false: {json}"
);
assert!(
json.contains("\"variants\":1"),
"variants must be pinned 1: {json}"
);
assert!(
!json.contains("cfg_scale"),
"an unset knob must be omitted: {json}"
);
assert!(!json.contains("null"), "no nulls belong in the body: {json}");
}
#[test]
fn edit_image_request_carries_model_and_base64_image() {
let request = EditImageRequest {
model: "firered-image-edit".to_owned(),
prompt: "make it a sunrise".to_owned(),
image: "aGVsbG8=".to_owned(),
output_format: None,
aspect_ratio: None,
resolution: None,
safe_mode: None,
};
let json = serde_json::to_string(&request).expect("serialize EditImageRequest");
assert!(
json.contains("\"model\":\"firered-image-edit\""),
"{json}"
);
assert!(
json.contains("\"image\":\"aGVsbG8=\""),
"the base64 image string must be present: {json}"
);
assert!(
!json.contains("output_format"),
"an unset knob must be omitted: {json}"
);
}
#[test]
fn web_search_mode_off_deserializes_from_bare_yaml_off() {
// `off` is a YAML-1.1 boolean but a plain string under serde_yaml_ng's YAML-1.2 core schema,
// so it deserializes straight into the lowercase `WebSearchMode::Off`. This pins that the
// sample config and docs can use the bare, unquoted `off` without it parsing as a boolean.
let params: VeniceParameters =
serde_yaml_ng::from_str("enable_web_search: off").expect("bare `off` should deserialize");
assert!(matches!(params.enable_web_search, Some(WebSearchMode::Off)));
}

View File

@@ -0,0 +1,55 @@
use crate::conversation::llm::{
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
};
use crate::utils::base64::base64_encode;
use super::wire::{ChatMessage, ContentPart, ImageUrl, MessageContent};
pub fn convert_llm_messages_to_venice(messages: Vec<LLMMessage>) -> Vec<ChatMessage> {
let mut venice_messages: Vec<ChatMessage> = Vec::with_capacity(messages.len());
for message in messages {
if let Some(venice_message) = convert_llm_message_to_venice(message) {
venice_messages.push(venice_message);
}
}
venice_messages
}
fn convert_llm_message_to_venice(message: LLMMessage) -> Option<ChatMessage> {
let role = match message.author {
LLMAuthor::Prompt => "system",
LLMAuthor::Assistant => "assistant",
LLMAuthor::User => "user",
};
match message.content {
LLMMessageContent::Text(text) => Some(ChatMessage {
role: role.to_owned(),
content: MessageContent::Text(text),
}),
LLMMessageContent::Image(image_details) => {
// Inline the image as a base64 data URI, the same shape the OpenAI vision content
// part uses. This is the gap the openai_compat provider can't fill (it drops images).
let data_uri = format!(
"data:{};base64,{}",
image_details.mime,
base64_encode(&image_details.data)
);
Some(ChatMessage {
role: role.to_owned(),
content: MessageContent::Parts(vec![ContentPart::ImageUrl {
image_url: ImageUrl { url: data_uri },
}]),
})
}
LLMMessageContent::File(_file_details) => {
tracing::warn!(
"The Venice provider does not support file content. This file message will be skipped."
);
None
}
}
}

View File

@@ -0,0 +1,212 @@
//! Serde structs modeling Venice's `/chat/completions`, `/audio/transcriptions`,
//! `/audio/speech`, `/image/generate`, and `/image/edit` wire shapes. Request types are
//! `Serialize`-only (we build them, Venice never sends them back); response types are
//! `Deserialize`-only. Keeping the split means the untagged request content enum is never on a
//! deserialize path, so a surprise response shape can't fail to match it.
//!
//! Field names match Venice's schema 1:1 (so the config's `model_id` becomes `model` here). Every
//! request body is `additionalProperties: false`, so optional knobs carry `skip_serializing_if`
//! to omit rather than send `null`. `/audio/speech` and `/image/edit` return raw binary (no
//! response struct); only `/image/generate` returns JSON (`GenerateImageResponse`).
use serde::{Deserialize, Serialize};
use super::config::VeniceParameters;
#[derive(Debug, Serialize)]
pub struct ChatCompletionRequest {
pub model: String,
pub messages: Vec<ChatMessage>,
#[serde(skip_serializing_if = "Option::is_none")]
pub temperature: Option<f32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub max_completion_tokens: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub venice_parameters: Option<VeniceParameters>,
}
#[derive(Debug, Serialize)]
pub struct ChatMessage {
pub role: String,
pub content: MessageContent,
}
/// A message body is either a bare string or a list of content parts. Venice accepts both; we
/// send the parts form only when a message carries an image (baibot keeps text and images in
/// separate messages, so a parts list only ever holds images in v1).
#[derive(Debug, Serialize)]
#[serde(untagged)]
pub enum MessageContent {
Text(String),
Parts(Vec<ContentPart>),
}
#[derive(Debug, Serialize)]
#[serde(tag = "type", rename_all = "snake_case")]
pub enum ContentPart {
ImageUrl { image_url: ImageUrl },
}
#[derive(Debug, Serialize)]
pub struct ImageUrl {
/// A `data:<mime>;base64,<data>` URI for inline images.
pub url: String,
}
/// Standard OpenAI-shaped chat completion response. We only read `choices[0].message.content`;
/// when web search is on, Venice inlines citations as `^n^` superscripts in that content and we
/// pass it through untouched.
#[derive(Debug, Deserialize)]
pub struct ChatCompletionResponse {
pub choices: Vec<ChatChoice>,
}
#[derive(Debug, Deserialize)]
pub struct ChatChoice {
pub message: ResponseMessage,
}
#[derive(Debug, Deserialize)]
pub struct ResponseMessage {
#[serde(default)]
pub content: Option<String>,
}
/// `/audio/transcriptions` response. We read `text`; the optional `duration`/`timestamps` the
/// API can return are not used in v1.
#[derive(Debug, Deserialize)]
pub struct TranscriptionResponse {
pub text: String,
}
/// `/audio/speech` (`CreateSpeechRequestSchema`) request. `input` and `model` are always sent;
/// the rest are omitted when unset. The response is raw binary audio, so there is no response
/// struct.
#[derive(Debug, Serialize)]
pub struct SpeechRequest {
pub model: String,
pub input: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub voice: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub speed: Option<f32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub response_format: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub prompt: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub temperature: Option<f32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub top_p: Option<f32>,
}
/// `/image/generate` (`GenerateImageRequest`) request. `return_binary` is pinned `false` and
/// `variants` to `1` by the builder: baibot wants exactly one image returned as base64-in-JSON,
/// which `GenerateImageResponse` then decodes. Flipping `return_binary` would make Venice answer
/// with raw binary and break that JSON decode, so it is not configurable.
#[derive(Debug, Serialize)]
pub struct GenerateImageRequest {
pub model: String,
pub prompt: String,
pub return_binary: bool,
pub variants: u32,
#[serde(skip_serializing_if = "Option::is_none")]
pub negative_prompt: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub cfg_scale: Option<f32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub steps: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub style_preset: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub seed: Option<i64>,
#[serde(skip_serializing_if = "Option::is_none")]
pub safe_mode: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub hide_watermark: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub format: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub width: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub height: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub aspect_ratio: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub resolution: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub quality: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub lora_strength: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub embed_exif_metadata: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub enable_web_search: Option<bool>,
}
/// `/image/generate` response when `return_binary` is false: a JSON envelope carrying the images
/// as base64 strings. We read `images[0]`; `request`/`timing` and other fields are ignored. `id`
/// is telemetry only (logged, never used for correctness), so it is optional: a response that
/// carries usable `images` must not fail to deserialize just because the telemetry field drifted.
#[derive(Debug, Deserialize)]
pub struct GenerateImageResponse {
#[serde(default)]
pub id: Option<String>,
pub images: Vec<String>,
}
/// `/image/edit` (`EditImageRequest`) request. The source `image` is a base64-encoded string
/// (Venice's `image` field is `anyOf` upload/base64/URL; we send base64-in-JSON, no multipart).
/// The response is raw binary, so there is no response struct.
#[derive(Debug, Serialize)]
pub struct EditImageRequest {
pub model: String,
pub prompt: String,
/// Base64-encoded source image bytes.
pub image: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub output_format: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub aspect_ratio: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub resolution: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub safe_mode: Option<bool>,
}

View File

@@ -9,7 +9,7 @@ fn agent_config_parsing_works() {
let sample_config = crate::agent::default_config_for_provider(&provider); let sample_config = crate::agent::default_config_for_provider(&provider);
let sample_config_pretty_yaml = serde_yaml_ng::to_string(&sample_config).unwrap(); let sample_config_pretty_yaml = serde_yaml_ng::to_string(&sample_config).unwrap();
let test_cases = vec![ let test_cases = [
// Invalid input // Invalid input
TestCase { TestCase {
input: r#"Hello"#.to_owned(), input: r#"Hello"#.to_owned(),

View File

@@ -105,7 +105,8 @@ pub mod test {
let bpe = super::get_bpe_for_model(model); let bpe = super::get_bpe_for_model(model);
let max_response_tokens: Option<u32> = Some(5); let max_response_tokens_value: u32 = 5;
let max_response_tokens: Option<u32> = Some(max_response_tokens_value);
let prompt = super::Message { let prompt = super::Message {
author: super::Author::Prompt, author: super::Author::Prompt,
@@ -193,7 +194,7 @@ pub mod test {
&Some(prompt), &Some(prompt),
conversation_messages, conversation_messages,
max_response_tokens, max_response_tokens,
prompt_length + max_response_tokens.unwrap_or(0) + forth_length + third_length, prompt_length + max_response_tokens_value + forth_length + third_length,
); );
assert_eq!(2, new_conversation_messages.len()); assert_eq!(2, new_conversation_messages.len());
@@ -215,7 +216,8 @@ pub mod test {
let bpe = super::get_bpe_for_model(model); let bpe = super::get_bpe_for_model(model);
let max_response_tokens: Option<u32> = Some(5); let max_response_tokens_value: u32 = 5;
let max_response_tokens: Option<u32> = Some(max_response_tokens_value);
let prompt = super::Message { let prompt = super::Message {
author: super::Author::User, author: super::Author::User,
@@ -303,7 +305,7 @@ pub mod test {
&Some(prompt), &Some(prompt),
conversation_messages, conversation_messages,
max_response_tokens, max_response_tokens,
prompt_length + max_response_tokens.unwrap_or(0) + forth_length + third_length, prompt_length + max_response_tokens_value + forth_length + third_length,
); );
assert_eq!(2, new_conversation_messages.len()); assert_eq!(2, new_conversation_messages.len());

View File

@@ -119,7 +119,7 @@ async fn get_matrix_messages_in_reply_chain_native(
AnySyncMessageLikeEvent::RoomMessage(room_message) => { AnySyncMessageLikeEvent::RoomMessage(room_message) => {
if let SyncMessageLikeEvent::Original(room_message_original) = room_message { if let SyncMessageLikeEvent::Original(room_message_original) = room_message {
match room_message_original.content.relates_to { match room_message_original.content.relates_to {
Some(Relation::Reply { in_reply_to }) => Some(in_reply_to.event_id.clone()), Some(Relation::Reply(reply)) => Some(reply.in_reply_to.event_id.clone()),
_ => None, _ => None,
} }
} else { } else {
@@ -420,11 +420,11 @@ pub async fn determine_interaction_context_for_room_event(
) )
.await .await
} }
Relation::Reply { in_reply_to } => { Relation::Reply(reply) => {
determine_interaction_context_for_room_event_related_to_reply( determine_interaction_context_for_room_event_related_to_reply(
current_event, current_event,
current_event_is_mentioning_bot, current_event_is_mentioning_bot,
in_reply_to.event_id.clone(), reply.in_reply_to.event_id.clone(),
) )
.await .await
} }

View File

@@ -1,3 +1,9 @@
// rustc 1.94+ trips a query-depth overflow when computing async layouts in
// the matrix-sdk timeline future graph. matrix-rust-sdk PR #6489 raises the
// limit, but `recursion_limit` is per-crate and applies to the crate currently
// being compiled — so the consumer has to repeat it.
#![recursion_limit = "256"]
mod agent; mod agent;
mod bot; mod bot;
mod controller; mod controller;