Compare commits
106 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
edbd72ece6 | ||
|
|
22906aa2d3 | ||
|
|
99bde53ef6 | ||
|
|
b3fd8e548f | ||
|
|
062fbbb8ef | ||
|
|
2801c78ad9 | ||
|
|
f4c698ad33 | ||
|
|
bd39001417 | ||
|
|
2692d0322e | ||
|
|
8eb70f0f2c | ||
|
|
5c0a7be7a2 | ||
|
|
0a8f9fc3e5 | ||
|
|
1ac3b2e060 | ||
|
|
a3ef9fd1bf | ||
|
|
df507eb201 | ||
|
|
4dcd9eff40 | ||
|
|
ea760ce755 | ||
|
|
1528df6a55 | ||
|
|
b0fa024297 | ||
|
|
3ec203128a | ||
|
|
da97361e1b | ||
|
|
b430fe0189 | ||
|
|
f03126a9e1 | ||
|
|
7d46b926c1 | ||
|
|
6f3c048195 | ||
|
|
b47cf598b5 | ||
|
|
265ad7e1cb | ||
|
|
a159f67e45 | ||
|
|
624b9de35b | ||
|
|
941bf7ca42 | ||
|
|
ef0f1671da | ||
|
|
b43f61f5ff | ||
|
|
1967d2b34c | ||
|
|
bb3734ad24 | ||
|
|
eb6db34177 | ||
|
|
6e845caa2e | ||
|
|
1004966785 | ||
|
|
7ae1864c2e | ||
|
|
68a2fb161f | ||
|
|
3a3eb58d7b | ||
|
|
74d988e650 | ||
|
|
ed8bedcd7e | ||
|
|
2842632969 | ||
|
|
10a5bd2abb | ||
|
|
5308b75f52 | ||
|
|
dad61e1270 | ||
|
|
91986a129c | ||
|
|
264f683d6a | ||
|
|
62f0f4fa0d | ||
|
|
69627abd74 | ||
|
|
d2660be33c | ||
|
|
ce81fe69bd | ||
|
|
1162636b88 | ||
|
|
8c90e13a79 | ||
|
|
274b614d25 | ||
|
|
7bd46821dc | ||
|
|
a84135ff32 | ||
|
|
231528a0d8 | ||
|
|
d8e47b0578 | ||
|
|
96c1542f4a | ||
|
|
2f9c3dfce0 | ||
|
|
de958208b2 | ||
|
|
ac4f2080ce | ||
|
|
3ffa50b7b9 | ||
|
|
8f86289373 | ||
|
|
e0dcc39a72 | ||
|
|
c94376109c | ||
|
|
256ed05662 | ||
|
|
8222681e27 | ||
|
|
f304b93c68 | ||
|
|
889d8a1d04 | ||
|
|
6082bfaf56 | ||
|
|
1d629e0859 | ||
|
|
49471c1df0 | ||
|
|
8f956d2329 | ||
|
|
ba4aa35987 | ||
|
|
06d699a17d | ||
|
|
77d41fb7eb | ||
|
|
aaf283dde3 | ||
|
|
17eafa86af | ||
|
|
4704934b06 | ||
|
|
6719538530 | ||
|
|
59e2746578 | ||
|
|
47d8edea70 | ||
|
|
692d61b239 | ||
|
|
05902f4c17 | ||
|
|
7e66068b16 | ||
|
|
406141cd7d | ||
|
|
c051da2f4a | ||
|
|
1ff7e8cf79 | ||
|
|
b3bca98e84 | ||
|
|
c07b712318 | ||
|
|
6741483056 | ||
|
|
06b2b6d776 | ||
|
|
a1bd292752 | ||
|
|
e4e1fe0e7b | ||
|
|
45a2d96029 | ||
|
|
ec1879d212 | ||
|
|
5e6a600895 | ||
|
|
3db924b124 | ||
|
|
ff7a5ef7af | ||
|
|
cd7d9137e8 | ||
|
|
0d509b2d0e | ||
|
|
3c47d40781 | ||
|
|
78893247e7 | ||
|
|
4847bd8ba8 |
85
.github/workflows/workflow.yml
vendored
85
.github/workflows/workflow.yml
vendored
@@ -3,13 +3,14 @@ on:
|
||||
push:
|
||||
branches: [ "main" ]
|
||||
tags: [ "v*" ]
|
||||
schedule:
|
||||
- cron: '0 0 * * 1'
|
||||
permissions:
|
||||
checks: write
|
||||
contents: write
|
||||
packages: write
|
||||
pull-requests: read
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
jobs:
|
||||
test-and-clippy:
|
||||
name: Unit testing and linting
|
||||
@@ -17,20 +18,46 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: dtolnay/rust-toolchain@stable
|
||||
- name: Install SQLite3
|
||||
run: sudo apt-get update && sudo apt-get install -y libsqlite3-dev
|
||||
- run: cargo test --all-features
|
||||
- run: cargo clippy
|
||||
|
||||
build-publish:
|
||||
name: Build and Publish
|
||||
runs-on: self-hosted
|
||||
docker-clean-metadata:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
json: ${{ steps.meta.outputs.json }}
|
||||
steps:
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
- name: Extract metadata (tags, labels) for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
platforms: arm64
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v1
|
||||
- name: Login to ghcr.io
|
||||
images: |
|
||||
ghcr.io/${{ github.repository }}
|
||||
tags: |
|
||||
type=raw,value=latest,enable=${{ github.ref_name == 'main' }}
|
||||
type=semver,pattern={{raw}}
|
||||
|
||||
docker-build:
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
attestations: write
|
||||
id-token: write
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- os: self-hosted
|
||||
arch: amd64
|
||||
- os: ubuntu-24.04-arm
|
||||
arch: arm64
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
- name: Log in to the GitHub Container registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
@@ -40,17 +67,41 @@ jobs:
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/${{ github.repository }}
|
||||
registry.etke.cc/${{ github.repository }}
|
||||
tags: |
|
||||
type=raw,value=latest,enable=${{ github.ref_name == 'main' }}
|
||||
type=semver,pattern={{raw}}
|
||||
- name: Build and push
|
||||
flavor: |
|
||||
latest=auto
|
||||
suffix=-${{ matrix.arch }},onlatest=true
|
||||
images: |
|
||||
ghcr.io/${{ github.repository }}
|
||||
|
||||
- name: Build and push Docker images
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
platforms: linux/amd64,linux/arm64
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
file: Dockerfile.ci
|
||||
|
||||
docker-manifest:
|
||||
needs:
|
||||
- docker-build
|
||||
- docker-clean-metadata
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
image: ${{ fromJson(needs.docker-clean-metadata.outputs.json).tags }}
|
||||
|
||||
steps:
|
||||
- name: Log in to the GitHub Container registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Create and push manifest
|
||||
run: |
|
||||
docker manifest create ${{ matrix.image }} ${{ matrix.image }}-amd64 ${{ matrix.image }}-arm64
|
||||
docker manifest push ${{ matrix.image }}
|
||||
|
||||
118
CHANGELOG.md
118
CHANGELOG.md
@@ -1,9 +1,127 @@
|
||||
# (2025-12-15) Version 1.11.0
|
||||
|
||||
- (**Feature**) Add support for custom avatars via file path and for keeping the already-set avatar (for those who wish to manage it by themselves via other means). See the [sample config](./etc/app/config.yml.dist) for details. ([062fbbb](https://github.com/etkecc/baibot/commit/062fbbb8ef9ad600db483a431c5c782402191023))
|
||||
|
||||
- (**Internal Improvement**) Dependency updates ([99bde53](https://github.com/etkecc/baibot/commit/99bde53ef648a5a9086a96778fde4a9dbc1ede58))
|
||||
|
||||
- (**Internal Improvement**) Documentation updates ([b3fd8e5](https://github.com/etkecc/baibot/commit/b3fd8e548f83fe46398ced4760d7e2bb7588c24d))
|
||||
|
||||
- (**Internal Improvement**) Upgrade Rust compiler (1.91.1 -> 1.92.0) ([22906aa](https://github.com/etkecc/baibot/commit/22906aa2d3cae51815fad2560a545eaa69c247b6))
|
||||
|
||||
|
||||
# (2025-12-06) Version 1.10.0
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.11.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.16.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.16.0).
|
||||
|
||||
# (2025-11-30) Version 1.9.0
|
||||
|
||||
- (**Internal Improvement**) Upgrade [async-openai](https://crates.io/crates/async-openai) from our own etkecc fork (0.28.1-patched) to the official upstream version 0.31.1. This upgrade required some code adaptations to the new module structure, etc. While tested, regressions are possible.
|
||||
|
||||
# (2025-11-28) Version 1.8.3
|
||||
|
||||
- (**Improvement**) Add support for the `BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY` environment variable for configuring `persistence.session_encryption_key`
|
||||
|
||||
- (**Improvement**) Add support for the `BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED` environment variable for configuring `user.encryption.recovery_reset_allowed`
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
# (2025-11-20) Version 1.8.2
|
||||
|
||||
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.89.0 -> 1.91.1).
|
||||
|
||||
# (2025-09-12) Version 1.8.1
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
# (2025-09-08) Version 1.8.0
|
||||
|
||||
- (**Internal Improvement**) Upgrade [mxlink](https://crates.io/crates/mxlink) (1.9.0 -> 1.10.0) and [matrix-sdk](https://crates.io/crates/matrix-sdk) (0.13.0 -> 0.14.0)
|
||||
|
||||
- (**Internal Improvement**) Upgrade [Rust](https://www.rust-lang.org/) (1.88.0 -> 1.89.0)
|
||||
|
||||
- (**Internal Improvement**) Upgrade Debian base for container images (12/bookworm -> 13/trixie)
|
||||
|
||||
# (2025-07-11) Version 1.7.6
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.9.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.13.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.13.0), which contains fixes for some security vulnerabilities)
|
||||
|
||||
# (2025-06-10) Version 1.7.5
|
||||
|
||||
- (**Internal Improvement**) Dependency and compiler updates (Rust 1.86 -> 1.86).
|
||||
|
||||
# (2025-06-10) Version 1.7.4
|
||||
|
||||
- (**Internal Improvement**) Dependency updates.
|
||||
|
||||
# (2025-06-10) Version 1.7.3
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.8.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.12.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.12.0), which contains fixes for important security vulnerabilities)
|
||||
|
||||
# (2025-05-11) Version 1.7.2
|
||||
|
||||
- (**Bugfix**) Allow `image_generation.size` configuration value for OpenAI to be `null` to allow the model to choose the size automatically and default to that
|
||||
|
||||
# (2025-05-11) Version 1.7.1
|
||||
|
||||
- (**Bugfix**) Fix lack of documentation for the new [image-editing](./docs/features.md#-image-editing) feature in the `!bai usage` command's output
|
||||
|
||||
# (2025-05-10) Version 1.7.0
|
||||
|
||||
- (**Feature**) Add vision support to the OpenAI and Anthropic providers. You can now mix text and images in your conversations - fixes [issue #5](https://github.com/etkecc/baibot/issues/5)
|
||||
|
||||
- (**Feature**) Add [image-editing](./docs/features.md#-image-editing) support to the OpenAI provider
|
||||
|
||||
- (**Improvement**) Add compatibility with OpenAI's `gpt-image-1` model - fixes [issue #40](https://github.com/etkecc/baibot/issues/40)
|
||||
|
||||
- (**Change**) Rework [image-creation](./docs/features.md#-image-creation) to avoid command conflicts with [image-editing](./docs/features.md#-image-editing). The image-creation command syntax is now `!bai image create <prompt>` (previously: `!bai image <prompt>`).
|
||||
|
||||
- (**Internal Improvement**) Dependency and compiler updates
|
||||
|
||||
> [!WARNING]
|
||||
> Unlike other releases, this release is not published to [crates.io](https://crates.io), because it relies on multiple library forks (`async-openai` and `anthropic-rs`) sourced from Github.
|
||||
|
||||
|
||||
# (2025-04-12) Version 1.6.0
|
||||
|
||||
- (**Internal Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.7.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.11.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.11.0))
|
||||
|
||||
|
||||
# (2025-03-31) Version 1.5.1
|
||||
|
||||
- (**Internal Improvement**) Dependency updates
|
||||
|
||||
# (2025-02-27) Version 1.5.0
|
||||
|
||||
- (**Feature**) Add support for sending Speech-to-Text replies for [Transcribe-only mode](./docs/features.md#transcribe-only-mode) as regular text messages instead of notices and doing it so by default ([a1bd292752](https://github.com/etkecc/baibot/commit/a1bd292752bdd37a196788c73d00b5619e843a78)) - improvement for [issue #14](https://github.com/etkecc/baibot/issues/14). See [🦻 Speech-to-Text / 🪄 Message Type for non-threaded only-transcribed messages](./docs/configuration/speech-to-text.md#-message-type-for-non-threaded-only-transcribed-messages) for details.
|
||||
|
||||
- (**Feature**) Add config setting controlling if a self-introduction message is posted after joining a room ([c051da2f4a](https://github.com/etkecc/baibot/commit/c051da2f4a161de0974ebb917f7a52d01f5a001f)) - fixes [issue #32](https://github.com/etkecc/baibot/issues/32). You may wish to add a `room.post_join_self_introduction_enabled` property to your configuration. See the [sample config](./etc/app/config.yml.dist) for details. If unspecified, it defaults to `true` anyway which preserves the old behavior.
|
||||
|
||||
- (**Feature**) Add support for configuring `max_completion_tokens` for OpenAI ([47d8edea70](https://github.com/etkecc/baibot/commit/47d8edea705a44aa25a9bfaec4888c0f9ea8700e))
|
||||
|
||||
- (**Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.6.1 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.10.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.10.0))
|
||||
|
||||
- (**Improvement**) Populate image/audio attachment `body` with a filename, not with text to avoid incorrect rendering in Element Web, etc. ([ec1879d212](https://github.com/etkecc/baibot/commit/ec1879d212fa8d6e5f8590486e94c72abfcb75a5))
|
||||
|
||||
- (**Improvement**) Replace Anthropic library ([anthropic-rs](https://crates.io/crates/anthropic-rs) -> [anthropic](https://crates.io/crates/anthropic)) and switch default recommended model (`claude-3-5-sonnet-20240620` -> `claude-3-7-sonnet-20250219`) ([692d61b239](https://github.com/etkecc/baibot/commit/692d61b2398f073b81d32d4cbe8145ab3929e48c)) - fixes [issue #22](https://github.com/etkecc/baibot/issues/22)
|
||||
|
||||
- (**Internal Improvement**) Switch to native building of `arm64` container images to decrease total build times from ~40 minutes to ~8 minutes ([6719538530b](https://github.com/etkecc/baibot/commit/6719538530bf76b3ff2d24077b2a7fa868276b79))
|
||||
|
||||
- (**Internal Improvement**) Various other internal changes, including upgrading [Rust from 1.82 to 1.85 and switching to Rust edition 2024](https://blog.rust-lang.org/2025/02/20/Rust-1.85.0.html)
|
||||
|
||||
|
||||
# (2024-12-12) Version 1.4.1
|
||||
|
||||
- (**Bugfix**) Fix detection for whether the bot is the last member in a room, to avoid incorrectly leaving multi-user rooms that have had at least one person `leave` ([3c47d40781](https://github.com/etkecc/baibot/commit/3c47d407819aa9c0121117a411858238724f06da))
|
||||
|
||||
|
||||
# (2024-11-19) Version 1.4.0
|
||||
|
||||
- (**Improvement**) Dependency updates. This version is based on [mxlink](https://crates.io/crates/mxlink)@1.4.0 (which is based on the newly released [matrix-sdk](https://crates.io/crates/matrix-sdk)@[0.8.0](https://github.com/matrix-org/matrix-rust-sdk/releases/tag/matrix-sdk-0.8.0)). Once you run this version at least once and your matrix-sdk datastore gets upgraded to the new schema, **you will not be able to downgrade to older baibot versions** (based on the older matrix-sdk), unless you start with an empty datastore.
|
||||
|
||||
- (**Bugfix**) Add missing typing notices sending functionality while generating images ([9d166e35ba](https://github.com/etkecc/baibot/commit/9d166e35ba6fc0daaf69318870e92436f3302056))
|
||||
|
||||
- (**Feature**) Support for [Matrix authenticated media](https://matrix.org/docs/spec-guides/authed-media-servers/), thanks to upgrading [mxlink](https://crates.io/crates/mxlink) / [matrix-sdk](https://crates.io/crates/matrix-sdk) - fixes [issue #12](https://github.com/etkecc/baibot/issues/12)
|
||||
|
||||
|
||||
# (2024-11-12) Version 1.3.2
|
||||
|
||||
|
||||
2726
Cargo.lock
generated
2726
Cargo.lock
generated
File diff suppressed because it is too large
Load Diff
22
Cargo.toml
22
Cargo.toml
@@ -7,32 +7,34 @@ license = "AGPL-3.0-or-later"
|
||||
readme = "README.md"
|
||||
keywords = ["matrix", "chat", "bot", "AI", "LLM"]
|
||||
include = ["/etc/assets/baibot-torso-768.png", "/src", "/README.md", "/CHANGELOG.md", "/LICENSE"]
|
||||
version = "1.4.0"
|
||||
edition = "2021"
|
||||
version = "1.11.0"
|
||||
edition = "2024"
|
||||
|
||||
[lib]
|
||||
name = "baibot"
|
||||
path = "src/lib.rs"
|
||||
|
||||
[dependencies]
|
||||
anthropic-rs = "0.1.*"
|
||||
anthropic = { git = "https://github.com/etkecc/anthropic-rs.git", branch = "fix-content-block-image" }
|
||||
anyhow = "1.0.*"
|
||||
async-openai = "0.26.*"
|
||||
async-openai = { version = "0.31.1", features = ["audio", "chat-completion", "image"] }
|
||||
base64 = "0.22.*"
|
||||
chrono = { version = "0.4.*", default-features = false, features = ["std", "now"] }
|
||||
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
|
||||
matrix-sdk = { version = "0.8.0", default-features = false }
|
||||
# We add the `native-tls` feature, because of https://github.com/etkecc/rust-mxlink/issues/1
|
||||
matrix-sdk = { version = "0.16.0", default-features = false, features = ["native-tls"] }
|
||||
mime_guess = "2.0.*"
|
||||
mxidwc = "1.0.*"
|
||||
mxlink = ">=1.4.0"
|
||||
mxlink = ">=1.11.0"
|
||||
etke_openai_api_rust = "0.1.*"
|
||||
quick_cache = "0.6.*"
|
||||
regex = "1.11.*"
|
||||
regex = "1.12.*"
|
||||
serde = { version = "1.0.*", features = ["derive"], default-features = false }
|
||||
serde_json = "1.0.*"
|
||||
serde_yaml = "0.9.*"
|
||||
tempfile = "3.14.*"
|
||||
tiktoken-rs = { version = "0.6.*", features = ["async-openai"] }
|
||||
tokio = { version = "1.41.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||
tempfile = "3.23.*"
|
||||
tiktoken-rs = { version = "0.9.*", default-features = false }
|
||||
tokio = { version = "1.48.*", features = ["rt", "rt-multi-thread", "macros"] }
|
||||
tracing = "0.1.*"
|
||||
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
|
||||
url = "2.5.*"
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/rust:1.82.0-slim-bookworm AS build
|
||||
FROM docker.io/rust:1.92.0-slim-trixie AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||
|
||||
@@ -39,7 +39,7 @@ RUN --mount=type=cache,target=/target,sharing=locked \
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/debian:bookworm-slim
|
||||
FROM docker.io/debian:trixie-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y ca-certificates sqlite3 && \
|
||||
apt-get clean && \
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/rust:1.82.0-slim-bookworm AS build
|
||||
FROM docker.io/rust:1.92.0-slim-trixie AS build
|
||||
|
||||
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
|
||||
|
||||
@@ -20,7 +20,7 @@ RUN cargo build --release
|
||||
# #
|
||||
#######################################
|
||||
|
||||
FROM docker.io/debian:bookworm-slim
|
||||
FROM docker.io/debian:trixie-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y ca-certificates sqlite3 && \
|
||||
apt-get clean && \
|
||||
|
||||
@@ -17,10 +17,10 @@ It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use
|
||||
|
||||
- Supports **different use purposes** (depending on the [☁️ provider](./docs/providers.md) & model):
|
||||
|
||||
- [💬 text-generation](./docs/features.md#-text-generation): communicating with you via text
|
||||
- [💬 text-generation](./docs/features.md#-text-generation): communicating with you via text (though certain models may "see" images as well)
|
||||
- [🦻 speech-to-text](./docs/features.md#-speech-to-text): turning your voice messages into text
|
||||
- [🗣️ text-to-speech](./docs/features.md#%EF%B8%8F-text-to-speech): turning bot or users text messages into voice messages
|
||||
- [🖌️ image-generation](./docs/features.md#%EF%B8%8F-image-generation): generating images based on instructions
|
||||
- [🖌️ image-generation](./docs/features.md#image-generation): creating and editing images based on instructions
|
||||
|
||||
- 🪄 Supports [seamless voice interaction](./docs/features.md#seamless-voice-interaction) (turning user voice messages into text, answering in text, then turning that text back into voice)
|
||||
|
||||
|
||||
@@ -43,7 +43,8 @@ Administrators cannot be changed without adjusting the bot's configuration on th
|
||||
|
||||
Room-local agent managers are users privileged to **create their own [agents](./agents.md)** (see `!bai agent`) in rooms.
|
||||
|
||||
**⚠️ WARNING**: Letting regular users create agents which contact arbitrary network services **may be a security issue**.
|
||||
> [!WARNING]
|
||||
> Letting regular users create agents which contact arbitrary network services **may be a security issue**.
|
||||
|
||||
The following commands are available:
|
||||
- **Show** the currently allowed users: `!bai access room-local-agent-managers`
|
||||
|
||||
@@ -35,7 +35,7 @@ Depending on where the agent is defined (within a room, globally, or [statically
|
||||
|
||||
When creating an agent, you will be given some sample [YAML](https://en.wikipedia.org/wiki/YAML) configuration which you can use to customize the agent's behavior.
|
||||
|
||||
This configuration varies depending on the [☁️ provider](./providers.md) used and the capabilities of the agent. Based on the configuration keys you pass, certain features will be enabled or disabled. For example, if you skip the `image_generation` key for an [OpenAI](./providers.md#openai) agent, it won't be able to generate images (see [🖌️ Image Generation](./features.md#-image-generation)).
|
||||
This configuration varies depending on the [☁️ provider](./providers.md) used and the capabilities of the agent. Based on the configuration keys you pass, certain features will be enabled or disabled. For example, if you skip the `image_generation` key for an [OpenAI](./providers.md#openai) agent, it won't be able to generate images (see [🖌️ Image Creation](./features.md#-image-creation), [🎨 Image Editing](./features.md#-image-editing), [🫵 Sticker Creation](./features.md#-sticker-creation)).
|
||||
|
||||
After making your modifications to the sample YAML, you submit it back to the bot and the new agent will be created.
|
||||
|
||||
|
||||
@@ -12,12 +12,15 @@ This file is created from the template found in [etc/app/config.yml.dist](../../
|
||||
|
||||
Certain keys can be left unset, in which case [📝 hardcoded defaults](../../src/entity/cfg/defaults.rs) would be used.
|
||||
|
||||
Each configuration key found in the YAML configuration can be overridden by setting an environment variable (dots should be replaced with `_`). Example:
|
||||
Some configuration keys found in the YAML configuration can be overridden by setting an environment variable (dots should be replaced with `_`). Example:
|
||||
|
||||
- to override `command_prefix`, set an environment variable `BAIBOT_COMMAND_PREFIX`
|
||||
- to override `homeserver.server_name`, set an environment variable `BAIBOT_HOMESERVER_SERVER_NAME`
|
||||
|
||||
The static configuration contains an `initial_global_config` key, which is used to populate the bot's global configuration (stored as [dynamic configuration](#dynamic-configuration)) the first time the bot starts. Modifying this subsequently will not have any effect. After initial global configuration creation, it's expected to be managed dynamically via chat commands.
|
||||
You can see the list of supported environment variables in the [🦀 src/entity/cfg/env.rs](../../src/entity/cfg/env.rs) file.
|
||||
|
||||
> [!WARNING]
|
||||
> The static configuration contains an `initial_global_config` key, which is used to populate the bot's global configuration (stored as [dynamic configuration](#dynamic-configuration)) the first time the bot starts. Modifying this subsequently will not have any effect. After initial global configuration creation, it's expected to be managed dynamically via chat commands.
|
||||
|
||||
|
||||
### Dynamic configuration
|
||||
@@ -40,7 +43,7 @@ You can adjust the following settings per room and/or globally:
|
||||
- [💬 Text Generation](text-generation.md)
|
||||
- [🦻 Speech-to-Text](speech-to-text.md)
|
||||
- [🗣️ Text-to-Speech](text-to-speech.md)
|
||||
- [🖌️ Image Generation](image-generation.md)
|
||||
- [🖌️ Image Creation](image-generation.md)
|
||||
- [🤝 Handlers](handlers.md)
|
||||
|
||||
Refer to the bot's help messages (as a response to a `!bai config` help command) for the most up-to-date information on what Room Settings can be configured.
|
||||
|
||||
@@ -8,10 +8,10 @@ You can also use **different models within the same room** (e.g. [💬 text-gene
|
||||
|
||||
The bot supports the following use-purposes:
|
||||
|
||||
- [💬 text-generation](../features.md#-text-generation): communicating with you via text
|
||||
- [💬 text-generation](../features.md#-text-generation): communicating with you via text (though certain models may "see" images as well)
|
||||
- [🦻 speech-to-text](../features.md#-speech-to-text): turning your voice messages into text
|
||||
- [🗣️ text-to-speech](../features.md#️-text-to-speech): turning bot or users text messages into voice messages
|
||||
- [🖌️ image-generation](../features.md#-image-generation): generating images based on instructions
|
||||
- [🖌️ image-generation](../features.md#image-generation): generating images based on instructions
|
||||
|
||||
In a given room, each different purpose can be served by a different [provider](../providers.md) and model. This combination of provider and model configuration is called an [🤖 agent](../agents.md). Each purpose can be served by a different **handler** agent.
|
||||
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
|
||||
## 🖌️ Image Generation
|
||||
## Image Generation
|
||||
|
||||
The Image Generation feature is not configurable at this moment.
|
||||
The Image Creation and Image Editing features are not configurable at this moment.
|
||||
|
||||
You may also wish to see:
|
||||
|
||||
- [🌟 Features / 🖌️ Image Generation](../features.md#-image-generation) for a higher-level introduction to the Image Generation features
|
||||
- [📖 Usage / 🖌️ Image Generation](../usage.md#-image-generation) section for more details on how to use the bot for Image Generation in a room
|
||||
- [🌟 Features / Image Generation / 🖌️ Image Creation](../features.md#-image-creation) for a higher-level introduction to the Image Creation features
|
||||
- [🌟 Features / Image Generation / 🎨 Image Editing](../features.md#-image-editing) for a higher-level introduction to the Image Editing features
|
||||
- [📖 Usage / Image Generation / 🖌️ Creating Images](../usage.md#-creating-images) section for more details on how to use the bot for Image Creation in a room
|
||||
- [📖 Usage / Image Generation / 🎨 Editing images](../usage.md#-editing-images) section for more details on how to use the bot for Image Editing in a room
|
||||
|
||||
@@ -23,6 +23,19 @@ The following configuration values are recognized:
|
||||
Example: `!bai config room speech-to-text set-flow-type ignore` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
|
||||
|
||||
|
||||
### 🪄 Message Type for non-threaded only-transcribed messages
|
||||
|
||||
Controls how the transcribed text of voice messages is sent to the chat when Flow Type = `only_transcribe`.
|
||||
|
||||
The following configuration values are recognized:
|
||||
|
||||
- (default) `text`: the transcribed text is sent as a regular message. This is more convenient if you'd like to forward the transcribed message to other rooms.
|
||||
|
||||
- `notice`: the transcribed text is sent as a notice message. This provides better compatibility with other bots in the room, as they are less likely to interact with messages of type notice.
|
||||
|
||||
Example: `!bai config room speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages notice` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
|
||||
|
||||
|
||||
### 🔤 Language
|
||||
|
||||
Lets you specify the language of the input voice messages, to avoid using auto-detection.
|
||||
|
||||
@@ -93,7 +93,7 @@ For getting started most quickly (and locally), we recommend using [LocalAI](#lo
|
||||
|
||||
**Ollama is most lightweight** (~2GB for the container image + ~1.6GB for the model), but supports only [💬 text-generation](./features.md#-text-generation).
|
||||
|
||||
**LocalAI requires 4x more disk space** (~6GB for the container image + ~12GB for the models), but supports [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text) and [🖼️ image-generation](./features.md#️-image-generation).
|
||||
**LocalAI requires 4x more disk space** (~6GB for the container image + ~12GB for the models), but supports [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text) and [🖼️ image-generation](./features.md#️-image-creation).
|
||||
|
||||
**OpenAI supports all of these capabilities** as well and does not require powerful hardware or lots of disk space. However, it requires signup and an API key.
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@ You can also use **different models within the same room** (e.g. [💬 text-gene
|
||||
|
||||
The bot supports the following use-purposes:
|
||||
|
||||
- [💬 text-generation](#-text-generation): communicating with you via text
|
||||
- [💬 text-generation](#-text-generation): communicating with you via text (though certain models may "see" images as well)
|
||||
- [🦻 speech-to-text](#-speech-to-text): turning your voice messages into text
|
||||
- [🗣️ text-to-speech](#%EF%B8%8F-text-to-speech): turning bot or users text messages into voice messages
|
||||
- [🖌️ image-generation](#%EF%B8%8F-image-generation): generating images based on instructions
|
||||
@@ -22,10 +22,12 @@ For more information about configuring handlers, see the [🤝 Handlers / Config
|
||||
|
||||
### 💬 Text Generation
|
||||
|
||||
Text Generation is the bot's ability to **respond to users' text messages with text**.
|
||||
Text Generation is the bot's ability to **respond to users' messages with text**.
|
||||
|
||||

|
||||
|
||||
Some models also support vision, so you may be able to mix text and images in the same conversation.
|
||||
|
||||
In multi-user (group) rooms, to avoid disturbing the normal conversation between people, the bot is auto-configured to only respond to messages starting with the command prefix (`!bai`) or direct mentions via the [💬 Text Generation / 🗟 Prefix Requirement Type](./configuration/text-generation.md#-prefix-requirement-type) setting.
|
||||
|
||||
Normally, the bot only responds to allowed [👥 Users](./access.md#-users). In certain cases, it's useful for an allowed user to provoke the bot to respond even in foreign threads or reply chains. You can learn more about this feature in the [On-demand involvement](./features.md#on-demand-involvement) section below.
|
||||
@@ -136,27 +138,45 @@ To operate in this mode, you can:
|
||||
|
||||
- adjust the [🦻 Speech-to-Text / 🪄 Flow Type](./configuration/speech-to-text.md#-flow-type) setting to make the bot only transcribe (without doing [💬 Text Generation](#-text-generation)): `!bai config room speech-to-text set-flow-type only_transcribe`
|
||||
|
||||
- optionally adjust [🦻 Speech-to-Text / 🪄 Message Type for non-threaded only-transcribed messages](./configuration/speech-to-text.md#-message-type-for-non-threaded-only-transcribed-messages), if you'd like to bot to send messages of type `notice` (for better compatibility with other bots in the room) instead of sending regular `text` messages (default)
|
||||
|
||||
### 🖌️ Image Generation
|
||||
|
||||
Image generation is the bot's ability to **generate images** based on text prompts.
|
||||
### Image Generation
|
||||
|
||||
See a [🖼️ Screenshot of the Image Generation feature](./screenshots/image-generation.webp).
|
||||
#### 🖌️ Image Creation
|
||||
|
||||
Image creation is the bot's ability to **create images** based on text prompts.
|
||||
|
||||
See a [🖼️ Screenshot of the Image Creation feature](./screenshots/image-creation.webp).
|
||||
|
||||
You may also wish to see:
|
||||
|
||||
- [🛠️ Configuration / 🖌️ Image Generation](./configuration/image-generation.md) for configuration options related to Image Generation
|
||||
- [📖 Usage / 🖌️ Image Generation](./usage.md#-image-generation) section for more details on how to use the bot for Image Generation in a room
|
||||
- [🫵 Sticker Generation](#-sticker-generation) - a special case of Image Generation
|
||||
- [📖 Usage / Image Generation / 🖌️ Creating Images](./usage.md#-creating-images) section for more details on how to use the bot for Image Creation in a room
|
||||
- [🖌️ Image Editing](#️-image-editing) - another image generation feature
|
||||
- [🫵 Sticker Creation](#-sticker-creation) - a special case of Image Creation
|
||||
|
||||
|
||||
### 🫵 Sticker Generation
|
||||
#### 🎨 Image Editing
|
||||
|
||||
Sticker generation is the bot's ability to **generate sticker** images based on text prompts. It's a special case of [🖌️ Image Generation](#️-image-generation).
|
||||
Image editing is the bot's ability to **edit images** based on a prompt and one or more existing images.
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Generation feature](./screenshots/sticker-generation.webp).
|
||||
See a [🖼️ Screenshot of the Image Editing feature (manipulating a single image)](./screenshots/image-editing-single-image.webp) and a [🖼️ Screenshot of the Image Editing feature (manipulating multiple images)](./screenshots/image-editing-multiple-images.webp).
|
||||
|
||||
See [📖 Usage / 🖌️ Image Generation / Generating Stickers](./usage.md#generating-stickers) for details.
|
||||
You may also wish to see:
|
||||
|
||||
- [🛠️ Configuration / 🖌️ Image Generation](./configuration/image-generation.md) for configuration options related to Image Generation
|
||||
- [📖 Usage / Image Generation / 🎨 Editing images](./usage.md#-editing-images) section for more details on how to use the bot for Image Editing in a room
|
||||
- [🖌️ Image Creation](#️-image-creation) - another image generation feature
|
||||
|
||||
|
||||
#### 🫵 Sticker Creation
|
||||
|
||||
Sticker generation is the bot's ability to **generate sticker** images based on text prompts. It's a special case of [🖌️ Image Creation](#️-image-creation).
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Creation feature](./screenshots/sticker-generation.webp).
|
||||
|
||||
See [📖 Usage / Image Generation / 🫵 Creating Stickers](./usage.md#-creating-stickers) for details.
|
||||
|
||||
|
||||
### 🔒 Encryption
|
||||
|
||||
@@ -53,6 +53,7 @@ CONTAINER_IMAGE_NAME=ghcr.io/etkecc/baibot:v1.0.0
|
||||
--env BAIBOT_PERSISTENCE_DATA_DIR_PATH=/data \
|
||||
--mount type=bind,src=/path/to/config.yml,dst=/app/config.yml,ro \
|
||||
--mount type=bind,src=/path/to/data,dst=/data \
|
||||
--tmpfs=/tmp:rw,noexec,nosuid,size=1024m \
|
||||
$CONTAINER_IMAGE_NAME
|
||||
```
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ The list of supported providers is below.
|
||||
|
||||
### How to choose a provider
|
||||
|
||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation), [🖌️ image-generation](./features.md#️-image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
||||
If you're not sure which provider to start with, **we recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation) (no vision), [🖌️ image-generation](./features.md#️image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
|
||||
|
||||
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
|
||||
|
||||
@@ -47,7 +47,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `anthropic`
|
||||
- 🔗 Links: [🏠 Home page](https://www.anthropic.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/Anthropic), [👤 Sign up](https://console.anthropic.com/), [📋 Models list](https://docs.anthropic.com/en/docs/about-claude/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (incl. vision)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local anthropic my-anthropic-agent`
|
||||
- create a global agent: `!bai agent create-global anthropic my-anthropic-agent`
|
||||
@@ -61,7 +61,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `groq`
|
||||
- 🔗 Links: [🏠 Home page](https://groq.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/Groq), [👤 Sign up](https://console.groq.com/login), [📋 Models list](https://console.groq.com/docs/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local groq my-groq-agent`
|
||||
- create a global agent: `!bai agent create-global groq my-groq-agent`
|
||||
@@ -75,7 +75,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `localai`
|
||||
- 🔗 Links: [🏠 Home page](https://localai.io/), [📋 Models list](https://localai.io/gallery.html)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local localai my-localai-agent`
|
||||
- create a global agent: `!bai agent create-global localai my-localai-agent`
|
||||
@@ -89,7 +89,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `mistral`
|
||||
- 🔗 Links: [🏠 Home page](https://mistral.ai/), [🌐 Wiki](https://en.wikipedia.org/wiki/Mistral_AI), [👤 Sign up](https://auth.mistral.ai/ui/registration), [📋 Models list](https://docs.mistral.ai/getting-started/models/)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local mistral my-mistral-agent`
|
||||
- create a global agent: `!bai agent create-global mistral my-mistral-agent`
|
||||
@@ -103,7 +103,7 @@ You don't need to choose just one though. The bot supports [mixing & matching mo
|
||||
|
||||
- 🆔 Identifier: `ollama`
|
||||
- 🔗 Links: [🏠 Home page](https://ollama.com/), [📋 Models list](https://ollama.com/library)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local ollama my-ollama-agent`
|
||||
- create a global agent: `!bai agent create-global ollama my-ollama-agent`
|
||||
@@ -120,15 +120,12 @@ For services which are not fully compatible with the OpenAI API, consider using
|
||||
|
||||
- 🆔 Identifier: `openai`
|
||||
- 🔗 Links: [🏠 Home page](https://openai.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/OpenAI), [👤 Sign up](https://platform.openai.com/signup), [📋 Models list](https://platform.openai.com/docs/models)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-generation), [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation), [💬 text-generation](./features.md#-text-generation) (incl. vision), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local openai my-openai-agent`
|
||||
- create a global agent: `!bai agent create-global openai my-openai-agent`
|
||||
|
||||
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which:
|
||||
|
||||
- in the general case looks [like this](./sample-provider-configs/openai.yml)
|
||||
- for the [o1](https://platform.openai.com/docs/models/o1) models needs to look [like this](./sample-provider-configs/openai-o1.yml)
|
||||
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](./sample-provider-configs/openai.yml).
|
||||
|
||||
|
||||
### OpenAI Compatible
|
||||
@@ -140,7 +137,7 @@ Some of these popular services already have **shortcut** providers (leading to t
|
||||
This provider is just as featureful as the [OpenAI](#openai) provider, but is more compatible with services which do not fully adhere to the [OpenAI API spec](https://github.com/openai/openai-openapi/).
|
||||
|
||||
- 🆔 Identifier: `openai-compatible`
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-generation), [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-creation), [💬 text-generation](./features.md#-text-generation) (no vision), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local openai-compatible my-openai-compatible-agent`
|
||||
- create a global agent: `!bai agent create-global openai-compatible my-openai-compatible-agent`
|
||||
@@ -154,7 +151,7 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
||||
|
||||
- 🆔 Identifier: `openrouter`
|
||||
- 🔗 Links: [🏠 Home page](https://openrouter.ai/), [👤 Sign up](https://openrouter.ai/), [📋 Models list](https://openrouter.ai/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local openrouter my-openrouter-agent`
|
||||
- create a global agent: `!bai agent create-global openrouter my-openrouter-agent`
|
||||
@@ -168,7 +165,7 @@ This provider is just as featureful as the [OpenAI](#openai) provider, but is mo
|
||||
|
||||
- 🆔 Identifier: `together-ai`
|
||||
- 🔗 Links: [🏠 Home page](https://www.together.ai/), [👤 Sign up](https://api.together.ai/signup), [📋 Models list](https://api.together.xyz/models)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
|
||||
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation) (no vision)
|
||||
- 🗲 Quick start:
|
||||
- create a room-local agent: `!bai agent create-room-local together-ai my-together-ai-agent`
|
||||
- create a global agent: `!bai agent create-global together-ai my-together-ai-agent`
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
base_url: https://api.anthropic.com/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: claude-3-5-sonnet-20240620
|
||||
model_id: claude-3-7-sonnet-20250219
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 8192
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
base_url: https://api.openai.com/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: o1-mini
|
||||
# o1 models do not support a system prompt
|
||||
prompt: null
|
||||
temperature: 1.0
|
||||
# o1 models do not support max_response_tokens.
|
||||
# They use `max_completion_tokens` as an alternative,
|
||||
# but we don't support it yet (see https://github.com/64bit/async-openai/issues/272).
|
||||
max_response_tokens: null
|
||||
max_context_tokens: 128000
|
||||
speech_to_text:
|
||||
model_id: whisper-1
|
||||
text_to_speech:
|
||||
model_id: tts-1-hd
|
||||
voice: onyx
|
||||
speed: 1.0
|
||||
response_format: opus
|
||||
image_generation:
|
||||
model_id: dall-e-3
|
||||
style: vivid
|
||||
size: 1024x1024
|
||||
quality: standard
|
||||
@@ -1,11 +1,14 @@
|
||||
base_url: https://api.openai.com/v1
|
||||
api_key: YOUR_API_KEY_HERE
|
||||
text_generation:
|
||||
model_id: gpt-4o
|
||||
model_id: gpt-5.2
|
||||
prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
temperature: 1.0
|
||||
max_response_tokens: 16384
|
||||
max_context_tokens: 128000
|
||||
# Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`.
|
||||
# If you're dealing with a non-reasoning model, specify `max_response_tokens` and unset `max_completion_tokens`.
|
||||
max_response_tokens: null
|
||||
max_completion_tokens: 128000
|
||||
max_context_tokens: 400000
|
||||
speech_to_text:
|
||||
model_id: whisper-1
|
||||
text_to_speech:
|
||||
@@ -14,7 +17,7 @@ text_to_speech:
|
||||
speed: 1.0
|
||||
response_format: opus
|
||||
image_generation:
|
||||
model_id: dall-e-3
|
||||
style: vivid
|
||||
size: 1024x1024
|
||||
quality: standard
|
||||
model_id: gpt-image-1
|
||||
style: null
|
||||
size: null
|
||||
quality: null
|
||||
|
||||
BIN
docs/screenshots/image-creation.webp
Normal file
BIN
docs/screenshots/image-creation.webp
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 298 KiB |
BIN
docs/screenshots/image-editing-multiple-images.webp
Normal file
BIN
docs/screenshots/image-editing-multiple-images.webp
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 339 KiB |
BIN
docs/screenshots/image-editing-single-image.webp
Normal file
BIN
docs/screenshots/image-editing-single-image.webp
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 285 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 684 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 13 KiB After Width: | Height: | Size: 22 KiB |
@@ -11,6 +11,8 @@ This is related to the [💬 Text Generation](./features.md#-text-generation) fe
|
||||
|
||||
If there's a text-generation handler agent configured, the bot **may** respond to messages sent in the room.
|
||||
|
||||
Some models also support vision, so you may be able to mix text and images in the same conversation.
|
||||
|
||||
See screenshots of:
|
||||
|
||||
- 🖼️ [the default Text Generation flow](./screenshots/text-generation.webp) in 1:1 rooms
|
||||
@@ -64,34 +66,48 @@ The speech-to-text feature triggers automatically by default, but can be adjuste
|
||||
If all your messages are in the same language, you can improve accuracy & latency by configuring the language (see [🦻 Speech-to-Text / 🔤 Language](./configuration/speech-to-text.md#-language)).
|
||||
|
||||
|
||||
### 🖌️ Image Generation
|
||||
|
||||
This is related to the [🖌️ Image Generation](./features.md#️-image-generation) feature.
|
||||
### Image Generation
|
||||
|
||||
This feature is not configurable at the moment. The configuration (size, quality, style) specified at the [🤖 agent](./agents.md) level will be used.
|
||||
|
||||
Capabilities depend on the [☁️ provider](./providers.md) and model used.
|
||||
|
||||
#### Generating images
|
||||
|
||||
Simply send a command like `!bai image A beautiful sunset over the ocean` and the bot will start a threaded conversation and post an image based on your prompt.
|
||||
#### 🖌️ Creating images
|
||||
|
||||
See a [🖼️ Screenshot of the Image Generation feature](./screenshots/image-generation.webp).
|
||||
Simply send a command like `!bai image create A beautiful sunset over the ocean` and the bot will start a threaded conversation and post an image based on your prompt.
|
||||
|
||||
You can then, respond in the same message thread with:
|
||||
See a [🖼️ Screenshot of the Image Creation feature](./screenshots/image-creation.webp).
|
||||
|
||||
You can then respond in the same message thread with:
|
||||
|
||||
- more messages, to add more criteria to your prompt.
|
||||
- a message saying `again`, to generate one more image with the current prompt.
|
||||
|
||||
|
||||
#### Generating stickers
|
||||
#### 🎨 Editing images
|
||||
|
||||
A variation of [generating images](#generating-images) is to generate "sticker images".
|
||||
Simply send a command like `!bai image edit Turn the following image into an anime-style drawing` and the bot will start a threaded conversation asking for more details.
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Generation feature](./screenshots/sticker-generation.webp).
|
||||
See a [🖼️ Screenshot of the Image Editing feature (manipulating a single image)](./screenshots/image-editing-single-image.webp) and a [🖼️ Screenshot of the Image Editing feature (manipulating multiple images)](./screenshots/image-editing-multiple-images.webp).
|
||||
|
||||
To generate a sticker, send a command like `!bai sticker A huge ramen bowl with lots of chashu and a mountain of beansprouts on top`.
|
||||
You can then respond in the same message thread with:
|
||||
|
||||
The difference from [generating images](#generating-images) is that the bot will:
|
||||
- more messages, to add more criteria to your prompt.
|
||||
- one or more images, to provide the images that the bot will operate on.
|
||||
- a message saying `go`, to start the image generation process.
|
||||
- a message saying `again`, to prompt the bot to generate one more image edit with the current prompt.
|
||||
|
||||
|
||||
#### 🫵 Creating stickers
|
||||
|
||||
A variation of [creating images](#creating-images) is creating "sticker images".
|
||||
|
||||
See a [🖼️ Screenshot of the Sticker Creation feature](./screenshots/sticker-generation.webp).
|
||||
|
||||
To create a sticker, send a command like `!bai sticker A huge ramen bowl with lots of chashu and a mountain of beansprouts on top`.
|
||||
|
||||
The difference from [creating images](#creating-images) is that the bot will:
|
||||
|
||||
- generate a smaller-resolution image (currently hardcoded to `256x256`) - smaller/quicker, but still good enough for a sticker
|
||||
- potentially switch to a different (cheaper or otherwise more suitable) model, if available
|
||||
|
||||
@@ -11,6 +11,12 @@ user:
|
||||
# Leave empty to use the default (baibot).
|
||||
name: baibot
|
||||
|
||||
# An optional path to an image file to be used as a custom avatar image.
|
||||
# - null or empty string: use the default avatar
|
||||
# - "keep": don't touch the avatar, keep whatever is already set
|
||||
# - any other value: path to a custom avatar image file
|
||||
avatar: null
|
||||
|
||||
encryption:
|
||||
# An optional passphrase to use for backing up and recovering the bot's encryption keys.
|
||||
# You can use any string here.
|
||||
@@ -32,6 +38,10 @@ user:
|
||||
# Command prefix. Leave empty to use the default (!bai).
|
||||
command_prefix: "!bai"
|
||||
|
||||
room:
|
||||
# Whether the bot should send an introduction message after joining a room.
|
||||
post_join_self_introduction_enabled: true
|
||||
|
||||
access:
|
||||
# Space-separated list of MXID patterns which specify who is an admin.
|
||||
admin_patterns:
|
||||
@@ -72,11 +82,14 @@ agents:
|
||||
# base_url: https://api.openai.com/v1
|
||||
# api_key: ""
|
||||
# text_generation:
|
||||
# model_id: gpt-4o
|
||||
# model_id: gpt-5.2
|
||||
# prompt: "You are a brief, but helpful bot called {{ baibot_name }} powered by the {{ baibot_model_id }} model. The date/time of this conversation's start is: {{ baibot_conversation_start_time_utc }}."
|
||||
# temperature: 1.0
|
||||
# max_response_tokens: 16384
|
||||
# max_context_tokens: 128000
|
||||
# # Reasoning models need to use `max_completion_tokens` instead of `max_response_tokens`.
|
||||
# # If you're dealing with a non-reasoning model, specify `max_response_tokens` and unset `max_completion_tokens`.
|
||||
# max_response_tokens: null
|
||||
# max_completion_tokens: 128000
|
||||
# max_context_tokens: 400000
|
||||
# speech_to_text:
|
||||
# model_id: whisper-1
|
||||
# text_to_speech:
|
||||
@@ -85,10 +98,10 @@ agents:
|
||||
# speed: 1.0
|
||||
# response_format: opus
|
||||
# image_generation:
|
||||
# model_id: dall-e-3
|
||||
# style: vivid
|
||||
# size: 1024x1024
|
||||
# quality: standard
|
||||
# model_id: gpt-image-1
|
||||
# style: null
|
||||
# size: null
|
||||
# quality: null
|
||||
#
|
||||
# - id: localai
|
||||
# provider: localai
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
postgres:
|
||||
image: docker.io/postgres:16.4-alpine
|
||||
image: docker.io/postgres:18.1-alpine
|
||||
user: ${UID}:${GID}
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
@@ -8,12 +8,13 @@ services:
|
||||
POSTGRES_PASSWORD: synapse-password
|
||||
POSTGRES_DB: homeserver
|
||||
POSTGRES_INITDB_ARGS: --lc-collate C --lc-ctype C --encoding UTF8
|
||||
PGDATA: /data
|
||||
volumes:
|
||||
- ./postgres:/var/lib/postgresql/data
|
||||
- ./postgres:/data
|
||||
- /etc/passwd:/etc/passwd:ro
|
||||
|
||||
synapse:
|
||||
image: ghcr.io/element-hq/synapse:v1.118.0
|
||||
image: ghcr.io/element-hq/synapse:v1.144.0
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
entrypoint: python
|
||||
@@ -26,14 +27,20 @@ services:
|
||||
- ./synapse/media-store:/media-store
|
||||
|
||||
element-web:
|
||||
image: docker.io/vectorim/element-web:v1.11.84
|
||||
image: ghcr.io/element-hq/element-web:v1.12.6
|
||||
user: "${UID}:${GID}"
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
ELEMENT_WEB_PORT: 8080
|
||||
ports:
|
||||
- "${SERVICE_ELEMENT_WEB_BIND_PORT_HTTP}:8080"
|
||||
volumes:
|
||||
- ../../etc/services/core/element-web/nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
- ../../etc/services/core/element-web/config.json:/app/config.json:ro
|
||||
tmpfs:
|
||||
- /var/cache/nginx:rw,mode=777
|
||||
- /var/run:rw,mode=777
|
||||
- /tmp/element-web-config:rw,mode=777
|
||||
- /etc/nginx/conf.d:rw,mode=777
|
||||
|
||||
networks:
|
||||
default:
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
"default_is_url": "https://vector.im",
|
||||
"integrations_ui_url": "https://scalar.vector.im/",
|
||||
"integrations_rest_url": "https://scalar.vector.im/api",
|
||||
"bug_report_endpoint_url": "https://riot.im/bugreports/submit",
|
||||
"bug_report_endpoint_url": "https://element.io/bugreports/submit",
|
||||
"enableLabs": true,
|
||||
"roomDirectory": {
|
||||
"servers": [
|
||||
|
||||
@@ -1,60 +0,0 @@
|
||||
# This is a custom nginx configuration file that we use in the container (instead of the default one),
|
||||
# because it allows us to run nginx with a non-root user.
|
||||
#
|
||||
# For this to work, the default vhost file (`/etc/nginx/conf.d/default.conf`) also needs to be removed.
|
||||
# (mounting `/dev/null` over `/etc/nginx/conf.d/default.conf` works well)
|
||||
#
|
||||
# The following changes have been done compared to a default nginx configuration file:
|
||||
# - default server port is changed (80 -> 8080), so that a non-root user can bind it
|
||||
# - various temp paths are changed to `/tmp`, so that a non-root user can write to them
|
||||
# - the `user` directive was removed, as we don't want nginx to switch users
|
||||
|
||||
worker_processes 1;
|
||||
|
||||
error_log /var/log/nginx/error.log warn;
|
||||
pid /tmp/nginx.pid;
|
||||
|
||||
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
|
||||
http {
|
||||
client_body_temp_path /tmp/client_body_temp;
|
||||
proxy_temp_path /tmp/proxy_temp;
|
||||
fastcgi_temp_path /tmp/fastcgi_temp;
|
||||
uwsgi_temp_path /tmp/uwsgi_temp;
|
||||
scgi_temp_path /tmp/scgi_temp;
|
||||
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
|
||||
log_format main '$remote_addr - $remote_user [$time_local] "$request" '
|
||||
'$status $body_bytes_sent "$http_referer" '
|
||||
'"$http_user_agent" "$http_x_forwarded_for"';
|
||||
|
||||
access_log /var/log/nginx/access.log main;
|
||||
|
||||
sendfile on;
|
||||
#tcp_nopush on;
|
||||
|
||||
keepalive_timeout 65;
|
||||
|
||||
#gzip on;
|
||||
|
||||
server {
|
||||
listen 8080;
|
||||
server_name localhost;
|
||||
|
||||
location / {
|
||||
root /usr/share/nginx/html;
|
||||
index index.html index.htm;
|
||||
}
|
||||
|
||||
error_page 500 502 503 504 /50x.html;
|
||||
location = /50x.html {
|
||||
root /usr/share/nginx/html;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -579,9 +579,7 @@ rc_login:
|
||||
#
|
||||
#federation_rr_transactions_per_room_per_second: 50
|
||||
|
||||
# Authenticated media is not supported yet.
|
||||
# See: https://github.com/etkecc/baibot/issues/12
|
||||
enable_authenticated_media: false
|
||||
enable_authenticated_media: true
|
||||
|
||||
# Directory where uploaded images and attachments are stored.
|
||||
#
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
ollama:
|
||||
image: docker.io/ollama/ollama:0.4.1
|
||||
image: docker.io/ollama/ollama:0.13.3
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
|
||||
|
||||
4
justfile
4
justfile
@@ -32,6 +32,10 @@ run-in-container *extra_args: app-container-prepare build-container-image-debug
|
||||
test *extra_args:
|
||||
RUST_BACKTRACE=1 cargo test {{ extra_args }}
|
||||
|
||||
# Formats the code
|
||||
fmt:
|
||||
RUST_BACKTRACE=1 cargo fmt --all
|
||||
|
||||
# Builds a debug binary (target/debug/*)
|
||||
build-debug *extra_args:
|
||||
RUST_BACKTRACE=1 cargo build {{ extra_args }}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use super::{
|
||||
provider::{self, ControllerType},
|
||||
AgentDefinition, AgentProvider, PublicIdentifier,
|
||||
provider::{self, ControllerType},
|
||||
};
|
||||
|
||||
// Dead-code is allowed. We do not use these enum struct payloads directly,
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use super::instantiation;
|
||||
use super::instantiation::AgentInstance;
|
||||
use super::AgentDefinition;
|
||||
use super::PublicIdentifier;
|
||||
use super::instantiation;
|
||||
use super::instantiation::AgentInstance;
|
||||
use crate::entity::RoomConfigContext;
|
||||
|
||||
#[derive(Debug)]
|
||||
|
||||
@@ -11,11 +11,11 @@ pub use manager::Manager;
|
||||
|
||||
pub use definition::AgentDefinition;
|
||||
|
||||
pub use instantiation::create_from_provider_and_yaml_value_config;
|
||||
pub use instantiation::default_config_for_provider;
|
||||
pub use instantiation::AgentInstance;
|
||||
pub use instantiation::Error as AgentInstantiationError;
|
||||
pub use instantiation::Result as AgentInstantiationResult;
|
||||
pub use instantiation::create_from_provider_and_yaml_value_config;
|
||||
pub use instantiation::default_config_for_provider;
|
||||
|
||||
pub use provider::{AgentProvider, AgentProviderInfo, ControllerTrait};
|
||||
pub use purpose::AgentPurpose;
|
||||
|
||||
@@ -1,7 +1,5 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use anthropic_rs::models::claude::ClaudeModel;
|
||||
|
||||
use crate::agent::{default_prompt, provider::ConfigTrait};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -28,6 +26,9 @@ impl ConfigTrait for Config {
|
||||
if self.base_url.is_empty() {
|
||||
return Err("The base URL must not be empty.".to_owned());
|
||||
}
|
||||
if !self.base_url.ends_with("/v1") {
|
||||
return Err("The base URL must end with '/v1'.".to_owned());
|
||||
}
|
||||
if self.api_key.is_empty() {
|
||||
return Err("The API key must not be empty.".to_owned());
|
||||
}
|
||||
@@ -67,5 +68,5 @@ impl Default for TextGenerationConfig {
|
||||
}
|
||||
|
||||
fn default_text_model_id() -> String {
|
||||
ClaudeModel::Claude35Sonnet.as_str().to_owned()
|
||||
"claude-3-7-sonnet-20250219".to_owned()
|
||||
}
|
||||
|
||||
@@ -1,30 +1,28 @@
|
||||
use std::fmt::Debug;
|
||||
use std::str::FromStr;
|
||||
use std::sync::Arc;
|
||||
|
||||
use anthropic_rs::completion::message::{ContentType, System};
|
||||
use anthropic_rs::{
|
||||
client::Client as AnthropicClient, config::Config as AnthropicConfig,
|
||||
models::claude::ClaudeModel,
|
||||
};
|
||||
use anthropic::client::{Client, ClientBuilder};
|
||||
use anthropic::types::ContentBlock;
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::agent::provider::entity::{
|
||||
ImageGenerationResult, PingResult, TextGenerationParams, TextGenerationResult,
|
||||
TextToSpeechParams, TextToSpeechResult,
|
||||
};
|
||||
use crate::agent::provider::{ImageGenerationParams, SpeechToTextParams, SpeechToTextResult};
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
|
||||
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
|
||||
};
|
||||
use crate::agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
};
|
||||
use crate::conversation::llm::{
|
||||
shorten_messages_list_to_context_size, Author as LLMAuthor, Conversation as LLMConversation,
|
||||
Message as LLMMessage,
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
};
|
||||
use crate::strings;
|
||||
|
||||
use super::config::Config;
|
||||
|
||||
struct ControllerInner {
|
||||
client: AnthropicClient,
|
||||
client: Client,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
@@ -43,18 +41,20 @@ impl Debug for Controller {
|
||||
|
||||
impl Controller {
|
||||
pub fn new(config: Config) -> anyhow::Result<Self> {
|
||||
let anthropic_config =
|
||||
AnthropicConfig::new(config.api_key.clone()).with_base_url(config.base_url.clone());
|
||||
// The previous library that we used expected a base URL that ends with "/v1"
|
||||
// (e.g. "https://api.anthropic.com/v1"), while the new one doesn't.
|
||||
//
|
||||
// To keep backward compatibility, we don't ask people to change their configuration
|
||||
// and rather adapt by removing the "/v1" from the base URL.
|
||||
if !config.base_url.ends_with("/v1") {
|
||||
return Err(anyhow::anyhow!("base_url must end with '/v1'"));
|
||||
}
|
||||
|
||||
let client = match AnthropicClient::new(anthropic_config) {
|
||||
Ok(client) => client,
|
||||
Err(err) => {
|
||||
return Err(anyhow::anyhow!(
|
||||
"Failed to create Anthropic client: {}",
|
||||
err.to_string()
|
||||
));
|
||||
}
|
||||
};
|
||||
let base_url = &config.base_url[..config.base_url.len() - 3];
|
||||
let client = ClientBuilder::default()
|
||||
.api_base(base_url.to_string())
|
||||
.api_key(config.api_key.clone())
|
||||
.build()?;
|
||||
|
||||
Ok(Self {
|
||||
config,
|
||||
@@ -71,7 +71,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
message_text: "Hello!".to_string(),
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
@@ -108,7 +108,7 @@ impl ControllerTrait for Controller {
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
message_text: prompt_text,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
@@ -142,29 +142,19 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let mut request = super::utils::create_anthropic_message_request(conversation_messages);
|
||||
|
||||
let model = match ClaudeModel::from_str(&text_generation_config.model_id) {
|
||||
Ok(model) => model,
|
||||
Err(err) => {
|
||||
tracing::debug!(?err, "Failed to parse model ID");
|
||||
|
||||
return Err(anyhow::anyhow!(
|
||||
"Failed to parse model ID: {}",
|
||||
&text_generation_config.model_id
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
let temperature = params
|
||||
.temperature_override
|
||||
.unwrap_or(text_generation_config.temperature);
|
||||
|
||||
if let Some(prompt_message) = prompt_message {
|
||||
request.system = Some(System::Text(prompt_message.message_text));
|
||||
if let LLMMessageContent::Text(text) = &prompt_message.content {
|
||||
request.system = text.clone();
|
||||
}
|
||||
}
|
||||
|
||||
request.model = model;
|
||||
request.temperature = Some(temperature);
|
||||
request.max_tokens = text_generation_config.max_response_tokens;
|
||||
request.model = text_generation_config.model_id.clone();
|
||||
request.temperature = Some(temperature as f64);
|
||||
request.max_tokens = text_generation_config.max_response_tokens as usize;
|
||||
|
||||
if let Ok(request_as_json) = serde_json::to_string(&request) {
|
||||
tracing::trace!(
|
||||
@@ -175,19 +165,20 @@ impl ControllerTrait for Controller {
|
||||
);
|
||||
}
|
||||
|
||||
let response = self.inner.client.create_message(request).await?;
|
||||
let response = self.inner.client.messages(request).await?;
|
||||
|
||||
tracing::trace!(?response, "Got response from Anthropic create message API");
|
||||
|
||||
// response.content usually contains a single element, but we support handling multiple to account for all possibilities
|
||||
let mut text_parts = vec![];
|
||||
for content in response.content {
|
||||
let content_type = content.content_type;
|
||||
|
||||
match content_type {
|
||||
ContentType::Text => {
|
||||
text_parts.push(content.text);
|
||||
} // There are no other content types to handle yet, but there may be in the future
|
||||
match content {
|
||||
ContentBlock::Text { text } => {
|
||||
text_parts.push(text);
|
||||
}
|
||||
ContentBlock::Image { .. } => {
|
||||
text_parts.push("The model responded with an image".to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -219,6 +210,15 @@ impl ControllerTrait for Controller {
|
||||
Err(anyhow::anyhow!("Image generation not supported"))
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
_prompt: &str,
|
||||
_images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
Err(anyhow::anyhow!("Image editing is not supported"))
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
_input: &str,
|
||||
|
||||
@@ -7,8 +7,8 @@ pub use controller::Controller;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::controller::ControllerType;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
use anthropic_rs::completion::message::{Content, ContentType, Message, MessageRequest, Role};
|
||||
use anthropic::types::{
|
||||
ContentBlock, ImageSource, Message, MessagesRequest, MessagesRequestBuilder, Role,
|
||||
};
|
||||
|
||||
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) -> MessageRequest {
|
||||
pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) -> MessagesRequest {
|
||||
let mut messages = vec![];
|
||||
|
||||
for message in llm_messages {
|
||||
@@ -14,19 +18,26 @@ pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) ->
|
||||
}
|
||||
};
|
||||
|
||||
let content = vec![Content {
|
||||
content_type: ContentType::Text,
|
||||
text: message.message_text,
|
||||
}];
|
||||
let content = match &message.content {
|
||||
LLMMessageContent::Text(text) => vec![ContentBlock::Text { text: text.clone() }],
|
||||
LLMMessageContent::Image(image_details) => {
|
||||
vec![ContentBlock::Image {
|
||||
source: ImageSource::Base64 {
|
||||
media_type: image_details.mime.to_string(),
|
||||
data: crate::utils::base64::base64_encode(&image_details.data),
|
||||
},
|
||||
}]
|
||||
}
|
||||
};
|
||||
|
||||
let message = Message { role, content };
|
||||
|
||||
messages.push(message);
|
||||
}
|
||||
|
||||
MessageRequest {
|
||||
stream: false,
|
||||
messages,
|
||||
..Default::default()
|
||||
}
|
||||
MessagesRequestBuilder::default()
|
||||
.messages(messages)
|
||||
.stream(false)
|
||||
.build()
|
||||
.expect("Failed to build messages request")
|
||||
}
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
use crate::{agent::AgentPurpose, conversation::llm::Conversation};
|
||||
|
||||
use super::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
entity::{
|
||||
ImageGenerationResult, PingResult, TextGenerationParams, TextGenerationResult,
|
||||
TextToSpeechParams, TextToSpeechResult,
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult, TextGenerationParams,
|
||||
TextGenerationResult, TextToSpeechParams, TextToSpeechResult,
|
||||
},
|
||||
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
};
|
||||
|
||||
pub trait ControllerTrait {
|
||||
@@ -42,6 +42,13 @@ pub trait ControllerTrait {
|
||||
params: ImageGenerationParams,
|
||||
) -> impl std::future::Future<Output = anyhow::Result<ImageGenerationResult>> + Send;
|
||||
|
||||
fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
params: ImageEditParams,
|
||||
) -> impl std::future::Future<Output = anyhow::Result<ImageEditResult>> + Send;
|
||||
|
||||
fn text_to_speech(
|
||||
&self,
|
||||
text: &str,
|
||||
@@ -166,6 +173,25 @@ impl ControllerTrait for ControllerType {
|
||||
}
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
match &self {
|
||||
ControllerType::OpenAI(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
ControllerType::OpenAICompat(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
ControllerType::Anthropic(controller) => {
|
||||
controller.create_image_edit(prompt, images, params).await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
text: &str,
|
||||
|
||||
@@ -67,9 +67,8 @@ impl AgentProvider {
|
||||
wiki_url: Some("https://en.wikipedia.org/wiki/Anthropic"),
|
||||
sign_up_url: Some("https://console.anthropic.com/"),
|
||||
models_list_url: Some("https://docs.anthropic.com/en/docs/about-claude/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: true,
|
||||
},
|
||||
Self::Groq => AgentProviderInfo {
|
||||
id: Self::Groq.to_static_str(),
|
||||
@@ -79,10 +78,8 @@ impl AgentProvider {
|
||||
wiki_url: Some("https://en.wikipedia.org/wiki/Groq"),
|
||||
sign_up_url: Some("https://console.groq.com/login"),
|
||||
models_list_url: Some("https://console.groq.com/docs/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration, AgentPurpose::SpeechToText],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
Self::LocalAI => AgentProviderInfo {
|
||||
id: Self::LocalAI.to_static_str(),
|
||||
@@ -97,6 +94,7 @@ impl AgentProvider {
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
Self::Mistral => AgentProviderInfo {
|
||||
id: Self::Mistral.to_static_str(),
|
||||
@@ -106,9 +104,8 @@ impl AgentProvider {
|
||||
wiki_url: Some("https://en.wikipedia.org/wiki/Mistral_AI"),
|
||||
sign_up_url: Some("https://auth.mistral.ai/ui/registration"),
|
||||
models_list_url: Some("https://docs.mistral.ai/getting-started/models/"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
Self::Ollama => AgentProviderInfo {
|
||||
id: Self::Ollama.to_static_str(),
|
||||
@@ -118,9 +115,8 @@ impl AgentProvider {
|
||||
wiki_url: None,
|
||||
sign_up_url: None,
|
||||
models_list_url: Some("https://ollama.com/library"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
Self::OpenAI => AgentProviderInfo {
|
||||
id: Self::OpenAI.to_static_str(),
|
||||
@@ -136,6 +132,7 @@ impl AgentProvider {
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: true,
|
||||
},
|
||||
Self::OpenAICompat => AgentProviderInfo {
|
||||
id: Self::OpenAICompat.to_static_str(),
|
||||
@@ -151,6 +148,7 @@ impl AgentProvider {
|
||||
AgentPurpose::TextToSpeech,
|
||||
AgentPurpose::SpeechToText,
|
||||
],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
Self::OpenRouter => AgentProviderInfo {
|
||||
id: Self::OpenRouter.to_static_str(),
|
||||
@@ -160,9 +158,8 @@ impl AgentProvider {
|
||||
wiki_url: None,
|
||||
sign_up_url: Some("https://openrouter.ai/"),
|
||||
models_list_url: Some("https://openrouter.ai/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
Self::TogetherAI => AgentProviderInfo {
|
||||
id: Self::TogetherAI.to_static_str(),
|
||||
@@ -172,9 +169,8 @@ impl AgentProvider {
|
||||
wiki_url: None,
|
||||
sign_up_url: Some("https://api.together.ai/signup"),
|
||||
models_list_url: Some("https://api.together.xyz/models"),
|
||||
supported_purposes: vec![
|
||||
AgentPurpose::TextGeneration,
|
||||
],
|
||||
supported_purposes: vec![AgentPurpose::TextGeneration],
|
||||
text_generation_supports_vision: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -195,4 +191,5 @@ pub struct AgentProviderInfo {
|
||||
pub sign_up_url: Option<&'static str>,
|
||||
pub models_list_url: Option<&'static str>,
|
||||
pub supported_purposes: Vec<AgentPurpose>,
|
||||
pub text_generation_supports_vision: bool,
|
||||
}
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
use mxlink::mime;
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ImageGenerationParams {
|
||||
pub size_override: Option<String>,
|
||||
@@ -26,6 +28,36 @@ impl ImageGenerationParams {
|
||||
|
||||
pub struct ImageGenerationResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mxlink::mime::Mime,
|
||||
pub mime_type: mime::Mime,
|
||||
pub revised_prompt: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ImageEditParams {}
|
||||
|
||||
pub struct ImageEditResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mime::Mime,
|
||||
}
|
||||
|
||||
pub struct ImageSource {
|
||||
pub filename: String,
|
||||
pub bytes: Vec<u8>,
|
||||
pub mime_type: mime::Mime,
|
||||
}
|
||||
|
||||
impl ImageSource {
|
||||
pub fn new(filename: String, bytes: Vec<u8>, mime_type: mime::Mime) -> Self {
|
||||
Self {
|
||||
filename,
|
||||
bytes,
|
||||
mime_type,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<ImageSource> for async_openai::types::images::ImageInput {
|
||||
fn from(value: ImageSource) -> Self {
|
||||
async_openai::types::images::ImageInput::from_vec_u8(value.filename, value.bytes)
|
||||
}
|
||||
}
|
||||
@@ -1,12 +1,14 @@
|
||||
mod agent_provider;
|
||||
mod image_generation;
|
||||
mod image;
|
||||
mod ping;
|
||||
mod speech_to_text;
|
||||
mod text_generation;
|
||||
mod text_to_speech;
|
||||
|
||||
pub use agent_provider::{AgentProvider, AgentProviderInfo};
|
||||
pub use image_generation::{ImageGenerationParams, ImageGenerationResult};
|
||||
pub use image::{
|
||||
ImageEditParams, ImageEditResult, ImageGenerationParams, ImageGenerationResult, ImageSource,
|
||||
};
|
||||
pub use ping::PingResult;
|
||||
pub use speech_to_text::{SpeechToTextParams, SpeechToTextResult};
|
||||
pub use text_generation::{
|
||||
|
||||
@@ -20,6 +20,7 @@ pub use controller::{ControllerTrait, ControllerType};
|
||||
pub use config::ConfigTrait;
|
||||
|
||||
pub use entity::{
|
||||
AgentProvider, AgentProviderInfo, ImageGenerationParams, PingResult, SpeechToTextParams,
|
||||
SpeechToTextResult, TextGenerationParams, TextGenerationPromptVariables, TextToSpeechParams,
|
||||
AgentProvider, AgentProviderInfo, ImageEditParams, ImageGenerationParams, ImageSource,
|
||||
PingResult, SpeechToTextParams, SpeechToTextResult, TextGenerationParams,
|
||||
TextGenerationPromptVariables, TextToSpeechParams,
|
||||
};
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::OPENAI_IMAGE_MODEL_GPT_IMAGE_1;
|
||||
use crate::agent::{default_prompt, provider::ConfigTrait};
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -58,6 +59,9 @@ pub struct TextGenerationConfig {
|
||||
#[serde(default)]
|
||||
pub max_response_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_completion_tokens: Option<u32>,
|
||||
|
||||
#[serde(default)]
|
||||
pub max_context_tokens: u32,
|
||||
}
|
||||
@@ -68,14 +72,15 @@ impl Default for TextGenerationConfig {
|
||||
model_id: default_text_model_id(),
|
||||
prompt: Some(default_prompt().to_owned()),
|
||||
temperature: super::super::default_temperature(),
|
||||
max_response_tokens: Some(16_384),
|
||||
max_context_tokens: 128_000,
|
||||
max_response_tokens: None,
|
||||
max_completion_tokens: Some(128_000),
|
||||
max_context_tokens: 400_000,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_model_id() -> String {
|
||||
"gpt-4o".to_owned()
|
||||
"gpt-5.2".to_owned()
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -99,16 +104,16 @@ fn default_speech_to_text_model_id() -> String {
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct TextToSpeechConfig {
|
||||
#[serde(default = "default_text_to_speech_model_id")]
|
||||
pub model_id: async_openai::types::SpeechModel,
|
||||
pub model_id: async_openai::types::audio::SpeechModel,
|
||||
|
||||
#[serde(default = "default_text_to_speech_voice")]
|
||||
pub voice: async_openai::types::Voice,
|
||||
pub voice: async_openai::types::audio::Voice,
|
||||
|
||||
#[serde(default = "default_text_to_speech_speed")]
|
||||
pub speed: f32,
|
||||
|
||||
#[serde(default = "default_text_to_speech_response_format")]
|
||||
pub response_format: async_openai::types::SpeechResponseFormat,
|
||||
pub response_format: async_openai::types::audio::SpeechResponseFormat,
|
||||
}
|
||||
|
||||
impl Default for TextToSpeechConfig {
|
||||
@@ -122,22 +127,22 @@ impl Default for TextToSpeechConfig {
|
||||
}
|
||||
}
|
||||
|
||||
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel {
|
||||
async_openai::types::SpeechModel::Tts1Hd
|
||||
fn default_text_to_speech_model_id() -> async_openai::types::audio::SpeechModel {
|
||||
async_openai::types::audio::SpeechModel::Tts1Hd
|
||||
}
|
||||
|
||||
fn default_text_to_speech_voice() -> async_openai::types::Voice {
|
||||
async_openai::types::Voice::Onyx
|
||||
fn default_text_to_speech_voice() -> async_openai::types::audio::Voice {
|
||||
async_openai::types::audio::Voice::Onyx
|
||||
}
|
||||
|
||||
fn default_text_to_speech_speed() -> f32 {
|
||||
1.0
|
||||
}
|
||||
|
||||
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat {
|
||||
fn default_text_to_speech_response_format() -> async_openai::types::audio::SpeechResponseFormat {
|
||||
// The API defaults to mp3, but we prefer Opus because it's smaller.
|
||||
// Our clients should all have support for it.
|
||||
async_openai::types::SpeechResponseFormat::Opus
|
||||
async_openai::types::audio::SpeechResponseFormat::Opus
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -145,19 +150,19 @@ pub struct ImageGenerationConfig {
|
||||
pub model_id: String,
|
||||
|
||||
#[serde(default = "default_image_style")]
|
||||
pub style: async_openai::types::ImageStyle,
|
||||
pub style: Option<async_openai::types::images::ImageStyle>,
|
||||
|
||||
#[serde(default = "default_image_size")]
|
||||
pub size: async_openai::types::ImageSize,
|
||||
pub size: Option<async_openai::types::images::ImageSize>,
|
||||
|
||||
#[serde(default = "default_image_quality")]
|
||||
pub quality: async_openai::types::ImageQuality,
|
||||
pub quality: Option<async_openai::types::images::ImageQuality>,
|
||||
}
|
||||
|
||||
impl Default for ImageGenerationConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
model_id: "dall-e-3".to_owned(),
|
||||
model_id: OPENAI_IMAGE_MODEL_GPT_IMAGE_1.to_owned(),
|
||||
style: default_image_style(),
|
||||
size: default_image_size(),
|
||||
quality: default_image_quality(),
|
||||
@@ -168,23 +173,25 @@ impl Default for ImageGenerationConfig {
|
||||
impl ImageGenerationConfig {
|
||||
pub fn model_id_as_openai_image_model(
|
||||
&self,
|
||||
) -> Result<async_openai::types::ImageModel, String> {
|
||||
) -> Result<async_openai::types::images::ImageModel, String> {
|
||||
match self.model_id.as_str() {
|
||||
"dall-e-2" => Ok(async_openai::types::ImageModel::DallE2),
|
||||
"dall-e-3" => Ok(async_openai::types::ImageModel::DallE3),
|
||||
other => Ok(async_openai::types::ImageModel::Other(other.to_owned())),
|
||||
"dall-e-2" => Ok(async_openai::types::images::ImageModel::DallE2),
|
||||
"dall-e-3" => Ok(async_openai::types::images::ImageModel::DallE3),
|
||||
"gpt-image-1" => Ok(async_openai::types::images::ImageModel::GptImage1),
|
||||
"gpt-image-1-mini" => Ok(async_openai::types::images::ImageModel::GptImage1Mini),
|
||||
other => Ok(async_openai::types::images::ImageModel::Other(other.to_owned())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn default_image_style() -> async_openai::types::ImageStyle {
|
||||
async_openai::types::ImageStyle::Vivid
|
||||
fn default_image_style() -> Option<async_openai::types::images::ImageStyle> {
|
||||
None
|
||||
}
|
||||
|
||||
fn default_image_size() -> async_openai::types::ImageSize {
|
||||
async_openai::types::ImageSize::S1024x1024
|
||||
fn default_image_size() -> Option<async_openai::types::images::ImageSize> {
|
||||
None
|
||||
}
|
||||
|
||||
fn default_image_quality() -> async_openai::types::ImageQuality {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
fn default_image_quality() -> Option<async_openai::types::images::ImageQuality> {
|
||||
None
|
||||
}
|
||||
|
||||
@@ -1,37 +1,42 @@
|
||||
use std::ops::Deref;
|
||||
|
||||
use async_openai::{
|
||||
Client as OpenAIClient,
|
||||
config::OpenAIConfig,
|
||||
types::{
|
||||
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageRequestArgs,
|
||||
CreateSpeechRequestArgs, CreateTranscriptionRequestArgs,
|
||||
audio::{AudioInput, CreateSpeechRequestArgs, CreateTranscriptionRequestArgs},
|
||||
chat::{ChatCompletionRequestMessage, CreateChatCompletionRequestArgs},
|
||||
images::{
|
||||
CreateImageEditRequestArgs, CreateImageRequestArgs,
|
||||
Image, ImageInput, ImageModel, ImageResponseFormat,
|
||||
},
|
||||
},
|
||||
Client as OpenAIClient,
|
||||
};
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::{
|
||||
agent::{
|
||||
provider::{
|
||||
entity::{ImageGenerationResult, PingResult, TextToSpeechParams, TextToSpeechResult},
|
||||
openai::utils::convert_string_to_enum,
|
||||
},
|
||||
AgentPurpose,
|
||||
agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
entity::{TextGenerationParams, TextGenerationResult},
|
||||
},
|
||||
strings,
|
||||
conversation::llm::{
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
},
|
||||
utils::base64::base64_decode,
|
||||
};
|
||||
use crate::{
|
||||
agent::{
|
||||
AgentPurpose,
|
||||
provider::{
|
||||
entity::{TextGenerationParams, TextGenerationResult},
|
||||
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
entity::{
|
||||
ImageEditResult, ImageGenerationResult, ImageSource, PingResult,
|
||||
TextToSpeechParams, TextToSpeechResult,
|
||||
},
|
||||
openai::utils::convert_string_to_enum,
|
||||
},
|
||||
utils::base64_decode,
|
||||
},
|
||||
conversation::llm::{
|
||||
shorten_messages_list_to_context_size, Author as LLMAuthor,
|
||||
Conversation as LLMConversation, Message as LLMMessage,
|
||||
},
|
||||
strings,
|
||||
};
|
||||
|
||||
use super::config::Config;
|
||||
@@ -62,7 +67,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
message_text: "Hello!".to_string(),
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
@@ -99,7 +104,7 @@ impl ControllerTrait for Controller {
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
message_text: prompt_text,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
@@ -144,6 +149,10 @@ impl ControllerTrait for Controller {
|
||||
request_builder.max_tokens(max_response_tokens);
|
||||
}
|
||||
|
||||
if let Some(max_completion_tokens) = text_generation_config.max_completion_tokens {
|
||||
request_builder.max_completion_tokens(max_completion_tokens);
|
||||
}
|
||||
|
||||
let request = request_builder.build()?;
|
||||
|
||||
if let Ok(request_as_json) = serde_json::to_string(&request) {
|
||||
@@ -201,12 +210,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let request = CreateTranscriptionRequestArgs::default()
|
||||
.model(&speech_to_text_config.model_id)
|
||||
.file(async_openai::types::AudioInput {
|
||||
source: async_openai::types::InputSource::VecU8 {
|
||||
filename,
|
||||
vec: media,
|
||||
},
|
||||
})
|
||||
.file(AudioInput::from_vec_u8(filename, media))
|
||||
.language(language.clone())
|
||||
.build()?;
|
||||
|
||||
@@ -216,7 +220,7 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI speech-to-text API request"
|
||||
);
|
||||
|
||||
let response = self.client.audio().transcribe(request).await?;
|
||||
let response = self.client.audio().transcription().create(request).await?;
|
||||
|
||||
tracing::trace!(
|
||||
?response,
|
||||
@@ -248,11 +252,12 @@ impl ControllerTrait for Controller {
|
||||
let model = if params.cheaper_model_switching_allowed {
|
||||
// Switch to a cheaper model
|
||||
match original_model {
|
||||
async_openai::types::ImageModel::DallE2 => async_openai::types::ImageModel::DallE2,
|
||||
async_openai::types::ImageModel::DallE3 => async_openai::types::ImageModel::DallE2,
|
||||
async_openai::types::ImageModel::Other(_) => {
|
||||
async_openai::types::ImageModel::DallE2
|
||||
ImageModel::DallE2 => ImageModel::DallE2,
|
||||
ImageModel::DallE3 => ImageModel::DallE2,
|
||||
ImageModel::Other(_) => {
|
||||
ImageModel::DallE2
|
||||
}
|
||||
_ => original_model.clone(),
|
||||
}
|
||||
} else {
|
||||
original_model
|
||||
@@ -261,12 +266,28 @@ impl ControllerTrait for Controller {
|
||||
let quality = if params.cheaper_quality_switching_allowed {
|
||||
// Switch to a cheaper quality
|
||||
match &image_generation_config.quality {
|
||||
async_openai::types::ImageQuality::Standard => {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
}
|
||||
async_openai::types::ImageQuality::HD => {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
}
|
||||
Some(quality) => match quality {
|
||||
async_openai::types::images::ImageQuality::Standard => {
|
||||
Some(async_openai::types::images::ImageQuality::Standard)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::HD => {
|
||||
Some(async_openai::types::images::ImageQuality::Standard)
|
||||
}
|
||||
// New quality levels - keep as-is or downgrade to Standard
|
||||
async_openai::types::images::ImageQuality::High => {
|
||||
Some(async_openai::types::images::ImageQuality::Standard)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::Medium => {
|
||||
Some(async_openai::types::images::ImageQuality::Medium)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::Low => {
|
||||
Some(async_openai::types::images::ImageQuality::Low)
|
||||
}
|
||||
async_openai::types::images::ImageQuality::Auto => {
|
||||
Some(async_openai::types::images::ImageQuality::Auto)
|
||||
}
|
||||
},
|
||||
None => None,
|
||||
}
|
||||
} else {
|
||||
image_generation_config.quality.clone()
|
||||
@@ -274,20 +295,40 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let size = params
|
||||
.size_override
|
||||
.map(|s| {
|
||||
convert_string_to_enum::<async_openai::types::ImageSize>(&s)
|
||||
.unwrap_or(image_generation_config.size)
|
||||
})
|
||||
.unwrap_or(image_generation_config.size);
|
||||
.map(|s| convert_string_to_enum::<async_openai::types::images::ImageSize>(&s).unwrap())
|
||||
.or(image_generation_config.size);
|
||||
|
||||
let request = CreateImageRequestArgs::default()
|
||||
.model(model)
|
||||
.prompt(prompt.to_owned())
|
||||
.response_format(async_openai::types::ImageResponseFormat::B64Json)
|
||||
.size(size)
|
||||
.style(image_generation_config.style.clone())
|
||||
.quality(quality)
|
||||
.build()?;
|
||||
let response_format = match model.clone() {
|
||||
ImageModel::DallE2 => Some(ImageResponseFormat::B64Json),
|
||||
ImageModel::DallE3 => Some(ImageResponseFormat::B64Json),
|
||||
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
||||
// In fact, specifying the response format results in an error.
|
||||
ImageModel::GptImage1 => None,
|
||||
ImageModel::GptImage1Mini => None,
|
||||
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
|
||||
};
|
||||
|
||||
let mut request_builder = CreateImageRequestArgs::default();
|
||||
|
||||
request_builder.model(model).prompt(prompt.to_owned());
|
||||
|
||||
if let Some(response_format) = response_format {
|
||||
request_builder.response_format(response_format);
|
||||
}
|
||||
|
||||
if let Some(style) = &image_generation_config.style {
|
||||
request_builder.style(style.clone());
|
||||
}
|
||||
|
||||
if let Some(quality) = quality {
|
||||
request_builder.quality(quality.clone());
|
||||
}
|
||||
|
||||
if let Some(size) = size {
|
||||
request_builder.size(size);
|
||||
}
|
||||
|
||||
let request = request_builder.build()?;
|
||||
|
||||
tracing::trace!(
|
||||
?prompt,
|
||||
@@ -298,15 +339,15 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI image generation API request"
|
||||
);
|
||||
|
||||
let response = self.client.images().create(request).await?;
|
||||
let response = self.client.images().generate(request).await?;
|
||||
|
||||
if let Some(image) = response.data.into_iter().next() {
|
||||
match image.deref() {
|
||||
async_openai::types::Image::B64Json {
|
||||
Image::B64Json {
|
||||
b64_json,
|
||||
revised_prompt,
|
||||
} => {
|
||||
let bytes = base64_decode(b64_json)?;
|
||||
let bytes = base64_decode(b64_json.as_ref())?;
|
||||
|
||||
return Ok(ImageGenerationResult {
|
||||
bytes,
|
||||
@@ -325,6 +366,105 @@ impl ControllerTrait for Controller {
|
||||
))
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
prompt: &str,
|
||||
images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
let Some(image_generation_config) = &self.config.image_generation else {
|
||||
return Err(anyhow::anyhow!(
|
||||
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
|
||||
&AgentPurpose::ImageGeneration
|
||||
),
|
||||
));
|
||||
};
|
||||
|
||||
if images.is_empty() {
|
||||
return Err(anyhow::anyhow!("No image sources provided"));
|
||||
}
|
||||
|
||||
let mut image_inputs: Vec<ImageInput> = Vec::new();
|
||||
for image in images {
|
||||
image_inputs.push(image.into());
|
||||
}
|
||||
|
||||
let dalle2_size = match image_generation_config.size {
|
||||
Some(async_openai::types::images::ImageSize::S256x256) => Some(async_openai::types::images::ImageSize::S256x256),
|
||||
Some(async_openai::types::images::ImageSize::S512x512) => Some(async_openai::types::images::ImageSize::S512x512),
|
||||
Some(async_openai::types::images::ImageSize::S1024x1024) => Some(async_openai::types::images::ImageSize::S1024x1024),
|
||||
_ => None,
|
||||
};
|
||||
|
||||
let model = image_generation_config
|
||||
.model_id_as_openai_image_model()
|
||||
.map_err(|err| anyhow::anyhow!(err))?;
|
||||
|
||||
let response_format = match model.clone() {
|
||||
ImageModel::DallE2 => {
|
||||
Some(ImageResponseFormat::B64Json)
|
||||
}
|
||||
ImageModel::DallE3 => {
|
||||
Some(ImageResponseFormat::B64Json)
|
||||
}
|
||||
// gpt-image-1 only outputs base64 and we don't need to specify the response format.
|
||||
// In fact, specifying the response format results in an error.
|
||||
ImageModel::GptImage1 => None,
|
||||
ImageModel::GptImage1Mini => None,
|
||||
ImageModel::Other(_) => Some(ImageResponseFormat::B64Json),
|
||||
};
|
||||
|
||||
let mut request_builder = CreateImageEditRequestArgs::default();
|
||||
|
||||
request_builder
|
||||
.image(image_inputs)
|
||||
.prompt(prompt.to_owned())
|
||||
.model(model);
|
||||
|
||||
if let Some(size) = dalle2_size {
|
||||
request_builder.size(size);
|
||||
}
|
||||
|
||||
if let Some(response_format) = response_format {
|
||||
request_builder.response_format(response_format);
|
||||
}
|
||||
|
||||
let request = request_builder
|
||||
.build()
|
||||
.map_err(|e| anyhow::anyhow!("Failed to build CreateImageEditRequest: {}", e))?;
|
||||
|
||||
tracing::trace!(
|
||||
model = format!("{:?}", request.model),
|
||||
size = format!("{:?}", request.size),
|
||||
response_format = format!("{:?}", request.response_format),
|
||||
"Sending OpenAI image edit API request"
|
||||
);
|
||||
|
||||
let response = self.client.images().edit(request).await?;
|
||||
|
||||
if let Some(image_data) = response.data.into_iter().next() {
|
||||
match image_data.deref() {
|
||||
Image::B64Json { b64_json, .. } => {
|
||||
let bytes = base64_decode(b64_json.as_ref())?;
|
||||
return Ok(ImageEditResult {
|
||||
bytes,
|
||||
mime_type: mxlink::mime::IMAGE_PNG,
|
||||
});
|
||||
}
|
||||
Image::Url { url, .. } => {
|
||||
tracing::warn!(?url, "Received URL instead of B64Json for image edit");
|
||||
return Err(anyhow::anyhow!(
|
||||
"Unexpected image type (URL) when B64Json was requested"
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(anyhow::anyhow!(
|
||||
"The OpenAI image edit API returned no images"
|
||||
))
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
input: &str,
|
||||
@@ -342,7 +482,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let voice = if let Some(voice_string) = params.voice_override {
|
||||
// This is a hacky way to construct a Voice enum from the string we have.
|
||||
let voice: serde_json::Result<async_openai::types::Voice> =
|
||||
let voice: serde_json::Result<async_openai::types::audio::Voice> =
|
||||
serde_json::from_str(&format!("\"{}\"", voice_string));
|
||||
match voice {
|
||||
Ok(voice) => voice,
|
||||
@@ -382,7 +522,7 @@ impl ControllerTrait for Controller {
|
||||
"Sending OpenAI text-to-speech API request"
|
||||
);
|
||||
|
||||
let result = self.client.audio().speech(request).await?;
|
||||
let result = self.client.audio().speech().create(request).await?;
|
||||
|
||||
Ok(TextToSpeechResult {
|
||||
bytes: result.bytes.into(),
|
||||
@@ -441,15 +581,15 @@ impl ControllerTrait for Controller {
|
||||
}
|
||||
|
||||
fn response_format_to_mime_type(
|
||||
response_format: &async_openai::types::SpeechResponseFormat,
|
||||
response_format: &async_openai::types::audio::SpeechResponseFormat,
|
||||
) -> Option<mxlink::mime::Mime> {
|
||||
let content_type = match response_format {
|
||||
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
|
||||
async_openai::types::audio::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
|
||||
};
|
||||
|
||||
match content_type.parse() {
|
||||
|
||||
@@ -13,8 +13,10 @@ pub(super) use config::TextToSpeechConfig;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::controller::ControllerType;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub const OPENAI_IMAGE_MODEL_GPT_IMAGE_1: &str = "gpt-image-1";
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
|
||||
@@ -1,9 +1,18 @@
|
||||
use async_openai::types::{
|
||||
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
|
||||
ChatCompletionRequestSystemMessageArgs, ChatCompletionRequestUserMessageArgs,
|
||||
chat::{
|
||||
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
|
||||
ChatCompletionRequestMessageContentPartImage,
|
||||
ChatCompletionRequestSystemMessageArgs,
|
||||
ChatCompletionRequestUserMessageArgs, ChatCompletionRequestUserMessageContent,
|
||||
ChatCompletionRequestUserMessageContentPart,
|
||||
ImageUrlArgs,
|
||||
},
|
||||
};
|
||||
|
||||
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
use crate::utils::base64::base64_encode;
|
||||
|
||||
pub fn convert_llm_messages_to_openai_messages(
|
||||
conversation_messages: Vec<LLMMessage>,
|
||||
@@ -12,29 +21,71 @@ pub fn convert_llm_messages_to_openai_messages(
|
||||
Vec::with_capacity(conversation_messages.len());
|
||||
|
||||
for message in conversation_messages {
|
||||
openai_conversation_messages.push(convert_llm_message_to_openai_message(message));
|
||||
let openai_message = convert_llm_message_to_openai_message(message);
|
||||
if let Some(openai_message) = openai_message {
|
||||
openai_conversation_messages.push(openai_message);
|
||||
}
|
||||
}
|
||||
|
||||
openai_conversation_messages
|
||||
}
|
||||
|
||||
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> ChatCompletionRequestMessage {
|
||||
match llm_message.author {
|
||||
LLMAuthor::Prompt => ChatCompletionRequestSystemMessageArgs::default()
|
||||
.content(llm_message.message_text)
|
||||
.build()
|
||||
.expect("Failed building OpenAI system message")
|
||||
.into(),
|
||||
LLMAuthor::Assistant => ChatCompletionRequestAssistantMessageArgs::default()
|
||||
.content(llm_message.message_text)
|
||||
.build()
|
||||
.expect("Failed building OpenAI assistant message")
|
||||
.into(),
|
||||
LLMAuthor::User => ChatCompletionRequestUserMessageArgs::default()
|
||||
.content(llm_message.message_text)
|
||||
.build()
|
||||
.expect("Failed building OpenAI user message")
|
||||
.into(),
|
||||
fn convert_llm_message_to_openai_message(
|
||||
llm_message: LLMMessage,
|
||||
) -> Option<ChatCompletionRequestMessage> {
|
||||
match &llm_message.content {
|
||||
LLMMessageContent::Text(text) => Some(match llm_message.author {
|
||||
LLMAuthor::Prompt => ChatCompletionRequestSystemMessageArgs::default()
|
||||
.content(text.clone())
|
||||
.build()
|
||||
.expect("Failed building OpenAI system message")
|
||||
.into(),
|
||||
LLMAuthor::Assistant => ChatCompletionRequestAssistantMessageArgs::default()
|
||||
.content(text.clone())
|
||||
.build()
|
||||
.expect("Failed building OpenAI assistant message")
|
||||
.into(),
|
||||
LLMAuthor::User => ChatCompletionRequestUserMessageArgs::default()
|
||||
.content(text.clone())
|
||||
.build()
|
||||
.expect("Failed building OpenAI user message")
|
||||
.into(),
|
||||
}),
|
||||
LLMMessageContent::Image(image_details) => {
|
||||
let image_url = format!(
|
||||
"data:{};base64,{}",
|
||||
image_details.mime,
|
||||
base64_encode(&image_details.data)
|
||||
);
|
||||
|
||||
let part = ChatCompletionRequestUserMessageContentPart::ImageUrl(
|
||||
ChatCompletionRequestMessageContentPartImage {
|
||||
image_url: ImageUrlArgs::default()
|
||||
.url(image_url)
|
||||
.build()
|
||||
.expect("Failed building OpenAI image url"),
|
||||
},
|
||||
);
|
||||
|
||||
let message_content = ChatCompletionRequestUserMessageContent::Array(vec![part]);
|
||||
|
||||
match llm_message.author {
|
||||
LLMAuthor::User => Some(
|
||||
ChatCompletionRequestUserMessageArgs::default()
|
||||
.content(message_content)
|
||||
.build()
|
||||
.expect("Failed building OpenAI user message")
|
||||
.into(),
|
||||
),
|
||||
_ => {
|
||||
tracing::warn!(
|
||||
"OpenAI API does not support image content for messages authored by {:?}. This message part will be skipped.",
|
||||
llm_message.author
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -93,6 +93,7 @@ impl TryInto<OpenAITextGenerationConfig> for TextGenerationConfig {
|
||||
prompt: self.prompt,
|
||||
temperature: self.temperature,
|
||||
max_response_tokens: self.max_response_tokens,
|
||||
max_completion_tokens: None,
|
||||
max_context_tokens: self.max_context_tokens,
|
||||
})
|
||||
}
|
||||
@@ -160,11 +161,11 @@ impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
|
||||
type Error = String;
|
||||
|
||||
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
|
||||
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?;
|
||||
let model_id = convert_string_to_enum::<async_openai::types::audio::SpeechModel>(&self.model_id)?;
|
||||
|
||||
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?;
|
||||
let voice = convert_string_to_enum::<async_openai::types::audio::Voice>(&self.voice)?;
|
||||
|
||||
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>(
|
||||
let response_format = convert_string_to_enum::<async_openai::types::audio::SpeechResponseFormat>(
|
||||
&self.response_format,
|
||||
)?;
|
||||
|
||||
@@ -223,21 +224,27 @@ impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
|
||||
|
||||
fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> {
|
||||
let size = if let Some(size) = &self.size {
|
||||
convert_string_to_enum::<async_openai::types::ImageSize>(size)?
|
||||
Some(convert_string_to_enum::<async_openai::types::images::ImageSize>(
|
||||
size,
|
||||
)?)
|
||||
} else {
|
||||
async_openai::types::ImageSize::S1024x1024
|
||||
None
|
||||
};
|
||||
|
||||
let style = if let Some(style) = &self.style {
|
||||
convert_string_to_enum::<async_openai::types::ImageStyle>(style)?
|
||||
Some(convert_string_to_enum::<async_openai::types::images::ImageStyle>(
|
||||
style,
|
||||
)?)
|
||||
} else {
|
||||
async_openai::types::ImageStyle::Vivid
|
||||
None
|
||||
};
|
||||
|
||||
let quality = if let Some(quality) = &self.quality {
|
||||
convert_string_to_enum::<async_openai::types::ImageQuality>(quality)?
|
||||
Some(convert_string_to_enum::<async_openai::types::images::ImageQuality>(
|
||||
quality,
|
||||
)?)
|
||||
} else {
|
||||
async_openai::types::ImageQuality::Standard
|
||||
None
|
||||
};
|
||||
|
||||
Ok(OpenAIImageGenerationConfig {
|
||||
|
||||
@@ -4,23 +4,25 @@ use etke_openai_api_rust::images::{ImagesApi, ImagesBody};
|
||||
use etke_openai_api_rust::{Auth, Message, OpenAI};
|
||||
|
||||
use super::super::ControllerTrait;
|
||||
use crate::agent::utils::base64_decode;
|
||||
use crate::utils::base64::base64_decode;
|
||||
use crate::{
|
||||
agent::provider::{
|
||||
ImageEditParams, ImageGenerationParams, ImageSource, SpeechToTextParams,
|
||||
SpeechToTextResult,
|
||||
entity::{TextGenerationParams, TextGenerationResult},
|
||||
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
|
||||
},
|
||||
conversation::llm::{
|
||||
shorten_messages_list_to_context_size, Author as LLMAuthor,
|
||||
Conversation as LLMConversation, Message as LLMMessage,
|
||||
Author as LLMAuthor, Conversation as LLMConversation, Message as LLMMessage,
|
||||
MessageContent as LLMMessageContent, shorten_messages_list_to_context_size,
|
||||
},
|
||||
};
|
||||
use crate::{
|
||||
agent::{
|
||||
provider::entity::{
|
||||
ImageGenerationResult, PingResult, TextToSpeechParams, TextToSpeechResult,
|
||||
},
|
||||
AgentPurpose,
|
||||
provider::entity::{
|
||||
ImageEditResult, ImageGenerationResult, PingResult, TextToSpeechParams,
|
||||
TextToSpeechResult,
|
||||
},
|
||||
},
|
||||
strings,
|
||||
};
|
||||
@@ -60,7 +62,7 @@ impl ControllerTrait for Controller {
|
||||
|
||||
let messages = vec![LLMMessage {
|
||||
author: LLMAuthor::User,
|
||||
message_text: "Hello!".to_string(),
|
||||
content: LLMMessageContent::Text("Hello!".to_string()),
|
||||
timestamp: chrono::Utc::now(),
|
||||
}];
|
||||
|
||||
@@ -97,7 +99,7 @@ impl ControllerTrait for Controller {
|
||||
} else {
|
||||
Some(LLMMessage {
|
||||
author: LLMAuthor::Prompt,
|
||||
message_text: prompt_text,
|
||||
content: LLMMessageContent::Text(prompt_text),
|
||||
timestamp: chrono::Utc::now(),
|
||||
})
|
||||
};
|
||||
@@ -366,6 +368,17 @@ impl ControllerTrait for Controller {
|
||||
))
|
||||
}
|
||||
|
||||
async fn create_image_edit(
|
||||
&self,
|
||||
_prompt: &str,
|
||||
_images: Vec<ImageSource>,
|
||||
_params: ImageEditParams,
|
||||
) -> anyhow::Result<ImageEditResult> {
|
||||
Err(anyhow::anyhow!(
|
||||
"The OpenAI image edit API is not supported by the OpenAI-compat provider"
|
||||
))
|
||||
}
|
||||
|
||||
async fn text_to_speech(
|
||||
&self,
|
||||
input: &str,
|
||||
|
||||
@@ -21,8 +21,8 @@ pub use controller::Controller;
|
||||
|
||||
use super::super::AgentInstantiationError;
|
||||
use super::super::AgentInstantiationResult;
|
||||
use super::controller::ControllerType;
|
||||
use super::ConfigTrait;
|
||||
use super::controller::ControllerType;
|
||||
|
||||
pub fn create_controller_from_yaml_value_config(
|
||||
agent_id: &str,
|
||||
|
||||
@@ -2,7 +2,9 @@ use etke_openai_api_rust::{Message, Role};
|
||||
|
||||
use crate::agent::provider::openai::Config as OpenAIConfig;
|
||||
|
||||
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
|
||||
use crate::conversation::llm::{
|
||||
Author as LLMAuthor, Message as LLMMessage, MessageContent as LLMMessageContent,
|
||||
};
|
||||
|
||||
pub fn convert_llm_messages_to_openai_messages(
|
||||
conversation_messages: Vec<LLMMessage>,
|
||||
@@ -11,22 +13,33 @@ pub fn convert_llm_messages_to_openai_messages(
|
||||
Vec::with_capacity(conversation_messages.len());
|
||||
|
||||
for message in conversation_messages {
|
||||
openai_conversation_messages.push(convert_llm_message_to_openai_message(message));
|
||||
let openai_message = convert_llm_message_to_openai_message(message);
|
||||
if let Some(openai_message) = openai_message {
|
||||
openai_conversation_messages.push(openai_message);
|
||||
}
|
||||
}
|
||||
|
||||
openai_conversation_messages
|
||||
}
|
||||
|
||||
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> Message {
|
||||
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> Option<Message> {
|
||||
let role = match llm_message.author {
|
||||
LLMAuthor::Prompt => Role::System,
|
||||
LLMAuthor::Assistant => Role::Assistant,
|
||||
LLMAuthor::User => Role::User,
|
||||
};
|
||||
|
||||
Message {
|
||||
role,
|
||||
content: llm_message.message_text,
|
||||
match &llm_message.content {
|
||||
LLMMessageContent::Text(text) => Some(Message {
|
||||
role,
|
||||
content: text.clone(),
|
||||
}),
|
||||
LLMMessageContent::Image(_image_details) => {
|
||||
tracing::warn!(
|
||||
"The OpenAI-compat provider's library does not support image content. This image message will be skipped."
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
use base64::{engine::general_purpose::STANDARD, Engine as _};
|
||||
|
||||
use crate::{
|
||||
agent::{
|
||||
AgentInstance, AgentPurpose, ControllerTrait, Manager as AgentManager, PublicIdentifier,
|
||||
@@ -140,7 +138,3 @@ async fn get_global_agent_id_for_purpose(
|
||||
.handler
|
||||
.get_by_purpose_with_catch_all_fallback(purpose)
|
||||
}
|
||||
|
||||
pub(crate) fn base64_decode(base64_string: &str) -> Result<Vec<u8>, base64::DecodeError> {
|
||||
STANDARD.decode(base64_string)
|
||||
}
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
use std::fs;
|
||||
use std::sync::Arc;
|
||||
use std::{future::Future, pin::Pin};
|
||||
|
||||
use mxlink::matrix_sdk::Room;
|
||||
use mxlink::matrix_sdk::media::{MediaFormat, MediaRequestParameters};
|
||||
use mxlink::matrix_sdk::ruma::{
|
||||
events::room::MediaSource, MilliSecondsSinceUnixEpoch, OwnedUserId,
|
||||
MilliSecondsSinceUnixEpoch, OwnedUserId, events::room::MediaSource,
|
||||
};
|
||||
use mxlink::matrix_sdk::Room;
|
||||
use mxlink::matrix_sdk::ruma::api::client::profile::{AvatarUrl, DisplayName};
|
||||
|
||||
use mxlink::{
|
||||
InitConfig, LoginConfig, LoginCredentials, LoginEncryption, MatrixLink, PersistenceConfig,
|
||||
@@ -17,12 +19,13 @@ use mxlink::helpers::account_data_config::{
|
||||
RoomConfigManager as AccountDataRoomConfigManager,
|
||||
};
|
||||
use mxlink::helpers::encryption::Manager as EncryptionManager;
|
||||
use mxlink::mime::Mime;
|
||||
|
||||
use crate::agent::Manager as AgentManager;
|
||||
use crate::entity::catch_up_marker::{
|
||||
CatchUpMarker, CatchUpMarkerManager, DelayedCatchUpMarkerManager,
|
||||
};
|
||||
use crate::entity::cfg::Config;
|
||||
use crate::entity::cfg::{Avatar, Config};
|
||||
use crate::entity::globalconfig::{GlobalConfig, GlobalConfigurationManager};
|
||||
use crate::entity::roomconfig::{RoomConfig, RoomConfigurationManager};
|
||||
|
||||
@@ -140,6 +143,10 @@ impl Bot {
|
||||
&self.inner.config.command_prefix
|
||||
}
|
||||
|
||||
pub(crate) fn post_join_self_introduction_enabled(&self) -> bool {
|
||||
self.inner.config.room.post_join_self_introduction_enabled
|
||||
}
|
||||
|
||||
pub(crate) fn homeserver_name(&self) -> &str {
|
||||
&self.inner.config.homeserver.server_name
|
||||
}
|
||||
@@ -281,24 +288,27 @@ impl Bot {
|
||||
async fn do_prepare_profile(&self) -> anyhow::Result<()> {
|
||||
tracing::debug!("Preparing profile..");
|
||||
|
||||
let desired_display_name = self.inner.config.user.name.clone();
|
||||
|
||||
let account = self.inner.matrix_link.client().account();
|
||||
let media = self.inner.matrix_link.client().media();
|
||||
|
||||
let desired_display_name = self.inner.config.user.name.clone();
|
||||
|
||||
let profile = account
|
||||
.fetch_user_profile()
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed fetching profile: {:?}", e))?;
|
||||
|
||||
let should_update_display_name = match &profile.displayname {
|
||||
let current_display_name = profile.get_static::<DisplayName>()?;
|
||||
let current_avatar_url = profile.get_static::<AvatarUrl>()?;
|
||||
|
||||
let should_update_display_name = match ¤t_display_name {
|
||||
Some(displayname) => displayname != &desired_display_name,
|
||||
None => true,
|
||||
};
|
||||
|
||||
if should_update_display_name {
|
||||
tracing::info!(
|
||||
?profile.displayname,
|
||||
?current_display_name,
|
||||
?desired_display_name,
|
||||
"Updating display name.."
|
||||
);
|
||||
@@ -308,34 +318,72 @@ impl Bot {
|
||||
}
|
||||
}
|
||||
|
||||
let should_update_avatar = match &profile.avatar_url {
|
||||
Some(avatar_url) => {
|
||||
let request = MediaRequestParameters {
|
||||
source: MediaSource::Plain(avatar_url.to_owned()),
|
||||
format: MediaFormat::File,
|
||||
};
|
||||
|
||||
let content = media
|
||||
.get_media_content(&request, true)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed fetching existing avatar: {:?}", e))?;
|
||||
|
||||
content.as_slice() != LOGO_BYTES
|
||||
let desired_avatar: Option<(Vec<u8>, Mime)> = match &self.inner.config.user.avatar {
|
||||
Avatar::Keep => {
|
||||
tracing::info!("Avatar configured to keep current, skipping avatar management");
|
||||
None
|
||||
}
|
||||
Avatar::Default => {
|
||||
tracing::info!("Avatar configured to use default");
|
||||
Some((
|
||||
LOGO_BYTES.to_vec(),
|
||||
LOGO_MIME_TYPE
|
||||
.parse()
|
||||
.expect("Failed parsing mime type for logo"),
|
||||
))
|
||||
}
|
||||
Avatar::Custom(avatar_path) => {
|
||||
tracing::info!(?avatar_path, "Avatar configured to use custom path");
|
||||
let bytes = fs::read(avatar_path).map_err(|e| {
|
||||
anyhow::anyhow!("Failed reading avatar from {:?}: {:?}", avatar_path, e)
|
||||
})?;
|
||||
let mime = mime_guess::from_path(avatar_path).first_or_octet_stream();
|
||||
tracing::debug!(?mime, bytes_len = bytes.len(), "Loaded custom avatar");
|
||||
Some((bytes, mime))
|
||||
}
|
||||
None => true,
|
||||
};
|
||||
|
||||
if should_update_avatar {
|
||||
tracing::info!("Updating avatar..");
|
||||
if let Some((desired_bytes, mime_type)) = desired_avatar {
|
||||
let should_update_avatar = match ¤t_avatar_url {
|
||||
Some(avatar_url) => {
|
||||
tracing::debug!(?avatar_url, "Fetching current avatar to compare");
|
||||
let request = MediaRequestParameters {
|
||||
source: MediaSource::Plain(avatar_url.to_owned()),
|
||||
format: MediaFormat::File,
|
||||
};
|
||||
|
||||
let mime_type = LOGO_MIME_TYPE
|
||||
.parse()
|
||||
.expect("Failed parsing mime type for logo");
|
||||
let content = media
|
||||
.get_media_content(&request, true)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed fetching existing avatar: {:?}", e))?;
|
||||
|
||||
account
|
||||
.upload_avatar(&mime_type, LOGO_BYTES.to_vec())
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed uploading avatar: {:?}", e))?;
|
||||
let needs_update = content.as_slice() != desired_bytes;
|
||||
|
||||
tracing::debug!(
|
||||
current_bytes_len = content.len(),
|
||||
desired_bytes_len = desired_bytes.len(),
|
||||
?needs_update,
|
||||
"Compared current and desired avatar"
|
||||
);
|
||||
|
||||
needs_update
|
||||
}
|
||||
None => {
|
||||
tracing::debug!("No current avatar set, will upload");
|
||||
true
|
||||
}
|
||||
};
|
||||
|
||||
if should_update_avatar {
|
||||
tracing::info!("Updating avatar..");
|
||||
account
|
||||
.upload_avatar(&mime_type, desired_bytes)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed uploading avatar: {:?}", e))?;
|
||||
tracing::info!("Avatar updated successfully");
|
||||
} else {
|
||||
tracing::debug!("Avatar already up to date, skipping upload");
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
|
||||
@@ -5,7 +5,7 @@ use anyhow::anyhow;
|
||||
|
||||
use crate::agent::AgentPurpose;
|
||||
|
||||
pub use crate::entity::cfg::{defaults as cfg_defaults, env as cfg_env, Config};
|
||||
pub use crate::entity::cfg::{Avatar, Config, defaults as cfg_defaults, env as cfg_env};
|
||||
|
||||
pub fn load() -> anyhow::Result<Config> {
|
||||
let config_file_path = env::var(cfg_env::BAIBOT_CONFIG_FILE_PATH)
|
||||
@@ -33,8 +33,17 @@ pub fn load() -> anyhow::Result<Config> {
|
||||
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE => {
|
||||
config.user.encryption.recovery_passphrase = Some(value);
|
||||
}
|
||||
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_RESET_ALLOWED => {
|
||||
config.user.encryption.recovery_reset_allowed = value.parse::<bool>()?;
|
||||
}
|
||||
cfg_env::BAIBOT_USER_NAME => config.user.name = value,
|
||||
cfg_env::BAIBOT_USER_AVATAR => {
|
||||
config.user.avatar = Avatar::from_string(value);
|
||||
}
|
||||
cfg_env::BAIBOT_COMMAND_PREFIX => config.command_prefix = value,
|
||||
cfg_env::BAIBOT_ROOM_POST_JOIN_SELF_INTRODUCTION_ENABLED => {
|
||||
config.room.post_join_self_introduction_enabled = value.parse::<bool>()?;
|
||||
}
|
||||
cfg_env::BAIBOT_LOGGING => {
|
||||
config.logging = value;
|
||||
}
|
||||
@@ -48,6 +57,9 @@ pub fn load() -> anyhow::Result<Config> {
|
||||
cfg_env::BAIBOT_PERSISTENCE_DATA_DIR_PATH => {
|
||||
config.persistence.data_dir_path = Some(value);
|
||||
}
|
||||
cfg_env::BAIBOT_PERSISTENCE_SESSION_ENCRYPTION_KEY => {
|
||||
config.persistence.session_encryption_key = Some(value);
|
||||
}
|
||||
cfg_env::BAIBOT_PERSISTENCE_CONFIG_ENCRYPTION_KEY => {
|
||||
config.persistence.config_encryption_key = Some(value);
|
||||
}
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use mxlink::matrix_sdk::{
|
||||
ruma::{
|
||||
api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::OriginalSyncRoomMessageEvent, OwnedEventId,
|
||||
},
|
||||
Room,
|
||||
ruma::{
|
||||
OwnedEventId, api::client::receipt::create_receipt::v3::ReceiptType,
|
||||
events::room::message::OriginalSyncRoomMessageEvent,
|
||||
},
|
||||
};
|
||||
|
||||
use mxlink::{CallbackError, MessageResponseType};
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
use mxlink::matrix_sdk::{
|
||||
ruma::{
|
||||
events::{
|
||||
room::message::Relation, AnySyncMessageLikeEvent, AnySyncTimelineEvent,
|
||||
SyncMessageLikeEvent,
|
||||
},
|
||||
OwnedEventId, OwnedUserId,
|
||||
},
|
||||
Room,
|
||||
ruma::{
|
||||
OwnedEventId, OwnedUserId,
|
||||
events::{
|
||||
AnySyncMessageLikeEvent, AnySyncTimelineEvent, SyncMessageLikeEvent,
|
||||
room::message::Relation,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
use mxlink::CallbackError;
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use mxlink::{
|
||||
matrix_sdk::{
|
||||
ruma::events::{room::member::StrippedRoomMemberEvent, AnySyncTimelineEvent},
|
||||
Room,
|
||||
},
|
||||
InvitationDecision,
|
||||
matrix_sdk::{
|
||||
Room,
|
||||
ruma::events::{AnySyncTimelineEvent, room::member::StrippedRoomMemberEvent},
|
||||
},
|
||||
};
|
||||
|
||||
use mxlink::CallbackError;
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
use super::AccessControllerType;
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
let mut message = String::new();
|
||||
|
||||
@@ -4,5 +4,5 @@ pub mod help;
|
||||
mod room_local_agent_managers;
|
||||
mod users;
|
||||
|
||||
pub use determination::{determine_controller, AccessControllerType};
|
||||
pub use determination::{AccessControllerType, determine_controller};
|
||||
pub use dispatching::dispatch_controller;
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle_get(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
let message = match &message_context
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle_get(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
let message = match &message_context.global_config().access.user_patterns {
|
||||
|
||||
@@ -3,15 +3,15 @@ mod tests;
|
||||
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::agent::provider::{ControllerTrait, PingResult};
|
||||
use crate::agent::PublicIdentifier;
|
||||
use crate::agent::{create_from_provider_and_yaml_value_config, AgentDefinition};
|
||||
use crate::agent::provider::{ControllerTrait, PingResult};
|
||||
use crate::agent::{AgentDefinition, create_from_provider_and_yaml_value_config};
|
||||
use crate::agent::{AgentInstance, AgentProvider};
|
||||
use crate::controller::utils::get_text_body_or_complain;
|
||||
use crate::entity::globalconfig::GlobalConfigurationManager;
|
||||
use crate::entity::roomconfig::RoomConfigurationManager;
|
||||
use crate::strings;
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
struct ParsedAgentConfig {
|
||||
agent: AgentInstance,
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::entity::{
|
||||
globalconfig::GlobalConfigurationManager, roomconfig::RoomConfigurationManager, MessageContext,
|
||||
MessageContext, globalconfig::GlobalConfigurationManager, roomconfig::RoomConfigurationManager,
|
||||
};
|
||||
use crate::{agent::PublicIdentifier, strings, Bot};
|
||||
use crate::{Bot, agent::PublicIdentifier, strings};
|
||||
|
||||
pub async fn handle(
|
||||
bot: &Bot,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{agent::PublicIdentifier, entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, agent::PublicIdentifier, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle(
|
||||
bot: &Bot,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
// Anyone can access this help command, because certain subcommands ("list")
|
||||
|
||||
@@ -2,7 +2,7 @@ use mxlink::MessageResponseType;
|
||||
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::strings;
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
let agents = bot
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
pub mod create;
|
||||
pub mod delete;
|
||||
@@ -7,7 +7,7 @@ pub mod determination;
|
||||
pub mod help;
|
||||
pub mod list;
|
||||
|
||||
pub use determination::{determine_controller, AgentControllerType};
|
||||
pub use determination::{AgentControllerType, determine_controller};
|
||||
|
||||
pub async fn dispatch_controller(
|
||||
handler: &AgentControllerType,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle_get<T>(
|
||||
bot: &Bot,
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
use crate::{
|
||||
agent::{AgentPurpose, PublicIdentifier},
|
||||
entity::roomconfig::{
|
||||
SpeechToTextFlowType, TextGenerationAutoUsage, TextGenerationPrefixRequirementType,
|
||||
SpeechToTextFlowType, SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
TextGenerationAutoUsage, TextGenerationPrefixRequirementType,
|
||||
TextToSpeechBotMessagesFlowType, TextToSpeechUserMessagesFlowType,
|
||||
},
|
||||
};
|
||||
@@ -54,6 +55,11 @@ pub enum ConfigSpeechToTextSettingRelatedControllerType {
|
||||
GetFlowType,
|
||||
SetFlowType(Option<SpeechToTextFlowType>),
|
||||
|
||||
GetMsgTypeForNonThreadedOnlyTranscribedMessages,
|
||||
SetMsgTypeForNonThreadedOnlyTranscribedMessages(
|
||||
Option<SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages>,
|
||||
),
|
||||
|
||||
GetLanguage,
|
||||
SetLanguage(Option<String>),
|
||||
}
|
||||
|
||||
@@ -1,7 +1,13 @@
|
||||
#[cfg(test)]
|
||||
mod tests;
|
||||
|
||||
use crate::{controller::ControllerType, entity::roomconfig::SpeechToTextFlowType, strings};
|
||||
use crate::{
|
||||
controller::ControllerType,
|
||||
entity::roomconfig::{
|
||||
SpeechToTextFlowType, SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
},
|
||||
strings,
|
||||
};
|
||||
|
||||
use super::super::controller_type::ConfigSpeechToTextSettingRelatedControllerType;
|
||||
|
||||
@@ -48,6 +54,53 @@ pub(super) fn determine(
|
||||
));
|
||||
}
|
||||
|
||||
// msg_type_for_non_threaded_only_transcribed_messages
|
||||
|
||||
if let Some(remaining_text) =
|
||||
text.strip_prefix("msg-type-for-non-threaded-only-transcribed-messages")
|
||||
{
|
||||
let remaining_text = remaining_text.trim();
|
||||
|
||||
if !remaining_text.is_empty() {
|
||||
return Err(ControllerType::Error(
|
||||
strings::cfg::configuration_getter_used_with_extra_text(
|
||||
"msg-type-for-non-threaded-only-transcribed-messages",
|
||||
remaining_text,
|
||||
)
|
||||
.to_owned(),
|
||||
));
|
||||
}
|
||||
|
||||
return Ok(ConfigSpeechToTextSettingRelatedControllerType::GetMsgTypeForNonThreadedOnlyTranscribedMessages);
|
||||
}
|
||||
|
||||
if let Some(value_string) =
|
||||
text.strip_prefix("set-msg-type-for-non-threaded-only-transcribed-messages")
|
||||
{
|
||||
let value_string = value_string.trim().to_owned();
|
||||
|
||||
let value_choice = if value_string.is_empty() {
|
||||
None
|
||||
} else {
|
||||
let value_choice =
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages::from_str(
|
||||
&value_string.to_lowercase(),
|
||||
);
|
||||
|
||||
if value_choice.is_none() {
|
||||
return Err(ControllerType::Error(
|
||||
strings::cfg::configuration_value_unrecognized(&value_string).to_owned(),
|
||||
));
|
||||
}
|
||||
|
||||
value_choice
|
||||
};
|
||||
|
||||
return Ok(ConfigSpeechToTextSettingRelatedControllerType::SetMsgTypeForNonThreadedOnlyTranscribedMessages(
|
||||
value_choice,
|
||||
));
|
||||
}
|
||||
|
||||
// Language
|
||||
|
||||
if let Some(remaining_text) = text.strip_prefix("language") {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
use crate::strings;
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use super::controller_type::{
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
use crate::entity::roomconfig::{RoomSettings, SpeechToTextFlowType};
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::entity::roomconfig::{
|
||||
RoomSettings, SpeechToTextFlowType,
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
use super::super::controller_type::{
|
||||
ConfigSpeechToTextSettingRelatedControllerType, SettingsStorageSource,
|
||||
@@ -52,6 +55,39 @@ pub(super) async fn dispatch(
|
||||
}
|
||||
}
|
||||
|
||||
ConfigSpeechToTextSettingRelatedControllerType::GetMsgTypeForNonThreadedOnlyTranscribedMessages => {
|
||||
let value = &room_settings.speech_to_text.msg_type_for_non_threaded_only_transcribed_messages;
|
||||
setting_get::<SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages>(bot, message_context, value).await
|
||||
}
|
||||
ConfigSpeechToTextSettingRelatedControllerType::SetMsgTypeForNonThreadedOnlyTranscribedMessages(value) => {
|
||||
let value = value.to_owned();
|
||||
|
||||
let setter_callback = Box::new(move |room_settings: &mut RoomSettings| {
|
||||
room_settings.speech_to_text.msg_type_for_non_threaded_only_transcribed_messages = value;
|
||||
});
|
||||
|
||||
match config_type {
|
||||
SettingsStorageSource::Room => {
|
||||
room_setting_set::<SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages>(
|
||||
bot,
|
||||
message_context,
|
||||
&value,
|
||||
setter_callback,
|
||||
)
|
||||
.await
|
||||
}
|
||||
SettingsStorageSource::Global => {
|
||||
global_setting_set::<SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages>(
|
||||
bot,
|
||||
message_context,
|
||||
&value,
|
||||
setter_callback,
|
||||
)
|
||||
.await
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ConfigSpeechToTextSettingRelatedControllerType::GetLanguage => {
|
||||
let value = &room_settings.speech_to_text.language;
|
||||
setting_get::<String>(bot, message_context, value).await
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use crate::entity::roomconfig::{
|
||||
RoomSettings, TextGenerationAutoUsage, TextGenerationPrefixRequirementType,
|
||||
};
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
use super::super::controller_type::{
|
||||
ConfigTextGenerationSettingRelatedControllerType, SettingsStorageSource,
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use crate::entity::roomconfig::{
|
||||
RoomSettings, TextToSpeechBotMessagesFlowType, TextToSpeechUserMessagesFlowType,
|
||||
};
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
use super::super::controller_type::{
|
||||
ConfigTextToSpeechSettingRelatedControllerType, SettingsStorageSource,
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::entity::{roomconfig::RoomSettings, MessageContext};
|
||||
use crate::{strings, Bot};
|
||||
use crate::entity::{MessageContext, roomconfig::RoomSettings};
|
||||
use crate::{Bot, strings};
|
||||
|
||||
pub async fn handle_set<T>(
|
||||
bot: &Bot,
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{
|
||||
Bot,
|
||||
agent::{AgentPurpose, PublicIdentifier},
|
||||
entity::{globalconfig::GlobalConfigurationManager, MessageContext},
|
||||
strings, Bot,
|
||||
entity::{MessageContext, globalconfig::GlobalConfigurationManager},
|
||||
strings,
|
||||
};
|
||||
|
||||
pub async fn handle_get(
|
||||
|
||||
@@ -1,14 +1,16 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{
|
||||
Bot,
|
||||
entity::{
|
||||
MessageContext,
|
||||
roomconfig::{
|
||||
SpeechToTextFlowType, TextGenerationAutoUsage, TextGenerationPrefixRequirementType,
|
||||
SpeechToTextFlowType, SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
TextGenerationAutoUsage, TextGenerationPrefixRequirementType,
|
||||
TextToSpeechBotMessagesFlowType, TextToSpeechUserMessagesFlowType,
|
||||
},
|
||||
MessageContext,
|
||||
},
|
||||
strings, Bot,
|
||||
strings,
|
||||
};
|
||||
|
||||
pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
@@ -346,6 +348,46 @@ fn build_section_speech_to_text(command_prefix: &str) -> String {
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Msg Type For Non Threaded Only Transcribed Messages
|
||||
|
||||
message.push_str(&format!(
|
||||
"#### {}",
|
||||
strings::help::cfg::speech_to_text_msg_type_for_non_threaded_only_transcribed_messages_heading()
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
message.push_str(strings::help::cfg::speech_to_text_msg_type_for_non_threaded_only_transcribed_messages_intro());
|
||||
message.push('\n');
|
||||
message.push_str(
|
||||
&strings::help::cfg::the_following_configuration_values_are_recognized(
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages::choices(),
|
||||
),
|
||||
);
|
||||
message.push_str("\n\n");
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_show(
|
||||
command_prefix,
|
||||
"speech-to-text msg-type-for-non-threaded-only-transcribed-messages"
|
||||
)
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_set(
|
||||
command_prefix,
|
||||
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages VALUE"
|
||||
)
|
||||
));
|
||||
message.push('\n');
|
||||
message.push_str(&format!(
|
||||
"- {}",
|
||||
&strings::help::cfg::current_setting_unset(
|
||||
command_prefix,
|
||||
"speech-to-text set-msg-type-for-non-threaded-only-transcribed-messages"
|
||||
)
|
||||
));
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Language
|
||||
|
||||
message.push_str(&format!(
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::entity::{roomconfig::RoomSettings, MessageContext};
|
||||
use crate::{strings, Bot};
|
||||
use crate::entity::{MessageContext, roomconfig::RoomSettings};
|
||||
use crate::{Bot, strings};
|
||||
|
||||
pub async fn handle_set<T>(
|
||||
bot: &Bot,
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{
|
||||
Bot,
|
||||
agent::{AgentPurpose, PublicIdentifier},
|
||||
entity::MessageContext,
|
||||
strings, Bot,
|
||||
strings,
|
||||
};
|
||||
|
||||
use crate::entity::roomconfig::RoomConfigurationManager;
|
||||
|
||||
@@ -1,15 +1,16 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{
|
||||
Bot,
|
||||
agent::{
|
||||
utils::get_effective_agent_for_purpose, AgentInstance, AgentPurpose, ControllerTrait,
|
||||
Manager as AgentManager, PublicIdentifier,
|
||||
AgentInstance, AgentPurpose, ControllerTrait, Manager as AgentManager, PublicIdentifier,
|
||||
utils::get_effective_agent_for_purpose,
|
||||
},
|
||||
entity::{
|
||||
roomconfig::{RoomConfig, RoomSettingsHandler},
|
||||
MessageContext, RoomConfigContext,
|
||||
roomconfig::{RoomConfig, RoomSettingsHandler},
|
||||
},
|
||||
strings, Bot,
|
||||
strings,
|
||||
};
|
||||
|
||||
pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
@@ -68,7 +69,7 @@ pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Resu
|
||||
);
|
||||
message.push_str("\n\n");
|
||||
|
||||
// Image Generation
|
||||
// Image Creation
|
||||
message.push_str(
|
||||
&generate_image_generation_section(agent_manager, message_context.room_config_context())
|
||||
.await,
|
||||
@@ -507,6 +508,35 @@ async fn generate_speech_to_text_section(
|
||||
flow_type_set_where,
|
||||
));
|
||||
|
||||
// Msg Type For Non Threaded Only Transcribed Messages
|
||||
|
||||
let effective_msg_type_for_non_threaded_only_transcribed_messages =
|
||||
room_config_context.speech_to_text_msg_type_for_non_threaded_only_transcribed_messages();
|
||||
let room_config_msg_type_for_non_threaded_only_transcribed_messages = room_config_context
|
||||
.room_config
|
||||
.settings
|
||||
.speech_to_text
|
||||
.msg_type_for_non_threaded_only_transcribed_messages;
|
||||
let global_config_msg_type_for_non_threaded_only_transcribed_messages = room_config_context
|
||||
.global_config
|
||||
.fallback_room_settings
|
||||
.speech_to_text
|
||||
.msg_type_for_non_threaded_only_transcribed_messages;
|
||||
|
||||
let msg_type_for_non_threaded_only_transcribed_messages_set_where =
|
||||
if room_config_msg_type_for_non_threaded_only_transcribed_messages.is_some() {
|
||||
strings::cfg::status_badge_set_in_room_config()
|
||||
} else if global_config_msg_type_for_non_threaded_only_transcribed_messages.is_some() {
|
||||
strings::cfg::status_badge_set_in_global_config()
|
||||
} else {
|
||||
strings::cfg::status_badge_using_hardcoded_default()
|
||||
};
|
||||
|
||||
message.push_str(&strings::cfg::status_speech_to_text_entry_msg_type_for_non_threaded_only_transcribed_messages(
|
||||
effective_msg_type_for_non_threaded_only_transcribed_messages,
|
||||
msg_type_for_non_threaded_only_transcribed_messages_set_where,
|
||||
));
|
||||
|
||||
// Language
|
||||
|
||||
let effective_language = room_config_context.speech_to_text_language();
|
||||
|
||||
@@ -1,30 +1,31 @@
|
||||
use mxlink::matrix_sdk::ruma::events::room::message::AudioMessageEventContent;
|
||||
use mxlink::matrix_sdk::ruma::OwnedEventId;
|
||||
use mxlink::matrix_sdk::ruma::events::room::message::AudioMessageEventContent;
|
||||
use mxlink::{MatrixLink, MessageResponseType};
|
||||
|
||||
use tracing::Instrument;
|
||||
|
||||
use crate::agent::provider::{
|
||||
SpeechToTextParams, TextGenerationParams, TextGenerationPromptVariables,
|
||||
};
|
||||
use crate::agent::AgentInstance;
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::ControllerTrait;
|
||||
use crate::agent::provider::{
|
||||
SpeechToTextParams, TextGenerationParams, TextGenerationPromptVariables,
|
||||
};
|
||||
use crate::controller::utils::agent::get_effective_agent_for_purpose_or_complain;
|
||||
use crate::conversation::matrix::MatrixMessageProcessingParams;
|
||||
use crate::entity::roomconfig::{
|
||||
SpeechToTextFlowType, TextToSpeechBotMessagesFlowType, TextToSpeechUserMessagesFlowType,
|
||||
};
|
||||
use crate::entity::MessagePayload;
|
||||
use crate::entity::roomconfig::{
|
||||
SpeechToTextFlowType, SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
TextToSpeechBotMessagesFlowType, TextToSpeechUserMessagesFlowType,
|
||||
};
|
||||
use crate::strings;
|
||||
use crate::utils::text_to_speech::create_transcribed_message_text;
|
||||
use crate::{
|
||||
Bot,
|
||||
conversation::{
|
||||
create_llm_conversation_for_matrix_reply_chain, create_llm_conversation_for_matrix_thread,
|
||||
matrix::create_list_of_bot_user_prefixes_to_strip,
|
||||
},
|
||||
entity::MessageContext,
|
||||
Bot,
|
||||
};
|
||||
|
||||
#[derive(Debug, PartialEq)]
|
||||
@@ -38,6 +39,8 @@ pub enum ChatCompletionControllerType {
|
||||
|
||||
Audio,
|
||||
|
||||
Image,
|
||||
|
||||
ThreadMention,
|
||||
ReplyMention,
|
||||
}
|
||||
@@ -71,21 +74,35 @@ pub async fn handle(
|
||||
if let MessagePayload::Audio(audio_content) = &message_context.payload() {
|
||||
original_message_is_audio = true;
|
||||
|
||||
let response_type = match speech_to_text_flow_type {
|
||||
let (response_type, msg_type) = match speech_to_text_flow_type {
|
||||
SpeechToTextFlowType::Ignore => {
|
||||
tracing::debug!("Intentionally ignoring audio message");
|
||||
return Ok(());
|
||||
}
|
||||
SpeechToTextFlowType::TranscribeAndGenerateText => {
|
||||
tracing::debug!("Will be transcribing and possibly generating text..");
|
||||
MessageResponseType::InThread(message_context.thread_info().clone())
|
||||
(
|
||||
MessageResponseType::InThread(message_context.thread_info().clone()),
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages::Notice,
|
||||
)
|
||||
}
|
||||
SpeechToTextFlowType::OnlyTranscribe => {
|
||||
tracing::debug!("Will only be transcribing audio to text..");
|
||||
if message_context.thread_info().is_thread_root_only() {
|
||||
MessageResponseType::Reply(message_context.thread_info().root_event_id.clone())
|
||||
let msg_type = message_context
|
||||
.room_config_context()
|
||||
.speech_to_text_msg_type_for_non_threaded_only_transcribed_messages();
|
||||
(
|
||||
MessageResponseType::Reply(
|
||||
message_context.thread_info().root_event_id.clone(),
|
||||
),
|
||||
msg_type,
|
||||
)
|
||||
} else {
|
||||
MessageResponseType::InThread(message_context.thread_info().clone())
|
||||
(
|
||||
MessageResponseType::InThread(message_context.thread_info().clone()),
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages::Notice,
|
||||
)
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -94,8 +111,14 @@ pub async fn handle(
|
||||
_typing_notice_guard = Some(bot.start_typing_notice(message_context.room()).await);
|
||||
}
|
||||
|
||||
let Some(speech_to_text_created_event_id_result) =
|
||||
handle_stage_speech_to_text(bot, message_context, audio_content, response_type).await
|
||||
let Some(speech_to_text_created_event_id_result) = handle_stage_speech_to_text(
|
||||
bot,
|
||||
message_context,
|
||||
audio_content,
|
||||
response_type,
|
||||
msg_type,
|
||||
)
|
||||
.await
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
@@ -282,6 +305,7 @@ async fn handle_stage_speech_to_text(
|
||||
message_context: &MessageContext,
|
||||
audio_content: &AudioMessageEventContent,
|
||||
response_type: MessageResponseType,
|
||||
msg_type: SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
) -> Option<OwnedEventId> {
|
||||
let agent = get_effective_agent_for_purpose_or_complain(
|
||||
bot,
|
||||
@@ -302,7 +326,7 @@ async fn handle_stage_speech_to_text(
|
||||
.react_no_fail(
|
||||
message_context.room(),
|
||||
message_context.event_id().clone(),
|
||||
AgentPurpose::SpeechToText.emoji().to_owned(),
|
||||
strings::PROGRESS_INDICATOR_EMOJI.to_owned(),
|
||||
)
|
||||
.await;
|
||||
|
||||
@@ -312,6 +336,7 @@ async fn handle_stage_speech_to_text(
|
||||
&agent,
|
||||
audio_content,
|
||||
response_type.clone(),
|
||||
msg_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
@@ -393,7 +418,8 @@ async fn handle_stage_text_generation(
|
||||
ChatCompletionControllerType::TextCommand
|
||||
| ChatCompletionControllerType::TextMention
|
||||
| ChatCompletionControllerType::TextDirect
|
||||
| ChatCompletionControllerType::Audio => {
|
||||
| ChatCompletionControllerType::Audio
|
||||
| ChatCompletionControllerType::Image => {
|
||||
Some(message_context.combined_admin_and_user_regexes())
|
||||
}
|
||||
|
||||
@@ -415,6 +441,7 @@ async fn handle_stage_text_generation(
|
||||
// When we're triggered via a reply mention, the context is the whole reply chain upward of the message that triggered us.
|
||||
ChatCompletionControllerType::ReplyMention => {
|
||||
create_llm_conversation_for_matrix_reply_chain(
|
||||
&matrix_link,
|
||||
&bot.room_event_fetcher().clone(),
|
||||
message_context.room(),
|
||||
message_context.thread_info().last_event_id.clone(),
|
||||
@@ -426,7 +453,7 @@ async fn handle_stage_text_generation(
|
||||
// Everything else is happening in a thread, so the context is the whole thread.
|
||||
_ => {
|
||||
create_llm_conversation_for_matrix_thread(
|
||||
matrix_link.clone(),
|
||||
&matrix_link,
|
||||
message_context.room(),
|
||||
message_context.thread_info().root_event_id.clone(),
|
||||
¶ms,
|
||||
@@ -572,6 +599,7 @@ async fn handle_stage_speech_to_text_actual_transcribing(
|
||||
agent: &AgentInstance,
|
||||
audio_content: &AudioMessageEventContent,
|
||||
response_type: MessageResponseType,
|
||||
msg_type: SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages,
|
||||
) -> anyhow::Result<OwnedEventId> {
|
||||
let src = &audio_content.source;
|
||||
|
||||
@@ -626,9 +654,6 @@ async fn handle_stage_speech_to_text_actual_transcribing(
|
||||
//
|
||||
// When sending a bare reply, we'd better annotate the message with a 🦻 reaction instead,
|
||||
// to make it clear to users that it's a transcription.
|
||||
//
|
||||
// Regardless of how we post this message, it will be posted as a notice,
|
||||
// which can indicate to the bot (for potential future text-generation purposes) that this message is not a bot message.
|
||||
let (transcribed_text, annotate_message_with_reaction) =
|
||||
if let MessageResponseType::InThread(_) = response_type {
|
||||
(
|
||||
@@ -639,10 +664,22 @@ async fn handle_stage_speech_to_text_actual_transcribing(
|
||||
(speech_to_text_result.text, true)
|
||||
};
|
||||
|
||||
let result = bot
|
||||
.messaging()
|
||||
.send_notice_markdown_no_fail(message_context.room(), transcribed_text, response_type)
|
||||
.await;
|
||||
let result = match msg_type {
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages::Text => {
|
||||
bot.messaging()
|
||||
.send_text_markdown_no_fail(message_context.room(), transcribed_text, response_type)
|
||||
.await
|
||||
}
|
||||
SpeechToTextMessageTypeForNonThreadedOnlyTranscribedMessages::Notice => {
|
||||
bot.messaging()
|
||||
.send_notice_markdown_no_fail(
|
||||
message_context.room(),
|
||||
transcribed_text,
|
||||
response_type,
|
||||
)
|
||||
.await
|
||||
}
|
||||
};
|
||||
|
||||
let event_id = result
|
||||
.map(|result| result.event_id)
|
||||
|
||||
@@ -23,5 +23,6 @@ pub enum ControllerType {
|
||||
ChatCompletion(super::chat_completion::ChatCompletionControllerType),
|
||||
|
||||
ImageGeneration(String),
|
||||
ImageEdit(String),
|
||||
StickerGeneration(String),
|
||||
}
|
||||
|
||||
@@ -4,8 +4,8 @@ mod tests;
|
||||
use super::chat_completion::ChatCompletionControllerType;
|
||||
use crate::{
|
||||
entity::{
|
||||
roomconfig::TextGenerationPrefixRequirementType, InteractionTrigger, MessageContext,
|
||||
MessagePayload,
|
||||
InteractionTrigger, MessageContext, MessagePayload,
|
||||
roomconfig::TextGenerationPrefixRequirementType,
|
||||
},
|
||||
strings,
|
||||
};
|
||||
@@ -36,6 +36,18 @@ pub fn determine_controller(
|
||||
first_thread_message.is_mentioning_bot,
|
||||
)
|
||||
}
|
||||
MessagePayload::Image(_image_message_content) => {
|
||||
let prefix_requirement_type = message_context
|
||||
.room_config_context()
|
||||
.text_generation_prefix_requirement_type();
|
||||
|
||||
match prefix_requirement_type {
|
||||
TextGenerationPrefixRequirementType::CommandPrefix => ControllerType::Ignore,
|
||||
TextGenerationPrefixRequirementType::No => {
|
||||
ControllerType::ChatCompletion(ChatCompletionControllerType::Image)
|
||||
}
|
||||
}
|
||||
}
|
||||
MessagePayload::Encrypted(thread_info) => {
|
||||
if thread_info.is_thread_root_only() {
|
||||
ControllerType::Error(strings::error::message_is_encrypted().to_owned())
|
||||
@@ -84,7 +96,7 @@ fn determine_text_controller(
|
||||
}
|
||||
|
||||
if let Some(prompt) = text.strip_prefix(&format!("{command_prefix} image")) {
|
||||
return ControllerType::ImageGeneration(prompt.trim().to_owned());
|
||||
return super::image::determine_controller(prompt.trim());
|
||||
}
|
||||
|
||||
if let Some(prompt) = text.strip_prefix(&format!("{command_prefix} sticker")) {
|
||||
|
||||
@@ -84,9 +84,17 @@ fn determine_text_controller() {
|
||||
expected: ControllerType::Config(controller::cfg::ConfigControllerType::Help),
|
||||
},
|
||||
TestCase {
|
||||
name: "Image generation",
|
||||
name: "Generic image command causes usage help",
|
||||
input: "!bai image Draw a cat!",
|
||||
is_mentioning_bot: false,
|
||||
room_text_generation_prefix_requirement_type:
|
||||
super::TextGenerationPrefixRequirementType::No,
|
||||
expected: ControllerType::UsageHelp,
|
||||
},
|
||||
TestCase {
|
||||
name: "Image generation",
|
||||
input: "!bai image create Draw a cat!",
|
||||
is_mentioning_bot: false,
|
||||
room_text_generation_prefix_requirement_type:
|
||||
super::TextGenerationPrefixRequirementType::No,
|
||||
expected: ControllerType::ImageGeneration("Draw a cat!".to_owned()),
|
||||
@@ -142,8 +150,7 @@ fn determine_text_controller() {
|
||||
// This test case is the same as the one above, just with a different prefix requirement setting.
|
||||
// We expect the same result.
|
||||
TestCase {
|
||||
name:
|
||||
"Regular message with bot mention triggers completion (command prefix requirement)",
|
||||
name: "Regular message with bot mention triggers completion (command prefix requirement)",
|
||||
input: "Regular text goes here",
|
||||
is_mentioning_bot: true,
|
||||
room_text_generation_prefix_requirement_type:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
use super::ControllerType;
|
||||
|
||||
@@ -77,6 +77,10 @@ pub async fn dispatch_controller(
|
||||
)
|
||||
.await
|
||||
}
|
||||
ControllerType::ImageEdit(prompt) => {
|
||||
super::image::edit::handle(bot, bot.matrix_link().clone(), message_context, prompt)
|
||||
.await
|
||||
}
|
||||
ControllerType::StickerGeneration(prompt) => {
|
||||
super::image::generation::handle_sticker(
|
||||
bot,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
pub async fn handle(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
let sender_can_manage_global_config = message_context.sender_can_manage_global_config();
|
||||
|
||||
16
src/controller/image/determination/mod.rs
Normal file
16
src/controller/image/determination/mod.rs
Normal file
@@ -0,0 +1,16 @@
|
||||
use crate::controller::ControllerType;
|
||||
mod tests;
|
||||
|
||||
pub fn determine_controller(text: &str) -> ControllerType {
|
||||
let text = text.trim();
|
||||
|
||||
if let Some(prompt) = text.strip_prefix("create") {
|
||||
return ControllerType::ImageGeneration(prompt.trim().to_owned());
|
||||
}
|
||||
|
||||
if let Some(prompt) = text.strip_prefix("edit") {
|
||||
return ControllerType::ImageEdit(prompt.trim().to_owned());
|
||||
}
|
||||
|
||||
ControllerType::UsageHelp
|
||||
}
|
||||
38
src/controller/image/determination/tests.rs
Normal file
38
src/controller/image/determination/tests.rs
Normal file
@@ -0,0 +1,38 @@
|
||||
#[test]
|
||||
fn determine_controller() {
|
||||
struct TestCase {
|
||||
name: &'static str,
|
||||
input: &'static str,
|
||||
expected: super::ControllerType,
|
||||
}
|
||||
|
||||
let test_cases = vec![
|
||||
TestCase {
|
||||
name: "Top-level is usage help",
|
||||
input: "",
|
||||
expected: super::ControllerType::UsageHelp,
|
||||
},
|
||||
TestCase {
|
||||
name: "Top-level with some text is usage help",
|
||||
input: "Some text",
|
||||
expected: super::ControllerType::UsageHelp,
|
||||
},
|
||||
TestCase {
|
||||
name: "Image generation triggered by create prefix",
|
||||
input: "create Some prompt",
|
||||
expected: super::ControllerType::ImageGeneration("Some prompt".to_owned()),
|
||||
},
|
||||
TestCase {
|
||||
name: "Image edit triggered by edit prefix",
|
||||
input: "edit Turn this into an anime-style image",
|
||||
expected: super::ControllerType::ImageEdit(
|
||||
"Turn this into an anime-style image".to_owned(),
|
||||
),
|
||||
},
|
||||
];
|
||||
|
||||
for test_case in test_cases {
|
||||
let result = super::determine_controller(test_case.input);
|
||||
assert_eq!(result, test_case.expected, "Test case: {}", test_case.name);
|
||||
}
|
||||
}
|
||||
162
src/controller/image/edit.rs
Normal file
162
src/controller/image/edit.rs
Normal file
@@ -0,0 +1,162 @@
|
||||
use mxlink::{MatrixLink, MessageResponseType};
|
||||
|
||||
use tracing::Instrument;
|
||||
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::ControllerTrait;
|
||||
use crate::agent::provider::ImageEditParams;
|
||||
use crate::agent::provider::ImageSource;
|
||||
use crate::controller::utils::agent::get_effective_agent_for_purpose_or_complain;
|
||||
use crate::conversation::create_llm_conversation_for_matrix_thread;
|
||||
use crate::conversation::matrix::MatrixMessageProcessingParams;
|
||||
use crate::strings;
|
||||
use crate::utils::mime::get_file_extension;
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
pub async fn handle(
|
||||
bot: &Bot,
|
||||
matrix_link: MatrixLink,
|
||||
message_context: &MessageContext,
|
||||
original_prompt: &str,
|
||||
) -> anyhow::Result<()> {
|
||||
let response_type = MessageResponseType::InThread(message_context.thread_info().clone());
|
||||
|
||||
let Some(agent) = get_effective_agent_for_purpose_or_complain(
|
||||
bot,
|
||||
message_context,
|
||||
AgentPurpose::ImageGeneration,
|
||||
response_type.clone(),
|
||||
true,
|
||||
)
|
||||
.await
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
|
||||
if message_context.thread_info().is_thread_root_only() {
|
||||
return send_guide(bot, message_context).await;
|
||||
}
|
||||
|
||||
let _typing_notice_guard = bot.start_typing_notice(message_context.room()).await;
|
||||
|
||||
let params = MatrixMessageProcessingParams::new(
|
||||
bot.user_id().to_owned(),
|
||||
Some(message_context.combined_admin_and_user_regexes()),
|
||||
);
|
||||
|
||||
let conversation = create_llm_conversation_for_matrix_thread(
|
||||
&matrix_link,
|
||||
message_context.room(),
|
||||
message_context.thread_info().root_event_id.clone(),
|
||||
¶ms,
|
||||
)
|
||||
.await?;
|
||||
|
||||
let prompt = if conversation.messages.len() >= 2 {
|
||||
// Skip the first message, which contains the original prompt (which we already have)
|
||||
let other_messages = conversation.messages.iter().skip(1).cloned().collect();
|
||||
|
||||
super::prompt::build(original_prompt, other_messages)
|
||||
} else {
|
||||
original_prompt.to_owned()
|
||||
};
|
||||
|
||||
let got_go_signal = conversation.messages.iter().any(|message| {
|
||||
if let crate::conversation::llm::MessageContent::Text(text) = &message.content {
|
||||
text.to_lowercase() == "go"
|
||||
} else {
|
||||
false
|
||||
}
|
||||
});
|
||||
|
||||
let image_sources: Vec<ImageSource> = conversation
|
||||
.messages
|
||||
.iter()
|
||||
.filter_map(|message| {
|
||||
if let crate::conversation::llm::MessageContent::Image(image_content) = &message.content
|
||||
{
|
||||
Some(image_content.clone().into())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
if !got_go_signal || image_sources.is_empty() {
|
||||
// We don't send the guide again here to avoid being annoying.
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let span = tracing::debug_span!("image_edit", agent_id = agent.identifier().as_string());
|
||||
|
||||
let result = agent
|
||||
.controller()
|
||||
.create_image_edit(&prompt, image_sources, ImageEditParams::default())
|
||||
.instrument(span)
|
||||
.await;
|
||||
|
||||
let response = match result {
|
||||
Ok(response) => response,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
"Error in room {} while trying to generate image edit via agent {}: {:?}",
|
||||
message_context.room_id(),
|
||||
agent.identifier(),
|
||||
err,
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::ImageGeneration,
|
||||
&err,
|
||||
),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
|
||||
let attachment_body_text = format!(
|
||||
"generated-image-edit.{}",
|
||||
get_file_extension(&response.mime_type)
|
||||
);
|
||||
|
||||
let mut event_content = matrix_link
|
||||
.media()
|
||||
.upload_and_prepare_event_content(
|
||||
message_context.room(),
|
||||
&response.mime_type,
|
||||
response.bytes,
|
||||
&attachment_body_text,
|
||||
)
|
||||
.await
|
||||
.map_err(|e| anyhow::anyhow!("Failed to upload and prepare event: {}", e))?;
|
||||
|
||||
matrix_link
|
||||
.messaging()
|
||||
.send_event(
|
||||
message_context.room(),
|
||||
&mut event_content,
|
||||
response_type.clone(),
|
||||
)
|
||||
.await?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn send_guide(bot: &Bot, message_context: &MessageContext) -> anyhow::Result<()> {
|
||||
bot.messaging()
|
||||
.send_text_markdown_no_fail(
|
||||
message_context.room(),
|
||||
strings::image_edit::guide_how_to_proceed(),
|
||||
MessageResponseType::InThread(message_context.thread_info().clone()),
|
||||
)
|
||||
.await;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -2,14 +2,15 @@ use mxlink::{MatrixLink, MessageResponseType};
|
||||
|
||||
use tracing::Instrument;
|
||||
|
||||
use crate::agent::provider::ImageGenerationParams;
|
||||
use crate::agent::AgentPurpose;
|
||||
use crate::agent::ControllerTrait;
|
||||
use crate::agent::provider::ImageGenerationParams;
|
||||
use crate::controller::utils::agent::get_effective_agent_for_purpose_or_complain;
|
||||
use crate::conversation::create_llm_conversation_for_matrix_thread;
|
||||
use crate::conversation::matrix::MatrixMessageProcessingParams;
|
||||
use crate::strings;
|
||||
use crate::{entity::MessageContext, Bot};
|
||||
use crate::utils::mime::get_file_extension;
|
||||
use crate::{Bot, entity::MessageContext};
|
||||
|
||||
// We may make this configurable (per room, etc.) in the future, but for now it's hardcoded.
|
||||
const STICKER_SIZE: &str = "256x256";
|
||||
@@ -42,7 +43,7 @@ pub async fn handle_image(
|
||||
);
|
||||
|
||||
let conversation = create_llm_conversation_for_matrix_thread(
|
||||
matrix_link.clone(),
|
||||
&matrix_link,
|
||||
message_context.room(),
|
||||
message_context.thread_info().root_event_id.clone(),
|
||||
¶ms,
|
||||
@@ -63,11 +64,37 @@ pub async fn handle_image(
|
||||
agent_id = agent.identifier().as_string()
|
||||
);
|
||||
|
||||
let response = agent
|
||||
let result = agent
|
||||
.controller()
|
||||
.generate_image(&prompt, ImageGenerationParams::default())
|
||||
.instrument(span)
|
||||
.await?;
|
||||
.await;
|
||||
|
||||
let response = match result {
|
||||
Ok(response) => response,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
"Error in room {} while trying to generate image via agent {}: {:?}",
|
||||
message_context.room_id(),
|
||||
agent.identifier(),
|
||||
err,
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::ImageGeneration,
|
||||
&err,
|
||||
),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
|
||||
let actual_prompt = response.revised_prompt.as_deref().unwrap_or(&prompt);
|
||||
|
||||
@@ -81,7 +108,10 @@ pub async fn handle_image(
|
||||
.await;
|
||||
}
|
||||
|
||||
let attachment_body_text = format!("Generated image based on: {}", actual_prompt);
|
||||
let attachment_body_text = format!(
|
||||
"generated-image.{}",
|
||||
get_file_extension(&response.mime_type)
|
||||
);
|
||||
|
||||
let mut event_content = matrix_link
|
||||
.media()
|
||||
@@ -151,13 +181,42 @@ pub async fn handle_sticker(
|
||||
.with_cheaper_model_switching_allowed(true)
|
||||
.with_cheaper_quality_switching_allowed(true);
|
||||
|
||||
let response = agent
|
||||
let result = agent
|
||||
.controller()
|
||||
.generate_image(original_prompt, params)
|
||||
.instrument(span)
|
||||
.await?;
|
||||
.await;
|
||||
|
||||
let attachment_body_text = format!("Generated sticker image based on: {}", original_prompt);
|
||||
let response = match result {
|
||||
Ok(response) => response,
|
||||
Err(err) => {
|
||||
tracing::warn!(
|
||||
"Error in room {} while trying to generate sticker via agent {}: {:?}",
|
||||
message_context.room_id(),
|
||||
agent.identifier(),
|
||||
err,
|
||||
);
|
||||
|
||||
bot.messaging()
|
||||
.send_error_markdown_no_fail(
|
||||
message_context.room(),
|
||||
&strings::agent::error_while_serving_purpose(
|
||||
agent.identifier(),
|
||||
&AgentPurpose::ImageGeneration,
|
||||
&err,
|
||||
),
|
||||
response_type,
|
||||
)
|
||||
.await;
|
||||
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
|
||||
let attachment_body_text = format!(
|
||||
"generated-sticker.{}",
|
||||
get_file_extension(&response.mime_type)
|
||||
);
|
||||
|
||||
let mut event_content = matrix_link
|
||||
.media()
|
||||
|
||||
@@ -1,2 +1,6 @@
|
||||
mod determination;
|
||||
pub mod edit;
|
||||
pub mod generation;
|
||||
mod prompt;
|
||||
|
||||
pub use determination::determine_controller;
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
use crate::conversation::llm::{Author, Message};
|
||||
use crate::conversation::llm::{Author, Message, MessageContent};
|
||||
|
||||
/// Builds a prompt from the original prompt and other messages in the conversation.
|
||||
///
|
||||
/// Only messages authored by the user are considered.
|
||||
///
|
||||
/// Messages that say "Again" (regardless of casing) are ignored. They are considered special messages
|
||||
/// which trigger re-generation, but do not need to be included in the prompt criteria.
|
||||
/// Messages that say "Again" or "Go" (regardless of casing) are ignored. They are considered special messages
|
||||
/// which trigger re-generation and "start" respectively, and do not need to be included in the prompt criteria.
|
||||
pub fn build(original_prompt: &str, other_messages: Vec<Message>) -> String {
|
||||
let mut prompt = original_prompt.to_owned();
|
||||
|
||||
@@ -14,7 +14,11 @@ pub fn build(original_prompt: &str, other_messages: Vec<Message>) -> String {
|
||||
.into_iter()
|
||||
.filter(|message| {
|
||||
if let Author::User = message.author {
|
||||
message.message_text.to_lowercase() != "again"
|
||||
if let MessageContent::Text(text) = &message.content {
|
||||
text.to_lowercase() != "again" && text.to_lowercase() != "go"
|
||||
} else {
|
||||
false
|
||||
}
|
||||
} else {
|
||||
false
|
||||
}
|
||||
@@ -24,9 +28,9 @@ pub fn build(original_prompt: &str, other_messages: Vec<Message>) -> String {
|
||||
if !other_messages.is_empty() {
|
||||
prompt.push_str("\nOther criteria:");
|
||||
for message in other_messages {
|
||||
prompt.push_str(
|
||||
format!("\n- {}", message.message_text.replace("\n", ". ").as_str()).as_str(),
|
||||
);
|
||||
if let MessageContent::Text(text) = &message.content {
|
||||
prompt.push_str(format!("\n- {}", text.replace("\n", ". ").as_str()).as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -36,7 +40,7 @@ pub fn build(original_prompt: &str, other_messages: Vec<Message>) -> String {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::build;
|
||||
use super::{Author, Message};
|
||||
use super::{Author, Message, MessageContent};
|
||||
|
||||
struct TestCase {
|
||||
original_prompt: &'static str,
|
||||
@@ -60,7 +64,7 @@ mod tests {
|
||||
original_prompt: "Generate a picture of a dog",
|
||||
messages: vec![Message {
|
||||
author: Author::User,
|
||||
message_text: "Must be blue".to_owned(),
|
||||
content: MessageContent::Text("Must be blue".to_owned()),
|
||||
timestamp,
|
||||
}],
|
||||
expected_prompt: "Generate a picture of a dog\nOther criteria:\n- Must be blue",
|
||||
@@ -68,46 +72,52 @@ mod tests {
|
||||
// Multiple complex user messages dispersed with assistant messages
|
||||
TestCase {
|
||||
original_prompt: "Generate a picture of an elephant",
|
||||
messages: vec![Message {
|
||||
author: Author::User,
|
||||
message_text: "Must be blue".to_owned(),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::Assistant,
|
||||
message_text: "Whatever".to_owned(),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::User,
|
||||
message_text: "Must be 3-legged.\nMust be flying.".to_owned(),
|
||||
timestamp,
|
||||
}],
|
||||
messages: vec![
|
||||
Message {
|
||||
author: Author::User,
|
||||
content: MessageContent::Text("Must be blue".to_owned()),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::Assistant,
|
||||
content: MessageContent::Text("Whatever".to_owned()),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::User,
|
||||
content: MessageContent::Text(
|
||||
"Must be 3-legged.\nMust be flying.".to_owned(),
|
||||
),
|
||||
timestamp,
|
||||
},
|
||||
],
|
||||
expected_prompt: "Generate a picture of an elephant\nOther criteria:\n- Must be blue\n- Must be 3-legged.. Must be flying.",
|
||||
},
|
||||
// "Again" is ignored.
|
||||
TestCase {
|
||||
original_prompt: "Generate a picture of a grizzly bear",
|
||||
messages: vec![Message {
|
||||
author: Author::User,
|
||||
message_text: "Must be blue".to_owned(),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::Assistant,
|
||||
message_text: "Whatever".to_owned(),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::User,
|
||||
message_text: "Again".to_owned(),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::User,
|
||||
message_text: "again".to_owned(),
|
||||
timestamp,
|
||||
}],
|
||||
messages: vec![
|
||||
Message {
|
||||
author: Author::User,
|
||||
content: MessageContent::Text("Must be blue".to_owned()),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::Assistant,
|
||||
content: MessageContent::Text("Whatever".to_owned()),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::User,
|
||||
content: MessageContent::Text("Again".to_owned()),
|
||||
timestamp,
|
||||
},
|
||||
Message {
|
||||
author: Author::User,
|
||||
content: MessageContent::Text("again".to_owned()),
|
||||
timestamp,
|
||||
},
|
||||
],
|
||||
expected_prompt: "Generate a picture of a grizzly bear\nOther criteria:\n- Must be blue",
|
||||
},
|
||||
];
|
||||
|
||||
@@ -1,13 +1,21 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::entity::RoomConfigContext;
|
||||
use crate::{strings, Bot};
|
||||
use crate::{Bot, strings};
|
||||
|
||||
pub async fn handle(
|
||||
bot: &Bot,
|
||||
room: &mxlink::matrix_sdk::Room,
|
||||
room_config_context: &RoomConfigContext,
|
||||
) -> anyhow::Result<()> {
|
||||
if !bot.post_join_self_introduction_enabled() {
|
||||
tracing::debug!(
|
||||
"Post-join self-introduction is disabled - not sending introduction message"
|
||||
);
|
||||
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let agent_manager = bot.agent_manager();
|
||||
|
||||
bot.messaging()
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{agent::AgentProvider, entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, agent::AgentProvider, entity::MessageContext, strings};
|
||||
|
||||
use super::ControllerType;
|
||||
|
||||
|
||||
@@ -3,9 +3,9 @@ use std::ops::Deref;
|
||||
use mxlink::MatrixLink;
|
||||
|
||||
use crate::{
|
||||
Bot,
|
||||
agent::AgentPurpose,
|
||||
entity::{MessageContext, MessagePayload},
|
||||
Bot,
|
||||
};
|
||||
|
||||
mod text_to_speech;
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use mxlink::{MatrixLink, MessageResponseType};
|
||||
|
||||
use mxlink::matrix_sdk::ruma::{
|
||||
events::room::message::TextMessageEventContent, OwnedEventId, OwnedUserId,
|
||||
OwnedEventId, OwnedUserId, events::room::message::TextMessageEventContent,
|
||||
};
|
||||
|
||||
use crate::entity::roomconfig::{
|
||||
@@ -9,8 +9,8 @@ use crate::entity::roomconfig::{
|
||||
};
|
||||
|
||||
use crate::{
|
||||
agent::AgentPurpose, controller::utils::agent::get_effective_agent_for_purpose_or_complain,
|
||||
entity::MessageContext, Bot,
|
||||
Bot, agent::AgentPurpose,
|
||||
controller::utils::agent::get_effective_agent_for_purpose_or_complain, entity::MessageContext,
|
||||
};
|
||||
|
||||
pub(super) async fn handle(
|
||||
@@ -34,7 +34,9 @@ pub(super) async fn handle(
|
||||
reacted_to_event_sender_id,
|
||||
matrix_link.user_id(),
|
||||
) {
|
||||
tracing::debug!("Ignoring request for on-demand text-to-speech (via reaction) due to room configuration");
|
||||
tracing::debug!(
|
||||
"Ignoring request for on-demand text-to-speech (via reaction) due to room configuration"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use mxlink::MessageResponseType;
|
||||
|
||||
use crate::{entity::MessageContext, strings, Bot};
|
||||
use crate::{Bot, entity::MessageContext, strings};
|
||||
|
||||
use super::ControllerType;
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user