Initial commit

This commit is contained in:
Slavi Pantaleev
2024-09-12 13:44:06 +03:00
commit 946aa9d9e9
220 changed files with 26033 additions and 0 deletions

2
.dockerignore Normal file
View File

@@ -0,0 +1,2 @@
/target
/var

32
.editorconfig Normal file
View File

@@ -0,0 +1,32 @@
# This file is the top-most EditorConfig file
root = true
# All Files
[*]
charset = utf-8
end_of_line = lf
indent_style = tab
indent_size = 4
insert_final_newline = true
trim_trailing_whitespace = true
#########################
# File Extension Settings
#########################
[*.{yml,yaml,yml.dist}]
indent_style = space
indent_size = 2
[*.rs]
indent_style = space
indent_size = 4
# Markdown Files
#
# Two spaces at the end of a line in Markdown mean "new line",
# so trimming trailing whitespace for such files can cause breakage.
[*.md]
trim_trailing_whitespace = false
indent_style = space
indent_size = 2

51
.github/workflows/workflow.yml vendored Normal file
View File

@@ -0,0 +1,51 @@
name: CI (main and tags)
on:
push:
branches: [ "main" ]
tags: [ "v*" ]
schedule:
- cron: '0 0 * * 1'
permissions:
checks: write
contents: write
packages: write
pull-requests: read
jobs:
test-and-clippy:
name: Unit testing and linting
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: dtolnay/rust-toolchain@stable
- run: cargo test --all-features
- run: cargo clippy
build-publish:
name: Build and Publish
runs-on: self-hosted
steps:
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v1
- name: Login to ghcr.io
uses: docker/login-action@v3
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Extract metadata (tags, labels) for Docker
id: meta
uses: docker/metadata-action@v5
with:
images: |
ghcr.io/${{ github.repository }}
registry.etke.cc/${{ github.repository }}
tags: |
type=raw,value=latest,enable=${{ github.ref_name == 'main' }}
type=semver,pattern={{raw}}
- name: Build and push
uses: docker/build-push-action@v6
with:
platforms: linux/amd64,linux/arm64
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}

2
.gitignore vendored Normal file
View File

@@ -0,0 +1,2 @@
/target
/var

1
CHANGELOG.md Normal file
View File

@@ -0,0 +1 @@
There's nothing here yet.

4380
Cargo.lock generated Normal file

File diff suppressed because it is too large Load Diff

42
Cargo.toml Normal file
View File

@@ -0,0 +1,42 @@
[package]
name = "baibot"
description = "A Matrix bot for using diffent capabilities (text-generation, text-to-speech, speech-to-text, image-generation, etc.) of AI / Large Language Models"
authors = ["Slavi Pantaleev <slavi@devture.com>"]
repository = "https://github.com/etkecc/baibot"
license = "AGPL-3.0-or-later"
readme = "README.md"
keywords = ["matrix", "chat", "LLM", "openai", "anthropic", "localai"]
exclude = [".editorconfig", "justfile", "docs/", "etc/", "Dockerfile", ".dockerignore", ".gitignore"]
version = "1.0.0"
edition = "2021"
[lib]
name = "baibot"
path = "src/lib.rs"
[dependencies]
anthropic-rs = "0.1.*"
anyhow = "1.0.*"
async-openai = "0.24.*"
base64 = "0.22.*"
# We'd rather not depend on this, but we cannot use the ruma-events EventContent macro without it.
matrix-sdk = { version = "0.7.1", default-features = false }
mxidwc = "1.0.*"
mxlink = "1.0.*"
openai_api_rust = { git = "https://github.com/etkecc/openai_api_rust.git", branch = "next" }
quick_cache = "0.6.*"
regex = "1.10.*"
serde = { version = "1.0.*", features = ["derive"], default-features = false }
serde_json = "1.0.*"
serde_yaml = "0.9.*"
tempfile = "3.12.*"
tiktoken-rs = { version = "0.5.*", features = ["async-openai"] }
tokio = { version = "1.40.*", features = ["rt", "rt-multi-thread", "macros"] }
tracing = "0.1.*"
tracing-subscriber = { version = "0.3.*", features = ["env-filter"] }
url = "2.5.*"
[profile.release]
strip = true
opt-level = "z"
lto = "thin"

43
Dockerfile Normal file
View File

@@ -0,0 +1,43 @@
#######################################
# #
# Stage 1: building #
# #
#######################################
FROM docker.io/rust:1.80.1-slim-bookworm AS build
RUN apt-get update && apt-get install -y build-essential pkg-config libssl-dev libsqlite3-dev
ENV CARGO_HOME=/cargo
ENV CARGO_TARGET_DIR=/target
WORKDIR /app
COPY . /app
RUN --mount=type=cache,target=/cargo,sharing=locked \
--mount=type=cache,target=/target,sharing=locked \
cargo build --release
# Move it out of the mounted cache, so we can copy it in the next stage.
RUN --mount=type=cache,target=/target,sharing=locked \
cp /target/release/baibot /baibot
#######################################
# #
# Stage 2: packaging #
# #
#######################################
FROM docker.io/debian:bookworm-slim
RUN apt-get update && apt-get install -y ca-certificates sqlite3
WORKDIR /app
COPY --from=build /baibot .
ENTRYPOINT ["/bin/sh", "-c"]
CMD ["/app/baibot"]

661
LICENSE Normal file
View File

@@ -0,0 +1,661 @@
GNU AFFERO GENERAL PUBLIC LICENSE
Version 3, 19 November 2007
Copyright (C) 2007 Free Software Foundation, Inc. <https://fsf.org/>
Everyone is permitted to copy and distribute verbatim copies
of this license document, but changing it is not allowed.
Preamble
The GNU Affero General Public License is a free, copyleft license for
software and other kinds of works, specifically designed to ensure
cooperation with the community in the case of network server software.
The licenses for most software and other practical works are designed
to take away your freedom to share and change the works. By contrast,
our General Public Licenses are intended to guarantee your freedom to
share and change all versions of a program--to make sure it remains free
software for all its users.
When we speak of free software, we are referring to freedom, not
price. Our General Public Licenses are designed to make sure that you
have the freedom to distribute copies of free software (and charge for
them if you wish), that you receive source code or can get it if you
want it, that you can change the software or use pieces of it in new
free programs, and that you know you can do these things.
Developers that use our General Public Licenses protect your rights
with two steps: (1) assert copyright on the software, and (2) offer
you this License which gives you legal permission to copy, distribute
and/or modify the software.
A secondary benefit of defending all users' freedom is that
improvements made in alternate versions of the program, if they
receive widespread use, become available for other developers to
incorporate. Many developers of free software are heartened and
encouraged by the resulting cooperation. However, in the case of
software used on network servers, this result may fail to come about.
The GNU General Public License permits making a modified version and
letting the public access it on a server without ever releasing its
source code to the public.
The GNU Affero General Public License is designed specifically to
ensure that, in such cases, the modified source code becomes available
to the community. It requires the operator of a network server to
provide the source code of the modified version running there to the
users of that server. Therefore, public use of a modified version, on
a publicly accessible server, gives the public access to the source
code of the modified version.
An older license, called the Affero General Public License and
published by Affero, was designed to accomplish similar goals. This is
a different license, not a version of the Affero GPL, but Affero has
released a new version of the Affero GPL which permits relicensing under
this license.
The precise terms and conditions for copying, distribution and
modification follow.
TERMS AND CONDITIONS
0. Definitions.
"This License" refers to version 3 of the GNU Affero General Public License.
"Copyright" also means copyright-like laws that apply to other kinds of
works, such as semiconductor masks.
"The Program" refers to any copyrightable work licensed under this
License. Each licensee is addressed as "you". "Licensees" and
"recipients" may be individuals or organizations.
To "modify" a work means to copy from or adapt all or part of the work
in a fashion requiring copyright permission, other than the making of an
exact copy. The resulting work is called a "modified version" of the
earlier work or a work "based on" the earlier work.
A "covered work" means either the unmodified Program or a work based
on the Program.
To "propagate" a work means to do anything with it that, without
permission, would make you directly or secondarily liable for
infringement under applicable copyright law, except executing it on a
computer or modifying a private copy. Propagation includes copying,
distribution (with or without modification), making available to the
public, and in some countries other activities as well.
To "convey" a work means any kind of propagation that enables other
parties to make or receive copies. Mere interaction with a user through
a computer network, with no transfer of a copy, is not conveying.
An interactive user interface displays "Appropriate Legal Notices"
to the extent that it includes a convenient and prominently visible
feature that (1) displays an appropriate copyright notice, and (2)
tells the user that there is no warranty for the work (except to the
extent that warranties are provided), that licensees may convey the
work under this License, and how to view a copy of this License. If
the interface presents a list of user commands or options, such as a
menu, a prominent item in the list meets this criterion.
1. Source Code.
The "source code" for a work means the preferred form of the work
for making modifications to it. "Object code" means any non-source
form of a work.
A "Standard Interface" means an interface that either is an official
standard defined by a recognized standards body, or, in the case of
interfaces specified for a particular programming language, one that
is widely used among developers working in that language.
The "System Libraries" of an executable work include anything, other
than the work as a whole, that (a) is included in the normal form of
packaging a Major Component, but which is not part of that Major
Component, and (b) serves only to enable use of the work with that
Major Component, or to implement a Standard Interface for which an
implementation is available to the public in source code form. A
"Major Component", in this context, means a major essential component
(kernel, window system, and so on) of the specific operating system
(if any) on which the executable work runs, or a compiler used to
produce the work, or an object code interpreter used to run it.
The "Corresponding Source" for a work in object code form means all
the source code needed to generate, install, and (for an executable
work) run the object code and to modify the work, including scripts to
control those activities. However, it does not include the work's
System Libraries, or general-purpose tools or generally available free
programs which are used unmodified in performing those activities but
which are not part of the work. For example, Corresponding Source
includes interface definition files associated with source files for
the work, and the source code for shared libraries and dynamically
linked subprograms that the work is specifically designed to require,
such as by intimate data communication or control flow between those
subprograms and other parts of the work.
The Corresponding Source need not include anything that users
can regenerate automatically from other parts of the Corresponding
Source.
The Corresponding Source for a work in source code form is that
same work.
2. Basic Permissions.
All rights granted under this License are granted for the term of
copyright on the Program, and are irrevocable provided the stated
conditions are met. This License explicitly affirms your unlimited
permission to run the unmodified Program. The output from running a
covered work is covered by this License only if the output, given its
content, constitutes a covered work. This License acknowledges your
rights of fair use or other equivalent, as provided by copyright law.
You may make, run and propagate covered works that you do not
convey, without conditions so long as your license otherwise remains
in force. You may convey covered works to others for the sole purpose
of having them make modifications exclusively for you, or provide you
with facilities for running those works, provided that you comply with
the terms of this License in conveying all material for which you do
not control copyright. Those thus making or running the covered works
for you must do so exclusively on your behalf, under your direction
and control, on terms that prohibit them from making any copies of
your copyrighted material outside their relationship with you.
Conveying under any other circumstances is permitted solely under
the conditions stated below. Sublicensing is not allowed; section 10
makes it unnecessary.
3. Protecting Users' Legal Rights From Anti-Circumvention Law.
No covered work shall be deemed part of an effective technological
measure under any applicable law fulfilling obligations under article
11 of the WIPO copyright treaty adopted on 20 December 1996, or
similar laws prohibiting or restricting circumvention of such
measures.
When you convey a covered work, you waive any legal power to forbid
circumvention of technological measures to the extent such circumvention
is effected by exercising rights under this License with respect to
the covered work, and you disclaim any intention to limit operation or
modification of the work as a means of enforcing, against the work's
users, your or third parties' legal rights to forbid circumvention of
technological measures.
4. Conveying Verbatim Copies.
You may convey verbatim copies of the Program's source code as you
receive it, in any medium, provided that you conspicuously and
appropriately publish on each copy an appropriate copyright notice;
keep intact all notices stating that this License and any
non-permissive terms added in accord with section 7 apply to the code;
keep intact all notices of the absence of any warranty; and give all
recipients a copy of this License along with the Program.
You may charge any price or no price for each copy that you convey,
and you may offer support or warranty protection for a fee.
5. Conveying Modified Source Versions.
You may convey a work based on the Program, or the modifications to
produce it from the Program, in the form of source code under the
terms of section 4, provided that you also meet all of these conditions:
a) The work must carry prominent notices stating that you modified
it, and giving a relevant date.
b) The work must carry prominent notices stating that it is
released under this License and any conditions added under section
7. This requirement modifies the requirement in section 4 to
"keep intact all notices".
c) You must license the entire work, as a whole, under this
License to anyone who comes into possession of a copy. This
License will therefore apply, along with any applicable section 7
additional terms, to the whole of the work, and all its parts,
regardless of how they are packaged. This License gives no
permission to license the work in any other way, but it does not
invalidate such permission if you have separately received it.
d) If the work has interactive user interfaces, each must display
Appropriate Legal Notices; however, if the Program has interactive
interfaces that do not display Appropriate Legal Notices, your
work need not make them do so.
A compilation of a covered work with other separate and independent
works, which are not by their nature extensions of the covered work,
and which are not combined with it such as to form a larger program,
in or on a volume of a storage or distribution medium, is called an
"aggregate" if the compilation and its resulting copyright are not
used to limit the access or legal rights of the compilation's users
beyond what the individual works permit. Inclusion of a covered work
in an aggregate does not cause this License to apply to the other
parts of the aggregate.
6. Conveying Non-Source Forms.
You may convey a covered work in object code form under the terms
of sections 4 and 5, provided that you also convey the
machine-readable Corresponding Source under the terms of this License,
in one of these ways:
a) Convey the object code in, or embodied in, a physical product
(including a physical distribution medium), accompanied by the
Corresponding Source fixed on a durable physical medium
customarily used for software interchange.
b) Convey the object code in, or embodied in, a physical product
(including a physical distribution medium), accompanied by a
written offer, valid for at least three years and valid for as
long as you offer spare parts or customer support for that product
model, to give anyone who possesses the object code either (1) a
copy of the Corresponding Source for all the software in the
product that is covered by this License, on a durable physical
medium customarily used for software interchange, for a price no
more than your reasonable cost of physically performing this
conveying of source, or (2) access to copy the
Corresponding Source from a network server at no charge.
c) Convey individual copies of the object code with a copy of the
written offer to provide the Corresponding Source. This
alternative is allowed only occasionally and noncommercially, and
only if you received the object code with such an offer, in accord
with subsection 6b.
d) Convey the object code by offering access from a designated
place (gratis or for a charge), and offer equivalent access to the
Corresponding Source in the same way through the same place at no
further charge. You need not require recipients to copy the
Corresponding Source along with the object code. If the place to
copy the object code is a network server, the Corresponding Source
may be on a different server (operated by you or a third party)
that supports equivalent copying facilities, provided you maintain
clear directions next to the object code saying where to find the
Corresponding Source. Regardless of what server hosts the
Corresponding Source, you remain obligated to ensure that it is
available for as long as needed to satisfy these requirements.
e) Convey the object code using peer-to-peer transmission, provided
you inform other peers where the object code and Corresponding
Source of the work are being offered to the general public at no
charge under subsection 6d.
A separable portion of the object code, whose source code is excluded
from the Corresponding Source as a System Library, need not be
included in conveying the object code work.
A "User Product" is either (1) a "consumer product", which means any
tangible personal property which is normally used for personal, family,
or household purposes, or (2) anything designed or sold for incorporation
into a dwelling. In determining whether a product is a consumer product,
doubtful cases shall be resolved in favor of coverage. For a particular
product received by a particular user, "normally used" refers to a
typical or common use of that class of product, regardless of the status
of the particular user or of the way in which the particular user
actually uses, or expects or is expected to use, the product. A product
is a consumer product regardless of whether the product has substantial
commercial, industrial or non-consumer uses, unless such uses represent
the only significant mode of use of the product.
"Installation Information" for a User Product means any methods,
procedures, authorization keys, or other information required to install
and execute modified versions of a covered work in that User Product from
a modified version of its Corresponding Source. The information must
suffice to ensure that the continued functioning of the modified object
code is in no case prevented or interfered with solely because
modification has been made.
If you convey an object code work under this section in, or with, or
specifically for use in, a User Product, and the conveying occurs as
part of a transaction in which the right of possession and use of the
User Product is transferred to the recipient in perpetuity or for a
fixed term (regardless of how the transaction is characterized), the
Corresponding Source conveyed under this section must be accompanied
by the Installation Information. But this requirement does not apply
if neither you nor any third party retains the ability to install
modified object code on the User Product (for example, the work has
been installed in ROM).
The requirement to provide Installation Information does not include a
requirement to continue to provide support service, warranty, or updates
for a work that has been modified or installed by the recipient, or for
the User Product in which it has been modified or installed. Access to a
network may be denied when the modification itself materially and
adversely affects the operation of the network or violates the rules and
protocols for communication across the network.
Corresponding Source conveyed, and Installation Information provided,
in accord with this section must be in a format that is publicly
documented (and with an implementation available to the public in
source code form), and must require no special password or key for
unpacking, reading or copying.
7. Additional Terms.
"Additional permissions" are terms that supplement the terms of this
License by making exceptions from one or more of its conditions.
Additional permissions that are applicable to the entire Program shall
be treated as though they were included in this License, to the extent
that they are valid under applicable law. If additional permissions
apply only to part of the Program, that part may be used separately
under those permissions, but the entire Program remains governed by
this License without regard to the additional permissions.
When you convey a copy of a covered work, you may at your option
remove any additional permissions from that copy, or from any part of
it. (Additional permissions may be written to require their own
removal in certain cases when you modify the work.) You may place
additional permissions on material, added by you to a covered work,
for which you have or can give appropriate copyright permission.
Notwithstanding any other provision of this License, for material you
add to a covered work, you may (if authorized by the copyright holders of
that material) supplement the terms of this License with terms:
a) Disclaiming warranty or limiting liability differently from the
terms of sections 15 and 16 of this License; or
b) Requiring preservation of specified reasonable legal notices or
author attributions in that material or in the Appropriate Legal
Notices displayed by works containing it; or
c) Prohibiting misrepresentation of the origin of that material, or
requiring that modified versions of such material be marked in
reasonable ways as different from the original version; or
d) Limiting the use for publicity purposes of names of licensors or
authors of the material; or
e) Declining to grant rights under trademark law for use of some
trade names, trademarks, or service marks; or
f) Requiring indemnification of licensors and authors of that
material by anyone who conveys the material (or modified versions of
it) with contractual assumptions of liability to the recipient, for
any liability that these contractual assumptions directly impose on
those licensors and authors.
All other non-permissive additional terms are considered "further
restrictions" within the meaning of section 10. If the Program as you
received it, or any part of it, contains a notice stating that it is
governed by this License along with a term that is a further
restriction, you may remove that term. If a license document contains
a further restriction but permits relicensing or conveying under this
License, you may add to a covered work material governed by the terms
of that license document, provided that the further restriction does
not survive such relicensing or conveying.
If you add terms to a covered work in accord with this section, you
must place, in the relevant source files, a statement of the
additional terms that apply to those files, or a notice indicating
where to find the applicable terms.
Additional terms, permissive or non-permissive, may be stated in the
form of a separately written license, or stated as exceptions;
the above requirements apply either way.
8. Termination.
You may not propagate or modify a covered work except as expressly
provided under this License. Any attempt otherwise to propagate or
modify it is void, and will automatically terminate your rights under
this License (including any patent licenses granted under the third
paragraph of section 11).
However, if you cease all violation of this License, then your
license from a particular copyright holder is reinstated (a)
provisionally, unless and until the copyright holder explicitly and
finally terminates your license, and (b) permanently, if the copyright
holder fails to notify you of the violation by some reasonable means
prior to 60 days after the cessation.
Moreover, your license from a particular copyright holder is
reinstated permanently if the copyright holder notifies you of the
violation by some reasonable means, this is the first time you have
received notice of violation of this License (for any work) from that
copyright holder, and you cure the violation prior to 30 days after
your receipt of the notice.
Termination of your rights under this section does not terminate the
licenses of parties who have received copies or rights from you under
this License. If your rights have been terminated and not permanently
reinstated, you do not qualify to receive new licenses for the same
material under section 10.
9. Acceptance Not Required for Having Copies.
You are not required to accept this License in order to receive or
run a copy of the Program. Ancillary propagation of a covered work
occurring solely as a consequence of using peer-to-peer transmission
to receive a copy likewise does not require acceptance. However,
nothing other than this License grants you permission to propagate or
modify any covered work. These actions infringe copyright if you do
not accept this License. Therefore, by modifying or propagating a
covered work, you indicate your acceptance of this License to do so.
10. Automatic Licensing of Downstream Recipients.
Each time you convey a covered work, the recipient automatically
receives a license from the original licensors, to run, modify and
propagate that work, subject to this License. You are not responsible
for enforcing compliance by third parties with this License.
An "entity transaction" is a transaction transferring control of an
organization, or substantially all assets of one, or subdividing an
organization, or merging organizations. If propagation of a covered
work results from an entity transaction, each party to that
transaction who receives a copy of the work also receives whatever
licenses to the work the party's predecessor in interest had or could
give under the previous paragraph, plus a right to possession of the
Corresponding Source of the work from the predecessor in interest, if
the predecessor has it or can get it with reasonable efforts.
You may not impose any further restrictions on the exercise of the
rights granted or affirmed under this License. For example, you may
not impose a license fee, royalty, or other charge for exercise of
rights granted under this License, and you may not initiate litigation
(including a cross-claim or counterclaim in a lawsuit) alleging that
any patent claim is infringed by making, using, selling, offering for
sale, or importing the Program or any portion of it.
11. Patents.
A "contributor" is a copyright holder who authorizes use under this
License of the Program or a work on which the Program is based. The
work thus licensed is called the contributor's "contributor version".
A contributor's "essential patent claims" are all patent claims
owned or controlled by the contributor, whether already acquired or
hereafter acquired, that would be infringed by some manner, permitted
by this License, of making, using, or selling its contributor version,
but do not include claims that would be infringed only as a
consequence of further modification of the contributor version. For
purposes of this definition, "control" includes the right to grant
patent sublicenses in a manner consistent with the requirements of
this License.
Each contributor grants you a non-exclusive, worldwide, royalty-free
patent license under the contributor's essential patent claims, to
make, use, sell, offer for sale, import and otherwise run, modify and
propagate the contents of its contributor version.
In the following three paragraphs, a "patent license" is any express
agreement or commitment, however denominated, not to enforce a patent
(such as an express permission to practice a patent or covenant not to
sue for patent infringement). To "grant" such a patent license to a
party means to make such an agreement or commitment not to enforce a
patent against the party.
If you convey a covered work, knowingly relying on a patent license,
and the Corresponding Source of the work is not available for anyone
to copy, free of charge and under the terms of this License, through a
publicly available network server or other readily accessible means,
then you must either (1) cause the Corresponding Source to be so
available, or (2) arrange to deprive yourself of the benefit of the
patent license for this particular work, or (3) arrange, in a manner
consistent with the requirements of this License, to extend the patent
license to downstream recipients. "Knowingly relying" means you have
actual knowledge that, but for the patent license, your conveying the
covered work in a country, or your recipient's use of the covered work
in a country, would infringe one or more identifiable patents in that
country that you have reason to believe are valid.
If, pursuant to or in connection with a single transaction or
arrangement, you convey, or propagate by procuring conveyance of, a
covered work, and grant a patent license to some of the parties
receiving the covered work authorizing them to use, propagate, modify
or convey a specific copy of the covered work, then the patent license
you grant is automatically extended to all recipients of the covered
work and works based on it.
A patent license is "discriminatory" if it does not include within
the scope of its coverage, prohibits the exercise of, or is
conditioned on the non-exercise of one or more of the rights that are
specifically granted under this License. You may not convey a covered
work if you are a party to an arrangement with a third party that is
in the business of distributing software, under which you make payment
to the third party based on the extent of your activity of conveying
the work, and under which the third party grants, to any of the
parties who would receive the covered work from you, a discriminatory
patent license (a) in connection with copies of the covered work
conveyed by you (or copies made from those copies), or (b) primarily
for and in connection with specific products or compilations that
contain the covered work, unless you entered into that arrangement,
or that patent license was granted, prior to 28 March 2007.
Nothing in this License shall be construed as excluding or limiting
any implied license or other defenses to infringement that may
otherwise be available to you under applicable patent law.
12. No Surrender of Others' Freedom.
If conditions are imposed on you (whether by court order, agreement or
otherwise) that contradict the conditions of this License, they do not
excuse you from the conditions of this License. If you cannot convey a
covered work so as to satisfy simultaneously your obligations under this
License and any other pertinent obligations, then as a consequence you may
not convey it at all. For example, if you agree to terms that obligate you
to collect a royalty for further conveying from those to whom you convey
the Program, the only way you could satisfy both those terms and this
License would be to refrain entirely from conveying the Program.
13. Remote Network Interaction; Use with the GNU General Public License.
Notwithstanding any other provision of this License, if you modify the
Program, your modified version must prominently offer all users
interacting with it remotely through a computer network (if your version
supports such interaction) an opportunity to receive the Corresponding
Source of your version by providing access to the Corresponding Source
from a network server at no charge, through some standard or customary
means of facilitating copying of software. This Corresponding Source
shall include the Corresponding Source for any work covered by version 3
of the GNU General Public License that is incorporated pursuant to the
following paragraph.
Notwithstanding any other provision of this License, you have
permission to link or combine any covered work with a work licensed
under version 3 of the GNU General Public License into a single
combined work, and to convey the resulting work. The terms of this
License will continue to apply to the part which is the covered work,
but the work with which it is combined will remain governed by version
3 of the GNU General Public License.
14. Revised Versions of this License.
The Free Software Foundation may publish revised and/or new versions of
the GNU Affero General Public License from time to time. Such new versions
will be similar in spirit to the present version, but may differ in detail to
address new problems or concerns.
Each version is given a distinguishing version number. If the
Program specifies that a certain numbered version of the GNU Affero General
Public License "or any later version" applies to it, you have the
option of following the terms and conditions either of that numbered
version or of any later version published by the Free Software
Foundation. If the Program does not specify a version number of the
GNU Affero General Public License, you may choose any version ever published
by the Free Software Foundation.
If the Program specifies that a proxy can decide which future
versions of the GNU Affero General Public License can be used, that proxy's
public statement of acceptance of a version permanently authorizes you
to choose that version for the Program.
Later license versions may give you additional or different
permissions. However, no additional obligations are imposed on any
author or copyright holder as a result of your choosing to follow a
later version.
15. Disclaimer of Warranty.
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY
APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT
HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY
OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO,
THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM
IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF
ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
16. Limitation of Liability.
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS
THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY
GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE
USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF
DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD
PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS),
EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF
SUCH DAMAGES.
17. Interpretation of Sections 15 and 16.
If the disclaimer of warranty and limitation of liability provided
above cannot be given local legal effect according to their terms,
reviewing courts shall apply local law that most closely approximates
an absolute waiver of all civil liability in connection with the
Program, unless a warranty or assumption of liability accompanies a
copy of the Program in return for a fee.
END OF TERMS AND CONDITIONS
How to Apply These Terms to Your New Programs
If you develop a new program, and you want it to be of the greatest
possible use to the public, the best way to achieve this is to make it
free software which everyone can redistribute and change under these terms.
To do so, attach the following notices to the program. It is safest
to attach them to the start of each source file to most effectively
state the exclusion of warranty; and each file should have at least
the "copyright" line and a pointer to where the full notice is found.
<one line to give the program's name and a brief idea of what it does.>
Copyright (C) <year> <name of author>
This program is free software: you can redistribute it and/or modify
it under the terms of the GNU Affero General Public License as published
by the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU Affero General Public License for more details.
You should have received a copy of the GNU Affero General Public License
along with this program. If not, see <https://www.gnu.org/licenses/>.
Also add information on how to contact you by electronic and paper mail.
If your software can interact with users remotely through a computer
network, you should also make sure that it provides a way for users to
get its source. For example, if your program is a web application, its
interface could display a "Source" link that leads users to an archive
of the code. There are many ways you could offer source, and different
solutions will be better for different programs; see section 13 for the
specific requirements.
You should also get your employer (if you work as a programmer) or school,
if any, to sign a "copyright disclaimer" for the program, if necessary.
For more information on this, and how to apply and follow the GNU AGPL, see
<https://www.gnu.org/licenses/>.

77
README.md Normal file
View File

@@ -0,0 +1,77 @@
<p align="center">
<img src="./etc/assets/baibot.svg" alt="baibot logo" width="150" />
<h1 align="center">baibot</h1>
</p>
🤖 baibot is an [AI](https://en.wikipedia.org/wiki/Artificial_intelligence) ([Large Language Model](https://en.wikipedia.org/wiki/Large_language_model)) bot for [Matrix](https://matrix.org/) built by [etke.cc](https://etke.cc/) (managed Matrix servers).
The name is pronounced 'bye' and is a play on [AI](https://en.wikipedia.org/wiki/Artificial_intelligence), referencing the fictional character [🇧🇬 Bai Ganyo](https://en.wikipedia.org/wiki/Bay_Ganyo).
It's designed as a more private and [featureful](#-features) alternative to [matrix-chatgpt-bot](https://github.com/matrixgpt/matrix-chatgpt-bot).
It's influenced by [chaz](https://github.com/arcuru/chaz), but does **not** use the [AIChat](https://github.com/sigoden/aichat) CLI tool and instead does everything in-process, without forking.
## 🌟 Features
- 🎨 Encourages **[provider](./docs/providers.md) choice** ([Anthropic](./docs/providers.md#anthropic), [Groq](./docs/providers.md#groq), [LocalAI](./docs/providers.md#localai), [OpenAI](./docs/providers.md#openai) and [☁️ many more](./docs/providers.md#️-providers)) as well as **[mixing & matching models](./docs/features.md#-mixing--matching-models)**:
- Supports **different use purposes** (depending on the [☁️ provider](./docs/providers.md) & model):
- [💬 text-generation](./docs/features.md#-text-generation): communicating with you via text
- [🦻 speech-to-text](./docs/features.md#-speech-to-text): turning your voice messages into text
- [🗣️ text-to-speech](./docs/features.md#️-text-to-speech): turning bot or users text messages into voice messages
- [🖌️ image-generation](./docs/features.md#-image-generation): generating images based on instructions
- 🪄 Supports [seamless voice interaction](./docs/features.md#seamless-voice-interaction) (turning user voice messages into text, answering in text, then turning that text back into voice)
- 🦻 Supports [transcribe-only mode](./docs/features.md#transcribe-only-mode) (turning user voice messages into text, without doing text-generation)
- 🗣️ Supports [text-to-speech-only mode](./docs/features.md#text-to-speech-only-mode) (turning user text messages into voice, without doing text-generation)
- 🔒 Supports [encryption](./docs/features.md#-encryption) for Matrix communication and Account-Data-stored configuration
- ♻️ Supports [context-management](./docs/configuration/text-generation.md#️-context-management) handling on some models (automatically adjusting the message history length, etc.)
- 🛠️ Allows **customizing much of the bot's [configuration](./docs/configuration/README.md)** at runtime (using commands sent via chat)
- 👥 **Actively maintained** by the team at [etke.cc](https://etke.cc/)
## 🖼️ Screenshots
![Introduction and general usage](./docs/screenshots/introduction-and-general-usage.webp)
You can find more screenshots on the the [🌟 Features](./docs/features.md) and other [📚 Documentation](./docs/README.md) pages, as well as in the [docs/screenshots](./docs/screenshots) directory.
## 🚀 Getting Started
🗲 For a quick experiment, you can refer to the [🧑‍💻 development documentation](./docs/development.md) which contains information on how to build and run the bot (and its various dependency services) locally.
For a real installation, see the [🚀 Installation](./docs/installation.md) documentation which contains information on [🐋 Running in a container](./docs/installation.md#-running-in-a-container) and [🖥️️️️️ Running a binary](./docs/installation.md#-running-a-binary).
## 📚 Documentation
See the bot's [📚 documentation](./docs/README.md) for more information on how to use and configure the bot.
## 💻 Development
See the bot's [🧑‍💻 development documentation](./docs/development.md) for more information on how to develop on the bot.
## 📜 Changes
This bot evolves over time, sometimes with backward-incompatible changes.
When updating the bot, refer to [the changelog](CHANGELOG.md) to catch up with what's new.
## 🆘 Support
- Matrix room: [#baibot:etke.cc](https://matrix.to/#/#baibot:etke.cc)
- GitHub issues: [etkecc/baibot/issues](https://github.com/etkecc/baibot/issues)
- (for [etke.cc](https://etke.cc/) customers): etke.cc [support](https://etke.cc/contacts/)

9
docs/README.md Normal file
View File

@@ -0,0 +1,9 @@
# Table of Contents
- [🔒 Access](./access.md)
- [🤖 Agents](./agents.md) and [☁️ Providers](./providers.md)
- [🛠️ Configuration](./configuration/README.md)
- [🌟 Features](./features.md)
- [📖 Usage](./usage.md)
- [🚀 Installation](./installation.md)
- [💻 Development](./development.md)

46
docs/access.md Normal file
View File

@@ -0,0 +1,46 @@
## 🔒 Access
This bot employs access control to decide who can use its services and manage its configuration.
### 👋 Joining rooms
The bot automatically joins rooms when invited by someone considered a bot [user](#-users).
### 👥 Users
The bot will ignore messages (and room invitations) from unallowed users.
Users can **use all the bot's [features](./features.md)** ([💬 Text Generation](./features.md#-text-generation), [🦻 Speech-to-Text](./features.md#-speech-to-text), etc.), but **cannot manage the bot's configuration**.
The bot can be used by users that match some [dynamically](./configuration/README.md#dynamic-configuration) configured [Matrix user id](https://spec.matrix.org/v1.11/#users) patterns.
The following commands are available:
- **Show** the currently allowed users: `!bai access users`
- **Set** the list of allowed users: `!bai access set-users SPACE_SEPARATED_PATTERNS`
Example patterns: `@*:example.com @*:another.com @someone:company.org`
### 👮‍♂️ Administrators
Administrators can **manage the bot's configuration and access control**.
The bot can be administrated by users that match some [statically](./configuration/README.md#static-configuration) configured [Matrix user id](https://spec.matrix.org/v1.11/#users) patterns.
Administrators cannot be changed without adjusting the bot's configuration on the server.
### 💼 Room-local agent managers
Room-local agent managers are users privileged to **create their own [agents](./agents.md)** (see `!bai agent`) in rooms.
Letting regular users create agents which contact arbitrary network services **may be a security issue**.
No room-local agent manager patterns are configured, so new agents can only be created by administrators.
The following commands are available:
- **Show** the currently allowed users: `!bai access room-local-agent-managers`
- **Set** the list of allowed users: `!bai access set-room-local-agent-managers SPACE_SEPARATED_PATTERNS`
Example patterns: `@*:synapse.127.0.0.1.nip.io @*:another.com @someone:company.org`

61
docs/agents.md Normal file
View File

@@ -0,0 +1,61 @@
## 🤖 Agents
An agent is an instantiation and configuration of some [☁️ provider](./providers.md).
It can support different capabilities (text-generation, speech-to-text, etc.) depending on the provider used and on the configuration of the agent.
Agents can be set as **[🤝 handlers](./configuration/handlers.md) for various purposes** (text-generation, speech-to-text, etc.) globally or in specific rooms. Send a `!bai config status` command to see the current configuration.
Agents can be **defined [statically](./configuration/README.md#static-configuration)** (in the server configuration) **or dynamically** (via commands sent to the bot).
When [creating agents](#creating-agents) dynamically, you can do it **per-room or globally**.
Globally-defined agents can be used by any authorized bot user in any room, while room-local agents can only be used in the room where they were defined.
Agent configuration (like all other configuration) is stored in the Matrix Account Data of the bot user and is **potentially encrypted** (if enabled in the configuration), so that your configuration data is safe even on untrusted homeservers.
### Listing agents
To **list** all available agents: `!bai agent list`
#### Creating agents
See a [🖼️ Screenshot of the agent creation process](./screenshots/agent-creation.webp).
To **create** a new agent, you need to specify the [provider](./providers.md) and an agent id of your choosing.
- **Create** a new agent:
- (Accessible in **this room only**) `!bai agent create-room-local PROVIDER_ID AGENT_ID`
- (Accessible in **all rooms**) `!bai agent create-global PROVIDER AGENT_ID`
- Example: `!bai agent create-room-local openai my-openai-agent`
The `AGENT_ID` is a unique identifier for the agent. It can be any string which **doesn't contain spaces and `/`**.
Depending on where the agent is defined (within a room, globally, or [statically](./configuration/README.md#static-configuration)), this id will get a prefix (e.g. `room-local/`, `global/` or `static/`). The combined id (prefix + agent id) makes the **full agent identifier** (refered to as `FULL_AGENT_IDENTIFIER` in commands below).
When creating an agent, you will be given some sample [YAML](https://en.wikipedia.org/wiki/YAML) configuration which you can use to customize the agent's behavior.
This configuration varies depending on the [☁️ provider](./providers.md) used and the capabilities of the agent. Based on the configuration keys you pass, certain features will be enabled or disabled. For example, if you skip the `image_generation` key for an [OpenAI](./providers.md#openai) agent, it won't be able to generate images (see [🖌️ Image Generation](./features.md#-image-generation)).
After making your modifications to the sample YAML, you submit it back to the bot and the new agent will be created.
**To make use of the agent**, you need to [🤝 configure it as a handler for a given purpose](./configuration/handlers.md).
### Showing agent details
To **show** full details for a given agent: `!bai agent details FULL_AGENT_IDENTIFIER`
This command requires a full agent identifier (e.g. `room-local/agent-id`).
### Deleting agents
To **delete** an agent: `!bai agent delete FULL_AGENT_IDENTIFIER`
This command requires a full agent identifier (e.g. `room-local/agent-id`).
### Updating agents
To **update** a given agent's configuration: show the agent's [details](#showing-agent-details) (current configuration), then [delete](#deleting-agents) it and finally [re-create](#creating-agents) it.

View File

@@ -0,0 +1,48 @@
## 🛠️ Configuration
The bot's behavior is controlled by a combination of [static](#static-configuration) and [dynamic](#dynamic-configuration) configuration.
### Static configuration
The bot can be configured using a [YAML](https://en.wikipedia.org/wiki/YAML) configuration file as well as [environment variables](https://en.wikipedia.org/wiki/Environment_variable).
When running the bot locally (during [🧑‍💻 development](../development.md)), the bot's configuration is read from the `var/app/config.yml` file.
This file is created from the template found in [etc/app/config.yml.dist](../../etc/app/config.yml.dist).
Certain keys can be left unset, in which case [📝 hardcoded defaults](../../src/entity/cfg/defaults.rs) would be used.
Each configuration key found in the YAML configuration can be overridden by setting an environment variable (dots should be replaced with `_`). Example:
- to override `command_prefix`, set an environment variable `BAIBOT_COMMAND_PREFIX`
- to override `homeserver.server_name`, set an environment variable `BAIBOT_HOMESERVER_SERVER_NAME`
The static configuration contains an `initial_global_config` key, which is used to populate the bot's global configuration (stored as [dynamic configuration](#dynamic-configuration)) the first time the bot starts. Modifying this subsequently will not have any effect. After initial global configuration creation, it's expected to be managed dynamically via chat commands.
### Dynamic configuration
Besides the bot's [static configuration](#static-configuration), **the bot can also be configured dynamically at runtime (via chat messages)**.
This includes changes to [🔒 Access](../access.md), [🤖 Agents](../agents.md) and [🛠️ Room Settings](#room-settings).
#### Room Settings
Room Settings come from 3 different levels with priority in the following order (higher to lower):
- 📍 per-room (`!bai config room ..` commands)
- 🌐 globally (`!bai config global ..` commands)
- 📝 as [hardcoded defaults](../../src/entity/cfg/defaults.rs)
You can adjust the following settings per room and/or globally:
- [💬 Text Generation](text-generation.md)
- [🦻 Speech-to-Text](speech-to-text.md)
- [🗣️ Text-to-Speech](text-to-speech.md)
- [🖌️ Image Generation](image-generation.md)
- [🤝 Handlers](handlers.md)
Refer to the bot's help messages (as a response to a `!bai config` help command) for the most up-to-date information on what Room Settings can be configured.
You can **get an overview of the configuration affecting the current room** (a mix of hardcoded defaults, agent defaults, global and room-level settings) by sending a `!bai config status` command to the room.

View File

@@ -0,0 +1,32 @@
## 🤝 Handlers
### Introduction
You can use **different models in different rooms** (e.g. [OpenAI](../providers.md#openai) GPT-4o alongside [Llama](https://en.wikipedia.org/wiki/Llama_(language_model)) running on [Groq](../providers.md#groq), etc.)
You can also use **different models within the same room** (e.g. [💬 text-generation](#-text-generation) handled by one [agent](./agents.md), [🦻 speech-to-text](#-speech-to-text) handled by another, [🗣️ text-to-speech](#️-text-to-speech) by a 3rd, etc.)
The bot supports the following use-purposes:
- [💬 text-generation](../features.md#-text-generation): communicating with you via text
- [🦻 speech-to-text](../features.md#-speech-to-text): turning your voice messages into text
- [🗣️ text-to-speech](../features.md#️-text-to-speech): turning bot or users text messages into voice messages
- [🖌️ image-generation](../features.md#-image-generation): generating images based on instructions
In a given room, each different purpose can be served by a different [provider](../providers.md) and model. This combination of provider and model configuration is called an [🤖 agent](../agents.md). Each purpose can be served by a different **handler** agent.
See a [🖼️ Screenshot of an example room configuration](./screenshots/config-status-handlers.webp).
### Configuring
Handlers can be configured [dynamically](./README.md#dynamic-configuration):
- either per-room (e.g. `!bai config room set-handler text-generation room-local/openai-gpt-4o`)
- or globally (e.g. `!bai config global set-handler text-generation global/openai-gpt-4o`)
The per-room configuration takes priority over the global configuration.
There's also a `catch-all` purpose that can be used as a fallback handler for messages that don't match any other handler.
💡 It's a good idea to globally-configure a powerful agent as a catch-all handler, so that the bot can always handle messages of any kind. You can then override individual handlers per room or globally.

View File

@@ -0,0 +1,9 @@
## 🖌️ Image Generation
The Image Generation feature is not configurable at this moment.
You may also wish to see:
- [🌟 Features / 🖌️ Image Generation](../features.md#-image-generation) for a higher-level introduction to the Image Generation features
- [📖 Usage / 🖌️ Image Generation](../usage.md#-image-generation) section for more details on how to use the bot for Image Generation in a room

View File

@@ -0,0 +1,39 @@
## 🦻 Speech-to-Text
Below are some configuration settings related to Speech-to-Text.
You may also wish to see:
- [🌟 Features / 🦻 Speech-to-Text](../features.md#-speech-to-text) for a higher-level introduction to the Speech-to-Text features
- [📖 Usage / 🦻 Speech-to-Text](../usage.md#-speech-to-text) section for more details on how to use the bot for Speech-to-Text in a room
### 🪄 Flow Type
Controls how voice messages sent by [👥 user](../access.md#-users) are handled.
The following configuration values are recognized:
- (default) `transcribe_and_generate_text`: the bot will turn [👥 user](../access.md#-users) voice messages into text and then generate text messages via [💬 Text Generation](../features.md#-text-generation). This is the default setting to allow for [Seamless voice interaction](../features.md#seamless-voice-interaction).
- `ignore`: the bot will ignore all audio messages
- `only_transcribe`: the bot will turn [👥 user](../access.md#-users) voice messages into text, but will **not** proceed with [💬 Text Generation](../features.md#-text-generation). Switching to this may be useful in some cases, as in [Transcribe-only mode](../features.md#transcribe-only-mode).
Example: `!bai config room speech-to-text set-flow-type ignore` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
### 🔤 Language
Lets you specify the language of the input voice messages, to avoid using auto-detection.
Supplying the input language using a 2-letter code (e.g. `ja`) as per [ISO-639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes) may improve accuracy & latency.
![Speech-to-Text Language setting usage example](../screenshots/speech-to-text-language.webp)
In the above example screenshot, even without a language specified, the voice was understood correctly as [Bulgarian](https://en.wikipedia.org/wiki/Bulgarian_language), but was produced in latin, not [Cyrillic](https://en.wikipedia.org/wiki/Cyrillic_script), which is wrong.
If different [👥 user](../access.md#-users) are using different languages, do not specify a language.
💡 Certain models (like [OpenAI](../providers.md#openai)'s Whisper) may perform auto-translation if you specify a language, but you're speaking another one. You may abuse this side-effect for performing voice-to-text translation, but be aware that not all models behave this way.
Example (setting it to Japanese): `!bai config room speech-to-text set-language ja` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))

View File

@@ -0,0 +1,75 @@
## 💬 Text Generation
Below are some [🛠️ dynamic configuration settings](./README.md#dynamic-configuration) related to Text Generation.
You may also wish to see:
- [🌟 Features / 💬 Text Generation](../features.md#-text-generation) for a higher-level introduction to the Text Generation features
- [📖 Usage / 💬 Text Generation](../usage.md#-text-generation) section for more details on how to use the bot for Text Generation in a room
### 🗟 Prefix Requirement Type
In Direct Message rooms with the bot (1:1 rooms), it most usually makes sense for the bot to respond to **all** of your messages, as shown on this [🖼️ screenshot](../screenshots/text-generation.webp).
In group rooms (with multiple users), it may be more appropriate for the bot to only respond to messages that are **prefixed** with the command prefix (e.g. `!bai`), so that other chat exchange in the room will not trigger it. Such a setup is shown on this [🖼️ screenshot](../screenshots/text-generation-prefix-requirement.webp).
There are exceptions to these rules, and you can configure the bot to respond only to prefixed messages in a 1:1 room, or to respond to all messages even in a multi-user group room.
To support such use-cases, the bot has a `text-generation prefix-requirement-type` setting, which can be set to:
- (default) `no`: indicates that the bot would not require a prefix and would respond to all messages
- `command_prefix`: indicates that the bot would require that messages be prefixed with the command prefix (e.g. `!bai`) and would ignore all messages that are not prefixed
By default, the bot is **auto-configured (upon joining a new room)** to use the `no` setting in rooms that only include 2 users (you and the bot), and `command_prefix` in rooms with more than 2 users. To prevent surprises, the bot will **not** adjust this setting subsequently. You can manually adjust it via `!bai config room set-prefix-requirement-type VALUE`.
Example: `!bai config room text-generation set-prefix-requirement-type command_prefix` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
Regardless of this configuration, **the bot will also respond to messages which directly [mention](https://spec.matrix.org/latest/client-server-api/#user-and-room-mentions) the bot** (e.g. `@baibot`), even if they are not prefixed. An example of this can be seen on this [🖼️ screenshot](../screenshots/text-generation-prefix-requirement.webp).
### 🪄 Auto Usage
Text generation is enabled by default (the `text-generation auto-usage` setting being set to `always`), but can be set to:
- (default) `always`: generate text for all messages (also see [🗟 Prefix Requirement Type](#-prefix-requirement-type))
- `never`: never generate text for messages
- `only_for_voice`: only generate text when the original user message was a voice message, later transcribed via [🦻 Speech-to-Text](../features.md#-speech-to-text)
- `only_for_text`: only generate text when original user message was a text message
Example: `!bai config room text-generation set-auto-usage only_for_voice` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
### ♻️ Context Management
The bot also supports ♻️ **context management**, which automatically adjusts the message history length, etc.
This feature relies on [tokenization](https://en.wikipedia.org/wiki/Large_language_model#Tokenization) performed by the [tiktoken-rs](https://github.com/zurawiki/tiktoken-rs) library which is [poorly well-maintained](https://github.com/zurawiki/tiktoken-rs/issues/50) and only works well for [OpenAI](../providers.md#openai) models.
This setting is **disabled by default**, but can be enabled via `!bai config room context-management-enabled true` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings)).
### ⌨️ Prompt Override
You can override the [system prompt](https://huggingface.co/docs/transformers/en/tasks/prompting) configured at the [🤖 agent](../agents.md) level.
Example (multi-line is supported):
```
!bai config room text-generation set-prompt-override You're a UI/UX expert. Everything you say needs to consider design and usability.
Where appropriate, you'll mention best practices and common pitfalls.
```
A prompt override can also be set globally, see [🛠️ Room Settings](./README.md#room-settings).
### 🌡️ Temperature Override
You can override the [temperature](https://blogs.novita.ai/what-are-large-language-model-settings-temperature-top-p-and-max-tokens/#what-is-llm-temperature) (randomness / creativity) parameter configured at the [🤖 agent](../agents.md) level.
Example: `!bai config room text-generation set-temperature-override 3.5` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))

View File

@@ -0,0 +1,63 @@
## 🗣️ Text-to-Speech
Below are some configuration settings related to Text-to-Speech.
You may also wish to see:
- [🌟 Features / 🗣️ Text-to-Speech](../features.md#-text-generation) for a higher-level introduction to the Text-to-Speech features
- [📖 Usage / 🗣️ Text-to-Speech](../usage.md#-text-generation) section for more details on how to use the bot for Text-to-Speech in a room
### 🪄 Bot Messages Flow Type
Controls how automatic text-to-speech functions for **messages sent by the bot**.
The following configuration values are recognized:
- (default) `on_demand_for_voice`: the bot will turn its own text messages into audio (voice) messages only after an allowed [👥 user](../access.md#-users) **reacts** to a bot's message with 🗣️. To make it easier for users to react without having to hunt for this emoji, the bot will automatically add a 🗣️ reaction to its own messages which are in response to a user audio (voice) message.
- `on_demand_always`: the bot will turn its own text messages into audio (voice) messages only after an allowed [👥 user](../access.md#-users) **reacts** to a bot's message with 🗣️. To make it easier for users to react without having to hunt for this emoji, the bot will automatically add a 🗣️ reaction to **all of its own messages**.
- `only_for_voice`: the bot will turn its own text messages into audio (voice) messages only if the original user message was a voice message. This is to allow for [Seamless voice interaction](../features.md#seamless-voice-interaction), where you can speak to the bot and then hear its responses
- `never`: the bot will never turn its own text messags into audio (voice) messages
- `always`: the bot will turn all its text messages into audio (voice) messages. This also allows for [Seamless voice interaction](../features.md#seamless-voice-interaction).
Example: `!bai config room text-to-speech set-bot-msgs-flow-type never` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
### 🪄 User Messages Flow Type
Controls how automatic text-to-speech functions for **messages sent by [👥 users](../access.md#-users)**.
**Only works when automatic text-generation is disabled** (see [💬 Text Generation / 🪄 Auto Usage](./text-generation.md#-auto-usage)).
The following configuration values are recognized:
- (default) `never`: the bot will never turn [👥 user](../access.md#-users) text messages into audio (voice) messages
- `on_demand`: the bot will turn [👥 user](../access.md#-users) text messages into audio (voice) messages if the text message receives a 🗣️ reaction
- `always`: the bot will turn all [👥 user](../access.md#-users) text messages into audio (voice) messages. This is to allow for [Text-to-Speech-only mode](../features.md/#text-to-speech-only-mode).
Example: `!bai config room text-to-speech set-user-msgs-flow-type always` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
### 🗲 Speed override
The speed override setting lets you speed up/down speech relative to the default speed configured at the [🤖 agent](../agents.md) level (usually `1.0`).
Values typically range from `0.25` to `4.0`, but may vary depending on the selected model.
Example: `!bai config room text-to-speech set-speed-override 1.5` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))
### 👫 Voice override
The voice override setting lets you change the voice being used by the text-to-speech model configured at the [🤖 agent](../agents.md) level (usually `onyx` when using [OpenAI](../providers.md#openai)).
Possible values (e.g. `onyx`) depend on the model you're using. For example, for [OpenAI](../providers.md#openai)'s Whisper model, [these voices](https://platform.openai.com/docs/guides/text-to-speech/voice-options) are available.
Example: `!bai config room text-to-speech set-voice-override nova` (this can also be set globally, see [🛠️ Room Settings](./README.md#room-settings))

134
docs/development.md Normal file
View File

@@ -0,0 +1,134 @@
## 🧑‍💻 Development
This documentation page contains information about **running the bot locally for development purposes**.
This can also **helpful for quickly testing the bot in a containerized environment, with all dependency services included**.
For running the bot against your Matrix server, see the [🚀 Installation](./installation.md) documentation.
This bot is built in [🦀 Rust](https://www.rust-lang.org/) and uses the [mxlink](https://github.com/etkecc/rust-mxlink) library (built on top of [matrix-rust-sdk](https://github.com/matrix-org/matrix-rust-sdk)).
For local development, we run all dependency services in [🐋 Docker](https://www.docker.com/) containers via [docker-compose](https://docs.docker.com/compose/).
### Prerequisites
- [🐋 Docker](https://www.docker.com/) and [docker-compose](https://docs.docker.com/compose/)
- [Just](https://github.com/casey/just)
- (Optional) [🦀 Rust](https://www.rust-lang.org/) - for compiling and running outside of a container
- (Optional) an API key for some Large Language Model [☁️ provider](./providers.md) (e.g. [OpenAI](./providers.md#openai)), though we recommend using [LocalAI](#localai) or [Ollama](#ollama) for local development
### Getting started guide
Developing [locally](#running-locally) is possible, but requires a [Rust](https://www.rust-lang.org/) toolchain.
If this dependency is problematic for you, consider [🐋 running in a container](#running-in-a-container).
In any case, you will need [🐋 Docker](https://www.docker.com/) as [dependency services](../etc/services/) run there.
#### Running locally
1. Start the core dependency services (Postgres, Synapse, Element Web): `just services-start`
2. (Only the first time around) Prepare initial app configuration in `var/app/local/config.yml`: `just app-local-prepare`
3. (Only the first time around) [Prepare your configuration file](#prepare-your-configuration-file)
4. (Only the first time around) Prepare initial default Matrix user accounts (`admin` and `baibot`): `just users-prepare`
5. (Optional) Start additional services depending on which [agent provider you've chosen](#choosing-an-agent-provider):
- for [LocalAI](#localai):
- Start services: `just localai-start`
- Wait a while for LocalAI to start up. It has a lot of models to download. Monitor progress using `just localai-tail-logs`
- When ready, you'll be able to reach LocalAI's web interface at http://localai.127.0.0.1.nip.io:42027/ (not that you really need it)
- for [Ollama](#ollama):
- Start services: `just ollama-start`
- (Only the first time around) Pull the model configured in `agents.static_definitions` in the configuration file: `just ollama-pull-model gemma2:2b`
6. Start the bot: `just run-locally`
7. Go to http://element.127.0.0.1.nip.io:42025/ and login with `admin` / `admin`
8. Create a new room and invite `@baibot:synapse.127.0.0.1.nip.io`
9. When done, stop the bot (`Ctrl` + `C`)
10. Stop the core dependency services: `just services-stop`
11. (Optional) Stop additional services:
- for [LocalAI](#localai): `just localai-stop`
- for [Ollama](#ollama): `just ollama-stop`
#### Running in a container
You can avoid having a [Rust](https://www.rust-lang.org/) toolchain installed locally and build/run this in a container.
1. Start the core dependency services (Postgres, Synapse, Element Web): `just services-start`
2. (Only the first time around) Prepare initial app configuration in `var/app/container/config.yml`: `just app-container-prepare`
3. (Only the first time around) [Prepare your configuration file](#prepare-your-configuration-file)
4. (Only the first time around) Prepare initial default Matrix user accounts (`admin` and `baibot`): `just users-prepare`
5. (Optional) Start additional services depending on which [agent provider you've chosen](#choosing-an-agent-provider):
- for [LocalAI](#localai):
- Start services: `just localai-start`
- Wait a while for LocalAI to start up. It has a lot of models to download. Monitor progress using `just localai-tail-logs`
- When ready, you'll be able to reach LocalAI's web interface at http://localai.127.0.0.1.nip.io:42027/ (not that you really need it)
- for [Ollama](#ollama):
- Start services: `just ollama-start`
- (Only the first time around) Pull the model configured in `agents.static_definitions` in the configuration file: `just ollama-pull-model gemma2:2b`
6. Start the bot: `just run-in-container`
7. Go to http://element.127.0.0.1.nip.io:42025/ and login with `admin` / `admin`
8. Create a new room and invite `@baibot:synapse.127.0.0.1.nip.io`
9. When done, stop the bot (`Ctrl` + `C`)
10. Stop the dependency services: `just services-stop`
11. (Optional) Stop additional services:
- for [LocalAI](#localai): `just localai-stop`
- for [Ollama](#ollama): `just ollama-stop`
#### Prepare your configuration file
This is about editing your configuration. The initial configuration is created based on `etc/app/config.yml.dist` when you run `just app-local-prepare` or `just app-container-prepare`.
Depending on whether you run locally or in a container, your configuration lives in a different file (`var/app/local/config.yml` and `var/app/container/config.yml`, respectively).
Before starting the bot, you may wish to adjust this configuration.
##### Choosing an agent provider
You can create [🤖 agents](./agents.md) either [statically](./configuration/README.md#static-configuration) or [dynamically](./configuration/README.md#dynamic-configuration) using any of the supported [☁️ providers](./providers.md).
For getting started most quickly (and locally), we recommend using [LocalAI](#localai) or [Ollama](#ollama). These services are already configured to run as [local services via docker-compose](../etc/services/).
**Ollama is most lightweight** (~2GB for the container image + ~1.6GB for the model), but supports only [💬 text-generation](./features.md#-text-generation).
**LocalAI requires 4x more disk space** (~6GB for the container image + ~12GB for the models), but supports [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text) and [🖼️ image-generation](./features.md#️-image-generation).
**OpenAI supports all of these capabilities** as well and does not require powerful hardware or lots of disk space. However, it requires signup and an API key.
For local testing, **we recommend LocalAI**, because it runs fully locally and supports more features than Ollama.
###### LocalAI
[LocalAI](./providers.md#localai) supports all [🌟 features](./features.md) of the bot.
If you decided to go with [LocalAI](./providers.md#localai):
- enable the `localai` entry in the `agents.static_definitions` list in the configuration file
- adjust the `initial_global_config.handler.catch_all` setting in the configuration file (`null` -> `static/localai`)
By default, we configure LocalAI to use the [All-In-One images](https://localai.io/basics/container/#all-in-one-images) running on the CPU.
Performance is not great, but it should work reasonably well on good hardware.
If you'd like to use GPU acceleration, you may adjust the `SERVICE_LOCALAI_IMAGE_NAME` variable in [var/services/env](../var/services/env) (this file is automatically prepared for you based on [etc/services/env.dist](../etc/services/env.dist)) to use [other available LocalAI All-In-One images](https://localai.io/basics/container/#available-aio-images).
###### Ollama
[Ollama](./providers.md#ollama) only supports [💬 text-generation](./features.md#-text-generation).
If you decided to go with [Ollama](./providers.md#ollama):
- enable the `ollama` entry in the `agents.static_definitions` list in the configuration file
- adjust the `initial_global_config.handler.catch_all` setting in the configuration file (`null` -> `static/ollama`)
The [gemma2:2b](https://ollama.com/library/gemma2:2b) model was chosen as a default, because it's smallest/lightest and should run well under [Ollama](./providers.md#ollama) on most machines.
###### OpenAI
[OpenAI](./providers.md#openai) supports all [🌟 features](./features.md) of the bot.
If you decided to go with [OpenAI](./providers.md#openai):
- enable the `openai` entry in the `agents.static_definitions` list in the configuration file
- adjust the `initial_global_config.handler.catch_all` setting in the configuration file (`null` -> `static/openai`)

150
docs/features.md Normal file
View File

@@ -0,0 +1,150 @@
## 🌟 Features
### 🎨 Mixing & matching models
You can use **different models in different rooms** (e.g. [OpenAI](./providers.md#openai) GPT-4o alongside [Llama](https://en.wikipedia.org/wiki/Llama_(language_model)) running on [Groq](./providers.md#groq), etc.)
You can also use **different models within the same room** (e.g. [💬 text-generation](#-text-generation) handled by one [🤖 agent](./agents.md), [🦻 speech-to-text](#-speech-to-text) handled by another, [🗣️ text-to-speech](#️-text-to-speech) by a 3rd, etc.)
The bot supports the following use-purposes:
- [💬 text-generation](#-text-generation): communicating with you via text
- [🦻 speech-to-text](#-speech-to-text): turning your voice messages into text
- [🗣️ text-to-speech](#️-text-to-speech): turning bot or users text messages into voice messages
- [🖌️ image-generation](#-image-generation): generating images based on instructions
In a given room, each different purpose can be served by a different [☁️ provider](./providers.md) and model. This combination of provider and model configuration is called an [🤖 agent](./agents.md). Each purpose can be served by a different **handler** agent.
See a [🖼️ Screenshot of an example room configuration](./screenshots/config-status-handlers.webp).
For more information about configuring handlers, see the [🤝 Handlers / Configuring](./configuration/handlers.md#configuring) documentation section.
### 💬 Text Generation
Text Generation is the bot's ability to **respond to users' text messages with text**.
![Screenshot of Text Generation - a user sends a message and the bot replies in a new conversation thread](./screenshots/text-generation.webp)
In multi-user (group) rooms, to avoid disturbing the normal conversation between people, the bot is auto-configured to only respond to messages starting with the command prefix (`!bai`) or direct mentions via the [💬 Text Generation / 🗟 Prefix Requirement Type](./configuration/text-generation.md#-prefix-requirement-type) setting.
A few other features (like [🗣️ Text-to-Speech](#️-text-to-speech) and [🦻 Speech-to-Text](#-speech-to-text)) combine well with Text Generation, so you **don't necessarily need to communicate with the bot via text** (with [Seamless voice interaction](#seamless-voice-interaction), you can communicate only with voice).
You may also wish to see:
- [🛠️ Configuration / 💬 Text Generation](./configuration/README.md#-text-generation) for configuration options related to Text Generation
- [📖 Usage / 💬 Text Generation](./usage.md#-text-generation) section for more details on how to use the bot for Text Generation in a room
### 🗣️ Text-to-Speech
Text-to-Speech is the bot's ability to **turn text messages into voice messages**.
It can be performed **on the bot's own text messages** (responses to yours due to [💬 Text Generation](#-text-generation)) and/or **on your own text messages**.
Text-to-Speech can be enabled to be done automatically or on-demand (only after reacting to a message with 🗣️), and is configurable for different message types ([🪄 Bot Messages Flow Type](./configuration/README.md#-bot-messages-flow-type) vs [🪄 User Messages Flow Type](./configuration/README.md#-user-messages-flow-type)).
By default, the bot **doesn't** perform text-to-speech. It can be configured for [Seamless voice interaction](#seamless-voice-interaction), where you can **speak to the bot** (instead of typing) and then **hear its responses**.
Another use-case is to have the bot operate in [Text-to-Speech-only mode](#text-to-speech-only-mode).
- [🛠️ Configuration / 🗣️ Text-to-Speech](./configuration/README.md#-text-to-speech) for configuration options related to Text-to-Speech
- [📖 Usage / 🗣️ Text-to-Speech](./usage.md#-text-to-speech) section for more details on how to use the bot for Text-to-Speech in a room
#### Text-to-Speech-only mode
You may wish to have the bot **automatically turn your text messages into voice messages**, but **without** doing [💬 Text Generation](#-text-generation).
![Screenshot of Text-to-Speech-only mode - text messages are turned to audio and posted as a reply, without Text Generation happening](./screenshots/text-to-speech-only-mode.webp)
This could be useful in a room with others, where you'd like to post text messages and have people in the room consume them more easily (by listening to audio).
To allow for this use-case, you can:
- disable [💬 Text Generation](#-text-generation) (via [💬 Text Generation / 🪄 Auto Usage](./configuration/text-generation.md#-auto-usage) setting): `!bai config room text-generation set-auto-usage never`
- enable [🗣️ Text-to-Speech](#️-text-to-speech) for user messages (via [🗣️ Text-to-Speech / 🪄 User Messages Flow Type](./configuration/text-to-speech.md#-user-messages-flow-type)): `!bai config room text-to-speech set-user-msgs-flow-type always` (or `on_demand`)
### 🦻 Speech-to-Text
Speech-to-Text is the bot's ability to **turn voice messages into text**.
![Default flow for Speech-to-Text and Text-Generation - your voice messages are transcribed to text and then answered via Text Generation](./screenshots/speech-to-text-default-flow.webp)
The default flow is shown in the screenshot above: your voice messages are transcribed to text and [💬 Text Generation](#-text-generation) is performed. By default, the bot offers [🗣️ Text-to-Speech](#️-text-to-speech) for its answers via a 🗣️ emoji. You can click it to trigger text-to-speech on-demand.
You may also configure the bot for [Seamless voice interaction](#seamless-voice-interaction) or [Transcribe-only mode](#transcribe-only-mode), etc.
#### Seamless voice interaction
The bot can perform seamless voice interaction (🗣️-to-🗣️), allowing you to **speak to the bot** (instead of typing) and then **hear its responses**.
![Screenshot of the Seamless voice interaction mode - your voice messages are transcribed to text, then answered via Text Generation, and finally the answer is turned into a voice message](./screenshots/text-to-speech-seamless-voice-interaction.webp)
The flow is like this:
1. 👤 You sending a voice message
2. 🤖 The bot:
- (default) first turning your **voice message into text** ([🦻 Speech-to-Text](#-speech-to-text)) and posting it as a reply. This lets you you see what the bot heard.
- (default) then **answering in text** ([💬 Text Generation](#-text-generation)). This lets you read/skim text, if you so prefer.
- (can be enabled) finally **turning the answer's text into a voice message** ([🗣️ Text-to-Speech](#️-text-to-speech))
3. 👤 You continuing the conversation via text or voice messages
⚠️ Certain clients (like [Element](https://element.io/)) only support sending voice messages as top-level room messages, not as thread replies. Until this client limitation is fixed, Element users can only send the 1st message as a voice message - subsequent replies in the same conversation thread will need to be sent as text messages.
By default, the last part of the aforementioned flow is **not enabled**, because we assume **a saner default is to reply with text and merely *offer* text-to-speech to those who want it**. Offering is done by the bot reacting to its own message with 🗣️, and letting you click this emoji to trigger text-to-speech on-demand.
To enable automatic text-to-speech for the bot's messages, set the [🗣️ Text-to-Speech / 🪄 Bot Messages Flow Type](./configuration/text-to-speech.md#-bot-messages-flow-type) setting to `only_for_voice` or `always` (e.g. `!bai config room text-to-speech set-bot-msgs-flow-type only_for_voice`).
#### Transcribe-only mode
If you'd like to have the bot **only turn voice messages into text** (without generating text messages or voice messages), you can configure the bot for that.
![Screenshot of Transcribe-only-mode for Speech-to-Text - your voice messages are transcribed to text, and the bot does not generate text messages or voice messages](./screenshots/speech-to-text-transcribe-only-mode.webp)
To operate in this mode, you can:
- disable [💬 Text Generation](#-text-generation) (via [💬 Text Generation / 🪄 Auto Usage](./configuration/text-generation.md#-auto-usage) setting): `!bai config room text-generation set-auto-usage never`
- adjust the [🦻 Speech-to-Text / 🪄 Flow Type](./configuration/speech-to-text.md#-flow-type) setting to make the bot only transcribe (without doing [💬 Text Generation](#-text-generation)): `!bai config room speech-to-text set-flow-type only_transcribe`
### 🖌️ Image Generation
Image generation is the bot's ability to **generate images** based on text prompts.
See a [🖼️ Screenshot of the Image Generation feature](./screenshots/image-generation.webp).
You may also wish to see:
- [🛠️ Configuration / 🖌️ Image Generation](./configuration/README.md#-image-generation) for configuration options related to Image Generation
- [📖 Usage / 🖌️ Image Generation](./usage.md#-image-generation) section for more details on how to use the bot for Image Generation in a room
- [🫵 Sticker Generation](#-sticker-generation) - a special case of Image Generation
### 🫵 Sticker Generation
Sticker generation is the bot's ability to **generate sticker** images based on text prompts. It's a special case of [🖌️ Image Generation](#️-image-generation).
See a [🖼️ Screenshot of the Sticker Generation feature](./screenshots/sticker-generation.webp).
See [📖 Usage / 🖌️ Image Generation / Generating Stickers](./usage.md#generating-stickers) for details.
### 🔒 Encryption
#### Message exchange
The bot works in both **unencrypted and encrypted Matrix rooms**.
If configured, the bot can make use of **Matrix's Secure Storage (Recovery) feature**, so that it can restore its encryption keys even its local database gets lost.
#### Configuration
The bot also stores its [🛠️ configuration](./configuration/README.md) (both 📍 per-room and 🌐globally) in Matrix Account Data, which is **generally stored as plain-text in the server**.
To overcome this Matrix limitation, the bot can **optionally encrypt the configuration data** before storing it in Account Data. This allows for the bot to be used securely even against untrusted servers, without leaking sensitive configuration data to them.

93
docs/installation.md Normal file
View File

@@ -0,0 +1,93 @@
## 🚀 Installation
☁️ The easiest way to use the bot is to **get a managed Matrix server from [etke.cc](https://etke.cc/)** and order baibot via the [order form](https://etke.cc/order/). Existing customers can request the inclusion of this additional service by [contacting support](https://etke.cc/contacts/).
💻 If you're managing your Matrix server with the help of the [matrix-docker-ansible-deploy](https://github.com/spantaleev/matrix-docker-ansible-deploy) Ansible playbook, you can easily **install the bot via the Ansible playbook**. See the playbook's [Setting up baibot](https://github.com/spantaleev/matrix-docker-ansible-deploy/blob/master/docs/configuring-playbook-bot-baibot.md) documentation page.
🐋 In other cases, we **recommend using our [prebuilt container images](https://github.com/etkecc/baibot/pkgs/container/baibot) and [running in a container](#-running-in-a-container)**. You can also [build a container image](#building-a-container-image) yourself.
🔨 If containers are not your thing, you can [build a binary](#-building-a-binary) yourself and [run it](#-running-a-binary).
🗲 For a quick experiment, you can refer to the [🧑‍💻 development documentation](./development.md) which contains information on how to build and run the bot (and its various dependency services) locally.
### 🐋 Building a container image
We provide prebuilt container images for the `amd64` and `arm64` architectures, so **you don't necessarily need to build images yourself** and can jump to [Running in a container](#-running-in-a-container).
If you nevertheless wish to build a container image yourself, you can do so by running `just build-container-image`.
This will build and tag your container image as `localhost/baibot:latest`.
### 🐋 Running in a container
We recommend using a **tagged-release** (e.g. `v1.0.0`, not `latest`) of our [prebuilt container images](https://github.com/etkecc/baibot/pkgs/container/baibot), but you can also [build a container image](#-building-a-container-image) yourself.
You should:
- [🛠️ prepare a configuration file](#-preparing-a-configuration-file) (e.g. `cp etc/app/config.yml.dist /path/to/config.yml` & edit it)
- prepare a data directory (`mkdir /path/to/data`)
The example below uses [🐋 Docker](https://www.docker.com/) to run the container, but other container runtimes like [Podman](https://podman.io/) should work as well.
```sh
# Adjust the version tag to point to the latest available tagged version.
# If building your own container image name, adjust to something like `localhost/baibot:latest`.
CONTAINER_IMAGE_NAME=ghcr.io/etkecc/baibot:v1.0.0
/usr/bin/env docker run \
-it \
--rm \
--name=baibot \
--user=$(id -u):$(id -g) \
--cap-drop=ALL \
--read-only \
--env BAIBOT_PERSISTENCE_DATA_DIR_PATH=/data \
--mount type=bind,src=/path/to/config.yml,dst=/app/config.yml,ro \
--mount type=bind,src=/path/to/data,dst=/data \
$CONTAINER_IMAGE_NAME
```
💡 If you've defined the `persistence.data_dir_path` setting in the `config.yml` file, you can skip the `BAIBOT_PERSISTENCE_DATA_DIR_PATH` environment variable.
### 🔨 Building a binary
To build a binary, you need a [🦀 Rust](https://www.rust-lang.org/) toolchain.
Consult the [Dockerfile](../Dockerfile) file to learn what some of the build dependencies are (e.g. `libssl-dev`, `libsqlite3-dev`, etc., on Debian-based distros).
You can build a binary from the current project's source code:
- in `debug` mode via: `just build-debug`, yielding a binary in `target/debug/baibot`
- (recommended) in `release` mode via: `just build-release`, yielding a binary in `target/release/baibot`
💡 Unless you're [🧑‍💻 developing](./development.md), you probably wish to build in release mode, as that provides a much smaller and more optimized binary.
📦 You can also install from the [baibot](https://crates.io/crates/baibot) crate published to [crates.io](https://crates.io) with the help of the [cargo](https://doc.rust-lang.org/cargo/) package manager by running: `cargo install baibot`.
### 🖥️ Running a binary
Once you've [🔨 built a binary](#-building-a-binary) and [🛠️ prepared a configuration file](#-preparing-a-configuration-file), you can run it.
Consult the [Dockerfile](../Dockerfile) file to learn what some of the runtime dependencies are (e.g. `ca-certificates`, `sqlite3`, etc., on Debian-based distros).
You can run the binary like this:
```sh
BAIBOT_CONFIG_FILE_PATH=/path/to/config.yml \
BAIBOT_PERSISTENCE_DATA_DIR_PATH=/path/to/data \
./target/release/baibot
```
💡 If you've defined the `persistence.data_dir_path` setting in the `config.yml` file, you can skip the `BAIBOT_PERSISTENCE_DATA_DIR_PATH` environment variable.
💡 If your `config.yml` file is in your working directory (which may be different than the directory the binary lives in), you can skip the `BAIBOT_CONFIG_FILE_PATH` environment variable.
### 🛠️ Preparing a configuration file
For an introduction to the configuration file, see the [🛠️ Configuration](./configuration/README.md) page.
Generally, you need to copy the configuration file template ([etc/app/config.yml.dist](../etc/app/config.yml.dist)) and make modifications as needed.

170
docs/providers.md Normal file
View File

@@ -0,0 +1,170 @@
## ☁️ Providers
[🤖 Agents](./agents.md) are powered by a provider. The provider could be a **local service** or a **cloud service**.
The list of supported providers is below.
### Table of contents
- [How to choose a provider](#how-to-choose-a-provider)
- [How to use a provider](#how-to-use-a-provider)
- [Supported providers](#supported-providers)
- [Anthropic](#anthropic)
- [Groq](#groq)
- [LocalAI](#localai)
- [Mistral](#mistral)
- [Ollama](#ollama)
- [OpenAI](#openai)
- [OpenAI Compatible](#openai-compatible)
- [OpenRouter](#openrouter)
- [Together AI](#together-ai)
### How to choose a provider
If you're not sure which provider to start with, we **recommend [OpenAI](#openai)** as it's the most popular and has the **widest range of capabilities**: [💬 text-generation](./features.md#-text-generation), [🖌️ image-generation](./features.md#️-image-generation), [🦻 speech-to-text](./features.md#-speech-to-text), [🗣️ text-to-speech](./features.md#️-text-to-speech).
You don't need to choose just one though. The bot supports [mixing & matching models](./features.md#-mixing--matching-models), so you can use multiple providers at the same time.
### How to use a provider
- sign up for it
- obtain an API key
- [create a new agent](./agents.md#creating-agents)
- set it as a handler for some types of messages (see [Mixing & matching models](./features.md#-mixing--matching-models)) for a specific room or globally
### Supported providers
### Anthropic
[Anthropic](https://www.anthropic.com/) is an American AI company founded by former OpenAI engineers and providing powerful language models.
- 🆔 Identifier: `anthropic`
- 🔗 Links: [🏠 Home page](https://www.anthropic.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/Anthropic), [👤 Sign up](https://console.anthropic.com/), [📋 Models list](https://docs.anthropic.com/en/docs/about-claude/models)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local anthropic my-anthropic-agent`
- create a global agent: `!bai agent create-global anthropic my-anthropic-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/anthropic.yml).
### Groq
[Groq](https://groq.com/) is an American company developing optimized Language Processing Units (LPU) and offering cloud service which runs various models (built by others) with very high performance.
- 🆔 Identifier: `groq`
- 🔗 Links: [🏠 Home page](https://groq.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/Groq), [👤 Sign up](https://console.groq.com/login), [📋 Models list](https://console.groq.com/docs/models)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation), [🦻 speech-to-text](./features.md#-speech-to-text)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local groq my-groq-agent`
- create a global agent: `!bai agent create-global groq my-groq-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/groq.yml).
### LocalAI
[LocalAI](https://localai.io/) is the free, Open Source OpenAI alternative. LocalAI act as a drop-in replacement REST API that’s compatible with OpenAI API specifications for local inferencing. It allows you to run LLMs, generate images, audio (and not only) locally or on-prem with consumer grade hardware, supporting multiple model families and architectures.
- 🆔 Identifier: `localai`
- 🔗 Links: [🏠 Home page](https://localai.io/), [📋 Models list](https://localai.io/gallery.html)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local localai my-localai-agent`
- create a global agent: `!bai agent create-global localai my-localai-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/localai.yml).
### Mistral
[Mistral AI](https://mistral.ai/) is a research lab based in Europe (France) which produces their own language models.
- 🆔 Identifier: `mistral`
- 🔗 Links: [🏠 Home page](https://mistral.ai/), [🌐 Wiki](https://en.wikipedia.org/wiki/Mistral_AI), [👤 Sign up](https://auth.mistral.ai/ui/registration), [📋 Models list](https://docs.mistral.ai/getting-started/models/)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local mistral my-mistral-agent`
- create a global agent: `!bai agent create-global mistral my-mistral-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/mistral.yml).
### Ollama
[Ollama](https://ollama.com/) lets you run various models in a [self-hosted](https://github.com/ollama/ollama?tab=readme-ov-file#ollama) way. This is more advanced and requires powerful hardware for running some of the better models, but ensures your data stays with you.
- 🆔 Identifier: `ollama`
- 🔗 Links: [🏠 Home page](https://ollama.com/), [📋 Models list](https://ollama.com/library)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local ollama my-ollama-agent`
- create a global agent: `!bai agent create-global ollama my-ollama-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/ollama.yml).
### OpenAI
[OpenAI](https://openai.com/) is an American AI company providing powerful language models.
Use this provider either with the OpenAI API or with other OpenAI-compatible API services which **fully** adhere to the [OpenAI API spec](https://github.com/openai/openai-openapi/).
For services which are not fully compatible with the OpenAI API, consider using the [OpenAI Compatible](#openai-compatible) provider.
- 🆔 Identifier: `openai`
- 🔗 Links: [🏠 Home page](https://openai.com/), [🌐 Wiki](https://en.wikipedia.org/wiki/OpenAI), [👤 Sign up](https://platform.openai.com/signup), [📋 Models list](https://platform.openai.com/docs/models)
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-generation), [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local openai my-openai-agent`
- create a global agent: `!bai agent create-global openai my-openai-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/openai.yml).
### OpenAI Compatible
This provider allows you to use OpenAI-compatible API services like [OpenRouter](https://openrouter.ai/), [Together AI](https://www.together.ai/), etc.
Some of these popular services already have **shortcut** providers (leading to this one behind the scenes) - this make it easier to get started.
This provider is just as featureful as the [OpenAI](#openai) provider, but is more compatible with services which do not fully adhere to the [OpenAI API spec](https://github.com/openai/openai-openapi/).
- 🆔 Identifier: `openai-compatible`
- 🌟 Capabilities: [🖌️ image-generation](./features.md#️-image-generation), [💬 text-generation](./features.md#-text-generation), [🗣️ text-to-speech](./features.md#️-text-to-speech), [🦻 speech-to-text](./features.md#-speech-to-text)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local openai-compatible my-openai-compatible-agent`
- create a global agent: `!bai agent create-global openai-compatible my-openai-compatible-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/openai-compatible.yml).
### OpenRouter
[OpenRouter](https://openrouter.ai/) is a unified interface for LLMs. The platform scouts for the lowest prices and best latencies/throughputs across dozens of providers, and lets you choose how to [prioritize](https://openrouter.ai/docs/provider-routing) them.
- 🆔 Identifier: `openrouter`
- 🔗 Links: [🏠 Home page](https://openrouter.ai/), [👤 Sign up](https://openrouter.ai/), [📋 Models list](https://openrouter.ai/models)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local openrouter my-openrouter-agent`
- create a global agent: `!bai agent create-global openrouter my-openrouter-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/openrouter.yml).
### Together AI
[Together AI](https://www.together.ai/) makes it easy to run or [fine-tune](https://docs.together.ai/docs/fine-tuning-overview) leading open source models with only a few lines of code.
- 🆔 Identifier: `together-ai`
- 🔗 Links: [🏠 Home page](https://www.together.ai/), [👤 Sign up](https://api.together.ai/signup), [📋 Models list](https://api.together.xyz/models)
- 🌟 Capabilities: [💬 text-generation](./features.md#-text-generation)
- 🗲 Quick start:
- create a room-local agent: `!bai agent create-room-local together-ai my-together-ai-agent`
- create a global agent: `!bai agent create-global together-ai my-together-ai-agent`
💡 When creating an agent, the bot will show you an up-to-date sample configuration for this provider which looks [like this](../sample-provider-configs/together-ai.yml).

View File

@@ -0,0 +1,8 @@
base_url: https://api.anthropic.com/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: claude-3-5-sonnet-20240620
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 8192
max_context_tokens: 204800

View File

@@ -0,0 +1,10 @@
base_url: https://api.groq.com/openai/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: llama3-70b-8192
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 4096
max_context_tokens: 131072
speech_to_text:
model_id: whisper-large-v3

View File

@@ -0,0 +1,20 @@
base_url: http://my-localai-self-hosted-service:8080/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: gpt-4
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 4096
max_context_tokens: 128000
speech_to_text:
model_id: whisper-1
text_to_speech:
model_id: tts-1
voice: onyx
speed: 1.0
response_format: opus
image_generation:
model_id: stablediffusion
style: vivid
size: 1024x1024
quality: standard

View File

@@ -0,0 +1,8 @@
base_url: https://api.mistral.ai/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: mistral-large-latest
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 4096
max_context_tokens: 128000

View File

@@ -0,0 +1,8 @@
base_url: http://my-ollama-self-hosted-service:11434/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: gemma2:2b
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 4096
max_context_tokens: 128000

View File

@@ -0,0 +1,10 @@
base_url: ''
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: some-model
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 4096
max_context_tokens: 128000
speech_to_text:
model_id: whisper-1

View File

@@ -0,0 +1,20 @@
base_url: https://api.openai.com/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: gpt-4o-2024-08-06
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 16384
max_context_tokens: 128000
speech_to_text:
model_id: whisper-1
text_to_speech:
model_id: tts-1-hd
voice: onyx
speed: 1.0
response_format: opus
image_generation:
model_id: dall-e-3
style: vivid
size: 1024x1024
quality: standard

View File

@@ -0,0 +1,8 @@
base_url: https://openrouter.ai/api/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: mattshumer/reflection-70b:free
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 2048
max_context_tokens: 8192

View File

@@ -0,0 +1,8 @@
base_url: https://api.together.xyz/v1
api_key: YOUR_API_KEY_HERE
text_generation:
model_id: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo
prompt: You are a brief, but helpful bot.
temperature: 1.0
max_response_tokens: 2048
max_context_tokens: 8192

Binary file not shown.

After

Width:  |  Height:  |  Size: 89 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 65 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 684 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 52 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 27 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 37 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 13 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 130 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 158 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 11 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 16 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 23 KiB

99
docs/usage.md Normal file
View File

@@ -0,0 +1,99 @@
## 📖 Usage
This document covers how to use the bot in a room.
The [🌟 Features](./features.md) page also includes details about how each feature works and can be configured.
### 💬 Text Generation
This is related to the [💬 Text Generation](./features.md#-text-generation) feature.
If there's a text-generation handler agent configured, the bot **may** respond to messages sent in the room.
🖼️ See screenshots of:
- the [default Text Generation flow](./screenshots/text-generation.webp) for 1:1 rooms
- the [Text Generation flow in multi-user rooms](./screenshots/text-generation-prefix-requirement.webp) (where the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting is auto-configured to "required")
Whether the bot responds depends on:
- ([🔒 access](./access.md)) whether you're a whitelisted bot [👥 user](./access.md#-users)
- [🛠️ configuration](./configuration/README.md) whether there's a configured `text-generation` handler agent (or a `catch-all` handler agent). See [Mixing & matching models](./features.md#-mixing--matching-models)
- (🎨 agent capabilities) whether the configured `text-generation` (or `catch-all`) handler agent actually supports text-generation. The provider may lack support for this feature or it may be disabled in the [🤖 agents](./agents.md) configuration
- (the [🗟 Prefix Requirement](./configuration/text-generation.md#-prefix-requirement-type) setting) whether a prefix (e.g. `!bai`) is required in front of messages sent to the room. For multi-user rooms, this setting defaults to "required"
Room messages start a threaded conversation where you can continue back-and-forth communication with the bot.
Unless you've enabled the [♻️ Context Management](./features.md#️-context-management) feature, all messages will be sent to the agent's API each time. If the context management feature is enabled, older messages may be dropped.
### 🗣️ Text-to-Speech
This is related to the [🗣️ Text-to-Speech](./features.md#️-text-to-speech) feature.
If there's a text-to-speech handler agent configured, the bot **may** convert text messages sent to the room to audio (voice).
See:
- a [🖼️ screenshot](./screenshots/text-to-speech-only-mode.webp) of the bot's [Text-to-Speech-only](./features.md#text-to-speech-only-mode) mode
- a [🖼️ screenshot](./screenshots/text-to-speech-seamless-voice-interaction.webp) of the bot's [Seamless voice interaction](./features.md#seamless-voice-interaction) mode
By default, the bot:
- will offer tex-to-speech for its own messages which are a response to voice message from your, as part of the [Seamless voice interaction](./features.md#seamless-voice-interaction) feature. This can be adjusted via the [🗣️ Text-to-Speech / 🪄 Bot Messages Flow Type](./configuration/text-to-speech.md#-bot-messages-flow-type) setting.
- does not turn your own text messages to audio (voice). If you'd like for the bot to operate in such a mode, use the [🗣️ Text-to-Speech / 🪄 User Messages Flow Type](./configuration/text-to-speech.md#-user-messages-flow-type) setting (see [Text-to-Speech-only mode](./features.md#text-to-speech-only-mode)).
### 🦻 Speech-to-Text
This is related to the [🦻 Speech-to-Text](./features.md#-speech-to-text) feature.
If there's a speech-to-text handler agent configured, the bot **may** transcribe voice messages sent to the room to text.
See a [🖼️ Screenshot of the default flow for Speech-to-Text and Text-Generation](./screenshots/speech-to-text-default-flow.webp).
The speech-to-text feature triggers automatically by default, but can be adjusted via the [🦻 Speech-to-Text / 🪄 Flow Type](./features.md#-speech-to-text-flow-type) setting.
If all your messages are in the same language, you can improve accuracy & latency by configuring the language (see [🦻 Speech-to-Text / 🔤 Language](./configuration/speech-to-text.md#-language)).
### 🖌️ Image Generation
This is related to the [🖌️ Image Generation](./features.md#️-image-generation) feature.
This feature is not configurable at the moment. The configuration (size, quality, style) specified at the [🤖 agent](./agents.md) level will be used.
#### Generating images
Simply send a command like `!bai image A beautiful sunset over the ocean` and the bot will start a threaded conversation and post an image based on your prompt.
See a [🖼️ Screenshot of the Image Generation feature](./screenshots/image-generation.webp).
You can then, respond in the same message thread with:
- more messages, to add more criteria to your prompt.
- a message saying `again`, to generate one more image with the current prompt.
#### Generating stickers
A variation of [generating images](#generating-images) is to generate "sticker images".
See a [🖼️ Screenshot of the Sticker Generation feature](./screenshots/sticker-generation.webp).
To generate a sticker, send a command like `!bai sticker A huge ramen bowl with lots of chashu and a mountain of beansprouts on top`.
The difference from [generating images](#generating-images) is that the bot will:
- generate a smaller-resolution image (currently hardcoded to `256x256`) - smaller/quicker, but still good enough for a sticker
- potentially switch to a different (cheaper or otherwise more suitable) model, if available
- post the image directly to the room (as a reply to your message), without starting a threaded conversation
Some models (like [OpenAI](./providers.md#openai)'s [Dall-E-3](https://openai.com/index/dall-e-3/)) can only generate larger images (`1024x1024`, etc., for a higher charge), so we switching to a smaller/cheaper model (like [Dall-E-2](https://openai.com/index/dall-e-2/)) is a way to generate a sticker cheaply.

153
etc/app/config.yml.dist Normal file
View File

@@ -0,0 +1,153 @@
homeserver:
# The canonical homeserver domain name
server_name: synapse.127.0.0.1.nip.io
url: http://synapse.127.0.0.1.nip.io:42020
user:
mxid_localpart: baibot
password: baibot
# The name the bot uses as a display name and when it refers to itself.
# Leave empty to use the default (baibot).
name: baibot
encryption:
# An optional passphrase to use for backing up and recovering the bot's encryption keys.
# You can use any string here.
#
# If set to null, the recovery module will not be used and losing your session/database (see persistence)
# will mean you lose access to old messages in encrypted room.
#
# Changing this subsequently will also cause you to lose access to old messages in encrypted rooms.
# If you really need to change this:
# - Set `encryption_recovery_reset_allowed` to `true` and adjust the passphrase
# - Remove your session file and database (see persistence)
# - Restart the bot
# - Then restore `encryption_recovery_reset_allowed` to `false` to prevent accidental resets in the future
recovery_passphrase: long-and-secure-passphrase-here
# An optional flag to reset the encryption recovery passphrase.
recovery_reset_allowed: false
# Command prefix. Leave empty to use the default (!bai).
command_prefix: "!bai"
access:
# Space-separated list of MXID patterns which specify who is an admin.
admin_patterns:
- "@admin:synapse.127.0.0.1.nip.io"
persistence:
# This is unset here, because we expect the configuration to come from an environment variable (BAIBOT_PERSISTENCE_DATA_DIR_PATH).
# In your setup, you may wish to set this to a directory path.
data_dir_path: null
# An optional secret for encrypting the bot's session data (stored in data_dir_path).
# This must be 32-bytes (64 characters when HEX-encoded).
# Generate it with: `openssl rand -hex 32`
# Leave null or empty to avoid using encryption.
# Changing this subsequently requires that you also throw away all data stored in data_dir_path.
session_encryption_key: 9701cd109ed56770687dd8410f7d7371a4390dd3feb8ed721f189a0756c40098
# An optional secret for encrypting bot configuration stored in Matrix's account data.
# This must be 32-bytes (64 characters when HEX-encoded).
# Generate it with: `openssl rand -hex 32`
# Leave null or empty to avoid using encryption.
# Changing this subsequently will make you lose your configuration.
config_encryption_key: a9f1df98d288802ead20a8be2c701a627eabd31cf3d9e2aea28867ccd7a4ded7
agents:
# A list of statically-defined agents.
#
# Below are a few common choices on popular providers, preconfigured for development purposes (see docs/development.md).
# You may enable some of the ones you see below or define others.
# You can also leave this list empty and only define agents dynamically (via chat).
#
# Uncomment one or more of these and potentially adjust their configuration (API key, etc).
# Consider setting `initial_global_config.handler.*` to an agent that you enable here.
static_definitions:
# - id: openai
# provider: openai
# config:
# base_url: https://api.openai.com/v1
# api_key: ""
# text_generation:
# model_id: gpt-4o-2024-08-06
# prompt: You are a brief, but helpful bot.
# temperature: 1.0
# max_response_tokens: 16384
# max_context_tokens: 128000
# speech_to_text:
# model_id: whisper-1
# text_to_speech:
# model_id: tts-1-hd
# voice: onyx
# speed: 1.0
# response_format: opus
# image_generation:
# model_id: dall-e-3
# style: vivid
# size: 1024x1024
# quality: standard
#
# - id: localai
# provider: localai
# config:
# base_url: http://127.0.0.1:42027/v1
# api_key: null
# text_generation:
# model_id: gpt-4
# prompt: You are a brief, but helpful bot.
# temperature: 1.0
# max_response_tokens: 16384
# max_context_tokens: 128000
# speech_to_text:
# model_id: whisper-1
# text_to_speech:
# model_id: tts-1
# voice: onyx
# speed: 1.0
# response_format: opus
# image_generation:
# model_id: stablediffusion
# style: vivid
# # Intentionally defaults to a small value to improve performance
# size: 256x256
# quality: standard
#
# - id: ollama
# provider: ollama
# config:
# base_url: "http://127.0.0.1:42026/v1"
# api_key: null
# text_generation:
# model_id: "gemma2:2b"
# prompt: "You are an assistant based on the gemma2:2b model. Be brief in your responses."
# temperature: 1.0
# max_response_tokens: 4096
# max_context_tokens: 128000
# Initial global configuration. This only affects the first run of the bot.
# Configuration is later managed at runtime.
initial_global_config:
handler:
catch_all: null
text_generation: null
text_to_speech: null
speech_to_text: null
image_generation: null
# Space-separated list of MXID patterns which specify who can use the bot.
# By default, we let anyone on the homeserver use the bot.
user_patterns:
- "@*:synapse.127.0.0.1.nip.io"
# Controls logging.
#
# Sets all tracing targets (external crates) to warn, and our own logs to debug.
# For even more verbose logging, one may also use trace.
#
# matrix_sdk_crypto may be chatty and could be added with an error level.
#
# Learn more here: https://stackoverflow.com/a/73735203
logging: warn,mxlink=debug,baibot=debug

Binary file not shown.

After

Width:  |  Height:  |  Size: 77 KiB

351
etc/assets/baibot-torso.svg Normal file

File diff suppressed because one or more lines are too long

After

Width:  |  Height:  |  Size: 67 KiB

BIN
etc/assets/baibot.png Normal file

Binary file not shown.

After

Width:  |  Height:  |  Size: 87 KiB

568
etc/assets/baibot.svg Normal file

File diff suppressed because one or more lines are too long

After

Width:  |  Height:  |  Size: 118 KiB

BIN
etc/assets/baibot.xcf Normal file

Binary file not shown.

View File

@@ -0,0 +1,41 @@
services:
postgres:
image: docker.io/postgres:16.3-alpine
user: ${UID}:${GID}
restart: unless-stopped
environment:
POSTGRES_USER: synapse
POSTGRES_PASSWORD: synapse-password
POSTGRES_DB: homeserver
POSTGRES_INITDB_ARGS: --lc-collate C --lc-ctype C --encoding UTF8
volumes:
- ./postgres:/var/lib/postgresql/data
- /etc/passwd:/etc/passwd:ro
synapse:
image: ghcr.io/element-hq/synapse:v1.114.0
user: "${UID}:${GID}"
restart: unless-stopped
entrypoint: python
command: "-m synapse.app.homeserver -c /config/homeserver.yaml"
ports:
- "${SERVICE_SYNAPSE_BIND_PORT_CLIENT_API}:8008"
- "${SERVICE_SYNAPSE_BIND_PORT_FEDERATION_API}:8008"
volumes:
- ../../etc/services/core/synapse/config:/config:ro
- ./synapse/media-store:/media-store
element-web:
image: docker.io/vectorim/element-web:v1.11.77
user: "${UID}:${GID}"
restart: unless-stopped
ports:
- "${SERVICE_ELEMENT_WEB_BIND_PORT_HTTP}:8080"
volumes:
- ../../etc/services/core/element-web/nginx.conf:/etc/nginx/nginx.conf:ro
- ../../etc/services/core/element-web/config.json:/app/config.json:ro
networks:
default:
name: ${NETWORK_NAME}
external: true

View File

@@ -0,0 +1,13 @@
{
"default_hs_url": "http://synapse.127.0.0.1.nip.io:42020",
"default_is_url": "https://vector.im",
"integrations_ui_url": "https://scalar.vector.im/",
"integrations_rest_url": "https://scalar.vector.im/api",
"bug_report_endpoint_url": "https://riot.im/bugreports/submit",
"enableLabs": true,
"roomDirectory": {
"servers": [
"matrix.org"
]
}
}

View File

@@ -0,0 +1,60 @@
# This is a custom nginx configuration file that we use in the container (instead of the default one),
# because it allows us to run nginx with a non-root user.
#
# For this to work, the default vhost file (`/etc/nginx/conf.d/default.conf`) also needs to be removed.
# (mounting `/dev/null` over `/etc/nginx/conf.d/default.conf` works well)
#
# The following changes have been done compared to a default nginx configuration file:
# - default server port is changed (80 -> 8080), so that a non-root user can bind it
# - various temp paths are changed to `/tmp`, so that a non-root user can write to them
# - the `user` directive was removed, as we don't want nginx to switch users
worker_processes 1;
error_log /var/log/nginx/error.log warn;
pid /tmp/nginx.pid;
events {
worker_connections 1024;
}
http {
client_body_temp_path /tmp/client_body_temp;
proxy_temp_path /tmp/proxy_temp;
fastcgi_temp_path /tmp/fastcgi_temp;
uwsgi_temp_path /tmp/uwsgi_temp;
scgi_temp_path /tmp/scgi_temp;
include /etc/nginx/mime.types;
default_type application/octet-stream;
log_format main '$remote_addr - $remote_user [$time_local] "$request" '
'$status $body_bytes_sent "$http_referer" '
'"$http_user_agent" "$http_x_forwarded_for"';
access_log /var/log/nginx/access.log main;
sendfile on;
#tcp_nopush on;
keepalive_timeout 65;
#gzip on;
server {
listen 8080;
server_name localhost;
location / {
root /usr/share/nginx/html;
index index.html index.htm;
}
error_page 500 502 503 504 /50x.html;
location = /50x.html {
root /usr/share/nginx/html;
}
}
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,36 @@
version: 1
formatters:
precise:
format: '%(asctime)s - %(name)s - %(lineno)d - %(levelname)s - %(request)s - %(message)s'
filters:
context:
(): synapse.util.logcontext.LoggingContextFilter
request: ""
handlers:
console:
class: logging.StreamHandler
formatter: precise
filters: [context]
loggers:
synapse:
level: INFO
shared_secret_authenticator:
level: INFO
rest_auth_provider:
level: INFO
synapse.storage.SQL:
# beware: increasing this to DEBUG will make synapse log sensitive
# information such as access tokens.
level: INFO
root:
level: INFO
handlers: [console]

View File

@@ -0,0 +1 @@
ed25519 a_FEMe JGs8Fk83GHIrVyhBYa/VRUFbU4+Fxtf8iOsJ7CMamcM

12
etc/services/env.dist Normal file
View File

@@ -0,0 +1,12 @@
SERVICE_SYNAPSE_BIND_PORT_CLIENT_API=127.0.0.1:42020
SERVICE_SYNAPSE_BIND_PORT_FEDERATION_API=127.0.0.1:42028
SERVICE_ELEMENT_WEB_BIND_PORT_HTTP=127.0.0.1:42025
SERVICE_OLLAMA_BIND_PORT_HTTP=127.0.0.1:42026
# See https://localai.io/basics/container/#all-in-one-images for the list of available images
SERVICE_LOCALAI_IMAGE_NAME=docker.io/localai/localai:latest-aio-cpu
SERVICE_LOCALAI_BIND_PORT_HTTP=127.0.0.1:42027
# Variables below are added later on, dynamically

View File

@@ -0,0 +1,20 @@
services:
localai:
image: ${SERVICE_LOCALAI_IMAGE_NAME}
restart: unless-stopped
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8080/readyz"]
interval: 1m
timeout: 20m
retries: 5
ports:
- ${SERVICE_LOCALAI_BIND_PORT_HTTP}:8080
environment:
- DEBUG=true
volumes:
- ./localai/models:/build/models:cached
networks:
default:
name: ${NETWORK_NAME}
external: true

View File

@@ -0,0 +1,13 @@
services:
ollama:
image: docker.io/ollama/ollama:0.3.9
restart: unless-stopped
ports:
- "${SERVICE_OLLAMA_BIND_PORT_HTTP}:11434"
volumes:
- ./ollama:/root/.ollama
networks:
default:
name: ${NETWORK_NAME}
external: true

239
justfile Normal file
View File

@@ -0,0 +1,239 @@
project_name := "baibot"
container_image_name := "localhost/baibot"
project_container_network := "baibot"
# Show help by default
default:
@just --list --justfile {{ justfile() }}
# Builds and runs a development binary
run-locally *extra_args: app-local-prepare
RUST_BACKTRACE=1 \
BAIBOT_CONFIG_FILE_PATH={{ justfile_directory() }}/var/app/local/config.yml \
BAIBOT_PERSISTENCE_DATA_DIR_PATH={{ justfile_directory() }}/var/app/local/data \
cargo run -- {{ extra_args }}
run-in-container *extra_args: app-container-prepare build-container-image
/usr/bin/env docker run \
-it \
--rm \
--name={{ project_name }} \
--user=$(id -u):$(id -g) \
--cap-drop=ALL \
--read-only \
--network={{ project_container_network }} \
--env BAIBOT_PERSISTENCE_DATA_DIR_PATH=/data \
--mount type=bind,src={{ justfile_directory() }}/var/app/container/config.yml,dst=/app/config.yml,ro \
--mount type=bind,src={{ justfile_directory() }}/var/app/container/data,dst=/data \
{{ container_image_name }} {{ extra_args }}
# Runs tests
test *extra_args:
RUST_BACKTRACE=1 cargo test {{ extra_args }}
# Builds a debug binary (target/debug/*)
build-debug *extra_args:
RUST_BACKTRACE=1 cargo build {{ extra_args }}
# Builds an optimized release binary (target/release/*)
build-release *extra_args: (build-debug "--release")
# Builds a container image
build-container-image tag='latest':
/usr/bin/env docker build \
-t {{ container_image_name }}:{{ tag }} \
.
# Runs a docker-compose command
docker-compose services_type *extra_args:
/usr/bin/docker compose \
--project-directory var/services \
--env-file var/services/env \
-f etc/services/{{ services_type }}/compose.yml \
-p {{ project_name }}-{{ services_type }} \
{{ extra_args }}
# Runs a docker-compose command against the core services
docker-compose-core *extra_args:
just docker-compose core {{ extra_args }}
# Runs a docker-compose command against the localai services
docker-compose-localai *extra_args:
just docker-compose localai {{ extra_args }}
# Runs a docker-compose command against the ollama services
docker-compose-ollama *extra_args:
just docker-compose ollama {{ extra_args }}
# Runs all core dependency components (in the background)
services-start: services-prepare (docker-compose-core "up" "-d")
# Stops all core dependency components
services-stop: (docker-compose-core "down")
# Tails the logs for all running core services
services-tail-logs: (docker-compose-core "logs" "-f")
# Prepares the core services for running
services-prepare: _prepare-var-services-env _prepare-var-services-postgres _prepare-var-services-synapse _prepare-container-network
# Runs LocalAI (in the background)
localai-start: localai-prepare (docker-compose-localai "up" "-d")
# Stops LocalAI
localai-stop: (docker-compose-localai "down")
# Tails the logs for LocalAI
localai-tail-logs: (docker-compose-localai "logs" "-f")
# Prepares LocalAI for running
localai-prepare: _prepare-var-services-env _prepare-var-services-localai _prepare-container-network
# Runs Ollama (in the background)
ollama-start: ollama-prepare (docker-compose-ollama "up" "-d")
# Stops Ollama
ollama-stop: (docker-compose-ollama "down")
# Tails the logs for Ollama
ollama-tail-logs: (docker-compose-ollama "logs" "-f")
# Prepares Ollama for running
ollama-prepare: _prepare-var-services-env _prepare-var-services-ollama _prepare-container-network
# Pulls an Ollama model
ollama-pull-model model_id:
just -f {{ justfile_directory() }}/justfile docker-compose-ollama \
exec ollama \
ollama pull {{ model_id }}
# Prepares the app for running locally
app-local-prepare: _prepare-var-app-local-config_yml _prepare-var-app-local-data
# Prepares the app for running in a container
app-container-prepare: _prepare-var-app-container-config_yml _prepare-var-app-container-data
# Prepares the user accounts
users-prepare: services-prepare
just -f {{ justfile_directory() }}/justfile synapse-register-admin-user "admin" "admin"
just -f {{ justfile_directory() }}/justfile synapse-register-regular-user "baibot" "baibot"
# Starts a Postgres CLI (psql)
postgres-cli: services-prepare (docker-compose-core "exec" "postgres" "/bin/sh" "-c" "'PGUSER=synapse PGPASSWORD=synapse-password PGDATABASE=homeserver psql -h postgres'")
# Creates an administrator user
synapse-register-admin-user username password: services-prepare
just -f {{ justfile_directory() }}/justfile docker-compose-core \
exec synapse \
register_new_matrix_user \
--admin \
-u {{ username }} \
-p {{ password }} \
-c /config/homeserver.yaml \
http://localhost:8008
# Create a regular user
synapse-register-regular-user username password: services-prepare
just -f {{ justfile_directory() }}/justfile docker-compose-core \
exec synapse \
register_new_matrix_user \
--no-admin \
-u {{ username }} \
-p {{ password }} \
-c /config/homeserver.yaml \
http://localhost:8008
# Runs the clippy linter
clippy *extra_args:
cargo clippy {{ extra_args }}
_prepare-var-services-env:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/services/env ]; then
mkdir -p var/services
cp {{ justfile_directory() }}/etc/services/env.dist var/services/env
echo 'UID='`id -u` >> var/services/env;
echo 'GID='`id -g` >> var/services/env;
echo 'NETWORK_NAME={{ project_container_network }}' >> var/services/env;
fi
_prepare-var-services-postgres:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/services/postgres ]; then
mkdir -p var/services/postgres
chown `id -u`:`id -g` var/services/postgres
fi
_prepare-var-services-synapse:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/services/synapse ]; then
mkdir -p var/services/synapse/media-store
fi
_prepare-var-services-ollama:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/services/ollama ]; then
mkdir -p var/services/ollama
fi
_prepare-var-services-localai:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/services/localai ]; then
mkdir -p var/services/localai
fi
_prepare-container-network:
#!/bin/sh
network_definition=$(/usr/bin/env docker network ls --filter='name={{ project_container_network }}' --format=json)
if [ "$network_definition" = "" ]; then
/usr/bin/docker network create {{ project_container_network }}
fi
_prepare-var-app-local-config_yml:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/app/local/config.yml ]; then
mkdir -p var/app/local
cp {{ justfile_directory() }}/etc/app/config.yml.dist var/app/local/config.yml
fi
_prepare-var-app-local-data:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/app/local/data ]; then
mkdir -p var/app/local/data
fi
_prepare-var-app-container-config_yml:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/app/container/config.yml ]; then
mkdir -p var/app/container
cp {{ justfile_directory() }}/etc/app/config.yml.dist var/app/container/config.yml
sed --in-place 's/synapse.127.0.0.1.nip.io:42020/synapse:8008/g' var/app/container/config.yml
sed --in-place 's/127.0.0.1:42026/ollama:11434/g' var/app/container/config.yml
sed --in-place 's/127.0.0.1:42027/localai:8080/g' var/app/container/config.yml
fi
_prepare-var-app-container-data:
#!/bin/sh
cd {{ justfile_directory() }};
if [ ! -f var/app/container/data ]; then
mkdir -p var/app/container/data
fi

55
src/agent/definition.rs Normal file
View File

@@ -0,0 +1,55 @@
use serde::de::Error as DeError;
use serde::{Deserialize, Deserializer, Serialize, Serializer};
use super::provider::AgentProvider;
// Custom serialization for AgentProvider
pub fn serialize_provider_to_string<S>(
value: &AgentProvider,
serializer: S,
) -> Result<S::Ok, S::Error>
where
S: Serializer,
{
serializer.serialize_str(value.to_static_str())
}
// Custom deserialization for AgentProvider
pub fn deserialize_provider_from_string<'de, D>(deserializer: D) -> Result<AgentProvider, D::Error>
where
D: Deserializer<'de>,
{
let s = String::deserialize(deserializer)?;
AgentProvider::from_string(&s).map_err(DeError::custom)
}
#[derive(Debug, Clone, Deserialize, Serialize)]
pub struct AgentDefinition {
pub id: String,
#[serde(
serialize_with = "serialize_provider_to_string",
deserialize_with = "deserialize_provider_from_string"
)]
pub provider: AgentProvider,
pub config: serde_yaml::Value,
}
impl AgentDefinition {
pub fn new(id: String, provider: AgentProvider, config: serde_yaml::Value) -> Self {
Self {
id,
provider,
config,
}
}
}
impl PartialEq for AgentDefinition {
fn eq(&self, other: &Self) -> bool {
self.id == other.id
}
}
impl Eq for AgentDefinition {}

101
src/agent/identifier.rs Normal file
View File

@@ -0,0 +1,101 @@
use std::fmt;
#[derive(Debug, PartialEq, Eq, Clone)]
pub enum PublicIdentifier {
Static(String),
DynamicGlobal(String),
DynamicRoomLocal(String),
}
impl PublicIdentifier {
pub fn from_str(s: &str) -> Option<Self> {
if let Some(rest) = s.strip_prefix("static/") {
return Some(PublicIdentifier::Static(rest.to_string()));
} else if let Some(rest) = s.strip_prefix("global/") {
return Some(PublicIdentifier::DynamicGlobal(rest.to_string()));
} else if let Some(rest) = s.strip_prefix("room-local/") {
return Some(PublicIdentifier::DynamicRoomLocal(rest.to_string()));
}
None
}
pub fn as_string(&self) -> String {
match self {
PublicIdentifier::Static(s) => format!("static/{}", s),
PublicIdentifier::DynamicGlobal(s) => format!("global/{}", s),
PublicIdentifier::DynamicRoomLocal(s) => format!("room-local/{}", s),
}
}
pub fn prefixless(&self) -> String {
match self {
PublicIdentifier::Static(s) => s.to_owned(),
PublicIdentifier::DynamicGlobal(s) => s.to_owned(),
PublicIdentifier::DynamicRoomLocal(s) => s.to_owned(),
}
}
pub fn validate(&self) -> Result<(), String> {
let prefixless = self.prefixless();
if prefixless.is_empty() {
return Err("The agent ID must not be empty.".to_owned());
}
// We use a slash to separate the agent type from the agent ID.
if prefixless.contains("/") {
return Err("The agent ID must not contain the `/` character.".to_owned());
}
// Spaces are used for separating command arguments, etc.
if prefixless.contains(" ") {
return Err("The agent ID must not contain spaces.".to_owned());
}
Ok(())
}
}
impl fmt::Display for PublicIdentifier {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "{}", self.as_string())
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_public_identifier_from_str() {
assert_eq!(
PublicIdentifier::from_str("static/abc"),
Some(PublicIdentifier::Static("abc".to_string()))
);
assert_eq!(
PublicIdentifier::from_str("global/abc"),
Some(PublicIdentifier::DynamicGlobal("abc".to_string()))
);
assert_eq!(
PublicIdentifier::from_str("room-local/abc"),
Some(PublicIdentifier::DynamicRoomLocal("abc".to_string()))
);
assert_eq!(PublicIdentifier::from_str("abc"), None);
}
#[test]
fn test_public_identifier_as_string() {
assert_eq!(
PublicIdentifier::Static("abc".to_string()).as_string(),
"static/abc"
);
assert_eq!(
PublicIdentifier::DynamicGlobal("abc".to_string()).as_string(),
"global/abc"
);
assert_eq!(
PublicIdentifier::DynamicRoomLocal("abc".to_string()).as_string(),
"room-local/abc"
);
}
}

154
src/agent/instantiation.rs Normal file
View File

@@ -0,0 +1,154 @@
use super::{
provider::{self, ControllerType},
AgentDefinition, AgentProvider, PublicIdentifier,
};
// Dead-code is allowed. We do not use these enum struct payloads directly,
// but these errors are being print-formatted (`{:?}`) in error messages, so we wish to keep them.
#[derive(Debug)]
#[allow(dead_code)]
pub enum Error {
// Contains the error message from the validation function
ConfigFailsValidation(String),
// Contains the agent ID
ConfigForAgentIsNotAMapping(String),
// Contains the error from the constructor function
ConstructionFailed(anyhow::Error),
// Contains the error from the YAML deserialization function
Yaml(serde_yaml::Error),
}
pub type Result<T> = std::result::Result<T, Error>;
#[derive(Debug, Clone)]
pub struct AgentInstance {
identifier: PublicIdentifier,
definition: AgentDefinition,
controller: ControllerType,
}
impl AgentInstance {
pub fn new(
identifier: PublicIdentifier,
definition: AgentDefinition,
controller: ControllerType,
) -> Self {
Self {
identifier,
definition,
controller,
}
}
pub fn identifier(&self) -> &PublicIdentifier {
&self.identifier
}
pub fn definition(&self) -> &AgentDefinition {
&self.definition
}
pub fn controller(&self) -> &ControllerType {
&self.controller
}
}
pub(super) fn create(
identifier: PublicIdentifier,
definition: AgentDefinition,
) -> Result<AgentInstance> {
let controller = create_controller_from_provider_and_json_value_config(
&definition.id,
&definition.provider,
definition.config.clone(),
)?;
Ok(AgentInstance::new(identifier, definition, controller))
}
pub fn create_from_provider_and_yaml_value_config(
provider: &AgentProvider,
identifier: &PublicIdentifier,
config: serde_yaml::Value,
) -> Result<AgentInstance> {
let definition = AgentDefinition::new(identifier.prefixless(), provider.to_owned(), config);
create(identifier.to_owned(), definition)
}
fn create_controller_from_provider_and_json_value_config(
agent_id: &str,
provider: &AgentProvider,
config: serde_yaml::Value,
) -> Result<ControllerType> {
match provider {
AgentProvider::Anthropic => {
provider::anthropic::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::Groq => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::Mistral => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::LocalAI => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::Ollama => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::OpenAI => {
provider::openai::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::OpenAICompat => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::OpenRouter => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
AgentProvider::TogetherAI => {
provider::openai_compat::create_controller_from_yaml_value_config(agent_id, config)
}
}
}
pub fn default_config_for_provider(provider: &AgentProvider) -> serde_yaml::Value {
match provider {
AgentProvider::Anthropic => {
let config = super::provider::anthropic::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::Groq => {
let config = super::provider::groq::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::LocalAI => {
let config = super::provider::localai::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::Mistral => {
let config = super::provider::mistral::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::Ollama => {
let config = super::provider::ollama::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::OpenAI => {
let config = super::provider::openai::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::OpenAICompat => {
let config = super::provider::openai_compat::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::OpenRouter => {
let config = super::provider::openrouter::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
AgentProvider::TogetherAI => {
let config = super::provider::togetherai::default_config();
serde_yaml::to_value(config).expect("Failed to serialize config")
}
}
}

68
src/agent/manager.rs Normal file
View File

@@ -0,0 +1,68 @@
use super::instantiation;
use super::instantiation::AgentInstance;
use super::AgentDefinition;
use super::PublicIdentifier;
use crate::entity::RoomConfigContext;
#[derive(Debug)]
pub struct Manager {
static_agents: Vec<AgentInstance>,
}
impl Manager {
pub fn new(static_agent_definitions: Vec<AgentDefinition>) -> anyhow::Result<Self> {
let mut static_agents = Vec::with_capacity(static_agent_definitions.len());
for definition in static_agent_definitions {
let identifier = PublicIdentifier::Static(definition.id.clone());
match instantiation::create(identifier.clone(), definition.to_owned()) {
Ok(instance) => static_agents.push(instance),
Err(e) => {
return Err(anyhow::anyhow!(
"Failed to create static agent {}: {:?}",
identifier,
e
));
}
}
}
Ok(Self { static_agents })
}
pub fn available_room_agents_by_room_config_context(
&self,
room_config_context: &RoomConfigContext,
) -> Vec<AgentInstance> {
let mut agents: Vec<AgentInstance> = vec![];
for agent in &self.static_agents {
agents.push(agent.clone());
}
for definition in &room_config_context.global_config.agents {
let identifier = PublicIdentifier::DynamicGlobal(definition.id.clone());
match instantiation::create(identifier.clone(), definition.to_owned()) {
Ok(instance) => agents.push(instance),
Err(e) => {
tracing::warn!("Failed to create {} agent: {:?}. Skipping.", identifier, e);
}
}
}
for definition in &room_config_context.room_config.agents {
let identifier = PublicIdentifier::DynamicRoomLocal(definition.id.clone());
match instantiation::create(identifier.clone(), definition.to_owned()) {
Ok(instance) => agents.push(instance),
Err(e) => {
tracing::warn!("Failed to create {} agent: {:?}. Skipping.", identifier, e);
}
}
}
agents
}
}

21
src/agent/mod.rs Normal file
View File

@@ -0,0 +1,21 @@
mod definition;
mod identifier;
mod instantiation;
mod manager;
pub mod provider;
mod purpose;
pub mod utils;
pub use identifier::PublicIdentifier;
pub use manager::Manager;
pub use definition::AgentDefinition;
pub use instantiation::create_from_provider_and_yaml_value_config;
pub use instantiation::default_config_for_provider;
pub use instantiation::AgentInstance;
pub use instantiation::Error as AgentInstantiationError;
pub use instantiation::Result as AgentInstantiationResult;
pub use provider::{AgentProvider, AgentProviderInfo, ControllerTrait};
pub use purpose::AgentPurpose;

View File

@@ -0,0 +1,71 @@
use serde::{Deserialize, Serialize};
use anthropic_rs::models::claude::ClaudeModel;
use crate::agent::provider::ConfigTrait;
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Config {
pub base_url: String,
pub api_key: String,
pub text_generation: Option<TextGenerationConfig>,
}
impl Default for Config {
fn default() -> Self {
Self {
base_url: "https://api.anthropic.com/v1".to_owned(),
api_key: "YOUR_API_KEY_HERE".to_owned(),
text_generation: Some(TextGenerationConfig::default()),
}
}
}
impl ConfigTrait for Config {
fn validate(&self) -> Result<(), String> {
if self.base_url.is_empty() {
return Err("The base URL must not be empty.".to_owned());
}
if self.api_key.is_empty() {
return Err("The API key must not be empty.".to_owned());
}
Ok(())
}
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextGenerationConfig {
#[serde(default = "default_text_model_id")]
pub model_id: String,
#[serde(default)]
pub prompt: Option<String>,
#[serde(default = "super::super::default_temperature")]
pub temperature: f32,
#[serde(default)]
pub max_response_tokens: u32,
#[serde(default)]
pub max_context_tokens: u32,
}
impl Default for TextGenerationConfig {
fn default() -> Self {
Self {
model_id: default_text_model_id(),
prompt: Some("You are a brief, but helpful bot.".to_owned()),
temperature: super::super::default_temperature(),
max_response_tokens: 8192,
max_context_tokens: 204_800,
}
}
}
fn default_text_model_id() -> String {
ClaudeModel::Claude35Sonnet.as_str().to_owned()
}

View File

@@ -0,0 +1,251 @@
use std::fmt::Debug;
use std::str::FromStr;
use std::sync::Arc;
use anthropic_rs::completion::message::ContentType;
use anthropic_rs::{
client::Client as AnthropicClient, config::Config as AnthropicConfig,
models::claude::ClaudeModel,
};
use super::super::ControllerTrait;
use crate::agent::provider::entity::{
ImageGenerationResult, PingResult, TextGenerationParams, TextGenerationResult,
TextToSpeechParams, TextToSpeechResult,
};
use crate::agent::provider::{ImageGenerationParams, SpeechToTextParams, SpeechToTextResult};
use crate::agent::AgentPurpose;
use crate::conversation::llm::{
shorten_messages_list_to_context_size, Author as LLMAuthor, Conversation as LLMConversation,
Message as LLMMessage,
};
use crate::strings;
use super::config::Config;
struct ControllerInner {
client: AnthropicClient,
}
#[derive(Clone)]
pub struct Controller {
config: Config,
inner: Arc<ControllerInner>,
}
impl Debug for Controller {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.debug_struct("Controller")
.field("config", &self.config)
.finish()
}
}
impl Controller {
pub fn new(config: Config) -> anyhow::Result<Self> {
let anthropic_config =
AnthropicConfig::new(config.api_key.clone()).with_base_url(config.base_url.clone());
let client = match AnthropicClient::new(anthropic_config) {
Ok(client) => client,
Err(err) => {
return Err(anyhow::anyhow!(
"Failed to create Anthropic client: {}",
err.to_string()
));
}
};
Ok(Self {
config,
inner: Arc::new(ControllerInner { client }),
})
}
}
impl ControllerTrait for Controller {
async fn ping(&self) -> anyhow::Result<PingResult> {
if !self.supports_purpose(AgentPurpose::TextGeneration) {
return Ok(PingResult::Inconclusive);
}
let messages = vec![LLMMessage {
author: LLMAuthor::User,
message_text: "Hello!".to_string(),
}];
let conversation = LLMConversation { messages };
self.generate_text(conversation, TextGenerationParams::default())
.await?;
Ok(PingResult::Successful)
}
async fn generate_text(
&self,
conversation: LLMConversation,
params: TextGenerationParams,
) -> anyhow::Result<TextGenerationResult> {
let Some(text_generation_config) = &self.config.text_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextGeneration
),
));
};
let prompt_text = params
.prompt_override
.unwrap_or(self.text_generation_prompt().unwrap_or("".to_owned()))
.trim()
.to_owned();
let prompt_message = if prompt_text.is_empty() {
None
} else {
Some(LLMMessage {
author: LLMAuthor::Prompt,
message_text: prompt_text,
})
};
let mut conversation_messages = conversation.messages;
if params.context_management_enabled {
tracing::trace!("Shortening messages list to context size");
conversation_messages = shorten_messages_list_to_context_size(
&text_generation_config.model_id,
&prompt_message,
conversation_messages,
text_generation_config.max_response_tokens,
text_generation_config.max_context_tokens,
);
tracing::trace!("Finished shortening messages list to context size");
};
let messages_count = conversation_messages.len();
let mut request = super::utils::create_anthropic_message_request(conversation_messages);
let model = match ClaudeModel::from_str(&text_generation_config.model_id) {
Ok(model) => model,
Err(err) => {
tracing::debug!(?err, "Failed to parse model ID");
return Err(anyhow::anyhow!(
"Failed to parse model ID: {}",
&text_generation_config.model_id
));
}
};
let temperature = params
.temperature_override
.unwrap_or(text_generation_config.temperature);
if let Some(prompt_message) = prompt_message {
request.system = Some(prompt_message.message_text);
}
request.model = model;
request.temperature = Some(temperature);
request.max_tokens = text_generation_config.max_response_tokens;
if let Ok(request_as_json) = serde_json::to_string(&request) {
tracing::trace!(
model = format!("{:?}", request.model),
?messages_count,
request = request_as_json,
"Sending Anthropic create message API request"
);
}
let response = self.inner.client.create_message(request).await?;
tracing::trace!(?response, "Got response from Anthropic create message API");
// response.content usually contains a single element, but we support handling multiple to account for all possibilities
let mut text_parts = vec![];
for content in response.content {
let content_type = content.content_type;
match content_type {
ContentType::Text => {
text_parts.push(content.text);
} // There are no other content types to handle yet, but there may be in the future
}
}
if text_parts.is_empty() {
return Err(anyhow::anyhow!(
"No text content in response from the Anthropic create message API"
));
}
Ok(TextGenerationResult {
text: text_parts.join("\n\n"),
})
}
async fn speech_to_text(
&self,
_mime_type: &mxlink::mime::Mime,
_media: Vec<u8>,
_params: SpeechToTextParams,
) -> anyhow::Result<SpeechToTextResult> {
Err(anyhow::anyhow!("Speech-to-Text not supported"))
}
async fn generate_image(
&self,
_prompt: &str,
_params: ImageGenerationParams,
) -> anyhow::Result<ImageGenerationResult> {
Err(anyhow::anyhow!("Image generation not supported"))
}
async fn text_to_speech(
&self,
_input: &str,
_params: TextToSpeechParams,
) -> anyhow::Result<TextToSpeechResult> {
Err(anyhow::anyhow!("Speech generation not supported"))
}
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
match purpose {
AgentPurpose::TextGeneration => self.config.text_generation.is_some(),
AgentPurpose::SpeechToText => false,
AgentPurpose::TextToSpeech => false,
AgentPurpose::ImageGeneration => false,
AgentPurpose::CatchAll => true,
}
}
fn text_generation_prompt(&self) -> Option<String> {
let Some(text_generation_config) = &self.config.text_generation else {
return None;
};
text_generation_config.prompt.clone()
}
fn text_generation_temperature(&self) -> Option<f32> {
let Some(text_generation_config) = &self.config.text_generation else {
return None;
};
Some(text_generation_config.temperature)
}
fn text_to_speech_voice(&self) -> Option<String> {
None
}
fn text_to_speech_speed(&self) -> Option<f32> {
None
}
}

View File

@@ -0,0 +1,43 @@
mod config;
mod controller;
mod utils;
pub use config::Config;
pub use controller::Controller;
use super::super::AgentInstantiationError;
use super::super::AgentInstantiationResult;
use super::controller::ControllerType;
use super::ConfigTrait;
pub fn create_controller_from_yaml_value_config(
agent_id: &str,
config: serde_yaml::Value,
) -> AgentInstantiationResult<ControllerType> {
let config = match &config {
serde_yaml::Value::Mapping(_) => {
let config: Config =
serde_yaml::from_value(config).map_err(AgentInstantiationError::Yaml)?;
config
.validate()
.map_err(AgentInstantiationError::ConfigFailsValidation)?;
config
}
_ => {
return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping(
agent_id.to_owned(),
));
}
};
let controller =
Controller::new(config).map_err(AgentInstantiationError::ConstructionFailed)?;
Ok(ControllerType::Anthropic(Box::new(controller)))
}
pub fn default_config() -> Config {
Config::default()
}

View File

@@ -0,0 +1,32 @@
use anthropic_rs::completion::message::{Content, ContentType, Message, MessageRequest, Role};
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
pub(super) fn create_anthropic_message_request(llm_messages: Vec<LLMMessage>) -> MessageRequest {
let mut messages = vec![];
for message in llm_messages {
let role = match message.author {
LLMAuthor::User => Role::User,
LLMAuthor::Assistant => Role::Assistant,
LLMAuthor::Prompt => {
continue;
}
};
let content = vec![Content {
content_type: ContentType::Text,
text: message.message_text,
}];
let message = Message { role, content };
messages.push(message);
}
MessageRequest {
stream: false,
messages,
..Default::default()
}
}

View File

@@ -0,0 +1,3 @@
pub trait ConfigTrait {
fn validate(&self) -> Result<(), String>;
}

View File

@@ -0,0 +1,172 @@
use crate::{agent::AgentPurpose, conversation::llm::Conversation};
use super::{
entity::{
ImageGenerationResult, PingResult, TextGenerationParams, TextGenerationResult,
TextToSpeechParams, TextToSpeechResult,
},
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
};
pub trait ControllerTrait {
fn supports_purpose(&self, purpose: AgentPurpose) -> bool;
fn ping(&self) -> impl std::future::Future<Output = anyhow::Result<PingResult>> + Send;
fn text_generation_prompt(&self) -> Option<String>;
fn text_generation_temperature(&self) -> Option<f32>;
fn text_to_speech_voice(&self) -> Option<String>;
fn text_to_speech_speed(&self) -> Option<f32>;
fn generate_text(
&self,
conversation: Conversation,
params: TextGenerationParams,
) -> impl std::future::Future<Output = anyhow::Result<TextGenerationResult>> + Send;
fn speech_to_text(
&self,
mime_type: &mxlink::mime::Mime,
media: Vec<u8>,
params: SpeechToTextParams,
) -> impl std::future::Future<Output = anyhow::Result<SpeechToTextResult>> + Send;
fn generate_image(
&self,
prompt: &str,
params: ImageGenerationParams,
) -> impl std::future::Future<Output = anyhow::Result<ImageGenerationResult>> + Send;
fn text_to_speech(
&self,
text: &str,
params: TextToSpeechParams,
) -> impl std::future::Future<Output = anyhow::Result<TextToSpeechResult>> + Send;
}
#[derive(Debug, Clone)]
pub enum ControllerType {
OpenAI(Box<super::openai::Controller>),
OpenAICompat(Box<super::openai_compat::Controller>),
Anthropic(Box<super::anthropic::Controller>),
}
impl ControllerTrait for ControllerType {
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
match &self {
ControllerType::OpenAI(controller) => controller.supports_purpose(purpose),
ControllerType::OpenAICompat(controller) => controller.supports_purpose(purpose),
ControllerType::Anthropic(controller) => controller.supports_purpose(purpose),
}
}
fn text_generation_prompt(&self) -> Option<String> {
match &self {
ControllerType::OpenAI(controller) => controller.text_generation_prompt(),
ControllerType::OpenAICompat(controller) => controller.text_generation_prompt(),
ControllerType::Anthropic(controller) => controller.text_generation_prompt(),
}
}
fn text_to_speech_voice(&self) -> Option<String> {
match &self {
ControllerType::OpenAI(controller) => controller.text_to_speech_voice(),
ControllerType::OpenAICompat(controller) => controller.text_to_speech_voice(),
ControllerType::Anthropic(controller) => controller.text_to_speech_voice(),
}
}
fn text_to_speech_speed(&self) -> Option<f32> {
match &self {
ControllerType::OpenAI(controller) => controller.text_to_speech_speed(),
ControllerType::OpenAICompat(controller) => controller.text_to_speech_speed(),
ControllerType::Anthropic(controller) => controller.text_to_speech_speed(),
}
}
fn text_generation_temperature(&self) -> Option<f32> {
match &self {
ControllerType::OpenAI(controller) => controller.text_generation_temperature(),
ControllerType::OpenAICompat(controller) => controller.text_generation_temperature(),
ControllerType::Anthropic(controller) => controller.text_generation_temperature(),
}
}
async fn ping(&self) -> anyhow::Result<PingResult> {
match &self {
ControllerType::OpenAI(controller) => controller.ping().await,
ControllerType::OpenAICompat(controller) => controller.ping().await,
ControllerType::Anthropic(controller) => controller.ping().await,
}
}
async fn generate_text(
&self,
conversation: Conversation,
params: TextGenerationParams,
) -> anyhow::Result<TextGenerationResult> {
match &self {
ControllerType::OpenAI(controller) => {
controller.generate_text(conversation, params).await
}
ControllerType::OpenAICompat(controller) => {
controller.generate_text(conversation, params).await
}
ControllerType::Anthropic(controller) => {
controller.generate_text(conversation, params).await
}
}
}
async fn speech_to_text(
&self,
mime_type: &mxlink::mime::Mime,
media: Vec<u8>,
params: SpeechToTextParams,
) -> anyhow::Result<SpeechToTextResult> {
match &self {
ControllerType::OpenAI(controller) => {
controller.speech_to_text(mime_type, media, params).await
}
ControllerType::OpenAICompat(controller) => {
controller.speech_to_text(mime_type, media, params).await
}
ControllerType::Anthropic(controller) => {
controller.speech_to_text(mime_type, media, params).await
}
}
}
async fn generate_image(
&self,
prompt: &str,
params: ImageGenerationParams,
) -> anyhow::Result<ImageGenerationResult> {
match &self {
ControllerType::OpenAI(controller) => controller.generate_image(prompt, params).await,
ControllerType::OpenAICompat(controller) => {
controller.generate_image(prompt, params).await
}
ControllerType::Anthropic(controller) => {
controller.generate_image(prompt, params).await
}
}
}
async fn text_to_speech(
&self,
text: &str,
params: TextToSpeechParams,
) -> anyhow::Result<TextToSpeechResult> {
match &self {
ControllerType::OpenAI(controller) => controller.text_to_speech(text, params).await,
ControllerType::OpenAICompat(controller) => {
controller.text_to_speech(text, params).await
}
ControllerType::Anthropic(controller) => controller.text_to_speech(text, params).await,
}
}
}

View File

@@ -0,0 +1,198 @@
use crate::agent::AgentPurpose;
#[derive(Debug, Clone)]
pub enum AgentProvider {
Anthropic,
Groq,
LocalAI,
Mistral,
Ollama,
OpenAI,
OpenAICompat,
OpenRouter,
TogetherAI,
}
impl AgentProvider {
pub fn choices() -> Vec<&'static Self> {
vec![
&Self::Anthropic,
&Self::Groq,
&Self::LocalAI,
&Self::Mistral,
&Self::Ollama,
&Self::OpenAI,
&Self::OpenAICompat,
&Self::OpenRouter,
&Self::TogetherAI,
]
}
pub fn to_static_str(&self) -> &'static str {
match &self {
Self::Anthropic => "anthropic",
Self::Groq => "groq",
Self::LocalAI => "localai",
Self::Mistral => "mistral",
Self::Ollama => "ollama",
Self::OpenAI => "openai",
Self::OpenAICompat => "openai-compatible",
Self::OpenRouter => "openrouter",
Self::TogetherAI => "together-ai",
}
}
pub fn from_string(s: &str) -> Result<Self, &'static str> {
match s {
"anthropic" => Ok(Self::Anthropic),
"groq" => Ok(Self::Groq),
"localai" => Ok(Self::LocalAI),
"mistral" => Ok(Self::Mistral),
"ollama" => Ok(Self::Ollama),
"openai" => Ok(Self::OpenAI),
"openai-compatible" => Ok(Self::OpenAICompat),
"openrouter" => Ok(Self::OpenRouter),
"together-ai" => Ok(Self::TogetherAI),
_ => Err("Unexpected string value"),
}
}
pub fn info(&self) -> AgentProviderInfo {
match &self {
Self::Anthropic => AgentProviderInfo {
id: Self::Anthropic.to_static_str(),
name: "Anthropic",
description: "Anthropic is an American AI company founded by former OpenAI engineers and providing powerful language models.",
homepage_url: Some("https://www.anthropic.com/"),
wiki_url: Some("https://en.wikipedia.org/wiki/Anthropic"),
sign_up_url: Some("https://console.anthropic.com/"),
models_list_url: Some("https://docs.anthropic.com/en/docs/about-claude/models"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
],
},
Self::Groq => AgentProviderInfo {
id: Self::Groq.to_static_str(),
name: "Groq",
description: "Groq is an American company developing optimized Language Processing Units (LPU) and offering cloud service which runs various models (built by others) with very high performance.",
homepage_url: Some("https://groq.com/"),
wiki_url: Some("https://en.wikipedia.org/wiki/Groq"),
sign_up_url: Some("https://console.groq.com/login"),
models_list_url: Some("https://console.groq.com/docs/models"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
AgentPurpose::SpeechToText,
],
},
Self::LocalAI => AgentProviderInfo {
id: Self::LocalAI.to_static_str(),
name: "LocalAI",
description: "LocalAI is the free, Open Source OpenAI alternative. LocalAI act as a drop-in replacement REST API that’s compatible with OpenAI API specifications for local inferencing. It allows you to run LLMs, generate images, audio (and not only) locally or on-prem with consumer grade hardware, supporting multiple model families and architectures.",
homepage_url: Some("https://localai.io/"),
wiki_url: None,
sign_up_url: None,
models_list_url: Some("https://localai.io/gallery.html"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
AgentPurpose::TextToSpeech,
AgentPurpose::SpeechToText,
],
},
Self::Mistral => AgentProviderInfo {
id: Self::Mistral.to_static_str(),
name: "Mistral",
description: "Mistral AI is a research lab based in Europe (France) which produces their own language models.",
homepage_url: Some("https://mistral.ai/"),
wiki_url: Some("https://en.wikipedia.org/wiki/Mistral_AI"),
sign_up_url: Some("https://auth.mistral.ai/ui/registration"),
models_list_url: Some("https://docs.mistral.ai/getting-started/models/"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
],
},
Self::Ollama => AgentProviderInfo {
id: Self::Ollama.to_static_str(),
name: "Ollama",
description: "Ollama lets you run various models in a [self-hosted](https://github.com/ollama/ollama?tab=readme-ov-file#ollama) way. This is more advanced and requires powerful hardware for running some of the better models, but ensures your data stays with you.",
homepage_url: Some("https://ollama.com/"),
wiki_url: None,
sign_up_url: None,
models_list_url: Some("https://ollama.com/library"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
],
},
Self::OpenAI => AgentProviderInfo {
id: Self::OpenAI.to_static_str(),
name: "OpenAI",
description: "OpenAI is an American AI company providing powerful language models.\n\nUse this provider either with the OpenAI API or with other OpenAI-compatible API services which **fully** adhere to the [OpenAI API spec](https://github.com/openai/openai-openapi/).\nFor services which are not fully compatible with the OpenAI API, consider using the **OpenAI Compatible** provider.",
homepage_url: Some("https://openai.com/"),
wiki_url: Some("https://en.wikipedia.org/wiki/OpenAI"),
sign_up_url: Some("https://platform.openai.com/signup"),
models_list_url: Some("https://platform.openai.com/docs/models"),
supported_purposes: vec![
AgentPurpose::ImageGeneration,
AgentPurpose::TextGeneration,
AgentPurpose::TextToSpeech,
AgentPurpose::SpeechToText,
],
},
Self::OpenAICompat => AgentProviderInfo {
id: Self::OpenAICompat.to_static_str(),
name: "OpenAI Compatible",
description: "This provider allows you to use OpenAI-compatible API services like [OpenRouter](https://openrouter.ai/), [Together AI](https://www.together.ai/), etc.\n\nSome of these popular services already have **shortcut** providers (leading to this one behind the scenes) - this make it easier to get started.\n\nThis provider just as featureful as the **OpenAI** provider, but is more compatible with services which do not fully adhere to the [OpenAI API spec](https://github.com/openai/openai-openapi/).",
homepage_url: None,
wiki_url: None,
sign_up_url: None,
models_list_url: None,
supported_purposes: vec![
AgentPurpose::ImageGeneration,
AgentPurpose::TextGeneration,
AgentPurpose::TextToSpeech,
AgentPurpose::SpeechToText,
],
},
Self::OpenRouter => AgentProviderInfo {
id: Self::OpenRouter.to_static_str(),
name: "OpenRouter",
description: "OpenRouter is a unified interface for LLMs. The platform scouts for the lowest prices and best latencies/throughputs across dozens of providers, and lets you choose how to [prioritize](https://openrouter.ai/docs/provider-routing) them.",
homepage_url: Some("https://openrouter.ai/"),
wiki_url: None,
sign_up_url: Some("https://openrouter.ai/"),
models_list_url: Some("https://openrouter.ai/models"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
],
},
Self::TogetherAI => AgentProviderInfo {
id: Self::TogetherAI.to_static_str(),
name: "Together AI",
description: "Together AI makes it easy to run or [fine-tune](https://docs.together.ai/docs/fine-tuning-overview) leading open source models with only a few lines of code.",
homepage_url: Some("https://www.together.ai/"),
wiki_url: None,
sign_up_url: Some("https://api.together.ai/signup"),
models_list_url: Some("https://api.together.xyz/models"),
supported_purposes: vec![
AgentPurpose::TextGeneration,
],
},
}
}
}
impl std::fmt::Display for AgentProvider {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
write!(f, "{}", self.to_static_str())
}
}
pub struct AgentProviderInfo {
pub id: &'static str,
pub name: &'static str,
pub description: &'static str,
pub homepage_url: Option<&'static str>,
pub wiki_url: Option<&'static str>,
pub sign_up_url: Option<&'static str>,
pub models_list_url: Option<&'static str>,
pub supported_purposes: Vec<AgentPurpose>,
}

View File

@@ -0,0 +1,31 @@
#[derive(Default)]
pub struct ImageGenerationParams {
pub size_override: Option<String>,
pub cheaper_model_switching_allowed: bool,
pub cheaper_quality_switching_allowed: bool,
}
impl ImageGenerationParams {
pub fn with_size_override(mut self, value: Option<String>) -> Self {
self.size_override = value;
self
}
pub fn with_cheaper_model_switching_allowed(mut self, value: bool) -> Self {
self.cheaper_model_switching_allowed = value;
self
}
pub fn with_cheaper_quality_switching_allowed(mut self, value: bool) -> Self {
self.cheaper_quality_switching_allowed = value;
self
}
}
pub struct ImageGenerationResult {
pub bytes: Vec<u8>,
pub mime_type: mxlink::mime::Mime,
pub revised_prompt: Option<String>,
}

View File

@@ -0,0 +1,13 @@
mod agent_provider;
mod image_generation;
mod ping;
mod speech_to_text;
mod text_generation;
mod text_to_speech;
pub use agent_provider::{AgentProvider, AgentProviderInfo};
pub use image_generation::{ImageGenerationParams, ImageGenerationResult};
pub use ping::PingResult;
pub use speech_to_text::{SpeechToTextParams, SpeechToTextResult};
pub use text_generation::{TextGenerationParams, TextGenerationResult};
pub use text_to_speech::{TextToSpeechParams, TextToSpeechResult};

View File

@@ -0,0 +1,4 @@
pub enum PingResult {
Inconclusive,
Successful,
}

View File

@@ -0,0 +1,8 @@
#[derive(Default)]
pub struct SpeechToTextParams {
pub language_override: Option<String>,
}
pub struct SpeechToTextResult {
pub text: String,
}

View File

@@ -0,0 +1,10 @@
#[derive(Default)]
pub struct TextGenerationParams {
pub context_management_enabled: bool,
pub prompt_override: Option<String>,
pub temperature_override: Option<f32>,
}
pub struct TextGenerationResult {
pub text: String,
}

View File

@@ -0,0 +1,10 @@
#[derive(Default)]
pub struct TextToSpeechParams {
pub speed_override: Option<f32>,
pub voice_override: Option<String>,
}
pub struct TextToSpeechResult {
pub bytes: Vec<u8>,
pub mime_type: mxlink::mime::Mime,
}

View File

@@ -0,0 +1,26 @@
// Groq is based on openai_compat, because it's not fully compatible with async-openai.
use super::openai_compat::Config;
pub fn default_config() -> Config {
let mut config = Config {
base_url: "https://api.groq.com/openai/v1".to_owned(),
text_to_speech: None,
image_generation: None,
..Default::default()
};
if let Some(ref mut config) = config.text_generation.as_mut() {
config.model_id = "llama3-70b-8192".to_owned();
config.max_context_tokens = 131_072;
config.max_response_tokens = 4096;
}
if let Some(ref mut config) = config.speech_to_text.as_mut() {
config.model_id = "whisper-large-v3".to_owned();
}
config
}

View File

@@ -0,0 +1,33 @@
// LocalAI is based on OpenAI (async-openai), because it seems to be fully compatible.
// Moreover, openai_api_rust does not support speech-to-text, so if we wish to use this feature
// we need to stick to async-openai.
use super::openai_compat::Config;
pub fn default_config() -> Config {
let mut config = Config {
base_url: "http://my-localai-self-hosted-service:8080/v1".to_owned(),
..Default::default()
};
if let Some(ref mut config) = config.text_generation.as_mut() {
config.model_id = "gpt-4".to_owned();
config.max_context_tokens = 128_000;
config.max_response_tokens = 4096;
}
if let Some(ref mut config) = config.text_to_speech.as_mut() {
config.model_id = "tts-1".to_owned();
}
if let Some(ref mut config) = config.speech_to_text.as_mut() {
config.model_id = "whisper-1".to_owned();
}
if let Some(ref mut config) = config.image_generation.as_mut() {
config.model_id = "stablediffusion".to_owned();
}
config
}

View File

@@ -0,0 +1,22 @@
// Mistral is based on openai_compat, because it's not fully compatible with async-openai.
use super::openai_compat::Config;
pub fn default_config() -> Config {
let mut config = Config {
base_url: "https://api.mistral.ai/v1".to_owned(),
speech_to_text: None,
text_to_speech: None,
image_generation: None,
..Default::default()
};
if let Some(ref mut config) = config.text_generation.as_mut() {
config.model_id = "mistral-large-latest".to_owned();
config.max_context_tokens = 128_000;
}
config
}

25
src/agent/provider/mod.rs Normal file
View File

@@ -0,0 +1,25 @@
pub mod anthropic;
mod config;
mod controller;
mod entity;
pub(super) mod groq;
pub mod localai;
pub(super) mod mistral;
pub mod ollama;
pub mod openai;
pub mod openai_compat;
pub(super) mod openrouter;
pub(super) mod togetherai;
fn default_temperature() -> f32 {
1.0
}
pub use controller::{ControllerTrait, ControllerType};
pub use config::ConfigTrait;
pub use entity::{
AgentProvider, AgentProviderInfo, ImageGenerationParams, PingResult, SpeechToTextParams,
SpeechToTextResult, TextGenerationParams, TextToSpeechParams,
};

View File

@@ -0,0 +1,24 @@
// At the time of testing, Ollama can be powered by `openai`, but we use `openai_compat` for better reliability
// in the event of future updates to `async-openai`.
use super::openai_compat::Config;
pub fn default_config() -> Config {
let mut config = Config {
base_url: "http://my-ollama-self-hosted-service:11434/v1".to_owned(),
text_to_speech: None,
image_generation: None,
speech_to_text: None,
..Default::default()
};
if let Some(ref mut config) = config.text_generation.as_mut() {
config.model_id = "gemma2:2b".to_owned();
config.max_context_tokens = 128_000;
config.max_response_tokens = 4096;
}
config
}

View File

@@ -0,0 +1,190 @@
use serde::{Deserialize, Serialize};
use crate::agent::provider::ConfigTrait;
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Config {
pub base_url: String,
pub api_key: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub text_generation: Option<TextGenerationConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub speech_to_text: Option<SpeechToTextConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub text_to_speech: Option<TextToSpeechConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub image_generation: Option<ImageGenerationConfig>,
}
impl Default for Config {
fn default() -> Self {
Self {
base_url: "https://api.openai.com/v1".to_owned(),
api_key: "YOUR_API_KEY_HERE".to_owned(),
text_generation: Some(TextGenerationConfig::default()),
speech_to_text: Some(SpeechToTextConfig::default()),
text_to_speech: Some(TextToSpeechConfig::default()),
image_generation: Some(ImageGenerationConfig::default()),
}
}
}
impl ConfigTrait for Config {
fn validate(&self) -> Result<(), String> {
if self.base_url.is_empty() {
return Err("The base URL must not be empty.".to_owned());
}
Ok(())
}
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextGenerationConfig {
#[serde(default = "default_text_model_id")]
pub model_id: String,
#[serde(default)]
pub prompt: Option<String>,
#[serde(default = "super::super::default_temperature")]
pub temperature: f32,
#[serde(default)]
pub max_response_tokens: u32,
#[serde(default)]
pub max_context_tokens: u32,
}
impl Default for TextGenerationConfig {
fn default() -> Self {
Self {
model_id: default_text_model_id(),
prompt: Some("You are a brief, but helpful bot.".to_owned()),
temperature: super::super::default_temperature(),
max_response_tokens: 16_384,
max_context_tokens: 128_000,
}
}
}
fn default_text_model_id() -> String {
"gpt-4o-2024-08-06".to_owned()
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct SpeechToTextConfig {
#[serde(default = "default_speech_to_text_model_id")]
pub model_id: String,
}
impl Default for SpeechToTextConfig {
fn default() -> Self {
Self {
model_id: default_speech_to_text_model_id(),
}
}
}
fn default_speech_to_text_model_id() -> String {
"whisper-1".to_owned()
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextToSpeechConfig {
#[serde(default = "default_text_to_speech_model_id")]
pub model_id: async_openai::types::SpeechModel,
#[serde(default = "default_text_to_speech_voice")]
pub voice: async_openai::types::Voice,
#[serde(default = "default_text_to_speech_speed")]
pub speed: f32,
#[serde(default = "default_text_to_speech_response_format")]
pub response_format: async_openai::types::SpeechResponseFormat,
}
impl Default for TextToSpeechConfig {
fn default() -> Self {
Self {
model_id: default_text_to_speech_model_id(),
voice: default_text_to_speech_voice(),
speed: default_text_to_speech_speed(),
response_format: default_text_to_speech_response_format(),
}
}
}
fn default_text_to_speech_model_id() -> async_openai::types::SpeechModel {
async_openai::types::SpeechModel::Tts1Hd
}
fn default_text_to_speech_voice() -> async_openai::types::Voice {
async_openai::types::Voice::Onyx
}
fn default_text_to_speech_speed() -> f32 {
1.0
}
fn default_text_to_speech_response_format() -> async_openai::types::SpeechResponseFormat {
// The API defaults to mp3, but we prefer Opus because it's smaller.
// Our clients should all have support for it.
async_openai::types::SpeechResponseFormat::Opus
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ImageGenerationConfig {
pub model_id: String,
#[serde(default = "default_image_style")]
pub style: async_openai::types::ImageStyle,
#[serde(default = "default_image_size")]
pub size: async_openai::types::ImageSize,
#[serde(default = "default_image_quality")]
pub quality: async_openai::types::ImageQuality,
}
impl Default for ImageGenerationConfig {
fn default() -> Self {
Self {
model_id: "dall-e-3".to_owned(),
style: default_image_style(),
size: default_image_size(),
quality: default_image_quality(),
}
}
}
impl ImageGenerationConfig {
pub fn model_id_as_openai_image_model(
&self,
) -> Result<async_openai::types::ImageModel, String> {
match self.model_id.as_str() {
"dall-e-2" => Ok(async_openai::types::ImageModel::DallE2),
"dall-e-3" => Ok(async_openai::types::ImageModel::DallE3),
other => Ok(async_openai::types::ImageModel::Other(other.to_owned())),
}
}
}
fn default_image_style() -> async_openai::types::ImageStyle {
async_openai::types::ImageStyle::Vivid
}
fn default_image_size() -> async_openai::types::ImageSize {
async_openai::types::ImageSize::S1024x1024
}
fn default_image_quality() -> async_openai::types::ImageQuality {
async_openai::types::ImageQuality::Standard
}

View File

@@ -0,0 +1,465 @@
use std::ops::Deref;
use async_openai::{
config::OpenAIConfig,
types::{
ChatCompletionRequestMessage, CreateChatCompletionRequestArgs, CreateImageRequestArgs,
CreateSpeechRequestArgs, CreateTranscriptionRequestArgs,
},
Client as OpenAIClient,
};
use super::super::ControllerTrait;
use crate::{
agent::{
provider::{
entity::{ImageGenerationResult, PingResult, TextToSpeechParams, TextToSpeechResult},
openai::utils::convert_string_to_enum,
},
AgentPurpose,
},
strings,
};
use crate::{
agent::{
provider::{
entity::{TextGenerationParams, TextGenerationResult},
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
},
utils::base64_decode,
},
conversation::llm::{
shorten_messages_list_to_context_size, Author as LLMAuthor,
Conversation as LLMConversation, Message as LLMMessage,
},
};
use super::config::Config;
#[derive(Debug, Clone)]
pub struct Controller {
config: Config,
client: OpenAIClient<OpenAIConfig>,
}
impl Controller {
pub fn new(config: Config) -> Self {
let openai_config = OpenAIConfig::new()
.with_api_base(config.base_url.clone())
.with_api_key(config.api_key.clone());
let client = OpenAIClient::with_config(openai_config);
Self { config, client }
}
}
impl ControllerTrait for Controller {
async fn ping(&self) -> anyhow::Result<PingResult> {
if !self.supports_purpose(AgentPurpose::TextGeneration) {
return Ok(PingResult::Inconclusive);
}
let messages = vec![LLMMessage {
author: LLMAuthor::User,
message_text: "Hello!".to_string(),
}];
let conversation = LLMConversation { messages };
self.generate_text(conversation, TextGenerationParams::default())
.await?;
Ok(PingResult::Successful)
}
async fn generate_text(
&self,
conversation: LLMConversation,
params: TextGenerationParams,
) -> anyhow::Result<TextGenerationResult> {
let Some(text_generation_config) = &self.config.text_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextGeneration
),
));
};
let prompt_text = params
.prompt_override
.unwrap_or(self.text_generation_prompt().unwrap_or("".to_owned()))
.trim()
.to_owned();
let prompt_message = if prompt_text.is_empty() {
None
} else {
Some(LLMMessage {
author: LLMAuthor::Prompt,
message_text: prompt_text,
})
};
let mut conversation_messages = conversation.messages;
if params.context_management_enabled {
tracing::trace!("Shortening messages list to context size");
conversation_messages = shorten_messages_list_to_context_size(
&text_generation_config.model_id,
&prompt_message,
conversation_messages,
text_generation_config.max_response_tokens,
text_generation_config.max_context_tokens,
);
tracing::trace!("Finished shortening messages list to context size");
};
if let Some(prompt_message) = prompt_message {
conversation_messages.insert(0, prompt_message);
}
let openai_conversation_messages: Vec<ChatCompletionRequestMessage> =
super::utils::convert_llm_messages_to_openai_messages(conversation_messages);
let messages_count = openai_conversation_messages.len();
let temperature = params
.temperature_override
.unwrap_or(text_generation_config.temperature);
let request = CreateChatCompletionRequestArgs::default()
.max_tokens(text_generation_config.max_response_tokens)
.model(&text_generation_config.model_id)
.temperature(temperature)
.messages(openai_conversation_messages)
.build()?;
if let Ok(request_as_json) = serde_json::to_string(&request) {
tracing::trace!(
model = format!("{:?}", request.model),
?messages_count,
request = request_as_json,
"Sending OpenAI chat completion API request"
);
}
let response = self.client.chat().create(request).await?;
tracing::trace!(
?response,
"Got response from the OpenAI chat completion API"
);
// We only request 1 result, so there should only be 1 choice.
if let Some(choice) = response.choices.into_iter().next() {
match choice.message.content {
Some(text) => {
return Ok(TextGenerationResult { text });
}
None => {
return Err(anyhow::anyhow!(
"No content was found in the response choice from the OpenAI chat completion API"
));
}
}
}
Err(anyhow::anyhow!(
"No response messages choices were returned from the OpenAI chat completion API"
))
}
async fn speech_to_text(
&self,
mime_type: &mxlink::mime::Mime,
media: Vec<u8>,
params: SpeechToTextParams,
) -> anyhow::Result<SpeechToTextResult> {
let Some(speech_to_text_config) = &self.config.speech_to_text else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::SpeechToText
),
));
};
let filename = audio_mime_type_to_file_name(mime_type).unwrap_or("audio.ogg".to_string());
let language = params.language_override.unwrap_or("".to_string());
let request = CreateTranscriptionRequestArgs::default()
.model(&speech_to_text_config.model_id)
.file(async_openai::types::AudioInput {
source: async_openai::types::InputSource::VecU8 {
filename,
vec: media,
},
})
.language(language.clone())
.build()?;
tracing::trace!(
model_id = speech_to_text_config.model_id,
?language,
"Sending OpenAI speech-to-text API request"
);
let response = self.client.audio().transcribe(request).await?;
tracing::trace!(
?response,
"Got response from the OpenAI audio transcription API"
);
Ok(SpeechToTextResult {
text: response.text,
})
}
async fn generate_image(
&self,
prompt: &str,
params: ImageGenerationParams,
) -> anyhow::Result<ImageGenerationResult> {
let Some(image_generation_config) = &self.config.image_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::ImageGeneration
),
));
};
let original_model = image_generation_config
.model_id_as_openai_image_model()
.map_err(|err| anyhow::anyhow!(err))?;
let model = if params.cheaper_model_switching_allowed {
// Switch to a cheaper model
match original_model {
async_openai::types::ImageModel::DallE2 => async_openai::types::ImageModel::DallE2,
async_openai::types::ImageModel::DallE3 => async_openai::types::ImageModel::DallE2,
async_openai::types::ImageModel::Other(_) => {
async_openai::types::ImageModel::DallE2
}
}
} else {
original_model
};
let quality = if params.cheaper_quality_switching_allowed {
// Switch to a cheaper quality
match &image_generation_config.quality {
async_openai::types::ImageQuality::Standard => {
async_openai::types::ImageQuality::Standard
}
async_openai::types::ImageQuality::HD => {
async_openai::types::ImageQuality::Standard
}
}
} else {
image_generation_config.quality.clone()
};
let size = params
.size_override
.map(|s| {
convert_string_to_enum::<async_openai::types::ImageSize>(&s)
.unwrap_or(image_generation_config.size)
})
.unwrap_or(image_generation_config.size);
let request = CreateImageRequestArgs::default()
.model(model)
.prompt(prompt.to_owned())
.response_format(async_openai::types::ImageResponseFormat::B64Json)
.size(size)
.style(image_generation_config.style.clone())
.quality(quality)
.build()?;
tracing::trace!(
?prompt,
model = format!("{:?}", request.model),
size = format!("{:?}", request.size),
style = format!("{:?}", request.style),
quality = format!("{:?}", request.quality),
"Sending OpenAI image generation API request"
);
let response = self.client.images().create(request).await?;
if let Some(image) = response.data.into_iter().next() {
match image.deref() {
async_openai::types::Image::B64Json {
b64_json,
revised_prompt,
} => {
let bytes = base64_decode(b64_json)?;
return Ok(ImageGenerationResult {
bytes,
mime_type: mxlink::mime::IMAGE_PNG,
revised_prompt: revised_prompt.clone(),
});
}
_ => {
return Err(anyhow::anyhow!("Unexpected image type"));
}
}
}
Err(anyhow::anyhow!(
"The OpenAI image generation API returned no images"
))
}
async fn text_to_speech(
&self,
input: &str,
params: TextToSpeechParams,
) -> anyhow::Result<TextToSpeechResult> {
let Some(text_to_speech_config) = &self.config.text_to_speech else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextToSpeech
),
));
};
let speed = params.speed_override.unwrap_or(text_to_speech_config.speed);
let voice = if let Some(voice_string) = params.voice_override {
// This is a hacky way to construct a Voice enum from the string we have.
let voice: serde_json::Result<async_openai::types::Voice> =
serde_json::from_str(&format!("\"{}\"", voice_string));
match voice {
Ok(voice) => voice,
Err(err) => {
tracing::debug!(?voice_string, ?err, "Failed to parse voice");
return Err(anyhow::anyhow!(
"The configured voice ({}) is not supported.",
voice_string
));
}
}
} else {
text_to_speech_config.voice.clone()
};
let response_format = text_to_speech_config.response_format;
let mime_type = response_format_to_mime_type(&response_format).unwrap_or(
"audio/mp3"
.parse()
.expect("Failed parsing default mime type"),
);
let request = CreateSpeechRequestArgs::default()
.model(text_to_speech_config.model_id.clone())
.voice(voice)
.speed(speed)
.response_format(response_format)
.input(input)
.build()?;
tracing::trace!(
model = format!("{:?}", request.model),
voice = format!("{:?}", request.voice),
speed = format!("{:?}", request.speed),
"Sending OpenAI text-to-speech API request"
);
let result = self.client.audio().speech(request).await?;
Ok(TextToSpeechResult {
bytes: result.bytes.into(),
mime_type,
})
}
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
match purpose {
AgentPurpose::TextGeneration => self.config.text_generation.is_some(),
AgentPurpose::SpeechToText => self.config.speech_to_text.is_some(),
AgentPurpose::TextToSpeech => self.config.text_to_speech.is_some(),
AgentPurpose::ImageGeneration => self.config.image_generation.is_some(),
AgentPurpose::CatchAll => true,
}
}
fn text_generation_prompt(&self) -> Option<String> {
let Some(text_generation_config) = &self.config.text_generation else {
return None;
};
text_generation_config.prompt.clone()
}
fn text_generation_temperature(&self) -> Option<f32> {
let Some(text_generation_config) = &self.config.text_generation else {
return None;
};
Some(text_generation_config.temperature)
}
fn text_to_speech_voice(&self) -> Option<String> {
let Some(text_to_speech_config) = &self.config.text_to_speech else {
return None;
};
// A hacky way to turn this enum to a string
let voice_as_string = serde_json::to_string(&text_to_speech_config.voice).ok()?;
Some(voice_as_string.replace("\"", ""))
}
fn text_to_speech_speed(&self) -> Option<f32> {
let Some(text_to_speech_config) = &self.config.text_to_speech else {
return None;
};
Some(text_to_speech_config.speed)
}
}
fn response_format_to_mime_type(
response_format: &async_openai::types::SpeechResponseFormat,
) -> Option<mxlink::mime::Mime> {
let content_type = match response_format {
async_openai::types::SpeechResponseFormat::Mp3 => "audio/mp3".to_owned(),
async_openai::types::SpeechResponseFormat::Wav => "audio/wav".to_owned(),
async_openai::types::SpeechResponseFormat::Opus => "audio/ogg".to_owned(),
async_openai::types::SpeechResponseFormat::Aac => "audio/aac".to_owned(),
async_openai::types::SpeechResponseFormat::Flac => "audio/flac".to_owned(),
async_openai::types::SpeechResponseFormat::Pcm => "audio/L8".to_owned(),
};
match content_type.parse() {
Ok(content_type) => Some(content_type),
Err(err) => {
tracing::error!(?err, "Failed to parse content type");
None
}
}
}
fn audio_mime_type_to_file_name(mime_type: &mxlink::mime::Mime) -> Option<String> {
let mime_type_string = mime_type.to_string();
let file_extension = match mime_type_string.as_str() {
"audio/flac" => "flac",
"audio/x-m4a" | "audio/m4a" => "m4a",
"audio/mp3" | "audio/mpeg" => "mp3",
"audio/mp4" => "mp4",
"application/ogg" | "audio/ogg" => "ogg",
"audio/wav" | "audio/x-wav" => "wav",
"audio/webm" => "webm",
_ => return None,
};
Some(format!("audio.{}", file_extension))
}

View File

@@ -0,0 +1,46 @@
mod config;
mod controller;
mod utils;
pub use config::Config;
pub use controller::Controller;
// openai_compat needs these, so it can convert from its own config types to these
pub(super) use config::ImageGenerationConfig;
pub(super) use config::SpeechToTextConfig;
pub(super) use config::TextGenerationConfig;
pub(super) use config::TextToSpeechConfig;
use super::super::AgentInstantiationError;
use super::super::AgentInstantiationResult;
use super::controller::ControllerType;
use super::ConfigTrait;
pub fn create_controller_from_yaml_value_config(
agent_id: &str,
config: serde_yaml::Value,
) -> AgentInstantiationResult<ControllerType> {
let config = match &config {
serde_yaml::Value::Mapping(_) => {
let config: Config =
serde_yaml::from_value(config).map_err(AgentInstantiationError::Yaml)?;
config
.validate()
.map_err(AgentInstantiationError::ConfigFailsValidation)?;
config
}
_ => {
return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping(
agent_id.to_owned(),
));
}
};
Ok(ControllerType::OpenAI(Box::new(Controller::new(config))))
}
pub fn default_config() -> Config {
Config::default()
}

View File

@@ -0,0 +1,55 @@
use async_openai::types::{
ChatCompletionRequestAssistantMessageArgs, ChatCompletionRequestMessage,
ChatCompletionRequestSystemMessageArgs, ChatCompletionRequestUserMessageArgs,
};
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
pub fn convert_llm_messages_to_openai_messages(
conversation_messages: Vec<LLMMessage>,
) -> Vec<ChatCompletionRequestMessage> {
let mut openai_conversation_messages: Vec<ChatCompletionRequestMessage> =
Vec::with_capacity(conversation_messages.len());
for message in conversation_messages {
openai_conversation_messages.push(convert_llm_message_to_openai_message(message));
}
openai_conversation_messages
}
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> ChatCompletionRequestMessage {
match llm_message.author {
LLMAuthor::Prompt => ChatCompletionRequestSystemMessageArgs::default()
.content(llm_message.message_text)
.build()
.expect("Failed building OpenAI system message")
.into(),
LLMAuthor::Assistant => ChatCompletionRequestAssistantMessageArgs::default()
.content(llm_message.message_text)
.build()
.expect("Failed building OpenAI assistant message")
.into(),
LLMAuthor::User => ChatCompletionRequestUserMessageArgs::default()
.content(llm_message.message_text)
.build()
.expect("Failed building OpenAI user message")
.into(),
}
}
pub(super) fn convert_string_to_enum<T>(value: &str) -> Result<T, String>
where
T: serde::de::DeserializeOwned,
{
// This is a hacky way to construct an enum from the string we have.
let enum_result: serde_json::Result<T> = serde_json::from_str(&format!("\"{}\"", value));
match enum_result {
Ok(enum_result) => Ok(enum_result),
Err(err) => {
tracing::debug!(?err, "Failed to parse into enum");
Err(format!("The value ({}) is not supported.", value))
}
}
}

View File

@@ -0,0 +1,261 @@
use serde::{Deserialize, Serialize};
use crate::agent::provider::openai::{
ImageGenerationConfig as OpenAIImageGenerationConfig,
SpeechToTextConfig as OpenAISpeechToTextConfig,
TextGenerationConfig as OpenAITextGenerationConfig,
TextToSpeechConfig as OpenAITextToSpeechConfig,
};
use crate::agent::provider::ConfigTrait;
use super::utils::convert_string_to_enum;
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Config {
pub base_url: String,
pub api_key: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub text_generation: Option<TextGenerationConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub speech_to_text: Option<SpeechToTextConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub text_to_speech: Option<TextToSpeechConfig>,
#[serde(skip_serializing_if = "Option::is_none")]
pub image_generation: Option<ImageGenerationConfig>,
}
impl Default for Config {
fn default() -> Self {
Self {
base_url: "".to_owned(),
api_key: Some("YOUR_API_KEY_HERE".to_owned()),
text_generation: Some(TextGenerationConfig::default()),
speech_to_text: Some(SpeechToTextConfig::default()),
text_to_speech: Some(TextToSpeechConfig::default()),
image_generation: Some(ImageGenerationConfig::default()),
}
}
}
impl ConfigTrait for Config {
fn validate(&self) -> Result<(), String> {
if self.base_url.is_empty() {
return Err("The base URL must not be empty.".to_owned());
}
Ok(())
}
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextGenerationConfig {
#[serde(default = "default_text_model_id")]
pub model_id: String,
#[serde(default)]
pub prompt: Option<String>,
#[serde(default = "super::super::default_temperature")]
pub temperature: f32,
#[serde(default)]
pub max_response_tokens: u32,
#[serde(default)]
pub max_context_tokens: u32,
}
impl Default for TextGenerationConfig {
fn default() -> Self {
Self {
model_id: default_text_model_id(),
prompt: Some("You are a brief, but helpful bot.".to_owned()),
temperature: super::super::default_temperature(),
max_response_tokens: 4096,
max_context_tokens: 128_000,
}
}
}
impl TryInto<OpenAITextGenerationConfig> for TextGenerationConfig {
type Error = anyhow::Error;
fn try_into(self) -> Result<OpenAITextGenerationConfig, Self::Error> {
Ok(OpenAITextGenerationConfig {
model_id: self.model_id,
prompt: self.prompt,
temperature: self.temperature,
max_response_tokens: self.max_response_tokens,
max_context_tokens: self.max_context_tokens,
})
}
}
fn default_text_model_id() -> String {
"some-model".to_owned()
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct SpeechToTextConfig {
#[serde(default = "default_speech_to_text_model_id")]
pub model_id: String,
}
impl Default for SpeechToTextConfig {
fn default() -> Self {
Self {
model_id: default_speech_to_text_model_id(),
}
}
}
impl TryInto<OpenAISpeechToTextConfig> for SpeechToTextConfig {
type Error = anyhow::Error;
fn try_into(self) -> Result<OpenAISpeechToTextConfig, Self::Error> {
Ok(OpenAISpeechToTextConfig {
model_id: self.model_id,
})
}
}
fn default_speech_to_text_model_id() -> String {
"whisper-1".to_owned()
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct TextToSpeechConfig {
#[serde(default = "default_text_to_speech_model_id")]
pub model_id: String,
#[serde(default = "default_text_to_speech_voice")]
pub voice: String,
#[serde(default = "default_text_to_speech_speed")]
pub speed: f32,
#[serde(default = "default_text_to_speech_response_format")]
pub response_format: String,
}
impl Default for TextToSpeechConfig {
fn default() -> Self {
Self {
model_id: default_text_to_speech_model_id(),
voice: default_text_to_speech_voice(),
speed: default_text_to_speech_speed(),
response_format: default_text_to_speech_response_format(),
}
}
}
impl TryInto<OpenAITextToSpeechConfig> for TextToSpeechConfig {
type Error = String;
fn try_into(self) -> Result<OpenAITextToSpeechConfig, Self::Error> {
let model_id = convert_string_to_enum::<async_openai::types::SpeechModel>(&self.model_id)?;
let voice = convert_string_to_enum::<async_openai::types::Voice>(&self.voice)?;
let response_format = convert_string_to_enum::<async_openai::types::SpeechResponseFormat>(
&self.response_format,
)?;
Ok(OpenAITextToSpeechConfig {
model_id,
voice,
speed: self.speed,
response_format,
})
}
}
fn default_text_to_speech_model_id() -> String {
"tts-1".to_owned()
}
fn default_text_to_speech_voice() -> String {
"onyx".to_owned()
}
fn default_text_to_speech_speed() -> f32 {
1.0
}
fn default_text_to_speech_response_format() -> String {
"opus".to_owned()
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ImageGenerationConfig {
pub model_id: String,
#[serde(default = "default_image_style")]
pub style: Option<String>,
#[serde(default = "default_image_size")]
pub size: Option<String>,
#[serde(default = "default_image_quality")]
pub quality: Option<String>,
}
impl Default for ImageGenerationConfig {
fn default() -> Self {
Self {
model_id: "stablediffusion".to_owned(),
style: default_image_style(),
size: default_image_size(),
quality: default_image_quality(),
}
}
}
impl TryInto<OpenAIImageGenerationConfig> for ImageGenerationConfig {
type Error = String;
fn try_into(self) -> Result<OpenAIImageGenerationConfig, Self::Error> {
let size = if let Some(size) = &self.size {
convert_string_to_enum::<async_openai::types::ImageSize>(size)?
} else {
async_openai::types::ImageSize::S1024x1024
};
let style = if let Some(style) = &self.style {
convert_string_to_enum::<async_openai::types::ImageStyle>(style)?
} else {
async_openai::types::ImageStyle::Vivid
};
let quality = if let Some(quality) = &self.quality {
convert_string_to_enum::<async_openai::types::ImageQuality>(quality)?
} else {
async_openai::types::ImageQuality::Standard
};
Ok(OpenAIImageGenerationConfig {
model_id: self.model_id,
style,
size,
quality,
})
}
}
fn default_image_style() -> Option<String> {
Some("vivid".to_owned())
}
fn default_image_size() -> Option<String> {
Some("1024x1024".to_owned())
}
fn default_image_quality() -> Option<String> {
Some("standard".to_owned())
}

View File

@@ -0,0 +1,445 @@
use openai_api_rust::audio::{AudioApi, AudioBody};
use openai_api_rust::chat::{ChatApi, ChatBody};
use openai_api_rust::images::{ImagesApi, ImagesBody};
use openai_api_rust::{Auth, Message, OpenAI};
use super::super::ControllerTrait;
use crate::agent::utils::base64_decode;
use crate::{
agent::provider::{
entity::{TextGenerationParams, TextGenerationResult},
ImageGenerationParams, SpeechToTextParams, SpeechToTextResult,
},
conversation::llm::{
shorten_messages_list_to_context_size, Author as LLMAuthor,
Conversation as LLMConversation, Message as LLMMessage,
},
};
use crate::{
agent::{
provider::entity::{
ImageGenerationResult, PingResult, TextToSpeechParams, TextToSpeechResult,
},
AgentPurpose,
},
strings,
};
use super::Config;
#[derive(Debug, Clone)]
pub struct Controller {
config: Config,
client: OpenAI,
}
impl Controller {
pub fn new(config: Config) -> Self {
let api_key = config.api_key.clone().unwrap_or("".to_owned());
let auth = Auth::new(&api_key);
// The library we use chokes if there's no trailing slash
let base_url = if config.base_url.ends_with("/") {
config.base_url.clone()
} else {
format!("{}/", config.base_url)
};
let client = OpenAI::new(auth, &base_url);
Self { config, client }
}
}
impl ControllerTrait for Controller {
async fn ping(&self) -> anyhow::Result<PingResult> {
if !self.supports_purpose(AgentPurpose::TextGeneration) {
return Ok(PingResult::Inconclusive);
}
let messages = vec![LLMMessage {
author: LLMAuthor::User,
message_text: "Hello!".to_string(),
}];
let conversation = LLMConversation { messages };
self.generate_text(conversation, TextGenerationParams::default())
.await?;
Ok(PingResult::Successful)
}
async fn generate_text(
&self,
conversation: LLMConversation,
params: TextGenerationParams,
) -> anyhow::Result<TextGenerationResult> {
let Some(text_generation_config) = &self.config.text_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextGeneration
),
));
};
let prompt_text = params
.prompt_override
.unwrap_or(self.text_generation_prompt().unwrap_or("".to_owned()))
.trim()
.to_owned();
let prompt_message = if prompt_text.is_empty() {
None
} else {
Some(LLMMessage {
author: LLMAuthor::Prompt,
message_text: prompt_text,
})
};
let mut conversation_messages = conversation.messages;
if params.context_management_enabled {
tracing::trace!("Shortening messages list to context size");
conversation_messages = shorten_messages_list_to_context_size(
&text_generation_config.model_id,
&prompt_message,
conversation_messages,
text_generation_config.max_response_tokens,
text_generation_config.max_context_tokens,
);
tracing::trace!("Finished shortening messages list to context size");
};
if let Some(prompt_message) = prompt_message {
conversation_messages.insert(0, prompt_message);
}
let openai_conversation_messages: Vec<Message> =
super::utils::convert_llm_messages_to_openai_messages(conversation_messages);
let messages_count = openai_conversation_messages.len();
let temperature = params
.temperature_override
.unwrap_or(text_generation_config.temperature);
let max_tokens = text_generation_config
.max_response_tokens
.try_into()
.expect("Failed converting max_response_tokens from u32 to i32");
let request = ChatBody {
model: text_generation_config.model_id.clone(),
max_tokens: Some(max_tokens),
temperature: Some(temperature),
top_p: None,
n: Some(1),
stream: Some(false),
stop: None,
presence_penalty: None,
frequency_penalty: None,
logit_bias: None,
user: None,
messages: openai_conversation_messages,
};
if let Ok(request_as_json) = serde_json::to_string(&request) {
tracing::trace!(
model = format!("{:?}", request.model),
?messages_count,
request = request_as_json,
"Sending OpenAI-compat chat completion API request"
);
}
// This library is not async-aware, so we need to use `spawn_blocking` to run the request on a separate thread.
let client = self.client.clone();
let response =
tokio::task::spawn_blocking(move || client.chat_completion_create(&request)).await?;
let response = match response {
Ok(response) => response,
Err(err) => {
return Err(anyhow::anyhow!(
"Failed to get response from the OpenAI-compat chat completion API: {:?}",
err
));
}
};
tracing::trace!(
?response,
"Got response from the OpenAI-compat chat completion API"
);
// We only request 1 result, so there should only be 1 choice.
if let Some(choice) = response.choices.into_iter().next() {
let Some(message) = choice.message else {
return Err(anyhow::anyhow!(
"No response message in choice was returned from the OpenAI-compat chat completion API"
));
};
return Ok(TextGenerationResult {
text: message.content,
});
}
Err(anyhow::anyhow!(
"No response messages choices were returned from the OpenAI-compat chat completion API"
))
}
async fn speech_to_text(
&self,
_mime_type: &mxlink::mime::Mime,
media: Vec<u8>,
params: SpeechToTextParams,
) -> anyhow::Result<SpeechToTextResult> {
let Some(speech_to_text_config) = &self.config.speech_to_text else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::SpeechToText
),
));
};
// This library does not support passing the audio data as a byte slice, so we need to write it to a temporary file :/
//
// This temporary file will get auto-deleted when the variable goes out of scope.
let temp_file = tokio::task::spawn_blocking(move || {
let mut temp_file = match tempfile::NamedTempFile::new() {
Ok(file) => file,
Err(e) => return Err(e),
};
match std::io::Write::write_all(&mut temp_file, &media) {
Ok(_) => (),
Err(e) => return Err(e),
}
Ok(temp_file)
})
.await??;
let file_path = temp_file
.path()
.to_str()
.ok_or_else(|| anyhow::anyhow!("Failed to get temporary file path"))?;
let language = params.language_override.clone();
let request = AudioBody {
file: std::fs::File::open(file_path)?,
model: speech_to_text_config.model_id.to_owned(),
prompt: None,
response_format: None,
temperature: None,
language: language.clone(),
};
tracing::trace!(
model_id = speech_to_text_config.model_id,
?language,
"Sending OpenAI-compat speech-to-text API request"
);
// This library is not async-aware, so we need to use `spawn_blocking` to run the request on a separate thread.
let client = self.client.clone();
let response =
tokio::task::spawn_blocking(move || client.audio_transcription_create(request)).await?;
let response = match response {
Ok(response) => response,
Err(err) => {
return Err(anyhow::anyhow!(
"Failed to get response from the OpenAI-compat audio transcription API: {:?}",
err
));
}
};
tracing::trace!(
?response,
"Got response from the OpenAI-compat audio transcription API"
);
let Some(text) = response.text else {
return Err(anyhow::anyhow!(
"No response text was returned from the OpenAI-compat audio transcription API"
));
};
Ok(SpeechToTextResult { text })
}
async fn generate_image(
&self,
prompt: &str,
params: ImageGenerationParams,
) -> anyhow::Result<ImageGenerationResult> {
let Some(image_generation_config) = &self.config.image_generation else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::ImageGeneration
),
));
};
// It seems like some OpenAI-compatible providers (e.g. LocalAI with StableDiffusion) skip some requirements
// when they span multiple lines.
let prompt = prompt.replace("\n", " ");
let size: Option<String> = params
.size_override
.or_else(|| image_generation_config.size.clone());
let request = ImagesBody {
model: Some(image_generation_config.model_id.to_owned()),
prompt: prompt.to_owned(),
n: Some(1),
quality: image_generation_config.quality.clone(),
size,
style: image_generation_config.style.clone(),
response_format: Some("b64_json".to_string()),
user: None,
};
tracing::trace!(
?prompt,
model = format!("{:?}", request.model),
size = format!("{:?}", request.size),
style = format!("{:?}", request.style),
quality = format!("{:?}", request.quality),
"Sending OpenAI-compat image generation API request"
);
// This library is not async-aware, so we need to use `spawn_blocking` to run the request on a separate thread.
let client = self.client.clone();
let response = tokio::task::spawn_blocking(move || client.image_create(&request)).await?;
let response = match response {
Ok(response) => response,
Err(err) => {
return Err(anyhow::anyhow!(
"Failed to get response from the OpenAI-compat image creation API: {:?}",
err
));
}
};
let Some(data) = response.data else {
return Err(anyhow::anyhow!(
"The OpenAI-compat image generationAPI returned no image data"
));
};
if let Some(image) = data.into_iter().next() {
let Some(b64_json) = &image.b64_json else {
return Err(anyhow::anyhow!(
"The OpenAI-compat image generation API returned no b64_json image data"
));
};
let bytes = base64_decode(b64_json)?;
return Ok(ImageGenerationResult {
bytes,
mime_type: mxlink::mime::IMAGE_PNG,
revised_prompt: image.revised_prompt,
});
}
Err(anyhow::anyhow!(
"The OpenAI image generation API returned no images"
))
}
async fn text_to_speech(
&self,
input: &str,
params: TextToSpeechParams,
) -> anyhow::Result<TextToSpeechResult> {
// openai_api_rust does not support text-to-speech, so our only bet is to do it via async-openai and hope it works.
// At the time of testing (2024-09-09), providers like LocalAI can be used for text-to-speech via async-openai.
//
// So.. below we try to convert our Config struct to the Config struct from the openai module
// and invoke the openai controller.
// Quick check to make sure doing work below is worth it
let Some(_text_to_speech_config) = &self.config.text_to_speech else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_so_cannot_be_used(
&AgentPurpose::TextToSpeech
),
));
};
tracing::debug!("Converting OpenAI-compact config to OpenAI config..");
let openai_config = super::utils::convert_config_to_openai_config_lossy(&self.config);
let Some(_text_to_speech_config) = &openai_config.text_to_speech else {
return Err(anyhow::anyhow!(
strings::agent::no_configuration_for_purpose_after_conversion_so_cannot_be_used(
&AgentPurpose::TextToSpeech
),
));
};
let openai_controller = super::super::openai::Controller::new(openai_config);
tracing::error!("Invoking text-to-speech via the OpenAI controller..");
openai_controller.text_to_speech(input, params).await
}
fn supports_purpose(&self, purpose: AgentPurpose) -> bool {
match purpose {
AgentPurpose::ImageGeneration => self.config.image_generation.is_some(),
AgentPurpose::TextGeneration => self.config.text_generation.is_some(),
AgentPurpose::SpeechToText => self.config.speech_to_text.is_some(),
AgentPurpose::TextToSpeech => self.config.text_to_speech.is_some(),
AgentPurpose::CatchAll => true,
}
}
fn text_generation_prompt(&self) -> Option<String> {
let Some(text_generation_config) = &self.config.text_generation else {
return None;
};
text_generation_config.prompt.clone()
}
fn text_generation_temperature(&self) -> Option<f32> {
let Some(text_generation_config) = &self.config.text_generation else {
return None;
};
Some(text_generation_config.temperature)
}
fn text_to_speech_voice(&self) -> Option<String> {
let Some(text_to_speech_config) = &self.config.text_to_speech else {
return None;
};
// A hacky way to turn this enum to a string
let voice_as_string = serde_json::to_string(&text_to_speech_config.voice).ok()?;
Some(voice_as_string.replace("\"", ""))
}
fn text_to_speech_speed(&self) -> Option<f32> {
let Some(text_to_speech_config) = &self.config.text_to_speech else {
return None;
};
Some(text_to_speech_config.speed)
}
}

View File

@@ -0,0 +1,70 @@
// The openai_compat provider aims to support a wider ranger of OpenAI-compatible providers.
//
// The `openai` provider is based on `async-openai`, which only aims to support the OpenAI API spec. See:
// - https://github.com/64bit/async-openai/issues/266
// - https://github.com/64bit/async-openai/blob/05d5a1b4fa6476829dd1a34447b80279cf89d4f8/async-openai/README.md#contributing
//
// This module uses its own configuration, which avoids using strict types tied to OpenAI,
// and thus allows for more flexibility.
//
// Communication with the OpenAI-compatible API is handled by the `openai_api_rust` crate.
// Since this crate is not async-aware, we need to use tokio's `spawn_blocking` to invoke it.
//
// Certain features (e.g. text-to-speech) are not supported by `openai_api_rust` yet, so we may try to delegate them to the `openai` provider.
mod config;
mod controller;
mod utils;
pub use config::Config;
pub use controller::Controller;
use super::super::AgentInstantiationError;
use super::super::AgentInstantiationResult;
use super::controller::ControllerType;
use super::ConfigTrait;
pub fn create_controller_from_yaml_value_config(
agent_id: &str,
config: serde_yaml::Value,
) -> AgentInstantiationResult<ControllerType> {
let config = match &config {
serde_yaml::Value::Mapping(_) => {
let config: Config =
serde_yaml::from_value(config).map_err(AgentInstantiationError::Yaml)?;
config
.validate()
.map_err(AgentInstantiationError::ConfigFailsValidation)?;
config
}
_ => {
return Err(AgentInstantiationError::ConfigForAgentIsNotAMapping(
agent_id.to_owned(),
));
}
};
Ok(ControllerType::OpenAICompat(Box::new(Controller::new(
config,
))))
}
pub fn default_config() -> Config {
let mut config = Config::default();
if let Some(text_generation) = &mut config.text_generation {
text_generation.model_id = "some-model".to_string();
text_generation.max_response_tokens = 4096;
text_generation.max_context_tokens = 128_000;
}
// We don't support these, so let's remove them from the configuration.
config.text_to_speech = None;
config.image_generation = None;
config.base_url = "".to_owned();
config
}

View File

@@ -0,0 +1,78 @@
use openai_api_rust::{Message, Role};
use crate::agent::provider::openai::Config as OpenAIConfig;
use crate::conversation::llm::{Author as LLMAuthor, Message as LLMMessage};
pub fn convert_llm_messages_to_openai_messages(
conversation_messages: Vec<LLMMessage>,
) -> Vec<Message> {
let mut openai_conversation_messages: Vec<Message> =
Vec::with_capacity(conversation_messages.len());
for message in conversation_messages {
openai_conversation_messages.push(convert_llm_message_to_openai_message(message));
}
openai_conversation_messages
}
fn convert_llm_message_to_openai_message(llm_message: LLMMessage) -> Message {
let role = match llm_message.author {
LLMAuthor::Prompt => Role::System,
LLMAuthor::Assistant => Role::Assistant,
LLMAuthor::User => Role::User,
};
Message {
role,
content: llm_message.message_text,
}
}
pub(super) fn convert_config_to_openai_config_lossy(config: &super::Config) -> OpenAIConfig {
let text_generation = config
.text_generation
.as_ref()
.and_then(|tg| tg.clone().try_into().ok());
let speech_to_text = config
.speech_to_text
.as_ref()
.and_then(|stt| stt.clone().try_into().ok());
let text_to_speech = config
.text_to_speech
.as_ref()
.and_then(|tts| tts.clone().try_into().ok());
let image_generation = config
.image_generation
.as_ref()
.and_then(|ig| ig.clone().try_into().ok());
OpenAIConfig {
api_key: config.api_key.clone().unwrap_or("".to_string()),
text_generation,
speech_to_text,
text_to_speech,
image_generation,
base_url: config.base_url.clone(),
}
}
pub(super) fn convert_string_to_enum<T>(value: &str) -> Result<T, String>
where
T: serde::de::DeserializeOwned,
{
// This is a hacky way to construct an enum from the string we have.
let enum_result: serde_json::Result<T> = serde_json::from_str(&format!("\"{}\"", value));
match enum_result {
Ok(enum_result) => Ok(enum_result),
Err(err) => {
tracing::debug!(?err, "Failed to parse into enum");
Err(format!("The value ({}) is not supported.", value))
}
}
}

View File

@@ -0,0 +1,21 @@
use super::openai_compat::Config;
pub fn default_config() -> Config {
let mut config = Config {
base_url: "https://openrouter.ai/api/v1".to_owned(),
text_to_speech: None,
image_generation: None,
speech_to_text: None,
..Default::default()
};
if let Some(ref mut config) = config.text_generation.as_mut() {
config.model_id = "mattshumer/reflection-70b:free".to_owned();
config.max_context_tokens = 8192;
config.max_response_tokens = 2048;
}
config
}

View File

@@ -0,0 +1,21 @@
use super::openai_compat::Config;
pub fn default_config() -> Config {
let mut config = Config {
base_url: "https://api.together.xyz/v1".to_owned(),
text_to_speech: None,
image_generation: None,
speech_to_text: None,
..Default::default()
};
if let Some(ref mut config) = config.text_generation.as_mut() {
config.model_id = "meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo".to_owned();
config.max_context_tokens = 8192;
config.max_response_tokens = 2048;
}
config
}

67
src/agent/purpose.rs Normal file
View File

@@ -0,0 +1,67 @@
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum AgentPurpose {
CatchAll,
ImageGeneration,
TextGeneration,
TextToSpeech,
SpeechToText,
}
impl AgentPurpose {
pub fn from_str(s: &str) -> Option<Self> {
match s {
"catch-all" => Some(Self::CatchAll),
"image-generation" => Some(Self::ImageGeneration),
"text-generation" => Some(Self::TextGeneration),
"text-to-speech" => Some(Self::TextToSpeech),
"speech-to-text" => Some(Self::SpeechToText),
_ => None,
}
}
pub fn as_str(&self) -> &'static str {
match self {
Self::CatchAll => "catch-all",
Self::ImageGeneration => "image-generation",
Self::TextGeneration => "text-generation",
Self::TextToSpeech => "text-to-speech",
Self::SpeechToText => "speech-to-text",
}
}
pub fn choices() -> Vec<&'static Self> {
vec![
&Self::TextGeneration,
&Self::SpeechToText,
&Self::TextToSpeech,
&Self::ImageGeneration,
&Self::CatchAll,
]
}
pub fn emoji(&self) -> &'static str {
match self {
Self::CatchAll => "❓",
Self::TextGeneration => "💬",
Self::SpeechToText => "🦻",
Self::TextToSpeech => "🗣️",
Self::ImageGeneration => "🖌️",
}
}
pub fn heading(&self) -> &'static str {
match self {
Self::CatchAll => "Catch-All",
Self::TextGeneration => "Text Generation",
Self::SpeechToText => "Speech-to-Text",
Self::TextToSpeech => "Text-to-Speech",
Self::ImageGeneration => "Image Generation",
}
}
}
impl std::fmt::Display for AgentPurpose {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(self.as_str())
}
}

146
src/agent/utils.rs Normal file
View File

@@ -0,0 +1,146 @@
use base64::{engine::general_purpose::STANDARD, Engine as _};
use crate::{
agent::{
AgentInstance, AgentPurpose, ControllerTrait, Manager as AgentManager, PublicIdentifier,
},
entity::RoomConfigContext,
strings,
};
#[derive(Debug)]
pub struct AgentForPurposeDeterminationInfo {
pub instance: AgentInstance,
pub configuration_source: AgentForPurposeDeterminationInfoConfigurationSource,
}
#[derive(Debug)]
pub enum AgentForPurposeDeterminationInfoConfigurationSource {
Room,
Global,
}
#[derive(Debug)]
pub enum AgentForPurposeDeterminationError {
Unknown(String),
NoneConfigured,
ConfiguredButMissing(PublicIdentifier),
ConfiguredButLacksSupport(PublicIdentifier),
}
pub async fn get_effective_agent_for_purpose(
agent_manager: &AgentManager,
room_config_context: &RoomConfigContext,
agent_purpose: AgentPurpose,
) -> Result<AgentForPurposeDeterminationInfo, AgentForPurposeDeterminationError> {
let (agent_identifier, configuration_source) =
match get_effective_room_agent_identifier_for_purpose(room_config_context, agent_purpose)
.await
{
Ok((agent_identifier, configuration_source)) => {
(agent_identifier, configuration_source)
}
Err(err) => {
return Err(AgentForPurposeDeterminationError::Unknown(err));
}
};
let Some(agent_identifier) = agent_identifier else {
return Err(AgentForPurposeDeterminationError::NoneConfigured);
};
let agents = agent_manager.available_room_agents_by_room_config_context(room_config_context);
let Some(agent_instance) = agents.iter().find(|a| *a.identifier() == agent_identifier) else {
return Err(AgentForPurposeDeterminationError::ConfiguredButMissing(
agent_identifier,
));
};
let agent_instance = agent_instance.clone();
let supports_purpose = agent_instance.controller().supports_purpose(agent_purpose);
if !supports_purpose {
return Err(AgentForPurposeDeterminationError::ConfiguredButLacksSupport(agent_identifier));
}
Ok(AgentForPurposeDeterminationInfo {
instance: agent_instance,
configuration_source,
})
}
async fn get_effective_room_agent_identifier_for_purpose(
room_config_context: &RoomConfigContext,
purpose: AgentPurpose,
) -> Result<
(
Option<PublicIdentifier>,
AgentForPurposeDeterminationInfoConfigurationSource,
),
String,
> {
let (agent_id, configuration_source) =
get_effective_room_agent_raw_id_for_purpose(room_config_context, purpose).await;
let Some(agent_id) = agent_id else {
return Ok((None, configuration_source));
};
let agent_identifier = match PublicIdentifier::from_str(agent_id.as_str()) {
Some(agent_identifier) => agent_identifier,
None => return Err(strings::agent::invalid_id_generic()),
};
Ok((Some(agent_identifier), configuration_source))
}
async fn get_effective_room_agent_raw_id_for_purpose(
room_config_context: &RoomConfigContext,
purpose: AgentPurpose,
) -> (
Option<String>,
AgentForPurposeDeterminationInfoConfigurationSource,
) {
let agent_id = room_config_context
.room_config
.settings
.handler
.get_by_purpose_with_catch_all_fallback(purpose);
if let Some(agent_id) = agent_id {
return (
Some(agent_id),
AgentForPurposeDeterminationInfoConfigurationSource::Room,
);
}
tracing::trace!(
?purpose,
"No specific agent found for purpose in room, falling back to global.",
);
(
get_global_agent_id_for_purpose(room_config_context, purpose).await,
AgentForPurposeDeterminationInfoConfigurationSource::Global,
)
}
async fn get_global_agent_id_for_purpose(
room_config_context: &RoomConfigContext,
purpose: AgentPurpose,
) -> Option<String> {
room_config_context
.global_config
.fallback_room_settings
.handler
.get_by_purpose_with_catch_all_fallback(purpose)
}
pub(crate) fn base64_decode(base64_string: &str) -> Result<Vec<u8>, base64::DecodeError> {
STANDARD.decode(base64_string)
}

419
src/bot/implementation.rs Normal file
View File

@@ -0,0 +1,419 @@
use std::sync::Arc;
use std::{future::Future, pin::Pin};
use mxlink::matrix_sdk::media::{MediaFormat, MediaRequest};
use mxlink::matrix_sdk::ruma::{
events::room::MediaSource, MilliSecondsSinceUnixEpoch, OwnedUserId,
};
use mxlink::matrix_sdk::Room;
use mxlink::{
InitConfig, LoginConfig, LoginCredentials, LoginEncryption, MatrixLink, PersistenceConfig,
};
use mxlink::helpers::account_data_config::{
ConfigError, GlobalConfigManager as AccountDataGlobalConfigManager,
RoomConfigManager as AccountDataRoomConfigManager,
};
use mxlink::helpers::encryption::Manager as EncryptionManager;
use crate::agent::Manager as AgentManager;
use crate::entity::catch_up_marker::{
CatchUpMarker, CatchUpMarkerManager, DelayedCatchUpMarkerManager,
};
use crate::entity::cfg::Config;
use crate::entity::globalconfig::{GlobalConfig, GlobalConfigurationManager};
use crate::entity::roomconfig::{RoomConfig, RoomConfigurationManager};
use crate::agent::Manager;
use crate::conversation::matrix::{RoomDisplayNameFetcher, RoomEventFetcher};
const ROOM_EVENT_FETCHER_LRU_CACHE_SIZE: usize = 1000;
const ROOM_DISPLAY_NAME_FETCHER_LRU_CACHE_SIZE: usize = 1000;
const ROOM_CONFIG_MANAGER_LRU_CACHE_SIZE: usize = 1000;
const LOGO_BYTES: &[u8] = include_bytes!("../../etc/assets/baibot-torso-768.png");
const LOGO_MIME_TYPE: &str = "image/png";
/// Controls how often we persist the catch-up marker to Account Data.
/// Consult the `DelayedCatchUpMarkerManager` documentation for more information.
const DELAYED_CATCH_UP_MARKER_MANAGER_PERSIST_INTERVAL_DURATION: std::time::Duration =
std::time::Duration::from_secs(10);
/// Controls what federation delay we will tolerate. The timestamp that gets persisted
/// will be based on the last seen event's `origin_server_ts` minus this duration.
/// Consult the `DelayedCatchUpMarkerManager` documentation for more information.
const DELAYED_CATCH_UP_MARKER_MANAGER_FEDERATION_DELAY_TOLERANCE_DURATION: std::time::Duration =
std::time::Duration::from_secs(90);
struct BotInner {
config: Config,
matrix_link: MatrixLink,
delayed_catch_up_marker_manager: DelayedCatchUpMarkerManager,
global_config_manager: tokio::sync::Mutex<GlobalConfigurationManager>,
room_config_manager: tokio::sync::Mutex<RoomConfigurationManager>,
room_event_fetcher: Arc<RoomEventFetcher>,
room_display_name_fetcher: Arc<RoomDisplayNameFetcher>,
agent_manager: Manager,
admin_pattern_regexes: Vec<regex::Regex>,
}
/// Bot represents a bot instance.
///
/// All of the state is held in an `Arc` so the `Bot` can be cloned freely.
#[derive(Clone)]
pub struct Bot {
inner: Arc<BotInner>,
}
impl Bot {
pub async fn new(config: Config) -> anyhow::Result<Self> {
// Take some potentially problematic configuration values out of the config early on.
// If we'd be failing, we'd like it to happen early, before we log in, etc.
let initial_global_config: GlobalConfig =
config.initial_global_config.clone().try_into()?;
let admin_pattern_regexes = config.access.admin_pattern_regexes()?;
let persistence_config_encryption_key = config.persistence.config_encryption_key()?;
let agent_manager = AgentManager::new(config.agents.static_definitions.clone())?;
let encryption_manager = EncryptionManager::new(persistence_config_encryption_key);
let matrix_link = create_matrix_link(&config).await?;
let catch_up_marker_manager = create_catch_up_marker_manager(matrix_link.clone());
let delayed_catch_up_marker_manager = DelayedCatchUpMarkerManager::new(
catch_up_marker_manager,
DELAYED_CATCH_UP_MARKER_MANAGER_PERSIST_INTERVAL_DURATION,
DELAYED_CATCH_UP_MARKER_MANAGER_FEDERATION_DELAY_TOLERANCE_DURATION,
);
let global_config_manager = tokio::sync::Mutex::new(create_global_configuration_manager(
matrix_link.clone(),
encryption_manager.clone(),
initial_global_config,
));
let room_config_manager = tokio::sync::Mutex::new(create_room_configuration_manager(
matrix_link.clone(),
encryption_manager.clone(),
));
let room_event_fetcher = RoomEventFetcher::new(Some(ROOM_EVENT_FETCHER_LRU_CACHE_SIZE));
let room_display_name_fetcher = RoomDisplayNameFetcher::new(
matrix_link.clone(),
Some(ROOM_DISPLAY_NAME_FETCHER_LRU_CACHE_SIZE),
);
Ok(Self {
inner: Arc::new(BotInner {
config,
matrix_link,
delayed_catch_up_marker_manager,
global_config_manager,
room_config_manager,
room_event_fetcher: Arc::new(room_event_fetcher),
room_display_name_fetcher: Arc::new(room_display_name_fetcher),
agent_manager,
admin_pattern_regexes,
}),
})
}
pub(crate) fn admin_patterns(&self) -> &Vec<String> {
&self.inner.config.access.admin_patterns
}
pub(crate) fn name(&self) -> &str {
&self.inner.config.user.name
}
pub(crate) fn command_prefix(&self) -> &str {
&self.inner.config.command_prefix
}
pub(crate) fn homeserver_name(&self) -> &str {
&self.inner.config.homeserver.server_name
}
pub(crate) fn global_config_manager(&self) -> &tokio::sync::Mutex<GlobalConfigurationManager> {
&self.inner.global_config_manager
}
pub(crate) fn room_config_manager(&self) -> &tokio::sync::Mutex<RoomConfigurationManager> {
&self.inner.room_config_manager
}
pub(crate) fn room_event_fetcher(&self) -> Arc<RoomEventFetcher> {
self.inner.room_event_fetcher.clone()
}
pub(crate) fn room_display_name_fetcher(&self) -> Arc<RoomDisplayNameFetcher> {
self.inner.room_display_name_fetcher.clone()
}
pub(crate) fn agent_manager(&self) -> &Manager {
&self.inner.agent_manager
}
pub(crate) fn matrix_link(&self) -> &MatrixLink {
&self.inner.matrix_link
}
pub(crate) fn user_id(&self) -> &OwnedUserId {
self.matrix_link().user_id()
}
pub(crate) fn reacting(&self) -> super::reacting::Reacting {
super::reacting::Reacting::new(self.clone())
}
pub(crate) fn rooms(&self) -> super::rooms::Rooms {
super::rooms::Rooms::new(self.clone())
}
pub(crate) fn messaging(&self) -> super::messaging::Messaging {
super::messaging::Messaging::new(self.clone())
}
pub(crate) fn admin_pattern_regexes(&self) -> &Vec<regex::Regex> {
&self.inner.admin_pattern_regexes
}
pub(crate) async fn global_config(&self) -> Result<GlobalConfig, ConfigError> {
let mut global_config_manager_guard = self.inner.global_config_manager.lock().await;
global_config_manager_guard.get_or_create().await
}
pub(crate) async fn is_caught_up(
&self,
event_origin_server_ts: MilliSecondsSinceUnixEpoch,
) -> Result<bool, ConfigError> {
self.inner
.delayed_catch_up_marker_manager
.is_caught_up(event_origin_server_ts.0.into())
.await
}
pub(crate) async fn catch_up(&self, event_origin_server_ts: MilliSecondsSinceUnixEpoch) {
self.inner
.delayed_catch_up_marker_manager
.catch_up(event_origin_server_ts.0.into())
.await
}
pub async fn start(&self) -> anyhow::Result<()> {
self.rooms().attach_event_handlers().await;
self.messaging().attach_event_handlers().await;
self.reacting().attach_event_handlers().await;
self.inner.delayed_catch_up_marker_manager.start().await;
self.prepare_profile().await?;
self.inner
.matrix_link
.start()
.await
.map_err(|e| anyhow::anyhow!("Failed to sync: {:?}", e))
}
async fn prepare_profile(&self) -> anyhow::Result<()> {
use std::time::Duration;
use tokio::time::sleep;
let mut delay = Duration::from_secs(3);
let max_delay = Duration::from_secs(30);
loop {
match self.do_prepare_profile().await {
Ok(_) => return Ok(()),
Err(err) => {
tracing::warn!(
?err,
?delay,
"Failed to prepare profile.. Will retry after delay..."
);
sleep(delay).await;
delay = std::cmp::min(delay * 2, max_delay);
}
}
}
}
async fn do_prepare_profile(&self) -> anyhow::Result<()> {
tracing::debug!("Preparing profile..");
let account = self.inner.matrix_link.client().account();
let media = self.inner.matrix_link.client().media();
let desired_display_name = self.inner.config.user.name.clone();
let profile = account
.get_profile()
.await
.map_err(|e| anyhow::anyhow!("Failed fetching profile: {:?}", e))?;
let should_update_display_name = match &profile.displayname {
Some(displayname) => displayname != &desired_display_name,
None => true,
};
if should_update_display_name {
tracing::info!(
?profile.displayname,
?desired_display_name,
"Updating display name.."
);
if let Err(err) = account.set_display_name(Some(&desired_display_name)).await {
return Err(anyhow::anyhow!("Failed setting display name: {:?}", err));
}
}
let should_update_avatar = match &profile.avatar_url {
Some(avatar_url) => {
let request = MediaRequest {
source: MediaSource::Plain(avatar_url.to_owned()),
format: MediaFormat::File,
};
let content = media
.get_media_content(&request, true)
.await
.map_err(|e| anyhow::anyhow!("Failed fetching existing avatar: {:?}", e))?;
content.as_slice() != LOGO_BYTES
}
None => true,
};
if should_update_avatar {
tracing::info!("Updating avatar..");
let mime_type = LOGO_MIME_TYPE
.parse()
.expect("Failed parsing mime type for logo");
account
.upload_avatar(&mime_type, LOGO_BYTES.to_vec())
.await
.map_err(|e| anyhow::anyhow!("Failed uploading avatar: {:?}", e))?;
}
Ok(())
}
}
async fn create_matrix_link(config: &Config) -> anyhow::Result<MatrixLink> {
let session_file_path = config.persistence.session_file_path()?;
let session_encryption_key = config.persistence.session_encryption_key()?;
let db_dir_path: std::path::PathBuf = config.persistence.db_dir_path()?;
let login_creds = LoginCredentials::UserPassword(
config.user.mxid_localpart.to_owned(),
config.user.password.to_owned(),
);
let login_encryption = LoginEncryption::new(
config.user.encryption.recovery_passphrase.clone(),
config.user.encryption.recovery_reset_allowed,
);
let login_config = LoginConfig::new(
config.homeserver.url.to_owned(),
login_creds,
Some(login_encryption),
config.user.name.to_owned(),
);
let persistence_config =
PersistenceConfig::new(session_file_path, session_encryption_key, db_dir_path);
let init_config = InitConfig::new(login_config, persistence_config);
mxlink::init(&init_config).await.map_err(|e| e.into())
}
pub fn create_global_configuration_manager(
matrix_link: MatrixLink,
encryption_manager: EncryptionManager,
initial_global_config: GlobalConfig,
) -> GlobalConfigurationManager {
let initial_global_config_callback = move || {
let initial_global_config = initial_global_config.clone();
let future = create_initial_global_config(initial_global_config);
// Explicitly box the future to match the expected type
Box::pin(future) as Pin<Box<dyn Future<Output = GlobalConfig> + Send>>
};
AccountDataGlobalConfigManager::new(
matrix_link,
encryption_manager,
initial_global_config_callback,
)
}
async fn create_initial_global_config(initial_global_config: GlobalConfig) -> GlobalConfig {
initial_global_config
}
pub fn create_room_configuration_manager(
matrix_link: MatrixLink,
encryption_manager: EncryptionManager,
) -> RoomConfigurationManager {
let initial_room_config_callback = |room: Room| {
let future = create_initial_room_config(room);
// Explicitly box the future to match the expected type
Box::pin(future) as Pin<Box<dyn Future<Output = RoomConfig> + Send>>
};
AccountDataRoomConfigManager::new(
matrix_link.user_id().clone(),
encryption_manager,
initial_room_config_callback,
Some(ROOM_CONFIG_MANAGER_LRU_CACHE_SIZE),
)
}
async fn create_initial_room_config(room: Room) -> RoomConfig {
RoomConfig::default().with_room(room).await
}
pub fn create_catch_up_marker_manager(matrix_link: MatrixLink) -> CatchUpMarkerManager {
let initial_global_config_callback = || {
let future = create_initial_catch_up_marker();
// Explicitly box the future to match the expected type
Box::pin(future) as Pin<Box<dyn Future<Output = CatchUpMarker> + Send>>
};
// Intentionally not using encryption, to make this resilient even if we lose our encryption key.
// We're not worried about the catch-up marker being read or tampered with, as it's not sensitive data.
let encryption_manager = EncryptionManager::new(None);
let catch_up_marker_manager: CatchUpMarkerManager = AccountDataGlobalConfigManager::new(
matrix_link.clone(),
encryption_manager,
initial_global_config_callback,
);
catch_up_marker_manager
}
async fn create_initial_catch_up_marker() -> CatchUpMarker {
CatchUpMarker::new(0)
}

110
src/bot/load_config.rs Normal file
View File

@@ -0,0 +1,110 @@
use std::env;
use std::path::PathBuf;
use anyhow::anyhow;
use crate::agent::AgentPurpose;
pub use crate::entity::cfg::{defaults as cfg_defaults, env as cfg_env, Config};
pub fn load() -> anyhow::Result<Config> {
let config_file_path = env::var(cfg_env::BAIBOT_CONFIG_FILE_PATH)
.unwrap_or_else(|_| cfg_defaults::config_file_path().to_owned());
let config_file_path = PathBuf::from(config_file_path);
if !config_file_path.exists() {
return Err(anyhow!(
"Config file ({}) not found. Adjust the {} environment variable to use another config file.",
config_file_path.display(),
cfg_env::BAIBOT_CONFIG_FILE_PATH,
));
}
let config_str = std::fs::read_to_string(config_file_path)?;
let mut config: Config = serde_yaml::from_str(&config_str)?;
// Allow environment variables to override some configuration keys
for (key, value) in env::vars() {
match key.as_str() {
cfg_env::BAIBOT_HOMESERVER_SERVER_NAME => config.homeserver.server_name = value,
cfg_env::BAIBOT_HOMESERVER_URL => config.homeserver.url = value,
cfg_env::BAIBOT_USER_MXID_LOCALPART => config.user.mxid_localpart = value,
cfg_env::BAIBOT_USER_PASSWORD => config.user.password = value,
cfg_env::BAIBOT_USER_ENCRYPTION_RECOVERY_PASSPHRASE => {
config.user.encryption.recovery_passphrase = Some(value);
}
cfg_env::BAIBOT_USER_NAME => config.user.name = value,
cfg_env::BAIBOT_COMMAND_PREFIX => config.command_prefix = value,
cfg_env::BAIBOT_LOGGING => {
config.logging = value;
}
cfg_env::BAIBOT_ACCESS_ADMIN_PATTERNS => {
config.access.admin_patterns = value
.split(' ')
.map(|s| s.trim().to_string())
.filter(|s| !s.is_empty())
.collect();
}
cfg_env::BAIBOT_PERSISTENCE_DATA_DIR_PATH => {
config.persistence.data_dir_path = Some(value);
}
cfg_env::BAIBOT_PERSISTENCE_CONFIG_ENCRYPTION_KEY => {
config.persistence.config_encryption_key = Some(value);
}
cfg_env::BAIBOT_INITIAL_GLOBAL_CONFIG_HANDLER_CATCH_ALL => {
let value = if value.is_empty() { None } else { Some(value) };
config
.initial_global_config
.handler
.set_by_purpose(AgentPurpose::CatchAll, value);
}
cfg_env::BAIBOT_INITIAL_GLOBAL_CONFIG_HANDLER_TEXT_GENERATION => {
let value = if value.is_empty() { None } else { Some(value) };
config
.initial_global_config
.handler
.set_by_purpose(AgentPurpose::TextGeneration, value);
}
cfg_env::BAIBOT_INITIAL_GLOBAL_CONFIG_HANDLER_TEXT_TO_SPEECH => {
let value = if value.is_empty() { None } else { Some(value) };
config
.initial_global_config
.handler
.set_by_purpose(AgentPurpose::TextToSpeech, value);
}
cfg_env::BAIBOT_INITIAL_GLOBAL_CONFIG_HANDLER_SPEECH_TO_TEXT => {
let value = if value.is_empty() { None } else { Some(value) };
config
.initial_global_config
.handler
.set_by_purpose(AgentPurpose::SpeechToText, value);
}
cfg_env::BAIBOT_INITIAL_GLOBAL_CONFIG_HANDLER_IMAGE_GENERATION => {
let value = if value.is_empty() { None } else { Some(value) };
config
.initial_global_config
.handler
.set_by_purpose(AgentPurpose::ImageGeneration, value);
}
cfg_env::BAIBOT_INITIAL_GLOBAL_CONFIG_USER_PATTERNS => {
config.initial_global_config.user_patterns = Some(
value
.split(' ')
.map(|s| s.trim().to_string())
.filter(|s| !s.is_empty())
.collect(),
);
}
_ => {}
}
}
config.validate().map_err(|s| anyhow!(s))?;
Ok(config)
}

334
src/bot/messaging.rs Normal file
View File

@@ -0,0 +1,334 @@
use mxlink::matrix_sdk::{
ruma::{
api::client::receipt::create_receipt::v3::ReceiptType,
events::room::message::OriginalSyncRoomMessageEvent, OwnedEventId,
},
Room,
};
use mxlink::{CallbackError, MessageResponseType};
use tracing::Instrument;
use crate::{
conversation::matrix::determine_thread_context_for_room_event,
entity::{MessageContext, MessagePayload, RoomConfigContext, TriggerEventInfo},
};
#[derive(Clone)]
pub struct Messaging {
bot: super::Bot,
}
impl Messaging {
pub fn new(bot: super::Bot) -> Self {
Self { bot }
}
pub async fn send_text_markdown_no_fail(
&self,
room: &Room,
message: String,
response_type: MessageResponseType,
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
{
let result = self
.bot
.matrix_link()
.messaging()
.send_text_markdown(room, message, response_type)
.await;
match result {
Ok(result) => Some(result),
Err(err) => {
tracing::error!(
room_id = format!("{:?}", room.room_id()),
?err,
"Failed to send text message to room",
);
None
}
}
}
pub async fn send_notice_markdown_no_fail(
&self,
room: &Room,
message: String,
response_type: MessageResponseType,
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
{
let result = self
.bot
.matrix_link()
.messaging()
.send_notice_markdown(room, message, response_type)
.await;
match result {
Ok(result) => Some(result),
Err(err) => {
tracing::error!(
room_id = format!("{:?}", room.room_id()),
?err,
"Failed to send notice message to room",
);
None
}
}
}
pub async fn send_tooltip_markdown_no_fail(
&self,
room: &Room,
message: &str,
response_type: MessageResponseType,
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
{
self.send_notice_markdown_no_fail(
room,
crate::utils::status::create_tooltip_message_text(message),
response_type,
)
.await
}
pub async fn send_success_markdown_no_fail(
&self,
room: &Room,
message: &str,
response_type: MessageResponseType,
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
{
self.send_notice_markdown_no_fail(
room,
crate::utils::status::create_success_message_text(message),
response_type,
)
.await
}
pub async fn send_error_markdown_no_fail(
&self,
room: &Room,
err: &str,
response_type: MessageResponseType,
) -> Option<mxlink::matrix_sdk::ruma::api::client::message::send_message_event::v3::Response>
{
self.send_notice_markdown_no_fail(
room,
crate::utils::status::create_error_message_text(err),
response_type,
)
.await
}
pub async fn redact_event_no_fail(
&self,
room: &Room,
target_event_id: OwnedEventId,
reason: Option<String>,
) -> Option<mxlink::matrix_sdk::ruma::api::client::redact::redact_event::v3::Response> {
let result = self
.bot
.matrix_link()
.messaging()
.redact_event(room, target_event_id.clone(), reason)
.await;
match result {
Ok(result) => Some(result),
Err(err) => {
tracing::error!(
room_id = format!("{:?}", room.room_id()),
?target_event_id,
?err,
"Failed to send redaction to room",
);
None
}
}
}
pub(super) async fn attach_event_handlers(&self) {
let matrix_link_messaging = self.bot.matrix_link().messaging();
let this = self.clone();
matrix_link_messaging.on_actionable_room_message(|event, room| async move {
this.on_actionable_message(event, room).await
});
}
#[tracing::instrument(name = "bot_on_actionable_message", skip_all, fields(room_id = room.room_id().as_str(), event_id = event.event_id.as_str()))]
async fn on_actionable_message(
&self,
event: OriginalSyncRoomMessageEvent,
room: Room,
) -> Result<(), CallbackError> {
if self
.bot
.is_caught_up(event.origin_server_ts)
.await
.map_err(|e| {
CallbackError::Unknown(
format!("Failed to determine catch-up state: {:?}", e).into(),
)
})?
{
tracing::debug!(
event_origin_server_ts = format!("{:?}", event.origin_server_ts),
"Ignoring old message event",
);
return Ok(());
}
tracing::info!("Processing message");
let global_config = self
.bot
.global_config()
.await
.map_err(|err| CallbackError::Unknown(err.into()))?;
tracing::trace!(?global_config, "Global config");
let room_config = self
.bot
.room_config_manager()
.lock()
.await
.get_or_create_for_room(&room)
.await
.map_err(|err| CallbackError::Unknown(err.into()))?;
tracing::trace!(?room_config, "Room config");
let trigger_event_sender_is_admin = mxidwc::match_user_id(
event.sender.clone().as_str(),
self.bot.admin_pattern_regexes(),
);
let trigger_event_sender_is_allowed_user = match &global_config.access.user_patterns {
Some(user_patterns) => {
let allowed_user_regexes = mxidwc::parse_patterns_vector(user_patterns)
.map_err(|err| CallbackError::Unknown(err.into()))?;
mxidwc::match_user_id(event.sender.clone().as_str(), &allowed_user_regexes)
}
None => false,
};
if !trigger_event_sender_is_admin && !trigger_event_sender_is_allowed_user {
tracing::debug!("Ignoring message from non-admin/non-allowed user");
return Ok(());
}
let payload: Result<MessagePayload, String> = event.content.msgtype.clone().try_into();
let payload = match payload {
Ok(payload) => payload,
Err(err) => {
tracing::debug!(
msg_type = event.content.msgtype(),
?err,
"Ignoring message not supported by us",
);
return Ok(());
}
};
let thread_context = determine_thread_context_for_room_event(
self.bot.user_id(),
&room,
&event,
&payload,
&self.bot.room_event_fetcher(),
)
.await;
let thread_context = match thread_context {
Ok(value) => value,
Err(err) => {
tracing::error!(?err, "Failed to determine thread context for event");
return Ok(());
}
};
let Some(thread_context) = thread_context else {
tracing::debug!("Ignoring message with unknown thread context (likely not a threaded message or a top-level message)");
return Ok(());
};
let room_config_context =
RoomConfigContext::new(global_config.clone(), room_config.clone());
let trigger_event_info = TriggerEventInfo::new(
event.event_id.clone(),
event.sender.clone(),
payload,
trigger_event_sender_is_admin,
);
let message_context = MessageContext::new(
room.clone(),
room_config_context,
self.bot.admin_pattern_regexes().clone(),
trigger_event_info,
thread_context.info.clone(),
);
let bot_display_name = self
.bot
.room_display_name_fetcher()
.own_display_name_in_room(message_context.room())
.await;
let bot_display_name = match bot_display_name {
Ok(value) => value,
Err(err) => {
tracing::warn!(
?err,
"Failed to fetch bot display name. Proceeding without it"
);
None
}
};
// The first event in the thread determines which handler processes the current event.
let controller_type = crate::controller::determine_controller(
self.bot.command_prefix(),
&thread_context.first_message,
&message_context,
self.bot.user_id(),
&bot_display_name,
);
tracing::info!(?controller_type, "Determined controller");
let _ = room
.send_single_receipt(
ReceiptType::Read,
thread_context.info.clone().into(),
event.event_id.clone(),
)
.await;
let start_time = std::time::Instant::now();
let event_span = tracing::error_span!("message_controller", ?controller_type);
crate::controller::dispatch_controller(&controller_type, &message_context, &self.bot)
.instrument(event_span)
.await;
let duration = std::time::Instant::now().duration_since(start_time);
tracing::debug!(?duration, "Controller finished");
self.bot.catch_up(event.origin_server_ts).await;
return Ok(());
}
}

8
src/bot/mod.rs Normal file
View File

@@ -0,0 +1,8 @@
mod implementation;
mod load_config;
mod messaging;
mod reacting;
mod rooms;
pub use implementation::Bot;
pub use load_config::load as load_config;

Some files were not shown because too many files have changed in this diff Show More