summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
-rw-r--r--README.md41
-rw-r--r--models.yaml62
-rw-r--r--src/serve.rs4
3 files changed, 47 insertions, 60 deletions
diff --git a/README.md b/README.md
index f7f6e23..9c5cb04 100644
--- a/README.md
+++ b/README.md
@@ -4,7 +4,7 @@
[![Crates](https://img.shields.io/crates/v/aichat.svg)](https://crates.io/crates/aichat)
[![Discord](https://img.shields.io/discord/1226737085453701222?label=Discord)](https://discord.gg/mr3ZZUB9hG)
-AIChat is an all-in-one AI CLI tool featuring chat REPL, RAG, function calling, and more.
+AIChat is an all-in-one AI CLI tool featuring chat REPL, RAG, function calling, agents, and more.
![AIChat Command](https://github.com/sigoden/aichat/assets/4012553/84ae8382-62be-41d0-a0f1-101b113c5bc7)
@@ -26,27 +26,27 @@ AIChat is an all-in-one AI CLI tool featuring chat REPL, RAG, function calling,
## Supported LLMs
-- OpenAI GPT-3.5/GPT-4 (paid, vision, embedding, function-calling)
-- Gemini: Gemini-1.0/Gemini-1.5 (free, paid, vision, embedding, function-calling)
-- Claude: Claude-3 (vision, paid, function-calling)
+- OpenAI: GPT-4/GPT-3.5 (paid, vision, embedding, function-calling)
+- Gemini: Gemini-1.5/Gemini-1.0 (free, paid, vision, embedding, function-calling)
+- Claude: Claude-3.5/Claude-3 (paid, vision, function-calling)
- Mistral (paid, embedding, function-calling)
-- Cohere: Command-R/Command-R+ (paid, embedding, function-calling)
+- Cohere: Command-R/Command-R+ (paid, embedding, rerank, function-calling)
- Reka (paid, vision)
- Perplexity: Llama-3/Mixtral (paid)
-- Groq: Llama-3/Mixtral/Gemma (free)
+- Groq: Llama-3/Mixtral/Gemma (free, function-calling)
- Ollama (free, local, embedding)
- Azure OpenAI (paid, vision, embedding, function-calling)
-- VertexAI: Gemini-1.0/Gemini-1.5 (paid, vision, embedding, function-calling)
-- VertexAI-Claude: Claude-3 (paid, vision)
-- Bedrock: Llama-3/Claude-3/Mistral (paid, vision)
-- Cloudflare (free, paid, vision)
+- VertexAI: Gemini-1.5/Gemini-1.0 (paid, vision, embedding, function-calling)
+- VertexAI-Claude: Claude-3.5/Claude-3 (paid, vision)
+- Bedrock: Llama-3/Claude-3.5/Claude-3/Mistral (paid, vision)
+- Cloudflare (free, vision, embedding)
- Replicate (paid)
- Ernie (paid)
-- Qianwen (paid, vision, embedding)
-- Moonshot (paid)
+- Qianwen: Qwen (paid, vision, embedding, function-calling)
+- Moonshot (paid, function-calling)
- Deepseek (paid)
-- ZhipuAI: GLM-3.5/GLM-4 (paid, vision)
-- LingYiWanWu (paid)
+- ZhipuAI: GLM-4 (paid, vision, function-calling)
+- LingYiWanWu: Yi-Large (paid, vision)
- Other openAI-compatible platforms
## Install
@@ -410,12 +410,23 @@ The LLM Arena is a web-based platform where you can compare different LLMs side-
Function calling supercharges LLMs by connecting them to external tools and data sources. This unlocks a world of possibilities, enabling LLMs to go beyond their core capabilities and tackle a wider range of tasks.
-We have created a new repository to help you make the most of this feature: [https://github.com/sigoden/llm-functions](https://github.com/sigoden/llm-functions)
+We have created a new repository [https://github.com/sigoden/llm-functions](https://github.com/sigoden/llm-functions) to help you make the most of this feature
Here's a glimpse of what function calling can do for you:
![image](https://github.com/sigoden/aichat/assets/4012553/c1b6b136-bbd3-4028-9b01-7d728390c0bf)
+## AI Agents
+
+Agent = Prompt (Role) + Tools (Function Callings) + Knowndge (RAG). It's also known as OpenAI's GPTs.
+
+The repository [https://github.com/sigoden/llm-functions](https://github.com/sigoden/llm-functions) provides utilities for developing agents and shares agents developed by the community.
+
+Here's a glimpse of what function calling can do for you:
+
+![image](https://github.com/sigoden/aichat/assets/4012553/d544f00d-5303-4393-a9fb-5e3f20f88412)
+
+
## Wikis
- [Role Guide](https://github.com/sigoden/aichat/wiki/Role-Guide)
diff --git a/models.yaml b/models.yaml
index be4ecc3..87e1b14 100644
--- a/models.yaml
+++ b/models.yaml
@@ -268,12 +268,10 @@
max_input_tokens: 32768
input_price: 0.24
output_price: 0.24
- supports_function_calling: true
- name: gemma-7b-it
max_input_tokens: 8192
input_price: 0.07
output_price: 0.07
- supports_function_calling: true
- platform: vertexai
# docs:
@@ -319,7 +317,6 @@
# - https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/use-claude
# notes:
# - get max_output_tokens info from models doc
- # - claude models have not been tested
models:
- name: claude-3-5-sonnet@20240620
max_input_tokens: 200000
@@ -360,7 +357,6 @@
# - https://aws.amazon.com/bedrock/pricing/
# notes:
# - get max_output_tokens info from playground
- # - claude/llama models have not been tested
models:
- name: anthropic.claude-3-5-sonnet-20240620-v1:0
max_input_tokens: 200000
@@ -429,8 +425,6 @@
# docs:
# - https://developers.cloudflare.com/workers-ai/models/
# - https://developers.cloudflare.com/workers-ai/platform/pricing/
- # notes:
- # - get max_output_tokens from playground
models:
- name: '@cf/meta/llama-3-8b-instruct'
max_input_tokens: 6144
@@ -472,8 +466,6 @@
# - https://replicate.com/explore
# - https://replicate.com/pricing
# - https://replicate.com/docs/reference/http
- # notes:
- # - max_output_tokens is required but unknown
models:
- name: meta/meta-llama-3-70b-instruct
max_input_tokens: 8192
@@ -504,29 +496,21 @@
# docs:
# - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/Nlks5zkzu
# - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/hlrk4akp7
- # notes:
- # - get max_output_tokens info from models doc
models:
- name: ernie-4.0-8k-0613
- max_input_tokens: 5120
- max_output_tokens: 2048
- require_max_tokens: true
+ max_input_tokens: 8192
input_price: 16.8
output_price: 16.8
- name: ernie-3.5-8k-0613
- max_input_tokens: 5120
- max_output_tokens: 2048
- require_max_tokens: true
+ max_input_tokens: 8192
input_price: 1.68
output_price: 1.68
- name: ernie-speed-128k
- max_input_tokens: 124000
- max_output_tokens: 4096
- require_max_tokens: true
+ max_input_tokens: 128000
input_price: 0
output_price: 0
- name: ernie-lite-8k
- max_input_tokens: 7168
+ max_input_tokens: 8192
max_output_tokens: 2048
require_max_tokens: true
input_price: 0
@@ -536,36 +520,30 @@
# docs:
# - https://help.aliyun.com/zh/dashscope/developer-reference/tongyiqianwen-large-language-models/
# - https://help.aliyun.com/zh/dashscope/developer-reference/qwen-vl-plus/
- # notes:
- # - get max_output_tokens info from models doc
models:
- name: qwen-long
max_input_tokens: 1000000
input_price: 0.07
output_price: 0.28
- name: qwen-turbo
- max_input_tokens: 6000
- max_output_tokens: 1500
+ max_input_tokens: 8000
input_price: 0.28
output_price: 0.84
supports_function_calling: true
- name: qwen-plus
- max_input_tokens: 30000
- max_output_tokens: 2000
+ max_input_tokens: 32000
input_price: 0.56
output_price: 1.68
supports_function_calling: true
- name: qwen-max
- max_input_tokens: 6000
- max_output_tokens: 2000
+ max_input_tokens: 8000
input_price: 5.6
output_price: 16.8
supports_function_calling: true
- name: qwen-max-longcontext
input_price: 5.6
output_price: 16.8
- max_input_tokens: 28000
- max_output_tokens: 2000
+ max_input_tokens: 30000
- name: qwen-vl-plus
input_price: 1.12
output_price: 1.12
@@ -585,8 +563,6 @@
# - https://platform.moonshot.cn/docs/intro
# - https://platform.moonshot.cn/docs/pricing
# - https://platform.moonshot.cn/docs/api-reference
- # notes:
- # - unable to get max_output_tokens info
models:
- name: moonshot-v1-8k
max_input_tokens: 8000
@@ -662,15 +638,23 @@
max_input_tokens: 32768
input_price: 2.8
output_price: 2.8
- - name: yi-medium
+ - name: yi-large-turbo
max_input_tokens: 16384
- input_price: 0.35
- output_price: 0.35
+ input_price: 1.68
+ output_price: 1.68
+ - name: yi-large-rag
+ max_input_tokens: 16384
+ input_price: 3.5
+ output_price: 3.5
- name: yi-vision
max_input_tokens: 4096
input_price: 0.84
output_price: 0.84
supports_vision: true
+ - name: yi-medium
+ max_input_tokens: 16384
+ input_price: 0.35
+ output_price: 0.35
- name: yi-medium-200k
max_input_tokens: 200000
input_price: 1.68
@@ -679,14 +663,6 @@
max_input_tokens: 16384
input_price: 0.14
output_price: 0.14
- - name: yi-large-rag
- max_input_tokens: 16384
- input_price: 3.5
- output_price: 3.5
- - name: yi-large-turbo
- max_input_tokens: 16384
- input_price: 1.68
- output_price: 1.68
- platform: anyscale
# docs:
diff --git a/src/serve.rs b/src/serve.rs
index 8371f55..344a53f 100644
--- a/src/serve.rs
+++ b/src/serve.rs
@@ -218,7 +218,7 @@ impl Server {
debug!("chat completions request: {req_body}");
let req_body = serde_json::from_value(req_body)
- .map_err(|err| anyhow!("Invalid requst body, {err}"))?;
+ .map_err(|err| anyhow!("Invalid request body, {err}"))?;
let ChatCompletionsReqBody {
model,
@@ -361,7 +361,7 @@ impl Server {
debug!("embeddings request: {req_body}");
let req_body = serde_json::from_value(req_body)
- .map_err(|err| anyhow!("Invalid requst body, {err}"))?;
+ .map_err(|err| anyhow!("Invalid request body, {err}"))?;
let EmbeddingsReqBody {
input,