diff options
| -rw-r--r-- | README.md | 41 | ||||
| -rw-r--r-- | models.yaml | 62 | ||||
| -rw-r--r-- | src/serve.rs | 4 |
3 files changed, 47 insertions, 60 deletions
@@ -4,7 +4,7 @@ [](https://crates.io/crates/aichat) [](https://discord.gg/mr3ZZUB9hG) -AIChat is an all-in-one AI CLI tool featuring chat REPL, RAG, function calling, and more. +AIChat is an all-in-one AI CLI tool featuring chat REPL, RAG, function calling, agents, and more.  @@ -26,27 +26,27 @@ AIChat is an all-in-one AI CLI tool featuring chat REPL, RAG, function calling, ## Supported LLMs -- OpenAI GPT-3.5/GPT-4 (paid, vision, embedding, function-calling) -- Gemini: Gemini-1.0/Gemini-1.5 (free, paid, vision, embedding, function-calling) -- Claude: Claude-3 (vision, paid, function-calling) +- OpenAI: GPT-4/GPT-3.5 (paid, vision, embedding, function-calling) +- Gemini: Gemini-1.5/Gemini-1.0 (free, paid, vision, embedding, function-calling) +- Claude: Claude-3.5/Claude-3 (paid, vision, function-calling) - Mistral (paid, embedding, function-calling) -- Cohere: Command-R/Command-R+ (paid, embedding, function-calling) +- Cohere: Command-R/Command-R+ (paid, embedding, rerank, function-calling) - Reka (paid, vision) - Perplexity: Llama-3/Mixtral (paid) -- Groq: Llama-3/Mixtral/Gemma (free) +- Groq: Llama-3/Mixtral/Gemma (free, function-calling) - Ollama (free, local, embedding) - Azure OpenAI (paid, vision, embedding, function-calling) -- VertexAI: Gemini-1.0/Gemini-1.5 (paid, vision, embedding, function-calling) -- VertexAI-Claude: Claude-3 (paid, vision) -- Bedrock: Llama-3/Claude-3/Mistral (paid, vision) -- Cloudflare (free, paid, vision) +- VertexAI: Gemini-1.5/Gemini-1.0 (paid, vision, embedding, function-calling) +- VertexAI-Claude: Claude-3.5/Claude-3 (paid, vision) +- Bedrock: Llama-3/Claude-3.5/Claude-3/Mistral (paid, vision) +- Cloudflare (free, vision, embedding) - Replicate (paid) - Ernie (paid) -- Qianwen (paid, vision, embedding) -- Moonshot (paid) +- Qianwen: Qwen (paid, vision, embedding, function-calling) +- Moonshot (paid, function-calling) - Deepseek (paid) -- ZhipuAI: GLM-3.5/GLM-4 (paid, vision) -- LingYiWanWu (paid) +- ZhipuAI: GLM-4 (paid, vision, function-calling) +- LingYiWanWu: Yi-Large (paid, vision) - Other openAI-compatible platforms ## Install @@ -410,12 +410,23 @@ The LLM Arena is a web-based platform where you can compare different LLMs side- Function calling supercharges LLMs by connecting them to external tools and data sources. This unlocks a world of possibilities, enabling LLMs to go beyond their core capabilities and tackle a wider range of tasks. -We have created a new repository to help you make the most of this feature: [https://github.com/sigoden/llm-functions](https://github.com/sigoden/llm-functions) +We have created a new repository [https://github.com/sigoden/llm-functions](https://github.com/sigoden/llm-functions) to help you make the most of this feature Here's a glimpse of what function calling can do for you:  +## AI Agents + +Agent = Prompt (Role) + Tools (Function Callings) + Knowndge (RAG). It's also known as OpenAI's GPTs. + +The repository [https://github.com/sigoden/llm-functions](https://github.com/sigoden/llm-functions) provides utilities for developing agents and shares agents developed by the community. + +Here's a glimpse of what function calling can do for you: + + + + ## Wikis - [Role Guide](https://github.com/sigoden/aichat/wiki/Role-Guide) diff --git a/models.yaml b/models.yaml index be4ecc3..87e1b14 100644 --- a/models.yaml +++ b/models.yaml @@ -268,12 +268,10 @@ max_input_tokens: 32768 input_price: 0.24 output_price: 0.24 - supports_function_calling: true - name: gemma-7b-it max_input_tokens: 8192 input_price: 0.07 output_price: 0.07 - supports_function_calling: true - platform: vertexai # docs: @@ -319,7 +317,6 @@ # - https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/use-claude # notes: # - get max_output_tokens info from models doc - # - claude models have not been tested models: - name: claude-3-5-sonnet@20240620 max_input_tokens: 200000 @@ -360,7 +357,6 @@ # - https://aws.amazon.com/bedrock/pricing/ # notes: # - get max_output_tokens info from playground - # - claude/llama models have not been tested models: - name: anthropic.claude-3-5-sonnet-20240620-v1:0 max_input_tokens: 200000 @@ -429,8 +425,6 @@ # docs: # - https://developers.cloudflare.com/workers-ai/models/ # - https://developers.cloudflare.com/workers-ai/platform/pricing/ - # notes: - # - get max_output_tokens from playground models: - name: '@cf/meta/llama-3-8b-instruct' max_input_tokens: 6144 @@ -472,8 +466,6 @@ # - https://replicate.com/explore # - https://replicate.com/pricing # - https://replicate.com/docs/reference/http - # notes: - # - max_output_tokens is required but unknown models: - name: meta/meta-llama-3-70b-instruct max_input_tokens: 8192 @@ -504,29 +496,21 @@ # docs: # - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/Nlks5zkzu # - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/hlrk4akp7 - # notes: - # - get max_output_tokens info from models doc models: - name: ernie-4.0-8k-0613 - max_input_tokens: 5120 - max_output_tokens: 2048 - require_max_tokens: true + max_input_tokens: 8192 input_price: 16.8 output_price: 16.8 - name: ernie-3.5-8k-0613 - max_input_tokens: 5120 - max_output_tokens: 2048 - require_max_tokens: true + max_input_tokens: 8192 input_price: 1.68 output_price: 1.68 - name: ernie-speed-128k - max_input_tokens: 124000 - max_output_tokens: 4096 - require_max_tokens: true + max_input_tokens: 128000 input_price: 0 output_price: 0 - name: ernie-lite-8k - max_input_tokens: 7168 + max_input_tokens: 8192 max_output_tokens: 2048 require_max_tokens: true input_price: 0 @@ -536,36 +520,30 @@ # docs: # - https://help.aliyun.com/zh/dashscope/developer-reference/tongyiqianwen-large-language-models/ # - https://help.aliyun.com/zh/dashscope/developer-reference/qwen-vl-plus/ - # notes: - # - get max_output_tokens info from models doc models: - name: qwen-long max_input_tokens: 1000000 input_price: 0.07 output_price: 0.28 - name: qwen-turbo - max_input_tokens: 6000 - max_output_tokens: 1500 + max_input_tokens: 8000 input_price: 0.28 output_price: 0.84 supports_function_calling: true - name: qwen-plus - max_input_tokens: 30000 - max_output_tokens: 2000 + max_input_tokens: 32000 input_price: 0.56 output_price: 1.68 supports_function_calling: true - name: qwen-max - max_input_tokens: 6000 - max_output_tokens: 2000 + max_input_tokens: 8000 input_price: 5.6 output_price: 16.8 supports_function_calling: true - name: qwen-max-longcontext input_price: 5.6 output_price: 16.8 - max_input_tokens: 28000 - max_output_tokens: 2000 + max_input_tokens: 30000 - name: qwen-vl-plus input_price: 1.12 output_price: 1.12 @@ -585,8 +563,6 @@ # - https://platform.moonshot.cn/docs/intro # - https://platform.moonshot.cn/docs/pricing # - https://platform.moonshot.cn/docs/api-reference - # notes: - # - unable to get max_output_tokens info models: - name: moonshot-v1-8k max_input_tokens: 8000 @@ -662,15 +638,23 @@ max_input_tokens: 32768 input_price: 2.8 output_price: 2.8 - - name: yi-medium + - name: yi-large-turbo max_input_tokens: 16384 - input_price: 0.35 - output_price: 0.35 + input_price: 1.68 + output_price: 1.68 + - name: yi-large-rag + max_input_tokens: 16384 + input_price: 3.5 + output_price: 3.5 - name: yi-vision max_input_tokens: 4096 input_price: 0.84 output_price: 0.84 supports_vision: true + - name: yi-medium + max_input_tokens: 16384 + input_price: 0.35 + output_price: 0.35 - name: yi-medium-200k max_input_tokens: 200000 input_price: 1.68 @@ -679,14 +663,6 @@ max_input_tokens: 16384 input_price: 0.14 output_price: 0.14 - - name: yi-large-rag - max_input_tokens: 16384 - input_price: 3.5 - output_price: 3.5 - - name: yi-large-turbo - max_input_tokens: 16384 - input_price: 1.68 - output_price: 1.68 - platform: anyscale # docs: diff --git a/src/serve.rs b/src/serve.rs index 8371f55..344a53f 100644 --- a/src/serve.rs +++ b/src/serve.rs @@ -218,7 +218,7 @@ impl Server { debug!("chat completions request: {req_body}"); let req_body = serde_json::from_value(req_body) - .map_err(|err| anyhow!("Invalid requst body, {err}"))?; + .map_err(|err| anyhow!("Invalid request body, {err}"))?; let ChatCompletionsReqBody { model, @@ -361,7 +361,7 @@ impl Server { debug!("embeddings request: {req_body}"); let req_body = serde_json::from_value(req_body) - .map_err(|err| anyhow!("Invalid requst body, {err}"))?; + .map_err(|err| anyhow!("Invalid request body, {err}"))?; let EmbeddingsReqBody { input, |
