summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorsigoden <sigoden@gmail.com>2025-01-22 20:51:10 +0800
committerGitHub <noreply@github.com>2025-01-22 20:51:10 +0800
commite0417e8d5bebe25476aaafa22a9ee23d9bd61457 (patch)
tree8edcd3aa9bcd1b012a3a429ad6240e6186caef36
parentdf4440a2a049d26c61a540d3254cd885345fbd6c (diff)
downloadaichat-e0417e8d5bebe25476aaafa22a9ee23d9bd61457.tar.gz
feat: add `--sync-models` cli option (#1114)
-rwxr-xr-xArgcfile.sh79
-rw-r--r--config.example.yaml2
-rw-r--r--models.yaml734
-rw-r--r--scripts/completions/aichat.bash2
-rw-r--r--scripts/completions/aichat.fish1
-rw-r--r--scripts/completions/aichat.nu2
-rw-r--r--scripts/completions/aichat.ps11
-rw-r--r--scripts/completions/aichat.zsh1
-rw-r--r--src/cli.rs3
-rw-r--r--src/client/common.rs13
-rw-r--r--src/client/macros.rs8
-rw-r--r--src/client/mod.rs2
-rw-r--r--src/client/model.rs23
-rw-r--r--src/client/openai_compatible.rs2
-rw-r--r--src/config/input.rs2
-rw-r--r--src/config/mod.rs118
-rw-r--r--src/main.rs6
-rw-r--r--src/utils/loader.rs2
-rw-r--r--src/utils/request.rs12
19 files changed, 533 insertions, 480 deletions
diff --git a/Argcfile.sh b/Argcfile.sh
index 43ebb59..c8450f7 100755
--- a/Argcfile.sh
+++ b/Argcfile.sh
@@ -4,7 +4,7 @@ set -e
# @meta dotenv
# @env DRY_RUN Dry run mode
-# @cmd Test first running
+# @cmd Test configuration initialization
# @env AICHAT_CONFIG_DIR=tmp/test-init-config
# @arg args~
test-init-config() {
@@ -17,10 +17,13 @@ test-init-config() {
cargo run -- "$@"
}
-# @cmd Test running with AICHAT_PLATFORM environment variable
-# @env AICHAT_PLATFORM!
+# @cmd Test running without configuration file
+# @env AICHAT_PROVIDER!
+# @env AICHAT_CONFIG_DIR=tmp/test-provider-env
# @arg args~
-test-platform-env() {
+test-no-config() {
+ mkdir -p "$AICHAT_CONFIG_DIR"
+ rm -rf "$AICHAT_CONFIG_DIR/config.yaml"
cargo run -- "$@"
}
@@ -80,27 +83,27 @@ test-server() {
# @cmd Chat with any LLM api
# @flag -S --no-stream
-# @arg platform_model![?`_choice_platform_model`]
+# @arg provider_model![?`_choice_provider_model`]
# @arg text~
chat() {
- if [[ "$argc_platform_model" == *':'* ]]; then
- model="${argc_platform_model##*:}"
- argc_platform="${argc_platform_model%:*}"
+ if [[ "$argc_provider_model" == *':'* ]]; then
+ model="${argc_provider_model##*:}"
+ argc_provider="${argc_provider_model%:*}"
else
- argc_platform="${argc_platform_model}"
+ argc_provider="${argc_provider_model}"
fi
- for platform_config in "${OPENAI_COMPATIBLE_PLATFORMS[@]}"; do
- if [[ "$argc_platform" == "${platform_config%%,*}" ]]; then
+ for provider_config in "${OPENAI_COMPATIBLE_PROVIDERS[@]}"; do
+ if [[ "$argc_provider" == "${provider_config%%,*}" ]]; then
_retrieve_api_base
break
fi
done
if [[ -n "$api_base" ]]; then
- env_prefix="$(echo "$argc_platform" | tr '[:lower:]' '[:upper:]')"
+ env_prefix="$(echo "$argc_provider" | tr '[:lower:]' '[:upper:]')"
api_key_env="${env_prefix}_API_KEY"
api_key="${!api_key_env}"
if [[ -z "$model" ]]; then
- model="$(echo "$platform_config" | cut -d, -f2)"
+ model="$(echo "$provider_config" | cut -d, -f2)"
fi
if [[ -z "$model" ]]; then
model_env="${env_prefix}_MODEL"
@@ -112,27 +115,27 @@ chat() {
--model "$model" \
"${argc_text[@]}"
else
- argc chat-$argc_platform "${argc_text[@]}"
+ argc chat-$argc_provider "${argc_text[@]}"
fi
}
# @cmd List models by openai-compatible api
# @flag --name-only Print model name only
-# @arg platform![`_choice_platform`]
+# @arg provider![`_choice_provider`]
models() {
- for platform_config in "${OPENAI_COMPATIBLE_PLATFORMS[@]}"; do
- if [[ "$argc_platform" == "${platform_config%%,*}" ]]; then
+ for provider_config in "${OPENAI_COMPATIBLE_PROVIDERS[@]}"; do
+ if [[ "$argc_provider" == "${provider_config%%,*}" ]]; then
_retrieve_api_base
break
fi
done
if [[ -n "$api_base" ]]; then
- env_prefix="$(echo "$argc_platform" | tr '[:lower:]' '[:upper:]')"
+ env_prefix="$(echo "$argc_provider" | tr '[:lower:]' '[:upper:]')"
api_key_env="${env_prefix}_API_KEY"
api_key="${!api_key_env}"
jq_args=()
if [[ -n "$argc_name_only" ]]; then
- case "$argc_platform" in
+ case "$argc_provider" in
cloudflare)
jq_args+=(-r '.result[].name')
;;
@@ -149,14 +152,14 @@ models() {
fi
_openai_compatible_models | jq "${jq_args[@]}"
else
- if ! cat "$0" | grep -q "^models-$argc_platform"; then
- _die "error: platform '$argc_platform' does not have a models api"
+ if ! cat "$0" | grep -q "^models-$argc_provider"; then
+ _die "error: provider '$argc_provider' does not have a models api"
fi
cli_args=()
if [[ -n "$argc_name_only" ]]; then
cli_args+=(--name-only)
fi
- argc models-$argc_platform "${cli_args[@]}"
+ argc models-$argc_provider "${cli_args[@]}"
fi
}
@@ -202,7 +205,7 @@ chat-azure-openai() {
# @cmd Chat with gemini api
# @env GEMINI_API_KEY!
-# @option -m --model=gemini-1.0-pro-latest $GEMINI_MODEL
+# @option -m --model=gemini-1.5-pro-latest $GEMINI_MODEL
# @flag -S --no-stream
# @arg text~
chat-gemini() {
@@ -246,7 +249,7 @@ chat-claude() {
# @cmd Chat with cohere api
# @env COHERE_API_KEY!
-# @option -m --model=command-r $COHERE_MODEL
+# @option -m --model=command-r-08-2024 $COHERE_MODEL
# @flag -S --no-stream
# @arg text~
chat-cohere() {
@@ -274,7 +277,7 @@ models-cohere() {
# @env require-tools gcloud
# @env VERTEXAI_PROJECT_ID!
# @env VERTEXAI_LOCATION!
-# @option -m --model=gemini-1.0-pro $VERTEXAI_GEMINI_MODEL
+# @option -m --model=gemini-1.5-flash-002 $VERTEXAI_GEMINI_MODEL
# @flag -S --no-stream
# @arg text~
chat-vertexai() {
@@ -307,7 +310,7 @@ chat-ernie() {
}
_argc_before() {
- OPENAI_COMPATIBLE_PLATFORMS=( \
+ OPENAI_COMPATIBLE_PROVIDERS=( \
openai,gpt-4o-mini,https://api.openai.com/v1 \
ai21,jamba-1.5-mini,https://api.ai21.com/studio/v1 \
cloudflare,@cf/meta/llama-3.1-8b-instruct,https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1 \
@@ -317,7 +320,7 @@ _argc_before() {
github,gpt-4o-mini,https://models.inference.ai.azure.com \
groq,llama-3.1-8b-instant,https://api.groq.com/openai/v1 \
hunyuan,hunyuan-large,https://api.hunyuan.cloud.tencent.com/v1 \
- lingyiwanwu,yi-large,https://api.lingyiwanwu.com/v1 \
+ lingyiwanwu,yi-lightning,https://api.lingyiwanwu.com/v1 \
minimax,MiniMax-Text-01,https://api.minimax.chat/v1 \
mistral,mistral-small-latest,https://api.mistral.ai/v1 \
moonshot,moonshot-v1-8k,https://api.moonshot.cn/v1 \
@@ -341,7 +344,7 @@ _openai_compatible_models() {
api_base="${api_base:-"$argc_api_base"}"
api_key="${api_key:-"$argc_api_key"}"
url="${api_base}/models"
- if [[ "$argc_platform" == "cloudflare" ]]; then
+ if [[ "$argc_provider" == "cloudflare" ]]; then
url="https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/models/search"
fi
@@ -351,12 +354,12 @@ _openai_compatible_models() {
}
_retrieve_api_base() {
- api_base="${platform_config##*,}"
+ api_base="${provider_config##*,}"
if [[ -z "$api_base" ]]; then
- key="$(echo $argc_platform | tr '[:lower:]' '[:upper:]')_API_BASE"
+ key="$(echo $argc_provider | tr '[:lower:]' '[:upper:]')_API_BASE"
api_base="${!key}"
if [[ -z "$api_base" ]]; then
- _die "error: miss api_base for $argc_platform; please set $key"
+ _die "error: miss api_base for $argc_provider; please set $key"
fi
fi
}
@@ -365,23 +368,23 @@ _choice_model() {
aichat --list-models
}
-_choice_platform_model() {
- _choice_platform
+_choice_provider_model() {
+ _choice_provider
_choice_model
}
-_choice_platform() {
+_choice_provider() {
_choice_client
- _choice_openai_compatible_platform
+ _choice_openai_compatible_provider
}
_choice_client() {
printf "%s\n" gemini claude cohere azure-openai vertexai bedrock ernie
}
-_choice_openai_compatible_platform() {
- for platform_config in "${OPENAI_COMPATIBLE_PLATFORMS[@]}"; do
- echo "${platform_config%%,*}"
+_choice_openai_compatible_provider() {
+ for provider_config in "${OPENAI_COMPATIBLE_PROVIDERS[@]}"; do
+ echo "${provider_config%%,*}"
done
}
diff --git a/config.example.yaml b/config.example.yaml
index ebf8853..e80c34e 100644
--- a/config.example.yaml
+++ b/config.example.yaml
@@ -82,6 +82,8 @@ right_prompt:
serve_addr: 127.0.0.1:8000 # Default serve listening address
user_agent: null # Set User-Agent HTTP header, use `auto` for aichat/<current-version>
save_shell_history: true # Whether to save shell execution command to the history file
+# Sync models changes from the specified URL, using a CDN link rather than a direct GitHub raw link due to the higher availability and reliability of the CDN.
+sync_models_url: https://cdn.jsdelivr.net/gh/sigoden/aichat/models.yaml # Or https://raw.githubusercontent.com/sigoden/aichat/refs/heads/main/models.yaml
# ---- clients ----
clients:
diff --git a/models.yaml b/models.yaml
index 60184de..6b505f1 100644
--- a/models.yaml
+++ b/models.yaml
@@ -1,11 +1,8 @@
-# Notes:
-# - do not submit pull requests to add new models; this list will be updated in batches with new releases.
-
# Links:
# - https://platform.openai.com/docs/models
# - https://openai.com/api/pricing/
# - https://platform.openai.com/docs/api-reference/chat
-- platform: openai
+- provider: openai
models:
- name: gpt-4o
max_input_tokens: 128000
@@ -50,7 +47,7 @@
supports_vision: true
supports_function_calling: true
- name: o1
- max_input_tokens: 128000
+ max_input_tokens: 200000
input_price: 15
output_price: 60
supports_vision: true
@@ -91,7 +88,7 @@
# - https://ai.google.dev/models/gemini
# - https://ai.google.dev/pricing
# - https://ai.google.dev/api/rest/v1beta/models/streamGenerateContent
-- platform: gemini
+- provider: gemini
models:
- name: gemini-1.5-pro-latest
max_input_tokens: 2097152
@@ -122,13 +119,13 @@
supports_vision: true
supports_function_calling: true
- name: gemini-2.0-flash-thinking-exp
- max_input_tokens: 32768
+ max_input_tokens: 32767
max_output_tokens: 8192
input_price: 0
output_price: 0
supports_vision: true
- name: gemini-exp-1206
- max_input_tokens: 32768
+ max_input_tokens: 2097152
max_output_tokens: 8192
input_price: 0
output_price: 0
@@ -144,7 +141,7 @@
# Links:
# - https://docs.anthropic.com/claude/docs/models-overview
# - https://docs.anthropic.com/claude/reference/messages-streaming
-- platform: claude
+- provider: claude
models:
- name: claude-3-5-sonnet-latest
max_input_tokens: 200000
@@ -207,7 +204,7 @@
# - https://docs.mistral.ai/getting-started/models/models_overview/
# - https://mistral.ai/technology/#pricing
# - https://docs.mistral.ai/api/
-- platform: mistral
+- provider: mistral
models:
- name: mistral-large-latest
max_input_tokens: 128000
@@ -250,7 +247,7 @@
# - https://docs.ai21.com/docs/jamba-15-models
# - https://www.ai21.com/pricing
# - https://docs.ai21.com/reference/jamba-15-api-ref
-- platform: ai21
+- provider: ai21
models:
- name: jamba-1.5-large
max_input_tokens: 256000
@@ -267,7 +264,7 @@
# - https://docs.cohere.com/docs/command-r-plus
# - https://cohere.com/pricing
# - https://docs.cohere.com/reference/chat
-- platform: cohere
+- provider: cohere
models:
- name: command-r-plus-08-2024
max_input_tokens: 128000
@@ -322,7 +319,7 @@
# Links:
# - https://docs.x.ai/docs/models
-- platform: xai
+- provider: xai
models:
- name: grok-2-latest
max_input_tokens: 131072
@@ -361,7 +358,7 @@
# - https://docs.perplexity.ai/guides/model-cards
# - https://docs.perplexity.ai/guides/pricing
# - https://docs.perplexity.ai/api-reference/chat-completions
-- platform: perplexity
+- provider: perplexity
models:
- name: llama-3.1-sonar-huge-128k-online
max_input_tokens: 127072
@@ -379,64 +376,57 @@
# Links:
# - https://console.groq.com/docs/models
# - https://console.groq.com/docs/api-reference#chat
-- platform: groq
+- provider: groq
models:
- name: llama-3.3-70b-versatile
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_function_calling: true
- name: llama-3.1-8b-instant
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_function_calling: true
- name: llama-3.2-90b-vision-preview
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_vision: true
- name: llama-3.2-11b-vision-preview
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_vision: true
- - name: gemma2-9b-it
- max_input_tokens: 8192
- input_price: 0
- output_price: 0
- supports_function_calling: true
# Links:
# - https://ollama.com/library
# - https://github.com/ollama/ollama/blob/main/docs/openai.md
-- platform: ollama
+- provider: ollama
models:
- name: llama3.1
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: llama3.2
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: llama3.2-vision
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_vision: true
- name: llama3.3
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: qwq
max_input_tokens: 32768
supports_function_calling: true
- name: qwen2.5
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: qwen2.5-coder
max_input_tokens: 32768
supports_function_calling: true
- name: phi4
max_input_tokens: 16384
- - name: gemma2
- max_input_tokens: 8192
- name: nomic-embed-text
type: embedding
max_tokens_per_chunk: 8192
@@ -448,7 +438,7 @@
# - https://cloud.google.com/vertex-ai/generative-ai/docs/model-garden/explore-models
# - https://cloud.google.com/vertex-ai/generative-ai/pricing
# - https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/gemini
-- platform: vertexai
+- provider: vertexai
models:
- name: gemini-1.5-pro-002
max_input_tokens: 2097152
@@ -470,7 +460,7 @@
supports_vision: true
supports_function_calling: true
- name: gemini-2.0-flash-thinking-exp-1219
- max_input_tokens: 32768
+ max_input_tokens: 32760
max_output_tokens: 8192
supports_vision: true
- name: claude-3-5-sonnet-v2@20241022
@@ -531,7 +521,7 @@
input_price: 0.3
output_price: 0.9
supports_function_calling: true
- - name: text-embedding-004
+ - name: text-embedding-005
type: embedding
max_input_tokens: 20000
input_price: 0.025
@@ -550,7 +540,7 @@
# - https://docs.aws.amazon.com/bedrock/latest/userguide/model-ids.html#model-ids-arns
# - https://aws.amazon.com/bedrock/pricing/
# - https://docs.aws.amazon.com/bedrock/latest/userguide/cross-region-inference-support.html
-- platform: bedrock
+- provider: bedrock
models:
- name: anthropic.claude-3-5-sonnet-20241022-v2:0
max_input_tokens: 200000
@@ -601,35 +591,35 @@
supports_vision: true
supports_function_calling: true
- name: us.meta.llama3-3-70b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.72
output_price: 0.72
supports_function_calling: true
- name: meta.llama3-1-405b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 4096
require_max_tokens: true
input_price: 2.4
output_price: 2.4
supports_function_calling: true
- name: meta.llama3-1-70b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.72
output_price: 0.72
supports_function_calling: true
- name: meta.llama3-1-8b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.22
output_price: 0.22
supports_function_calling: true
- name: us.meta.llama3-2-90b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.72
@@ -637,7 +627,7 @@
supports_function_calling: true
supports_vision: true
- name: us.meta.llama3-2-11b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.16
@@ -702,7 +692,7 @@
# Links:
# - https://developers.cloudflare.com/workers-ai/models/
# - https://developers.cloudflare.com/workers-ai/configuration/open-ai-compatibility/
-- platform: cloudflare
+- provider: cloudflare
models:
- name: '@cf/meta/llama-3.3-70b-instruct-fp8-fast'
max_input_tokens: 6144
@@ -738,7 +728,7 @@
# Links:
# - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/Nlks5zkzu
# - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/hlrk4akp7
-- platform: ernie
+- provider: ernie
models:
- name: ernie-4.0-turbo-8k-latest
max_input_tokens: 8192
@@ -781,10 +771,11 @@
max_input_tokens: 1024
input_price: 0.07
+
# Links:
# - https://help.aliyun.com/zh/model-studio/getting-started/models
# - https://help.aliyun.com/zh/model-studio/developer-reference/use-qwen-by-calling-api
-- platform: qianwen
+- provider: qianwen
models:
- name: qwen-max-latest
max_input_tokens: 30720
@@ -793,13 +784,13 @@
output_price: 8.4
supports_function_calling: true
- name: qwen-plus-latest
- max_input_tokens: 128000
+ max_input_tokens: 129024
max_output_tokens: 8192
input_price: 0.112
output_price: 0.28
supports_function_calling: true
- name: qwen-turbo-latest
- max_input_tokens: 129024
+ max_input_tokens: 1000000
max_output_tokens: 8192
input_price: 0.042
output_price: 0.084
@@ -831,12 +822,16 @@
output_price: 0.98
supports_function_calling: true
- name: qwen-vl-max-latest
- input_price: 2.8
- output_price: 2.8
+ max_input_tokens: 30720
+ max_output_tokens: 2048
+ input_price: 0.42
+ output_price: 1.26
supports_vision: true
- name: qwen-vl-plus-latest
- input_price: 1.12
- output_price: 1.12
+ max_input_tokens: 30000
+ max_output_tokens: 2048
+ input_price: 0.21
+ output_price: 0.63
supports_vision: true
- name: qwen2.5-72b-instruct
max_input_tokens: 129024
@@ -867,7 +862,7 @@
# - https://cloud.tencent.com/document/product/1729/104753
# - https://cloud.tencent.com/document/product/1729/97731
# - https://cloud.tencent.com/document/product/1729/111007
-- platform: hunyuan
+- provider: hunyuan
models:
- name: hunyuan-turbo-latest
max_input_tokens: 28000
@@ -878,10 +873,14 @@
- name: hunyuan-large
max_input_tokens: 28000
max_output_tokens: 4096
+ input_price: 0.56
+ output_price: 1.68
supports_function_calling: true
- name: hunyuan-large-longcontext
max_input_tokens: 128000
max_output_tokens: 6144
+ input_price: 0.84
+ output_price: 2.52
supports_function_calling: true
- name: hunyuan-standard
max_input_tokens: 30000
@@ -927,38 +926,37 @@
max_batch_size: 100
# Links:
-# - https://platform.moonshot.cn/docs/intro
-# - https://platform.moonshot.cn/docs/pricing/chat
-# - https://platform.moonshot.cn/docs/api/chat
-- platform: moonshot
+# - https://platform.moonshot.cn/docs/pricing/chat#%E8%AE%A1%E8%B4%B9%E5%9F%BA%E6%9C%AC%E6%A6%82%E5%BF%B5
+# - https://platform.moonshot.cn/docs/api/chat#%E5%85%AC%E5%BC%80%E7%9A%84%E6%9C%8D%E5%8A%A1%E5%9C%B0%E5%9D%80
+- provider: moonshot
models:
- name: moonshot-v1-8k
- max_input_tokens: 8000
+ max_input_tokens: 8192
input_price: 1.68
output_price: 1.68
supports_function_calling: true
- name: moonshot-v1-32k
- max_input_tokens: 32000
+ max_input_tokens: 32768
input_price: 3.36
output_price: 3.36
supports_function_calling: true
- name: moonshot-v1-128k
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 8.4
output_price: 8.4
supports_function_calling: true
- name: moonshot-v1-8k-vision-preview
- max_input_tokens: 8000
+ max_input_tokens: 8192
input_price: 1.68
output_price: 1.68
supports_vision: true
- name: moonshot-v1-32k-vision-preview
- max_input_tokens: 32000
+ max_input_tokens: 32768
input_price: 3.36
output_price: 3.36
supports_vision: true
- name: moonshot-v1-128k-vision-preview
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 8.4
output_price: 8.4
supports_vision: true
@@ -966,38 +964,46 @@
# Links:
# - https://api-docs.deepseek.com/quick_start/pricing
# - https://platform.deepseek.com/api-docs/api/create-chat-completion
-- platform: deepseek
+- provider: deepseek
models:
- name: deepseek-chat
- max_input_tokens: 65536
+ max_input_tokens: 64000
max_output_tokens: 8192
input_price: 0.14
output_price: 0.28
supports_function_calling: true
+ - name: deepseek-reasoner
+ max_input_tokens: 64000
+ max_output_tokens: 8192
+ input_price: 0.55
+ output_price: 2.19
# Links:
-# - https://open.bigmodel.cn/dev/howuse/model
# - https://open.bigmodel.cn/pricing
# - https://open.bigmodel.cn/dev/api#glm-4
-- platform: zhipuai
+- provider: zhipuai
models:
- name: glm-4-plus
max_input_tokens: 128000
+ max_output_tokens: 4096
input_price: 7
output_price: 7
supports_function_calling: true
- name: glm-4-alltools
max_input_tokens: 128000
+ max_output_tokens: 4096
input_price: 14
output_price: 14
supports_function_calling: true
- name: glm-4-long
max_input_tokens: 1000000
+ max_output_tokens: 4096
input_price: 0.14
output_price: 0.14
supports_function_calling: true
- name: glm-4-flash
max_input_tokens: 128000
+ max_output_tokens: 4096
input_price: 0
output_price: 0
supports_function_calling: true
@@ -1011,6 +1017,10 @@
input_price: 0
output_price: 0
supports_vision: true
+ - name: glm-zero-preview
+ max_input_tokens: 16384
+ input_price: 1.4
+ output_price: 1.4
- name: embedding-3
type: embedding
max_input_tokens: 8192
@@ -1021,7 +1031,7 @@
# Links:
# - https://platform.lingyiwanwu.com/docs#%E6%A8%A1%E5%9E%8B%E4%B8%8E%E8%AE%A1%E8%B4%B9
# - https://platform.lingyiwanwu.com/docs/api-reference#create-chat-completion
-- platform: lingyiwanwu
+- provider: lingyiwanwu
models:
- name: yi-lightning
max_input_tokens: 16384
@@ -1036,7 +1046,7 @@
# Links:
# - https://platform.minimaxi.com/document/Price
# - https://platform.minimaxi.com/document/ChatCompletion%20v2
-- platform: minimax
+- provider: minimax
models:
- name: minimax-text-01
max_input_tokens: 1000192
@@ -1052,273 +1062,8 @@
# supports_function_calling: true
# Links:
-# - https://github.com/marketplace/models
-- platform: github
- models:
- - name: gpt-4o
- max_input_tokens: 128000
- supports_function_calling: true
- - name: gpt-4o-mini
- max_input_tokens: 128000
- supports_function_calling: true
- - name: o1
- max_input_tokens: 128000
- supports_function_calling: true
- supports_vision: true
- no_stream: true
- no_system_message: true
- - name: o1-preview
- max_input_tokens: 128000
- no_stream: true
- no_system_message: true
- - name: o1-mini
- max_input_tokens: 128000
- no_stream: true
- no_system_message: true
- - name: text-embedding-3-large
- type: embedding
- max_tokens_per_chunk: 8191
- default_chunk_size: 2000
- max_batch_size: 100
- - name: text-embedding-3-small
- type: embedding
- max_tokens_per_chunk: 8191
- default_chunk_size: 2000
- max_batch_size: 100
- - name: llama-3.3-70b-instruct
- max_input_tokens: 128000
- - name: meta-llama-3.1-405b-instruct
- max_input_tokens: 128000
- - name: meta-llama-3.1-70b-instruct
- max_input_tokens: 128000
- - name: meta-llama-3.1-8b-instruct
- max_input_tokens: 128000
- - name: llama-3.2-90b-vision-instruct
- max_input_tokens: 8192
- supports_vision: true
- - name: llama-3.2-11b-vision-instruct
- max_input_tokens: 8192
- supports_vision: true
- - name: mistral-large-2411
- max_input_tokens: 128000
- supports_function_calling: true
- - name: codestral-2501
- max_input_tokens: 256000
- supports_function_calling: true
- - name: cohere-command-r-plus-08-2024
- max_input_tokens: 128000
- supports_function_calling: true
- - name: cohere-command-r-08-2024
- max_input_tokens: 128000
- supports_function_calling: true
- - name: cohere-embed-v3-english
- type: embedding
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 96
- - name: cohere-embed-v3-multilingual
- type: embedding
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 96
- - name: ai21-jamba-1.5-large
- max_input_tokens: 256000
- supports_function_calling: true
- - name: ai21-jamba-1.5-mini
- max_input_tokens: 256000
- supports_function_calling: true
- - name: phi-4
- max_input_tokens: 16384
- - name: phi-3.5-moe-instruct
- max_input_tokens: 128000
- - name: phi-3.5-mini-instruct
- max_input_tokens: 128000
- - name: phi-3.5-vision-instruct
- max_input_tokens: 128000
- supports_vision: true
-
-# Links:
-# - https://deepinfra.com/models
-- platform: deepinfra
- models:
- - name: meta-llama/Llama-3.3-70B-Instruct
- max_input_tokens: 128000
- input_price: 0.23
- output_price: 0.40
- - name: meta-llama/Meta-Llama-3.1-405B-Instruct
- max_input_tokens: 32000
- input_price: 0.8
- output_price: 0.8
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-70B-Instruct
- max_input_tokens: 128000
- input_price: 0.23
- output_price: 0.4
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-8B-Instruct
- max_input_tokens: 128000
- input_price: 0.03
- output_price: 0.05
- supports_function_calling: true
- - name: meta-llama/Llama-3.2-90B-Vision-Instruct
- max_input_tokens: 128000
- input_price: 0.35
- output_price: 0.4
- - name: meta-llama/Llama-3.2-11B-Vision-Instruct
- max_input_tokens: 128000
- input_price: 0.055
- output_price: 0.055
- - name: mistralai/Mistral-Nemo-Instruct-2407
- max_input_tokens: 128000
- input_price: 0.035
- output_price: 0.08
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.27
- output_price: 0.27
- - name: google/gemma-2-9b-it
- max_input_tokens: 8192
- input_price: 0.03
- output_price: 0.06
- - name: Qwen/Qwen2.5-72B-Instruct
- max_input_tokens: 32768
- input_price: 0.23
- output_price: 0.40
- supports_function_calling: true
- - name: Qwen/Qwen2.5-Coder-32B-Instruct
- max_input_tokens: 32768
- input_price: 0.07
- output_price: 0.16
- - name: Qwen/QVQ-72B-Preview
- max_input_tokens: 32768
- input_price: 0.25
- output_price: 0.50
- supports_vision: true
- - name: Qwen/QwQ-32B-Preview
- max_input_tokens: 32768
- input_price: 0.12
- output_price: 0.18
- - name: deepseek-ai/DeepSeek-V3
- max_input_tokens: 32768
- input_price: 0.85
- output_price: 0.9
- - name: microsoft/phi-4
- max_input_tokens: 16384
- input_price: 0.07
- output_price: 0.14
- - name: nvidia/Llama-3.1-Nemotron-70B-Instruct
- max_input_tokens: 128000
- input_price: 0.12
- output_price: 0.30
- supports_function_calling: true
- - name: BAAI/bge-large-en-v1.5
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: BAAI/bge-m3
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 8192
- default_chunk_size: 2000
- max_batch_size: 100
- - name: intfloat/e5-large-v2
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: intfloat/multilingual-e5-large
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: thenlper/gte-large
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
-
-# Links:
-# - https://fireworks.ai/models
-# - https://fireworks.ai/pricing
-- platform: fireworks
- models:
- - name: accounts/fireworks/models/llama-v3p3-70b-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/llama-v3p1-405b-instruct
- max_input_tokens: 131072
- input_price: 3
- output_price: 3
- supports_function_calling: true
- - name: accounts/fireworks/models/llama-v3p1-70b-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- supports_function_calling: true
- - name: accounts/fireworks/models/llama-v3p1-8b-instruct
- max_input_tokens: 131072
- input_price: 0.2
- output_price: 0.2
- - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- supports_vision: true
- - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct
- max_input_tokens: 131072
- input_price: 0.2
- output_price: 0.2
- supports_vision: true
- - name: accounts/fireworks/models/qwen2p5-72b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- supports_function_calling: true
- - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/qwen-qwq-32b-preview
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/qwen2-vl-72b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- supports_vision: true
- - name: accounts/fireworks/models/deepseek-v3
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: nomic-ai/nomic-embed-text-v1.5
- type: embedding
- input_price: 0.008
- max_tokens_per_chunk: 8192
- default_chunk_size: 1500
- max_batch_size: 100
- - name: WhereIsAI/UAE-Large-V1
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: thenlper/gte-large
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
-
-# Links:
# - https://openrouter.ai/models
-- platform: openrouter
+- provider: openrouter
models:
- name: openai/gpt-4o
max_input_tokens: 128000
@@ -1396,14 +1141,6 @@
output_price: 0.15
supports_vision: true
supports_function_calling: true
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.27
- output_price: 0.27
- - name: google/gemma-2-9b-it
- max_input_tokens: 4096
- input_price: 0.06
- output_price: 0.06
- name: anthropic/claude-3.5-sonnet
max_input_tokens: 200000
max_output_tokens: 8192
@@ -1449,7 +1186,7 @@
input_price: 0.12
output_price: 0.3
- name: meta-llama/llama-3.1-405b-instruct
- max_input_tokens: 131072
+ max_input_tokens: 32768
input_price: 0.8
output_price: 0.8
supports_function_calling: true
@@ -1528,10 +1265,14 @@
input_price: 0.0375
output_price: 0.15
- name: deepseek/deepseek-chat
- max_input_tokens: 32768
+ max_input_tokens: 64000
input_price: 0.14
output_price: 0.28
supports_function_calling: true
+ - name: deepseek/deepseek-r1
+ max_input_tokens: 163840
+ input_price: 0.55
+ output_price: 2.19
- name: perplexity/llama-3.1-sonar-huge-128k-online
max_input_tokens: 127072
input_price: 5
@@ -1544,12 +1285,8 @@
max_input_tokens: 127072
input_price: 0.2
output_price: 0.2
- - name: 01-ai/yi-large
- max_input_tokens: 32768
- input_price: 3
- output_price: 3
- name: microsoft/phi-4
- max_input_tokens: 16000
+ max_input_tokens: 16384
input_price: 0.07
output_price: 0.14
- name: microsoft/phi-3.5-mini-128k-instruct
@@ -1578,11 +1315,6 @@
input_price: 0.25
output_price: 0.5
supports_vision: true
- - name: nvidia/llama-3.1-nemotron-70b-instruct
- max_input_tokens: 131072
- input_price: 0.35
- output_price: 0.4
- supports_function_calling: true
- name: x-ai/grok-2-1212
max_input_tokens: 131072
input_price: 2
@@ -1626,10 +1358,262 @@
input_price: 0.2
output_price: 1.1
+
+# Links:
+# - https://github.com/marketplace/models
+- provider: github
+ models:
+ - name: gpt-4o
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: gpt-4o-mini
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: o1
+ max_input_tokens: 200000
+ supports_function_calling: true
+ supports_vision: true
+ no_stream: true
+ no_system_message: true
+ - name: o1-preview
+ max_input_tokens: 128000
+ no_stream: true
+ no_system_message: true
+ - name: o1-mini
+ max_input_tokens: 128000
+ no_stream: true
+ no_system_message: true
+ - name: text-embedding-3-large
+ type: embedding
+ max_tokens_per_chunk: 8191
+ default_chunk_size: 2000
+ max_batch_size: 100
+ - name: text-embedding-3-small
+ type: embedding
+ max_tokens_per_chunk: 8191
+ default_chunk_size: 2000
+ max_batch_size: 100
+ - name: llama-3.3-70b-instruct
+ max_input_tokens: 131072
+ - name: meta-llama-3.1-405b-instruct
+ max_input_tokens: 131072
+ - name: meta-llama-3.1-70b-instruct
+ max_input_tokens: 131072
+ - name: meta-llama-3.1-8b-instruct
+ max_input_tokens: 131072
+ - name: llama-3.2-90b-vision-instruct
+ max_input_tokens: 131072
+ supports_vision: true
+ - name: llama-3.2-11b-vision-instruct
+ max_input_tokens: 131072
+ supports_vision: true
+ - name: mistral-large-2411
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: codestral-2501
+ max_input_tokens: 256000
+ supports_function_calling: true
+ - name: cohere-command-r-plus-08-2024
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: cohere-command-r-08-2024
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: cohere-embed-v3-english
+ type: embedding
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 96
+ - name: cohere-embed-v3-multilingual
+ type: embedding
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 96
+ - name: ai21-jamba-1.5-large
+ max_input_tokens: 256000
+ supports_function_calling: true
+ - name: ai21-jamba-1.5-mini
+ max_input_tokens: 256000
+ supports_function_calling: true
+ - name: phi-4
+ max_input_tokens: 16384
+ - name: phi-3.5-moe-instruct
+ max_input_tokens: 128000
+ - name: phi-3.5-mini-instruct
+ max_input_tokens: 128000
+ - name: phi-3.5-vision-instruct
+ max_input_tokens: 128000
+ supports_vision: true
+
+# Links:
+# - https://deepinfra.com/models
+- provider: deepinfra
+ models:
+ - name: meta-llama/Llama-3.3-70B-Instruct
+ max_input_tokens: 131072
+ input_price: 0.23
+ output_price: 0.40
+ - name: meta-llama/Meta-Llama-3.1-405B-Instruct
+ max_input_tokens: 32768
+ input_price: 0.8
+ output_price: 0.8
+ supports_function_calling: true
+ - name: meta-llama/Meta-Llama-3.1-70B-Instruct
+ max_input_tokens: 131072
+ input_price: 0.23
+ output_price: 0.4
+ supports_function_calling: true
+ - name: meta-llama/Meta-Llama-3.1-8B-Instruct
+ max_input_tokens: 131072
+ input_price: 0.03
+ output_price: 0.05
+ supports_function_calling: true
+ - name: meta-llama/Llama-3.2-90B-Vision-Instruct
+ max_input_tokens: 131072
+ input_price: 0.35
+ output_price: 0.4
+ - name: meta-llama/Llama-3.2-11B-Vision-Instruct
+ max_input_tokens: 131072
+ input_price: 0.055
+ output_price: 0.055
+ - name: Qwen/Qwen2.5-72B-Instruct
+ max_input_tokens: 32768
+ input_price: 0.23
+ output_price: 0.40
+ supports_function_calling: true
+ - name: Qwen/Qwen2.5-Coder-32B-Instruct
+ max_input_tokens: 32768
+ input_price: 0.07
+ output_price: 0.16
+ - name: Qwen/QVQ-72B-Preview
+ max_input_tokens: 32768
+ input_price: 0.25
+ output_price: 0.50
+ supports_vision: true
+ - name: Qwen/QwQ-32B-Preview
+ max_input_tokens: 32768
+ input_price: 0.12
+ output_price: 0.18
+ - name: deepseek-ai/DeepSeek-V3
+ max_input_tokens: 32768
+ input_price: 0.85
+ output_price: 0.9
+ - name: microsoft/phi-4
+ max_input_tokens: 16384
+ input_price: 0.07
+ output_price: 0.14
+ - name: BAAI/bge-large-en-v1.5
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: BAAI/bge-m3
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 8192
+ default_chunk_size: 2000
+ max_batch_size: 100
+ - name: intfloat/e5-large-v2
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: intfloat/multilingual-e5-large
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: thenlper/gte-large
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+
+# Links:
+# - https://fireworks.ai/models
+# - https://fireworks.ai/pricing
+- provider: fireworks
+ models:
+ - name: accounts/fireworks/models/llama-v3p3-70b-instruct
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/llama-v3p1-405b-instruct
+ max_input_tokens: 131072
+ input_price: 3
+ output_price: 3
+ supports_function_calling: true
+ - name: accounts/fireworks/models/llama-v3p1-70b-instruct
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ supports_function_calling: true
+ - name: accounts/fireworks/models/llama-v3p1-8b-instruct
+ max_input_tokens: 131072
+ input_price: 0.2
+ output_price: 0.2
+ - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ supports_vision: true
+ - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct
+ max_input_tokens: 131072
+ input_price: 0.2
+ output_price: 0.2
+ supports_vision: true
+ - name: accounts/fireworks/models/qwen2p5-72b-instruct
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ supports_function_calling: true
+ - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/qwen-qwq-32b-preview
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/qwen2-vl-72b-instruct
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ supports_vision: true
+ - name: accounts/fireworks/models/deepseek-v3
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/deepseek-r1
+ max_input_tokens: 160000
+ input_price: 8
+ output_price: 8
+ - name: nomic-ai/nomic-embed-text-v1.5
+ type: embedding
+ input_price: 0.008
+ max_tokens_per_chunk: 8192
+ default_chunk_size: 1500
+ max_batch_size: 100
+ - name: WhereIsAI/UAE-Large-V1
+ type: embedding
+ input_price: 0.016
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: thenlper/gte-large
+ type: embedding
+ input_price: 0.016
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
# Links
# - https://cloud.siliconflow.cn/models
# - https://docs.siliconflow.cn/api-reference/chat-completions/chat-completions
-- platform: siliconflow
+- provider: siliconflow
models:
- name: meta-llama/Llama-3.3-70B-Instruct
max_input_tokens: 32768
@@ -1653,7 +1637,7 @@
output_price: 0.578
supports_function_calling: true
- name: Qwen/Qwen2.5-72B-Instruct-128K
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0.578
output_price: 0.578
supports_function_calling: true
@@ -1684,14 +1668,6 @@
max_input_tokens: 32768
input_price: 0.176
output_price: 0.176
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.176
- output_price: 0.176
- - name: google/gemma-2-9b-it
- max_input_tokens: 8192
- input_price: 0
- output_price: 0
- name: deepseek-ai/DeepSeek-V2.5
max_input_tokens: 32768
input_price: 0.186
@@ -1728,25 +1704,25 @@
# Links:
# - https://docs.together.ai/docs/serverless-models
# - https://www.together.ai/pricing
-- platform: together
+- provider: together
models:
- name: meta-llama/Llama-3.3-70B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 131072
input_price: 0.88
output_price: 0.88
supports_function_calling: true
- name: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 130815
input_price: 3.5
output_price: 3.5
supports_function_calling: true
- name: meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 131072
input_price: 0.88
output_price: 0.88
supports_function_calling: true
- name: meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 131072
input_price: 0.18
output_price: 0.18
supports_function_calling: true
@@ -1760,14 +1736,6 @@
input_price: 0.18
output_price: 0.18
supports_vision: true
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.8
- output_price: 0.8
- - name: google/gemma-2-9b-it
- max_input_tokens: 8192
- input_price: 0.3
- output_price: 0.3
- name: Qwen/Qwen2.5-72B-Instruct-Turbo
max_input_tokens: 32768
input_price: 1.2
@@ -1777,7 +1745,7 @@
input_price: 0.3
output_price: 0.3
- name: Qwen/Qwen2.5-Coder-32B-Instruct
- max_input_tokens: 16384
+ max_input_tokens: 32768
input_price: 0.8
output_price: 0.8
- name: Qwen/QwQ-32B-Preview
@@ -1793,6 +1761,10 @@
max_input_tokens: 131072
input_price: 1.25
output_price: 1.25
+ - name: deepseek-ai/DeepSeek-R1
+ max_input_tokens: 163840
+ input_price: 7
+ output_price: 7
- name: WhereIsAI/UAE-Large-V1
type: embedding
input_price: 0.016
@@ -1811,9 +1783,9 @@
input_price: 0.1
# Links:
-# - https://jina.ai/
+# - https://jina.ai/models
# - https://api.jina.ai/redoc
-- platform: jina
+- provider: jina
models:
- name: jina-embeddings-v3
type: embedding
@@ -1846,7 +1818,7 @@
# - https://docs.voyageai.com/docs/embeddings
# - https://docs.voyageai.com/docs/pricing
# - https://docs.voyageai.com/reference/
-- platform: voyageai
+- provider: voyageai
models:
- name: voyage-3-large
type: embedding
diff --git a/scripts/completions/aichat.bash b/scripts/completions/aichat.bash
index 4708077..9ea4f9a 100644
--- a/scripts/completions/aichat.bash
+++ b/scripts/completions/aichat.bash
@@ -17,7 +17,7 @@ _aichat() {
case "${cmd}" in
aichat)
- opts="-m -r -s -a -e -c -f -S -h -V --model --prompt --role --session --empty-session --save-session --agent --agent-variable --rag --rebuild-rag --macro --serve --execute --code --file --no-stream --dry-run --info --list-models --list-roles --list-sessions --list-agents --list-rags --list-macros --help --version"
+ opts="-m -r -s -a -e -c -f -S -h -V --model --prompt --role --session --empty-session --save-session --agent --agent-variable --rag --rebuild-rag --macro --serve --execute --code --file --no-stream --dry-run --info --sync-models --list-models --list-roles --list-sessions --list-agents --list-rags --list-macros --help --version"
if [[ ${cur} == -* || ${cword} -eq 1 ]] ; then
COMPREPLY=( $(compgen -W "${opts}" -- "${cur}") )
return 0
diff --git a/scripts/completions/aichat.fish b/scripts/completions/aichat.fish
index 3618a5d..d3c336c 100644
--- a/scripts/completions/aichat.fish
+++ b/scripts/completions/aichat.fish
@@ -16,6 +16,7 @@ complete -c aichat -s f -l file -d 'Include files, directories, or URLs' -r -F
complete -c aichat -s S -l no-stream -d 'Turn off stream mode'
complete -c aichat -l dry-run -d 'Display the message without sending it'
complete -c aichat -l info -d 'Display information'
+complete -c aichat -l sync-models -d 'Sync models updates'
complete -c aichat -l list-models -d 'List all available chat models'
complete -c aichat -l list-roles -d 'List all roles'
complete -c aichat -l list-sessions -d 'List all sessions'
diff --git a/scripts/completions/aichat.nu b/scripts/completions/aichat.nu
index bbea5e2..365f0cc 100644
--- a/scripts/completions/aichat.nu
+++ b/scripts/completions/aichat.nu
@@ -40,7 +40,6 @@ module completions {
| parse "{value}"
}
- # All-in-one chat and copilot CLI that integrates 10+ AI platforms
export extern aichat [
--model(-m): string@"nu-complete aichat model" # Select a LLM model
--prompt # Use the system prompt
@@ -60,6 +59,7 @@ module completions {
--no-stream(-S) # Turn off stream mode
--dry-run # Display the message without sending it
--info # Display information
+ --sync-models # Sync models updates
--list-models # List all available chat models
--list-roles # List all roles
--list-sessions # List all sessions
diff --git a/scripts/completions/aichat.ps1 b/scripts/completions/aichat.ps1
index dc4ef64..c8c3187 100644
--- a/scripts/completions/aichat.ps1
+++ b/scripts/completions/aichat.ps1
@@ -46,6 +46,7 @@ Register-ArgumentCompleter -Native -CommandName 'aichat' -ScriptBlock {
[CompletionResult]::new('--no-stream', '--no-stream', [CompletionResultType]::ParameterName, 'Turn off stream mode')
[CompletionResult]::new('--dry-run', '--dry-run', [CompletionResultType]::ParameterName, 'Display the message without sending it')
[CompletionResult]::new('--info', '--info', [CompletionResultType]::ParameterName, 'Display information')
+ [CompletionResult]::new('--sync-models', '--sync-models', [CompletionResultType]::ParameterName, 'Sync models updates')
[CompletionResult]::new('--list-models', '--list-models', [CompletionResultType]::ParameterName, 'List all available chat models')
[CompletionResult]::new('--list-roles', '--list-roles', [CompletionResultType]::ParameterName, 'List all roles')
[CompletionResult]::new('--list-sessions', '--list-sessions', [CompletionResultType]::ParameterName, 'List all sessions')
diff --git a/scripts/completions/aichat.zsh b/scripts/completions/aichat.zsh
index 15cbdce..1349081 100644
--- a/scripts/completions/aichat.zsh
+++ b/scripts/completions/aichat.zsh
@@ -41,6 +41,7 @@ _aichat() {
'--no-stream[Turn off stream mode]' \
'--dry-run[Display the message without sending it]' \
'--info[Display information]' \
+'--sync-models[Sync models updates]' \
'--list-models[List all available chat models]' \
'--list-roles[List all roles]' \
'--list-sessions[List all sessions]' \
diff --git a/src/cli.rs b/src/cli.rs
index de88776..3204c58 100644
--- a/src/cli.rs
+++ b/src/cli.rs
@@ -60,6 +60,9 @@ pub struct Cli {
/// Display information
#[clap(long)]
pub info: bool,
+ /// Sync models updates
+ #[clap(long)]
+ pub sync_models: bool,
/// List all available chat models
#[clap(long)]
pub list_models: bool,
diff --git a/src/client/common.rs b/src/client/common.rs
index b4d01ce..80f585d 100644
--- a/src/client/common.rs
+++ b/src/client/common.rs
@@ -1,7 +1,7 @@
use super::*;
use crate::{
- config::{GlobalConfig, Input},
+ config::{Config, GlobalConfig, Input},
function::{eval_tool_calls, FunctionDeclaration, ToolCall, ToolResult},
render::render_stream,
utils::*,
@@ -20,7 +20,9 @@ use tokio::sync::mpsc::unbounded_channel;
const MODELS_YAML: &str = include_str!("../../models.yaml");
lazy_static::lazy_static! {
- pub static ref ALL_PREDEFINED_MODELS: Vec<PredefinedModels> = serde_yaml::from_str(MODELS_YAML).unwrap();
+ pub static ref ALL_PROVIDER_MODELS: Vec<ProviderModels> = {
+ Config::loal_models_override().ok().unwrap_or_else(|| serde_yaml::from_str(MODELS_YAML).unwrap())
+ };
static ref ESCAPE_SLASH_RE: Regex = Regex::new(r"(?<!\\)/").unwrap();
}
@@ -338,14 +340,15 @@ pub fn create_config(prompts: &[PromptAction], client: &str) -> Result<(String,
}
pub fn create_openai_compatible_client_config(client: &str) -> Result<Option<(String, Value)>> {
- let api_base = super::OPENAI_COMPATIBLE_PLATFORMS
+ let api_base = super::OPENAI_COMPATIBLE_PROVIDERS
.into_iter()
.find(|(name, _)| client == *name)
.map(|(_, api_base)| api_base)
.unwrap_or("http(s)://{API_ADDR}/v1");
let name = if client == OpenAICompatibleClient::NAME {
- prompt_input_string("Provider Name", true, None)?
+ let value = prompt_input_string("Provider Name", true, None)?;
+ value.replace(' ', "-")
} else {
client.to_string()
};
@@ -548,7 +551,7 @@ fn set_client_config(list: &[PromptAction], client_config: &mut Value, client: &
}
fn set_client_models_config(client_config: &mut Value, client: &str) -> Result<()> {
- if ALL_PREDEFINED_MODELS.iter().any(|v| v.platform == client) {
+ if ALL_PROVIDER_MODELS.iter().any(|v| v.provider == client) {
return Ok(());
}
diff --git a/src/client/macros.rs b/src/client/macros.rs
index a76e62b..97171db 100644
--- a/src/client/macros.rs
+++ b/src/client/macros.rs
@@ -52,10 +52,10 @@ macro_rules! register_client {
pub fn list_models(local_config: &$config) -> Vec<Model> {
let client_name = Self::name(local_config);
if local_config.models.is_empty() {
- if let Some(models) = $crate::client::ALL_PREDEFINED_MODELS.iter().find(|v| {
- v.platform == $name ||
+ if let Some(models) = $crate::client::ALL_PROVIDER_MODELS.iter().find(|v| {
+ v.provider == $name ||
($name == OpenAICompatibleClient::NAME
- && local_config.name.as_ref().map(|name| name.starts_with(&v.platform)).unwrap_or_default())
+ && local_config.name.as_ref().map(|name| name.starts_with(&v.provider)).unwrap_or_default())
}) {
return Model::from_config(client_name, &models.models);
}
@@ -83,7 +83,7 @@ macro_rules! register_client {
pub fn list_client_types() -> Vec<&'static str> {
let mut client_types: Vec<_> = vec![$($client::NAME,)+];
- client_types.extend($crate::client::OPENAI_COMPATIBLE_PLATFORMS.iter().map(|(name, _)| *name));
+ client_types.extend($crate::client::OPENAI_COMPATIBLE_PROVIDERS.iter().map(|(name, _)| *name));
client_types
}
diff --git a/src/client/mod.rs b/src/client/mod.rs
index 3d8d4da..bf11107 100644
--- a/src/client/mod.rs
+++ b/src/client/mod.rs
@@ -34,7 +34,7 @@ register_client!(
(ernie, "ernie", ErnieConfig, ErnieClient),
);
-pub const OPENAI_COMPATIBLE_PLATFORMS: [(&str, &str); 22] = [
+pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [
("ai21", "https://api.ai21.com/studio/v1"),
(
"cloudflare",
diff --git a/src/client/model.rs b/src/client/model.rs
index 4b0457f..b562705 100644
--- a/src/client/model.rs
+++ b/src/client/model.rs
@@ -277,26 +277,33 @@ pub struct ModelData {
pub name: String,
#[serde(default = "default_model_type", rename = "type")]
pub model_type: String,
+ #[serde(skip_serializing_if = "Option::is_none")]
pub max_input_tokens: Option<usize>,
+ #[serde(skip_serializing_if = "Option::is_none")]
pub input_price: Option<f64>,
+ #[serde(skip_serializing_if = "Option::is_none")]
pub output_price: Option<f64>,
// chat-only properties
+ #[serde(skip_serializing_if = "Option::is_none")]
pub max_output_tokens: Option<isize>,
- #[serde(default)]
+ #[serde(default, skip_serializing_if = "std::ops::Not::not")]
pub require_max_tokens: bool,
- #[serde(default)]
+ #[serde(default, skip_serializing_if = "std::ops::Not::not")]
pub supports_vision: bool,
- #[serde(default)]
+ #[serde(default, skip_serializing_if = "std::ops::Not::not")]
pub supports_function_calling: bool,
- #[serde(default)]
+ #[serde(default, skip_serializing_if = "std::ops::Not::not")]
no_stream: bool,
- #[serde(default)]
+ #[serde(default, skip_serializing_if = "std::ops::Not::not")]
no_system_message: bool,
// embedding-only properties
+ #[serde(skip_serializing_if = "Option::is_none")]
pub max_tokens_per_chunk: Option<usize>,
+ #[serde(skip_serializing_if = "Option::is_none")]
pub default_chunk_size: Option<usize>,
+ #[serde(skip_serializing_if = "Option::is_none")]
pub max_batch_size: Option<usize>,
}
@@ -310,9 +317,9 @@ impl ModelData {
}
}
-#[derive(Debug, Clone, Deserialize)]
-pub struct PredefinedModels {
- pub platform: String,
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct ProviderModels {
+ pub provider: String,
pub models: Vec<ModelData>,
}
diff --git a/src/client/openai_compatible.rs b/src/client/openai_compatible.rs
index 18acafb..ce1eea3 100644
--- a/src/client/openai_compatible.rs
+++ b/src/client/openai_compatible.rs
@@ -96,7 +96,7 @@ fn get_api_base_ext(self_: &OpenAICompatibleClient) -> Result<String> {
let api_base = match self_.get_api_base() {
Ok(v) => v,
Err(err) => {
- match OPENAI_COMPATIBLE_PLATFORMS
+ match OPENAI_COMPATIBLE_PROVIDERS
.into_iter()
.find_map(|(name, api_base)| {
if name == self_.model.client_name() {
diff --git a/src/config/input.rs b/src/config/input.rs
index 58c0065..e468f19 100644
--- a/src/config/input.rs
+++ b/src/config/input.rs
@@ -442,7 +442,7 @@ async fn load_documents(
}
for file_url in remote_urls {
- let (contents, extension) = fetch(&loaders, &file_url, true)
+ let (contents, extension) = fetch_with_loaders(&loaders, &file_url, true)
.await
.with_context(|| format!("Failed to load url '{file_url}'"))?;
if extension == MEDIA_URL_EXTENSION {
diff --git a/src/config/mod.rs b/src/config/mod.rs
index 8813930..341470f 100644
--- a/src/config/mod.rs
+++ b/src/config/mod.rs
@@ -12,7 +12,7 @@ use self::session::Session;
use crate::client::{
create_client_config, list_client_types, list_models, ClientConfig, MessageContentToolCalls,
- Model, ModelType, OPENAI_COMPATIBLE_PLATFORMS,
+ Model, ModelType, ProviderModels, OPENAI_COMPATIBLE_PROVIDERS,
};
use crate::function::{FunctionDeclaration, Functions, ToolResult};
use crate::rag::Rag;
@@ -24,7 +24,7 @@ use anyhow::{anyhow, bail, Context, Result};
use indexmap::IndexMap;
use inquire::{list_option::ListOption, validator::Validation, Confirm, MultiSelect, Select, Text};
use parking_lot::RwLock;
-use serde::Deserialize;
+use serde::{Deserialize, Serialize};
use serde_json::json;
use simplelog::LevelFilter;
use std::collections::{HashMap, HashSet};
@@ -64,6 +64,8 @@ const CLIENTS_FIELD: &str = "clients";
const SERVE_ADDR: &str = "127.0.0.1:8000";
+const SYNC_MODELS_URL: &str = "https://cdn.jsdelivr.net/gh/sigoden/aichat/models.yaml";
+
const SUMMARIZE_PROMPT: &str =
"Summarize the discussion briefly in 200 words or less to use as a prompt for future context.";
const SUMMARY_PROMPT: &str = "This is a summary of the chat history as a recap: ";
@@ -140,6 +142,7 @@ pub struct Config {
pub serve_addr: Option<String>,
pub user_agent: Option<String>,
pub save_shell_history: bool,
+ pub sync_models_url: Option<String>,
pub clients: Vec<ClientConfig>,
@@ -214,6 +217,7 @@ impl Default for Config {
serve_addr: None,
user_agent: None,
save_shell_history: true,
+ sync_models_url: None,
clients: vec![],
@@ -240,9 +244,12 @@ impl Config {
pub fn init(working_mode: WorkingMode, info_flag: bool) -> Result<Self> {
let config_path = Self::config_file();
let mut config = if !config_path.exists() {
- match env::var(get_env_name("platform")) {
- Ok(v) => Self::load_dynamic(&v)?,
- Err(_) => {
+ match env::var(get_env_name("provider"))
+ .ok()
+ .or_else(|| env::var(get_env_name("platform")).ok())
+ {
+ Some(v) => Self::load_dynamic(&v)?,
+ None => {
if *IS_STDOUT_TERMINAL {
create_config_file(&config_path)?;
}
@@ -417,6 +424,10 @@ impl Config {
}
}
+ pub fn models_override_file() -> PathBuf {
+ Self::local_path("models-override.json")
+ }
+
pub fn state(&self) -> StateFlags {
let mut flags = StateFlags::empty();
if let Some(session) = &self.session {
@@ -1362,23 +1373,12 @@ impl Config {
let temp_file = temp_file(&format!("-rag-{}", rag.name()), ".txt");
tokio::fs::write(&temp_file, &document_paths.join("\n"))
.await
- .with_context(|| {
- format!(
- "Failed to write current document paths to '{}'",
- temp_file.display()
- )
- })?;
+ .with_context(|| format!("Failed to write to '{}'", temp_file.display()))?;
let editor = config.read().editor()?;
edit_file(&editor, &temp_file)?;
- let new_document_paths =
- tokio::fs::read_to_string(&temp_file)
- .await
- .with_context(|| {
- format!(
- "Failed to read new document paths from '{}'",
- temp_file.display()
- )
- })?;
+ let new_document_paths = tokio::fs::read_to_string(&temp_file)
+ .await
+ .with_context(|| format!("Failed to read '{}'", temp_file.display()))?;
let new_document_paths = new_document_paths
.split('\n')
.filter_map(|v| {
@@ -1535,12 +1535,7 @@ impl Config {
&agent_config_path,
"# see https://github.com/sigoden/aichat/blob/main/config.agent.example.yaml\n",
)
- .with_context(|| {
- format!(
- "Failed to write to agent config file at '{}'",
- agent_config_path.display()
- )
- })?;
+ .with_context(|| format!("Failed to write to '{}'", agent_config_path.display()))?;
}
let editor = self.editor()?;
edit_file(&editor, &agent_config_path)?;
@@ -1860,6 +1855,50 @@ impl Config {
.collect()
}
+ pub fn sync_models_url(&self) -> String {
+ self.sync_models_url
+ .clone()
+ .unwrap_or_else(|| SYNC_MODELS_URL.into())
+ }
+
+ pub async fn sync_models(url: &str, abort_signal: AbortSignal) -> Result<()> {
+ let content = abortable_run_with_spinner(fetch(url), "Fetching models.yaml", abort_signal)
+ .await
+ .with_context(|| format!("Failed to fetch '{url}'"))?;
+ println!("✓ Fetched '{url}'");
+ let list = serde_yaml::from_str::<Vec<ProviderModels>>(&content)
+ .with_context(|| "Failed to parse models.yaml")?;
+ let models_override = ModelsOverride {
+ version: env!("CARGO_PKG_VERSION").to_string(),
+ list,
+ };
+ let models_override_data =
+ serde_json::to_string_pretty(&models_override).with_context(|| "Failed to serde {}")?;
+
+ let model_override_path = Self::models_override_file();
+ ensure_parent_exists(&model_override_path)?;
+ std::fs::write(&model_override_path, models_override_data)
+ .with_context(|| format!("Failed to write to '{}'", model_override_path.display()))?;
+ println!("✓ Updated '{}'", model_override_path.display());
+ Ok(())
+ }
+
+ pub fn loal_models_override() -> Result<Vec<ProviderModels>> {
+ let model_override_path = Self::models_override_file();
+ let err = || {
+ format!(
+ "Failed to load models at '{}'",
+ model_override_path.display()
+ )
+ };
+ let content = read_to_string(&model_override_path).with_context(err)?;
+ let models_override: ModelsOverride = serde_json::from_str(&content).with_context(err)?;
+ if models_override.version != env!("CARGO_PKG_VERSION") {
+ bail!("Incompatible version")
+ }
+ Ok(models_override.list)
+ }
+
pub fn render_options(&self) -> Result<RenderOptions> {
let theme = if self.highlight {
let theme_mode = if self.light_theme { "light" } else { "dark" };
@@ -2175,17 +2214,17 @@ impl Config {
}
fn load_dynamic(model_id: &str) -> Result<Self> {
- let platform = match model_id.split_once(':') {
+ let provider = match model_id.split_once(':') {
Some((v, _)) => v,
_ => model_id,
};
- let is_openai_compatible = OPENAI_COMPATIBLE_PLATFORMS
+ let is_openai_compatible = OPENAI_COMPATIBLE_PROVIDERS
.into_iter()
- .any(|(name, _)| platform == name);
+ .any(|(name, _)| provider == name);
let client = if is_openai_compatible {
- json!({ "type": "openai-compatible", "name": platform })
+ json!({ "type": "openai-compatible", "name": provider })
} else {
- json!({ "type": platform })
+ json!({ "type": provider })
};
let config = json!({
"model": model_id.to_string(),
@@ -2323,6 +2362,9 @@ impl Config {
if let Some(Some(v)) = read_env_bool(&get_env_name("save_shell_history")) {
self.save_shell_history = v;
}
+ if let Some(v) = read_env_value::<String>(&get_env_name("sync_models_url")) {
+ self.sync_models_url = v;
+ }
}
fn load_functions(&mut self) -> Result<()> {
@@ -2502,6 +2544,12 @@ pub struct MacroVariable {
pub default: Option<String>,
}
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct ModelsOverride {
+ pub version: String,
+ pub list: Vec<ProviderModels>,
+}
+
#[derive(Debug, Clone)]
pub struct LastMessage {
pub input: Input,
@@ -2581,12 +2629,8 @@ fn create_config_file(config_path: &Path) -> Result<()> {
);
ensure_parent_exists(config_path)?;
- std::fs::write(config_path, config_data).with_context(|| {
- format!(
- "Failed to write to config file at '{}'",
- config_path.display()
- )
- })?;
+ std::fs::write(config_path, config_data)
+ .with_context(|| format!("Failed to write to '{}'", config_path.display()))?;
#[cfg(unix)]
{
use std::os::unix::prelude::PermissionsExt;
diff --git a/src/main.rs b/src/main.rs
index e30efde..5251bad 100644
--- a/src/main.rs
+++ b/src/main.rs
@@ -46,6 +46,7 @@ async fn main() -> Result<()> {
WorkingMode::Cmd
};
let info_flag = cli.info
+ || cli.sync_models
|| cli.list_models
|| cli.list_roles
|| cli.list_agents
@@ -64,6 +65,11 @@ async fn main() -> Result<()> {
async fn run(config: GlobalConfig, cli: Cli, text: Option<String>) -> Result<()> {
let abort_signal = create_abort_signal();
+ if cli.sync_models {
+ let url = config.read().sync_models_url();
+ return Config::sync_models(&url, abort_signal.clone()).await;
+ }
+
if cli.list_models {
for model in list_models(&config.read(), ModelType::Chat) {
println!("{}", model.id());
diff --git a/src/utils/loader.rs b/src/utils/loader.rs
index 519563c..2ac671d 100644
--- a/src/utils/loader.rs
+++ b/src/utils/loader.rs
@@ -61,7 +61,7 @@ pub async fn load_file(loaders: &HashMap<String, String>, path: &str) -> Result<
}
pub async fn load_url(loaders: &HashMap<String, String>, path: &str) -> Result<LoadedDocument> {
- let (contents, extension) = fetch(loaders, path, false).await?;
+ let (contents, extension) = fetch_with_loaders(loaders, path, false).await?;
let mut metadata: DocumentMetadata = Default::default();
metadata.insert(EXTENSION_METADATA.into(), extension);
Ok(LoadedDocument::new(path.into(), contents, metadata))
diff --git a/src/utils/request.rs b/src/utils/request.rs
index 9f2804b..a479e0e 100644
--- a/src/utils/request.rs
+++ b/src/utils/request.rs
@@ -50,7 +50,17 @@ lazy_static::lazy_static! {
static ref GITHUB_REPO_RE: Regex = Regex::new(r"^https://github\.com/([^/]+)/([^/]+)/tree/([^/]+)").unwrap();
}
-pub async fn fetch(
+pub async fn fetch(url: &str) -> Result<String> {
+ let client = match *CLIENT {
+ Ok(ref client) => client,
+ Err(ref err) => bail!("{err}"),
+ };
+ let res = client.get(url).send().await?;
+ let output = res.text().await?;
+ Ok(output)
+}
+
+pub async fn fetch_with_loaders(
loaders: &HashMap<String, String>,
path: &str,
allow_media: bool,