summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorsigoden <sigoden@gmail.com>2024-09-01 10:05:53 +0800
committerGitHub <noreply@github.com>2024-09-01 10:05:53 +0800
commitef0810434cf60eeb33e8f61712c4023efe71876d (patch)
tree77450abe9dbe9bebd83172003f4710dc3883bcb4
parent573e0d58b44cd0686c9e7405723e5cc3a5c5126f (diff)
downloadaichat-ef0810434cf60eeb33e8f61712c4023efe71876d.tar.gz
refactor: minor improvement (#818)
-rwxr-xr-xArgcfile.sh50
-rw-r--r--config.example.yaml13
-rw-r--r--models.yaml359
-rw-r--r--src/client/model.rs7
-rw-r--r--src/client/openai.rs2
5 files changed, 186 insertions, 245 deletions
diff --git a/Argcfile.sh b/Argcfile.sh
index e1af57f..138f422 100755
--- a/Argcfile.sh
+++ b/Argcfile.sh
@@ -89,7 +89,9 @@ OPENAI_COMPATIBLE_PLATFORMS=( \
moonshot,moonshot-v1-8k,https://api.moonshot.cn/v1 \
openrouter,openai/gpt-4o-mini,https://openrouter.ai/api/v1 \
octoai,meta-llama-3.1-8b-instruct,https://text.octoai.run/v1 \
+ ollama,llama3.1:latest,http://localhost:11434/v1 \
perplexity,llama-3.1-8b-instruct,https://api.perplexity.ai \
+ qianwen,qwen-turbo,https://dashscope.aliyuncs.com/compatible-mode/v1 \
together,meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo,https://api.together.xyz/v1 \
zhipuai,glm-4-0520,https://open.bigmodel.cn/api/paas/v4 \
lingyiwanwu,yi-large,https://api.lingyiwanwu.com/v1 \
@@ -248,18 +250,6 @@ models-cohere() {
}
-# @cmd Chat with ollama api
-# @env OLLAMA_BASE_URL=http://127.0.0.1:11434
-# @option -m --model=llama3.1:latest $OLLAMA_MODEL
-# @flag -S --no-stream
-# @arg text~
-chat-ollama() {
- _wrapper curl -i $OLLAMA_BASE_URL/api/chat \
--X POST \
--H 'Content-Type: application/json' \
--d "$(_build_body ollama "$@")"
-}
-
# @cmd Chat with vertexai api
# @env require-tools gcloud
# @env VERTEXAI_PROJECT_ID!
@@ -347,26 +337,6 @@ chat-ernie() {
}
-# @cmd Chat with qianwen api
-# @env QIANWEN_API_KEY!
-# @option -m --model=qwen-turbo $QIANWEN_MODEL
-# @flag -S --no-stream
-# @arg text~
-chat-qianwen() {
- stream_args="-H X-DashScope-SSE:enable"
- parameters_args='{"incremental_output": true}'
- if [[ -n "$argc_no_stream" ]]; then
- stream_args=""
- parameters_args='{}'
- fi
- url=https://dashscope.aliyuncs.com/api/v1/services/aigc/text-generation/generation
- _wrapper curl -i "$url" \
--X POST \
--H "Authorization: Bearer $QIANWEN_API_KEY" \
--H 'Content-Type: application/json' $stream_args \
--d "$(_build_body qianwen "$@")"
-}
-
_argc_before() {
stream="true"
if [[ -n "$argc_no_stream" ]]; then
@@ -420,7 +390,7 @@ _build_body() {
else
shift
case "$kind" in
- openai|ollama)
+ openai)
echo '{
"model": "'$argc_model'",
"messages": [
@@ -485,20 +455,6 @@ _build_body() {
}
}'
;;
- qianwen)
- echo '{
- "model": "'$argc_model'",
- "parameters": '"$parameters_args"',
- "input":{
- "messages": [
- {
- "role": "user",
- "content": "'"$*"'"
- }
- ]
- }
-}'
- ;;
*)
_die "Unsupported build body for $kind"
;;
diff --git a/config.example.yaml b/config.example.yaml
index db61817..a93a0ce 100644
--- a/config.example.yaml
+++ b/config.example.yaml
@@ -185,19 +185,6 @@ clients:
- type: openai-compatible
name: ollama
api_base: http://localhost:11434/v1
- models:
- - name: llama3.1
- max_input_tokens: 128000
- supports_function_calling: true
- - name: gemma2
- max_input_tokens: 8192
- - name: mistral-nemo
- max_input_tokens: 128000
- supports_function_calling: true
- - name: nomic-embed-text
- type: embedding
- default_chunk_size: 1000
- max_batch_size: 50
# See https://learn.microsoft.com/en-us/azure/ai-services/openai/chatgpt-quickstart
- type: azure-openai
diff --git a/models.yaml b/models.yaml
index ca27645..106adbd 100644
--- a/models.yaml
+++ b/models.yaml
@@ -11,16 +11,16 @@
models:
- name: gpt-4o
max_input_tokens: 128000
- max_output_tokens: 16384
- input_price: 2.5
- output_price: 10
+ max_output_tokens: 4096
+ input_price: 5
+ output_price: 15
supports_vision: true
supports_function_calling: true
- - name: gpt-4o-mini
+ - name: gpt-4o-2024-08-06
max_input_tokens: 128000
max_output_tokens: 16384
- input_price: 0.15
- output_price: 0.6
+ input_price: 2.5
+ output_price: 10
supports_vision: true
supports_function_calling: true
- name: chatgpt-4o-latest
@@ -30,6 +30,13 @@
output_price: 15
supports_vision: true
supports_function_calling: true
+ - name: gpt-4o-mini
+ max_input_tokens: 128000
+ max_output_tokens: 16384
+ input_price: 0.15
+ output_price: 0.6
+ supports_vision: true
+ supports_function_calling: true
- name: gpt-4-turbo
max_input_tokens: 128000
max_output_tokens: 4096
@@ -47,14 +54,12 @@
type: embedding
max_input_tokens: 8191
input_price: 0.13
- output_vector_size: 3072
default_chunk_size: 3000
max_batch_size: 100
- name: text-embedding-3-small
type: embedding
max_input_tokens: 8191
input_price: 0.02
- output_vector_size: 1536
default_chunk_size: 3000
max_batch_size: 100
@@ -69,11 +74,11 @@
- name: gemini-1.5-pro-latest
max_input_tokens: 2097152
max_output_tokens: 8192
- input_price: 3.5
- output_price: 10.5
+ input_price: 0
+ output_price: 0
supports_vision: true
supports_function_calling: true
- - name: gemini-1.5-pro-exp-0801
+ - name: gemini-1.5-pro-exp-0827
max_input_tokens: 2097152
max_output_tokens: 8192
supports_vision: true
@@ -81,15 +86,29 @@
- name: gemini-1.5-flash-latest
max_input_tokens: 1048576
max_output_tokens: 8192
- input_price: 0.075
- output_price: 0.3
+ input_price: 0
+ output_price: 0
+ supports_vision: true
+ supports_function_calling: true
+ - name: gemini-1.5-flash-exp-0827
+ max_input_tokens: 1048576
+ max_output_tokens: 8192
+ input_price: 0
+ output_price: 0
+ supports_vision: true
+ supports_function_calling: true
+ - name: gemini-1.5-flash-8b-exp-0827
+ max_input_tokens: 1048576
+ max_output_tokens: 8192
+ input_price: 0
+ output_price: 0
supports_vision: true
supports_function_calling: true
- name: gemini-1.0-pro-latest
max_input_tokens: 30720
max_output_tokens: 2048
- input_price: 0.5
- output_price: 1.5
+ input_price: 0
+ output_price: 0
supports_function_calling: true
- name: text-embedding-004
type: embedding
@@ -165,19 +184,10 @@
max_input_tokens: 256000
input_price: 0.25
output_price: 0.25
- - name: open-mixtral-8x22b
- max_input_tokens: 64000
- input_price: 2
- output_price: 6
- - name: open-mixtral-8x7b
- max_input_tokens: 32000
- input_price: 0.7
- output_price: 0.7
- name: mistral-embed
type: embedding
max_input_tokens: 8092
input_price: 0.1
- output_vector_size: 1024
default_chunk_size: 2000
max_batch_size: 3
@@ -199,32 +209,40 @@
- platform: cohere
# docs:
- # - https://docs.cohere.com/docs/command-r
+ # - https://docs.cohere.com/docs/command-r-plus
# - https://cohere.com/pricing
# - https://docs.cohere.com/reference/chat
models:
- name: command-r-plus
max_input_tokens: 128000
- input_price: 3
- output_price: 15
+ input_price: 2.5
+ output_price: 10
+ supports_function_calling: true
+ - name: command-r-plus-08-2024
+ max_input_tokens: 128000
+ input_price: 2.5
+ output_price: 10
supports_function_calling: true
- name: command-r
max_input_tokens: 128000
- input_price: 0.5
- output_price: 1.5
+ input_price: 0.15
+ output_price: 0.6
+ supports_function_calling: true
+ - name: command-r-08-2024
+ max_input_tokens: 128000
+ input_price: 0.15
+ output_price: 0.6
supports_function_calling: true
- name: embed-english-v3.0
type: embedding
max_input_tokens: 512
input_price: 0.1
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 96
- name: embed-multilingual-v3.0
type: embedding
max_input_tokens: 512
input_price: 0.1
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 96
- name: rerank-english-v3.0
@@ -236,9 +254,9 @@
- platform: perplexity
# docs:
- # - https://docs.perplexity.ai/docs/model-cards
- # - https://docs.perplexity.ai/docs/pricing
- # - https://docs.perplexity.ai/reference/post_chat_completions
+ # - https://docs.perplexity.ai/guides/model-cards
+ # - https://docs.perplexity.ai/guides/pricing
+ # - https://docs.perplexity.ai/api-reference/chat-completions
models:
- name: llama-3.1-sonar-huge-128k-online
max_input_tokens: 127072
@@ -279,40 +297,62 @@
models:
- name: llama3-70b-8192
max_input_tokens: 8192
- input_price: 0.59
- output_price: 0.79
+ input_price: 0
+ output_price: 0
supports_function_calling: true
- name: llama3-8b-8192
max_input_tokens: 8192
- input_price: 0.05
- output_price: 0.08
+ input_price: 0
+ output_price: 0
supports_function_calling: true
- name: llama3-groq-70b-8192-tool-use-preview
max_input_tokens: 8192
- input_price: 0.89
- output_price: 0.89
+ input_price: 0
+ output_price: 0
supports_function_calling: true
- name: llama3-groq-8b-8192-tool-use-preview
max_input_tokens: 8192
- input_price: 0.19
- output_price: 0.19
+ input_price: 0
+ output_price: 0
supports_function_calling: true
- - name: llama-3.1-405b-reasoning
- max_input_tokens: 8192
- name: llama-3.1-70b-versatile
max_input_tokens: 8192
+ input_price: 0
+ output_price: 0
- name: llama-3.1-8b-instant
max_input_tokens: 8192
- - name: mixtral-8x7b-32768
- max_input_tokens: 32768
- input_price: 0.24
- output_price: 0.24
+ input_price: 0
+ output_price: 0
- name: gemma2-9b-it
max_input_tokens: 8192
- input_price: 0.2
- output_price: 0.2
+ input_price: 0
+ output_price: 0
supports_function_calling: true
+- platform: ollama
+ # docs:
+ # - https://ollama.com/library
+ # - https://github.com/ollama/ollama/blob/main/docs/openai.md
+ models:
+ - name: llama3.1
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: gemma2
+ max_input_tokens: 8192
+ - name: mistral-nemo
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: mistral-large
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: phi3
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: nomic-embed-text
+ type: embedding
+ default_chunk_size: 1000
+ max_batch_size: 50
+
- platform: vertexai
# docs:
# - https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models
@@ -392,14 +432,12 @@
type: embedding
max_input_tokens: 3072
input_price: 0.025
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 5
- name: text-multilingual-embedding-002
type: embedding
max_input_tokens: 3072
input_price: 0.2
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 5
@@ -485,14 +523,12 @@
type: embedding
max_input_tokens: 512
input_price: 0.1
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 96
- name: cohere.embed-multilingual-v3
type: embedding
max_input_tokens: 512
input_price: 0.1
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 96
@@ -516,7 +552,6 @@
type: embedding
max_input_tokens: 512
input_price: 0
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
@@ -543,12 +578,6 @@
require_max_tokens: true
input_price: 0.05
output_price: 0.25
- - name: mistralai/mixtral-8x7b-instruct-v0.1
- max_input_tokens: 32000
- max_output_tokens: 8192
- require_max_tokens: true
- input_price: 0.3
- output_price: 1
- platform: ernie
# docs:
@@ -578,14 +607,12 @@
type: embedding
max_input_tokens: 512
input_price: 0.28
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 16
- name: bge_large_en
type: embedding
max_input_tokens: 512
input_price: 0.28
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 16
- name: bce_reranker_base
@@ -631,11 +658,16 @@
input_price: 1.12
output_price: 1.12
supports_vision: true
+ - name: text-embedding-v3
+ type: embedding
+ max_input_tokens: 8192
+ input_price: 0.1
+ default_chunk_size: 2000
+ max_batch_size: 6
- name: text-embedding-v2
type: embedding
max_input_tokens: 2048
input_price: 0.1
- output_vector_size: 1536
default_chunk_size: 1500
max_batch_size: 25
@@ -682,6 +714,11 @@
# - https://open.bigmodel.cn/dev/howuse/model
# - https://open.bigmodel.cn/pricing
models:
+ - name: glm-4-plus
+ max_input_tokens: 128000
+ input_price: 7
+ output_price: 7
+ supports_function_calling: true
- name: glm-4-0520
max_input_tokens: 128000
input_price: 14
@@ -697,21 +734,16 @@
input_price: 14
output_price: 14
supports_function_calling: true
- - name: glm-4-airx
- max_input_tokens: 8092
- input_price: 1.4
- output_price: 1.4
- supports_function_calling: true
- - name: glm-4-air
- max_input_tokens: 128000
- input_price: 0.14
- output_price: 0.14
- supports_function_calling: true
- name: glm-4-flash
max_input_tokens: 128000
- input_price: 0.014
- output_price: 0.014
+ input_price: 0
+ output_price: 0
supports_function_calling: true
+ - name: glm-4v-plus
+ max_input_tokens: 8192
+ input_price: 1.4
+ output_price: 1.4
+ supports_vision: true
- name: glm-4v
max_input_tokens: 2048
input_price: 7
@@ -721,7 +753,6 @@
type: embedding
max_input_tokens: 8192
input_price: 0.07
- output_vector_size: 2048
default_chunk_size: 2000
max_batch_size: 3
@@ -751,11 +782,6 @@
max_input_tokens: 200000
input_price: 1.68
output_price: 1.68
- - name: yi-vision
- max_input_tokens: 16384
- input_price: 0.84
- output_price: 0.84
- supports_vision: true
- name: yi-medium
max_input_tokens: 16384
input_price: 0.35
@@ -764,6 +790,11 @@
max_input_tokens: 16384
input_price: 0.14
output_price: 0.14
+ - name: yi-vision
+ max_input_tokens: 16384
+ input_price: 0.84
+ output_price: 0.84
+ supports_vision: true
- platform: github
# docs:
@@ -775,6 +806,16 @@
- name: gpt-4o-mini
max_input_tokens: 128000
supports_function_calling: true
+ - name: text-embedding-3-large
+ type: embedding
+ max_input_tokens: 8191
+ default_chunk_size: 3000
+ max_batch_size: 100
+ - name: text-embedding-3-small
+ type: embedding
+ max_input_tokens: 8191
+ default_chunk_size: 3000
+ max_batch_size: 100
- name: meta-llama-3.1-405b-instruct
max_input_tokens: 128000
- name: meta-llama-3.1-70b-instruct
@@ -804,13 +845,11 @@
- name: cohere-embed-v3-english
type: embedding
max_input_tokens: 512
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 96
- name: cohere-embed-v3-multilingual
type: embedding
max_input_tokens: 512
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 96
@@ -841,15 +880,10 @@
input_price: 0.08
output_price: 0.08
supports_function_calling: true
- - name: mistralai/Mixtral-8x22B-Instruct-v0.1
- max_input_tokens: 65536
- input_price: 0.65
- output_price: 0.65
- supports_function_calling: true
- - name: mistralai/Mixtral-8x7B-Instruct-v0.1
- max_input_tokens: 32768
- input_price: 0.24
- output_price: 0.24
+ - name: mistralai/Mistral-Nemo-Instruct-2407
+ max_input_tokens: 128000
+ input_price: 0.13
+ output_price: 0.13
supports_function_calling: true
- name: google/gemma-2-27b-it
max_input_tokens: 8192
@@ -868,35 +902,30 @@
type: embedding
max_input_tokens: 512
input_price: 0.01
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
- name: BAAI/bge-m3
type: embedding
max_input_tokens: 8192
input_price: 0.01
- output_vector_size: 1024
default_chunk_size: 2000
max_batch_size: 100
- name: intfloat/e5-large-v2
type: embedding
max_input_tokens: 512
input_price: 0.01
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
- name: intfloat/multilingual-e5-large
type: embedding
max_input_tokens: 512
input_price: 0.01
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
- name: thenlper/gte-large
type: embedding
max_input_tokens: 512
input_price: 0.01
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
@@ -925,26 +954,10 @@
max_input_tokens: 8192
input_price: 0.2
output_price: 0.2
- - name: accounts/fireworks/models/mixtral-8x22b-instruct
- max_input_tokens: 65536
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/mixtral-8x7b-instruct
- max_input_tokens: 32768
- input_price: 0.5
- output_price: 0.5
- name: accounts/fireworks/models/gemma2-9b-it
max_input_tokens: 8192
input_price: 0.2
output_price: 0.2
- - name: accounts/fireworks/models/deepseek-coder-v2-instruct
- max_input_tokens: 131072
- input_price: 2.7
- output_price: 2.7
- - name: accounts/fireworks/models/deepseek-coder-v2-lite-instruct
- max_input_tokens: 163840
- input_price: 0.2
- output_price: 0.2
- name: accounts/fireworks/models/phi-3-vision-128k-instruct
max_input_tokens: 131072
input_price: 0.2
@@ -964,21 +977,18 @@
type: embedding
max_input_tokens: 8192
input_price: 0.008
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 100
- name: WhereIsAI/UAE-Large-V1
type: embedding
max_input_tokens: 512
input_price: 0.016
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
- name: thenlper/gte-large
type: embedding
max_input_tokens: 512
input_price: 0.016
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
@@ -988,14 +998,14 @@
models:
- name: openai/gpt-4o
max_input_tokens: 128000
- input_price: 2.5
- output_price: 10
+ input_price: 5
+ output_price: 15
supports_vision: true
supports_function_calling: true
- - name: openai/gpt-4o-mini
+ - name: openai/gpt-4o-2024-08-06
max_input_tokens: 128000
- input_price: 0.15
- output_price: 0.6
+ input_price: 2.5
+ output_price: 10
supports_vision: true
supports_function_calling: true
- name: openai/chatgpt-4o-latest
@@ -1004,6 +1014,12 @@
output_price: 15
supports_vision: true
supports_function_calling: true
+ - name: openai/gpt-4o-mini
+ max_input_tokens: 128000
+ input_price: 0.15
+ output_price: 0.6
+ supports_vision: true
+ supports_function_calling: true
- name: openai/gpt-4-turbo
max_input_tokens: 128000
input_price: 10
@@ -1016,32 +1032,48 @@
output_price: 1.5
supports_function_calling: true
- name: google/gemini-pro-1.5
- max_input_tokens: 2800000
+ max_input_tokens: 4000000
input_price: 2.5
output_price: 7.5
supports_vision: true
supports_function_calling: true
- name: google/gemini-pro-1.5-exp
max_input_tokens: 4000000
- input_price: 2.5
- output_price: 7.5
+ input_price: 0
+ output_price: 0
supports_vision: true
supports_function_calling: true
- name: google/gemini-flash-1.5
- max_input_tokens: 2800000
- input_price: 0.25
- output_price: 0.75
+ max_input_tokens: 4000000
+ input_price: 0.0375
+ output_price: 0.15
+ supports_vision: true
+ supports_function_calling: true
+ - name: google/gemini-flash-1.5-exp
+ max_input_tokens: 4000000
+ input_price: 0
+ output_price: 0
+ supports_vision: true
+ supports_function_calling: true
+ - name: google/gemini-flash-8b-1.5-exp
+ max_input_tokens: 4000000
+ input_price: 0
+ output_price: 0
supports_vision: true
supports_function_calling: true
- name: google/gemini-pro
- max_input_tokens: 91728
+ max_input_tokens: 131040
input_price: 0.125
output_price: 0.375
supports_function_calling: true
- - name: google/gemma-2-9b-it
+ - name: google/gemma-2-27b-it
max_input_tokens: 2800000
- input_price: 0.2
- output_price: 0.2
+ input_price: 0.27
+ output_price: 0.27
+ - name: google/gemma-2-9b-it
+ max_input_tokens: 8192
+ input_price: 0.06
+ output_price: 0.06
- name: anthropic/claude-3.5-sonnet
max_input_tokens: 200000
max_output_tokens: 4096
@@ -1108,14 +1140,6 @@
max_input_tokens: 256000
input_price: 0.25
output_price: 0.25
- - name: mistralai/mixtral-8x22b-instruct
- max_input_tokens: 65536
- input_price: 0.65
- output_price: 0.65
- - name: mistralai/mixtral-8x7b-instruct
- max_input_tokens: 32768
- input_price: 0.24
- output_price: 0.24
- name: ai21/jamba-1-5-large
max_input_tokens: 256000
input_price: 2
@@ -1128,13 +1152,23 @@
supports_function_calling: true
- name: cohere/command-r-plus
max_input_tokens: 128000
- input_price: 3
- output_price: 15
+ input_price: 2.5
+ output_price: 10
+ supports_function_calling: true
+ - name: cohere/command-r-plus-08-2024
+ max_input_tokens: 128000
+ input_price: 2.5
+ output_price: 10
supports_function_calling: true
- name: cohere/command-r
max_input_tokens: 128000
- input_price: 0.5
- output_price: 1.5
+ input_price: 0.15
+ output_price: 0.6
+ supports_function_calling: true
+ - name: cohere/command-r-08-2024
+ max_input_tokens: 128000
+ input_price: 0.15
+ output_price: 0.6
supports_function_calling: true
- name: deepseek/deepseek-chat
max_input_tokens: 32768
@@ -1216,23 +1250,10 @@
max_input_tokens: 8192
input_price: 0.9
output_price: 0.9
- - name: meta-llama-3-8b-instruct
- max_input_tokens: 8192
- input_price: 0.15
- output_price: 0.15
- - name: mixtral-8x22b-instruct
- max_input_tokens: 65536
- input_price: 1.2
- output_price: 1.2
- - name: mixtral-8x7b-instruct
- max_input_tokens: 32768
- input_price: 0.45
- output_price: 0.45
- name: thenlper/gte-large
type: embedding
max_input_tokens: 512
input_price: 0.05
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
@@ -1262,14 +1283,6 @@
max_input_tokens: 8192
input_price: 0.18
output_price: 0.18
- - name: mistralai/Mixtral-8x22B-Instruct-v0.1
- max_input_tokens: 65536
- input_price: 1.2
- output_price: 1.2
- - name: mistralai/Mixtral-8x7B-Instruct-v0.1
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- name: Qwen/Qwen2-72B-Instruct
max_input_tokens: 32768
input_price: 0.9
@@ -1278,14 +1291,12 @@
type: embedding
max_input_tokens: 512
input_price: 0.016
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
- name: BAAI/bge-large-en-v1.5
type: embedding
max_input_tokens: 512
input_price: 0.016
- output_vector_size: 1024
default_chunk_size: 1000
max_batch_size: 100
@@ -1298,28 +1309,24 @@
type: embedding
max_input_tokens: 8192
input_price: 0.02
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 100
- name: jina-embeddings-v2-base-en
type: embedding
max_input_tokens: 8192
input_price: 0.02
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 100
- name: jina-embeddings-v2-base-zh
type: embedding
max_input_tokens: 8192
input_price: 0.02
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 100
- - name: jina-colbert-v1-en
+ - name: jina-colbert-v2
type: embedding
max_input_tokens: 8192
input_price: 0.02
- output_vector_size: 768
default_chunk_size: 1500
max_batch_size: 100
- name: jina-reranker-v2-base-multilingual
@@ -1334,7 +1341,7 @@
type: reranker
max_input_tokens: 8192
input_price: 0.02
- - name: jina-colbert-v1-en
+ - name: jina-colbert-v2
type: reranker
max_input_tokens: 8192
input_price: 0.02
@@ -1349,31 +1356,27 @@
type: embedding
max_input_tokens: 16000
input_price: 0.12
- output_vector_size: 1024
default_chunk_size: 2000
max_batch_size: 128
- name: voyage-large-2
type: embedding
max_input_tokens: 16000
input_price: 0.12
- output_vector_size: 1536
default_chunk_size: 3000
max_batch_size: 128
- name: voyage-multilingual-2
type: embedding
max_input_tokens: 32000
input_price: 0.12
- output_vector_size: 1024
default_chunk_size: 2000
max_batch_size: 128
- name: voyage-code-2
type: embedding
max_input_tokens: 16000
input_price: 0.12
- output_vector_size: 1536
default_chunk_size: 3000
max_batch_size: 128
- name: rerank-1
type: reranker
max_input_tokens: 8000
- input_price: 0.05 \ No newline at end of file
+ input_price: 0.05
diff --git a/src/client/model.rs b/src/client/model.rs
index bea96a2..50dabeb 100644
--- a/src/client/model.rs
+++ b/src/client/model.rs
@@ -159,17 +159,13 @@ impl Model {
let ModelData {
max_input_tokens,
input_price,
- output_vector_size,
max_batch_size,
..
} = &self.data;
- let dimension = format_option_value(output_vector_size);
let max_tokens = format_option_value(max_input_tokens);
let price = format_option_value(input_price);
let batch = format_option_value(max_batch_size);
- format!(
- "dimension:{dimension}; max-tokens:{max_tokens}; price:{price}; batch:{batch}"
- )
+ format!("max-tokens:{max_tokens}; price:{price}; batch:{batch}")
}
_ => String::new(),
}
@@ -270,7 +266,6 @@ pub struct ModelData {
pub supports_function_calling: bool,
// embedding-only properties
- pub output_vector_size: Option<usize>,
pub default_chunk_size: Option<usize>,
pub max_batch_size: Option<usize>,
}
diff --git a/src/client/openai.rs b/src/client/openai.rs
index e47ad4c..902a215 100644
--- a/src/client/openai.rs
+++ b/src/client/openai.rs
@@ -249,7 +249,7 @@ pub fn openai_build_chat_completions_body(data: ChatCompletionsData, model: &Mod
})
]
- }).collect()
+ }).collect()
}
},
_ => vec![json!({ "role": role, "content": content })]