summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorsigoden <sigoden@gmail.com>2025-02-17 08:45:21 +0800
committerGitHub <noreply@github.com>2025-02-17 08:45:21 +0800
commita6eda31de56044319e0207a4ffc1869e26224b26 (patch)
tree1d288dbfad92546305d03a0a1294d2cced450658
parentf14f1ad01bf31bb4bb4322661691be9392c30acf (diff)
downloadaichat-a6eda31de56044319e0207a4ffc1869e26224b26.tar.gz
feat: remove supports for fireworks/siliconflow/together (#1181)
-rwxr-xr-xArgcfile.sh6
-rw-r--r--config.example.yaml29
-rw-r--r--models.yaml288
-rw-r--r--src/client/mod.rs5
4 files changed, 7 insertions, 321 deletions
diff --git a/Argcfile.sh b/Argcfile.sh
index ea0b9c1..e03c3a1 100755
--- a/Argcfile.sh
+++ b/Argcfile.sh
@@ -142,9 +142,6 @@ models() {
github)
jq_args+=(-r '.[].name')
;;
- together)
- jq_args+=(-r '.[].id')
- ;;
*)
jq_args+=(-r '.data[].id')
;;
@@ -317,7 +314,6 @@ _argc_before() {
deepinfra,meta-llama/Meta-Llama-3.1-8B-Instruct,https://api.deepinfra.com/v1/openai \
deepseek,deepseek-chat,https://api.deepseek.com \
ernie,ernie-4.0-turbo-8k-latest,https://qianfan.baidubce.com/v2 \
- fireworks,accounts/fireworks/models/llama-v3p1-8b-instruct,https://api.fireworks.ai/inference/v1 \
github,gpt-4o-mini,https://models.inference.ai.azure.com \
groq,llama-3.1-8b-instant,https://api.groq.com/openai/v1 \
hunyuan,hunyuan-large,https://api.hunyuan.cloud.tencent.com/v1 \
@@ -328,8 +324,6 @@ _argc_before() {
openrouter,openai/gpt-4o-mini,https://openrouter.ai/api/v1 \
perplexity,llama-3.1-8b-instruct,https://api.perplexity.ai \
qianwen,qwen-turbo-latest,https://dashscope.aliyuncs.com/compatible-mode/v1 \
- siliconflow,meta-llama/Meta-Llama-3.1-8B-Instruct,https://api.siliconflow.cn/v1 \
- together,meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo,https://api.together.xyz/v1 \
xai,grok-beta,https://api.x.ai/v1 \
zhipuai,glm-4-0520,https://open.bigmodel.cn/api/paas/v4 \
)
diff --git a/config.example.yaml b/config.example.yaml
index d0b6111..6284d76 100644
--- a/config.example.yaml
+++ b/config.example.yaml
@@ -288,10 +288,10 @@ clients:
api_base: https://api.minimax.chat/v1
api_key: xxx
- # See https://deepinfra.com/docs
+ # See https://openrouter.ai/docs#quick-start
- type: openai-compatible
- name: deepinfra
- api_base: https://api.deepinfra.com/v1/openai
+ name: openrouter
+ api_base: https://openrouter.ai/api/v1
api_key: xxx
# See https://github.com/marketplace/models
@@ -300,29 +300,12 @@ clients:
api_base: https://models.inference.ai.azure.com
api_key: xxx
- # See https://readme.fireworks.ai/docs/quickstart
- - type: openai-compatible
- name: fireworks
- api_base: https://api.fireworks.ai/inference/v1
- api_key: xxx
-
- # See https://openrouter.ai/docs#quick-start
- - type: openai-compatible
- name: openrouter
- api_base: https://openrouter.ai/api/v1
- api_key: xxx
-
- # See https://docs.siliconflow.cn/docs/getting-started
+ # See https://deepinfra.com/docs
- type: openai-compatible
- name: siliconflow
- api_base: https://api.siliconflow.cn/v1
+ name: deepinfra
+ api_base: https://api.deepinfra.com/v1/openai
api_key: xxx
- # See https://docs.together.ai/docs/quickstart
- - type: openai-compatible
- name: together
- api_base: https://api.together.xyz/v1
- api_key: xxx
# ----- RAG dedicated -----
diff --git a/models.yaml b/models.yaml
index e3bd1b2..217a716 100644
--- a/models.yaml
+++ b/models.yaml
@@ -1695,294 +1695,6 @@
max_batch_size: 100
# Links:
-# - https://fireworks.ai/models
-# - https://fireworks.ai/pricing
-# - https://docs.fireworks.ai/api-reference/post-chatcompletions
-- provider: fireworks
- models:
- - name: accounts/fireworks/models/llama-v3p3-70b-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/llama-v3p1-405b-instruct
- max_input_tokens: 131072
- input_price: 3
- output_price: 3
- supports_function_calling: true
- - name: accounts/fireworks/models/llama-v3p1-70b-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- supports_function_calling: true
- - name: accounts/fireworks/models/llama-v3p1-8b-instruct
- max_input_tokens: 131072
- input_price: 0.2
- output_price: 0.2
- - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- supports_vision: true
- - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct
- max_input_tokens: 131072
- input_price: 0.2
- output_price: 0.2
- supports_vision: true
- - name: accounts/fireworks/models/qwen2p5-72b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- supports_function_calling: true
- - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/qwen-qwq-32b-preview
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/qwen2-vl-72b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- supports_vision: true
- - name: accounts/fireworks/models/deepseek-v3
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/deepseek-r1
- max_input_tokens: 160000
- input_price: 8
- output_price: 8
- - name: accounts/fireworks/models/deepseek-r1-distill-llama-70b
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/deepseek-r1-distill-qwen-32b
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/mistral-small-24b-instruct-2501
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: nomic-ai/nomic-embed-text-v1.5
- type: embedding
- input_price: 0.008
- max_tokens_per_chunk: 8192
- default_chunk_size: 1500
- max_batch_size: 100
- - name: WhereIsAI/UAE-Large-V1
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: thenlper/gte-large
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
-
-# Links
-# - https://cloud.siliconflow.cn/models
-# - https://docs.siliconflow.cn/api-reference/chat-completions/chat-completions
-- provider: siliconflow
- models:
- - name: meta-llama/Llama-3.3-70B-Instruct
- max_input_tokens: 32768
- input_price: 0.578
- output_price: 0.578
- - name: meta-llama/Meta-Llama-3.1-405B-Instruct
- max_input_tokens: 32768
- input_price: 2.94
- output_price: 2.94
- - name: meta-llama/Meta-Llama-3.1-70B-Instruct
- max_input_tokens: 32768
- input_price: 0.578
- output_price: 0.578
- - name: meta-llama/Meta-Llama-3.1-8B-Instruct
- max_input_tokens: 32768
- input_price: 0
- output_price: 0
- - name: Qwen/Qwen2.5-72B-Instruct
- max_input_tokens: 32768
- input_price: 0.578
- output_price: 0.578
- supports_function_calling: true
- - name: Qwen/Qwen2.5-72B-Instruct-128K
- max_input_tokens: 131072
- input_price: 0.578
- output_price: 0.578
- supports_function_calling: true
- - name: Qwen/Qwen2.5-7B-Instruct
- max_input_tokens: 32768
- input_price: 0
- output_price: 0
- supports_function_calling: true
- - name: Qwen/Qwen2.5-Coder-32B-Instruct
- max_input_tokens: 32768
- input_price: 0.176
- output_price: 0.176
- - name: Qwen/Qwen2.5-Coder-7B-Instruct
- max_input_tokens: 32768
- input_price: 0
- output_price: 0
- - name: Qwen/Qwen2-VL-72B-Instruct
- max_input_tokens: 32768
- input_price: 0.5782
- output_price: 0.5782
- supports_vision: true
- - name: Qwen/QVQ-72B-Preview
- max_input_tokens: 32768
- input_price: 1.386
- output_price: 1.386
- supports_vision: true
- - name: Qwen/QwQ-32B-Preview
- max_input_tokens: 32768
- input_price: 0.176
- output_price: 0.176
- - name: deepseek-ai/DeepSeek-V3
- max_input_tokens: 65536
- input_price: 0.28
- output_price: 0.28
- - name: deepseek-ai/DeepSeek-R1
- max_input_tokens: 65536
- input_price: 2.24
- output_price: 2.24
- - name: deepseek-ai/DeepSeek-R1-Distill-Llama-70B
- max_input_tokens: 32768
- input_price: 0.578
- output_price: 0.578
- - name: deepseek-ai/DeepSeek-R1-Distill-Qwen-32B
- max_input_tokens: 32768
- input_price: 0.176
- output_price: 0.176
- - name: deepseek-ai/DeepSeek-V2.5
- max_input_tokens: 32768
- input_price: 0.7
- output_price: 0.7
- supports_function_calling: true
- - name: deepseek-ai/deepseek-vl2
- max_input_tokens: 32768
- input_price: 0.138
- output_price: 0.138
- supports_vision: true
- - name: BAAI/bge-large-en-v1.5
- type: embedding
- input_price: 0
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: BAAI/bge-large-zh-v1.5
- type: embedding
- input_price: 0
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: BAAI/bge-m3
- type: embedding
- input_price: 0
- max_tokens_per_chunk: 8192
- default_chunk_size: 2000
- max_batch_size: 100
- - name: BAAI/bge-reranker-v2-m3
- type: reranker
- max_input_tokens: 8192
- input_price: 0
-
-# Links:
-# - https://docs.together.ai/docs/serverless-models
-# - https://www.together.ai/pricing
-# - https://docs.together.ai/reference/chat-completions-1
-- provider: together
- models:
- - name: meta-llama/Llama-3.3-70B-Instruct-Turbo
- max_input_tokens: 131072
- input_price: 0.88
- output_price: 0.88
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo
- max_input_tokens: 130815
- input_price: 3.5
- output_price: 3.5
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo
- max_input_tokens: 131072
- input_price: 0.88
- output_price: 0.88
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo
- max_input_tokens: 131072
- input_price: 0.18
- output_price: 0.18
- supports_function_calling: true
- - name: meta-llama/Llama-3.2-90B-Vision-Instruct-Turbo
- max_input_tokens: 131072
- input_price: 0.88
- output_price: 0.88
- supports_vision: true
- - name: meta-llama/Llama-3.2-11B-Vision-Instruct-Turbo
- max_input_tokens: 131072
- input_price: 0.18
- output_price: 0.18
- supports_vision: true
- - name: Qwen/Qwen2.5-72B-Instruct-Turbo
- max_input_tokens: 32768
- input_price: 1.2
- output_price: 1.2
- - name: Qwen/Qwen2.5-7B-Instruct-Turbo
- max_input_tokens: 32768
- input_price: 0.3
- output_price: 0.3
- - name: Qwen/Qwen2.5-Coder-32B-Instruct
- max_input_tokens: 32768
- input_price: 0.8
- output_price: 0.8
- - name: Qwen/QwQ-32B-Preview
- max_input_tokens: 32768
- input_price: 1.2
- output_price: 1.2
- - name: Qwen/Qwen2-VL-72B-Instruct
- max_input_tokens: 32768
- input_price: 1.2
- output_price: 1.2
- supports_vision: true
- - name: deepseek-ai/DeepSeek-V3
- max_input_tokens: 131072
- input_price: 1.25
- output_price: 1.25
- - name: deepseek-ai/DeepSeek-R1
- max_input_tokens: 163840
- input_price: 7
- output_price: 7
- - name: deepseek-ai/DeepSeek-R1-Distill-Llama-70B
- max_input_tokens: 131072
- input_price: 2
- output_price: 2
- - name: mistralai/Mistral-Small-24B-Instruct-2501
- max_input_tokens: 32768
- input_price: 0.8
- output_price: 0.8
- - name: WhereIsAI/UAE-Large-V1
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: BAAI/bge-large-en-v1.5
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: Salesforce/Llama-Rank-V1
- type: reranker
- max_input_tokens: 8192
- input_price: 0.1
-
-# Links:
# - https://jina.ai/models
# - https://api.jina.ai/redoc
- provider: jina
diff --git a/src/client/mod.rs b/src/client/mod.rs
index 92fd31f..5492497 100644
--- a/src/client/mod.rs
+++ b/src/client/mod.rs
@@ -33,7 +33,7 @@ register_client!(
(bedrock, "bedrock", BedrockConfig, BedrockClient),
);
-pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [
+pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 19] = [
("ai21", "https://api.ai21.com/studio/v1"),
(
"cloudflare",
@@ -42,7 +42,6 @@ pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [
("deepinfra", "https://api.deepinfra.com/v1/openai"),
("deepseek", "https://api.deepseek.com"),
("ernie", "https://qianfan.baidubce.com/v2"),
- ("fireworks", "https://api.fireworks.ai/inference/v1"),
("github", "https://models.inference.ai.azure.com"),
("groq", "https://api.groq.com/openai/v1"),
("hunyuan", "https://api.hunyuan.cloud.tencent.com/v1"),
@@ -56,8 +55,6 @@ pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [
"qianwen",
"https://dashscope.aliyuncs.com/compatible-mode/v1",
),
- ("siliconflow", "https://api.siliconflow.cn/v1"),
- ("together", "https://api.together.xyz/v1"),
("xai", "https://api.x.ai/v1"),
("zhipuai", "https://open.bigmodel.cn/api/paas/v4"),
// RAG-dedicated