From a6eda31de56044319e0207a4ffc1869e26224b26 Mon Sep 17 00:00:00 2001 From: sigoden Date: Mon, 17 Feb 2025 08:45:21 +0800 Subject: feat: remove supports for fireworks/siliconflow/together (#1181) --- Argcfile.sh | 6 -- config.example.yaml | 29 ++---- models.yaml | 288 ---------------------------------------------------- src/client/mod.rs | 5 +- 4 files changed, 7 insertions(+), 321 deletions(-) diff --git a/Argcfile.sh b/Argcfile.sh index ea0b9c1..e03c3a1 100755 --- a/Argcfile.sh +++ b/Argcfile.sh @@ -142,9 +142,6 @@ models() { github) jq_args+=(-r '.[].name') ;; - together) - jq_args+=(-r '.[].id') - ;; *) jq_args+=(-r '.data[].id') ;; @@ -317,7 +314,6 @@ _argc_before() { deepinfra,meta-llama/Meta-Llama-3.1-8B-Instruct,https://api.deepinfra.com/v1/openai \ deepseek,deepseek-chat,https://api.deepseek.com \ ernie,ernie-4.0-turbo-8k-latest,https://qianfan.baidubce.com/v2 \ - fireworks,accounts/fireworks/models/llama-v3p1-8b-instruct,https://api.fireworks.ai/inference/v1 \ github,gpt-4o-mini,https://models.inference.ai.azure.com \ groq,llama-3.1-8b-instant,https://api.groq.com/openai/v1 \ hunyuan,hunyuan-large,https://api.hunyuan.cloud.tencent.com/v1 \ @@ -328,8 +324,6 @@ _argc_before() { openrouter,openai/gpt-4o-mini,https://openrouter.ai/api/v1 \ perplexity,llama-3.1-8b-instruct,https://api.perplexity.ai \ qianwen,qwen-turbo-latest,https://dashscope.aliyuncs.com/compatible-mode/v1 \ - siliconflow,meta-llama/Meta-Llama-3.1-8B-Instruct,https://api.siliconflow.cn/v1 \ - together,meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo,https://api.together.xyz/v1 \ xai,grok-beta,https://api.x.ai/v1 \ zhipuai,glm-4-0520,https://open.bigmodel.cn/api/paas/v4 \ ) diff --git a/config.example.yaml b/config.example.yaml index d0b6111..6284d76 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -288,10 +288,10 @@ clients: api_base: https://api.minimax.chat/v1 api_key: xxx - # See https://deepinfra.com/docs + # See https://openrouter.ai/docs#quick-start - type: openai-compatible - name: deepinfra - api_base: https://api.deepinfra.com/v1/openai + name: openrouter + api_base: https://openrouter.ai/api/v1 api_key: xxx # See https://github.com/marketplace/models @@ -300,29 +300,12 @@ clients: api_base: https://models.inference.ai.azure.com api_key: xxx - # See https://readme.fireworks.ai/docs/quickstart - - type: openai-compatible - name: fireworks - api_base: https://api.fireworks.ai/inference/v1 - api_key: xxx - - # See https://openrouter.ai/docs#quick-start - - type: openai-compatible - name: openrouter - api_base: https://openrouter.ai/api/v1 - api_key: xxx - - # See https://docs.siliconflow.cn/docs/getting-started + # See https://deepinfra.com/docs - type: openai-compatible - name: siliconflow - api_base: https://api.siliconflow.cn/v1 + name: deepinfra + api_base: https://api.deepinfra.com/v1/openai api_key: xxx - # See https://docs.together.ai/docs/quickstart - - type: openai-compatible - name: together - api_base: https://api.together.xyz/v1 - api_key: xxx # ----- RAG dedicated ----- diff --git a/models.yaml b/models.yaml index e3bd1b2..217a716 100644 --- a/models.yaml +++ b/models.yaml @@ -1694,294 +1694,6 @@ default_chunk_size: 1000 max_batch_size: 100 -# Links: -# - https://fireworks.ai/models -# - https://fireworks.ai/pricing -# - https://docs.fireworks.ai/api-reference/post-chatcompletions -- provider: fireworks - models: - - name: accounts/fireworks/models/llama-v3p3-70b-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/llama-v3p1-405b-instruct - max_input_tokens: 131072 - input_price: 3 - output_price: 3 - supports_function_calling: true - - name: accounts/fireworks/models/llama-v3p1-70b-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - supports_function_calling: true - - name: accounts/fireworks/models/llama-v3p1-8b-instruct - max_input_tokens: 131072 - input_price: 0.2 - output_price: 0.2 - - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - supports_vision: true - - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct - max_input_tokens: 131072 - input_price: 0.2 - output_price: 0.2 - supports_vision: true - - name: accounts/fireworks/models/qwen2p5-72b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - supports_function_calling: true - - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/qwen-qwq-32b-preview - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/qwen2-vl-72b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - supports_vision: true - - name: accounts/fireworks/models/deepseek-v3 - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/deepseek-r1 - max_input_tokens: 160000 - input_price: 8 - output_price: 8 - - name: accounts/fireworks/models/deepseek-r1-distill-llama-70b - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/deepseek-r1-distill-qwen-32b - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/mistral-small-24b-instruct-2501 - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: nomic-ai/nomic-embed-text-v1.5 - type: embedding - input_price: 0.008 - max_tokens_per_chunk: 8192 - default_chunk_size: 1500 - max_batch_size: 100 - - name: WhereIsAI/UAE-Large-V1 - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: thenlper/gte-large - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - -# Links -# - https://cloud.siliconflow.cn/models -# - https://docs.siliconflow.cn/api-reference/chat-completions/chat-completions -- provider: siliconflow - models: - - name: meta-llama/Llama-3.3-70B-Instruct - max_input_tokens: 32768 - input_price: 0.578 - output_price: 0.578 - - name: meta-llama/Meta-Llama-3.1-405B-Instruct - max_input_tokens: 32768 - input_price: 2.94 - output_price: 2.94 - - name: meta-llama/Meta-Llama-3.1-70B-Instruct - max_input_tokens: 32768 - input_price: 0.578 - output_price: 0.578 - - name: meta-llama/Meta-Llama-3.1-8B-Instruct - max_input_tokens: 32768 - input_price: 0 - output_price: 0 - - name: Qwen/Qwen2.5-72B-Instruct - max_input_tokens: 32768 - input_price: 0.578 - output_price: 0.578 - supports_function_calling: true - - name: Qwen/Qwen2.5-72B-Instruct-128K - max_input_tokens: 131072 - input_price: 0.578 - output_price: 0.578 - supports_function_calling: true - - name: Qwen/Qwen2.5-7B-Instruct - max_input_tokens: 32768 - input_price: 0 - output_price: 0 - supports_function_calling: true - - name: Qwen/Qwen2.5-Coder-32B-Instruct - max_input_tokens: 32768 - input_price: 0.176 - output_price: 0.176 - - name: Qwen/Qwen2.5-Coder-7B-Instruct - max_input_tokens: 32768 - input_price: 0 - output_price: 0 - - name: Qwen/Qwen2-VL-72B-Instruct - max_input_tokens: 32768 - input_price: 0.5782 - output_price: 0.5782 - supports_vision: true - - name: Qwen/QVQ-72B-Preview - max_input_tokens: 32768 - input_price: 1.386 - output_price: 1.386 - supports_vision: true - - name: Qwen/QwQ-32B-Preview - max_input_tokens: 32768 - input_price: 0.176 - output_price: 0.176 - - name: deepseek-ai/DeepSeek-V3 - max_input_tokens: 65536 - input_price: 0.28 - output_price: 0.28 - - name: deepseek-ai/DeepSeek-R1 - max_input_tokens: 65536 - input_price: 2.24 - output_price: 2.24 - - name: deepseek-ai/DeepSeek-R1-Distill-Llama-70B - max_input_tokens: 32768 - input_price: 0.578 - output_price: 0.578 - - name: deepseek-ai/DeepSeek-R1-Distill-Qwen-32B - max_input_tokens: 32768 - input_price: 0.176 - output_price: 0.176 - - name: deepseek-ai/DeepSeek-V2.5 - max_input_tokens: 32768 - input_price: 0.7 - output_price: 0.7 - supports_function_calling: true - - name: deepseek-ai/deepseek-vl2 - max_input_tokens: 32768 - input_price: 0.138 - output_price: 0.138 - supports_vision: true - - name: BAAI/bge-large-en-v1.5 - type: embedding - input_price: 0 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: BAAI/bge-large-zh-v1.5 - type: embedding - input_price: 0 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: BAAI/bge-m3 - type: embedding - input_price: 0 - max_tokens_per_chunk: 8192 - default_chunk_size: 2000 - max_batch_size: 100 - - name: BAAI/bge-reranker-v2-m3 - type: reranker - max_input_tokens: 8192 - input_price: 0 - -# Links: -# - https://docs.together.ai/docs/serverless-models -# - https://www.together.ai/pricing -# - https://docs.together.ai/reference/chat-completions-1 -- provider: together - models: - - name: meta-llama/Llama-3.3-70B-Instruct-Turbo - max_input_tokens: 131072 - input_price: 0.88 - output_price: 0.88 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo - max_input_tokens: 130815 - input_price: 3.5 - output_price: 3.5 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo - max_input_tokens: 131072 - input_price: 0.88 - output_price: 0.88 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo - max_input_tokens: 131072 - input_price: 0.18 - output_price: 0.18 - supports_function_calling: true - - name: meta-llama/Llama-3.2-90B-Vision-Instruct-Turbo - max_input_tokens: 131072 - input_price: 0.88 - output_price: 0.88 - supports_vision: true - - name: meta-llama/Llama-3.2-11B-Vision-Instruct-Turbo - max_input_tokens: 131072 - input_price: 0.18 - output_price: 0.18 - supports_vision: true - - name: Qwen/Qwen2.5-72B-Instruct-Turbo - max_input_tokens: 32768 - input_price: 1.2 - output_price: 1.2 - - name: Qwen/Qwen2.5-7B-Instruct-Turbo - max_input_tokens: 32768 - input_price: 0.3 - output_price: 0.3 - - name: Qwen/Qwen2.5-Coder-32B-Instruct - max_input_tokens: 32768 - input_price: 0.8 - output_price: 0.8 - - name: Qwen/QwQ-32B-Preview - max_input_tokens: 32768 - input_price: 1.2 - output_price: 1.2 - - name: Qwen/Qwen2-VL-72B-Instruct - max_input_tokens: 32768 - input_price: 1.2 - output_price: 1.2 - supports_vision: true - - name: deepseek-ai/DeepSeek-V3 - max_input_tokens: 131072 - input_price: 1.25 - output_price: 1.25 - - name: deepseek-ai/DeepSeek-R1 - max_input_tokens: 163840 - input_price: 7 - output_price: 7 - - name: deepseek-ai/DeepSeek-R1-Distill-Llama-70B - max_input_tokens: 131072 - input_price: 2 - output_price: 2 - - name: mistralai/Mistral-Small-24B-Instruct-2501 - max_input_tokens: 32768 - input_price: 0.8 - output_price: 0.8 - - name: WhereIsAI/UAE-Large-V1 - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: BAAI/bge-large-en-v1.5 - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: Salesforce/Llama-Rank-V1 - type: reranker - max_input_tokens: 8192 - input_price: 0.1 - # Links: # - https://jina.ai/models # - https://api.jina.ai/redoc diff --git a/src/client/mod.rs b/src/client/mod.rs index 92fd31f..5492497 100644 --- a/src/client/mod.rs +++ b/src/client/mod.rs @@ -33,7 +33,7 @@ register_client!( (bedrock, "bedrock", BedrockConfig, BedrockClient), ); -pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [ +pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 19] = [ ("ai21", "https://api.ai21.com/studio/v1"), ( "cloudflare", @@ -42,7 +42,6 @@ pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [ ("deepinfra", "https://api.deepinfra.com/v1/openai"), ("deepseek", "https://api.deepseek.com"), ("ernie", "https://qianfan.baidubce.com/v2"), - ("fireworks", "https://api.fireworks.ai/inference/v1"), ("github", "https://models.inference.ai.azure.com"), ("groq", "https://api.groq.com/openai/v1"), ("hunyuan", "https://api.hunyuan.cloud.tencent.com/v1"), @@ -56,8 +55,6 @@ pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [ "qianwen", "https://dashscope.aliyuncs.com/compatible-mode/v1", ), - ("siliconflow", "https://api.siliconflow.cn/v1"), - ("together", "https://api.together.xyz/v1"), ("xai", "https://api.x.ai/v1"), ("zhipuai", "https://open.bigmodel.cn/api/paas/v4"), // RAG-dedicated -- cgit v1.2.3