diff options
| author | sigoden <sigoden@gmail.com> | 2025-01-22 20:51:10 +0800 |
|---|---|---|
| committer | GitHub <noreply@github.com> | 2025-01-22 20:51:10 +0800 |
| commit | e0417e8d5bebe25476aaafa22a9ee23d9bd61457 (patch) | |
| tree | 8edcd3aa9bcd1b012a3a429ad6240e6186caef36 /models.yaml | |
| parent | df4440a2a049d26c61a540d3254cd885345fbd6c (diff) | |
| download | aichat-e0417e8d5bebe25476aaafa22a9ee23d9bd61457.tar.gz | |
feat: add `--sync-models` cli option (#1114)
Diffstat (limited to 'models.yaml')
| -rw-r--r-- | models.yaml | 734 |
1 files changed, 353 insertions, 381 deletions
diff --git a/models.yaml b/models.yaml index 60184de..6b505f1 100644 --- a/models.yaml +++ b/models.yaml @@ -1,11 +1,8 @@ -# Notes: -# - do not submit pull requests to add new models; this list will be updated in batches with new releases. - # Links: # - https://platform.openai.com/docs/models # - https://openai.com/api/pricing/ # - https://platform.openai.com/docs/api-reference/chat -- platform: openai +- provider: openai models: - name: gpt-4o max_input_tokens: 128000 @@ -50,7 +47,7 @@ supports_vision: true supports_function_calling: true - name: o1 - max_input_tokens: 128000 + max_input_tokens: 200000 input_price: 15 output_price: 60 supports_vision: true @@ -91,7 +88,7 @@ # - https://ai.google.dev/models/gemini # - https://ai.google.dev/pricing # - https://ai.google.dev/api/rest/v1beta/models/streamGenerateContent -- platform: gemini +- provider: gemini models: - name: gemini-1.5-pro-latest max_input_tokens: 2097152 @@ -122,13 +119,13 @@ supports_vision: true supports_function_calling: true - name: gemini-2.0-flash-thinking-exp - max_input_tokens: 32768 + max_input_tokens: 32767 max_output_tokens: 8192 input_price: 0 output_price: 0 supports_vision: true - name: gemini-exp-1206 - max_input_tokens: 32768 + max_input_tokens: 2097152 max_output_tokens: 8192 input_price: 0 output_price: 0 @@ -144,7 +141,7 @@ # Links: # - https://docs.anthropic.com/claude/docs/models-overview # - https://docs.anthropic.com/claude/reference/messages-streaming -- platform: claude +- provider: claude models: - name: claude-3-5-sonnet-latest max_input_tokens: 200000 @@ -207,7 +204,7 @@ # - https://docs.mistral.ai/getting-started/models/models_overview/ # - https://mistral.ai/technology/#pricing # - https://docs.mistral.ai/api/ -- platform: mistral +- provider: mistral models: - name: mistral-large-latest max_input_tokens: 128000 @@ -250,7 +247,7 @@ # - https://docs.ai21.com/docs/jamba-15-models # - https://www.ai21.com/pricing # - https://docs.ai21.com/reference/jamba-15-api-ref -- platform: ai21 +- provider: ai21 models: - name: jamba-1.5-large max_input_tokens: 256000 @@ -267,7 +264,7 @@ # - https://docs.cohere.com/docs/command-r-plus # - https://cohere.com/pricing # - https://docs.cohere.com/reference/chat -- platform: cohere +- provider: cohere models: - name: command-r-plus-08-2024 max_input_tokens: 128000 @@ -322,7 +319,7 @@ # Links: # - https://docs.x.ai/docs/models -- platform: xai +- provider: xai models: - name: grok-2-latest max_input_tokens: 131072 @@ -361,7 +358,7 @@ # - https://docs.perplexity.ai/guides/model-cards # - https://docs.perplexity.ai/guides/pricing # - https://docs.perplexity.ai/api-reference/chat-completions -- platform: perplexity +- provider: perplexity models: - name: llama-3.1-sonar-huge-128k-online max_input_tokens: 127072 @@ -379,64 +376,57 @@ # Links: # - https://console.groq.com/docs/models # - https://console.groq.com/docs/api-reference#chat -- platform: groq +- provider: groq models: - name: llama-3.3-70b-versatile - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_function_calling: true - name: llama-3.1-8b-instant - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_function_calling: true - name: llama-3.2-90b-vision-preview - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_vision: true - name: llama-3.2-11b-vision-preview - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_vision: true - - name: gemma2-9b-it - max_input_tokens: 8192 - input_price: 0 - output_price: 0 - supports_function_calling: true # Links: # - https://ollama.com/library # - https://github.com/ollama/ollama/blob/main/docs/openai.md -- platform: ollama +- provider: ollama models: - name: llama3.1 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: llama3.2 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: llama3.2-vision - max_input_tokens: 128000 + max_input_tokens: 131072 supports_vision: true - name: llama3.3 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: qwq max_input_tokens: 32768 supports_function_calling: true - name: qwen2.5 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: qwen2.5-coder max_input_tokens: 32768 supports_function_calling: true - name: phi4 max_input_tokens: 16384 - - name: gemma2 - max_input_tokens: 8192 - name: nomic-embed-text type: embedding max_tokens_per_chunk: 8192 @@ -448,7 +438,7 @@ # - https://cloud.google.com/vertex-ai/generative-ai/docs/model-garden/explore-models # - https://cloud.google.com/vertex-ai/generative-ai/pricing # - https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/gemini -- platform: vertexai +- provider: vertexai models: - name: gemini-1.5-pro-002 max_input_tokens: 2097152 @@ -470,7 +460,7 @@ supports_vision: true supports_function_calling: true - name: gemini-2.0-flash-thinking-exp-1219 - max_input_tokens: 32768 + max_input_tokens: 32760 max_output_tokens: 8192 supports_vision: true - name: claude-3-5-sonnet-v2@20241022 @@ -531,7 +521,7 @@ input_price: 0.3 output_price: 0.9 supports_function_calling: true - - name: text-embedding-004 + - name: text-embedding-005 type: embedding max_input_tokens: 20000 input_price: 0.025 @@ -550,7 +540,7 @@ # - https://docs.aws.amazon.com/bedrock/latest/userguide/model-ids.html#model-ids-arns # - https://aws.amazon.com/bedrock/pricing/ # - https://docs.aws.amazon.com/bedrock/latest/userguide/cross-region-inference-support.html -- platform: bedrock +- provider: bedrock models: - name: anthropic.claude-3-5-sonnet-20241022-v2:0 max_input_tokens: 200000 @@ -601,35 +591,35 @@ supports_vision: true supports_function_calling: true - name: us.meta.llama3-3-70b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.72 output_price: 0.72 supports_function_calling: true - name: meta.llama3-1-405b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 4096 require_max_tokens: true input_price: 2.4 output_price: 2.4 supports_function_calling: true - name: meta.llama3-1-70b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.72 output_price: 0.72 supports_function_calling: true - name: meta.llama3-1-8b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.22 output_price: 0.22 supports_function_calling: true - name: us.meta.llama3-2-90b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.72 @@ -637,7 +627,7 @@ supports_function_calling: true supports_vision: true - name: us.meta.llama3-2-11b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.16 @@ -702,7 +692,7 @@ # Links: # - https://developers.cloudflare.com/workers-ai/models/ # - https://developers.cloudflare.com/workers-ai/configuration/open-ai-compatibility/ -- platform: cloudflare +- provider: cloudflare models: - name: '@cf/meta/llama-3.3-70b-instruct-fp8-fast' max_input_tokens: 6144 @@ -738,7 +728,7 @@ # Links: # - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/Nlks5zkzu # - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/hlrk4akp7 -- platform: ernie +- provider: ernie models: - name: ernie-4.0-turbo-8k-latest max_input_tokens: 8192 @@ -781,10 +771,11 @@ max_input_tokens: 1024 input_price: 0.07 + # Links: # - https://help.aliyun.com/zh/model-studio/getting-started/models # - https://help.aliyun.com/zh/model-studio/developer-reference/use-qwen-by-calling-api -- platform: qianwen +- provider: qianwen models: - name: qwen-max-latest max_input_tokens: 30720 @@ -793,13 +784,13 @@ output_price: 8.4 supports_function_calling: true - name: qwen-plus-latest - max_input_tokens: 128000 + max_input_tokens: 129024 max_output_tokens: 8192 input_price: 0.112 output_price: 0.28 supports_function_calling: true - name: qwen-turbo-latest - max_input_tokens: 129024 + max_input_tokens: 1000000 max_output_tokens: 8192 input_price: 0.042 output_price: 0.084 @@ -831,12 +822,16 @@ output_price: 0.98 supports_function_calling: true - name: qwen-vl-max-latest - input_price: 2.8 - output_price: 2.8 + max_input_tokens: 30720 + max_output_tokens: 2048 + input_price: 0.42 + output_price: 1.26 supports_vision: true - name: qwen-vl-plus-latest - input_price: 1.12 - output_price: 1.12 + max_input_tokens: 30000 + max_output_tokens: 2048 + input_price: 0.21 + output_price: 0.63 supports_vision: true - name: qwen2.5-72b-instruct max_input_tokens: 129024 @@ -867,7 +862,7 @@ # - https://cloud.tencent.com/document/product/1729/104753 # - https://cloud.tencent.com/document/product/1729/97731 # - https://cloud.tencent.com/document/product/1729/111007 -- platform: hunyuan +- provider: hunyuan models: - name: hunyuan-turbo-latest max_input_tokens: 28000 @@ -878,10 +873,14 @@ - name: hunyuan-large max_input_tokens: 28000 max_output_tokens: 4096 + input_price: 0.56 + output_price: 1.68 supports_function_calling: true - name: hunyuan-large-longcontext max_input_tokens: 128000 max_output_tokens: 6144 + input_price: 0.84 + output_price: 2.52 supports_function_calling: true - name: hunyuan-standard max_input_tokens: 30000 @@ -927,38 +926,37 @@ max_batch_size: 100 # Links: -# - https://platform.moonshot.cn/docs/intro -# - https://platform.moonshot.cn/docs/pricing/chat -# - https://platform.moonshot.cn/docs/api/chat -- platform: moonshot +# - https://platform.moonshot.cn/docs/pricing/chat#%E8%AE%A1%E8%B4%B9%E5%9F%BA%E6%9C%AC%E6%A6%82%E5%BF%B5 +# - https://platform.moonshot.cn/docs/api/chat#%E5%85%AC%E5%BC%80%E7%9A%84%E6%9C%8D%E5%8A%A1%E5%9C%B0%E5%9D%80 +- provider: moonshot models: - name: moonshot-v1-8k - max_input_tokens: 8000 + max_input_tokens: 8192 input_price: 1.68 output_price: 1.68 supports_function_calling: true - name: moonshot-v1-32k - max_input_tokens: 32000 + max_input_tokens: 32768 input_price: 3.36 output_price: 3.36 supports_function_calling: true - name: moonshot-v1-128k - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 8.4 output_price: 8.4 supports_function_calling: true - name: moonshot-v1-8k-vision-preview - max_input_tokens: 8000 + max_input_tokens: 8192 input_price: 1.68 output_price: 1.68 supports_vision: true - name: moonshot-v1-32k-vision-preview - max_input_tokens: 32000 + max_input_tokens: 32768 input_price: 3.36 output_price: 3.36 supports_vision: true - name: moonshot-v1-128k-vision-preview - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 8.4 output_price: 8.4 supports_vision: true @@ -966,38 +964,46 @@ # Links: # - https://api-docs.deepseek.com/quick_start/pricing # - https://platform.deepseek.com/api-docs/api/create-chat-completion -- platform: deepseek +- provider: deepseek models: - name: deepseek-chat - max_input_tokens: 65536 + max_input_tokens: 64000 max_output_tokens: 8192 input_price: 0.14 output_price: 0.28 supports_function_calling: true + - name: deepseek-reasoner + max_input_tokens: 64000 + max_output_tokens: 8192 + input_price: 0.55 + output_price: 2.19 # Links: -# - https://open.bigmodel.cn/dev/howuse/model # - https://open.bigmodel.cn/pricing # - https://open.bigmodel.cn/dev/api#glm-4 -- platform: zhipuai +- provider: zhipuai models: - name: glm-4-plus max_input_tokens: 128000 + max_output_tokens: 4096 input_price: 7 output_price: 7 supports_function_calling: true - name: glm-4-alltools max_input_tokens: 128000 + max_output_tokens: 4096 input_price: 14 output_price: 14 supports_function_calling: true - name: glm-4-long max_input_tokens: 1000000 + max_output_tokens: 4096 input_price: 0.14 output_price: 0.14 supports_function_calling: true - name: glm-4-flash max_input_tokens: 128000 + max_output_tokens: 4096 input_price: 0 output_price: 0 supports_function_calling: true @@ -1011,6 +1017,10 @@ input_price: 0 output_price: 0 supports_vision: true + - name: glm-zero-preview + max_input_tokens: 16384 + input_price: 1.4 + output_price: 1.4 - name: embedding-3 type: embedding max_input_tokens: 8192 @@ -1021,7 +1031,7 @@ # Links: # - https://platform.lingyiwanwu.com/docs#%E6%A8%A1%E5%9E%8B%E4%B8%8E%E8%AE%A1%E8%B4%B9 # - https://platform.lingyiwanwu.com/docs/api-reference#create-chat-completion -- platform: lingyiwanwu +- provider: lingyiwanwu models: - name: yi-lightning max_input_tokens: 16384 @@ -1036,7 +1046,7 @@ # Links: # - https://platform.minimaxi.com/document/Price # - https://platform.minimaxi.com/document/ChatCompletion%20v2 -- platform: minimax +- provider: minimax models: - name: minimax-text-01 max_input_tokens: 1000192 @@ -1052,273 +1062,8 @@ # supports_function_calling: true # Links: -# - https://github.com/marketplace/models -- platform: github - models: - - name: gpt-4o - max_input_tokens: 128000 - supports_function_calling: true - - name: gpt-4o-mini - max_input_tokens: 128000 - supports_function_calling: true - - name: o1 - max_input_tokens: 128000 - supports_function_calling: true - supports_vision: true - no_stream: true - no_system_message: true - - name: o1-preview - max_input_tokens: 128000 - no_stream: true - no_system_message: true - - name: o1-mini - max_input_tokens: 128000 - no_stream: true - no_system_message: true - - name: text-embedding-3-large - type: embedding - max_tokens_per_chunk: 8191 - default_chunk_size: 2000 - max_batch_size: 100 - - name: text-embedding-3-small - type: embedding - max_tokens_per_chunk: 8191 - default_chunk_size: 2000 - max_batch_size: 100 - - name: llama-3.3-70b-instruct - max_input_tokens: 128000 - - name: meta-llama-3.1-405b-instruct - max_input_tokens: 128000 - - name: meta-llama-3.1-70b-instruct - max_input_tokens: 128000 - - name: meta-llama-3.1-8b-instruct - max_input_tokens: 128000 - - name: llama-3.2-90b-vision-instruct - max_input_tokens: 8192 - supports_vision: true - - name: llama-3.2-11b-vision-instruct - max_input_tokens: 8192 - supports_vision: true - - name: mistral-large-2411 - max_input_tokens: 128000 - supports_function_calling: true - - name: codestral-2501 - max_input_tokens: 256000 - supports_function_calling: true - - name: cohere-command-r-plus-08-2024 - max_input_tokens: 128000 - supports_function_calling: true - - name: cohere-command-r-08-2024 - max_input_tokens: 128000 - supports_function_calling: true - - name: cohere-embed-v3-english - type: embedding - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 96 - - name: cohere-embed-v3-multilingual - type: embedding - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 96 - - name: ai21-jamba-1.5-large - max_input_tokens: 256000 - supports_function_calling: true - - name: ai21-jamba-1.5-mini - max_input_tokens: 256000 - supports_function_calling: true - - name: phi-4 - max_input_tokens: 16384 - - name: phi-3.5-moe-instruct - max_input_tokens: 128000 - - name: phi-3.5-mini-instruct - max_input_tokens: 128000 - - name: phi-3.5-vision-instruct - max_input_tokens: 128000 - supports_vision: true - -# Links: -# - https://deepinfra.com/models -- platform: deepinfra - models: - - name: meta-llama/Llama-3.3-70B-Instruct - max_input_tokens: 128000 - input_price: 0.23 - output_price: 0.40 - - name: meta-llama/Meta-Llama-3.1-405B-Instruct - max_input_tokens: 32000 - input_price: 0.8 - output_price: 0.8 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-70B-Instruct - max_input_tokens: 128000 - input_price: 0.23 - output_price: 0.4 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-8B-Instruct - max_input_tokens: 128000 - input_price: 0.03 - output_price: 0.05 - supports_function_calling: true - - name: meta-llama/Llama-3.2-90B-Vision-Instruct - max_input_tokens: 128000 - input_price: 0.35 - output_price: 0.4 - - name: meta-llama/Llama-3.2-11B-Vision-Instruct - max_input_tokens: 128000 - input_price: 0.055 - output_price: 0.055 - - name: mistralai/Mistral-Nemo-Instruct-2407 - max_input_tokens: 128000 - input_price: 0.035 - output_price: 0.08 - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.27 - output_price: 0.27 - - name: google/gemma-2-9b-it - max_input_tokens: 8192 - input_price: 0.03 - output_price: 0.06 - - name: Qwen/Qwen2.5-72B-Instruct - max_input_tokens: 32768 - input_price: 0.23 - output_price: 0.40 - supports_function_calling: true - - name: Qwen/Qwen2.5-Coder-32B-Instruct - max_input_tokens: 32768 - input_price: 0.07 - output_price: 0.16 - - name: Qwen/QVQ-72B-Preview - max_input_tokens: 32768 - input_price: 0.25 - output_price: 0.50 - supports_vision: true - - name: Qwen/QwQ-32B-Preview - max_input_tokens: 32768 - input_price: 0.12 - output_price: 0.18 - - name: deepseek-ai/DeepSeek-V3 - max_input_tokens: 32768 - input_price: 0.85 - output_price: 0.9 - - name: microsoft/phi-4 - max_input_tokens: 16384 - input_price: 0.07 - output_price: 0.14 - - name: nvidia/Llama-3.1-Nemotron-70B-Instruct - max_input_tokens: 128000 - input_price: 0.12 - output_price: 0.30 - supports_function_calling: true - - name: BAAI/bge-large-en-v1.5 - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: BAAI/bge-m3 - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 8192 - default_chunk_size: 2000 - max_batch_size: 100 - - name: intfloat/e5-large-v2 - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: intfloat/multilingual-e5-large - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: thenlper/gte-large - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - -# Links: -# - https://fireworks.ai/models -# - https://fireworks.ai/pricing -- platform: fireworks - models: - - name: accounts/fireworks/models/llama-v3p3-70b-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/llama-v3p1-405b-instruct - max_input_tokens: 131072 - input_price: 3 - output_price: 3 - supports_function_calling: true - - name: accounts/fireworks/models/llama-v3p1-70b-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - supports_function_calling: true - - name: accounts/fireworks/models/llama-v3p1-8b-instruct - max_input_tokens: 131072 - input_price: 0.2 - output_price: 0.2 - - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - supports_vision: true - - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct - max_input_tokens: 131072 - input_price: 0.2 - output_price: 0.2 - supports_vision: true - - name: accounts/fireworks/models/qwen2p5-72b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - supports_function_calling: true - - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/qwen-qwq-32b-preview - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/qwen2-vl-72b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - supports_vision: true - - name: accounts/fireworks/models/deepseek-v3 - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: nomic-ai/nomic-embed-text-v1.5 - type: embedding - input_price: 0.008 - max_tokens_per_chunk: 8192 - default_chunk_size: 1500 - max_batch_size: 100 - - name: WhereIsAI/UAE-Large-V1 - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: thenlper/gte-large - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - -# Links: # - https://openrouter.ai/models -- platform: openrouter +- provider: openrouter models: - name: openai/gpt-4o max_input_tokens: 128000 @@ -1396,14 +1141,6 @@ output_price: 0.15 supports_vision: true supports_function_calling: true - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.27 - output_price: 0.27 - - name: google/gemma-2-9b-it - max_input_tokens: 4096 - input_price: 0.06 - output_price: 0.06 - name: anthropic/claude-3.5-sonnet max_input_tokens: 200000 max_output_tokens: 8192 @@ -1449,7 +1186,7 @@ input_price: 0.12 output_price: 0.3 - name: meta-llama/llama-3.1-405b-instruct - max_input_tokens: 131072 + max_input_tokens: 32768 input_price: 0.8 output_price: 0.8 supports_function_calling: true @@ -1528,10 +1265,14 @@ input_price: 0.0375 output_price: 0.15 - name: deepseek/deepseek-chat - max_input_tokens: 32768 + max_input_tokens: 64000 input_price: 0.14 output_price: 0.28 supports_function_calling: true + - name: deepseek/deepseek-r1 + max_input_tokens: 163840 + input_price: 0.55 + output_price: 2.19 - name: perplexity/llama-3.1-sonar-huge-128k-online max_input_tokens: 127072 input_price: 5 @@ -1544,12 +1285,8 @@ max_input_tokens: 127072 input_price: 0.2 output_price: 0.2 - - name: 01-ai/yi-large - max_input_tokens: 32768 - input_price: 3 - output_price: 3 - name: microsoft/phi-4 - max_input_tokens: 16000 + max_input_tokens: 16384 input_price: 0.07 output_price: 0.14 - name: microsoft/phi-3.5-mini-128k-instruct @@ -1578,11 +1315,6 @@ input_price: 0.25 output_price: 0.5 supports_vision: true - - name: nvidia/llama-3.1-nemotron-70b-instruct - max_input_tokens: 131072 - input_price: 0.35 - output_price: 0.4 - supports_function_calling: true - name: x-ai/grok-2-1212 max_input_tokens: 131072 input_price: 2 @@ -1626,10 +1358,262 @@ input_price: 0.2 output_price: 1.1 + +# Links: +# - https://github.com/marketplace/models +- provider: github + models: + - name: gpt-4o + max_input_tokens: 128000 + supports_function_calling: true + - name: gpt-4o-mini + max_input_tokens: 128000 + supports_function_calling: true + - name: o1 + max_input_tokens: 200000 + supports_function_calling: true + supports_vision: true + no_stream: true + no_system_message: true + - name: o1-preview + max_input_tokens: 128000 + no_stream: true + no_system_message: true + - name: o1-mini + max_input_tokens: 128000 + no_stream: true + no_system_message: true + - name: text-embedding-3-large + type: embedding + max_tokens_per_chunk: 8191 + default_chunk_size: 2000 + max_batch_size: 100 + - name: text-embedding-3-small + type: embedding + max_tokens_per_chunk: 8191 + default_chunk_size: 2000 + max_batch_size: 100 + - name: llama-3.3-70b-instruct + max_input_tokens: 131072 + - name: meta-llama-3.1-405b-instruct + max_input_tokens: 131072 + - name: meta-llama-3.1-70b-instruct + max_input_tokens: 131072 + - name: meta-llama-3.1-8b-instruct + max_input_tokens: 131072 + - name: llama-3.2-90b-vision-instruct + max_input_tokens: 131072 + supports_vision: true + - name: llama-3.2-11b-vision-instruct + max_input_tokens: 131072 + supports_vision: true + - name: mistral-large-2411 + max_input_tokens: 128000 + supports_function_calling: true + - name: codestral-2501 + max_input_tokens: 256000 + supports_function_calling: true + - name: cohere-command-r-plus-08-2024 + max_input_tokens: 128000 + supports_function_calling: true + - name: cohere-command-r-08-2024 + max_input_tokens: 128000 + supports_function_calling: true + - name: cohere-embed-v3-english + type: embedding + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 96 + - name: cohere-embed-v3-multilingual + type: embedding + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 96 + - name: ai21-jamba-1.5-large + max_input_tokens: 256000 + supports_function_calling: true + - name: ai21-jamba-1.5-mini + max_input_tokens: 256000 + supports_function_calling: true + - name: phi-4 + max_input_tokens: 16384 + - name: phi-3.5-moe-instruct + max_input_tokens: 128000 + - name: phi-3.5-mini-instruct + max_input_tokens: 128000 + - name: phi-3.5-vision-instruct + max_input_tokens: 128000 + supports_vision: true + +# Links: +# - https://deepinfra.com/models +- provider: deepinfra + models: + - name: meta-llama/Llama-3.3-70B-Instruct + max_input_tokens: 131072 + input_price: 0.23 + output_price: 0.40 + - name: meta-llama/Meta-Llama-3.1-405B-Instruct + max_input_tokens: 32768 + input_price: 0.8 + output_price: 0.8 + supports_function_calling: true + - name: meta-llama/Meta-Llama-3.1-70B-Instruct + max_input_tokens: 131072 + input_price: 0.23 + output_price: 0.4 + supports_function_calling: true + - name: meta-llama/Meta-Llama-3.1-8B-Instruct + max_input_tokens: 131072 + input_price: 0.03 + output_price: 0.05 + supports_function_calling: true + - name: meta-llama/Llama-3.2-90B-Vision-Instruct + max_input_tokens: 131072 + input_price: 0.35 + output_price: 0.4 + - name: meta-llama/Llama-3.2-11B-Vision-Instruct + max_input_tokens: 131072 + input_price: 0.055 + output_price: 0.055 + - name: Qwen/Qwen2.5-72B-Instruct + max_input_tokens: 32768 + input_price: 0.23 + output_price: 0.40 + supports_function_calling: true + - name: Qwen/Qwen2.5-Coder-32B-Instruct + max_input_tokens: 32768 + input_price: 0.07 + output_price: 0.16 + - name: Qwen/QVQ-72B-Preview + max_input_tokens: 32768 + input_price: 0.25 + output_price: 0.50 + supports_vision: true + - name: Qwen/QwQ-32B-Preview + max_input_tokens: 32768 + input_price: 0.12 + output_price: 0.18 + - name: deepseek-ai/DeepSeek-V3 + max_input_tokens: 32768 + input_price: 0.85 + output_price: 0.9 + - name: microsoft/phi-4 + max_input_tokens: 16384 + input_price: 0.07 + output_price: 0.14 + - name: BAAI/bge-large-en-v1.5 + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: BAAI/bge-m3 + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 8192 + default_chunk_size: 2000 + max_batch_size: 100 + - name: intfloat/e5-large-v2 + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: intfloat/multilingual-e5-large + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: thenlper/gte-large + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + +# Links: +# - https://fireworks.ai/models +# - https://fireworks.ai/pricing +- provider: fireworks + models: + - name: accounts/fireworks/models/llama-v3p3-70b-instruct + max_input_tokens: 131072 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/llama-v3p1-405b-instruct + max_input_tokens: 131072 + input_price: 3 + output_price: 3 + supports_function_calling: true + - name: accounts/fireworks/models/llama-v3p1-70b-instruct + max_input_tokens: 131072 + input_price: 0.9 + output_price: 0.9 + supports_function_calling: true + - name: accounts/fireworks/models/llama-v3p1-8b-instruct + max_input_tokens: 131072 + input_price: 0.2 + output_price: 0.2 + - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct + max_input_tokens: 131072 + input_price: 0.9 + output_price: 0.9 + supports_vision: true + - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct + max_input_tokens: 131072 + input_price: 0.2 + output_price: 0.2 + supports_vision: true + - name: accounts/fireworks/models/qwen2p5-72b-instruct + max_input_tokens: 32768 + input_price: 0.9 + output_price: 0.9 + supports_function_calling: true + - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct + max_input_tokens: 32768 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/qwen-qwq-32b-preview + max_input_tokens: 32768 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/qwen2-vl-72b-instruct + max_input_tokens: 32768 + input_price: 0.9 + output_price: 0.9 + supports_vision: true + - name: accounts/fireworks/models/deepseek-v3 + max_input_tokens: 131072 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/deepseek-r1 + max_input_tokens: 160000 + input_price: 8 + output_price: 8 + - name: nomic-ai/nomic-embed-text-v1.5 + type: embedding + input_price: 0.008 + max_tokens_per_chunk: 8192 + default_chunk_size: 1500 + max_batch_size: 100 + - name: WhereIsAI/UAE-Large-V1 + type: embedding + input_price: 0.016 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: thenlper/gte-large + type: embedding + input_price: 0.016 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 # Links # - https://cloud.siliconflow.cn/models # - https://docs.siliconflow.cn/api-reference/chat-completions/chat-completions -- platform: siliconflow +- provider: siliconflow models: - name: meta-llama/Llama-3.3-70B-Instruct max_input_tokens: 32768 @@ -1653,7 +1637,7 @@ output_price: 0.578 supports_function_calling: true - name: Qwen/Qwen2.5-72B-Instruct-128K - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0.578 output_price: 0.578 supports_function_calling: true @@ -1684,14 +1668,6 @@ max_input_tokens: 32768 input_price: 0.176 output_price: 0.176 - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.176 - output_price: 0.176 - - name: google/gemma-2-9b-it - max_input_tokens: 8192 - input_price: 0 - output_price: 0 - name: deepseek-ai/DeepSeek-V2.5 max_input_tokens: 32768 input_price: 0.186 @@ -1728,25 +1704,25 @@ # Links: # - https://docs.together.ai/docs/serverless-models # - https://www.together.ai/pricing -- platform: together +- provider: together models: - name: meta-llama/Llama-3.3-70B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 131072 input_price: 0.88 output_price: 0.88 supports_function_calling: true - name: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 130815 input_price: 3.5 output_price: 3.5 supports_function_calling: true - name: meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 131072 input_price: 0.88 output_price: 0.88 supports_function_calling: true - name: meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 131072 input_price: 0.18 output_price: 0.18 supports_function_calling: true @@ -1760,14 +1736,6 @@ input_price: 0.18 output_price: 0.18 supports_vision: true - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.8 - output_price: 0.8 - - name: google/gemma-2-9b-it - max_input_tokens: 8192 - input_price: 0.3 - output_price: 0.3 - name: Qwen/Qwen2.5-72B-Instruct-Turbo max_input_tokens: 32768 input_price: 1.2 @@ -1777,7 +1745,7 @@ input_price: 0.3 output_price: 0.3 - name: Qwen/Qwen2.5-Coder-32B-Instruct - max_input_tokens: 16384 + max_input_tokens: 32768 input_price: 0.8 output_price: 0.8 - name: Qwen/QwQ-32B-Preview @@ -1793,6 +1761,10 @@ max_input_tokens: 131072 input_price: 1.25 output_price: 1.25 + - name: deepseek-ai/DeepSeek-R1 + max_input_tokens: 163840 + input_price: 7 + output_price: 7 - name: WhereIsAI/UAE-Large-V1 type: embedding input_price: 0.016 @@ -1811,9 +1783,9 @@ input_price: 0.1 # Links: -# - https://jina.ai/ +# - https://jina.ai/models # - https://api.jina.ai/redoc -- platform: jina +- provider: jina models: - name: jina-embeddings-v3 type: embedding @@ -1846,7 +1818,7 @@ # - https://docs.voyageai.com/docs/embeddings # - https://docs.voyageai.com/docs/pricing # - https://docs.voyageai.com/reference/ -- platform: voyageai +- provider: voyageai models: - name: voyage-3-large type: embedding |
