summaryrefslogtreecommitdiffstats
path: root/models.yaml
diff options
context:
space:
mode:
Diffstat (limited to 'models.yaml')
-rw-r--r--models.yaml734
1 files changed, 353 insertions, 381 deletions
diff --git a/models.yaml b/models.yaml
index 60184de..6b505f1 100644
--- a/models.yaml
+++ b/models.yaml
@@ -1,11 +1,8 @@
-# Notes:
-# - do not submit pull requests to add new models; this list will be updated in batches with new releases.
-
# Links:
# - https://platform.openai.com/docs/models
# - https://openai.com/api/pricing/
# - https://platform.openai.com/docs/api-reference/chat
-- platform: openai
+- provider: openai
models:
- name: gpt-4o
max_input_tokens: 128000
@@ -50,7 +47,7 @@
supports_vision: true
supports_function_calling: true
- name: o1
- max_input_tokens: 128000
+ max_input_tokens: 200000
input_price: 15
output_price: 60
supports_vision: true
@@ -91,7 +88,7 @@
# - https://ai.google.dev/models/gemini
# - https://ai.google.dev/pricing
# - https://ai.google.dev/api/rest/v1beta/models/streamGenerateContent
-- platform: gemini
+- provider: gemini
models:
- name: gemini-1.5-pro-latest
max_input_tokens: 2097152
@@ -122,13 +119,13 @@
supports_vision: true
supports_function_calling: true
- name: gemini-2.0-flash-thinking-exp
- max_input_tokens: 32768
+ max_input_tokens: 32767
max_output_tokens: 8192
input_price: 0
output_price: 0
supports_vision: true
- name: gemini-exp-1206
- max_input_tokens: 32768
+ max_input_tokens: 2097152
max_output_tokens: 8192
input_price: 0
output_price: 0
@@ -144,7 +141,7 @@
# Links:
# - https://docs.anthropic.com/claude/docs/models-overview
# - https://docs.anthropic.com/claude/reference/messages-streaming
-- platform: claude
+- provider: claude
models:
- name: claude-3-5-sonnet-latest
max_input_tokens: 200000
@@ -207,7 +204,7 @@
# - https://docs.mistral.ai/getting-started/models/models_overview/
# - https://mistral.ai/technology/#pricing
# - https://docs.mistral.ai/api/
-- platform: mistral
+- provider: mistral
models:
- name: mistral-large-latest
max_input_tokens: 128000
@@ -250,7 +247,7 @@
# - https://docs.ai21.com/docs/jamba-15-models
# - https://www.ai21.com/pricing
# - https://docs.ai21.com/reference/jamba-15-api-ref
-- platform: ai21
+- provider: ai21
models:
- name: jamba-1.5-large
max_input_tokens: 256000
@@ -267,7 +264,7 @@
# - https://docs.cohere.com/docs/command-r-plus
# - https://cohere.com/pricing
# - https://docs.cohere.com/reference/chat
-- platform: cohere
+- provider: cohere
models:
- name: command-r-plus-08-2024
max_input_tokens: 128000
@@ -322,7 +319,7 @@
# Links:
# - https://docs.x.ai/docs/models
-- platform: xai
+- provider: xai
models:
- name: grok-2-latest
max_input_tokens: 131072
@@ -361,7 +358,7 @@
# - https://docs.perplexity.ai/guides/model-cards
# - https://docs.perplexity.ai/guides/pricing
# - https://docs.perplexity.ai/api-reference/chat-completions
-- platform: perplexity
+- provider: perplexity
models:
- name: llama-3.1-sonar-huge-128k-online
max_input_tokens: 127072
@@ -379,64 +376,57 @@
# Links:
# - https://console.groq.com/docs/models
# - https://console.groq.com/docs/api-reference#chat
-- platform: groq
+- provider: groq
models:
- name: llama-3.3-70b-versatile
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_function_calling: true
- name: llama-3.1-8b-instant
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_function_calling: true
- name: llama-3.2-90b-vision-preview
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_vision: true
- name: llama-3.2-11b-vision-preview
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0
output_price: 0
supports_vision: true
- - name: gemma2-9b-it
- max_input_tokens: 8192
- input_price: 0
- output_price: 0
- supports_function_calling: true
# Links:
# - https://ollama.com/library
# - https://github.com/ollama/ollama/blob/main/docs/openai.md
-- platform: ollama
+- provider: ollama
models:
- name: llama3.1
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: llama3.2
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: llama3.2-vision
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_vision: true
- name: llama3.3
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: qwq
max_input_tokens: 32768
supports_function_calling: true
- name: qwen2.5
- max_input_tokens: 128000
+ max_input_tokens: 131072
supports_function_calling: true
- name: qwen2.5-coder
max_input_tokens: 32768
supports_function_calling: true
- name: phi4
max_input_tokens: 16384
- - name: gemma2
- max_input_tokens: 8192
- name: nomic-embed-text
type: embedding
max_tokens_per_chunk: 8192
@@ -448,7 +438,7 @@
# - https://cloud.google.com/vertex-ai/generative-ai/docs/model-garden/explore-models
# - https://cloud.google.com/vertex-ai/generative-ai/pricing
# - https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/gemini
-- platform: vertexai
+- provider: vertexai
models:
- name: gemini-1.5-pro-002
max_input_tokens: 2097152
@@ -470,7 +460,7 @@
supports_vision: true
supports_function_calling: true
- name: gemini-2.0-flash-thinking-exp-1219
- max_input_tokens: 32768
+ max_input_tokens: 32760
max_output_tokens: 8192
supports_vision: true
- name: claude-3-5-sonnet-v2@20241022
@@ -531,7 +521,7 @@
input_price: 0.3
output_price: 0.9
supports_function_calling: true
- - name: text-embedding-004
+ - name: text-embedding-005
type: embedding
max_input_tokens: 20000
input_price: 0.025
@@ -550,7 +540,7 @@
# - https://docs.aws.amazon.com/bedrock/latest/userguide/model-ids.html#model-ids-arns
# - https://aws.amazon.com/bedrock/pricing/
# - https://docs.aws.amazon.com/bedrock/latest/userguide/cross-region-inference-support.html
-- platform: bedrock
+- provider: bedrock
models:
- name: anthropic.claude-3-5-sonnet-20241022-v2:0
max_input_tokens: 200000
@@ -601,35 +591,35 @@
supports_vision: true
supports_function_calling: true
- name: us.meta.llama3-3-70b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.72
output_price: 0.72
supports_function_calling: true
- name: meta.llama3-1-405b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 4096
require_max_tokens: true
input_price: 2.4
output_price: 2.4
supports_function_calling: true
- name: meta.llama3-1-70b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.72
output_price: 0.72
supports_function_calling: true
- name: meta.llama3-1-8b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.22
output_price: 0.22
supports_function_calling: true
- name: us.meta.llama3-2-90b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.72
@@ -637,7 +627,7 @@
supports_function_calling: true
supports_vision: true
- name: us.meta.llama3-2-11b-instruct-v1:0
- max_input_tokens: 128000
+ max_input_tokens: 131072
max_output_tokens: 8192
require_max_tokens: true
input_price: 0.16
@@ -702,7 +692,7 @@
# Links:
# - https://developers.cloudflare.com/workers-ai/models/
# - https://developers.cloudflare.com/workers-ai/configuration/open-ai-compatibility/
-- platform: cloudflare
+- provider: cloudflare
models:
- name: '@cf/meta/llama-3.3-70b-instruct-fp8-fast'
max_input_tokens: 6144
@@ -738,7 +728,7 @@
# Links:
# - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/Nlks5zkzu
# - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/hlrk4akp7
-- platform: ernie
+- provider: ernie
models:
- name: ernie-4.0-turbo-8k-latest
max_input_tokens: 8192
@@ -781,10 +771,11 @@
max_input_tokens: 1024
input_price: 0.07
+
# Links:
# - https://help.aliyun.com/zh/model-studio/getting-started/models
# - https://help.aliyun.com/zh/model-studio/developer-reference/use-qwen-by-calling-api
-- platform: qianwen
+- provider: qianwen
models:
- name: qwen-max-latest
max_input_tokens: 30720
@@ -793,13 +784,13 @@
output_price: 8.4
supports_function_calling: true
- name: qwen-plus-latest
- max_input_tokens: 128000
+ max_input_tokens: 129024
max_output_tokens: 8192
input_price: 0.112
output_price: 0.28
supports_function_calling: true
- name: qwen-turbo-latest
- max_input_tokens: 129024
+ max_input_tokens: 1000000
max_output_tokens: 8192
input_price: 0.042
output_price: 0.084
@@ -831,12 +822,16 @@
output_price: 0.98
supports_function_calling: true
- name: qwen-vl-max-latest
- input_price: 2.8
- output_price: 2.8
+ max_input_tokens: 30720
+ max_output_tokens: 2048
+ input_price: 0.42
+ output_price: 1.26
supports_vision: true
- name: qwen-vl-plus-latest
- input_price: 1.12
- output_price: 1.12
+ max_input_tokens: 30000
+ max_output_tokens: 2048
+ input_price: 0.21
+ output_price: 0.63
supports_vision: true
- name: qwen2.5-72b-instruct
max_input_tokens: 129024
@@ -867,7 +862,7 @@
# - https://cloud.tencent.com/document/product/1729/104753
# - https://cloud.tencent.com/document/product/1729/97731
# - https://cloud.tencent.com/document/product/1729/111007
-- platform: hunyuan
+- provider: hunyuan
models:
- name: hunyuan-turbo-latest
max_input_tokens: 28000
@@ -878,10 +873,14 @@
- name: hunyuan-large
max_input_tokens: 28000
max_output_tokens: 4096
+ input_price: 0.56
+ output_price: 1.68
supports_function_calling: true
- name: hunyuan-large-longcontext
max_input_tokens: 128000
max_output_tokens: 6144
+ input_price: 0.84
+ output_price: 2.52
supports_function_calling: true
- name: hunyuan-standard
max_input_tokens: 30000
@@ -927,38 +926,37 @@
max_batch_size: 100
# Links:
-# - https://platform.moonshot.cn/docs/intro
-# - https://platform.moonshot.cn/docs/pricing/chat
-# - https://platform.moonshot.cn/docs/api/chat
-- platform: moonshot
+# - https://platform.moonshot.cn/docs/pricing/chat#%E8%AE%A1%E8%B4%B9%E5%9F%BA%E6%9C%AC%E6%A6%82%E5%BF%B5
+# - https://platform.moonshot.cn/docs/api/chat#%E5%85%AC%E5%BC%80%E7%9A%84%E6%9C%8D%E5%8A%A1%E5%9C%B0%E5%9D%80
+- provider: moonshot
models:
- name: moonshot-v1-8k
- max_input_tokens: 8000
+ max_input_tokens: 8192
input_price: 1.68
output_price: 1.68
supports_function_calling: true
- name: moonshot-v1-32k
- max_input_tokens: 32000
+ max_input_tokens: 32768
input_price: 3.36
output_price: 3.36
supports_function_calling: true
- name: moonshot-v1-128k
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 8.4
output_price: 8.4
supports_function_calling: true
- name: moonshot-v1-8k-vision-preview
- max_input_tokens: 8000
+ max_input_tokens: 8192
input_price: 1.68
output_price: 1.68
supports_vision: true
- name: moonshot-v1-32k-vision-preview
- max_input_tokens: 32000
+ max_input_tokens: 32768
input_price: 3.36
output_price: 3.36
supports_vision: true
- name: moonshot-v1-128k-vision-preview
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 8.4
output_price: 8.4
supports_vision: true
@@ -966,38 +964,46 @@
# Links:
# - https://api-docs.deepseek.com/quick_start/pricing
# - https://platform.deepseek.com/api-docs/api/create-chat-completion
-- platform: deepseek
+- provider: deepseek
models:
- name: deepseek-chat
- max_input_tokens: 65536
+ max_input_tokens: 64000
max_output_tokens: 8192
input_price: 0.14
output_price: 0.28
supports_function_calling: true
+ - name: deepseek-reasoner
+ max_input_tokens: 64000
+ max_output_tokens: 8192
+ input_price: 0.55
+ output_price: 2.19
# Links:
-# - https://open.bigmodel.cn/dev/howuse/model
# - https://open.bigmodel.cn/pricing
# - https://open.bigmodel.cn/dev/api#glm-4
-- platform: zhipuai
+- provider: zhipuai
models:
- name: glm-4-plus
max_input_tokens: 128000
+ max_output_tokens: 4096
input_price: 7
output_price: 7
supports_function_calling: true
- name: glm-4-alltools
max_input_tokens: 128000
+ max_output_tokens: 4096
input_price: 14
output_price: 14
supports_function_calling: true
- name: glm-4-long
max_input_tokens: 1000000
+ max_output_tokens: 4096
input_price: 0.14
output_price: 0.14
supports_function_calling: true
- name: glm-4-flash
max_input_tokens: 128000
+ max_output_tokens: 4096
input_price: 0
output_price: 0
supports_function_calling: true
@@ -1011,6 +1017,10 @@
input_price: 0
output_price: 0
supports_vision: true
+ - name: glm-zero-preview
+ max_input_tokens: 16384
+ input_price: 1.4
+ output_price: 1.4
- name: embedding-3
type: embedding
max_input_tokens: 8192
@@ -1021,7 +1031,7 @@
# Links:
# - https://platform.lingyiwanwu.com/docs#%E6%A8%A1%E5%9E%8B%E4%B8%8E%E8%AE%A1%E8%B4%B9
# - https://platform.lingyiwanwu.com/docs/api-reference#create-chat-completion
-- platform: lingyiwanwu
+- provider: lingyiwanwu
models:
- name: yi-lightning
max_input_tokens: 16384
@@ -1036,7 +1046,7 @@
# Links:
# - https://platform.minimaxi.com/document/Price
# - https://platform.minimaxi.com/document/ChatCompletion%20v2
-- platform: minimax
+- provider: minimax
models:
- name: minimax-text-01
max_input_tokens: 1000192
@@ -1052,273 +1062,8 @@
# supports_function_calling: true
# Links:
-# - https://github.com/marketplace/models
-- platform: github
- models:
- - name: gpt-4o
- max_input_tokens: 128000
- supports_function_calling: true
- - name: gpt-4o-mini
- max_input_tokens: 128000
- supports_function_calling: true
- - name: o1
- max_input_tokens: 128000
- supports_function_calling: true
- supports_vision: true
- no_stream: true
- no_system_message: true
- - name: o1-preview
- max_input_tokens: 128000
- no_stream: true
- no_system_message: true
- - name: o1-mini
- max_input_tokens: 128000
- no_stream: true
- no_system_message: true
- - name: text-embedding-3-large
- type: embedding
- max_tokens_per_chunk: 8191
- default_chunk_size: 2000
- max_batch_size: 100
- - name: text-embedding-3-small
- type: embedding
- max_tokens_per_chunk: 8191
- default_chunk_size: 2000
- max_batch_size: 100
- - name: llama-3.3-70b-instruct
- max_input_tokens: 128000
- - name: meta-llama-3.1-405b-instruct
- max_input_tokens: 128000
- - name: meta-llama-3.1-70b-instruct
- max_input_tokens: 128000
- - name: meta-llama-3.1-8b-instruct
- max_input_tokens: 128000
- - name: llama-3.2-90b-vision-instruct
- max_input_tokens: 8192
- supports_vision: true
- - name: llama-3.2-11b-vision-instruct
- max_input_tokens: 8192
- supports_vision: true
- - name: mistral-large-2411
- max_input_tokens: 128000
- supports_function_calling: true
- - name: codestral-2501
- max_input_tokens: 256000
- supports_function_calling: true
- - name: cohere-command-r-plus-08-2024
- max_input_tokens: 128000
- supports_function_calling: true
- - name: cohere-command-r-08-2024
- max_input_tokens: 128000
- supports_function_calling: true
- - name: cohere-embed-v3-english
- type: embedding
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 96
- - name: cohere-embed-v3-multilingual
- type: embedding
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 96
- - name: ai21-jamba-1.5-large
- max_input_tokens: 256000
- supports_function_calling: true
- - name: ai21-jamba-1.5-mini
- max_input_tokens: 256000
- supports_function_calling: true
- - name: phi-4
- max_input_tokens: 16384
- - name: phi-3.5-moe-instruct
- max_input_tokens: 128000
- - name: phi-3.5-mini-instruct
- max_input_tokens: 128000
- - name: phi-3.5-vision-instruct
- max_input_tokens: 128000
- supports_vision: true
-
-# Links:
-# - https://deepinfra.com/models
-- platform: deepinfra
- models:
- - name: meta-llama/Llama-3.3-70B-Instruct
- max_input_tokens: 128000
- input_price: 0.23
- output_price: 0.40
- - name: meta-llama/Meta-Llama-3.1-405B-Instruct
- max_input_tokens: 32000
- input_price: 0.8
- output_price: 0.8
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-70B-Instruct
- max_input_tokens: 128000
- input_price: 0.23
- output_price: 0.4
- supports_function_calling: true
- - name: meta-llama/Meta-Llama-3.1-8B-Instruct
- max_input_tokens: 128000
- input_price: 0.03
- output_price: 0.05
- supports_function_calling: true
- - name: meta-llama/Llama-3.2-90B-Vision-Instruct
- max_input_tokens: 128000
- input_price: 0.35
- output_price: 0.4
- - name: meta-llama/Llama-3.2-11B-Vision-Instruct
- max_input_tokens: 128000
- input_price: 0.055
- output_price: 0.055
- - name: mistralai/Mistral-Nemo-Instruct-2407
- max_input_tokens: 128000
- input_price: 0.035
- output_price: 0.08
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.27
- output_price: 0.27
- - name: google/gemma-2-9b-it
- max_input_tokens: 8192
- input_price: 0.03
- output_price: 0.06
- - name: Qwen/Qwen2.5-72B-Instruct
- max_input_tokens: 32768
- input_price: 0.23
- output_price: 0.40
- supports_function_calling: true
- - name: Qwen/Qwen2.5-Coder-32B-Instruct
- max_input_tokens: 32768
- input_price: 0.07
- output_price: 0.16
- - name: Qwen/QVQ-72B-Preview
- max_input_tokens: 32768
- input_price: 0.25
- output_price: 0.50
- supports_vision: true
- - name: Qwen/QwQ-32B-Preview
- max_input_tokens: 32768
- input_price: 0.12
- output_price: 0.18
- - name: deepseek-ai/DeepSeek-V3
- max_input_tokens: 32768
- input_price: 0.85
- output_price: 0.9
- - name: microsoft/phi-4
- max_input_tokens: 16384
- input_price: 0.07
- output_price: 0.14
- - name: nvidia/Llama-3.1-Nemotron-70B-Instruct
- max_input_tokens: 128000
- input_price: 0.12
- output_price: 0.30
- supports_function_calling: true
- - name: BAAI/bge-large-en-v1.5
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: BAAI/bge-m3
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 8192
- default_chunk_size: 2000
- max_batch_size: 100
- - name: intfloat/e5-large-v2
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: intfloat/multilingual-e5-large
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: thenlper/gte-large
- type: embedding
- input_price: 0.01
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
-
-# Links:
-# - https://fireworks.ai/models
-# - https://fireworks.ai/pricing
-- platform: fireworks
- models:
- - name: accounts/fireworks/models/llama-v3p3-70b-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/llama-v3p1-405b-instruct
- max_input_tokens: 131072
- input_price: 3
- output_price: 3
- supports_function_calling: true
- - name: accounts/fireworks/models/llama-v3p1-70b-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- supports_function_calling: true
- - name: accounts/fireworks/models/llama-v3p1-8b-instruct
- max_input_tokens: 131072
- input_price: 0.2
- output_price: 0.2
- - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- supports_vision: true
- - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct
- max_input_tokens: 131072
- input_price: 0.2
- output_price: 0.2
- supports_vision: true
- - name: accounts/fireworks/models/qwen2p5-72b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- supports_function_calling: true
- - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/qwen-qwq-32b-preview
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- - name: accounts/fireworks/models/qwen2-vl-72b-instruct
- max_input_tokens: 32768
- input_price: 0.9
- output_price: 0.9
- supports_vision: true
- - name: accounts/fireworks/models/deepseek-v3
- max_input_tokens: 131072
- input_price: 0.9
- output_price: 0.9
- - name: nomic-ai/nomic-embed-text-v1.5
- type: embedding
- input_price: 0.008
- max_tokens_per_chunk: 8192
- default_chunk_size: 1500
- max_batch_size: 100
- - name: WhereIsAI/UAE-Large-V1
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
- - name: thenlper/gte-large
- type: embedding
- input_price: 0.016
- max_tokens_per_chunk: 512
- default_chunk_size: 1000
- max_batch_size: 100
-
-# Links:
# - https://openrouter.ai/models
-- platform: openrouter
+- provider: openrouter
models:
- name: openai/gpt-4o
max_input_tokens: 128000
@@ -1396,14 +1141,6 @@
output_price: 0.15
supports_vision: true
supports_function_calling: true
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.27
- output_price: 0.27
- - name: google/gemma-2-9b-it
- max_input_tokens: 4096
- input_price: 0.06
- output_price: 0.06
- name: anthropic/claude-3.5-sonnet
max_input_tokens: 200000
max_output_tokens: 8192
@@ -1449,7 +1186,7 @@
input_price: 0.12
output_price: 0.3
- name: meta-llama/llama-3.1-405b-instruct
- max_input_tokens: 131072
+ max_input_tokens: 32768
input_price: 0.8
output_price: 0.8
supports_function_calling: true
@@ -1528,10 +1265,14 @@
input_price: 0.0375
output_price: 0.15
- name: deepseek/deepseek-chat
- max_input_tokens: 32768
+ max_input_tokens: 64000
input_price: 0.14
output_price: 0.28
supports_function_calling: true
+ - name: deepseek/deepseek-r1
+ max_input_tokens: 163840
+ input_price: 0.55
+ output_price: 2.19
- name: perplexity/llama-3.1-sonar-huge-128k-online
max_input_tokens: 127072
input_price: 5
@@ -1544,12 +1285,8 @@
max_input_tokens: 127072
input_price: 0.2
output_price: 0.2
- - name: 01-ai/yi-large
- max_input_tokens: 32768
- input_price: 3
- output_price: 3
- name: microsoft/phi-4
- max_input_tokens: 16000
+ max_input_tokens: 16384
input_price: 0.07
output_price: 0.14
- name: microsoft/phi-3.5-mini-128k-instruct
@@ -1578,11 +1315,6 @@
input_price: 0.25
output_price: 0.5
supports_vision: true
- - name: nvidia/llama-3.1-nemotron-70b-instruct
- max_input_tokens: 131072
- input_price: 0.35
- output_price: 0.4
- supports_function_calling: true
- name: x-ai/grok-2-1212
max_input_tokens: 131072
input_price: 2
@@ -1626,10 +1358,262 @@
input_price: 0.2
output_price: 1.1
+
+# Links:
+# - https://github.com/marketplace/models
+- provider: github
+ models:
+ - name: gpt-4o
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: gpt-4o-mini
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: o1
+ max_input_tokens: 200000
+ supports_function_calling: true
+ supports_vision: true
+ no_stream: true
+ no_system_message: true
+ - name: o1-preview
+ max_input_tokens: 128000
+ no_stream: true
+ no_system_message: true
+ - name: o1-mini
+ max_input_tokens: 128000
+ no_stream: true
+ no_system_message: true
+ - name: text-embedding-3-large
+ type: embedding
+ max_tokens_per_chunk: 8191
+ default_chunk_size: 2000
+ max_batch_size: 100
+ - name: text-embedding-3-small
+ type: embedding
+ max_tokens_per_chunk: 8191
+ default_chunk_size: 2000
+ max_batch_size: 100
+ - name: llama-3.3-70b-instruct
+ max_input_tokens: 131072
+ - name: meta-llama-3.1-405b-instruct
+ max_input_tokens: 131072
+ - name: meta-llama-3.1-70b-instruct
+ max_input_tokens: 131072
+ - name: meta-llama-3.1-8b-instruct
+ max_input_tokens: 131072
+ - name: llama-3.2-90b-vision-instruct
+ max_input_tokens: 131072
+ supports_vision: true
+ - name: llama-3.2-11b-vision-instruct
+ max_input_tokens: 131072
+ supports_vision: true
+ - name: mistral-large-2411
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: codestral-2501
+ max_input_tokens: 256000
+ supports_function_calling: true
+ - name: cohere-command-r-plus-08-2024
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: cohere-command-r-08-2024
+ max_input_tokens: 128000
+ supports_function_calling: true
+ - name: cohere-embed-v3-english
+ type: embedding
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 96
+ - name: cohere-embed-v3-multilingual
+ type: embedding
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 96
+ - name: ai21-jamba-1.5-large
+ max_input_tokens: 256000
+ supports_function_calling: true
+ - name: ai21-jamba-1.5-mini
+ max_input_tokens: 256000
+ supports_function_calling: true
+ - name: phi-4
+ max_input_tokens: 16384
+ - name: phi-3.5-moe-instruct
+ max_input_tokens: 128000
+ - name: phi-3.5-mini-instruct
+ max_input_tokens: 128000
+ - name: phi-3.5-vision-instruct
+ max_input_tokens: 128000
+ supports_vision: true
+
+# Links:
+# - https://deepinfra.com/models
+- provider: deepinfra
+ models:
+ - name: meta-llama/Llama-3.3-70B-Instruct
+ max_input_tokens: 131072
+ input_price: 0.23
+ output_price: 0.40
+ - name: meta-llama/Meta-Llama-3.1-405B-Instruct
+ max_input_tokens: 32768
+ input_price: 0.8
+ output_price: 0.8
+ supports_function_calling: true
+ - name: meta-llama/Meta-Llama-3.1-70B-Instruct
+ max_input_tokens: 131072
+ input_price: 0.23
+ output_price: 0.4
+ supports_function_calling: true
+ - name: meta-llama/Meta-Llama-3.1-8B-Instruct
+ max_input_tokens: 131072
+ input_price: 0.03
+ output_price: 0.05
+ supports_function_calling: true
+ - name: meta-llama/Llama-3.2-90B-Vision-Instruct
+ max_input_tokens: 131072
+ input_price: 0.35
+ output_price: 0.4
+ - name: meta-llama/Llama-3.2-11B-Vision-Instruct
+ max_input_tokens: 131072
+ input_price: 0.055
+ output_price: 0.055
+ - name: Qwen/Qwen2.5-72B-Instruct
+ max_input_tokens: 32768
+ input_price: 0.23
+ output_price: 0.40
+ supports_function_calling: true
+ - name: Qwen/Qwen2.5-Coder-32B-Instruct
+ max_input_tokens: 32768
+ input_price: 0.07
+ output_price: 0.16
+ - name: Qwen/QVQ-72B-Preview
+ max_input_tokens: 32768
+ input_price: 0.25
+ output_price: 0.50
+ supports_vision: true
+ - name: Qwen/QwQ-32B-Preview
+ max_input_tokens: 32768
+ input_price: 0.12
+ output_price: 0.18
+ - name: deepseek-ai/DeepSeek-V3
+ max_input_tokens: 32768
+ input_price: 0.85
+ output_price: 0.9
+ - name: microsoft/phi-4
+ max_input_tokens: 16384
+ input_price: 0.07
+ output_price: 0.14
+ - name: BAAI/bge-large-en-v1.5
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: BAAI/bge-m3
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 8192
+ default_chunk_size: 2000
+ max_batch_size: 100
+ - name: intfloat/e5-large-v2
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: intfloat/multilingual-e5-large
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: thenlper/gte-large
+ type: embedding
+ input_price: 0.01
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+
+# Links:
+# - https://fireworks.ai/models
+# - https://fireworks.ai/pricing
+- provider: fireworks
+ models:
+ - name: accounts/fireworks/models/llama-v3p3-70b-instruct
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/llama-v3p1-405b-instruct
+ max_input_tokens: 131072
+ input_price: 3
+ output_price: 3
+ supports_function_calling: true
+ - name: accounts/fireworks/models/llama-v3p1-70b-instruct
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ supports_function_calling: true
+ - name: accounts/fireworks/models/llama-v3p1-8b-instruct
+ max_input_tokens: 131072
+ input_price: 0.2
+ output_price: 0.2
+ - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ supports_vision: true
+ - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct
+ max_input_tokens: 131072
+ input_price: 0.2
+ output_price: 0.2
+ supports_vision: true
+ - name: accounts/fireworks/models/qwen2p5-72b-instruct
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ supports_function_calling: true
+ - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/qwen-qwq-32b-preview
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/qwen2-vl-72b-instruct
+ max_input_tokens: 32768
+ input_price: 0.9
+ output_price: 0.9
+ supports_vision: true
+ - name: accounts/fireworks/models/deepseek-v3
+ max_input_tokens: 131072
+ input_price: 0.9
+ output_price: 0.9
+ - name: accounts/fireworks/models/deepseek-r1
+ max_input_tokens: 160000
+ input_price: 8
+ output_price: 8
+ - name: nomic-ai/nomic-embed-text-v1.5
+ type: embedding
+ input_price: 0.008
+ max_tokens_per_chunk: 8192
+ default_chunk_size: 1500
+ max_batch_size: 100
+ - name: WhereIsAI/UAE-Large-V1
+ type: embedding
+ input_price: 0.016
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
+ - name: thenlper/gte-large
+ type: embedding
+ input_price: 0.016
+ max_tokens_per_chunk: 512
+ default_chunk_size: 1000
+ max_batch_size: 100
# Links
# - https://cloud.siliconflow.cn/models
# - https://docs.siliconflow.cn/api-reference/chat-completions/chat-completions
-- platform: siliconflow
+- provider: siliconflow
models:
- name: meta-llama/Llama-3.3-70B-Instruct
max_input_tokens: 32768
@@ -1653,7 +1637,7 @@
output_price: 0.578
supports_function_calling: true
- name: Qwen/Qwen2.5-72B-Instruct-128K
- max_input_tokens: 128000
+ max_input_tokens: 131072
input_price: 0.578
output_price: 0.578
supports_function_calling: true
@@ -1684,14 +1668,6 @@
max_input_tokens: 32768
input_price: 0.176
output_price: 0.176
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.176
- output_price: 0.176
- - name: google/gemma-2-9b-it
- max_input_tokens: 8192
- input_price: 0
- output_price: 0
- name: deepseek-ai/DeepSeek-V2.5
max_input_tokens: 32768
input_price: 0.186
@@ -1728,25 +1704,25 @@
# Links:
# - https://docs.together.ai/docs/serverless-models
# - https://www.together.ai/pricing
-- platform: together
+- provider: together
models:
- name: meta-llama/Llama-3.3-70B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 131072
input_price: 0.88
output_price: 0.88
supports_function_calling: true
- name: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 130815
input_price: 3.5
output_price: 3.5
supports_function_calling: true
- name: meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 131072
input_price: 0.88
output_price: 0.88
supports_function_calling: true
- name: meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo
- max_input_tokens: 32768
+ max_input_tokens: 131072
input_price: 0.18
output_price: 0.18
supports_function_calling: true
@@ -1760,14 +1736,6 @@
input_price: 0.18
output_price: 0.18
supports_vision: true
- - name: google/gemma-2-27b-it
- max_input_tokens: 8192
- input_price: 0.8
- output_price: 0.8
- - name: google/gemma-2-9b-it
- max_input_tokens: 8192
- input_price: 0.3
- output_price: 0.3
- name: Qwen/Qwen2.5-72B-Instruct-Turbo
max_input_tokens: 32768
input_price: 1.2
@@ -1777,7 +1745,7 @@
input_price: 0.3
output_price: 0.3
- name: Qwen/Qwen2.5-Coder-32B-Instruct
- max_input_tokens: 16384
+ max_input_tokens: 32768
input_price: 0.8
output_price: 0.8
- name: Qwen/QwQ-32B-Preview
@@ -1793,6 +1761,10 @@
max_input_tokens: 131072
input_price: 1.25
output_price: 1.25
+ - name: deepseek-ai/DeepSeek-R1
+ max_input_tokens: 163840
+ input_price: 7
+ output_price: 7
- name: WhereIsAI/UAE-Large-V1
type: embedding
input_price: 0.016
@@ -1811,9 +1783,9 @@
input_price: 0.1
# Links:
-# - https://jina.ai/
+# - https://jina.ai/models
# - https://api.jina.ai/redoc
-- platform: jina
+- provider: jina
models:
- name: jina-embeddings-v3
type: embedding
@@ -1846,7 +1818,7 @@
# - https://docs.voyageai.com/docs/embeddings
# - https://docs.voyageai.com/docs/pricing
# - https://docs.voyageai.com/reference/
-- platform: voyageai
+- provider: voyageai
models:
- name: voyage-3-large
type: embedding