From e0417e8d5bebe25476aaafa22a9ee23d9bd61457 Mon Sep 17 00:00:00 2001 From: sigoden Date: Wed, 22 Jan 2025 20:51:10 +0800 Subject: feat: add `--sync-models` cli option (#1114) --- Argcfile.sh | 79 +-- config.example.yaml | 2 + models.yaml | 1090 +++++++++++++++++++-------------------- scripts/completions/aichat.bash | 2 +- scripts/completions/aichat.fish | 1 + scripts/completions/aichat.nu | 2 +- scripts/completions/aichat.ps1 | 1 + scripts/completions/aichat.zsh | 1 + src/cli.rs | 3 + src/client/common.rs | 13 +- src/client/macros.rs | 8 +- src/client/mod.rs | 2 +- src/client/model.rs | 23 +- src/client/openai_compatible.rs | 2 +- src/config/input.rs | 2 +- src/config/mod.rs | 118 +++-- src/main.rs | 6 + src/utils/loader.rs | 2 +- src/utils/request.rs | 12 +- 19 files changed, 711 insertions(+), 658 deletions(-) diff --git a/Argcfile.sh b/Argcfile.sh index 43ebb59..c8450f7 100755 --- a/Argcfile.sh +++ b/Argcfile.sh @@ -4,7 +4,7 @@ set -e # @meta dotenv # @env DRY_RUN Dry run mode -# @cmd Test first running +# @cmd Test configuration initialization # @env AICHAT_CONFIG_DIR=tmp/test-init-config # @arg args~ test-init-config() { @@ -17,10 +17,13 @@ test-init-config() { cargo run -- "$@" } -# @cmd Test running with AICHAT_PLATFORM environment variable -# @env AICHAT_PLATFORM! +# @cmd Test running without configuration file +# @env AICHAT_PROVIDER! +# @env AICHAT_CONFIG_DIR=tmp/test-provider-env # @arg args~ -test-platform-env() { +test-no-config() { + mkdir -p "$AICHAT_CONFIG_DIR" + rm -rf "$AICHAT_CONFIG_DIR/config.yaml" cargo run -- "$@" } @@ -80,27 +83,27 @@ test-server() { # @cmd Chat with any LLM api # @flag -S --no-stream -# @arg platform_model![?`_choice_platform_model`] +# @arg provider_model![?`_choice_provider_model`] # @arg text~ chat() { - if [[ "$argc_platform_model" == *':'* ]]; then - model="${argc_platform_model##*:}" - argc_platform="${argc_platform_model%:*}" + if [[ "$argc_provider_model" == *':'* ]]; then + model="${argc_provider_model##*:}" + argc_provider="${argc_provider_model%:*}" else - argc_platform="${argc_platform_model}" + argc_provider="${argc_provider_model}" fi - for platform_config in "${OPENAI_COMPATIBLE_PLATFORMS[@]}"; do - if [[ "$argc_platform" == "${platform_config%%,*}" ]]; then + for provider_config in "${OPENAI_COMPATIBLE_PROVIDERS[@]}"; do + if [[ "$argc_provider" == "${provider_config%%,*}" ]]; then _retrieve_api_base break fi done if [[ -n "$api_base" ]]; then - env_prefix="$(echo "$argc_platform" | tr '[:lower:]' '[:upper:]')" + env_prefix="$(echo "$argc_provider" | tr '[:lower:]' '[:upper:]')" api_key_env="${env_prefix}_API_KEY" api_key="${!api_key_env}" if [[ -z "$model" ]]; then - model="$(echo "$platform_config" | cut -d, -f2)" + model="$(echo "$provider_config" | cut -d, -f2)" fi if [[ -z "$model" ]]; then model_env="${env_prefix}_MODEL" @@ -112,27 +115,27 @@ chat() { --model "$model" \ "${argc_text[@]}" else - argc chat-$argc_platform "${argc_text[@]}" + argc chat-$argc_provider "${argc_text[@]}" fi } # @cmd List models by openai-compatible api # @flag --name-only Print model name only -# @arg platform![`_choice_platform`] +# @arg provider![`_choice_provider`] models() { - for platform_config in "${OPENAI_COMPATIBLE_PLATFORMS[@]}"; do - if [[ "$argc_platform" == "${platform_config%%,*}" ]]; then + for provider_config in "${OPENAI_COMPATIBLE_PROVIDERS[@]}"; do + if [[ "$argc_provider" == "${provider_config%%,*}" ]]; then _retrieve_api_base break fi done if [[ -n "$api_base" ]]; then - env_prefix="$(echo "$argc_platform" | tr '[:lower:]' '[:upper:]')" + env_prefix="$(echo "$argc_provider" | tr '[:lower:]' '[:upper:]')" api_key_env="${env_prefix}_API_KEY" api_key="${!api_key_env}" jq_args=() if [[ -n "$argc_name_only" ]]; then - case "$argc_platform" in + case "$argc_provider" in cloudflare) jq_args+=(-r '.result[].name') ;; @@ -149,14 +152,14 @@ models() { fi _openai_compatible_models | jq "${jq_args[@]}" else - if ! cat "$0" | grep -q "^models-$argc_platform"; then - _die "error: platform '$argc_platform' does not have a models api" + if ! cat "$0" | grep -q "^models-$argc_provider"; then + _die "error: provider '$argc_provider' does not have a models api" fi cli_args=() if [[ -n "$argc_name_only" ]]; then cli_args+=(--name-only) fi - argc models-$argc_platform "${cli_args[@]}" + argc models-$argc_provider "${cli_args[@]}" fi } @@ -202,7 +205,7 @@ chat-azure-openai() { # @cmd Chat with gemini api # @env GEMINI_API_KEY! -# @option -m --model=gemini-1.0-pro-latest $GEMINI_MODEL +# @option -m --model=gemini-1.5-pro-latest $GEMINI_MODEL # @flag -S --no-stream # @arg text~ chat-gemini() { @@ -246,7 +249,7 @@ chat-claude() { # @cmd Chat with cohere api # @env COHERE_API_KEY! -# @option -m --model=command-r $COHERE_MODEL +# @option -m --model=command-r-08-2024 $COHERE_MODEL # @flag -S --no-stream # @arg text~ chat-cohere() { @@ -274,7 +277,7 @@ models-cohere() { # @env require-tools gcloud # @env VERTEXAI_PROJECT_ID! # @env VERTEXAI_LOCATION! -# @option -m --model=gemini-1.0-pro $VERTEXAI_GEMINI_MODEL +# @option -m --model=gemini-1.5-flash-002 $VERTEXAI_GEMINI_MODEL # @flag -S --no-stream # @arg text~ chat-vertexai() { @@ -307,7 +310,7 @@ chat-ernie() { } _argc_before() { - OPENAI_COMPATIBLE_PLATFORMS=( \ + OPENAI_COMPATIBLE_PROVIDERS=( \ openai,gpt-4o-mini,https://api.openai.com/v1 \ ai21,jamba-1.5-mini,https://api.ai21.com/studio/v1 \ cloudflare,@cf/meta/llama-3.1-8b-instruct,https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1 \ @@ -317,7 +320,7 @@ _argc_before() { github,gpt-4o-mini,https://models.inference.ai.azure.com \ groq,llama-3.1-8b-instant,https://api.groq.com/openai/v1 \ hunyuan,hunyuan-large,https://api.hunyuan.cloud.tencent.com/v1 \ - lingyiwanwu,yi-large,https://api.lingyiwanwu.com/v1 \ + lingyiwanwu,yi-lightning,https://api.lingyiwanwu.com/v1 \ minimax,MiniMax-Text-01,https://api.minimax.chat/v1 \ mistral,mistral-small-latest,https://api.mistral.ai/v1 \ moonshot,moonshot-v1-8k,https://api.moonshot.cn/v1 \ @@ -341,7 +344,7 @@ _openai_compatible_models() { api_base="${api_base:-"$argc_api_base"}" api_key="${api_key:-"$argc_api_key"}" url="${api_base}/models" - if [[ "$argc_platform" == "cloudflare" ]]; then + if [[ "$argc_provider" == "cloudflare" ]]; then url="https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/models/search" fi @@ -351,12 +354,12 @@ _openai_compatible_models() { } _retrieve_api_base() { - api_base="${platform_config##*,}" + api_base="${provider_config##*,}" if [[ -z "$api_base" ]]; then - key="$(echo $argc_platform | tr '[:lower:]' '[:upper:]')_API_BASE" + key="$(echo $argc_provider | tr '[:lower:]' '[:upper:]')_API_BASE" api_base="${!key}" if [[ -z "$api_base" ]]; then - _die "error: miss api_base for $argc_platform; please set $key" + _die "error: miss api_base for $argc_provider; please set $key" fi fi } @@ -365,23 +368,23 @@ _choice_model() { aichat --list-models } -_choice_platform_model() { - _choice_platform +_choice_provider_model() { + _choice_provider _choice_model } -_choice_platform() { +_choice_provider() { _choice_client - _choice_openai_compatible_platform + _choice_openai_compatible_provider } _choice_client() { printf "%s\n" gemini claude cohere azure-openai vertexai bedrock ernie } -_choice_openai_compatible_platform() { - for platform_config in "${OPENAI_COMPATIBLE_PLATFORMS[@]}"; do - echo "${platform_config%%,*}" +_choice_openai_compatible_provider() { + for provider_config in "${OPENAI_COMPATIBLE_PROVIDERS[@]}"; do + echo "${provider_config%%,*}" done } diff --git a/config.example.yaml b/config.example.yaml index ebf8853..e80c34e 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -82,6 +82,8 @@ right_prompt: serve_addr: 127.0.0.1:8000 # Default serve listening address user_agent: null # Set User-Agent HTTP header, use `auto` for aichat/ save_shell_history: true # Whether to save shell execution command to the history file +# Sync models changes from the specified URL, using a CDN link rather than a direct GitHub raw link due to the higher availability and reliability of the CDN. +sync_models_url: https://cdn.jsdelivr.net/gh/sigoden/aichat/models.yaml # Or https://raw.githubusercontent.com/sigoden/aichat/refs/heads/main/models.yaml # ---- clients ---- clients: diff --git a/models.yaml b/models.yaml index 60184de..6b505f1 100644 --- a/models.yaml +++ b/models.yaml @@ -1,11 +1,8 @@ -# Notes: -# - do not submit pull requests to add new models; this list will be updated in batches with new releases. - # Links: # - https://platform.openai.com/docs/models # - https://openai.com/api/pricing/ # - https://platform.openai.com/docs/api-reference/chat -- platform: openai +- provider: openai models: - name: gpt-4o max_input_tokens: 128000 @@ -50,7 +47,7 @@ supports_vision: true supports_function_calling: true - name: o1 - max_input_tokens: 128000 + max_input_tokens: 200000 input_price: 15 output_price: 60 supports_vision: true @@ -91,7 +88,7 @@ # - https://ai.google.dev/models/gemini # - https://ai.google.dev/pricing # - https://ai.google.dev/api/rest/v1beta/models/streamGenerateContent -- platform: gemini +- provider: gemini models: - name: gemini-1.5-pro-latest max_input_tokens: 2097152 @@ -122,13 +119,13 @@ supports_vision: true supports_function_calling: true - name: gemini-2.0-flash-thinking-exp - max_input_tokens: 32768 + max_input_tokens: 32767 max_output_tokens: 8192 input_price: 0 output_price: 0 supports_vision: true - name: gemini-exp-1206 - max_input_tokens: 32768 + max_input_tokens: 2097152 max_output_tokens: 8192 input_price: 0 output_price: 0 @@ -144,7 +141,7 @@ # Links: # - https://docs.anthropic.com/claude/docs/models-overview # - https://docs.anthropic.com/claude/reference/messages-streaming -- platform: claude +- provider: claude models: - name: claude-3-5-sonnet-latest max_input_tokens: 200000 @@ -207,7 +204,7 @@ # - https://docs.mistral.ai/getting-started/models/models_overview/ # - https://mistral.ai/technology/#pricing # - https://docs.mistral.ai/api/ -- platform: mistral +- provider: mistral models: - name: mistral-large-latest max_input_tokens: 128000 @@ -250,7 +247,7 @@ # - https://docs.ai21.com/docs/jamba-15-models # - https://www.ai21.com/pricing # - https://docs.ai21.com/reference/jamba-15-api-ref -- platform: ai21 +- provider: ai21 models: - name: jamba-1.5-large max_input_tokens: 256000 @@ -267,7 +264,7 @@ # - https://docs.cohere.com/docs/command-r-plus # - https://cohere.com/pricing # - https://docs.cohere.com/reference/chat -- platform: cohere +- provider: cohere models: - name: command-r-plus-08-2024 max_input_tokens: 128000 @@ -322,7 +319,7 @@ # Links: # - https://docs.x.ai/docs/models -- platform: xai +- provider: xai models: - name: grok-2-latest max_input_tokens: 131072 @@ -361,7 +358,7 @@ # - https://docs.perplexity.ai/guides/model-cards # - https://docs.perplexity.ai/guides/pricing # - https://docs.perplexity.ai/api-reference/chat-completions -- platform: perplexity +- provider: perplexity models: - name: llama-3.1-sonar-huge-128k-online max_input_tokens: 127072 @@ -379,64 +376,57 @@ # Links: # - https://console.groq.com/docs/models # - https://console.groq.com/docs/api-reference#chat -- platform: groq +- provider: groq models: - name: llama-3.3-70b-versatile - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_function_calling: true - name: llama-3.1-8b-instant - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_function_calling: true - name: llama-3.2-90b-vision-preview - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_vision: true - name: llama-3.2-11b-vision-preview - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0 output_price: 0 supports_vision: true - - name: gemma2-9b-it - max_input_tokens: 8192 - input_price: 0 - output_price: 0 - supports_function_calling: true # Links: # - https://ollama.com/library # - https://github.com/ollama/ollama/blob/main/docs/openai.md -- platform: ollama +- provider: ollama models: - name: llama3.1 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: llama3.2 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: llama3.2-vision - max_input_tokens: 128000 + max_input_tokens: 131072 supports_vision: true - name: llama3.3 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: qwq max_input_tokens: 32768 supports_function_calling: true - name: qwen2.5 - max_input_tokens: 128000 + max_input_tokens: 131072 supports_function_calling: true - name: qwen2.5-coder max_input_tokens: 32768 supports_function_calling: true - name: phi4 max_input_tokens: 16384 - - name: gemma2 - max_input_tokens: 8192 - name: nomic-embed-text type: embedding max_tokens_per_chunk: 8192 @@ -448,7 +438,7 @@ # - https://cloud.google.com/vertex-ai/generative-ai/docs/model-garden/explore-models # - https://cloud.google.com/vertex-ai/generative-ai/pricing # - https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/gemini -- platform: vertexai +- provider: vertexai models: - name: gemini-1.5-pro-002 max_input_tokens: 2097152 @@ -470,7 +460,7 @@ supports_vision: true supports_function_calling: true - name: gemini-2.0-flash-thinking-exp-1219 - max_input_tokens: 32768 + max_input_tokens: 32760 max_output_tokens: 8192 supports_vision: true - name: claude-3-5-sonnet-v2@20241022 @@ -531,7 +521,7 @@ input_price: 0.3 output_price: 0.9 supports_function_calling: true - - name: text-embedding-004 + - name: text-embedding-005 type: embedding max_input_tokens: 20000 input_price: 0.025 @@ -550,7 +540,7 @@ # - https://docs.aws.amazon.com/bedrock/latest/userguide/model-ids.html#model-ids-arns # - https://aws.amazon.com/bedrock/pricing/ # - https://docs.aws.amazon.com/bedrock/latest/userguide/cross-region-inference-support.html -- platform: bedrock +- provider: bedrock models: - name: anthropic.claude-3-5-sonnet-20241022-v2:0 max_input_tokens: 200000 @@ -601,35 +591,35 @@ supports_vision: true supports_function_calling: true - name: us.meta.llama3-3-70b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.72 output_price: 0.72 supports_function_calling: true - name: meta.llama3-1-405b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 4096 require_max_tokens: true input_price: 2.4 output_price: 2.4 supports_function_calling: true - name: meta.llama3-1-70b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.72 output_price: 0.72 supports_function_calling: true - name: meta.llama3-1-8b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.22 output_price: 0.22 supports_function_calling: true - name: us.meta.llama3-2-90b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.72 @@ -637,7 +627,7 @@ supports_function_calling: true supports_vision: true - name: us.meta.llama3-2-11b-instruct-v1:0 - max_input_tokens: 128000 + max_input_tokens: 131072 max_output_tokens: 8192 require_max_tokens: true input_price: 0.16 @@ -702,7 +692,7 @@ # Links: # - https://developers.cloudflare.com/workers-ai/models/ # - https://developers.cloudflare.com/workers-ai/configuration/open-ai-compatibility/ -- platform: cloudflare +- provider: cloudflare models: - name: '@cf/meta/llama-3.3-70b-instruct-fp8-fast' max_input_tokens: 6144 @@ -738,7 +728,7 @@ # Links: # - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/Nlks5zkzu # - https://cloud.baidu.com/doc/WENXINWORKSHOP/s/hlrk4akp7 -- platform: ernie +- provider: ernie models: - name: ernie-4.0-turbo-8k-latest max_input_tokens: 8192 @@ -781,10 +771,11 @@ max_input_tokens: 1024 input_price: 0.07 + # Links: # - https://help.aliyun.com/zh/model-studio/getting-started/models # - https://help.aliyun.com/zh/model-studio/developer-reference/use-qwen-by-calling-api -- platform: qianwen +- provider: qianwen models: - name: qwen-max-latest max_input_tokens: 30720 @@ -793,13 +784,13 @@ output_price: 8.4 supports_function_calling: true - name: qwen-plus-latest - max_input_tokens: 128000 + max_input_tokens: 129024 max_output_tokens: 8192 input_price: 0.112 output_price: 0.28 supports_function_calling: true - name: qwen-turbo-latest - max_input_tokens: 129024 + max_input_tokens: 1000000 max_output_tokens: 8192 input_price: 0.042 output_price: 0.084 @@ -831,12 +822,16 @@ output_price: 0.98 supports_function_calling: true - name: qwen-vl-max-latest - input_price: 2.8 - output_price: 2.8 + max_input_tokens: 30720 + max_output_tokens: 2048 + input_price: 0.42 + output_price: 1.26 supports_vision: true - name: qwen-vl-plus-latest - input_price: 1.12 - output_price: 1.12 + max_input_tokens: 30000 + max_output_tokens: 2048 + input_price: 0.21 + output_price: 0.63 supports_vision: true - name: qwen2.5-72b-instruct max_input_tokens: 129024 @@ -867,7 +862,7 @@ # - https://cloud.tencent.com/document/product/1729/104753 # - https://cloud.tencent.com/document/product/1729/97731 # - https://cloud.tencent.com/document/product/1729/111007 -- platform: hunyuan +- provider: hunyuan models: - name: hunyuan-turbo-latest max_input_tokens: 28000 @@ -878,10 +873,14 @@ - name: hunyuan-large max_input_tokens: 28000 max_output_tokens: 4096 + input_price: 0.56 + output_price: 1.68 supports_function_calling: true - name: hunyuan-large-longcontext max_input_tokens: 128000 max_output_tokens: 6144 + input_price: 0.84 + output_price: 2.52 supports_function_calling: true - name: hunyuan-standard max_input_tokens: 30000 @@ -927,38 +926,37 @@ max_batch_size: 100 # Links: -# - https://platform.moonshot.cn/docs/intro -# - https://platform.moonshot.cn/docs/pricing/chat -# - https://platform.moonshot.cn/docs/api/chat -- platform: moonshot +# - https://platform.moonshot.cn/docs/pricing/chat#%E8%AE%A1%E8%B4%B9%E5%9F%BA%E6%9C%AC%E6%A6%82%E5%BF%B5 +# - https://platform.moonshot.cn/docs/api/chat#%E5%85%AC%E5%BC%80%E7%9A%84%E6%9C%8D%E5%8A%A1%E5%9C%B0%E5%9D%80 +- provider: moonshot models: - name: moonshot-v1-8k - max_input_tokens: 8000 + max_input_tokens: 8192 input_price: 1.68 output_price: 1.68 supports_function_calling: true - name: moonshot-v1-32k - max_input_tokens: 32000 + max_input_tokens: 32768 input_price: 3.36 output_price: 3.36 supports_function_calling: true - name: moonshot-v1-128k - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 8.4 output_price: 8.4 supports_function_calling: true - name: moonshot-v1-8k-vision-preview - max_input_tokens: 8000 + max_input_tokens: 8192 input_price: 1.68 output_price: 1.68 supports_vision: true - name: moonshot-v1-32k-vision-preview - max_input_tokens: 32000 + max_input_tokens: 32768 input_price: 3.36 output_price: 3.36 supports_vision: true - name: moonshot-v1-128k-vision-preview - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 8.4 output_price: 8.4 supports_vision: true @@ -966,38 +964,46 @@ # Links: # - https://api-docs.deepseek.com/quick_start/pricing # - https://platform.deepseek.com/api-docs/api/create-chat-completion -- platform: deepseek +- provider: deepseek models: - name: deepseek-chat - max_input_tokens: 65536 + max_input_tokens: 64000 max_output_tokens: 8192 input_price: 0.14 output_price: 0.28 supports_function_calling: true + - name: deepseek-reasoner + max_input_tokens: 64000 + max_output_tokens: 8192 + input_price: 0.55 + output_price: 2.19 # Links: -# - https://open.bigmodel.cn/dev/howuse/model # - https://open.bigmodel.cn/pricing # - https://open.bigmodel.cn/dev/api#glm-4 -- platform: zhipuai +- provider: zhipuai models: - name: glm-4-plus max_input_tokens: 128000 + max_output_tokens: 4096 input_price: 7 output_price: 7 supports_function_calling: true - name: glm-4-alltools max_input_tokens: 128000 + max_output_tokens: 4096 input_price: 14 output_price: 14 supports_function_calling: true - name: glm-4-long max_input_tokens: 1000000 + max_output_tokens: 4096 input_price: 0.14 output_price: 0.14 supports_function_calling: true - name: glm-4-flash max_input_tokens: 128000 + max_output_tokens: 4096 input_price: 0 output_price: 0 supports_function_calling: true @@ -1011,6 +1017,10 @@ input_price: 0 output_price: 0 supports_vision: true + - name: glm-zero-preview + max_input_tokens: 16384 + input_price: 1.4 + output_price: 1.4 - name: embedding-3 type: embedding max_input_tokens: 8192 @@ -1021,7 +1031,7 @@ # Links: # - https://platform.lingyiwanwu.com/docs#%E6%A8%A1%E5%9E%8B%E4%B8%8E%E8%AE%A1%E8%B4%B9 # - https://platform.lingyiwanwu.com/docs/api-reference#create-chat-completion -- platform: lingyiwanwu +- provider: lingyiwanwu models: - name: yi-lightning max_input_tokens: 16384 @@ -1036,7 +1046,7 @@ # Links: # - https://platform.minimaxi.com/document/Price # - https://platform.minimaxi.com/document/ChatCompletion%20v2 -- platform: minimax +- provider: minimax models: - name: minimax-text-01 max_input_tokens: 1000192 @@ -1052,404 +1062,131 @@ # supports_function_calling: true # Links: -# - https://github.com/marketplace/models -- platform: github +# - https://openrouter.ai/models +- provider: openrouter models: - - name: gpt-4o + - name: openai/gpt-4o max_input_tokens: 128000 + input_price: 2.5 + output_price: 10 + supports_vision: true supports_function_calling: true - - name: gpt-4o-mini + - name: openai/gpt-4o-2024-11-20 max_input_tokens: 128000 + input_price: 2.5 + output_price: 10 + supports_vision: true supports_function_calling: true - - name: o1 + - name: openai/gpt-4o-2024-08-06 max_input_tokens: 128000 - supports_function_calling: true + input_price: 2.5 + output_price: 10 supports_vision: true - no_stream: true - no_system_message: true - - name: o1-preview + supports_function_calling: true + - name: openai/chatgpt-4o-latest max_input_tokens: 128000 - no_stream: true - no_system_message: true - - name: o1-mini + input_price: 5 + output_price: 15 + supports_vision: true + supports_function_calling: true + - name: openai/gpt-4o-mini max_input_tokens: 128000 - no_stream: true - no_system_message: true - - name: text-embedding-3-large - type: embedding - max_tokens_per_chunk: 8191 - default_chunk_size: 2000 - max_batch_size: 100 - - name: text-embedding-3-small - type: embedding - max_tokens_per_chunk: 8191 - default_chunk_size: 2000 - max_batch_size: 100 - - name: llama-3.3-70b-instruct + input_price: 0.15 + output_price: 0.6 + supports_vision: true + supports_function_calling: true + - name: openai/gpt-4-turbo max_input_tokens: 128000 - - name: meta-llama-3.1-405b-instruct + input_price: 10 + output_price: 30 + supports_vision: true + supports_function_calling: true + - name: openai/o1 max_input_tokens: 128000 - - name: meta-llama-3.1-70b-instruct + input_price: 15 + output_price: 60 + supports_vision: true + supports_function_calling: true + no_system_message: true + - name: openai/o1-preview max_input_tokens: 128000 - - name: meta-llama-3.1-8b-instruct + input_price: 15 + output_price: 60 + no_system_message: true + - name: openai/o1-mini max_input_tokens: 128000 - - name: llama-3.2-90b-vision-instruct - max_input_tokens: 8192 + input_price: 3 + output_price: 12 + no_system_message: true + - name: openai/gpt-3.5-turbo + max_input_tokens: 16385 + input_price: 0.5 + output_price: 1.5 + supports_function_calling: true + - name: google/gemini-pro-1.5 + max_input_tokens: 2000000 + input_price: 1.25 + output_price: 5 supports_vision: true - - name: llama-3.2-11b-vision-instruct - max_input_tokens: 8192 + supports_function_calling: true + - name: google/gemini-flash-1.5 + max_input_tokens: 1000000 + input_price: 0.075 + output_price: 0.3 supports_vision: true - - name: mistral-large-2411 - max_input_tokens: 128000 supports_function_calling: true - - name: codestral-2501 - max_input_tokens: 256000 + - name: google/gemini-flash-1.5-8b + max_input_tokens: 1000000 + input_price: 0.0375 + output_price: 0.15 + supports_vision: true supports_function_calling: true - - name: cohere-command-r-plus-08-2024 - max_input_tokens: 128000 + - name: anthropic/claude-3.5-sonnet + max_input_tokens: 200000 + max_output_tokens: 8192 + require_max_tokens: true + input_price: 3 + output_price: 15 + supports_vision: true supports_function_calling: true - - name: cohere-command-r-08-2024 - max_input_tokens: 128000 + - name: anthropic/claude-3-5-haiku + max_input_tokens: 200000 + max_output_tokens: 8192 + require_max_tokens: true + input_price: 0.8 + output_price: 4 + supports_vision: true supports_function_calling: true - - name: cohere-embed-v3-english - type: embedding - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 96 - - name: cohere-embed-v3-multilingual - type: embedding - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 96 - - name: ai21-jamba-1.5-large - max_input_tokens: 256000 + - name: anthropic/claude-3-opus + max_input_tokens: 200000 + max_output_tokens: 4096 + require_max_tokens: true + input_price: 15 + output_price: 75 + supports_vision: true supports_function_calling: true - - name: ai21-jamba-1.5-mini - max_input_tokens: 256000 + - name: anthropic/claude-3-sonnet + max_input_tokens: 200000 + max_output_tokens: 4096 + require_max_tokens: true + input_price: 3 + output_price: 15 + supports_vision: true supports_function_calling: true - - name: phi-4 - max_input_tokens: 16384 - - name: phi-3.5-moe-instruct - max_input_tokens: 128000 - - name: phi-3.5-mini-instruct - max_input_tokens: 128000 - - name: phi-3.5-vision-instruct - max_input_tokens: 128000 + - name: anthropic/claude-3-haiku + max_input_tokens: 200000 + max_output_tokens: 4096 + require_max_tokens: true + input_price: 0.25 + output_price: 1.25 supports_vision: true - -# Links: -# - https://deepinfra.com/models -- platform: deepinfra - models: - - name: meta-llama/Llama-3.3-70B-Instruct - max_input_tokens: 128000 - input_price: 0.23 - output_price: 0.40 - - name: meta-llama/Meta-Llama-3.1-405B-Instruct - max_input_tokens: 32000 - input_price: 0.8 - output_price: 0.8 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-70B-Instruct - max_input_tokens: 128000 - input_price: 0.23 - output_price: 0.4 - supports_function_calling: true - - name: meta-llama/Meta-Llama-3.1-8B-Instruct - max_input_tokens: 128000 - input_price: 0.03 - output_price: 0.05 - supports_function_calling: true - - name: meta-llama/Llama-3.2-90B-Vision-Instruct - max_input_tokens: 128000 - input_price: 0.35 - output_price: 0.4 - - name: meta-llama/Llama-3.2-11B-Vision-Instruct - max_input_tokens: 128000 - input_price: 0.055 - output_price: 0.055 - - name: mistralai/Mistral-Nemo-Instruct-2407 - max_input_tokens: 128000 - input_price: 0.035 - output_price: 0.08 - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.27 - output_price: 0.27 - - name: google/gemma-2-9b-it - max_input_tokens: 8192 - input_price: 0.03 - output_price: 0.06 - - name: Qwen/Qwen2.5-72B-Instruct - max_input_tokens: 32768 - input_price: 0.23 - output_price: 0.40 - supports_function_calling: true - - name: Qwen/Qwen2.5-Coder-32B-Instruct - max_input_tokens: 32768 - input_price: 0.07 - output_price: 0.16 - - name: Qwen/QVQ-72B-Preview - max_input_tokens: 32768 - input_price: 0.25 - output_price: 0.50 - supports_vision: true - - name: Qwen/QwQ-32B-Preview - max_input_tokens: 32768 - input_price: 0.12 - output_price: 0.18 - - name: deepseek-ai/DeepSeek-V3 - max_input_tokens: 32768 - input_price: 0.85 - output_price: 0.9 - - name: microsoft/phi-4 - max_input_tokens: 16384 - input_price: 0.07 - output_price: 0.14 - - name: nvidia/Llama-3.1-Nemotron-70B-Instruct - max_input_tokens: 128000 - input_price: 0.12 - output_price: 0.30 - supports_function_calling: true - - name: BAAI/bge-large-en-v1.5 - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: BAAI/bge-m3 - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 8192 - default_chunk_size: 2000 - max_batch_size: 100 - - name: intfloat/e5-large-v2 - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: intfloat/multilingual-e5-large - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: thenlper/gte-large - type: embedding - input_price: 0.01 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - -# Links: -# - https://fireworks.ai/models -# - https://fireworks.ai/pricing -- platform: fireworks - models: - - name: accounts/fireworks/models/llama-v3p3-70b-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/llama-v3p1-405b-instruct - max_input_tokens: 131072 - input_price: 3 - output_price: 3 - supports_function_calling: true - - name: accounts/fireworks/models/llama-v3p1-70b-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - supports_function_calling: true - - name: accounts/fireworks/models/llama-v3p1-8b-instruct - max_input_tokens: 131072 - input_price: 0.2 - output_price: 0.2 - - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - supports_vision: true - - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct - max_input_tokens: 131072 - input_price: 0.2 - output_price: 0.2 - supports_vision: true - - name: accounts/fireworks/models/qwen2p5-72b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - supports_function_calling: true - - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/qwen-qwq-32b-preview - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - - name: accounts/fireworks/models/qwen2-vl-72b-instruct - max_input_tokens: 32768 - input_price: 0.9 - output_price: 0.9 - supports_vision: true - - name: accounts/fireworks/models/deepseek-v3 - max_input_tokens: 131072 - input_price: 0.9 - output_price: 0.9 - - name: nomic-ai/nomic-embed-text-v1.5 - type: embedding - input_price: 0.008 - max_tokens_per_chunk: 8192 - default_chunk_size: 1500 - max_batch_size: 100 - - name: WhereIsAI/UAE-Large-V1 - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - - name: thenlper/gte-large - type: embedding - input_price: 0.016 - max_tokens_per_chunk: 512 - default_chunk_size: 1000 - max_batch_size: 100 - -# Links: -# - https://openrouter.ai/models -- platform: openrouter - models: - - name: openai/gpt-4o - max_input_tokens: 128000 - input_price: 2.5 - output_price: 10 - supports_vision: true - supports_function_calling: true - - name: openai/gpt-4o-2024-11-20 - max_input_tokens: 128000 - input_price: 2.5 - output_price: 10 - supports_vision: true - supports_function_calling: true - - name: openai/gpt-4o-2024-08-06 - max_input_tokens: 128000 - input_price: 2.5 - output_price: 10 - supports_vision: true - supports_function_calling: true - - name: openai/chatgpt-4o-latest - max_input_tokens: 128000 - input_price: 5 - output_price: 15 - supports_vision: true - supports_function_calling: true - - name: openai/gpt-4o-mini - max_input_tokens: 128000 - input_price: 0.15 - output_price: 0.6 - supports_vision: true - supports_function_calling: true - - name: openai/gpt-4-turbo - max_input_tokens: 128000 - input_price: 10 - output_price: 30 - supports_vision: true - supports_function_calling: true - - name: openai/o1 - max_input_tokens: 128000 - input_price: 15 - output_price: 60 - supports_vision: true - supports_function_calling: true - no_system_message: true - - name: openai/o1-preview - max_input_tokens: 128000 - input_price: 15 - output_price: 60 - no_system_message: true - - name: openai/o1-mini - max_input_tokens: 128000 - input_price: 3 - output_price: 12 - no_system_message: true - - name: openai/gpt-3.5-turbo - max_input_tokens: 16385 - input_price: 0.5 - output_price: 1.5 - supports_function_calling: true - - name: google/gemini-pro-1.5 - max_input_tokens: 2000000 - input_price: 1.25 - output_price: 5 - supports_vision: true - supports_function_calling: true - - name: google/gemini-flash-1.5 - max_input_tokens: 1000000 - input_price: 0.075 - output_price: 0.3 - supports_vision: true - supports_function_calling: true - - name: google/gemini-flash-1.5-8b - max_input_tokens: 1000000 - input_price: 0.0375 - output_price: 0.15 - supports_vision: true - supports_function_calling: true - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.27 - output_price: 0.27 - - name: google/gemma-2-9b-it - max_input_tokens: 4096 - input_price: 0.06 - output_price: 0.06 - - name: anthropic/claude-3.5-sonnet - max_input_tokens: 200000 - max_output_tokens: 8192 - require_max_tokens: true - input_price: 3 - output_price: 15 - supports_vision: true - supports_function_calling: true - - name: anthropic/claude-3-5-haiku - max_input_tokens: 200000 - max_output_tokens: 8192 - require_max_tokens: true - input_price: 0.8 - output_price: 4 - supports_vision: true - supports_function_calling: true - - name: anthropic/claude-3-opus - max_input_tokens: 200000 - max_output_tokens: 4096 - require_max_tokens: true - input_price: 15 - output_price: 75 - supports_vision: true - supports_function_calling: true - - name: anthropic/claude-3-sonnet - max_input_tokens: 200000 - max_output_tokens: 4096 - require_max_tokens: true - input_price: 3 - output_price: 15 - supports_vision: true - supports_function_calling: true - - name: anthropic/claude-3-haiku - max_input_tokens: 200000 - max_output_tokens: 4096 - require_max_tokens: true - input_price: 0.25 - output_price: 1.25 - supports_vision: true - supports_function_calling: true - - name: meta-llama/llama-3.3-70b-instruct - max_input_tokens: 131072 - input_price: 0.12 - output_price: 0.3 - - name: meta-llama/llama-3.1-405b-instruct - max_input_tokens: 131072 + supports_function_calling: true + - name: meta-llama/llama-3.3-70b-instruct + max_input_tokens: 131072 + input_price: 0.12 + output_price: 0.3 + - name: meta-llama/llama-3.1-405b-instruct + max_input_tokens: 32768 input_price: 0.8 output_price: 0.8 supports_function_calling: true @@ -1519,117 +1256,364 @@ supports_function_calling: true - name: cohere/command-r-08-2024 max_input_tokens: 128000 - input_price: 0.15 - output_price: 0.6 - supports_function_calling: true - - name: cohere/command-r7b-12-2024 + input_price: 0.15 + output_price: 0.6 + supports_function_calling: true + - name: cohere/command-r7b-12-2024 + max_input_tokens: 128000 + max_output_tokens: 4096 + input_price: 0.0375 + output_price: 0.15 + - name: deepseek/deepseek-chat + max_input_tokens: 64000 + input_price: 0.14 + output_price: 0.28 + supports_function_calling: true + - name: deepseek/deepseek-r1 + max_input_tokens: 163840 + input_price: 0.55 + output_price: 2.19 + - name: perplexity/llama-3.1-sonar-huge-128k-online + max_input_tokens: 127072 + input_price: 5 + output_price: 5 + - name: perplexity/llama-3.1-sonar-large-128k-online + max_input_tokens: 127072 + input_price: 1 + output_price: 1 + - name: perplexity/llama-3.1-sonar-small-128k-online + max_input_tokens: 127072 + input_price: 0.2 + output_price: 0.2 + - name: microsoft/phi-4 + max_input_tokens: 16384 + input_price: 0.07 + output_price: 0.14 + - name: microsoft/phi-3.5-mini-128k-instruct + max_input_tokens: 128000 + input_price: 0.1 + output_price: 0.1 + - name: qwen/qwen-2.5-72b-instruct + max_input_tokens: 131072 + input_price: 0.35 + output_price: 0.4 + supports_function_calling: true + - name: qwen/qwen-2.5-coder-32b-instruct + max_input_tokens: 32768 + input_price: 0.18 + output_price: 0.18 + - name: qwen/qwen-2-vl-72b-instruct + max_input_tokens: 32768 + input_price: 0.4 + output_price: 0.4 + - name: qwen/qwq-32b-preview + max_input_tokens: 32768 + input_price: 0.15 + output_price: 0.6 + - name: qwen/qvq-72b-preview + max_input_tokens: 128000 + input_price: 0.25 + output_price: 0.5 + supports_vision: true + - name: x-ai/grok-2-1212 + max_input_tokens: 131072 + input_price: 2 + output_price: 10 + supports_function_calling: true + - name: x-ai/grok-beta + max_input_tokens: 32768 + input_price: 5 + output_price: 15 + supports_function_calling: true + - name: x-ai/grok-2-vision-1212 + max_input_tokens: 32768 + input_price: 2 + output_price: 10 + supports_vision: true + supports_function_calling: true + - name: x-ai/grok-vision-beta + max_input_tokens: 8192 + input_price: 5 + output_price: 15 + supports_vision: true + - name: amazon/nova-pro-v1 + max_input_tokens: 300000 + max_output_tokens: 5120 + input_price: 0.8 + output_price: 3.2 + supports_vision: true + - name: amazon/nova-lite-v1 + max_input_tokens: 300000 + max_output_tokens: 5120 + input_price: 0.06 + output_price: 0.24 + supports_vision: true + - name: amazon/nova-micro-v1 + max_input_tokens: 128000 + max_output_tokens: 5120 + input_price: 0.035 + output_price: 0.14 + - name: minimax/minimax-01 + max_input_tokens: 1000192 + input_price: 0.2 + output_price: 1.1 + + +# Links: +# - https://github.com/marketplace/models +- provider: github + models: + - name: gpt-4o + max_input_tokens: 128000 + supports_function_calling: true + - name: gpt-4o-mini + max_input_tokens: 128000 + supports_function_calling: true + - name: o1 + max_input_tokens: 200000 + supports_function_calling: true + supports_vision: true + no_stream: true + no_system_message: true + - name: o1-preview + max_input_tokens: 128000 + no_stream: true + no_system_message: true + - name: o1-mini + max_input_tokens: 128000 + no_stream: true + no_system_message: true + - name: text-embedding-3-large + type: embedding + max_tokens_per_chunk: 8191 + default_chunk_size: 2000 + max_batch_size: 100 + - name: text-embedding-3-small + type: embedding + max_tokens_per_chunk: 8191 + default_chunk_size: 2000 + max_batch_size: 100 + - name: llama-3.3-70b-instruct + max_input_tokens: 131072 + - name: meta-llama-3.1-405b-instruct + max_input_tokens: 131072 + - name: meta-llama-3.1-70b-instruct + max_input_tokens: 131072 + - name: meta-llama-3.1-8b-instruct + max_input_tokens: 131072 + - name: llama-3.2-90b-vision-instruct + max_input_tokens: 131072 + supports_vision: true + - name: llama-3.2-11b-vision-instruct + max_input_tokens: 131072 + supports_vision: true + - name: mistral-large-2411 + max_input_tokens: 128000 + supports_function_calling: true + - name: codestral-2501 + max_input_tokens: 256000 + supports_function_calling: true + - name: cohere-command-r-plus-08-2024 + max_input_tokens: 128000 + supports_function_calling: true + - name: cohere-command-r-08-2024 + max_input_tokens: 128000 + supports_function_calling: true + - name: cohere-embed-v3-english + type: embedding + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 96 + - name: cohere-embed-v3-multilingual + type: embedding + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 96 + - name: ai21-jamba-1.5-large + max_input_tokens: 256000 + supports_function_calling: true + - name: ai21-jamba-1.5-mini + max_input_tokens: 256000 + supports_function_calling: true + - name: phi-4 + max_input_tokens: 16384 + - name: phi-3.5-moe-instruct + max_input_tokens: 128000 + - name: phi-3.5-mini-instruct + max_input_tokens: 128000 + - name: phi-3.5-vision-instruct max_input_tokens: 128000 - max_output_tokens: 4096 - input_price: 0.0375 - output_price: 0.15 - - name: deepseek/deepseek-chat + supports_vision: true + +# Links: +# - https://deepinfra.com/models +- provider: deepinfra + models: + - name: meta-llama/Llama-3.3-70B-Instruct + max_input_tokens: 131072 + input_price: 0.23 + output_price: 0.40 + - name: meta-llama/Meta-Llama-3.1-405B-Instruct max_input_tokens: 32768 - input_price: 0.14 - output_price: 0.28 + input_price: 0.8 + output_price: 0.8 supports_function_calling: true - - name: perplexity/llama-3.1-sonar-huge-128k-online - max_input_tokens: 127072 - input_price: 5 - output_price: 5 - - name: perplexity/llama-3.1-sonar-large-128k-online - max_input_tokens: 127072 - input_price: 1 - output_price: 1 - - name: perplexity/llama-3.1-sonar-small-128k-online - max_input_tokens: 127072 - input_price: 0.2 - output_price: 0.2 - - name: 01-ai/yi-large - max_input_tokens: 32768 - input_price: 3 - output_price: 3 - - name: microsoft/phi-4 - max_input_tokens: 16000 - input_price: 0.07 - output_price: 0.14 - - name: microsoft/phi-3.5-mini-128k-instruct - max_input_tokens: 128000 - input_price: 0.1 - output_price: 0.1 - - name: qwen/qwen-2.5-72b-instruct + - name: meta-llama/Meta-Llama-3.1-70B-Instruct max_input_tokens: 131072 - input_price: 0.35 + input_price: 0.23 output_price: 0.4 supports_function_calling: true - - name: qwen/qwen-2.5-coder-32b-instruct + - name: meta-llama/Meta-Llama-3.1-8B-Instruct + max_input_tokens: 131072 + input_price: 0.03 + output_price: 0.05 + supports_function_calling: true + - name: meta-llama/Llama-3.2-90B-Vision-Instruct + max_input_tokens: 131072 + input_price: 0.35 + output_price: 0.4 + - name: meta-llama/Llama-3.2-11B-Vision-Instruct + max_input_tokens: 131072 + input_price: 0.055 + output_price: 0.055 + - name: Qwen/Qwen2.5-72B-Instruct max_input_tokens: 32768 - input_price: 0.18 - output_price: 0.18 - - name: qwen/qwen-2-vl-72b-instruct + input_price: 0.23 + output_price: 0.40 + supports_function_calling: true + - name: Qwen/Qwen2.5-Coder-32B-Instruct max_input_tokens: 32768 - input_price: 0.4 - output_price: 0.4 - - name: qwen/qwq-32b-preview + input_price: 0.07 + output_price: 0.16 + - name: Qwen/QVQ-72B-Preview max_input_tokens: 32768 - input_price: 0.15 - output_price: 0.6 - - name: qwen/qvq-72b-preview - max_input_tokens: 128000 input_price: 0.25 - output_price: 0.5 + output_price: 0.50 supports_vision: true - - name: nvidia/llama-3.1-nemotron-70b-instruct + - name: Qwen/QwQ-32B-Preview + max_input_tokens: 32768 + input_price: 0.12 + output_price: 0.18 + - name: deepseek-ai/DeepSeek-V3 + max_input_tokens: 32768 + input_price: 0.85 + output_price: 0.9 + - name: microsoft/phi-4 + max_input_tokens: 16384 + input_price: 0.07 + output_price: 0.14 + - name: BAAI/bge-large-en-v1.5 + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: BAAI/bge-m3 + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 8192 + default_chunk_size: 2000 + max_batch_size: 100 + - name: intfloat/e5-large-v2 + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: intfloat/multilingual-e5-large + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: thenlper/gte-large + type: embedding + input_price: 0.01 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + +# Links: +# - https://fireworks.ai/models +# - https://fireworks.ai/pricing +- provider: fireworks + models: + - name: accounts/fireworks/models/llama-v3p3-70b-instruct max_input_tokens: 131072 - input_price: 0.35 - output_price: 0.4 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/llama-v3p1-405b-instruct + max_input_tokens: 131072 + input_price: 3 + output_price: 3 supports_function_calling: true - - name: x-ai/grok-2-1212 + - name: accounts/fireworks/models/llama-v3p1-70b-instruct max_input_tokens: 131072 - input_price: 2 - output_price: 10 + input_price: 0.9 + output_price: 0.9 supports_function_calling: true - - name: x-ai/grok-beta + - name: accounts/fireworks/models/llama-v3p1-8b-instruct + max_input_tokens: 131072 + input_price: 0.2 + output_price: 0.2 + - name: accounts/fireworks/models/llama-v3p2-90b-vision-instruct + max_input_tokens: 131072 + input_price: 0.9 + output_price: 0.9 + supports_vision: true + - name: accounts/fireworks/models/llama-v3p2-11b-vision-instruct + max_input_tokens: 131072 + input_price: 0.2 + output_price: 0.2 + supports_vision: true + - name: accounts/fireworks/models/qwen2p5-72b-instruct max_input_tokens: 32768 - input_price: 5 - output_price: 15 + input_price: 0.9 + output_price: 0.9 supports_function_calling: true - - name: x-ai/grok-2-vision-1212 + - name: accounts/fireworks/models/qwen2p5-coder-32b-instruct max_input_tokens: 32768 - input_price: 2 - output_price: 10 - supports_vision: true - supports_function_calling: true - - name: x-ai/grok-vision-beta - max_input_tokens: 8192 - input_price: 5 - output_price: 15 - supports_vision: true - - name: amazon/nova-pro-v1 - max_input_tokens: 300000 - max_output_tokens: 5120 - input_price: 0.8 - output_price: 3.2 - supports_vision: true - - name: amazon/nova-lite-v1 - max_input_tokens: 300000 - max_output_tokens: 5120 - input_price: 0.06 - output_price: 0.24 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/qwen-qwq-32b-preview + max_input_tokens: 32768 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/qwen2-vl-72b-instruct + max_input_tokens: 32768 + input_price: 0.9 + output_price: 0.9 supports_vision: true - - name: amazon/nova-micro-v1 - max_input_tokens: 128000 - max_output_tokens: 5120 - input_price: 0.035 - output_price: 0.14 - - name: minimax/minimax-01 - max_input_tokens: 1000192 - input_price: 0.2 - output_price: 1.1 - + - name: accounts/fireworks/models/deepseek-v3 + max_input_tokens: 131072 + input_price: 0.9 + output_price: 0.9 + - name: accounts/fireworks/models/deepseek-r1 + max_input_tokens: 160000 + input_price: 8 + output_price: 8 + - name: nomic-ai/nomic-embed-text-v1.5 + type: embedding + input_price: 0.008 + max_tokens_per_chunk: 8192 + default_chunk_size: 1500 + max_batch_size: 100 + - name: WhereIsAI/UAE-Large-V1 + type: embedding + input_price: 0.016 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 + - name: thenlper/gte-large + type: embedding + input_price: 0.016 + max_tokens_per_chunk: 512 + default_chunk_size: 1000 + max_batch_size: 100 # Links # - https://cloud.siliconflow.cn/models # - https://docs.siliconflow.cn/api-reference/chat-completions/chat-completions -- platform: siliconflow +- provider: siliconflow models: - name: meta-llama/Llama-3.3-70B-Instruct max_input_tokens: 32768 @@ -1653,7 +1637,7 @@ output_price: 0.578 supports_function_calling: true - name: Qwen/Qwen2.5-72B-Instruct-128K - max_input_tokens: 128000 + max_input_tokens: 131072 input_price: 0.578 output_price: 0.578 supports_function_calling: true @@ -1684,14 +1668,6 @@ max_input_tokens: 32768 input_price: 0.176 output_price: 0.176 - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.176 - output_price: 0.176 - - name: google/gemma-2-9b-it - max_input_tokens: 8192 - input_price: 0 - output_price: 0 - name: deepseek-ai/DeepSeek-V2.5 max_input_tokens: 32768 input_price: 0.186 @@ -1728,25 +1704,25 @@ # Links: # - https://docs.together.ai/docs/serverless-models # - https://www.together.ai/pricing -- platform: together +- provider: together models: - name: meta-llama/Llama-3.3-70B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 131072 input_price: 0.88 output_price: 0.88 supports_function_calling: true - name: meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 130815 input_price: 3.5 output_price: 3.5 supports_function_calling: true - name: meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 131072 input_price: 0.88 output_price: 0.88 supports_function_calling: true - name: meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo - max_input_tokens: 32768 + max_input_tokens: 131072 input_price: 0.18 output_price: 0.18 supports_function_calling: true @@ -1760,14 +1736,6 @@ input_price: 0.18 output_price: 0.18 supports_vision: true - - name: google/gemma-2-27b-it - max_input_tokens: 8192 - input_price: 0.8 - output_price: 0.8 - - name: google/gemma-2-9b-it - max_input_tokens: 8192 - input_price: 0.3 - output_price: 0.3 - name: Qwen/Qwen2.5-72B-Instruct-Turbo max_input_tokens: 32768 input_price: 1.2 @@ -1777,7 +1745,7 @@ input_price: 0.3 output_price: 0.3 - name: Qwen/Qwen2.5-Coder-32B-Instruct - max_input_tokens: 16384 + max_input_tokens: 32768 input_price: 0.8 output_price: 0.8 - name: Qwen/QwQ-32B-Preview @@ -1793,6 +1761,10 @@ max_input_tokens: 131072 input_price: 1.25 output_price: 1.25 + - name: deepseek-ai/DeepSeek-R1 + max_input_tokens: 163840 + input_price: 7 + output_price: 7 - name: WhereIsAI/UAE-Large-V1 type: embedding input_price: 0.016 @@ -1811,9 +1783,9 @@ input_price: 0.1 # Links: -# - https://jina.ai/ +# - https://jina.ai/models # - https://api.jina.ai/redoc -- platform: jina +- provider: jina models: - name: jina-embeddings-v3 type: embedding @@ -1846,7 +1818,7 @@ # - https://docs.voyageai.com/docs/embeddings # - https://docs.voyageai.com/docs/pricing # - https://docs.voyageai.com/reference/ -- platform: voyageai +- provider: voyageai models: - name: voyage-3-large type: embedding diff --git a/scripts/completions/aichat.bash b/scripts/completions/aichat.bash index 4708077..9ea4f9a 100644 --- a/scripts/completions/aichat.bash +++ b/scripts/completions/aichat.bash @@ -17,7 +17,7 @@ _aichat() { case "${cmd}" in aichat) - opts="-m -r -s -a -e -c -f -S -h -V --model --prompt --role --session --empty-session --save-session --agent --agent-variable --rag --rebuild-rag --macro --serve --execute --code --file --no-stream --dry-run --info --list-models --list-roles --list-sessions --list-agents --list-rags --list-macros --help --version" + opts="-m -r -s -a -e -c -f -S -h -V --model --prompt --role --session --empty-session --save-session --agent --agent-variable --rag --rebuild-rag --macro --serve --execute --code --file --no-stream --dry-run --info --sync-models --list-models --list-roles --list-sessions --list-agents --list-rags --list-macros --help --version" if [[ ${cur} == -* || ${cword} -eq 1 ]] ; then COMPREPLY=( $(compgen -W "${opts}" -- "${cur}") ) return 0 diff --git a/scripts/completions/aichat.fish b/scripts/completions/aichat.fish index 3618a5d..d3c336c 100644 --- a/scripts/completions/aichat.fish +++ b/scripts/completions/aichat.fish @@ -16,6 +16,7 @@ complete -c aichat -s f -l file -d 'Include files, directories, or URLs' -r -F complete -c aichat -s S -l no-stream -d 'Turn off stream mode' complete -c aichat -l dry-run -d 'Display the message without sending it' complete -c aichat -l info -d 'Display information' +complete -c aichat -l sync-models -d 'Sync models updates' complete -c aichat -l list-models -d 'List all available chat models' complete -c aichat -l list-roles -d 'List all roles' complete -c aichat -l list-sessions -d 'List all sessions' diff --git a/scripts/completions/aichat.nu b/scripts/completions/aichat.nu index bbea5e2..365f0cc 100644 --- a/scripts/completions/aichat.nu +++ b/scripts/completions/aichat.nu @@ -40,7 +40,6 @@ module completions { | parse "{value}" } - # All-in-one chat and copilot CLI that integrates 10+ AI platforms export extern aichat [ --model(-m): string@"nu-complete aichat model" # Select a LLM model --prompt # Use the system prompt @@ -60,6 +59,7 @@ module completions { --no-stream(-S) # Turn off stream mode --dry-run # Display the message without sending it --info # Display information + --sync-models # Sync models updates --list-models # List all available chat models --list-roles # List all roles --list-sessions # List all sessions diff --git a/scripts/completions/aichat.ps1 b/scripts/completions/aichat.ps1 index dc4ef64..c8c3187 100644 --- a/scripts/completions/aichat.ps1 +++ b/scripts/completions/aichat.ps1 @@ -46,6 +46,7 @@ Register-ArgumentCompleter -Native -CommandName 'aichat' -ScriptBlock { [CompletionResult]::new('--no-stream', '--no-stream', [CompletionResultType]::ParameterName, 'Turn off stream mode') [CompletionResult]::new('--dry-run', '--dry-run', [CompletionResultType]::ParameterName, 'Display the message without sending it') [CompletionResult]::new('--info', '--info', [CompletionResultType]::ParameterName, 'Display information') + [CompletionResult]::new('--sync-models', '--sync-models', [CompletionResultType]::ParameterName, 'Sync models updates') [CompletionResult]::new('--list-models', '--list-models', [CompletionResultType]::ParameterName, 'List all available chat models') [CompletionResult]::new('--list-roles', '--list-roles', [CompletionResultType]::ParameterName, 'List all roles') [CompletionResult]::new('--list-sessions', '--list-sessions', [CompletionResultType]::ParameterName, 'List all sessions') diff --git a/scripts/completions/aichat.zsh b/scripts/completions/aichat.zsh index 15cbdce..1349081 100644 --- a/scripts/completions/aichat.zsh +++ b/scripts/completions/aichat.zsh @@ -41,6 +41,7 @@ _aichat() { '--no-stream[Turn off stream mode]' \ '--dry-run[Display the message without sending it]' \ '--info[Display information]' \ +'--sync-models[Sync models updates]' \ '--list-models[List all available chat models]' \ '--list-roles[List all roles]' \ '--list-sessions[List all sessions]' \ diff --git a/src/cli.rs b/src/cli.rs index de88776..3204c58 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -60,6 +60,9 @@ pub struct Cli { /// Display information #[clap(long)] pub info: bool, + /// Sync models updates + #[clap(long)] + pub sync_models: bool, /// List all available chat models #[clap(long)] pub list_models: bool, diff --git a/src/client/common.rs b/src/client/common.rs index b4d01ce..80f585d 100644 --- a/src/client/common.rs +++ b/src/client/common.rs @@ -1,7 +1,7 @@ use super::*; use crate::{ - config::{GlobalConfig, Input}, + config::{Config, GlobalConfig, Input}, function::{eval_tool_calls, FunctionDeclaration, ToolCall, ToolResult}, render::render_stream, utils::*, @@ -20,7 +20,9 @@ use tokio::sync::mpsc::unbounded_channel; const MODELS_YAML: &str = include_str!("../../models.yaml"); lazy_static::lazy_static! { - pub static ref ALL_PREDEFINED_MODELS: Vec = serde_yaml::from_str(MODELS_YAML).unwrap(); + pub static ref ALL_PROVIDER_MODELS: Vec = { + Config::loal_models_override().ok().unwrap_or_else(|| serde_yaml::from_str(MODELS_YAML).unwrap()) + }; static ref ESCAPE_SLASH_RE: Regex = Regex::new(r"(? Result<(String, } pub fn create_openai_compatible_client_config(client: &str) -> Result> { - let api_base = super::OPENAI_COMPATIBLE_PLATFORMS + let api_base = super::OPENAI_COMPATIBLE_PROVIDERS .into_iter() .find(|(name, _)| client == *name) .map(|(_, api_base)| api_base) .unwrap_or("http(s)://{API_ADDR}/v1"); let name = if client == OpenAICompatibleClient::NAME { - prompt_input_string("Provider Name", true, None)? + let value = prompt_input_string("Provider Name", true, None)?; + value.replace(' ', "-") } else { client.to_string() }; @@ -548,7 +551,7 @@ fn set_client_config(list: &[PromptAction], client_config: &mut Value, client: & } fn set_client_models_config(client_config: &mut Value, client: &str) -> Result<()> { - if ALL_PREDEFINED_MODELS.iter().any(|v| v.platform == client) { + if ALL_PROVIDER_MODELS.iter().any(|v| v.provider == client) { return Ok(()); } diff --git a/src/client/macros.rs b/src/client/macros.rs index a76e62b..97171db 100644 --- a/src/client/macros.rs +++ b/src/client/macros.rs @@ -52,10 +52,10 @@ macro_rules! register_client { pub fn list_models(local_config: &$config) -> Vec { let client_name = Self::name(local_config); if local_config.models.is_empty() { - if let Some(models) = $crate::client::ALL_PREDEFINED_MODELS.iter().find(|v| { - v.platform == $name || + if let Some(models) = $crate::client::ALL_PROVIDER_MODELS.iter().find(|v| { + v.provider == $name || ($name == OpenAICompatibleClient::NAME - && local_config.name.as_ref().map(|name| name.starts_with(&v.platform)).unwrap_or_default()) + && local_config.name.as_ref().map(|name| name.starts_with(&v.provider)).unwrap_or_default()) }) { return Model::from_config(client_name, &models.models); } @@ -83,7 +83,7 @@ macro_rules! register_client { pub fn list_client_types() -> Vec<&'static str> { let mut client_types: Vec<_> = vec![$($client::NAME,)+]; - client_types.extend($crate::client::OPENAI_COMPATIBLE_PLATFORMS.iter().map(|(name, _)| *name)); + client_types.extend($crate::client::OPENAI_COMPATIBLE_PROVIDERS.iter().map(|(name, _)| *name)); client_types } diff --git a/src/client/mod.rs b/src/client/mod.rs index 3d8d4da..bf11107 100644 --- a/src/client/mod.rs +++ b/src/client/mod.rs @@ -34,7 +34,7 @@ register_client!( (ernie, "ernie", ErnieConfig, ErnieClient), ); -pub const OPENAI_COMPATIBLE_PLATFORMS: [(&str, &str); 22] = [ +pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 22] = [ ("ai21", "https://api.ai21.com/studio/v1"), ( "cloudflare", diff --git a/src/client/model.rs b/src/client/model.rs index 4b0457f..b562705 100644 --- a/src/client/model.rs +++ b/src/client/model.rs @@ -277,26 +277,33 @@ pub struct ModelData { pub name: String, #[serde(default = "default_model_type", rename = "type")] pub model_type: String, + #[serde(skip_serializing_if = "Option::is_none")] pub max_input_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub input_price: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub output_price: Option, // chat-only properties + #[serde(skip_serializing_if = "Option::is_none")] pub max_output_tokens: Option, - #[serde(default)] + #[serde(default, skip_serializing_if = "std::ops::Not::not")] pub require_max_tokens: bool, - #[serde(default)] + #[serde(default, skip_serializing_if = "std::ops::Not::not")] pub supports_vision: bool, - #[serde(default)] + #[serde(default, skip_serializing_if = "std::ops::Not::not")] pub supports_function_calling: bool, - #[serde(default)] + #[serde(default, skip_serializing_if = "std::ops::Not::not")] no_stream: bool, - #[serde(default)] + #[serde(default, skip_serializing_if = "std::ops::Not::not")] no_system_message: bool, // embedding-only properties + #[serde(skip_serializing_if = "Option::is_none")] pub max_tokens_per_chunk: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub default_chunk_size: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub max_batch_size: Option, } @@ -310,9 +317,9 @@ impl ModelData { } } -#[derive(Debug, Clone, Deserialize)] -pub struct PredefinedModels { - pub platform: String, +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ProviderModels { + pub provider: String, pub models: Vec, } diff --git a/src/client/openai_compatible.rs b/src/client/openai_compatible.rs index 18acafb..ce1eea3 100644 --- a/src/client/openai_compatible.rs +++ b/src/client/openai_compatible.rs @@ -96,7 +96,7 @@ fn get_api_base_ext(self_: &OpenAICompatibleClient) -> Result { let api_base = match self_.get_api_base() { Ok(v) => v, Err(err) => { - match OPENAI_COMPATIBLE_PLATFORMS + match OPENAI_COMPATIBLE_PROVIDERS .into_iter() .find_map(|(name, api_base)| { if name == self_.model.client_name() { diff --git a/src/config/input.rs b/src/config/input.rs index 58c0065..e468f19 100644 --- a/src/config/input.rs +++ b/src/config/input.rs @@ -442,7 +442,7 @@ async fn load_documents( } for file_url in remote_urls { - let (contents, extension) = fetch(&loaders, &file_url, true) + let (contents, extension) = fetch_with_loaders(&loaders, &file_url, true) .await .with_context(|| format!("Failed to load url '{file_url}'"))?; if extension == MEDIA_URL_EXTENSION { diff --git a/src/config/mod.rs b/src/config/mod.rs index 8813930..341470f 100644 --- a/src/config/mod.rs +++ b/src/config/mod.rs @@ -12,7 +12,7 @@ use self::session::Session; use crate::client::{ create_client_config, list_client_types, list_models, ClientConfig, MessageContentToolCalls, - Model, ModelType, OPENAI_COMPATIBLE_PLATFORMS, + Model, ModelType, ProviderModels, OPENAI_COMPATIBLE_PROVIDERS, }; use crate::function::{FunctionDeclaration, Functions, ToolResult}; use crate::rag::Rag; @@ -24,7 +24,7 @@ use anyhow::{anyhow, bail, Context, Result}; use indexmap::IndexMap; use inquire::{list_option::ListOption, validator::Validation, Confirm, MultiSelect, Select, Text}; use parking_lot::RwLock; -use serde::Deserialize; +use serde::{Deserialize, Serialize}; use serde_json::json; use simplelog::LevelFilter; use std::collections::{HashMap, HashSet}; @@ -64,6 +64,8 @@ const CLIENTS_FIELD: &str = "clients"; const SERVE_ADDR: &str = "127.0.0.1:8000"; +const SYNC_MODELS_URL: &str = "https://cdn.jsdelivr.net/gh/sigoden/aichat/models.yaml"; + const SUMMARIZE_PROMPT: &str = "Summarize the discussion briefly in 200 words or less to use as a prompt for future context."; const SUMMARY_PROMPT: &str = "This is a summary of the chat history as a recap: "; @@ -140,6 +142,7 @@ pub struct Config { pub serve_addr: Option, pub user_agent: Option, pub save_shell_history: bool, + pub sync_models_url: Option, pub clients: Vec, @@ -214,6 +217,7 @@ impl Default for Config { serve_addr: None, user_agent: None, save_shell_history: true, + sync_models_url: None, clients: vec![], @@ -240,9 +244,12 @@ impl Config { pub fn init(working_mode: WorkingMode, info_flag: bool) -> Result { let config_path = Self::config_file(); let mut config = if !config_path.exists() { - match env::var(get_env_name("platform")) { - Ok(v) => Self::load_dynamic(&v)?, - Err(_) => { + match env::var(get_env_name("provider")) + .ok() + .or_else(|| env::var(get_env_name("platform")).ok()) + { + Some(v) => Self::load_dynamic(&v)?, + None => { if *IS_STDOUT_TERMINAL { create_config_file(&config_path)?; } @@ -417,6 +424,10 @@ impl Config { } } + pub fn models_override_file() -> PathBuf { + Self::local_path("models-override.json") + } + pub fn state(&self) -> StateFlags { let mut flags = StateFlags::empty(); if let Some(session) = &self.session { @@ -1362,23 +1373,12 @@ impl Config { let temp_file = temp_file(&format!("-rag-{}", rag.name()), ".txt"); tokio::fs::write(&temp_file, &document_paths.join("\n")) .await - .with_context(|| { - format!( - "Failed to write current document paths to '{}'", - temp_file.display() - ) - })?; + .with_context(|| format!("Failed to write to '{}'", temp_file.display()))?; let editor = config.read().editor()?; edit_file(&editor, &temp_file)?; - let new_document_paths = - tokio::fs::read_to_string(&temp_file) - .await - .with_context(|| { - format!( - "Failed to read new document paths from '{}'", - temp_file.display() - ) - })?; + let new_document_paths = tokio::fs::read_to_string(&temp_file) + .await + .with_context(|| format!("Failed to read '{}'", temp_file.display()))?; let new_document_paths = new_document_paths .split('\n') .filter_map(|v| { @@ -1535,12 +1535,7 @@ impl Config { &agent_config_path, "# see https://github.com/sigoden/aichat/blob/main/config.agent.example.yaml\n", ) - .with_context(|| { - format!( - "Failed to write to agent config file at '{}'", - agent_config_path.display() - ) - })?; + .with_context(|| format!("Failed to write to '{}'", agent_config_path.display()))?; } let editor = self.editor()?; edit_file(&editor, &agent_config_path)?; @@ -1860,6 +1855,50 @@ impl Config { .collect() } + pub fn sync_models_url(&self) -> String { + self.sync_models_url + .clone() + .unwrap_or_else(|| SYNC_MODELS_URL.into()) + } + + pub async fn sync_models(url: &str, abort_signal: AbortSignal) -> Result<()> { + let content = abortable_run_with_spinner(fetch(url), "Fetching models.yaml", abort_signal) + .await + .with_context(|| format!("Failed to fetch '{url}'"))?; + println!("✓ Fetched '{url}'"); + let list = serde_yaml::from_str::>(&content) + .with_context(|| "Failed to parse models.yaml")?; + let models_override = ModelsOverride { + version: env!("CARGO_PKG_VERSION").to_string(), + list, + }; + let models_override_data = + serde_json::to_string_pretty(&models_override).with_context(|| "Failed to serde {}")?; + + let model_override_path = Self::models_override_file(); + ensure_parent_exists(&model_override_path)?; + std::fs::write(&model_override_path, models_override_data) + .with_context(|| format!("Failed to write to '{}'", model_override_path.display()))?; + println!("✓ Updated '{}'", model_override_path.display()); + Ok(()) + } + + pub fn loal_models_override() -> Result> { + let model_override_path = Self::models_override_file(); + let err = || { + format!( + "Failed to load models at '{}'", + model_override_path.display() + ) + }; + let content = read_to_string(&model_override_path).with_context(err)?; + let models_override: ModelsOverride = serde_json::from_str(&content).with_context(err)?; + if models_override.version != env!("CARGO_PKG_VERSION") { + bail!("Incompatible version") + } + Ok(models_override.list) + } + pub fn render_options(&self) -> Result { let theme = if self.highlight { let theme_mode = if self.light_theme { "light" } else { "dark" }; @@ -2175,17 +2214,17 @@ impl Config { } fn load_dynamic(model_id: &str) -> Result { - let platform = match model_id.split_once(':') { + let provider = match model_id.split_once(':') { Some((v, _)) => v, _ => model_id, }; - let is_openai_compatible = OPENAI_COMPATIBLE_PLATFORMS + let is_openai_compatible = OPENAI_COMPATIBLE_PROVIDERS .into_iter() - .any(|(name, _)| platform == name); + .any(|(name, _)| provider == name); let client = if is_openai_compatible { - json!({ "type": "openai-compatible", "name": platform }) + json!({ "type": "openai-compatible", "name": provider }) } else { - json!({ "type": platform }) + json!({ "type": provider }) }; let config = json!({ "model": model_id.to_string(), @@ -2323,6 +2362,9 @@ impl Config { if let Some(Some(v)) = read_env_bool(&get_env_name("save_shell_history")) { self.save_shell_history = v; } + if let Some(v) = read_env_value::(&get_env_name("sync_models_url")) { + self.sync_models_url = v; + } } fn load_functions(&mut self) -> Result<()> { @@ -2502,6 +2544,12 @@ pub struct MacroVariable { pub default: Option, } +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ModelsOverride { + pub version: String, + pub list: Vec, +} + #[derive(Debug, Clone)] pub struct LastMessage { pub input: Input, @@ -2581,12 +2629,8 @@ fn create_config_file(config_path: &Path) -> Result<()> { ); ensure_parent_exists(config_path)?; - std::fs::write(config_path, config_data).with_context(|| { - format!( - "Failed to write to config file at '{}'", - config_path.display() - ) - })?; + std::fs::write(config_path, config_data) + .with_context(|| format!("Failed to write to '{}'", config_path.display()))?; #[cfg(unix)] { use std::os::unix::prelude::PermissionsExt; diff --git a/src/main.rs b/src/main.rs index e30efde..5251bad 100644 --- a/src/main.rs +++ b/src/main.rs @@ -46,6 +46,7 @@ async fn main() -> Result<()> { WorkingMode::Cmd }; let info_flag = cli.info + || cli.sync_models || cli.list_models || cli.list_roles || cli.list_agents @@ -64,6 +65,11 @@ async fn main() -> Result<()> { async fn run(config: GlobalConfig, cli: Cli, text: Option) -> Result<()> { let abort_signal = create_abort_signal(); + if cli.sync_models { + let url = config.read().sync_models_url(); + return Config::sync_models(&url, abort_signal.clone()).await; + } + if cli.list_models { for model in list_models(&config.read(), ModelType::Chat) { println!("{}", model.id()); diff --git a/src/utils/loader.rs b/src/utils/loader.rs index 519563c..2ac671d 100644 --- a/src/utils/loader.rs +++ b/src/utils/loader.rs @@ -61,7 +61,7 @@ pub async fn load_file(loaders: &HashMap, path: &str) -> Result< } pub async fn load_url(loaders: &HashMap, path: &str) -> Result { - let (contents, extension) = fetch(loaders, path, false).await?; + let (contents, extension) = fetch_with_loaders(loaders, path, false).await?; let mut metadata: DocumentMetadata = Default::default(); metadata.insert(EXTENSION_METADATA.into(), extension); Ok(LoadedDocument::new(path.into(), contents, metadata)) diff --git a/src/utils/request.rs b/src/utils/request.rs index 9f2804b..a479e0e 100644 --- a/src/utils/request.rs +++ b/src/utils/request.rs @@ -50,7 +50,17 @@ lazy_static::lazy_static! { static ref GITHUB_REPO_RE: Regex = Regex::new(r"^https://github\.com/([^/]+)/([^/]+)/tree/([^/]+)").unwrap(); } -pub async fn fetch( +pub async fn fetch(url: &str) -> Result { + let client = match *CLIENT { + Ok(ref client) => client, + Err(ref err) => bail!("{err}"), + }; + let res = client.get(url).send().await?; + let output = res.text().await?; + Ok(output) +} + +pub async fn fetch_with_loaders( loaders: &HashMap, path: &str, allow_media: bool, -- cgit v1.2.3