summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorsigoden <sigoden@gmail.com>2025-02-10 17:46:24 +0800
committerGitHub <noreply@github.com>2025-02-10 17:46:24 +0800
commit78f3d4ff49679da251a035f783d5b8380730cce5 (patch)
tree28777a63e04b6be8a97bf5e6de39af8ef673bfb4
parent4fab4c54b2916931bf5705b5aa5e560e57f7f289 (diff)
downloadaichat-78f3d4ff49679da251a035f783d5b8380730cce5.tar.gz
feat: drop predefined models for ollama (#1165)
-rwxr-xr-xArgcfile.sh1
-rw-r--r--config.example.yaml26
-rw-r--r--models.yaml29
-rw-r--r--src/client/common.rs29
-rw-r--r--src/client/mod.rs3
5 files changed, 29 insertions, 59 deletions
diff --git a/Argcfile.sh b/Argcfile.sh
index 4ef3d16..52101b0 100755
--- a/Argcfile.sh
+++ b/Argcfile.sh
@@ -313,7 +313,6 @@ _argc_before() {
moonshot,moonshot-v1-8k,https://api.moonshot.cn/v1 \
novita,meta-llama/llama-3.1-8b-instruct,https://api.novita.ai/v3/openai \
openrouter,openai/gpt-4o-mini,https://openrouter.ai/api/v1 \
- ollama,llama3.1:latest,http://${OLLAMA_HOST:-"127.0.0.1:11434"}/v1 \
perplexity,llama-3.1-8b-instruct,https://api.perplexity.ai \
qianwen,qwen-turbo-latest,https://dashscope.aliyuncs.com/compatible-mode/v1 \
siliconflow,meta-llama/Meta-Llama-3.1-8B-Instruct,https://api.siliconflow.cn/v1 \
diff --git a/config.example.yaml b/config.example.yaml
index 7e81083..43cec93 100644
--- a/config.example.yaml
+++ b/config.example.yaml
@@ -98,13 +98,10 @@ clients:
# supports_reasoning: true
# - name: xxxx # Embedding model
# type: embedding
- # max_input_tokens: 200000
- # max_tokens_per_chunk: 2000
# default_chunk_size: 1500
# max_batch_size: 100
# - name: xxxx # Reranker model
# type: reranker
- # max_input_tokens: 2048
# patch: # Patch api
# chat_completions: # Api type, possible values: chat_completions, embeddings, and rerank
# <regex>: # The regex to match model names, e.g. '.*' 'gpt-4o' 'gpt-4o|gpt-4-.*'
@@ -125,19 +122,23 @@ clients:
# For any platform compatible with OpenAI's API
- type: openai-compatible
- name: local
- api_base: http://localhost:8080/v1
+ name: ollama
+ api_base: http://localhost:11434/v1
api_key: xxx # Optional
models:
+ - name: deepseek-r1
+ max_input_tokens: 131072
+ supports_reasoning: true
- name: llama3.1
max_input_tokens: 128000
supports_function_calling: true
- - name: jina-embeddings-v2-base-en
+ - name: llama3.2-vision
+ max_input_tokens: 131072
+ supports_vision: true
+ - name: nomic-embed-text
type: embedding
- default_chunk_size: 1500
- max_batch_size: 100
- - name: jina-reranker-v2-base-multilingual
- type: reranker
+ default_chunk_size: 1000
+ max_batch_size: 50
# See https://ai.google.dev/docs
- type: gemini
@@ -197,11 +198,6 @@ clients:
api_base: https://api.groq.com/openai/v1
api_key: xxx
- # See https://github.com/jmorganca/ollama
- - type: openai-compatible
- name: ollama
- api_base: http://localhost:11434/v1
-
# See https://learn.microsoft.com/en-us/azure/ai-services/openai/chatgpt-quickstart
- type: azure-openai
api_base: https://{RESOURCE}.openai.azure.com
diff --git a/models.yaml b/models.yaml
index a2e8e97..4875ce7 100644
--- a/models.yaml
+++ b/models.yaml
@@ -441,35 +441,6 @@
supports_reasoning: true
# Links:
-# - https://ollama.com/library
-# - https://github.com/ollama/ollama/blob/main/docs/openai.md
-- provider: ollama
- models:
- - name: llama3.1
- max_input_tokens: 131072
- supports_function_calling: true
- - name: llama3.2
- max_input_tokens: 131072
- supports_function_calling: true
- - name: llama3.2-vision
- max_input_tokens: 131072
- supports_vision: true
- - name: qwen2.5
- max_input_tokens: 131072
- supports_function_calling: true
- - name: qwen2.5-coder
- max_input_tokens: 32768
- supports_function_calling: true
- - name: deepseek-r1
- max_input_tokens: 131072
- supports_reasoning: true
- - name: nomic-embed-text
- type: embedding
- max_tokens_per_chunk: 8192
- default_chunk_size: 1000
- max_batch_size: 50
-
-# Links:
# - https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models
# - https://cloud.google.com/vertex-ai/generative-ai/docs/model-garden/explore-models
# - https://cloud.google.com/vertex-ai/generative-ai/pricing
diff --git a/src/client/common.rs b/src/client/common.rs
index cfbb9d6..99b84b9 100644
--- a/src/client/common.rs
+++ b/src/client/common.rs
@@ -553,24 +553,29 @@ async fn set_client_models_config(client_config: &mut Value, client: &str) -> Re
std::env::var(&env_name).ok()
}),
) {
- if let Ok(fetched_models) = abortable_run_with_spinner(
+ match abortable_run_with_spinner(
fetch_models(api_base, api_key.as_deref()),
"Fetching models",
create_abort_signal(),
)
.await
{
- model_names = MultiSelect::new("LLM models (required):", fetched_models)
- .with_validator(|list: &[ListOption<&String>]| {
- if list.is_empty() {
- Ok(Validation::Invalid(
- "At least one item must be selected".into(),
- ))
- } else {
- Ok(Validation::Valid)
- }
- })
- .prompt()?;
+ Ok(fetched_models) => {
+ model_names = MultiSelect::new("LLM models (required):", fetched_models)
+ .with_validator(|list: &[ListOption<&String>]| {
+ if list.is_empty() {
+ Ok(Validation::Invalid(
+ "At least one item must be selected".into(),
+ ))
+ } else {
+ Ok(Validation::Valid)
+ }
+ })
+ .prompt()?;
+ }
+ Err(err) => {
+ eprintln!("✗ Unable to fetch models: {err}");
+ }
}
}
if model_names.is_empty() {
diff --git a/src/client/mod.rs b/src/client/mod.rs
index c6b65bc..af0ace1 100644
--- a/src/client/mod.rs
+++ b/src/client/mod.rs
@@ -33,7 +33,7 @@ register_client!(
(bedrock, "bedrock", BedrockConfig, BedrockClient),
);
-pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 25] = [
+pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 24] = [
("ai21", "https://api.ai21.com/studio/v1"),
(
"cloudflare",
@@ -53,7 +53,6 @@ pub const OPENAI_COMPATIBLE_PROVIDERS: [(&str, &str); 25] = [
("moonshot", "https://api.moonshot.cn/v1"),
("novita", "https://api.novita.ai/v3/openai"),
("openrouter", "https://openrouter.ai/api/v1"),
- ("ollama", "http://{OLLAMA_HOST}:11434/v1"),
("perplexity", "https://api.perplexity.ai"),
(
"qianwen",