diff options
| author | sigoden <sigoden@gmail.com> | 2024-09-02 10:38:53 +0800 |
|---|---|---|
| committer | GitHub <noreply@github.com> | 2024-09-02 10:38:53 +0800 |
| commit | dc78636129427111e8949dcb0160aee15c0f3e91 (patch) | |
| tree | ab4ea2661830e61911cebfc16c7ac41216f9f392 | |
| parent | e39498e340a192077ec64d14c72470cd99716a71 (diff) | |
| download | aichat-dc78636129427111e8949dcb0160aee15c0f3e91.tar.gz | |
feat: add huggingface client (#822)
| -rwxr-xr-x | Argcfile.sh | 5 | ||||
| -rw-r--r-- | config.example.yaml | 8 | ||||
| -rw-r--r-- | models.yaml | 18 | ||||
| -rw-r--r-- | src/client/mod.rs | 6 |
4 files changed, 33 insertions, 4 deletions
diff --git a/Argcfile.sh b/Argcfile.sh index ac8f5da..31d34bd 100755 --- a/Argcfile.sh +++ b/Argcfile.sh @@ -85,7 +85,10 @@ OPENAI_COMPATIBLE_PLATFORMS=( \ deepinfra,meta-llama/Meta-Llama-3.1-8B-Instruct,https://api.deepinfra.com/v1/openai \ deepseek,deepseek-chat,https://api.deepseek.com \ fireworks,accounts/fireworks/models/llama-v3p1-8b-instruct,https://api.fireworks.ai/inference/v1 \ + github,gpt-4o-mini,https://models.inference.ai.azure.com \ groq,llama3-8b-8192,https://api.groq.com/openai/v1 \ + huggingface,meta-llama/Meta-Llama-3-8B-Instruct,https://api-inference.huggingface.co/v1 \ + lingyiwanwu,yi-large,https://api.lingyiwanwu.com/v1 \ mistral,open-mistral-nemo,https://api.mistral.ai/v1 \ moonshot,moonshot-v1-8k,https://api.moonshot.cn/v1 \ openrouter,openai/gpt-4o-mini,https://openrouter.ai/api/v1 \ @@ -95,8 +98,6 @@ OPENAI_COMPATIBLE_PLATFORMS=( \ qianwen,qwen-turbo,https://dashscope.aliyuncs.com/compatible-mode/v1 \ together,meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo,https://api.together.xyz/v1 \ zhipuai,glm-4-0520,https://open.bigmodel.cn/api/paas/v4 \ - lingyiwanwu,yi-large,https://api.lingyiwanwu.com/v1 \ - github,gpt-4o-mini,https://models.inference.ai.azure.com \ ) # @cmd Chat with any LLM api diff --git a/config.example.yaml b/config.example.yaml index 35ebf45..65e4d71 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -228,6 +228,12 @@ clients: api_base: https://api.cloudflare.com/client/v4/accounts/{ACCOUNT_ID}/ai/v1 api_key: xxx + # See https://huggingface.co/inference-api/serverless + - type: openai-compatible + name: huggingface + api_base: https://api-inference.huggingface.co/v1 + api_key: hf_xxx + # See https://replicate.com/docs - type: replicate api_key: xxx @@ -303,6 +309,8 @@ clients: api_base: https://api.together.xyz/v1 api_key: xxx + # ----- RAG dedicated ----- + # See https://jina.ai - type: openai-compatible name: jina diff --git a/models.yaml b/models.yaml index 3b30583..1a02da1 100644 --- a/models.yaml +++ b/models.yaml @@ -543,6 +543,24 @@ max_batch_size: 100 # Links: +# - https://huggingface.co/models?inference=warm&pipeline_tag=text-generation&other=text-generation-inference&sort=trending +# - https://huggingface.co/docs/text-generation-inference/en/reference/api_reference +- platform: huggingface + models: + - name: meta-llama/Meta-Llama-3-8B-Instruct + max_input_tokens: 8192 + max_output_tokens: 4096 + require_max_tokens: true + input_price: 0 + output_price: 0 + - name: mistralai/Mistral-Nemo-Instruct-2407 + max_input_tokens: 128000 + max_output_tokens: 4096 + require_max_tokens: true + input_price: 0 + output_price: 0 + +# Links: # - https://replicate.com/explore # - https://replicate.com/pricing # - https://replicate.com/docs/reference/http#create-a-prediction-using-an-official-model diff --git a/src/client/mod.rs b/src/client/mod.rs index ecafd3c..eb9d3e9 100644 --- a/src/client/mod.rs +++ b/src/client/mod.rs @@ -37,7 +37,7 @@ register_client!( (ernie, "ernie", ErnieConfig, ErnieClient), ); -pub const OPENAI_COMPATIBLE_PLATFORMS: [(&str, &str); 19] = [ +pub const OPENAI_COMPATIBLE_PLATFORMS: [(&str, &str); 20] = [ ("ai21", "https://api.ai21.com/studio/v1"), ("cloudflare", ""), ("deepinfra", "https://api.deepinfra.com/v1/openai"), @@ -45,7 +45,7 @@ pub const OPENAI_COMPATIBLE_PLATFORMS: [(&str, &str); 19] = [ ("fireworks", "https://api.fireworks.ai/inference/v1"), ("github", "https://models.inference.ai.azure.com"), ("groq", "https://api.groq.com/openai/v1"), - ("jina", "https://api.jina.ai/v1"), + ("huggingface", "https://api-inference.huggingface.co/v1"), ("lingyiwanwu", "https://api.lingyiwanwu.com/v1"), ("mistral", "https://api.mistral.ai/v1"), ("moonshot", "https://api.moonshot.cn/v1"), @@ -59,5 +59,7 @@ pub const OPENAI_COMPATIBLE_PLATFORMS: [(&str, &str); 19] = [ ), ("together", "https://api.together.xyz/v1"), ("zhipuai", "https://open.bigmodel.cn/api/paas/v4"), + // RAG-dedicated + ("jina", "https://api.jina.ai/v1"), ("voyageai", "https://api.voyageai.com/v1"), ]; |
