summaryrefslogtreecommitdiffstats
path: root/config.example.yaml
diff options
context:
space:
mode:
authorsigoden <sigoden@gmail.com>2024-06-05 09:02:23 +0800
committerGitHub <noreply@github.com>2024-06-05 09:02:23 +0800
commit1ec6abfaee2fdc189b348b7e3a8145bd9a84da74 (patch)
tree6f68a860a39b0fbe87784de925b5a31e00e74e33 /config.example.yaml
parent71f2e94579511d7524f5534377001ab3f02a9597 (diff)
downloadaichat-1ec6abfaee2fdc189b348b7e3a8145bd9a84da74.tar.gz
feat: support RAG (#560)
* feat: support RAG * support more embeddings models and implement concurrent embedding api * show the progress of addings paths * ignore embedding context when saving message * embedding model max_chunk_size => default_chunk_size * support pdf and pandoc formats (docx, epub, ipynb)
Diffstat (limited to 'config.example.yaml')
-rw-r--r--config.example.yaml37
1 files changed, 30 insertions, 7 deletions
diff --git a/config.example.yaml b/config.example.yaml
index b6ec6fe..86fe221 100644
--- a/config.example.yaml
+++ b/config.example.yaml
@@ -15,9 +15,24 @@ prelude: null # Set a default role or session to start with (
# if unset fallback to $EDITOR and $VISUAL
buffer_editor: null
-# Controls the function calling feature. For setup instructions, visit https://github.com/sigoden/llm-functions.
+# Controls the function calling feature. For setup instructions, visit https://github.com/sigoden/llm-functions
function_calling: false
+# Specifies the embedding model to use
+embedding_model: null
+
+# Determines how many relevant documents are retrieved
+rag_top_k: 4
+
+# Defines the query structure using variables like __CONTEXT__ and __INPUT__ to tailor searches to specific needs
+rag_template: |
+ Answer the following question based only on the provided context:
+ <context>
+ __CONTEXT__
+ </context>
+
+ Question: __INPUT__
+
# Compress session when token count reaches or exceeds this threshold (must be at least 1000)
compress_threshold: 4000
# Text prompt used for creating a concise summary of session message
@@ -26,7 +41,7 @@ summarize_prompt: 'Summarize the discussion briefly in 200 words or less to use
summary_prompt: 'This is a summary of the chat history as a recap: '
# Custom REPL prompt, see https://github.com/sigoden/aichat/wiki/Custom-REPL-Prompt
-left_prompt: '{color.green}{?session {session}{?role /}}{role}{color.cyan}{?session )}{!session >}{color.reset} '
+left_prompt: '{color.green}{?session {session}{?role /}}{role}{?rag #{rag}{color.cyan}{?session )}{!session >}{color.reset} '
right_prompt: '{color.purple}{?session {?consume_tokens {consume_tokens}({consume_percent}%)}{!consume_tokens {consume_tokens}}}{color.reset}'
clients:
@@ -34,13 +49,19 @@ clients:
# - type: xxxx
# name: xxxx # Only use it to distinguish clients with the same client type. Optional
# models:
- # - name: xxxx # The model name
+ # - name: xxxx
+ # mode: chat # Chat model
# max_input_tokens: 100000
# supports_vision: true
# supports_function_calling: true
+ # - name: xxxx
+ # mode: embedding # Embedding model
+ # max_input_tokens: 2048
+ # default_chunk_size: 2000
+ # max_concurrent_chunks: 100
# patches:
# <regex>: # The regex to match model names, e.g. '.*' 'gpt-4o' 'gpt-4o|gpt-4-.*'
- # request_body: # The JSON to be merged with the request body.
+ # chat_completions_body: # The JSON to be merged with the chat completions request body.
# extra:
# proxy: socks5://127.0.0.1:1080 # Set https/socks5 proxy. ENV: HTTPS_PROXY/https_proxy/ALL_PROXY/all_proxy
# connect_timeout: 10 # Set timeout in seconds for connect to api
@@ -66,7 +87,7 @@ clients:
api_key: xxx # ENV: {client}_API_KEY
patches:
'.*':
- request_body: # Override safetySettings for all models
+ chat_completions_body: # Override safetySettings for all models
safetySettings:
- category: HARM_CATEGORY_HARASSMENT
threshold: BLOCK_NONE
@@ -107,10 +128,12 @@ clients:
- type: ollama
api_base: http://localhost:11434 # ENV: {client}_API_BASE
api_auth: Basic xxx # ENV: {client}_API_AUTH
- chat_endpoint: /api/chat # Optional
models: # Required
- name: llama3
max_input_tokens: 8192
+ - name: all-minilm:l6-v2
+ mode: embedding
+ max_chunk_size: 1000
# See https://learn.microsoft.com/en-us/azure/ai-services/openai/chatgpt-quickstart
- type: azure-openai
@@ -130,7 +153,7 @@ clients:
adc_file: <path-to/gcloud/application_default_credentials.json>
patches:
'gemini-.*':
- request_body: # Override safetySettings for all gemini models
+ chat_completions_body: # Override safetySettings for all gemini models
safetySettings:
- category: HARM_CATEGORY_HARASSMENT
threshold: BLOCK_ONLY_HIGH