From f82524fd154a1bf4a7277db868147afdb0cd507a Mon Sep 17 00:00:00 2001 From: sigoden Date: Thu, 27 Jun 2024 12:30:09 +0800 Subject: feat: implement native rag url loader (#660) --- config.example.yaml | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) (limited to 'config.example.yaml') diff --git a/config.example.yaml b/config.example.yaml index 262f41b..c7c5375 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -51,11 +51,12 @@ rag_min_score_keyword_search: 0 # Specifies the minimum relevance sc rag_min_score_rerank: 0 # Specifies the minimum relevance score for reranking # Defines document loaders rag_document_loaders: - # You can add more loaders, here is the syntax: - # : + # You can add custom loaders using the following syntax: + # : + # Note: Use `$1` for input filepath and `$2` for output filepath. If `$2` is not provided, output to stdout. pdf: 'pdftotext $1 -' # Load .pdf file, see https://poppler.freedesktop.org docx: 'pandoc --to plain $1' # Load .docx file - url: 'curl -fsSL $1' # Load url + # xlsx: 'ssconvert $1 $2' # Load .xlsx file # recursive_url: 'crawler $1 $2' # Load websites # Defines the query structure using variables like __CONTEXT__ and __INPUT__ to tailor searches to specific needs -- cgit v1.2.3