From 555f4f5d128d018912144b0e7f2b3a50223bf713 Mon Sep 17 00:00:00 2001 From: sigoden Date: Fri, 6 Sep 2024 08:08:55 +0800 Subject: feat: better html to markdown converter (#840) --- src/utils/request.rs | 16 ++++++---------- 1 file changed, 6 insertions(+), 10 deletions(-) (limited to 'src/utils/request.rs') diff --git a/src/utils/request.rs b/src/utils/request.rs index 73919d8..9f2804b 100644 --- a/src/utils/request.rs +++ b/src/utils/request.rs @@ -8,8 +8,11 @@ use reqwest::Url; use scraper::{Html, Selector}; use serde::Deserialize; use serde_json::Value; -use std::{collections::HashMap, time::Duration}; -use std::{collections::HashSet, sync::Arc}; +use std::{ + collections::{HashMap, HashSet}, + sync::Arc, + time::Duration, +}; use tokio::io::AsyncWriteExt; use tokio::sync::Semaphore; @@ -136,10 +139,7 @@ pub async fn fetch( None => { let contents = res.text().await?; if extension == "html" { - ( - html2text::from_read(contents.as_bytes(), usize::MAX), - "md".into(), - ) + (html_to_md(&contents), "md".into()) } else { (contents, extension) } @@ -387,10 +387,6 @@ async fn crawl_page( Ok((path.to_string(), text, links.into_iter().collect())) } -fn html_to_md(html: &str) -> String { - html2text::from_read(html.as_bytes(), usize::MAX) -} - fn should_exclude_link(link: &str, exclude: &[String]) -> bool { if link.contains("#") { return true; -- cgit v1.2.3