fix clippy warnigns
This commit is contained in:
+57
-61
@@ -13,22 +13,33 @@ use std::fs;
|
||||
use std::time::Duration;
|
||||
use url::Url;
|
||||
|
||||
const DOC_TYPES: &[&str] = &[
|
||||
"application/pdf",
|
||||
"application/msword",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"application/vnd.ms-word.document.macroenabled.12",
|
||||
"application/vnd.ms-powerpoint",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
||||
"application/vnd.ms-excel",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
"application/vnd.oasis.opendocument.spreadsheet",
|
||||
"application/vnd.oasis.opendocument.presentation",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"text/csv",
|
||||
];
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct DocStats {
|
||||
pub converted: usize,
|
||||
pub raw: usize,
|
||||
pub errors: usize,
|
||||
}
|
||||
|
||||
impl Default for DocStats {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
converted: 0,
|
||||
raw: 0,
|
||||
errors: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct Scraper {
|
||||
fetcher: Fetcher,
|
||||
seen: HashSet<String>,
|
||||
@@ -45,6 +56,7 @@ pub struct Scraper {
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn new(
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
@@ -94,6 +106,20 @@ impl Scraper {
|
||||
}
|
||||
}
|
||||
|
||||
fn extract_new_links(
|
||||
&self,
|
||||
html: &str,
|
||||
final_url: &Url,
|
||||
) -> (usize, Vec<Url>) {
|
||||
let links = extract_links(html, final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
let new_links: Vec<Url> = links
|
||||
.into_iter()
|
||||
.filter(|link| !self.seen.contains(link.as_str()))
|
||||
.collect();
|
||||
(new_count, new_links)
|
||||
}
|
||||
|
||||
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
|
||||
let mut queue: VecDeque<Url> = VecDeque::new();
|
||||
let start_url_owned = start_url.clone();
|
||||
@@ -106,9 +132,8 @@ impl Scraper {
|
||||
let mut skipped_type = 0usize;
|
||||
|
||||
while let Some(raw_url) = queue.pop_front() {
|
||||
let url = match normalize_url(raw_url.as_str()) {
|
||||
Some(u) => u,
|
||||
None => continue,
|
||||
let Some(url) = normalize_url(raw_url.as_str()) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let url_key = url.as_str().to_string();
|
||||
@@ -118,19 +143,19 @@ impl Scraper {
|
||||
}
|
||||
self.seen.insert(url_key.clone());
|
||||
|
||||
if !is_in_scope(&url, &self.scope_path) {
|
||||
log::debug!("[skip] out of scope: {}", url);
|
||||
if !is_in_scope(&url, self.scope_path.as_ref()) {
|
||||
log::debug!("[skip] out of scope: {url}");
|
||||
skipped_scope += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
if !self.robots.is_allowed(url.path()) {
|
||||
log::info!("[skip] robots.txt disallows: {}", url);
|
||||
log::info!("[skip] robots.txt disallows: {url}");
|
||||
skipped_robots += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
log::info!("[{}] fetching: {}", count, url);
|
||||
log::info!("[{count}] fetching: {url}");
|
||||
|
||||
match self.fetcher.fetch_with_retry(&url).await {
|
||||
Ok(result) => {
|
||||
@@ -141,12 +166,12 @@ impl Scraper {
|
||||
self.should_save(&result.final_url, &result.content_type);
|
||||
|
||||
if !should_save {
|
||||
log::info!(" skipped (type filter: {})", ext_label);
|
||||
log::info!(" skipped (type filter: {ext_label})");
|
||||
skipped_type += 1;
|
||||
}
|
||||
|
||||
if should_save {
|
||||
match self.save(&result, is_html).await {
|
||||
match self.save(&result, is_html) {
|
||||
Ok(()) => {
|
||||
if !is_html {
|
||||
log::info!(" binary: {}", result.content_type);
|
||||
@@ -167,24 +192,14 @@ impl Scraper {
|
||||
}
|
||||
|
||||
if is_html {
|
||||
if !self.single_page {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links = extract_links(html, &final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
|
||||
let new_links: Vec<Url> = links
|
||||
.into_iter()
|
||||
.filter(|link| !self.seen.contains(link.as_str()))
|
||||
.collect();
|
||||
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
|
||||
log::info!(" found {} links ({} new)", new_count, new_links.len());
|
||||
}
|
||||
} else {
|
||||
if self.single_page {
|
||||
log::info!(" [single-page mode] not crawling for links");
|
||||
} else if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let (new_count, new_links) = self.extract_new_links(html, &final_url);
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
log::info!(" found {new_count} links ({} new)", new_links.len());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -206,19 +221,19 @@ impl Scraper {
|
||||
}
|
||||
|
||||
if skipped_robots > 0 {
|
||||
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
log::info!("Skipped {skipped_robots} URLs due to robots.txt");
|
||||
}
|
||||
if skipped_scope > 0 {
|
||||
log::info!("Skipped {} URLs due to path scope", skipped_scope);
|
||||
log::info!("Skipped {skipped_scope} URLs due to path scope");
|
||||
}
|
||||
if skipped_type > 0 {
|
||||
log::info!("Skipped {} URLs due to type filter", skipped_type);
|
||||
log::info!("Skipped {skipped_type} URLs due to type filter");
|
||||
}
|
||||
|
||||
(count, error_count)
|
||||
}
|
||||
|
||||
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
||||
let content: Vec<u8>;
|
||||
|
||||
@@ -226,7 +241,7 @@ impl Scraper {
|
||||
let html = std::str::from_utf8(&result.bytes)?;
|
||||
let md = html_to_markdown(html);
|
||||
if let Some(stripped) = filename.strip_suffix(".html") {
|
||||
filename = format!("{}.md", stripped);
|
||||
filename = format!("{stripped}.md");
|
||||
}
|
||||
content = md.into_bytes();
|
||||
} else if self.convert_docs {
|
||||
@@ -263,25 +278,6 @@ impl Scraper {
|
||||
|
||||
fn is_document_content_type(ct: &str) -> bool {
|
||||
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
|
||||
const DOC_TYPES: &[&str] = &[
|
||||
"application/pdf",
|
||||
"application/msword",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"application/vnd.ms-word.document.macroenabled.12",
|
||||
"application/vnd.ms-powerpoint",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
||||
"application/vnd.ms-excel",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
"application/vnd.oasis.opendocument.spreadsheet",
|
||||
"application/vnd.oasis.opendocument.presentation",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"text/csv",
|
||||
];
|
||||
DOC_TYPES.contains(&ct.as_str())
|
||||
}
|
||||
|
||||
@@ -296,7 +292,7 @@ mod tests {
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
) -> Scraper {
|
||||
Scraper {
|
||||
fetcher: Fetcher::new(Client::new()).unwrap(),
|
||||
fetcher: Fetcher::new(Client::new()),
|
||||
seen: HashSet::new(),
|
||||
output_dir: std::path::PathBuf::new(),
|
||||
log_path: std::path::PathBuf::new(),
|
||||
|
||||
Reference in New Issue
Block a user