#![warn(clippy::all, clippy::pedantic)] use crate::converter::html_to_markdown; use crate::doc_processor::{DocProcessResult, try_convert}; use crate::error_logger::log_error; use crate::extractor::extract_links; use crate::fetcher::{FetchResult, Fetcher}; use crate::robots::RobotsRule; use crate::url_utils::{get_extension, is_in_scope, normalize_url, url_to_filename}; use anyhow::Result; use std::collections::{HashSet, VecDeque}; use std::fs; use std::time::Duration; use url::Url; const DOC_TYPES: &[&str] = &[ "application/pdf", "application/msword", "application/vnd.openxmlformats-officedocument.wordprocessingml.document", "application/vnd.ms-word.document.macroenabled.12", "application/vnd.ms-powerpoint", "application/vnd.openxmlformats-officedocument.presentationml.presentation", "application/vnd.ms-powerpoint.presentation.macroenabled.12", "application/vnd.ms-excel", "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "application/vnd.ms-excel.sheet.macroenabled.12", "application/vnd.ms-excel.sheet.binary.macroenabled.12", "application/vnd.oasis.opendocument.text", "application/vnd.oasis.opendocument.spreadsheet", "application/vnd.oasis.opendocument.presentation", "application/rtf", "application/epub+zip", "text/csv", ]; #[derive(Default)] pub struct DocStats { pub converted: usize, pub raw: usize, pub errors: usize, } pub struct Scraper { fetcher: Fetcher, seen: HashSet, output_dir: std::path::PathBuf, log_path: std::path::PathBuf, base_host: String, robots: RobotsRule, doc_stats: DocStats, convert_docs: bool, include_types: Option>, exclude_types: Option>, scope_path: Option, single_page: bool, allow_subdomains: bool, delay_ms: u64, } impl Scraper { #[allow(clippy::too_many_arguments)] pub fn new( output_dir: std::path::PathBuf, log_path: std::path::PathBuf, base_host: String, robots: RobotsRule, convert_docs: bool, fetcher: Fetcher, include_types: Option>, exclude_types: Option>, scope_path: Option, single_page: bool, allow_subdomains: bool, delay_ms: u64, ) -> Self { Self { fetcher, seen: HashSet::new(), output_dir, log_path, base_host, robots, doc_stats: DocStats::default(), convert_docs, include_types, exclude_types, scope_path, single_page, allow_subdomains, delay_ms, } } fn should_save(&self, url: &Url, content_type: &str) -> (bool, String) { let ext = get_extension(url, content_type); let ext_clean = ext.trim_start_matches('.').to_lowercase(); if let Some(include) = &self.include_types { if include.contains(&ext_clean) { (true, ext_clean) } else { (false, ext_clean) } } else if let Some(exclude) = &self.exclude_types { if exclude.contains(&ext_clean) { (false, ext_clean) } else { (true, ext_clean) } } else { (true, ext_clean) } } fn extract_new_links(&self, html: &str, final_url: &Url) -> (usize, Vec) { let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains); let new_count = links.len(); let new_links: Vec = links .into_iter() .filter(|link| !self.seen.contains(link.as_str())) .collect(); (new_count, new_links) } pub async fn run(&mut self, start_url: &Url) -> (usize, usize) { let mut queue: VecDeque = VecDeque::new(); let start_url_owned = start_url.clone(); queue.push_back(start_url_owned); let mut count = 0usize; let mut error_count = 0usize; let mut skipped_robots = 0usize; let mut skipped_scope = 0usize; let mut skipped_type = 0usize; while let Some(raw_url) = queue.pop_front() { let Some(url) = normalize_url(raw_url.as_str()) else { continue; }; let url_key = url.as_str().to_string(); if self.seen.contains(&url_key) { continue; } self.seen.insert(url_key.clone()); if !is_in_scope(&url, self.scope_path.as_ref()) { log::debug!("[skip] out of scope: {url}"); skipped_scope += 1; continue; } if !self.robots.is_allowed(url.path()) { log::info!("[skip] robots.txt disallows: {url}"); skipped_robots += 1; continue; } log::info!("[{count}] fetching: {url}"); match self.fetcher.fetch_with_retry(&url).await { Ok(result) => { let final_url = result.final_url.clone(); let is_html = result.content_type.contains("text/html"); let (should_save, ext_label) = self.should_save(&result.final_url, &result.content_type); if !should_save { log::info!(" skipped (type filter: {ext_label})"); skipped_type += 1; } if should_save { match self.save(&result, is_html) { Ok(()) => { if !is_html { log::info!(" binary: {}", result.content_type); } } Err(e) => { log_error( &self.log_path, &final_url, "save_error", &e.to_string(), None, ); error_count += 1; self.doc_stats.errors += 1; } } } if is_html { if self.single_page { log::info!(" [single-page mode] not crawling for links"); } else if let Ok(html) = std::str::from_utf8(&result.bytes) { let (new_count, new_links) = self.extract_new_links(html, &final_url); for link in &new_links { queue.push_back(link.clone()); } log::info!(" found {new_count} links ({} new)", new_links.len()); } } } Err(e) => { log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None); error_count += 1; } } count += 1; tokio::time::sleep(Duration::from_millis(self.delay_ms)).await; } if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 { log::info!("\nDocument Statistics:"); log::info!(" Converted to Markdown: {}", self.doc_stats.converted); log::info!(" Kept as-is: {}", self.doc_stats.raw); log::info!(" Errors: {}", self.doc_stats.errors); } if skipped_robots > 0 { log::info!("Skipped {skipped_robots} URLs due to robots.txt"); } if skipped_scope > 0 { log::info!("Skipped {skipped_scope} URLs due to path scope"); } if skipped_type > 0 { log::info!("Skipped {skipped_type} URLs due to type filter"); } (count, error_count) } fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> { let mut filename = url_to_filename(&result.final_url, &result.content_type); let content: Vec; if is_html { let html = std::str::from_utf8(&result.bytes)?; let md = html_to_markdown(html); if let Some(stripped) = filename.strip_suffix(".html") { filename = format!("{stripped}.md"); } content = md.into_bytes(); } else if self.convert_docs { match try_convert(&result.bytes, &result.final_url) { DocProcessResult::Markdown(md) => { if let Some(pos) = filename.rfind('.') { filename.truncate(pos); } filename.push_str(".md"); content = md.into_bytes(); log::info!(" [DOC] converted to Markdown"); self.doc_stats.converted += 1; } DocProcessResult::Raw => { content = result.bytes.to_vec(); if is_document_content_type(&result.content_type) { self.doc_stats.raw += 1; } } } } else { content = result.bytes.to_vec(); } let filepath = self.output_dir.join(&filename); fs::write(&filepath, &content)?; log::info!( " saved -> {}", filepath.file_name().unwrap_or_default().to_string_lossy() ); Ok(()) } } fn is_document_content_type(ct: &str) -> bool { let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase(); DOC_TYPES.contains(&ct.as_str()) } #[cfg(test)] mod tests { use super::*; use reqwest::Client; use url::Url; fn make_scraper( include_types: Option>, exclude_types: Option>, ) -> Scraper { Scraper { fetcher: Fetcher::new(Client::new()), seen: HashSet::new(), output_dir: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(), base_host: "example.com".to_string(), allow_subdomains: false, delay_ms: 1000, robots: RobotsRule { allowed: Vec::new(), disallowed: Vec::new(), }, doc_stats: DocStats::default(), convert_docs: true, include_types, exclude_types, scope_path: None, single_page: false, } } #[test] fn test_should_save_include_only_matching() { let mut include = HashSet::new(); include.insert("pdf".to_string()); let scraper = make_scraper(Some(include), None); let url = Url::parse("https://example.com/doc.pdf").unwrap(); let (should_save, ext) = scraper.should_save(&url, "application/pdf"); assert!(should_save); assert_eq!(ext, "pdf"); } #[test] fn test_should_save_include_non_matching() { let mut include = HashSet::new(); include.insert("pdf".to_string()); let scraper = make_scraper(Some(include), None); let url = Url::parse("https://example.com/doc.docx").unwrap(); let (should_save, ext) = scraper.should_save( &url, "application/vnd.openxmlformats-officedocument.wordprocessingml.document", ); assert!(!should_save); assert_eq!(ext, "docx"); } #[test] fn test_should_save_exclude_matching() { let mut exclude = HashSet::new(); exclude.insert("pdf".to_string()); let scraper = make_scraper(None, Some(exclude)); let url = Url::parse("https://example.com/doc.pdf").unwrap(); let (should_save, ext) = scraper.should_save(&url, "application/pdf"); assert!(!should_save); assert_eq!(ext, "pdf"); } #[test] fn test_should_save_exclude_non_matching() { let mut exclude = HashSet::new(); exclude.insert("pdf".to_string()); let scraper = make_scraper(None, Some(exclude)); let url = Url::parse("https://example.com/doc.docx").unwrap(); let (should_save, ext) = scraper.should_save( &url, "application/vnd.openxmlformats-officedocument.wordprocessingml.document", ); assert!(should_save); assert_eq!(ext, "docx"); } #[test] fn test_should_save_no_filters_allows_all() { let scraper = make_scraper(None, None); let url = Url::parse("https://example.com/doc.pdf").unwrap(); let (should_save, ext) = scraper.should_save(&url, "application/pdf"); assert!(should_save); assert_eq!(ext, "pdf"); } }