Files
web-scraper-rust/src/scraper.rs
T
2026-08-25 13:14:19 +01:00

376 lines
13 KiB
Rust

#![warn(clippy::all, clippy::pedantic)]
use crate::converter::html_to_markdown;
use crate::doc_processor::{DocProcessResult, try_convert};
use crate::error_logger::log_error;
use crate::extractor::extract_links;
use crate::fetcher::{FetchResult, Fetcher};
use crate::robots::RobotsRule;
use crate::url_utils::{get_extension, is_in_scope, normalize_url, url_to_filename};
use anyhow::Result;
use std::collections::{HashSet, VecDeque};
use std::fs;
use std::time::Duration;
use url::Url;
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
#[derive(Default)]
pub struct DocStats {
pub converted: usize,
pub raw: usize,
pub errors: usize,
}
pub struct Scraper {
fetcher: Fetcher,
seen: HashSet<String>,
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_host: String,
robots: RobotsRule,
doc_stats: DocStats,
convert_docs: bool,
include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>,
scope_path: Option<String>,
single_page: bool,
allow_subdomains: bool,
delay_ms: u64,
}
impl Scraper {
#[allow(clippy::too_many_arguments)]
pub fn new(
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_host: String,
robots: RobotsRule,
convert_docs: bool,
fetcher: Fetcher,
include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>,
scope_path: Option<String>,
single_page: bool,
allow_subdomains: bool,
delay_ms: u64,
) -> Self {
Self {
fetcher,
seen: HashSet::new(),
output_dir,
log_path,
base_host,
robots,
doc_stats: DocStats::default(),
convert_docs,
include_types,
exclude_types,
scope_path,
single_page,
allow_subdomains,
delay_ms,
}
}
fn should_save(&self, url: &Url, content_type: &str) -> (bool, String) {
let ext = get_extension(url, content_type);
let ext_clean = ext.trim_start_matches('.').to_lowercase();
if let Some(include) = &self.include_types {
if include.contains(&ext_clean) {
(true, ext_clean)
} else {
(false, ext_clean)
}
} else if let Some(exclude) = &self.exclude_types {
if exclude.contains(&ext_clean) {
(false, ext_clean)
} else {
(true, ext_clean)
}
} else {
(true, ext_clean)
}
}
fn extract_new_links(&self, html: &str, final_url: &Url) -> (usize, Vec<Url>) {
let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
(new_count, new_links)
}
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque<Url> = VecDeque::new();
let start_url_owned = start_url.clone();
queue.push_back(start_url_owned);
let mut count = 0usize;
let mut error_count = 0usize;
let mut skipped_robots = 0usize;
let mut skipped_scope = 0usize;
let mut skipped_type = 0usize;
while let Some(raw_url) = queue.pop_front() {
let Some(url) = normalize_url(raw_url.as_str()) else {
continue;
};
let url_key = url.as_str().to_string();
if self.seen.contains(&url_key) {
continue;
}
self.seen.insert(url_key.clone());
if !is_in_scope(&url, self.scope_path.as_ref()) {
log::debug!("[skip] out of scope: {url}");
skipped_scope += 1;
continue;
}
if !self.robots.is_allowed(url.path()) {
log::info!("[skip] robots.txt disallows: {url}");
skipped_robots += 1;
continue;
}
log::info!("[{count}] fetching: {url}");
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
let final_url = result.final_url.clone();
let is_html = result.content_type.contains("text/html");
let (should_save, ext_label) =
self.should_save(&result.final_url, &result.content_type);
if !should_save {
log::info!(" skipped (type filter: {ext_label})");
skipped_type += 1;
}
if should_save {
match self.save(&result, is_html) {
Ok(()) => {
if !is_html {
log::info!(" binary: {}", result.content_type);
}
}
Err(e) => {
log_error(
&self.log_path,
&final_url,
"save_error",
&e.to_string(),
None,
);
error_count += 1;
self.doc_stats.errors += 1;
}
}
}
if is_html {
if self.single_page {
log::info!(" [single-page mode] not crawling for links");
} else if let Ok(html) = std::str::from_utf8(&result.bytes) {
let (new_count, new_links) = self.extract_new_links(html, &final_url);
for link in &new_links {
queue.push_back(link.clone());
}
log::info!(" found {new_count} links ({} new)", new_links.len());
}
}
}
Err(e) => {
log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None);
error_count += 1;
}
}
count += 1;
tokio::time::sleep(Duration::from_millis(self.delay_ms)).await;
}
if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 {
log::info!("\nDocument Statistics:");
log::info!(" Converted to Markdown: {}", self.doc_stats.converted);
log::info!(" Kept as-is: {}", self.doc_stats.raw);
log::info!(" Errors: {}", self.doc_stats.errors);
}
if skipped_robots > 0 {
log::info!("Skipped {skipped_robots} URLs due to robots.txt");
}
if skipped_scope > 0 {
log::info!("Skipped {skipped_scope} URLs due to path scope");
}
if skipped_type > 0 {
log::info!("Skipped {skipped_type} URLs due to type filter");
}
(count, error_count)
}
fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
if is_html {
let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html);
if let Some(stripped) = filename.strip_suffix(".html") {
filename = format!("{stripped}.md");
}
content = md.into_bytes();
} else if self.convert_docs {
match try_convert(&result.bytes, &result.final_url) {
DocProcessResult::Markdown(md) => {
if let Some(pos) = filename.rfind('.') {
filename.truncate(pos);
}
filename.push_str(".md");
content = md.into_bytes();
log::info!(" [DOC] converted to Markdown");
self.doc_stats.converted += 1;
}
DocProcessResult::Raw => {
content = result.bytes.to_vec();
if is_document_content_type(&result.content_type) {
self.doc_stats.raw += 1;
}
}
}
} else {
content = result.bytes.to_vec();
}
let filepath = self.output_dir.join(&filename);
fs::write(&filepath, &content)?;
log::info!(
" saved -> {}",
filepath.file_name().unwrap_or_default().to_string_lossy()
);
Ok(())
}
}
fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
DOC_TYPES.contains(&ct.as_str())
}
#[cfg(test)]
mod tests {
use super::*;
use reqwest::Client;
use url::Url;
fn make_scraper(
include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>,
) -> Scraper {
Scraper {
fetcher: Fetcher::new(Client::new()),
seen: HashSet::new(),
output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(),
base_host: "example.com".to_string(),
allow_subdomains: false,
delay_ms: 1000,
robots: RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
},
doc_stats: DocStats::default(),
convert_docs: true,
include_types,
exclude_types,
scope_path: None,
single_page: false,
}
}
#[test]
fn test_should_save_include_only_matching() {
let mut include = HashSet::new();
include.insert("pdf".to_string());
let scraper = make_scraper(Some(include), None);
let url = Url::parse("https://example.com/doc.pdf").unwrap();
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
assert!(should_save);
assert_eq!(ext, "pdf");
}
#[test]
fn test_should_save_include_non_matching() {
let mut include = HashSet::new();
include.insert("pdf".to_string());
let scraper = make_scraper(Some(include), None);
let url = Url::parse("https://example.com/doc.docx").unwrap();
let (should_save, ext) = scraper.should_save(
&url,
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
);
assert!(!should_save);
assert_eq!(ext, "docx");
}
#[test]
fn test_should_save_exclude_matching() {
let mut exclude = HashSet::new();
exclude.insert("pdf".to_string());
let scraper = make_scraper(None, Some(exclude));
let url = Url::parse("https://example.com/doc.pdf").unwrap();
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
assert!(!should_save);
assert_eq!(ext, "pdf");
}
#[test]
fn test_should_save_exclude_non_matching() {
let mut exclude = HashSet::new();
exclude.insert("pdf".to_string());
let scraper = make_scraper(None, Some(exclude));
let url = Url::parse("https://example.com/doc.docx").unwrap();
let (should_save, ext) = scraper.should_save(
&url,
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
);
assert!(should_save);
assert_eq!(ext, "docx");
}
#[test]
fn test_should_save_no_filters_allows_all() {
let scraper = make_scraper(None, None);
let url = Url::parse("https://example.com/doc.pdf").unwrap();
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
assert!(should_save);
assert_eq!(ext, "pdf");
}
}