376 lines
13 KiB
Rust
376 lines
13 KiB
Rust
#![warn(clippy::all, clippy::pedantic)]
|
|
|
|
use crate::converter::html_to_markdown;
|
|
use crate::doc_processor::{DocProcessResult, try_convert};
|
|
use crate::error_logger::log_error;
|
|
use crate::extractor::extract_links;
|
|
use crate::fetcher::{FetchResult, Fetcher};
|
|
use crate::robots::RobotsRule;
|
|
use crate::url_utils::{get_extension, is_in_scope, normalize_url, url_to_filename};
|
|
use anyhow::Result;
|
|
use std::collections::{HashSet, VecDeque};
|
|
use std::fs;
|
|
use std::time::Duration;
|
|
use url::Url;
|
|
|
|
const DOC_TYPES: &[&str] = &[
|
|
"application/pdf",
|
|
"application/msword",
|
|
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
"application/vnd.ms-word.document.macroenabled.12",
|
|
"application/vnd.ms-powerpoint",
|
|
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
|
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
|
"application/vnd.ms-excel",
|
|
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
"application/vnd.ms-excel.sheet.macroenabled.12",
|
|
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
|
"application/vnd.oasis.opendocument.text",
|
|
"application/vnd.oasis.opendocument.spreadsheet",
|
|
"application/vnd.oasis.opendocument.presentation",
|
|
"application/rtf",
|
|
"application/epub+zip",
|
|
"text/csv",
|
|
];
|
|
|
|
#[derive(Default)]
|
|
pub struct DocStats {
|
|
pub converted: usize,
|
|
pub raw: usize,
|
|
pub errors: usize,
|
|
}
|
|
|
|
pub struct Scraper {
|
|
fetcher: Fetcher,
|
|
seen: HashSet<String>,
|
|
output_dir: std::path::PathBuf,
|
|
log_path: std::path::PathBuf,
|
|
base_host: String,
|
|
robots: RobotsRule,
|
|
doc_stats: DocStats,
|
|
convert_docs: bool,
|
|
include_types: Option<HashSet<String>>,
|
|
exclude_types: Option<HashSet<String>>,
|
|
scope_path: Option<String>,
|
|
single_page: bool,
|
|
allow_subdomains: bool,
|
|
delay_ms: u64,
|
|
}
|
|
|
|
impl Scraper {
|
|
#[allow(clippy::too_many_arguments)]
|
|
pub fn new(
|
|
output_dir: std::path::PathBuf,
|
|
log_path: std::path::PathBuf,
|
|
base_host: String,
|
|
robots: RobotsRule,
|
|
convert_docs: bool,
|
|
fetcher: Fetcher,
|
|
include_types: Option<HashSet<String>>,
|
|
exclude_types: Option<HashSet<String>>,
|
|
scope_path: Option<String>,
|
|
single_page: bool,
|
|
allow_subdomains: bool,
|
|
delay_ms: u64,
|
|
) -> Self {
|
|
Self {
|
|
fetcher,
|
|
seen: HashSet::new(),
|
|
output_dir,
|
|
log_path,
|
|
base_host,
|
|
robots,
|
|
doc_stats: DocStats::default(),
|
|
convert_docs,
|
|
include_types,
|
|
exclude_types,
|
|
scope_path,
|
|
single_page,
|
|
allow_subdomains,
|
|
delay_ms,
|
|
}
|
|
}
|
|
|
|
fn should_save(&self, url: &Url, content_type: &str) -> (bool, String) {
|
|
let ext = get_extension(url, content_type);
|
|
let ext_clean = ext.trim_start_matches('.').to_lowercase();
|
|
|
|
if let Some(include) = &self.include_types {
|
|
if include.contains(&ext_clean) {
|
|
(true, ext_clean)
|
|
} else {
|
|
(false, ext_clean)
|
|
}
|
|
} else if let Some(exclude) = &self.exclude_types {
|
|
if exclude.contains(&ext_clean) {
|
|
(false, ext_clean)
|
|
} else {
|
|
(true, ext_clean)
|
|
}
|
|
} else {
|
|
(true, ext_clean)
|
|
}
|
|
}
|
|
|
|
fn extract_new_links(&self, html: &str, final_url: &Url) -> (usize, Vec<Url>) {
|
|
let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains);
|
|
let new_count = links.len();
|
|
let new_links: Vec<Url> = links
|
|
.into_iter()
|
|
.filter(|link| !self.seen.contains(link.as_str()))
|
|
.collect();
|
|
(new_count, new_links)
|
|
}
|
|
|
|
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
|
|
let mut queue: VecDeque<Url> = VecDeque::new();
|
|
let start_url_owned = start_url.clone();
|
|
queue.push_back(start_url_owned);
|
|
|
|
let mut count = 0usize;
|
|
let mut error_count = 0usize;
|
|
let mut skipped_robots = 0usize;
|
|
let mut skipped_scope = 0usize;
|
|
let mut skipped_type = 0usize;
|
|
|
|
while let Some(raw_url) = queue.pop_front() {
|
|
let Some(url) = normalize_url(raw_url.as_str()) else {
|
|
continue;
|
|
};
|
|
|
|
let url_key = url.as_str().to_string();
|
|
|
|
if self.seen.contains(&url_key) {
|
|
continue;
|
|
}
|
|
self.seen.insert(url_key.clone());
|
|
|
|
if !is_in_scope(&url, self.scope_path.as_ref()) {
|
|
log::debug!("[skip] out of scope: {url}");
|
|
skipped_scope += 1;
|
|
continue;
|
|
}
|
|
|
|
if !self.robots.is_allowed(url.path()) {
|
|
log::info!("[skip] robots.txt disallows: {url}");
|
|
skipped_robots += 1;
|
|
continue;
|
|
}
|
|
|
|
log::info!("[{count}] fetching: {url}");
|
|
|
|
match self.fetcher.fetch_with_retry(&url).await {
|
|
Ok(result) => {
|
|
let final_url = result.final_url.clone();
|
|
let is_html = result.content_type.contains("text/html");
|
|
|
|
let (should_save, ext_label) =
|
|
self.should_save(&result.final_url, &result.content_type);
|
|
|
|
if !should_save {
|
|
log::info!(" skipped (type filter: {ext_label})");
|
|
skipped_type += 1;
|
|
}
|
|
|
|
if should_save {
|
|
match self.save(&result, is_html) {
|
|
Ok(()) => {
|
|
if !is_html {
|
|
log::info!(" binary: {}", result.content_type);
|
|
}
|
|
}
|
|
Err(e) => {
|
|
log_error(
|
|
&self.log_path,
|
|
&final_url,
|
|
"save_error",
|
|
&e.to_string(),
|
|
None,
|
|
);
|
|
error_count += 1;
|
|
self.doc_stats.errors += 1;
|
|
}
|
|
}
|
|
}
|
|
|
|
if is_html {
|
|
if self.single_page {
|
|
log::info!(" [single-page mode] not crawling for links");
|
|
} else if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
|
let (new_count, new_links) = self.extract_new_links(html, &final_url);
|
|
for link in &new_links {
|
|
queue.push_back(link.clone());
|
|
}
|
|
log::info!(" found {new_count} links ({} new)", new_links.len());
|
|
}
|
|
}
|
|
}
|
|
Err(e) => {
|
|
log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None);
|
|
error_count += 1;
|
|
}
|
|
}
|
|
|
|
count += 1;
|
|
tokio::time::sleep(Duration::from_millis(self.delay_ms)).await;
|
|
}
|
|
|
|
if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 {
|
|
log::info!("\nDocument Statistics:");
|
|
log::info!(" Converted to Markdown: {}", self.doc_stats.converted);
|
|
log::info!(" Kept as-is: {}", self.doc_stats.raw);
|
|
log::info!(" Errors: {}", self.doc_stats.errors);
|
|
}
|
|
|
|
if skipped_robots > 0 {
|
|
log::info!("Skipped {skipped_robots} URLs due to robots.txt");
|
|
}
|
|
if skipped_scope > 0 {
|
|
log::info!("Skipped {skipped_scope} URLs due to path scope");
|
|
}
|
|
if skipped_type > 0 {
|
|
log::info!("Skipped {skipped_type} URLs due to type filter");
|
|
}
|
|
|
|
(count, error_count)
|
|
}
|
|
|
|
fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
|
|
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
|
let content: Vec<u8>;
|
|
|
|
if is_html {
|
|
let html = std::str::from_utf8(&result.bytes)?;
|
|
let md = html_to_markdown(html);
|
|
if let Some(stripped) = filename.strip_suffix(".html") {
|
|
filename = format!("{stripped}.md");
|
|
}
|
|
content = md.into_bytes();
|
|
} else if self.convert_docs {
|
|
match try_convert(&result.bytes, &result.final_url) {
|
|
DocProcessResult::Markdown(md) => {
|
|
if let Some(pos) = filename.rfind('.') {
|
|
filename.truncate(pos);
|
|
}
|
|
filename.push_str(".md");
|
|
content = md.into_bytes();
|
|
log::info!(" [DOC] converted to Markdown");
|
|
self.doc_stats.converted += 1;
|
|
}
|
|
DocProcessResult::Raw => {
|
|
content = result.bytes.to_vec();
|
|
if is_document_content_type(&result.content_type) {
|
|
self.doc_stats.raw += 1;
|
|
}
|
|
}
|
|
}
|
|
} else {
|
|
content = result.bytes.to_vec();
|
|
}
|
|
|
|
let filepath = self.output_dir.join(&filename);
|
|
fs::write(&filepath, &content)?;
|
|
log::info!(
|
|
" saved -> {}",
|
|
filepath.file_name().unwrap_or_default().to_string_lossy()
|
|
);
|
|
Ok(())
|
|
}
|
|
}
|
|
|
|
fn is_document_content_type(ct: &str) -> bool {
|
|
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
|
|
DOC_TYPES.contains(&ct.as_str())
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
use reqwest::Client;
|
|
use url::Url;
|
|
|
|
fn make_scraper(
|
|
include_types: Option<HashSet<String>>,
|
|
exclude_types: Option<HashSet<String>>,
|
|
) -> Scraper {
|
|
Scraper {
|
|
fetcher: Fetcher::new(Client::new()),
|
|
seen: HashSet::new(),
|
|
output_dir: std::path::PathBuf::new(),
|
|
log_path: std::path::PathBuf::new(),
|
|
base_host: "example.com".to_string(),
|
|
allow_subdomains: false,
|
|
delay_ms: 1000,
|
|
robots: RobotsRule {
|
|
allowed: Vec::new(),
|
|
disallowed: Vec::new(),
|
|
},
|
|
doc_stats: DocStats::default(),
|
|
convert_docs: true,
|
|
include_types,
|
|
exclude_types,
|
|
scope_path: None,
|
|
single_page: false,
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_should_save_include_only_matching() {
|
|
let mut include = HashSet::new();
|
|
include.insert("pdf".to_string());
|
|
let scraper = make_scraper(Some(include), None);
|
|
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
|
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
|
|
assert!(should_save);
|
|
assert_eq!(ext, "pdf");
|
|
}
|
|
|
|
#[test]
|
|
fn test_should_save_include_non_matching() {
|
|
let mut include = HashSet::new();
|
|
include.insert("pdf".to_string());
|
|
let scraper = make_scraper(Some(include), None);
|
|
let url = Url::parse("https://example.com/doc.docx").unwrap();
|
|
let (should_save, ext) = scraper.should_save(
|
|
&url,
|
|
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
);
|
|
assert!(!should_save);
|
|
assert_eq!(ext, "docx");
|
|
}
|
|
|
|
#[test]
|
|
fn test_should_save_exclude_matching() {
|
|
let mut exclude = HashSet::new();
|
|
exclude.insert("pdf".to_string());
|
|
let scraper = make_scraper(None, Some(exclude));
|
|
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
|
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
|
|
assert!(!should_save);
|
|
assert_eq!(ext, "pdf");
|
|
}
|
|
|
|
#[test]
|
|
fn test_should_save_exclude_non_matching() {
|
|
let mut exclude = HashSet::new();
|
|
exclude.insert("pdf".to_string());
|
|
let scraper = make_scraper(None, Some(exclude));
|
|
let url = Url::parse("https://example.com/doc.docx").unwrap();
|
|
let (should_save, ext) = scraper.should_save(
|
|
&url,
|
|
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
);
|
|
assert!(should_save);
|
|
assert_eq!(ext, "docx");
|
|
}
|
|
|
|
#[test]
|
|
fn test_should_save_no_filters_allows_all() {
|
|
let scraper = make_scraper(None, None);
|
|
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
|
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
|
|
assert!(should_save);
|
|
assert_eq!(ext, "pdf");
|
|
}
|
|
}
|