review and QA
This commit is contained in:
+196
-67
@@ -1,16 +1,16 @@
|
||||
#![warn(clippy::all, clippy::pedantic)]
|
||||
|
||||
use crate::converter::html_to_markdown;
|
||||
use crate::doc_processor::{DocProcessResult, try_convert};
|
||||
use crate::error_logger::log_error;
|
||||
use crate::extractor::extract_links;
|
||||
use crate::fetcher::{FetchResult, Fetcher};
|
||||
use crate::robots::RobotsRule;
|
||||
use crate::url_utils::{normalize_url, url_to_filename};
|
||||
use crate::url_utils::{get_extension, is_in_scope, normalize_url, url_to_filename};
|
||||
use anyhow::Result;
|
||||
use std::collections::{HashSet, VecDeque};
|
||||
use std::fs;
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
use tokio::sync::Mutex;
|
||||
use url::Url;
|
||||
|
||||
pub struct DocStats {
|
||||
@@ -31,13 +31,16 @@ impl Default for DocStats {
|
||||
|
||||
pub struct Scraper {
|
||||
fetcher: Fetcher,
|
||||
seen: Arc<Mutex<HashSet<String>>>,
|
||||
seen: HashSet<String>,
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
robots: RobotsRule,
|
||||
doc_stats: Arc<Mutex<DocStats>>,
|
||||
doc_stats: DocStats,
|
||||
convert_docs: bool,
|
||||
include_types: Option<HashSet<String>>,
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
scope_path: Option<String>,
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
@@ -48,26 +51,56 @@ impl Scraper {
|
||||
robots: RobotsRule,
|
||||
convert_docs: bool,
|
||||
fetcher: Fetcher,
|
||||
include_types: Option<HashSet<String>>,
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
scope_path: Option<String>,
|
||||
) -> Self {
|
||||
Self {
|
||||
fetcher,
|
||||
seen: Arc::new(Mutex::new(HashSet::new())),
|
||||
seen: HashSet::new(),
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
robots,
|
||||
doc_stats: Arc::new(Mutex::new(DocStats::default())),
|
||||
doc_stats: DocStats::default(),
|
||||
convert_docs,
|
||||
include_types,
|
||||
exclude_types,
|
||||
scope_path,
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn run(&self, start_url: &Url) -> (usize, usize) {
|
||||
fn should_save(&self, url: &Url, content_type: &str) -> (bool, String) {
|
||||
let ext = get_extension(url, content_type);
|
||||
let ext_clean = ext.trim_start_matches('.').to_lowercase();
|
||||
|
||||
if let Some(include) = &self.include_types {
|
||||
if include.contains(&ext_clean) {
|
||||
(true, ext_clean)
|
||||
} else {
|
||||
(false, ext_clean)
|
||||
}
|
||||
} else if let Some(exclude) = &self.exclude_types {
|
||||
if exclude.contains(&ext_clean) {
|
||||
(false, ext_clean)
|
||||
} else {
|
||||
(true, ext_clean)
|
||||
}
|
||||
} else {
|
||||
(true, ext_clean)
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
|
||||
let mut queue: VecDeque<Url> = VecDeque::new();
|
||||
queue.push_back(start_url.clone());
|
||||
let start_url_owned = start_url.clone();
|
||||
queue.push_back(start_url_owned);
|
||||
|
||||
let mut count = 0usize;
|
||||
let mut error_count = 0usize;
|
||||
let mut skipped_robots = 0usize;
|
||||
let mut skipped_scope = 0usize;
|
||||
let mut skipped_type = 0usize;
|
||||
|
||||
while let Some(raw_url) = queue.pop_front() {
|
||||
let url = match normalize_url(raw_url.as_str()) {
|
||||
@@ -77,12 +110,15 @@ impl Scraper {
|
||||
|
||||
let url_key = url.as_str().to_string();
|
||||
|
||||
{
|
||||
let mut seen = self.seen.lock().await;
|
||||
if seen.contains(&url_key) {
|
||||
continue;
|
||||
}
|
||||
seen.insert(url_key.clone());
|
||||
if self.seen.contains(&url_key) {
|
||||
continue;
|
||||
}
|
||||
self.seen.insert(url_key.clone());
|
||||
|
||||
if !is_in_scope(&url, &self.scope_path) {
|
||||
log::debug!("[skip] out of scope: {}", url);
|
||||
skipped_scope += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
if !self.robots.is_allowed(url.path()) {
|
||||
@@ -98,46 +134,50 @@ impl Scraper {
|
||||
let final_url = result.final_url.clone();
|
||||
let is_html = result.content_type.contains("text/html");
|
||||
|
||||
match self.save(&result, is_html).await {
|
||||
Ok(()) => {
|
||||
if is_html {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links = extract_links(html, &final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
let (should_save, ext_label) =
|
||||
self.should_save(&result.final_url, &result.content_type);
|
||||
|
||||
let new_links: Vec<Url> = {
|
||||
let seen = self.seen.lock().await;
|
||||
links
|
||||
.into_iter()
|
||||
.filter(|link| !seen.contains(link.as_str()))
|
||||
.collect()
|
||||
};
|
||||
if !should_save {
|
||||
log::info!(" skipped (type filter: {})", ext_label);
|
||||
skipped_type += 1;
|
||||
}
|
||||
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
|
||||
log::info!(
|
||||
" found {} links ({} new)",
|
||||
new_count,
|
||||
new_links.len()
|
||||
);
|
||||
if should_save {
|
||||
match self.save(&result, is_html).await {
|
||||
Ok(()) => {
|
||||
if !is_html {
|
||||
log::info!(" binary: {}", result.content_type);
|
||||
}
|
||||
} else {
|
||||
log::info!(" binary: {}", result.content_type);
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(
|
||||
&self.log_path,
|
||||
&final_url,
|
||||
"save_error",
|
||||
&e.to_string(),
|
||||
None,
|
||||
);
|
||||
error_count += 1;
|
||||
self.doc_stats.errors += 1;
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(
|
||||
&self.log_path,
|
||||
&final_url,
|
||||
"save_error",
|
||||
&e.to_string(),
|
||||
None,
|
||||
);
|
||||
error_count += 1;
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
}
|
||||
|
||||
if is_html {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links = extract_links(html, &final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
|
||||
let new_links: Vec<Url> = links
|
||||
.into_iter()
|
||||
.filter(|link| !self.seen.contains(link.as_str()))
|
||||
.collect();
|
||||
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
|
||||
log::info!(" found {} links ({} new)", new_count, new_links.len());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -151,32 +191,35 @@ impl Scraper {
|
||||
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await;
|
||||
}
|
||||
|
||||
{
|
||||
let stats = self.doc_stats.lock().await;
|
||||
if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 {
|
||||
log::info!("\nDocument Statistics:");
|
||||
log::info!(" Converted to Markdown: {}", stats.converted);
|
||||
log::info!(" Kept as-is: {}", stats.raw);
|
||||
log::info!(" Errors: {}", stats.errors);
|
||||
}
|
||||
if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 {
|
||||
log::info!("\nDocument Statistics:");
|
||||
log::info!(" Converted to Markdown: {}", self.doc_stats.converted);
|
||||
log::info!(" Kept as-is: {}", self.doc_stats.raw);
|
||||
log::info!(" Errors: {}", self.doc_stats.errors);
|
||||
}
|
||||
|
||||
if skipped_robots > 0 {
|
||||
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
}
|
||||
if skipped_scope > 0 {
|
||||
log::info!("Skipped {} URLs due to path scope", skipped_scope);
|
||||
}
|
||||
if skipped_type > 0 {
|
||||
log::info!("Skipped {} URLs due to type filter", skipped_type);
|
||||
}
|
||||
|
||||
(count, error_count)
|
||||
}
|
||||
|
||||
async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
||||
let content: Vec<u8>;
|
||||
|
||||
if is_html {
|
||||
let html = std::str::from_utf8(&result.bytes)?;
|
||||
let md = html_to_markdown(html);
|
||||
if filename.ends_with(".html") {
|
||||
filename = filename.replace(".html", ".md");
|
||||
if let Some(stripped) = filename.strip_suffix(".html") {
|
||||
filename = format!("{}.md", stripped);
|
||||
}
|
||||
content = md.into_bytes();
|
||||
} else if self.convert_docs {
|
||||
@@ -188,19 +231,17 @@ impl Scraper {
|
||||
filename.push_str(".md");
|
||||
content = md.into_bytes();
|
||||
log::info!(" [DOC] converted to Markdown");
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.converted += 1;
|
||||
self.doc_stats.converted += 1;
|
||||
}
|
||||
DocProcessResult::Raw => {
|
||||
content = result.bytes.clone();
|
||||
content = result.bytes.to_vec();
|
||||
if is_document_content_type(&result.content_type) {
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.raw += 1;
|
||||
self.doc_stats.raw += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
content = result.bytes.clone();
|
||||
content = result.bytes.to_vec();
|
||||
}
|
||||
|
||||
let filepath = self.output_dir.join(&filename);
|
||||
@@ -236,3 +277,91 @@ fn is_document_content_type(ct: &str) -> bool {
|
||||
];
|
||||
DOC_TYPES.contains(&ct.as_str())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use reqwest::Client;
|
||||
use url::Url;
|
||||
|
||||
fn make_scraper(
|
||||
include_types: Option<HashSet<String>>,
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
) -> Scraper {
|
||||
Scraper {
|
||||
fetcher: Fetcher::new(Client::new()).unwrap(),
|
||||
seen: HashSet::new(),
|
||||
output_dir: std::path::PathBuf::new(),
|
||||
log_path: std::path::PathBuf::new(),
|
||||
base_domain: "example.com".to_string(),
|
||||
robots: RobotsRule {
|
||||
allowed: Vec::new(),
|
||||
disallowed: Vec::new(),
|
||||
},
|
||||
doc_stats: DocStats::default(),
|
||||
convert_docs: true,
|
||||
include_types,
|
||||
exclude_types,
|
||||
scope_path: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_save_include_only_matching() {
|
||||
let mut include = HashSet::new();
|
||||
include.insert("pdf".to_string());
|
||||
let scraper = make_scraper(Some(include), None);
|
||||
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
||||
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
|
||||
assert!(should_save);
|
||||
assert_eq!(ext, "pdf");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_save_include_non_matching() {
|
||||
let mut include = HashSet::new();
|
||||
include.insert("pdf".to_string());
|
||||
let scraper = make_scraper(Some(include), None);
|
||||
let url = Url::parse("https://example.com/doc.docx").unwrap();
|
||||
let (should_save, ext) = scraper.should_save(
|
||||
&url,
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
);
|
||||
assert!(!should_save);
|
||||
assert_eq!(ext, "docx");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_save_exclude_matching() {
|
||||
let mut exclude = HashSet::new();
|
||||
exclude.insert("pdf".to_string());
|
||||
let scraper = make_scraper(None, Some(exclude));
|
||||
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
||||
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
|
||||
assert!(!should_save);
|
||||
assert_eq!(ext, "pdf");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_save_exclude_non_matching() {
|
||||
let mut exclude = HashSet::new();
|
||||
exclude.insert("pdf".to_string());
|
||||
let scraper = make_scraper(None, Some(exclude));
|
||||
let url = Url::parse("https://example.com/doc.docx").unwrap();
|
||||
let (should_save, ext) = scraper.should_save(
|
||||
&url,
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
);
|
||||
assert!(should_save);
|
||||
assert_eq!(ext, "docx");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_save_no_filters_allows_all() {
|
||||
let scraper = make_scraper(None, None);
|
||||
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
||||
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
|
||||
assert!(should_save);
|
||||
assert_eq!(ext, "pdf");
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user