review and QA

This commit is contained in:
2026-08-15 22:49:18 +01:00
parent 3c7043562e
commit 7f01f1a341
10 changed files with 477 additions and 220 deletions
+196 -67
View File
@@ -1,16 +1,16 @@
#![warn(clippy::all, clippy::pedantic)]
use crate::converter::html_to_markdown;
use crate::doc_processor::{DocProcessResult, try_convert};
use crate::error_logger::log_error;
use crate::extractor::extract_links;
use crate::fetcher::{FetchResult, Fetcher};
use crate::robots::RobotsRule;
use crate::url_utils::{normalize_url, url_to_filename};
use crate::url_utils::{get_extension, is_in_scope, normalize_url, url_to_filename};
use anyhow::Result;
use std::collections::{HashSet, VecDeque};
use std::fs;
use std::sync::Arc;
use std::time::Duration;
use tokio::sync::Mutex;
use url::Url;
pub struct DocStats {
@@ -31,13 +31,16 @@ impl Default for DocStats {
pub struct Scraper {
fetcher: Fetcher,
seen: Arc<Mutex<HashSet<String>>>,
seen: HashSet<String>,
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
doc_stats: Arc<Mutex<DocStats>>,
doc_stats: DocStats,
convert_docs: bool,
include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>,
scope_path: Option<String>,
}
impl Scraper {
@@ -48,26 +51,56 @@ impl Scraper {
robots: RobotsRule,
convert_docs: bool,
fetcher: Fetcher,
include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>,
scope_path: Option<String>,
) -> Self {
Self {
fetcher,
seen: Arc::new(Mutex::new(HashSet::new())),
seen: HashSet::new(),
output_dir,
log_path,
base_domain,
robots,
doc_stats: Arc::new(Mutex::new(DocStats::default())),
doc_stats: DocStats::default(),
convert_docs,
include_types,
exclude_types,
scope_path,
}
}
pub async fn run(&self, start_url: &Url) -> (usize, usize) {
fn should_save(&self, url: &Url, content_type: &str) -> (bool, String) {
let ext = get_extension(url, content_type);
let ext_clean = ext.trim_start_matches('.').to_lowercase();
if let Some(include) = &self.include_types {
if include.contains(&ext_clean) {
(true, ext_clean)
} else {
(false, ext_clean)
}
} else if let Some(exclude) = &self.exclude_types {
if exclude.contains(&ext_clean) {
(false, ext_clean)
} else {
(true, ext_clean)
}
} else {
(true, ext_clean)
}
}
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque<Url> = VecDeque::new();
queue.push_back(start_url.clone());
let start_url_owned = start_url.clone();
queue.push_back(start_url_owned);
let mut count = 0usize;
let mut error_count = 0usize;
let mut skipped_robots = 0usize;
let mut skipped_scope = 0usize;
let mut skipped_type = 0usize;
while let Some(raw_url) = queue.pop_front() {
let url = match normalize_url(raw_url.as_str()) {
@@ -77,12 +110,15 @@ impl Scraper {
let url_key = url.as_str().to_string();
{
let mut seen = self.seen.lock().await;
if seen.contains(&url_key) {
continue;
}
seen.insert(url_key.clone());
if self.seen.contains(&url_key) {
continue;
}
self.seen.insert(url_key.clone());
if !is_in_scope(&url, &self.scope_path) {
log::debug!("[skip] out of scope: {}", url);
skipped_scope += 1;
continue;
}
if !self.robots.is_allowed(url.path()) {
@@ -98,46 +134,50 @@ impl Scraper {
let final_url = result.final_url.clone();
let is_html = result.content_type.contains("text/html");
match self.save(&result, is_html).await {
Ok(()) => {
if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let (should_save, ext_label) =
self.should_save(&result.final_url, &result.content_type);
let new_links: Vec<Url> = {
let seen = self.seen.lock().await;
links
.into_iter()
.filter(|link| !seen.contains(link.as_str()))
.collect()
};
if !should_save {
log::info!(" skipped (type filter: {})", ext_label);
skipped_type += 1;
}
for link in &new_links {
queue.push_back(link.clone());
}
log::info!(
" found {} links ({} new)",
new_count,
new_links.len()
);
if should_save {
match self.save(&result, is_html).await {
Ok(()) => {
if !is_html {
log::info!(" binary: {}", result.content_type);
}
} else {
log::info!(" binary: {}", result.content_type);
}
Err(e) => {
log_error(
&self.log_path,
&final_url,
"save_error",
&e.to_string(),
None,
);
error_count += 1;
self.doc_stats.errors += 1;
}
}
Err(e) => {
log_error(
&self.log_path,
&final_url,
"save_error",
&e.to_string(),
None,
);
error_count += 1;
let mut stats = self.doc_stats.lock().await;
stats.errors += 1;
}
if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
for link in &new_links {
queue.push_back(link.clone());
}
log::info!(" found {} links ({} new)", new_count, new_links.len());
}
}
}
@@ -151,32 +191,35 @@ impl Scraper {
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await;
}
{
let stats = self.doc_stats.lock().await;
if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 {
log::info!("\nDocument Statistics:");
log::info!(" Converted to Markdown: {}", stats.converted);
log::info!(" Kept as-is: {}", stats.raw);
log::info!(" Errors: {}", stats.errors);
}
if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 {
log::info!("\nDocument Statistics:");
log::info!(" Converted to Markdown: {}", self.doc_stats.converted);
log::info!(" Kept as-is: {}", self.doc_stats.raw);
log::info!(" Errors: {}", self.doc_stats.errors);
}
if skipped_robots > 0 {
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
}
if skipped_scope > 0 {
log::info!("Skipped {} URLs due to path scope", skipped_scope);
}
if skipped_type > 0 {
log::info!("Skipped {} URLs due to type filter", skipped_type);
}
(count, error_count)
}
async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> {
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
if is_html {
let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html);
if filename.ends_with(".html") {
filename = filename.replace(".html", ".md");
if let Some(stripped) = filename.strip_suffix(".html") {
filename = format!("{}.md", stripped);
}
content = md.into_bytes();
} else if self.convert_docs {
@@ -188,19 +231,17 @@ impl Scraper {
filename.push_str(".md");
content = md.into_bytes();
log::info!(" [DOC] converted to Markdown");
let mut stats = self.doc_stats.lock().await;
stats.converted += 1;
self.doc_stats.converted += 1;
}
DocProcessResult::Raw => {
content = result.bytes.clone();
content = result.bytes.to_vec();
if is_document_content_type(&result.content_type) {
let mut stats = self.doc_stats.lock().await;
stats.raw += 1;
self.doc_stats.raw += 1;
}
}
}
} else {
content = result.bytes.clone();
content = result.bytes.to_vec();
}
let filepath = self.output_dir.join(&filename);
@@ -236,3 +277,91 @@ fn is_document_content_type(ct: &str) -> bool {
];
DOC_TYPES.contains(&ct.as_str())
}
#[cfg(test)]
mod tests {
use super::*;
use reqwest::Client;
use url::Url;
fn make_scraper(
include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>,
) -> Scraper {
Scraper {
fetcher: Fetcher::new(Client::new()).unwrap(),
seen: HashSet::new(),
output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(),
base_domain: "example.com".to_string(),
robots: RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
},
doc_stats: DocStats::default(),
convert_docs: true,
include_types,
exclude_types,
scope_path: None,
}
}
#[test]
fn test_should_save_include_only_matching() {
let mut include = HashSet::new();
include.insert("pdf".to_string());
let scraper = make_scraper(Some(include), None);
let url = Url::parse("https://example.com/doc.pdf").unwrap();
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
assert!(should_save);
assert_eq!(ext, "pdf");
}
#[test]
fn test_should_save_include_non_matching() {
let mut include = HashSet::new();
include.insert("pdf".to_string());
let scraper = make_scraper(Some(include), None);
let url = Url::parse("https://example.com/doc.docx").unwrap();
let (should_save, ext) = scraper.should_save(
&url,
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
);
assert!(!should_save);
assert_eq!(ext, "docx");
}
#[test]
fn test_should_save_exclude_matching() {
let mut exclude = HashSet::new();
exclude.insert("pdf".to_string());
let scraper = make_scraper(None, Some(exclude));
let url = Url::parse("https://example.com/doc.pdf").unwrap();
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
assert!(!should_save);
assert_eq!(ext, "pdf");
}
#[test]
fn test_should_save_exclude_non_matching() {
let mut exclude = HashSet::new();
exclude.insert("pdf".to_string());
let scraper = make_scraper(None, Some(exclude));
let url = Url::parse("https://example.com/doc.docx").unwrap();
let (should_save, ext) = scraper.should_save(
&url,
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
);
assert!(should_save);
assert_eq!(ext, "docx");
}
#[test]
fn test_should_save_no_filters_allows_all() {
let scraper = make_scraper(None, None);
let url = Url::parse("https://example.com/doc.pdf").unwrap();
let (should_save, ext) = scraper.should_save(&url, "application/pdf");
assert!(should_save);
assert_eq!(ext, "pdf");
}
}