This commit is contained in:
2026-08-14 22:18:11 +01:00
parent c655b939a4
commit a60cd5deae
10 changed files with 1136 additions and 35 deletions
+108 -10
View File
@@ -1,24 +1,43 @@
// src/scraper.rs
use crate::converter::html_to_markdown;
use crate::error_logger::log_error;
use crate::extractor::extract_links;
use crate::fetcher::{FetchResult, Fetcher};
use crate::pdf_processor::{PdfProcessResult, process_pdf};
use crate::robots::RobotsRule;
use crate::url_utils::{normalize_url, url_to_filename};
use anyhow::Result;
use std::collections::HashSet;
use std::collections::VecDeque;
use std::collections::{HashSet, VecDeque};
use std::fs;
use std::sync::Arc;
use std::time::Duration;
use tokio::sync::Mutex;
use url::Url;
pub struct PdfStats {
pub text_based: usize,
pub scanned: usize,
pub errors: usize,
}
impl Default for PdfStats {
fn default() -> Self {
Self {
text_based: 0,
scanned: 0,
errors: 0,
}
}
}
pub struct Scraper {
fetcher: Fetcher,
seen: Arc<Mutex<HashSet<String>>>,
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
pdf_stats: Arc<Mutex<PdfStats>>,
convert_pdfs: bool,
}
impl Scraper {
@@ -26,6 +45,8 @@ impl Scraper {
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
convert_pdfs: bool,
) -> Self {
Self {
fetcher: Fetcher::new(),
@@ -33,6 +54,9 @@ impl Scraper {
output_dir,
log_path,
base_domain,
robots,
pdf_stats: Arc::new(Mutex::new(PdfStats::default())),
convert_pdfs,
}
}
@@ -42,6 +66,7 @@ impl Scraper {
let mut count = 0usize;
let mut error_count = 0usize;
let mut skipped_robots = 0usize;
while let Some(raw_url) = queue.pop_front() {
let url = match normalize_url(raw_url.as_str()) {
@@ -59,21 +84,26 @@ impl Scraper {
seen.insert(url_key.clone());
}
if !self.robots.is_allowed(url.path()) {
println!("[skip] robots.txt disallows: {}", url);
skipped_robots += 1;
continue;
}
println!("[{}] fetching: {}", count, url);
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
let is_html = result.content_type.contains("text/html");
let is_pdf = result.content_type.contains("application/pdf");
match self.save(&result).await {
match self.save(&result, is_html, is_pdf).await {
Ok(()) => {
if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links =
extract_links(html, &result.final_url, &self.base_domain);
// Only queue links not already seen — do NOT insert into seen here.
// They get inserted when actually fetched, matching the Python version.
let mut new_links = Vec::new();
{
@@ -106,6 +136,11 @@ impl Scraper {
None,
);
error_count += 1;
if is_pdf {
let mut stats = self.pdf_stats.lock().await;
stats.errors += 1;
}
}
}
}
@@ -119,13 +154,76 @@ impl Scraper {
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
}
{
let stats = self.pdf_stats.lock().await;
if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 {
println!("\nPDF Statistics:");
println!(" Text-based (converted): {}", stats.text_based);
println!(" Scanned (kept as PDF): {}", stats.scanned);
println!(" Errors: {}", stats.errors);
}
}
if skipped_robots > 0 {
println!("Skipped {} URLs due to robots.txt", skipped_robots);
}
(count, error_count)
}
async fn save(&self, result: &FetchResult) -> Result<()> {
let filename = url_to_filename(&result.final_url, &result.content_type);
async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
if is_html {
let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html);
if filename.ends_with(".html") {
filename = filename.replace(".html", ".md");
}
content = md.into_bytes();
} else if is_pdf && self.convert_pdfs {
// Save PDF to temp file for pdf-inspector processing
let temp_path = self.output_dir.join("temp_pdf.tmp");
fs::write(&temp_path, &result.bytes)?;
match process_pdf(&temp_path) {
Ok(PdfProcessResult::Markdown(md)) => {
if filename.ends_with(".pdf") {
filename = filename.replace(".pdf", ".md");
}
content = md.into_bytes();
fs::remove_file(&temp_path)?;
println!(" [PDF] text-based — converted to Markdown");
let mut stats = self.pdf_stats.lock().await;
stats.text_based += 1;
}
Ok(PdfProcessResult::Scanned) => {
println!(" [PDF] scanned document — saved as-is");
let mut stats = self.pdf_stats.lock().await;
stats.scanned += 1;
fs::rename(&temp_path, self.output_dir.join(&filename))?;
return Ok(());
}
Err(e) => {
println!(" [PDF] processing error ({}), saved as-is", e);
let mut stats = self.pdf_stats.lock().await;
stats.errors += 1;
fs::rename(&temp_path, self.output_dir.join(&filename))?;
return Ok(());
}
}
} else {
// Non-HTML, non-PDF, or PDF conversion disabled — save as-is
content = result.bytes.clone();
}
let filepath = self.output_dir.join(&filename);
fs::write(&filepath, &result.bytes)?;
fs::write(&filepath, &content)?;
println!(
" saved -> {}",
filepath.file_name().unwrap_or_default().to_string_lossy()