qa and review
This commit is contained in:
+87
-82
@@ -1,8 +1,8 @@
|
||||
use crate::converter::html_to_markdown;
|
||||
use crate::doc_processor::{DocProcessResult, try_convert};
|
||||
use crate::error_logger::log_error;
|
||||
use crate::extractor::extract_links;
|
||||
use crate::fetcher::{FetchResult, Fetcher};
|
||||
use crate::pdf_processor::{PdfProcessResult, process_pdf};
|
||||
use crate::robots::RobotsRule;
|
||||
use crate::url_utils::{normalize_url, url_to_filename};
|
||||
use anyhow::Result;
|
||||
@@ -13,17 +13,17 @@ use std::time::Duration;
|
||||
use tokio::sync::Mutex;
|
||||
use url::Url;
|
||||
|
||||
pub struct PdfStats {
|
||||
pub text_based: usize,
|
||||
pub scanned: usize,
|
||||
pub struct DocStats {
|
||||
pub converted: usize,
|
||||
pub raw: usize,
|
||||
pub errors: usize,
|
||||
}
|
||||
|
||||
impl Default for PdfStats {
|
||||
impl Default for DocStats {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
text_based: 0,
|
||||
scanned: 0,
|
||||
converted: 0,
|
||||
raw: 0,
|
||||
errors: 0,
|
||||
}
|
||||
}
|
||||
@@ -36,8 +36,8 @@ pub struct Scraper {
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
robots: RobotsRule,
|
||||
pdf_stats: Arc<Mutex<PdfStats>>,
|
||||
convert_pdfs: bool,
|
||||
doc_stats: Arc<Mutex<DocStats>>,
|
||||
convert_docs: bool,
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
@@ -46,17 +46,18 @@ impl Scraper {
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
robots: RobotsRule,
|
||||
convert_pdfs: bool,
|
||||
convert_docs: bool,
|
||||
fetcher: Fetcher,
|
||||
) -> Self {
|
||||
Self {
|
||||
fetcher: Fetcher::new(),
|
||||
fetcher,
|
||||
seen: Arc::new(Mutex::new(HashSet::new())),
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
robots,
|
||||
pdf_stats: Arc::new(Mutex::new(PdfStats::default())),
|
||||
convert_pdfs,
|
||||
doc_stats: Arc::new(Mutex::new(DocStats::default())),
|
||||
convert_docs,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -85,62 +86,58 @@ impl Scraper {
|
||||
}
|
||||
|
||||
if !self.robots.is_allowed(url.path()) {
|
||||
println!("[skip] robots.txt disallows: {}", url);
|
||||
log::info!("[skip] robots.txt disallows: {}", url);
|
||||
skipped_robots += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
println!("[{}] fetching: {}", count, url);
|
||||
log::info!("[{}] fetching: {}", count, url);
|
||||
|
||||
match self.fetcher.fetch_with_retry(&url).await {
|
||||
Ok(result) => {
|
||||
let final_url = result.final_url.clone();
|
||||
let is_html = result.content_type.contains("text/html");
|
||||
let is_pdf = result.content_type.contains("application/pdf");
|
||||
|
||||
match self.save(&result, is_html, is_pdf).await {
|
||||
match self.save(&result, is_html).await {
|
||||
Ok(()) => {
|
||||
if is_html {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links =
|
||||
extract_links(html, &result.final_url, &self.base_domain);
|
||||
let links = extract_links(html, &final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
|
||||
let mut new_links = Vec::new();
|
||||
|
||||
{
|
||||
let new_links: Vec<Url> = {
|
||||
let seen = self.seen.lock().await;
|
||||
for link in &links {
|
||||
if !seen.contains(link.as_str()) {
|
||||
new_links.push(link.clone());
|
||||
}
|
||||
}
|
||||
links
|
||||
.into_iter()
|
||||
.filter(|link| !seen.contains(link.as_str()))
|
||||
.collect()
|
||||
};
|
||||
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
|
||||
let new_count = new_links.len();
|
||||
|
||||
for link in new_links {
|
||||
queue.push_back(link);
|
||||
}
|
||||
|
||||
println!(" found {} links ({} new)", links.len(), new_count);
|
||||
log::info!(
|
||||
" found {} links ({} new)",
|
||||
new_count,
|
||||
new_links.len()
|
||||
);
|
||||
}
|
||||
} else {
|
||||
println!(" binary: {}", result.content_type);
|
||||
log::info!(" binary: {}", result.content_type);
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(
|
||||
&self.log_path,
|
||||
&result.final_url,
|
||||
&final_url,
|
||||
"save_error",
|
||||
&e.to_string(),
|
||||
None,
|
||||
);
|
||||
error_count += 1;
|
||||
|
||||
if is_pdf {
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
}
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -151,29 +148,28 @@ impl Scraper {
|
||||
}
|
||||
|
||||
count += 1;
|
||||
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
|
||||
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await;
|
||||
}
|
||||
|
||||
{
|
||||
let stats = self.pdf_stats.lock().await;
|
||||
if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 {
|
||||
println!("\nPDF Statistics:");
|
||||
println!(" Text-based (converted): {}", stats.text_based);
|
||||
println!(" Scanned (kept as PDF): {}", stats.scanned);
|
||||
println!(" Errors: {}", stats.errors);
|
||||
let stats = self.doc_stats.lock().await;
|
||||
if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 {
|
||||
log::info!("\nDocument Statistics:");
|
||||
log::info!(" Converted to Markdown: {}", stats.converted);
|
||||
log::info!(" Kept as-is: {}", stats.raw);
|
||||
log::info!(" Errors: {}", stats.errors);
|
||||
}
|
||||
}
|
||||
|
||||
if skipped_robots > 0 {
|
||||
println!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
}
|
||||
|
||||
(count, error_count)
|
||||
}
|
||||
|
||||
async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> {
|
||||
async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
||||
|
||||
let content: Vec<u8>;
|
||||
|
||||
if is_html {
|
||||
@@ -183,51 +179,60 @@ impl Scraper {
|
||||
filename = filename.replace(".html", ".md");
|
||||
}
|
||||
content = md.into_bytes();
|
||||
} else if is_pdf && self.convert_pdfs {
|
||||
// Save PDF to temp file for pdf-inspector processing
|
||||
let temp_path = self.output_dir.join("temp_pdf.tmp");
|
||||
fs::write(&temp_path, &result.bytes)?;
|
||||
|
||||
match process_pdf(&temp_path) {
|
||||
Ok(PdfProcessResult::Markdown(md)) => {
|
||||
if filename.ends_with(".pdf") {
|
||||
filename = filename.replace(".pdf", ".md");
|
||||
} else if self.convert_docs {
|
||||
match try_convert(&result.bytes, &result.final_url) {
|
||||
DocProcessResult::Markdown(md) => {
|
||||
if let Some(pos) = filename.rfind('.') {
|
||||
filename.truncate(pos);
|
||||
}
|
||||
filename.push_str(".md");
|
||||
content = md.into_bytes();
|
||||
fs::remove_file(&temp_path)?;
|
||||
|
||||
println!(" [PDF] text-based — converted to Markdown");
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.text_based += 1;
|
||||
log::info!(" [DOC] converted to Markdown");
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.converted += 1;
|
||||
}
|
||||
Ok(PdfProcessResult::Scanned) => {
|
||||
println!(" [PDF] scanned document — saved as-is");
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.scanned += 1;
|
||||
|
||||
fs::rename(&temp_path, self.output_dir.join(&filename))?;
|
||||
return Ok(());
|
||||
}
|
||||
Err(e) => {
|
||||
println!(" [PDF] processing error ({}), saved as-is", e);
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
|
||||
fs::rename(&temp_path, self.output_dir.join(&filename))?;
|
||||
return Ok(());
|
||||
DocProcessResult::Raw => {
|
||||
content = result.bytes.clone();
|
||||
if is_document_content_type(&result.content_type) {
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.raw += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Non-HTML, non-PDF, or PDF conversion disabled — save as-is
|
||||
content = result.bytes.clone();
|
||||
}
|
||||
|
||||
let filepath = self.output_dir.join(&filename);
|
||||
fs::write(&filepath, &content)?;
|
||||
println!(
|
||||
log::info!(
|
||||
" saved -> {}",
|
||||
filepath.file_name().unwrap_or_default().to_string_lossy()
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn is_document_content_type(ct: &str) -> bool {
|
||||
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
|
||||
const DOC_TYPES: &[&str] = &[
|
||||
"application/pdf",
|
||||
"application/msword",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"application/vnd.ms-word.document.macroenabled.12",
|
||||
"application/vnd.ms-powerpoint",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
||||
"application/vnd.ms-excel",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
"application/vnd.oasis.opendocument.spreadsheet",
|
||||
"application/vnd.oasis.opendocument.presentation",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"text/csv",
|
||||
];
|
||||
DOC_TYPES.contains(&ct.as_str())
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user