From fe4b5ed18a833fe71a7d32598be8b013466f04b8 Mon Sep 17 00:00:00 2001 From: Bartal Laearsson Date: Tue, 25 Aug 2026 12:01:07 +0100 Subject: [PATCH] fix clippy warnigns --- .gitignore | 3 +- Cargo.lock | 7 --- Cargo.toml | 2 - src/converter.rs | 7 +-- src/doc_processor.rs | 15 +++--- src/error_logger.rs | 15 +++--- src/fetcher.rs | 27 ++++------ src/main.rs | 35 ++++++------- src/robots.rs | 11 +--- src/scraper.rs | 118 +++++++++++++++++++++---------------------- src/url_utils.rs | 21 +++----- 11 files changed, 106 insertions(+), 155 deletions(-) diff --git a/.gitignore b/.gitignore index 4e540c8..3343de3 100644 --- a/.gitignore +++ b/.gitignore @@ -1,2 +1,3 @@ /target -scraped* +*scraped/ +*scraped_logs/ diff --git a/Cargo.lock b/Cargo.lock index 7e2e645..f473db1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1220,12 +1220,6 @@ dependencies = [ "wasm-bindgen", ] -[[package]] -name = "lazy_static" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" - [[package]] name = "libc" version = "0.2.189" @@ -2617,7 +2611,6 @@ dependencies = [ "env_logger", "hex", "htmd", - "lazy_static", "log", "md-5", "reqwest", diff --git a/Cargo.toml b/Cargo.toml index a7d42b6..8228616 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,4 +1,3 @@ -# Cargo.toml [package] name = "web-scraper" version = "0.1.1" @@ -11,7 +10,6 @@ chrono = { version = "0.4", features = ["serde"] } clap = { version = "4", features = ["derive"] } hex = "0.4" htmd = "0.1" -lazy_static = "1.4" log = "0.4" env_logger = "0.11" md-5 = "0.10" diff --git a/src/converter.rs b/src/converter.rs index c276e12..af60490 100644 --- a/src/converter.rs +++ b/src/converter.rs @@ -21,10 +21,6 @@ fn get_body_selector() -> &'static Selector { BODY_SELECTOR.get_or_init(|| Selector::parse("body").expect("hardcoded selector is valid")) } -/// Extract the `` element's HTML from a full HTML document. -/// The HTML5 parser always synthesizes a `` element, so this -/// returns `Some` for any well-formed document. Returns `None` only -/// if the parser fails entirely. fn extract_body(html: &str) -> Option { let document = Html::parse_document(html); document @@ -40,7 +36,7 @@ pub fn html_to_markdown(html: &str) -> String { match converter.convert(&source) { Ok(md) => md, Err(e) => { - log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e); + log::warn!("HTML-to-MD conversion failed ({e}), saving raw HTML"); source } } @@ -60,7 +56,6 @@ mod tests { #[test] fn test_extract_body_without_body_tag_falls_back_to_synthetic() { - // HTML5 parser synthesizes a element even when absent in source. let html = "

Hello

"; let result = extract_body(html); assert!(result.is_some(), "parser should synthesize a body element"); diff --git a/src/doc_processor.rs b/src/doc_processor.rs index 00e06d4..ba67573 100644 --- a/src/doc_processor.rs +++ b/src/doc_processor.rs @@ -1,4 +1,3 @@ -// src/doc_processor.rs #![warn(clippy::all, clippy::pedantic)] use anyhow::Result; @@ -29,7 +28,7 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult { DocProcessResult::Raw } Err(e) => { - log::warn!("anydoc conversion failed ({}), saving raw", e); + log::warn!("anydoc conversion failed ({e}), saving raw"); DocProcessResult::Raw } } @@ -37,23 +36,23 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult { #[allow(dead_code)] pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, usize, usize)> { - let supported_exts = [ + const SUPPORTED_EXTS: &[&str] = &[ "pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls", "xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv", ]; let entries: Vec<_> = std::fs::read_dir(dir)? - .filter_map(|e| e.ok()) + .filter_map(std::result::Result::ok) .filter(|e| { e.path().extension().is_some_and(|ext| { let ext = ext.to_string_lossy().to_lowercase(); - supported_exts.contains(&ext.as_str()) + SUPPORTED_EXTS.contains(&ext.as_str()) }) }) .collect(); let total = entries.len(); - log::info!("Found {} documents to process", total); + log::info!("Found {total} documents to process"); let mut converted = 0usize; let mut raw = 0usize; @@ -67,7 +66,7 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, let bytes = match std::fs::read(&path) { Ok(b) => b, Err(e) => { - log::error!("error reading file: {}", e); + log::error!("error reading file: {e}"); errors += 1; continue; } @@ -102,7 +101,6 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, Ok((converted, raw, errors)) } -// src/doc_processor.rs - tests section added at end #[cfg(test)] mod tests { use super::*; @@ -113,7 +111,6 @@ mod tests { let html_bytes = b"

Test

"; let url = Url::parse("https://example.com/test.html").unwrap(); let result = try_convert(html_bytes, &url); - // Will likely return Raw since anydoc may not handle HTML match result { DocProcessResult::Markdown(_) | DocProcessResult::Raw => {} } diff --git a/src/error_logger.rs b/src/error_logger.rs index 3eac01a..67b584b 100644 --- a/src/error_logger.rs +++ b/src/error_logger.rs @@ -1,4 +1,3 @@ -// src/error_logger.rs #![warn(clippy::all, clippy::pedantic)] use chrono::{DateTime, Utc}; @@ -6,7 +5,7 @@ use serde::Serialize; use std::fs::OpenOptions; use std::io::{BufWriter, Write}; use std::path::Path; -use std::sync::Mutex; +use std::sync::{LazyLock, Mutex}; use url::Url; #[derive(Serialize)] @@ -19,9 +18,8 @@ pub struct ErrorEntry { pub status_code: Option, } -lazy_static::lazy_static! { - static ref LOG_WRITER: Mutex>> = Mutex::new(None); -} +static LOG_WRITER: LazyLock>>> = + LazyLock::new(|| Mutex::new(None)); pub fn log_error( log_path: &Path, @@ -41,18 +39,17 @@ pub fn log_error( let line = match serde_json::to_string(&entry) { Ok(json) => json + "\n", Err(e) => { - log::error!("Failed to serialize error entry: {}", e); + log::error!("Failed to serialize error entry: {e}"); return; } }; - // Buffered writer approach { let mut guard = LOG_WRITER.lock().unwrap(); match guard.as_mut() { Some(writer) => { if let Err(e) = writer.write_all(line.as_bytes()) { - log::error!("Failed to write error log: {}", e); + log::error!("Failed to write error log: {e}"); } } None => { @@ -68,5 +65,5 @@ pub fn log_error( } } - log::info!("LOGGED ERROR [{}]: {}", error_type, message); + log::info!("LOGGED ERROR [{error_type}]: {message}"); } diff --git a/src/fetcher.rs b/src/fetcher.rs index 01cbd3e..5dabce8 100644 --- a/src/fetcher.rs +++ b/src/fetcher.rs @@ -11,8 +11,8 @@ pub struct Fetcher { } impl Fetcher { - pub fn new(client: Client) -> Result { - Ok(Self { client }) + pub fn new(client: Client) -> Self { + Self { client } } pub async fn fetch_with_retry(&self, url: &Url) -> Result { @@ -33,21 +33,16 @@ impl Fetcher { Err(e) => { if attempt < MAX_RETRIES { log::warn!( - "retry {}/{}: body read failed ({})", - attempt, - MAX_RETRIES, - e + "retry {attempt}/{MAX_RETRIES}: body read failed ({e})" ); tokio::time::sleep(Duration::from_secs( - RETRY_BACKOFF_SECS * attempt as u64, + RETRY_BACKOFF_SECS * u64::from(attempt), )) .await; continue; } return Err(anyhow!( - "Body read failed after {} retries: {}", - MAX_RETRIES, - e + "Body read failed after {MAX_RETRIES} retries: {e}" )); } }; @@ -64,23 +59,19 @@ impl Fetcher { } Err(e) => { if attempt < MAX_RETRIES { - log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e); + log::warn!("retry {attempt}/{MAX_RETRIES}: request failed ({e})"); tokio::time::sleep(Duration::from_secs( - RETRY_BACKOFF_SECS * attempt as u64, + RETRY_BACKOFF_SECS * u64::from(attempt), )) .await; continue; } - return Err(anyhow!( - "Request failed after {} retries: {}", - MAX_RETRIES, - e - )); + return Err(anyhow!("Request failed after {MAX_RETRIES} retries: {e}")); } } } - Err(anyhow!("Exhausted all {} retries", MAX_RETRIES)) + Err(anyhow!("Exhausted all {MAX_RETRIES} retries")) } } diff --git a/src/main.rs b/src/main.rs index 4416f44..f6f76cd 100644 --- a/src/main.rs +++ b/src/main.rs @@ -80,26 +80,23 @@ async fn main() -> Result<()> { let url = Url::parse(&start_url)?; let base_domain = url_utils::derive_base_domain(&url) - .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?; + .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?; - let output_dir = match &args.output { - Some(p) => PathBuf::from(p), - None => { - let default_name = format!("{}_scraped", base_domain.replace('.', "_")); - PathBuf::from(default_name) - } + let output_dir = if let Some(p) = &args.output { + PathBuf::from(p) + } else { + let default_name = format!("{}_scraped", base_domain.replace('.', "_")); + PathBuf::from(default_name) }; let logs_dir = { let output_name = output_dir .file_name() - .map(|n| n.to_string_lossy().into_owned()) - .unwrap_or_else(|| "output".to_string()); - let logs_name = format!("{}_logs", output_name); + .map_or_else(|| "output".to_string(), |n| n.to_string_lossy().into_owned()); + let logs_name = format!("{output_name}_logs"); output_dir .parent() - .map(|p| p.join(&logs_name)) - .unwrap_or_else(|| PathBuf::from(&logs_name)) + .map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name)) }; fs::create_dir_all(&output_dir)?; @@ -126,8 +123,8 @@ async fn main() -> Result<()> { } }; - log::info!("Scraping: {}", start_url); - log::info!("Base domain: {}", base_domain); + log::info!("Scraping: {start_url}"); + log::info!("Base domain: {base_domain}"); log::info!("Output dir: {}", output_dir.display()); log::info!("Logs dir: {}", logs_dir.display()); log::info!( @@ -140,13 +137,13 @@ async fn main() -> Result<()> { ); match (&include_types, &exclude_types) { - (Some(t), _) => log::info!("Type filter: include {:?}", t), - (_, Some(t)) => log::info!("Type filter: exclude {:?}", t), + (Some(t), _) => log::info!("Type filter: include {t:?}"), + (_, Some(t)) => log::info!("Type filter: exclude {t:?}"), _ => log::info!("Type filter: none (all types)"), } match &scope_path { - Some(s) => log::info!("Path scope: {} (only this path and deeper)", s), + Some(s) => log::info!("Path scope: {s} (only this path and deeper)"), None => log::info!("Path scope: none (full site)"), } @@ -159,7 +156,7 @@ async fn main() -> Result<()> { let robots = fetch_robots(&client, &url, config::USER_AGENT).await; log::info!(""); - let fetcher = fetcher::Fetcher::new(client)?; + let fetcher = fetcher::Fetcher::new(client); let mut scraper = Scraper::new( output_dir, log_path.clone(), @@ -174,7 +171,7 @@ async fn main() -> Result<()> { ); let (count, errors) = scraper.run(&url).await; - log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors); + log::info!("\nDone. Fetched {count} URLs. Errors: {errors}."); log::info!("Error log: {}", log_path.display()); Ok(()) diff --git a/src/robots.rs b/src/robots.rs index d3ef40b..6f498dd 100644 --- a/src/robots.rs +++ b/src/robots.rs @@ -93,10 +93,7 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> ); } Err(e) => { - log::warn!( - "Failed to fetch robots.txt ({}): assuming no restrictions", - e - ); + log::warn!("Failed to fetch robots.txt ({e}): assuming no restrictions"); } } @@ -133,11 +130,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule { let agent = agent.trim().to_lowercase(); current_agents.push(agent.clone()); - if agent == "*" || agent == target_lower || agent == "web-scraper" { - collecting = true; - } else { - collecting = false; - } + collecting = agent == "*" || agent == target_lower || agent == "web-scraper"; } else if collecting { if let Some(path) = line.strip_prefix("Disallow:") { let path = path.trim().to_string(); diff --git a/src/scraper.rs b/src/scraper.rs index e48d796..93f48ae 100644 --- a/src/scraper.rs +++ b/src/scraper.rs @@ -13,22 +13,33 @@ use std::fs; use std::time::Duration; use url::Url; +const DOC_TYPES: &[&str] = &[ + "application/pdf", + "application/msword", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + "application/vnd.ms-word.document.macroenabled.12", + "application/vnd.ms-powerpoint", + "application/vnd.openxmlformats-officedocument.presentationml.presentation", + "application/vnd.ms-powerpoint.presentation.macroenabled.12", + "application/vnd.ms-excel", + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + "application/vnd.ms-excel.sheet.macroenabled.12", + "application/vnd.ms-excel.sheet.binary.macroenabled.12", + "application/vnd.oasis.opendocument.text", + "application/vnd.oasis.opendocument.spreadsheet", + "application/vnd.oasis.opendocument.presentation", + "application/rtf", + "application/epub+zip", + "text/csv", +]; + +#[derive(Default)] pub struct DocStats { pub converted: usize, pub raw: usize, pub errors: usize, } -impl Default for DocStats { - fn default() -> Self { - Self { - converted: 0, - raw: 0, - errors: 0, - } - } -} - pub struct Scraper { fetcher: Fetcher, seen: HashSet, @@ -45,6 +56,7 @@ pub struct Scraper { } impl Scraper { + #[allow(clippy::too_many_arguments)] pub fn new( output_dir: std::path::PathBuf, log_path: std::path::PathBuf, @@ -94,6 +106,20 @@ impl Scraper { } } + fn extract_new_links( + &self, + html: &str, + final_url: &Url, + ) -> (usize, Vec) { + let links = extract_links(html, final_url, &self.base_domain); + let new_count = links.len(); + let new_links: Vec = links + .into_iter() + .filter(|link| !self.seen.contains(link.as_str())) + .collect(); + (new_count, new_links) + } + pub async fn run(&mut self, start_url: &Url) -> (usize, usize) { let mut queue: VecDeque = VecDeque::new(); let start_url_owned = start_url.clone(); @@ -106,9 +132,8 @@ impl Scraper { let mut skipped_type = 0usize; while let Some(raw_url) = queue.pop_front() { - let url = match normalize_url(raw_url.as_str()) { - Some(u) => u, - None => continue, + let Some(url) = normalize_url(raw_url.as_str()) else { + continue; }; let url_key = url.as_str().to_string(); @@ -118,19 +143,19 @@ impl Scraper { } self.seen.insert(url_key.clone()); - if !is_in_scope(&url, &self.scope_path) { - log::debug!("[skip] out of scope: {}", url); + if !is_in_scope(&url, self.scope_path.as_ref()) { + log::debug!("[skip] out of scope: {url}"); skipped_scope += 1; continue; } if !self.robots.is_allowed(url.path()) { - log::info!("[skip] robots.txt disallows: {}", url); + log::info!("[skip] robots.txt disallows: {url}"); skipped_robots += 1; continue; } - log::info!("[{}] fetching: {}", count, url); + log::info!("[{count}] fetching: {url}"); match self.fetcher.fetch_with_retry(&url).await { Ok(result) => { @@ -141,12 +166,12 @@ impl Scraper { self.should_save(&result.final_url, &result.content_type); if !should_save { - log::info!(" skipped (type filter: {})", ext_label); + log::info!(" skipped (type filter: {ext_label})"); skipped_type += 1; } if should_save { - match self.save(&result, is_html).await { + match self.save(&result, is_html) { Ok(()) => { if !is_html { log::info!(" binary: {}", result.content_type); @@ -167,24 +192,14 @@ impl Scraper { } if is_html { - if !self.single_page { - if let Ok(html) = std::str::from_utf8(&result.bytes) { - let links = extract_links(html, &final_url, &self.base_domain); - let new_count = links.len(); - - let new_links: Vec = links - .into_iter() - .filter(|link| !self.seen.contains(link.as_str())) - .collect(); - - for link in &new_links { - queue.push_back(link.clone()); - } - - log::info!(" found {} links ({} new)", new_count, new_links.len()); - } - } else { + if self.single_page { log::info!(" [single-page mode] not crawling for links"); + } else if let Ok(html) = std::str::from_utf8(&result.bytes) { + let (new_count, new_links) = self.extract_new_links(html, &final_url); + for link in &new_links { + queue.push_back(link.clone()); + } + log::info!(" found {new_count} links ({} new)", new_links.len()); } } } @@ -206,19 +221,19 @@ impl Scraper { } if skipped_robots > 0 { - log::info!("Skipped {} URLs due to robots.txt", skipped_robots); + log::info!("Skipped {skipped_robots} URLs due to robots.txt"); } if skipped_scope > 0 { - log::info!("Skipped {} URLs due to path scope", skipped_scope); + log::info!("Skipped {skipped_scope} URLs due to path scope"); } if skipped_type > 0 { - log::info!("Skipped {} URLs due to type filter", skipped_type); + log::info!("Skipped {skipped_type} URLs due to type filter"); } (count, error_count) } - async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> { + fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> { let mut filename = url_to_filename(&result.final_url, &result.content_type); let content: Vec; @@ -226,7 +241,7 @@ impl Scraper { let html = std::str::from_utf8(&result.bytes)?; let md = html_to_markdown(html); if let Some(stripped) = filename.strip_suffix(".html") { - filename = format!("{}.md", stripped); + filename = format!("{stripped}.md"); } content = md.into_bytes(); } else if self.convert_docs { @@ -263,25 +278,6 @@ impl Scraper { fn is_document_content_type(ct: &str) -> bool { let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase(); - const DOC_TYPES: &[&str] = &[ - "application/pdf", - "application/msword", - "application/vnd.openxmlformats-officedocument.wordprocessingml.document", - "application/vnd.ms-word.document.macroenabled.12", - "application/vnd.ms-powerpoint", - "application/vnd.openxmlformats-officedocument.presentationml.presentation", - "application/vnd.ms-powerpoint.presentation.macroenabled.12", - "application/vnd.ms-excel", - "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", - "application/vnd.ms-excel.sheet.macroenabled.12", - "application/vnd.ms-excel.sheet.binary.macroenabled.12", - "application/vnd.oasis.opendocument.text", - "application/vnd.oasis.opendocument.spreadsheet", - "application/vnd.oasis.opendocument.presentation", - "application/rtf", - "application/epub+zip", - "text/csv", - ]; DOC_TYPES.contains(&ct.as_str()) } @@ -296,7 +292,7 @@ mod tests { exclude_types: Option>, ) -> Scraper { Scraper { - fetcher: Fetcher::new(Client::new()).unwrap(), + fetcher: Fetcher::new(Client::new()), seen: HashSet::new(), output_dir: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(), diff --git a/src/url_utils.rs b/src/url_utils.rs index 111b0de..1315f37 100644 --- a/src/url_utils.rs +++ b/src/url_utils.rs @@ -1,4 +1,3 @@ -// src/url_utils.rs #![warn(clippy::all, clippy::pedantic)] use md5::Md5; @@ -35,11 +34,7 @@ pub fn normalize_url(url: &str) -> Option { Some(parsed) } -/// Returns true if `url`'s path falls within the given scope. -/// A scope of `/blog/2021` matches `/blog/2021` and `/blog/2021/...` -/// but not `/blog/20212`. -/// `None` scope means no restriction. -pub fn is_in_scope(url: &Url, scope_path: &Option) -> bool { +pub fn is_in_scope(url: &Url, scope_path: Option<&String>) -> bool { match scope_path { None => true, Some(scope) => { @@ -48,7 +43,7 @@ pub fn is_in_scope(url: &Url, scope_path: &Option) -> bool { return true; } let path = url.path(); - path == scope || path.starts_with(&format!("{}/", scope)) + path == scope || path.starts_with(&format!("{scope}/")) } } } @@ -57,10 +52,10 @@ pub fn get_extension(url: &Url, content_type: &str) -> String { let path = url.path(); let last_segment = path.rsplit('/').next().unwrap_or(path); - if let Some(dot_pos) = last_segment.rfind('.') { - if dot_pos > 0 { - return last_segment[dot_pos..].to_lowercase(); - } + if let Some(dot_pos) = last_segment.rfind('.') + && dot_pos > 0 + { + return last_segment[dot_pos..].to_lowercase(); } let ct = content_type @@ -100,7 +95,6 @@ pub fn get_extension(url: &Url, content_type: &str) -> String { "application/vnd.oasis.opendocument.spreadsheet" => ".ods", "application/vnd.oasis.opendocument.presentation" => ".odp", "application/zip" => ".zip", - "application/octet-stream" => ".bin", _ => ".bin", } .to_string() @@ -120,8 +114,7 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String { let host = url .host_str() .unwrap_or("unknown") - .replace('.', "_") - .replace(':', "_"); + .replace(['.', ':'], "_"); let path = url.path().trim_start_matches('/'); let path = if path.is_empty() { "index" } else { path }; let path = path.trim_end_matches('/');