diff --git a/.gitignore b/.gitignore
index 4e540c8..3343de3 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,2 +1,3 @@
/target
-scraped*
+*scraped/
+*scraped_logs/
diff --git a/Cargo.lock b/Cargo.lock
index 7e2e645..f473db1 100644
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -1220,12 +1220,6 @@ dependencies = [
"wasm-bindgen",
]
-[[package]]
-name = "lazy_static"
-version = "1.5.0"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
-
[[package]]
name = "libc"
version = "0.2.189"
@@ -2617,7 +2611,6 @@ dependencies = [
"env_logger",
"hex",
"htmd",
- "lazy_static",
"log",
"md-5",
"reqwest",
diff --git a/Cargo.toml b/Cargo.toml
index a7d42b6..8228616 100644
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -1,4 +1,3 @@
-# Cargo.toml
[package]
name = "web-scraper"
version = "0.1.1"
@@ -11,7 +10,6 @@ chrono = { version = "0.4", features = ["serde"] }
clap = { version = "4", features = ["derive"] }
hex = "0.4"
htmd = "0.1"
-lazy_static = "1.4"
log = "0.4"
env_logger = "0.11"
md-5 = "0.10"
diff --git a/src/converter.rs b/src/converter.rs
index c276e12..af60490 100644
--- a/src/converter.rs
+++ b/src/converter.rs
@@ -21,10 +21,6 @@ fn get_body_selector() -> &'static Selector {
BODY_SELECTOR.get_or_init(|| Selector::parse("body").expect("hardcoded selector is valid"))
}
-/// Extract the `
` element's HTML from a full HTML document.
-/// The HTML5 parser always synthesizes a `` element, so this
-/// returns `Some` for any well-formed document. Returns `None` only
-/// if the parser fails entirely.
fn extract_body(html: &str) -> Option {
let document = Html::parse_document(html);
document
@@ -40,7 +36,7 @@ pub fn html_to_markdown(html: &str) -> String {
match converter.convert(&source) {
Ok(md) => md,
Err(e) => {
- log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e);
+ log::warn!("HTML-to-MD conversion failed ({e}), saving raw HTML");
source
}
}
@@ -60,7 +56,6 @@ mod tests {
#[test]
fn test_extract_body_without_body_tag_falls_back_to_synthetic() {
- // HTML5 parser synthesizes a element even when absent in source.
let html = "Hello
";
let result = extract_body(html);
assert!(result.is_some(), "parser should synthesize a body element");
diff --git a/src/doc_processor.rs b/src/doc_processor.rs
index 00e06d4..ba67573 100644
--- a/src/doc_processor.rs
+++ b/src/doc_processor.rs
@@ -1,4 +1,3 @@
-// src/doc_processor.rs
#![warn(clippy::all, clippy::pedantic)]
use anyhow::Result;
@@ -29,7 +28,7 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
DocProcessResult::Raw
}
Err(e) => {
- log::warn!("anydoc conversion failed ({}), saving raw", e);
+ log::warn!("anydoc conversion failed ({e}), saving raw");
DocProcessResult::Raw
}
}
@@ -37,23 +36,23 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
#[allow(dead_code)]
pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, usize, usize)> {
- let supported_exts = [
+ const SUPPORTED_EXTS: &[&str] = &[
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
];
let entries: Vec<_> = std::fs::read_dir(dir)?
- .filter_map(|e| e.ok())
+ .filter_map(std::result::Result::ok)
.filter(|e| {
e.path().extension().is_some_and(|ext| {
let ext = ext.to_string_lossy().to_lowercase();
- supported_exts.contains(&ext.as_str())
+ SUPPORTED_EXTS.contains(&ext.as_str())
})
})
.collect();
let total = entries.len();
- log::info!("Found {} documents to process", total);
+ log::info!("Found {total} documents to process");
let mut converted = 0usize;
let mut raw = 0usize;
@@ -67,7 +66,7 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
let bytes = match std::fs::read(&path) {
Ok(b) => b,
Err(e) => {
- log::error!("error reading file: {}", e);
+ log::error!("error reading file: {e}");
errors += 1;
continue;
}
@@ -102,7 +101,6 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
Ok((converted, raw, errors))
}
-// src/doc_processor.rs - tests section added at end
#[cfg(test)]
mod tests {
use super::*;
@@ -113,7 +111,6 @@ mod tests {
let html_bytes = b"Test
";
let url = Url::parse("https://example.com/test.html").unwrap();
let result = try_convert(html_bytes, &url);
- // Will likely return Raw since anydoc may not handle HTML
match result {
DocProcessResult::Markdown(_) | DocProcessResult::Raw => {}
}
diff --git a/src/error_logger.rs b/src/error_logger.rs
index 3eac01a..67b584b 100644
--- a/src/error_logger.rs
+++ b/src/error_logger.rs
@@ -1,4 +1,3 @@
-// src/error_logger.rs
#![warn(clippy::all, clippy::pedantic)]
use chrono::{DateTime, Utc};
@@ -6,7 +5,7 @@ use serde::Serialize;
use std::fs::OpenOptions;
use std::io::{BufWriter, Write};
use std::path::Path;
-use std::sync::Mutex;
+use std::sync::{LazyLock, Mutex};
use url::Url;
#[derive(Serialize)]
@@ -19,9 +18,8 @@ pub struct ErrorEntry {
pub status_code: Option,
}
-lazy_static::lazy_static! {
- static ref LOG_WRITER: Mutex>> = Mutex::new(None);
-}
+static LOG_WRITER: LazyLock>>> =
+ LazyLock::new(|| Mutex::new(None));
pub fn log_error(
log_path: &Path,
@@ -41,18 +39,17 @@ pub fn log_error(
let line = match serde_json::to_string(&entry) {
Ok(json) => json + "\n",
Err(e) => {
- log::error!("Failed to serialize error entry: {}", e);
+ log::error!("Failed to serialize error entry: {e}");
return;
}
};
- // Buffered writer approach
{
let mut guard = LOG_WRITER.lock().unwrap();
match guard.as_mut() {
Some(writer) => {
if let Err(e) = writer.write_all(line.as_bytes()) {
- log::error!("Failed to write error log: {}", e);
+ log::error!("Failed to write error log: {e}");
}
}
None => {
@@ -68,5 +65,5 @@ pub fn log_error(
}
}
- log::info!("LOGGED ERROR [{}]: {}", error_type, message);
+ log::info!("LOGGED ERROR [{error_type}]: {message}");
}
diff --git a/src/fetcher.rs b/src/fetcher.rs
index 01cbd3e..5dabce8 100644
--- a/src/fetcher.rs
+++ b/src/fetcher.rs
@@ -11,8 +11,8 @@ pub struct Fetcher {
}
impl Fetcher {
- pub fn new(client: Client) -> Result {
- Ok(Self { client })
+ pub fn new(client: Client) -> Self {
+ Self { client }
}
pub async fn fetch_with_retry(&self, url: &Url) -> Result {
@@ -33,21 +33,16 @@ impl Fetcher {
Err(e) => {
if attempt < MAX_RETRIES {
log::warn!(
- "retry {}/{}: body read failed ({})",
- attempt,
- MAX_RETRIES,
- e
+ "retry {attempt}/{MAX_RETRIES}: body read failed ({e})"
);
tokio::time::sleep(Duration::from_secs(
- RETRY_BACKOFF_SECS * attempt as u64,
+ RETRY_BACKOFF_SECS * u64::from(attempt),
))
.await;
continue;
}
return Err(anyhow!(
- "Body read failed after {} retries: {}",
- MAX_RETRIES,
- e
+ "Body read failed after {MAX_RETRIES} retries: {e}"
));
}
};
@@ -64,23 +59,19 @@ impl Fetcher {
}
Err(e) => {
if attempt < MAX_RETRIES {
- log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e);
+ log::warn!("retry {attempt}/{MAX_RETRIES}: request failed ({e})");
tokio::time::sleep(Duration::from_secs(
- RETRY_BACKOFF_SECS * attempt as u64,
+ RETRY_BACKOFF_SECS * u64::from(attempt),
))
.await;
continue;
}
- return Err(anyhow!(
- "Request failed after {} retries: {}",
- MAX_RETRIES,
- e
- ));
+ return Err(anyhow!("Request failed after {MAX_RETRIES} retries: {e}"));
}
}
}
- Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
+ Err(anyhow!("Exhausted all {MAX_RETRIES} retries"))
}
}
diff --git a/src/main.rs b/src/main.rs
index 4416f44..f6f76cd 100644
--- a/src/main.rs
+++ b/src/main.rs
@@ -80,26 +80,23 @@ async fn main() -> Result<()> {
let url = Url::parse(&start_url)?;
let base_domain = url_utils::derive_base_domain(&url)
- .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
+ .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
- let output_dir = match &args.output {
- Some(p) => PathBuf::from(p),
- None => {
- let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
- PathBuf::from(default_name)
- }
+ let output_dir = if let Some(p) = &args.output {
+ PathBuf::from(p)
+ } else {
+ let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
+ PathBuf::from(default_name)
};
let logs_dir = {
let output_name = output_dir
.file_name()
- .map(|n| n.to_string_lossy().into_owned())
- .unwrap_or_else(|| "output".to_string());
- let logs_name = format!("{}_logs", output_name);
+ .map_or_else(|| "output".to_string(), |n| n.to_string_lossy().into_owned());
+ let logs_name = format!("{output_name}_logs");
output_dir
.parent()
- .map(|p| p.join(&logs_name))
- .unwrap_or_else(|| PathBuf::from(&logs_name))
+ .map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
};
fs::create_dir_all(&output_dir)?;
@@ -126,8 +123,8 @@ async fn main() -> Result<()> {
}
};
- log::info!("Scraping: {}", start_url);
- log::info!("Base domain: {}", base_domain);
+ log::info!("Scraping: {start_url}");
+ log::info!("Base domain: {base_domain}");
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!(
@@ -140,13 +137,13 @@ async fn main() -> Result<()> {
);
match (&include_types, &exclude_types) {
- (Some(t), _) => log::info!("Type filter: include {:?}", t),
- (_, Some(t)) => log::info!("Type filter: exclude {:?}", t),
+ (Some(t), _) => log::info!("Type filter: include {t:?}"),
+ (_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
match &scope_path {
- Some(s) => log::info!("Path scope: {} (only this path and deeper)", s),
+ Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
@@ -159,7 +156,7 @@ async fn main() -> Result<()> {
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
log::info!("");
- let fetcher = fetcher::Fetcher::new(client)?;
+ let fetcher = fetcher::Fetcher::new(client);
let mut scraper = Scraper::new(
output_dir,
log_path.clone(),
@@ -174,7 +171,7 @@ async fn main() -> Result<()> {
);
let (count, errors) = scraper.run(&url).await;
- log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
+ log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
log::info!("Error log: {}", log_path.display());
Ok(())
diff --git a/src/robots.rs b/src/robots.rs
index d3ef40b..6f498dd 100644
--- a/src/robots.rs
+++ b/src/robots.rs
@@ -93,10 +93,7 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
);
}
Err(e) => {
- log::warn!(
- "Failed to fetch robots.txt ({}): assuming no restrictions",
- e
- );
+ log::warn!("Failed to fetch robots.txt ({e}): assuming no restrictions");
}
}
@@ -133,11 +130,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
let agent = agent.trim().to_lowercase();
current_agents.push(agent.clone());
- if agent == "*" || agent == target_lower || agent == "web-scraper" {
- collecting = true;
- } else {
- collecting = false;
- }
+ collecting = agent == "*" || agent == target_lower || agent == "web-scraper";
} else if collecting {
if let Some(path) = line.strip_prefix("Disallow:") {
let path = path.trim().to_string();
diff --git a/src/scraper.rs b/src/scraper.rs
index e48d796..93f48ae 100644
--- a/src/scraper.rs
+++ b/src/scraper.rs
@@ -13,22 +13,33 @@ use std::fs;
use std::time::Duration;
use url::Url;
+const DOC_TYPES: &[&str] = &[
+ "application/pdf",
+ "application/msword",
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
+ "application/vnd.ms-word.document.macroenabled.12",
+ "application/vnd.ms-powerpoint",
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation",
+ "application/vnd.ms-powerpoint.presentation.macroenabled.12",
+ "application/vnd.ms-excel",
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
+ "application/vnd.ms-excel.sheet.macroenabled.12",
+ "application/vnd.ms-excel.sheet.binary.macroenabled.12",
+ "application/vnd.oasis.opendocument.text",
+ "application/vnd.oasis.opendocument.spreadsheet",
+ "application/vnd.oasis.opendocument.presentation",
+ "application/rtf",
+ "application/epub+zip",
+ "text/csv",
+];
+
+#[derive(Default)]
pub struct DocStats {
pub converted: usize,
pub raw: usize,
pub errors: usize,
}
-impl Default for DocStats {
- fn default() -> Self {
- Self {
- converted: 0,
- raw: 0,
- errors: 0,
- }
- }
-}
-
pub struct Scraper {
fetcher: Fetcher,
seen: HashSet,
@@ -45,6 +56,7 @@ pub struct Scraper {
}
impl Scraper {
+ #[allow(clippy::too_many_arguments)]
pub fn new(
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
@@ -94,6 +106,20 @@ impl Scraper {
}
}
+ fn extract_new_links(
+ &self,
+ html: &str,
+ final_url: &Url,
+ ) -> (usize, Vec) {
+ let links = extract_links(html, final_url, &self.base_domain);
+ let new_count = links.len();
+ let new_links: Vec = links
+ .into_iter()
+ .filter(|link| !self.seen.contains(link.as_str()))
+ .collect();
+ (new_count, new_links)
+ }
+
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque = VecDeque::new();
let start_url_owned = start_url.clone();
@@ -106,9 +132,8 @@ impl Scraper {
let mut skipped_type = 0usize;
while let Some(raw_url) = queue.pop_front() {
- let url = match normalize_url(raw_url.as_str()) {
- Some(u) => u,
- None => continue,
+ let Some(url) = normalize_url(raw_url.as_str()) else {
+ continue;
};
let url_key = url.as_str().to_string();
@@ -118,19 +143,19 @@ impl Scraper {
}
self.seen.insert(url_key.clone());
- if !is_in_scope(&url, &self.scope_path) {
- log::debug!("[skip] out of scope: {}", url);
+ if !is_in_scope(&url, self.scope_path.as_ref()) {
+ log::debug!("[skip] out of scope: {url}");
skipped_scope += 1;
continue;
}
if !self.robots.is_allowed(url.path()) {
- log::info!("[skip] robots.txt disallows: {}", url);
+ log::info!("[skip] robots.txt disallows: {url}");
skipped_robots += 1;
continue;
}
- log::info!("[{}] fetching: {}", count, url);
+ log::info!("[{count}] fetching: {url}");
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
@@ -141,12 +166,12 @@ impl Scraper {
self.should_save(&result.final_url, &result.content_type);
if !should_save {
- log::info!(" skipped (type filter: {})", ext_label);
+ log::info!(" skipped (type filter: {ext_label})");
skipped_type += 1;
}
if should_save {
- match self.save(&result, is_html).await {
+ match self.save(&result, is_html) {
Ok(()) => {
if !is_html {
log::info!(" binary: {}", result.content_type);
@@ -167,24 +192,14 @@ impl Scraper {
}
if is_html {
- if !self.single_page {
- if let Ok(html) = std::str::from_utf8(&result.bytes) {
- let links = extract_links(html, &final_url, &self.base_domain);
- let new_count = links.len();
-
- let new_links: Vec = links
- .into_iter()
- .filter(|link| !self.seen.contains(link.as_str()))
- .collect();
-
- for link in &new_links {
- queue.push_back(link.clone());
- }
-
- log::info!(" found {} links ({} new)", new_count, new_links.len());
- }
- } else {
+ if self.single_page {
log::info!(" [single-page mode] not crawling for links");
+ } else if let Ok(html) = std::str::from_utf8(&result.bytes) {
+ let (new_count, new_links) = self.extract_new_links(html, &final_url);
+ for link in &new_links {
+ queue.push_back(link.clone());
+ }
+ log::info!(" found {new_count} links ({} new)", new_links.len());
}
}
}
@@ -206,19 +221,19 @@ impl Scraper {
}
if skipped_robots > 0 {
- log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
+ log::info!("Skipped {skipped_robots} URLs due to robots.txt");
}
if skipped_scope > 0 {
- log::info!("Skipped {} URLs due to path scope", skipped_scope);
+ log::info!("Skipped {skipped_scope} URLs due to path scope");
}
if skipped_type > 0 {
- log::info!("Skipped {} URLs due to type filter", skipped_type);
+ log::info!("Skipped {skipped_type} URLs due to type filter");
}
(count, error_count)
}
- async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
+ fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec;
@@ -226,7 +241,7 @@ impl Scraper {
let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html);
if let Some(stripped) = filename.strip_suffix(".html") {
- filename = format!("{}.md", stripped);
+ filename = format!("{stripped}.md");
}
content = md.into_bytes();
} else if self.convert_docs {
@@ -263,25 +278,6 @@ impl Scraper {
fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
- const DOC_TYPES: &[&str] = &[
- "application/pdf",
- "application/msword",
- "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
- "application/vnd.ms-word.document.macroenabled.12",
- "application/vnd.ms-powerpoint",
- "application/vnd.openxmlformats-officedocument.presentationml.presentation",
- "application/vnd.ms-powerpoint.presentation.macroenabled.12",
- "application/vnd.ms-excel",
- "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
- "application/vnd.ms-excel.sheet.macroenabled.12",
- "application/vnd.ms-excel.sheet.binary.macroenabled.12",
- "application/vnd.oasis.opendocument.text",
- "application/vnd.oasis.opendocument.spreadsheet",
- "application/vnd.oasis.opendocument.presentation",
- "application/rtf",
- "application/epub+zip",
- "text/csv",
- ];
DOC_TYPES.contains(&ct.as_str())
}
@@ -296,7 +292,7 @@ mod tests {
exclude_types: Option>,
) -> Scraper {
Scraper {
- fetcher: Fetcher::new(Client::new()).unwrap(),
+ fetcher: Fetcher::new(Client::new()),
seen: HashSet::new(),
output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(),
diff --git a/src/url_utils.rs b/src/url_utils.rs
index 111b0de..1315f37 100644
--- a/src/url_utils.rs
+++ b/src/url_utils.rs
@@ -1,4 +1,3 @@
-// src/url_utils.rs
#![warn(clippy::all, clippy::pedantic)]
use md5::Md5;
@@ -35,11 +34,7 @@ pub fn normalize_url(url: &str) -> Option {
Some(parsed)
}
-/// Returns true if `url`'s path falls within the given scope.
-/// A scope of `/blog/2021` matches `/blog/2021` and `/blog/2021/...`
-/// but not `/blog/20212`.
-/// `None` scope means no restriction.
-pub fn is_in_scope(url: &Url, scope_path: &Option) -> bool {
+pub fn is_in_scope(url: &Url, scope_path: Option<&String>) -> bool {
match scope_path {
None => true,
Some(scope) => {
@@ -48,7 +43,7 @@ pub fn is_in_scope(url: &Url, scope_path: &Option) -> bool {
return true;
}
let path = url.path();
- path == scope || path.starts_with(&format!("{}/", scope))
+ path == scope || path.starts_with(&format!("{scope}/"))
}
}
}
@@ -57,10 +52,10 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
let path = url.path();
let last_segment = path.rsplit('/').next().unwrap_or(path);
- if let Some(dot_pos) = last_segment.rfind('.') {
- if dot_pos > 0 {
- return last_segment[dot_pos..].to_lowercase();
- }
+ if let Some(dot_pos) = last_segment.rfind('.')
+ && dot_pos > 0
+ {
+ return last_segment[dot_pos..].to_lowercase();
}
let ct = content_type
@@ -100,7 +95,6 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
"application/vnd.oasis.opendocument.spreadsheet" => ".ods",
"application/vnd.oasis.opendocument.presentation" => ".odp",
"application/zip" => ".zip",
- "application/octet-stream" => ".bin",
_ => ".bin",
}
.to_string()
@@ -120,8 +114,7 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
let host = url
.host_str()
.unwrap_or("unknown")
- .replace('.', "_")
- .replace(':', "_");
+ .replace(['.', ':'], "_");
let path = url.path().trim_start_matches('/');
let path = if path.is_empty() { "index" } else { path };
let path = path.trim_end_matches('/');