fix clippy warnigns

This commit is contained in:
2026-08-25 12:01:07 +01:00
parent da8f7302c1
commit fe4b5ed18a
11 changed files with 106 additions and 155 deletions
+2 -1
View File
@@ -1,2 +1,3 @@
/target
scraped*
*scraped/
*scraped_logs/
Generated
-7
View File
@@ -1220,12 +1220,6 @@ dependencies = [
"wasm-bindgen",
]
[[package]]
name = "lazy_static"
version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
[[package]]
name = "libc"
version = "0.2.189"
@@ -2617,7 +2611,6 @@ dependencies = [
"env_logger",
"hex",
"htmd",
"lazy_static",
"log",
"md-5",
"reqwest",
-2
View File
@@ -1,4 +1,3 @@
# Cargo.toml
[package]
name = "web-scraper"
version = "0.1.1"
@@ -11,7 +10,6 @@ chrono = { version = "0.4", features = ["serde"] }
clap = { version = "4", features = ["derive"] }
hex = "0.4"
htmd = "0.1"
lazy_static = "1.4"
log = "0.4"
env_logger = "0.11"
md-5 = "0.10"
+1 -6
View File
@@ -21,10 +21,6 @@ fn get_body_selector() -> &'static Selector {
BODY_SELECTOR.get_or_init(|| Selector::parse("body").expect("hardcoded selector is valid"))
}
/// Extract the `<body>` element's HTML from a full HTML document.
/// The HTML5 parser always synthesizes a `<body>` element, so this
/// returns `Some` for any well-formed document. Returns `None` only
/// if the parser fails entirely.
fn extract_body(html: &str) -> Option<String> {
let document = Html::parse_document(html);
document
@@ -40,7 +36,7 @@ pub fn html_to_markdown(html: &str) -> String {
match converter.convert(&source) {
Ok(md) => md,
Err(e) => {
log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e);
log::warn!("HTML-to-MD conversion failed ({e}), saving raw HTML");
source
}
}
@@ -60,7 +56,6 @@ mod tests {
#[test]
fn test_extract_body_without_body_tag_falls_back_to_synthetic() {
// HTML5 parser synthesizes a <body> element even when absent in source.
let html = "<html><head></head><p>Hello</p></html>";
let result = extract_body(html);
assert!(result.is_some(), "parser should synthesize a body element");
+6 -9
View File
@@ -1,4 +1,3 @@
// src/doc_processor.rs
#![warn(clippy::all, clippy::pedantic)]
use anyhow::Result;
@@ -29,7 +28,7 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
DocProcessResult::Raw
}
Err(e) => {
log::warn!("anydoc conversion failed ({}), saving raw", e);
log::warn!("anydoc conversion failed ({e}), saving raw");
DocProcessResult::Raw
}
}
@@ -37,23 +36,23 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
#[allow(dead_code)]
pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, usize, usize)> {
let supported_exts = [
const SUPPORTED_EXTS: &[&str] = &[
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
];
let entries: Vec<_> = std::fs::read_dir(dir)?
.filter_map(|e| e.ok())
.filter_map(std::result::Result::ok)
.filter(|e| {
e.path().extension().is_some_and(|ext| {
let ext = ext.to_string_lossy().to_lowercase();
supported_exts.contains(&ext.as_str())
SUPPORTED_EXTS.contains(&ext.as_str())
})
})
.collect();
let total = entries.len();
log::info!("Found {} documents to process", total);
log::info!("Found {total} documents to process");
let mut converted = 0usize;
let mut raw = 0usize;
@@ -67,7 +66,7 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
let bytes = match std::fs::read(&path) {
Ok(b) => b,
Err(e) => {
log::error!("error reading file: {}", e);
log::error!("error reading file: {e}");
errors += 1;
continue;
}
@@ -102,7 +101,6 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
Ok((converted, raw, errors))
}
// src/doc_processor.rs - tests section added at end
#[cfg(test)]
mod tests {
use super::*;
@@ -113,7 +111,6 @@ mod tests {
let html_bytes = b"<html><body><p>Test</p></body></html>";
let url = Url::parse("https://example.com/test.html").unwrap();
let result = try_convert(html_bytes, &url);
// Will likely return Raw since anydoc may not handle HTML
match result {
DocProcessResult::Markdown(_) | DocProcessResult::Raw => {}
}
+6 -9
View File
@@ -1,4 +1,3 @@
// src/error_logger.rs
#![warn(clippy::all, clippy::pedantic)]
use chrono::{DateTime, Utc};
@@ -6,7 +5,7 @@ use serde::Serialize;
use std::fs::OpenOptions;
use std::io::{BufWriter, Write};
use std::path::Path;
use std::sync::Mutex;
use std::sync::{LazyLock, Mutex};
use url::Url;
#[derive(Serialize)]
@@ -19,9 +18,8 @@ pub struct ErrorEntry {
pub status_code: Option<u16>,
}
lazy_static::lazy_static! {
static ref LOG_WRITER: Mutex<Option<BufWriter<std::fs::File>>> = Mutex::new(None);
}
static LOG_WRITER: LazyLock<Mutex<Option<BufWriter<std::fs::File>>>> =
LazyLock::new(|| Mutex::new(None));
pub fn log_error(
log_path: &Path,
@@ -41,18 +39,17 @@ pub fn log_error(
let line = match serde_json::to_string(&entry) {
Ok(json) => json + "\n",
Err(e) => {
log::error!("Failed to serialize error entry: {}", e);
log::error!("Failed to serialize error entry: {e}");
return;
}
};
// Buffered writer approach
{
let mut guard = LOG_WRITER.lock().unwrap();
match guard.as_mut() {
Some(writer) => {
if let Err(e) = writer.write_all(line.as_bytes()) {
log::error!("Failed to write error log: {}", e);
log::error!("Failed to write error log: {e}");
}
}
None => {
@@ -68,5 +65,5 @@ pub fn log_error(
}
}
log::info!("LOGGED ERROR [{}]: {}", error_type, message);
log::info!("LOGGED ERROR [{error_type}]: {message}");
}
+9 -18
View File
@@ -11,8 +11,8 @@ pub struct Fetcher {
}
impl Fetcher {
pub fn new(client: Client) -> Result<Self> {
Ok(Self { client })
pub fn new(client: Client) -> Self {
Self { client }
}
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
@@ -33,21 +33,16 @@ impl Fetcher {
Err(e) => {
if attempt < MAX_RETRIES {
log::warn!(
"retry {}/{}: body read failed ({})",
attempt,
MAX_RETRIES,
e
"retry {attempt}/{MAX_RETRIES}: body read failed ({e})"
);
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
RETRY_BACKOFF_SECS * u64::from(attempt),
))
.await;
continue;
}
return Err(anyhow!(
"Body read failed after {} retries: {}",
MAX_RETRIES,
e
"Body read failed after {MAX_RETRIES} retries: {e}"
));
}
};
@@ -64,23 +59,19 @@ impl Fetcher {
}
Err(e) => {
if attempt < MAX_RETRIES {
log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e);
log::warn!("retry {attempt}/{MAX_RETRIES}: request failed ({e})");
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
RETRY_BACKOFF_SECS * u64::from(attempt),
))
.await;
continue;
}
return Err(anyhow!(
"Request failed after {} retries: {}",
MAX_RETRIES,
e
));
return Err(anyhow!("Request failed after {MAX_RETRIES} retries: {e}"));
}
}
}
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
Err(anyhow!("Exhausted all {MAX_RETRIES} retries"))
}
}
+14 -17
View File
@@ -80,26 +80,23 @@ async fn main() -> Result<()> {
let url = Url::parse(&start_url)?;
let base_domain = url_utils::derive_base_domain(&url)
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
let output_dir = match &args.output {
Some(p) => PathBuf::from(p),
None => {
let output_dir = if let Some(p) = &args.output {
PathBuf::from(p)
} else {
let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
PathBuf::from(default_name)
}
};
let logs_dir = {
let output_name = output_dir
.file_name()
.map(|n| n.to_string_lossy().into_owned())
.unwrap_or_else(|| "output".to_string());
let logs_name = format!("{}_logs", output_name);
.map_or_else(|| "output".to_string(), |n| n.to_string_lossy().into_owned());
let logs_name = format!("{output_name}_logs");
output_dir
.parent()
.map(|p| p.join(&logs_name))
.unwrap_or_else(|| PathBuf::from(&logs_name))
.map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
};
fs::create_dir_all(&output_dir)?;
@@ -126,8 +123,8 @@ async fn main() -> Result<()> {
}
};
log::info!("Scraping: {}", start_url);
log::info!("Base domain: {}", base_domain);
log::info!("Scraping: {start_url}");
log::info!("Base domain: {base_domain}");
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!(
@@ -140,13 +137,13 @@ async fn main() -> Result<()> {
);
match (&include_types, &exclude_types) {
(Some(t), _) => log::info!("Type filter: include {:?}", t),
(_, Some(t)) => log::info!("Type filter: exclude {:?}", t),
(Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
match &scope_path {
Some(s) => log::info!("Path scope: {} (only this path and deeper)", s),
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
@@ -159,7 +156,7 @@ async fn main() -> Result<()> {
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
log::info!("");
let fetcher = fetcher::Fetcher::new(client)?;
let fetcher = fetcher::Fetcher::new(client);
let mut scraper = Scraper::new(
output_dir,
log_path.clone(),
@@ -174,7 +171,7 @@ async fn main() -> Result<()> {
);
let (count, errors) = scraper.run(&url).await;
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
log::info!("Error log: {}", log_path.display());
Ok(())
+2 -9
View File
@@ -93,10 +93,7 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
);
}
Err(e) => {
log::warn!(
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
log::warn!("Failed to fetch robots.txt ({e}): assuming no restrictions");
}
}
@@ -133,11 +130,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
let agent = agent.trim().to_lowercase();
current_agents.push(agent.clone());
if agent == "*" || agent == target_lower || agent == "web-scraper" {
collecting = true;
} else {
collecting = false;
}
collecting = agent == "*" || agent == target_lower || agent == "web-scraper";
} else if collecting {
if let Some(path) = line.strip_prefix("Disallow:") {
let path = path.trim().to_string();
+55 -59
View File
@@ -13,22 +13,33 @@ use std::fs;
use std::time::Duration;
use url::Url;
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
#[derive(Default)]
pub struct DocStats {
pub converted: usize,
pub raw: usize,
pub errors: usize,
}
impl Default for DocStats {
fn default() -> Self {
Self {
converted: 0,
raw: 0,
errors: 0,
}
}
}
pub struct Scraper {
fetcher: Fetcher,
seen: HashSet<String>,
@@ -45,6 +56,7 @@ pub struct Scraper {
}
impl Scraper {
#[allow(clippy::too_many_arguments)]
pub fn new(
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
@@ -94,6 +106,20 @@ impl Scraper {
}
}
fn extract_new_links(
&self,
html: &str,
final_url: &Url,
) -> (usize, Vec<Url>) {
let links = extract_links(html, final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
(new_count, new_links)
}
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque<Url> = VecDeque::new();
let start_url_owned = start_url.clone();
@@ -106,9 +132,8 @@ impl Scraper {
let mut skipped_type = 0usize;
while let Some(raw_url) = queue.pop_front() {
let url = match normalize_url(raw_url.as_str()) {
Some(u) => u,
None => continue,
let Some(url) = normalize_url(raw_url.as_str()) else {
continue;
};
let url_key = url.as_str().to_string();
@@ -118,19 +143,19 @@ impl Scraper {
}
self.seen.insert(url_key.clone());
if !is_in_scope(&url, &self.scope_path) {
log::debug!("[skip] out of scope: {}", url);
if !is_in_scope(&url, self.scope_path.as_ref()) {
log::debug!("[skip] out of scope: {url}");
skipped_scope += 1;
continue;
}
if !self.robots.is_allowed(url.path()) {
log::info!("[skip] robots.txt disallows: {}", url);
log::info!("[skip] robots.txt disallows: {url}");
skipped_robots += 1;
continue;
}
log::info!("[{}] fetching: {}", count, url);
log::info!("[{count}] fetching: {url}");
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
@@ -141,12 +166,12 @@ impl Scraper {
self.should_save(&result.final_url, &result.content_type);
if !should_save {
log::info!(" skipped (type filter: {})", ext_label);
log::info!(" skipped (type filter: {ext_label})");
skipped_type += 1;
}
if should_save {
match self.save(&result, is_html).await {
match self.save(&result, is_html) {
Ok(()) => {
if !is_html {
log::info!(" binary: {}", result.content_type);
@@ -167,24 +192,14 @@ impl Scraper {
}
if is_html {
if !self.single_page {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
if self.single_page {
log::info!(" [single-page mode] not crawling for links");
} else if let Ok(html) = std::str::from_utf8(&result.bytes) {
let (new_count, new_links) = self.extract_new_links(html, &final_url);
for link in &new_links {
queue.push_back(link.clone());
}
log::info!(" found {} links ({} new)", new_count, new_links.len());
}
} else {
log::info!(" [single-page mode] not crawling for links");
log::info!(" found {new_count} links ({} new)", new_links.len());
}
}
}
@@ -206,19 +221,19 @@ impl Scraper {
}
if skipped_robots > 0 {
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
log::info!("Skipped {skipped_robots} URLs due to robots.txt");
}
if skipped_scope > 0 {
log::info!("Skipped {} URLs due to path scope", skipped_scope);
log::info!("Skipped {skipped_scope} URLs due to path scope");
}
if skipped_type > 0 {
log::info!("Skipped {} URLs due to type filter", skipped_type);
log::info!("Skipped {skipped_type} URLs due to type filter");
}
(count, error_count)
}
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
@@ -226,7 +241,7 @@ impl Scraper {
let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html);
if let Some(stripped) = filename.strip_suffix(".html") {
filename = format!("{}.md", stripped);
filename = format!("{stripped}.md");
}
content = md.into_bytes();
} else if self.convert_docs {
@@ -263,25 +278,6 @@ impl Scraper {
fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
DOC_TYPES.contains(&ct.as_str())
}
@@ -296,7 +292,7 @@ mod tests {
exclude_types: Option<HashSet<String>>,
) -> Scraper {
Scraper {
fetcher: Fetcher::new(Client::new()).unwrap(),
fetcher: Fetcher::new(Client::new()),
seen: HashSet::new(),
output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(),
+6 -13
View File
@@ -1,4 +1,3 @@
// src/url_utils.rs
#![warn(clippy::all, clippy::pedantic)]
use md5::Md5;
@@ -35,11 +34,7 @@ pub fn normalize_url(url: &str) -> Option<Url> {
Some(parsed)
}
/// Returns true if `url`'s path falls within the given scope.
/// A scope of `/blog/2021` matches `/blog/2021` and `/blog/2021/...`
/// but not `/blog/20212`.
/// `None` scope means no restriction.
pub fn is_in_scope(url: &Url, scope_path: &Option<String>) -> bool {
pub fn is_in_scope(url: &Url, scope_path: Option<&String>) -> bool {
match scope_path {
None => true,
Some(scope) => {
@@ -48,7 +43,7 @@ pub fn is_in_scope(url: &Url, scope_path: &Option<String>) -> bool {
return true;
}
let path = url.path();
path == scope || path.starts_with(&format!("{}/", scope))
path == scope || path.starts_with(&format!("{scope}/"))
}
}
}
@@ -57,11 +52,11 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
let path = url.path();
let last_segment = path.rsplit('/').next().unwrap_or(path);
if let Some(dot_pos) = last_segment.rfind('.') {
if dot_pos > 0 {
if let Some(dot_pos) = last_segment.rfind('.')
&& dot_pos > 0
{
return last_segment[dot_pos..].to_lowercase();
}
}
let ct = content_type
.split(';')
@@ -100,7 +95,6 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
"application/vnd.oasis.opendocument.spreadsheet" => ".ods",
"application/vnd.oasis.opendocument.presentation" => ".odp",
"application/zip" => ".zip",
"application/octet-stream" => ".bin",
_ => ".bin",
}
.to_string()
@@ -120,8 +114,7 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
let host = url
.host_str()
.unwrap_or("unknown")
.replace('.', "_")
.replace(':', "_");
.replace(['.', ':'], "_");
let path = url.path().trim_start_matches('/');
let path = if path.is_empty() { "index" } else { path };
let path = path.trim_end_matches('/');