fix clippy warnigns

This commit is contained in:
2026-08-25 12:01:07 +01:00
parent da8f7302c1
commit fe4b5ed18a
11 changed files with 106 additions and 155 deletions
+2 -1
View File
@@ -1,2 +1,3 @@
/target /target
scraped* *scraped/
*scraped_logs/
Generated
-7
View File
@@ -1220,12 +1220,6 @@ dependencies = [
"wasm-bindgen", "wasm-bindgen",
] ]
[[package]]
name = "lazy_static"
version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
[[package]] [[package]]
name = "libc" name = "libc"
version = "0.2.189" version = "0.2.189"
@@ -2617,7 +2611,6 @@ dependencies = [
"env_logger", "env_logger",
"hex", "hex",
"htmd", "htmd",
"lazy_static",
"log", "log",
"md-5", "md-5",
"reqwest", "reqwest",
-2
View File
@@ -1,4 +1,3 @@
# Cargo.toml
[package] [package]
name = "web-scraper" name = "web-scraper"
version = "0.1.1" version = "0.1.1"
@@ -11,7 +10,6 @@ chrono = { version = "0.4", features = ["serde"] }
clap = { version = "4", features = ["derive"] } clap = { version = "4", features = ["derive"] }
hex = "0.4" hex = "0.4"
htmd = "0.1" htmd = "0.1"
lazy_static = "1.4"
log = "0.4" log = "0.4"
env_logger = "0.11" env_logger = "0.11"
md-5 = "0.10" md-5 = "0.10"
+1 -6
View File
@@ -21,10 +21,6 @@ fn get_body_selector() -> &'static Selector {
BODY_SELECTOR.get_or_init(|| Selector::parse("body").expect("hardcoded selector is valid")) BODY_SELECTOR.get_or_init(|| Selector::parse("body").expect("hardcoded selector is valid"))
} }
/// Extract the `<body>` element's HTML from a full HTML document.
/// The HTML5 parser always synthesizes a `<body>` element, so this
/// returns `Some` for any well-formed document. Returns `None` only
/// if the parser fails entirely.
fn extract_body(html: &str) -> Option<String> { fn extract_body(html: &str) -> Option<String> {
let document = Html::parse_document(html); let document = Html::parse_document(html);
document document
@@ -40,7 +36,7 @@ pub fn html_to_markdown(html: &str) -> String {
match converter.convert(&source) { match converter.convert(&source) {
Ok(md) => md, Ok(md) => md,
Err(e) => { Err(e) => {
log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e); log::warn!("HTML-to-MD conversion failed ({e}), saving raw HTML");
source source
} }
} }
@@ -60,7 +56,6 @@ mod tests {
#[test] #[test]
fn test_extract_body_without_body_tag_falls_back_to_synthetic() { fn test_extract_body_without_body_tag_falls_back_to_synthetic() {
// HTML5 parser synthesizes a <body> element even when absent in source.
let html = "<html><head></head><p>Hello</p></html>"; let html = "<html><head></head><p>Hello</p></html>";
let result = extract_body(html); let result = extract_body(html);
assert!(result.is_some(), "parser should synthesize a body element"); assert!(result.is_some(), "parser should synthesize a body element");
+6 -9
View File
@@ -1,4 +1,3 @@
// src/doc_processor.rs
#![warn(clippy::all, clippy::pedantic)] #![warn(clippy::all, clippy::pedantic)]
use anyhow::Result; use anyhow::Result;
@@ -29,7 +28,7 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
DocProcessResult::Raw DocProcessResult::Raw
} }
Err(e) => { Err(e) => {
log::warn!("anydoc conversion failed ({}), saving raw", e); log::warn!("anydoc conversion failed ({e}), saving raw");
DocProcessResult::Raw DocProcessResult::Raw
} }
} }
@@ -37,23 +36,23 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
#[allow(dead_code)] #[allow(dead_code)]
pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, usize, usize)> { pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, usize, usize)> {
let supported_exts = [ const SUPPORTED_EXTS: &[&str] = &[
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls", "pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv", "xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
]; ];
let entries: Vec<_> = std::fs::read_dir(dir)? let entries: Vec<_> = std::fs::read_dir(dir)?
.filter_map(|e| e.ok()) .filter_map(std::result::Result::ok)
.filter(|e| { .filter(|e| {
e.path().extension().is_some_and(|ext| { e.path().extension().is_some_and(|ext| {
let ext = ext.to_string_lossy().to_lowercase(); let ext = ext.to_string_lossy().to_lowercase();
supported_exts.contains(&ext.as_str()) SUPPORTED_EXTS.contains(&ext.as_str())
}) })
}) })
.collect(); .collect();
let total = entries.len(); let total = entries.len();
log::info!("Found {} documents to process", total); log::info!("Found {total} documents to process");
let mut converted = 0usize; let mut converted = 0usize;
let mut raw = 0usize; let mut raw = 0usize;
@@ -67,7 +66,7 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
let bytes = match std::fs::read(&path) { let bytes = match std::fs::read(&path) {
Ok(b) => b, Ok(b) => b,
Err(e) => { Err(e) => {
log::error!("error reading file: {}", e); log::error!("error reading file: {e}");
errors += 1; errors += 1;
continue; continue;
} }
@@ -102,7 +101,6 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
Ok((converted, raw, errors)) Ok((converted, raw, errors))
} }
// src/doc_processor.rs - tests section added at end
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
@@ -113,7 +111,6 @@ mod tests {
let html_bytes = b"<html><body><p>Test</p></body></html>"; let html_bytes = b"<html><body><p>Test</p></body></html>";
let url = Url::parse("https://example.com/test.html").unwrap(); let url = Url::parse("https://example.com/test.html").unwrap();
let result = try_convert(html_bytes, &url); let result = try_convert(html_bytes, &url);
// Will likely return Raw since anydoc may not handle HTML
match result { match result {
DocProcessResult::Markdown(_) | DocProcessResult::Raw => {} DocProcessResult::Markdown(_) | DocProcessResult::Raw => {}
} }
+6 -9
View File
@@ -1,4 +1,3 @@
// src/error_logger.rs
#![warn(clippy::all, clippy::pedantic)] #![warn(clippy::all, clippy::pedantic)]
use chrono::{DateTime, Utc}; use chrono::{DateTime, Utc};
@@ -6,7 +5,7 @@ use serde::Serialize;
use std::fs::OpenOptions; use std::fs::OpenOptions;
use std::io::{BufWriter, Write}; use std::io::{BufWriter, Write};
use std::path::Path; use std::path::Path;
use std::sync::Mutex; use std::sync::{LazyLock, Mutex};
use url::Url; use url::Url;
#[derive(Serialize)] #[derive(Serialize)]
@@ -19,9 +18,8 @@ pub struct ErrorEntry {
pub status_code: Option<u16>, pub status_code: Option<u16>,
} }
lazy_static::lazy_static! { static LOG_WRITER: LazyLock<Mutex<Option<BufWriter<std::fs::File>>>> =
static ref LOG_WRITER: Mutex<Option<BufWriter<std::fs::File>>> = Mutex::new(None); LazyLock::new(|| Mutex::new(None));
}
pub fn log_error( pub fn log_error(
log_path: &Path, log_path: &Path,
@@ -41,18 +39,17 @@ pub fn log_error(
let line = match serde_json::to_string(&entry) { let line = match serde_json::to_string(&entry) {
Ok(json) => json + "\n", Ok(json) => json + "\n",
Err(e) => { Err(e) => {
log::error!("Failed to serialize error entry: {}", e); log::error!("Failed to serialize error entry: {e}");
return; return;
} }
}; };
// Buffered writer approach
{ {
let mut guard = LOG_WRITER.lock().unwrap(); let mut guard = LOG_WRITER.lock().unwrap();
match guard.as_mut() { match guard.as_mut() {
Some(writer) => { Some(writer) => {
if let Err(e) = writer.write_all(line.as_bytes()) { if let Err(e) = writer.write_all(line.as_bytes()) {
log::error!("Failed to write error log: {}", e); log::error!("Failed to write error log: {e}");
} }
} }
None => { None => {
@@ -68,5 +65,5 @@ pub fn log_error(
} }
} }
log::info!("LOGGED ERROR [{}]: {}", error_type, message); log::info!("LOGGED ERROR [{error_type}]: {message}");
} }
+9 -18
View File
@@ -11,8 +11,8 @@ pub struct Fetcher {
} }
impl Fetcher { impl Fetcher {
pub fn new(client: Client) -> Result<Self> { pub fn new(client: Client) -> Self {
Ok(Self { client }) Self { client }
} }
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> { pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
@@ -33,21 +33,16 @@ impl Fetcher {
Err(e) => { Err(e) => {
if attempt < MAX_RETRIES { if attempt < MAX_RETRIES {
log::warn!( log::warn!(
"retry {}/{}: body read failed ({})", "retry {attempt}/{MAX_RETRIES}: body read failed ({e})"
attempt,
MAX_RETRIES,
e
); );
tokio::time::sleep(Duration::from_secs( tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64, RETRY_BACKOFF_SECS * u64::from(attempt),
)) ))
.await; .await;
continue; continue;
} }
return Err(anyhow!( return Err(anyhow!(
"Body read failed after {} retries: {}", "Body read failed after {MAX_RETRIES} retries: {e}"
MAX_RETRIES,
e
)); ));
} }
}; };
@@ -64,23 +59,19 @@ impl Fetcher {
} }
Err(e) => { Err(e) => {
if attempt < MAX_RETRIES { if attempt < MAX_RETRIES {
log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e); log::warn!("retry {attempt}/{MAX_RETRIES}: request failed ({e})");
tokio::time::sleep(Duration::from_secs( tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64, RETRY_BACKOFF_SECS * u64::from(attempt),
)) ))
.await; .await;
continue; continue;
} }
return Err(anyhow!( return Err(anyhow!("Request failed after {MAX_RETRIES} retries: {e}"));
"Request failed after {} retries: {}",
MAX_RETRIES,
e
));
} }
} }
} }
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES)) Err(anyhow!("Exhausted all {MAX_RETRIES} retries"))
} }
} }
+14 -17
View File
@@ -80,26 +80,23 @@ async fn main() -> Result<()> {
let url = Url::parse(&start_url)?; let url = Url::parse(&start_url)?;
let base_domain = url_utils::derive_base_domain(&url) let base_domain = url_utils::derive_base_domain(&url)
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?; .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
let output_dir = match &args.output { let output_dir = if let Some(p) = &args.output {
Some(p) => PathBuf::from(p), PathBuf::from(p)
None => { } else {
let default_name = format!("{}_scraped", base_domain.replace('.', "_")); let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
PathBuf::from(default_name) PathBuf::from(default_name)
}
}; };
let logs_dir = { let logs_dir = {
let output_name = output_dir let output_name = output_dir
.file_name() .file_name()
.map(|n| n.to_string_lossy().into_owned()) .map_or_else(|| "output".to_string(), |n| n.to_string_lossy().into_owned());
.unwrap_or_else(|| "output".to_string()); let logs_name = format!("{output_name}_logs");
let logs_name = format!("{}_logs", output_name);
output_dir output_dir
.parent() .parent()
.map(|p| p.join(&logs_name)) .map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
.unwrap_or_else(|| PathBuf::from(&logs_name))
}; };
fs::create_dir_all(&output_dir)?; fs::create_dir_all(&output_dir)?;
@@ -126,8 +123,8 @@ async fn main() -> Result<()> {
} }
}; };
log::info!("Scraping: {}", start_url); log::info!("Scraping: {start_url}");
log::info!("Base domain: {}", base_domain); log::info!("Base domain: {base_domain}");
log::info!("Output dir: {}", output_dir.display()); log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display()); log::info!("Logs dir: {}", logs_dir.display());
log::info!( log::info!(
@@ -140,13 +137,13 @@ async fn main() -> Result<()> {
); );
match (&include_types, &exclude_types) { match (&include_types, &exclude_types) {
(Some(t), _) => log::info!("Type filter: include {:?}", t), (Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {:?}", t), (_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"), _ => log::info!("Type filter: none (all types)"),
} }
match &scope_path { match &scope_path {
Some(s) => log::info!("Path scope: {} (only this path and deeper)", s), Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"), None => log::info!("Path scope: none (full site)"),
} }
@@ -159,7 +156,7 @@ async fn main() -> Result<()> {
let robots = fetch_robots(&client, &url, config::USER_AGENT).await; let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
log::info!(""); log::info!("");
let fetcher = fetcher::Fetcher::new(client)?; let fetcher = fetcher::Fetcher::new(client);
let mut scraper = Scraper::new( let mut scraper = Scraper::new(
output_dir, output_dir,
log_path.clone(), log_path.clone(),
@@ -174,7 +171,7 @@ async fn main() -> Result<()> {
); );
let (count, errors) = scraper.run(&url).await; let (count, errors) = scraper.run(&url).await;
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors); log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
log::info!("Error log: {}", log_path.display()); log::info!("Error log: {}", log_path.display());
Ok(()) Ok(())
+2 -9
View File
@@ -93,10 +93,7 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
); );
} }
Err(e) => { Err(e) => {
log::warn!( log::warn!("Failed to fetch robots.txt ({e}): assuming no restrictions");
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
} }
} }
@@ -133,11 +130,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
let agent = agent.trim().to_lowercase(); let agent = agent.trim().to_lowercase();
current_agents.push(agent.clone()); current_agents.push(agent.clone());
if agent == "*" || agent == target_lower || agent == "web-scraper" { collecting = agent == "*" || agent == target_lower || agent == "web-scraper";
collecting = true;
} else {
collecting = false;
}
} else if collecting { } else if collecting {
if let Some(path) = line.strip_prefix("Disallow:") { if let Some(path) = line.strip_prefix("Disallow:") {
let path = path.trim().to_string(); let path = path.trim().to_string();
+55 -59
View File
@@ -13,22 +13,33 @@ use std::fs;
use std::time::Duration; use std::time::Duration;
use url::Url; use url::Url;
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
#[derive(Default)]
pub struct DocStats { pub struct DocStats {
pub converted: usize, pub converted: usize,
pub raw: usize, pub raw: usize,
pub errors: usize, pub errors: usize,
} }
impl Default for DocStats {
fn default() -> Self {
Self {
converted: 0,
raw: 0,
errors: 0,
}
}
}
pub struct Scraper { pub struct Scraper {
fetcher: Fetcher, fetcher: Fetcher,
seen: HashSet<String>, seen: HashSet<String>,
@@ -45,6 +56,7 @@ pub struct Scraper {
} }
impl Scraper { impl Scraper {
#[allow(clippy::too_many_arguments)]
pub fn new( pub fn new(
output_dir: std::path::PathBuf, output_dir: std::path::PathBuf,
log_path: std::path::PathBuf, log_path: std::path::PathBuf,
@@ -94,6 +106,20 @@ impl Scraper {
} }
} }
fn extract_new_links(
&self,
html: &str,
final_url: &Url,
) -> (usize, Vec<Url>) {
let links = extract_links(html, final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
(new_count, new_links)
}
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) { pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque<Url> = VecDeque::new(); let mut queue: VecDeque<Url> = VecDeque::new();
let start_url_owned = start_url.clone(); let start_url_owned = start_url.clone();
@@ -106,9 +132,8 @@ impl Scraper {
let mut skipped_type = 0usize; let mut skipped_type = 0usize;
while let Some(raw_url) = queue.pop_front() { while let Some(raw_url) = queue.pop_front() {
let url = match normalize_url(raw_url.as_str()) { let Some(url) = normalize_url(raw_url.as_str()) else {
Some(u) => u, continue;
None => continue,
}; };
let url_key = url.as_str().to_string(); let url_key = url.as_str().to_string();
@@ -118,19 +143,19 @@ impl Scraper {
} }
self.seen.insert(url_key.clone()); self.seen.insert(url_key.clone());
if !is_in_scope(&url, &self.scope_path) { if !is_in_scope(&url, self.scope_path.as_ref()) {
log::debug!("[skip] out of scope: {}", url); log::debug!("[skip] out of scope: {url}");
skipped_scope += 1; skipped_scope += 1;
continue; continue;
} }
if !self.robots.is_allowed(url.path()) { if !self.robots.is_allowed(url.path()) {
log::info!("[skip] robots.txt disallows: {}", url); log::info!("[skip] robots.txt disallows: {url}");
skipped_robots += 1; skipped_robots += 1;
continue; continue;
} }
log::info!("[{}] fetching: {}", count, url); log::info!("[{count}] fetching: {url}");
match self.fetcher.fetch_with_retry(&url).await { match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => { Ok(result) => {
@@ -141,12 +166,12 @@ impl Scraper {
self.should_save(&result.final_url, &result.content_type); self.should_save(&result.final_url, &result.content_type);
if !should_save { if !should_save {
log::info!(" skipped (type filter: {})", ext_label); log::info!(" skipped (type filter: {ext_label})");
skipped_type += 1; skipped_type += 1;
} }
if should_save { if should_save {
match self.save(&result, is_html).await { match self.save(&result, is_html) {
Ok(()) => { Ok(()) => {
if !is_html { if !is_html {
log::info!(" binary: {}", result.content_type); log::info!(" binary: {}", result.content_type);
@@ -167,24 +192,14 @@ impl Scraper {
} }
if is_html { if is_html {
if !self.single_page { if self.single_page {
if let Ok(html) = std::str::from_utf8(&result.bytes) { log::info!(" [single-page mode] not crawling for links");
let links = extract_links(html, &final_url, &self.base_domain); } else if let Ok(html) = std::str::from_utf8(&result.bytes) {
let new_count = links.len(); let (new_count, new_links) = self.extract_new_links(html, &final_url);
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
for link in &new_links { for link in &new_links {
queue.push_back(link.clone()); queue.push_back(link.clone());
} }
log::info!(" found {new_count} links ({} new)", new_links.len());
log::info!(" found {} links ({} new)", new_count, new_links.len());
}
} else {
log::info!(" [single-page mode] not crawling for links");
} }
} }
} }
@@ -206,19 +221,19 @@ impl Scraper {
} }
if skipped_robots > 0 { if skipped_robots > 0 {
log::info!("Skipped {} URLs due to robots.txt", skipped_robots); log::info!("Skipped {skipped_robots} URLs due to robots.txt");
} }
if skipped_scope > 0 { if skipped_scope > 0 {
log::info!("Skipped {} URLs due to path scope", skipped_scope); log::info!("Skipped {skipped_scope} URLs due to path scope");
} }
if skipped_type > 0 { if skipped_type > 0 {
log::info!("Skipped {} URLs due to type filter", skipped_type); log::info!("Skipped {skipped_type} URLs due to type filter");
} }
(count, error_count) (count, error_count)
} }
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> { fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type); let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>; let content: Vec<u8>;
@@ -226,7 +241,7 @@ impl Scraper {
let html = std::str::from_utf8(&result.bytes)?; let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html); let md = html_to_markdown(html);
if let Some(stripped) = filename.strip_suffix(".html") { if let Some(stripped) = filename.strip_suffix(".html") {
filename = format!("{}.md", stripped); filename = format!("{stripped}.md");
} }
content = md.into_bytes(); content = md.into_bytes();
} else if self.convert_docs { } else if self.convert_docs {
@@ -263,25 +278,6 @@ impl Scraper {
fn is_document_content_type(ct: &str) -> bool { fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase(); let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
DOC_TYPES.contains(&ct.as_str()) DOC_TYPES.contains(&ct.as_str())
} }
@@ -296,7 +292,7 @@ mod tests {
exclude_types: Option<HashSet<String>>, exclude_types: Option<HashSet<String>>,
) -> Scraper { ) -> Scraper {
Scraper { Scraper {
fetcher: Fetcher::new(Client::new()).unwrap(), fetcher: Fetcher::new(Client::new()),
seen: HashSet::new(), seen: HashSet::new(),
output_dir: std::path::PathBuf::new(), output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(),
+6 -13
View File
@@ -1,4 +1,3 @@
// src/url_utils.rs
#![warn(clippy::all, clippy::pedantic)] #![warn(clippy::all, clippy::pedantic)]
use md5::Md5; use md5::Md5;
@@ -35,11 +34,7 @@ pub fn normalize_url(url: &str) -> Option<Url> {
Some(parsed) Some(parsed)
} }
/// Returns true if `url`'s path falls within the given scope. pub fn is_in_scope(url: &Url, scope_path: Option<&String>) -> bool {
/// A scope of `/blog/2021` matches `/blog/2021` and `/blog/2021/...`
/// but not `/blog/20212`.
/// `None` scope means no restriction.
pub fn is_in_scope(url: &Url, scope_path: &Option<String>) -> bool {
match scope_path { match scope_path {
None => true, None => true,
Some(scope) => { Some(scope) => {
@@ -48,7 +43,7 @@ pub fn is_in_scope(url: &Url, scope_path: &Option<String>) -> bool {
return true; return true;
} }
let path = url.path(); let path = url.path();
path == scope || path.starts_with(&format!("{}/", scope)) path == scope || path.starts_with(&format!("{scope}/"))
} }
} }
} }
@@ -57,11 +52,11 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
let path = url.path(); let path = url.path();
let last_segment = path.rsplit('/').next().unwrap_or(path); let last_segment = path.rsplit('/').next().unwrap_or(path);
if let Some(dot_pos) = last_segment.rfind('.') { if let Some(dot_pos) = last_segment.rfind('.')
if dot_pos > 0 { && dot_pos > 0
{
return last_segment[dot_pos..].to_lowercase(); return last_segment[dot_pos..].to_lowercase();
} }
}
let ct = content_type let ct = content_type
.split(';') .split(';')
@@ -100,7 +95,6 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
"application/vnd.oasis.opendocument.spreadsheet" => ".ods", "application/vnd.oasis.opendocument.spreadsheet" => ".ods",
"application/vnd.oasis.opendocument.presentation" => ".odp", "application/vnd.oasis.opendocument.presentation" => ".odp",
"application/zip" => ".zip", "application/zip" => ".zip",
"application/octet-stream" => ".bin",
_ => ".bin", _ => ".bin",
} }
.to_string() .to_string()
@@ -120,8 +114,7 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
let host = url let host = url
.host_str() .host_str()
.unwrap_or("unknown") .unwrap_or("unknown")
.replace('.', "_") .replace(['.', ':'], "_");
.replace(':', "_");
let path = url.path().trim_start_matches('/'); let path = url.path().trim_start_matches('/');
let path = if path.is_empty() { "index" } else { path }; let path = if path.is_empty() { "index" } else { path };
let path = path.trim_end_matches('/'); let path = path.trim_end_matches('/');