From 7837b08c86ca157326c6086d619f317cfadfa0c9 Mon Sep 17 00:00:00 2001 From: Bartal Laearsson Date: Tue, 25 Aug 2026 12:17:17 +0100 Subject: [PATCH] fix test and more clippy --- src/extractor.rs | 9 +++++++-- src/main.rs | 20 ++++++++++++++++++-- src/scraper.rs | 14 +++++++++----- src/url_utils.rs | 10 +++++----- 4 files changed, 39 insertions(+), 14 deletions(-) diff --git a/src/extractor.rs b/src/extractor.rs index 3287d8b..a935a1a 100644 --- a/src/extractor.rs +++ b/src/extractor.rs @@ -9,7 +9,12 @@ fn get_link_selector() -> &'static Selector { LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid")) } -pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet { +pub fn extract_links( + html: &str, + base_url: &Url, + base_host: &str, + allow_subdomains: bool, +) -> HashSet { let document = Html::parse_document(html); let selector = get_link_selector(); @@ -23,7 +28,7 @@ pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet Result<()> { let url = Url::parse(&start_url)?; + let base_host = url + .host_str() + .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))? + .to_string(); + let base_domain = url_utils::derive_base_domain(&url) .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?; @@ -124,7 +135,11 @@ async fn main() -> Result<()> { }; log::info!("Scraping: {start_url}"); - log::info!("Base domain: {base_domain}"); + log::info!("Base host: {base_host}"); + log::info!( + "Subdomain crawling: {}", + if args.subdomains { "enabled" } else { "disabled" } + ); log::info!("Output dir: {}", output_dir.display()); log::info!("Logs dir: {}", logs_dir.display()); log::info!( @@ -160,7 +175,7 @@ async fn main() -> Result<()> { let mut scraper = Scraper::new( output_dir, log_path.clone(), - base_domain, + base_host, robots, convert_docs, fetcher, @@ -168,6 +183,7 @@ async fn main() -> Result<()> { exclude_types, scope_path, args.single, + args.subdomains, ); let (count, errors) = scraper.run(&url).await; diff --git a/src/scraper.rs b/src/scraper.rs index 93f48ae..6718651 100644 --- a/src/scraper.rs +++ b/src/scraper.rs @@ -45,7 +45,7 @@ pub struct Scraper { seen: HashSet, output_dir: std::path::PathBuf, log_path: std::path::PathBuf, - base_domain: String, + base_host: String, robots: RobotsRule, doc_stats: DocStats, convert_docs: bool, @@ -53,6 +53,7 @@ pub struct Scraper { exclude_types: Option>, scope_path: Option, single_page: bool, + allow_subdomains: bool, } impl Scraper { @@ -60,7 +61,7 @@ impl Scraper { pub fn new( output_dir: std::path::PathBuf, log_path: std::path::PathBuf, - base_domain: String, + base_host: String, robots: RobotsRule, convert_docs: bool, fetcher: Fetcher, @@ -68,13 +69,14 @@ impl Scraper { exclude_types: Option>, scope_path: Option, single_page: bool, + allow_subdomains: bool, ) -> Self { Self { fetcher, seen: HashSet::new(), output_dir, log_path, - base_domain, + base_host, robots, doc_stats: DocStats::default(), convert_docs, @@ -82,6 +84,7 @@ impl Scraper { exclude_types, scope_path, single_page, + allow_subdomains, } } @@ -111,7 +114,7 @@ impl Scraper { html: &str, final_url: &Url, ) -> (usize, Vec) { - let links = extract_links(html, final_url, &self.base_domain); + let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains); let new_count = links.len(); let new_links: Vec = links .into_iter() @@ -296,7 +299,7 @@ mod tests { seen: HashSet::new(), output_dir: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(), - base_domain: "example.com".to_string(), + base_host: "example.com".to_string(), robots: RobotsRule { allowed: Vec::new(), disallowed: Vec::new(), @@ -307,6 +310,7 @@ mod tests { exclude_types, scope_path: None, single_page: false, + allow_subdomains: false, } } diff --git a/src/url_utils.rs b/src/url_utils.rs index 1315f37..de93fce 100644 --- a/src/url_utils.rs +++ b/src/url_utils.rs @@ -14,13 +14,13 @@ pub fn derive_base_domain(url: &Url) -> Option { } } -pub fn is_internal(url: &Url, base_domain: &str) -> bool { +pub fn is_internal(url: &Url, base_host: &str, allow_subdomains: bool) -> bool { match url.host_str() { Some(host) => { - host == base_domain - || host - .strip_suffix(base_domain) - .is_some_and(|rest| rest.ends_with('.')) + if host == base_host { + return true; + } + allow_subdomains && host.ends_with(&format!(".{base_host}")) } None => false, }