fix test and more clippy

This commit is contained in:
2026-08-25 12:17:17 +01:00
parent fe4b5ed18a
commit 7837b08c86
4 changed files with 39 additions and 14 deletions
+7 -2
View File
@@ -9,7 +9,12 @@ fn get_link_selector() -> &'static Selector {
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid")) LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
} }
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> { pub fn extract_links(
html: &str,
base_url: &Url,
base_host: &str,
allow_subdomains: bool,
) -> HashSet<Url> {
let document = Html::parse_document(html); let document = Html::parse_document(html);
let selector = get_link_selector(); let selector = get_link_selector();
@@ -23,7 +28,7 @@ pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<U
&& !href.starts_with("javascript:") && !href.starts_with("javascript:")
}) })
.filter_map(|href| base_url.join(href).ok()) .filter_map(|href| base_url.join(href).ok())
.filter(|url| crate::url_utils::is_internal(url, base_domain)) .filter(|url| crate::url_utils::is_internal(url, base_host, allow_subdomains))
.map(|mut url| { .map(|mut url| {
url.set_fragment(None); url.set_fragment(None);
url url
+18 -2
View File
@@ -30,6 +30,12 @@ struct Args {
#[arg(help = "Starting URL to scrape")] #[arg(help = "Starting URL to scrape")]
start_url: String, start_url: String,
#[arg(
long,
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
)]
subdomains: bool,
#[arg( #[arg(
long, long,
help = "Only scrape the single page at START_URL, do not crawl for links" help = "Only scrape the single page at START_URL, do not crawl for links"
@@ -79,6 +85,11 @@ async fn main() -> Result<()> {
let url = Url::parse(&start_url)?; let url = Url::parse(&start_url)?;
let base_host = url
.host_str()
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?
.to_string();
let base_domain = url_utils::derive_base_domain(&url) let base_domain = url_utils::derive_base_domain(&url)
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?; .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
@@ -124,7 +135,11 @@ async fn main() -> Result<()> {
}; };
log::info!("Scraping: {start_url}"); log::info!("Scraping: {start_url}");
log::info!("Base domain: {base_domain}"); log::info!("Base host: {base_host}");
log::info!(
"Subdomain crawling: {}",
if args.subdomains { "enabled" } else { "disabled" }
);
log::info!("Output dir: {}", output_dir.display()); log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display()); log::info!("Logs dir: {}", logs_dir.display());
log::info!( log::info!(
@@ -160,7 +175,7 @@ async fn main() -> Result<()> {
let mut scraper = Scraper::new( let mut scraper = Scraper::new(
output_dir, output_dir,
log_path.clone(), log_path.clone(),
base_domain, base_host,
robots, robots,
convert_docs, convert_docs,
fetcher, fetcher,
@@ -168,6 +183,7 @@ async fn main() -> Result<()> {
exclude_types, exclude_types,
scope_path, scope_path,
args.single, args.single,
args.subdomains,
); );
let (count, errors) = scraper.run(&url).await; let (count, errors) = scraper.run(&url).await;
+9 -5
View File
@@ -45,7 +45,7 @@ pub struct Scraper {
seen: HashSet<String>, seen: HashSet<String>,
output_dir: std::path::PathBuf, output_dir: std::path::PathBuf,
log_path: std::path::PathBuf, log_path: std::path::PathBuf,
base_domain: String, base_host: String,
robots: RobotsRule, robots: RobotsRule,
doc_stats: DocStats, doc_stats: DocStats,
convert_docs: bool, convert_docs: bool,
@@ -53,6 +53,7 @@ pub struct Scraper {
exclude_types: Option<HashSet<String>>, exclude_types: Option<HashSet<String>>,
scope_path: Option<String>, scope_path: Option<String>,
single_page: bool, single_page: bool,
allow_subdomains: bool,
} }
impl Scraper { impl Scraper {
@@ -60,7 +61,7 @@ impl Scraper {
pub fn new( pub fn new(
output_dir: std::path::PathBuf, output_dir: std::path::PathBuf,
log_path: std::path::PathBuf, log_path: std::path::PathBuf,
base_domain: String, base_host: String,
robots: RobotsRule, robots: RobotsRule,
convert_docs: bool, convert_docs: bool,
fetcher: Fetcher, fetcher: Fetcher,
@@ -68,13 +69,14 @@ impl Scraper {
exclude_types: Option<HashSet<String>>, exclude_types: Option<HashSet<String>>,
scope_path: Option<String>, scope_path: Option<String>,
single_page: bool, single_page: bool,
allow_subdomains: bool,
) -> Self { ) -> Self {
Self { Self {
fetcher, fetcher,
seen: HashSet::new(), seen: HashSet::new(),
output_dir, output_dir,
log_path, log_path,
base_domain, base_host,
robots, robots,
doc_stats: DocStats::default(), doc_stats: DocStats::default(),
convert_docs, convert_docs,
@@ -82,6 +84,7 @@ impl Scraper {
exclude_types, exclude_types,
scope_path, scope_path,
single_page, single_page,
allow_subdomains,
} }
} }
@@ -111,7 +114,7 @@ impl Scraper {
html: &str, html: &str,
final_url: &Url, final_url: &Url,
) -> (usize, Vec<Url>) { ) -> (usize, Vec<Url>) {
let links = extract_links(html, final_url, &self.base_domain); let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains);
let new_count = links.len(); let new_count = links.len();
let new_links: Vec<Url> = links let new_links: Vec<Url> = links
.into_iter() .into_iter()
@@ -296,7 +299,7 @@ mod tests {
seen: HashSet::new(), seen: HashSet::new(),
output_dir: std::path::PathBuf::new(), output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(),
base_domain: "example.com".to_string(), base_host: "example.com".to_string(),
robots: RobotsRule { robots: RobotsRule {
allowed: Vec::new(), allowed: Vec::new(),
disallowed: Vec::new(), disallowed: Vec::new(),
@@ -307,6 +310,7 @@ mod tests {
exclude_types, exclude_types,
scope_path: None, scope_path: None,
single_page: false, single_page: false,
allow_subdomains: false,
} }
} }
+5 -5
View File
@@ -14,13 +14,13 @@ pub fn derive_base_domain(url: &Url) -> Option<String> {
} }
} }
pub fn is_internal(url: &Url, base_domain: &str) -> bool { pub fn is_internal(url: &Url, base_host: &str, allow_subdomains: bool) -> bool {
match url.host_str() { match url.host_str() {
Some(host) => { Some(host) => {
host == base_domain if host == base_host {
|| host return true;
.strip_suffix(base_domain) }
.is_some_and(|rest| rest.ends_with('.')) allow_subdomains && host.ends_with(&format!(".{base_host}"))
} }
None => false, None => false,
} }