fix test and more clippy
This commit is contained in:
+7
-2
@@ -9,7 +9,12 @@ fn get_link_selector() -> &'static Selector {
|
||||
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
|
||||
}
|
||||
|
||||
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
|
||||
pub fn extract_links(
|
||||
html: &str,
|
||||
base_url: &Url,
|
||||
base_host: &str,
|
||||
allow_subdomains: bool,
|
||||
) -> HashSet<Url> {
|
||||
let document = Html::parse_document(html);
|
||||
let selector = get_link_selector();
|
||||
|
||||
@@ -23,7 +28,7 @@ pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<U
|
||||
&& !href.starts_with("javascript:")
|
||||
})
|
||||
.filter_map(|href| base_url.join(href).ok())
|
||||
.filter(|url| crate::url_utils::is_internal(url, base_domain))
|
||||
.filter(|url| crate::url_utils::is_internal(url, base_host, allow_subdomains))
|
||||
.map(|mut url| {
|
||||
url.set_fragment(None);
|
||||
url
|
||||
|
||||
+18
-2
@@ -30,6 +30,12 @@ struct Args {
|
||||
#[arg(help = "Starting URL to scrape")]
|
||||
start_url: String,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
|
||||
)]
|
||||
subdomains: bool,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
help = "Only scrape the single page at START_URL, do not crawl for links"
|
||||
@@ -79,6 +85,11 @@ async fn main() -> Result<()> {
|
||||
|
||||
let url = Url::parse(&start_url)?;
|
||||
|
||||
let base_host = url
|
||||
.host_str()
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?
|
||||
.to_string();
|
||||
|
||||
let base_domain = url_utils::derive_base_domain(&url)
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
|
||||
|
||||
@@ -124,7 +135,11 @@ async fn main() -> Result<()> {
|
||||
};
|
||||
|
||||
log::info!("Scraping: {start_url}");
|
||||
log::info!("Base domain: {base_domain}");
|
||||
log::info!("Base host: {base_host}");
|
||||
log::info!(
|
||||
"Subdomain crawling: {}",
|
||||
if args.subdomains { "enabled" } else { "disabled" }
|
||||
);
|
||||
log::info!("Output dir: {}", output_dir.display());
|
||||
log::info!("Logs dir: {}", logs_dir.display());
|
||||
log::info!(
|
||||
@@ -160,7 +175,7 @@ async fn main() -> Result<()> {
|
||||
let mut scraper = Scraper::new(
|
||||
output_dir,
|
||||
log_path.clone(),
|
||||
base_domain,
|
||||
base_host,
|
||||
robots,
|
||||
convert_docs,
|
||||
fetcher,
|
||||
@@ -168,6 +183,7 @@ async fn main() -> Result<()> {
|
||||
exclude_types,
|
||||
scope_path,
|
||||
args.single,
|
||||
args.subdomains,
|
||||
);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
|
||||
+9
-5
@@ -45,7 +45,7 @@ pub struct Scraper {
|
||||
seen: HashSet<String>,
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
base_host: String,
|
||||
robots: RobotsRule,
|
||||
doc_stats: DocStats,
|
||||
convert_docs: bool,
|
||||
@@ -53,6 +53,7 @@ pub struct Scraper {
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
scope_path: Option<String>,
|
||||
single_page: bool,
|
||||
allow_subdomains: bool,
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
@@ -60,7 +61,7 @@ impl Scraper {
|
||||
pub fn new(
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
base_host: String,
|
||||
robots: RobotsRule,
|
||||
convert_docs: bool,
|
||||
fetcher: Fetcher,
|
||||
@@ -68,13 +69,14 @@ impl Scraper {
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
scope_path: Option<String>,
|
||||
single_page: bool,
|
||||
allow_subdomains: bool,
|
||||
) -> Self {
|
||||
Self {
|
||||
fetcher,
|
||||
seen: HashSet::new(),
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
base_host,
|
||||
robots,
|
||||
doc_stats: DocStats::default(),
|
||||
convert_docs,
|
||||
@@ -82,6 +84,7 @@ impl Scraper {
|
||||
exclude_types,
|
||||
scope_path,
|
||||
single_page,
|
||||
allow_subdomains,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -111,7 +114,7 @@ impl Scraper {
|
||||
html: &str,
|
||||
final_url: &Url,
|
||||
) -> (usize, Vec<Url>) {
|
||||
let links = extract_links(html, final_url, &self.base_domain);
|
||||
let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains);
|
||||
let new_count = links.len();
|
||||
let new_links: Vec<Url> = links
|
||||
.into_iter()
|
||||
@@ -296,7 +299,7 @@ mod tests {
|
||||
seen: HashSet::new(),
|
||||
output_dir: std::path::PathBuf::new(),
|
||||
log_path: std::path::PathBuf::new(),
|
||||
base_domain: "example.com".to_string(),
|
||||
base_host: "example.com".to_string(),
|
||||
robots: RobotsRule {
|
||||
allowed: Vec::new(),
|
||||
disallowed: Vec::new(),
|
||||
@@ -307,6 +310,7 @@ mod tests {
|
||||
exclude_types,
|
||||
scope_path: None,
|
||||
single_page: false,
|
||||
allow_subdomains: false,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+5
-5
@@ -14,13 +14,13 @@ pub fn derive_base_domain(url: &Url) -> Option<String> {
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
|
||||
pub fn is_internal(url: &Url, base_host: &str, allow_subdomains: bool) -> bool {
|
||||
match url.host_str() {
|
||||
Some(host) => {
|
||||
host == base_domain
|
||||
|| host
|
||||
.strip_suffix(base_domain)
|
||||
.is_some_and(|rest| rest.ends_with('.'))
|
||||
if host == base_host {
|
||||
return true;
|
||||
}
|
||||
allow_subdomains && host.ends_with(&format!(".{base_host}"))
|
||||
}
|
||||
None => false,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user