fix test and more clippy
This commit is contained in:
+7
-2
@@ -9,7 +9,12 @@ fn get_link_selector() -> &'static Selector {
|
|||||||
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
|
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
|
pub fn extract_links(
|
||||||
|
html: &str,
|
||||||
|
base_url: &Url,
|
||||||
|
base_host: &str,
|
||||||
|
allow_subdomains: bool,
|
||||||
|
) -> HashSet<Url> {
|
||||||
let document = Html::parse_document(html);
|
let document = Html::parse_document(html);
|
||||||
let selector = get_link_selector();
|
let selector = get_link_selector();
|
||||||
|
|
||||||
@@ -23,7 +28,7 @@ pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<U
|
|||||||
&& !href.starts_with("javascript:")
|
&& !href.starts_with("javascript:")
|
||||||
})
|
})
|
||||||
.filter_map(|href| base_url.join(href).ok())
|
.filter_map(|href| base_url.join(href).ok())
|
||||||
.filter(|url| crate::url_utils::is_internal(url, base_domain))
|
.filter(|url| crate::url_utils::is_internal(url, base_host, allow_subdomains))
|
||||||
.map(|mut url| {
|
.map(|mut url| {
|
||||||
url.set_fragment(None);
|
url.set_fragment(None);
|
||||||
url
|
url
|
||||||
|
|||||||
+18
-2
@@ -30,6 +30,12 @@ struct Args {
|
|||||||
#[arg(help = "Starting URL to scrape")]
|
#[arg(help = "Starting URL to scrape")]
|
||||||
start_url: String,
|
start_url: String,
|
||||||
|
|
||||||
|
#[arg(
|
||||||
|
long,
|
||||||
|
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
|
||||||
|
)]
|
||||||
|
subdomains: bool,
|
||||||
|
|
||||||
#[arg(
|
#[arg(
|
||||||
long,
|
long,
|
||||||
help = "Only scrape the single page at START_URL, do not crawl for links"
|
help = "Only scrape the single page at START_URL, do not crawl for links"
|
||||||
@@ -79,6 +85,11 @@ async fn main() -> Result<()> {
|
|||||||
|
|
||||||
let url = Url::parse(&start_url)?;
|
let url = Url::parse(&start_url)?;
|
||||||
|
|
||||||
|
let base_host = url
|
||||||
|
.host_str()
|
||||||
|
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?
|
||||||
|
.to_string();
|
||||||
|
|
||||||
let base_domain = url_utils::derive_base_domain(&url)
|
let base_domain = url_utils::derive_base_domain(&url)
|
||||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
|
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
|
||||||
|
|
||||||
@@ -124,7 +135,11 @@ async fn main() -> Result<()> {
|
|||||||
};
|
};
|
||||||
|
|
||||||
log::info!("Scraping: {start_url}");
|
log::info!("Scraping: {start_url}");
|
||||||
log::info!("Base domain: {base_domain}");
|
log::info!("Base host: {base_host}");
|
||||||
|
log::info!(
|
||||||
|
"Subdomain crawling: {}",
|
||||||
|
if args.subdomains { "enabled" } else { "disabled" }
|
||||||
|
);
|
||||||
log::info!("Output dir: {}", output_dir.display());
|
log::info!("Output dir: {}", output_dir.display());
|
||||||
log::info!("Logs dir: {}", logs_dir.display());
|
log::info!("Logs dir: {}", logs_dir.display());
|
||||||
log::info!(
|
log::info!(
|
||||||
@@ -160,7 +175,7 @@ async fn main() -> Result<()> {
|
|||||||
let mut scraper = Scraper::new(
|
let mut scraper = Scraper::new(
|
||||||
output_dir,
|
output_dir,
|
||||||
log_path.clone(),
|
log_path.clone(),
|
||||||
base_domain,
|
base_host,
|
||||||
robots,
|
robots,
|
||||||
convert_docs,
|
convert_docs,
|
||||||
fetcher,
|
fetcher,
|
||||||
@@ -168,6 +183,7 @@ async fn main() -> Result<()> {
|
|||||||
exclude_types,
|
exclude_types,
|
||||||
scope_path,
|
scope_path,
|
||||||
args.single,
|
args.single,
|
||||||
|
args.subdomains,
|
||||||
);
|
);
|
||||||
let (count, errors) = scraper.run(&url).await;
|
let (count, errors) = scraper.run(&url).await;
|
||||||
|
|
||||||
|
|||||||
+9
-5
@@ -45,7 +45,7 @@ pub struct Scraper {
|
|||||||
seen: HashSet<String>,
|
seen: HashSet<String>,
|
||||||
output_dir: std::path::PathBuf,
|
output_dir: std::path::PathBuf,
|
||||||
log_path: std::path::PathBuf,
|
log_path: std::path::PathBuf,
|
||||||
base_domain: String,
|
base_host: String,
|
||||||
robots: RobotsRule,
|
robots: RobotsRule,
|
||||||
doc_stats: DocStats,
|
doc_stats: DocStats,
|
||||||
convert_docs: bool,
|
convert_docs: bool,
|
||||||
@@ -53,6 +53,7 @@ pub struct Scraper {
|
|||||||
exclude_types: Option<HashSet<String>>,
|
exclude_types: Option<HashSet<String>>,
|
||||||
scope_path: Option<String>,
|
scope_path: Option<String>,
|
||||||
single_page: bool,
|
single_page: bool,
|
||||||
|
allow_subdomains: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Scraper {
|
impl Scraper {
|
||||||
@@ -60,7 +61,7 @@ impl Scraper {
|
|||||||
pub fn new(
|
pub fn new(
|
||||||
output_dir: std::path::PathBuf,
|
output_dir: std::path::PathBuf,
|
||||||
log_path: std::path::PathBuf,
|
log_path: std::path::PathBuf,
|
||||||
base_domain: String,
|
base_host: String,
|
||||||
robots: RobotsRule,
|
robots: RobotsRule,
|
||||||
convert_docs: bool,
|
convert_docs: bool,
|
||||||
fetcher: Fetcher,
|
fetcher: Fetcher,
|
||||||
@@ -68,13 +69,14 @@ impl Scraper {
|
|||||||
exclude_types: Option<HashSet<String>>,
|
exclude_types: Option<HashSet<String>>,
|
||||||
scope_path: Option<String>,
|
scope_path: Option<String>,
|
||||||
single_page: bool,
|
single_page: bool,
|
||||||
|
allow_subdomains: bool,
|
||||||
) -> Self {
|
) -> Self {
|
||||||
Self {
|
Self {
|
||||||
fetcher,
|
fetcher,
|
||||||
seen: HashSet::new(),
|
seen: HashSet::new(),
|
||||||
output_dir,
|
output_dir,
|
||||||
log_path,
|
log_path,
|
||||||
base_domain,
|
base_host,
|
||||||
robots,
|
robots,
|
||||||
doc_stats: DocStats::default(),
|
doc_stats: DocStats::default(),
|
||||||
convert_docs,
|
convert_docs,
|
||||||
@@ -82,6 +84,7 @@ impl Scraper {
|
|||||||
exclude_types,
|
exclude_types,
|
||||||
scope_path,
|
scope_path,
|
||||||
single_page,
|
single_page,
|
||||||
|
allow_subdomains,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -111,7 +114,7 @@ impl Scraper {
|
|||||||
html: &str,
|
html: &str,
|
||||||
final_url: &Url,
|
final_url: &Url,
|
||||||
) -> (usize, Vec<Url>) {
|
) -> (usize, Vec<Url>) {
|
||||||
let links = extract_links(html, final_url, &self.base_domain);
|
let links = extract_links(html, final_url, &self.base_host, self.allow_subdomains);
|
||||||
let new_count = links.len();
|
let new_count = links.len();
|
||||||
let new_links: Vec<Url> = links
|
let new_links: Vec<Url> = links
|
||||||
.into_iter()
|
.into_iter()
|
||||||
@@ -296,7 +299,7 @@ mod tests {
|
|||||||
seen: HashSet::new(),
|
seen: HashSet::new(),
|
||||||
output_dir: std::path::PathBuf::new(),
|
output_dir: std::path::PathBuf::new(),
|
||||||
log_path: std::path::PathBuf::new(),
|
log_path: std::path::PathBuf::new(),
|
||||||
base_domain: "example.com".to_string(),
|
base_host: "example.com".to_string(),
|
||||||
robots: RobotsRule {
|
robots: RobotsRule {
|
||||||
allowed: Vec::new(),
|
allowed: Vec::new(),
|
||||||
disallowed: Vec::new(),
|
disallowed: Vec::new(),
|
||||||
@@ -307,6 +310,7 @@ mod tests {
|
|||||||
exclude_types,
|
exclude_types,
|
||||||
scope_path: None,
|
scope_path: None,
|
||||||
single_page: false,
|
single_page: false,
|
||||||
|
allow_subdomains: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+5
-5
@@ -14,13 +14,13 @@ pub fn derive_base_domain(url: &Url) -> Option<String> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
|
pub fn is_internal(url: &Url, base_host: &str, allow_subdomains: bool) -> bool {
|
||||||
match url.host_str() {
|
match url.host_str() {
|
||||||
Some(host) => {
|
Some(host) => {
|
||||||
host == base_domain
|
if host == base_host {
|
||||||
|| host
|
return true;
|
||||||
.strip_suffix(base_domain)
|
}
|
||||||
.is_some_and(|rest| rest.ends_with('.'))
|
allow_subdomains && host.ends_with(&format!(".{base_host}"))
|
||||||
}
|
}
|
||||||
None => false,
|
None => false,
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user