scrape ms flag

This commit is contained in:
2026-08-25 12:29:57 +01:00
parent 7837b08c86
commit d4d220b18b
3 changed files with 16 additions and 5 deletions
-3
View File
@@ -1,7 +1,4 @@
pub const DELAY_MS: u64 = 1000;
pub const TIMEOUT_SECS: u64 = 60; pub const TIMEOUT_SECS: u64 = 60;
pub const MAX_RETRIES: u32 = 3; pub const MAX_RETRIES: u32 = 3;
pub const RETRY_BACKOFF_SECS: u64 = 5; pub const RETRY_BACKOFF_SECS: u64 = 5;
pub const USER_AGENT: &str = "web-scraper/1.0 (research)"; pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
pub const RATE_LIMIT_MS: u64 = DELAY_MS;
+10
View File
@@ -30,6 +30,13 @@ struct Args {
#[arg(help = "Starting URL to scrape")] #[arg(help = "Starting URL to scrape")]
start_url: String, start_url: String,
#[arg(
long,
default_value = "1000",
help = "Delay in milliseconds between requests (default: 1000)"
)]
delay_ms: u64,
#[arg( #[arg(
long, long,
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)" help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
@@ -82,6 +89,7 @@ async fn main() -> Result<()> {
let start_url = args.start_url; let start_url = args.start_url;
let convert_docs = !args.no_doc_conversion; let convert_docs = !args.no_doc_conversion;
let delay_ms = args.delay_ms;
let url = Url::parse(&start_url)?; let url = Url::parse(&start_url)?;
@@ -135,6 +143,7 @@ async fn main() -> Result<()> {
}; };
log::info!("Scraping: {start_url}"); log::info!("Scraping: {start_url}");
log::info!("Request delay: {delay_ms} ms");
log::info!("Base host: {base_host}"); log::info!("Base host: {base_host}");
log::info!( log::info!(
"Subdomain crawling: {}", "Subdomain crawling: {}",
@@ -184,6 +193,7 @@ async fn main() -> Result<()> {
scope_path, scope_path,
args.single, args.single,
args.subdomains, args.subdomains,
delay_ms,
); );
let (count, errors) = scraper.run(&url).await; let (count, errors) = scraper.run(&url).await;
+6 -2
View File
@@ -54,6 +54,7 @@ pub struct Scraper {
scope_path: Option<String>, scope_path: Option<String>,
single_page: bool, single_page: bool,
allow_subdomains: bool, allow_subdomains: bool,
delay_ms: u64,
} }
impl Scraper { impl Scraper {
@@ -70,6 +71,7 @@ impl Scraper {
scope_path: Option<String>, scope_path: Option<String>,
single_page: bool, single_page: bool,
allow_subdomains: bool, allow_subdomains: bool,
delay_ms: u64,
) -> Self { ) -> Self {
Self { Self {
fetcher, fetcher,
@@ -85,6 +87,7 @@ impl Scraper {
scope_path, scope_path,
single_page, single_page,
allow_subdomains, allow_subdomains,
delay_ms,
} }
} }
@@ -213,7 +216,7 @@ impl Scraper {
} }
count += 1; count += 1;
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await; tokio::time::sleep(Duration::from_millis(self.delay_ms)).await;
} }
if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 { if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 {
@@ -300,6 +303,8 @@ mod tests {
output_dir: std::path::PathBuf::new(), output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(),
base_host: "example.com".to_string(), base_host: "example.com".to_string(),
allow_subdomains: false,
delay_ms: 1000,
robots: RobotsRule { robots: RobotsRule {
allowed: Vec::new(), allowed: Vec::new(),
disallowed: Vec::new(), disallowed: Vec::new(),
@@ -310,7 +315,6 @@ mod tests {
exclude_types, exclude_types,
scope_path: None, scope_path: None,
single_page: false, single_page: false,
allow_subdomains: false,
} }
} }