From d4d220b18b267ae8979f769da73648c17bdae81a Mon Sep 17 00:00:00 2001 From: Bartal Laearsson Date: Tue, 25 Aug 2026 12:29:57 +0100 Subject: [PATCH] scrape ms flag --- src/config.rs | 3 --- src/main.rs | 10 ++++++++++ src/scraper.rs | 8 ++++++-- 3 files changed, 16 insertions(+), 5 deletions(-) diff --git a/src/config.rs b/src/config.rs index ef82dcc..8bce36b 100644 --- a/src/config.rs +++ b/src/config.rs @@ -1,7 +1,4 @@ -pub const DELAY_MS: u64 = 1000; pub const TIMEOUT_SECS: u64 = 60; pub const MAX_RETRIES: u32 = 3; pub const RETRY_BACKOFF_SECS: u64 = 5; pub const USER_AGENT: &str = "web-scraper/1.0 (research)"; - -pub const RATE_LIMIT_MS: u64 = DELAY_MS; diff --git a/src/main.rs b/src/main.rs index 32c94e9..1054305 100644 --- a/src/main.rs +++ b/src/main.rs @@ -30,6 +30,13 @@ struct Args { #[arg(help = "Starting URL to scrape")] start_url: String, + #[arg( + long, + default_value = "1000", + help = "Delay in milliseconds between requests (default: 1000)" + )] + delay_ms: u64, + #[arg( long, help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)" @@ -82,6 +89,7 @@ async fn main() -> Result<()> { let start_url = args.start_url; let convert_docs = !args.no_doc_conversion; + let delay_ms = args.delay_ms; let url = Url::parse(&start_url)?; @@ -135,6 +143,7 @@ async fn main() -> Result<()> { }; log::info!("Scraping: {start_url}"); + log::info!("Request delay: {delay_ms} ms"); log::info!("Base host: {base_host}"); log::info!( "Subdomain crawling: {}", @@ -184,6 +193,7 @@ async fn main() -> Result<()> { scope_path, args.single, args.subdomains, + delay_ms, ); let (count, errors) = scraper.run(&url).await; diff --git a/src/scraper.rs b/src/scraper.rs index 6718651..f5b97e5 100644 --- a/src/scraper.rs +++ b/src/scraper.rs @@ -54,6 +54,7 @@ pub struct Scraper { scope_path: Option, single_page: bool, allow_subdomains: bool, + delay_ms: u64, } impl Scraper { @@ -70,6 +71,7 @@ impl Scraper { scope_path: Option, single_page: bool, allow_subdomains: bool, + delay_ms: u64, ) -> Self { Self { fetcher, @@ -85,6 +87,7 @@ impl Scraper { scope_path, single_page, allow_subdomains, + delay_ms, } } @@ -213,7 +216,7 @@ impl Scraper { } count += 1; - tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await; + tokio::time::sleep(Duration::from_millis(self.delay_ms)).await; } if self.doc_stats.converted > 0 || self.doc_stats.raw > 0 || self.doc_stats.errors > 0 { @@ -300,6 +303,8 @@ mod tests { output_dir: std::path::PathBuf::new(), log_path: std::path::PathBuf::new(), base_host: "example.com".to_string(), + allow_subdomains: false, + delay_ms: 1000, robots: RobotsRule { allowed: Vec::new(), disallowed: Vec::new(), @@ -310,7 +315,6 @@ mod tests { exclude_types, scope_path: None, single_page: false, - allow_subdomains: false, } }