use reqwest::Client; use std::time::Duration; use url::Url; pub struct RobotsRule { pub allowed: Vec, pub disallowed: Vec, } impl RobotsRule { pub fn is_allowed(&self, path: &str) -> bool { let mut matched_disallow_len = 0; let mut matched_allow_len = 0; for rule in &self.disallowed { if path_matches(rule, path) && rule.len() > matched_disallow_len { matched_disallow_len = rule.len(); } } for rule in &self.allowed { if path_matches(rule, path) && rule.len() > matched_allow_len { matched_allow_len = rule.len(); } } matched_allow_len >= matched_disallow_len } } fn path_matches(pattern: &str, path: &str) -> bool { if pattern.is_empty() { return false; } if !pattern.contains('*') { return path.starts_with(pattern) || path == pattern; } let parts: Vec<&str> = pattern.split('*').collect(); let mut pos = 0; for (i, part) in parts.iter().enumerate() { if i == 0 { if !path.starts_with(part) { return false; } pos = part.len(); } else { match path[pos..].find(part) { Some(idx) => pos += idx + part.len(), None => return false, } } } true } pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule { let host_str = base_url.host_str().unwrap_or(""); if host_str.is_empty() { log::warn!("No host in base URL, skipping robots.txt"); return RobotsRule { allowed: Vec::new(), disallowed: Vec::new(), }; } let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str); let mut rule = RobotsRule { allowed: Vec::new(), disallowed: Vec::new(), }; match client .get(&robots_url) .header("User-Agent", user_agent) .timeout(Duration::from_secs(30)) .send() .await { Ok(resp) if resp.status().is_success() => { if let Ok(text) = resp.text().await { rule = parse_robots_txt(&text, user_agent); } } Ok(resp) => { log::warn!( "robots.txt returned HTTP {} — assuming no restrictions", resp.status() ); } Err(e) => { log::warn!( "Failed to fetch robots.txt ({}): assuming no restrictions", e ); } } rule } fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule { let target_lower = target_agent.to_lowercase(); let mut rule = RobotsRule { allowed: Vec::new(), disallowed: Vec::new(), }; let mut current_agents: Vec = Vec::new(); let mut collecting = false; for line in text.lines() { let line = line.trim(); if line.is_empty() { current_agents.clear(); collecting = false; continue; } if line.starts_with('#') { continue; } if let Some(agent) = line .strip_prefix("User-agent:") .or_else(|| line.strip_prefix("User-Agent:")) { let agent = agent.trim().to_lowercase(); current_agents.push(agent.clone()); if agent == "*" || agent == target_lower || agent == "web-scraper" { collecting = true; } else { collecting = false; } } else if collecting { if let Some(path) = line.strip_prefix("Disallow:") { let path = path.trim().to_string(); if !path.is_empty() { rule.disallowed.push(path); } } else if let Some(path) = line.strip_prefix("Allow:") { let path = path.trim().to_string(); if !path.is_empty() { rule.allowed.push(path); } } } } if !rule.disallowed.is_empty() || !rule.allowed.is_empty() { log::info!( "robots.txt: {} disallow rules, {} allow rules", rule.disallowed.len(), rule.allowed.len() ); } rule } #[cfg(test)] mod tests { use super::*; #[test] fn test_path_matches_exact() { assert!(path_matches("/admin", "/admin")); assert!(path_matches("/admin", "/admin/users")); assert!(path_matches("/admin", "/administrator")); } #[test] fn test_path_matches_wildcard() { assert!(path_matches("/private/*", "/private/data")); assert!(path_matches("/private/*/secret", "/private/x/secret")); assert!(!path_matches("/private/*/secret", "/private/secret")); } #[test] fn test_path_matches_empty_pattern() { assert!(!path_matches("", "/anything")); } #[test] fn test_is_allowed_disallow_takes_precedence_when_longer() { let rule = RobotsRule { allowed: vec!["/a".into()], disallowed: vec!["/a/b".into()], }; assert!(!rule.is_allowed("/a/b")); assert!(rule.is_allowed("/a/c")); } #[test] fn test_is_allowed_allow_overrides_disallow_when_longer() { let rule = RobotsRule { allowed: vec!["/public/pages".into()], disallowed: vec!["/public".into()], }; assert!(rule.is_allowed("/public/pages")); } #[test] fn test_parse_robots_txt_wildcard_agent() { let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n"; let rule = parse_robots_txt(txt, "web-scraper"); assert!(rule.disallowed.contains(&"/admin".to_string())); assert!(rule.allowed.contains(&"/admin/public".to_string())); } #[test] fn test_parse_robots_txt_specific_agent() { let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n"; let rule = parse_robots_txt(txt, "web-scraper"); assert!(rule.disallowed.contains(&"/private".to_string())); assert!(!rule.disallowed.contains(&"/".to_string())); } }