Files
web-scraper-rust/src/robots.rs
T
2026-08-15 21:21:43 +01:00

225 lines
6.2 KiB
Rust

use reqwest::Client;
use std::time::Duration;
use url::Url;
pub struct RobotsRule {
pub allowed: Vec<String>,
pub disallowed: Vec<String>,
}
impl RobotsRule {
pub fn is_allowed(&self, path: &str) -> bool {
let mut matched_disallow_len = 0;
let mut matched_allow_len = 0;
for rule in &self.disallowed {
if path_matches(rule, path) && rule.len() > matched_disallow_len {
matched_disallow_len = rule.len();
}
}
for rule in &self.allowed {
if path_matches(rule, path) && rule.len() > matched_allow_len {
matched_allow_len = rule.len();
}
}
matched_allow_len >= matched_disallow_len
}
}
fn path_matches(pattern: &str, path: &str) -> bool {
if pattern.is_empty() {
return false;
}
if !pattern.contains('*') {
return path.starts_with(pattern) || path == pattern;
}
let parts: Vec<&str> = pattern.split('*').collect();
let mut pos = 0;
for (i, part) in parts.iter().enumerate() {
if i == 0 {
if !path.starts_with(part) {
return false;
}
pos = part.len();
} else {
match path[pos..].find(part) {
Some(idx) => pos += idx + part.len(),
None => return false,
}
}
}
true
}
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
let host_str = base_url.host_str().unwrap_or("");
if host_str.is_empty() {
log::warn!("No host in base URL, skipping robots.txt");
return RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
}
let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str);
let mut rule = RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
match client
.get(&robots_url)
.header("User-Agent", user_agent)
.timeout(Duration::from_secs(30))
.send()
.await
{
Ok(resp) if resp.status().is_success() => {
if let Ok(text) = resp.text().await {
rule = parse_robots_txt(&text, user_agent);
}
}
Ok(resp) => {
log::warn!(
"robots.txt returned HTTP {} — assuming no restrictions",
resp.status()
);
}
Err(e) => {
log::warn!(
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
}
}
rule
}
fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
let target_lower = target_agent.to_lowercase();
let mut rule = RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
let mut current_agents: Vec<String> = Vec::new();
let mut collecting = false;
for line in text.lines() {
let line = line.trim();
if line.is_empty() {
current_agents.clear();
collecting = false;
continue;
}
if line.starts_with('#') {
continue;
}
if let Some(agent) = line
.strip_prefix("User-agent:")
.or_else(|| line.strip_prefix("User-Agent:"))
{
let agent = agent.trim().to_lowercase();
current_agents.push(agent.clone());
if agent == "*" || agent == target_lower || agent == "web-scraper" {
collecting = true;
} else {
collecting = false;
}
} else if collecting {
if let Some(path) = line.strip_prefix("Disallow:") {
let path = path.trim().to_string();
if !path.is_empty() {
rule.disallowed.push(path);
}
} else if let Some(path) = line.strip_prefix("Allow:") {
let path = path.trim().to_string();
if !path.is_empty() {
rule.allowed.push(path);
}
}
}
}
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
log::info!(
"robots.txt: {} disallow rules, {} allow rules",
rule.disallowed.len(),
rule.allowed.len()
);
}
rule
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_path_matches_exact() {
assert!(path_matches("/admin", "/admin"));
assert!(path_matches("/admin", "/admin/users"));
assert!(path_matches("/admin", "/administrator"));
}
#[test]
fn test_path_matches_wildcard() {
assert!(path_matches("/private/*", "/private/data"));
assert!(path_matches("/private/*/secret", "/private/x/secret"));
assert!(!path_matches("/private/*/secret", "/private/secret"));
}
#[test]
fn test_path_matches_empty_pattern() {
assert!(!path_matches("", "/anything"));
}
#[test]
fn test_is_allowed_disallow_takes_precedence_when_longer() {
let rule = RobotsRule {
allowed: vec!["/a".into()],
disallowed: vec!["/a/b".into()],
};
assert!(!rule.is_allowed("/a/b"));
assert!(rule.is_allowed("/a/c"));
}
#[test]
fn test_is_allowed_allow_overrides_disallow_when_longer() {
let rule = RobotsRule {
allowed: vec!["/public/pages".into()],
disallowed: vec!["/public".into()],
};
assert!(rule.is_allowed("/public/pages"));
}
#[test]
fn test_parse_robots_txt_wildcard_agent() {
let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/admin".to_string()));
assert!(rule.allowed.contains(&"/admin/public".to_string()));
}
#[test]
fn test_parse_robots_txt_specific_agent() {
let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/private".to_string()));
assert!(!rule.disallowed.contains(&"/".to_string()));
}
}