218 lines
6.0 KiB
Rust
218 lines
6.0 KiB
Rust
use reqwest::Client;
|
|
use std::time::Duration;
|
|
use url::Url;
|
|
|
|
pub struct RobotsRule {
|
|
pub allowed: Vec<String>,
|
|
pub disallowed: Vec<String>,
|
|
}
|
|
|
|
impl RobotsRule {
|
|
pub fn is_allowed(&self, path: &str) -> bool {
|
|
let mut matched_disallow_len = 0;
|
|
let mut matched_allow_len = 0;
|
|
|
|
for rule in &self.disallowed {
|
|
if path_matches(rule, path) && rule.len() > matched_disallow_len {
|
|
matched_disallow_len = rule.len();
|
|
}
|
|
}
|
|
|
|
for rule in &self.allowed {
|
|
if path_matches(rule, path) && rule.len() > matched_allow_len {
|
|
matched_allow_len = rule.len();
|
|
}
|
|
}
|
|
|
|
matched_allow_len >= matched_disallow_len
|
|
}
|
|
}
|
|
|
|
fn path_matches(pattern: &str, path: &str) -> bool {
|
|
if pattern.is_empty() {
|
|
return false;
|
|
}
|
|
|
|
if !pattern.contains('*') {
|
|
return path.starts_with(pattern) || path == pattern;
|
|
}
|
|
|
|
let parts: Vec<&str> = pattern.split('*').collect();
|
|
let mut pos = 0;
|
|
|
|
for (i, part) in parts.iter().enumerate() {
|
|
if i == 0 {
|
|
if !path.starts_with(part) {
|
|
return false;
|
|
}
|
|
pos = part.len();
|
|
} else {
|
|
match path[pos..].find(part) {
|
|
Some(idx) => pos += idx + part.len(),
|
|
None => return false,
|
|
}
|
|
}
|
|
}
|
|
|
|
true
|
|
}
|
|
|
|
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
|
|
let host_str = base_url.host_str().unwrap_or("");
|
|
if host_str.is_empty() {
|
|
log::warn!("No host in base URL, skipping robots.txt");
|
|
return RobotsRule {
|
|
allowed: Vec::new(),
|
|
disallowed: Vec::new(),
|
|
};
|
|
}
|
|
|
|
let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str);
|
|
|
|
let mut rule = RobotsRule {
|
|
allowed: Vec::new(),
|
|
disallowed: Vec::new(),
|
|
};
|
|
|
|
match client
|
|
.get(&robots_url)
|
|
.header("User-Agent", user_agent)
|
|
.timeout(Duration::from_secs(30))
|
|
.send()
|
|
.await
|
|
{
|
|
Ok(resp) if resp.status().is_success() => {
|
|
if let Ok(text) = resp.text().await {
|
|
rule = parse_robots_txt(&text, user_agent);
|
|
}
|
|
}
|
|
Ok(resp) => {
|
|
log::warn!(
|
|
"robots.txt returned HTTP {} — assuming no restrictions",
|
|
resp.status()
|
|
);
|
|
}
|
|
Err(e) => {
|
|
log::warn!("Failed to fetch robots.txt ({e}): assuming no restrictions");
|
|
}
|
|
}
|
|
|
|
rule
|
|
}
|
|
|
|
fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
|
|
let target_lower = target_agent.to_lowercase();
|
|
let mut rule = RobotsRule {
|
|
allowed: Vec::new(),
|
|
disallowed: Vec::new(),
|
|
};
|
|
|
|
let mut current_agents: Vec<String> = Vec::new();
|
|
let mut collecting = false;
|
|
|
|
for line in text.lines() {
|
|
let line = line.trim();
|
|
|
|
if line.is_empty() {
|
|
current_agents.clear();
|
|
collecting = false;
|
|
continue;
|
|
}
|
|
|
|
if line.starts_with('#') {
|
|
continue;
|
|
}
|
|
|
|
if let Some(agent) = line
|
|
.strip_prefix("User-agent:")
|
|
.or_else(|| line.strip_prefix("User-Agent:"))
|
|
{
|
|
let agent = agent.trim().to_lowercase();
|
|
current_agents.push(agent.clone());
|
|
|
|
collecting = agent == "*" || agent == target_lower || agent == "web-scraper";
|
|
} else if collecting {
|
|
if let Some(path) = line.strip_prefix("Disallow:") {
|
|
let path = path.trim().to_string();
|
|
if !path.is_empty() {
|
|
rule.disallowed.push(path);
|
|
}
|
|
} else if let Some(path) = line.strip_prefix("Allow:") {
|
|
let path = path.trim().to_string();
|
|
if !path.is_empty() {
|
|
rule.allowed.push(path);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
|
|
log::info!(
|
|
"robots.txt: {} disallow rules, {} allow rules",
|
|
rule.disallowed.len(),
|
|
rule.allowed.len()
|
|
);
|
|
}
|
|
|
|
rule
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
#[test]
|
|
fn test_path_matches_exact() {
|
|
assert!(path_matches("/admin", "/admin"));
|
|
assert!(path_matches("/admin", "/admin/users"));
|
|
assert!(path_matches("/admin", "/administrator"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_path_matches_wildcard() {
|
|
assert!(path_matches("/private/*", "/private/data"));
|
|
assert!(path_matches("/private/*/secret", "/private/x/secret"));
|
|
assert!(!path_matches("/private/*/secret", "/private/secret"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_path_matches_empty_pattern() {
|
|
assert!(!path_matches("", "/anything"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_is_allowed_disallow_takes_precedence_when_longer() {
|
|
let rule = RobotsRule {
|
|
allowed: vec!["/a".into()],
|
|
disallowed: vec!["/a/b".into()],
|
|
};
|
|
assert!(!rule.is_allowed("/a/b"));
|
|
assert!(rule.is_allowed("/a/c"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_is_allowed_allow_overrides_disallow_when_longer() {
|
|
let rule = RobotsRule {
|
|
allowed: vec!["/public/pages".into()],
|
|
disallowed: vec!["/public".into()],
|
|
};
|
|
assert!(rule.is_allowed("/public/pages"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_parse_robots_txt_wildcard_agent() {
|
|
let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n";
|
|
let rule = parse_robots_txt(txt, "web-scraper");
|
|
assert!(rule.disallowed.contains(&"/admin".to_string()));
|
|
assert!(rule.allowed.contains(&"/admin/public".to_string()));
|
|
}
|
|
|
|
#[test]
|
|
fn test_parse_robots_txt_specific_agent() {
|
|
let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n";
|
|
let rule = parse_robots_txt(txt, "web-scraper");
|
|
assert!(rule.disallowed.contains(&"/private".to_string()));
|
|
assert!(!rule.disallowed.contains(&"/".to_string()));
|
|
}
|
|
}
|