This commit is contained in:
2026-08-14 22:18:11 +01:00
parent c655b939a4
commit a60cd5deae
10 changed files with 1136 additions and 35 deletions
+160
View File
@@ -0,0 +1,160 @@
use reqwest::Client;
use std::time::Duration;
use url::Url;
pub struct RobotsRule {
pub allowed: Vec<String>,
pub disallowed: Vec<String>,
}
impl RobotsRule {
pub fn is_allowed(&self, path: &str) -> bool {
let mut matched_disallow_len = 0;
let mut matched_allow_len = 0;
for rule in &self.disallowed {
if path_matches(rule, path) && rule.len() > matched_disallow_len {
matched_disallow_len = rule.len();
}
}
for rule in &self.allowed {
if path_matches(rule, path) && rule.len() > matched_allow_len {
matched_allow_len = rule.len();
}
}
matched_allow_len >= matched_disallow_len
}
}
fn path_matches(pattern: &str, path: &str) -> bool {
if pattern.is_empty() {
return false;
}
if !pattern.contains('*') {
return path.starts_with(pattern) || path == pattern;
}
let parts: Vec<&str> = pattern.split('*').collect();
let mut pos = 0;
for (i, part) in parts.iter().enumerate() {
if i == 0 {
if !path.starts_with(part) {
return false;
}
pos = part.len();
} else {
match path[pos..].find(part) {
Some(idx) => pos += idx + part.len(),
None => return false,
}
}
}
true
}
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
let robots_url = format!(
"{}://{}/robots.txt",
base_url.scheme(),
base_url.host_str().unwrap_or("")
);
let mut rule = RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
match client
.get(&robots_url)
.header("User-Agent", user_agent)
.timeout(Duration::from_secs(30))
.send()
.await
{
Ok(resp) if resp.status().is_success() => {
if let Ok(text) = resp.text().await {
rule = parse_robots_txt(&text, user_agent);
}
}
Ok(resp) => {
println!(
"robots.txt returned HTTP {} — assuming no restrictions",
resp.status()
);
}
Err(e) => {
println!(
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
}
}
rule
}
fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
let target_lower = target_agent.to_lowercase();
let mut rule = RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
let mut current_agents: Vec<String> = Vec::new();
let mut collecting = false;
for line in text.lines() {
let line = line.trim();
if line.is_empty() {
current_agents.clear();
collecting = false;
continue;
}
if line.starts_with('#') {
continue;
}
if let Some(agent) = line
.strip_prefix("User-agent:")
.or_else(|| line.strip_prefix("User-Agent:"))
{
let agent = agent.trim().to_lowercase();
current_agents.push(agent.clone());
if agent == "*" || agent == target_lower || agent == "web-scraper" {
collecting = true;
} else {
collecting = false;
}
} else if collecting {
if let Some(path) = line.strip_prefix("Disallow:") {
let path = path.trim().to_string();
if !path.is_empty() {
rule.disallowed.push(path);
}
} else if let Some(path) = line.strip_prefix("Allow:") {
let path = path.trim().to_string();
if !path.is_empty() {
rule.allowed.push(path);
}
}
}
}
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
println!(
"robots.txt: {} disallow rules, {} allow rules",
rule.disallowed.len(),
rule.allowed.len()
);
}
rule
}