qa and review

This commit is contained in:
2026-08-15 21:21:43 +01:00
parent a60cd5deae
commit 3c7043562e
13 changed files with 753 additions and 297 deletions
+72 -8
View File
@@ -58,11 +58,16 @@ fn path_matches(pattern: &str, path: &str) -> bool {
}
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
let robots_url = format!(
"{}://{}/robots.txt",
base_url.scheme(),
base_url.host_str().unwrap_or("")
);
let host_str = base_url.host_str().unwrap_or("");
if host_str.is_empty() {
log::warn!("No host in base URL, skipping robots.txt");
return RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
}
let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str);
let mut rule = RobotsRule {
allowed: Vec::new(),
@@ -82,13 +87,13 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
}
}
Ok(resp) => {
println!(
log::warn!(
"robots.txt returned HTTP {} — assuming no restrictions",
resp.status()
);
}
Err(e) => {
println!(
log::warn!(
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
@@ -149,7 +154,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
}
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
println!(
log::info!(
"robots.txt: {} disallow rules, {} allow rules",
rule.disallowed.len(),
rule.allowed.len()
@@ -158,3 +163,62 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
rule
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_path_matches_exact() {
assert!(path_matches("/admin", "/admin"));
assert!(path_matches("/admin", "/admin/users"));
assert!(path_matches("/admin", "/administrator"));
}
#[test]
fn test_path_matches_wildcard() {
assert!(path_matches("/private/*", "/private/data"));
assert!(path_matches("/private/*/secret", "/private/x/secret"));
assert!(!path_matches("/private/*/secret", "/private/secret"));
}
#[test]
fn test_path_matches_empty_pattern() {
assert!(!path_matches("", "/anything"));
}
#[test]
fn test_is_allowed_disallow_takes_precedence_when_longer() {
let rule = RobotsRule {
allowed: vec!["/a".into()],
disallowed: vec!["/a/b".into()],
};
assert!(!rule.is_allowed("/a/b"));
assert!(rule.is_allowed("/a/c"));
}
#[test]
fn test_is_allowed_allow_overrides_disallow_when_longer() {
let rule = RobotsRule {
allowed: vec!["/public/pages".into()],
disallowed: vec!["/public".into()],
};
assert!(rule.is_allowed("/public/pages"));
}
#[test]
fn test_parse_robots_txt_wildcard_agent() {
let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/admin".to_string()));
assert!(rule.allowed.contains(&"/admin/public".to_string()));
}
#[test]
fn test_parse_robots_txt_specific_agent() {
let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/private".to_string()));
assert!(!rule.disallowed.contains(&"/".to_string()));
}
}