yeah works

This commit is contained in:
2026-08-13 14:51:28 +01:00
commit c655b939a4
10 changed files with 2515 additions and 0 deletions
+5
View File
@@ -0,0 +1,5 @@
pub const DELAY_MS: u64 = 100;
pub const TIMEOUT_SECS: u64 = 60;
pub const MAX_RETRIES: u32 = 3;
pub const RETRY_BACKOFF_SECS: u64 = 5;
pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
+41
View File
@@ -0,0 +1,41 @@
use chrono::{DateTime, Utc};
use serde::Serialize;
use std::fs::OpenOptions;
use std::io::Write;
use std::path::Path;
use url::Url;
#[derive(Serialize)]
pub struct ErrorEntry {
pub timestamp: DateTime<Utc>,
pub url: String,
pub error_type: String,
pub message: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub status_code: Option<u16>,
}
pub fn log_error(
log_path: &Path,
url: &Url,
error_type: &str,
message: &str,
status_code: Option<u16>,
) {
let entry = ErrorEntry {
timestamp: Utc::now(),
url: url.to_string(),
error_type: error_type.to_string(),
message: message.to_string(),
status_code,
};
let line = serde_json::to_string(&entry).unwrap_or_default() + "\n";
if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) {
let _ = file.write_all(line.as_bytes());
let _ = file.flush();
}
println!(" LOGGED ERROR [{}]: {}", error_type, message);
}
+25
View File
@@ -0,0 +1,25 @@
use scraper::{Html, Selector};
use std::collections::HashSet;
use url::Url;
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
let document = Html::parse_document(html);
let selector = Selector::parse("a[href]").unwrap();
document
.select(&selector)
.filter_map(|el| el.value().attr("href"))
.filter(|href| !href.is_empty())
.filter(|href| {
!href.starts_with("mailto:")
&& !href.starts_with("tel:")
&& !href.starts_with("javascript:")
})
.filter_map(|href| base_url.join(href).ok())
.filter(|url| crate::url_utils::is_internal(url, base_domain))
.map(|mut url| {
url.set_fragment(None);
url
})
.collect()
}
+99
View File
@@ -0,0 +1,99 @@
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
use anyhow::{Result, anyhow};
use reqwest::Client;
use std::time::Duration;
use url::Url;
pub struct Fetcher {
client: Client,
}
impl Fetcher {
pub fn new() -> Self {
let client = Client::builder()
.timeout(Duration::from_secs(TIMEOUT_SECS))
.user_agent(USER_AGENT)
.redirect(reqwest::redirect::Policy::limited(10))
.build()
.expect("Failed to create HTTP client");
Self { client }
}
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
for attempt in 1..=MAX_RETRIES {
match self.client.get(url.as_str()).send().await {
Ok(resp) => {
let status = resp.status();
let final_url = resp.url().clone();
let content_type = resp
.headers()
.get("content-type")
.and_then(|h| h.to_str().ok())
.unwrap_or("")
.to_string();
let bytes = match resp.bytes().await {
Ok(b) => b.to_vec(),
Err(e) => {
if attempt < MAX_RETRIES {
self.print_retry(attempt, &e.to_string());
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
))
.await;
continue;
}
return Err(anyhow!(
"Body read failed after {} retries: {}",
MAX_RETRIES,
e
));
}
};
if !status.is_success() {
return Err(anyhow!("HTTP {}", status.as_u16()));
}
return Ok(FetchResult {
bytes,
content_type,
final_url,
});
}
Err(e) => {
if attempt < MAX_RETRIES {
self.print_retry(attempt, &e.to_string());
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
))
.await;
continue;
}
return Err(anyhow!(
"Request failed after {} retries: {}",
MAX_RETRIES,
e
));
}
}
}
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
}
fn print_retry(&self, attempt: u32, error: &str) {
let wait = RETRY_BACKOFF_SECS * attempt as u64;
println!(
" retry {}/{} in {}s... ({})",
attempt, MAX_RETRIES, wait, error
);
}
}
pub struct FetchResult {
pub bytes: Vec<u8>,
pub content_type: String,
pub final_url: Url,
}
+52
View File
@@ -0,0 +1,52 @@
// src/main.rs
mod config;
mod error_logger;
mod extractor;
mod fetcher;
mod scraper;
mod url_utils;
use scraper::Scraper;
use std::fs;
use std::path::PathBuf;
use url::Url;
#[tokio::main]
async fn main() -> anyhow::Result<()> {
let args: Vec<String> = std::env::args().collect();
if args.len() < 2 {
eprintln!("Usage: site-scraper <start-url>");
eprintln!("Example: site-scraper https://www.logting.fo");
std::process::exit(1);
}
let start_url = &args[1];
let url = Url::parse(start_url)?;
let base_domain = url_utils::derive_base_domain(&url)
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
let output_dir_name = format!("scraped_{}", base_domain.replace('.', "_"));
let output_dir = PathBuf::from(&output_dir_name);
let logs_dir = output_dir.join("logs");
fs::create_dir_all(&logs_dir)?;
let log_path = logs_dir.join("scrape_errors.jsonl");
fs::remove_file(&log_path).ok();
println!("Scraping: {}", start_url);
println!("Base domain: {}", base_domain);
println!("Output dir: {}", output_dir.display());
println!();
let scraper = Scraper::new(output_dir, log_path, base_domain);
let (count, errors) = scraper.run(&url).await;
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
Ok(())
}
+135
View File
@@ -0,0 +1,135 @@
// src/scraper.rs
use crate::error_logger::log_error;
use crate::extractor::extract_links;
use crate::fetcher::{FetchResult, Fetcher};
use crate::url_utils::{normalize_url, url_to_filename};
use anyhow::Result;
use std::collections::HashSet;
use std::collections::VecDeque;
use std::fs;
use std::sync::Arc;
use std::time::Duration;
use tokio::sync::Mutex;
use url::Url;
pub struct Scraper {
fetcher: Fetcher,
seen: Arc<Mutex<HashSet<String>>>,
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_domain: String,
}
impl Scraper {
pub fn new(
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
base_domain: String,
) -> Self {
Self {
fetcher: Fetcher::new(),
seen: Arc::new(Mutex::new(HashSet::new())),
output_dir,
log_path,
base_domain,
}
}
pub async fn run(&self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque<Url> = VecDeque::new();
queue.push_back(start_url.clone());
let mut count = 0usize;
let mut error_count = 0usize;
while let Some(raw_url) = queue.pop_front() {
let url = match normalize_url(raw_url.as_str()) {
Some(u) => u,
None => continue,
};
let url_key = url.as_str().to_string();
{
let mut seen = self.seen.lock().await;
if seen.contains(&url_key) {
continue;
}
seen.insert(url_key.clone());
}
println!("[{}] fetching: {}", count, url);
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
let is_html = result.content_type.contains("text/html");
match self.save(&result).await {
Ok(()) => {
if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links =
extract_links(html, &result.final_url, &self.base_domain);
// Only queue links not already seen — do NOT insert into seen here.
// They get inserted when actually fetched, matching the Python version.
let mut new_links = Vec::new();
{
let seen = self.seen.lock().await;
for link in &links {
if !seen.contains(link.as_str()) {
new_links.push(link.clone());
}
}
}
let new_count = new_links.len();
for link in new_links {
queue.push_back(link);
}
println!(" found {} links ({} new)", links.len(), new_count);
}
} else {
println!(" binary: {}", result.content_type);
}
}
Err(e) => {
log_error(
&self.log_path,
&result.final_url,
"save_error",
&e.to_string(),
None,
);
error_count += 1;
}
}
}
Err(e) => {
log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None);
error_count += 1;
}
}
count += 1;
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
}
(count, error_count)
}
async fn save(&self, result: &FetchResult) -> Result<()> {
let filename = url_to_filename(&result.final_url, &result.content_type);
let filepath = self.output_dir.join(&filename);
fs::write(&filepath, &result.bytes)?;
println!(
" saved -> {}",
filepath.file_name().unwrap_or_default().to_string_lossy()
);
Ok(())
}
}
+127
View File
@@ -0,0 +1,127 @@
// src/url_utils.rs
use md5::Md5;
use sha2::{Digest, Sha256};
use url::Url;
pub fn derive_base_domain(url: &Url) -> Option<String> {
let host = url.host_str()?;
let parts: Vec<&str> = host.split('.').collect();
if parts.len() >= 2 {
Some(parts[parts.len() - 2..].join("."))
} else {
Some(host.to_string())
}
}
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
match url.host_str() {
Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)),
None => false,
}
}
pub fn normalize_url(url: &str) -> Option<Url> {
let cleaned = url.split('#').next().unwrap_or(url);
let cleaned = cleaned.trim_end_matches(['?', '&']);
let mut parsed = Url::parse(cleaned).ok()?;
parsed.set_fragment(None);
Some(parsed)
}
fn get_extension(url: &Url, content_type: &str) -> String {
let path = url.path();
let last_segment = path.rsplit('/').next().unwrap_or(path);
if let Some(dot_pos) = last_segment.rfind('.') {
if dot_pos > 0 {
return last_segment[dot_pos..].to_lowercase();
}
}
let ct = content_type
.split(';')
.next()
.unwrap_or("")
.trim()
.to_lowercase();
match ct.as_str() {
"text/html" => ".html",
"application/pdf" => ".pdf",
"application/json" => ".json",
"application/xml" | "text/xml" => ".xml",
"text/plain" => ".txt",
"text/css" => ".css",
"application/javascript" | "text/javascript" => ".js",
"image/png" => ".png",
"image/jpeg" => ".jpg",
"image/gif" => ".gif",
"image/svg+xml" => ".svg",
"image/webp" => ".webp",
"application/vnd.ms-excel" => ".xls",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
"application/msword" => ".doc",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx",
"application/zip" => ".zip",
"application/octet-stream" => ".bin",
_ => ".bin",
}
.to_string()
}
pub fn url_to_filename(url: &Url, content_type: &str) -> String {
let url_str = url.as_str();
if url_str.len() > 200 {
let mut hasher = Sha256::new();
hasher.update(url_str.as_bytes());
let hash = hex::encode(hasher.finalize());
let ext = get_extension(url, content_type);
return format!("{}{}", &hash[..16], ext);
}
let host = url
.host_str()
.unwrap_or("unknown")
.replace('.', "_")
.replace(':', "_");
let path = url.path().trim_start_matches('/');
let path = if path.is_empty() { "index" } else { path };
let last_segment = path.rsplit('/').next().unwrap_or(path);
let (stem, ext_from_path) = match last_segment.rfind('.') {
Some(pos) if pos > 0 => (&last_segment[..pos], Some(&last_segment[pos..])),
_ => (last_segment, None),
};
let ext = match ext_from_path {
Some(e) => e.to_lowercase(),
None => get_extension(url, content_type),
};
let query_suffix = match url.query() {
Some(q) if !q.is_empty() => {
let mut hasher = Md5::new();
hasher.update(q.as_bytes());
let hash = hex::encode(hasher.finalize());
format!("_{}", &hash[..8])
}
_ => String::new(),
};
let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext);
filename
.chars()
.map(|c| {
if c.is_alphanumeric() || "._-".contains(c) {
c
} else {
'_'
}
})
.collect()
}