yeah works
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
pub const DELAY_MS: u64 = 100;
|
||||
pub const TIMEOUT_SECS: u64 = 60;
|
||||
pub const MAX_RETRIES: u32 = 3;
|
||||
pub const RETRY_BACKOFF_SECS: u64 = 5;
|
||||
pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
|
||||
@@ -0,0 +1,41 @@
|
||||
use chrono::{DateTime, Utc};
|
||||
use serde::Serialize;
|
||||
use std::fs::OpenOptions;
|
||||
use std::io::Write;
|
||||
use std::path::Path;
|
||||
use url::Url;
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct ErrorEntry {
|
||||
pub timestamp: DateTime<Utc>,
|
||||
pub url: String,
|
||||
pub error_type: String,
|
||||
pub message: String,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub status_code: Option<u16>,
|
||||
}
|
||||
|
||||
pub fn log_error(
|
||||
log_path: &Path,
|
||||
url: &Url,
|
||||
error_type: &str,
|
||||
message: &str,
|
||||
status_code: Option<u16>,
|
||||
) {
|
||||
let entry = ErrorEntry {
|
||||
timestamp: Utc::now(),
|
||||
url: url.to_string(),
|
||||
error_type: error_type.to_string(),
|
||||
message: message.to_string(),
|
||||
status_code,
|
||||
};
|
||||
|
||||
let line = serde_json::to_string(&entry).unwrap_or_default() + "\n";
|
||||
|
||||
if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) {
|
||||
let _ = file.write_all(line.as_bytes());
|
||||
let _ = file.flush();
|
||||
}
|
||||
|
||||
println!(" LOGGED ERROR [{}]: {}", error_type, message);
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
use scraper::{Html, Selector};
|
||||
use std::collections::HashSet;
|
||||
use url::Url;
|
||||
|
||||
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
|
||||
let document = Html::parse_document(html);
|
||||
let selector = Selector::parse("a[href]").unwrap();
|
||||
|
||||
document
|
||||
.select(&selector)
|
||||
.filter_map(|el| el.value().attr("href"))
|
||||
.filter(|href| !href.is_empty())
|
||||
.filter(|href| {
|
||||
!href.starts_with("mailto:")
|
||||
&& !href.starts_with("tel:")
|
||||
&& !href.starts_with("javascript:")
|
||||
})
|
||||
.filter_map(|href| base_url.join(href).ok())
|
||||
.filter(|url| crate::url_utils::is_internal(url, base_domain))
|
||||
.map(|mut url| {
|
||||
url.set_fragment(None);
|
||||
url
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
|
||||
use anyhow::{Result, anyhow};
|
||||
use reqwest::Client;
|
||||
use std::time::Duration;
|
||||
use url::Url;
|
||||
|
||||
pub struct Fetcher {
|
||||
client: Client,
|
||||
}
|
||||
|
||||
impl Fetcher {
|
||||
pub fn new() -> Self {
|
||||
let client = Client::builder()
|
||||
.timeout(Duration::from_secs(TIMEOUT_SECS))
|
||||
.user_agent(USER_AGENT)
|
||||
.redirect(reqwest::redirect::Policy::limited(10))
|
||||
.build()
|
||||
.expect("Failed to create HTTP client");
|
||||
|
||||
Self { client }
|
||||
}
|
||||
|
||||
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
|
||||
for attempt in 1..=MAX_RETRIES {
|
||||
match self.client.get(url.as_str()).send().await {
|
||||
Ok(resp) => {
|
||||
let status = resp.status();
|
||||
let final_url = resp.url().clone();
|
||||
let content_type = resp
|
||||
.headers()
|
||||
.get("content-type")
|
||||
.and_then(|h| h.to_str().ok())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
|
||||
let bytes = match resp.bytes().await {
|
||||
Ok(b) => b.to_vec(),
|
||||
Err(e) => {
|
||||
if attempt < MAX_RETRIES {
|
||||
self.print_retry(attempt, &e.to_string());
|
||||
tokio::time::sleep(Duration::from_secs(
|
||||
RETRY_BACKOFF_SECS * attempt as u64,
|
||||
))
|
||||
.await;
|
||||
continue;
|
||||
}
|
||||
return Err(anyhow!(
|
||||
"Body read failed after {} retries: {}",
|
||||
MAX_RETRIES,
|
||||
e
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
if !status.is_success() {
|
||||
return Err(anyhow!("HTTP {}", status.as_u16()));
|
||||
}
|
||||
|
||||
return Ok(FetchResult {
|
||||
bytes,
|
||||
content_type,
|
||||
final_url,
|
||||
});
|
||||
}
|
||||
Err(e) => {
|
||||
if attempt < MAX_RETRIES {
|
||||
self.print_retry(attempt, &e.to_string());
|
||||
tokio::time::sleep(Duration::from_secs(
|
||||
RETRY_BACKOFF_SECS * attempt as u64,
|
||||
))
|
||||
.await;
|
||||
continue;
|
||||
}
|
||||
return Err(anyhow!(
|
||||
"Request failed after {} retries: {}",
|
||||
MAX_RETRIES,
|
||||
e
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
|
||||
}
|
||||
|
||||
fn print_retry(&self, attempt: u32, error: &str) {
|
||||
let wait = RETRY_BACKOFF_SECS * attempt as u64;
|
||||
println!(
|
||||
" retry {}/{} in {}s... ({})",
|
||||
attempt, MAX_RETRIES, wait, error
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
pub struct FetchResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub content_type: String,
|
||||
pub final_url: Url,
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
// src/main.rs
|
||||
|
||||
mod config;
|
||||
mod error_logger;
|
||||
mod extractor;
|
||||
mod fetcher;
|
||||
mod scraper;
|
||||
mod url_utils;
|
||||
|
||||
use scraper::Scraper;
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use url::Url;
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> anyhow::Result<()> {
|
||||
let args: Vec<String> = std::env::args().collect();
|
||||
|
||||
if args.len() < 2 {
|
||||
eprintln!("Usage: site-scraper <start-url>");
|
||||
eprintln!("Example: site-scraper https://www.logting.fo");
|
||||
std::process::exit(1);
|
||||
}
|
||||
|
||||
let start_url = &args[1];
|
||||
let url = Url::parse(start_url)?;
|
||||
|
||||
let base_domain = url_utils::derive_base_domain(&url)
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
||||
|
||||
let output_dir_name = format!("scraped_{}", base_domain.replace('.', "_"));
|
||||
let output_dir = PathBuf::from(&output_dir_name);
|
||||
let logs_dir = output_dir.join("logs");
|
||||
|
||||
fs::create_dir_all(&logs_dir)?;
|
||||
|
||||
let log_path = logs_dir.join("scrape_errors.jsonl");
|
||||
fs::remove_file(&log_path).ok();
|
||||
|
||||
println!("Scraping: {}", start_url);
|
||||
println!("Base domain: {}", base_domain);
|
||||
println!("Output dir: {}", output_dir.display());
|
||||
println!();
|
||||
|
||||
let scraper = Scraper::new(output_dir, log_path, base_domain);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
+135
@@ -0,0 +1,135 @@
|
||||
// src/scraper.rs
|
||||
|
||||
use crate::error_logger::log_error;
|
||||
use crate::extractor::extract_links;
|
||||
use crate::fetcher::{FetchResult, Fetcher};
|
||||
use crate::url_utils::{normalize_url, url_to_filename};
|
||||
use anyhow::Result;
|
||||
use std::collections::HashSet;
|
||||
use std::collections::VecDeque;
|
||||
use std::fs;
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
use tokio::sync::Mutex;
|
||||
use url::Url;
|
||||
|
||||
pub struct Scraper {
|
||||
fetcher: Fetcher,
|
||||
seen: Arc<Mutex<HashSet<String>>>,
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
pub fn new(
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
) -> Self {
|
||||
Self {
|
||||
fetcher: Fetcher::new(),
|
||||
seen: Arc::new(Mutex::new(HashSet::new())),
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn run(&self, start_url: &Url) -> (usize, usize) {
|
||||
let mut queue: VecDeque<Url> = VecDeque::new();
|
||||
queue.push_back(start_url.clone());
|
||||
|
||||
let mut count = 0usize;
|
||||
let mut error_count = 0usize;
|
||||
|
||||
while let Some(raw_url) = queue.pop_front() {
|
||||
let url = match normalize_url(raw_url.as_str()) {
|
||||
Some(u) => u,
|
||||
None => continue,
|
||||
};
|
||||
|
||||
let url_key = url.as_str().to_string();
|
||||
|
||||
{
|
||||
let mut seen = self.seen.lock().await;
|
||||
if seen.contains(&url_key) {
|
||||
continue;
|
||||
}
|
||||
seen.insert(url_key.clone());
|
||||
}
|
||||
|
||||
println!("[{}] fetching: {}", count, url);
|
||||
|
||||
match self.fetcher.fetch_with_retry(&url).await {
|
||||
Ok(result) => {
|
||||
let is_html = result.content_type.contains("text/html");
|
||||
|
||||
match self.save(&result).await {
|
||||
Ok(()) => {
|
||||
if is_html {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links =
|
||||
extract_links(html, &result.final_url, &self.base_domain);
|
||||
|
||||
// Only queue links not already seen — do NOT insert into seen here.
|
||||
// They get inserted when actually fetched, matching the Python version.
|
||||
let mut new_links = Vec::new();
|
||||
|
||||
{
|
||||
let seen = self.seen.lock().await;
|
||||
for link in &links {
|
||||
if !seen.contains(link.as_str()) {
|
||||
new_links.push(link.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let new_count = new_links.len();
|
||||
|
||||
for link in new_links {
|
||||
queue.push_back(link);
|
||||
}
|
||||
|
||||
println!(" found {} links ({} new)", links.len(), new_count);
|
||||
}
|
||||
} else {
|
||||
println!(" binary: {}", result.content_type);
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(
|
||||
&self.log_path,
|
||||
&result.final_url,
|
||||
"save_error",
|
||||
&e.to_string(),
|
||||
None,
|
||||
);
|
||||
error_count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None);
|
||||
error_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
count += 1;
|
||||
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
|
||||
}
|
||||
|
||||
(count, error_count)
|
||||
}
|
||||
|
||||
async fn save(&self, result: &FetchResult) -> Result<()> {
|
||||
let filename = url_to_filename(&result.final_url, &result.content_type);
|
||||
let filepath = self.output_dir.join(&filename);
|
||||
fs::write(&filepath, &result.bytes)?;
|
||||
println!(
|
||||
" saved -> {}",
|
||||
filepath.file_name().unwrap_or_default().to_string_lossy()
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
// src/url_utils.rs
|
||||
|
||||
use md5::Md5;
|
||||
use sha2::{Digest, Sha256};
|
||||
use url::Url;
|
||||
|
||||
pub fn derive_base_domain(url: &Url) -> Option<String> {
|
||||
let host = url.host_str()?;
|
||||
let parts: Vec<&str> = host.split('.').collect();
|
||||
if parts.len() >= 2 {
|
||||
Some(parts[parts.len() - 2..].join("."))
|
||||
} else {
|
||||
Some(host.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
|
||||
match url.host_str() {
|
||||
Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)),
|
||||
None => false,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn normalize_url(url: &str) -> Option<Url> {
|
||||
let cleaned = url.split('#').next().unwrap_or(url);
|
||||
let cleaned = cleaned.trim_end_matches(['?', '&']);
|
||||
let mut parsed = Url::parse(cleaned).ok()?;
|
||||
parsed.set_fragment(None);
|
||||
Some(parsed)
|
||||
}
|
||||
|
||||
fn get_extension(url: &Url, content_type: &str) -> String {
|
||||
let path = url.path();
|
||||
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||
|
||||
if let Some(dot_pos) = last_segment.rfind('.') {
|
||||
if dot_pos > 0 {
|
||||
return last_segment[dot_pos..].to_lowercase();
|
||||
}
|
||||
}
|
||||
|
||||
let ct = content_type
|
||||
.split(';')
|
||||
.next()
|
||||
.unwrap_or("")
|
||||
.trim()
|
||||
.to_lowercase();
|
||||
|
||||
match ct.as_str() {
|
||||
"text/html" => ".html",
|
||||
"application/pdf" => ".pdf",
|
||||
"application/json" => ".json",
|
||||
"application/xml" | "text/xml" => ".xml",
|
||||
"text/plain" => ".txt",
|
||||
"text/css" => ".css",
|
||||
"application/javascript" | "text/javascript" => ".js",
|
||||
"image/png" => ".png",
|
||||
"image/jpeg" => ".jpg",
|
||||
"image/gif" => ".gif",
|
||||
"image/svg+xml" => ".svg",
|
||||
"image/webp" => ".webp",
|
||||
"application/vnd.ms-excel" => ".xls",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
|
||||
"application/msword" => ".doc",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx",
|
||||
"application/zip" => ".zip",
|
||||
"application/octet-stream" => ".bin",
|
||||
_ => ".bin",
|
||||
}
|
||||
.to_string()
|
||||
}
|
||||
|
||||
pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
||||
let url_str = url.as_str();
|
||||
|
||||
if url_str.len() > 200 {
|
||||
let mut hasher = Sha256::new();
|
||||
hasher.update(url_str.as_bytes());
|
||||
let hash = hex::encode(hasher.finalize());
|
||||
let ext = get_extension(url, content_type);
|
||||
return format!("{}{}", &hash[..16], ext);
|
||||
}
|
||||
|
||||
let host = url
|
||||
.host_str()
|
||||
.unwrap_or("unknown")
|
||||
.replace('.', "_")
|
||||
.replace(':', "_");
|
||||
|
||||
let path = url.path().trim_start_matches('/');
|
||||
let path = if path.is_empty() { "index" } else { path };
|
||||
|
||||
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||
|
||||
let (stem, ext_from_path) = match last_segment.rfind('.') {
|
||||
Some(pos) if pos > 0 => (&last_segment[..pos], Some(&last_segment[pos..])),
|
||||
_ => (last_segment, None),
|
||||
};
|
||||
|
||||
let ext = match ext_from_path {
|
||||
Some(e) => e.to_lowercase(),
|
||||
None => get_extension(url, content_type),
|
||||
};
|
||||
|
||||
let query_suffix = match url.query() {
|
||||
Some(q) if !q.is_empty() => {
|
||||
let mut hasher = Md5::new();
|
||||
hasher.update(q.as_bytes());
|
||||
let hash = hex::encode(hasher.finalize());
|
||||
format!("_{}", &hash[..8])
|
||||
}
|
||||
_ => String::new(),
|
||||
};
|
||||
|
||||
let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext);
|
||||
|
||||
filename
|
||||
.chars()
|
||||
.map(|c| {
|
||||
if c.is_alphanumeric() || "._-".contains(c) {
|
||||
c
|
||||
} else {
|
||||
'_'
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
Reference in New Issue
Block a user