yeah works
This commit is contained in:
@@ -0,0 +1,2 @@
|
|||||||
|
/target
|
||||||
|
scraped*
|
||||||
Generated
+2012
File diff suppressed because it is too large
Load Diff
+17
@@ -0,0 +1,17 @@
|
|||||||
|
[package]
|
||||||
|
name = "rs-scraper"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2024"
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
anyhow = "1"
|
||||||
|
chrono = { version = "0.4", features = ["serde"] }
|
||||||
|
hex = "0.4"
|
||||||
|
md-5 = "0.10"
|
||||||
|
reqwest = "0.12"
|
||||||
|
scraper = "0.22"
|
||||||
|
serde = { version = "1", features = ["derive"] }
|
||||||
|
serde_json = "1"
|
||||||
|
sha2 = "0.10"
|
||||||
|
tokio = { version = "1", features = ["full"] }
|
||||||
|
url = "2"
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
pub const DELAY_MS: u64 = 100;
|
||||||
|
pub const TIMEOUT_SECS: u64 = 60;
|
||||||
|
pub const MAX_RETRIES: u32 = 3;
|
||||||
|
pub const RETRY_BACKOFF_SECS: u64 = 5;
|
||||||
|
pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
use chrono::{DateTime, Utc};
|
||||||
|
use serde::Serialize;
|
||||||
|
use std::fs::OpenOptions;
|
||||||
|
use std::io::Write;
|
||||||
|
use std::path::Path;
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
#[derive(Serialize)]
|
||||||
|
pub struct ErrorEntry {
|
||||||
|
pub timestamp: DateTime<Utc>,
|
||||||
|
pub url: String,
|
||||||
|
pub error_type: String,
|
||||||
|
pub message: String,
|
||||||
|
#[serde(skip_serializing_if = "Option::is_none")]
|
||||||
|
pub status_code: Option<u16>,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn log_error(
|
||||||
|
log_path: &Path,
|
||||||
|
url: &Url,
|
||||||
|
error_type: &str,
|
||||||
|
message: &str,
|
||||||
|
status_code: Option<u16>,
|
||||||
|
) {
|
||||||
|
let entry = ErrorEntry {
|
||||||
|
timestamp: Utc::now(),
|
||||||
|
url: url.to_string(),
|
||||||
|
error_type: error_type.to_string(),
|
||||||
|
message: message.to_string(),
|
||||||
|
status_code,
|
||||||
|
};
|
||||||
|
|
||||||
|
let line = serde_json::to_string(&entry).unwrap_or_default() + "\n";
|
||||||
|
|
||||||
|
if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) {
|
||||||
|
let _ = file.write_all(line.as_bytes());
|
||||||
|
let _ = file.flush();
|
||||||
|
}
|
||||||
|
|
||||||
|
println!(" LOGGED ERROR [{}]: {}", error_type, message);
|
||||||
|
}
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
use scraper::{Html, Selector};
|
||||||
|
use std::collections::HashSet;
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
|
||||||
|
let document = Html::parse_document(html);
|
||||||
|
let selector = Selector::parse("a[href]").unwrap();
|
||||||
|
|
||||||
|
document
|
||||||
|
.select(&selector)
|
||||||
|
.filter_map(|el| el.value().attr("href"))
|
||||||
|
.filter(|href| !href.is_empty())
|
||||||
|
.filter(|href| {
|
||||||
|
!href.starts_with("mailto:")
|
||||||
|
&& !href.starts_with("tel:")
|
||||||
|
&& !href.starts_with("javascript:")
|
||||||
|
})
|
||||||
|
.filter_map(|href| base_url.join(href).ok())
|
||||||
|
.filter(|url| crate::url_utils::is_internal(url, base_domain))
|
||||||
|
.map(|mut url| {
|
||||||
|
url.set_fragment(None);
|
||||||
|
url
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
|
||||||
|
use anyhow::{Result, anyhow};
|
||||||
|
use reqwest::Client;
|
||||||
|
use std::time::Duration;
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
pub struct Fetcher {
|
||||||
|
client: Client,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Fetcher {
|
||||||
|
pub fn new() -> Self {
|
||||||
|
let client = Client::builder()
|
||||||
|
.timeout(Duration::from_secs(TIMEOUT_SECS))
|
||||||
|
.user_agent(USER_AGENT)
|
||||||
|
.redirect(reqwest::redirect::Policy::limited(10))
|
||||||
|
.build()
|
||||||
|
.expect("Failed to create HTTP client");
|
||||||
|
|
||||||
|
Self { client }
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
|
||||||
|
for attempt in 1..=MAX_RETRIES {
|
||||||
|
match self.client.get(url.as_str()).send().await {
|
||||||
|
Ok(resp) => {
|
||||||
|
let status = resp.status();
|
||||||
|
let final_url = resp.url().clone();
|
||||||
|
let content_type = resp
|
||||||
|
.headers()
|
||||||
|
.get("content-type")
|
||||||
|
.and_then(|h| h.to_str().ok())
|
||||||
|
.unwrap_or("")
|
||||||
|
.to_string();
|
||||||
|
|
||||||
|
let bytes = match resp.bytes().await {
|
||||||
|
Ok(b) => b.to_vec(),
|
||||||
|
Err(e) => {
|
||||||
|
if attempt < MAX_RETRIES {
|
||||||
|
self.print_retry(attempt, &e.to_string());
|
||||||
|
tokio::time::sleep(Duration::from_secs(
|
||||||
|
RETRY_BACKOFF_SECS * attempt as u64,
|
||||||
|
))
|
||||||
|
.await;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
return Err(anyhow!(
|
||||||
|
"Body read failed after {} retries: {}",
|
||||||
|
MAX_RETRIES,
|
||||||
|
e
|
||||||
|
));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
if !status.is_success() {
|
||||||
|
return Err(anyhow!("HTTP {}", status.as_u16()));
|
||||||
|
}
|
||||||
|
|
||||||
|
return Ok(FetchResult {
|
||||||
|
bytes,
|
||||||
|
content_type,
|
||||||
|
final_url,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
if attempt < MAX_RETRIES {
|
||||||
|
self.print_retry(attempt, &e.to_string());
|
||||||
|
tokio::time::sleep(Duration::from_secs(
|
||||||
|
RETRY_BACKOFF_SECS * attempt as u64,
|
||||||
|
))
|
||||||
|
.await;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
return Err(anyhow!(
|
||||||
|
"Request failed after {} retries: {}",
|
||||||
|
MAX_RETRIES,
|
||||||
|
e
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn print_retry(&self, attempt: u32, error: &str) {
|
||||||
|
let wait = RETRY_BACKOFF_SECS * attempt as u64;
|
||||||
|
println!(
|
||||||
|
" retry {}/{} in {}s... ({})",
|
||||||
|
attempt, MAX_RETRIES, wait, error
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct FetchResult {
|
||||||
|
pub bytes: Vec<u8>,
|
||||||
|
pub content_type: String,
|
||||||
|
pub final_url: Url,
|
||||||
|
}
|
||||||
+52
@@ -0,0 +1,52 @@
|
|||||||
|
// src/main.rs
|
||||||
|
|
||||||
|
mod config;
|
||||||
|
mod error_logger;
|
||||||
|
mod extractor;
|
||||||
|
mod fetcher;
|
||||||
|
mod scraper;
|
||||||
|
mod url_utils;
|
||||||
|
|
||||||
|
use scraper::Scraper;
|
||||||
|
use std::fs;
|
||||||
|
use std::path::PathBuf;
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
#[tokio::main]
|
||||||
|
async fn main() -> anyhow::Result<()> {
|
||||||
|
let args: Vec<String> = std::env::args().collect();
|
||||||
|
|
||||||
|
if args.len() < 2 {
|
||||||
|
eprintln!("Usage: site-scraper <start-url>");
|
||||||
|
eprintln!("Example: site-scraper https://www.logting.fo");
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
|
||||||
|
let start_url = &args[1];
|
||||||
|
let url = Url::parse(start_url)?;
|
||||||
|
|
||||||
|
let base_domain = url_utils::derive_base_domain(&url)
|
||||||
|
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
||||||
|
|
||||||
|
let output_dir_name = format!("scraped_{}", base_domain.replace('.', "_"));
|
||||||
|
let output_dir = PathBuf::from(&output_dir_name);
|
||||||
|
let logs_dir = output_dir.join("logs");
|
||||||
|
|
||||||
|
fs::create_dir_all(&logs_dir)?;
|
||||||
|
|
||||||
|
let log_path = logs_dir.join("scrape_errors.jsonl");
|
||||||
|
fs::remove_file(&log_path).ok();
|
||||||
|
|
||||||
|
println!("Scraping: {}", start_url);
|
||||||
|
println!("Base domain: {}", base_domain);
|
||||||
|
println!("Output dir: {}", output_dir.display());
|
||||||
|
println!();
|
||||||
|
|
||||||
|
let scraper = Scraper::new(output_dir, log_path, base_domain);
|
||||||
|
let (count, errors) = scraper.run(&url).await;
|
||||||
|
|
||||||
|
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||||
|
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
+135
@@ -0,0 +1,135 @@
|
|||||||
|
// src/scraper.rs
|
||||||
|
|
||||||
|
use crate::error_logger::log_error;
|
||||||
|
use crate::extractor::extract_links;
|
||||||
|
use crate::fetcher::{FetchResult, Fetcher};
|
||||||
|
use crate::url_utils::{normalize_url, url_to_filename};
|
||||||
|
use anyhow::Result;
|
||||||
|
use std::collections::HashSet;
|
||||||
|
use std::collections::VecDeque;
|
||||||
|
use std::fs;
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::Duration;
|
||||||
|
use tokio::sync::Mutex;
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
pub struct Scraper {
|
||||||
|
fetcher: Fetcher,
|
||||||
|
seen: Arc<Mutex<HashSet<String>>>,
|
||||||
|
output_dir: std::path::PathBuf,
|
||||||
|
log_path: std::path::PathBuf,
|
||||||
|
base_domain: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Scraper {
|
||||||
|
pub fn new(
|
||||||
|
output_dir: std::path::PathBuf,
|
||||||
|
log_path: std::path::PathBuf,
|
||||||
|
base_domain: String,
|
||||||
|
) -> Self {
|
||||||
|
Self {
|
||||||
|
fetcher: Fetcher::new(),
|
||||||
|
seen: Arc::new(Mutex::new(HashSet::new())),
|
||||||
|
output_dir,
|
||||||
|
log_path,
|
||||||
|
base_domain,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn run(&self, start_url: &Url) -> (usize, usize) {
|
||||||
|
let mut queue: VecDeque<Url> = VecDeque::new();
|
||||||
|
queue.push_back(start_url.clone());
|
||||||
|
|
||||||
|
let mut count = 0usize;
|
||||||
|
let mut error_count = 0usize;
|
||||||
|
|
||||||
|
while let Some(raw_url) = queue.pop_front() {
|
||||||
|
let url = match normalize_url(raw_url.as_str()) {
|
||||||
|
Some(u) => u,
|
||||||
|
None => continue,
|
||||||
|
};
|
||||||
|
|
||||||
|
let url_key = url.as_str().to_string();
|
||||||
|
|
||||||
|
{
|
||||||
|
let mut seen = self.seen.lock().await;
|
||||||
|
if seen.contains(&url_key) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
seen.insert(url_key.clone());
|
||||||
|
}
|
||||||
|
|
||||||
|
println!("[{}] fetching: {}", count, url);
|
||||||
|
|
||||||
|
match self.fetcher.fetch_with_retry(&url).await {
|
||||||
|
Ok(result) => {
|
||||||
|
let is_html = result.content_type.contains("text/html");
|
||||||
|
|
||||||
|
match self.save(&result).await {
|
||||||
|
Ok(()) => {
|
||||||
|
if is_html {
|
||||||
|
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||||
|
let links =
|
||||||
|
extract_links(html, &result.final_url, &self.base_domain);
|
||||||
|
|
||||||
|
// Only queue links not already seen — do NOT insert into seen here.
|
||||||
|
// They get inserted when actually fetched, matching the Python version.
|
||||||
|
let mut new_links = Vec::new();
|
||||||
|
|
||||||
|
{
|
||||||
|
let seen = self.seen.lock().await;
|
||||||
|
for link in &links {
|
||||||
|
if !seen.contains(link.as_str()) {
|
||||||
|
new_links.push(link.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let new_count = new_links.len();
|
||||||
|
|
||||||
|
for link in new_links {
|
||||||
|
queue.push_back(link);
|
||||||
|
}
|
||||||
|
|
||||||
|
println!(" found {} links ({} new)", links.len(), new_count);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
println!(" binary: {}", result.content_type);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
log_error(
|
||||||
|
&self.log_path,
|
||||||
|
&result.final_url,
|
||||||
|
"save_error",
|
||||||
|
&e.to_string(),
|
||||||
|
None,
|
||||||
|
);
|
||||||
|
error_count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None);
|
||||||
|
error_count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
count += 1;
|
||||||
|
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
|
||||||
|
}
|
||||||
|
|
||||||
|
(count, error_count)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn save(&self, result: &FetchResult) -> Result<()> {
|
||||||
|
let filename = url_to_filename(&result.final_url, &result.content_type);
|
||||||
|
let filepath = self.output_dir.join(&filename);
|
||||||
|
fs::write(&filepath, &result.bytes)?;
|
||||||
|
println!(
|
||||||
|
" saved -> {}",
|
||||||
|
filepath.file_name().unwrap_or_default().to_string_lossy()
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,127 @@
|
|||||||
|
// src/url_utils.rs
|
||||||
|
|
||||||
|
use md5::Md5;
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
pub fn derive_base_domain(url: &Url) -> Option<String> {
|
||||||
|
let host = url.host_str()?;
|
||||||
|
let parts: Vec<&str> = host.split('.').collect();
|
||||||
|
if parts.len() >= 2 {
|
||||||
|
Some(parts[parts.len() - 2..].join("."))
|
||||||
|
} else {
|
||||||
|
Some(host.to_string())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
|
||||||
|
match url.host_str() {
|
||||||
|
Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)),
|
||||||
|
None => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn normalize_url(url: &str) -> Option<Url> {
|
||||||
|
let cleaned = url.split('#').next().unwrap_or(url);
|
||||||
|
let cleaned = cleaned.trim_end_matches(['?', '&']);
|
||||||
|
let mut parsed = Url::parse(cleaned).ok()?;
|
||||||
|
parsed.set_fragment(None);
|
||||||
|
Some(parsed)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_extension(url: &Url, content_type: &str) -> String {
|
||||||
|
let path = url.path();
|
||||||
|
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||||
|
|
||||||
|
if let Some(dot_pos) = last_segment.rfind('.') {
|
||||||
|
if dot_pos > 0 {
|
||||||
|
return last_segment[dot_pos..].to_lowercase();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let ct = content_type
|
||||||
|
.split(';')
|
||||||
|
.next()
|
||||||
|
.unwrap_or("")
|
||||||
|
.trim()
|
||||||
|
.to_lowercase();
|
||||||
|
|
||||||
|
match ct.as_str() {
|
||||||
|
"text/html" => ".html",
|
||||||
|
"application/pdf" => ".pdf",
|
||||||
|
"application/json" => ".json",
|
||||||
|
"application/xml" | "text/xml" => ".xml",
|
||||||
|
"text/plain" => ".txt",
|
||||||
|
"text/css" => ".css",
|
||||||
|
"application/javascript" | "text/javascript" => ".js",
|
||||||
|
"image/png" => ".png",
|
||||||
|
"image/jpeg" => ".jpg",
|
||||||
|
"image/gif" => ".gif",
|
||||||
|
"image/svg+xml" => ".svg",
|
||||||
|
"image/webp" => ".webp",
|
||||||
|
"application/vnd.ms-excel" => ".xls",
|
||||||
|
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
|
||||||
|
"application/msword" => ".doc",
|
||||||
|
"application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx",
|
||||||
|
"application/zip" => ".zip",
|
||||||
|
"application/octet-stream" => ".bin",
|
||||||
|
_ => ".bin",
|
||||||
|
}
|
||||||
|
.to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
||||||
|
let url_str = url.as_str();
|
||||||
|
|
||||||
|
if url_str.len() > 200 {
|
||||||
|
let mut hasher = Sha256::new();
|
||||||
|
hasher.update(url_str.as_bytes());
|
||||||
|
let hash = hex::encode(hasher.finalize());
|
||||||
|
let ext = get_extension(url, content_type);
|
||||||
|
return format!("{}{}", &hash[..16], ext);
|
||||||
|
}
|
||||||
|
|
||||||
|
let host = url
|
||||||
|
.host_str()
|
||||||
|
.unwrap_or("unknown")
|
||||||
|
.replace('.', "_")
|
||||||
|
.replace(':', "_");
|
||||||
|
|
||||||
|
let path = url.path().trim_start_matches('/');
|
||||||
|
let path = if path.is_empty() { "index" } else { path };
|
||||||
|
|
||||||
|
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||||
|
|
||||||
|
let (stem, ext_from_path) = match last_segment.rfind('.') {
|
||||||
|
Some(pos) if pos > 0 => (&last_segment[..pos], Some(&last_segment[pos..])),
|
||||||
|
_ => (last_segment, None),
|
||||||
|
};
|
||||||
|
|
||||||
|
let ext = match ext_from_path {
|
||||||
|
Some(e) => e.to_lowercase(),
|
||||||
|
None => get_extension(url, content_type),
|
||||||
|
};
|
||||||
|
|
||||||
|
let query_suffix = match url.query() {
|
||||||
|
Some(q) if !q.is_empty() => {
|
||||||
|
let mut hasher = Md5::new();
|
||||||
|
hasher.update(q.as_bytes());
|
||||||
|
let hash = hex::encode(hasher.finalize());
|
||||||
|
format!("_{}", &hash[..8])
|
||||||
|
}
|
||||||
|
_ => String::new(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext);
|
||||||
|
|
||||||
|
filename
|
||||||
|
.chars()
|
||||||
|
.map(|c| {
|
||||||
|
if c.is_alphanumeric() || "._-".contains(c) {
|
||||||
|
c
|
||||||
|
} else {
|
||||||
|
'_'
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user