qa and review
This commit is contained in:
+5
-1
@@ -1,5 +1,9 @@
|
||||
pub const DELAY_MS: u64 = 0;
|
||||
pub const DELAY_MS: u64 = 1000;
|
||||
pub const TIMEOUT_SECS: u64 = 60;
|
||||
pub const MAX_RETRIES: u32 = 3;
|
||||
pub const RETRY_BACKOFF_SECS: u64 = 5;
|
||||
pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
|
||||
|
||||
/// Rate limit in milliseconds between requests. Set to 0 only for trusted/internal sites.
|
||||
/// Default is 1000ms to avoid IP bans on public sites like logting.fo.
|
||||
pub const RATE_LIMIT_MS: u64 = DELAY_MS;
|
||||
|
||||
+5
-5
@@ -1,14 +1,14 @@
|
||||
use htmd::HtmlToMarkdown;
|
||||
use std::sync::OnceLock;
|
||||
|
||||
static CONVERTER: OnceLock<HtmlToMarkdown> = OnceLock::new();
|
||||
|
||||
pub fn html_to_markdown(html: &str) -> String {
|
||||
let converter = HtmlToMarkdown::new();
|
||||
let converter = CONVERTER.get_or_init(HtmlToMarkdown::new);
|
||||
match converter.convert(html) {
|
||||
Ok(md) => md,
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
" WARN: HTML-to-MD conversion failed ({}), saving raw HTML",
|
||||
e
|
||||
);
|
||||
log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e);
|
||||
html.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
use anyhow::Result;
|
||||
use std::path::Path;
|
||||
use url::Url;
|
||||
|
||||
pub enum DocProcessResult {
|
||||
Markdown(String),
|
||||
Raw,
|
||||
}
|
||||
|
||||
pub fn detect_format(bytes: &[u8], url: &Url) -> Option<anydoc::Format> {
|
||||
if let Some(fmt) = anydoc::Format::from_bytes(bytes) {
|
||||
return Some(fmt);
|
||||
}
|
||||
anydoc::Format::from_path(Path::new(url.path()))
|
||||
}
|
||||
|
||||
pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
|
||||
let Some(format) = detect_format(bytes, url) else {
|
||||
return DocProcessResult::Raw;
|
||||
};
|
||||
|
||||
match anydoc::to_markdown_bytes(bytes, Some(format)) {
|
||||
Ok(md) if !md.trim().is_empty() => DocProcessResult::Markdown(md),
|
||||
Ok(_) => {
|
||||
log::warn!("anydoc produced empty output, saving raw");
|
||||
DocProcessResult::Raw
|
||||
}
|
||||
Err(e) => {
|
||||
log::warn!("anydoc conversion failed ({}), saving raw", e);
|
||||
DocProcessResult::Raw
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub fn batch_convert_docs(dir: &Path) -> Result<(usize, usize, usize)> {
|
||||
let supported_exts = [
|
||||
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
|
||||
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
|
||||
];
|
||||
|
||||
let entries: Vec<_> = std::fs::read_dir(dir)?
|
||||
.filter_map(|e| e.ok())
|
||||
.filter(|e| {
|
||||
e.path().extension().is_some_and(|ext| {
|
||||
let ext = ext.to_string_lossy().to_lowercase();
|
||||
supported_exts.contains(&ext.as_str())
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
|
||||
let total = entries.len();
|
||||
log::info!("Found {} documents to process", total);
|
||||
|
||||
let mut converted = 0usize;
|
||||
let mut raw = 0usize;
|
||||
let mut errors = 0usize;
|
||||
|
||||
for (i, entry) in entries.iter().enumerate() {
|
||||
let path = entry.path();
|
||||
let filename = path.file_name().unwrap_or_default().to_string_lossy();
|
||||
log::info!("[{}/{}] processing: {}", i + 1, total, filename);
|
||||
|
||||
let bytes = match std::fs::read(&path) {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
log::error!("error reading file: {}", e);
|
||||
errors += 1;
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
let ext = path.extension().unwrap_or_default().to_string_lossy();
|
||||
let Some(format) = anydoc::Format::from_extension(ext.as_ref()) else {
|
||||
log::info!(" unrecognized format — keeping as-is");
|
||||
raw += 1;
|
||||
continue;
|
||||
};
|
||||
|
||||
match anydoc::to_markdown_bytes(&bytes, Some(format)) {
|
||||
Ok(md) if !md.trim().is_empty() => {
|
||||
let md_path = path.with_extension("md");
|
||||
std::fs::write(&md_path, md)?;
|
||||
std::fs::remove_file(&path)?;
|
||||
log::info!(" converted -> {}", md_path.display());
|
||||
converted += 1;
|
||||
}
|
||||
_ => {
|
||||
log::info!(" could not convert — keeping as-is");
|
||||
raw += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok((converted, raw, errors))
|
||||
}
|
||||
+15
-5
@@ -30,12 +30,22 @@ pub fn log_error(
|
||||
status_code,
|
||||
};
|
||||
|
||||
let line = serde_json::to_string(&entry).unwrap_or_default() + "\n";
|
||||
let line = match serde_json::to_string(&entry) {
|
||||
Ok(json) => json + "\n",
|
||||
Err(e) => {
|
||||
log::error!("Failed to serialize error entry: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) {
|
||||
let _ = file.write_all(line.as_bytes());
|
||||
let _ = file.flush();
|
||||
if let Err(e) = file.write_all(line.as_bytes()) {
|
||||
log::error!("Failed to write error log: {}", e);
|
||||
} else {
|
||||
let _ = file.flush();
|
||||
log::info!("LOGGED ERROR [{}]: {}", error_type, message);
|
||||
}
|
||||
} else {
|
||||
log::error!("Failed to open error log file: {}", log_path.display());
|
||||
}
|
||||
|
||||
println!(" LOGGED ERROR [{}]: {}", error_type, message);
|
||||
}
|
||||
|
||||
+9
-2
@@ -1,13 +1,20 @@
|
||||
use scraper::{Html, Selector};
|
||||
use std::collections::HashSet;
|
||||
use std::sync::OnceLock;
|
||||
use url::Url;
|
||||
|
||||
static LINK_SELECTOR: OnceLock<Selector> = OnceLock::new();
|
||||
|
||||
fn get_link_selector() -> &'static Selector {
|
||||
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
|
||||
}
|
||||
|
||||
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
|
||||
let document = Html::parse_document(html);
|
||||
let selector = Selector::parse("a[href]").unwrap();
|
||||
let selector = get_link_selector();
|
||||
|
||||
document
|
||||
.select(&selector)
|
||||
.select(selector)
|
||||
.filter_map(|el| el.value().attr("href"))
|
||||
.filter(|href| !href.is_empty())
|
||||
.filter(|href| {
|
||||
|
||||
+10
-17
@@ -1,5 +1,3 @@
|
||||
// src/fetcher.rs
|
||||
|
||||
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
|
||||
use anyhow::{Result, anyhow};
|
||||
use reqwest::Client;
|
||||
@@ -11,15 +9,15 @@ pub struct Fetcher {
|
||||
}
|
||||
|
||||
impl Fetcher {
|
||||
pub fn new() -> Self {
|
||||
pub fn new() -> Result<Self> {
|
||||
let client = Client::builder()
|
||||
.timeout(Duration::from_secs(TIMEOUT_SECS))
|
||||
.user_agent(USER_AGENT)
|
||||
.redirect(reqwest::redirect::Policy::limited(10))
|
||||
.build()
|
||||
.expect("Failed to create HTTP client");
|
||||
.map_err(|e| anyhow!("Failed to create HTTP client: {}", e))?;
|
||||
|
||||
Self { client }
|
||||
Ok(Self { client })
|
||||
}
|
||||
|
||||
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
|
||||
@@ -39,7 +37,12 @@ impl Fetcher {
|
||||
Ok(b) => b.to_vec(),
|
||||
Err(e) => {
|
||||
if attempt < MAX_RETRIES {
|
||||
self.print_retry(attempt, &e.to_string());
|
||||
log::warn!(
|
||||
"retry {}/{}: body read failed ({})",
|
||||
attempt,
|
||||
MAX_RETRIES,
|
||||
e
|
||||
);
|
||||
tokio::time::sleep(Duration::from_secs(
|
||||
RETRY_BACKOFF_SECS * attempt as u64,
|
||||
))
|
||||
@@ -62,12 +65,11 @@ impl Fetcher {
|
||||
bytes,
|
||||
content_type,
|
||||
final_url,
|
||||
_status_code: status.as_u16(),
|
||||
});
|
||||
}
|
||||
Err(e) => {
|
||||
if attempt < MAX_RETRIES {
|
||||
self.print_retry(attempt, &e.to_string());
|
||||
log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e);
|
||||
tokio::time::sleep(Duration::from_secs(
|
||||
RETRY_BACKOFF_SECS * attempt as u64,
|
||||
))
|
||||
@@ -85,19 +87,10 @@ impl Fetcher {
|
||||
|
||||
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
|
||||
}
|
||||
|
||||
fn print_retry(&self, attempt: u32, error: &str) {
|
||||
let wait = RETRY_BACKOFF_SECS * attempt as u64;
|
||||
println!(
|
||||
" retry {}/{} in {}s... ({})",
|
||||
attempt, MAX_RETRIES, wait, error
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
pub struct FetchResult {
|
||||
pub bytes: Vec<u8>,
|
||||
pub content_type: String,
|
||||
pub final_url: Url,
|
||||
pub _status_code: u16,
|
||||
}
|
||||
|
||||
+42
-53
@@ -1,60 +1,42 @@
|
||||
mod config;
|
||||
mod converter;
|
||||
mod doc_processor;
|
||||
mod error_logger;
|
||||
mod extractor;
|
||||
mod fetcher;
|
||||
mod pdf_processor;
|
||||
mod robots;
|
||||
mod scraper;
|
||||
mod url_utils;
|
||||
|
||||
use crate::config::USER_AGENT;
|
||||
use crate::robots::fetch_robots;
|
||||
use crate::scraper::Scraper;
|
||||
use anyhow::Result;
|
||||
use clap::Parser;
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use url::Url;
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
#[command(name = "rs-scraper")]
|
||||
#[command(author = "FLÓ")]
|
||||
#[command(version = "0.1.0")]
|
||||
#[command(about = "Web scraper for Faroese public sector sites")]
|
||||
struct Args {
|
||||
#[arg(help = "Starting URL to scrape")]
|
||||
start_url: String,
|
||||
|
||||
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
|
||||
no_doc_conversion: bool,
|
||||
}
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> anyhow::Result<()> {
|
||||
let args: Vec<String> = std::env::args().collect();
|
||||
async fn main() -> Result<()> {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
|
||||
|
||||
// Parse flags
|
||||
let mut convert_pdfs = true;
|
||||
let mut url_arg: Option<String> = None;
|
||||
let args = Args::parse();
|
||||
|
||||
for arg in args.iter().skip(1) {
|
||||
match arg.as_str() {
|
||||
"--no-pdf-conversion" => convert_pdfs = false,
|
||||
"--help" | "-h" => {
|
||||
println!("Usage: site-scraper [OPTIONS] <start-url>");
|
||||
println!();
|
||||
println!("Options:");
|
||||
println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction");
|
||||
println!(" -h, --help Show this help message");
|
||||
println!();
|
||||
println!("Example:");
|
||||
println!(" site-scraper https://www.logting.fo");
|
||||
println!(" site-scraper --no-pdf-conversion https://taks.fo");
|
||||
std::process::exit(0);
|
||||
}
|
||||
_ => {
|
||||
if url_arg.is_none() {
|
||||
url_arg = Some(arg.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let start_url = match url_arg {
|
||||
Some(u) => u,
|
||||
None => {
|
||||
eprintln!("Usage: site-scraper [OPTIONS] <start-url>");
|
||||
eprintln!("Example: site-scraper https://www.logting.fo");
|
||||
eprintln!("Run with --help for options.");
|
||||
std::process::exit(1);
|
||||
}
|
||||
};
|
||||
let start_url = args.start_url;
|
||||
let convert_docs = !args.no_doc_conversion;
|
||||
|
||||
let url = Url::parse(&start_url)?;
|
||||
|
||||
@@ -70,29 +52,36 @@ async fn main() -> anyhow::Result<()> {
|
||||
let log_path = logs_dir.join("scrape_errors.jsonl");
|
||||
fs::remove_file(&log_path).ok();
|
||||
|
||||
println!("Scraping: {}", start_url);
|
||||
println!("Base domain: {}", base_domain);
|
||||
println!("Output dir: {}", output_dir.display());
|
||||
println!(
|
||||
"PDF conversion: {}",
|
||||
if convert_pdfs { "enabled" } else { "disabled" }
|
||||
log::info!("Scraping: {}", start_url);
|
||||
log::info!("Base domain: {}", base_domain);
|
||||
log::info!("Output dir: {}", output_dir.display());
|
||||
log::info!(
|
||||
"Document conversion: {}",
|
||||
if convert_docs { "enabled" } else { "disabled" }
|
||||
);
|
||||
|
||||
// Fetch robots.txt
|
||||
let client = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(30))
|
||||
.user_agent(USER_AGENT)
|
||||
.user_agent(config::USER_AGENT)
|
||||
.build()?;
|
||||
|
||||
println!("Fetching robots.txt...");
|
||||
let robots = fetch_robots(&client, &url, "web-scraper").await;
|
||||
println!();
|
||||
log::info!("Fetching robots.txt...");
|
||||
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
|
||||
log::info!("");
|
||||
|
||||
let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs);
|
||||
let fetcher = fetcher::Fetcher::new()?;
|
||||
let scraper = Scraper::new(
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
robots,
|
||||
convert_docs,
|
||||
fetcher,
|
||||
);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
log::info!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1,71 +0,0 @@
|
||||
// src/pdf_processor.rs
|
||||
|
||||
use anyhow::Result;
|
||||
use std::path::Path;
|
||||
|
||||
pub enum PdfProcessResult {
|
||||
Markdown(String),
|
||||
Scanned,
|
||||
}
|
||||
|
||||
pub fn process_pdf(pdf_path: &Path) -> Result<PdfProcessResult> {
|
||||
use pdf_inspector::process_pdf;
|
||||
|
||||
let result = process_pdf(pdf_path)?;
|
||||
|
||||
match result.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased | pdf_inspector::PdfType::Mixed => {
|
||||
match &result.markdown {
|
||||
Some(md) if !md.trim().is_empty() => Ok(PdfProcessResult::Markdown(md.clone())),
|
||||
_ => Ok(PdfProcessResult::Scanned),
|
||||
}
|
||||
}
|
||||
pdf_inspector::PdfType::Scanned | pdf_inspector::PdfType::ImageBased => {
|
||||
Ok(PdfProcessResult::Scanned)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn batch_convert_pdfs(dir: &Path) -> Result<(usize, usize, usize)> {
|
||||
let mut text_based = 0usize;
|
||||
let mut scanned = 0usize;
|
||||
let mut errors = 0usize;
|
||||
|
||||
let entries: Vec<_> = std::fs::read_dir(dir)?
|
||||
.filter_map(|e| e.ok())
|
||||
.filter(|e| {
|
||||
e.path()
|
||||
.extension()
|
||||
.is_some_and(|ext| ext.eq_ignore_ascii_case("pdf"))
|
||||
})
|
||||
.collect();
|
||||
|
||||
let total = entries.len();
|
||||
println!("Found {} PDFs to process", total);
|
||||
|
||||
for (i, entry) in entries.iter().enumerate() {
|
||||
let path = entry.path();
|
||||
let filename = path.file_name().unwrap_or_default().to_string_lossy();
|
||||
println!("[{}/{}] processing: {}", i + 1, total, filename);
|
||||
|
||||
match process_pdf(&path) {
|
||||
Ok(PdfProcessResult::Markdown(md)) => {
|
||||
let md_path = path.with_extension("md");
|
||||
std::fs::write(&md_path, md)?;
|
||||
std::fs::remove_file(&path)?;
|
||||
println!(" text-based -> {}", md_path.display());
|
||||
text_based += 1;
|
||||
}
|
||||
Ok(PdfProcessResult::Scanned) => {
|
||||
println!(" scanned — keeping as PDF");
|
||||
scanned += 1;
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" error: {}", e);
|
||||
errors += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok((text_based, scanned, errors))
|
||||
}
|
||||
+72
-8
@@ -58,11 +58,16 @@ fn path_matches(pattern: &str, path: &str) -> bool {
|
||||
}
|
||||
|
||||
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
|
||||
let robots_url = format!(
|
||||
"{}://{}/robots.txt",
|
||||
base_url.scheme(),
|
||||
base_url.host_str().unwrap_or("")
|
||||
);
|
||||
let host_str = base_url.host_str().unwrap_or("");
|
||||
if host_str.is_empty() {
|
||||
log::warn!("No host in base URL, skipping robots.txt");
|
||||
return RobotsRule {
|
||||
allowed: Vec::new(),
|
||||
disallowed: Vec::new(),
|
||||
};
|
||||
}
|
||||
|
||||
let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str);
|
||||
|
||||
let mut rule = RobotsRule {
|
||||
allowed: Vec::new(),
|
||||
@@ -82,13 +87,13 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
|
||||
}
|
||||
}
|
||||
Ok(resp) => {
|
||||
println!(
|
||||
log::warn!(
|
||||
"robots.txt returned HTTP {} — assuming no restrictions",
|
||||
resp.status()
|
||||
);
|
||||
}
|
||||
Err(e) => {
|
||||
println!(
|
||||
log::warn!(
|
||||
"Failed to fetch robots.txt ({}): assuming no restrictions",
|
||||
e
|
||||
);
|
||||
@@ -149,7 +154,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
|
||||
}
|
||||
|
||||
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
|
||||
println!(
|
||||
log::info!(
|
||||
"robots.txt: {} disallow rules, {} allow rules",
|
||||
rule.disallowed.len(),
|
||||
rule.allowed.len()
|
||||
@@ -158,3 +163,62 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
|
||||
|
||||
rule
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_path_matches_exact() {
|
||||
assert!(path_matches("/admin", "/admin"));
|
||||
assert!(path_matches("/admin", "/admin/users"));
|
||||
assert!(path_matches("/admin", "/administrator"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_matches_wildcard() {
|
||||
assert!(path_matches("/private/*", "/private/data"));
|
||||
assert!(path_matches("/private/*/secret", "/private/x/secret"));
|
||||
assert!(!path_matches("/private/*/secret", "/private/secret"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_matches_empty_pattern() {
|
||||
assert!(!path_matches("", "/anything"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_allowed_disallow_takes_precedence_when_longer() {
|
||||
let rule = RobotsRule {
|
||||
allowed: vec!["/a".into()],
|
||||
disallowed: vec!["/a/b".into()],
|
||||
};
|
||||
assert!(!rule.is_allowed("/a/b"));
|
||||
assert!(rule.is_allowed("/a/c"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_allowed_allow_overrides_disallow_when_longer() {
|
||||
let rule = RobotsRule {
|
||||
allowed: vec!["/public/pages".into()],
|
||||
disallowed: vec!["/public".into()],
|
||||
};
|
||||
assert!(rule.is_allowed("/public/pages"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_robots_txt_wildcard_agent() {
|
||||
let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n";
|
||||
let rule = parse_robots_txt(txt, "web-scraper");
|
||||
assert!(rule.disallowed.contains(&"/admin".to_string()));
|
||||
assert!(rule.allowed.contains(&"/admin/public".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_robots_txt_specific_agent() {
|
||||
let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n";
|
||||
let rule = parse_robots_txt(txt, "web-scraper");
|
||||
assert!(rule.disallowed.contains(&"/private".to_string()));
|
||||
assert!(!rule.disallowed.contains(&"/".to_string()));
|
||||
}
|
||||
}
|
||||
|
||||
+87
-82
@@ -1,8 +1,8 @@
|
||||
use crate::converter::html_to_markdown;
|
||||
use crate::doc_processor::{DocProcessResult, try_convert};
|
||||
use crate::error_logger::log_error;
|
||||
use crate::extractor::extract_links;
|
||||
use crate::fetcher::{FetchResult, Fetcher};
|
||||
use crate::pdf_processor::{PdfProcessResult, process_pdf};
|
||||
use crate::robots::RobotsRule;
|
||||
use crate::url_utils::{normalize_url, url_to_filename};
|
||||
use anyhow::Result;
|
||||
@@ -13,17 +13,17 @@ use std::time::Duration;
|
||||
use tokio::sync::Mutex;
|
||||
use url::Url;
|
||||
|
||||
pub struct PdfStats {
|
||||
pub text_based: usize,
|
||||
pub scanned: usize,
|
||||
pub struct DocStats {
|
||||
pub converted: usize,
|
||||
pub raw: usize,
|
||||
pub errors: usize,
|
||||
}
|
||||
|
||||
impl Default for PdfStats {
|
||||
impl Default for DocStats {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
text_based: 0,
|
||||
scanned: 0,
|
||||
converted: 0,
|
||||
raw: 0,
|
||||
errors: 0,
|
||||
}
|
||||
}
|
||||
@@ -36,8 +36,8 @@ pub struct Scraper {
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
robots: RobotsRule,
|
||||
pdf_stats: Arc<Mutex<PdfStats>>,
|
||||
convert_pdfs: bool,
|
||||
doc_stats: Arc<Mutex<DocStats>>,
|
||||
convert_docs: bool,
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
@@ -46,17 +46,18 @@ impl Scraper {
|
||||
log_path: std::path::PathBuf,
|
||||
base_domain: String,
|
||||
robots: RobotsRule,
|
||||
convert_pdfs: bool,
|
||||
convert_docs: bool,
|
||||
fetcher: Fetcher,
|
||||
) -> Self {
|
||||
Self {
|
||||
fetcher: Fetcher::new(),
|
||||
fetcher,
|
||||
seen: Arc::new(Mutex::new(HashSet::new())),
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
robots,
|
||||
pdf_stats: Arc::new(Mutex::new(PdfStats::default())),
|
||||
convert_pdfs,
|
||||
doc_stats: Arc::new(Mutex::new(DocStats::default())),
|
||||
convert_docs,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -85,62 +86,58 @@ impl Scraper {
|
||||
}
|
||||
|
||||
if !self.robots.is_allowed(url.path()) {
|
||||
println!("[skip] robots.txt disallows: {}", url);
|
||||
log::info!("[skip] robots.txt disallows: {}", url);
|
||||
skipped_robots += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
println!("[{}] fetching: {}", count, url);
|
||||
log::info!("[{}] fetching: {}", count, url);
|
||||
|
||||
match self.fetcher.fetch_with_retry(&url).await {
|
||||
Ok(result) => {
|
||||
let final_url = result.final_url.clone();
|
||||
let is_html = result.content_type.contains("text/html");
|
||||
let is_pdf = result.content_type.contains("application/pdf");
|
||||
|
||||
match self.save(&result, is_html, is_pdf).await {
|
||||
match self.save(&result, is_html).await {
|
||||
Ok(()) => {
|
||||
if is_html {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links =
|
||||
extract_links(html, &result.final_url, &self.base_domain);
|
||||
let links = extract_links(html, &final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
|
||||
let mut new_links = Vec::new();
|
||||
|
||||
{
|
||||
let new_links: Vec<Url> = {
|
||||
let seen = self.seen.lock().await;
|
||||
for link in &links {
|
||||
if !seen.contains(link.as_str()) {
|
||||
new_links.push(link.clone());
|
||||
}
|
||||
}
|
||||
links
|
||||
.into_iter()
|
||||
.filter(|link| !seen.contains(link.as_str()))
|
||||
.collect()
|
||||
};
|
||||
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
|
||||
let new_count = new_links.len();
|
||||
|
||||
for link in new_links {
|
||||
queue.push_back(link);
|
||||
}
|
||||
|
||||
println!(" found {} links ({} new)", links.len(), new_count);
|
||||
log::info!(
|
||||
" found {} links ({} new)",
|
||||
new_count,
|
||||
new_links.len()
|
||||
);
|
||||
}
|
||||
} else {
|
||||
println!(" binary: {}", result.content_type);
|
||||
log::info!(" binary: {}", result.content_type);
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
log_error(
|
||||
&self.log_path,
|
||||
&result.final_url,
|
||||
&final_url,
|
||||
"save_error",
|
||||
&e.to_string(),
|
||||
None,
|
||||
);
|
||||
error_count += 1;
|
||||
|
||||
if is_pdf {
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
}
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -151,29 +148,28 @@ impl Scraper {
|
||||
}
|
||||
|
||||
count += 1;
|
||||
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
|
||||
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await;
|
||||
}
|
||||
|
||||
{
|
||||
let stats = self.pdf_stats.lock().await;
|
||||
if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 {
|
||||
println!("\nPDF Statistics:");
|
||||
println!(" Text-based (converted): {}", stats.text_based);
|
||||
println!(" Scanned (kept as PDF): {}", stats.scanned);
|
||||
println!(" Errors: {}", stats.errors);
|
||||
let stats = self.doc_stats.lock().await;
|
||||
if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 {
|
||||
log::info!("\nDocument Statistics:");
|
||||
log::info!(" Converted to Markdown: {}", stats.converted);
|
||||
log::info!(" Kept as-is: {}", stats.raw);
|
||||
log::info!(" Errors: {}", stats.errors);
|
||||
}
|
||||
}
|
||||
|
||||
if skipped_robots > 0 {
|
||||
println!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
}
|
||||
|
||||
(count, error_count)
|
||||
}
|
||||
|
||||
async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> {
|
||||
async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
||||
|
||||
let content: Vec<u8>;
|
||||
|
||||
if is_html {
|
||||
@@ -183,51 +179,60 @@ impl Scraper {
|
||||
filename = filename.replace(".html", ".md");
|
||||
}
|
||||
content = md.into_bytes();
|
||||
} else if is_pdf && self.convert_pdfs {
|
||||
// Save PDF to temp file for pdf-inspector processing
|
||||
let temp_path = self.output_dir.join("temp_pdf.tmp");
|
||||
fs::write(&temp_path, &result.bytes)?;
|
||||
|
||||
match process_pdf(&temp_path) {
|
||||
Ok(PdfProcessResult::Markdown(md)) => {
|
||||
if filename.ends_with(".pdf") {
|
||||
filename = filename.replace(".pdf", ".md");
|
||||
} else if self.convert_docs {
|
||||
match try_convert(&result.bytes, &result.final_url) {
|
||||
DocProcessResult::Markdown(md) => {
|
||||
if let Some(pos) = filename.rfind('.') {
|
||||
filename.truncate(pos);
|
||||
}
|
||||
filename.push_str(".md");
|
||||
content = md.into_bytes();
|
||||
fs::remove_file(&temp_path)?;
|
||||
|
||||
println!(" [PDF] text-based — converted to Markdown");
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.text_based += 1;
|
||||
log::info!(" [DOC] converted to Markdown");
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.converted += 1;
|
||||
}
|
||||
Ok(PdfProcessResult::Scanned) => {
|
||||
println!(" [PDF] scanned document — saved as-is");
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.scanned += 1;
|
||||
|
||||
fs::rename(&temp_path, self.output_dir.join(&filename))?;
|
||||
return Ok(());
|
||||
}
|
||||
Err(e) => {
|
||||
println!(" [PDF] processing error ({}), saved as-is", e);
|
||||
let mut stats = self.pdf_stats.lock().await;
|
||||
stats.errors += 1;
|
||||
|
||||
fs::rename(&temp_path, self.output_dir.join(&filename))?;
|
||||
return Ok(());
|
||||
DocProcessResult::Raw => {
|
||||
content = result.bytes.clone();
|
||||
if is_document_content_type(&result.content_type) {
|
||||
let mut stats = self.doc_stats.lock().await;
|
||||
stats.raw += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Non-HTML, non-PDF, or PDF conversion disabled — save as-is
|
||||
content = result.bytes.clone();
|
||||
}
|
||||
|
||||
let filepath = self.output_dir.join(&filename);
|
||||
fs::write(&filepath, &content)?;
|
||||
println!(
|
||||
log::info!(
|
||||
" saved -> {}",
|
||||
filepath.file_name().unwrap_or_default().to_string_lossy()
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn is_document_content_type(ct: &str) -> bool {
|
||||
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
|
||||
const DOC_TYPES: &[&str] = &[
|
||||
"application/pdf",
|
||||
"application/msword",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"application/vnd.ms-word.document.macroenabled.12",
|
||||
"application/vnd.ms-powerpoint",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
||||
"application/vnd.ms-excel",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
"application/vnd.oasis.opendocument.spreadsheet",
|
||||
"application/vnd.oasis.opendocument.presentation",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"text/csv",
|
||||
];
|
||||
DOC_TYPES.contains(&ct.as_str())
|
||||
}
|
||||
|
||||
+143
-10
@@ -14,7 +14,12 @@ pub fn derive_base_domain(url: &Url) -> Option<String> {
|
||||
|
||||
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
|
||||
match url.host_str() {
|
||||
Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)),
|
||||
Some(host) => {
|
||||
host == base_domain
|
||||
|| host
|
||||
.strip_suffix(base_domain)
|
||||
.is_some_and(|rest| rest.ends_with('.'))
|
||||
}
|
||||
None => false,
|
||||
}
|
||||
}
|
||||
@@ -50,6 +55,7 @@ fn get_extension(url: &Url, content_type: &str) -> String {
|
||||
"application/json" => ".json",
|
||||
"application/xml" | "text/xml" => ".xml",
|
||||
"text/plain" => ".txt",
|
||||
"text/csv" => ".csv",
|
||||
"text/css" => ".css",
|
||||
"application/javascript" | "text/javascript" => ".js",
|
||||
"image/png" => ".png",
|
||||
@@ -57,10 +63,21 @@ fn get_extension(url: &Url, content_type: &str) -> String {
|
||||
"image/gif" => ".gif",
|
||||
"image/svg+xml" => ".svg",
|
||||
"image/webp" => ".webp",
|
||||
"application/vnd.ms-excel" => ".xls",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
|
||||
"application/rtf" => ".rtf",
|
||||
"application/epub+zip" => ".epub",
|
||||
"application/msword" => ".doc",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx",
|
||||
"application/vnd.ms-word.document.macroenabled.12" => ".docm",
|
||||
"application/vnd.ms-powerpoint" => ".ppt",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation" => ".pptx",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12" => ".pptm",
|
||||
"application/vnd.ms-excel" => ".xls",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12" => ".xlsm",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12" => ".xlsb",
|
||||
"application/vnd.oasis.opendocument.text" => ".odt",
|
||||
"application/vnd.oasis.opendocument.spreadsheet" => ".ods",
|
||||
"application/vnd.oasis.opendocument.presentation" => ".odp",
|
||||
"application/zip" => ".zip",
|
||||
"application/octet-stream" => ".bin",
|
||||
_ => ".bin",
|
||||
@@ -84,14 +101,10 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
||||
.unwrap_or("unknown")
|
||||
.replace('.', "_")
|
||||
.replace(':', "_");
|
||||
|
||||
let path = url.path().trim_start_matches('/');
|
||||
let path = if path.is_empty() { "index" } else { path };
|
||||
|
||||
// FIX: strip trailing slash so /support/ -> "support" not ""
|
||||
let path = path.trim_end_matches('/');
|
||||
let path = if path.is_empty() { "index" } else { path };
|
||||
|
||||
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||
|
||||
let (stem, ext_from_path) = match last_segment.rfind('.') {
|
||||
@@ -114,9 +127,11 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
||||
_ => String::new(),
|
||||
};
|
||||
|
||||
let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext);
|
||||
let mut hasher = Sha256::new();
|
||||
hasher.update(url_str.as_bytes());
|
||||
let url_hash = hex::encode(hasher.finalize());
|
||||
|
||||
filename
|
||||
let safe_stem: String = stem
|
||||
.chars()
|
||||
.map(|c| {
|
||||
if c.is_alphanumeric() || "._-".contains(c) {
|
||||
@@ -125,5 +140,123 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
||||
'_'
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
.collect();
|
||||
|
||||
format!(
|
||||
"{}_{}{}{:}{}",
|
||||
host,
|
||||
safe_stem,
|
||||
query_suffix,
|
||||
&url_hash[..8],
|
||||
ext
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_derive_base_domain() {
|
||||
let url = Url::parse("https://www.logting.fo/").unwrap();
|
||||
assert_eq!(derive_base_domain(&url), Some("logting.fo".into()));
|
||||
|
||||
let url = Url::parse("https://taks.fo/en/skatur/").unwrap();
|
||||
assert_eq!(derive_base_domain(&url), Some("taks.fo".into()));
|
||||
|
||||
let url = Url::parse("https://localhost:8080/").unwrap();
|
||||
assert_eq!(derive_base_domain(&url), Some("localhost".into()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_internal_subdomain() {
|
||||
let base = "logting.fo";
|
||||
assert!(is_internal(
|
||||
&Url::parse("https://logting.fo/").unwrap(),
|
||||
base
|
||||
));
|
||||
assert!(is_internal(
|
||||
&Url::parse("https://www.logting.fo/").unwrap(),
|
||||
base
|
||||
));
|
||||
assert!(is_internal(
|
||||
&Url::parse("https://sub.www.logting.fo/").unwrap(),
|
||||
base
|
||||
));
|
||||
assert!(!is_internal(
|
||||
&Url::parse("https://evilogting.fo/").unwrap(),
|
||||
base
|
||||
));
|
||||
assert!(!is_internal(
|
||||
&Url::parse("https://logting.com/").unwrap(),
|
||||
base
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_normalize_url_strips_fragment() {
|
||||
let url = normalize_url("https://example.com/page#section").unwrap();
|
||||
assert_eq!(url.as_str(), "https://example.com/page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_normalize_url_strips_trailing_query_chars() {
|
||||
let url = normalize_url("https://example.com/page?&").unwrap();
|
||||
assert_eq!(url.as_str(), "https://example.com/page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_url_to_filename_faroese_unicode() {
|
||||
let url = Url::parse("https://logting.fo/lov/tinglýsing/2024/").unwrap();
|
||||
let filename = url_to_filename(&url, "text/html");
|
||||
assert!(filename.starts_with("logting_fo_"));
|
||||
assert!(filename.ends_with(".html") || filename.ends_with(".md"));
|
||||
assert!(!filename.contains('ý'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_url_to_filename_long_url_hashed() {
|
||||
let long = format!("https://example.com/{}", "a".repeat(200));
|
||||
let url = Url::parse(&long).unwrap();
|
||||
let filename = url_to_filename(&url, "text/html");
|
||||
assert!(filename.len() < 30);
|
||||
assert!(filename.ends_with(".html"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_url_to_filename_with_query() {
|
||||
let url = Url::parse("https://example.com/page?id=42&sort=desc").unwrap();
|
||||
let filename = url_to_filename(&url, "text/html");
|
||||
assert!(filename.contains('_'));
|
||||
assert!(filename.ends_with(".html"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_get_extension_from_content_type() {
|
||||
let url = Url::parse("https://example.com/").unwrap();
|
||||
assert_eq!(get_extension(&url, "text/html"), ".html");
|
||||
assert_eq!(
|
||||
get_extension(&url, "application/pdf; charset=binary"),
|
||||
".pdf"
|
||||
);
|
||||
assert_eq!(get_extension(&url, "application/octet-stream"), ".bin");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_get_extension_from_url_path() {
|
||||
let url = Url::parse("https://example.com/doc.pdf").unwrap();
|
||||
assert_eq!(get_extension(&url, "application/octet-stream"), ".pdf");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_url_collision_avoidance() {
|
||||
let url1 = Url::parse("https://example.com/foo-bar").unwrap();
|
||||
let url2 = Url::parse("https://example.com/foo.bar").unwrap();
|
||||
let fn1 = url_to_filename(&url1, "text/html");
|
||||
let fn2 = url_to_filename(&url2, "text/html");
|
||||
assert_ne!(
|
||||
fn1, fn2,
|
||||
"Different URLs should produce different filenames"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user