qa and review

This commit is contained in:
2026-08-15 21:21:43 +01:00
parent a60cd5deae
commit 3c7043562e
13 changed files with 753 additions and 297 deletions
+5 -1
View File
@@ -1,5 +1,9 @@
pub const DELAY_MS: u64 = 0;
pub const DELAY_MS: u64 = 1000;
pub const TIMEOUT_SECS: u64 = 60;
pub const MAX_RETRIES: u32 = 3;
pub const RETRY_BACKOFF_SECS: u64 = 5;
pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
/// Rate limit in milliseconds between requests. Set to 0 only for trusted/internal sites.
/// Default is 1000ms to avoid IP bans on public sites like logting.fo.
pub const RATE_LIMIT_MS: u64 = DELAY_MS;
+5 -5
View File
@@ -1,14 +1,14 @@
use htmd::HtmlToMarkdown;
use std::sync::OnceLock;
static CONVERTER: OnceLock<HtmlToMarkdown> = OnceLock::new();
pub fn html_to_markdown(html: &str) -> String {
let converter = HtmlToMarkdown::new();
let converter = CONVERTER.get_or_init(HtmlToMarkdown::new);
match converter.convert(html) {
Ok(md) => md,
Err(e) => {
eprintln!(
" WARN: HTML-to-MD conversion failed ({}), saving raw HTML",
e
);
log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e);
html.to_string()
}
}
+96
View File
@@ -0,0 +1,96 @@
use anyhow::Result;
use std::path::Path;
use url::Url;
pub enum DocProcessResult {
Markdown(String),
Raw,
}
pub fn detect_format(bytes: &[u8], url: &Url) -> Option<anydoc::Format> {
if let Some(fmt) = anydoc::Format::from_bytes(bytes) {
return Some(fmt);
}
anydoc::Format::from_path(Path::new(url.path()))
}
pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
let Some(format) = detect_format(bytes, url) else {
return DocProcessResult::Raw;
};
match anydoc::to_markdown_bytes(bytes, Some(format)) {
Ok(md) if !md.trim().is_empty() => DocProcessResult::Markdown(md),
Ok(_) => {
log::warn!("anydoc produced empty output, saving raw");
DocProcessResult::Raw
}
Err(e) => {
log::warn!("anydoc conversion failed ({}), saving raw", e);
DocProcessResult::Raw
}
}
}
#[allow(dead_code)]
pub fn batch_convert_docs(dir: &Path) -> Result<(usize, usize, usize)> {
let supported_exts = [
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
];
let entries: Vec<_> = std::fs::read_dir(dir)?
.filter_map(|e| e.ok())
.filter(|e| {
e.path().extension().is_some_and(|ext| {
let ext = ext.to_string_lossy().to_lowercase();
supported_exts.contains(&ext.as_str())
})
})
.collect();
let total = entries.len();
log::info!("Found {} documents to process", total);
let mut converted = 0usize;
let mut raw = 0usize;
let mut errors = 0usize;
for (i, entry) in entries.iter().enumerate() {
let path = entry.path();
let filename = path.file_name().unwrap_or_default().to_string_lossy();
log::info!("[{}/{}] processing: {}", i + 1, total, filename);
let bytes = match std::fs::read(&path) {
Ok(b) => b,
Err(e) => {
log::error!("error reading file: {}", e);
errors += 1;
continue;
}
};
let ext = path.extension().unwrap_or_default().to_string_lossy();
let Some(format) = anydoc::Format::from_extension(ext.as_ref()) else {
log::info!(" unrecognized format — keeping as-is");
raw += 1;
continue;
};
match anydoc::to_markdown_bytes(&bytes, Some(format)) {
Ok(md) if !md.trim().is_empty() => {
let md_path = path.with_extension("md");
std::fs::write(&md_path, md)?;
std::fs::remove_file(&path)?;
log::info!(" converted -> {}", md_path.display());
converted += 1;
}
_ => {
log::info!(" could not convert — keeping as-is");
raw += 1;
}
}
}
Ok((converted, raw, errors))
}
+15 -5
View File
@@ -30,12 +30,22 @@ pub fn log_error(
status_code,
};
let line = serde_json::to_string(&entry).unwrap_or_default() + "\n";
let line = match serde_json::to_string(&entry) {
Ok(json) => json + "\n",
Err(e) => {
log::error!("Failed to serialize error entry: {}", e);
return;
}
};
if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) {
let _ = file.write_all(line.as_bytes());
let _ = file.flush();
if let Err(e) = file.write_all(line.as_bytes()) {
log::error!("Failed to write error log: {}", e);
} else {
let _ = file.flush();
log::info!("LOGGED ERROR [{}]: {}", error_type, message);
}
} else {
log::error!("Failed to open error log file: {}", log_path.display());
}
println!(" LOGGED ERROR [{}]: {}", error_type, message);
}
+9 -2
View File
@@ -1,13 +1,20 @@
use scraper::{Html, Selector};
use std::collections::HashSet;
use std::sync::OnceLock;
use url::Url;
static LINK_SELECTOR: OnceLock<Selector> = OnceLock::new();
fn get_link_selector() -> &'static Selector {
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
}
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
let document = Html::parse_document(html);
let selector = Selector::parse("a[href]").unwrap();
let selector = get_link_selector();
document
.select(&selector)
.select(selector)
.filter_map(|el| el.value().attr("href"))
.filter(|href| !href.is_empty())
.filter(|href| {
+10 -17
View File
@@ -1,5 +1,3 @@
// src/fetcher.rs
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
use anyhow::{Result, anyhow};
use reqwest::Client;
@@ -11,15 +9,15 @@ pub struct Fetcher {
}
impl Fetcher {
pub fn new() -> Self {
pub fn new() -> Result<Self> {
let client = Client::builder()
.timeout(Duration::from_secs(TIMEOUT_SECS))
.user_agent(USER_AGENT)
.redirect(reqwest::redirect::Policy::limited(10))
.build()
.expect("Failed to create HTTP client");
.map_err(|e| anyhow!("Failed to create HTTP client: {}", e))?;
Self { client }
Ok(Self { client })
}
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
@@ -39,7 +37,12 @@ impl Fetcher {
Ok(b) => b.to_vec(),
Err(e) => {
if attempt < MAX_RETRIES {
self.print_retry(attempt, &e.to_string());
log::warn!(
"retry {}/{}: body read failed ({})",
attempt,
MAX_RETRIES,
e
);
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
))
@@ -62,12 +65,11 @@ impl Fetcher {
bytes,
content_type,
final_url,
_status_code: status.as_u16(),
});
}
Err(e) => {
if attempt < MAX_RETRIES {
self.print_retry(attempt, &e.to_string());
log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e);
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
))
@@ -85,19 +87,10 @@ impl Fetcher {
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
}
fn print_retry(&self, attempt: u32, error: &str) {
let wait = RETRY_BACKOFF_SECS * attempt as u64;
println!(
" retry {}/{} in {}s... ({})",
attempt, MAX_RETRIES, wait, error
);
}
}
pub struct FetchResult {
pub bytes: Vec<u8>,
pub content_type: String,
pub final_url: Url,
pub _status_code: u16,
}
+42 -53
View File
@@ -1,60 +1,42 @@
mod config;
mod converter;
mod doc_processor;
mod error_logger;
mod extractor;
mod fetcher;
mod pdf_processor;
mod robots;
mod scraper;
mod url_utils;
use crate::config::USER_AGENT;
use crate::robots::fetch_robots;
use crate::scraper::Scraper;
use anyhow::Result;
use clap::Parser;
use std::fs;
use std::path::PathBuf;
use url::Url;
#[derive(Parser, Debug)]
#[command(name = "rs-scraper")]
#[command(author = "FLÓ")]
#[command(version = "0.1.0")]
#[command(about = "Web scraper for Faroese public sector sites")]
struct Args {
#[arg(help = "Starting URL to scrape")]
start_url: String,
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
no_doc_conversion: bool,
}
#[tokio::main]
async fn main() -> anyhow::Result<()> {
let args: Vec<String> = std::env::args().collect();
async fn main() -> Result<()> {
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
// Parse flags
let mut convert_pdfs = true;
let mut url_arg: Option<String> = None;
let args = Args::parse();
for arg in args.iter().skip(1) {
match arg.as_str() {
"--no-pdf-conversion" => convert_pdfs = false,
"--help" | "-h" => {
println!("Usage: site-scraper [OPTIONS] <start-url>");
println!();
println!("Options:");
println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction");
println!(" -h, --help Show this help message");
println!();
println!("Example:");
println!(" site-scraper https://www.logting.fo");
println!(" site-scraper --no-pdf-conversion https://taks.fo");
std::process::exit(0);
}
_ => {
if url_arg.is_none() {
url_arg = Some(arg.clone());
}
}
}
}
let start_url = match url_arg {
Some(u) => u,
None => {
eprintln!("Usage: site-scraper [OPTIONS] <start-url>");
eprintln!("Example: site-scraper https://www.logting.fo");
eprintln!("Run with --help for options.");
std::process::exit(1);
}
};
let start_url = args.start_url;
let convert_docs = !args.no_doc_conversion;
let url = Url::parse(&start_url)?;
@@ -70,29 +52,36 @@ async fn main() -> anyhow::Result<()> {
let log_path = logs_dir.join("scrape_errors.jsonl");
fs::remove_file(&log_path).ok();
println!("Scraping: {}", start_url);
println!("Base domain: {}", base_domain);
println!("Output dir: {}", output_dir.display());
println!(
"PDF conversion: {}",
if convert_pdfs { "enabled" } else { "disabled" }
log::info!("Scraping: {}", start_url);
log::info!("Base domain: {}", base_domain);
log::info!("Output dir: {}", output_dir.display());
log::info!(
"Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" }
);
// Fetch robots.txt
let client = reqwest::Client::builder()
.timeout(std::time::Duration::from_secs(30))
.user_agent(USER_AGENT)
.user_agent(config::USER_AGENT)
.build()?;
println!("Fetching robots.txt...");
let robots = fetch_robots(&client, &url, "web-scraper").await;
println!();
log::info!("Fetching robots.txt...");
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
log::info!("");
let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs);
let fetcher = fetcher::Fetcher::new()?;
let scraper = Scraper::new(
output_dir,
log_path,
base_domain,
robots,
convert_docs,
fetcher,
);
let (count, errors) = scraper.run(&url).await;
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
log::info!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
Ok(())
}
-71
View File
@@ -1,71 +0,0 @@
// src/pdf_processor.rs
use anyhow::Result;
use std::path::Path;
pub enum PdfProcessResult {
Markdown(String),
Scanned,
}
pub fn process_pdf(pdf_path: &Path) -> Result<PdfProcessResult> {
use pdf_inspector::process_pdf;
let result = process_pdf(pdf_path)?;
match result.pdf_type {
pdf_inspector::PdfType::TextBased | pdf_inspector::PdfType::Mixed => {
match &result.markdown {
Some(md) if !md.trim().is_empty() => Ok(PdfProcessResult::Markdown(md.clone())),
_ => Ok(PdfProcessResult::Scanned),
}
}
pdf_inspector::PdfType::Scanned | pdf_inspector::PdfType::ImageBased => {
Ok(PdfProcessResult::Scanned)
}
}
}
pub fn batch_convert_pdfs(dir: &Path) -> Result<(usize, usize, usize)> {
let mut text_based = 0usize;
let mut scanned = 0usize;
let mut errors = 0usize;
let entries: Vec<_> = std::fs::read_dir(dir)?
.filter_map(|e| e.ok())
.filter(|e| {
e.path()
.extension()
.is_some_and(|ext| ext.eq_ignore_ascii_case("pdf"))
})
.collect();
let total = entries.len();
println!("Found {} PDFs to process", total);
for (i, entry) in entries.iter().enumerate() {
let path = entry.path();
let filename = path.file_name().unwrap_or_default().to_string_lossy();
println!("[{}/{}] processing: {}", i + 1, total, filename);
match process_pdf(&path) {
Ok(PdfProcessResult::Markdown(md)) => {
let md_path = path.with_extension("md");
std::fs::write(&md_path, md)?;
std::fs::remove_file(&path)?;
println!(" text-based -> {}", md_path.display());
text_based += 1;
}
Ok(PdfProcessResult::Scanned) => {
println!(" scanned — keeping as PDF");
scanned += 1;
}
Err(e) => {
eprintln!(" error: {}", e);
errors += 1;
}
}
}
Ok((text_based, scanned, errors))
}
+72 -8
View File
@@ -58,11 +58,16 @@ fn path_matches(pattern: &str, path: &str) -> bool {
}
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
let robots_url = format!(
"{}://{}/robots.txt",
base_url.scheme(),
base_url.host_str().unwrap_or("")
);
let host_str = base_url.host_str().unwrap_or("");
if host_str.is_empty() {
log::warn!("No host in base URL, skipping robots.txt");
return RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
}
let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str);
let mut rule = RobotsRule {
allowed: Vec::new(),
@@ -82,13 +87,13 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
}
}
Ok(resp) => {
println!(
log::warn!(
"robots.txt returned HTTP {} — assuming no restrictions",
resp.status()
);
}
Err(e) => {
println!(
log::warn!(
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
@@ -149,7 +154,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
}
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
println!(
log::info!(
"robots.txt: {} disallow rules, {} allow rules",
rule.disallowed.len(),
rule.allowed.len()
@@ -158,3 +163,62 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
rule
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_path_matches_exact() {
assert!(path_matches("/admin", "/admin"));
assert!(path_matches("/admin", "/admin/users"));
assert!(path_matches("/admin", "/administrator"));
}
#[test]
fn test_path_matches_wildcard() {
assert!(path_matches("/private/*", "/private/data"));
assert!(path_matches("/private/*/secret", "/private/x/secret"));
assert!(!path_matches("/private/*/secret", "/private/secret"));
}
#[test]
fn test_path_matches_empty_pattern() {
assert!(!path_matches("", "/anything"));
}
#[test]
fn test_is_allowed_disallow_takes_precedence_when_longer() {
let rule = RobotsRule {
allowed: vec!["/a".into()],
disallowed: vec!["/a/b".into()],
};
assert!(!rule.is_allowed("/a/b"));
assert!(rule.is_allowed("/a/c"));
}
#[test]
fn test_is_allowed_allow_overrides_disallow_when_longer() {
let rule = RobotsRule {
allowed: vec!["/public/pages".into()],
disallowed: vec!["/public".into()],
};
assert!(rule.is_allowed("/public/pages"));
}
#[test]
fn test_parse_robots_txt_wildcard_agent() {
let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/admin".to_string()));
assert!(rule.allowed.contains(&"/admin/public".to_string()));
}
#[test]
fn test_parse_robots_txt_specific_agent() {
let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/private".to_string()));
assert!(!rule.disallowed.contains(&"/".to_string()));
}
}
+87 -82
View File
@@ -1,8 +1,8 @@
use crate::converter::html_to_markdown;
use crate::doc_processor::{DocProcessResult, try_convert};
use crate::error_logger::log_error;
use crate::extractor::extract_links;
use crate::fetcher::{FetchResult, Fetcher};
use crate::pdf_processor::{PdfProcessResult, process_pdf};
use crate::robots::RobotsRule;
use crate::url_utils::{normalize_url, url_to_filename};
use anyhow::Result;
@@ -13,17 +13,17 @@ use std::time::Duration;
use tokio::sync::Mutex;
use url::Url;
pub struct PdfStats {
pub text_based: usize,
pub scanned: usize,
pub struct DocStats {
pub converted: usize,
pub raw: usize,
pub errors: usize,
}
impl Default for PdfStats {
impl Default for DocStats {
fn default() -> Self {
Self {
text_based: 0,
scanned: 0,
converted: 0,
raw: 0,
errors: 0,
}
}
@@ -36,8 +36,8 @@ pub struct Scraper {
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
pdf_stats: Arc<Mutex<PdfStats>>,
convert_pdfs: bool,
doc_stats: Arc<Mutex<DocStats>>,
convert_docs: bool,
}
impl Scraper {
@@ -46,17 +46,18 @@ impl Scraper {
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
convert_pdfs: bool,
convert_docs: bool,
fetcher: Fetcher,
) -> Self {
Self {
fetcher: Fetcher::new(),
fetcher,
seen: Arc::new(Mutex::new(HashSet::new())),
output_dir,
log_path,
base_domain,
robots,
pdf_stats: Arc::new(Mutex::new(PdfStats::default())),
convert_pdfs,
doc_stats: Arc::new(Mutex::new(DocStats::default())),
convert_docs,
}
}
@@ -85,62 +86,58 @@ impl Scraper {
}
if !self.robots.is_allowed(url.path()) {
println!("[skip] robots.txt disallows: {}", url);
log::info!("[skip] robots.txt disallows: {}", url);
skipped_robots += 1;
continue;
}
println!("[{}] fetching: {}", count, url);
log::info!("[{}] fetching: {}", count, url);
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
let final_url = result.final_url.clone();
let is_html = result.content_type.contains("text/html");
let is_pdf = result.content_type.contains("application/pdf");
match self.save(&result, is_html, is_pdf).await {
match self.save(&result, is_html).await {
Ok(()) => {
if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links =
extract_links(html, &result.final_url, &self.base_domain);
let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let mut new_links = Vec::new();
{
let new_links: Vec<Url> = {
let seen = self.seen.lock().await;
for link in &links {
if !seen.contains(link.as_str()) {
new_links.push(link.clone());
}
}
links
.into_iter()
.filter(|link| !seen.contains(link.as_str()))
.collect()
};
for link in &new_links {
queue.push_back(link.clone());
}
let new_count = new_links.len();
for link in new_links {
queue.push_back(link);
}
println!(" found {} links ({} new)", links.len(), new_count);
log::info!(
" found {} links ({} new)",
new_count,
new_links.len()
);
}
} else {
println!(" binary: {}", result.content_type);
log::info!(" binary: {}", result.content_type);
}
}
Err(e) => {
log_error(
&self.log_path,
&result.final_url,
&final_url,
"save_error",
&e.to_string(),
None,
);
error_count += 1;
if is_pdf {
let mut stats = self.pdf_stats.lock().await;
stats.errors += 1;
}
let mut stats = self.doc_stats.lock().await;
stats.errors += 1;
}
}
}
@@ -151,29 +148,28 @@ impl Scraper {
}
count += 1;
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await;
}
{
let stats = self.pdf_stats.lock().await;
if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 {
println!("\nPDF Statistics:");
println!(" Text-based (converted): {}", stats.text_based);
println!(" Scanned (kept as PDF): {}", stats.scanned);
println!(" Errors: {}", stats.errors);
let stats = self.doc_stats.lock().await;
if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 {
log::info!("\nDocument Statistics:");
log::info!(" Converted to Markdown: {}", stats.converted);
log::info!(" Kept as-is: {}", stats.raw);
log::info!(" Errors: {}", stats.errors);
}
}
if skipped_robots > 0 {
println!("Skipped {} URLs due to robots.txt", skipped_robots);
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
}
(count, error_count)
}
async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> {
async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
if is_html {
@@ -183,51 +179,60 @@ impl Scraper {
filename = filename.replace(".html", ".md");
}
content = md.into_bytes();
} else if is_pdf && self.convert_pdfs {
// Save PDF to temp file for pdf-inspector processing
let temp_path = self.output_dir.join("temp_pdf.tmp");
fs::write(&temp_path, &result.bytes)?;
match process_pdf(&temp_path) {
Ok(PdfProcessResult::Markdown(md)) => {
if filename.ends_with(".pdf") {
filename = filename.replace(".pdf", ".md");
} else if self.convert_docs {
match try_convert(&result.bytes, &result.final_url) {
DocProcessResult::Markdown(md) => {
if let Some(pos) = filename.rfind('.') {
filename.truncate(pos);
}
filename.push_str(".md");
content = md.into_bytes();
fs::remove_file(&temp_path)?;
println!(" [PDF] text-based — converted to Markdown");
let mut stats = self.pdf_stats.lock().await;
stats.text_based += 1;
log::info!(" [DOC] converted to Markdown");
let mut stats = self.doc_stats.lock().await;
stats.converted += 1;
}
Ok(PdfProcessResult::Scanned) => {
println!(" [PDF] scanned document — saved as-is");
let mut stats = self.pdf_stats.lock().await;
stats.scanned += 1;
fs::rename(&temp_path, self.output_dir.join(&filename))?;
return Ok(());
}
Err(e) => {
println!(" [PDF] processing error ({}), saved as-is", e);
let mut stats = self.pdf_stats.lock().await;
stats.errors += 1;
fs::rename(&temp_path, self.output_dir.join(&filename))?;
return Ok(());
DocProcessResult::Raw => {
content = result.bytes.clone();
if is_document_content_type(&result.content_type) {
let mut stats = self.doc_stats.lock().await;
stats.raw += 1;
}
}
}
} else {
// Non-HTML, non-PDF, or PDF conversion disabled — save as-is
content = result.bytes.clone();
}
let filepath = self.output_dir.join(&filename);
fs::write(&filepath, &content)?;
println!(
log::info!(
" saved -> {}",
filepath.file_name().unwrap_or_default().to_string_lossy()
);
Ok(())
}
}
fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
DOC_TYPES.contains(&ct.as_str())
}
+143 -10
View File
@@ -14,7 +14,12 @@ pub fn derive_base_domain(url: &Url) -> Option<String> {
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
match url.host_str() {
Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)),
Some(host) => {
host == base_domain
|| host
.strip_suffix(base_domain)
.is_some_and(|rest| rest.ends_with('.'))
}
None => false,
}
}
@@ -50,6 +55,7 @@ fn get_extension(url: &Url, content_type: &str) -> String {
"application/json" => ".json",
"application/xml" | "text/xml" => ".xml",
"text/plain" => ".txt",
"text/csv" => ".csv",
"text/css" => ".css",
"application/javascript" | "text/javascript" => ".js",
"image/png" => ".png",
@@ -57,10 +63,21 @@ fn get_extension(url: &Url, content_type: &str) -> String {
"image/gif" => ".gif",
"image/svg+xml" => ".svg",
"image/webp" => ".webp",
"application/vnd.ms-excel" => ".xls",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
"application/rtf" => ".rtf",
"application/epub+zip" => ".epub",
"application/msword" => ".doc",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx",
"application/vnd.ms-word.document.macroenabled.12" => ".docm",
"application/vnd.ms-powerpoint" => ".ppt",
"application/vnd.openxmlformats-officedocument.presentationml.presentation" => ".pptx",
"application/vnd.ms-powerpoint.presentation.macroenabled.12" => ".pptm",
"application/vnd.ms-excel" => ".xls",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
"application/vnd.ms-excel.sheet.macroenabled.12" => ".xlsm",
"application/vnd.ms-excel.sheet.binary.macroenabled.12" => ".xlsb",
"application/vnd.oasis.opendocument.text" => ".odt",
"application/vnd.oasis.opendocument.spreadsheet" => ".ods",
"application/vnd.oasis.opendocument.presentation" => ".odp",
"application/zip" => ".zip",
"application/octet-stream" => ".bin",
_ => ".bin",
@@ -84,14 +101,10 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
.unwrap_or("unknown")
.replace('.', "_")
.replace(':', "_");
let path = url.path().trim_start_matches('/');
let path = if path.is_empty() { "index" } else { path };
// FIX: strip trailing slash so /support/ -> "support" not ""
let path = path.trim_end_matches('/');
let path = if path.is_empty() { "index" } else { path };
let last_segment = path.rsplit('/').next().unwrap_or(path);
let (stem, ext_from_path) = match last_segment.rfind('.') {
@@ -114,9 +127,11 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
_ => String::new(),
};
let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext);
let mut hasher = Sha256::new();
hasher.update(url_str.as_bytes());
let url_hash = hex::encode(hasher.finalize());
filename
let safe_stem: String = stem
.chars()
.map(|c| {
if c.is_alphanumeric() || "._-".contains(c) {
@@ -125,5 +140,123 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
'_'
}
})
.collect()
.collect();
format!(
"{}_{}{}{:}{}",
host,
safe_stem,
query_suffix,
&url_hash[..8],
ext
)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_derive_base_domain() {
let url = Url::parse("https://www.logting.fo/").unwrap();
assert_eq!(derive_base_domain(&url), Some("logting.fo".into()));
let url = Url::parse("https://taks.fo/en/skatur/").unwrap();
assert_eq!(derive_base_domain(&url), Some("taks.fo".into()));
let url = Url::parse("https://localhost:8080/").unwrap();
assert_eq!(derive_base_domain(&url), Some("localhost".into()));
}
#[test]
fn test_is_internal_subdomain() {
let base = "logting.fo";
assert!(is_internal(
&Url::parse("https://logting.fo/").unwrap(),
base
));
assert!(is_internal(
&Url::parse("https://www.logting.fo/").unwrap(),
base
));
assert!(is_internal(
&Url::parse("https://sub.www.logting.fo/").unwrap(),
base
));
assert!(!is_internal(
&Url::parse("https://evilogting.fo/").unwrap(),
base
));
assert!(!is_internal(
&Url::parse("https://logting.com/").unwrap(),
base
));
}
#[test]
fn test_normalize_url_strips_fragment() {
let url = normalize_url("https://example.com/page#section").unwrap();
assert_eq!(url.as_str(), "https://example.com/page");
}
#[test]
fn test_normalize_url_strips_trailing_query_chars() {
let url = normalize_url("https://example.com/page?&").unwrap();
assert_eq!(url.as_str(), "https://example.com/page");
}
#[test]
fn test_url_to_filename_faroese_unicode() {
let url = Url::parse("https://logting.fo/lov/tinglýsing/2024/").unwrap();
let filename = url_to_filename(&url, "text/html");
assert!(filename.starts_with("logting_fo_"));
assert!(filename.ends_with(".html") || filename.ends_with(".md"));
assert!(!filename.contains('ý'));
}
#[test]
fn test_url_to_filename_long_url_hashed() {
let long = format!("https://example.com/{}", "a".repeat(200));
let url = Url::parse(&long).unwrap();
let filename = url_to_filename(&url, "text/html");
assert!(filename.len() < 30);
assert!(filename.ends_with(".html"));
}
#[test]
fn test_url_to_filename_with_query() {
let url = Url::parse("https://example.com/page?id=42&sort=desc").unwrap();
let filename = url_to_filename(&url, "text/html");
assert!(filename.contains('_'));
assert!(filename.ends_with(".html"));
}
#[test]
fn test_get_extension_from_content_type() {
let url = Url::parse("https://example.com/").unwrap();
assert_eq!(get_extension(&url, "text/html"), ".html");
assert_eq!(
get_extension(&url, "application/pdf; charset=binary"),
".pdf"
);
assert_eq!(get_extension(&url, "application/octet-stream"), ".bin");
}
#[test]
fn test_get_extension_from_url_path() {
let url = Url::parse("https://example.com/doc.pdf").unwrap();
assert_eq!(get_extension(&url, "application/octet-stream"), ".pdf");
}
#[test]
fn test_url_collision_avoidance() {
let url1 = Url::parse("https://example.com/foo-bar").unwrap();
let url2 = Url::parse("https://example.com/foo.bar").unwrap();
let fn1 = url_to_filename(&url1, "text/html");
let fn2 = url_to_filename(&url2, "text/html");
assert_ne!(
fn1, fn2,
"Different URLs should produce different filenames"
);
}
}