Files
web-scraper-rust/src/main.rs
T
2026-08-25 12:17:17 +01:00

195 lines
5.5 KiB
Rust

#![warn(clippy::all, clippy::pedantic)]
mod config;
mod converter;
mod doc_processor;
mod error_logger;
mod extractor;
mod fetcher;
mod robots;
mod scraper;
mod url_utils;
use crate::robots::fetch_robots;
use crate::scraper::Scraper;
use anyhow::{Result, bail};
use clap::Parser;
use reqwest::Client;
use std::collections::HashSet;
use std::fs;
use std::path::PathBuf;
use std::time::Duration;
use url::Url;
#[derive(Parser, Debug)]
#[command(name = "rs-scraper")]
#[command(author = "FLÓ")]
#[command(version = "0.2.0")]
#[command(about = "Web scraper for Faroese public sector sites")]
struct Args {
#[arg(help = "Starting URL to scrape")]
start_url: String,
#[arg(
long,
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
)]
subdomains: bool,
#[arg(
long,
help = "Only scrape the single page at START_URL, do not crawl for links"
)]
single: bool,
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
no_doc_conversion: bool,
#[arg(
long,
help = "Comma-separated extensions to include (e.g. pdf,docx,html). \
Only files matching these types are saved. \
HTML pages are still crawled for links even if excluded. \
Mutually exclusive with --exclude-types."
)]
types: Option<String>,
#[arg(
long,
help = "Comma-separated extensions to exclude (e.g. pdf,xlsx). \
All file types are saved except these. \
HTML pages are still crawled for links. \
Mutually exclusive with --types."
)]
exclude_types: Option<String>,
#[arg(short = 'o', long, help = "Output directory for scraped files")]
output: Option<String>,
}
fn parse_ext_list(s: &str) -> HashSet<String> {
s.split(',')
.map(|s| s.trim().trim_start_matches('.').to_lowercase())
.filter(|s| !s.is_empty())
.collect()
}
#[tokio::main]
async fn main() -> Result<()> {
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
let args = Args::parse();
let start_url = args.start_url;
let convert_docs = !args.no_doc_conversion;
let url = Url::parse(&start_url)?;
let base_host = url
.host_str()
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?
.to_string();
let base_domain = url_utils::derive_base_domain(&url)
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
let output_dir = if let Some(p) = &args.output {
PathBuf::from(p)
} else {
let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
PathBuf::from(default_name)
};
let logs_dir = {
let output_name = output_dir
.file_name()
.map_or_else(|| "output".to_string(), |n| n.to_string_lossy().into_owned());
let logs_name = format!("{output_name}_logs");
output_dir
.parent()
.map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
};
fs::create_dir_all(&output_dir)?;
fs::create_dir_all(&logs_dir)?;
let log_path = logs_dir.join("scrape_errors.jsonl");
fs::remove_file(&log_path).ok();
let (include_types, exclude_types) = match (&args.types, &args.exclude_types) {
(Some(_), Some(_)) => {
bail!("--types and --exclude-types are mutually exclusive");
}
(Some(t), None) => (Some(parse_ext_list(t)), None),
(None, Some(t)) => (None, Some(parse_ext_list(t))),
(None, None) => (None, None),
};
let scope_path = {
let path = url.path().trim_end_matches('/');
if path.is_empty() {
None
} else {
Some(path.to_string())
}
};
log::info!("Scraping: {start_url}");
log::info!("Base host: {base_host}");
log::info!(
"Subdomain crawling: {}",
if args.subdomains { "enabled" } else { "disabled" }
);
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!(
"Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" }
);
log::info!(
"Single page mode: {}",
if args.single { "enabled" } else { "disabled" }
);
match (&include_types, &exclude_types) {
(Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
match &scope_path {
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
let client = Client::builder()
.timeout(Duration::from_secs(config::TIMEOUT_SECS))
.user_agent(config::USER_AGENT)
.build()?;
log::info!("Fetching robots.txt...");
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
log::info!("");
let fetcher = fetcher::Fetcher::new(client);
let mut scraper = Scraper::new(
output_dir,
log_path.clone(),
base_host,
robots,
convert_docs,
fetcher,
include_types,
exclude_types,
scope_path,
args.single,
args.subdomains,
);
let (count, errors) = scraper.run(&url).await;
log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
log::info!("Error log: {}", log_path.display());
Ok(())
}