#![warn(clippy::all, clippy::pedantic)] mod config; mod converter; mod doc_processor; mod error_logger; mod extractor; mod fetcher; mod robots; mod scraper; mod tui; mod url_utils; use crate::robots::fetch_robots; use crate::scraper::Scraper; use crate::tui::TuiConfig; use anyhow::{Result, bail}; use clap::Parser; use reqwest::Client; use std::collections::HashSet; use std::fs; use std::path::{Path, PathBuf}; use std::time::Duration; use url::Url; #[derive(Parser, Debug)] #[command(name = "rs-scraper")] #[command(author = "FLÓ")] #[command(version = "0.2.0")] #[command(about = "Web scraper for Faroese public sector sites")] #[allow(clippy::struct_excessive_bools)] struct Args { #[arg(help = "Starting URL to scrape")] start_url: Option, #[arg( long = "interactive", short = 'i', help = "Launch interactive TUI configuration instead of CLI" )] interactive: bool, #[arg( long, help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)" )] subdomains: bool, #[arg( long, help = "Only scrape the single page at START_URL, do not crawl for links" )] single: bool, #[arg(long, help = "Disable path scoping — crawl any path on the same host")] no_scope: bool, #[arg(long, help = "Save documents as-is, skip Markdown conversion")] no_doc_conversion: bool, #[arg( long, default_value = "1000", help = "Delay in milliseconds between requests (default: 1000)" )] delay_ms: u64, #[arg( long, help = "Comma-separated extensions to include (e.g. pdf,docx,html). \ Only files matching these types are saved. \ HTML pages are still crawled for links even if excluded. \ Mutually exclusive with --exclude-types." )] types: Option, #[arg( long, help = "Comma-separated extensions to exclude (e.g. pdf,xlsx). \ All file types are saved except these. \ HTML pages are still crawled for links. \ Mutually exclusive with --types." )] exclude_types: Option, #[arg(short = 'o', long, help = "Output directory for scraped files")] output: Option, } fn parse_ext_list(s: &str) -> HashSet { s.split(',') .map(|s| s.trim().trim_start_matches('.').to_lowercase()) .filter(|s| !s.is_empty()) .collect() } #[allow(clippy::too_many_arguments)] fn print_config( start_url: &str, base_host: &str, subdomains: bool, output_dir: &Path, logs_dir: &Path, convert_docs: bool, single: bool, delay_ms: u64, include_types: Option<&HashSet>, exclude_types: Option<&HashSet>, scope_path: Option<&String>, ) { log::info!("Scraping: {start_url}"); log::info!("Base host: {base_host}"); log::info!("Subdomain crawling: {}", if subdomains { "enabled" } else { "disabled" }); log::info!("Output dir: {}", output_dir.display()); log::info!("Logs dir: {}", logs_dir.display()); log::info!("Document conversion: {}", if convert_docs { "enabled" } else { "disabled" }); log::info!("Single page mode: {}", if single { "enabled" } else { "disabled" }); log::info!("Request delay: {delay_ms} ms"); match (include_types, exclude_types) { (Some(t), _) => log::info!("Type filter: include {t:?}"), (_, Some(t)) => log::info!("Type filter: exclude {t:?}"), _ => log::info!("Type filter: none (all types)"), } match scope_path { Some(s) => log::info!("Path scope: {s} (only this path and deeper)"), None => log::info!("Path scope: none (full site)"), } } fn build_config_from_args(args: &Args) -> Result<(String, TuiConfig)> { let url_str = args.start_url.as_ref() .ok_or_else(|| anyhow::anyhow!("START_URL is required when not using --interactive"))?; let tui_config = TuiConfig { start_url: url_str.clone(), output_dir: args.output.clone().unwrap_or_default(), subdomains: args.subdomains, single: args.single, no_scope: args.no_scope, no_doc_conversion: args.no_doc_conversion, delay_ms: args.delay_ms, types: args.types.clone(), exclude_types: args.exclude_types.clone(), }; Ok((url_str.clone(), tui_config)) } #[allow(clippy::too_many_arguments)] #[allow(clippy::fn_params_excessive_bools)] async fn run_scraper( start_url: &str, subdomains: bool, single: bool, no_scope: bool, no_doc_conversion: bool, delay_ms: u64, types: Option<&str>, exclude_types: Option<&str>, output: Option<&str>, ) -> Result<()> { let url = Url::parse(start_url)?; let base_host = url.host_str() .ok_or_else(|| anyhow::anyhow!("Could not parse hostname"))? .to_string(); let base_domain = url_utils::derive_base_domain(&url) .ok_or_else(|| anyhow::anyhow!("Could not parse base domain"))?; let output_dir = match output { Some(p) if !p.is_empty() => PathBuf::from(p), _ => PathBuf::from(format!("{}_scraped", base_domain.replace('.', "_"))), }; let logs_dir = { let output_name = output_dir.file_name().map_or_else( || "output".to_string(), |n| n.to_string_lossy().into_owned(), ); let logs_name = format!("{output_name}_logs"); output_dir.parent() .map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name)) }; fs::create_dir_all(&output_dir)?; fs::create_dir_all(&logs_dir)?; let log_path = logs_dir.join("scrape_errors.jsonl"); fs::remove_file(&log_path).ok(); let (include_types, exclude_types) = match (types, exclude_types) { (Some(_), Some(_)) => bail!("--types and --exclude-types are mutually exclusive"), (Some(t), None) => (Some(parse_ext_list(t)), None), (None, Some(t)) => (None, Some(parse_ext_list(t))), (None, None) => (None, None), }; let scope_path = if no_scope { None } else { let path = url.path().trim_end_matches('/'); if path.is_empty() { None } else { Some(path.to_string()) } }; print_config( start_url, &base_host, subdomains, &output_dir, &logs_dir, !no_doc_conversion, single, delay_ms, include_types.as_ref(), exclude_types.as_ref(), scope_path.as_ref(), ); let client = Client::builder() .timeout(Duration::from_secs(config::TIMEOUT_SECS)) .user_agent(config::USER_AGENT) .build()?; log::info!("Fetching robots.txt..."); let robots = fetch_robots(&client, &url, config::USER_AGENT).await; log::info!(""); let fetcher = fetcher::Fetcher::new(client); let mut scraper = Scraper::new( output_dir, log_path.clone(), base_host, robots, !no_doc_conversion, fetcher, include_types, exclude_types, scope_path, single, subdomains, delay_ms, ); let (count, errors) = scraper.run(&url).await; log::info!("\nDone. Fetched {count} URLs. Errors: {errors}."); log::info!("Error log: {}", log_path.display()); Ok(()) } #[tokio::main] async fn main() -> Result<()> { env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init(); let args = Args::parse(); let cli_command; if args.interactive { log::info!("Launching interactive TUI..."); if let Some(config) = tui::run_tui()? { cli_command = config.build_command(); log::info!("Running: {cli_command}"); run_scraper( &config.start_url, config.subdomains, config.single, config.no_scope, config.no_doc_conversion, config.delay_ms, config.types.as_deref(), config.exclude_types.as_deref(), if config.output_dir.is_empty() { None } else { Some(&config.output_dir) }, ).await?; } else { log::info!("Cancelled by user."); return Ok(()); } } else { let (start_url, tui_config) = build_config_from_args(&args)?; cli_command = tui_config.build_command(); log::info!("Running: {cli_command}"); run_scraper( &start_url, args.subdomains, args.single, args.no_scope, args.no_doc_conversion, args.delay_ms, args.types.as_deref(), args.exclude_types.as_deref(), args.output.as_deref(), ).await?; } eprintln!("Command used: {cli_command}"); Ok(()) }