mod config; mod converter; mod error_logger; mod extractor; mod fetcher; mod pdf_processor; mod robots; mod scraper; mod url_utils; use crate::config::USER_AGENT; use crate::robots::fetch_robots; use crate::scraper::Scraper; use std::fs; use std::path::PathBuf; use url::Url; #[tokio::main] async fn main() -> anyhow::Result<()> { let args: Vec = std::env::args().collect(); // Parse flags let mut convert_pdfs = true; let mut url_arg: Option = None; for arg in args.iter().skip(1) { match arg.as_str() { "--no-pdf-conversion" => convert_pdfs = false, "--help" | "-h" => { println!("Usage: site-scraper [OPTIONS] "); println!(); println!("Options:"); println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction"); println!(" -h, --help Show this help message"); println!(); println!("Example:"); println!(" site-scraper https://www.logting.fo"); println!(" site-scraper --no-pdf-conversion https://taks.fo"); std::process::exit(0); } _ => { if url_arg.is_none() { url_arg = Some(arg.clone()); } } } } let start_url = match url_arg { Some(u) => u, None => { eprintln!("Usage: site-scraper [OPTIONS] "); eprintln!("Example: site-scraper https://www.logting.fo"); eprintln!("Run with --help for options."); std::process::exit(1); } }; let url = Url::parse(&start_url)?; let base_domain = url_utils::derive_base_domain(&url) .ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?; let output_dir_name = format!("scraped_{}", base_domain.replace('.', "_")); let output_dir = PathBuf::from(&output_dir_name); let logs_dir = output_dir.join("logs"); fs::create_dir_all(&logs_dir)?; let log_path = logs_dir.join("scrape_errors.jsonl"); fs::remove_file(&log_path).ok(); println!("Scraping: {}", start_url); println!("Base domain: {}", base_domain); println!("Output dir: {}", output_dir.display()); println!( "PDF conversion: {}", if convert_pdfs { "enabled" } else { "disabled" } ); // Fetch robots.txt let client = reqwest::Client::builder() .timeout(std::time::Duration::from_secs(30)) .user_agent(USER_AGENT) .build()?; println!("Fetching robots.txt..."); let robots = fetch_robots(&client, &url, "web-scraper").await; println!(); let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs); let (count, errors) = scraper.run(&url).await; println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors); println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name); Ok(()) }