yeah works
This commit is contained in:
+52
@@ -0,0 +1,52 @@
|
||||
// src/main.rs
|
||||
|
||||
mod config;
|
||||
mod error_logger;
|
||||
mod extractor;
|
||||
mod fetcher;
|
||||
mod scraper;
|
||||
mod url_utils;
|
||||
|
||||
use scraper::Scraper;
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use url::Url;
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> anyhow::Result<()> {
|
||||
let args: Vec<String> = std::env::args().collect();
|
||||
|
||||
if args.len() < 2 {
|
||||
eprintln!("Usage: site-scraper <start-url>");
|
||||
eprintln!("Example: site-scraper https://www.logting.fo");
|
||||
std::process::exit(1);
|
||||
}
|
||||
|
||||
let start_url = &args[1];
|
||||
let url = Url::parse(start_url)?;
|
||||
|
||||
let base_domain = url_utils::derive_base_domain(&url)
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
||||
|
||||
let output_dir_name = format!("scraped_{}", base_domain.replace('.', "_"));
|
||||
let output_dir = PathBuf::from(&output_dir_name);
|
||||
let logs_dir = output_dir.join("logs");
|
||||
|
||||
fs::create_dir_all(&logs_dir)?;
|
||||
|
||||
let log_path = logs_dir.join("scrape_errors.jsonl");
|
||||
fs::remove_file(&log_path).ok();
|
||||
|
||||
println!("Scraping: {}", start_url);
|
||||
println!("Base domain: {}", base_domain);
|
||||
println!("Output dir: {}", output_dir.display());
|
||||
println!();
|
||||
|
||||
let scraper = Scraper::new(output_dir, log_path, base_domain);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
Reference in New Issue
Block a user