qa and review
This commit is contained in:
+42
-53
@@ -1,60 +1,42 @@
|
||||
mod config;
|
||||
mod converter;
|
||||
mod doc_processor;
|
||||
mod error_logger;
|
||||
mod extractor;
|
||||
mod fetcher;
|
||||
mod pdf_processor;
|
||||
mod robots;
|
||||
mod scraper;
|
||||
mod url_utils;
|
||||
|
||||
use crate::config::USER_AGENT;
|
||||
use crate::robots::fetch_robots;
|
||||
use crate::scraper::Scraper;
|
||||
use anyhow::Result;
|
||||
use clap::Parser;
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use url::Url;
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
#[command(name = "rs-scraper")]
|
||||
#[command(author = "FLÓ")]
|
||||
#[command(version = "0.1.0")]
|
||||
#[command(about = "Web scraper for Faroese public sector sites")]
|
||||
struct Args {
|
||||
#[arg(help = "Starting URL to scrape")]
|
||||
start_url: String,
|
||||
|
||||
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
|
||||
no_doc_conversion: bool,
|
||||
}
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> anyhow::Result<()> {
|
||||
let args: Vec<String> = std::env::args().collect();
|
||||
async fn main() -> Result<()> {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
|
||||
|
||||
// Parse flags
|
||||
let mut convert_pdfs = true;
|
||||
let mut url_arg: Option<String> = None;
|
||||
let args = Args::parse();
|
||||
|
||||
for arg in args.iter().skip(1) {
|
||||
match arg.as_str() {
|
||||
"--no-pdf-conversion" => convert_pdfs = false,
|
||||
"--help" | "-h" => {
|
||||
println!("Usage: site-scraper [OPTIONS] <start-url>");
|
||||
println!();
|
||||
println!("Options:");
|
||||
println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction");
|
||||
println!(" -h, --help Show this help message");
|
||||
println!();
|
||||
println!("Example:");
|
||||
println!(" site-scraper https://www.logting.fo");
|
||||
println!(" site-scraper --no-pdf-conversion https://taks.fo");
|
||||
std::process::exit(0);
|
||||
}
|
||||
_ => {
|
||||
if url_arg.is_none() {
|
||||
url_arg = Some(arg.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let start_url = match url_arg {
|
||||
Some(u) => u,
|
||||
None => {
|
||||
eprintln!("Usage: site-scraper [OPTIONS] <start-url>");
|
||||
eprintln!("Example: site-scraper https://www.logting.fo");
|
||||
eprintln!("Run with --help for options.");
|
||||
std::process::exit(1);
|
||||
}
|
||||
};
|
||||
let start_url = args.start_url;
|
||||
let convert_docs = !args.no_doc_conversion;
|
||||
|
||||
let url = Url::parse(&start_url)?;
|
||||
|
||||
@@ -70,29 +52,36 @@ async fn main() -> anyhow::Result<()> {
|
||||
let log_path = logs_dir.join("scrape_errors.jsonl");
|
||||
fs::remove_file(&log_path).ok();
|
||||
|
||||
println!("Scraping: {}", start_url);
|
||||
println!("Base domain: {}", base_domain);
|
||||
println!("Output dir: {}", output_dir.display());
|
||||
println!(
|
||||
"PDF conversion: {}",
|
||||
if convert_pdfs { "enabled" } else { "disabled" }
|
||||
log::info!("Scraping: {}", start_url);
|
||||
log::info!("Base domain: {}", base_domain);
|
||||
log::info!("Output dir: {}", output_dir.display());
|
||||
log::info!(
|
||||
"Document conversion: {}",
|
||||
if convert_docs { "enabled" } else { "disabled" }
|
||||
);
|
||||
|
||||
// Fetch robots.txt
|
||||
let client = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(30))
|
||||
.user_agent(USER_AGENT)
|
||||
.user_agent(config::USER_AGENT)
|
||||
.build()?;
|
||||
|
||||
println!("Fetching robots.txt...");
|
||||
let robots = fetch_robots(&client, &url, "web-scraper").await;
|
||||
println!();
|
||||
log::info!("Fetching robots.txt...");
|
||||
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
|
||||
log::info!("");
|
||||
|
||||
let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs);
|
||||
let fetcher = fetcher::Fetcher::new()?;
|
||||
let scraper = Scraper::new(
|
||||
output_dir,
|
||||
log_path,
|
||||
base_domain,
|
||||
robots,
|
||||
convert_docs,
|
||||
fetcher,
|
||||
);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
log::info!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user