review and QA
This commit is contained in:
+94
-11
@@ -1,3 +1,5 @@
|
||||
#![warn(clippy::all, clippy::pedantic)]
|
||||
|
||||
mod config;
|
||||
mod converter;
|
||||
mod doc_processor;
|
||||
@@ -10,16 +12,19 @@ mod url_utils;
|
||||
|
||||
use crate::robots::fetch_robots;
|
||||
use crate::scraper::Scraper;
|
||||
use anyhow::Result;
|
||||
use anyhow::{Result, bail};
|
||||
use clap::Parser;
|
||||
use reqwest::Client;
|
||||
use std::collections::HashSet;
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use std::time::Duration;
|
||||
use url::Url;
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
#[command(name = "rs-scraper")]
|
||||
#[command(author = "FLÓ")]
|
||||
#[command(version = "0.1.0")]
|
||||
#[command(version = "0.2.0")]
|
||||
#[command(about = "Web scraper for Faroese public sector sites")]
|
||||
struct Args {
|
||||
#[arg(help = "Starting URL to scrape")]
|
||||
@@ -27,6 +32,34 @@ struct Args {
|
||||
|
||||
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
|
||||
no_doc_conversion: bool,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
help = "Comma-separated extensions to include (e.g. pdf,docx,html). \
|
||||
Only files matching these types are saved. \
|
||||
HTML pages are still crawled for links even if excluded. \
|
||||
Mutually exclusive with --exclude-types."
|
||||
)]
|
||||
types: Option<String>,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
help = "Comma-separated extensions to exclude (e.g. pdf,xlsx). \
|
||||
All file types are saved except these. \
|
||||
HTML pages are still crawled for links. \
|
||||
Mutually exclusive with --types."
|
||||
)]
|
||||
exclude_types: Option<String>,
|
||||
|
||||
#[arg(short = 'o', long, help = "Output directory for scraped files")]
|
||||
output: Option<String>,
|
||||
}
|
||||
|
||||
fn parse_ext_list(s: &str) -> HashSet<String> {
|
||||
s.split(',')
|
||||
.map(|s| s.trim().trim_start_matches('.').to_lowercase())
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[tokio::main]
|
||||
@@ -43,25 +76,72 @@ async fn main() -> Result<()> {
|
||||
let base_domain = url_utils::derive_base_domain(&url)
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
||||
|
||||
let output_dir_name = format!("scraped_{}", base_domain.replace('.', "_"));
|
||||
let output_dir = PathBuf::from(&output_dir_name);
|
||||
let logs_dir = output_dir.join("logs");
|
||||
let output_dir = match &args.output {
|
||||
Some(p) => PathBuf::from(p),
|
||||
None => {
|
||||
let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
|
||||
PathBuf::from(default_name)
|
||||
}
|
||||
};
|
||||
|
||||
let logs_dir = {
|
||||
let output_name = output_dir
|
||||
.file_name()
|
||||
.map(|n| n.to_string_lossy().into_owned())
|
||||
.unwrap_or_else(|| "output".to_string());
|
||||
let logs_name = format!("{}_logs", output_name);
|
||||
output_dir
|
||||
.parent()
|
||||
.map(|p| p.join(&logs_name))
|
||||
.unwrap_or_else(|| PathBuf::from(&logs_name))
|
||||
};
|
||||
|
||||
fs::create_dir_all(&output_dir)?;
|
||||
fs::create_dir_all(&logs_dir)?;
|
||||
|
||||
let log_path = logs_dir.join("scrape_errors.jsonl");
|
||||
fs::remove_file(&log_path).ok();
|
||||
|
||||
let (include_types, exclude_types) = match (&args.types, &args.exclude_types) {
|
||||
(Some(_), Some(_)) => {
|
||||
bail!("--types and --exclude-types are mutually exclusive");
|
||||
}
|
||||
(Some(t), None) => (Some(parse_ext_list(t)), None),
|
||||
(None, Some(t)) => (None, Some(parse_ext_list(t))),
|
||||
(None, None) => (None, None),
|
||||
};
|
||||
|
||||
let scope_path = {
|
||||
let path = url.path().trim_end_matches('/');
|
||||
if path.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(path.to_string())
|
||||
}
|
||||
};
|
||||
|
||||
log::info!("Scraping: {}", start_url);
|
||||
log::info!("Base domain: {}", base_domain);
|
||||
log::info!("Output dir: {}", output_dir.display());
|
||||
log::info!("Logs dir: {}", logs_dir.display());
|
||||
log::info!(
|
||||
"Document conversion: {}",
|
||||
if convert_docs { "enabled" } else { "disabled" }
|
||||
);
|
||||
|
||||
let client = reqwest::Client::builder()
|
||||
.timeout(std::time::Duration::from_secs(30))
|
||||
match (&include_types, &exclude_types) {
|
||||
(Some(t), _) => log::info!("Type filter: include {:?}", t),
|
||||
(_, Some(t)) => log::info!("Type filter: exclude {:?}", t),
|
||||
_ => log::info!("Type filter: none (all types)"),
|
||||
}
|
||||
|
||||
match &scope_path {
|
||||
Some(s) => log::info!("Path scope: {} (only this path and deeper)", s),
|
||||
None => log::info!("Path scope: none (full site)"),
|
||||
}
|
||||
|
||||
let client = Client::builder()
|
||||
.timeout(Duration::from_secs(config::TIMEOUT_SECS))
|
||||
.user_agent(config::USER_AGENT)
|
||||
.build()?;
|
||||
|
||||
@@ -69,19 +149,22 @@ async fn main() -> Result<()> {
|
||||
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
|
||||
log::info!("");
|
||||
|
||||
let fetcher = fetcher::Fetcher::new()?;
|
||||
let scraper = Scraper::new(
|
||||
let fetcher = fetcher::Fetcher::new(client)?;
|
||||
let mut scraper = Scraper::new(
|
||||
output_dir,
|
||||
log_path,
|
||||
log_path.clone(),
|
||||
base_domain,
|
||||
robots,
|
||||
convert_docs,
|
||||
fetcher,
|
||||
include_types,
|
||||
exclude_types,
|
||||
scope_path,
|
||||
);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
log::info!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
|
||||
log::info!("Error log: {}", log_path.display());
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user