306 lines
8.7 KiB
Rust
306 lines
8.7 KiB
Rust
#![warn(clippy::all, clippy::pedantic)]
|
|
|
|
mod config;
|
|
mod converter;
|
|
mod doc_processor;
|
|
mod error_logger;
|
|
mod extractor;
|
|
mod fetcher;
|
|
mod robots;
|
|
mod scraper;
|
|
mod tui;
|
|
mod url_utils;
|
|
|
|
use crate::robots::fetch_robots;
|
|
use crate::scraper::Scraper;
|
|
use crate::tui::TuiConfig;
|
|
use anyhow::{Result, bail};
|
|
use clap::Parser;
|
|
use reqwest::Client;
|
|
use std::collections::HashSet;
|
|
use std::fs;
|
|
use std::path::{Path, PathBuf};
|
|
use std::time::Duration;
|
|
use url::Url;
|
|
|
|
#[derive(Parser, Debug)]
|
|
#[command(name = "rs-scraper")]
|
|
#[command(author = "FLÓ")]
|
|
#[command(version = "0.2.0")]
|
|
#[command(about = "Web scraper for Faroese public sector sites")]
|
|
#[allow(clippy::struct_excessive_bools)]
|
|
struct Args {
|
|
#[arg(help = "Starting URL to scrape")]
|
|
start_url: Option<String>,
|
|
|
|
#[arg(
|
|
long = "interactive",
|
|
short = 'i',
|
|
help = "Launch interactive TUI configuration instead of CLI"
|
|
)]
|
|
interactive: bool,
|
|
|
|
#[arg(
|
|
long,
|
|
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
|
|
)]
|
|
subdomains: bool,
|
|
|
|
#[arg(
|
|
long,
|
|
help = "Only scrape the single page at START_URL, do not crawl for links"
|
|
)]
|
|
single: bool,
|
|
|
|
#[arg(long, help = "Disable path scoping — crawl any path on the same host")]
|
|
no_scope: bool,
|
|
|
|
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
|
|
no_doc_conversion: bool,
|
|
|
|
#[arg(
|
|
long,
|
|
default_value = "1000",
|
|
help = "Delay in milliseconds between requests (default: 1000)"
|
|
)]
|
|
delay_ms: u64,
|
|
|
|
#[arg(
|
|
long,
|
|
help = "Comma-separated extensions to include (e.g. pdf,docx,html). \
|
|
Only files matching these types are saved. \
|
|
HTML pages are still crawled for links even if excluded. \
|
|
Mutually exclusive with --exclude-types."
|
|
)]
|
|
types: Option<String>,
|
|
|
|
#[arg(
|
|
long,
|
|
help = "Comma-separated extensions to exclude (e.g. pdf,xlsx). \
|
|
All file types are saved except these. \
|
|
HTML pages are still crawled for links. \
|
|
Mutually exclusive with --types."
|
|
)]
|
|
exclude_types: Option<String>,
|
|
|
|
#[arg(short = 'o', long, help = "Output directory for scraped files")]
|
|
output: Option<String>,
|
|
}
|
|
|
|
fn parse_ext_list(s: &str) -> HashSet<String> {
|
|
s.split(',')
|
|
.map(|s| s.trim().trim_start_matches('.').to_lowercase())
|
|
.filter(|s| !s.is_empty())
|
|
.collect()
|
|
}
|
|
|
|
#[allow(clippy::too_many_arguments)]
|
|
fn print_config(
|
|
start_url: &str,
|
|
base_host: &str,
|
|
subdomains: bool,
|
|
output_dir: &Path,
|
|
logs_dir: &Path,
|
|
convert_docs: bool,
|
|
single: bool,
|
|
delay_ms: u64,
|
|
include_types: Option<&HashSet<String>>,
|
|
exclude_types: Option<&HashSet<String>>,
|
|
scope_path: Option<&String>,
|
|
) {
|
|
log::info!("Scraping: {start_url}");
|
|
log::info!("Base host: {base_host}");
|
|
log::info!("Subdomain crawling: {}", if subdomains { "enabled" } else { "disabled" });
|
|
log::info!("Output dir: {}", output_dir.display());
|
|
log::info!("Logs dir: {}", logs_dir.display());
|
|
log::info!("Document conversion: {}", if convert_docs { "enabled" } else { "disabled" });
|
|
log::info!("Single page mode: {}", if single { "enabled" } else { "disabled" });
|
|
log::info!("Request delay: {delay_ms} ms");
|
|
|
|
match (include_types, exclude_types) {
|
|
(Some(t), _) => log::info!("Type filter: include {t:?}"),
|
|
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
|
|
_ => log::info!("Type filter: none (all types)"),
|
|
}
|
|
|
|
match scope_path {
|
|
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
|
|
None => log::info!("Path scope: none (full site)"),
|
|
}
|
|
}
|
|
|
|
fn build_config_from_args(args: &Args) -> Result<(String, TuiConfig)> {
|
|
let url_str = args.start_url.as_ref()
|
|
.ok_or_else(|| anyhow::anyhow!("START_URL is required when not using --interactive"))?;
|
|
|
|
let tui_config = TuiConfig {
|
|
start_url: url_str.clone(),
|
|
output_dir: args.output.clone().unwrap_or_default(),
|
|
subdomains: args.subdomains,
|
|
single: args.single,
|
|
no_scope: args.no_scope,
|
|
no_doc_conversion: args.no_doc_conversion,
|
|
delay_ms: args.delay_ms,
|
|
types: args.types.clone(),
|
|
exclude_types: args.exclude_types.clone(),
|
|
};
|
|
|
|
Ok((url_str.clone(), tui_config))
|
|
}
|
|
|
|
#[allow(clippy::too_many_arguments)]
|
|
#[allow(clippy::fn_params_excessive_bools)]
|
|
async fn run_scraper(
|
|
start_url: &str,
|
|
subdomains: bool,
|
|
single: bool,
|
|
no_scope: bool,
|
|
no_doc_conversion: bool,
|
|
delay_ms: u64,
|
|
types: Option<&str>,
|
|
exclude_types: Option<&str>,
|
|
output: Option<&str>,
|
|
) -> Result<()> {
|
|
let url = Url::parse(start_url)?;
|
|
|
|
let base_host = url.host_str()
|
|
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname"))?
|
|
.to_string();
|
|
|
|
let base_domain = url_utils::derive_base_domain(&url)
|
|
.ok_or_else(|| anyhow::anyhow!("Could not parse base domain"))?;
|
|
|
|
let output_dir = match output {
|
|
Some(p) if !p.is_empty() => PathBuf::from(p),
|
|
_ => PathBuf::from(format!("{}_scraped", base_domain.replace('.', "_"))),
|
|
};
|
|
|
|
let logs_dir = {
|
|
let output_name = output_dir.file_name().map_or_else(
|
|
|| "output".to_string(),
|
|
|n| n.to_string_lossy().into_owned(),
|
|
);
|
|
let logs_name = format!("{output_name}_logs");
|
|
output_dir.parent()
|
|
.map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
|
|
};
|
|
|
|
fs::create_dir_all(&output_dir)?;
|
|
fs::create_dir_all(&logs_dir)?;
|
|
|
|
let log_path = logs_dir.join("scrape_errors.jsonl");
|
|
fs::remove_file(&log_path).ok();
|
|
|
|
let (include_types, exclude_types) = match (types, exclude_types) {
|
|
(Some(_), Some(_)) => bail!("--types and --exclude-types are mutually exclusive"),
|
|
(Some(t), None) => (Some(parse_ext_list(t)), None),
|
|
(None, Some(t)) => (None, Some(parse_ext_list(t))),
|
|
(None, None) => (None, None),
|
|
};
|
|
|
|
let scope_path = if no_scope {
|
|
None
|
|
} else {
|
|
let path = url.path().trim_end_matches('/');
|
|
if path.is_empty() { None } else { Some(path.to_string()) }
|
|
};
|
|
|
|
print_config(
|
|
start_url,
|
|
&base_host,
|
|
subdomains,
|
|
&output_dir,
|
|
&logs_dir,
|
|
!no_doc_conversion,
|
|
single,
|
|
delay_ms,
|
|
include_types.as_ref(),
|
|
exclude_types.as_ref(),
|
|
scope_path.as_ref(),
|
|
);
|
|
|
|
let client = Client::builder()
|
|
.timeout(Duration::from_secs(config::TIMEOUT_SECS))
|
|
.user_agent(config::USER_AGENT)
|
|
.build()?;
|
|
|
|
log::info!("Fetching robots.txt...");
|
|
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
|
|
log::info!("");
|
|
|
|
let fetcher = fetcher::Fetcher::new(client);
|
|
let mut scraper = Scraper::new(
|
|
output_dir,
|
|
log_path.clone(),
|
|
base_host,
|
|
robots,
|
|
!no_doc_conversion,
|
|
fetcher,
|
|
include_types,
|
|
exclude_types,
|
|
scope_path,
|
|
single,
|
|
subdomains,
|
|
delay_ms,
|
|
);
|
|
|
|
let (count, errors) = scraper.run(&url).await;
|
|
log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
|
|
log::info!("Error log: {}", log_path.display());
|
|
|
|
Ok(())
|
|
}
|
|
|
|
#[tokio::main]
|
|
async fn main() -> Result<()> {
|
|
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
|
|
|
|
let args = Args::parse();
|
|
|
|
let cli_command;
|
|
|
|
if args.interactive {
|
|
log::info!("Launching interactive TUI...");
|
|
|
|
if let Some(config) = tui::run_tui()? {
|
|
cli_command = config.build_command();
|
|
log::info!("Running: {cli_command}");
|
|
|
|
run_scraper(
|
|
&config.start_url,
|
|
config.subdomains,
|
|
config.single,
|
|
config.no_scope,
|
|
config.no_doc_conversion,
|
|
config.delay_ms,
|
|
config.types.as_deref(),
|
|
config.exclude_types.as_deref(),
|
|
if config.output_dir.is_empty() { None } else { Some(&config.output_dir) },
|
|
).await?;
|
|
} else {
|
|
log::info!("Cancelled by user.");
|
|
return Ok(());
|
|
}
|
|
} else {
|
|
let (start_url, tui_config) = build_config_from_args(&args)?;
|
|
cli_command = tui_config.build_command();
|
|
log::info!("Running: {cli_command}");
|
|
|
|
run_scraper(
|
|
&start_url,
|
|
args.subdomains,
|
|
args.single,
|
|
args.no_scope,
|
|
args.no_doc_conversion,
|
|
args.delay_ms,
|
|
args.types.as_deref(),
|
|
args.exclude_types.as_deref(),
|
|
args.output.as_deref(),
|
|
).await?;
|
|
}
|
|
|
|
eprintln!("Command used: {cli_command}");
|
|
|
|
Ok(())
|
|
}
|