add no restriction

This commit is contained in:
2026-08-25 14:11:18 +01:00
parent d7208f98cf
commit 8df9afabbb
3 changed files with 72 additions and 41 deletions
Generated
+1 -1
View File
@@ -2601,7 +2601,7 @@ dependencies = [
[[package]] [[package]]
name = "web-scraper" name = "web-scraper"
version = "0.1.1" version = "0.1.2"
dependencies = [ dependencies = [
"anydoc", "anydoc",
"anyhow", "anyhow",
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "web-scraper" name = "web-scraper"
version = "0.1.1" version = "0.1.2"
edition = "2024" edition = "2024"
[dependencies] [dependencies]
+70 -39
View File
@@ -26,17 +26,11 @@ use url::Url;
#[command(author = "FLÓ")] #[command(author = "FLÓ")]
#[command(version = "0.2.0")] #[command(version = "0.2.0")]
#[command(about = "Web scraper for Faroese public sector sites")] #[command(about = "Web scraper for Faroese public sector sites")]
#[allow(clippy::struct_excessive_bools)]
struct Args { struct Args {
#[arg(help = "Starting URL to scrape")] #[arg(help = "Starting URL to scrape")]
start_url: String, start_url: String,
#[arg(
long,
default_value = "1000",
help = "Delay in milliseconds between requests (default: 1000)"
)]
delay_ms: u64,
#[arg( #[arg(
long, long,
help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)" help = "Also crawl subdomains of the start URL's host (e.g. git.flo.fo when scraping flo.fo)"
@@ -49,9 +43,19 @@ struct Args {
)] )]
single: bool, single: bool,
#[arg(long, help = "Disable path scoping — crawl any path on the same host")]
no_scope: bool,
#[arg(long, help = "Save documents as-is, skip Markdown conversion")] #[arg(long, help = "Save documents as-is, skip Markdown conversion")]
no_doc_conversion: bool, no_doc_conversion: bool,
#[arg(
long,
default_value = "1000",
help = "Delay in milliseconds between requests (default: 1000)"
)]
delay_ms: u64,
#[arg( #[arg(
long, long,
help = "Comma-separated extensions to include (e.g. pdf,docx,html). \ help = "Comma-separated extensions to include (e.g. pdf,docx,html). \
@@ -81,6 +85,50 @@ fn parse_ext_list(s: &str) -> HashSet<String> {
.collect() .collect()
} }
#[allow(clippy::too_many_arguments)]
fn print_config(
start_url: &str,
base_host: &str,
subdomains: bool,
output_dir: &std::path::Path,
logs_dir: &std::path::Path,
convert_docs: bool,
single: bool,
delay_ms: u64,
include_types: Option<&HashSet<String>>,
exclude_types: Option<&HashSet<String>>,
scope_path: Option<&String>,
) {
log::info!("Scraping: {start_url}");
log::info!("Base host: {base_host}");
log::info!(
"Subdomain crawling: {}",
if subdomains { "enabled" } else { "disabled" }
);
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!(
"Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" }
);
log::info!(
"Single page mode: {}",
if single { "enabled" } else { "disabled" }
);
log::info!("Request delay: {delay_ms} ms");
match (include_types, exclude_types) {
(Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
match scope_path {
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
}
#[tokio::main] #[tokio::main]
async fn main() -> Result<()> { async fn main() -> Result<()> {
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init(); env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
@@ -134,7 +182,9 @@ async fn main() -> Result<()> {
(None, None) => (None, None), (None, None) => (None, None),
}; };
let scope_path = { let scope_path = if args.no_scope {
None
} else {
let path = url.path().trim_end_matches('/'); let path = url.path().trim_end_matches('/');
if path.is_empty() { if path.is_empty() {
None None
@@ -143,38 +193,19 @@ async fn main() -> Result<()> {
} }
}; };
log::info!("Scraping: {start_url}"); print_config(
log::info!("Request delay: {delay_ms} ms"); &start_url,
log::info!("Base host: {base_host}"); &base_host,
log::info!( args.subdomains,
"Subdomain crawling: {}", &output_dir,
if args.subdomains { &logs_dir,
"enabled" convert_docs,
} else { args.single,
"disabled" delay_ms,
} include_types.as_ref(),
exclude_types.as_ref(),
scope_path.as_ref(),
); );
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!(
"Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" }
);
log::info!(
"Single page mode: {}",
if args.single { "enabled" } else { "disabled" }
);
match (&include_types, &exclude_types) {
(Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
match &scope_path {
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
let client = Client::builder() let client = Client::builder()
.timeout(Duration::from_secs(config::TIMEOUT_SECS)) .timeout(Duration::from_secs(config::TIMEOUT_SECS))