5 Commits
Author SHA1 Message Date
bartal-lsn fb3cdf5322 pass clippy 2026-08-28 00:18:23 +01:00
bartal-lsn eb39513960 feat(add tui): add ratatui tui.rs 2026-08-27 22:52:27 +01:00
bartal-lsn 30ec5b1bcc start on ratatui. fixed clippy 2026-08-27 22:51:40 +01:00
bartal-lsn 33ef5251d5 readem 2026-08-27 10:46:08 +01:00
bartal-lsn 8df9afabbb add no restriction 2026-08-25 14:11:18 +01:00
5 changed files with 1876 additions and 116 deletions
Generated
+997 -45
View File
File diff suppressed because it is too large Load Diff
+3 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "web-scraper"
version = "0.1.1"
version = "0.1.2"
edition = "2024"
[dependencies]
@@ -8,9 +8,11 @@ anydoc = "0.1"
anyhow = "1"
chrono = { version = "0.4", features = ["serde"] }
clap = { version = "4", features = ["derive"] }
crossterm = "0.29"
hex = "0.4"
htmd = "0.1"
log = "0.4"
ratatui = "0.30"
env_logger = "0.11"
md-5 = "0.10"
reqwest = "0.12"
+83
View File
@@ -0,0 +1,83 @@
# web-scraper-rust
A polite, single-binary Rust web scraper that crawls pages, converts HTML to Markdown, and tries its best to turn Office documents into something readable too.
It respects `robots.txt`. It waits between requests. It hashes long URLs into filenames because filesystems have feelings too. It does not do cookies, JavaScript rendering, or anything fancy. If you need a headless browser, this is not your tool.
## What It Does
- Crawls internal links from a starting URL (BFS queue)
- Converts HTML pages to Markdown via `htmd`
- Detects and converts documents (PDF, DOCX, XLSX, PPTX, ODT, RTF, EPUB, CSV...) to Markdown via `anydoc`
- Respects `robots.txt` with wildcard pattern matching
- Deduplicates URLs, normalizes fragments and trailing query params
- Configurable request delay, retry with exponential backoff
- File type filtering via `--include-types` / `--exclude-types`
- Path scoping (stay within a subdirectory) or full-site crawl
- Subdomain crawling opt-in
- Single-page mode (grab one page, don't crawl further)
- Structured error logging to JSONL, kept in a separate `_logs` directory
## Tech Stack
| Thing | Choice | Why |
|---|---|---|
| Language | Rust 2024 | Single static binary. No runtime. No GC. No regrets. |
| HTTP client | reqwest + tokio | Async, fast, well-maintained. |
| HTML parsing | scraper | CSS selectors on server-side HTML. |
| HTML→MD | htmd | Strips nav, footer, script tags. Produces clean Markdown. |
| Document conversion | anydoc | Detects format from bytes, converts to Markdown. Falls back to raw. |
| CLI | clap | Derive macros, typed args, good help text. |
| Logging | env_logger | Simple, sufficient. |
## Usage
```bash
# Scrape a single page, save Markdown
web-scraper https://example.com/page --single
# Crawl entire site, only PDFs and HTML
web-scraper https://example.com --types pdf,html
# Crawl a subdirectory only, exclude images
web-scraper https://example.com/docs --exclude-types png,jpg,gif
# Include subdomains, custom delay
web-scraper https://example.com --subdomains --delay-ms 500
```
Output goes to `./<domain>_scraped/`, logs to `./<domain>_scraped_logs/`.
## CLI Flags
```
START_URL Starting URL to scrape
-o, --output DIR Output directory (default: <domain>_scraped)
--subdomains Also crawl subdomains of the base host
--single Only scrape the start URL, don't crawl
--no-scope Crawl any path on the same host
--no-doc-conversion Save documents as-is, skip conversion
--delay-ms N Delay between requests in ms (default: 1000)
--types a,b,c Only save these file types (HTML always crawled)
--exclude-types a,b Skip these file types
```
`--types` and `--exclude-types` are mutually exclusive. Both still crawl HTML pages for links regardless.
## Build
```bash
just build # clippy + fmt + release build
# or
cargo build --release
```
Release profile is optimized for size: `opt-level = "z"`, LTO, single codegen unit, stripped.
## Error Logs
Errors are logged to `scrape_errors.jsonl` in the logs directory. Each entry includes timestamp, URL, error type, message, and HTTP status code if applicable. The log file is cleared at the start of each run.
---
**Built to scrape things. Respects robots.txt. Does not require a DevOps team.**
+166 -70
View File
@@ -8,16 +8,18 @@ mod extractor;
mod fetcher;
mod robots;
mod scraper;
mod tui;
mod url_utils;
use crate::robots::fetch_robots;
use crate::scraper::Scraper;
use crate::tui::TuiConfig;
use anyhow::{Result, bail};
use clap::Parser;
use reqwest::Client;
use std::collections::HashSet;
use std::fs;
use std::path::PathBuf;
use std::path::{Path, PathBuf};
use std::time::Duration;
use url::Url;
@@ -26,16 +28,17 @@ use url::Url;
#[command(author = "FLÓ")]
#[command(version = "0.2.0")]
#[command(about = "Web scraper for Faroese public sector sites")]
#[allow(clippy::struct_excessive_bools)]
struct Args {
#[arg(help = "Starting URL to scrape")]
start_url: String,
start_url: Option<String>,
#[arg(
long,
default_value = "1000",
help = "Delay in milliseconds between requests (default: 1000)"
long = "interactive",
short = 'i',
help = "Launch interactive TUI configuration instead of CLI"
)]
delay_ms: u64,
interactive: bool,
#[arg(
long,
@@ -49,9 +52,19 @@ struct Args {
)]
single: bool,
#[arg(long, help = "Disable path scoping — crawl any path on the same host")]
no_scope: bool,
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
no_doc_conversion: bool,
#[arg(
long,
default_value = "1000",
help = "Delay in milliseconds between requests (default: 1000)"
)]
delay_ms: u64,
#[arg(
long,
help = "Comma-separated extensions to include (e.g. pdf,docx,html). \
@@ -81,31 +94,85 @@ fn parse_ext_list(s: &str) -> HashSet<String> {
.collect()
}
#[tokio::main]
async fn main() -> Result<()> {
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
#[allow(clippy::too_many_arguments)]
fn print_config(
start_url: &str,
base_host: &str,
subdomains: bool,
output_dir: &Path,
logs_dir: &Path,
convert_docs: bool,
single: bool,
delay_ms: u64,
include_types: Option<&HashSet<String>>,
exclude_types: Option<&HashSet<String>>,
scope_path: Option<&String>,
) {
log::info!("Scraping: {start_url}");
log::info!("Base host: {base_host}");
log::info!("Subdomain crawling: {}", if subdomains { "enabled" } else { "disabled" });
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!("Document conversion: {}", if convert_docs { "enabled" } else { "disabled" });
log::info!("Single page mode: {}", if single { "enabled" } else { "disabled" });
log::info!("Request delay: {delay_ms} ms");
let args = Args::parse();
match (include_types, exclude_types) {
(Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
let start_url = args.start_url;
let convert_docs = !args.no_doc_conversion;
let delay_ms = args.delay_ms;
match scope_path {
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
}
let url = Url::parse(&start_url)?;
fn build_config_from_args(args: &Args) -> Result<(String, TuiConfig)> {
let url_str = args.start_url.as_ref()
.ok_or_else(|| anyhow::anyhow!("START_URL is required when not using --interactive"))?;
let base_host = url
.host_str()
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?
let tui_config = TuiConfig {
start_url: url_str.clone(),
output_dir: args.output.clone().unwrap_or_default(),
subdomains: args.subdomains,
single: args.single,
no_scope: args.no_scope,
no_doc_conversion: args.no_doc_conversion,
delay_ms: args.delay_ms,
types: args.types.clone(),
exclude_types: args.exclude_types.clone(),
};
Ok((url_str.clone(), tui_config))
}
#[allow(clippy::too_many_arguments)]
#[allow(clippy::fn_params_excessive_bools)]
async fn run_scraper(
start_url: &str,
subdomains: bool,
single: bool,
no_scope: bool,
no_doc_conversion: bool,
delay_ms: u64,
types: Option<&str>,
exclude_types: Option<&str>,
output: Option<&str>,
) -> Result<()> {
let url = Url::parse(start_url)?;
let base_host = url.host_str()
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname"))?
.to_string();
let base_domain = url_utils::derive_base_domain(&url)
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
.ok_or_else(|| anyhow::anyhow!("Could not parse base domain"))?;
let output_dir = if let Some(p) = &args.output {
PathBuf::from(p)
} else {
let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
PathBuf::from(default_name)
let output_dir = match output {
Some(p) if !p.is_empty() => PathBuf::from(p),
_ => PathBuf::from(format!("{}_scraped", base_domain.replace('.', "_"))),
};
let logs_dir = {
@@ -114,8 +181,7 @@ async fn main() -> Result<()> {
|n| n.to_string_lossy().into_owned(),
);
let logs_name = format!("{output_name}_logs");
output_dir
.parent()
output_dir.parent()
.map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
};
@@ -125,56 +191,33 @@ async fn main() -> Result<()> {
let log_path = logs_dir.join("scrape_errors.jsonl");
fs::remove_file(&log_path).ok();
let (include_types, exclude_types) = match (&args.types, &args.exclude_types) {
(Some(_), Some(_)) => {
bail!("--types and --exclude-types are mutually exclusive");
}
let (include_types, exclude_types) = match (types, exclude_types) {
(Some(_), Some(_)) => bail!("--types and --exclude-types are mutually exclusive"),
(Some(t), None) => (Some(parse_ext_list(t)), None),
(None, Some(t)) => (None, Some(parse_ext_list(t))),
(None, None) => (None, None),
};
let scope_path = {
let scope_path = if no_scope {
None
} else {
let path = url.path().trim_end_matches('/');
if path.is_empty() {
None
} else {
Some(path.to_string())
}
if path.is_empty() { None } else { Some(path.to_string()) }
};
log::info!("Scraping: {start_url}");
log::info!("Request delay: {delay_ms} ms");
log::info!("Base host: {base_host}");
log::info!(
"Subdomain crawling: {}",
if args.subdomains {
"enabled"
} else {
"disabled"
}
print_config(
start_url,
&base_host,
subdomains,
&output_dir,
&logs_dir,
!no_doc_conversion,
single,
delay_ms,
include_types.as_ref(),
exclude_types.as_ref(),
scope_path.as_ref(),
);
log::info!("Output dir: {}", output_dir.display());
log::info!("Logs dir: {}", logs_dir.display());
log::info!(
"Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" }
);
log::info!(
"Single page mode: {}",
if args.single { "enabled" } else { "disabled" }
);
match (&include_types, &exclude_types) {
(Some(t), _) => log::info!("Type filter: include {t:?}"),
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
_ => log::info!("Type filter: none (all types)"),
}
match &scope_path {
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
None => log::info!("Path scope: none (full site)"),
}
let client = Client::builder()
.timeout(Duration::from_secs(config::TIMEOUT_SECS))
@@ -191,19 +234,72 @@ async fn main() -> Result<()> {
log_path.clone(),
base_host,
robots,
convert_docs,
!no_doc_conversion,
fetcher,
include_types,
exclude_types,
scope_path,
args.single,
args.subdomains,
single,
subdomains,
delay_ms,
);
let (count, errors) = scraper.run(&url).await;
let (count, errors) = scraper.run(&url).await;
log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
log::info!("Error log: {}", log_path.display());
Ok(())
}
#[tokio::main]
async fn main() -> Result<()> {
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
let args = Args::parse();
let cli_command;
if args.interactive {
log::info!("Launching interactive TUI...");
if let Some(config) = tui::run_tui()? {
cli_command = config.build_command();
log::info!("Running: {cli_command}");
run_scraper(
&config.start_url,
config.subdomains,
config.single,
config.no_scope,
config.no_doc_conversion,
config.delay_ms,
config.types.as_deref(),
config.exclude_types.as_deref(),
if config.output_dir.is_empty() { None } else { Some(&config.output_dir) },
).await?;
} else {
log::info!("Cancelled by user.");
return Ok(());
}
} else {
let (start_url, tui_config) = build_config_from_args(&args)?;
cli_command = tui_config.build_command();
log::info!("Running: {cli_command}");
run_scraper(
&start_url,
args.subdomains,
args.single,
args.no_scope,
args.no_doc_conversion,
args.delay_ms,
args.types.as_deref(),
args.exclude_types.as_deref(),
args.output.as_deref(),
).await?;
}
eprintln!("Command used: {cli_command}");
Ok(())
}
+627
View File
@@ -0,0 +1,627 @@
#![allow(clippy::struct_excessive_bools)]
use crossterm::{
event::{self, Event, KeyCode, KeyEventKind, KeyModifiers},
execute,
terminal::{disable_raw_mode, enable_raw_mode, EnterAlternateScreen, LeaveAlternateScreen},
};
use ratatui::{
backend::CrosstermBackend,
layout::{Alignment, Constraint, Direction, Layout},
style::{Color, Modifier, Style},
text::{Line, Span},
widgets::{Block, Paragraph},
Frame, Terminal,
};
use std::io;
#[derive(Debug, Clone)]
pub struct TuiConfig {
pub start_url: String,
pub output_dir: String,
pub subdomains: bool,
pub single: bool,
pub no_scope: bool,
pub no_doc_conversion: bool,
pub delay_ms: u64,
pub types: Option<String>,
pub exclude_types: Option<String>,
}
impl Default for TuiConfig {
fn default() -> Self {
Self {
start_url: String::new(),
output_dir: String::new(),
subdomains: false,
single: false,
no_scope: false,
no_doc_conversion: false,
delay_ms: 1000,
types: None,
exclude_types: None,
}
}
}
impl TuiConfig {
pub fn to_cli_args(&self) -> Vec<String> {
let mut args = vec![self.start_url.clone()];
if !self.output_dir.is_empty() {
args.extend(["-o".to_string(), self.output_dir.clone()]);
}
if self.subdomains {
args.push("--subdomains".to_string());
}
if self.single {
args.push("--single".to_string());
}
if self.no_scope {
args.push("--no-scope".to_string());
}
if self.no_doc_conversion {
args.push("--no-doc-conversion".to_string());
}
if self.delay_ms != 1000 {
args.extend(["--delay-ms".to_string(), self.delay_ms.to_string()]);
}
if let Some(ref t) = self.types {
args.extend(["--types".to_string(), t.clone()]);
}
if let Some(ref t) = self.exclude_types {
args.extend(["--exclude-types".to_string(), t.clone()]);
}
args
}
pub fn build_command(&self) -> String {
format!("web-scraper {}", self.to_cli_args().join(" "))
}
}
fn validate_comma_list(input: &str) -> Result<(), String> {
let trimmed = input.trim();
if trimmed.is_empty() {
return Ok(());
}
let items: Vec<&str> = trimmed.split(',').map(str::trim).filter(|s| !s.is_empty()).collect();
if items.is_empty() {
return Err("List cannot be empty".to_string());
}
for item in &items {
let check = item.strip_prefix('.').unwrap_or(item);
if check.is_empty() || !check.chars().all(char::is_alphanumeric) {
return Err(format!("Invalid item '{item}': use extensions like 'pdf' or '.pdf'"));
}
}
Ok(())
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum FieldIndex {
StartUrl,
OutputDir,
Subdomains,
Single,
NoScope,
NoDocConversion,
DelayMs,
TypesInclude,
TypesExclude,
Submit,
}
impl FieldIndex {
fn next(self) -> Self {
match self {
FieldIndex::StartUrl => FieldIndex::OutputDir,
FieldIndex::OutputDir => FieldIndex::Subdomains,
FieldIndex::Subdomains => FieldIndex::Single,
FieldIndex::Single => FieldIndex::NoScope,
FieldIndex::NoScope => FieldIndex::NoDocConversion,
FieldIndex::NoDocConversion => FieldIndex::DelayMs,
FieldIndex::DelayMs => FieldIndex::TypesInclude,
FieldIndex::TypesInclude => FieldIndex::TypesExclude,
FieldIndex::TypesExclude => FieldIndex::Submit,
FieldIndex::Submit => FieldIndex::StartUrl,
}
}
fn prev(self) -> Self {
match self {
FieldIndex::StartUrl => FieldIndex::Submit,
FieldIndex::OutputDir => FieldIndex::StartUrl,
FieldIndex::Subdomains => FieldIndex::OutputDir,
FieldIndex::Single => FieldIndex::Subdomains,
FieldIndex::NoScope => FieldIndex::Single,
FieldIndex::NoDocConversion => FieldIndex::NoScope,
FieldIndex::DelayMs => FieldIndex::NoDocConversion,
FieldIndex::TypesInclude => FieldIndex::DelayMs,
FieldIndex::TypesExclude => FieldIndex::TypesInclude,
FieldIndex::Submit => FieldIndex::TypesExclude,
}
}
fn label(self) -> &'static str {
match self {
FieldIndex::StartUrl => "Start URL",
FieldIndex::OutputDir => "Output directory",
FieldIndex::Subdomains => "Include subdomains?",
FieldIndex::Single => "Single page mode?",
FieldIndex::NoScope => "No path scoping?",
FieldIndex::NoDocConversion => "Skip document conversion?",
FieldIndex::DelayMs => "Delay (ms)",
FieldIndex::TypesInclude => "Include types",
FieldIndex::TypesExclude => "Exclude types",
FieldIndex::Submit => "Submit",
}
}
fn hint(self) -> &'static str {
match self {
FieldIndex::StartUrl => "https://example.com/docs",
FieldIndex::OutputDir => "<domain>_scraped",
FieldIndex::Subdomains
| FieldIndex::Single
| FieldIndex::NoScope
| FieldIndex::NoDocConversion => "[y/N]",
FieldIndex::DelayMs => "1000",
FieldIndex::TypesInclude => ".pdf,docx,.html",
FieldIndex::TypesExclude => "png,jpg,.gif",
FieldIndex::Submit => "",
}
}
fn is_text_field(self) -> bool {
matches!(
self,
FieldIndex::StartUrl
| FieldIndex::OutputDir
| FieldIndex::DelayMs
| FieldIndex::TypesInclude
| FieldIndex::TypesExclude
)
}
fn is_boolean_field(self) -> bool {
matches!(
self,
FieldIndex::Subdomains
| FieldIndex::Single
| FieldIndex::NoScope
| FieldIndex::NoDocConversion
)
}
fn subtitle(self) -> &'static str {
match self {
FieldIndex::DelayMs => "[enter positive integer]",
FieldIndex::TypesInclude | FieldIndex::TypesExclude => "[comma-separated, dots optional]",
_ => "",
}
}
}
const ALL_FIELDS: [FieldIndex; 10] = [
FieldIndex::StartUrl,
FieldIndex::OutputDir,
FieldIndex::Subdomains,
FieldIndex::Single,
FieldIndex::NoScope,
FieldIndex::NoDocConversion,
FieldIndex::DelayMs,
FieldIndex::TypesInclude,
FieldIndex::TypesExclude,
FieldIndex::Submit,
];
enum Action {
Continue,
Cancel,
Submit,
}
pub fn run_tui() -> io::Result<Option<TuiConfig>> {
enable_raw_mode()?;
let mut stdout = io::stdout();
execute!(stdout, EnterAlternateScreen)?;
let backend = CrosstermBackend::new(stdout);
let mut terminal = Terminal::new(backend)?;
let mut config = TuiConfig::default();
let mut current_field = FieldIndex::StartUrl;
let mut text_input = String::new();
let mut errors: Vec<(FieldIndex, String)> = Vec::new();
load_field_into_input(&mut text_input, &config, current_field);
loop {
terminal.draw(|frame| ui(frame, &config, current_field, &errors))?;
if event::poll(std::time::Duration::from_millis(50))?
&& let Event::Key(key) = event::read()?
&& key.kind == KeyEventKind::Press
{
let action = handle_key_event(
&mut config,
&mut current_field,
&mut text_input,
&mut errors,
key.code,
key.modifiers,
);
match action {
Action::Continue => {}
Action::Cancel => {
disable_raw_mode()?;
execute!(terminal.backend_mut(), LeaveAlternateScreen)?;
terminal.show_cursor()?;
return Ok(None);
}
Action::Submit => {
disable_raw_mode()?;
execute!(terminal.backend_mut(), LeaveAlternateScreen)?;
terminal.show_cursor()?;
return Ok(Some(config));
}
}
}
}
}
#[allow(clippy::too_many_lines)]
fn ui(frame: &mut Frame, config: &TuiConfig, current_field: FieldIndex, errors: &[(FieldIndex, String)]) {
let title_line = Line::from(Span::styled(
"Interactive Web Scraper Configuration",
Style::default().fg(Color::LightCyan).add_modifier(Modifier::BOLD),
));
let controls_lines: Vec<Line> = vec![
Line::from(Span::styled("Controls:", Style::default().fg(Color::DarkGray))),
Line::from(" Enter : Next field / Toggle boolean / Submit"),
Line::from(" Tab : Next field"),
Line::from(" Shift+Tab: Previous field"),
Line::from(" Ctrl+C : Exit without running"),
Line::from(" Esc : Exit without running"),
];
let cmd_preview = config.build_command();
let cmd_lines: Vec<Line> = vec![
Line::from(Span::styled(
"Command preview:",
Style::default().fg(Color::Yellow).add_modifier(Modifier::BOLD),
)),
Line::from(Span::styled(format!(" {cmd_preview}"), Style::default().fg(Color::Yellow))),
];
let status_text = if config.start_url.is_empty() {
"Waiting for Start URL..."
} else {
"Ready. Press Enter on Submit to run."
};
let status_style = if config.start_url.is_empty() {
Style::default().fg(Color::Yellow)
} else {
Style::default().fg(Color::Green)
};
let status_line = Line::from(Span::styled(status_text, status_style));
let chunks = Layout::default()
.direction(Direction::Vertical)
.margin(1)
.constraints([
Constraint::Length(1),
Constraint::Min(10),
Constraint::Length(6),
Constraint::Length(3),
Constraint::Length(1),
])
.split(frame.area());
frame.render_widget(Paragraph::new(title_line), chunks[0]);
let field_chunks = Layout::default()
.direction(Direction::Vertical)
.constraints(std::iter::repeat_n(Constraint::Length(2), ALL_FIELDS.len()).collect::<Vec<_>>())
.split(chunks[1]);
for (i, &field) in ALL_FIELDS.iter().enumerate() {
render_field(frame, field_chunks[i], config, current_field, field, errors);
}
frame.render_widget(Paragraph::new(controls_lines), chunks[2]);
frame.render_widget(Paragraph::new(cmd_lines), chunks[3]);
let error_line = if let Some((_, err)) = errors.iter().find(|(f, _)| *f == FieldIndex::Submit) {
Line::from(Span::styled(format!("[ERR: {err}]"), Style::default().fg(Color::Red)))
} else {
status_line
};
frame.render_widget(
Paragraph::new(error_line).alignment(Alignment::Center),
chunks[4],
);
}
fn render_field(frame: &mut Frame, area: ratatui::layout::Rect, config: &TuiConfig, current_field: FieldIndex, field: FieldIndex, errors: &[(FieldIndex, String)]) {
let value_text = get_field_display(config, field);
let hint = field.hint();
let is_active = current_field == field;
let display_value = if !value_text.is_empty() {
value_text.as_str()
} else if !hint.is_empty() {
hint
} else {
""
};
let bool_state = if field.is_boolean_field() {
format!(" [{}]", if get_bool_field(config, field) { "enabled" } else { "disabled" })
} else {
String::new()
};
let prefix = if is_active { ">" } else { " " };
let label = format!("{:30} ", format!("{}:", field.label()));
let subtitle = field.subtitle();
let value_style = if is_active {
Style::default()
} else if display_value.is_empty() || display_value == hint {
Style::default().fg(Color::DarkGray)
} else {
Style::default()
};
let mut spans = vec![
Span::styled(
format!("{prefix} {label}"),
if is_active {
Style::default().add_modifier(Modifier::BOLD)
} else {
Style::default()
},
),
Span::styled(display_value.to_string(), value_style),
Span::styled(bool_state, Style::default().fg(Color::DarkGray)),
Span::styled(format!(" {subtitle}"), Style::default().fg(Color::DarkGray)),
];
if let Some((_, err)) = errors.iter().find(|(f, _)| *f == field) {
spans.push(Span::styled(format!(" [ERR: {err}]"), Style::default().fg(Color::Red)));
}
let line = Line::from(spans);
let block = if is_active {
Block::default().style(Style::default().bg(Color::DarkGray))
} else {
Block::default()
};
let widget = Paragraph::new(line).block(block);
frame.render_widget(widget, area);
}
fn get_bool_field(config: &TuiConfig, field: FieldIndex) -> bool {
match field {
FieldIndex::Subdomains => config.subdomains,
FieldIndex::Single => config.single,
FieldIndex::NoScope => config.no_scope,
FieldIndex::NoDocConversion => config.no_doc_conversion,
_ => false,
}
}
fn get_field_display(config: &TuiConfig, field: FieldIndex) -> String {
match field {
FieldIndex::StartUrl => config.start_url.clone(),
FieldIndex::OutputDir => {
if config.output_dir.is_empty() {
String::new()
} else {
config.output_dir.clone()
}
}
FieldIndex::DelayMs => config.delay_ms.to_string(),
FieldIndex::TypesInclude => config.types.clone().unwrap_or_default(),
FieldIndex::TypesExclude => config.exclude_types.clone().unwrap_or_default(),
_ => String::new(),
}
}
fn load_field_into_input(text_input: &mut String, config: &TuiConfig, field: FieldIndex) {
*text_input = match field {
FieldIndex::StartUrl => config.start_url.clone(),
FieldIndex::OutputDir => config.output_dir.clone(),
FieldIndex::DelayMs => config.delay_ms.to_string(),
FieldIndex::TypesInclude => config.types.clone().unwrap_or_default(),
FieldIndex::TypesExclude => config.exclude_types.clone().unwrap_or_default(),
_ => String::new(),
};
}
fn save_input_to_field(config: &mut TuiConfig, field: FieldIndex, text_input: &str) {
match field {
FieldIndex::StartUrl => {
config.start_url = text_input.to_string();
}
FieldIndex::OutputDir => {
config.output_dir = text_input.to_string();
}
FieldIndex::DelayMs => {
if let Ok(val) = text_input.trim().parse::<u64>() {
config.delay_ms = val;
}
}
FieldIndex::TypesInclude => {
let trimmed = text_input.trim();
config.types = if trimmed.is_empty() {
None
} else {
Some(trimmed.to_string())
};
}
FieldIndex::TypesExclude => {
let trimmed = text_input.trim();
config.exclude_types = if trimmed.is_empty() {
None
} else {
Some(trimmed.to_string())
};
}
_ => {}
}
}
fn handle_key_event(
config: &mut TuiConfig,
current_field: &mut FieldIndex,
text_input: &mut String,
errors: &mut Vec<(FieldIndex, String)>,
code: KeyCode,
modifiers: KeyModifiers,
) -> Action {
match code {
KeyCode::Esc
| KeyCode::Char('c') if modifiers.contains(KeyModifiers::CONTROL) => {
Action::Cancel
}
KeyCode::Tab => {
save_input_to_field(config, *current_field, text_input);
*current_field = current_field.next();
load_field_into_input(text_input, config, *current_field);
errors.clear();
Action::Continue
}
KeyCode::BackTab => {
save_input_to_field(config, *current_field, text_input);
*current_field = current_field.prev();
load_field_into_input(text_input, config, *current_field);
errors.clear();
Action::Continue
}
KeyCode::Enter => {
process_enter(config, current_field, text_input, errors)
}
KeyCode::Char(ch) if current_field.is_text_field() => {
text_input.push(ch);
Action::Continue
}
KeyCode::Backspace if current_field.is_text_field() => {
text_input.pop();
Action::Continue
}
KeyCode::Delete if current_field.is_text_field() => {
text_input.clear();
Action::Continue
}
_ => Action::Continue,
}
}
fn process_enter(
config: &mut TuiConfig,
current_field: &mut FieldIndex,
text_input: &mut String,
errors: &mut Vec<(FieldIndex, String)>,
) -> Action {
match *current_field {
FieldIndex::Subdomains => {
config.subdomains = !config.subdomains;
Action::Continue
}
FieldIndex::Single => {
config.single = !config.single;
Action::Continue
}
FieldIndex::NoScope => {
config.no_scope = !config.no_scope;
Action::Continue
}
FieldIndex::NoDocConversion => {
config.no_doc_conversion = !config.no_doc_conversion;
Action::Continue
}
FieldIndex::StartUrl => {
save_input_to_field(config, *current_field, text_input);
if config.start_url.trim().is_empty() {
errors.retain(|(f, _)| *f != FieldIndex::StartUrl);
errors.push((FieldIndex::StartUrl, "Required".into()));
} else {
errors.clear();
*current_field = current_field.next();
load_field_into_input(text_input, config, *current_field);
}
Action::Continue
}
FieldIndex::OutputDir | FieldIndex::DelayMs => {
save_input_to_field(config, *current_field, text_input);
errors.clear();
*current_field = current_field.next();
load_field_into_input(text_input, config, *current_field);
Action::Continue
}
FieldIndex::TypesInclude => {
save_input_to_field(config, *current_field, text_input);
let val = config.types.as_deref().unwrap_or("");
match validate_comma_list(val) {
Ok(()) => {
errors.clear();
*current_field = current_field.next();
load_field_into_input(text_input, config, *current_field);
}
Err(e) => {
errors.retain(|(f, _)| *f != FieldIndex::TypesInclude);
errors.push((FieldIndex::TypesInclude, e));
}
}
Action::Continue
}
FieldIndex::TypesExclude => {
save_input_to_field(config, *current_field, text_input);
let val = config.exclude_types.as_deref().unwrap_or("");
match validate_comma_list(val) {
Ok(()) => {
errors.clear();
*current_field = current_field.next();
load_field_into_input(text_input, config, *current_field);
}
Err(e) => {
errors.retain(|(f, _)| *f != FieldIndex::TypesExclude);
errors.push((FieldIndex::TypesExclude, e));
}
}
Action::Continue
}
FieldIndex::Submit => {
if config.start_url.trim().is_empty() {
errors.retain(|(f, _)| *f != FieldIndex::StartUrl);
errors.push((FieldIndex::StartUrl, "Required".into()));
return Action::Continue;
}
if let Some(ref t) = config.types
&& let Err(e) = validate_comma_list(t)
{
errors.retain(|(f, _)| *f != FieldIndex::TypesInclude);
errors.push((FieldIndex::TypesInclude, e));
return Action::Continue;
}
if let Some(ref t) = config.exclude_types
&& let Err(e) = validate_comma_list(t)
{
errors.retain(|(f, _)| *f != FieldIndex::TypesExclude);
errors.push((FieldIndex::TypesExclude, e));
return Action::Continue;
}
if config.types.is_some() && config.exclude_types.is_some() {
errors.push((
FieldIndex::Submit,
"types and exclude-types are mutually exclusive".into(),
));
return Action::Continue;
}
Action::Submit
}
}
}