fix clippy warnigns
This commit is contained in:
+1
-6
@@ -21,10 +21,6 @@ fn get_body_selector() -> &'static Selector {
|
||||
BODY_SELECTOR.get_or_init(|| Selector::parse("body").expect("hardcoded selector is valid"))
|
||||
}
|
||||
|
||||
/// Extract the `<body>` element's HTML from a full HTML document.
|
||||
/// The HTML5 parser always synthesizes a `<body>` element, so this
|
||||
/// returns `Some` for any well-formed document. Returns `None` only
|
||||
/// if the parser fails entirely.
|
||||
fn extract_body(html: &str) -> Option<String> {
|
||||
let document = Html::parse_document(html);
|
||||
document
|
||||
@@ -40,7 +36,7 @@ pub fn html_to_markdown(html: &str) -> String {
|
||||
match converter.convert(&source) {
|
||||
Ok(md) => md,
|
||||
Err(e) => {
|
||||
log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e);
|
||||
log::warn!("HTML-to-MD conversion failed ({e}), saving raw HTML");
|
||||
source
|
||||
}
|
||||
}
|
||||
@@ -60,7 +56,6 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_extract_body_without_body_tag_falls_back_to_synthetic() {
|
||||
// HTML5 parser synthesizes a <body> element even when absent in source.
|
||||
let html = "<html><head></head><p>Hello</p></html>";
|
||||
let result = extract_body(html);
|
||||
assert!(result.is_some(), "parser should synthesize a body element");
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
// src/doc_processor.rs
|
||||
#![warn(clippy::all, clippy::pedantic)]
|
||||
|
||||
use anyhow::Result;
|
||||
@@ -29,7 +28,7 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
|
||||
DocProcessResult::Raw
|
||||
}
|
||||
Err(e) => {
|
||||
log::warn!("anydoc conversion failed ({}), saving raw", e);
|
||||
log::warn!("anydoc conversion failed ({e}), saving raw");
|
||||
DocProcessResult::Raw
|
||||
}
|
||||
}
|
||||
@@ -37,23 +36,23 @@ pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize, usize, usize)> {
|
||||
let supported_exts = [
|
||||
const SUPPORTED_EXTS: &[&str] = &[
|
||||
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
|
||||
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
|
||||
];
|
||||
|
||||
let entries: Vec<_> = std::fs::read_dir(dir)?
|
||||
.filter_map(|e| e.ok())
|
||||
.filter_map(std::result::Result::ok)
|
||||
.filter(|e| {
|
||||
e.path().extension().is_some_and(|ext| {
|
||||
let ext = ext.to_string_lossy().to_lowercase();
|
||||
supported_exts.contains(&ext.as_str())
|
||||
SUPPORTED_EXTS.contains(&ext.as_str())
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
|
||||
let total = entries.len();
|
||||
log::info!("Found {} documents to process", total);
|
||||
log::info!("Found {total} documents to process");
|
||||
|
||||
let mut converted = 0usize;
|
||||
let mut raw = 0usize;
|
||||
@@ -67,7 +66,7 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
|
||||
let bytes = match std::fs::read(&path) {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
log::error!("error reading file: {}", e);
|
||||
log::error!("error reading file: {e}");
|
||||
errors += 1;
|
||||
continue;
|
||||
}
|
||||
@@ -102,7 +101,6 @@ pub fn batch_convert_docs(dir: &Path, delete_originals: bool) -> Result<(usize,
|
||||
Ok((converted, raw, errors))
|
||||
}
|
||||
|
||||
// src/doc_processor.rs - tests section added at end
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -113,7 +111,6 @@ mod tests {
|
||||
let html_bytes = b"<html><body><p>Test</p></body></html>";
|
||||
let url = Url::parse("https://example.com/test.html").unwrap();
|
||||
let result = try_convert(html_bytes, &url);
|
||||
// Will likely return Raw since anydoc may not handle HTML
|
||||
match result {
|
||||
DocProcessResult::Markdown(_) | DocProcessResult::Raw => {}
|
||||
}
|
||||
|
||||
+6
-9
@@ -1,4 +1,3 @@
|
||||
// src/error_logger.rs
|
||||
#![warn(clippy::all, clippy::pedantic)]
|
||||
|
||||
use chrono::{DateTime, Utc};
|
||||
@@ -6,7 +5,7 @@ use serde::Serialize;
|
||||
use std::fs::OpenOptions;
|
||||
use std::io::{BufWriter, Write};
|
||||
use std::path::Path;
|
||||
use std::sync::Mutex;
|
||||
use std::sync::{LazyLock, Mutex};
|
||||
use url::Url;
|
||||
|
||||
#[derive(Serialize)]
|
||||
@@ -19,9 +18,8 @@ pub struct ErrorEntry {
|
||||
pub status_code: Option<u16>,
|
||||
}
|
||||
|
||||
lazy_static::lazy_static! {
|
||||
static ref LOG_WRITER: Mutex<Option<BufWriter<std::fs::File>>> = Mutex::new(None);
|
||||
}
|
||||
static LOG_WRITER: LazyLock<Mutex<Option<BufWriter<std::fs::File>>>> =
|
||||
LazyLock::new(|| Mutex::new(None));
|
||||
|
||||
pub fn log_error(
|
||||
log_path: &Path,
|
||||
@@ -41,18 +39,17 @@ pub fn log_error(
|
||||
let line = match serde_json::to_string(&entry) {
|
||||
Ok(json) => json + "\n",
|
||||
Err(e) => {
|
||||
log::error!("Failed to serialize error entry: {}", e);
|
||||
log::error!("Failed to serialize error entry: {e}");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
// Buffered writer approach
|
||||
{
|
||||
let mut guard = LOG_WRITER.lock().unwrap();
|
||||
match guard.as_mut() {
|
||||
Some(writer) => {
|
||||
if let Err(e) = writer.write_all(line.as_bytes()) {
|
||||
log::error!("Failed to write error log: {}", e);
|
||||
log::error!("Failed to write error log: {e}");
|
||||
}
|
||||
}
|
||||
None => {
|
||||
@@ -68,5 +65,5 @@ pub fn log_error(
|
||||
}
|
||||
}
|
||||
|
||||
log::info!("LOGGED ERROR [{}]: {}", error_type, message);
|
||||
log::info!("LOGGED ERROR [{error_type}]: {message}");
|
||||
}
|
||||
|
||||
+9
-18
@@ -11,8 +11,8 @@ pub struct Fetcher {
|
||||
}
|
||||
|
||||
impl Fetcher {
|
||||
pub fn new(client: Client) -> Result<Self> {
|
||||
Ok(Self { client })
|
||||
pub fn new(client: Client) -> Self {
|
||||
Self { client }
|
||||
}
|
||||
|
||||
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
|
||||
@@ -33,21 +33,16 @@ impl Fetcher {
|
||||
Err(e) => {
|
||||
if attempt < MAX_RETRIES {
|
||||
log::warn!(
|
||||
"retry {}/{}: body read failed ({})",
|
||||
attempt,
|
||||
MAX_RETRIES,
|
||||
e
|
||||
"retry {attempt}/{MAX_RETRIES}: body read failed ({e})"
|
||||
);
|
||||
tokio::time::sleep(Duration::from_secs(
|
||||
RETRY_BACKOFF_SECS * attempt as u64,
|
||||
RETRY_BACKOFF_SECS * u64::from(attempt),
|
||||
))
|
||||
.await;
|
||||
continue;
|
||||
}
|
||||
return Err(anyhow!(
|
||||
"Body read failed after {} retries: {}",
|
||||
MAX_RETRIES,
|
||||
e
|
||||
"Body read failed after {MAX_RETRIES} retries: {e}"
|
||||
));
|
||||
}
|
||||
};
|
||||
@@ -64,23 +59,19 @@ impl Fetcher {
|
||||
}
|
||||
Err(e) => {
|
||||
if attempt < MAX_RETRIES {
|
||||
log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e);
|
||||
log::warn!("retry {attempt}/{MAX_RETRIES}: request failed ({e})");
|
||||
tokio::time::sleep(Duration::from_secs(
|
||||
RETRY_BACKOFF_SECS * attempt as u64,
|
||||
RETRY_BACKOFF_SECS * u64::from(attempt),
|
||||
))
|
||||
.await;
|
||||
continue;
|
||||
}
|
||||
return Err(anyhow!(
|
||||
"Request failed after {} retries: {}",
|
||||
MAX_RETRIES,
|
||||
e
|
||||
));
|
||||
return Err(anyhow!("Request failed after {MAX_RETRIES} retries: {e}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
|
||||
Err(anyhow!("Exhausted all {MAX_RETRIES} retries"))
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+16
-19
@@ -80,26 +80,23 @@ async fn main() -> Result<()> {
|
||||
let url = Url::parse(&start_url)?;
|
||||
|
||||
let base_domain = url_utils::derive_base_domain(&url)
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {start_url}"))?;
|
||||
|
||||
let output_dir = match &args.output {
|
||||
Some(p) => PathBuf::from(p),
|
||||
None => {
|
||||
let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
|
||||
PathBuf::from(default_name)
|
||||
}
|
||||
let output_dir = if let Some(p) = &args.output {
|
||||
PathBuf::from(p)
|
||||
} else {
|
||||
let default_name = format!("{}_scraped", base_domain.replace('.', "_"));
|
||||
PathBuf::from(default_name)
|
||||
};
|
||||
|
||||
let logs_dir = {
|
||||
let output_name = output_dir
|
||||
.file_name()
|
||||
.map(|n| n.to_string_lossy().into_owned())
|
||||
.unwrap_or_else(|| "output".to_string());
|
||||
let logs_name = format!("{}_logs", output_name);
|
||||
.map_or_else(|| "output".to_string(), |n| n.to_string_lossy().into_owned());
|
||||
let logs_name = format!("{output_name}_logs");
|
||||
output_dir
|
||||
.parent()
|
||||
.map(|p| p.join(&logs_name))
|
||||
.unwrap_or_else(|| PathBuf::from(&logs_name))
|
||||
.map_or_else(|| PathBuf::from(&logs_name), |p| p.join(&logs_name))
|
||||
};
|
||||
|
||||
fs::create_dir_all(&output_dir)?;
|
||||
@@ -126,8 +123,8 @@ async fn main() -> Result<()> {
|
||||
}
|
||||
};
|
||||
|
||||
log::info!("Scraping: {}", start_url);
|
||||
log::info!("Base domain: {}", base_domain);
|
||||
log::info!("Scraping: {start_url}");
|
||||
log::info!("Base domain: {base_domain}");
|
||||
log::info!("Output dir: {}", output_dir.display());
|
||||
log::info!("Logs dir: {}", logs_dir.display());
|
||||
log::info!(
|
||||
@@ -140,13 +137,13 @@ async fn main() -> Result<()> {
|
||||
);
|
||||
|
||||
match (&include_types, &exclude_types) {
|
||||
(Some(t), _) => log::info!("Type filter: include {:?}", t),
|
||||
(_, Some(t)) => log::info!("Type filter: exclude {:?}", t),
|
||||
(Some(t), _) => log::info!("Type filter: include {t:?}"),
|
||||
(_, Some(t)) => log::info!("Type filter: exclude {t:?}"),
|
||||
_ => log::info!("Type filter: none (all types)"),
|
||||
}
|
||||
|
||||
match &scope_path {
|
||||
Some(s) => log::info!("Path scope: {} (only this path and deeper)", s),
|
||||
Some(s) => log::info!("Path scope: {s} (only this path and deeper)"),
|
||||
None => log::info!("Path scope: none (full site)"),
|
||||
}
|
||||
|
||||
@@ -159,7 +156,7 @@ async fn main() -> Result<()> {
|
||||
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
|
||||
log::info!("");
|
||||
|
||||
let fetcher = fetcher::Fetcher::new(client)?;
|
||||
let fetcher = fetcher::Fetcher::new(client);
|
||||
let mut scraper = Scraper::new(
|
||||
output_dir,
|
||||
log_path.clone(),
|
||||
@@ -174,7 +171,7 @@ async fn main() -> Result<()> {
|
||||
);
|
||||
let (count, errors) = scraper.run(&url).await;
|
||||
|
||||
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||
log::info!("\nDone. Fetched {count} URLs. Errors: {errors}.");
|
||||
log::info!("Error log: {}", log_path.display());
|
||||
|
||||
Ok(())
|
||||
|
||||
+2
-9
@@ -93,10 +93,7 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
|
||||
);
|
||||
}
|
||||
Err(e) => {
|
||||
log::warn!(
|
||||
"Failed to fetch robots.txt ({}): assuming no restrictions",
|
||||
e
|
||||
);
|
||||
log::warn!("Failed to fetch robots.txt ({e}): assuming no restrictions");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -133,11 +130,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
|
||||
let agent = agent.trim().to_lowercase();
|
||||
current_agents.push(agent.clone());
|
||||
|
||||
if agent == "*" || agent == target_lower || agent == "web-scraper" {
|
||||
collecting = true;
|
||||
} else {
|
||||
collecting = false;
|
||||
}
|
||||
collecting = agent == "*" || agent == target_lower || agent == "web-scraper";
|
||||
} else if collecting {
|
||||
if let Some(path) = line.strip_prefix("Disallow:") {
|
||||
let path = path.trim().to_string();
|
||||
|
||||
+57
-61
@@ -13,22 +13,33 @@ use std::fs;
|
||||
use std::time::Duration;
|
||||
use url::Url;
|
||||
|
||||
const DOC_TYPES: &[&str] = &[
|
||||
"application/pdf",
|
||||
"application/msword",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"application/vnd.ms-word.document.macroenabled.12",
|
||||
"application/vnd.ms-powerpoint",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
||||
"application/vnd.ms-excel",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
"application/vnd.oasis.opendocument.spreadsheet",
|
||||
"application/vnd.oasis.opendocument.presentation",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"text/csv",
|
||||
];
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct DocStats {
|
||||
pub converted: usize,
|
||||
pub raw: usize,
|
||||
pub errors: usize,
|
||||
}
|
||||
|
||||
impl Default for DocStats {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
converted: 0,
|
||||
raw: 0,
|
||||
errors: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct Scraper {
|
||||
fetcher: Fetcher,
|
||||
seen: HashSet<String>,
|
||||
@@ -45,6 +56,7 @@ pub struct Scraper {
|
||||
}
|
||||
|
||||
impl Scraper {
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn new(
|
||||
output_dir: std::path::PathBuf,
|
||||
log_path: std::path::PathBuf,
|
||||
@@ -94,6 +106,20 @@ impl Scraper {
|
||||
}
|
||||
}
|
||||
|
||||
fn extract_new_links(
|
||||
&self,
|
||||
html: &str,
|
||||
final_url: &Url,
|
||||
) -> (usize, Vec<Url>) {
|
||||
let links = extract_links(html, final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
let new_links: Vec<Url> = links
|
||||
.into_iter()
|
||||
.filter(|link| !self.seen.contains(link.as_str()))
|
||||
.collect();
|
||||
(new_count, new_links)
|
||||
}
|
||||
|
||||
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
|
||||
let mut queue: VecDeque<Url> = VecDeque::new();
|
||||
let start_url_owned = start_url.clone();
|
||||
@@ -106,9 +132,8 @@ impl Scraper {
|
||||
let mut skipped_type = 0usize;
|
||||
|
||||
while let Some(raw_url) = queue.pop_front() {
|
||||
let url = match normalize_url(raw_url.as_str()) {
|
||||
Some(u) => u,
|
||||
None => continue,
|
||||
let Some(url) = normalize_url(raw_url.as_str()) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let url_key = url.as_str().to_string();
|
||||
@@ -118,19 +143,19 @@ impl Scraper {
|
||||
}
|
||||
self.seen.insert(url_key.clone());
|
||||
|
||||
if !is_in_scope(&url, &self.scope_path) {
|
||||
log::debug!("[skip] out of scope: {}", url);
|
||||
if !is_in_scope(&url, self.scope_path.as_ref()) {
|
||||
log::debug!("[skip] out of scope: {url}");
|
||||
skipped_scope += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
if !self.robots.is_allowed(url.path()) {
|
||||
log::info!("[skip] robots.txt disallows: {}", url);
|
||||
log::info!("[skip] robots.txt disallows: {url}");
|
||||
skipped_robots += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
log::info!("[{}] fetching: {}", count, url);
|
||||
log::info!("[{count}] fetching: {url}");
|
||||
|
||||
match self.fetcher.fetch_with_retry(&url).await {
|
||||
Ok(result) => {
|
||||
@@ -141,12 +166,12 @@ impl Scraper {
|
||||
self.should_save(&result.final_url, &result.content_type);
|
||||
|
||||
if !should_save {
|
||||
log::info!(" skipped (type filter: {})", ext_label);
|
||||
log::info!(" skipped (type filter: {ext_label})");
|
||||
skipped_type += 1;
|
||||
}
|
||||
|
||||
if should_save {
|
||||
match self.save(&result, is_html).await {
|
||||
match self.save(&result, is_html) {
|
||||
Ok(()) => {
|
||||
if !is_html {
|
||||
log::info!(" binary: {}", result.content_type);
|
||||
@@ -167,24 +192,14 @@ impl Scraper {
|
||||
}
|
||||
|
||||
if is_html {
|
||||
if !self.single_page {
|
||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let links = extract_links(html, &final_url, &self.base_domain);
|
||||
let new_count = links.len();
|
||||
|
||||
let new_links: Vec<Url> = links
|
||||
.into_iter()
|
||||
.filter(|link| !self.seen.contains(link.as_str()))
|
||||
.collect();
|
||||
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
|
||||
log::info!(" found {} links ({} new)", new_count, new_links.len());
|
||||
}
|
||||
} else {
|
||||
if self.single_page {
|
||||
log::info!(" [single-page mode] not crawling for links");
|
||||
} else if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||
let (new_count, new_links) = self.extract_new_links(html, &final_url);
|
||||
for link in &new_links {
|
||||
queue.push_back(link.clone());
|
||||
}
|
||||
log::info!(" found {new_count} links ({} new)", new_links.len());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -206,19 +221,19 @@ impl Scraper {
|
||||
}
|
||||
|
||||
if skipped_robots > 0 {
|
||||
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||
log::info!("Skipped {skipped_robots} URLs due to robots.txt");
|
||||
}
|
||||
if skipped_scope > 0 {
|
||||
log::info!("Skipped {} URLs due to path scope", skipped_scope);
|
||||
log::info!("Skipped {skipped_scope} URLs due to path scope");
|
||||
}
|
||||
if skipped_type > 0 {
|
||||
log::info!("Skipped {} URLs due to type filter", skipped_type);
|
||||
log::info!("Skipped {skipped_type} URLs due to type filter");
|
||||
}
|
||||
|
||||
(count, error_count)
|
||||
}
|
||||
|
||||
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
|
||||
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
||||
let content: Vec<u8>;
|
||||
|
||||
@@ -226,7 +241,7 @@ impl Scraper {
|
||||
let html = std::str::from_utf8(&result.bytes)?;
|
||||
let md = html_to_markdown(html);
|
||||
if let Some(stripped) = filename.strip_suffix(".html") {
|
||||
filename = format!("{}.md", stripped);
|
||||
filename = format!("{stripped}.md");
|
||||
}
|
||||
content = md.into_bytes();
|
||||
} else if self.convert_docs {
|
||||
@@ -263,25 +278,6 @@ impl Scraper {
|
||||
|
||||
fn is_document_content_type(ct: &str) -> bool {
|
||||
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
|
||||
const DOC_TYPES: &[&str] = &[
|
||||
"application/pdf",
|
||||
"application/msword",
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"application/vnd.ms-word.document.macroenabled.12",
|
||||
"application/vnd.ms-powerpoint",
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
|
||||
"application/vnd.ms-excel",
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/vnd.ms-excel.sheet.macroenabled.12",
|
||||
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
"application/vnd.oasis.opendocument.spreadsheet",
|
||||
"application/vnd.oasis.opendocument.presentation",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"text/csv",
|
||||
];
|
||||
DOC_TYPES.contains(&ct.as_str())
|
||||
}
|
||||
|
||||
@@ -296,7 +292,7 @@ mod tests {
|
||||
exclude_types: Option<HashSet<String>>,
|
||||
) -> Scraper {
|
||||
Scraper {
|
||||
fetcher: Fetcher::new(Client::new()).unwrap(),
|
||||
fetcher: Fetcher::new(Client::new()),
|
||||
seen: HashSet::new(),
|
||||
output_dir: std::path::PathBuf::new(),
|
||||
log_path: std::path::PathBuf::new(),
|
||||
|
||||
+7
-14
@@ -1,4 +1,3 @@
|
||||
// src/url_utils.rs
|
||||
#![warn(clippy::all, clippy::pedantic)]
|
||||
|
||||
use md5::Md5;
|
||||
@@ -35,11 +34,7 @@ pub fn normalize_url(url: &str) -> Option<Url> {
|
||||
Some(parsed)
|
||||
}
|
||||
|
||||
/// Returns true if `url`'s path falls within the given scope.
|
||||
/// A scope of `/blog/2021` matches `/blog/2021` and `/blog/2021/...`
|
||||
/// but not `/blog/20212`.
|
||||
/// `None` scope means no restriction.
|
||||
pub fn is_in_scope(url: &Url, scope_path: &Option<String>) -> bool {
|
||||
pub fn is_in_scope(url: &Url, scope_path: Option<&String>) -> bool {
|
||||
match scope_path {
|
||||
None => true,
|
||||
Some(scope) => {
|
||||
@@ -48,7 +43,7 @@ pub fn is_in_scope(url: &Url, scope_path: &Option<String>) -> bool {
|
||||
return true;
|
||||
}
|
||||
let path = url.path();
|
||||
path == scope || path.starts_with(&format!("{}/", scope))
|
||||
path == scope || path.starts_with(&format!("{scope}/"))
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -57,10 +52,10 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
|
||||
let path = url.path();
|
||||
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||
|
||||
if let Some(dot_pos) = last_segment.rfind('.') {
|
||||
if dot_pos > 0 {
|
||||
return last_segment[dot_pos..].to_lowercase();
|
||||
}
|
||||
if let Some(dot_pos) = last_segment.rfind('.')
|
||||
&& dot_pos > 0
|
||||
{
|
||||
return last_segment[dot_pos..].to_lowercase();
|
||||
}
|
||||
|
||||
let ct = content_type
|
||||
@@ -100,7 +95,6 @@ pub fn get_extension(url: &Url, content_type: &str) -> String {
|
||||
"application/vnd.oasis.opendocument.spreadsheet" => ".ods",
|
||||
"application/vnd.oasis.opendocument.presentation" => ".odp",
|
||||
"application/zip" => ".zip",
|
||||
"application/octet-stream" => ".bin",
|
||||
_ => ".bin",
|
||||
}
|
||||
.to_string()
|
||||
@@ -120,8 +114,7 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
||||
let host = url
|
||||
.host_str()
|
||||
.unwrap_or("unknown")
|
||||
.replace('.', "_")
|
||||
.replace(':', "_");
|
||||
.replace(['.', ':'], "_");
|
||||
let path = url.path().trim_start_matches('/');
|
||||
let path = if path.is_empty() { "index" } else { path };
|
||||
let path = path.trim_end_matches('/');
|
||||
|
||||
Reference in New Issue
Block a user