fix clippy warnigns

This commit is contained in:
2026-08-25 12:01:07 +01:00
parent da8f7302c1
commit fe4b5ed18a
11 changed files with 106 additions and 155 deletions
+57 -61
View File
@@ -13,22 +13,33 @@ use std::fs;
use std::time::Duration;
use url::Url;
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
#[derive(Default)]
pub struct DocStats {
pub converted: usize,
pub raw: usize,
pub errors: usize,
}
impl Default for DocStats {
fn default() -> Self {
Self {
converted: 0,
raw: 0,
errors: 0,
}
}
}
pub struct Scraper {
fetcher: Fetcher,
seen: HashSet<String>,
@@ -45,6 +56,7 @@ pub struct Scraper {
}
impl Scraper {
#[allow(clippy::too_many_arguments)]
pub fn new(
output_dir: std::path::PathBuf,
log_path: std::path::PathBuf,
@@ -94,6 +106,20 @@ impl Scraper {
}
}
fn extract_new_links(
&self,
html: &str,
final_url: &Url,
) -> (usize, Vec<Url>) {
let links = extract_links(html, final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
(new_count, new_links)
}
pub async fn run(&mut self, start_url: &Url) -> (usize, usize) {
let mut queue: VecDeque<Url> = VecDeque::new();
let start_url_owned = start_url.clone();
@@ -106,9 +132,8 @@ impl Scraper {
let mut skipped_type = 0usize;
while let Some(raw_url) = queue.pop_front() {
let url = match normalize_url(raw_url.as_str()) {
Some(u) => u,
None => continue,
let Some(url) = normalize_url(raw_url.as_str()) else {
continue;
};
let url_key = url.as_str().to_string();
@@ -118,19 +143,19 @@ impl Scraper {
}
self.seen.insert(url_key.clone());
if !is_in_scope(&url, &self.scope_path) {
log::debug!("[skip] out of scope: {}", url);
if !is_in_scope(&url, self.scope_path.as_ref()) {
log::debug!("[skip] out of scope: {url}");
skipped_scope += 1;
continue;
}
if !self.robots.is_allowed(url.path()) {
log::info!("[skip] robots.txt disallows: {}", url);
log::info!("[skip] robots.txt disallows: {url}");
skipped_robots += 1;
continue;
}
log::info!("[{}] fetching: {}", count, url);
log::info!("[{count}] fetching: {url}");
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
@@ -141,12 +166,12 @@ impl Scraper {
self.should_save(&result.final_url, &result.content_type);
if !should_save {
log::info!(" skipped (type filter: {})", ext_label);
log::info!(" skipped (type filter: {ext_label})");
skipped_type += 1;
}
if should_save {
match self.save(&result, is_html).await {
match self.save(&result, is_html) {
Ok(()) => {
if !is_html {
log::info!(" binary: {}", result.content_type);
@@ -167,24 +192,14 @@ impl Scraper {
}
if is_html {
if !self.single_page {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links
.into_iter()
.filter(|link| !self.seen.contains(link.as_str()))
.collect();
for link in &new_links {
queue.push_back(link.clone());
}
log::info!(" found {} links ({} new)", new_count, new_links.len());
}
} else {
if self.single_page {
log::info!(" [single-page mode] not crawling for links");
} else if let Ok(html) = std::str::from_utf8(&result.bytes) {
let (new_count, new_links) = self.extract_new_links(html, &final_url);
for link in &new_links {
queue.push_back(link.clone());
}
log::info!(" found {new_count} links ({} new)", new_links.len());
}
}
}
@@ -206,19 +221,19 @@ impl Scraper {
}
if skipped_robots > 0 {
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
log::info!("Skipped {skipped_robots} URLs due to robots.txt");
}
if skipped_scope > 0 {
log::info!("Skipped {} URLs due to path scope", skipped_scope);
log::info!("Skipped {skipped_scope} URLs due to path scope");
}
if skipped_type > 0 {
log::info!("Skipped {} URLs due to type filter", skipped_type);
log::info!("Skipped {skipped_type} URLs due to type filter");
}
(count, error_count)
}
async fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
fn save(&mut self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
@@ -226,7 +241,7 @@ impl Scraper {
let html = std::str::from_utf8(&result.bytes)?;
let md = html_to_markdown(html);
if let Some(stripped) = filename.strip_suffix(".html") {
filename = format!("{}.md", stripped);
filename = format!("{stripped}.md");
}
content = md.into_bytes();
} else if self.convert_docs {
@@ -263,25 +278,6 @@ impl Scraper {
fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
DOC_TYPES.contains(&ct.as_str())
}
@@ -296,7 +292,7 @@ mod tests {
exclude_types: Option<HashSet<String>>,
) -> Scraper {
Scraper {
fetcher: Fetcher::new(Client::new()).unwrap(),
fetcher: Fetcher::new(Client::new()),
seen: HashSet::new(),
output_dir: std::path::PathBuf::new(),
log_path: std::path::PathBuf::new(),