fixings
This commit is contained in:
Generated
+715
-12
File diff suppressed because it is too large
Load Diff
@@ -7,7 +7,9 @@ edition = "2024"
|
|||||||
anyhow = "1"
|
anyhow = "1"
|
||||||
chrono = { version = "0.4", features = ["serde"] }
|
chrono = { version = "0.4", features = ["serde"] }
|
||||||
hex = "0.4"
|
hex = "0.4"
|
||||||
|
htmd = "0.1"
|
||||||
md-5 = "0.10"
|
md-5 = "0.10"
|
||||||
|
pdf-inspector = "1"
|
||||||
reqwest = "0.12"
|
reqwest = "0.12"
|
||||||
scraper = "0.22"
|
scraper = "0.22"
|
||||||
serde = { version = "1", features = ["derive"] }
|
serde = { version = "1", features = ["derive"] }
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
pub const DELAY_MS: u64 = 100;
|
pub const DELAY_MS: u64 = 0;
|
||||||
pub const TIMEOUT_SECS: u64 = 60;
|
pub const TIMEOUT_SECS: u64 = 60;
|
||||||
pub const MAX_RETRIES: u32 = 3;
|
pub const MAX_RETRIES: u32 = 3;
|
||||||
pub const RETRY_BACKOFF_SECS: u64 = 5;
|
pub const RETRY_BACKOFF_SECS: u64 = 5;
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
use htmd::HtmlToMarkdown;
|
||||||
|
|
||||||
|
pub fn html_to_markdown(html: &str) -> String {
|
||||||
|
let converter = HtmlToMarkdown::new();
|
||||||
|
match converter.convert(html) {
|
||||||
|
Ok(md) => md,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!(
|
||||||
|
" WARN: HTML-to-MD conversion failed ({}), saving raw HTML",
|
||||||
|
e
|
||||||
|
);
|
||||||
|
html.to_string()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,3 +1,5 @@
|
|||||||
|
// src/fetcher.rs
|
||||||
|
|
||||||
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
|
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
|
||||||
use anyhow::{Result, anyhow};
|
use anyhow::{Result, anyhow};
|
||||||
use reqwest::Client;
|
use reqwest::Client;
|
||||||
@@ -60,6 +62,7 @@ impl Fetcher {
|
|||||||
bytes,
|
bytes,
|
||||||
content_type,
|
content_type,
|
||||||
final_url,
|
final_url,
|
||||||
|
_status_code: status.as_u16(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
@@ -96,4 +99,5 @@ pub struct FetchResult {
|
|||||||
pub bytes: Vec<u8>,
|
pub bytes: Vec<u8>,
|
||||||
pub content_type: String,
|
pub content_type: String,
|
||||||
pub final_url: Url,
|
pub final_url: Url,
|
||||||
|
pub _status_code: u16,
|
||||||
}
|
}
|
||||||
|
|||||||
+56
-10
@@ -1,13 +1,16 @@
|
|||||||
// src/main.rs
|
|
||||||
|
|
||||||
mod config;
|
mod config;
|
||||||
|
mod converter;
|
||||||
mod error_logger;
|
mod error_logger;
|
||||||
mod extractor;
|
mod extractor;
|
||||||
mod fetcher;
|
mod fetcher;
|
||||||
|
mod pdf_processor;
|
||||||
|
mod robots;
|
||||||
mod scraper;
|
mod scraper;
|
||||||
mod url_utils;
|
mod url_utils;
|
||||||
|
|
||||||
use scraper::Scraper;
|
use crate::config::USER_AGENT;
|
||||||
|
use crate::robots::fetch_robots;
|
||||||
|
use crate::scraper::Scraper;
|
||||||
use std::fs;
|
use std::fs;
|
||||||
use std::path::PathBuf;
|
use std::path::PathBuf;
|
||||||
use url::Url;
|
use url::Url;
|
||||||
@@ -16,14 +19,44 @@ use url::Url;
|
|||||||
async fn main() -> anyhow::Result<()> {
|
async fn main() -> anyhow::Result<()> {
|
||||||
let args: Vec<String> = std::env::args().collect();
|
let args: Vec<String> = std::env::args().collect();
|
||||||
|
|
||||||
if args.len() < 2 {
|
// Parse flags
|
||||||
eprintln!("Usage: site-scraper <start-url>");
|
let mut convert_pdfs = true;
|
||||||
eprintln!("Example: site-scraper https://www.logting.fo");
|
let mut url_arg: Option<String> = None;
|
||||||
std::process::exit(1);
|
|
||||||
|
for arg in args.iter().skip(1) {
|
||||||
|
match arg.as_str() {
|
||||||
|
"--no-pdf-conversion" => convert_pdfs = false,
|
||||||
|
"--help" | "-h" => {
|
||||||
|
println!("Usage: site-scraper [OPTIONS] <start-url>");
|
||||||
|
println!();
|
||||||
|
println!("Options:");
|
||||||
|
println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction");
|
||||||
|
println!(" -h, --help Show this help message");
|
||||||
|
println!();
|
||||||
|
println!("Example:");
|
||||||
|
println!(" site-scraper https://www.logting.fo");
|
||||||
|
println!(" site-scraper --no-pdf-conversion https://taks.fo");
|
||||||
|
std::process::exit(0);
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
if url_arg.is_none() {
|
||||||
|
url_arg = Some(arg.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
let start_url = &args[1];
|
let start_url = match url_arg {
|
||||||
let url = Url::parse(start_url)?;
|
Some(u) => u,
|
||||||
|
None => {
|
||||||
|
eprintln!("Usage: site-scraper [OPTIONS] <start-url>");
|
||||||
|
eprintln!("Example: site-scraper https://www.logting.fo");
|
||||||
|
eprintln!("Run with --help for options.");
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
let url = Url::parse(&start_url)?;
|
||||||
|
|
||||||
let base_domain = url_utils::derive_base_domain(&url)
|
let base_domain = url_utils::derive_base_domain(&url)
|
||||||
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
.ok_or_else(|| anyhow::anyhow!("Could not parse hostname from {}", start_url))?;
|
||||||
@@ -40,9 +73,22 @@ async fn main() -> anyhow::Result<()> {
|
|||||||
println!("Scraping: {}", start_url);
|
println!("Scraping: {}", start_url);
|
||||||
println!("Base domain: {}", base_domain);
|
println!("Base domain: {}", base_domain);
|
||||||
println!("Output dir: {}", output_dir.display());
|
println!("Output dir: {}", output_dir.display());
|
||||||
|
println!(
|
||||||
|
"PDF conversion: {}",
|
||||||
|
if convert_pdfs { "enabled" } else { "disabled" }
|
||||||
|
);
|
||||||
|
|
||||||
|
// Fetch robots.txt
|
||||||
|
let client = reqwest::Client::builder()
|
||||||
|
.timeout(std::time::Duration::from_secs(30))
|
||||||
|
.user_agent(USER_AGENT)
|
||||||
|
.build()?;
|
||||||
|
|
||||||
|
println!("Fetching robots.txt...");
|
||||||
|
let robots = fetch_robots(&client, &url, "web-scraper").await;
|
||||||
println!();
|
println!();
|
||||||
|
|
||||||
let scraper = Scraper::new(output_dir, log_path, base_domain);
|
let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs);
|
||||||
let (count, errors) = scraper.run(&url).await;
|
let (count, errors) = scraper.run(&url).await;
|
||||||
|
|
||||||
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
|
||||||
|
|||||||
@@ -0,0 +1,71 @@
|
|||||||
|
// src/pdf_processor.rs
|
||||||
|
|
||||||
|
use anyhow::Result;
|
||||||
|
use std::path::Path;
|
||||||
|
|
||||||
|
pub enum PdfProcessResult {
|
||||||
|
Markdown(String),
|
||||||
|
Scanned,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn process_pdf(pdf_path: &Path) -> Result<PdfProcessResult> {
|
||||||
|
use pdf_inspector::process_pdf;
|
||||||
|
|
||||||
|
let result = process_pdf(pdf_path)?;
|
||||||
|
|
||||||
|
match result.pdf_type {
|
||||||
|
pdf_inspector::PdfType::TextBased | pdf_inspector::PdfType::Mixed => {
|
||||||
|
match &result.markdown {
|
||||||
|
Some(md) if !md.trim().is_empty() => Ok(PdfProcessResult::Markdown(md.clone())),
|
||||||
|
_ => Ok(PdfProcessResult::Scanned),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pdf_inspector::PdfType::Scanned | pdf_inspector::PdfType::ImageBased => {
|
||||||
|
Ok(PdfProcessResult::Scanned)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn batch_convert_pdfs(dir: &Path) -> Result<(usize, usize, usize)> {
|
||||||
|
let mut text_based = 0usize;
|
||||||
|
let mut scanned = 0usize;
|
||||||
|
let mut errors = 0usize;
|
||||||
|
|
||||||
|
let entries: Vec<_> = std::fs::read_dir(dir)?
|
||||||
|
.filter_map(|e| e.ok())
|
||||||
|
.filter(|e| {
|
||||||
|
e.path()
|
||||||
|
.extension()
|
||||||
|
.is_some_and(|ext| ext.eq_ignore_ascii_case("pdf"))
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let total = entries.len();
|
||||||
|
println!("Found {} PDFs to process", total);
|
||||||
|
|
||||||
|
for (i, entry) in entries.iter().enumerate() {
|
||||||
|
let path = entry.path();
|
||||||
|
let filename = path.file_name().unwrap_or_default().to_string_lossy();
|
||||||
|
println!("[{}/{}] processing: {}", i + 1, total, filename);
|
||||||
|
|
||||||
|
match process_pdf(&path) {
|
||||||
|
Ok(PdfProcessResult::Markdown(md)) => {
|
||||||
|
let md_path = path.with_extension("md");
|
||||||
|
std::fs::write(&md_path, md)?;
|
||||||
|
std::fs::remove_file(&path)?;
|
||||||
|
println!(" text-based -> {}", md_path.display());
|
||||||
|
text_based += 1;
|
||||||
|
}
|
||||||
|
Ok(PdfProcessResult::Scanned) => {
|
||||||
|
println!(" scanned — keeping as PDF");
|
||||||
|
scanned += 1;
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!(" error: {}", e);
|
||||||
|
errors += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((text_based, scanned, errors))
|
||||||
|
}
|
||||||
+160
@@ -0,0 +1,160 @@
|
|||||||
|
use reqwest::Client;
|
||||||
|
use std::time::Duration;
|
||||||
|
use url::Url;
|
||||||
|
|
||||||
|
pub struct RobotsRule {
|
||||||
|
pub allowed: Vec<String>,
|
||||||
|
pub disallowed: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RobotsRule {
|
||||||
|
pub fn is_allowed(&self, path: &str) -> bool {
|
||||||
|
let mut matched_disallow_len = 0;
|
||||||
|
let mut matched_allow_len = 0;
|
||||||
|
|
||||||
|
for rule in &self.disallowed {
|
||||||
|
if path_matches(rule, path) && rule.len() > matched_disallow_len {
|
||||||
|
matched_disallow_len = rule.len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for rule in &self.allowed {
|
||||||
|
if path_matches(rule, path) && rule.len() > matched_allow_len {
|
||||||
|
matched_allow_len = rule.len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
matched_allow_len >= matched_disallow_len
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn path_matches(pattern: &str, path: &str) -> bool {
|
||||||
|
if pattern.is_empty() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if !pattern.contains('*') {
|
||||||
|
return path.starts_with(pattern) || path == pattern;
|
||||||
|
}
|
||||||
|
|
||||||
|
let parts: Vec<&str> = pattern.split('*').collect();
|
||||||
|
let mut pos = 0;
|
||||||
|
|
||||||
|
for (i, part) in parts.iter().enumerate() {
|
||||||
|
if i == 0 {
|
||||||
|
if !path.starts_with(part) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
pos = part.len();
|
||||||
|
} else {
|
||||||
|
match path[pos..].find(part) {
|
||||||
|
Some(idx) => pos += idx + part.len(),
|
||||||
|
None => return false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
true
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
|
||||||
|
let robots_url = format!(
|
||||||
|
"{}://{}/robots.txt",
|
||||||
|
base_url.scheme(),
|
||||||
|
base_url.host_str().unwrap_or("")
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut rule = RobotsRule {
|
||||||
|
allowed: Vec::new(),
|
||||||
|
disallowed: Vec::new(),
|
||||||
|
};
|
||||||
|
|
||||||
|
match client
|
||||||
|
.get(&robots_url)
|
||||||
|
.header("User-Agent", user_agent)
|
||||||
|
.timeout(Duration::from_secs(30))
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(resp) if resp.status().is_success() => {
|
||||||
|
if let Ok(text) = resp.text().await {
|
||||||
|
rule = parse_robots_txt(&text, user_agent);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(resp) => {
|
||||||
|
println!(
|
||||||
|
"robots.txt returned HTTP {} — assuming no restrictions",
|
||||||
|
resp.status()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
println!(
|
||||||
|
"Failed to fetch robots.txt ({}): assuming no restrictions",
|
||||||
|
e
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
rule
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
|
||||||
|
let target_lower = target_agent.to_lowercase();
|
||||||
|
let mut rule = RobotsRule {
|
||||||
|
allowed: Vec::new(),
|
||||||
|
disallowed: Vec::new(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut current_agents: Vec<String> = Vec::new();
|
||||||
|
let mut collecting = false;
|
||||||
|
|
||||||
|
for line in text.lines() {
|
||||||
|
let line = line.trim();
|
||||||
|
|
||||||
|
if line.is_empty() {
|
||||||
|
current_agents.clear();
|
||||||
|
collecting = false;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if line.starts_with('#') {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(agent) = line
|
||||||
|
.strip_prefix("User-agent:")
|
||||||
|
.or_else(|| line.strip_prefix("User-Agent:"))
|
||||||
|
{
|
||||||
|
let agent = agent.trim().to_lowercase();
|
||||||
|
current_agents.push(agent.clone());
|
||||||
|
|
||||||
|
if agent == "*" || agent == target_lower || agent == "web-scraper" {
|
||||||
|
collecting = true;
|
||||||
|
} else {
|
||||||
|
collecting = false;
|
||||||
|
}
|
||||||
|
} else if collecting {
|
||||||
|
if let Some(path) = line.strip_prefix("Disallow:") {
|
||||||
|
let path = path.trim().to_string();
|
||||||
|
if !path.is_empty() {
|
||||||
|
rule.disallowed.push(path);
|
||||||
|
}
|
||||||
|
} else if let Some(path) = line.strip_prefix("Allow:") {
|
||||||
|
let path = path.trim().to_string();
|
||||||
|
if !path.is_empty() {
|
||||||
|
rule.allowed.push(path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
|
||||||
|
println!(
|
||||||
|
"robots.txt: {} disallow rules, {} allow rules",
|
||||||
|
rule.disallowed.len(),
|
||||||
|
rule.allowed.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
rule
|
||||||
|
}
|
||||||
+108
-10
@@ -1,24 +1,43 @@
|
|||||||
// src/scraper.rs
|
use crate::converter::html_to_markdown;
|
||||||
|
|
||||||
use crate::error_logger::log_error;
|
use crate::error_logger::log_error;
|
||||||
use crate::extractor::extract_links;
|
use crate::extractor::extract_links;
|
||||||
use crate::fetcher::{FetchResult, Fetcher};
|
use crate::fetcher::{FetchResult, Fetcher};
|
||||||
|
use crate::pdf_processor::{PdfProcessResult, process_pdf};
|
||||||
|
use crate::robots::RobotsRule;
|
||||||
use crate::url_utils::{normalize_url, url_to_filename};
|
use crate::url_utils::{normalize_url, url_to_filename};
|
||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
use std::collections::HashSet;
|
use std::collections::{HashSet, VecDeque};
|
||||||
use std::collections::VecDeque;
|
|
||||||
use std::fs;
|
use std::fs;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
use tokio::sync::Mutex;
|
use tokio::sync::Mutex;
|
||||||
use url::Url;
|
use url::Url;
|
||||||
|
|
||||||
|
pub struct PdfStats {
|
||||||
|
pub text_based: usize,
|
||||||
|
pub scanned: usize,
|
||||||
|
pub errors: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Default for PdfStats {
|
||||||
|
fn default() -> Self {
|
||||||
|
Self {
|
||||||
|
text_based: 0,
|
||||||
|
scanned: 0,
|
||||||
|
errors: 0,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pub struct Scraper {
|
pub struct Scraper {
|
||||||
fetcher: Fetcher,
|
fetcher: Fetcher,
|
||||||
seen: Arc<Mutex<HashSet<String>>>,
|
seen: Arc<Mutex<HashSet<String>>>,
|
||||||
output_dir: std::path::PathBuf,
|
output_dir: std::path::PathBuf,
|
||||||
log_path: std::path::PathBuf,
|
log_path: std::path::PathBuf,
|
||||||
base_domain: String,
|
base_domain: String,
|
||||||
|
robots: RobotsRule,
|
||||||
|
pdf_stats: Arc<Mutex<PdfStats>>,
|
||||||
|
convert_pdfs: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Scraper {
|
impl Scraper {
|
||||||
@@ -26,6 +45,8 @@ impl Scraper {
|
|||||||
output_dir: std::path::PathBuf,
|
output_dir: std::path::PathBuf,
|
||||||
log_path: std::path::PathBuf,
|
log_path: std::path::PathBuf,
|
||||||
base_domain: String,
|
base_domain: String,
|
||||||
|
robots: RobotsRule,
|
||||||
|
convert_pdfs: bool,
|
||||||
) -> Self {
|
) -> Self {
|
||||||
Self {
|
Self {
|
||||||
fetcher: Fetcher::new(),
|
fetcher: Fetcher::new(),
|
||||||
@@ -33,6 +54,9 @@ impl Scraper {
|
|||||||
output_dir,
|
output_dir,
|
||||||
log_path,
|
log_path,
|
||||||
base_domain,
|
base_domain,
|
||||||
|
robots,
|
||||||
|
pdf_stats: Arc::new(Mutex::new(PdfStats::default())),
|
||||||
|
convert_pdfs,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -42,6 +66,7 @@ impl Scraper {
|
|||||||
|
|
||||||
let mut count = 0usize;
|
let mut count = 0usize;
|
||||||
let mut error_count = 0usize;
|
let mut error_count = 0usize;
|
||||||
|
let mut skipped_robots = 0usize;
|
||||||
|
|
||||||
while let Some(raw_url) = queue.pop_front() {
|
while let Some(raw_url) = queue.pop_front() {
|
||||||
let url = match normalize_url(raw_url.as_str()) {
|
let url = match normalize_url(raw_url.as_str()) {
|
||||||
@@ -59,21 +84,26 @@ impl Scraper {
|
|||||||
seen.insert(url_key.clone());
|
seen.insert(url_key.clone());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if !self.robots.is_allowed(url.path()) {
|
||||||
|
println!("[skip] robots.txt disallows: {}", url);
|
||||||
|
skipped_robots += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
println!("[{}] fetching: {}", count, url);
|
println!("[{}] fetching: {}", count, url);
|
||||||
|
|
||||||
match self.fetcher.fetch_with_retry(&url).await {
|
match self.fetcher.fetch_with_retry(&url).await {
|
||||||
Ok(result) => {
|
Ok(result) => {
|
||||||
let is_html = result.content_type.contains("text/html");
|
let is_html = result.content_type.contains("text/html");
|
||||||
|
let is_pdf = result.content_type.contains("application/pdf");
|
||||||
|
|
||||||
match self.save(&result).await {
|
match self.save(&result, is_html, is_pdf).await {
|
||||||
Ok(()) => {
|
Ok(()) => {
|
||||||
if is_html {
|
if is_html {
|
||||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||||
let links =
|
let links =
|
||||||
extract_links(html, &result.final_url, &self.base_domain);
|
extract_links(html, &result.final_url, &self.base_domain);
|
||||||
|
|
||||||
// Only queue links not already seen — do NOT insert into seen here.
|
|
||||||
// They get inserted when actually fetched, matching the Python version.
|
|
||||||
let mut new_links = Vec::new();
|
let mut new_links = Vec::new();
|
||||||
|
|
||||||
{
|
{
|
||||||
@@ -106,6 +136,11 @@ impl Scraper {
|
|||||||
None,
|
None,
|
||||||
);
|
);
|
||||||
error_count += 1;
|
error_count += 1;
|
||||||
|
|
||||||
|
if is_pdf {
|
||||||
|
let mut stats = self.pdf_stats.lock().await;
|
||||||
|
stats.errors += 1;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -119,13 +154,76 @@ impl Scraper {
|
|||||||
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
|
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
let stats = self.pdf_stats.lock().await;
|
||||||
|
if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 {
|
||||||
|
println!("\nPDF Statistics:");
|
||||||
|
println!(" Text-based (converted): {}", stats.text_based);
|
||||||
|
println!(" Scanned (kept as PDF): {}", stats.scanned);
|
||||||
|
println!(" Errors: {}", stats.errors);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if skipped_robots > 0 {
|
||||||
|
println!("Skipped {} URLs due to robots.txt", skipped_robots);
|
||||||
|
}
|
||||||
|
|
||||||
(count, error_count)
|
(count, error_count)
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn save(&self, result: &FetchResult) -> Result<()> {
|
async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> {
|
||||||
let filename = url_to_filename(&result.final_url, &result.content_type);
|
let mut filename = url_to_filename(&result.final_url, &result.content_type);
|
||||||
|
|
||||||
|
let content: Vec<u8>;
|
||||||
|
|
||||||
|
if is_html {
|
||||||
|
let html = std::str::from_utf8(&result.bytes)?;
|
||||||
|
let md = html_to_markdown(html);
|
||||||
|
if filename.ends_with(".html") {
|
||||||
|
filename = filename.replace(".html", ".md");
|
||||||
|
}
|
||||||
|
content = md.into_bytes();
|
||||||
|
} else if is_pdf && self.convert_pdfs {
|
||||||
|
// Save PDF to temp file for pdf-inspector processing
|
||||||
|
let temp_path = self.output_dir.join("temp_pdf.tmp");
|
||||||
|
fs::write(&temp_path, &result.bytes)?;
|
||||||
|
|
||||||
|
match process_pdf(&temp_path) {
|
||||||
|
Ok(PdfProcessResult::Markdown(md)) => {
|
||||||
|
if filename.ends_with(".pdf") {
|
||||||
|
filename = filename.replace(".pdf", ".md");
|
||||||
|
}
|
||||||
|
content = md.into_bytes();
|
||||||
|
fs::remove_file(&temp_path)?;
|
||||||
|
|
||||||
|
println!(" [PDF] text-based — converted to Markdown");
|
||||||
|
let mut stats = self.pdf_stats.lock().await;
|
||||||
|
stats.text_based += 1;
|
||||||
|
}
|
||||||
|
Ok(PdfProcessResult::Scanned) => {
|
||||||
|
println!(" [PDF] scanned document — saved as-is");
|
||||||
|
let mut stats = self.pdf_stats.lock().await;
|
||||||
|
stats.scanned += 1;
|
||||||
|
|
||||||
|
fs::rename(&temp_path, self.output_dir.join(&filename))?;
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
println!(" [PDF] processing error ({}), saved as-is", e);
|
||||||
|
let mut stats = self.pdf_stats.lock().await;
|
||||||
|
stats.errors += 1;
|
||||||
|
|
||||||
|
fs::rename(&temp_path, self.output_dir.join(&filename))?;
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// Non-HTML, non-PDF, or PDF conversion disabled — save as-is
|
||||||
|
content = result.bytes.clone();
|
||||||
|
}
|
||||||
|
|
||||||
let filepath = self.output_dir.join(&filename);
|
let filepath = self.output_dir.join(&filename);
|
||||||
fs::write(&filepath, &result.bytes)?;
|
fs::write(&filepath, &content)?;
|
||||||
println!(
|
println!(
|
||||||
" saved -> {}",
|
" saved -> {}",
|
||||||
filepath.file_name().unwrap_or_default().to_string_lossy()
|
filepath.file_name().unwrap_or_default().to_string_lossy()
|
||||||
|
|||||||
+4
-2
@@ -1,5 +1,3 @@
|
|||||||
// src/url_utils.rs
|
|
||||||
|
|
||||||
use md5::Md5;
|
use md5::Md5;
|
||||||
use sha2::{Digest, Sha256};
|
use sha2::{Digest, Sha256};
|
||||||
use url::Url;
|
use url::Url;
|
||||||
@@ -90,6 +88,10 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
|
|||||||
let path = url.path().trim_start_matches('/');
|
let path = url.path().trim_start_matches('/');
|
||||||
let path = if path.is_empty() { "index" } else { path };
|
let path = if path.is_empty() { "index" } else { path };
|
||||||
|
|
||||||
|
// FIX: strip trailing slash so /support/ -> "support" not ""
|
||||||
|
let path = path.trim_end_matches('/');
|
||||||
|
let path = if path.is_empty() { "index" } else { path };
|
||||||
|
|
||||||
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
let last_segment = path.rsplit('/').next().unwrap_or(path);
|
||||||
|
|
||||||
let (stem, ext_from_path) = match last_segment.rfind('.') {
|
let (stem, ext_from_path) = match last_segment.rfind('.') {
|
||||||
|
|||||||
Reference in New Issue
Block a user