add single page html to md

This commit is contained in:
2026-08-18 11:18:03 +01:00
parent 7f01f1a341
commit 9932b81627
2 changed files with 30 additions and 11 deletions
+11
View File
@@ -30,6 +30,12 @@ struct Args {
#[arg(help = "Starting URL to scrape")] #[arg(help = "Starting URL to scrape")]
start_url: String, start_url: String,
#[arg(
long,
help = "Only scrape the single page at START_URL, do not crawl for links"
)]
single: bool,
#[arg(long, help = "Save documents as-is, skip Markdown conversion")] #[arg(long, help = "Save documents as-is, skip Markdown conversion")]
no_doc_conversion: bool, no_doc_conversion: bool,
@@ -128,6 +134,10 @@ async fn main() -> Result<()> {
"Document conversion: {}", "Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" } if convert_docs { "enabled" } else { "disabled" }
); );
log::info!(
"Single page mode: {}",
if args.single { "enabled" } else { "disabled" }
);
match (&include_types, &exclude_types) { match (&include_types, &exclude_types) {
(Some(t), _) => log::info!("Type filter: include {:?}", t), (Some(t), _) => log::info!("Type filter: include {:?}", t),
@@ -160,6 +170,7 @@ async fn main() -> Result<()> {
include_types, include_types,
exclude_types, exclude_types,
scope_path, scope_path,
args.single,
); );
let (count, errors) = scraper.run(&url).await; let (count, errors) = scraper.run(&url).await;
+19 -11
View File
@@ -41,6 +41,7 @@ pub struct Scraper {
include_types: Option<HashSet<String>>, include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>, exclude_types: Option<HashSet<String>>,
scope_path: Option<String>, scope_path: Option<String>,
single_page: bool,
} }
impl Scraper { impl Scraper {
@@ -54,6 +55,7 @@ impl Scraper {
include_types: Option<HashSet<String>>, include_types: Option<HashSet<String>>,
exclude_types: Option<HashSet<String>>, exclude_types: Option<HashSet<String>>,
scope_path: Option<String>, scope_path: Option<String>,
single_page: bool,
) -> Self { ) -> Self {
Self { Self {
fetcher, fetcher,
@@ -67,6 +69,7 @@ impl Scraper {
include_types, include_types,
exclude_types, exclude_types,
scope_path, scope_path,
single_page,
} }
} }
@@ -164,20 +167,24 @@ impl Scraper {
} }
if is_html { if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) { if !self.single_page {
let links = extract_links(html, &final_url, &self.base_domain); if let Ok(html) = std::str::from_utf8(&result.bytes) {
let new_count = links.len(); let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let new_links: Vec<Url> = links let new_links: Vec<Url> = links
.into_iter() .into_iter()
.filter(|link| !self.seen.contains(link.as_str())) .filter(|link| !self.seen.contains(link.as_str()))
.collect(); .collect();
for link in &new_links { for link in &new_links {
queue.push_back(link.clone()); queue.push_back(link.clone());
}
log::info!(" found {} links ({} new)", new_count, new_links.len());
} }
} else {
log::info!(" found {} links ({} new)", new_count, new_links.len()); log::info!(" [single-page mode] not crawling for links");
} }
} }
} }
@@ -303,6 +310,7 @@ mod tests {
include_types, include_types,
exclude_types, exclude_types,
scope_path: None, scope_path: None,
single_page: false,
} }
} }