add single page html to md
This commit is contained in:
+11
@@ -30,6 +30,12 @@ struct Args {
|
|||||||
#[arg(help = "Starting URL to scrape")]
|
#[arg(help = "Starting URL to scrape")]
|
||||||
start_url: String,
|
start_url: String,
|
||||||
|
|
||||||
|
#[arg(
|
||||||
|
long,
|
||||||
|
help = "Only scrape the single page at START_URL, do not crawl for links"
|
||||||
|
)]
|
||||||
|
single: bool,
|
||||||
|
|
||||||
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
|
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
|
||||||
no_doc_conversion: bool,
|
no_doc_conversion: bool,
|
||||||
|
|
||||||
@@ -128,6 +134,10 @@ async fn main() -> Result<()> {
|
|||||||
"Document conversion: {}",
|
"Document conversion: {}",
|
||||||
if convert_docs { "enabled" } else { "disabled" }
|
if convert_docs { "enabled" } else { "disabled" }
|
||||||
);
|
);
|
||||||
|
log::info!(
|
||||||
|
"Single page mode: {}",
|
||||||
|
if args.single { "enabled" } else { "disabled" }
|
||||||
|
);
|
||||||
|
|
||||||
match (&include_types, &exclude_types) {
|
match (&include_types, &exclude_types) {
|
||||||
(Some(t), _) => log::info!("Type filter: include {:?}", t),
|
(Some(t), _) => log::info!("Type filter: include {:?}", t),
|
||||||
@@ -160,6 +170,7 @@ async fn main() -> Result<()> {
|
|||||||
include_types,
|
include_types,
|
||||||
exclude_types,
|
exclude_types,
|
||||||
scope_path,
|
scope_path,
|
||||||
|
args.single,
|
||||||
);
|
);
|
||||||
let (count, errors) = scraper.run(&url).await;
|
let (count, errors) = scraper.run(&url).await;
|
||||||
|
|
||||||
|
|||||||
@@ -41,6 +41,7 @@ pub struct Scraper {
|
|||||||
include_types: Option<HashSet<String>>,
|
include_types: Option<HashSet<String>>,
|
||||||
exclude_types: Option<HashSet<String>>,
|
exclude_types: Option<HashSet<String>>,
|
||||||
scope_path: Option<String>,
|
scope_path: Option<String>,
|
||||||
|
single_page: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Scraper {
|
impl Scraper {
|
||||||
@@ -54,6 +55,7 @@ impl Scraper {
|
|||||||
include_types: Option<HashSet<String>>,
|
include_types: Option<HashSet<String>>,
|
||||||
exclude_types: Option<HashSet<String>>,
|
exclude_types: Option<HashSet<String>>,
|
||||||
scope_path: Option<String>,
|
scope_path: Option<String>,
|
||||||
|
single_page: bool,
|
||||||
) -> Self {
|
) -> Self {
|
||||||
Self {
|
Self {
|
||||||
fetcher,
|
fetcher,
|
||||||
@@ -67,6 +69,7 @@ impl Scraper {
|
|||||||
include_types,
|
include_types,
|
||||||
exclude_types,
|
exclude_types,
|
||||||
scope_path,
|
scope_path,
|
||||||
|
single_page,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -164,6 +167,7 @@ impl Scraper {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if is_html {
|
if is_html {
|
||||||
|
if !self.single_page {
|
||||||
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
if let Ok(html) = std::str::from_utf8(&result.bytes) {
|
||||||
let links = extract_links(html, &final_url, &self.base_domain);
|
let links = extract_links(html, &final_url, &self.base_domain);
|
||||||
let new_count = links.len();
|
let new_count = links.len();
|
||||||
@@ -179,6 +183,9 @@ impl Scraper {
|
|||||||
|
|
||||||
log::info!(" found {} links ({} new)", new_count, new_links.len());
|
log::info!(" found {} links ({} new)", new_count, new_links.len());
|
||||||
}
|
}
|
||||||
|
} else {
|
||||||
|
log::info!(" [single-page mode] not crawling for links");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
@@ -303,6 +310,7 @@ mod tests {
|
|||||||
include_types,
|
include_types,
|
||||||
exclude_types,
|
exclude_types,
|
||||||
scope_path: None,
|
scope_path: None,
|
||||||
|
single_page: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user