diff --git a/Cargo.lock b/Cargo.lock index 28d98e2..f42c20d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -87,12 +87,39 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "anydoc" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93a8b0dbecf79be9329111093b493d256db2a573ed3f1cc827c96bdbbe32a936" +dependencies = [ + "calamine", + "cfb", + "csv", + "encoding_rs", + "flate2", + "log", + "pdf-inspector", + "quick-xml", + "zip", +] + [[package]] name = "anyhow" version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" +[[package]] +name = "atoi_simd" +version = "0.18.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f3cdb3708a128e559a30fb830e8a77a5022ee6902806925c216658652b452a44" +dependencies = [ + "debug_unsafe", + "rustversion", +] + [[package]] name = "atomic-waker" version = "1.1.2" @@ -159,6 +186,24 @@ version = "1.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" +[[package]] +name = "calamine" +version = "0.36.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fa68281b1a76b54a62156474adb06bb380a67e07dd60656e3217152b42183f3" +dependencies = [ + "atoi_simd", + "byteorder", + "chrono", + "codepage", + "encoding_rs", + "fast-float2", + "log", + "quick-xml", + "serde", + "zip", +] + [[package]] name = "cbc" version = "0.1.2" @@ -170,14 +215,25 @@ dependencies = [ [[package]] name = "cc" -version = "1.4.2" +version = "1.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d262e149917187838d5b42777c8253bcb64500067342904e7d429499a6f277e" +checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d" dependencies = [ "find-msvc-tools", "shlex", ] +[[package]] +name = "cfb" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a347dcabdae9c31b0825fd6a8bed285ec9c2acb89c47827126d52fa4f59cece3" +dependencies = [ + "fnv", + "uuid", + "web-time", +] + [[package]] name = "cfg-if" version = "1.0.4" @@ -219,6 +275,55 @@ dependencies = [ "inout", ] +[[package]] +name = "clap" +version = "4.6.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.6.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" +dependencies = [ + "anstream", + "anstyle", + "clap_lex", + "strsim", +] + +[[package]] +name = "clap_derive" +version = "4.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "clap_lex" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9" + +[[package]] +name = "codepage" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "48f68d061bc2828ae826206326e61251aca94c1e4a5305cf52d9138639c918b4" +dependencies = [ + "encoding_rs", +] + [[package]] name = "colorchoice" version = "1.0.5" @@ -336,6 +441,33 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "csv" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938" +dependencies = [ + "csv-core", + "itoa", + "ryu", + "serde_core", +] + +[[package]] +name = "csv-core" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782" +dependencies = [ + "memchr", +] + +[[package]] +name = "debug_unsafe" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7eed2c4702fa172d1ce21078faa7c5203e69f5394d48cc436d25928394a867a2" + [[package]] name = "defmt" version = "1.1.1" @@ -489,6 +621,12 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "fast-float2" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6e8948ce679d00a02a94739ea185595dca7118ed04feb991127e443bd3d761f" + [[package]] name = "fastrand" version = "2.5.0" @@ -497,9 +635,9 @@ checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "find-msvc-tools" -version = "0.1.10" +version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "26b73573e6edcd2af0cdf47bd6cb58f0b3839491263c314eaad1ccf24430e1de" +checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" [[package]] name = "flate2" @@ -509,6 +647,7 @@ checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" dependencies = [ "crc32fast", "miniz_oxide", + "zlib-rs", ] [[package]] @@ -668,6 +807,12 @@ version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + [[package]] name = "hex" version = "0.4.3" @@ -732,9 +877,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.4" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" dependencies = [ "bytes", "futures-core", @@ -852,9 +997,9 @@ dependencies = [ [[package]] name = "icu_collections" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" dependencies = [ "displaydoc", "potential_utf", @@ -866,9 +1011,9 @@ dependencies = [ [[package]] name = "icu_locale_core" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" dependencies = [ "displaydoc", "litemap", @@ -879,9 +1024,9 @@ dependencies = [ [[package]] name = "icu_normalizer" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" dependencies = [ "icu_collections", "icu_normalizer_data", @@ -893,16 +1038,17 @@ dependencies = [ [[package]] name = "icu_normalizer_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" [[package]] name = "icu_properties" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" dependencies = [ + "displaydoc", "icu_collections", "icu_locale_core", "icu_properties_data", @@ -913,15 +1059,15 @@ dependencies = [ [[package]] name = "icu_properties_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" [[package]] name = "icu_provider" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +checksum = "92a7ed671a6aad807a8651a2e1782a6598fda9ce5185dd8158549e95a91c6428" dependencies = [ "displaydoc", "icu_locale_core", @@ -1088,9 +1234,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" [[package]] name = "litemap" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" [[package]] name = "lock_api" @@ -1365,9 +1511,9 @@ dependencies = [ [[package]] name = "pdf-inspector" -version = "1.14.1" +version = "1.14.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f323996b5c71c5f48860be70c265965135c9bedb39a33271dfc95d5f53dbdb34" +checksum = "1e024ae242c514e2adf6aee186678e0eabdc2e5ecfbb2159186881b4498593cb" dependencies = [ "env_logger", "include_dir", @@ -1447,9 +1593,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "pkg-config" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" [[package]] name = "portable-atomic" @@ -1468,9 +1614,9 @@ dependencies = [ [[package]] name = "potential_utf" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" dependencies = [ "zerovec", ] @@ -1496,6 +1642,16 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "quick-xml" +version = "0.41.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" +dependencies = [ + "encoding_rs", + "memchr", +] + [[package]] name = "quote" version = "1.0.47" @@ -1545,9 +1701,9 @@ checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" [[package]] name = "rangemap" -version = "1.7.1" +version = "1.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "973443cf09a9c8656b574a866ab68dfa19f0867d0340648c7d2f6a71b8a8ea68" +checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d" [[package]] name = "rayon" @@ -1665,12 +1821,15 @@ dependencies = [ name = "rs-scraper" version = "0.1.0" dependencies = [ + "anydoc", "anyhow", "chrono", + "clap", + "env_logger", "hex", "htmd", + "log", "md-5", - "pdf-inspector", "reqwest", "scraper", "serde", @@ -1977,6 +2136,12 @@ dependencies = [ "unicode-properties", ] +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + [[package]] name = "subtle" version = "2.6.1" @@ -2122,9 +2287,9 @@ dependencies = [ [[package]] name = "tinystr" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" dependencies = [ "displaydoc", "zerovec", @@ -2283,6 +2448,12 @@ version = "0.25.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d2df906b07856748fa3f6e0ad0cbaa047052d4a7dd609e231c4f72cee8c36f31" +[[package]] +name = "typed-path" +version = "0.12.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e28f89b80c87b8fb0cf04ab448d5dd0dd0ade2f8891bae878de66a75a28600e" + [[package]] name = "typenum" version = "1.20.1" @@ -2358,6 +2529,16 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" +[[package]] +name = "uuid" +version = "1.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2cefc03fd367c0c6d4305de1b312cf00248c4114f4a0418ce6a6af769e3b0bd9" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + [[package]] name = "vcpkg" version = "0.2.15" @@ -2450,6 +2631,16 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "web-time" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + [[package]] name = "weezl" version = "0.1.12" @@ -2610,9 +2801,9 @@ checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" [[package]] name = "writeable" -version = "0.6.3" +version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" [[package]] name = "xml5ever" @@ -2677,9 +2868,9 @@ checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" [[package]] name = "zerotrie" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" dependencies = [ "displaydoc", "yoke", @@ -2688,9 +2879,9 @@ dependencies = [ [[package]] name = "zerovec" -version = "0.11.6" +version = "0.11.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +checksum = "94b5c6b5976d66c1d703c4fd17d3f5e43c8cedaacf604961b171adc7130896d8" dependencies = [ "yoke", "zerofrom", @@ -2699,17 +2890,49 @@ dependencies = [ [[package]] name = "zerovec-derive" -version = "0.11.3" +version = "0.11.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +checksum = "47402523226a02bfe5230160dc3ccc089aa6f6f19e7fcbb4e6f824bbb1b4aa62" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] +[[package]] +name = "zip" +version = "8.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d04a6b5381502aa6087c94c669499eb1602eb9c5e8198e534de571f7154809b" +dependencies = [ + "crc32fast", + "flate2", + "indexmap", + "memchr", + "typed-path", + "zopfli", +] + +[[package]] +name = "zlib-rs" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" + [[package]] name = "zmij" version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" + +[[package]] +name = "zopfli" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f05cd8797d63865425ff89b5c4a48804f35ba0ce8d125800027ad6017d2b5249" +dependencies = [ + "bumpalo", + "crc32fast", + "log", + "simd-adler32", +] diff --git a/Cargo.toml b/Cargo.toml index c2964c5..3b38b25 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,12 +4,15 @@ version = "0.1.0" edition = "2024" [dependencies] +anydoc = "0.1" anyhow = "1" chrono = { version = "0.4", features = ["serde"] } +clap = { version = "4", features = ["derive"] } hex = "0.4" htmd = "0.1" +log = "0.4" +env_logger = "0.11" md-5 = "0.10" -pdf-inspector = "1" reqwest = "0.12" scraper = "0.22" serde = { version = "1", features = ["derive"] } diff --git a/src/config.rs b/src/config.rs index 791281b..ab85a86 100644 --- a/src/config.rs +++ b/src/config.rs @@ -1,5 +1,9 @@ -pub const DELAY_MS: u64 = 0; +pub const DELAY_MS: u64 = 1000; pub const TIMEOUT_SECS: u64 = 60; pub const MAX_RETRIES: u32 = 3; pub const RETRY_BACKOFF_SECS: u64 = 5; pub const USER_AGENT: &str = "web-scraper/1.0 (research)"; + +/// Rate limit in milliseconds between requests. Set to 0 only for trusted/internal sites. +/// Default is 1000ms to avoid IP bans on public sites like logting.fo. +pub const RATE_LIMIT_MS: u64 = DELAY_MS; diff --git a/src/converter.rs b/src/converter.rs index 5b723da..bfa0b86 100644 --- a/src/converter.rs +++ b/src/converter.rs @@ -1,14 +1,14 @@ use htmd::HtmlToMarkdown; +use std::sync::OnceLock; + +static CONVERTER: OnceLock = OnceLock::new(); pub fn html_to_markdown(html: &str) -> String { - let converter = HtmlToMarkdown::new(); + let converter = CONVERTER.get_or_init(HtmlToMarkdown::new); match converter.convert(html) { Ok(md) => md, Err(e) => { - eprintln!( - " WARN: HTML-to-MD conversion failed ({}), saving raw HTML", - e - ); + log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e); html.to_string() } } diff --git a/src/doc_processor.rs b/src/doc_processor.rs new file mode 100644 index 0000000..9d762b4 --- /dev/null +++ b/src/doc_processor.rs @@ -0,0 +1,96 @@ +use anyhow::Result; +use std::path::Path; +use url::Url; + +pub enum DocProcessResult { + Markdown(String), + Raw, +} + +pub fn detect_format(bytes: &[u8], url: &Url) -> Option { + if let Some(fmt) = anydoc::Format::from_bytes(bytes) { + return Some(fmt); + } + anydoc::Format::from_path(Path::new(url.path())) +} + +pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult { + let Some(format) = detect_format(bytes, url) else { + return DocProcessResult::Raw; + }; + + match anydoc::to_markdown_bytes(bytes, Some(format)) { + Ok(md) if !md.trim().is_empty() => DocProcessResult::Markdown(md), + Ok(_) => { + log::warn!("anydoc produced empty output, saving raw"); + DocProcessResult::Raw + } + Err(e) => { + log::warn!("anydoc conversion failed ({}), saving raw", e); + DocProcessResult::Raw + } + } +} + +#[allow(dead_code)] +pub fn batch_convert_docs(dir: &Path) -> Result<(usize, usize, usize)> { + let supported_exts = [ + "pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls", + "xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv", + ]; + + let entries: Vec<_> = std::fs::read_dir(dir)? + .filter_map(|e| e.ok()) + .filter(|e| { + e.path().extension().is_some_and(|ext| { + let ext = ext.to_string_lossy().to_lowercase(); + supported_exts.contains(&ext.as_str()) + }) + }) + .collect(); + + let total = entries.len(); + log::info!("Found {} documents to process", total); + + let mut converted = 0usize; + let mut raw = 0usize; + let mut errors = 0usize; + + for (i, entry) in entries.iter().enumerate() { + let path = entry.path(); + let filename = path.file_name().unwrap_or_default().to_string_lossy(); + log::info!("[{}/{}] processing: {}", i + 1, total, filename); + + let bytes = match std::fs::read(&path) { + Ok(b) => b, + Err(e) => { + log::error!("error reading file: {}", e); + errors += 1; + continue; + } + }; + + let ext = path.extension().unwrap_or_default().to_string_lossy(); + let Some(format) = anydoc::Format::from_extension(ext.as_ref()) else { + log::info!(" unrecognized format — keeping as-is"); + raw += 1; + continue; + }; + + match anydoc::to_markdown_bytes(&bytes, Some(format)) { + Ok(md) if !md.trim().is_empty() => { + let md_path = path.with_extension("md"); + std::fs::write(&md_path, md)?; + std::fs::remove_file(&path)?; + log::info!(" converted -> {}", md_path.display()); + converted += 1; + } + _ => { + log::info!(" could not convert — keeping as-is"); + raw += 1; + } + } + } + + Ok((converted, raw, errors)) +} diff --git a/src/error_logger.rs b/src/error_logger.rs index f5aab5f..3c343aa 100644 --- a/src/error_logger.rs +++ b/src/error_logger.rs @@ -30,12 +30,22 @@ pub fn log_error( status_code, }; - let line = serde_json::to_string(&entry).unwrap_or_default() + "\n"; + let line = match serde_json::to_string(&entry) { + Ok(json) => json + "\n", + Err(e) => { + log::error!("Failed to serialize error entry: {}", e); + return; + } + }; if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) { - let _ = file.write_all(line.as_bytes()); - let _ = file.flush(); + if let Err(e) = file.write_all(line.as_bytes()) { + log::error!("Failed to write error log: {}", e); + } else { + let _ = file.flush(); + log::info!("LOGGED ERROR [{}]: {}", error_type, message); + } + } else { + log::error!("Failed to open error log file: {}", log_path.display()); } - - println!(" LOGGED ERROR [{}]: {}", error_type, message); } diff --git a/src/extractor.rs b/src/extractor.rs index 4a35f13..3287d8b 100644 --- a/src/extractor.rs +++ b/src/extractor.rs @@ -1,13 +1,20 @@ use scraper::{Html, Selector}; use std::collections::HashSet; +use std::sync::OnceLock; use url::Url; +static LINK_SELECTOR: OnceLock = OnceLock::new(); + +fn get_link_selector() -> &'static Selector { + LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid")) +} + pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet { let document = Html::parse_document(html); - let selector = Selector::parse("a[href]").unwrap(); + let selector = get_link_selector(); document - .select(&selector) + .select(selector) .filter_map(|el| el.value().attr("href")) .filter(|href| !href.is_empty()) .filter(|href| { diff --git a/src/fetcher.rs b/src/fetcher.rs index 51f13ce..7367ae7 100644 --- a/src/fetcher.rs +++ b/src/fetcher.rs @@ -1,5 +1,3 @@ -// src/fetcher.rs - use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT}; use anyhow::{Result, anyhow}; use reqwest::Client; @@ -11,15 +9,15 @@ pub struct Fetcher { } impl Fetcher { - pub fn new() -> Self { + pub fn new() -> Result { let client = Client::builder() .timeout(Duration::from_secs(TIMEOUT_SECS)) .user_agent(USER_AGENT) .redirect(reqwest::redirect::Policy::limited(10)) .build() - .expect("Failed to create HTTP client"); + .map_err(|e| anyhow!("Failed to create HTTP client: {}", e))?; - Self { client } + Ok(Self { client }) } pub async fn fetch_with_retry(&self, url: &Url) -> Result { @@ -39,7 +37,12 @@ impl Fetcher { Ok(b) => b.to_vec(), Err(e) => { if attempt < MAX_RETRIES { - self.print_retry(attempt, &e.to_string()); + log::warn!( + "retry {}/{}: body read failed ({})", + attempt, + MAX_RETRIES, + e + ); tokio::time::sleep(Duration::from_secs( RETRY_BACKOFF_SECS * attempt as u64, )) @@ -62,12 +65,11 @@ impl Fetcher { bytes, content_type, final_url, - _status_code: status.as_u16(), }); } Err(e) => { if attempt < MAX_RETRIES { - self.print_retry(attempt, &e.to_string()); + log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e); tokio::time::sleep(Duration::from_secs( RETRY_BACKOFF_SECS * attempt as u64, )) @@ -85,19 +87,10 @@ impl Fetcher { Err(anyhow!("Exhausted all {} retries", MAX_RETRIES)) } - - fn print_retry(&self, attempt: u32, error: &str) { - let wait = RETRY_BACKOFF_SECS * attempt as u64; - println!( - " retry {}/{} in {}s... ({})", - attempt, MAX_RETRIES, wait, error - ); - } } pub struct FetchResult { pub bytes: Vec, pub content_type: String, pub final_url: Url, - pub _status_code: u16, } diff --git a/src/main.rs b/src/main.rs index ee3529a..6f788a3 100644 --- a/src/main.rs +++ b/src/main.rs @@ -1,60 +1,42 @@ mod config; mod converter; +mod doc_processor; mod error_logger; mod extractor; mod fetcher; -mod pdf_processor; mod robots; mod scraper; mod url_utils; -use crate::config::USER_AGENT; use crate::robots::fetch_robots; use crate::scraper::Scraper; +use anyhow::Result; +use clap::Parser; use std::fs; use std::path::PathBuf; use url::Url; +#[derive(Parser, Debug)] +#[command(name = "rs-scraper")] +#[command(author = "FLÓ")] +#[command(version = "0.1.0")] +#[command(about = "Web scraper for Faroese public sector sites")] +struct Args { + #[arg(help = "Starting URL to scrape")] + start_url: String, + + #[arg(long, help = "Save documents as-is, skip Markdown conversion")] + no_doc_conversion: bool, +} + #[tokio::main] -async fn main() -> anyhow::Result<()> { - let args: Vec = std::env::args().collect(); +async fn main() -> Result<()> { + env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init(); - // Parse flags - let mut convert_pdfs = true; - let mut url_arg: Option = None; + let args = Args::parse(); - for arg in args.iter().skip(1) { - match arg.as_str() { - "--no-pdf-conversion" => convert_pdfs = false, - "--help" | "-h" => { - println!("Usage: site-scraper [OPTIONS] "); - println!(); - println!("Options:"); - println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction"); - println!(" -h, --help Show this help message"); - println!(); - println!("Example:"); - println!(" site-scraper https://www.logting.fo"); - println!(" site-scraper --no-pdf-conversion https://taks.fo"); - std::process::exit(0); - } - _ => { - if url_arg.is_none() { - url_arg = Some(arg.clone()); - } - } - } - } - - let start_url = match url_arg { - Some(u) => u, - None => { - eprintln!("Usage: site-scraper [OPTIONS] "); - eprintln!("Example: site-scraper https://www.logting.fo"); - eprintln!("Run with --help for options."); - std::process::exit(1); - } - }; + let start_url = args.start_url; + let convert_docs = !args.no_doc_conversion; let url = Url::parse(&start_url)?; @@ -70,29 +52,36 @@ async fn main() -> anyhow::Result<()> { let log_path = logs_dir.join("scrape_errors.jsonl"); fs::remove_file(&log_path).ok(); - println!("Scraping: {}", start_url); - println!("Base domain: {}", base_domain); - println!("Output dir: {}", output_dir.display()); - println!( - "PDF conversion: {}", - if convert_pdfs { "enabled" } else { "disabled" } + log::info!("Scraping: {}", start_url); + log::info!("Base domain: {}", base_domain); + log::info!("Output dir: {}", output_dir.display()); + log::info!( + "Document conversion: {}", + if convert_docs { "enabled" } else { "disabled" } ); - // Fetch robots.txt let client = reqwest::Client::builder() .timeout(std::time::Duration::from_secs(30)) - .user_agent(USER_AGENT) + .user_agent(config::USER_AGENT) .build()?; - println!("Fetching robots.txt..."); - let robots = fetch_robots(&client, &url, "web-scraper").await; - println!(); + log::info!("Fetching robots.txt..."); + let robots = fetch_robots(&client, &url, config::USER_AGENT).await; + log::info!(""); - let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs); + let fetcher = fetcher::Fetcher::new()?; + let scraper = Scraper::new( + output_dir, + log_path, + base_domain, + robots, + convert_docs, + fetcher, + ); let (count, errors) = scraper.run(&url).await; - println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors); - println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name); + log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors); + log::info!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name); Ok(()) } diff --git a/src/pdf_processor.rs b/src/pdf_processor.rs deleted file mode 100644 index e8a1f33..0000000 --- a/src/pdf_processor.rs +++ /dev/null @@ -1,71 +0,0 @@ -// src/pdf_processor.rs - -use anyhow::Result; -use std::path::Path; - -pub enum PdfProcessResult { - Markdown(String), - Scanned, -} - -pub fn process_pdf(pdf_path: &Path) -> Result { - use pdf_inspector::process_pdf; - - let result = process_pdf(pdf_path)?; - - match result.pdf_type { - pdf_inspector::PdfType::TextBased | pdf_inspector::PdfType::Mixed => { - match &result.markdown { - Some(md) if !md.trim().is_empty() => Ok(PdfProcessResult::Markdown(md.clone())), - _ => Ok(PdfProcessResult::Scanned), - } - } - pdf_inspector::PdfType::Scanned | pdf_inspector::PdfType::ImageBased => { - Ok(PdfProcessResult::Scanned) - } - } -} - -pub fn batch_convert_pdfs(dir: &Path) -> Result<(usize, usize, usize)> { - let mut text_based = 0usize; - let mut scanned = 0usize; - let mut errors = 0usize; - - let entries: Vec<_> = std::fs::read_dir(dir)? - .filter_map(|e| e.ok()) - .filter(|e| { - e.path() - .extension() - .is_some_and(|ext| ext.eq_ignore_ascii_case("pdf")) - }) - .collect(); - - let total = entries.len(); - println!("Found {} PDFs to process", total); - - for (i, entry) in entries.iter().enumerate() { - let path = entry.path(); - let filename = path.file_name().unwrap_or_default().to_string_lossy(); - println!("[{}/{}] processing: {}", i + 1, total, filename); - - match process_pdf(&path) { - Ok(PdfProcessResult::Markdown(md)) => { - let md_path = path.with_extension("md"); - std::fs::write(&md_path, md)?; - std::fs::remove_file(&path)?; - println!(" text-based -> {}", md_path.display()); - text_based += 1; - } - Ok(PdfProcessResult::Scanned) => { - println!(" scanned — keeping as PDF"); - scanned += 1; - } - Err(e) => { - eprintln!(" error: {}", e); - errors += 1; - } - } - } - - Ok((text_based, scanned, errors)) -} diff --git a/src/robots.rs b/src/robots.rs index 9d13ee0..d3ef40b 100644 --- a/src/robots.rs +++ b/src/robots.rs @@ -58,11 +58,16 @@ fn path_matches(pattern: &str, path: &str) -> bool { } pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule { - let robots_url = format!( - "{}://{}/robots.txt", - base_url.scheme(), - base_url.host_str().unwrap_or("") - ); + let host_str = base_url.host_str().unwrap_or(""); + if host_str.is_empty() { + log::warn!("No host in base URL, skipping robots.txt"); + return RobotsRule { + allowed: Vec::new(), + disallowed: Vec::new(), + }; + } + + let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str); let mut rule = RobotsRule { allowed: Vec::new(), @@ -82,13 +87,13 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> } } Ok(resp) => { - println!( + log::warn!( "robots.txt returned HTTP {} — assuming no restrictions", resp.status() ); } Err(e) => { - println!( + log::warn!( "Failed to fetch robots.txt ({}): assuming no restrictions", e ); @@ -149,7 +154,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule { } if !rule.disallowed.is_empty() || !rule.allowed.is_empty() { - println!( + log::info!( "robots.txt: {} disallow rules, {} allow rules", rule.disallowed.len(), rule.allowed.len() @@ -158,3 +163,62 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule { rule } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_path_matches_exact() { + assert!(path_matches("/admin", "/admin")); + assert!(path_matches("/admin", "/admin/users")); + assert!(path_matches("/admin", "/administrator")); + } + + #[test] + fn test_path_matches_wildcard() { + assert!(path_matches("/private/*", "/private/data")); + assert!(path_matches("/private/*/secret", "/private/x/secret")); + assert!(!path_matches("/private/*/secret", "/private/secret")); + } + + #[test] + fn test_path_matches_empty_pattern() { + assert!(!path_matches("", "/anything")); + } + + #[test] + fn test_is_allowed_disallow_takes_precedence_when_longer() { + let rule = RobotsRule { + allowed: vec!["/a".into()], + disallowed: vec!["/a/b".into()], + }; + assert!(!rule.is_allowed("/a/b")); + assert!(rule.is_allowed("/a/c")); + } + + #[test] + fn test_is_allowed_allow_overrides_disallow_when_longer() { + let rule = RobotsRule { + allowed: vec!["/public/pages".into()], + disallowed: vec!["/public".into()], + }; + assert!(rule.is_allowed("/public/pages")); + } + + #[test] + fn test_parse_robots_txt_wildcard_agent() { + let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n"; + let rule = parse_robots_txt(txt, "web-scraper"); + assert!(rule.disallowed.contains(&"/admin".to_string())); + assert!(rule.allowed.contains(&"/admin/public".to_string())); + } + + #[test] + fn test_parse_robots_txt_specific_agent() { + let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n"; + let rule = parse_robots_txt(txt, "web-scraper"); + assert!(rule.disallowed.contains(&"/private".to_string())); + assert!(!rule.disallowed.contains(&"/".to_string())); + } +} diff --git a/src/scraper.rs b/src/scraper.rs index 9d5257d..065cca3 100644 --- a/src/scraper.rs +++ b/src/scraper.rs @@ -1,8 +1,8 @@ use crate::converter::html_to_markdown; +use crate::doc_processor::{DocProcessResult, try_convert}; use crate::error_logger::log_error; use crate::extractor::extract_links; use crate::fetcher::{FetchResult, Fetcher}; -use crate::pdf_processor::{PdfProcessResult, process_pdf}; use crate::robots::RobotsRule; use crate::url_utils::{normalize_url, url_to_filename}; use anyhow::Result; @@ -13,17 +13,17 @@ use std::time::Duration; use tokio::sync::Mutex; use url::Url; -pub struct PdfStats { - pub text_based: usize, - pub scanned: usize, +pub struct DocStats { + pub converted: usize, + pub raw: usize, pub errors: usize, } -impl Default for PdfStats { +impl Default for DocStats { fn default() -> Self { Self { - text_based: 0, - scanned: 0, + converted: 0, + raw: 0, errors: 0, } } @@ -36,8 +36,8 @@ pub struct Scraper { log_path: std::path::PathBuf, base_domain: String, robots: RobotsRule, - pdf_stats: Arc>, - convert_pdfs: bool, + doc_stats: Arc>, + convert_docs: bool, } impl Scraper { @@ -46,17 +46,18 @@ impl Scraper { log_path: std::path::PathBuf, base_domain: String, robots: RobotsRule, - convert_pdfs: bool, + convert_docs: bool, + fetcher: Fetcher, ) -> Self { Self { - fetcher: Fetcher::new(), + fetcher, seen: Arc::new(Mutex::new(HashSet::new())), output_dir, log_path, base_domain, robots, - pdf_stats: Arc::new(Mutex::new(PdfStats::default())), - convert_pdfs, + doc_stats: Arc::new(Mutex::new(DocStats::default())), + convert_docs, } } @@ -85,62 +86,58 @@ impl Scraper { } if !self.robots.is_allowed(url.path()) { - println!("[skip] robots.txt disallows: {}", url); + log::info!("[skip] robots.txt disallows: {}", url); skipped_robots += 1; continue; } - println!("[{}] fetching: {}", count, url); + log::info!("[{}] fetching: {}", count, url); match self.fetcher.fetch_with_retry(&url).await { Ok(result) => { + let final_url = result.final_url.clone(); let is_html = result.content_type.contains("text/html"); - let is_pdf = result.content_type.contains("application/pdf"); - match self.save(&result, is_html, is_pdf).await { + match self.save(&result, is_html).await { Ok(()) => { if is_html { if let Ok(html) = std::str::from_utf8(&result.bytes) { - let links = - extract_links(html, &result.final_url, &self.base_domain); + let links = extract_links(html, &final_url, &self.base_domain); + let new_count = links.len(); - let mut new_links = Vec::new(); - - { + let new_links: Vec = { let seen = self.seen.lock().await; - for link in &links { - if !seen.contains(link.as_str()) { - new_links.push(link.clone()); - } - } + links + .into_iter() + .filter(|link| !seen.contains(link.as_str())) + .collect() + }; + + for link in &new_links { + queue.push_back(link.clone()); } - let new_count = new_links.len(); - - for link in new_links { - queue.push_back(link); - } - - println!(" found {} links ({} new)", links.len(), new_count); + log::info!( + " found {} links ({} new)", + new_count, + new_links.len() + ); } } else { - println!(" binary: {}", result.content_type); + log::info!(" binary: {}", result.content_type); } } Err(e) => { log_error( &self.log_path, - &result.final_url, + &final_url, "save_error", &e.to_string(), None, ); error_count += 1; - - if is_pdf { - let mut stats = self.pdf_stats.lock().await; - stats.errors += 1; - } + let mut stats = self.doc_stats.lock().await; + stats.errors += 1; } } } @@ -151,29 +148,28 @@ impl Scraper { } count += 1; - tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await; + tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await; } { - let stats = self.pdf_stats.lock().await; - if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 { - println!("\nPDF Statistics:"); - println!(" Text-based (converted): {}", stats.text_based); - println!(" Scanned (kept as PDF): {}", stats.scanned); - println!(" Errors: {}", stats.errors); + let stats = self.doc_stats.lock().await; + if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 { + log::info!("\nDocument Statistics:"); + log::info!(" Converted to Markdown: {}", stats.converted); + log::info!(" Kept as-is: {}", stats.raw); + log::info!(" Errors: {}", stats.errors); } } if skipped_robots > 0 { - println!("Skipped {} URLs due to robots.txt", skipped_robots); + log::info!("Skipped {} URLs due to robots.txt", skipped_robots); } (count, error_count) } - async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> { + async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> { let mut filename = url_to_filename(&result.final_url, &result.content_type); - let content: Vec; if is_html { @@ -183,51 +179,60 @@ impl Scraper { filename = filename.replace(".html", ".md"); } content = md.into_bytes(); - } else if is_pdf && self.convert_pdfs { - // Save PDF to temp file for pdf-inspector processing - let temp_path = self.output_dir.join("temp_pdf.tmp"); - fs::write(&temp_path, &result.bytes)?; - - match process_pdf(&temp_path) { - Ok(PdfProcessResult::Markdown(md)) => { - if filename.ends_with(".pdf") { - filename = filename.replace(".pdf", ".md"); + } else if self.convert_docs { + match try_convert(&result.bytes, &result.final_url) { + DocProcessResult::Markdown(md) => { + if let Some(pos) = filename.rfind('.') { + filename.truncate(pos); } + filename.push_str(".md"); content = md.into_bytes(); - fs::remove_file(&temp_path)?; - - println!(" [PDF] text-based — converted to Markdown"); - let mut stats = self.pdf_stats.lock().await; - stats.text_based += 1; + log::info!(" [DOC] converted to Markdown"); + let mut stats = self.doc_stats.lock().await; + stats.converted += 1; } - Ok(PdfProcessResult::Scanned) => { - println!(" [PDF] scanned document — saved as-is"); - let mut stats = self.pdf_stats.lock().await; - stats.scanned += 1; - - fs::rename(&temp_path, self.output_dir.join(&filename))?; - return Ok(()); - } - Err(e) => { - println!(" [PDF] processing error ({}), saved as-is", e); - let mut stats = self.pdf_stats.lock().await; - stats.errors += 1; - - fs::rename(&temp_path, self.output_dir.join(&filename))?; - return Ok(()); + DocProcessResult::Raw => { + content = result.bytes.clone(); + if is_document_content_type(&result.content_type) { + let mut stats = self.doc_stats.lock().await; + stats.raw += 1; + } } } } else { - // Non-HTML, non-PDF, or PDF conversion disabled — save as-is content = result.bytes.clone(); } let filepath = self.output_dir.join(&filename); fs::write(&filepath, &content)?; - println!( + log::info!( " saved -> {}", filepath.file_name().unwrap_or_default().to_string_lossy() ); Ok(()) } } + +fn is_document_content_type(ct: &str) -> bool { + let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase(); + const DOC_TYPES: &[&str] = &[ + "application/pdf", + "application/msword", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + "application/vnd.ms-word.document.macroenabled.12", + "application/vnd.ms-powerpoint", + "application/vnd.openxmlformats-officedocument.presentationml.presentation", + "application/vnd.ms-powerpoint.presentation.macroenabled.12", + "application/vnd.ms-excel", + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + "application/vnd.ms-excel.sheet.macroenabled.12", + "application/vnd.ms-excel.sheet.binary.macroenabled.12", + "application/vnd.oasis.opendocument.text", + "application/vnd.oasis.opendocument.spreadsheet", + "application/vnd.oasis.opendocument.presentation", + "application/rtf", + "application/epub+zip", + "text/csv", + ]; + DOC_TYPES.contains(&ct.as_str()) +} diff --git a/src/url_utils.rs b/src/url_utils.rs index eb33ecf..c2c3d88 100644 --- a/src/url_utils.rs +++ b/src/url_utils.rs @@ -14,7 +14,12 @@ pub fn derive_base_domain(url: &Url) -> Option { pub fn is_internal(url: &Url, base_domain: &str) -> bool { match url.host_str() { - Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)), + Some(host) => { + host == base_domain + || host + .strip_suffix(base_domain) + .is_some_and(|rest| rest.ends_with('.')) + } None => false, } } @@ -50,6 +55,7 @@ fn get_extension(url: &Url, content_type: &str) -> String { "application/json" => ".json", "application/xml" | "text/xml" => ".xml", "text/plain" => ".txt", + "text/csv" => ".csv", "text/css" => ".css", "application/javascript" | "text/javascript" => ".js", "image/png" => ".png", @@ -57,10 +63,21 @@ fn get_extension(url: &Url, content_type: &str) -> String { "image/gif" => ".gif", "image/svg+xml" => ".svg", "image/webp" => ".webp", - "application/vnd.ms-excel" => ".xls", - "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx", + "application/rtf" => ".rtf", + "application/epub+zip" => ".epub", "application/msword" => ".doc", "application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx", + "application/vnd.ms-word.document.macroenabled.12" => ".docm", + "application/vnd.ms-powerpoint" => ".ppt", + "application/vnd.openxmlformats-officedocument.presentationml.presentation" => ".pptx", + "application/vnd.ms-powerpoint.presentation.macroenabled.12" => ".pptm", + "application/vnd.ms-excel" => ".xls", + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx", + "application/vnd.ms-excel.sheet.macroenabled.12" => ".xlsm", + "application/vnd.ms-excel.sheet.binary.macroenabled.12" => ".xlsb", + "application/vnd.oasis.opendocument.text" => ".odt", + "application/vnd.oasis.opendocument.spreadsheet" => ".ods", + "application/vnd.oasis.opendocument.presentation" => ".odp", "application/zip" => ".zip", "application/octet-stream" => ".bin", _ => ".bin", @@ -84,14 +101,10 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String { .unwrap_or("unknown") .replace('.', "_") .replace(':', "_"); - let path = url.path().trim_start_matches('/'); let path = if path.is_empty() { "index" } else { path }; - - // FIX: strip trailing slash so /support/ -> "support" not "" let path = path.trim_end_matches('/'); let path = if path.is_empty() { "index" } else { path }; - let last_segment = path.rsplit('/').next().unwrap_or(path); let (stem, ext_from_path) = match last_segment.rfind('.') { @@ -114,9 +127,11 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String { _ => String::new(), }; - let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext); + let mut hasher = Sha256::new(); + hasher.update(url_str.as_bytes()); + let url_hash = hex::encode(hasher.finalize()); - filename + let safe_stem: String = stem .chars() .map(|c| { if c.is_alphanumeric() || "._-".contains(c) { @@ -125,5 +140,123 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String { '_' } }) - .collect() + .collect(); + + format!( + "{}_{}{}{:}{}", + host, + safe_stem, + query_suffix, + &url_hash[..8], + ext + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_derive_base_domain() { + let url = Url::parse("https://www.logting.fo/").unwrap(); + assert_eq!(derive_base_domain(&url), Some("logting.fo".into())); + + let url = Url::parse("https://taks.fo/en/skatur/").unwrap(); + assert_eq!(derive_base_domain(&url), Some("taks.fo".into())); + + let url = Url::parse("https://localhost:8080/").unwrap(); + assert_eq!(derive_base_domain(&url), Some("localhost".into())); + } + + #[test] + fn test_is_internal_subdomain() { + let base = "logting.fo"; + assert!(is_internal( + &Url::parse("https://logting.fo/").unwrap(), + base + )); + assert!(is_internal( + &Url::parse("https://www.logting.fo/").unwrap(), + base + )); + assert!(is_internal( + &Url::parse("https://sub.www.logting.fo/").unwrap(), + base + )); + assert!(!is_internal( + &Url::parse("https://evilogting.fo/").unwrap(), + base + )); + assert!(!is_internal( + &Url::parse("https://logting.com/").unwrap(), + base + )); + } + + #[test] + fn test_normalize_url_strips_fragment() { + let url = normalize_url("https://example.com/page#section").unwrap(); + assert_eq!(url.as_str(), "https://example.com/page"); + } + + #[test] + fn test_normalize_url_strips_trailing_query_chars() { + let url = normalize_url("https://example.com/page?&").unwrap(); + assert_eq!(url.as_str(), "https://example.com/page"); + } + + #[test] + fn test_url_to_filename_faroese_unicode() { + let url = Url::parse("https://logting.fo/lov/tinglýsing/2024/").unwrap(); + let filename = url_to_filename(&url, "text/html"); + assert!(filename.starts_with("logting_fo_")); + assert!(filename.ends_with(".html") || filename.ends_with(".md")); + assert!(!filename.contains('ý')); + } + + #[test] + fn test_url_to_filename_long_url_hashed() { + let long = format!("https://example.com/{}", "a".repeat(200)); + let url = Url::parse(&long).unwrap(); + let filename = url_to_filename(&url, "text/html"); + assert!(filename.len() < 30); + assert!(filename.ends_with(".html")); + } + + #[test] + fn test_url_to_filename_with_query() { + let url = Url::parse("https://example.com/page?id=42&sort=desc").unwrap(); + let filename = url_to_filename(&url, "text/html"); + assert!(filename.contains('_')); + assert!(filename.ends_with(".html")); + } + + #[test] + fn test_get_extension_from_content_type() { + let url = Url::parse("https://example.com/").unwrap(); + assert_eq!(get_extension(&url, "text/html"), ".html"); + assert_eq!( + get_extension(&url, "application/pdf; charset=binary"), + ".pdf" + ); + assert_eq!(get_extension(&url, "application/octet-stream"), ".bin"); + } + + #[test] + fn test_get_extension_from_url_path() { + let url = Url::parse("https://example.com/doc.pdf").unwrap(); + assert_eq!(get_extension(&url, "application/octet-stream"), ".pdf"); + } + + #[test] + fn test_url_collision_avoidance() { + let url1 = Url::parse("https://example.com/foo-bar").unwrap(); + let url2 = Url::parse("https://example.com/foo.bar").unwrap(); + let fn1 = url_to_filename(&url1, "text/html"); + let fn2 = url_to_filename(&url2, "text/html"); + assert_ne!( + fn1, fn2, + "Different URLs should produce different filenames" + ); + } }