qa and review

This commit is contained in:
2026-08-15 21:21:43 +01:00
parent a60cd5deae
commit 3c7043562e
13 changed files with 753 additions and 297 deletions
Generated
+265 -42
View File
@@ -87,12 +87,39 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "anydoc"
version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "93a8b0dbecf79be9329111093b493d256db2a573ed3f1cc827c96bdbbe32a936"
dependencies = [
"calamine",
"cfb",
"csv",
"encoding_rs",
"flate2",
"log",
"pdf-inspector",
"quick-xml",
"zip",
]
[[package]]
name = "anyhow"
version = "1.0.104"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470"
[[package]]
name = "atoi_simd"
version = "0.18.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f3cdb3708a128e559a30fb830e8a77a5022ee6902806925c216658652b452a44"
dependencies = [
"debug_unsafe",
"rustversion",
]
[[package]]
name = "atomic-waker"
version = "1.1.2"
@@ -159,6 +186,24 @@ version = "1.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04"
[[package]]
name = "calamine"
version = "0.36.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5fa68281b1a76b54a62156474adb06bb380a67e07dd60656e3217152b42183f3"
dependencies = [
"atoi_simd",
"byteorder",
"chrono",
"codepage",
"encoding_rs",
"fast-float2",
"log",
"quick-xml",
"serde",
"zip",
]
[[package]]
name = "cbc"
version = "0.1.2"
@@ -170,14 +215,25 @@ dependencies = [
[[package]]
name = "cc"
version = "1.4.2"
version = "1.4.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5d262e149917187838d5b42777c8253bcb64500067342904e7d429499a6f277e"
checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d"
dependencies = [
"find-msvc-tools",
"shlex",
]
[[package]]
name = "cfb"
version = "0.14.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a347dcabdae9c31b0825fd6a8bed285ec9c2acb89c47827126d52fa4f59cece3"
dependencies = [
"fnv",
"uuid",
"web-time",
]
[[package]]
name = "cfg-if"
version = "1.0.4"
@@ -219,6 +275,55 @@ dependencies = [
"inout",
]
[[package]]
name = "clap"
version = "4.6.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca"
dependencies = [
"clap_builder",
"clap_derive",
]
[[package]]
name = "clap_builder"
version = "4.6.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889"
dependencies = [
"anstream",
"anstyle",
"clap_lex",
"strsim",
]
[[package]]
name = "clap_derive"
version = "4.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061"
dependencies = [
"heck",
"proc-macro2",
"quote",
"syn 3.0.3",
]
[[package]]
name = "clap_lex"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9"
[[package]]
name = "codepage"
version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "48f68d061bc2828ae826206326e61251aca94c1e4a5305cf52d9138639c918b4"
dependencies = [
"encoding_rs",
]
[[package]]
name = "colorchoice"
version = "1.0.5"
@@ -336,6 +441,33 @@ dependencies = [
"syn 2.0.119",
]
[[package]]
name = "csv"
version = "1.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938"
dependencies = [
"csv-core",
"itoa",
"ryu",
"serde_core",
]
[[package]]
name = "csv-core"
version = "0.1.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782"
dependencies = [
"memchr",
]
[[package]]
name = "debug_unsafe"
version = "0.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7eed2c4702fa172d1ce21078faa7c5203e69f5394d48cc436d25928394a867a2"
[[package]]
name = "defmt"
version = "1.1.1"
@@ -489,6 +621,12 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "fast-float2"
version = "0.2.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c6e8948ce679d00a02a94739ea185595dca7118ed04feb991127e443bd3d761f"
[[package]]
name = "fastrand"
version = "2.5.0"
@@ -497,9 +635,9 @@ checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223"
[[package]]
name = "find-msvc-tools"
version = "0.1.10"
version = "0.1.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "26b73573e6edcd2af0cdf47bd6cb58f0b3839491263c314eaad1ccf24430e1de"
checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890"
[[package]]
name = "flate2"
@@ -509,6 +647,7 @@ checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c"
dependencies = [
"crc32fast",
"miniz_oxide",
"zlib-rs",
]
[[package]]
@@ -668,6 +807,12 @@ version = "0.17.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a"
[[package]]
name = "heck"
version = "0.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
[[package]]
name = "hex"
version = "0.4.3"
@@ -732,9 +877,9 @@ dependencies = [
[[package]]
name = "http-body-util"
version = "0.1.4"
version = "0.1.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2"
checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c"
dependencies = [
"bytes",
"futures-core",
@@ -852,9 +997,9 @@ dependencies = [
[[package]]
name = "icu_collections"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c"
checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513"
dependencies = [
"displaydoc",
"potential_utf",
@@ -866,9 +1011,9 @@ dependencies = [
[[package]]
name = "icu_locale_core"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29"
checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb"
dependencies = [
"displaydoc",
"litemap",
@@ -879,9 +1024,9 @@ dependencies = [
[[package]]
name = "icu_normalizer"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4"
checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f"
dependencies = [
"icu_collections",
"icu_normalizer_data",
@@ -893,16 +1038,17 @@ dependencies = [
[[package]]
name = "icu_normalizer_data"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38"
checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0"
[[package]]
name = "icu_properties"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de"
checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148"
dependencies = [
"displaydoc",
"icu_collections",
"icu_locale_core",
"icu_properties_data",
@@ -913,15 +1059,15 @@ dependencies = [
[[package]]
name = "icu_properties_data"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14"
checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa"
[[package]]
name = "icu_provider"
version = "2.2.0"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421"
checksum = "92a7ed671a6aad807a8651a2e1782a6598fda9ce5185dd8158549e95a91c6428"
dependencies = [
"displaydoc",
"icu_locale_core",
@@ -1088,9 +1234,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
[[package]]
name = "litemap"
version = "0.8.2"
version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0"
checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae"
[[package]]
name = "lock_api"
@@ -1365,9 +1511,9 @@ dependencies = [
[[package]]
name = "pdf-inspector"
version = "1.14.1"
version = "1.14.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f323996b5c71c5f48860be70c265965135c9bedb39a33271dfc95d5f53dbdb34"
checksum = "1e024ae242c514e2adf6aee186678e0eabdc2e5ecfbb2159186881b4498593cb"
dependencies = [
"env_logger",
"include_dir",
@@ -1447,9 +1593,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd"
[[package]]
name = "pkg-config"
version = "0.3.33"
version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e"
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
[[package]]
name = "portable-atomic"
@@ -1468,9 +1614,9 @@ dependencies = [
[[package]]
name = "potential_utf"
version = "0.1.5"
version = "0.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564"
checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661"
dependencies = [
"zerovec",
]
@@ -1496,6 +1642,16 @@ dependencies = [
"unicode-ident",
]
[[package]]
name = "quick-xml"
version = "0.41.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1"
dependencies = [
"encoding_rs",
"memchr",
]
[[package]]
name = "quote"
version = "1.0.47"
@@ -1545,9 +1701,9 @@ checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69"
[[package]]
name = "rangemap"
version = "1.7.1"
version = "1.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "973443cf09a9c8656b574a866ab68dfa19f0867d0340648c7d2f6a71b8a8ea68"
checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d"
[[package]]
name = "rayon"
@@ -1665,12 +1821,15 @@ dependencies = [
name = "rs-scraper"
version = "0.1.0"
dependencies = [
"anydoc",
"anyhow",
"chrono",
"clap",
"env_logger",
"hex",
"htmd",
"log",
"md-5",
"pdf-inspector",
"reqwest",
"scraper",
"serde",
@@ -1977,6 +2136,12 @@ dependencies = [
"unicode-properties",
]
[[package]]
name = "strsim"
version = "0.11.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f"
[[package]]
name = "subtle"
version = "2.6.1"
@@ -2122,9 +2287,9 @@ dependencies = [
[[package]]
name = "tinystr"
version = "0.8.3"
version = "0.8.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d"
checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643"
dependencies = [
"displaydoc",
"zerovec",
@@ -2283,6 +2448,12 @@ version = "0.25.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d2df906b07856748fa3f6e0ad0cbaa047052d4a7dd609e231c4f72cee8c36f31"
[[package]]
name = "typed-path"
version = "0.12.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8e28f89b80c87b8fb0cf04ab448d5dd0dd0ade2f8891bae878de66a75a28600e"
[[package]]
name = "typenum"
version = "1.20.1"
@@ -2358,6 +2529,16 @@ version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
[[package]]
name = "uuid"
version = "1.24.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2cefc03fd367c0c6d4305de1b312cf00248c4114f4a0418ce6a6af769e3b0bd9"
dependencies = [
"js-sys",
"wasm-bindgen",
]
[[package]]
name = "vcpkg"
version = "0.2.15"
@@ -2450,6 +2631,16 @@ dependencies = [
"wasm-bindgen",
]
[[package]]
name = "web-time"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb"
dependencies = [
"js-sys",
"wasm-bindgen",
]
[[package]]
name = "weezl"
version = "0.1.12"
@@ -2610,9 +2801,9 @@ checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec"
[[package]]
name = "writeable"
version = "0.6.3"
version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4"
checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc"
[[package]]
name = "xml5ever"
@@ -2677,9 +2868,9 @@ checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e"
[[package]]
name = "zerotrie"
version = "0.2.4"
version = "0.2.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf"
checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f"
dependencies = [
"displaydoc",
"yoke",
@@ -2688,9 +2879,9 @@ dependencies = [
[[package]]
name = "zerovec"
version = "0.11.6"
version = "0.11.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239"
checksum = "94b5c6b5976d66c1d703c4fd17d3f5e43c8cedaacf604961b171adc7130896d8"
dependencies = [
"yoke",
"zerofrom",
@@ -2699,17 +2890,49 @@ dependencies = [
[[package]]
name = "zerovec-derive"
version = "0.11.3"
version = "0.11.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555"
checksum = "47402523226a02bfe5230160dc3ccc089aa6f6f19e7fcbb4e6f824bbb1b4aa62"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.119",
"syn 3.0.3",
]
[[package]]
name = "zip"
version = "8.6.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2d04a6b5381502aa6087c94c669499eb1602eb9c5e8198e534de571f7154809b"
dependencies = [
"crc32fast",
"flate2",
"indexmap",
"memchr",
"typed-path",
"zopfli",
]
[[package]]
name = "zlib-rs"
version = "0.6.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12"
[[package]]
name = "zmij"
version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
[[package]]
name = "zopfli"
version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f05cd8797d63865425ff89b5c4a48804f35ba0ce8d125800027ad6017d2b5249"
dependencies = [
"bumpalo",
"crc32fast",
"log",
"simd-adler32",
]
+4 -1
View File
@@ -4,12 +4,15 @@ version = "0.1.0"
edition = "2024"
[dependencies]
anydoc = "0.1"
anyhow = "1"
chrono = { version = "0.4", features = ["serde"] }
clap = { version = "4", features = ["derive"] }
hex = "0.4"
htmd = "0.1"
log = "0.4"
env_logger = "0.11"
md-5 = "0.10"
pdf-inspector = "1"
reqwest = "0.12"
scraper = "0.22"
serde = { version = "1", features = ["derive"] }
+5 -1
View File
@@ -1,5 +1,9 @@
pub const DELAY_MS: u64 = 0;
pub const DELAY_MS: u64 = 1000;
pub const TIMEOUT_SECS: u64 = 60;
pub const MAX_RETRIES: u32 = 3;
pub const RETRY_BACKOFF_SECS: u64 = 5;
pub const USER_AGENT: &str = "web-scraper/1.0 (research)";
/// Rate limit in milliseconds between requests. Set to 0 only for trusted/internal sites.
/// Default is 1000ms to avoid IP bans on public sites like logting.fo.
pub const RATE_LIMIT_MS: u64 = DELAY_MS;
+5 -5
View File
@@ -1,14 +1,14 @@
use htmd::HtmlToMarkdown;
use std::sync::OnceLock;
static CONVERTER: OnceLock<HtmlToMarkdown> = OnceLock::new();
pub fn html_to_markdown(html: &str) -> String {
let converter = HtmlToMarkdown::new();
let converter = CONVERTER.get_or_init(HtmlToMarkdown::new);
match converter.convert(html) {
Ok(md) => md,
Err(e) => {
eprintln!(
" WARN: HTML-to-MD conversion failed ({}), saving raw HTML",
e
);
log::warn!("HTML-to-MD conversion failed ({}), saving raw HTML", e);
html.to_string()
}
}
+96
View File
@@ -0,0 +1,96 @@
use anyhow::Result;
use std::path::Path;
use url::Url;
pub enum DocProcessResult {
Markdown(String),
Raw,
}
pub fn detect_format(bytes: &[u8], url: &Url) -> Option<anydoc::Format> {
if let Some(fmt) = anydoc::Format::from_bytes(bytes) {
return Some(fmt);
}
anydoc::Format::from_path(Path::new(url.path()))
}
pub fn try_convert(bytes: &[u8], url: &Url) -> DocProcessResult {
let Some(format) = detect_format(bytes, url) else {
return DocProcessResult::Raw;
};
match anydoc::to_markdown_bytes(bytes, Some(format)) {
Ok(md) if !md.trim().is_empty() => DocProcessResult::Markdown(md),
Ok(_) => {
log::warn!("anydoc produced empty output, saving raw");
DocProcessResult::Raw
}
Err(e) => {
log::warn!("anydoc conversion failed ({}), saving raw", e);
DocProcessResult::Raw
}
}
}
#[allow(dead_code)]
pub fn batch_convert_docs(dir: &Path) -> Result<(usize, usize, usize)> {
let supported_exts = [
"pdf", "doc", "docx", "docm", "ppt", "pps", "pot", "pptx", "pptm", "ppsx", "ppsm", "xls",
"xlsx", "xlsm", "xlsb", "odt", "ods", "odp", "rtf", "epub", "csv",
];
let entries: Vec<_> = std::fs::read_dir(dir)?
.filter_map(|e| e.ok())
.filter(|e| {
e.path().extension().is_some_and(|ext| {
let ext = ext.to_string_lossy().to_lowercase();
supported_exts.contains(&ext.as_str())
})
})
.collect();
let total = entries.len();
log::info!("Found {} documents to process", total);
let mut converted = 0usize;
let mut raw = 0usize;
let mut errors = 0usize;
for (i, entry) in entries.iter().enumerate() {
let path = entry.path();
let filename = path.file_name().unwrap_or_default().to_string_lossy();
log::info!("[{}/{}] processing: {}", i + 1, total, filename);
let bytes = match std::fs::read(&path) {
Ok(b) => b,
Err(e) => {
log::error!("error reading file: {}", e);
errors += 1;
continue;
}
};
let ext = path.extension().unwrap_or_default().to_string_lossy();
let Some(format) = anydoc::Format::from_extension(ext.as_ref()) else {
log::info!(" unrecognized format — keeping as-is");
raw += 1;
continue;
};
match anydoc::to_markdown_bytes(&bytes, Some(format)) {
Ok(md) if !md.trim().is_empty() => {
let md_path = path.with_extension("md");
std::fs::write(&md_path, md)?;
std::fs::remove_file(&path)?;
log::info!(" converted -> {}", md_path.display());
converted += 1;
}
_ => {
log::info!(" could not convert — keeping as-is");
raw += 1;
}
}
}
Ok((converted, raw, errors))
}
+14 -4
View File
@@ -30,12 +30,22 @@ pub fn log_error(
status_code,
};
let line = serde_json::to_string(&entry).unwrap_or_default() + "\n";
let line = match serde_json::to_string(&entry) {
Ok(json) => json + "\n",
Err(e) => {
log::error!("Failed to serialize error entry: {}", e);
return;
}
};
if let Ok(mut file) = OpenOptions::new().append(true).create(true).open(log_path) {
let _ = file.write_all(line.as_bytes());
if let Err(e) = file.write_all(line.as_bytes()) {
log::error!("Failed to write error log: {}", e);
} else {
let _ = file.flush();
log::info!("LOGGED ERROR [{}]: {}", error_type, message);
}
} else {
log::error!("Failed to open error log file: {}", log_path.display());
}
println!(" LOGGED ERROR [{}]: {}", error_type, message);
}
+9 -2
View File
@@ -1,13 +1,20 @@
use scraper::{Html, Selector};
use std::collections::HashSet;
use std::sync::OnceLock;
use url::Url;
static LINK_SELECTOR: OnceLock<Selector> = OnceLock::new();
fn get_link_selector() -> &'static Selector {
LINK_SELECTOR.get_or_init(|| Selector::parse("a[href]").expect("hardcoded selector is valid"))
}
pub fn extract_links(html: &str, base_url: &Url, base_domain: &str) -> HashSet<Url> {
let document = Html::parse_document(html);
let selector = Selector::parse("a[href]").unwrap();
let selector = get_link_selector();
document
.select(&selector)
.select(selector)
.filter_map(|el| el.value().attr("href"))
.filter(|href| !href.is_empty())
.filter(|href| {
+10 -17
View File
@@ -1,5 +1,3 @@
// src/fetcher.rs
use crate::config::{MAX_RETRIES, RETRY_BACKOFF_SECS, TIMEOUT_SECS, USER_AGENT};
use anyhow::{Result, anyhow};
use reqwest::Client;
@@ -11,15 +9,15 @@ pub struct Fetcher {
}
impl Fetcher {
pub fn new() -> Self {
pub fn new() -> Result<Self> {
let client = Client::builder()
.timeout(Duration::from_secs(TIMEOUT_SECS))
.user_agent(USER_AGENT)
.redirect(reqwest::redirect::Policy::limited(10))
.build()
.expect("Failed to create HTTP client");
.map_err(|e| anyhow!("Failed to create HTTP client: {}", e))?;
Self { client }
Ok(Self { client })
}
pub async fn fetch_with_retry(&self, url: &Url) -> Result<FetchResult> {
@@ -39,7 +37,12 @@ impl Fetcher {
Ok(b) => b.to_vec(),
Err(e) => {
if attempt < MAX_RETRIES {
self.print_retry(attempt, &e.to_string());
log::warn!(
"retry {}/{}: body read failed ({})",
attempt,
MAX_RETRIES,
e
);
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
))
@@ -62,12 +65,11 @@ impl Fetcher {
bytes,
content_type,
final_url,
_status_code: status.as_u16(),
});
}
Err(e) => {
if attempt < MAX_RETRIES {
self.print_retry(attempt, &e.to_string());
log::warn!("retry {}/{}: request failed ({})", attempt, MAX_RETRIES, e);
tokio::time::sleep(Duration::from_secs(
RETRY_BACKOFF_SECS * attempt as u64,
))
@@ -85,19 +87,10 @@ impl Fetcher {
Err(anyhow!("Exhausted all {} retries", MAX_RETRIES))
}
fn print_retry(&self, attempt: u32, error: &str) {
let wait = RETRY_BACKOFF_SECS * attempt as u64;
println!(
" retry {}/{} in {}s... ({})",
attempt, MAX_RETRIES, wait, error
);
}
}
pub struct FetchResult {
pub bytes: Vec<u8>,
pub content_type: String,
pub final_url: Url,
pub _status_code: u16,
}
+42 -53
View File
@@ -1,60 +1,42 @@
mod config;
mod converter;
mod doc_processor;
mod error_logger;
mod extractor;
mod fetcher;
mod pdf_processor;
mod robots;
mod scraper;
mod url_utils;
use crate::config::USER_AGENT;
use crate::robots::fetch_robots;
use crate::scraper::Scraper;
use anyhow::Result;
use clap::Parser;
use std::fs;
use std::path::PathBuf;
use url::Url;
#[derive(Parser, Debug)]
#[command(name = "rs-scraper")]
#[command(author = "FLÓ")]
#[command(version = "0.1.0")]
#[command(about = "Web scraper for Faroese public sector sites")]
struct Args {
#[arg(help = "Starting URL to scrape")]
start_url: String,
#[arg(long, help = "Save documents as-is, skip Markdown conversion")]
no_doc_conversion: bool,
}
#[tokio::main]
async fn main() -> anyhow::Result<()> {
let args: Vec<String> = std::env::args().collect();
async fn main() -> Result<()> {
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
// Parse flags
let mut convert_pdfs = true;
let mut url_arg: Option<String> = None;
let args = Args::parse();
for arg in args.iter().skip(1) {
match arg.as_str() {
"--no-pdf-conversion" => convert_pdfs = false,
"--help" | "-h" => {
println!("Usage: site-scraper [OPTIONS] <start-url>");
println!();
println!("Options:");
println!(" --no-pdf-conversion Save PDFs as-is, skip text extraction");
println!(" -h, --help Show this help message");
println!();
println!("Example:");
println!(" site-scraper https://www.logting.fo");
println!(" site-scraper --no-pdf-conversion https://taks.fo");
std::process::exit(0);
}
_ => {
if url_arg.is_none() {
url_arg = Some(arg.clone());
}
}
}
}
let start_url = match url_arg {
Some(u) => u,
None => {
eprintln!("Usage: site-scraper [OPTIONS] <start-url>");
eprintln!("Example: site-scraper https://www.logting.fo");
eprintln!("Run with --help for options.");
std::process::exit(1);
}
};
let start_url = args.start_url;
let convert_docs = !args.no_doc_conversion;
let url = Url::parse(&start_url)?;
@@ -70,29 +52,36 @@ async fn main() -> anyhow::Result<()> {
let log_path = logs_dir.join("scrape_errors.jsonl");
fs::remove_file(&log_path).ok();
println!("Scraping: {}", start_url);
println!("Base domain: {}", base_domain);
println!("Output dir: {}", output_dir.display());
println!(
"PDF conversion: {}",
if convert_pdfs { "enabled" } else { "disabled" }
log::info!("Scraping: {}", start_url);
log::info!("Base domain: {}", base_domain);
log::info!("Output dir: {}", output_dir.display());
log::info!(
"Document conversion: {}",
if convert_docs { "enabled" } else { "disabled" }
);
// Fetch robots.txt
let client = reqwest::Client::builder()
.timeout(std::time::Duration::from_secs(30))
.user_agent(USER_AGENT)
.user_agent(config::USER_AGENT)
.build()?;
println!("Fetching robots.txt...");
let robots = fetch_robots(&client, &url, "web-scraper").await;
println!();
log::info!("Fetching robots.txt...");
let robots = fetch_robots(&client, &url, config::USER_AGENT).await;
log::info!("");
let scraper = Scraper::new(output_dir, log_path, base_domain, robots, convert_pdfs);
let fetcher = fetcher::Fetcher::new()?;
let scraper = Scraper::new(
output_dir,
log_path,
base_domain,
robots,
convert_docs,
fetcher,
);
let (count, errors) = scraper.run(&url).await;
println!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
println!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
log::info!("\nDone. Fetched {} URLs. Errors: {}.", count, errors);
log::info!("Error log: {}/logs/scrape_errors.jsonl", output_dir_name);
Ok(())
}
-71
View File
@@ -1,71 +0,0 @@
// src/pdf_processor.rs
use anyhow::Result;
use std::path::Path;
pub enum PdfProcessResult {
Markdown(String),
Scanned,
}
pub fn process_pdf(pdf_path: &Path) -> Result<PdfProcessResult> {
use pdf_inspector::process_pdf;
let result = process_pdf(pdf_path)?;
match result.pdf_type {
pdf_inspector::PdfType::TextBased | pdf_inspector::PdfType::Mixed => {
match &result.markdown {
Some(md) if !md.trim().is_empty() => Ok(PdfProcessResult::Markdown(md.clone())),
_ => Ok(PdfProcessResult::Scanned),
}
}
pdf_inspector::PdfType::Scanned | pdf_inspector::PdfType::ImageBased => {
Ok(PdfProcessResult::Scanned)
}
}
}
pub fn batch_convert_pdfs(dir: &Path) -> Result<(usize, usize, usize)> {
let mut text_based = 0usize;
let mut scanned = 0usize;
let mut errors = 0usize;
let entries: Vec<_> = std::fs::read_dir(dir)?
.filter_map(|e| e.ok())
.filter(|e| {
e.path()
.extension()
.is_some_and(|ext| ext.eq_ignore_ascii_case("pdf"))
})
.collect();
let total = entries.len();
println!("Found {} PDFs to process", total);
for (i, entry) in entries.iter().enumerate() {
let path = entry.path();
let filename = path.file_name().unwrap_or_default().to_string_lossy();
println!("[{}/{}] processing: {}", i + 1, total, filename);
match process_pdf(&path) {
Ok(PdfProcessResult::Markdown(md)) => {
let md_path = path.with_extension("md");
std::fs::write(&md_path, md)?;
std::fs::remove_file(&path)?;
println!(" text-based -> {}", md_path.display());
text_based += 1;
}
Ok(PdfProcessResult::Scanned) => {
println!(" scanned — keeping as PDF");
scanned += 1;
}
Err(e) => {
eprintln!(" error: {}", e);
errors += 1;
}
}
}
Ok((text_based, scanned, errors))
}
+72 -8
View File
@@ -58,11 +58,16 @@ fn path_matches(pattern: &str, path: &str) -> bool {
}
pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) -> RobotsRule {
let robots_url = format!(
"{}://{}/robots.txt",
base_url.scheme(),
base_url.host_str().unwrap_or("")
);
let host_str = base_url.host_str().unwrap_or("");
if host_str.is_empty() {
log::warn!("No host in base URL, skipping robots.txt");
return RobotsRule {
allowed: Vec::new(),
disallowed: Vec::new(),
};
}
let robots_url = format!("{}://{}/robots.txt", base_url.scheme(), host_str);
let mut rule = RobotsRule {
allowed: Vec::new(),
@@ -82,13 +87,13 @@ pub async fn fetch_robots(client: &Client, base_url: &Url, user_agent: &str) ->
}
}
Ok(resp) => {
println!(
log::warn!(
"robots.txt returned HTTP {} — assuming no restrictions",
resp.status()
);
}
Err(e) => {
println!(
log::warn!(
"Failed to fetch robots.txt ({}): assuming no restrictions",
e
);
@@ -149,7 +154,7 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
}
if !rule.disallowed.is_empty() || !rule.allowed.is_empty() {
println!(
log::info!(
"robots.txt: {} disallow rules, {} allow rules",
rule.disallowed.len(),
rule.allowed.len()
@@ -158,3 +163,62 @@ fn parse_robots_txt(text: &str, target_agent: &str) -> RobotsRule {
rule
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_path_matches_exact() {
assert!(path_matches("/admin", "/admin"));
assert!(path_matches("/admin", "/admin/users"));
assert!(path_matches("/admin", "/administrator"));
}
#[test]
fn test_path_matches_wildcard() {
assert!(path_matches("/private/*", "/private/data"));
assert!(path_matches("/private/*/secret", "/private/x/secret"));
assert!(!path_matches("/private/*/secret", "/private/secret"));
}
#[test]
fn test_path_matches_empty_pattern() {
assert!(!path_matches("", "/anything"));
}
#[test]
fn test_is_allowed_disallow_takes_precedence_when_longer() {
let rule = RobotsRule {
allowed: vec!["/a".into()],
disallowed: vec!["/a/b".into()],
};
assert!(!rule.is_allowed("/a/b"));
assert!(rule.is_allowed("/a/c"));
}
#[test]
fn test_is_allowed_allow_overrides_disallow_when_longer() {
let rule = RobotsRule {
allowed: vec!["/public/pages".into()],
disallowed: vec!["/public".into()],
};
assert!(rule.is_allowed("/public/pages"));
}
#[test]
fn test_parse_robots_txt_wildcard_agent() {
let txt = "User-agent: *\nDisallow: /admin\nAllow: /admin/public\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/admin".to_string()));
assert!(rule.allowed.contains(&"/admin/public".to_string()));
}
#[test]
fn test_parse_robots_txt_specific_agent() {
let txt = "User-agent: badbot\nDisallow: /\n\nUser-agent: *\nDisallow: /private\n";
let rule = parse_robots_txt(txt, "web-scraper");
assert!(rule.disallowed.contains(&"/private".to_string()));
assert!(!rule.disallowed.contains(&"/".to_string()));
}
}
+85 -80
View File
@@ -1,8 +1,8 @@
use crate::converter::html_to_markdown;
use crate::doc_processor::{DocProcessResult, try_convert};
use crate::error_logger::log_error;
use crate::extractor::extract_links;
use crate::fetcher::{FetchResult, Fetcher};
use crate::pdf_processor::{PdfProcessResult, process_pdf};
use crate::robots::RobotsRule;
use crate::url_utils::{normalize_url, url_to_filename};
use anyhow::Result;
@@ -13,17 +13,17 @@ use std::time::Duration;
use tokio::sync::Mutex;
use url::Url;
pub struct PdfStats {
pub text_based: usize,
pub scanned: usize,
pub struct DocStats {
pub converted: usize,
pub raw: usize,
pub errors: usize,
}
impl Default for PdfStats {
impl Default for DocStats {
fn default() -> Self {
Self {
text_based: 0,
scanned: 0,
converted: 0,
raw: 0,
errors: 0,
}
}
@@ -36,8 +36,8 @@ pub struct Scraper {
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
pdf_stats: Arc<Mutex<PdfStats>>,
convert_pdfs: bool,
doc_stats: Arc<Mutex<DocStats>>,
convert_docs: bool,
}
impl Scraper {
@@ -46,17 +46,18 @@ impl Scraper {
log_path: std::path::PathBuf,
base_domain: String,
robots: RobotsRule,
convert_pdfs: bool,
convert_docs: bool,
fetcher: Fetcher,
) -> Self {
Self {
fetcher: Fetcher::new(),
fetcher,
seen: Arc::new(Mutex::new(HashSet::new())),
output_dir,
log_path,
base_domain,
robots,
pdf_stats: Arc::new(Mutex::new(PdfStats::default())),
convert_pdfs,
doc_stats: Arc::new(Mutex::new(DocStats::default())),
convert_docs,
}
}
@@ -85,65 +86,61 @@ impl Scraper {
}
if !self.robots.is_allowed(url.path()) {
println!("[skip] robots.txt disallows: {}", url);
log::info!("[skip] robots.txt disallows: {}", url);
skipped_robots += 1;
continue;
}
println!("[{}] fetching: {}", count, url);
log::info!("[{}] fetching: {}", count, url);
match self.fetcher.fetch_with_retry(&url).await {
Ok(result) => {
let final_url = result.final_url.clone();
let is_html = result.content_type.contains("text/html");
let is_pdf = result.content_type.contains("application/pdf");
match self.save(&result, is_html, is_pdf).await {
match self.save(&result, is_html).await {
Ok(()) => {
if is_html {
if let Ok(html) = std::str::from_utf8(&result.bytes) {
let links =
extract_links(html, &result.final_url, &self.base_domain);
let links = extract_links(html, &final_url, &self.base_domain);
let new_count = links.len();
let mut new_links = Vec::new();
{
let new_links: Vec<Url> = {
let seen = self.seen.lock().await;
for link in &links {
if !seen.contains(link.as_str()) {
new_links.push(link.clone());
}
}
links
.into_iter()
.filter(|link| !seen.contains(link.as_str()))
.collect()
};
for link in &new_links {
queue.push_back(link.clone());
}
let new_count = new_links.len();
for link in new_links {
queue.push_back(link);
}
println!(" found {} links ({} new)", links.len(), new_count);
log::info!(
" found {} links ({} new)",
new_count,
new_links.len()
);
}
} else {
println!(" binary: {}", result.content_type);
log::info!(" binary: {}", result.content_type);
}
}
Err(e) => {
log_error(
&self.log_path,
&result.final_url,
&final_url,
"save_error",
&e.to_string(),
None,
);
error_count += 1;
if is_pdf {
let mut stats = self.pdf_stats.lock().await;
let mut stats = self.doc_stats.lock().await;
stats.errors += 1;
}
}
}
}
Err(e) => {
log_error(&self.log_path, &url, "fetch_error", &e.to_string(), None);
error_count += 1;
@@ -151,29 +148,28 @@ impl Scraper {
}
count += 1;
tokio::time::sleep(Duration::from_millis(crate::config::DELAY_MS)).await;
tokio::time::sleep(Duration::from_millis(crate::config::RATE_LIMIT_MS)).await;
}
{
let stats = self.pdf_stats.lock().await;
if stats.text_based > 0 || stats.scanned > 0 || stats.errors > 0 {
println!("\nPDF Statistics:");
println!(" Text-based (converted): {}", stats.text_based);
println!(" Scanned (kept as PDF): {}", stats.scanned);
println!(" Errors: {}", stats.errors);
let stats = self.doc_stats.lock().await;
if stats.converted > 0 || stats.raw > 0 || stats.errors > 0 {
log::info!("\nDocument Statistics:");
log::info!(" Converted to Markdown: {}", stats.converted);
log::info!(" Kept as-is: {}", stats.raw);
log::info!(" Errors: {}", stats.errors);
}
}
if skipped_robots > 0 {
println!("Skipped {} URLs due to robots.txt", skipped_robots);
log::info!("Skipped {} URLs due to robots.txt", skipped_robots);
}
(count, error_count)
}
async fn save(&self, result: &FetchResult, is_html: bool, is_pdf: bool) -> Result<()> {
async fn save(&self, result: &FetchResult, is_html: bool) -> Result<()> {
let mut filename = url_to_filename(&result.final_url, &result.content_type);
let content: Vec<u8>;
if is_html {
@@ -183,51 +179,60 @@ impl Scraper {
filename = filename.replace(".html", ".md");
}
content = md.into_bytes();
} else if is_pdf && self.convert_pdfs {
// Save PDF to temp file for pdf-inspector processing
let temp_path = self.output_dir.join("temp_pdf.tmp");
fs::write(&temp_path, &result.bytes)?;
match process_pdf(&temp_path) {
Ok(PdfProcessResult::Markdown(md)) => {
if filename.ends_with(".pdf") {
filename = filename.replace(".pdf", ".md");
} else if self.convert_docs {
match try_convert(&result.bytes, &result.final_url) {
DocProcessResult::Markdown(md) => {
if let Some(pos) = filename.rfind('.') {
filename.truncate(pos);
}
filename.push_str(".md");
content = md.into_bytes();
fs::remove_file(&temp_path)?;
println!(" [PDF] text-based — converted to Markdown");
let mut stats = self.pdf_stats.lock().await;
stats.text_based += 1;
log::info!(" [DOC] converted to Markdown");
let mut stats = self.doc_stats.lock().await;
stats.converted += 1;
}
Ok(PdfProcessResult::Scanned) => {
println!(" [PDF] scanned document — saved as-is");
let mut stats = self.pdf_stats.lock().await;
stats.scanned += 1;
fs::rename(&temp_path, self.output_dir.join(&filename))?;
return Ok(());
DocProcessResult::Raw => {
content = result.bytes.clone();
if is_document_content_type(&result.content_type) {
let mut stats = self.doc_stats.lock().await;
stats.raw += 1;
}
Err(e) => {
println!(" [PDF] processing error ({}), saved as-is", e);
let mut stats = self.pdf_stats.lock().await;
stats.errors += 1;
fs::rename(&temp_path, self.output_dir.join(&filename))?;
return Ok(());
}
}
} else {
// Non-HTML, non-PDF, or PDF conversion disabled — save as-is
content = result.bytes.clone();
}
let filepath = self.output_dir.join(&filename);
fs::write(&filepath, &content)?;
println!(
log::info!(
" saved -> {}",
filepath.file_name().unwrap_or_default().to_string_lossy()
);
Ok(())
}
}
fn is_document_content_type(ct: &str) -> bool {
let ct = ct.split(';').next().unwrap_or("").trim().to_lowercase();
const DOC_TYPES: &[&str] = &[
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-word.document.macroenabled.12",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.ms-powerpoint.presentation.macroenabled.12",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel.sheet.macroenabled.12",
"application/vnd.ms-excel.sheet.binary.macroenabled.12",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
"application/vnd.oasis.opendocument.presentation",
"application/rtf",
"application/epub+zip",
"text/csv",
];
DOC_TYPES.contains(&ct.as_str())
}
+143 -10
View File
@@ -14,7 +14,12 @@ pub fn derive_base_domain(url: &Url) -> Option<String> {
pub fn is_internal(url: &Url, base_domain: &str) -> bool {
match url.host_str() {
Some(host) => host == base_domain || host.ends_with(&format!(".{}", base_domain)),
Some(host) => {
host == base_domain
|| host
.strip_suffix(base_domain)
.is_some_and(|rest| rest.ends_with('.'))
}
None => false,
}
}
@@ -50,6 +55,7 @@ fn get_extension(url: &Url, content_type: &str) -> String {
"application/json" => ".json",
"application/xml" | "text/xml" => ".xml",
"text/plain" => ".txt",
"text/csv" => ".csv",
"text/css" => ".css",
"application/javascript" | "text/javascript" => ".js",
"image/png" => ".png",
@@ -57,10 +63,21 @@ fn get_extension(url: &Url, content_type: &str) -> String {
"image/gif" => ".gif",
"image/svg+xml" => ".svg",
"image/webp" => ".webp",
"application/vnd.ms-excel" => ".xls",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
"application/rtf" => ".rtf",
"application/epub+zip" => ".epub",
"application/msword" => ".doc",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document" => ".docx",
"application/vnd.ms-word.document.macroenabled.12" => ".docm",
"application/vnd.ms-powerpoint" => ".ppt",
"application/vnd.openxmlformats-officedocument.presentationml.presentation" => ".pptx",
"application/vnd.ms-powerpoint.presentation.macroenabled.12" => ".pptm",
"application/vnd.ms-excel" => ".xls",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" => ".xlsx",
"application/vnd.ms-excel.sheet.macroenabled.12" => ".xlsm",
"application/vnd.ms-excel.sheet.binary.macroenabled.12" => ".xlsb",
"application/vnd.oasis.opendocument.text" => ".odt",
"application/vnd.oasis.opendocument.spreadsheet" => ".ods",
"application/vnd.oasis.opendocument.presentation" => ".odp",
"application/zip" => ".zip",
"application/octet-stream" => ".bin",
_ => ".bin",
@@ -84,14 +101,10 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
.unwrap_or("unknown")
.replace('.', "_")
.replace(':', "_");
let path = url.path().trim_start_matches('/');
let path = if path.is_empty() { "index" } else { path };
// FIX: strip trailing slash so /support/ -> "support" not ""
let path = path.trim_end_matches('/');
let path = if path.is_empty() { "index" } else { path };
let last_segment = path.rsplit('/').next().unwrap_or(path);
let (stem, ext_from_path) = match last_segment.rfind('.') {
@@ -114,9 +127,11 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
_ => String::new(),
};
let filename = format!("{}_{}{}{}", host, stem, query_suffix, ext);
let mut hasher = Sha256::new();
hasher.update(url_str.as_bytes());
let url_hash = hex::encode(hasher.finalize());
filename
let safe_stem: String = stem
.chars()
.map(|c| {
if c.is_alphanumeric() || "._-".contains(c) {
@@ -125,5 +140,123 @@ pub fn url_to_filename(url: &Url, content_type: &str) -> String {
'_'
}
})
.collect()
.collect();
format!(
"{}_{}{}{:}{}",
host,
safe_stem,
query_suffix,
&url_hash[..8],
ext
)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_derive_base_domain() {
let url = Url::parse("https://www.logting.fo/").unwrap();
assert_eq!(derive_base_domain(&url), Some("logting.fo".into()));
let url = Url::parse("https://taks.fo/en/skatur/").unwrap();
assert_eq!(derive_base_domain(&url), Some("taks.fo".into()));
let url = Url::parse("https://localhost:8080/").unwrap();
assert_eq!(derive_base_domain(&url), Some("localhost".into()));
}
#[test]
fn test_is_internal_subdomain() {
let base = "logting.fo";
assert!(is_internal(
&Url::parse("https://logting.fo/").unwrap(),
base
));
assert!(is_internal(
&Url::parse("https://www.logting.fo/").unwrap(),
base
));
assert!(is_internal(
&Url::parse("https://sub.www.logting.fo/").unwrap(),
base
));
assert!(!is_internal(
&Url::parse("https://evilogting.fo/").unwrap(),
base
));
assert!(!is_internal(
&Url::parse("https://logting.com/").unwrap(),
base
));
}
#[test]
fn test_normalize_url_strips_fragment() {
let url = normalize_url("https://example.com/page#section").unwrap();
assert_eq!(url.as_str(), "https://example.com/page");
}
#[test]
fn test_normalize_url_strips_trailing_query_chars() {
let url = normalize_url("https://example.com/page?&").unwrap();
assert_eq!(url.as_str(), "https://example.com/page");
}
#[test]
fn test_url_to_filename_faroese_unicode() {
let url = Url::parse("https://logting.fo/lov/tinglýsing/2024/").unwrap();
let filename = url_to_filename(&url, "text/html");
assert!(filename.starts_with("logting_fo_"));
assert!(filename.ends_with(".html") || filename.ends_with(".md"));
assert!(!filename.contains('ý'));
}
#[test]
fn test_url_to_filename_long_url_hashed() {
let long = format!("https://example.com/{}", "a".repeat(200));
let url = Url::parse(&long).unwrap();
let filename = url_to_filename(&url, "text/html");
assert!(filename.len() < 30);
assert!(filename.ends_with(".html"));
}
#[test]
fn test_url_to_filename_with_query() {
let url = Url::parse("https://example.com/page?id=42&sort=desc").unwrap();
let filename = url_to_filename(&url, "text/html");
assert!(filename.contains('_'));
assert!(filename.ends_with(".html"));
}
#[test]
fn test_get_extension_from_content_type() {
let url = Url::parse("https://example.com/").unwrap();
assert_eq!(get_extension(&url, "text/html"), ".html");
assert_eq!(
get_extension(&url, "application/pdf; charset=binary"),
".pdf"
);
assert_eq!(get_extension(&url, "application/octet-stream"), ".bin");
}
#[test]
fn test_get_extension_from_url_path() {
let url = Url::parse("https://example.com/doc.pdf").unwrap();
assert_eq!(get_extension(&url, "application/octet-stream"), ".pdf");
}
#[test]
fn test_url_collision_avoidance() {
let url1 = Url::parse("https://example.com/foo-bar").unwrap();
let url2 = Url::parse("https://example.com/foo.bar").unwrap();
let fn1 = url_to_filename(&url1, "text/html");
let fn2 = url_to_filename(&url2, "text/html");
assert_ne!(
fn1, fn2,
"Different URLs should produce different filenames"
);
}
}