Files
2026-08-11 20:56:31 +01:00

299 lines
8.5 KiB
Python

# /// script
# requires-python = ">=3.12"
# dependencies = [
# "requests",
# "beautifulsoup4",
# ]
# ///
"""
Generic scraper. Saves all files flat in ./scraped_<domain>/ with sanitized filenames.
Writes structured error log to ./scraped_<domain>/scrape_errors.jsonl.
Usage:
uv run scraper.py https://www.logting.fo
uv run scraper.py https://taks.fo
"""
import hashlib
import json
import sys
import time
from collections import deque
from datetime import datetime, timezone
from pathlib import Path
from urllib.parse import urljoin, urlparse
import requests
from bs4 import BeautifulSoup
DELAY = 0.1
TIMEOUT = 60
MAX_RETRIES = 3
RETRY_BACKOFF = 5.0
USER_AGENT = "web-scraper/1.0 (research)"
CONTENT_TYPE_EXT = {
"text/html": ".html",
"application/pdf": ".pdf",
"application/json": ".json",
"application/xml": ".xml",
"text/xml": ".xml",
"text/plain": ".txt",
"text/css": ".css",
"application/javascript": ".js",
"text/javascript": ".js",
"image/png": ".png",
"image/jpeg": ".jpg",
"image/gif": ".gif",
"image/svg+xml": ".svg",
"image/webp": ".webp",
"application/vnd.ms-excel": ".xls",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": ".xlsx",
"application/msword": ".doc",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document": ".docx",
"application/zip": ".zip",
"application/octet-stream": ".bin",
}
SEEN: set[str] = set()
def log_error(
log_path: Path,
url: str,
error_type: str,
message: str,
status_code: int | None = None,
) -> None:
entry = {
"timestamp": datetime.now(timezone.utc).isoformat(),
"url": url,
"error_type": error_type,
"message": message,
"status_code": status_code,
}
with open(log_path, "a", encoding="utf-8") as f:
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
print(f" LOGGED ERROR [{error_type}]: {message}")
def derive_base_domain(url: str) -> str:
parsed = urlparse(url)
hostname = parsed.hostname
if not hostname:
print(f"Could not parse hostname from {url}")
sys.exit(1)
parts = hostname.split(".")
if len(parts) >= 2:
return ".".join(parts[-2:])
return hostname
def is_internal(url: str, base_domain: str) -> bool:
parsed = urlparse(url)
hostname = parsed.hostname
if not hostname:
return False
return hostname == base_domain or hostname.endswith(f".{base_domain}")
def normalize_url(url: str) -> str:
url = url.split("#")[0].rstrip("?&")
parsed = urlparse(url)
scheme = parsed.scheme.lower()
netloc = parsed.netloc.lower()
path = parsed.path
query = parsed.query
if query:
return f"{scheme}://{netloc}{path}?{query}"
return f"{scheme}://{netloc}{path}"
def get_extension(url: str, content_type: str) -> str:
path_suffix = Path(urlparse(url).path).suffix.lower()
if path_suffix:
return path_suffix
ct = content_type.lower().split(";")[0].strip()
if ct in CONTENT_TYPE_EXT:
return CONTENT_TYPE_EXT[ct]
return ".bin"
def url_to_filename(url: str, content_type: str = "") -> str:
parsed = urlparse(url)
if len(url) > 200:
hash_part = hashlib.sha256(url.encode()).hexdigest()[:16]
ext = get_extension(url, content_type)
return f"{hash_part}{ext}"
host_prefix = parsed.netloc.replace(".", "_").replace(":", "_")
path = parsed.path.lstrip("/") or "index"
# Split extension from path so query hash goes before it
path_obj = Path(path)
stem = path_obj.stem
ext = path_obj.suffix.lower()
if not ext:
ext = get_extension(url, content_type)
query_suffix = ""
if parsed.query:
query_hash = hashlib.md5(parsed.query.encode()).hexdigest()[:8]
query_suffix = f"_{query_hash}"
filename = f"{host_prefix}_{stem}{query_suffix}{ext}"
filename = "".join(c if c.isalnum() or c in "._-" else "_" for c in filename)
return filename
def save(output_dir: Path, url: str, content: bytes, content_type: str = "") -> None:
filename = url_to_filename(url, content_type)
filepath = output_dir / filename
filepath.write_bytes(content)
print(f" saved -> {filepath.name}")
def extract_links(base_url: str, html: str, base_domain: str) -> set[str]:
soup = BeautifulSoup(html, "html.parser")
links: set[str] = set()
for tag in soup.find_all("a", href=True):
href = tag["href"].strip()
if not href or href.startswith(("mailto:", "tel:", "javascript:")):
continue
full_url = urljoin(base_url, href)
full_url = normalize_url(full_url)
if is_internal(full_url, base_domain):
links.add(full_url)
return links
def fetch(url: str, log_path: Path) -> requests.Response | None:
for attempt in range(1, MAX_RETRIES + 1):
try:
return requests.get(
url,
timeout=TIMEOUT,
headers={"User-Agent": USER_AGENT},
allow_redirects=True,
)
except requests.Timeout:
if attempt < MAX_RETRIES:
wait = RETRY_BACKOFF * attempt
print(f" timeout, retry {attempt}/{MAX_RETRIES} in {wait}s...")
time.sleep(wait)
else:
log_error(
log_path,
url,
"timeout",
f"Timed out after {MAX_RETRIES} retries ({TIMEOUT}s each)",
)
return None
except requests.ConnectionError as e:
if attempt < MAX_RETRIES:
wait = RETRY_BACKOFF * attempt
print(
f" connection error, retry {attempt}/{MAX_RETRIES} in {wait}s..."
)
time.sleep(wait)
else:
log_error(log_path, url, "connection_error", str(e))
return None
except requests.RequestException as e:
log_error(log_path, url, "request_error", str(e))
return None
return None
def main() -> None:
if len(sys.argv) < 2:
print("Usage: uv run scraper.py <start-url>")
print("Example: uv run scraper.py https://www.logting.fo")
sys.exit(1)
start_url = sys.argv[1]
base_domain = derive_base_domain(start_url)
output_dir = Path(f"./scraped_{base_domain.replace('.', '_')}")
log_dir = output_dir / "logs"
log_dir.mkdir(parents=True, exist_ok=True)
log_path = log_dir / "scrape_errors.jsonl"
output_dir.mkdir(parents=True, exist_ok=True)
log_path.unlink(missing_ok=True)
queue: deque[str] = deque([start_url])
count = 0
error_count = 0
print(f"Scraping: {start_url}")
print(f"Base domain: {base_domain}")
print(f"Output dir: {output_dir}")
print()
while queue:
url = normalize_url(queue.popleft())
if url in SEEN:
continue
SEEN.add(url)
print(f"[{count}] fetching: {url}")
resp = fetch(url, log_path)
if resp is None:
error_count += 1
time.sleep(DELAY)
continue
if resp.status_code != 200:
log_error(
log_path,
url,
"http_error",
f"HTTP {resp.status_code}",
resp.status_code,
)
error_count += 1
time.sleep(DELAY)
continue
final_url = normalize_url(resp.url)
content_type = resp.headers.get("Content-Type", "")
if "text/html" in content_type:
try:
html = resp.text
save(output_dir, final_url, html.encode("utf-8"), content_type)
links = extract_links(final_url, html, base_domain)
new_links = [u for u in links if u not in SEEN]
queue.extend(new_links)
print(f" found {len(links)} links ({len(new_links)} new)")
except Exception as e:
log_error(log_path, final_url, "parse_error", str(e))
error_count += 1
else:
try:
save(output_dir, final_url, resp.content, content_type)
print(f" binary: {content_type}")
except Exception as e:
log_error(log_path, final_url, "save_error", str(e))
error_count += 1
count += 1
time.sleep(DELAY)
print(
f"\nDone. Fetched {count} URLs ({len(SEEN)} seen total). Errors: {error_count}."
)
print(f"Error log: {log_path}")
if __name__ == "__main__":
main()