from __future__ import annotations import threading from dataclasses import replace from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from app.config import settings from app.crawler import CaptureOptions, SiteCrawler class FixtureHandler(BaseHTTPRequestHandler): fixtures = { "/robots.txt": ("text/plain", b"User-agent: *\nAllow: /\n"), "/": ( "text/html", b""" About """, ), "/about/": ("text/html", b"Home"), "/assets/site.css": ( "text/css", b"@import '/assets/nested.css'; .hero { background: url('/assets/hero.png'); }", ), "/assets/nested.css": ("text/css", b".logo { background: url('/assets/logo.svg'); }"), "/assets/app.js": ("application/javascript", b"import('/assets/chunk.js');"), "/assets/chunk.js": ("application/javascript", b"export const loaded = true;"), "/assets/logo.svg": ("image/svg+xml", b""), "/assets/hero.png": ("image/png", b"not-a-real-png-but-a-static-asset"), } def do_GET(self) -> None: # noqa: N802 content_type, body = self.fixtures.get(self.path, ("text/plain", b"not found")) status = 200 if self.path in self.fixtures else 404 self.send_response(status) self.send_header("Content-Type", content_type) self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) def log_message(self, format: str, *args) -> None: # type: ignore[override] return def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None: server = ThreadingHTTPServer(("127.0.0.1", 0), FixtureHandler) thread = threading.Thread(target=server.serve_forever, daemon=True) thread.start() port = server.server_address[1] test_settings = replace( settings, data_dir=tmp_path, work_dir=tmp_path / "work", artifacts_dir=tmp_path / "artifacts", reports_dir=tmp_path / "reports", allow_private_networks=True, allow_nonstandard_ports=True, respect_robots=True, fetch_concurrency=3, ) events: list[tuple[str, str]] = [] crawler = SiteCrawler( settings=test_settings, options=CaptureOptions( source_url=f"http://127.0.0.1:{port}/", include_external_assets=False, max_pages=10, max_depth=3, max_bytes=2_000_000, max_duration_seconds=30, ), cancelled=lambda: False, on_progress=lambda phase, message, stats, warnings: events.append((phase, message)), ) try: result = crawler.run(tmp_path / "work") finally: server.shutdown() server.server_close() host_dir = result.site_dir / "127.0.0.1" root_html = (host_dir / "index.html").read_text(encoding="utf-8") css = (host_dir / "assets" / "site.css").read_text(encoding="utf-8") javascript = (host_dir / "assets" / "app.js").read_text(encoding="utf-8") assert result.entry_point == "127.0.0.1/index.html" assert result.stats["pages_fetched"] == 2 assert result.stats["assets_fetched"] >= 6 assert "about/index.html" in root_html assert "assets/site.css" in root_html assert "hero.png" in css assert "import('chunk.js')" in javascript assert (host_dir / "assets" / "chunk.js").is_file() assert (result.site_dir / "siteharbor-manifest.json").is_file() assert events