from __future__ import annotations
import threading
from dataclasses import replace
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from app.config import settings
from app.crawler import CaptureOptions, SiteCrawler
class FixtureHandler(BaseHTTPRequestHandler):
fixtures = {
"/robots.txt": ("text/plain", b"User-agent: *\nAllow: /\n"),
"/": (
"text/html",
b"""
About
""",
),
"/about/": ("text/html", b"Home"),
"/assets/site.css": (
"text/css",
b"@import '/assets/nested.css'; .hero { background: url('/assets/hero.png'); }",
),
"/assets/nested.css": ("text/css", b".logo { background: url('/assets/logo.svg'); }"),
"/assets/app.js": ("application/javascript", b"import('/assets/chunk.js');"),
"/assets/chunk.js": ("application/javascript", b"export const loaded = true;"),
"/assets/logo.svg": ("image/svg+xml", b""),
"/assets/hero.png": ("image/png", b"not-a-real-png-but-a-static-asset"),
}
def do_GET(self) -> None: # noqa: N802
content_type, body = self.fixtures.get(self.path, ("text/plain", b"not found"))
status = 200 if self.path in self.fixtures else 404
self.send_response(status)
self.send_header("Content-Type", content_type)
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def log_message(self, format: str, *args) -> None: # type: ignore[override]
return
def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
server = ThreadingHTTPServer(("127.0.0.1", 0), FixtureHandler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
port = server.server_address[1]
test_settings = replace(
settings,
data_dir=tmp_path,
work_dir=tmp_path / "work",
artifacts_dir=tmp_path / "artifacts",
reports_dir=tmp_path / "reports",
allow_private_networks=True,
allow_nonstandard_ports=True,
respect_robots=True,
fetch_concurrency=3,
)
events: list[tuple[str, str]] = []
crawler = SiteCrawler(
settings=test_settings,
options=CaptureOptions(
source_url=f"http://127.0.0.1:{port}/",
include_external_assets=False,
max_pages=10,
max_depth=3,
max_bytes=2_000_000,
max_duration_seconds=30,
),
cancelled=lambda: False,
on_progress=lambda phase, message, stats, warnings: events.append((phase, message)),
)
try:
result = crawler.run(tmp_path / "work")
finally:
server.shutdown()
server.server_close()
host_dir = result.site_dir / "127.0.0.1"
root_html = (host_dir / "index.html").read_text(encoding="utf-8")
css = (host_dir / "assets" / "site.css").read_text(encoding="utf-8")
javascript = (host_dir / "assets" / "app.js").read_text(encoding="utf-8")
assert result.entry_point == "127.0.0.1/index.html"
assert result.stats["pages_fetched"] == 2
assert result.stats["assets_fetched"] >= 6
assert "about/index.html" in root_html
assert "assets/site.css" in root_html
assert "hero.png" in css
assert "import('chunk.js')" in javascript
assert (host_dir / "assets" / "chunk.js").is_file()
assert (result.site_dir / "siteharbor-manifest.json").is_file()
assert events