Enhance job logging and UI updates

- Implement job logging functionality to track progress and events for each job.
- Add a new section in the job card to display session logs with detailed entries.
- Update job statistics to include pages fetched, pages found, and assets processed.
- Modify the UI theme and text for better clarity and aesthetics.
- Adjust tests to validate new logging features and ensure proper event handling.
This commit is contained in:
kstyagi23
2026-09-05 07:36:34 +05:30
parent 8f9210d6be
commit 2059c275c2
10 changed files with 762 additions and 593 deletions
+8 -2
View File
@@ -184,12 +184,17 @@ class CaptureService:
)
def update(
phase: str, message: str, stats: dict[str, int], warnings: list[str]
phase: str,
message: str,
stats: dict[str, int],
warnings: list[str],
detail: dict[str, object] | None = None,
) -> None:
nonlocal last_progress_at, last_progress_phase
now = time.monotonic()
if (
phase == last_progress_phase
detail is None
and phase == last_progress_phase
and now - last_progress_at < PROGRESS_WRITE_INTERVAL_SECONDS
):
return
@@ -203,6 +208,7 @@ class CaptureService:
message=message,
stats=stats,
warnings=warnings,
detail=detail,
)
if not renewed:
raise CaptureCancelled("Capture lease was reassigned")
+62 -4
View File
@@ -227,7 +227,9 @@ class SiteCrawler:
settings: Settings,
options: CaptureOptions,
cancelled: Callable[[], bool],
on_progress: Callable[[str, str, dict[str, int], list[str]], None],
on_progress: Callable[
[str, str, dict[str, int], list[str], dict[str, object] | None], None
],
) -> None:
self.settings = settings
self.options = options
@@ -273,6 +275,8 @@ class SiteCrawler:
self._root_error: str | None = None
self._limit_reached = False
self._site_dir: Path | None = None
self._inflight_pages = 0
self._inflight_assets = 0
self._address_book = PinnedAddressBook()
limits = httpx.Limits(
max_connections=max(4, self.parallel_connections * 2),
@@ -307,6 +311,12 @@ class SiteCrawler:
self.pending.popleft()
for _ in range(min(len(self.pending), self.parallel_connections))
]
self._inflight_pages = sum(
request.kind == "document" for request in batch
)
self._inflight_assets = len(batch) - self._inflight_pages
for request in batch:
self._emit_page_started(request)
pending_futures = {executor.submit(self._fetch, request) for request in batch}
while pending_futures:
completed, pending_futures = wait(
@@ -316,7 +326,12 @@ class SiteCrawler:
)
self._ensure_not_cancelled()
for future in completed:
self._handle_fetch_result(future.result())
result = future.result()
if result.request.kind == "document":
self._inflight_pages -= 1
else:
self._inflight_assets -= 1
self._handle_fetch_result(result)
self._emit_progress("Capturing", "Discovering pages and assets")
finally:
@@ -473,6 +488,9 @@ class SiteCrawler:
if is_root:
self._root_error = result.message or "The starting page could not be fetched"
self._record_problem(result)
self._emit_page_progress(
result, "skipped" if result.outcome == "skipped" else "failed"
)
return
local_path = self.mapper.assign(result.final_url, result.content_type)
@@ -491,6 +509,7 @@ class SiteCrawler:
if is_root:
self._root_error = failed_result.message
self._record_problem(failed_result)
self._emit_page_progress(failed_result, "failed")
return
record = ResourceRecord(
@@ -526,6 +545,7 @@ class SiteCrawler:
self._discover_css(text, record.final_url, result.request.depth)
elif self._is_javascript(record):
self._discover_javascript(text, record.final_url, result.request.depth)
self._emit_page_progress(result, "captured")
def _record_problem(self, result: FetchResult) -> None:
if result.outcome == "skipped" and result.message:
@@ -960,8 +980,46 @@ class SiteCrawler:
)
(self._site_dir / "README.txt").write_text(readme, encoding="utf-8")
def _emit_progress(self, phase: str, message: str) -> None:
self.on_progress(phase, message, self.stats.copy(), self._visible_warnings())
def _emit_page_started(self, request: CrawlRequest) -> None:
if request.kind != "document":
return
queued_pages = sum(request.kind == "document" for request in self.pending)
queued_assets = len(self.pending) - queued_pages
self._emit_progress(
"Capturing",
"Discovering pages and assets",
detail={
"kind": "page",
"action": "fetching",
"url": request.url,
"depth": request.depth,
"pages_remaining": queued_pages + self._inflight_pages,
"assets_remaining": queued_assets + self._inflight_assets,
},
)
def _emit_page_progress(self, result: FetchResult, action: str) -> None:
if result.request.kind != "document":
return
queued_pages = sum(request.kind == "document" for request in self.pending)
queued_assets = len(self.pending) - queued_pages
self._emit_progress(
"Capturing",
"Discovering pages and assets",
detail={
"kind": "page",
"action": action,
"url": result.final_url or result.request.url,
"depth": result.request.depth,
"pages_remaining": queued_pages + self._inflight_pages,
"assets_remaining": queued_assets + self._inflight_assets,
},
)
def _emit_progress(
self, phase: str, message: str, detail: dict[str, object] | None = None
) -> None:
self.on_progress(phase, message, self.stats.copy(), self._visible_warnings(), detail)
def _ensure_not_cancelled(self) -> None:
if self.cancelled():
+3 -1
View File
@@ -255,7 +255,7 @@ class Database:
FROM job_events
WHERE job_id = ? AND id > ?
ORDER BY id ASC
LIMIT 100
LIMIT 250
""",
(job_id, after_id),
).fetchall()
@@ -382,6 +382,7 @@ class Database:
message: str,
stats: dict[str, int],
warnings: list[str] | None = None,
detail: dict[str, object] | None = None,
) -> bool:
now = utcnow()
values = {
@@ -426,6 +427,7 @@ class Database:
"message": message,
"stats": stats,
"warnings": warnings or [],
"detail": detail,
},
)
return True
+7 -2
View File
@@ -59,7 +59,8 @@ async def secure_headers(_: Request, call_next: Any) -> Response:
response.headers.setdefault(
"Content-Security-Policy",
"default-src 'self'; base-uri 'none'; object-src 'none'; frame-ancestors 'none'; "
"form-action 'self'; connect-src 'self'; img-src 'self' data:; style-src 'self'; "
"form-action 'self'; connect-src 'self'; img-src 'self' data:; "
"style-src 'self' https://fonts.googleapis.com; font-src 'self' https://fonts.gstatic.com; "
"script-src 'self'",
)
return response
@@ -224,8 +225,12 @@ async def _event_stream(request: Request, job_id: str, after_id: int) -> AsyncIt
events = await asyncio.to_thread(database.get_events, job_id, last_id)
for event in events:
last_id = int(event["id"])
payload = json.dumps(event["payload"], separators=(",", ":"))
payload = json.dumps(
{**event["payload"], "logged_at": event["created_at"]}, separators=(",", ":")
)
yield f"id: {last_id}\nevent: {event['kind']}\ndata: {payload}\n\n"
if events:
continue
job = await asyncio.to_thread(database.get_job, job_id)
if not job or job["state"] in terminal_states:
+518 -567
View File
File diff suppressed because it is too large Load Diff
+129 -5
View File
@@ -43,6 +43,9 @@ const TOKEN_PATTERN = /^[A-Za-z0-9_-]{24,200}$/;
const jobs = new Map();
const streams = new Map();
const jobLogs = new Map();
const jobLogIds = new Map();
const refreshTimers = new Map();
let storageAvailable = true;
const ownerToken = getOwnerToken();
@@ -130,6 +133,8 @@ function rememberJob(job) {
function forgetJob(jobId) {
jobs.delete(jobId);
closeEvents(jobId);
jobLogs.delete(jobId);
jobLogIds.delete(jobId);
writeStoredSessions(readStoredSessions().filter((session) => session.id !== jobId));
renderJobs();
}
@@ -161,6 +166,14 @@ function formatTime(value) {
: date.toLocaleString([], { dateStyle: "medium", timeStyle: "short" });
}
function formatLogTime(value) {
if (!value) return "--:--:--";
const date = new Date(value);
return Number.isNaN(date.getTime())
? "--:--:--"
: date.toLocaleTimeString([], { hour: "2-digit", minute: "2-digit", second: "2-digit" });
}
function hostname(url) {
try {
return new URL(url).hostname;
@@ -311,6 +324,8 @@ function jobProgress(job) {
function createJobCard(job) {
const card = createElement("article", "job-card");
card.dataset.state = job.state;
card.dataset.jobId = job.id;
card.classList.toggle("is-active", isActive(job));
const topLine = createElement("div", "job-topline");
topLine.append(
@@ -326,11 +341,19 @@ function createJobCard(job) {
progress.firstElementChild.style.setProperty("--progress", `${jobProgress(job)}%`);
card.append(progress);
const latestPage = latestPageLog(job.id);
const pagesFound = Number(job.stats?.pages_found) || 0;
const pagesFetched = Number(job.stats?.pages_fetched) || 0;
const queuedPages = Number(latestPage?.detail?.pages_remaining);
const pagesRemaining = Number.isFinite(queuedPages)
? queuedPages
: Math.max(0, pagesFound - pagesFetched);
const stats = createElement("div", "job-stats");
const statItems = [
[job.stats?.pages_fetched || 0, "pages"],
[job.stats?.files_written || 0, "files"],
[formatBytes(job.stats?.bytes_downloaded || 0), "collected"],
[pagesFetched, "pages fetched"],
[pagesFound, "pages found"],
[pagesRemaining, "pages remaining"],
[formatBytes(job.stats?.bytes_downloaded || 0), "downloaded"],
];
statItems.forEach(([value, label]) => {
const item = document.createElement("div");
@@ -355,6 +378,11 @@ function createJobCard(job) {
const tags = createElement("div", "job-tags");
const parallel = job.options?.parallel_connections;
if (parallel) tags.append(createElement("span", "", `${parallel} connections`));
const assetsFound = Number(job.stats?.assets_found) || 0;
const assetsFetched = Number(job.stats?.assets_fetched) || 0;
if (assetsFound || assetsFetched) {
tags.append(createElement("span", "", `${assetsFetched}/${assetsFound} assets`));
}
tags.append(
createElement(
"span",
@@ -369,6 +397,8 @@ function createJobCard(job) {
if (job.warnings?.length) card.append(createElement("p", "job-warning", job.warnings[0]));
if (isActive(job)) card.append(createJobLog(job.id));
const actions = createElement("div", "job-actions");
if (job.archive?.available) {
const download = createElement("button", "job-action primary", "Download ZIP");
@@ -403,15 +433,106 @@ function createJobCard(job) {
return card;
}
function latestPageLog(jobId) {
const entries = jobLogs.get(jobId) || [];
return [...entries].reverse().find((entry) => entry.detail?.kind === "page") || null;
}
function createJobLog(jobId) {
const section = createElement("section", "job-log");
const heading = createElement("div", "job-log-heading");
heading.append(
createElement("span", "job-log-title", "Session log"),
createElement("span", "job-log-count", `${(jobLogs.get(jobId) || []).length} events`),
);
const list = createElement("ol", "job-log-list");
const entries = jobLogs.get(jobId) || [];
if (entries.length) entries.forEach((entry) => list.append(createLogEntry(entry)));
else list.append(createElement("li", "job-log-empty", "Waiting for worker activity..."));
section.append(heading, list);
return section;
}
function createLogEntry(entry) {
const item = createElement("li", "job-log-entry");
const detail = entry.detail || {};
item.dataset.kind = detail.kind || entry.kind;
item.append(createElement("time", "job-log-time", formatLogTime(entry.logged_at)));
if (detail.kind === "page") {
const action = createElement("span", "job-log-action", `page ${detail.action || "seen"}`);
const url = createElement("span", "job-log-url", String(detail.url || "unknown page"));
const queue = createElement(
"span",
"job-log-queue",
`queue ${detail.pages_remaining ?? 0}p / ${detail.assets_remaining ?? 0}a`,
);
item.append(action, url, queue);
return item;
}
item.append(
createElement("span", "job-log-action", String(entry.phase || entry.kind)),
createElement("span", "job-log-url", String(entry.message || "Worker update")),
);
return item;
}
function recordStreamEvent(jobId, kind, event) {
let payload;
try {
payload = JSON.parse(event.data);
} catch {
return;
}
if (kind === "progress" && !payload.detail) return;
const eventId = event.lastEventId || `${kind}:${payload.logged_at}:${payload.message}`;
const seenIds = jobLogIds.get(jobId) || new Set();
if (seenIds.has(eventId)) return;
seenIds.add(eventId);
jobLogIds.set(jobId, seenIds);
const entries = jobLogs.get(jobId) || [];
const entry = { id: eventId, kind, ...payload };
entries.push(entry);
jobLogs.set(jobId, entries);
appendLogEntry(jobId, entry, entries.length);
}
function appendLogEntry(jobId, entry, count) {
const card = jobGrid.querySelector(`[data-job-id="${jobId}"]`);
const list = card?.querySelector(".job-log-list");
if (!list) return;
const followTail = list.scrollTop + list.clientHeight >= list.scrollHeight - 16;
list.querySelector(".job-log-empty")?.remove();
list.append(createLogEntry(entry));
const counter = card.querySelector(".job-log-count");
if (counter) counter.textContent = `${count} events`;
if (followTail) list.scrollTop = list.scrollHeight;
}
function scheduleJobRefresh(jobId) {
if (refreshTimers.has(jobId)) return;
const timer = window.setTimeout(() => {
refreshTimers.delete(jobId);
void refreshJob(jobId);
}, 350);
refreshTimers.set(jobId, timer);
}
function connectEvents(jobId) {
if (streams.has(jobId)) return;
const eventUrl = `/api/captures/${encodeURIComponent(jobId)}/events?session=${encodeURIComponent(ownerToken)}`;
const stream = new EventSource(eventUrl);
streams.set(jobId, stream);
["state", "progress"].forEach((eventName) => {
stream.addEventListener(eventName, () => {
stream.addEventListener("state", (event) => {
recordStreamEvent(jobId, "state", event);
void refreshJob(jobId);
});
stream.addEventListener("progress", (event) => {
recordStreamEvent(jobId, "progress", event);
scheduleJobRefresh(jobId);
});
stream.addEventListener("terminal", () => {
closeEvents(jobId);
@@ -429,6 +550,9 @@ function closeEvents(jobId) {
const stream = streams.get(jobId);
if (stream) stream.close();
streams.delete(jobId);
const timer = refreshTimers.get(jobId);
if (timer) window.clearTimeout(timer);
refreshTimers.delete(jobId);
}
async function refreshJob(jobId) {
+13 -7
View File
@@ -3,7 +3,7 @@
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<meta name="theme-color" content="#f3f5f1">
<meta name="theme-color" content="#171717">
<meta name="application-name" content="SiteHarbor">
<meta name="robots" content="index, follow">
<meta
@@ -29,6 +29,12 @@
}
</script>
<title>Website Downloader &amp; Offline Archive Tool | {{ app_name }}</title>
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link
href="https://fonts.googleapis.com/css2?family=Geist+Mono:wght@400;500;600&amp;family=Geist:wght@400;500;600;700&amp;display=swap"
rel="stylesheet"
>
<link rel="stylesheet" href="{{ url_for('static', path='app.css') }}">
<script defer src="{{ url_for('static', path='app.js') }}"></script>
</head>
@@ -40,7 +46,7 @@
<span class="brand-mark" aria-hidden="true"></span>
<span class="brand-copy">
<strong>SiteHarbor</strong>
<span>Private website archiver</span>
<span>v0.1 / private archive utility</span>
</span>
</a>
<nav class="site-nav" aria-label="Primary navigation">
@@ -66,8 +72,8 @@
<section class="workbench" id="capture-workspace" aria-label="Website capture workspace">
<div class="observatory">
<div class="observatory-copy">
<p class="eyebrow">PRIVATE WEBSITE ARCHIVES</p>
<h1>Archive the web.<br><em>Keep what matters.</em></h1>
<p class="eyebrow">// PRIVATE WEBSITE ARCHIVES</p>
<h1>Archive the web.<br><em>Keep the trail.</em></h1>
<p class="lede">
Turn a public website into a portable ZIP for offline review. Set the boundaries,
start the crawl, and download it when it is ready.
@@ -102,8 +108,8 @@
<form class="capture-console" id="capture-form" novalidate>
<div class="console-head">
<div>
<p class="eyebrow">NEW ARCHIVE</p>
<h2>Create an archive.</h2>
<p class="eyebrow">// NEW ARCHIVE</p>
<h2>Start a capture.</h2>
</div>
<span class="console-step">STEP 01</span>
</div>
@@ -235,7 +241,7 @@
<section class="activity-section" id="recent-captures" aria-labelledby="activity-title">
<div class="section-heading">
<div>
<p class="eyebrow">THIS BROWSER</p>
<p class="eyebrow">// THIS BROWSER</p>
<h2 id="activity-title">Recent archives</h2>
<p>Only this browser can view its captures. Come back to check progress or download a finished archive.</p>
</div>
+9
View File
@@ -121,6 +121,15 @@ def test_capture_service_creates_private_archive(tmp_path) -> None:
names = archive.namelist()
assert any(name.endswith("siteharbor-manifest.json") for name in names)
assert any(name.endswith("site.css") for name in names)
page_logs = [
event["payload"]["detail"]
for event in database.get_events(job["id"])
if event["kind"] == "progress"
and isinstance(event["payload"].get("detail"), dict)
and event["payload"]["detail"].get("kind") == "page"
]
assert {log["action"] for log in page_logs} == {"fetching", "captured"}
assert {log["url"] for log in page_logs} == {source_url}
def test_capture_cancels_while_a_resource_is_stalled(tmp_path) -> None:
+9 -2
View File
@@ -63,7 +63,7 @@ def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
respect_robots=True,
fetch_concurrency=3,
)
events: list[tuple[str, str]] = []
events: list[tuple[str, str, dict[str, object] | None]] = []
crawler = SiteCrawler(
settings=test_settings,
options=CaptureOptions(
@@ -75,7 +75,9 @@ def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
max_duration_seconds=30,
),
cancelled=lambda: False,
on_progress=lambda phase, message, stats, warnings: events.append((phase, message)),
on_progress=lambda phase, message, stats, warnings, detail: events.append(
(phase, message, detail)
),
)
try:
@@ -99,3 +101,8 @@ def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
assert (host_dir / "assets" / "chunk.js").is_file()
assert (result.site_dir / "siteharbor-manifest.json").is_file()
assert events
page_events = [detail for _, _, detail in events if detail and detail["kind"] == "page"]
assert len(page_events) == 4
assert {str(detail["action"]) for detail in page_events} == {"fetching", "captured"}
assert all(isinstance(detail["url"], str) for detail in page_events)
assert all(isinstance(detail["pages_remaining"], int) for detail in page_events)
+1
View File
@@ -36,6 +36,7 @@ def test_capture_routes_are_scoped_to_the_browser_session(tmp_path, monkeypatch)
assert "Archive the web." in homepage.text
assert "Website Downloader &amp; Offline Archive Tool" in homepage.text
assert "https://git.zerofucks.io/kstyagi/website-downloader" in homepage.text
assert "Geist+Mono" in homepage.text
assert client.get("/api/captures").status_code == 401
created = client.post(