Enhance job logging and UI updates
- Implement job logging functionality to track progress and events for each job. - Add a new section in the job card to display session logs with detailed entries. - Update job statistics to include pages fetched, pages found, and assets processed. - Modify the UI theme and text for better clarity and aesthetics. - Adjust tests to validate new logging features and ensure proper event handling.
This commit is contained in:
+8
-2
@@ -184,12 +184,17 @@ class CaptureService:
|
|||||||
)
|
)
|
||||||
|
|
||||||
def update(
|
def update(
|
||||||
phase: str, message: str, stats: dict[str, int], warnings: list[str]
|
phase: str,
|
||||||
|
message: str,
|
||||||
|
stats: dict[str, int],
|
||||||
|
warnings: list[str],
|
||||||
|
detail: dict[str, object] | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
nonlocal last_progress_at, last_progress_phase
|
nonlocal last_progress_at, last_progress_phase
|
||||||
now = time.monotonic()
|
now = time.monotonic()
|
||||||
if (
|
if (
|
||||||
phase == last_progress_phase
|
detail is None
|
||||||
|
and phase == last_progress_phase
|
||||||
and now - last_progress_at < PROGRESS_WRITE_INTERVAL_SECONDS
|
and now - last_progress_at < PROGRESS_WRITE_INTERVAL_SECONDS
|
||||||
):
|
):
|
||||||
return
|
return
|
||||||
@@ -203,6 +208,7 @@ class CaptureService:
|
|||||||
message=message,
|
message=message,
|
||||||
stats=stats,
|
stats=stats,
|
||||||
warnings=warnings,
|
warnings=warnings,
|
||||||
|
detail=detail,
|
||||||
)
|
)
|
||||||
if not renewed:
|
if not renewed:
|
||||||
raise CaptureCancelled("Capture lease was reassigned")
|
raise CaptureCancelled("Capture lease was reassigned")
|
||||||
|
|||||||
+62
-4
@@ -227,7 +227,9 @@ class SiteCrawler:
|
|||||||
settings: Settings,
|
settings: Settings,
|
||||||
options: CaptureOptions,
|
options: CaptureOptions,
|
||||||
cancelled: Callable[[], bool],
|
cancelled: Callable[[], bool],
|
||||||
on_progress: Callable[[str, str, dict[str, int], list[str]], None],
|
on_progress: Callable[
|
||||||
|
[str, str, dict[str, int], list[str], dict[str, object] | None], None
|
||||||
|
],
|
||||||
) -> None:
|
) -> None:
|
||||||
self.settings = settings
|
self.settings = settings
|
||||||
self.options = options
|
self.options = options
|
||||||
@@ -273,6 +275,8 @@ class SiteCrawler:
|
|||||||
self._root_error: str | None = None
|
self._root_error: str | None = None
|
||||||
self._limit_reached = False
|
self._limit_reached = False
|
||||||
self._site_dir: Path | None = None
|
self._site_dir: Path | None = None
|
||||||
|
self._inflight_pages = 0
|
||||||
|
self._inflight_assets = 0
|
||||||
self._address_book = PinnedAddressBook()
|
self._address_book = PinnedAddressBook()
|
||||||
limits = httpx.Limits(
|
limits = httpx.Limits(
|
||||||
max_connections=max(4, self.parallel_connections * 2),
|
max_connections=max(4, self.parallel_connections * 2),
|
||||||
@@ -307,6 +311,12 @@ class SiteCrawler:
|
|||||||
self.pending.popleft()
|
self.pending.popleft()
|
||||||
for _ in range(min(len(self.pending), self.parallel_connections))
|
for _ in range(min(len(self.pending), self.parallel_connections))
|
||||||
]
|
]
|
||||||
|
self._inflight_pages = sum(
|
||||||
|
request.kind == "document" for request in batch
|
||||||
|
)
|
||||||
|
self._inflight_assets = len(batch) - self._inflight_pages
|
||||||
|
for request in batch:
|
||||||
|
self._emit_page_started(request)
|
||||||
pending_futures = {executor.submit(self._fetch, request) for request in batch}
|
pending_futures = {executor.submit(self._fetch, request) for request in batch}
|
||||||
while pending_futures:
|
while pending_futures:
|
||||||
completed, pending_futures = wait(
|
completed, pending_futures = wait(
|
||||||
@@ -316,7 +326,12 @@ class SiteCrawler:
|
|||||||
)
|
)
|
||||||
self._ensure_not_cancelled()
|
self._ensure_not_cancelled()
|
||||||
for future in completed:
|
for future in completed:
|
||||||
self._handle_fetch_result(future.result())
|
result = future.result()
|
||||||
|
if result.request.kind == "document":
|
||||||
|
self._inflight_pages -= 1
|
||||||
|
else:
|
||||||
|
self._inflight_assets -= 1
|
||||||
|
self._handle_fetch_result(result)
|
||||||
|
|
||||||
self._emit_progress("Capturing", "Discovering pages and assets")
|
self._emit_progress("Capturing", "Discovering pages and assets")
|
||||||
finally:
|
finally:
|
||||||
@@ -473,6 +488,9 @@ class SiteCrawler:
|
|||||||
if is_root:
|
if is_root:
|
||||||
self._root_error = result.message or "The starting page could not be fetched"
|
self._root_error = result.message or "The starting page could not be fetched"
|
||||||
self._record_problem(result)
|
self._record_problem(result)
|
||||||
|
self._emit_page_progress(
|
||||||
|
result, "skipped" if result.outcome == "skipped" else "failed"
|
||||||
|
)
|
||||||
return
|
return
|
||||||
|
|
||||||
local_path = self.mapper.assign(result.final_url, result.content_type)
|
local_path = self.mapper.assign(result.final_url, result.content_type)
|
||||||
@@ -491,6 +509,7 @@ class SiteCrawler:
|
|||||||
if is_root:
|
if is_root:
|
||||||
self._root_error = failed_result.message
|
self._root_error = failed_result.message
|
||||||
self._record_problem(failed_result)
|
self._record_problem(failed_result)
|
||||||
|
self._emit_page_progress(failed_result, "failed")
|
||||||
return
|
return
|
||||||
|
|
||||||
record = ResourceRecord(
|
record = ResourceRecord(
|
||||||
@@ -526,6 +545,7 @@ class SiteCrawler:
|
|||||||
self._discover_css(text, record.final_url, result.request.depth)
|
self._discover_css(text, record.final_url, result.request.depth)
|
||||||
elif self._is_javascript(record):
|
elif self._is_javascript(record):
|
||||||
self._discover_javascript(text, record.final_url, result.request.depth)
|
self._discover_javascript(text, record.final_url, result.request.depth)
|
||||||
|
self._emit_page_progress(result, "captured")
|
||||||
|
|
||||||
def _record_problem(self, result: FetchResult) -> None:
|
def _record_problem(self, result: FetchResult) -> None:
|
||||||
if result.outcome == "skipped" and result.message:
|
if result.outcome == "skipped" and result.message:
|
||||||
@@ -960,8 +980,46 @@ class SiteCrawler:
|
|||||||
)
|
)
|
||||||
(self._site_dir / "README.txt").write_text(readme, encoding="utf-8")
|
(self._site_dir / "README.txt").write_text(readme, encoding="utf-8")
|
||||||
|
|
||||||
def _emit_progress(self, phase: str, message: str) -> None:
|
def _emit_page_started(self, request: CrawlRequest) -> None:
|
||||||
self.on_progress(phase, message, self.stats.copy(), self._visible_warnings())
|
if request.kind != "document":
|
||||||
|
return
|
||||||
|
queued_pages = sum(request.kind == "document" for request in self.pending)
|
||||||
|
queued_assets = len(self.pending) - queued_pages
|
||||||
|
self._emit_progress(
|
||||||
|
"Capturing",
|
||||||
|
"Discovering pages and assets",
|
||||||
|
detail={
|
||||||
|
"kind": "page",
|
||||||
|
"action": "fetching",
|
||||||
|
"url": request.url,
|
||||||
|
"depth": request.depth,
|
||||||
|
"pages_remaining": queued_pages + self._inflight_pages,
|
||||||
|
"assets_remaining": queued_assets + self._inflight_assets,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
def _emit_page_progress(self, result: FetchResult, action: str) -> None:
|
||||||
|
if result.request.kind != "document":
|
||||||
|
return
|
||||||
|
queued_pages = sum(request.kind == "document" for request in self.pending)
|
||||||
|
queued_assets = len(self.pending) - queued_pages
|
||||||
|
self._emit_progress(
|
||||||
|
"Capturing",
|
||||||
|
"Discovering pages and assets",
|
||||||
|
detail={
|
||||||
|
"kind": "page",
|
||||||
|
"action": action,
|
||||||
|
"url": result.final_url or result.request.url,
|
||||||
|
"depth": result.request.depth,
|
||||||
|
"pages_remaining": queued_pages + self._inflight_pages,
|
||||||
|
"assets_remaining": queued_assets + self._inflight_assets,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
def _emit_progress(
|
||||||
|
self, phase: str, message: str, detail: dict[str, object] | None = None
|
||||||
|
) -> None:
|
||||||
|
self.on_progress(phase, message, self.stats.copy(), self._visible_warnings(), detail)
|
||||||
|
|
||||||
def _ensure_not_cancelled(self) -> None:
|
def _ensure_not_cancelled(self) -> None:
|
||||||
if self.cancelled():
|
if self.cancelled():
|
||||||
|
|||||||
@@ -255,7 +255,7 @@ class Database:
|
|||||||
FROM job_events
|
FROM job_events
|
||||||
WHERE job_id = ? AND id > ?
|
WHERE job_id = ? AND id > ?
|
||||||
ORDER BY id ASC
|
ORDER BY id ASC
|
||||||
LIMIT 100
|
LIMIT 250
|
||||||
""",
|
""",
|
||||||
(job_id, after_id),
|
(job_id, after_id),
|
||||||
).fetchall()
|
).fetchall()
|
||||||
@@ -382,6 +382,7 @@ class Database:
|
|||||||
message: str,
|
message: str,
|
||||||
stats: dict[str, int],
|
stats: dict[str, int],
|
||||||
warnings: list[str] | None = None,
|
warnings: list[str] | None = None,
|
||||||
|
detail: dict[str, object] | None = None,
|
||||||
) -> bool:
|
) -> bool:
|
||||||
now = utcnow()
|
now = utcnow()
|
||||||
values = {
|
values = {
|
||||||
@@ -426,6 +427,7 @@ class Database:
|
|||||||
"message": message,
|
"message": message,
|
||||||
"stats": stats,
|
"stats": stats,
|
||||||
"warnings": warnings or [],
|
"warnings": warnings or [],
|
||||||
|
"detail": detail,
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
return True
|
return True
|
||||||
|
|||||||
+7
-2
@@ -59,7 +59,8 @@ async def secure_headers(_: Request, call_next: Any) -> Response:
|
|||||||
response.headers.setdefault(
|
response.headers.setdefault(
|
||||||
"Content-Security-Policy",
|
"Content-Security-Policy",
|
||||||
"default-src 'self'; base-uri 'none'; object-src 'none'; frame-ancestors 'none'; "
|
"default-src 'self'; base-uri 'none'; object-src 'none'; frame-ancestors 'none'; "
|
||||||
"form-action 'self'; connect-src 'self'; img-src 'self' data:; style-src 'self'; "
|
"form-action 'self'; connect-src 'self'; img-src 'self' data:; "
|
||||||
|
"style-src 'self' https://fonts.googleapis.com; font-src 'self' https://fonts.gstatic.com; "
|
||||||
"script-src 'self'",
|
"script-src 'self'",
|
||||||
)
|
)
|
||||||
return response
|
return response
|
||||||
@@ -224,8 +225,12 @@ async def _event_stream(request: Request, job_id: str, after_id: int) -> AsyncIt
|
|||||||
events = await asyncio.to_thread(database.get_events, job_id, last_id)
|
events = await asyncio.to_thread(database.get_events, job_id, last_id)
|
||||||
for event in events:
|
for event in events:
|
||||||
last_id = int(event["id"])
|
last_id = int(event["id"])
|
||||||
payload = json.dumps(event["payload"], separators=(",", ":"))
|
payload = json.dumps(
|
||||||
|
{**event["payload"], "logged_at": event["created_at"]}, separators=(",", ":")
|
||||||
|
)
|
||||||
yield f"id: {last_id}\nevent: {event['kind']}\ndata: {payload}\n\n"
|
yield f"id: {last_id}\nevent: {event['kind']}\ndata: {payload}\n\n"
|
||||||
|
if events:
|
||||||
|
continue
|
||||||
|
|
||||||
job = await asyncio.to_thread(database.get_job, job_id)
|
job = await asyncio.to_thread(database.get_job, job_id)
|
||||||
if not job or job["state"] in terminal_states:
|
if not job or job["state"] in terminal_states:
|
||||||
|
|||||||
+519
-568
File diff suppressed because it is too large
Load Diff
+129
-5
@@ -43,6 +43,9 @@ const TOKEN_PATTERN = /^[A-Za-z0-9_-]{24,200}$/;
|
|||||||
|
|
||||||
const jobs = new Map();
|
const jobs = new Map();
|
||||||
const streams = new Map();
|
const streams = new Map();
|
||||||
|
const jobLogs = new Map();
|
||||||
|
const jobLogIds = new Map();
|
||||||
|
const refreshTimers = new Map();
|
||||||
let storageAvailable = true;
|
let storageAvailable = true;
|
||||||
const ownerToken = getOwnerToken();
|
const ownerToken = getOwnerToken();
|
||||||
|
|
||||||
@@ -130,6 +133,8 @@ function rememberJob(job) {
|
|||||||
function forgetJob(jobId) {
|
function forgetJob(jobId) {
|
||||||
jobs.delete(jobId);
|
jobs.delete(jobId);
|
||||||
closeEvents(jobId);
|
closeEvents(jobId);
|
||||||
|
jobLogs.delete(jobId);
|
||||||
|
jobLogIds.delete(jobId);
|
||||||
writeStoredSessions(readStoredSessions().filter((session) => session.id !== jobId));
|
writeStoredSessions(readStoredSessions().filter((session) => session.id !== jobId));
|
||||||
renderJobs();
|
renderJobs();
|
||||||
}
|
}
|
||||||
@@ -161,6 +166,14 @@ function formatTime(value) {
|
|||||||
: date.toLocaleString([], { dateStyle: "medium", timeStyle: "short" });
|
: date.toLocaleString([], { dateStyle: "medium", timeStyle: "short" });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function formatLogTime(value) {
|
||||||
|
if (!value) return "--:--:--";
|
||||||
|
const date = new Date(value);
|
||||||
|
return Number.isNaN(date.getTime())
|
||||||
|
? "--:--:--"
|
||||||
|
: date.toLocaleTimeString([], { hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
||||||
|
}
|
||||||
|
|
||||||
function hostname(url) {
|
function hostname(url) {
|
||||||
try {
|
try {
|
||||||
return new URL(url).hostname;
|
return new URL(url).hostname;
|
||||||
@@ -311,6 +324,8 @@ function jobProgress(job) {
|
|||||||
function createJobCard(job) {
|
function createJobCard(job) {
|
||||||
const card = createElement("article", "job-card");
|
const card = createElement("article", "job-card");
|
||||||
card.dataset.state = job.state;
|
card.dataset.state = job.state;
|
||||||
|
card.dataset.jobId = job.id;
|
||||||
|
card.classList.toggle("is-active", isActive(job));
|
||||||
|
|
||||||
const topLine = createElement("div", "job-topline");
|
const topLine = createElement("div", "job-topline");
|
||||||
topLine.append(
|
topLine.append(
|
||||||
@@ -326,11 +341,19 @@ function createJobCard(job) {
|
|||||||
progress.firstElementChild.style.setProperty("--progress", `${jobProgress(job)}%`);
|
progress.firstElementChild.style.setProperty("--progress", `${jobProgress(job)}%`);
|
||||||
card.append(progress);
|
card.append(progress);
|
||||||
|
|
||||||
|
const latestPage = latestPageLog(job.id);
|
||||||
|
const pagesFound = Number(job.stats?.pages_found) || 0;
|
||||||
|
const pagesFetched = Number(job.stats?.pages_fetched) || 0;
|
||||||
|
const queuedPages = Number(latestPage?.detail?.pages_remaining);
|
||||||
|
const pagesRemaining = Number.isFinite(queuedPages)
|
||||||
|
? queuedPages
|
||||||
|
: Math.max(0, pagesFound - pagesFetched);
|
||||||
const stats = createElement("div", "job-stats");
|
const stats = createElement("div", "job-stats");
|
||||||
const statItems = [
|
const statItems = [
|
||||||
[job.stats?.pages_fetched || 0, "pages"],
|
[pagesFetched, "pages fetched"],
|
||||||
[job.stats?.files_written || 0, "files"],
|
[pagesFound, "pages found"],
|
||||||
[formatBytes(job.stats?.bytes_downloaded || 0), "collected"],
|
[pagesRemaining, "pages remaining"],
|
||||||
|
[formatBytes(job.stats?.bytes_downloaded || 0), "downloaded"],
|
||||||
];
|
];
|
||||||
statItems.forEach(([value, label]) => {
|
statItems.forEach(([value, label]) => {
|
||||||
const item = document.createElement("div");
|
const item = document.createElement("div");
|
||||||
@@ -355,6 +378,11 @@ function createJobCard(job) {
|
|||||||
const tags = createElement("div", "job-tags");
|
const tags = createElement("div", "job-tags");
|
||||||
const parallel = job.options?.parallel_connections;
|
const parallel = job.options?.parallel_connections;
|
||||||
if (parallel) tags.append(createElement("span", "", `${parallel} connections`));
|
if (parallel) tags.append(createElement("span", "", `${parallel} connections`));
|
||||||
|
const assetsFound = Number(job.stats?.assets_found) || 0;
|
||||||
|
const assetsFetched = Number(job.stats?.assets_fetched) || 0;
|
||||||
|
if (assetsFound || assetsFetched) {
|
||||||
|
tags.append(createElement("span", "", `${assetsFetched}/${assetsFound} assets`));
|
||||||
|
}
|
||||||
tags.append(
|
tags.append(
|
||||||
createElement(
|
createElement(
|
||||||
"span",
|
"span",
|
||||||
@@ -369,6 +397,8 @@ function createJobCard(job) {
|
|||||||
|
|
||||||
if (job.warnings?.length) card.append(createElement("p", "job-warning", job.warnings[0]));
|
if (job.warnings?.length) card.append(createElement("p", "job-warning", job.warnings[0]));
|
||||||
|
|
||||||
|
if (isActive(job)) card.append(createJobLog(job.id));
|
||||||
|
|
||||||
const actions = createElement("div", "job-actions");
|
const actions = createElement("div", "job-actions");
|
||||||
if (job.archive?.available) {
|
if (job.archive?.available) {
|
||||||
const download = createElement("button", "job-action primary", "Download ZIP");
|
const download = createElement("button", "job-action primary", "Download ZIP");
|
||||||
@@ -403,15 +433,106 @@ function createJobCard(job) {
|
|||||||
return card;
|
return card;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function latestPageLog(jobId) {
|
||||||
|
const entries = jobLogs.get(jobId) || [];
|
||||||
|
return [...entries].reverse().find((entry) => entry.detail?.kind === "page") || null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function createJobLog(jobId) {
|
||||||
|
const section = createElement("section", "job-log");
|
||||||
|
const heading = createElement("div", "job-log-heading");
|
||||||
|
heading.append(
|
||||||
|
createElement("span", "job-log-title", "Session log"),
|
||||||
|
createElement("span", "job-log-count", `${(jobLogs.get(jobId) || []).length} events`),
|
||||||
|
);
|
||||||
|
const list = createElement("ol", "job-log-list");
|
||||||
|
const entries = jobLogs.get(jobId) || [];
|
||||||
|
if (entries.length) entries.forEach((entry) => list.append(createLogEntry(entry)));
|
||||||
|
else list.append(createElement("li", "job-log-empty", "Waiting for worker activity..."));
|
||||||
|
section.append(heading, list);
|
||||||
|
return section;
|
||||||
|
}
|
||||||
|
|
||||||
|
function createLogEntry(entry) {
|
||||||
|
const item = createElement("li", "job-log-entry");
|
||||||
|
const detail = entry.detail || {};
|
||||||
|
item.dataset.kind = detail.kind || entry.kind;
|
||||||
|
item.append(createElement("time", "job-log-time", formatLogTime(entry.logged_at)));
|
||||||
|
|
||||||
|
if (detail.kind === "page") {
|
||||||
|
const action = createElement("span", "job-log-action", `page ${detail.action || "seen"}`);
|
||||||
|
const url = createElement("span", "job-log-url", String(detail.url || "unknown page"));
|
||||||
|
const queue = createElement(
|
||||||
|
"span",
|
||||||
|
"job-log-queue",
|
||||||
|
`queue ${detail.pages_remaining ?? 0}p / ${detail.assets_remaining ?? 0}a`,
|
||||||
|
);
|
||||||
|
item.append(action, url, queue);
|
||||||
|
return item;
|
||||||
|
}
|
||||||
|
|
||||||
|
item.append(
|
||||||
|
createElement("span", "job-log-action", String(entry.phase || entry.kind)),
|
||||||
|
createElement("span", "job-log-url", String(entry.message || "Worker update")),
|
||||||
|
);
|
||||||
|
return item;
|
||||||
|
}
|
||||||
|
|
||||||
|
function recordStreamEvent(jobId, kind, event) {
|
||||||
|
let payload;
|
||||||
|
try {
|
||||||
|
payload = JSON.parse(event.data);
|
||||||
|
} catch {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (kind === "progress" && !payload.detail) return;
|
||||||
|
|
||||||
|
const eventId = event.lastEventId || `${kind}:${payload.logged_at}:${payload.message}`;
|
||||||
|
const seenIds = jobLogIds.get(jobId) || new Set();
|
||||||
|
if (seenIds.has(eventId)) return;
|
||||||
|
seenIds.add(eventId);
|
||||||
|
jobLogIds.set(jobId, seenIds);
|
||||||
|
|
||||||
|
const entries = jobLogs.get(jobId) || [];
|
||||||
|
const entry = { id: eventId, kind, ...payload };
|
||||||
|
entries.push(entry);
|
||||||
|
jobLogs.set(jobId, entries);
|
||||||
|
appendLogEntry(jobId, entry, entries.length);
|
||||||
|
}
|
||||||
|
|
||||||
|
function appendLogEntry(jobId, entry, count) {
|
||||||
|
const card = jobGrid.querySelector(`[data-job-id="${jobId}"]`);
|
||||||
|
const list = card?.querySelector(".job-log-list");
|
||||||
|
if (!list) return;
|
||||||
|
const followTail = list.scrollTop + list.clientHeight >= list.scrollHeight - 16;
|
||||||
|
list.querySelector(".job-log-empty")?.remove();
|
||||||
|
list.append(createLogEntry(entry));
|
||||||
|
const counter = card.querySelector(".job-log-count");
|
||||||
|
if (counter) counter.textContent = `${count} events`;
|
||||||
|
if (followTail) list.scrollTop = list.scrollHeight;
|
||||||
|
}
|
||||||
|
|
||||||
|
function scheduleJobRefresh(jobId) {
|
||||||
|
if (refreshTimers.has(jobId)) return;
|
||||||
|
const timer = window.setTimeout(() => {
|
||||||
|
refreshTimers.delete(jobId);
|
||||||
|
void refreshJob(jobId);
|
||||||
|
}, 350);
|
||||||
|
refreshTimers.set(jobId, timer);
|
||||||
|
}
|
||||||
|
|
||||||
function connectEvents(jobId) {
|
function connectEvents(jobId) {
|
||||||
if (streams.has(jobId)) return;
|
if (streams.has(jobId)) return;
|
||||||
const eventUrl = `/api/captures/${encodeURIComponent(jobId)}/events?session=${encodeURIComponent(ownerToken)}`;
|
const eventUrl = `/api/captures/${encodeURIComponent(jobId)}/events?session=${encodeURIComponent(ownerToken)}`;
|
||||||
const stream = new EventSource(eventUrl);
|
const stream = new EventSource(eventUrl);
|
||||||
streams.set(jobId, stream);
|
streams.set(jobId, stream);
|
||||||
["state", "progress"].forEach((eventName) => {
|
stream.addEventListener("state", (event) => {
|
||||||
stream.addEventListener(eventName, () => {
|
recordStreamEvent(jobId, "state", event);
|
||||||
void refreshJob(jobId);
|
void refreshJob(jobId);
|
||||||
});
|
});
|
||||||
|
stream.addEventListener("progress", (event) => {
|
||||||
|
recordStreamEvent(jobId, "progress", event);
|
||||||
|
scheduleJobRefresh(jobId);
|
||||||
});
|
});
|
||||||
stream.addEventListener("terminal", () => {
|
stream.addEventListener("terminal", () => {
|
||||||
closeEvents(jobId);
|
closeEvents(jobId);
|
||||||
@@ -429,6 +550,9 @@ function closeEvents(jobId) {
|
|||||||
const stream = streams.get(jobId);
|
const stream = streams.get(jobId);
|
||||||
if (stream) stream.close();
|
if (stream) stream.close();
|
||||||
streams.delete(jobId);
|
streams.delete(jobId);
|
||||||
|
const timer = refreshTimers.get(jobId);
|
||||||
|
if (timer) window.clearTimeout(timer);
|
||||||
|
refreshTimers.delete(jobId);
|
||||||
}
|
}
|
||||||
|
|
||||||
async function refreshJob(jobId) {
|
async function refreshJob(jobId) {
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
<head>
|
<head>
|
||||||
<meta charset="utf-8">
|
<meta charset="utf-8">
|
||||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||||
<meta name="theme-color" content="#f3f5f1">
|
<meta name="theme-color" content="#171717">
|
||||||
<meta name="application-name" content="SiteHarbor">
|
<meta name="application-name" content="SiteHarbor">
|
||||||
<meta name="robots" content="index, follow">
|
<meta name="robots" content="index, follow">
|
||||||
<meta
|
<meta
|
||||||
@@ -29,6 +29,12 @@
|
|||||||
}
|
}
|
||||||
</script>
|
</script>
|
||||||
<title>Website Downloader & Offline Archive Tool | {{ app_name }}</title>
|
<title>Website Downloader & Offline Archive Tool | {{ app_name }}</title>
|
||||||
|
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||||
|
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||||
|
<link
|
||||||
|
href="https://fonts.googleapis.com/css2?family=Geist+Mono:wght@400;500;600&family=Geist:wght@400;500;600;700&display=swap"
|
||||||
|
rel="stylesheet"
|
||||||
|
>
|
||||||
<link rel="stylesheet" href="{{ url_for('static', path='app.css') }}">
|
<link rel="stylesheet" href="{{ url_for('static', path='app.css') }}">
|
||||||
<script defer src="{{ url_for('static', path='app.js') }}"></script>
|
<script defer src="{{ url_for('static', path='app.js') }}"></script>
|
||||||
</head>
|
</head>
|
||||||
@@ -40,7 +46,7 @@
|
|||||||
<span class="brand-mark" aria-hidden="true"></span>
|
<span class="brand-mark" aria-hidden="true"></span>
|
||||||
<span class="brand-copy">
|
<span class="brand-copy">
|
||||||
<strong>SiteHarbor</strong>
|
<strong>SiteHarbor</strong>
|
||||||
<span>Private website archiver</span>
|
<span>v0.1 / private archive utility</span>
|
||||||
</span>
|
</span>
|
||||||
</a>
|
</a>
|
||||||
<nav class="site-nav" aria-label="Primary navigation">
|
<nav class="site-nav" aria-label="Primary navigation">
|
||||||
@@ -66,8 +72,8 @@
|
|||||||
<section class="workbench" id="capture-workspace" aria-label="Website capture workspace">
|
<section class="workbench" id="capture-workspace" aria-label="Website capture workspace">
|
||||||
<div class="observatory">
|
<div class="observatory">
|
||||||
<div class="observatory-copy">
|
<div class="observatory-copy">
|
||||||
<p class="eyebrow">PRIVATE WEBSITE ARCHIVES</p>
|
<p class="eyebrow">// PRIVATE WEBSITE ARCHIVES</p>
|
||||||
<h1>Archive the web.<br><em>Keep what matters.</em></h1>
|
<h1>Archive the web.<br><em>Keep the trail.</em></h1>
|
||||||
<p class="lede">
|
<p class="lede">
|
||||||
Turn a public website into a portable ZIP for offline review. Set the boundaries,
|
Turn a public website into a portable ZIP for offline review. Set the boundaries,
|
||||||
start the crawl, and download it when it is ready.
|
start the crawl, and download it when it is ready.
|
||||||
@@ -102,8 +108,8 @@
|
|||||||
<form class="capture-console" id="capture-form" novalidate>
|
<form class="capture-console" id="capture-form" novalidate>
|
||||||
<div class="console-head">
|
<div class="console-head">
|
||||||
<div>
|
<div>
|
||||||
<p class="eyebrow">NEW ARCHIVE</p>
|
<p class="eyebrow">// NEW ARCHIVE</p>
|
||||||
<h2>Create an archive.</h2>
|
<h2>Start a capture.</h2>
|
||||||
</div>
|
</div>
|
||||||
<span class="console-step">STEP 01</span>
|
<span class="console-step">STEP 01</span>
|
||||||
</div>
|
</div>
|
||||||
@@ -235,7 +241,7 @@
|
|||||||
<section class="activity-section" id="recent-captures" aria-labelledby="activity-title">
|
<section class="activity-section" id="recent-captures" aria-labelledby="activity-title">
|
||||||
<div class="section-heading">
|
<div class="section-heading">
|
||||||
<div>
|
<div>
|
||||||
<p class="eyebrow">THIS BROWSER</p>
|
<p class="eyebrow">// THIS BROWSER</p>
|
||||||
<h2 id="activity-title">Recent archives</h2>
|
<h2 id="activity-title">Recent archives</h2>
|
||||||
<p>Only this browser can view its captures. Come back to check progress or download a finished archive.</p>
|
<p>Only this browser can view its captures. Come back to check progress or download a finished archive.</p>
|
||||||
</div>
|
</div>
|
||||||
|
|||||||
@@ -121,6 +121,15 @@ def test_capture_service_creates_private_archive(tmp_path) -> None:
|
|||||||
names = archive.namelist()
|
names = archive.namelist()
|
||||||
assert any(name.endswith("siteharbor-manifest.json") for name in names)
|
assert any(name.endswith("siteharbor-manifest.json") for name in names)
|
||||||
assert any(name.endswith("site.css") for name in names)
|
assert any(name.endswith("site.css") for name in names)
|
||||||
|
page_logs = [
|
||||||
|
event["payload"]["detail"]
|
||||||
|
for event in database.get_events(job["id"])
|
||||||
|
if event["kind"] == "progress"
|
||||||
|
and isinstance(event["payload"].get("detail"), dict)
|
||||||
|
and event["payload"]["detail"].get("kind") == "page"
|
||||||
|
]
|
||||||
|
assert {log["action"] for log in page_logs} == {"fetching", "captured"}
|
||||||
|
assert {log["url"] for log in page_logs} == {source_url}
|
||||||
|
|
||||||
|
|
||||||
def test_capture_cancels_while_a_resource_is_stalled(tmp_path) -> None:
|
def test_capture_cancels_while_a_resource_is_stalled(tmp_path) -> None:
|
||||||
|
|||||||
@@ -63,7 +63,7 @@ def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
|
|||||||
respect_robots=True,
|
respect_robots=True,
|
||||||
fetch_concurrency=3,
|
fetch_concurrency=3,
|
||||||
)
|
)
|
||||||
events: list[tuple[str, str]] = []
|
events: list[tuple[str, str, dict[str, object] | None]] = []
|
||||||
crawler = SiteCrawler(
|
crawler = SiteCrawler(
|
||||||
settings=test_settings,
|
settings=test_settings,
|
||||||
options=CaptureOptions(
|
options=CaptureOptions(
|
||||||
@@ -75,7 +75,9 @@ def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
|
|||||||
max_duration_seconds=30,
|
max_duration_seconds=30,
|
||||||
),
|
),
|
||||||
cancelled=lambda: False,
|
cancelled=lambda: False,
|
||||||
on_progress=lambda phase, message, stats, warnings: events.append((phase, message)),
|
on_progress=lambda phase, message, stats, warnings, detail: events.append(
|
||||||
|
(phase, message, detail)
|
||||||
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -99,3 +101,8 @@ def test_crawler_captures_and_rewrites_static_fixture(tmp_path) -> None:
|
|||||||
assert (host_dir / "assets" / "chunk.js").is_file()
|
assert (host_dir / "assets" / "chunk.js").is_file()
|
||||||
assert (result.site_dir / "siteharbor-manifest.json").is_file()
|
assert (result.site_dir / "siteharbor-manifest.json").is_file()
|
||||||
assert events
|
assert events
|
||||||
|
page_events = [detail for _, _, detail in events if detail and detail["kind"] == "page"]
|
||||||
|
assert len(page_events) == 4
|
||||||
|
assert {str(detail["action"]) for detail in page_events} == {"fetching", "captured"}
|
||||||
|
assert all(isinstance(detail["url"], str) for detail in page_events)
|
||||||
|
assert all(isinstance(detail["pages_remaining"], int) for detail in page_events)
|
||||||
|
|||||||
@@ -36,6 +36,7 @@ def test_capture_routes_are_scoped_to_the_browser_session(tmp_path, monkeypatch)
|
|||||||
assert "Archive the web." in homepage.text
|
assert "Archive the web." in homepage.text
|
||||||
assert "Website Downloader & Offline Archive Tool" in homepage.text
|
assert "Website Downloader & Offline Archive Tool" in homepage.text
|
||||||
assert "https://git.zerofucks.io/kstyagi/website-downloader" in homepage.text
|
assert "https://git.zerofucks.io/kstyagi/website-downloader" in homepage.text
|
||||||
|
assert "Geist+Mono" in homepage.text
|
||||||
assert client.get("/api/captures").status_code == 401
|
assert client.get("/api/captures").status_code == 401
|
||||||
|
|
||||||
created = client.post(
|
created = client.post(
|
||||||
|
|||||||
Reference in New Issue
Block a user