Enhance job logging and UI updates
- Implement job logging functionality to track progress and events for each job. - Add a new section in the job card to display session logs with detailed entries. - Update job statistics to include pages fetched, pages found, and assets processed. - Modify the UI theme and text for better clarity and aesthetics. - Adjust tests to validate new logging features and ensure proper event handling.
This commit is contained in:
+62
-4
@@ -227,7 +227,9 @@ class SiteCrawler:
|
||||
settings: Settings,
|
||||
options: CaptureOptions,
|
||||
cancelled: Callable[[], bool],
|
||||
on_progress: Callable[[str, str, dict[str, int], list[str]], None],
|
||||
on_progress: Callable[
|
||||
[str, str, dict[str, int], list[str], dict[str, object] | None], None
|
||||
],
|
||||
) -> None:
|
||||
self.settings = settings
|
||||
self.options = options
|
||||
@@ -273,6 +275,8 @@ class SiteCrawler:
|
||||
self._root_error: str | None = None
|
||||
self._limit_reached = False
|
||||
self._site_dir: Path | None = None
|
||||
self._inflight_pages = 0
|
||||
self._inflight_assets = 0
|
||||
self._address_book = PinnedAddressBook()
|
||||
limits = httpx.Limits(
|
||||
max_connections=max(4, self.parallel_connections * 2),
|
||||
@@ -307,6 +311,12 @@ class SiteCrawler:
|
||||
self.pending.popleft()
|
||||
for _ in range(min(len(self.pending), self.parallel_connections))
|
||||
]
|
||||
self._inflight_pages = sum(
|
||||
request.kind == "document" for request in batch
|
||||
)
|
||||
self._inflight_assets = len(batch) - self._inflight_pages
|
||||
for request in batch:
|
||||
self._emit_page_started(request)
|
||||
pending_futures = {executor.submit(self._fetch, request) for request in batch}
|
||||
while pending_futures:
|
||||
completed, pending_futures = wait(
|
||||
@@ -316,7 +326,12 @@ class SiteCrawler:
|
||||
)
|
||||
self._ensure_not_cancelled()
|
||||
for future in completed:
|
||||
self._handle_fetch_result(future.result())
|
||||
result = future.result()
|
||||
if result.request.kind == "document":
|
||||
self._inflight_pages -= 1
|
||||
else:
|
||||
self._inflight_assets -= 1
|
||||
self._handle_fetch_result(result)
|
||||
|
||||
self._emit_progress("Capturing", "Discovering pages and assets")
|
||||
finally:
|
||||
@@ -473,6 +488,9 @@ class SiteCrawler:
|
||||
if is_root:
|
||||
self._root_error = result.message or "The starting page could not be fetched"
|
||||
self._record_problem(result)
|
||||
self._emit_page_progress(
|
||||
result, "skipped" if result.outcome == "skipped" else "failed"
|
||||
)
|
||||
return
|
||||
|
||||
local_path = self.mapper.assign(result.final_url, result.content_type)
|
||||
@@ -491,6 +509,7 @@ class SiteCrawler:
|
||||
if is_root:
|
||||
self._root_error = failed_result.message
|
||||
self._record_problem(failed_result)
|
||||
self._emit_page_progress(failed_result, "failed")
|
||||
return
|
||||
|
||||
record = ResourceRecord(
|
||||
@@ -526,6 +545,7 @@ class SiteCrawler:
|
||||
self._discover_css(text, record.final_url, result.request.depth)
|
||||
elif self._is_javascript(record):
|
||||
self._discover_javascript(text, record.final_url, result.request.depth)
|
||||
self._emit_page_progress(result, "captured")
|
||||
|
||||
def _record_problem(self, result: FetchResult) -> None:
|
||||
if result.outcome == "skipped" and result.message:
|
||||
@@ -960,8 +980,46 @@ class SiteCrawler:
|
||||
)
|
||||
(self._site_dir / "README.txt").write_text(readme, encoding="utf-8")
|
||||
|
||||
def _emit_progress(self, phase: str, message: str) -> None:
|
||||
self.on_progress(phase, message, self.stats.copy(), self._visible_warnings())
|
||||
def _emit_page_started(self, request: CrawlRequest) -> None:
|
||||
if request.kind != "document":
|
||||
return
|
||||
queued_pages = sum(request.kind == "document" for request in self.pending)
|
||||
queued_assets = len(self.pending) - queued_pages
|
||||
self._emit_progress(
|
||||
"Capturing",
|
||||
"Discovering pages and assets",
|
||||
detail={
|
||||
"kind": "page",
|
||||
"action": "fetching",
|
||||
"url": request.url,
|
||||
"depth": request.depth,
|
||||
"pages_remaining": queued_pages + self._inflight_pages,
|
||||
"assets_remaining": queued_assets + self._inflight_assets,
|
||||
},
|
||||
)
|
||||
|
||||
def _emit_page_progress(self, result: FetchResult, action: str) -> None:
|
||||
if result.request.kind != "document":
|
||||
return
|
||||
queued_pages = sum(request.kind == "document" for request in self.pending)
|
||||
queued_assets = len(self.pending) - queued_pages
|
||||
self._emit_progress(
|
||||
"Capturing",
|
||||
"Discovering pages and assets",
|
||||
detail={
|
||||
"kind": "page",
|
||||
"action": action,
|
||||
"url": result.final_url or result.request.url,
|
||||
"depth": result.request.depth,
|
||||
"pages_remaining": queued_pages + self._inflight_pages,
|
||||
"assets_remaining": queued_assets + self._inflight_assets,
|
||||
},
|
||||
)
|
||||
|
||||
def _emit_progress(
|
||||
self, phase: str, message: str, detail: dict[str, object] | None = None
|
||||
) -> None:
|
||||
self.on_progress(phase, message, self.stats.copy(), self._visible_warnings(), detail)
|
||||
|
||||
def _ensure_not_cancelled(self) -> None:
|
||||
if self.cancelled():
|
||||
|
||||
Reference in New Issue
Block a user