implement cancellation handling in CaptureService and Database, update job states accordingly
This commit is contained in:
+21
-9
@@ -9,7 +9,7 @@ import threading
|
||||
import time
|
||||
from collections import deque
|
||||
from collections.abc import Callable
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
@@ -60,6 +60,8 @@ JS_SOURCE_MAP_RE = re.compile(r"(?://[#@]\s*sourceMappingURL=)(?P<url>\S+)")
|
||||
META_REFRESH_RE = re.compile(r"^(?P<delay>\s*\d+(?:\.\d+)?\s*;?\s*url\s*=\s*)(?P<url>.+)$", re.I)
|
||||
MAX_RECORDED_WARNINGS = 100
|
||||
MAX_RECORDED_ERRORS = 250
|
||||
MAX_CONNECTION_TIMEOUT_SECONDS = 5.0
|
||||
MAX_READ_TIMEOUT_SECONDS = 5.0
|
||||
WINDOWS_RESERVED_NAMES = {
|
||||
"CON",
|
||||
"PRN",
|
||||
@@ -294,7 +296,8 @@ class SiteCrawler:
|
||||
if self._proxy_url:
|
||||
self._pin_proxy()
|
||||
self._enqueue(CrawlRequest(self.source_url, "document", 0))
|
||||
with ThreadPoolExecutor(max_workers=self.parallel_connections) as executor:
|
||||
executor = ThreadPoolExecutor(max_workers=self.parallel_connections)
|
||||
try:
|
||||
while self.pending and not self._limit_reached:
|
||||
self._ensure_not_cancelled()
|
||||
if self._duration_exceeded():
|
||||
@@ -304,12 +307,21 @@ class SiteCrawler:
|
||||
self.pending.popleft()
|
||||
for _ in range(min(len(self.pending), self.parallel_connections))
|
||||
]
|
||||
futures = [executor.submit(self._fetch, request) for request in batch]
|
||||
for future in as_completed(futures):
|
||||
pending_futures = {executor.submit(self._fetch, request) for request in batch}
|
||||
while pending_futures:
|
||||
completed, pending_futures = wait(
|
||||
pending_futures,
|
||||
timeout=0.1,
|
||||
return_when=FIRST_COMPLETED,
|
||||
)
|
||||
self._ensure_not_cancelled()
|
||||
self._handle_fetch_result(future.result())
|
||||
for future in completed:
|
||||
self._handle_fetch_result(future.result())
|
||||
|
||||
self._emit_progress("Capturing", "Discovering pages and assets")
|
||||
finally:
|
||||
# Do not wait for a stalled remote server after a user cancellation.
|
||||
executor.shutdown(wait=not self.cancelled(), cancel_futures=True)
|
||||
|
||||
if not self._root_record:
|
||||
raise CaptureError(self._root_error or "The starting page could not be captured.")
|
||||
@@ -969,10 +981,10 @@ class SiteCrawler:
|
||||
def _request_timeout(self) -> httpx.Timeout:
|
||||
remaining = max(0.05, self._remaining_seconds())
|
||||
return httpx.Timeout(
|
||||
connect=min(10.0, remaining),
|
||||
read=min(20.0, remaining),
|
||||
write=min(10.0, remaining),
|
||||
pool=min(10.0, remaining),
|
||||
connect=min(MAX_CONNECTION_TIMEOUT_SECONDS, remaining),
|
||||
read=min(MAX_READ_TIMEOUT_SECONDS, remaining),
|
||||
write=min(MAX_CONNECTION_TIMEOUT_SECONDS, remaining),
|
||||
pool=min(MAX_CONNECTION_TIMEOUT_SECONDS, remaining),
|
||||
)
|
||||
|
||||
def _pin_target(self, url: str) -> None:
|
||||
|
||||
Reference in New Issue
Block a user