Refactor terminology and update UI for archiving feature

- Changed button text and messages from "crawl" to "archive" for clarity.
- Updated HTML elements to reflect the new terminology and improve user guidance.
- Modified footer content to emphasize local-first archiving and provide source link.
- Added a link to the project's repository in the index page and test to ensure it appears.
This commit is contained in:
kstyagi23
2026-09-05 07:16:04 +05:30
parent f0fc211a42
commit 8f9210d6be
5 changed files with 543 additions and 687 deletions
+2
View File
@@ -2,6 +2,8 @@
SiteHarbor is a standalone Python web-capture service. It creates a bounded offline archive of a public website, preserves discovered static assets, rewrites supported local links, records a manifest, and keeps completed ZIP files outside the public web root. SiteHarbor is a standalone Python web-capture service. It creates a bounded offline archive of a public website, preserves discovered static assets, rewrites supported local links, records a manifest, and keeps completed ZIP files outside the public web root.
Source: <https://git.zerofucks.io/kstyagi/website-downloader>
It is intentionally designed as a local or trusted-network application. It uses SQLite for durable job state and does not require Redis, a queue server, or object storage. It is intentionally designed as a local or trusted-network application. It uses SQLite for durable job state and does not require Redis, a queue server, or object storage.
## What It Does ## What It Does
+490 -625
View File
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -606,7 +606,7 @@ form.addEventListener("submit", async (event) => {
setMessage(""); setMessage("");
submitButton.disabled = true; submitButton.disabled = true;
submitButton.querySelector("span").textContent = "Creating crawl"; submitButton.querySelector("span").textContent = "Preparing archive";
try { try {
const job = await request("/api/captures", { const job = await request("/api/captures", {
method: "POST", method: "POST",
@@ -625,14 +625,14 @@ form.addEventListener("submit", async (event) => {
}), }),
}); });
upsertJob(job); upsertJob(job);
setMessage("Crawl queued. This browser will reconnect to it automatically.", "success"); setMessage("Archive queued. This browser will reconnect to it automatically.", "success");
urlInput.value = ""; urlInput.value = "";
void loadPublicStats(); void loadPublicStats();
} catch (error) { } catch (error) {
setMessage(error.message); setMessage(error.message);
} finally { } finally {
submitButton.disabled = false; submitButton.disabled = false;
submitButton.querySelector("span").textContent = "Start private crawl"; submitButton.querySelector("span").textContent = "Start archive";
} }
}); });
+47 -59
View File
@@ -3,7 +3,7 @@
<head> <head>
<meta charset="utf-8"> <meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1"> <meta name="viewport" content="width=device-width, initial-scale=1">
<meta name="theme-color" content="#151715"> <meta name="theme-color" content="#f3f5f1">
<meta name="application-name" content="SiteHarbor"> <meta name="application-name" content="SiteHarbor">
<meta name="robots" content="index, follow"> <meta name="robots" content="index, follow">
<meta <meta
@@ -24,7 +24,8 @@
"name": "SiteHarbor", "name": "SiteHarbor",
"applicationCategory": "UtilitiesApplication", "applicationCategory": "UtilitiesApplication",
"operatingSystem": "Web", "operatingSystem": "Web",
"description": "A local website downloader for creating bounded offline archives of public websites." "description": "A local website downloader for creating bounded offline archives of public websites.",
"sameAs": "https://git.zerofucks.io/kstyagi/website-downloader"
} }
</script> </script>
<title>Website Downloader &amp; Offline Archive Tool | {{ app_name }}</title> <title>Website Downloader &amp; Offline Archive Tool | {{ app_name }}</title>
@@ -32,22 +33,29 @@
<script defer src="{{ url_for('static', path='app.js') }}"></script> <script defer src="{{ url_for('static', path='app.js') }}"></script>
</head> </head>
<body> <body>
<a class="skip-link" href="#capture-workspace">Skip to archive form</a>
<div class="site-shell"> <div class="site-shell">
<header class="site-header"> <header class="site-header">
<a class="brand" href="/" aria-label="SiteHarbor home"> <a class="brand" href="/" aria-label="SiteHarbor home">
<span class="brand-mark" aria-hidden="true"><i></i><i></i><i></i></span> <span class="brand-mark" aria-hidden="true"></span>
<span class="brand-copy"> <span class="brand-copy">
<span class="brand-kicker">LOCAL WEB ARCHIVER</span>
<strong>SiteHarbor</strong> <strong>SiteHarbor</strong>
<span>Private website archiver</span>
</span> </span>
</a> </a>
<nav class="site-nav" aria-label="Primary navigation"> <nav class="site-nav" aria-label="Primary navigation">
<a href="#capture-workspace">Archive a site</a> <a href="#capture-workspace">New archive</a>
<a href="#recent-captures">Your captures</a> <a href="#recent-captures">Recent archives</a>
<a href="#how-it-works">How it works</a>
</nav> </nav>
<div class="service-panel"> <div class="header-tools">
<span class="service-panel-label">LOCAL SERVICE</span> <a
class="source-link"
href="https://git.zerofucks.io/kstyagi/website-downloader"
target="_blank"
rel="noopener noreferrer"
>
View source
</a>
<span class="service-status" id="service-status" aria-live="polite"> <span class="service-status" id="service-status" aria-live="polite">
<i></i><span>Checking service</span> <i></i><span>Checking service</span>
</span> </span>
@@ -58,46 +66,46 @@
<section class="workbench" id="capture-workspace" aria-label="Website capture workspace"> <section class="workbench" id="capture-workspace" aria-label="Website capture workspace">
<div class="observatory"> <div class="observatory">
<div class="observatory-copy"> <div class="observatory-copy">
<p class="eyebrow">OFFLINE WEBSITE DOWNLOADER</p> <p class="eyebrow">PRIVATE WEBSITE ARCHIVES</p>
<h1>Archive the web.<br><em>Keep it offline.</em></h1> <h1>Archive the web.<br><em>Keep what matters.</em></h1>
<p class="lede"> <p class="lede">
Create a private, offline-ready ZIP archive of a public website. SiteHarbor captures Turn a public website into a portable ZIP for offline review. Set the boundaries,
supported pages and assets, then prepares local links for offline browsing. start the crawl, and download it when it is ready.
</p> </p>
</div> </div>
<div class="metric-grid" aria-live="polite" aria-label="Public service statistics"> <div class="metric-grid" aria-live="polite" aria-label="Public service statistics">
<article class="metric-card metric-primary"> <article class="metric-card metric-primary">
<span class="metric-value" id="stat-websites">--</span> <span class="metric-value" id="stat-websites">--</span>
<span class="metric-label">Websites cloned</span> <span class="metric-label">Archived sites</span>
</article> </article>
<article class="metric-card"> <article class="metric-card">
<span class="metric-value" id="stat-data">--</span> <span class="metric-value" id="stat-data">--</span>
<span class="metric-label">Data collected</span> <span class="metric-label">Data captured</span>
</article> </article>
<article class="metric-card"> <article class="metric-card">
<span class="metric-value" id="stat-active">--</span> <span class="metric-value" id="stat-active">--</span>
<span class="metric-label">Crawls running</span> <span class="metric-label">Active crawls</span>
</article> </article>
<article class="metric-card"> <article class="metric-card">
<span class="metric-value" id="stat-files">--</span> <span class="metric-value" id="stat-files">--</span>
<span class="metric-label">Files preserved</span> <span class="metric-label">Files captured</span>
</article> </article>
</div> </div>
<div class="observatory-foot"> <div class="observatory-foot">
<span class="pulse-dot" aria-hidden="true"></span> <span class="pulse-dot" aria-hidden="true"></span>
<p>Metrics are shared totals. Crawl URLs and archives stay private to the browser session that created them.</p> <p>Shared totals are public. Your crawl URLs, status, and downloads stay scoped to this browser.</p>
</div> </div>
</div> </div>
<form class="capture-console" id="capture-form" novalidate> <form class="capture-console" id="capture-form" novalidate>
<div class="console-head"> <div class="console-head">
<div> <div>
<p class="eyebrow">NEW WEBSITE ARCHIVE</p> <p class="eyebrow">NEW ARCHIVE</p>
<h2>Set your crawl.</h2> <h2>Create an archive.</h2>
</div> </div>
<span class="console-step">01 / 01</span> <span class="console-step">STEP 01</span>
</div> </div>
<label class="url-field" for="capture-url"> <label class="url-field" for="capture-url">
@@ -216,10 +224,10 @@
<p class="form-message" id="form-message" aria-live="polite"></p> <p class="form-message" id="form-message" aria-live="polite"></p>
<div class="console-actions"> <div class="console-actions">
<button class="capture-button" id="capture-submit" type="submit"> <button class="capture-button" id="capture-submit" type="submit">
<span>Start private crawl</span> <span>Start archive</span>
<span aria-hidden="true">-&gt;</span> <span aria-hidden="true">-&gt;</span>
</button> </button>
<p>Only crawl sites you are allowed to archive. Private networks and unsafe ports are blocked.</p> <p>Archive only sites you are allowed to copy. Private networks and unsafe ports stay blocked.</p>
</div> </div>
</form> </form>
</section> </section>
@@ -228,54 +236,34 @@
<div class="section-heading"> <div class="section-heading">
<div> <div>
<p class="eyebrow">THIS BROWSER</p> <p class="eyebrow">THIS BROWSER</p>
<h2 id="activity-title">Your website archives</h2> <h2 id="activity-title">Recent archives</h2>
<p>Private session records stay in this browser. Return to monitor a crawl or download an available offline archive.</p> <p>Only this browser can view its captures. Come back to check progress or download a finished archive.</p>
</div> </div>
<button class="quiet-button" id="refresh-jobs" type="button">Refresh archives</button> <button class="quiet-button" id="refresh-jobs" type="button">Refresh</button>
</div> </div>
<div class="empty-state" id="empty-state"> <div class="empty-state" id="empty-state">
<span class="empty-mark" aria-hidden="true">+</span> <span class="empty-mark" aria-hidden="true">+</span>
<div> <div>
<h3>No website archives yet</h3> <h3>No archives yet</h3>
<p>Start a crawl above. This browser can reconnect to its status and download it when ready.</p> <p>Start with a public website URL. Your archive will appear here as soon as it is queued.</p>
</div> </div>
</div> </div>
<div class="job-grid" id="job-grid" aria-live="polite"></div> <div class="job-grid" id="job-grid" aria-live="polite"></div>
</section> </section>
</main> </main>
<footer class="site-footer" id="how-it-works"> <footer class="site-footer">
<div class="footer-top"> <p>Local-first website archiving for research, reference, and offline review.</p>
<section class="footer-intro" aria-labelledby="footer-title"> <div class="footer-links">
<p class="footer-kicker">SITEHARBOR / OFFLINE WEB ARCHIVES</p> <span>Archives expire after two days.</span>
<h2 id="footer-title">A practical website downloader for offline review.</h2> <a
<p> href="https://git.zerofucks.io/kstyagi/website-downloader"
Save a bounded, browser-owned copy of a public website without sending your archive target="_blank"
to a third-party storage service. rel="noopener noreferrer"
</p> >
</section> Source code
<div class="footer-points"> </a>
<section class="footer-point">
<span class="footer-index">01</span>
<h3>Bounded crawl</h3>
<p>Set page, depth, size, and time limits before SiteHarbor begins.</p>
</section>
<section class="footer-point">
<span class="footer-index">02</span>
<h3>Portable ZIP</h3>
<p>Download captured pages, supported assets, and a manifest for offline browsing.</p>
</section>
<section class="footer-point">
<span class="footer-index">03</span>
<h3>Private by default</h3>
<p>Archives stay scoped to this browser session and are removed after two days.</p>
</section>
</div>
</div>
<div class="footer-bottom">
<span>Local-first web archiving with FastAPI, Uvicorn, and SQLite.</span>
<span>Archive only websites you are authorized to copy.</span>
</div> </div>
</footer> </footer>
</div> </div>
+1
View File
@@ -35,6 +35,7 @@ def test_capture_routes_are_scoped_to_the_browser_session(tmp_path, monkeypatch)
assert homepage.status_code == 200 assert homepage.status_code == 200
assert "Archive the web." in homepage.text assert "Archive the web." in homepage.text
assert "Website Downloader &amp; Offline Archive Tool" in homepage.text assert "Website Downloader &amp; Offline Archive Tool" in homepage.text
assert "https://git.zerofucks.io/kstyagi/website-downloader" in homepage.text
assert client.get("/api/captures").status_code == 401 assert client.get("/api/captures").status_code == 401
created = client.post( created = client.post(