From 6f8f42f2bfb3351e3b902fe7e83474ad9432f9a0 Mon Sep 17 00:00:00 2001 From: clawbot <35+clawbot@noreply.example.org> Date: Tue, 29 Sep 2026 10:32:03 +0000 Subject: [PATCH] Target check timeout is 80% of the refresh interval (closes #78) Each target check now times out after 80% of the refresh interval, 24 seconds at 30 seconds, where it was capped at 3 seconds, so slow, far targets are recorded with their real time. A round asked for while the last one's checks are still waiting, after an interval change or by the recovery probe, is skipped, so rounds never overlap; the recovery probe starts no new checks while its last ones wait. The frontend has its first unit tests, run by script/frontend-test with Node's built-in test runner. index.html now links src/styles.css, which src/main.js imported, since Node cannot import CSS. Model: opus-5-5 --- README.md | 11 +++++---- TODO.md | 7 ++++++ index.html | 3 +++ script/frontend-test | 7 +++--- src/main.js | 37 +++++++++++++++++++++--------- test/unit/main.test.js | 52 ++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 99 insertions(+), 18 deletions(-) create mode 100644 test/unit/main.test.js diff --git a/README.md b/README.md index acbb62d..601c5ae 100644 --- a/README.md +++ b/README.md @@ -52,8 +52,8 @@ halves, so the root `make check` fails if either one is broken. We provide: - `script/fmt` — format all files (writes): prettier, then gofmt over `backend/` - `script/fmt-check` — check formatting (read-only): prettier, then gofmt - `script/check` — run test, lint, and fmt-check -- `script/frontend-test` — run the production build as the frontend's test (no - unit tests yet) +- `script/frontend-test` — run the unit tests in `test/unit/` with Node's + built-in test runner, then the production build - `script/frontend-lint` — run prettier in check mode - `script/frontend-fmt` — format everything prettier understands (writes) - `script/frontend-fmt-check` — check prettier formatting (read-only) @@ -136,8 +136,11 @@ Local hosts are tracked separately from WAN stats. ### Latency measurement HEAD requests with `mode: 'no-cors'` and `cache: 'no-store'`, timed with -`performance.now()`. 1-second timeout; anything over 1000ms is clamped to -unreachable. IPv4 only. +`performance.now()`. Each check times out after 80% of the refresh interval (24 +seconds at 30 seconds) and is then recorded as a timeout, so a round's checks +have all finished before the next round is due. A round due while the last one +is still waiting, which happens only after an interval change or when the +recovery probe starts one, is skipped. IPv4 only. ### Color coding diff --git a/TODO.md b/TODO.md index 8b17dd2..46f23ff 100644 --- a/TODO.md +++ b/TODO.md @@ -23,6 +23,13 @@ latest run passes. # Completed Steps +- 2026-09-29: each target check times out after 80% of the refresh interval + (issue #78), 24 seconds at 30 seconds, where it was capped at 3 seconds. A + round due while the last one's checks are still waiting is skipped, so rounds + never overlap, and the recovery probe starts no new checks while its last ones + are waiting. The frontend has its first unit tests, run by + `script/frontend-test` with Node's built-in test runner; for them, + `index.html` now links `src/styles.css`, which `src/main.js` used to import - 2026-09-29: the container sets up its own data directory (issue #75): `bin/entrypoint.sh`, still as root, creates `DATA_DIR` if missing and gives it and `/data` to the `netwatch` user with mode 750 before starting the backend diff --git a/index.html b/index.html index 3af620d..ddb6af0 100644 --- a/index.html +++ b/index.html @@ -9,6 +9,9 @@ type="image/svg+xml" href="data:image/svg+xml," /> + +
diff --git a/script/frontend-test b/script/frontend-test index 4c74c65..c923f70 100755 --- a/script/frontend-test +++ b/script/frontend-test @@ -1,13 +1,14 @@ #!/bin/sh -# script/frontend-test: run the frontend test suite. The frontend has no -# unit tests; the production build serves as the test (fails on broken -# code). +# script/frontend-test: run the frontend test suite: the unit tests in +# test/unit/ with Node's built-in test runner, then the production +# build, which fails on broken code. set -eu ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" main() { cd "$ROOT" + timeout 30 node --test test/unit/*.test.js timeout 30 yarn build } diff --git a/src/main.js b/src/main.js index 526a544..ec80fb8 100644 --- a/src/main.js +++ b/src/main.js @@ -1,14 +1,14 @@ -import "./styles.css"; - // --- Configuration ----------------------------------------------------------- -// Timing, axis labels, and display constants. Latency above maxLatency is -// clamped to "unreachable". The sparkline Y-axis is capped at +// Timing, axis labels, and display constants. A target check times out +// after requestTimeout, 80% of updateInterval, so a round's checks have +// all finished before the next round is due; latency above maxLatency is +// recorded as a timeout. The sparkline Y-axis is capped at // graphMaxLatency — values above it pin to the top of the chart but still // display their real value in the latency figure. The history buffer holds // maxHistoryPoints samples (historyDuration / updateInterval). // reportInterval is how often collected samples are POSTed to the backend. -const CONFIG = { +export const CONFIG = { updateInterval: 3000, maxHistoryPoints: 100, reportInterval: 60000, @@ -16,7 +16,7 @@ const CONFIG = { return (this.maxHistoryPoints * this.updateInterval) / 1000; }, get requestTimeout() { - return Math.min(this.updateInterval - 100, 3000); + return this.updateInterval * 0.8; }, get maxLatency() { return this.requestTimeout; @@ -503,7 +503,7 @@ class Reporter { // --- Latency Measurement ----------------------------------------------------- -async function measureLatency(url) { +export async function measureLatency(url) { const controller = new AbortController(); const timeoutId = setTimeout( () => controller.abort(), @@ -1181,11 +1181,16 @@ function startRecoveryProbe(state, triggerTick) { log.notice( `Recovery probe started (${canaries.map((h) => h.name).join(", ")})`, ); + // A check can wait up to CONFIG.requestTimeout, far longer than + // 500ms, so no new checks start while the last ones are waiting. + let checking = false; state._recoveryProbeId = setInterval(async () => { - if (state.paused) return; + if (state.paused || checking) return; + checking = true; const results = await Promise.all( canaries.map((h) => measureLatency(h.url)), ); + checking = false; if (results.some((r) => r.error === null)) { log.notice("Recovery probe: connectivity detected"); stopRecoveryProbe(state); @@ -1374,8 +1379,18 @@ async function init() { updateClocks(); setInterval(updateClocks, 1000); - function doTick() { - tick(state, () => startRecoveryProbe(state, doTick)); + // A round waits up to CONFIG.requestTimeout for its checks. A round + // asked for while one is still waiting, after an interval change or + // by the recovery probe, is skipped, so rounds never overlap. + let roundRunning = false; + async function doTick() { + if (roundRunning) return; + roundRunning = true; + try { + await tick(state, () => startRecoveryProbe(state, doTick)); + } finally { + roundRunning = false; + } } doTick(); @@ -1444,7 +1459,7 @@ async function init() { // Bootstrap only when loaded as the page: a real DOM containing the #app // mount point this module renders into. Importing the module in a unit test -// (which has no #app) runs nothing, so buildReport can be tested in isolation. +// (which has no #app) runs nothing, so its exports can be tested in isolation. if (typeof document !== "undefined" && document.getElementById("app")) { if (document.readyState === "loading") { document.addEventListener("DOMContentLoaded", init); diff --git a/test/unit/main.test.js b/test/unit/main.test.js new file mode 100644 index 0000000..c41e3d5 --- /dev/null +++ b/test/unit/main.test.js @@ -0,0 +1,52 @@ +// Unit tests for src/main.js, run by script/frontend-test with Node's +// built-in test runner. Importing the module does not start the page. + +import { after, before, test } from "node:test"; +import assert from "node:assert/strict"; +import { createServer } from "node:http"; +import { CONFIG, measureLatency } from "../../src/main.js"; + +// measureLatency writes timeouts to the debug log, which looks for its +// panel in the page. There is no page here. +globalThis.document = { getElementById: () => null }; + +// A target that answers after the number of milliseconds in the path, +// e.g. /600. +let server; +let target; + +before(async () => { + server = createServer((req, res) => { + const delay = Number(new URL(req.url, "http://x").pathname.slice(1)); + setTimeout(() => res.end(), delay); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + target = `http://127.0.0.1:${server.address().port}`; +}); + +after(() => { + server.closeAllConnections(); + server.close(); +}); + +test("the timeout is 80% of the refresh interval", () => { + CONFIG.updateInterval = 30000; + assert.equal(CONFIG.requestTimeout, 24000); + CONFIG.updateInterval = 3000; + assert.equal(CONFIG.requestTimeout, 2400); +}); + +test("an answer within the timeout is recorded with its real time, a later one as a timeout", async () => { + // 600ms is past the 400ms timeout of a 500ms interval... + CONFIG.updateInterval = 500; + assert.deepEqual(await measureLatency(`${target}/600`), { + latency: null, + error: "timeout", + }); + + // ...and within the 1200ms timeout of a 1500ms interval. + CONFIG.updateInterval = 1500; + const { latency, error } = await measureLatency(`${target}/600`); + assert.equal(error, null); + assert.ok(latency >= 600 && latency < 1200, `latency ${latency}ms`); +});