From 0ab6418e3dada7c3e210aeb27ff5ea001396be53 Mon Sep 17 00:00:00 2001 From: clawbot <35+clawbot@noreply.example.org> Date: Sat, 3 Oct 2026 15:36:47 +0200 Subject: [PATCH] Target check timeout is 80% of the refresh interval (closes #78) Each target check now times out after 80% of the refresh interval, 24 seconds at 30 seconds, where it was capped at 3 seconds, so slow, far targets are recorded with their real time. Rounds never overlap: a round gives up the last round's checks if they are still waiting, which happens only when a round starts early, after an interval change or when the recovery probe finds a target answering. The probe still checks every half second, giving up its previous checks, so they do not pile up. The frontend has its first unit tests, run by script/frontend-test with Node's built-in test runner on a mocked clock. index.html now links src/styles.css, which src/main.js imported, since Node cannot import CSS. Model: opus-5-5 --- Dockerfile | 5 ++- README.md | 14 +++++-- TODO.md | 8 ++++ index.html | 3 ++ script/frontend-test | 7 ++-- src/main.js | 83 +++++++++++++++++++++++++++--------------- test/unit/main.test.js | 66 +++++++++++++++++++++++++++++++++ 7 files changed, 148 insertions(+), 38 deletions(-) create mode 100644 test/unit/main.test.js diff --git a/Dockerfile b/Dockerfile index 4203274..612d8a7 100644 --- a/Dockerfile +++ b/Dockerfile @@ -65,8 +65,9 @@ RUN yarn install --frozen-lockfile RUN apk add --no-cache git make COPY . . # make frontend-check is the frontend half of make check (test + lint + -# fmt-check); its test step is the production yarn build, so this both -# produces dist/ and gates the image on lint/fmt-check/test regressions. +# fmt-check); its test step runs the unit tests, then the production +# yarn build, so this both produces dist/ and gates the image on +# lint/fmt-check/test regressions. # This node stage has neither Go nor Docker; the lint and builder stages # above gate the backend half. RUN make frontend-check diff --git a/README.md b/README.md index acbb62d..0db2a61 100644 --- a/README.md +++ b/README.md @@ -52,8 +52,8 @@ halves, so the root `make check` fails if either one is broken. We provide: - `script/fmt` — format all files (writes): prettier, then gofmt over `backend/` - `script/fmt-check` — check formatting (read-only): prettier, then gofmt - `script/check` — run test, lint, and fmt-check -- `script/frontend-test` — run the production build as the frontend's test (no - unit tests yet) +- `script/frontend-test` — run the unit tests in `test/unit/` with Node's + built-in test runner, then the production build - `script/frontend-lint` — run prettier in check mode - `script/frontend-fmt` — format everything prettier understands (writes) - `script/frontend-fmt-check` — check prettier formatting (read-only) @@ -136,8 +136,14 @@ Local hosts are tracked separately from WAN stats. ### Latency measurement HEAD requests with `mode: 'no-cors'` and `cache: 'no-store'`, timed with -`performance.now()`. 1-second timeout; anything over 1000ms is clamped to -unreachable. IPv4 only. +`performance.now()`. Each check times out after 80% of the refresh interval (24 +seconds at 30 seconds) and is then recorded as a timeout, so a round's checks +have all finished before the next round is due. When no WAN host answers, a +recovery probe checks 4 random WAN hosts every half second, giving up the checks +it started half a second before. As soon as one answers, a new round starts at +once, as it does after an interval change. A round started early gives up the +last round's checks if they are still waiting, and that round records nothing, +so rounds never overlap. IPv4 only. ### Color coding diff --git a/TODO.md b/TODO.md index 8b17dd2..a5c802e 100644 --- a/TODO.md +++ b/TODO.md @@ -23,6 +23,14 @@ latest run passes. # Completed Steps +- 2026-10-03: each target check times out after 80% of the refresh interval + (issue #78), 24 seconds at 30 seconds, where it was capped at 3 seconds. A + round started early, after an interval change or when the recovery probe finds + a target answering, gives up the last round's checks if they are still + waiting, so rounds never overlap; the recovery probe gives up its own checks + after half a second. The frontend has its first unit tests, run by + `script/frontend-test` with Node's built-in test runner; for them, + `index.html` now links `src/styles.css`, which `src/main.js` used to import - 2026-09-29: the container sets up its own data directory (issue #75): `bin/entrypoint.sh`, still as root, creates `DATA_DIR` if missing and gives it and `/data` to the `netwatch` user with mode 750 before starting the backend diff --git a/index.html b/index.html index 3af620d..ddb6af0 100644 --- a/index.html +++ b/index.html @@ -9,6 +9,9 @@ type="image/svg+xml" href="data:image/svg+xml,📡" /> + +
diff --git a/script/frontend-test b/script/frontend-test index 4c74c65..c923f70 100755 --- a/script/frontend-test +++ b/script/frontend-test @@ -1,13 +1,14 @@ #!/bin/sh -# script/frontend-test: run the frontend test suite. The frontend has no -# unit tests; the production build serves as the test (fails on broken -# code). +# script/frontend-test: run the frontend test suite: the unit tests in +# test/unit/ with Node's built-in test runner, then the production +# build, which fails on broken code. set -eu ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" main() { cd "$ROOT" + timeout 30 node --test test/unit/*.test.js timeout 30 yarn build } diff --git a/src/main.js b/src/main.js index 526a544..82f1dbd 100644 --- a/src/main.js +++ b/src/main.js @@ -1,14 +1,14 @@ -import "./styles.css"; - // --- Configuration ----------------------------------------------------------- -// Timing, axis labels, and display constants. Latency above maxLatency is -// clamped to "unreachable". The sparkline Y-axis is capped at +// Timing, axis labels, and display constants. A target check times out +// after requestTimeout, 80% of updateInterval, so a round's checks have +// all finished before the next round is due; latency above maxLatency is +// recorded as a timeout. The sparkline Y-axis is capped at // graphMaxLatency — values above it pin to the top of the chart but still // display their real value in the latency figure. The history buffer holds // maxHistoryPoints samples (historyDuration / updateInterval). // reportInterval is how often collected samples are POSTed to the backend. -const CONFIG = { +export const CONFIG = { updateInterval: 3000, maxHistoryPoints: 100, reportInterval: 60000, @@ -16,7 +16,7 @@ const CONFIG = { return (this.maxHistoryPoints * this.updateInterval) / 1000; }, get requestTimeout() { - return Math.min(this.updateInterval - 100, 3000); + return this.updateInterval * 0.8; }, get maxLatency() { return this.requestTimeout; @@ -503,12 +503,16 @@ class Reporter { // --- Latency Measurement ----------------------------------------------------- -async function measureLatency(url) { +// Checks one target. The check times out after CONFIG.requestTimeout; the +// caller can give it up sooner through the optional signal, which also ends +// it as a timeout. +export async function measureLatency(url, signal) { const controller = new AbortController(); const timeoutId = setTimeout( () => controller.abort(), CONFIG.requestTimeout, ); + signal?.addEventListener("abort", () => controller.abort()); const targetUrl = new URL(url); targetUrl.searchParams.set("_cb", Date.now().toString()); @@ -1101,7 +1105,7 @@ function sortAndRebuildWAN(state) { // --- Main Loop --------------------------------------------------------------- -async function tick(state, onOffline) { +async function tick(state, signal, onOffline) { const ts = Date.now(); if (state.paused) { @@ -1123,11 +1127,12 @@ async function tick(state, onOffline) { log.debug(`Tick #${state.tickCount + 1} started`); const results = await Promise.all( - state.allHosts.map((h) => measureLatency(h.url)), + state.allHosts.map((h) => measureLatency(h.url, signal)), ); - // User may have paused while awaiting results — discard them - if (state.paused) return; + // User may have paused, or the next round may have given up this + // one's checks, while awaiting results — discard them + if (state.paused || signal.aborted) return; state.tickCount++; @@ -1168,9 +1173,10 @@ async function tick(state, onOffline) { // --- Recovery Probe ---------------------------------------------------------- -// When offline, rapidly poll 4 random WAN hosts every 500ms. As soon as any -// responds, stop probing and fire a normal tick to refresh all hosts. -function startRecoveryProbe(state, triggerTick) { +// When offline, check 4 random WAN hosts every 500ms, giving up the checks +// started 500ms before, so at most 4 are ever waiting. As soon as one +// answers, stop probing and start a new round at once. +function startRecoveryProbe(state, startRounds) { if (state._recoveryProbeId) return; // already running const candidates = [...state.wan]; for (let i = candidates.length - 1; i > 0; i--) { @@ -1181,15 +1187,18 @@ function startRecoveryProbe(state, triggerTick) { log.notice( `Recovery probe started (${canaries.map((h) => h.name).join(", ")})`, ); - state._recoveryProbeId = setInterval(async () => { + state._recoveryProbeId = setInterval(() => { if (state.paused) return; - const results = await Promise.all( - canaries.map((h) => measureLatency(h.url)), - ); - if (results.some((r) => r.error === null)) { - log.notice("Recovery probe: connectivity detected"); - stopRecoveryProbe(state); - triggerTick(); + state._recoveryProbeChecks?.abort(); + const checks = new AbortController(); + state._recoveryProbeChecks = checks; + for (const host of canaries) { + measureLatency(host.url, checks.signal).then((r) => { + if (r.error !== null || checks.signal.aborted) return; + log.notice("Recovery probe: connectivity detected"); + stopRecoveryProbe(state); + startRounds(); + }); } }, 500); } @@ -1198,6 +1207,7 @@ function stopRecoveryProbe(state) { if (state._recoveryProbeId) { clearInterval(state._recoveryProbeId); state._recoveryProbeId = null; + state._recoveryProbeChecks?.abort(); } } @@ -1374,18 +1384,34 @@ async function init() { updateClocks(); setInterval(updateClocks, 1000); + // Rounds never overlap: a round first gives up the last round's checks + // if they are still waiting, and the last round then records nothing. + // At a steady interval they never are, as they time out at 80% of it; + // they can be when a round starts early, after an interval change or + // when the recovery probe finds a target answering. + let roundChecks = new AbortController(); function doTick() { - tick(state, () => startRecoveryProbe(state, doTick)); + roundChecks.abort(); + roundChecks = new AbortController(); + tick(state, roundChecks.signal, () => + startRecoveryProbe(state, startRounds), + ); } - doTick(); - let tickIntervalId = setInterval(doTick, CONFIG.updateInterval); + // Starts a round now and then one every CONFIG.updateInterval. + let tickIntervalId; + function startRounds() { + clearInterval(tickIntervalId); + doTick(); + tickIntervalId = setInterval(doTick, CONFIG.updateInterval); + } + + startRounds(); document .getElementById("interval-select") .addEventListener("change", (e) => { const newInterval = parseInt(e.target.value, 10); - clearInterval(tickIntervalId); CONFIG.updateInterval = newInterval; log.notice( `Interval changed to ${humanDuration(newInterval / 1000)}, history reset`, @@ -1434,8 +1460,7 @@ async function init() { // Start immediately with new interval stopRecoveryProbe(state); - doTick(); - tickIntervalId = setInterval(doTick, CONFIG.updateInterval); + startRounds(); }); window.addEventListener("resize", () => handleResize(state)); @@ -1444,7 +1469,7 @@ async function init() { // Bootstrap only when loaded as the page: a real DOM containing the #app // mount point this module renders into. Importing the module in a unit test -// (which has no #app) runs nothing, so buildReport can be tested in isolation. +// (which has no #app) runs nothing, so its exports can be tested in isolation. if (typeof document !== "undefined" && document.getElementById("app")) { if (document.readyState === "loading") { document.addEventListener("DOMContentLoaded", init); diff --git a/test/unit/main.test.js b/test/unit/main.test.js new file mode 100644 index 0000000..c49ef63 --- /dev/null +++ b/test/unit/main.test.js @@ -0,0 +1,66 @@ +// Unit tests for src/main.js, run by script/frontend-test with Node's +// built-in test runner. Importing the module does not start the page. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { CONFIG, measureLatency } from "../../src/main.js"; + +// measureLatency writes timeouts to the debug log, which looks for its +// panel in the page. There is no page here. +globalThis.document = { getElementById: () => null }; + +// Mocks the clock for test t, so that a check lasting seconds takes no real +// time, and replaces fetch with a target that answers after answerAfter +// milliseconds of that clock, or never when answerAfter is Infinity. Both +// are restored when the test ends. +function mockTarget(t, answerAfter) { + t.mock.timers.enable({ apis: ["setTimeout", "Date"] }); + t.mock.method(performance, "now", () => Date.now()); + t.mock.method( + globalThis, + "fetch", + (url, { signal }) => + new Promise((resolve, reject) => { + if (answerAfter !== Infinity) setTimeout(resolve, answerAfter); + signal.addEventListener("abort", () => reject(signal.reason)); + }), + ); +} + +// The result of check if it has ended, otherwise "still waiting". +function settled(check) { + return Promise.race([ + check, + new Promise((resolve) => setImmediate(resolve, "still waiting")), + ]); +} + +for (const interval of [10000, 30000]) { + const timeout = interval * 0.8; + // Over 3 seconds, which the timeout used to be capped at. + const slowAnswer = timeout - 1000; + + test(`at a ${interval}ms interval, an answer after ${slowAnswer}ms is recorded with its real time`, async (t) => { + CONFIG.updateInterval = interval; + mockTarget(t, slowAnswer); + const check = measureLatency("https://target.test"); + t.mock.timers.tick(slowAnswer); + assert.deepEqual(await settled(check), { + latency: slowAnswer, + error: null, + }); + }); + + test(`at a ${interval}ms interval, a target that never answers is recorded as a timeout after ${timeout}ms`, async (t) => { + CONFIG.updateInterval = interval; + mockTarget(t, Infinity); + const check = measureLatency("https://target.test"); + t.mock.timers.tick(timeout - 1); + assert.equal(await settled(check), "still waiting"); + t.mock.timers.tick(1); + assert.deepEqual(await settled(check), { + latency: null, + error: "timeout", + }); + }); +}