Target check timeout is 80% of the refresh interval (closes #78) #79

Open
clawbot wants to merge 1 commits from fix/timeout-80-percent into next
6 changed files with 99 additions and 18 deletions
Showing only changes of commit 6f8f42f2bf - Show all commits
+7 -4
View File
@@ -52,8 +52,8 @@ halves, so the root `make check` fails if either one is broken. We provide:
- `script/fmt` — format all files (writes): prettier, then gofmt over `backend/`
- `script/fmt-check` — check formatting (read-only): prettier, then gofmt
- `script/check` — run test, lint, and fmt-check
- `script/frontend-test` — run the production build as the frontend's test (no
unit tests yet)
- `script/frontend-test` — run the unit tests in `test/unit/` with Node's
built-in test runner, then the production build
- `script/frontend-lint` — run prettier in check mode
- `script/frontend-fmt` — format everything prettier understands (writes)
- `script/frontend-fmt-check` — check prettier formatting (read-only)
@@ -136,8 +136,11 @@ Local hosts are tracked separately from WAN stats.
### Latency measurement
HEAD requests with `mode: 'no-cors'` and `cache: 'no-store'`, timed with
`performance.now()`. 1-second timeout; anything over 1000ms is clamped to
unreachable. IPv4 only.
`performance.now()`. Each check times out after 80% of the refresh interval (24
seconds at 30 seconds) and is then recorded as a timeout, so a round's checks
have all finished before the next round is due. A round due while the last one
is still waiting, which happens only after an interval change or when the
recovery probe starts one, is skipped. IPv4 only.
### Color coding
+7
View File
@@ -23,6 +23,13 @@ latest run passes.
# Completed Steps
- 2026-09-29: each target check times out after 80% of the refresh interval
(issue #78), 24 seconds at 30 seconds, where it was capped at 3 seconds. A
round due while the last one's checks are still waiting is skipped, so rounds
never overlap, and the recovery probe starts no new checks while its last ones
are waiting. The frontend has its first unit tests, run by
`script/frontend-test` with Node's built-in test runner; for them,
`index.html` now links `src/styles.css`, which `src/main.js` used to import
- 2026-09-29: the container sets up its own data directory (issue #75):
`bin/entrypoint.sh`, still as root, creates `DATA_DIR` if missing and gives it
and `/data` to the `netwatch` user with mode 750 before starting the backend
+3
View File
@@ -9,6 +9,9 @@
type="image/svg+xml"
href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'><text y='.9em' font-size='90'>📡</text></svg>"
/>
<!-- Linked here, not imported by src/main.js, so the unit tests can
import that module in Node, which cannot import CSS. -->
<link rel="stylesheet" href="/src/styles.css" />
</head>
<body class="bg-gray-900 text-white min-h-screen">
<div id="app"></div>
+4 -3
View File
@@ -1,13 +1,14 @@
#!/bin/sh
# script/frontend-test: run the frontend test suite. The frontend has no
# unit tests; the production build serves as the test (fails on broken
# code).
# script/frontend-test: run the frontend test suite: the unit tests in
# test/unit/ with Node's built-in test runner, then the production
# build, which fails on broken code.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
timeout 30 node --test test/unit/*.test.js
timeout 30 yarn build
}
+26 -11
View File
@@ -1,14 +1,14 @@
import "./styles.css";
// --- Configuration -----------------------------------------------------------
// Timing, axis labels, and display constants. Latency above maxLatency is
// clamped to "unreachable". The sparkline Y-axis is capped at
// Timing, axis labels, and display constants. A target check times out
// after requestTimeout, 80% of updateInterval, so a round's checks have
// all finished before the next round is due; latency above maxLatency is
// recorded as a timeout. The sparkline Y-axis is capped at
// graphMaxLatency — values above it pin to the top of the chart but still
// display their real value in the latency figure. The history buffer holds
// maxHistoryPoints samples (historyDuration / updateInterval).
// reportInterval is how often collected samples are POSTed to the backend.
const CONFIG = {
export const CONFIG = {
updateInterval: 3000,
maxHistoryPoints: 100,
reportInterval: 60000,
@@ -16,7 +16,7 @@ const CONFIG = {
return (this.maxHistoryPoints * this.updateInterval) / 1000;
},
get requestTimeout() {
return Math.min(this.updateInterval - 100, 3000);
return this.updateInterval * 0.8;
},
get maxLatency() {
return this.requestTimeout;
@@ -503,7 +503,7 @@ class Reporter {
// --- Latency Measurement -----------------------------------------------------
async function measureLatency(url) {
export async function measureLatency(url) {
const controller = new AbortController();
const timeoutId = setTimeout(
() => controller.abort(),
@@ -1181,11 +1181,16 @@ function startRecoveryProbe(state, triggerTick) {
log.notice(
`Recovery probe started (${canaries.map((h) => h.name).join(", ")})`,
);
// A check can wait up to CONFIG.requestTimeout, far longer than
// 500ms, so no new checks start while the last ones are waiting.
let checking = false;
state._recoveryProbeId = setInterval(async () => {
if (state.paused) return;
if (state.paused || checking) return;
checking = true;
const results = await Promise.all(
canaries.map((h) => measureLatency(h.url)),
);
checking = false;
if (results.some((r) => r.error === null)) {
log.notice("Recovery probe: connectivity detected");
stopRecoveryProbe(state);
@@ -1374,8 +1379,18 @@ async function init() {
updateClocks();
setInterval(updateClocks, 1000);
function doTick() {
tick(state, () => startRecoveryProbe(state, doTick));
// A round waits up to CONFIG.requestTimeout for its checks. A round
// asked for while one is still waiting, after an interval change or
// by the recovery probe, is skipped, so rounds never overlap.
let roundRunning = false;
async function doTick() {
if (roundRunning) return;
roundRunning = true;
try {
await tick(state, () => startRecoveryProbe(state, doTick));
} finally {
roundRunning = false;
}
}
doTick();
@@ -1444,7 +1459,7 @@ async function init() {
// Bootstrap only when loaded as the page: a real DOM containing the #app
// mount point this module renders into. Importing the module in a unit test
// (which has no #app) runs nothing, so buildReport can be tested in isolation.
// (which has no #app) runs nothing, so its exports can be tested in isolation.
if (typeof document !== "undefined" && document.getElementById("app")) {
if (document.readyState === "loading") {
document.addEventListener("DOMContentLoaded", init);
+52
View File
@@ -0,0 +1,52 @@
// Unit tests for src/main.js, run by script/frontend-test with Node's
// built-in test runner. Importing the module does not start the page.
import { after, before, test } from "node:test";
import assert from "node:assert/strict";
import { createServer } from "node:http";
import { CONFIG, measureLatency } from "../../src/main.js";
// measureLatency writes timeouts to the debug log, which looks for its
// panel in the page. There is no page here.
globalThis.document = { getElementById: () => null };
// A target that answers after the number of milliseconds in the path,
// e.g. /600.
let server;
let target;
before(async () => {
server = createServer((req, res) => {
const delay = Number(new URL(req.url, "http://x").pathname.slice(1));
setTimeout(() => res.end(), delay);
});
await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve));
target = `http://127.0.0.1:${server.address().port}`;
});
after(() => {
server.closeAllConnections();
server.close();
});
test("the timeout is 80% of the refresh interval", () => {
CONFIG.updateInterval = 30000;
assert.equal(CONFIG.requestTimeout, 24000);
CONFIG.updateInterval = 3000;
assert.equal(CONFIG.requestTimeout, 2400);
});
test("an answer within the timeout is recorded with its real time, a later one as a timeout", async () => {
// 600ms is past the 400ms timeout of a 500ms interval...
CONFIG.updateInterval = 500;
assert.deepEqual(await measureLatency(`${target}/600`), {
latency: null,
error: "timeout",
});
// ...and within the 1200ms timeout of a 1500ms interval.
CONFIG.updateInterval = 1500;
const { latency, error } = await measureLatency(`${target}/600`);
assert.equal(error, null);
assert.ok(latency >= 600 && latency < 1200, `latency ${latency}ms`);
});