Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
17470dc4a4 | ||
|
|
161f955ae2 |
@@ -16,9 +16,8 @@ Docker image, and the Gitea workflow `.gitea/workflows/check.yml` runs
|
||||
|
||||
# Next Step
|
||||
|
||||
Write the latency statistics once and move the thresholds written inline in
|
||||
`src/main.js` into `CONFIG`
|
||||
([#102](https://git.eeqj.de/sneak/netwatch/issues/102)).
|
||||
Take "IPv4 only" out of the page's footer, as nothing in the page limits a check
|
||||
to IPv4 ([#111](https://git.eeqj.de/sneak/netwatch/issues/111)).
|
||||
|
||||
# Completed Steps
|
||||
|
||||
@@ -34,6 +33,16 @@ Write the latency statistics once and move the thresholds written inline in
|
||||
where the repo stands, and Next Step and Future Steps hold only open work,
|
||||
linked to its issue where one exists. The viewport harness README names Node's
|
||||
test runner, not `vitest`
|
||||
- 2026-10-04: in `src/main.js` (issue #102), a target's min, max, median and
|
||||
average latency come from one list of its answers, through the same function
|
||||
the summary's figures use, so the median is written once. The latency color
|
||||
limits are one table in `CONFIG`, read by both the figure's and the
|
||||
sparkline's color. The health thresholds, the debug log's length, the gateway
|
||||
check's timeout, the recovery probe's number of hosts and interval, how often
|
||||
the rows are sorted and the delay before the first sparkline resize are
|
||||
`CONFIG` entries too. A unit test checks the summary's figures. Nothing the
|
||||
page does or shows changed; the footer's color legend still writes the limits
|
||||
out as text
|
||||
- 2026-10-04: password guesses at `/metrics` are rate limited (issue #104): each
|
||||
client address, resolved through `TRUSTED_PROXIES` as for reports, may make 60
|
||||
requests to `/metrics` a minute, counted by `go-chi/httprate` apart from its
|
||||
@@ -375,8 +384,6 @@ Write the latency statistics once and move the thresholds written inline in
|
||||
|
||||
# Future Steps
|
||||
|
||||
- Take "IPv4 only" out of the page's footer, as nothing in the page limits a
|
||||
check to IPv4 ([#111](https://git.eeqj.de/sneak/netwatch/issues/111))
|
||||
- Decide whether the repo moves to the layout `REPO_POLICIES.md` gives, with
|
||||
`backend/` no longer repeating files from the root
|
||||
([#30](https://git.eeqj.de/sneak/netwatch/issues/30))
|
||||
|
||||
+93
-82
@@ -30,6 +30,39 @@ export const CONFIG = {
|
||||
return [0, 1, 2, 3, 4, 5].map((i) => Math.round((d * i) / 5));
|
||||
},
|
||||
canvasHeight: 96,
|
||||
// A latency figure and its sparkline take the color of the first entry
|
||||
// whose limit, in ms, the latency is below.
|
||||
latencyColors: [
|
||||
{ below: 50, hex: "#22c55e", className: "text-green-500" },
|
||||
{ below: 100, hex: "#84cc16", className: "text-lime-500" },
|
||||
{ below: 200, hex: "#eab308", className: "text-yellow-500" },
|
||||
{ below: 500, hex: "#f97316", className: "text-orange-500" },
|
||||
{ below: Infinity, hex: "#ef4444", className: "text-red-500" },
|
||||
],
|
||||
// The health is offline when more than offlineTimeouts WAN hosts timed
|
||||
// out or were unreachable and at most offlineReachable answered;
|
||||
// otherwise degraded when more than degradedTimeouts timed out or were
|
||||
// unreachable; otherwise slow when more than slowHosts answered after
|
||||
// more than slowLatency ms.
|
||||
offlineTimeouts: 10,
|
||||
offlineReachable: 4,
|
||||
degradedTimeouts: 4,
|
||||
slowHosts: 3,
|
||||
slowLatency: 1000,
|
||||
// The debug log keeps its last maxLogEntries lines.
|
||||
maxLogEntries: 1000,
|
||||
// A gateway candidate that has not answered after gatewayTimeout ms is
|
||||
// passed over.
|
||||
gatewayTimeout: 1500,
|
||||
// When no WAN host answers, the recovery probe checks recoveryProbeHosts
|
||||
// random ones every recoveryProbeInterval ms.
|
||||
recoveryProbeHosts: 4,
|
||||
recoveryProbeInterval: 500,
|
||||
// The rows are sorted after the first round that is not discarded, then
|
||||
// every roundsPerSort rounds.
|
||||
roundsPerSort: 10,
|
||||
// The sparklines are sized and drawn again resizeDelay ms after start.
|
||||
resizeDelay: 100,
|
||||
};
|
||||
|
||||
// WAN endpoints to monitor. These are used for the aggregate health/stats
|
||||
@@ -114,7 +147,8 @@ const debugLog = [];
|
||||
const log = (() => {
|
||||
function append(level, message) {
|
||||
debugLog.push({ timestamp: new Date(), level, message });
|
||||
if (debugLog.length > 1000) debugLog.splice(0, debugLog.length - 1000);
|
||||
if (debugLog.length > CONFIG.maxLogEntries)
|
||||
debugLog.splice(0, debugLog.length - CONFIG.maxLogEntries);
|
||||
const panel = document.getElementById("debug-panel");
|
||||
if (panel && !panel.classList.contains("hidden")) renderDebugLog();
|
||||
}
|
||||
@@ -174,7 +208,10 @@ async function detectGateway() {
|
||||
const result = await Promise.any(
|
||||
GATEWAY_CANDIDATES.map(async (url) => {
|
||||
const controller = new AbortController();
|
||||
const timeoutId = setTimeout(() => controller.abort(), 1500);
|
||||
const timeoutId = setTimeout(
|
||||
() => controller.abort(),
|
||||
CONFIG.gatewayTimeout,
|
||||
);
|
||||
try {
|
||||
await fetch(url, {
|
||||
method: "GET",
|
||||
@@ -199,6 +236,27 @@ async function detectGateway() {
|
||||
|
||||
// --- App State ---------------------------------------------------------------
|
||||
|
||||
// The min, max, median and average of latencies, a list of numbers, or all
|
||||
// null when it is empty. The median of an even count is the mean of the
|
||||
// middle two; it and the average are rounded.
|
||||
function latencyStats(latencies) {
|
||||
if (latencies.length === 0)
|
||||
return { min: null, max: null, med: null, avg: null };
|
||||
const sorted = [...latencies].sort((a, b) => a - b);
|
||||
const mid = Math.floor(sorted.length / 2);
|
||||
return {
|
||||
min: sorted[0],
|
||||
max: sorted[sorted.length - 1],
|
||||
med:
|
||||
sorted.length % 2
|
||||
? sorted[mid]
|
||||
: Math.round((sorted[mid - 1] + sorted[mid]) / 2),
|
||||
avg: Math.round(
|
||||
latencies.reduce((a, b) => a + b, 0) / latencies.length,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
export class HostState {
|
||||
constructor(host, pinned = false) {
|
||||
this.name = host.name;
|
||||
@@ -230,36 +288,14 @@ export class HostState {
|
||||
this._trim();
|
||||
}
|
||||
|
||||
averageLatency() {
|
||||
const valid = this.history.filter((p) => p.latency !== null);
|
||||
if (valid.length === 0) return null;
|
||||
return Math.round(
|
||||
valid.reduce((s, p) => s + p.latency, 0) / valid.length,
|
||||
);
|
||||
}
|
||||
|
||||
minLatency() {
|
||||
const valid = this.history.filter((p) => p.latency !== null);
|
||||
if (valid.length === 0) return null;
|
||||
return Math.min(...valid.map((p) => p.latency));
|
||||
}
|
||||
|
||||
maxLatency() {
|
||||
const valid = this.history.filter((p) => p.latency !== null);
|
||||
if (valid.length === 0) return null;
|
||||
return Math.max(...valid.map((p) => p.latency));
|
||||
}
|
||||
|
||||
medianLatency() {
|
||||
const sorted = this.history
|
||||
// The min, max, median and average latency of the checks in the history
|
||||
// that got an answer.
|
||||
historyStats() {
|
||||
return latencyStats(
|
||||
this.history
|
||||
.filter((p) => p.latency !== null)
|
||||
.map((p) => p.latency)
|
||||
.sort((a, b) => a - b);
|
||||
if (sorted.length === 0) return null;
|
||||
const mid = Math.floor(sorted.length / 2);
|
||||
return sorted.length % 2
|
||||
? sorted[mid]
|
||||
: Math.round((sorted[mid - 1] + sorted[mid]) / 2);
|
||||
.map((p) => p.latency),
|
||||
);
|
||||
}
|
||||
|
||||
_trim() {
|
||||
@@ -288,33 +324,13 @@ export class AppState {
|
||||
|
||||
/** WAN-only stats from latest sample (excludes local) */
|
||||
wanStats() {
|
||||
const reachable = this.wan.filter((h) => h.lastLatency !== null);
|
||||
const latencies = reachable.map((h) => h.lastLatency);
|
||||
const total = this.wan.length;
|
||||
if (latencies.length === 0)
|
||||
return {
|
||||
reachable: 0,
|
||||
total,
|
||||
min: null,
|
||||
max: null,
|
||||
med: null,
|
||||
avg: null,
|
||||
};
|
||||
const sorted = [...latencies].sort((a, b) => a - b);
|
||||
const mid = Math.floor(sorted.length / 2);
|
||||
const med =
|
||||
sorted.length % 2
|
||||
? sorted[mid]
|
||||
: Math.round((sorted[mid - 1] + sorted[mid]) / 2);
|
||||
const latencies = this.wan
|
||||
.filter((h) => h.lastLatency !== null)
|
||||
.map((h) => h.lastLatency);
|
||||
return {
|
||||
reachable: latencies.length,
|
||||
total,
|
||||
min: Math.min(...latencies),
|
||||
max: Math.max(...latencies),
|
||||
med,
|
||||
avg: Math.round(
|
||||
latencies.reduce((a, b) => a + b, 0) / latencies.length,
|
||||
),
|
||||
total: this.wan.length,
|
||||
...latencyStats(latencies),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -340,12 +356,16 @@ export class AppState {
|
||||
const timeouts = this.wan.filter(
|
||||
(h) => h.status === "error" || h.status === "offline",
|
||||
).length;
|
||||
if (timeouts > 10 && reachable <= 4) return "offline";
|
||||
if (timeouts > 4) return "degraded";
|
||||
if (
|
||||
timeouts > CONFIG.offlineTimeouts &&
|
||||
reachable <= CONFIG.offlineReachable
|
||||
)
|
||||
return "offline";
|
||||
if (timeouts > CONFIG.degradedTimeouts) return "degraded";
|
||||
const slow = this.wan.filter(
|
||||
(h) => h.lastLatency !== null && h.lastLatency > 1000,
|
||||
(h) => h.lastLatency !== null && h.lastLatency > CONFIG.slowLatency,
|
||||
).length;
|
||||
if (slow > 3) return "slow";
|
||||
if (slow > CONFIG.slowHosts) return "slow";
|
||||
return "healthy";
|
||||
}
|
||||
|
||||
@@ -557,21 +577,13 @@ export async function measureLatency(url, signal) {
|
||||
|
||||
export function latencyHex(latency) {
|
||||
if (latency === null) return "#6b7280";
|
||||
if (latency < 50) return "#22c55e";
|
||||
if (latency < 100) return "#84cc16";
|
||||
if (latency < 200) return "#eab308";
|
||||
if (latency < 500) return "#f97316";
|
||||
return "#ef4444";
|
||||
return CONFIG.latencyColors.find((c) => latency < c.below).hex;
|
||||
}
|
||||
|
||||
export function latencyClass(latency, status) {
|
||||
if (status === "offline" || status === "error" || latency === null)
|
||||
return "text-gray-500";
|
||||
if (latency < 50) return "text-green-500";
|
||||
if (latency < 100) return "text-lime-500";
|
||||
if (latency < 200) return "text-yellow-500";
|
||||
if (latency < 500) return "text-orange-500";
|
||||
return "text-red-500";
|
||||
return CONFIG.latencyColors.find((c) => latency < c.below).className;
|
||||
}
|
||||
|
||||
// --- Sparkline Renderer ------------------------------------------------------
|
||||
@@ -912,10 +924,7 @@ function updateHostRow(host, index) {
|
||||
latencyEl.innerHTML = `<span class="text-gray-500">---</span>`;
|
||||
}
|
||||
|
||||
const avg = host.averageLatency();
|
||||
const med = host.medianLatency();
|
||||
const min = host.minLatency();
|
||||
const max = host.maxLatency();
|
||||
const { min, med, avg, max } = host.historyStats();
|
||||
if (host.status === "online" && avg !== null) {
|
||||
statusEl.innerHTML = statusStatsHTML([
|
||||
["min", min],
|
||||
@@ -1182,8 +1191,9 @@ export async function tick(state, signal, onOffline) {
|
||||
// rows whose check ended before the resume still read "paused"
|
||||
state.allHosts.forEach((host, i) => updateHostRow(host, i));
|
||||
|
||||
// Sort after the first real check, then every 10 ticks thereafter
|
||||
if (state.tickCount === 2 || state.tickCount % 10 === 1) {
|
||||
// Sort after the first real check, then every CONFIG.roundsPerSort
|
||||
// ticks thereafter
|
||||
if (state.tickCount === 2 || state.tickCount % CONFIG.roundsPerSort === 1) {
|
||||
sortAndRebuildWAN(state);
|
||||
}
|
||||
|
||||
@@ -1206,9 +1216,10 @@ export async function tick(state, signal, onOffline) {
|
||||
|
||||
// --- Recovery Probe ----------------------------------------------------------
|
||||
|
||||
// When offline, check 4 random WAN hosts every 500ms, giving up the checks
|
||||
// started 500ms before, so at most 4 are ever waiting. As soon as one
|
||||
// answers, stop probing and start a new round at once.
|
||||
// When offline, check CONFIG.recoveryProbeHosts random WAN hosts every
|
||||
// CONFIG.recoveryProbeInterval ms, giving up the checks started one interval
|
||||
// before, so at most that many are ever waiting. As soon as one answers,
|
||||
// stop probing and start a new round at once.
|
||||
function startRecoveryProbe(state, startRounds) {
|
||||
if (state._recoveryProbeId) return; // already running
|
||||
const candidates = [...state.wan];
|
||||
@@ -1216,7 +1227,7 @@ function startRecoveryProbe(state, startRounds) {
|
||||
const j = Math.floor(Math.random() * (i + 1));
|
||||
[candidates[i], candidates[j]] = [candidates[j], candidates[i]];
|
||||
}
|
||||
const canaries = candidates.slice(0, 4);
|
||||
const canaries = candidates.slice(0, CONFIG.recoveryProbeHosts);
|
||||
log.notice(
|
||||
`Recovery probe started (${canaries.map((h) => h.name).join(", ")})`,
|
||||
);
|
||||
@@ -1233,7 +1244,7 @@ function startRecoveryProbe(state, startRounds) {
|
||||
startRounds();
|
||||
});
|
||||
}
|
||||
}, 500);
|
||||
}, CONFIG.recoveryProbeInterval);
|
||||
}
|
||||
|
||||
function stopRecoveryProbe(state) {
|
||||
@@ -1497,7 +1508,7 @@ async function init() {
|
||||
});
|
||||
|
||||
window.addEventListener("resize", () => handleResize(state));
|
||||
setTimeout(() => handleResize(state), 100);
|
||||
setTimeout(() => handleResize(state), CONFIG.resizeDelay);
|
||||
}
|
||||
|
||||
// Bootstrap only when loaded as the page: a real DOM containing the #app
|
||||
|
||||
+26
-10
@@ -347,19 +347,35 @@ for (const { history, latencies, statistics } of [
|
||||
},
|
||||
]) {
|
||||
test(`a target's min, max, average and median latency over ${history}`, () => {
|
||||
const host = hostAfter(latencies);
|
||||
assert.deepEqual(
|
||||
{
|
||||
min: host.minLatency(),
|
||||
max: host.maxLatency(),
|
||||
average: host.averageLatency(),
|
||||
median: host.medianLatency(),
|
||||
},
|
||||
statistics,
|
||||
);
|
||||
const { min, max, avg, med } = hostAfter(latencies).historyStats();
|
||||
assert.deepEqual({ min, max, average: avg, median: med }, statistics);
|
||||
});
|
||||
}
|
||||
|
||||
// The summary's figures come from each WAN target's last check, by the same
|
||||
// rules as a target's own: here four answered, one was found unreachable
|
||||
// and the rest have not been checked yet. The median, 22.5, and the
|
||||
// average, 21.25, are rounded.
|
||||
test("the summary's min, max, median and average latency over the WAN targets' last checks", () => {
|
||||
const state = new AppState([]);
|
||||
[30, 10, null, 25, 20].forEach((latency, i) =>
|
||||
state.wan[i].pushSample(
|
||||
Date.now(),
|
||||
latency === null
|
||||
? { latency: null, error: "unreachable" }
|
||||
: { latency, error: null },
|
||||
),
|
||||
);
|
||||
assert.deepEqual(state.wanStats(), {
|
||||
reachable: 4,
|
||||
total: state.wan.length,
|
||||
min: 10,
|
||||
max: 30,
|
||||
med: 23,
|
||||
avg: 21,
|
||||
});
|
||||
});
|
||||
|
||||
// An app state in which, of the WAN targets, the first timedOut timed out,
|
||||
// the next unreachable were found unreachable, the next answered answered
|
||||
// after latency ms, and the rest have not been checked yet.
|
||||
|
||||
Reference in New Issue
Block a user