1 Commits
Author SHA1 Message Date
sneak 871b986978 next into main (#73)
check / check (push) Successful in 2m3s
Reviewed-on: #73
2026-09-29 12:04:30 +02:00
26 changed files with 212 additions and 1059 deletions
-3
View File
@@ -4,6 +4,3 @@ tmp
.DS_Store
*.log
.claude
# .git is sent so the build can stamp the version, without its config.
.git/config
+6 -23
View File
@@ -20,10 +20,7 @@ RUN make lint
# golang:1.25-alpine (2026-02-27)
FROM golang:1.25-alpine@sha256:f6751d823c26342f9506c03797d2527668d095b0a15f1862cddb4d927a7a4ced AS builder
# gcc and musl-dev are for make test: its race detector needs cgo, which
# Go turns on by itself once a C compiler is present. make build still
# sets CGO_ENABLED=0, so the binary stays static.
RUN apk add --no-cache gcc git make musl-dev
RUN apk add --no-cache make
WORKDIR /src
@@ -40,24 +37,11 @@ RUN make test
# make build is a shim around backend/script/build, the one definition
# of the build command:
# CGO_ENABLED=0 go build -trimpath -ldflags "-s -w -X main.Version=..."
# CGO_ENABLED=0 go build -trimpath -ldflags "-s -w -X main.Version=... -X main.Buildarch=..."
# That script reads VERSION from the environment, so it is handed over
# there rather than as a make variable.
#
# The version is the VERSION build argument when one is given, otherwise
# `git describe --tags --always` of the repo's .git: the tag on a tagged
# commit, tag-N-gHASH on a commit after one, the short commit when no
# tag is reachable. A version that still comes out empty, dev or unknown
# fails the build. .git goes to /git, not /src/.git, where go build would
# find it and record VCS details of a work tree holding only backend/.
COPY .git /git
ARG VERSION
RUN version="${VERSION:-$(git --git-dir=/git describe --tags --always)}"; \
case "$version" in ""|dev|unknown) \
echo "version is '$version' although .git is present" >&2; \
exit 1 ;; \
esac; \
VERSION="$version" make build
ARG VERSION=dev
RUN VERSION="${VERSION}" make build
# Frontend stage
# node:22-alpine as of 2026-02-22
@@ -68,9 +52,8 @@ RUN yarn install --frozen-lockfile
RUN apk add --no-cache git make
COPY . .
# make frontend-check is the frontend half of make check (test + lint +
# fmt-check); its test step runs the unit tests, then the production
# yarn build, so this both produces dist/ and gates the image on
# lint/fmt-check/test regressions.
# fmt-check); its test step is the production yarn build, so this both
# produces dist/ and gates the image on lint/fmt-check/test regressions.
# This node stage has neither Go nor Docker; the lint and builder stages
# above gate the backend half.
RUN make frontend-check
+10 -18
View File
@@ -39,23 +39,21 @@ halves, so the root `make check` fails if either one is broken. We provide:
- `script/bootstrap` — install all dependencies (the pinned node via nvm unless
one new enough for the frontend's dependencies is installed, yarn via
corepack, `yarn install --frozen-lockfile`, the pinned Go unless one at least
as new as `backend/go.mod` asks for is installed, the Go modules, and gcc with
the C library headers unless gcc is installed, for the race detector in
`make test`), linking what it installs itself into `~/.local/bin`, which has
to be on `PATH`. It installs no Go linter and not Docker: `make lint` runs the
linter in Docker
as new as `backend/go.mod` asks for is installed, and the Go modules), linking
what it installs itself into `~/.local/bin`, which has to be on `PATH`. It
installs no Go linter and not Docker: `make lint` runs the linter in Docker
- `script/setup` — make a fresh clone ready for development: bootstrap plus the
git pre-commit hook
- `script/projectname` — print the project name (used for the Docker image tag)
- `script/test` — run `script/frontend-test`, then the backend's Go tests, each
under its own 30-second timeout
- `script/test` — run `script/frontend-test`, then the backend's Go tests, both
within one 30-second timeout
- `script/lint` — run `script/frontend-lint`, then golangci-lint in Docker, by
building the lint stage of `Dockerfile` without the cache
- `script/fmt` — format all files (writes): prettier, then gofmt over `backend/`
- `script/fmt-check` — check formatting (read-only): prettier, then gofmt
- `script/check` — run test, lint, and fmt-check
- `script/frontend-test` — run the unit tests in `test/unit/` with Node's
built-in test runner, then the production build
- `script/frontend-test` — run the production build as the frontend's test (no
unit tests yet)
- `script/frontend-lint` — run prettier in check mode
- `script/frontend-fmt` — format everything prettier understands (writes)
- `script/frontend-fmt-check` — check prettier formatting (read-only)
@@ -138,14 +136,8 @@ Local hosts are tracked separately from WAN stats.
### Latency measurement
HEAD requests with `mode: 'no-cors'` and `cache: 'no-store'`, timed with
`performance.now()`. Each check times out after 80% of the refresh interval (24
seconds at 30 seconds) and is then recorded as a timeout, so a round's checks
have all finished before the next round is due. When no WAN host answers, a
recovery probe checks 4 random WAN hosts every half second, giving up the checks
it started half a second before. As soon as one answers, a new round starts at
once, as it does after an interval change. A round started early gives up the
last round's checks if they are still waiting, and that round records nothing,
so rounds never overlap. IPv4 only.
`performance.now()`. 1-second timeout; anything over 1000ms is clamped to
unreachable. IPv4 only.
### Color coding
@@ -220,7 +212,7 @@ What the [upaas](https://git.eeqj.de/sneak/upaas) app for netwatch needs:
- `REPORTS_PER_MINUTE`, default `60`: reports each client address may send a
minute
- `DATA_DIR_MAX_BYTES`, default `1073741824` (1 GiB): the most room the
report files may take; the oldest are deleted to stay under it
report files may take
- `CORS_ALLOWED_ORIGINS`, default empty: other origins whose pages may call
the API
- `DEBUG`, default `false`: debug logging
-28
View File
@@ -23,34 +23,6 @@ latest run passes.
# Completed Steps
- 2026-10-03: `DATA_DIR_MAX_BYTES` is now how much of the report files is kept
(issue #54): when a report would take them past it, the oldest report files
are deleted to make room, each deletion logged, and at start files already
past it are deleted the same way. A file still being written is never deleted.
A report is refused with 507 only when the reports waiting to be written fill
the cap on their own, and then no file is deleted. The reports of a failed
write stop counting, and the part of its file written is removed. A file that
cannot be deleted still counts until the next start; one already deleted by
hand counts as freed
- 2026-10-03: the Go tests run with the race detector and coverage (issue #88):
`backend/script/test` runs `go test -timeout 30s -race -cover ./...` and, if
that fails, runs it again with `-v` and fails. Go's `-timeout` bounds the
tests, not their compile; the root `script/test` no longer puts one 30-second
timeout around both halves, which a cold Go build cache could use up on
compiling alone. The race detector needs a C compiler: the builder stage of
`Dockerfile` has gcc and musl-dev, and `script/bootstrap` installs gcc, with
the C library headers on apt and apk, when gcc is missing; the binary is still
built with `CGO_ENABLED=0`. New tests cover the health check's answer, a valid
report's answer, a report file's exact contents, and the flush when the buffer
reaches 10 MiB; the handlers' `TestImport` stub is gone
- 2026-10-03: each target check times out after 80% of the refresh interval
(issue #78), 24 seconds at 30 seconds, where it was capped at 3 seconds. A
round started early, after an interval change or when the recovery probe finds
a target answering, gives up the last round's checks if they are still
waiting, so rounds never overlap; the recovery probe gives up its own checks
after half a second. The frontend has its first unit tests, run by
`script/frontend-test` with Node's built-in test runner; for them,
`index.html` now links `src/styles.css`, which `src/main.js` used to import
- 2026-09-29: the container sets up its own data directory (issue #75):
`bin/entrypoint.sh`, still as root, creates `DATA_DIR` if missing and gives it
and `/data` to the `netwatch` user with mode 750 before starting the backend
+13 -22
View File
@@ -32,13 +32,10 @@ pattern as the repo root: the targets in `backend/Makefile` are thin shims over
`test`, `fmt` and `fmt-check`:
- `script/build` — compile the static `netwatch-server` binary with its version
stamped in. The version is `VERSION` from the environment;
and architecture stamped in. The version is `VERSION` from the environment;
when that is unset or empty, it falls back to `git describe` inside a git
checkout, then to `dev`
- `script/test` — run the Go tests with the race detector and coverage. Go's
`-timeout 30s` bounds the tests, not their compile. If they fail, they run
again with `-v` for the details, and the script fails. The race detector needs
a C compiler
- `script/test` — run the Go tests under a 30-second timeout
- `script/lint` — check `.golangci.yml` against its pinned sha256, then run
golangci-lint. It runs inside the golangci-lint image of the lint stage of
the root `Dockerfile`; from a checkout, run `make lint` at the repo root,
@@ -83,7 +80,7 @@ Internal packages in `internal/` follow standard Go project layout:
| `BIND_ADDRESS` | empty | IP address to listen on; empty listens on every interface |
| `PORT` | `8080` | HTTP listen port |
| `DATA_DIR` | `./data/reports` | Directory for compressed reports |
| `DATA_DIR_MAX_BYTES` | `1073741824` (1 GiB) | Most bytes of report files kept in `DATA_DIR`, oldest deleted first; see [Report limits](#report-limits) |
| `DATA_DIR_MAX_BYTES` | `1073741824` (1 GiB) | Largest total size of the report files in `DATA_DIR`; see [Report limits](#report-limits) |
| `DEBUG` | `false` | Enable debug logging |
| `TRUSTED_PROXIES` | loopback + RFC1918 | Comma-separated CIDRs whose `X-Forwarded-For` / `X-Real-IP` headers are trusted for client IP resolution |
| `REPORTS_PER_MINUTE` | `60` | Reports each client address may send a minute; see [Report limits](#report-limits) |
@@ -128,11 +125,10 @@ Reports are written as `reports-<timestamp>-<number>.jsonl.zst` files in
`DATA_DIR`. The timestamp is in UTC to the millisecond, so the names sort by
time. The number starts at 1 when the server starts and goes up by one for each
file the server starts to write, so two files written in the same millisecond
still get different names. A failed write uses up its number and leaves a gap in
the numbers: its file, if it was created, is removed. The file stays, counted
toward `DATA_DIR_MAX_BYTES` from the next start, only if removing it fails too.
Each file contains one JSON object per line, compressed with zstd. Files are
created with `O_EXCL` to prevent overwrites.
still get different names. A failed write uses up its number, leaving a gap in
the numbers if the file could not be created and otherwise a file under that
number that may be incomplete. Each file contains one JSON object per line,
compressed with zstd. Files are created with `O_EXCL` to prevent overwrites.
### Report limits
@@ -152,17 +148,11 @@ credentials, so it is bounded instead. Both refusals below answer with the same
`X-RateLimit-Reset` headers.
- **Size cap.** The report files in `DATA_DIR` may total at most
`DATA_DIR_MAX_BYTES`, counting the files already there at start. Reports
waiting in memory count at their uncompressed size until they are written;
those lost to a failed write stop counting, and the part of its file written
is removed. When a report would take the total past the cap, the oldest report
files are deleted to make room, and each deletion is logged with the file's
name and size; a file still being written is never deleted. A report is
refused with 507, and nothing of it is stored, only when the reports waiting
to be written fill the cap on their own, and then no file is deleted. At
start, report files past the cap, as after lowering it, are deleted the same
way. So the cap is how much of the newest reports is kept: the default of 1
GiB is small enough for any host; set it to the space you can give
`DATA_DIR`.
waiting in memory count at their uncompressed size until they are written, so
a report that would take the total past the cap is refused with 507, and
nothing of it is stored. Deleting report files frees room only at the next
start, when the files are counted again. The default of 1 GiB is small enough
for any host; set it to the space you can give `DATA_DIR`.
### CORS
@@ -180,6 +170,7 @@ starting, with an error naming `CORS_ALLOWED_ORIGINS`.
- Add integration test that POSTs a report and verifies the compressed output
- Add report decompression/query endpoint
- Add metrics (Prometheus) for buffer size, flush count, report count
- Add retention policy to prune old report files
## License
+2
View File
@@ -21,6 +21,7 @@ import (
var (
Appname = "netwatch-server"
Version string
Buildarch string
)
func main() {
@@ -39,6 +40,7 @@ func main() {
globals.Appname = Appname
globals.Version = Version
globals.Buildarch = Buildarch
fx.New(
fx.Provide(
+4
View File
@@ -10,18 +10,22 @@ var (
Appname string
// Version is the git version tag.
Version string
// Buildarch is the build architecture.
Buildarch string
)
// Globals holds build-time metadata for the application.
type Globals struct {
Appname string
Version string
Buildarch string
}
// New creates a Globals instance from package-level variables.
func New(_ fx.Lifecycle) (*Globals, error) {
return &Globals{
Appname: Appname,
Buildarch: Buildarch,
Version: Version,
}, nil
}
@@ -0,0 +1,13 @@
package handlers_test
import (
"testing"
_ "sneak.berlin/go/netwatch/internal/handlers"
)
func TestImport(t *testing.T) {
t.Parallel()
// Compilation check — verifies the package parses
// and all imports resolve.
}
@@ -1,120 +0,0 @@
package handlers_test
import (
"encoding/json"
"maps"
"net/http"
"net/http/httptest"
"slices"
"testing"
"time"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/handlers"
"sneak.berlin/go/netwatch/internal/healthcheck"
"sneak.berlin/go/netwatch/internal/logger"
"go.uber.org/fx/fxtest"
)
// newStartedHandlers builds Handlers with a real health check for the
// server named in g, and starts them, which records the time the
// uptime counts from.
func newStartedHandlers(t *testing.T, g *globals.Globals) *handlers.Handlers {
t.Helper()
lc := fxtest.NewLifecycle(t)
log, err := logger.New(lc, logger.Params{Globals: g})
if err != nil {
t.Fatalf("logger: %v", err)
}
hc, err := healthcheck.New(lc,
healthcheck.Params{Globals: g, Logger: log})
if err != nil {
t.Fatalf("health check: %v", err)
}
h, err := handlers.New(lc,
handlers.Params{Globals: g, Healthcheck: hc, Logger: log})
if err != nil {
t.Fatalf("handlers: %v", err)
}
lc.RequireStart()
t.Cleanup(lc.RequireStop)
return h
}
// TestHandleHealthCheck checks the health check's answer: 200, a JSON
// content type, and a JSON object with exactly the fields of
// healthcheck.Response, carrying this server's name and version and
// an uptime counted from its start.
func TestHandleHealthCheck(t *testing.T) {
t.Parallel()
g := &globals.Globals{Appname: "netwatch-server", Version: "v1.2.3"}
h := newStartedHandlers(t, g)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/.well-known/healthcheck", http.NoBody)
h.HandleHealthCheck().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusOK)
}
contentType := rec.Header().Get("Content-Type")
if contentType != "application/json; charset=utf-8" {
t.Errorf("Content-Type = %q, want %q",
contentType, "application/json; charset=utf-8")
}
var body map[string]any
err := json.Unmarshal(rec.Body.Bytes(), &body)
if err != nil {
t.Fatalf("body not a JSON object: %v (%q)", err, rec.Body.String())
}
fields := []string{
"appname", "now", "status", "uptimeHuman", "uptimeSeconds", "version",
}
if got := slices.Sorted(maps.Keys(body)); !slices.Equal(got, fields) {
t.Fatalf("fields = %v, want %v", got, fields)
}
for field, want := range map[string]string{
"appname": g.Appname, "status": "ok", "version": g.Version,
} {
if body[field] != want {
t.Errorf("%s = %v, want %q", field, body[field], want)
}
}
now, _ := body["now"].(string)
at, err := time.Parse(time.RFC3339Nano, now)
if err != nil || time.Since(at).Abs() > time.Minute {
t.Errorf("now = %q, want the current time in RFC 3339 (%v)", now, err)
}
// Started just now, so the uptime is well under a minute.
human, _ := body["uptimeHuman"].(string)
uptime, err := time.ParseDuration(human)
if err != nil || uptime > time.Minute {
t.Errorf("uptimeHuman = %q, want a duration under a minute (%v)",
human, err)
}
seconds, ok := body["uptimeSeconds"].(float64)
if !ok || seconds < 0 || seconds > time.Minute.Seconds() {
t.Errorf("uptimeSeconds = %v, want a number of seconds under a minute",
body["uptimeSeconds"])
}
}
+3 -4
View File
@@ -86,12 +86,11 @@ func (s *Handlers) decodeErrorStatus(err error) int {
}
// appendErrorStatus logs a failure to store a report and returns
// the status to send: 507 when the reports waiting to be written fill
// the size cap, otherwise 500.
// the status to send: 507 when the report files are at their size
// cap, otherwise 500.
func (s *Handlers) appendErrorStatus(err error) int {
if errors.Is(err, reportbuf.ErrFull) {
s.log.Warn("report refused: " +
"reports waiting to be written fill the size cap")
s.log.Warn("report refused: report files at their size cap")
return http.StatusInsufficientStorage
}
+3 -60
View File
@@ -46,63 +46,6 @@ func decodeStatus(t *testing.T, body []byte) string {
return resp.Status
}
// TestHandleReportAcceptsValidReports checks the answer to a valid
// report, one with no hosts and one shaped as the frontend sends them:
// 200 and {"status":"ok"} as JSON.
func TestHandleReportAcceptsValidReports(t *testing.T) {
t.Parallel()
tests := []struct {
name string
body string
}{
{
name: "no hosts",
body: `{"clientId":"c1","geo":null,"hosts":[],` +
`"timestamp":"2026-10-03T12:00:00.000Z"}`,
},
{
name: "a host with a latency and an error sample",
body: `{"clientId":"c1","geo":null,"hosts":[{` +
`"name":"Example","url":"https://example.com/",` +
`"status":"error","history":[` +
`{"t":1790000000000,"latency":42,"error":null},` +
`{"t":1790000003000,"latency":null,"error":"timeout"}]}],` +
`"timestamp":"2026-10-03T12:00:00.000Z"}`,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
h := newTestHandlers(stubAppender{}, io.Discard)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(tt.body),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusOK)
}
contentType := rec.Header().Get("Content-Type")
if contentType != "application/json; charset=utf-8" {
t.Errorf("Content-Type = %q, want %q",
contentType, "application/json; charset=utf-8")
}
if got := rec.Body.String(); got != "{\"status\":\"ok\"}\n" {
t.Errorf("body = %q, want %q", got, "{\"status\":\"ok\"}\n")
}
})
}
}
func TestHandleReportStorageFailureIsNon2xx(t *testing.T) {
t.Parallel()
@@ -125,9 +68,9 @@ func TestHandleReportStorageFailureIsNon2xx(t *testing.T) {
}
}
// TestHandleReportFullIs507 checks the answer when the reports waiting
// to be written fill the size cap: 507 and the usual error body, which
// tells the client nothing more.
// TestHandleReportFullIs507 checks the answer when the report files
// are at their size cap: 507 and the usual error body, which tells
// the client nothing more.
func TestHandleReportFullIs507(t *testing.T) {
t.Parallel()
+1 -2
View File
@@ -5,7 +5,6 @@ package logger
import (
"log/slog"
"os"
"runtime"
"sneak.berlin/go/netwatch/internal/globals"
@@ -96,6 +95,6 @@ func (l *Logger) Identify() {
l.log.Info("starting",
"appname", l.params.Globals.Appname,
"version", l.params.Globals.Version,
"arch", runtime.GOARCH,
"buildarch", l.params.Globals.Buildarch,
)
}
+1 -15
View File
@@ -1,13 +1,6 @@
package reportbuf
import (
"os"
"time"
)
// FlushSizeThreshold exposes the buffer size at which Append starts
// writing a report file to the external tests.
const FlushSizeThreshold = flushSizeThreshold
import "time"
// Flush writes the buffered reports to a file now, as the periodic
// flush does, so tests need not wait a minute for it.
@@ -20,10 +13,3 @@ func (b *Buffer) Flush() error {
func (b *Buffer) StopClock(at time.Time) {
b.now = func() time.Time { return at }
}
// OnFileCreated makes the buffer call fn with each report file it
// writes from now on, once the file is created and before anything is
// written to it.
func (b *Buffer) OnFileCreated(fn func(f *os.File)) {
b.fileCreated = fn
}
+45 -145
View File
@@ -1,6 +1,5 @@
// Package reportbuf accumulates telemetry reports in memory
// and periodically flushes them to zstd-compressed JSONL files,
// deleting the oldest files to keep them under a size cap.
// and periodically flushes them to zstd-compressed JSONL files.
package reportbuf
import (
@@ -38,10 +37,9 @@ const (
fileSuffix = ".jsonl.zst"
)
// ErrFull is returned by Append when the reports waiting to be
// written leave no room for the report under the configured maximum
// size, however many report files are deleted.
var ErrFull = errors.New("reports waiting to be written fill the size cap")
// ErrFull is returned by Append when storing the report would
// take the report files past the configured maximum size.
var ErrFull = errors.New("report files at their size cap")
// Params defines the dependencies for Buffer.
type Params struct {
@@ -51,29 +49,12 @@ type Params struct {
Logger *logger.Logger
}
// reportFile is a report file that may be deleted to make room, with
// the size it counts for in usedBytes.
type reportFile struct {
name string
size int64
}
// Buffer accumulates JSON lines in memory and flushes them
// to zstd-compressed files on disk.
type Buffer struct {
buf bytes.Buffer
dataDir string
done chan struct{}
// fileCreated is called with each report file once it is
// created, before anything is written to it: it does nothing,
// except in tests that hold the write open or make it fail.
fileCreated func(f *os.File)
// files are the report files that may be deleted to make room,
// oldest first: those in dataDir at start, then each one this
// buffer writes, once it is complete. A file still being written
// is not among them. filesBytes is their total size.
files []reportFile
filesBytes int64
log *slog.Logger
maxBytes int64
mu sync.Mutex
@@ -85,8 +66,8 @@ type Buffer struct {
seq atomic.Uint64
stopOnce sync.Once
// usedBytes is what Append checks against maxBytes: the size
// of the report files in dataDir, plus the reports waiting to
// be written to one at their uncompressed size.
// of the report files in dataDir, plus the reports not yet
// written to one at their uncompressed size.
usedBytes int64
}
@@ -104,7 +85,6 @@ func New(
b := &Buffer{
dataDir: dir,
done: make(chan struct{}),
fileCreated: func(*os.File) {},
log: params.Logger.Get(),
maxBytes: params.Config.DataDirMaxBytes,
now: time.Now,
@@ -117,28 +97,12 @@ func New(
return fmt.Errorf("create data dir: %w", err)
}
// Report files left by earlier runs count too, and are
// the first to be deleted to make room.
files, err := reportFiles(b.dataDir)
// Report files left by earlier runs count too.
b.usedBytes, err = reportFilesSize(b.dataDir)
if err != nil {
return err
}
b.mu.Lock()
b.files = files
for _, f := range files {
b.filesBytes += f.size
}
b.usedBytes = b.filesBytes
// The files may be past the cap, if it was lowered since
// the last run.
b.deleteOldestFiles(0)
b.mu.Unlock()
go b.flushLoop()
return nil
@@ -164,10 +128,9 @@ func New(
}
// Append marshals v as a single JSON line and appends it to
// the buffer. If the line would take usedBytes past maxBytes, the
// oldest report files are deleted to make room; it stores nothing
// and returns ErrFull if that cannot make room. If the buffer
// reaches the size threshold, it is drained and written to disk
// the buffer. It stores nothing and returns ErrFull if the line
// would take usedBytes past maxBytes. If the buffer reaches the
// size threshold, it is drained and written to disk
// asynchronously.
func (b *Buffer) Append(v any) error {
line, err := json.Marshal(v)
@@ -179,8 +142,6 @@ func (b *Buffer) Append(v any) error {
b.mu.Lock()
b.deleteOldestFiles(lineBytes)
if b.usedBytes+lineBytes > b.maxBytes {
b.mu.Unlock()
@@ -210,37 +171,6 @@ func (b *Buffer) Append(v any) error {
return nil
}
// deleteOldestFiles deletes report files, oldest first, until n more
// bytes fit under maxBytes. It deletes none when the reports waiting
// to be written leave no room for n even with every file gone, since
// that would lose the files for nothing. The caller must hold b.mu.
func (b *Buffer) deleteOldestFiles(n int64) {
for b.usedBytes+n > b.maxBytes && len(b.files) > 0 {
if b.usedBytes-b.filesBytes+n > b.maxBytes {
return
}
f := b.files[0]
b.files = b.files[1:]
b.filesBytes -= f.size
// A file already gone, deleted by hand, has freed its room too.
err := os.Remove(filepath.Join(b.dataDir, f.name))
if err != nil && !errors.Is(err, fs.ErrNotExist) {
// The file is still there, so it still counts. It is
// not tried again until the next start.
b.log.Error("delete report file failed",
"file", f.name, "error", err)
continue
}
b.usedBytes -= f.size
b.log.Info("deleted report file to make room",
"file", f.name, "bytes", f.size)
}
}
// flushLoop runs a ticker that periodically flushes buffered
// data to disk until the done channel is closed.
func (b *Buffer) flushLoop() {
@@ -287,41 +217,15 @@ func (b *Buffer) drainBuf() []byte {
return data
}
// writeFile writes data, reports drained from the buffer, to a new
// timestamped zstd-compressed JSONL file in the data directory.
// writeFile creates a timestamped zstd-compressed JSONL file
// in the data directory.
func (b *Buffer) writeFile(data []byte) error {
// The timestamp comes first, so the names sort by time; the number
// after it tells apart files named in the same millisecond.
ts := b.now().UTC().Format("2006-01-02T15-04-05.000Z")
name := fmt.Sprintf("%s%s-%d%s", filePrefix, ts, b.seq.Add(1), fileSuffix)
path := filepath.Join(b.dataDir, name)
size, err := b.createFile(filepath.Join(b.dataDir, name), data)
// The reports no longer wait to be written, so they stop counting
// at their uncompressed size. If the write failed they are lost;
// otherwise they count as the file, which from here on may be
// deleted to make room.
b.mu.Lock()
defer b.mu.Unlock()
b.usedBytes -= int64(len(data))
if err != nil {
return err
}
b.usedBytes += size
b.files = append(b.files, reportFile{name: name, size: size})
b.filesBytes += size
return nil
}
// createFile creates the file at path holding data compressed with
// zstd, and returns its size. If the write fails once the file is
// created, it removes the file, so that a failed write leaves nothing
// behind to take room.
func (b *Buffer) createFile(path string, data []byte) (int64, error) {
// path is built from the operator-supplied dataDir plus a
// generated timestamp and number, so it carries no external input.
f, err := os.OpenFile( //nolint:gosec // see comment above
@@ -330,65 +234,61 @@ func (b *Buffer) createFile(path string, data []byte) (int64, error) {
filePerms,
)
if err != nil {
return 0, fmt.Errorf("create report file: %w", err)
return fmt.Errorf("create report file: %w", err)
}
b.fileCreated(f)
// Closes the file on the early returns below. The success
// path closes it explicitly to check the error; closing it
// a second time here is harmless.
defer func() { _ = f.Close() }()
size, err := writeCompressed(f, data)
if err != nil {
_ = f.Close()
return 0, errors.Join(err, os.Remove(path))
}
err = f.Close()
if err != nil {
err = fmt.Errorf("close report file: %w", err)
return 0, errors.Join(err, os.Remove(path))
}
return size, nil
}
// writeCompressed writes data to f compressed with zstd, and returns
// the size of f.
func writeCompressed(f *os.File, data []byte) (int64, error) {
enc, err := zstd.NewWriter(f)
if err != nil {
return 0, fmt.Errorf("create zstd encoder: %w", err)
return fmt.Errorf("create zstd encoder: %w", err)
}
_, err = enc.Write(data)
if err != nil {
_ = enc.Close()
return 0, fmt.Errorf("write compressed data: %w", err)
return fmt.Errorf("write compressed data: %w", err)
}
err = enc.Close()
if err != nil {
return 0, fmt.Errorf("close zstd encoder: %w", err)
return fmt.Errorf("close zstd encoder: %w", err)
}
info, err := f.Stat()
if err != nil {
return 0, fmt.Errorf("stat report file: %w", err)
return fmt.Errorf("stat report file: %w", err)
}
return info.Size(), nil
err = f.Close()
if err != nil {
return fmt.Errorf("close report file: %w", err)
}
// The reports counted at their uncompressed size while they
// waited; now they count as the file. After a failed write they
// stay counted as they were, which errs toward refusing reports
// early rather than letting the files pass the cap.
b.mu.Lock()
b.usedBytes += info.Size() - int64(len(data))
b.mu.Unlock()
return nil
}
// reportFiles returns the report files in dir, oldest first:
// os.ReadDir sorts them by name, and the names sort by time.
func reportFiles(dir string) ([]reportFile, error) {
// reportFilesSize returns the total size of the report files in
// dir.
func reportFilesSize(dir string) (int64, error) {
entries, err := os.ReadDir(dir)
if err != nil {
return nil, fmt.Errorf("read data dir: %w", err)
return 0, fmt.Errorf("read data dir: %w", err)
}
files := make([]reportFile, 0, len(entries))
var total int64
for _, entry := range entries {
name := entry.Name()
@@ -399,11 +299,11 @@ func reportFiles(dir string) ([]reportFile, error) {
info, err := entry.Info()
if err != nil {
return nil, fmt.Errorf("stat report file: %w", err)
return 0, fmt.Errorf("stat report file: %w", err)
}
files = append(files, reportFile{name: name, size: info.Size()})
total += info.Size()
}
return files, nil
return total, nil
}
+45 -441
View File
@@ -139,19 +139,11 @@ func lineBytes(t *testing.T, report any) int {
return len(line) + 1
}
// TestAppendPastCapIsRefused fills the cap with a report not yet
// written. The next report is refused, and the report file already in
// DATA_DIR is kept: it is smaller than a report, so deleting it could
// not make room.
func TestAppendPastCapIsRefused(t *testing.T) {
report := map[string]string{"id": "cap"}
dir := t.TempDir()
earlier := reportFilePath(dir, 1)
writeBytes(t, earlier, 1)
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(1+lineBytes(t, report)))
t.Setenv("DATA_DIR", t.TempDir())
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(lineBytes(t, report)))
buf := startBuffer(t)
@@ -164,38 +156,24 @@ func TestAppendPastCapIsRefused(t *testing.T) {
if !errors.Is(err, reportbuf.ErrFull) {
t.Fatalf("report past the cap: error = %v, want ErrFull", err)
}
if !exists(t, earlier) {
t.Fatal("report file deleted, though that could not make room")
}
}
// TestOldestReportFileDeletedFirst starts on a data directory holding
// report files from an earlier run, and a file that is not a report,
// which neither counts nor is ever deleted. Nothing is deleted while
// there is room; then only the oldest report file is.
func TestOldestReportFileDeletedFirst(t *testing.T) {
const fileBytes = 100
// TestCapCountsReportFilesAlreadyInDataDir starts on a data
// directory holding a report file from an earlier run, and a file
// that is not a report, which must not count.
func TestCapCountsReportFilesAlreadyInDataDir(t *testing.T) {
const earlierBytes = 100
report := map[string]string{"id": "oldest"}
report := map[string]string{"id": "cap"}
dir := t.TempDir()
oldest := reportFilePath(dir, 1)
kept := []string{
reportFilePath(dir, 2),
reportFilePath(dir, 3),
filepath.Join(dir, "notes.txt"),
}
writeBytes(t, oldest, fileBytes)
for _, path := range kept {
writeBytes(t, path, fileBytes)
}
writeBytes(t, filepath.Join(dir, "reports-2026-01-01T00-00-00.000Z.jsonl.zst"),
earlierBytes)
writeBytes(t, filepath.Join(dir, "notes.txt"), 10*earlierBytes)
t.Setenv("DATA_DIR", dir)
// Room for the three report files and one report.
t.Setenv("DATA_DIR_MAX_BYTES",
strconv.Itoa(3*fileBytes+lineBytes(t, report)))
strconv.Itoa(earlierBytes+lineBytes(t, report)))
buf := startBuffer(t)
@@ -204,55 +182,9 @@ func TestOldestReportFileDeletedFirst(t *testing.T) {
t.Fatalf("report that fills the cap exactly: %v", err)
}
if !exists(t, oldest) {
t.Fatal("oldest report file deleted while there was room")
}
err = buf.Append(report)
if err != nil {
t.Fatalf("report past the cap: %v", err)
}
if exists(t, oldest) {
t.Fatal("oldest report file kept when room was needed")
}
for _, path := range kept {
if !exists(t, path) {
t.Fatalf("%s deleted; only the oldest report file should be", path)
}
}
}
// TestStartDeletesFilesPastCap starts on report files past the cap,
// as after the cap is lowered: the oldest are deleted until the rest
// fit.
func TestStartDeletesFilesPastCap(t *testing.T) {
const fileBytes = 100
dir := t.TempDir()
oldest := reportFilePath(dir, 1)
kept := []string{reportFilePath(dir, 2), reportFilePath(dir, 3)}
writeBytes(t, oldest, fileBytes)
for _, path := range kept {
writeBytes(t, path, fileBytes)
}
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(len(kept)*fileBytes))
startBuffer(t)
if exists(t, oldest) {
t.Fatal("oldest report file kept, though the files were past the cap")
}
for _, path := range kept {
if !exists(t, path) {
t.Fatalf("%s deleted, though the rest fit without it", path)
}
if !errors.Is(err, reportbuf.ErrFull) {
t.Fatalf("report past the cap: error = %v, want ErrFull", err)
}
}
@@ -263,9 +195,8 @@ func TestWrittenReportsCountAtFileSize(t *testing.T) {
// Repetitive, so its file is far smaller than its JSON.
report := map[string]string{"id": strings.Repeat("a", 1000)}
size := lineBytes(t, report)
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR", t.TempDir())
// Room for the report twice over only if the first one counts
// at its file's size by the time the second arrives.
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(2*size-1))
@@ -286,22 +217,13 @@ func TestWrittenReportsCountAtFileSize(t *testing.T) {
if err != nil {
t.Fatalf("second report, after the first was written: %v", err)
}
if !hasReportFile(t, dir) {
t.Fatal("first report's file deleted, though the second fit beside it")
}
}
// TestWrittenFilesDeletedToMakeRoom writes one report file after
// another under a small cap. Every report must be taken; files are
// deleted only when the report would not fit beside them, and the
// files kept leave room for it.
func TestWrittenFilesDeletedToMakeRoom(t *testing.T) {
const (
maxBytes = 200
// One file each, which take far more than maxBytes together.
reports = 50
)
// TestWrittenReportsKeepCounting writes one report file after another
// under a small cap: each report must be taken while the files on disk
// leave room for it, and refused once they do not.
func TestWrittenReportsKeepCounting(t *testing.T) {
const maxBytes = 200
report := map[string]string{"id": "written"}
size := int64(lineBytes(t, report))
@@ -312,24 +234,23 @@ func TestWrittenFilesDeletedToMakeRoom(t *testing.T) {
buf := startBuffer(t)
for range reports {
before := reportFilesBytes(t, dir)
// Every file takes at least a byte, so they fill the cap within
// maxBytes rounds.
for range maxBytes {
used := reportFilesBytes(t, dir)
err := buf.Append(report)
if used+size > maxBytes {
if !errors.Is(err, reportbuf.ErrFull) {
t.Fatalf("with %d bytes of report files: error = %v, "+
"want ErrFull", used, err)
}
return
}
if err != nil {
t.Fatalf("with %d bytes of report files: %v", before, err)
}
after := reportFilesBytes(t, dir)
if before+size <= maxBytes && after != before {
t.Fatalf("files deleted, though the report fit beside "+
"their %d bytes", before)
}
if after+size > maxBytes {
t.Fatalf("%d bytes of report files kept, leaving no room "+
"for the report", after)
t.Fatalf("with %d bytes of report files: %v", used, err)
}
err = buf.Flush()
@@ -337,217 +258,8 @@ func TestWrittenFilesDeletedToMakeRoom(t *testing.T) {
t.Fatalf("flush: %v", err)
}
}
}
// TestFailedDeletionStillCounts makes deleting the oldest report file
// fail. It is still there, so it still takes room, and the next oldest
// is deleted in its place.
func TestFailedDeletionStillCounts(t *testing.T) {
const fileBytes = 100
report := map[string]string{"id": "stuck"}
dir := t.TempDir()
stuck := reportFilePath(dir, 1)
next := reportFilePath(dir, 2)
newest := reportFilePath(dir, 3)
for _, path := range []string{stuck, next, newest} {
writeBytes(t, path, fileBytes)
}
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(3*fileBytes))
buf := startBuffer(t)
// A directory that is not empty cannot be deleted, even by root,
// which the tests run as in the backend image.
err := os.Remove(stuck)
if err != nil {
t.Fatalf("remove %s: %v", stuck, err)
}
err = os.Mkdir(stuck, 0o750)
if err != nil {
t.Fatalf("make directory %s: %v", stuck, err)
}
writeBytes(t, filepath.Join(stuck, "file"), 1)
err = buf.Append(report)
if err != nil {
t.Fatalf("report past the cap: %v", err)
}
if exists(t, next) {
t.Fatal("next oldest report file kept: the failed deletion " +
"counted as making room")
}
if !exists(t, newest) {
t.Fatal("newest report file deleted, though deleting one made room")
}
}
// TestFileDeletedByHandFreesRoom deletes the oldest report file by
// hand after start. When room is needed, its room counts as freed, so
// no other file is deleted.
func TestFileDeletedByHandFreesRoom(t *testing.T) {
const fileBytes = 100
report := map[string]string{"id": "by-hand"}
dir := t.TempDir()
gone := reportFilePath(dir, 1)
kept := reportFilePath(dir, 2)
writeBytes(t, gone, fileBytes)
writeBytes(t, kept, fileBytes)
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(2*fileBytes))
buf := startBuffer(t)
err := os.Remove(gone)
if err != nil {
t.Fatalf("remove %s: %v", gone, err)
}
err = buf.Append(report)
if err != nil {
t.Fatalf("report past the cap: %v", err)
}
if !exists(t, kept) {
t.Fatal("report file deleted, though the one deleted by hand " +
"had made room")
}
}
// TestFileBeingWrittenIsNeverDeleted holds the write of one report file
// open while a second write completes, then sends a report that needs
// room. Deleting either file would make it, and the one being written is
// the older, but only the complete one may be deleted. Once the first
// write is complete, its file is deleted when room is needed.
func TestFileBeingWrittenIsNeverDeleted(t *testing.T) {
report := map[string]string{"id": "writing"}
t.Setenv("DATA_DIR", t.TempDir())
// Room for two reports waiting to be written, but not for two
// beside a report file.
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(2*lineBytes(t, report)))
buf := startBuffer(t)
created := make(chan string)
release := make(chan struct{})
buf.OnFileCreated(func(f *os.File) {
created <- f.Name()
<-release
})
err := buf.Append(report)
if err != nil {
t.Fatalf("first report: %v", err)
}
flushed := make(chan error)
go func() { flushed <- buf.Flush() }()
writing := <-created
// Only the first write is held; the second goes through, and
// so do the writes after it, the final one at stop included.
var complete string
buf.OnFileCreated(func(f *os.File) { complete = f.Name() })
// Errorf, not Fatalf, until the first write is released, so that a
// failure here does not leave it held.
err = buf.Append(report)
if err != nil {
t.Errorf("second report: %v", err)
}
err = buf.Flush()
if err != nil {
t.Errorf("flush of the second report: %v", err)
}
err = buf.Append(report)
if err != nil {
t.Errorf("report that needs room: %v", err)
}
if !exists(t, writing) {
t.Error("report file deleted while it was being written")
}
if exists(t, complete) {
t.Error("complete report file kept, though room was needed")
}
close(release)
err = <-flushed
if err != nil {
t.Fatalf("flush of the first report: %v", err)
}
err = buf.Append(report)
if err != nil {
t.Fatalf("report after the first write was complete: %v", err)
}
if exists(t, writing) {
t.Fatal("complete report file kept when room was needed")
}
}
// TestFailedWriteStopsCounting makes a write fail once its file is
// created. Its reports are lost, so they stop counting, and the part of
// the file written is removed, so it takes no room.
func TestFailedWriteStopsCounting(t *testing.T) {
report := map[string]string{"id": "failed"}
t.Setenv("DATA_DIR", t.TempDir())
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(lineBytes(t, report)))
buf := startBuffer(t)
var failed string
// Closing the file under the write makes the write fail.
buf.OnFileCreated(func(f *os.File) {
failed = f.Name()
_ = f.Close()
})
err := buf.Append(report)
if err != nil {
t.Fatalf("report that fills the cap exactly: %v", err)
}
err = buf.Flush()
if err == nil {
t.Fatal("flush succeeded, though its file was closed under it")
}
if exists(t, failed) {
t.Fatal("file of the failed write kept")
}
// Writes from here on, the final one at stop included, succeed.
buf.OnFileCreated(func(*os.File) {})
err = buf.Append(report)
if err != nil {
t.Fatalf("report after the failed write: %v", err)
}
t.Fatal("the report files never filled the cap")
}
// TestConcurrentAppendsStopAtCap appends from many goroutines at once
@@ -622,11 +334,7 @@ func TestTwoFlushesInOneMillisecond(t *testing.T) {
}
}
files, err := readReportFiles(dir)
if err != nil {
t.Fatalf("read report files: %v", err)
}
files := readReportFiles(t, dir)
if len(files) != flushes {
t.Fatalf("%d report files after %d flushes", len(files), flushes)
}
@@ -639,88 +347,6 @@ func TestTwoFlushesInOneMillisecond(t *testing.T) {
}
}
// TestReportFileHoldsTheLinesAppended flushes three reports and reads
// their file back: it must decompress to exactly their JSON lines, in
// the order they were appended.
func TestReportFileHoldsTheLinesAppended(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
buf := startBuffer(t)
for _, id := range []int{1, 2, 3} {
err := buf.Append(map[string]int{"id": id})
if err != nil {
t.Fatalf("append report %d: %v", id, err)
}
}
err := buf.Flush()
if err != nil {
t.Fatalf("flush: %v", err)
}
files, err := readReportFiles(dir)
if err != nil {
t.Fatalf("read report files: %v", err)
}
want := `{"id":1}` + "\n" + `{"id":2}` + "\n" + `{"id":3}` + "\n"
if len(files) != 1 || files[0] != want {
t.Fatalf("report files = %q, want one holding %q", files, want)
}
}
// TestFlushAtSizeThreshold appends reports until the buffer holds
// FlushSizeThreshold bytes. The append that gets it there must write
// them all to one report file, with no call to Flush and the periodic
// flush a minute away, and no earlier append may write one.
func TestFlushAtSizeThreshold(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
buf := startBuffer(t)
// Large reports, so the threshold takes a few hundred appends.
pad := strings.Repeat("a", 64<<10)
var appended strings.Builder
for id := 0; appended.Len() < reportbuf.FlushSizeThreshold; id++ {
report := map[string]any{"id": id, "pad": pad}
err := buf.Append(report)
if err != nil {
t.Fatalf("append report %d: %v", id, err)
}
line, err := json.Marshal(report)
if err != nil {
t.Fatalf("marshal report %d: %v", id, err)
}
appended.Write(line)
appended.WriteByte('\n')
}
// Append writes the file in the background, so wait for it.
deadline := time.Now().Add(10 * time.Second)
for {
files, err := readReportFiles(dir)
if err == nil && len(files) == 1 && files[0] == appended.String() {
return
}
if time.Now().After(deadline) {
t.Fatalf("%d report files (error: %v), want one holding the "+
"%d bytes appended", len(files), err, appended.Len())
}
time.Sleep(10 * time.Millisecond)
}
}
// reportFilesBytes returns the total size of the report files in dir.
func reportFilesBytes(t *testing.T, dir string) int64 {
t.Helper()
@@ -745,19 +371,20 @@ func reportFilesBytes(t *testing.T, dir string) int64 {
}
// readReportFiles returns the decompressed contents of each report
// file in dir. A file still being written does not decompress, so it
// gives an error.
func readReportFiles(dir string) ([]string, error) {
// file in dir.
func readReportFiles(t *testing.T, dir string) []string {
t.Helper()
files := os.DirFS(dir)
names, err := fs.Glob(files, "reports-*.jsonl.zst")
if err != nil {
return nil, fmt.Errorf("list report files: %w", err)
t.Fatalf("list report files: %v", err)
}
dec, err := zstd.NewReader(nil)
if err != nil {
return nil, fmt.Errorf("create zstd decoder: %w", err)
t.Fatalf("create zstd decoder: %v", err)
}
defer dec.Close()
@@ -766,26 +393,18 @@ func readReportFiles(dir string) ([]string, error) {
for _, name := range names {
compressed, readErr := fs.ReadFile(files, name)
if readErr != nil {
return nil, fmt.Errorf("read %s: %w", name, readErr)
t.Fatalf("read %s: %v", name, readErr)
}
data, decErr := dec.DecodeAll(compressed, nil)
if decErr != nil {
return nil, fmt.Errorf("decompress %s: %w", name, decErr)
t.Fatalf("decompress %s: %v", name, decErr)
}
contents = append(contents, string(data))
}
return contents, nil
}
// reportFilePath returns the path in dir of a report file named as
// written on the given day of January 2026, so that a lower day sorts
// as older.
func reportFilePath(dir string, day int) string {
return filepath.Join(dir,
fmt.Sprintf("reports-2026-01-%02dT00-00-00.000Z-1.jsonl.zst", day))
return contents
}
func writeBytes(t *testing.T, path string, n int) {
@@ -797,21 +416,6 @@ func writeBytes(t *testing.T, path string, n int) {
}
}
func exists(t *testing.T, path string) bool {
t.Helper()
_, err := os.Stat(path)
if errors.Is(err, fs.ErrNotExist) {
return false
}
if err != nil {
t.Fatalf("stat %s: %v", path, err)
}
return true
}
func hasReportFile(t *testing.T, dir string) bool {
t.Helper()
+1 -2
View File
@@ -4,7 +4,6 @@ import (
"errors"
"net"
"net/http"
"runtime"
"strconv"
"time"
@@ -54,7 +53,7 @@ func (s *Server) listenAndServe() {
s.log.Info("http begin listen",
"listenaddr", s.httpServer.Addr,
"version", s.params.Globals.Version,
"arch", runtime.GOARCH,
"buildarch", s.params.Globals.Buildarch,
)
err := s.httpServer.ListenAndServe()
+2 -2
View File
@@ -1,6 +1,6 @@
#!/bin/sh
# script/build: compile the static netwatch-server binary into the
# backend project root, with its version stamped in.
# backend project root, with its version and architecture stamped in.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
@@ -14,7 +14,7 @@ main() {
version="${VERSION:-$(git describe --always --dirty 2>/dev/null || echo dev)}"
CGO_ENABLED=0 go build -trimpath \
-ldflags "-s -w -X main.Version=$version" \
-ldflags "-s -w -X main.Version=$version -X main.Buildarch=$(uname -m)" \
-o netwatch-server ./cmd/netwatch-server/
}
+2 -10
View File
@@ -1,20 +1,12 @@
#!/bin/sh
# script/test: run the backend test suite with the race detector and
# coverage. Go's own -timeout bounds the tests and not their compile,
# so a cold build cache cannot fail it. The race detector needs cgo,
# and so a C compiler. If the tests fail, they run again with -v for
# the details, and the script fails even if that run passes.
# script/test: run the backend test suite.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
go test -timeout 30s -race -cover ./... || {
echo "--- Rerunning with -v for details ---"
go test -timeout 30s -race -v ./...
exit 1
}
timeout 30 go test ./...
}
main "$@"
-3
View File
@@ -9,9 +9,6 @@
type="image/svg+xml"
href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'><text y='.9em' font-size='90'>📡</text></svg>"
/>
<!-- Linked here, not imported by src/main.js, so the unit tests can
import that module in Node, which cannot import CSS. -->
<link rel="stylesheet" href="/src/styles.css" />
</head>
<body class="bg-gray-900 text-white min-h-screen">
<div id="app"></div>
+3 -10
View File
@@ -64,8 +64,7 @@ detect_pkgmgr() {
fi
}
# pkg_install <nix-attr> <apt-pkgs> <brew-formula> <apk-pkgs>: the apt
# and apk arguments may each list several packages, separated by spaces.
# pkg_install <nix-attr> <apt-pkg> <brew-formula> <apk-pkg>
pkg_install() {
detect_pkgmgr
case "$PKGMGR" in
@@ -75,10 +74,10 @@ pkg_install() {
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get update
APT_UPDATED=1
fi
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get install -y $2
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get install -y "$2"
;;
brew) brew install "$3" ;;
apk) apk add --no-cache $4 ;;
apk) apk add --no-cache "$4" ;;
esac
}
@@ -254,12 +253,6 @@ main() {
if missing make; then pkg_install gnumake make make make; fi
if missing git; then pkg_install git git git git; fi
# The race detector in make test needs cgo, which Go turns on only
# when it finds its C compiler, gcc on Linux. apt and apk ship the C
# library headers apart from gcc.
if missing gcc; then
pkg_install gcc "gcc libc6-dev" gcc "gcc musl-dev"
fi
ensure_node
ensure_yarn
+3 -2
View File
@@ -16,8 +16,9 @@ main() {
"$SCRIPT_DIR/check"
# Own line: a failing command substitution inside an argument does
# not trip `set -e`, so the inline form degrades silently to an
# empty constant. The VERSION build argument takes precedence over
# the version a build stage derives from the .git in the context.
# empty constant. VERSION is computed here because .dockerignore
# excludes .git, so `git describe` in a build stage yields an empty
# version without failing.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
+3 -2
View File
@@ -12,8 +12,9 @@ main() {
cd "$ROOT"
# Own line: a failing command substitution inside an argument does
# not trip `set -e`, so the inline form degrades silently to an
# empty constant. The VERSION build argument takes precedence over
# the version a build stage derives from the .git in the context.
# empty constant. VERSION is computed here because .dockerignore
# excludes .git, so `git describe` in a build stage yields an empty
# version without failing.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
+3 -4
View File
@@ -1,14 +1,13 @@
#!/bin/sh
# script/frontend-test: run the frontend test suite: the unit tests in
# test/unit/ with Node's built-in test runner, then the production
# build, which fails on broken code.
# script/frontend-test: run the frontend test suite. The frontend has no
# unit tests; the production build serves as the test (fails on broken
# code).
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
timeout 30 node --test test/unit/*.test.js
timeout 30 yarn build
}
+3 -6
View File
@@ -1,17 +1,14 @@
#!/bin/sh
# script/test: run the test suite for the whole repo: the frontend at
# the repo root, then the Go backend in backend/. Each half has its own
# 30-second limit, and there is none around both: from a cold Go build
# cache, compiling the backend's tests with the race detector can take
# 30 seconds on its own, and Go's -timeout leaves the compile out.
# the repo root, then the Go backend in backend/. Both halves together
# get 30 seconds; each also keeps its own limit for the Dockerfiles.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
script/frontend-test
backend/script/test
timeout 30 sh -c 'script/frontend-test && backend/script/test'
}
main "$@"
+26 -51
View File
@@ -1,14 +1,14 @@
import "./styles.css";
// --- Configuration -----------------------------------------------------------
// Timing, axis labels, and display constants. A target check times out
// after requestTimeout, 80% of updateInterval, so a round's checks have
// all finished before the next round is due; latency above maxLatency is
// recorded as a timeout. The sparkline Y-axis is capped at
// Timing, axis labels, and display constants. Latency above maxLatency is
// clamped to "unreachable". The sparkline Y-axis is capped at
// graphMaxLatency — values above it pin to the top of the chart but still
// display their real value in the latency figure. The history buffer holds
// maxHistoryPoints samples (historyDuration / updateInterval).
// reportInterval is how often collected samples are POSTed to the backend.
export const CONFIG = {
const CONFIG = {
updateInterval: 3000,
maxHistoryPoints: 100,
reportInterval: 60000,
@@ -16,7 +16,7 @@ export const CONFIG = {
return (this.maxHistoryPoints * this.updateInterval) / 1000;
},
get requestTimeout() {
return this.updateInterval * 0.8;
return Math.min(this.updateInterval - 100, 3000);
},
get maxLatency() {
return this.requestTimeout;
@@ -503,16 +503,12 @@ class Reporter {
// --- Latency Measurement -----------------------------------------------------
// Checks one target. The check times out after CONFIG.requestTimeout; the
// caller can give it up sooner through the optional signal, which also ends
// it as a timeout.
export async function measureLatency(url, signal) {
async function measureLatency(url) {
const controller = new AbortController();
const timeoutId = setTimeout(
() => controller.abort(),
CONFIG.requestTimeout,
);
signal?.addEventListener("abort", () => controller.abort());
const targetUrl = new URL(url);
targetUrl.searchParams.set("_cb", Date.now().toString());
@@ -1105,7 +1101,7 @@ function sortAndRebuildWAN(state) {
// --- Main Loop ---------------------------------------------------------------
async function tick(state, signal, onOffline) {
async function tick(state, onOffline) {
const ts = Date.now();
if (state.paused) {
@@ -1127,12 +1123,11 @@ async function tick(state, signal, onOffline) {
log.debug(`Tick #${state.tickCount + 1} started`);
const results = await Promise.all(
state.allHosts.map((h) => measureLatency(h.url, signal)),
state.allHosts.map((h) => measureLatency(h.url)),
);
// User may have paused, or the next round may have given up this
// one's checks, while awaiting results — discard them
if (state.paused || signal.aborted) return;
// User may have paused while awaiting results — discard them
if (state.paused) return;
state.tickCount++;
@@ -1173,10 +1168,9 @@ async function tick(state, signal, onOffline) {
// --- Recovery Probe ----------------------------------------------------------
// When offline, check 4 random WAN hosts every 500ms, giving up the checks
// started 500ms before, so at most 4 are ever waiting. As soon as one
// answers, stop probing and start a new round at once.
function startRecoveryProbe(state, startRounds) {
// When offline, rapidly poll 4 random WAN hosts every 500ms. As soon as any
// responds, stop probing and fire a normal tick to refresh all hosts.
function startRecoveryProbe(state, triggerTick) {
if (state._recoveryProbeId) return; // already running
const candidates = [...state.wan];
for (let i = candidates.length - 1; i > 0; i--) {
@@ -1187,18 +1181,15 @@ function startRecoveryProbe(state, startRounds) {
log.notice(
`Recovery probe started (${canaries.map((h) => h.name).join(", ")})`,
);
state._recoveryProbeId = setInterval(() => {
state._recoveryProbeId = setInterval(async () => {
if (state.paused) return;
state._recoveryProbeChecks?.abort();
const checks = new AbortController();
state._recoveryProbeChecks = checks;
for (const host of canaries) {
measureLatency(host.url, checks.signal).then((r) => {
if (r.error !== null || checks.signal.aborted) return;
const results = await Promise.all(
canaries.map((h) => measureLatency(h.url)),
);
if (results.some((r) => r.error === null)) {
log.notice("Recovery probe: connectivity detected");
stopRecoveryProbe(state);
startRounds();
});
triggerTick();
}
}, 500);
}
@@ -1207,7 +1198,6 @@ function stopRecoveryProbe(state) {
if (state._recoveryProbeId) {
clearInterval(state._recoveryProbeId);
state._recoveryProbeId = null;
state._recoveryProbeChecks?.abort();
}
}
@@ -1384,34 +1374,18 @@ async function init() {
updateClocks();
setInterval(updateClocks, 1000);
// Rounds never overlap: a round first gives up the last round's checks
// if they are still waiting, and the last round then records nothing.
// At a steady interval they never are, as they time out at 80% of it;
// they can be when a round starts early, after an interval change or
// when the recovery probe finds a target answering.
let roundChecks = new AbortController();
function doTick() {
roundChecks.abort();
roundChecks = new AbortController();
tick(state, roundChecks.signal, () =>
startRecoveryProbe(state, startRounds),
);
tick(state, () => startRecoveryProbe(state, doTick));
}
// Starts a round now and then one every CONFIG.updateInterval.
let tickIntervalId;
function startRounds() {
clearInterval(tickIntervalId);
doTick();
tickIntervalId = setInterval(doTick, CONFIG.updateInterval);
}
startRounds();
let tickIntervalId = setInterval(doTick, CONFIG.updateInterval);
document
.getElementById("interval-select")
.addEventListener("change", (e) => {
const newInterval = parseInt(e.target.value, 10);
clearInterval(tickIntervalId);
CONFIG.updateInterval = newInterval;
log.notice(
`Interval changed to ${humanDuration(newInterval / 1000)}, history reset`,
@@ -1460,7 +1434,8 @@ async function init() {
// Start immediately with new interval
stopRecoveryProbe(state);
startRounds();
doTick();
tickIntervalId = setInterval(doTick, CONFIG.updateInterval);
});
window.addEventListener("resize", () => handleResize(state));
@@ -1469,7 +1444,7 @@ async function init() {
// Bootstrap only when loaded as the page: a real DOM containing the #app
// mount point this module renders into. Importing the module in a unit test
// (which has no #app) runs nothing, so its exports can be tested in isolation.
// (which has no #app) runs nothing, so buildReport can be tested in isolation.
if (typeof document !== "undefined" && document.getElementById("app")) {
if (document.readyState === "loading") {
document.addEventListener("DOMContentLoaded", init);
-66
View File
@@ -1,66 +0,0 @@
// Unit tests for src/main.js, run by script/frontend-test with Node's
// built-in test runner. Importing the module does not start the page.
import { test } from "node:test";
import assert from "node:assert/strict";
import { CONFIG, measureLatency } from "../../src/main.js";
// measureLatency writes timeouts to the debug log, which looks for its
// panel in the page. There is no page here.
globalThis.document = { getElementById: () => null };
// Mocks the clock for test t, so that a check lasting seconds takes no real
// time, and replaces fetch with a target that answers after answerAfter
// milliseconds of that clock, or never when answerAfter is Infinity. Both
// are restored when the test ends.
function mockTarget(t, answerAfter) {
t.mock.timers.enable({ apis: ["setTimeout", "Date"] });
t.mock.method(performance, "now", () => Date.now());
t.mock.method(
globalThis,
"fetch",
(url, { signal }) =>
new Promise((resolve, reject) => {
if (answerAfter !== Infinity) setTimeout(resolve, answerAfter);
signal.addEventListener("abort", () => reject(signal.reason));
}),
);
}
// The result of check if it has ended, otherwise "still waiting".
function settled(check) {
return Promise.race([
check,
new Promise((resolve) => setImmediate(resolve, "still waiting")),
]);
}
for (const interval of [10000, 30000]) {
const timeout = interval * 0.8;
// Over 3 seconds, which the timeout used to be capped at.
const slowAnswer = timeout - 1000;
test(`at a ${interval}ms interval, an answer after ${slowAnswer}ms is recorded with its real time`, async (t) => {
CONFIG.updateInterval = interval;
mockTarget(t, slowAnswer);
const check = measureLatency("https://target.test");
t.mock.timers.tick(slowAnswer);
assert.deepEqual(await settled(check), {
latency: slowAnswer,
error: null,
});
});
test(`at a ${interval}ms interval, a target that never answers is recorded as a timeout after ${timeout}ms`, async (t) => {
CONFIG.updateInterval = interval;
mockTarget(t, Infinity);
const check = measureLatency("https://target.test");
t.mock.timers.tick(timeout - 1);
assert.equal(await settled(check), "still waiting");
t.mock.timers.tick(1);
assert.deepEqual(await settled(check), {
latency: null,
error: "timeout",
});
});
}