Author SHA1 Message Date
clawbot 4ecdd8027b feat: add mobile detection with friendly unsupported message
check / check (push) Successful in 28s
Detect mobile viewport (innerWidth < 768) at init and show a centered
'Not yet available on mobile.' box instead of starting the full monitor.
No polling, no host rows, no sparklines on mobile.

Desktop behavior is completely unchanged.

Ref #2, ref #4
2026-02-27 02:02:04 -08:00
89 changed files with 827 additions and 9789 deletions
-4
View File
@@ -1,9 +1,5 @@
node_modules
dist
tmp
.DS_Store
*.log
.claude
# .git is sent so the build can stamp the version, without its config.
.git/config
+2 -4
View File
@@ -6,7 +6,5 @@ jobs:
steps:
# actions/checkout v4.2.2, 2026-02-22
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
# script/cibuild bootstraps, runs every check and builds the
# image. script/bootstrap links what it installs into
# ~/.local/bin, so that has to be on PATH for the rest.
- run: PATH="$HOME/.local/bin:$PATH" script/cibuild
- run: docker build .
- run: docker build -f Dockerfile.backend .
+1 -25
View File
@@ -1,28 +1,4 @@
# OS
.DS_Store
Thumbs.db
# Editors
*.swp
*.swo
*~
*.bak
.idea/
.vscode/
*.sublime-*
# Node
node_modules/
# Environment / secrets
.env
.env.*
*.pem
*.key
# Build output
dist/
tmp/
# Logs
.DS_Store
*.log
+1 -3
View File
@@ -1,7 +1,5 @@
backend/
dist/
node_modules/
tmp/
yarn.lock
.claude/
# The org standard file, copied verbatim; backend/script/lint checks its sha256.
backend/.golangci.yml
+6 -117
View File
@@ -1,129 +1,18 @@
# The one image netwatch ships: nginx serves the built frontend and
# passes /api/, /.well-known/healthcheck and /metrics to netwatch-server,
# the Go backend, which runs in the same container on loopback only.
# bin/entrypoint.sh starts and watches both.
# Lint stage — fast feedback on formatting and lint issues. The
# golangci/golangci-lint image ships Go, gofmt, make and the linter, so
# nothing is installed here. The root make lint builds this stage alone.
# golangci/golangci-lint:v2.12.2 (2026-08-10)
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
WORKDIR /src
COPY backend/go.mod backend/go.sum ./
RUN go mod download
COPY backend/ .
RUN make fmt-check
RUN make lint
# Backend build stage
# golang:1.25-alpine (2026-02-27)
FROM golang:1.25-alpine@sha256:f6751d823c26342f9506c03797d2527668d095b0a15f1862cddb4d927a7a4ced AS builder
# gcc and musl-dev are for make test: its race detector needs cgo, which
# Go turns on by itself once a C compiler is present. make build still
# sets CGO_ENABLED=0, so the binary stays static.
RUN apk add --no-cache gcc git make musl-dev
WORKDIR /src
# Force BuildKit to run the lint stage before proceeding. BuildKit runs
# stages in parallel by default; without this no-op copy a lint failure
# would not gate compilation.
COPY --from=lint /src/go.sum /dev/null
COPY backend/go.mod backend/go.sum ./
RUN go mod download
COPY backend/ .
RUN make test
# make build is a shim around backend/script/build, the one definition
# of the build command:
# CGO_ENABLED=0 go build -trimpath -ldflags "-s -w -X main.Version=..."
# That script reads VERSION from the environment, so it is handed over
# there rather than as a make variable.
#
# The version is the VERSION build argument when one is given, otherwise
# `git describe --tags --always` of the repo's .git: the tag on a tagged
# commit, tag-N-gHASH on a commit after one, the short commit when no
# tag is reachable. A version that still comes out empty, dev or unknown
# fails the build. .git goes to /git, not /src/.git, where go build would
# find it and record VCS details of a work tree holding only backend/.
COPY .git /git
ARG VERSION
RUN version="${VERSION:-$(git --git-dir=/git describe --tags --always)}"; \
case "$version" in ""|dev|unknown) \
echo "version is '$version' although .git is present" >&2; \
exit 1 ;; \
esac; \
VERSION="$version" make build
# Frontend lint stage — eslint over the JavaScript, as the lint stage
# above lints the Go. The root make lint builds this stage alone too.
# node:22-alpine as of 2026-02-22
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS frontend-lint
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS build
WORKDIR /app
COPY package.json yarn.lock ./
RUN yarn install --frozen-lockfile
RUN apk add --no-cache git
COPY . .
RUN script/frontend-lint
RUN yarn build
# Frontend stage
# node:22-alpine as of 2026-02-22
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS frontend
WORKDIR /app
# Force BuildKit to run the frontend-lint stage before proceeding, as
# the builder stage does with the lint stage: without this no-op copy an
# eslint failure would not gate the image.
COPY --from=frontend-lint /app/yarn.lock /dev/null
COPY package.json yarn.lock ./
RUN yarn install --frozen-lockfile
RUN apk add --no-cache git make
COPY . .
# make frontend-check runs the frontend tests and format check; its test
# step runs the unit tests, then the production yarn build, so this both
# produces dist/ and gates the image on test and formatting regressions.
# This node stage has neither Go nor Docker; the frontend-lint, lint and
# builder stages above gate the rest.
RUN make frontend-check
# Runtime stage
# nginx:stable-alpine as of 2026-02-22
FROM nginx@sha256:15e96e59aa3b0aada3a121296e3bce117721f42d88f5f64217ef4b18f458c6ab
# netwatch-server runs as this user, which owns the report directory.
# nginx keeps the image's own arrangement: its main process runs as
# root, its worker processes as the nginx user.
RUN addgroup -g 1000 -S netwatch && \
adduser -u 1000 -S netwatch -G netwatch
# At start-up the nginx image renders every template here into
# conf.d; bin/entrypoint.sh says how.
RUN rm /etc/nginx/conf.d/default.conf
COPY nginx.conf /etc/nginx/templates/netwatch.conf.template
COPY security-headers.conf /etc/nginx/security-headers.conf
COPY --from=frontend /app/dist /usr/share/nginx/html
COPY --from=builder /src/netwatch-server /usr/local/bin/netwatch-server
COPY bin/entrypoint.sh /usr/local/bin/entrypoint.sh
COPY nginx.conf /etc/nginx/conf.d/netwatch.conf
COPY --from=build /app/dist /usr/share/nginx/html
# bin/entrypoint.sh creates DATA_DIR at start and gives it and /data to
# the netwatch user, whatever is mounted there.
ENV DATA_DIR=/data/reports
VOLUME /data
# The default public port; PORT changes it.
EXPOSE 8080
# Requests the backend's health check through nginx, on the port from
# PORT, so it fails unless both answer. upaas reads the result 60
# seconds after a deploy and fails the deploy unless it is healthy.
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
CMD wget -q -O /dev/null "http://127.0.0.1:${PORT:-8080}/.well-known/healthcheck"
# The nginx image stops its container with SIGQUIT; the entrypoint
# acts on TERM and INT.
STOPSIGNAL SIGTERM
ENTRYPOINT ["/usr/local/bin/entrypoint.sh"]
CMD ["nginx", "-g", "daemon off;"]
+25
View File
@@ -0,0 +1,25 @@
# golang:1.25-alpine (2026-02-27)
FROM golang:1.25-alpine@sha256:f6751d823c26342f9506c03797d2527668d095b0a15f1862cddb4d927a7a4ced AS builder
RUN apk add --no-cache git make gcc musl-dev
# golangci-lint v2.7.2 (2026-02-27)
RUN CGO_ENABLED=0 go install github.com/golangci/golangci-lint/v2/cmd/golangci-lint@9f61b0f53f80672872fced07b6874397c3ed197b
WORKDIR /repo/backend
COPY backend/go.mod backend/go.sum ./
RUN go mod download
COPY .git /repo/.git
COPY backend/ .
RUN make check
RUN make build
# alpine:3.23 (2026-02-27)
FROM alpine:3.23@sha256:25109184c71bdad752c8312a8623239686a9a2071e8825f20acb8f2198c3f659
RUN apk add --no-cache ca-certificates
COPY --from=builder /repo/backend/netwatch-server /usr/local/bin/netwatch-server
EXPOSE 8080
ENTRYPOINT ["netwatch-server"]
+8 -47
View File
@@ -1,60 +1,21 @@
.PHONY: bootstrap setup dev build test lint fmt fmt-check check \
add-dependency tidy frontend-check frontend-viewport-test docker hooks
# Standard targets are thin shims; the implementations live in script/
# per the scripts-to-rule-them-all pattern (see the Entrypoints section
# of README.md). test, lint, fmt, fmt-check and check cover the whole
# repo: the frontend here and the Go backend in backend/.
bootstrap:
@script/bootstrap
setup:
@script/setup
.PHONY: dev test lint fmt fmt-check check docker
dev:
@script/dev
# The frontend only; backend/Makefile's build target builds the Go server.
build:
@script/build
yarn dev
test:
@script/test
timeout 30 yarn build
lint:
@script/lint
yarn prettier --check .
fmt:
@script/fmt
yarn prettier --write .
fmt-check:
@script/fmt-check
yarn prettier --check .
check:
@script/check
# make add-dependency PACKAGE=<name>@<version>. PACKAGE reaches the
# script through the environment, so the shell never reads it as code.
add-dependency:
@script/add-dependency "$$PACKAGE"
tidy:
@script/tidy
# The frontend tests and format check, for Dockerfile's frontend stage,
# which has neither Go nor Docker. Use check everywhere else.
frontend-check:
@script/frontend-check
# Responsive-layout verification in a containerised browser. Kept out of
# check: it needs Docker and takes minutes, where make test has to stay
# under 20 seconds.
frontend-viewport-test:
@script/frontend-viewport-test
check: test lint fmt-check
docker:
@script/docker
hooks:
@script/install-precommit
timeout 300 docker build -t netwatch .
+47 -251
View File
@@ -1,126 +1,42 @@
NetWatch is an MIT-licensed JavaScript single-page application by
[@sneak](https://sneak.berlin) that provides real-time network latency
monitoring to common internet hosts, displayed with color-coded figures and
sparkline graphs, served from a static bucket or from its Docker image, where a
small Go backend stores the measurements the page reports.
sparkline graphs, served from a static bucket or Docker container.
## Getting Started
```bash
# Install the dependencies and the git pre-commit hook
make setup
# Install dependencies
yarn install
# Run the page on the Vite dev server
make dev
# Development server
yarn dev
# Run the tests, both linters and the format check
make check
# Production build
yarn build
# Build the page into dist/
make build
# Preview production build
yarn preview
# Build the image and run it
make docker
# Docker
docker build -t netwatch .
docker run -p 8080:8080 netwatch
```
`make check` and `make docker` need Docker. `make dev` passes `/api` to
`http://127.0.0.1:8080`, where `make run` in `backend/` starts `netwatch-server`
with its defaults, so the reports the page posts are stored in
`backend/data/reports`.
## Entrypoints
This repository adheres to the
[Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all)
standard: normalized scripts in `script/` are the entrypoints for the
development workflow, and the Makefile targets are thin shims that call them.
The Go backend in `backend/` has its own `script/` directory and shim Makefile
(see [backend/README.md](backend/README.md)). The root scripts cover both
halves, so the root `make check` fails if either one is broken. We provide:
- `script/bootstrap` — install all dependencies (the pinned node via nvm unless
one new enough for the frontend's dependencies is installed, yarn via
corepack, `yarn install --frozen-lockfile`, the pinned Go unless one at least
as new as `backend/go.mod` asks for is installed, the Go modules, and gcc with
the C library headers unless gcc is installed, for the race detector in
`make test`), linking what it installs itself into `~/.local/bin`, which has
to be on `PATH`. It installs no Go linter and not Docker: `make lint` runs
both linters in Docker
- `script/setup` — make a fresh clone ready for development: bootstrap plus the
git pre-commit hook
- `script/dev` — run the Vite dev server, which proxies `/api` to a locally
running `netwatch-server`
- `script/build` — build the frontend for production into `dist/`;
`backend/script/build` builds the Go server
- `script/projectname` — print the project name (used for the Docker image tag)
- `script/test` — run `script/frontend-test`, then `backend/script/test`, the
backend's Go tests with the race detector and coverage
- `script/lint` — run eslint, then golangci-lint, both in Docker, by building
the `frontend-lint` and `lint` stages of `Dockerfile` without the cache
- `script/fmt` — format all files (writes): prettier, then gofmt over `backend/`
- `script/fmt-check` — check formatting (read-only): prettier, then gofmt
- `script/check` — run test, lint, and fmt-check
- `script/add-dependency` — add a frontend package, or move one to another
version: `make add-dependency PACKAGE=<name>@<version>` runs `yarn add --dev`,
which changes `package.json` and `yarn.lock` together, then
`yarn install --frozen-lockfile`
- `script/tidy` — run `go mod tidy` in `backend/`: to add a Go module, import it
and run `make tidy`; to move one to another version, edit its `require` line
in `backend/go.mod`, then run `make tidy`
- `script/frontend-test` — run the unit tests in `test/unit/` with Node's
built-in test runner, through the `test` script in `package.json`, and if any
fails, run them again listing every test, and fail; then the production build.
Each run has a 30-second timeout
- `script/frontend-lint` — run eslint with the rules in `eslint.config.js`; it
runs inside the `frontend-lint` stage of `Dockerfile`, which `make lint`
builds
- `script/frontend-fmt` — format everything prettier understands (writes), the
markdown in `backend/` included
- `script/frontend-fmt-check` — check prettier formatting (read-only)
- `script/frontend-check` — run `script/frontend-test` and
`script/frontend-fmt-check`, for the frontend stage of `Dockerfile`, which has
neither Go nor Docker
- `script/frontend-viewport-test` — responsive-layout verification of the built
frontend in a containerised headless Chrome (see
[test/viewport/README.md](test/viewport/README.md)). Not part of
`script/check`: it needs Docker and takes minutes.
- `script/docker` — build the image from `Dockerfile` without the build cache,
tagged `netwatch` via `script/projectname`
- `script/cibuild` — CI entrypoint: runs `script/bootstrap` and `script/check`,
then builds the image as `script/docker` does, without the build cache
- `script/precommit` — run by the git pre-commit hook; runs `script/check`
- `script/install-precommit` — install the git pre-commit hook
## Responsive layout
The narrow-viewport layout lives in the `max-width: 768px` media block in
`src/styles.css`. It is verified automatically by `make frontend-viewport-test`,
which drives a digest-pinned headless Chrome against the built `dist/` and
asserts on computed layout at widths derived from that CSS — on every breakpoint
it declares and one pixel either side of it, plus a 320px floor, a desktop
baseline and two landscape sizes. See
[test/viewport/README.md](test/viewport/README.md) for what it covers and what
it genuinely cannot.
## Rationale
When debugging network issues, it's useful to have a persistent at-a-glance view
of latency and reachability to multiple well-known internet endpoints. NetWatch
provides this as a single page that does all its measuring in the browser, so it
can be served from anywhere static files are served. The backend in its Docker
image only stores the measurements the page reports; without it, the page works
the same and nothing is stored.
provides this as a zero-dependency SPA that can be deployed anywhere static
files are served, with no backend required.
## Design
The page is built with Vite and Tailwind CSS v4. Its code is all in
`src/main.js`, with a class-based architecture:
The application is a single-page app built with Vite and Tailwind CSS v4. All
code lives in `src/main.js` with a class-based architecture:
- **`CONFIG`**: Configuration object (update interval, timeouts, axis ticks,
etc.). The interval menu sets `updateInterval`, the one value the page writes
into `CONFIG`; the timeouts, the time the history spans and the x-axis ticks
are computed from it
- **`CONFIG`**: Frozen configuration object (update interval, timeouts, axis
ticks, etc.)
- **`HostState`**: Per-host state management — history buffer, latency tracking,
status transitions
- **`AppState`**: Top-level state container — WAN hosts, local hosts, pause
@@ -129,54 +45,16 @@ The page is built with Vite and Tailwind CSS v4. Its code is all in
color-coded line segments, error regions, and DPR-aware scaling
- **UI functions**: `buildUI()` constructs the DOM, `updateHostRow()` /
`updateSummary()` / `updateHealthBox()` handle incremental updates
- **`tick()`**: Main loop — measures all hosts in parallel, pushing each host's
sample and redrawing its row as soon as its check ends, then redraws every
row, the summary and the health box once the last check ends. The first round,
after loading or an interval change, is discarded. The rows are sorted when
the last check ends in round 2, the first one kept, and in rounds 11, 21, 31
and so on. When paused, pushes blank markers (no probes, no false outage)
- **`Reporter`**: Posts collected samples to the backend
### Reporting
Every `reportInterval` (default 60s) the page POSTs a JSON report to the
same-origin path `/api/v1/reports`: a random per-browser `clientId` kept in
`localStorage`, `geo` sent as null, and each host's unreported, non-paused
samples (timestamp, latency, error). A per-host high-water mark makes every
report a delta, so only new samples are sent; the mark advances only on a
delivered report, and while paused nothing is sent. Delivery failure is quiet —
one debug-log line per outage, retried at the next interval, never blocking
probing. The report-building step is a pure function of host state.
### Backend
`netwatch-server`, in `backend/`, is a small Go HTTP server that stores the
reports the page posts. It keeps them in memory and writes them to `DATA_DIR` as
zstd-compressed files of JSON lines: every minute, whenever 10 MiB are waiting,
and when it stops. Its routes:
- `POST /api/v1/reports` — takes a report, without credentials; each client
address may send a limited number a minute, and the report files are capped in
size, the oldest deleted first
- `GET /.well-known/healthcheck` — answers 200 with `"status":"ok"`, the
server's version and its uptime
- `GET /metrics` — Prometheus metrics behind basic auth, only when
`METRICS_USERNAME` and `METRICS_PASSWORD` are set; each client address may
make a limited number of requests to it a minute
In the image, the `builder` stage of `Dockerfile` tests it and builds it with
`backend/script/build`, and `bin/entrypoint.sh` runs it as user `netwatch` on
`127.0.0.1:8081`, behind nginx. Outside the image, `make run` in `backend/`
builds it and runs it on port 8080. Its settings, report storage and limits are
in [backend/README.md](backend/README.md).
- **`tick()`**: Main loop — measures all hosts in parallel via `Promise.all`,
pushes samples, redraws UI. When paused, pushes blank markers (no probes, no
false outage)
### Monitoring targets
- **26 WAN hosts**: datavi.be (pinned at start), Anthropic API, OpenAI API, AWS
Console, Google Cloud Console, Microsoft Azure, Cloudflare, Fastly CDN,
Akamai, Google, GitHub, B2, 8 S3 regional endpoints (Cape Town, London,
Bahrain, Tokyo, Singapore, Sydney, Oregon, São Paulo) and 6 Hetzner speed test
servers (Nuremberg, Falkenstein, Helsinki, Ashburn, Hillsboro, Singapore)
- **22 WAN hosts**: datavi.be, Anthropic API, OpenAI API, AWS Console, GCP
Console, Azure, Cloudflare, Fastly, Akamai, GitHub, B2, 7 S3 regional
endpoints (Cape Town, London, Bahrain, Tokyo, Sydney, Oregon, São Paulo), 4
GCS locational endpoints (Iowa, Belgium, Singapore, Sydney)
- **Local CPE**: Cable modem at 192.168.100.1 (always monitored)
- **Local Gateway**: Auto-detected on startup by probing common default gateway
addresses (192.168.1.1, 192.168.0.1, 192.168.8.1, 10.0.0.1); first responder
@@ -188,17 +66,9 @@ Local hosts are tracked separately from WAN stats.
### Latency measurement
GET requests with `mode: 'no-cors'`, `cache: 'no-store'` and a cache-busting
query parameter, timed with `performance.now()`. Each check times out after 80%
of the refresh interval (24 seconds at 30 seconds) and is then recorded as a
timeout, so a round's checks have all finished before the next round is due.
When no WAN host answers, a recovery probe checks 4 WAN hosts, picked at random
when it starts, every half second, giving up the checks it started half a second
before. As soon as one answers, a new round starts at once, as it does after an
interval change. A round started early gives up the last round's checks if they
are still waiting, and that round records nothing more, so rounds never overlap.
The browser chooses between IPv4 and IPv6 for each target, as for any request;
the local targets are IPv4 addresses.
HEAD requests with `mode: 'no-cors'` and `cache: 'no-store'`, timed with
`performance.now()`. 1-second timeout; anything over 1000ms is clamped to
unreachable. IPv4 only.
### Color coding
@@ -223,106 +93,30 @@ dist/
## Features
- A round of checks every 3 seconds by default; the interval menu sets 1, 2, 3,
5, 10, 15, 30 or 60 seconds and clears the history
- Sparklines of each target's last 100 rounds: 300 seconds at 3 seconds
- The first round after loading or an interval change is discarded, as DNS and
TLS setup inflate its latencies
- Health indicator from the WAN hosts' latest results: OFFLINE (red) when more
than 10 fail and at most 4 answer, otherwise DEGRADED (orange) when more than
4 fail, otherwise SLOW (yellow) when more than 3 take over 1000ms, otherwise
HEALTHY (green)
- Summary stats across WAN hosts only: how many answered, the min, median,
average and max of their latest latencies, the min and max over the whole
history, and the number of rounds run (`Checks`)
- Fixed chart axes: Y-axis 0–1000ms, higher latencies drawn at the top; X-axis
the time the history spans
- Real-time monitoring with 2s update interval and 300s history sparklines
- Health indicator: green (HEALTHY) or red (DEGRADED) based on WAN reachability
- Summary stats: reachable count, min/max/avg latency across WAN hosts only
- Fixed chart axes: Y-axis 0–1000ms, X-axis 0–300s
- Color-coded latency figures and sparkline line segments
- WAN host rows sorted by latest latency, unreachable last; pinned rows stay on
top, in name order
- Play/pause: pause stops probes but history keeps scrolling (blank gaps, no
false outage)
- Debug log panel, behind a checkbox in the footer, with five levels (error,
warning, notice, info, debug) and the last 1000 lines
- Local and UTC clocks
- Clickable service URLs
- A footer link to the commit the page was built from
- Canvas-based sparkline rendering with devicePixelRatio scaling
- Zero runtime dependencies: all resources bundled into build artifacts
## Deployment
`make build` writes the page to `dist/`, which any static file host (S3, GCS,
Cloudflare Pages, Vercel, Netlify, GitHub Pages) can serve; with no backend
there, its reports fail quietly and nothing is stored. Or run the Docker image
behind a reverse proxy.
After running `yarn build`, deploy the contents of the `dist/` directory to any
static file host (S3, GCS, Cloudflare Pages, Vercel, Netlify, GitHub Pages) or
use the Docker image behind a reverse proxy.
The Docker image, built from `Dockerfile` by `make docker`, is the whole service
in one container: nginx serves the built frontend and passes `/api/`,
`/.well-known/healthcheck` and `/metrics` to the Go backend, `netwatch-server`,
which listens only inside the container, on `127.0.0.1:8081`. The image:
The Docker image:
- Listens on port 8080 by default (override with `PORT` env var)
- Takes the client address from `X-Forwarded-For` only on requests from the
reverse proxies named in `TRUSTED_PROXIES`, and by default from none
- Trusts `X-Forwarded-For` from RFC1918 reverse proxies (10/8, 172.16/12,
192.168/16)
- Sends access logs to stdout
- Caches static assets with immutable headers
- Sends the security headers `REPO_POLICIES.md` requires on every response, as
`security-headers.conf` sets them, in place of the backend's own
- Stores reports in `DATA_DIR`, `/data/reports` by default, on the `/data`
volume. Before the backend starts, the image creates `DATA_DIR` and gives it
and `/data` to user `netwatch` (uid 1000), which the backend runs as, so a
host directory bind-mounted at `/data` ends up owned by uid 1000
- Writes buffered reports to disk on `docker stop`, and exits non-zero if nginx
or the backend exits on its own, so the platform restarts it
## Running under upaas
What the [upaas](https://git.eeqj.de/sneak/upaas) app for netwatch needs:
- **Port:** container port `8080`.
- **Volume:** container path `/data`; the reports are kept in `/data/reports`.
- **Environment variables:** none is required. An empty one counts as unset, and
one set to a value netwatch cannot use stops the container at start, with the
reason in its log.
- `PORT`, default `8080`: the container port, from 1 to 65535. `8081` cannot
be used: the backend listens on it inside the container
- `REPORTS_PER_MINUTE`, default `60`: reports each client address may send a
minute
- `DATA_DIR_MAX_BYTES`, default `1073741824` (1 GiB): the most room the
report files may take; the oldest are deleted to stay under it
- `CORS_ALLOWED_ORIGINS`, default empty: other origins whose pages may call
the API
- `DEBUG`, default `false`: debug logging
- `DATA_DIR`, default `/data/reports`: the directory the reports are kept
in: `/data` or a path below it, with no `.` or `..` part and no extra `/`.
The container also stops if the path goes through a symbolic link that
leads out of `/data` or is written as a full path
- `TRUSTED_PROXIES`, default empty: set it to the address the reverse proxy
in front of the container connects from, as an IP address or CIDR; several
are separated by commas. nginx takes the client address from
`X-Forwarded-For` only on a request from one of them, and the rate limit
counts that address. Unset, `X-Forwarded-For` is ignored and every client
behind the proxy shares the proxy's one allowance of `REPORTS_PER_MINUTE`.
Name only addresses nothing but the proxy connects from: any client that
connects from one can write its own `X-Forwarded-For`, and through a port
Docker publishes, every client may connect from the Docker network's
gateway, such as `172.17.0.1`.
- `METRICS_USERNAME` and `METRICS_PASSWORD`, default empty: with both set,
the backend records Prometheus metrics of its requests and serves them at
`/metrics` on the container port, to requests with this user name and
password as their basic auth credentials. With neither set, there are no
metrics and `/metrics` is not found. One set without the other, or a user
name containing `:`, stops the container
- `SENTRY_DSN`, default empty: set to a Sentry project's DSN, the backend
sends its errors to that Sentry project: each request whose handling
crashes, which still gets a 500 response. A value Sentry does not accept
stops the container. Empty, the backend sends nothing to Sentry
- **Health check:** the image's `HEALTHCHECK` requests
`/.well-known/healthcheck` through nginx every 30 seconds, so it fails unless
both nginx and the backend answer. upaas reads the container's health 60
seconds after a deploy and fails the deploy unless it is `healthy`. The
container also stops when either process exits.
## Browser Compatibility
@@ -331,19 +125,21 @@ properties.
## Limitations
- **CORS**: The checks are cross-origin requests in `no-cors` mode, so the page
cannot read the answer, only time it: any answer counts as reachable, an error
page included.
- **Local targets**: The cable modem at 192.168.100.1 and the detected gateway
answer only on a network that has them, and only when NetWatch is served from
localhost or a private address (see Monitoring targets).
- **CORS**: Some hosts may block cross-origin HEAD requests. The app uses
`no-cors` mode which allows the request but provides opaque responses. Latency
is still measurable based on request timing.
- **Local gateway**: The 192.168.100.1 endpoint requires the host to be
accessible from your network.
- **Network conditions**: Measurements reflect browser-to-endpoint latency,
which includes your local network, ISP, and internet routing.
## TODO
The to-do list is [TODO.md](TODO.md): where the work stands, the next step, the
open work, and what has been done.
- Add unit tests
- Add eslint for JS linting (currently lint target runs prettier only)
- Add configurable host list (environment variable or config file)
- Add latency history export (CSV/JSON)
- Add notification/alert when status changes to DEGRADED
## License
+83 -383
View File
@@ -1,408 +1,108 @@
---
title: Repository Policies
last_modified: 2026-07-06
---
# Development Policies
This document covers repository structure, tooling, and workflow standards. Code
style conventions are in separate documents:
- Docker image references by tag are server-mutable, therefore using them is an
RCE vulnerability. All docker image references must use cryptographic hashes
to securely specify the exact image that is expected.
- [Code Styleguide](https://git.eeqj.de/sneak/prompts/raw/branch/main/prompts/CODE_STYLEGUIDE.md)
(general, bash, Docker)
- [Go](https://git.eeqj.de/sneak/prompts/raw/branch/main/prompts/CODE_STYLEGUIDE_GO.md)
- [JavaScript](https://git.eeqj.de/sneak/prompts/raw/branch/main/prompts/CODE_STYLEGUIDE_JS.md)
- [Python](https://git.eeqj.de/sneak/prompts/raw/branch/main/prompts/CODE_STYLEGUIDE_PYTHON.md)
- [Go HTTP Server Conventions](https://git.eeqj.de/sneak/prompts/raw/branch/main/prompts/GO_HTTP_SERVER_CONVENTIONS.md)
- Correspondingly, `go install` commands using things like '@latest' are also
dangerous RCE. Whenever writing scripts or tools, ALWAYS specify go install
targets using commit hashes which are cryptographically secure.
---
- Every repo with software in it must have a Makefile in the root. Each such
Makefile should support `make test` (runs the project-specific tests),
`make lint`, `make fmt` (writes), `make fmt-check` (readonly), and
`make check` (has `test`, `lint`, and `fmt-check` as prereqs), `make docker`
(builds docker image).
- Cross-project documentation (such as this file) must include
`last_modified: YYYY-MM-DD` in the YAML front matter so it can be kept in sync
with the authoritative source as policies evolve.
- Every repo should have a Dockerfile. If the repo contains non-server software,
the Dockerfile should bring up a development environment and `make check`
(i.e. the docker build should fail if the branch is not green).
- **ALL external references must be pinned by cryptographic hash.** This
includes Docker base images, Go modules, npm packages, GitHub Actions, and
anything else fetched from a remote source. Version tags (`@v4`, `@latest`,
`:3.21`, etc.) are server-mutable and therefore remote code execution
vulnerabilities. The ONLY acceptable way to reference an external dependency
is by its content hash (Docker `@sha256:...`, Go module hash in `go.sum`, npm
integrity hash in lockfile, GitHub Actions `@<commit-sha>`). No exceptions.
This also means never `curl | bash` to install tools like pyenv, nvm, rustup,
etc. Instead, download a specific release archive from GitHub, verify its hash
(hardcoded in the Dockerfile or script), and only then install. Unverified
install scripts are arbitrary remote code execution. This is the single most
important rule in this document. Double-check every external reference in
every file before committing. There are zero exceptions to this rule.
- Platform-specific standard formatting should be used. `black` for python,
`prettier` for js/css/etc, `go fmt` for go. The only changes to default
settings should be to specify four-space indents where applicable (i.e.
everything except `go fmt`).
- Every repo with software must have a root `Makefile` with these targets:
`make bootstrap`, `make setup`, `make test`, `make lint`, `make fmt` (writes),
`make fmt-check` (read-only), `make check` (runs `test`, `lint`, `fmt-check`),
`make docker`, and `make hooks` (installs pre-commit hook). A model Makefile
is at `https://git.eeqj.de/sneak/prompts/raw/branch/main/Makefile`.
- If local testing is possible (it is not always), `make check` should be a
pre-commit hook. If it is not possible, `make lint && make fmt-check` should
be a pre-commit hook.
- Repos follow the
[Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all)
pattern: the implementation of each Makefile target lives in an executable
script in `script/` (`script/bootstrap`, `script/setup`, `script/test`,
`script/lint`, `script/fmt`, `script/fmt-check`, `script/check`,
`script/docker`), and the Makefile targets are thin shims that call them. The
scripts must be POSIX sh (`#!/bin/sh`, `set -eu`, no bashisms) so they run in
minimal containers (e.g. alpine images have no bash); locate the repo root
with `$(cd "$(dirname "$0")/.." && pwd -P)` and `cd` there before acting. From
the standard's canonical set we use `bootstrap`, `setup` (make the repo ready
for development after a fresh clone: runs `bootstrap`, then
`install-precommit`, plus any repo-specific initialization), `test`, and
`cibuild`. `script/bootstrap` installs all dependencies idempotently and
assumes nothing is present: base tools come from nix, apt, brew, or apk
(detected in that order; apt runs noninteractive). For node it uses the
installed node if present; otherwise it installs a PINNED node version via
nvm, first installing nvm itself if missing — from a hash-verified GitHub
release archive (never `curl | sh`), with bash installed as an explicit
prerequisite since nvm requires bash. yarn is then pinned via
`corepack prepare yarn@<version> --activate`. Never install "latest" or "lts";
always exact versions. `script/cibuild` runs the CI build: it changes to the
repo root and runs `docker build .`; the Gitea workflow calls it. Four further
scripts are our own extensions to the standard: `script/check` runs
`script/test`, `script/lint`, and `script/fmt-check`; `script/precommit` is
what the git pre-commit hook runs, and it calls `script/check`;
`script/install-precommit` installs the git pre-commit hook (the `make hooks`
target shims to it); and `script/projectname` (literally that filename) simply
outputs the project's name. Scripts that need the name call
`script/projectname` — e.g. `script/docker` assembles its image tag from it —
so those scripts stay byte-identical across all repos. Repo-type-specific
pre-commit extras (e.g. `go mod tidy` verification in Go repos) belong in
`script/precommit`, not in the hook itself. Model scripts are at
`https://git.eeqj.de/sneak/prompts/raw/branch/main/script/<name>`. The README
must document the provided scripts in an **Entrypoints** section (see the
README requirements below).
- If a working `make test` takes more than 20 seconds, that's a bug that needs
fixing. In fact, there should be a timeout specified in the `Makefile` that
fails it automatically if it takes >30s.
- Always use Makefile targets (`make fmt`, `make test`, `make lint`, etc.)
instead of invoking the underlying tools directly. The Makefile is the single
source of truth for how these operations are run.
- The Makefile is authoritative documentation for how the repo is used. Beyond
the required targets above, it should have targets for every common operation:
running a local development server (`make run`, `make dev`), re-initializing
or migrating the database (`make db-reset`, `make migrate`), building
artifacts (`make build`), generating code, seeding data, or anything else a
developer would do regularly. If someone checks out the repo and types
`make<tab>`, they should see every meaningful operation available. A new
contributor should be able to understand the entire development workflow by
reading the Makefile.
- Every repo should have a `Dockerfile`. All Dockerfiles must run `make check`
as a build step so the build fails if the branch is not green. For non-server
repos, the Dockerfile should bring up a development environment and run
`make check`. For server repos, `make check` should run as an early build
stage before the final image is assembled. Dockerfiles install development
prerequisites by running `script/bootstrap` rather than duplicating installs
inline; COPY `script/` and the dependency manifests (`package.json` +
`yarn.lock`, `go.mod` + `go.sum`, etc.) before running it so the bootstrap
layer stays cached until dependencies change.
- **Dockerfiles must use a separate lint stage for fail-fast feedback.** Go
repos use a multistage build where linting runs in an independent stage based
on the `golangci/golangci-lint` image (pinned by hash). This stage runs
`make fmt-check` and `make lint` before the full build begins. The build stage
then declares an explicit dependency on the lint stage via
`COPY --from=lint /src/go.sum /dev/null`, which forces BuildKit to complete
linting before proceeding to compilation and tests. This ensures lint failures
surface in seconds rather than minutes, without blocking on dependency
download or compilation in the build stage.
The standard pattern for a Go repo Dockerfile is:
```dockerfile
# Lint stage — fast feedback on formatting and lint issues
# golangci/golangci-lint:v2.x.x, YYYY-MM-DD
FROM golangci/golangci-lint@sha256:... AS lint
WORKDIR /src
COPY go.mod go.sum ./
RUN go mod download
COPY . .
RUN make fmt-check
RUN make lint
# Build stage
# golang:1.x-alpine, YYYY-MM-DD
FROM golang@sha256:... AS builder
WORKDIR /src
# Force BuildKit to run the lint stage before proceeding
COPY --from=lint /src/go.sum /dev/null
COPY go.mod go.sum ./
RUN go mod download
COPY . .
RUN make test
ARG VERSION=dev
RUN CGO_ENABLED=0 go build -trimpath \
-ldflags="-s -w -X main.Version=${VERSION}" \
-o /app ./cmd/app/
# Runtime stage
FROM alpine@sha256:...
COPY --from=builder /app /usr/local/bin/app
ENTRYPOINT ["app"]
```
Key points:
- The lint stage uses the `golangci/golangci-lint` image directly (it
includes both Go and the linter), so there is no need to install the
linter separately.
- `COPY --from=lint /src/go.sum /dev/null` is a no-op file copy that creates
a stage dependency. BuildKit runs stages in parallel by default; without
this line, the build stage would not wait for lint to finish and a lint
failure might not fail the overall build.
- If the project uses `//go:embed` directives that reference build artifacts
(e.g. a web frontend compiled in a separate stage), the lint stage must
create placeholder files so the embed directives resolve. Example:
`RUN mkdir -p web/dist && touch web/dist/index.html web/dist/style.css`.
The lint stage should not depend on the actual build output — it exists to
fail fast.
- If the project requires CGO or system libraries for linting (e.g.
`vips-dev`), install them in the lint stage with `apk add`.
- The build stage runs `make test` after compilation setup. Tests run in the
build stage, not the lint stage, because they may require compiled
artifacts or heavier dependencies.
- Every repo should have a Gitea Actions workflow (`.gitea/workflows/`) that
runs `script/cibuild` (which runs `docker build .`) on push. Since the
Dockerfile already runs `make check`, a successful build implies all checks
pass.
- Use platform-standard formatters: `black` for Python, `prettier` for
JS/CSS/Markdown/HTML, `go fmt` for Go. Always use default configuration with
two exceptions: four-space indents (except Go), and `proseWrap: always` for
Markdown (hard-wrap at 80 columns). Documentation and writing repos (Markdown,
HTML, CSS) should also have `.prettierrc` and `.prettierignore`.
- Pre-commit hook: runs `script/precommit`, which calls `script/check`. If local
testing is not possible in the repo, `script/precommit` may skip `script/test`
and run only `script/lint` and `script/fmt-check`. The hook is installed by
`script/install-precommit`; the Makefile must provide a `make hooks` target
that shims to it.
- All repos with software must have tests that run via the platform-standard
test framework (`go test`, `pytest`, `jest`/`vitest`, etc.). If no meaningful
tests exist yet, add the most minimal test possible — e.g. importing the
module under test to verify it compiles/parses. There is no excuse for
`make test` to be a no-op.
- `make test` must complete in under 20 seconds. Add a 30-second timeout in the
Makefile.
- **`make test` should use the conditional verbose rerun pattern.** Run tests
without `-v` (verbose) first. If tests fail, automatically rerun with `-v` to
show full output. This keeps CI logs and `docker build` output clean on
success (just package/suite summaries) while providing full diagnostic detail
on failure (every test case, every assertion). The general shell pattern:
```makefile
test:
@<test-command> || \
{ echo "--- Rerunning with -v for details ---"; \
<test-command-with-v>; exit 1; }
```
Go example:
```makefile
test:
@go test -timeout 30s -race -cover ./... || \
{ echo "--- Rerunning with -v for details ---"; \
go test -timeout 30s -race -v ./...; exit 1; }
```
Python example:
```makefile
test:
@python -m pytest || \
{ echo "--- Rerunning with -v for details ---"; \
python -m pytest -v; exit 1; }
```
The `exit 1` ensures the target always fails after a rerun — the first run
already proved the tests are broken, so the build must not pass even if a
flaky test happens to succeed on the second attempt. The rerun exists solely
for diagnostic output.
- Docker builds must complete in under 5 minutes.
- `make check` must not modify any files in the repo. Tests may use temporary
directories.
- Docker builds should time out in 5 minutes or less.
- `main` must always pass `make check`, no exceptions.
- Never commit secrets. `.env` files, credentials, API keys, and private keys
must be in `.gitignore`. No exceptions.
- Do all changes on a feature branch. You can do whatever you want on a feature
branch.
- `.gitignore` should be comprehensive from the start: OS files (`.DS_Store`),
editor files (`.swp`, `*~`), language build artifacts, and `node_modules/`.
Fetch the standard `.gitignore` from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when setting up
a new repo.
- We have a standardized `.golangci.yml` which we reuse and is _NEVER_ to be
modified by an agent, only manually by the user. It can be copied from
`~/dev/upaas/.golangci.yml` if it exists at that location.
- **No build artifacts in version control.** Code-derived data (compiled
bundles, minified output, generated assets) must never be committed to the
repository if it can be avoided. The build process (e.g. Dockerfile, Makefile)
should generate these at build time. Notable exception: Go protobuf generated
files (`.pb.go`) ARE committed because repos need to work with `go get`, which
downloads code but does not execute code generation.
- When specifying images or packages by hash in Dockerfiles or
`docker-compose.yml`, put a comment above the line and show the version and
date at which it was current.
- Never use `git add -A` or `git add .`. Always stage files explicitly by name.
- For javascript, always use `yarn` over `npm`.
- Never force-push to `main`.
- Whenever writing dates, ALWAYS write YYYY-MM-DD (ISO 8601).
- Make all changes on a feature branch. You can do whatever you want on a
feature branch.
- Simple projects should be configured with environment variables, as is
standard for Dockerized applications.
- `.golangci.yml` is standardized and must _NEVER_ be modified by an agent, only
manually by the user. Fetch from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml`.
- Dockerized web services should listen on the default HTTP port of 8080 unless
overridden with the `PORT` environment variable.
- When pinning images or packages by hash, add a comment above the reference
with the version and date (YYYY-MM-DD).
- Use `yarn`, not `npm`.
- Write all dates as YYYY-MM-DD (ISO 8601).
- Simple projects should be configured with environment variables.
- Dockerized web services listen on port 8080 by default, overridable with
`PORT`.
- **HTTP/web services must be hardened for production internet exposure before
tagging 1.0.** This means full compliance with security best practices
including, without limitation, all of the following:
- **Security headers** on every response:
- `Strict-Transport-Security` (HSTS) with `max-age` of at least one year
and `includeSubDomains`.
- `Content-Security-Policy` (CSP) with a restrictive default policy
(`default-src 'self'` as a baseline, tightened per-resource as
needed). Never use `unsafe-inline` or `unsafe-eval` unless
unavoidable, and document the reason.
- `X-Frame-Options: DENY` (or `SAMEORIGIN` if framing is required).
Prefer the `frame-ancestors` CSP directive as the primary control.
- `X-Content-Type-Options: nosniff`.
- `Referrer-Policy: strict-origin-when-cross-origin` (or stricter).
- `Permissions-Policy` restricting access to browser features the
application does not use (camera, microphone, geolocation, etc.).
- **Request and response limits:**
- Maximum request body size enforced on all endpoints (e.g. Go
`http.MaxBytesReader`). Choose a sane default per-route; never accept
unbounded input.
- Maximum response body size where applicable (e.g. paginated APIs).
- `ReadTimeout` and `ReadHeaderTimeout` on the `http.Server` to defend
against slowloris attacks.
- `WriteTimeout` on the `http.Server`.
- `IdleTimeout` on the `http.Server`.
- Per-handler execution time limits via `context.WithTimeout` or
chi/stdlib `middleware.Timeout`.
- **Authentication and session security:**
- Rate limiting on password-based authentication endpoints. API keys are
high-entropy and not susceptible to brute force, so they are exempt.
- CSRF tokens on all state-mutating HTML forms. API endpoints
authenticated via `Authorization` header (Bearer token, API key) are
exempt because the browser does not attach these automatically.
- Passwords stored using bcrypt, scrypt, or argon2 — never plain-text,
MD5, or SHA.
- Session cookies set with `HttpOnly`, `Secure`, and `SameSite=Lax` (or
`Strict`) attributes.
- **Reverse proxy awareness:**
- True client IP detection when behind a reverse proxy
(`X-Forwarded-For`, `X-Real-IP`). The application must accept
forwarded headers only from a configured set of trusted proxy
addresses — never trust `X-Forwarded-For` unconditionally.
- **CORS:**
- Authenticated endpoints must restrict `Access-Control-Allow-Origin` to
an explicit allowlist of known origins. Wildcard (`*`) is acceptable
only for public, unauthenticated read-only APIs.
- **Error handling:**
- Internal errors must never leak stack traces, SQL queries, file paths,
or other implementation details to the client. Return generic error
messages in production; detailed errors only when `DEBUG` is enabled.
- **TLS:**
- Services never terminate TLS directly. They are always deployed behind
a TLS-terminating reverse proxy. The service itself listens on plain
HTTP. However, HSTS headers and `Secure` cookie flags must still be
set by the application so that the browser enforces HTTPS end-to-end.
This list is non-exhaustive. Apply defense-in-depth: if a standard security
hardening measure exists for HTTP services and is not listed here, it is
still expected. When in doubt, harden.
- `README.md` is the primary documentation. Required sections:
- **Description**: First line must include the project name, purpose,
category (web server, SPA, CLI tool, etc.), license, and author. Example:
"µPaaS is an MIT-licensed Go web application by @sneak that receives
git-frontend webhooks and deploys applications via Docker in realtime."
- **Getting Started**: Copy-pasteable install/usage code block.
- **Entrypoints**: Opens by stating that the repo adheres to the
[Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all)
standard (with that link), then documents each provided `script/`
entrypoint and its purpose.
- **Rationale**: Why does this exist?
- **Design**: How is the program structured?
- **TODO**: Update meticulously, even between commits. When planning, put
the todo list in the README so a new agent can pick up where the last one
- The `README.md` is a project's primary documentation. It should contain at a
minimum the following sections:
- Description
- Include a short and complete description of the functionality and
purpose of the software as the first line in the readme. It must
include:
- the name
- the purpose
- the category (web server, SPA, command line tool, etc)
- the license
- the author
- eg: "µPaaS is an MIT-licensed Go web application by @sneak that
receives git-frontend webhooks and interacts with a Docker server
to build and deploy applications in realtime as certain branches
are updated."
- Getting Started
- a code block with copy-pasteable installation/use sections
- Rationale
- why does this exist?
- Design
- how is the program structured?
- TODO
- This is your TODO list for the project - update it meticulously, even
in between commits. Whenever planning, put your todo list in the
README so that a separate agent with new context can pick up where you
left off.
- **License**: MIT, GPL, or WTFPL. Ask the user for new projects. Include a
`LICENSE` file in the repo root and a License section in the README.
- **Author**: [@sneak](https://sneak.berlin).
- License
- GPL or MIT or WTFPL - ask the user when beginning a new project and
include a LICENSE file in the root and in a section in the README.
- Author
- @sneak (link `@sneak` to `https://sneak.berlin`).
- First commit of a new repo should contain only `README.md`.
- When beginning a new project, initialize a git repo and make the first commit
simply the first version of the README.md in the root of the repo.
- Go module root: `sneak.berlin/go/<name>`. Always run `go mod tidy` before
committing.
- For Go packages, the module root is `sneak.berlin/go/...`, such as
`sneak.berlin/go/dnswatcher`.
- Use SemVer.
- We use SemVer always.
- Database migrations live in `internal/db/migrations/` and must be embedded in
the binary.
- `000_migration.sql` — contains ONLY the creation of the migrations
tracking table itself. Nothing else.
- `001_schema.sql` — the full application schema.
- **Pre-1.0.0:** never add additional migration files (002, 003, etc.).
There is no installed base to migrate. Edit `001_schema.sql` directly.
- **Post-1.0.0:** add new numbered migration files for each schema change.
Never edit existing migrations after release.
- If no tag `1.0.0` or greater exists in the repository, modify the existing
migrations and assume no installed base or existing databases. If `>=1.0.0`,
database changes add new migration files.
- All repos should have an `.editorconfig` enforcing the project's indentation
settings.
- Avoid putting files in the repo root unless necessary. Root should contain
only project-level config files (`README.md`, `Makefile`, `Dockerfile`,
`LICENSE`, `.gitignore`, `.editorconfig`, `REPO_POLICIES.md`, and
language-specific config). Everything else goes in a subdirectory. Canonical
subdirectory names:
- `bin/` — executable scripts and tools
- `cmd/` — Go command entrypoints
- `configs/` — configuration templates and examples
- `deploy/` — deployment manifests (k8s, compose, terraform)
- `docs/` — documentation and markdown (README.md stays in root)
- `internal/` — Go internal packages
- `internal/db/migrations/` — database migrations
- `pkg/` — Go library packages
- `share/` — systemd units, data files
- `static/` — static assets (images, fonts, etc.)
- `web/` — web frontend source
- When setting up a new repo, files from the `prompts` repo may be used as
templates. Fetch them from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/<path>`.
- New repos must contain at minimum:
- `README.md`, `.git`, `.gitignore`, `.editorconfig`
- `LICENSE`, `REPO_POLICIES.md` (copy from the `prompts` repo)
- `Makefile`
- `script/` entrypoints (`bootstrap`, `setup`, `projectname`, `test`,
`lint`, `fmt`, `fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
`install-precommit`)
- New repos must have at a minimum the following files:
- `README.md`, `.git`, `.gitignore`
- `POLICIES.md` (copy from `~/Documents/_PROMPTS/POLICIES.md`)
- `Dockerfile`, `.dockerignore`
- `.gitea/workflows/check.yml`
- Go: `go.mod`, `go.sum`, `.golangci.yml`
- JS: `package.json`, `yarn.lock`, `.prettierrc`, `.prettierignore`
- Python: `pyproject.toml`
- for go: `go.mod`, `go.sum`, `.golangci.yml`
- for js: `package.json`
-393
View File
@@ -1,393 +0,0 @@
# Workflow
- branch from `next`
- do the work in Next Step
- move Next Step to the top of Completed Steps
- move the top item of Future Steps into Next Step
- commit (`TODO.md` changes in the same commit as the work)
- push the branch and open a PR against `next`
# Status
pre-1.0. No git tags. `main` is the stable branch and `next` the development
branch, which every PR targets. The frontend and the Go backend ship as one
Docker image, and the Gitea workflow `.gitea/workflows/check.yml` runs
`script/cibuild` on every push. Working toward 1.0.0.
# Next Step
Write the latency statistics once and move the thresholds written inline in
`src/main.js` into `CONFIG`
([#102](https://git.eeqj.de/sneak/netwatch/issues/102)).
# Completed Steps
- 2026-10-04: `README.md`, `TODO.md` and `test/viewport/README.md` say what the
tree does (issue #24). The README's Getting Started leads with `make` targets;
a new Backend section says what `netwatch-server` stores, its routes and how
the image builds and runs it, and points to `backend/README.md` for its
settings; the checks are GET requests; the 26 WAN hosts, the four health
states, the summary's figures and the features the list lacked are described
as the page has them; and its TODO section points here, as does the one in
`backend/README.md`, whose open items moved to Future Steps. This file's
Workflow branches from `next` and opens the PR against `next`, Status says
where the repo stands, and Next Step and Future Steps hold only open work,
linked to its issue where one exists. The viewport harness README names Node's
test runner, not `vitest`
- 2026-10-04: password guesses at `/metrics` are rate limited (issue #104): each
client address, resolved through `TRUSTED_PROXIES` as for reports, may make 60
requests to `/metrics` a minute, counted by `go-chi/httprate` apart from its
reports; past that it gets 429 and its basic auth credentials are not checked.
The limit is a constant in `backend/internal/server/routes.go`. A test uses up
one client's allowance on wrong passwords, gets 429 with the right one, and
checks that another client behind the same nginx still gets in
- 2026-10-04: the backend reports errors to Sentry (issue #95). With
`SENTRY_DSN` set, it sets up `sentry-go` with the release `netwatch-server-`
and its version, reports each panic in a handler through `sentryhttp`, the
last of the middleware every request goes through, which panics again so the
request still gets the 500 from the panic recovery, and waits up to 2 seconds
on shutdown for Sentry to finish sending. A DSN Sentry refuses stops the start
with an error naming `SENTRY_DSN`. With it empty, Sentry is not set up and
nothing is sent to it
- 2026-10-04: a target's name and URL and a debug log message show as the
characters they are and are never read as HTML (issue #29): a host row escapes
the name and URL it writes into its markup, and the debug log sets each line
as text. A unit test checks a target whose name and URL hold `<`, `>`, `"`,
`&` and `'`. `README.md` no longer calls `CONFIG` frozen: the interval menu
sets its `updateInterval`, and the values computed from it follow. `AppState`
declares the recovery probe's two properties, the sparkline axis functions
lose the parameters they did not use, and the comment on a target's history
names both kinds of entry it holds. Nothing the page does changed
- 2026-10-04: a dependency can be added without running yarn or go by hand
(issue #45): `make add-dependency PACKAGE=<name>@<version>` shims to the new
`script/add-dependency`, which runs `yarn add --dev`, so `package.json` and
`yarn.lock` change together, then `yarn install --frozen-lockfile`; the same
command moves a package to another version. `make tidy` shims to the new
`script/tidy`, which runs `go mod tidy` in `backend/`: a Go module is added by
importing it, or moved by editing its `require` line, then `make tidy`.
`script/bootstrap` still installs with `--frozen-lockfile`
- 2026-10-04: the backend serves Prometheus metrics (issue #94). With
`METRICS_USERNAME` and `METRICS_PASSWORD` both set, it records request
duration and response size through `go-http-metrics` and serves them, with
Go's runtime and process metrics, at `GET /metrics` behind basic auth with
those credentials; nginx passes `/metrics` to it as it does `/api/`. With
neither set there are no metrics and `/metrics` is 404; one without the other
stops the start with an error naming both, and so does a `METRICS_USERNAME`
containing `:`, with an error naming it. Only requests that reach the health
check or `POST /api/v1/reports` are recorded, not `/metrics` itself and not
every request as `GO_HTTP_SERVER_CONVENTIONS.md` shows, because the labels are
the request's path and method, which clients can make up without end. For
that, `POST /api/v1/reports` is now registered by its full path instead of
inside a `/api/v1` route group; it answers as before
- 2026-10-04: `script/` and `Makefile` follow the org models (issue #28):
`make dev` shims to the new `script/dev`, the Vite dev server, and the new
`make build` to `script/build`, the frontend production build.
`.prettierignore` no longer leaves out `backend/`, so `make fmt` and
`make fmt-check` cover `backend/README.md`; it leaves out
`backend/.golangci.yml` by name, the org standard file whose sha256
`backend/script/lint` checks. `script/install-precommit` and the date on
`script/bootstrap`'s pins are the org model again; `script/bootstrap`,
`script/fmt` and `script/fmt-check` each say in a comment why they differ from
it
- 2026-10-04: a frontend build on Node 26 or newer, such as `make test` on a
host with Node 26, no longer prints Node's warning that `module.register()` is
deprecated (issue #32); the build in `Dockerfile` runs on Node 22, which never
printed it. The call was in `@tailwindcss/node`, which `@tailwindcss/vite`
brings in at its own exact version; tailwind 4.3.1 calls
`module.registerHooks()` instead where Node has it. `yarn.lock` now has
`@tailwindcss/vite` and `tailwindcss` at 4.3.3, inside the ranges
`package.json` already allowed, and tailwind's own dependencies moved with
them. The built CSS changes only in how it is written out, in tailwind's
Firefox focus-ring rule, which no longer applies to iframes (the page has
none, so nothing on it looks different), and in tailwind's default sans-serif
font list, which the page does not use: `body` sets a monospace font
- 2026-10-03: the frontend's unit tests cover what the page computes (issue
#21): `humanDuration`, the latency colours of a figure and of a sparkline
either side of each boundary, a target's min, max, average and median latency
over an empty history, an all-unreachable one and a mixed one, and each of the
four health states either side of its thresholds. `package.json` has a `test`
script, so `yarn run test` and `npm run test` run them, and
`script/frontend-test` runs it: quietly, and if a test fails, again with every
test listed, and then fails. `src/main.js` now exports `humanDuration`,
`HostState`, `latencyClass` and `latencyHex` for the tests; nothing it does
changed
- 2026-10-03: the backend's logs are one stream (issue #27): fx logs its own
steps of starting and stopping through the backend's logger, so off a terminal
every line the backend's own logger and fx write is JSON, where fx used to
write plain text to stderr. A config file that is found but cannot be read now
stops the start with its error, logged as JSON like a bad setting, where it
used to end in a Go panic. A test runs the server as a child process and
checks both. The backend logs its name, version and architecture once at
start. The health check's uptime keys are now `uptime_seconds` and
`uptime_human`; its path, content type, `"status":"ok"` and 200 are unchanged.
`SENTRY_DSN`, `METRICS_USERNAME` and `METRICS_PASSWORD` are still read and
still unused, and left out of `backend/README.md`, until issues #94 and #95
wire them up
- 2026-10-03: the frontend has a real linter (issue #47, and item 2 of issue
#28): `eslint` with its recommended rules, set in `eslint.config.js`, runs in
a new `frontend-lint` stage of `Dockerfile`, which the frontend stage waits
on, as the builder stage waits on the Go `lint` stage. `script/lint` builds
both stages without the cache and runs no linter on the host.
`script/frontend-lint` runs eslint where it used to repeat the prettier check
that `script/fmt-check` runs on the host, and `script/frontend-check`, run by
the frontend stage, is now the tests and the format check. `script/bootstrap`
wants node 22.13.0 or newer, as eslint 10 does
- 2026-10-03: the tap-target check in `make frontend-viewport-test` expects one
visible pin button per WAN host row (issue #46), where it expected at least 10
of the 26, so pin buttons missing from only some rows now fail it. The host
row count the harness gathers, which the `app-rendered` check also reads, now
counts only the WAN host rows: the local host rows have no pin button
- 2026-10-03: each target's row shows its result as soon as its check ends
(issue #91), where every row waited for the round's slowest check, up to 24
seconds at a 30-second interval. Every row is still redrawn, and sorting, the
summary, the health box and offline detection still run, once, when the
round's last check ends, so no row reads "paused" after a pause and resume
during the round. A check that ends after the user pauses or after its round
is given up shows nothing, and the first round is still discarded as a whole
- 2026-10-03: root no longer acts outside `/data` when it prepares `DATA_DIR`
(issue #80): `bin/entrypoint.sh` runs `netwatch-server prepare-data-dir`,
which refuses a `DATA_DIR` that is not `/data` or a path below it written in
full, then creates `DATA_DIR`, gives `/data` and everything in it to
`netwatch` and sets the modes, all through a Go `os.Root` opened on `/data`.
That refuses any path leading out of `/data`, so neither a symbolic link
already there nor one a host process swaps in during the start can make root
create or change anything elsewhere, and `DATA_DIR=/etc` no longer gives
`/etc` to `netwatch`. The `README.md` section "Running under upaas" says which
values are accepted
- 2026-10-03: `DATA_DIR_MAX_BYTES` is now how much of the report files is kept
(issue #54): when a report would take them past it, the oldest report files
are deleted to make room, each deletion logged, and at start files already
past it are deleted the same way. A file still being written is never deleted.
A report is refused with 507 only when the reports waiting to be written fill
the cap on their own, and then no file is deleted. The reports of a failed
write stop counting, and the part of its file written is removed. A file that
cannot be deleted still counts until the next start; one already deleted by
hand counts as freed
- 2026-10-03: `backend/script/lint` says what went wrong with its
`.golangci.yml` check (issue #34). On a hash mismatch it says to compare the
file with the org standard: if they differ, restore the org standard; if they
are the same, the org standard changed, so update `GOLANGCI_CONFIG_SHA256` in
that script. It used to say only to restore the file, which loops once the org
standard itself has moved. A missing `.golangci.yml`, and a `sha256sum` that
is missing or prints no hash, each get their own message instead of being
reported as a mismatch; every one still fails the lint
- 2026-10-03: the Go tests run with the race detector and coverage (issue #88):
`backend/script/test` runs `go test -timeout 30s -race -cover ./...` and, if
that fails, runs it again with `-v` and fails. Go's `-timeout` bounds the
tests, not their compile; the root `script/test` no longer puts one 30-second
timeout around both halves, which a cold Go build cache could use up on
compiling alone. The race detector needs a C compiler: the builder stage of
`Dockerfile` has gcc and musl-dev, and `script/bootstrap` installs gcc, with
the C library headers on apt and apk, when gcc is missing; the binary is still
built with `CGO_ENABLED=0`. New tests cover the health check's answer, a valid
report's answer, a report file's exact contents, and the flush when the buffer
reaches 10 MiB; the handlers' `TestImport` stub is gone
- 2026-10-03: each target check times out after 80% of the refresh interval
(issue #78), 24 seconds at 30 seconds, where it was capped at 3 seconds. A
round started early, after an interval change or when the recovery probe finds
a target answering, gives up the last round's checks if they are still
waiting, so rounds never overlap; the recovery probe gives up its own checks
after half a second. The frontend has its first unit tests, run by
`script/frontend-test` with Node's built-in test runner; for them,
`index.html` now links `src/styles.css`, which `src/main.js` used to import
- 2026-09-29: the container sets up its own data directory (issue #75):
`bin/entrypoint.sh`, still as root, creates `DATA_DIR` if missing and gives it
and `/data` to the `netwatch` user with mode 750 before starting the backend
as that user, so an empty host directory owned by root, or one holding files
from another uid, works with no step on the host. It stops the start instead
when a symbolic link is on the path to `DATA_DIR`, since root would change
whatever the link points to. The `README.md` first-run step that created and
chowned the host directory is gone, and the image no longer sets that
ownership at build time
- 2026-09-29: CI can no longer pass on checks that did not run (issue #37):
`script/cibuild` is now the org model, byte for byte. It runs
`script/bootstrap` and `script/check`, then builds the image with `--no-cache`
and the version from `git describe` as the `VERSION` build argument, where it
used to be a plain `docker build .` whose check steps could come from the
build cache. The workflow puts `~/.local/bin`, where bootstrap links what it
installs, on the step's `PATH`, and bootstrap now installs its pinned node
when the installed one is older than the frontend's dependencies need
- 2026-09-29: `backend/.golangci.yml` re-vendored from `sneak/prompts` (issue
#41): `gomodguard`, deprecated in golangci-lint v2.12.0, is disabled and its
successor `gomodguard_v2` enabled with the org block list, so lint runs print
no deprecation warning. The new file also turns `depguard` on with its
`test-support` rule, which keeps `net/http/httptest` out of non-test code;
netwatch adds no entries of its own to that rule. `backend/script/lint` checks
the new sha256
- 2026-09-29: nginx sends the security headers `REPO_POLICIES.md` requires on
every response (issue #18), including errors, `/assets/` and what it passes on
from the backend, whose own copies it drops so each header goes out once. They
live in `security-headers.conf`, which `nginx.conf` includes. The content
security policy allows no inline script or style, so the status dot's grey in
`src/main.js` is now a class; `connect-src` is `*` because probed hosts
redirect to others, and the browser checks each redirect against it
- 2026-09-29: the request log is bounded (issue #60): the method, URL, protocol,
`User-Agent`, `Referer`, request ID (which chi takes from the client's
`X-Request-Id` header) and client address it writes are each cut to 128 bytes,
the bound the report handler already used, so one request can no longer put
about 1 MiB per field into a log line. That bound and its helper now live in
the `logger` package, shared by both
- 2026-09-29: nginx takes the client address from `X-Forwarded-For` only on
requests from the reverse proxies named in the container's `TRUSTED_PROXIES`
(issue #64), and by default from none, where it trusted every RFC1918 address
before, so a client could write a new address on each request and escape the
rate limit. `bin/entrypoint.sh` writes one `set_real_ip_from` line per entry
into `/etc/nginx/trusted-proxies.conf`, which `nginx.conf` includes, refusing
an entry that is not an IP address or CIDR, as `netwatch-server check-cidr`
finds; it starts the backend with `TRUSTED_PROXIES=127.0.0.1/32`, since nginx
is its only client
- 2026-09-29: report file names can no longer collide (issue #61): each is
`reports-<timestamp>-<number>.jsonl.zst`, where the number goes up by one for
each file the server starts to write, so two flushes in the same millisecond,
such as a flush for size and the final flush at shutdown, each get a file of
their own instead of the second one failing. A failed write uses up its
number, leaving a gap if the file could not be created and otherwise a file
under that number that may be incomplete.
- 2026-09-29: ready to run under upaas (issue #59): the image has a
`HEALTHCHECK` that requests `/.well-known/healthcheck` through nginx on the
port from `PORT`. The backend no longer reads a bad `PORT` as 0 or a bad
`DEBUG` as false: those, and a `BIND_ADDRESS` that is not an IP address, stop
it from starting with an error naming the variable, as the limits,
`CORS_ALLOWED_ORIGINS` and, now by name, `TRUSTED_PROXIES` already did.
`bin/entrypoint.sh` also refuses a `PORT` outside 1 to 65535, and `8081`,
where the backend listens inside the container, naming `PORT`. `README.md` has
a "Running under upaas" section, whose first-run steps create the host
directory for `/data` owned by uid 1000; the image does not change its owner
- 2026-09-29: nginx listens on `PORT` (issue #26), 8080 when unset or empty: the
nginx image renders `nginx.conf` as a template at container start, filling in
`PORT` and no other variable. `bin/entrypoint.sh` refuses to start when `PORT`
is not digits only. `server_tokens off` keeps the nginx version out of
responses. `script/frontend-viewport-test` renders the template the same way.
Gzip and a `50x.html` error page are not added
- 2026-09-29: bounded the report endpoint (issue #20): `POST /api/v1/reports`
still needs no credentials, but each client address, as resolved through
`TRUSTED_PROXIES`, may send `REPORTS_PER_MINUTE` (default 60) reports a
minute, counted by `go-chi/httprate`, and past that gets 429 with
`Retry-After`; the report files in `DATA_DIR`, counted from start with those
already there, may total at most `DATA_DIR_MAX_BYTES` (default 1 GiB), past
which reports get 507; and the wildcard CORS is gone: no CORS headers unless
`CORS_ALLOWED_ORIGINS` lists origins, and an entry that is not a plain
`scheme://host[:port]` origin, `*` included, stops the server from starting.
Deleting report files frees room only at the next start; pruning is issue #54
- 2026-09-28: one container image (issue #52): the root `Dockerfile` builds the
only image, and `Dockerfile.backend` is gone. nginx serves the frontend on
port 8080 and proxies `/api/` and `/.well-known/healthcheck` to the backend,
which listens on `127.0.0.1:8081` in the same container; the new
`BIND_ADDRESS` setting sets its listen address. `bin/entrypoint.sh` starts
both, passes TERM and INT on to both, and exits non-zero when either exits on
its own. The backend runs as user `netwatch` and stores reports on the `/data`
volume. `script/docker` is the org model again
- 2026-09-28: unified the gate (issue #16): the root `make check` covers the Go
backend as well as the frontend, and the pre-commit hook with it; the backend
moved onto scripts-to-rule-them-all (`backend/script/*`, `backend/Makefile` as
shims, its duplicate hook installer removed); `script/cibuild` builds both
images and is the workflow's only build step. The root `make lint` runs
golangci-lint only in Docker, by building the lint stage of
`Dockerfile.backend` without the cache. `script/bootstrap` installs the pinned
Go unless the installed one is at least what `backend/go.mod` asks for, links
what it installs into `~/.local/bin` without replacing anything it did not
create, and installs no linter. Root `make test` runs both halves within one
30-second timeout. When `VERSION` is unset or empty, the backend binary's
version falls back to `git describe` inside a git checkout, then to `dev`
- 2026-09-28: frontend reporting client (issue #53): a `Reporter` class posts
collected samples to `/api/v1/reports` every `reportInterval` (default 60s) as
a per-host delta, with the report-building step a pure exported function of
host state; a per-host mark advances only on a delivered POST; at most one
report POST is pending at a time and it is abandoned after half the interval,
so a slow POST never overlaps the next report and a mark never moves
backwards; the samples of an abandoned POST are sent again at the next
interval, so a backend that stored them but answered late receives them twice;
the per-browser client id works in insecure (plain-HTTP) contexts;
`vite.config.js` proxies `/api` to the local backend for `yarn dev`
- 2026-09-28: report ingest correctness (issue #23): a storage failure now
returns 500 instead of a false `ok`; oversize bodies return 413 (distinguished
from malformed JSON, which stays 400); a `MaxBodyBytes` middleware caps every
route, not just the report route; the raw attacker-controlled `geo` blob is no
longer logged (only its length), and `client_id`, `timestamp` and decode error
text are length-bounded before logging; a `decodeJSON` handler helper was
added; panic recovery now routes the stack through slog instead of chi's
plain-text stderr; and writing a report file now returns its error, so a
failed final flush on shutdown makes the process exit non-zero instead of
losing the buffered reports silently
- 2026-09-21: shutdown lifecycle correctness. The process now shuts down through
fx instead of `os.Exit`, so every component's `OnStop` runs and buffered
reports are flushed to disk on `SIGTERM` — previously a full flush window of
telemetry was silently lost on every restart. The `http.Server` is now built
before its serving goroutine starts, so shutdown can no longer race or
nil-deref it; a listen failure exits non-zero via `fx.Shutdowner`; `reportbuf`
`OnStop` is idempotent; and `writeTimeout` now exceeds the chi per-request
budget so that budget is actually reachable. Dead `startupTime`, `exitCode`,
and `cancelFunc` fields were removed
- 2026-09-21: backend HTTP hardening (issue #19): added `ReadHeaderTimeout` and
`IdleTimeout` to the server, a `SecurityHeaders` middleware (HSTS, tight CSP,
frame/sniff/referrer/permissions headers) registered before CORS, and
trusted-proxy client IP resolution honouring `X-Forwarded-For` / `X-Real-IP`
only from a `TRUSTED_PROXIES` allowlist (loopback plus RFC1918 by default)
- 2026-08-10: adopted the org-standard `backend/.golangci.yml` verbatim and
moved the pinned golangci-lint from v2.7.2 to v2.12.2 (the `lint` stage of
`Dockerfile.backend` now pins the `golangci/golangci-lint:v2.12.2` image by
digest); the previous config declared `version: "2"` but used v1 schema keys,
so every threshold in it was inert and its green result was meaningless.
`backend/Makefile`'s `lint` target now asserts the config's sha256 against the
canonical file first, so drift from the org standard fails the build instead
of silently degrading to defaults
- 2026-08-10: every interactive control now meets the 44x44 CSS px minimum tap
target (`.pin-btn`, `#interval-select`, the debug-log label and, on narrow
viewports, `#pause-btn`). The pin button's hit area grows via matching
negative margins, so its layout footprint and row density are unchanged
- 2026-08-10: per-host status line wraps below the 768px breakpoint instead of
forcing horizontal page scroll at 320px
- 2026-08-09: `Dockerfile.backend` reworked to the mandated Go multistage
lint-stage pattern: separate `lint` stage on the hash-pinned
`golangci/golangci-lint` image, `COPY --from=lint` stage dependency,
`CGO_ENABLED=0` static build driven by `ARG VERSION`, and no more `COPY .git`
- 2026-08-09: dotfile compliance — lifted `backend/.editorconfig` to the repo
root so `root = true` covers the frontend too, and replaced `.gitignore` with
the org model (OS, editor, node, and environment/secrets sections) plus this
repo's `dist/` and `*.log`. `.env`, `.env.*`, `*.pem`, and `*.key` are now
ignored repo-wide, not just under `backend/`. Excluding `.git` from
`.dockerignore` stays deferred: both images read git metadata at build time
(`COPY .git` in `Dockerfile.backend`, `git rev-parse` in `vite.config.js`)
- 2026-08-09: automated responsive-layout harness
(`make frontend-viewport-test`): digest-pinned headless Chrome driven over CDP
against the built `dist/`, viewport widths derived from the breakpoints in
`src/styles.css` ([#13](https://git.eeqj.de/sneak/netwatch/issues/13)). The
tap-target and host-row checks each fail when they measured nothing; the
overflow, viewport-edge and clipped-text checks have no such guard of their
own and rely on the `app-rendered` check, which fails the run when the app did
not render. Found two real layout defects, filed as
[#42](https://git.eeqj.de/sneak/netwatch/issues/42) and
[#43](https://git.eeqj.de/sneak/netwatch/issues/43)
- 2026-07-07 Adopted scripts-to-rule-them-all: `script/` entrypoints, Makefile
shims, README Entrypoints section
- 2026-02-27: backend with buffered zstd-compressed report storage; CI workflow
and backend repo standard files; backend Dockerfile fixed (Go 1.25,
golangci-lint) and moved to repo root (feat/reportbuf-storage)
- 2026-02-26: host row layout redesigned with CSS grid; overflow and spacing
fixes; nginx config extracted; port hardcoded to 8080
- 2026-02-26: debug log panel, median stats, recovery probe, Docker build fix,
S3 Singapore endpoint added
- 2026-02-23: summary box redesign, host pinning, local and UTC clocks, checks
counter
- 2026-02-23: hosts sorted by latency; GET instead of HEAD for latency; timeout
derived from interval; Hetzner regional endpoints; 3s interval
- 2026-01-29: initial NetWatch network latency monitor
# Future Steps
- Take "IPv4 only" out of the page's footer, as nothing in the page limits a
check to IPv4 ([#111](https://git.eeqj.de/sneak/netwatch/issues/111))
- Decide whether the repo moves to the layout `REPO_POLICIES.md` gives, with
`backend/` no longer repeating files from the root
([#30](https://git.eeqj.de/sneak/netwatch/issues/30))
- Run `make frontend-viewport-test` in CI as its own step; it is not part of
`make check`, as it needs Docker and takes minutes
- A backend test that posts a report to `POST /api/v1/reports` and checks the
compressed file it is written to
- A backend route that decompresses the stored reports and answers queries on
them
- Prometheus metrics for the backend's in-memory buffer: its size, the number of
flushes and the number of reports
- A configurable host list (an environment variable or a config file)
- Export of the latency history (CSV or JSON)
- A notification when the health status changes to DEGRADED
+5 -71
View File
@@ -1,30 +1,21 @@
version: "2"
# Config schema uses the golangci-lint v2 layout (settings live under
# linters.settings, not top-level linters-settings) so that the
# thresholds below are actually applied by golangci-lint >= v2.
run:
timeout: 5m
modules-download-mode: readonly
linters:
default: all
enable:
# Successor to the deprecated gomodguard. Named explicitly, rather than
# left to `default: all`, because it carries the module policy below.
- gomodguard_v2
disable:
# Genuinely incompatible with project patterns
- exhaustruct # Requires all struct fields
- depguard # Dependency allow/block lists
- godot # Requires comments to end with periods
- wsl # Deprecated, replaced by wsl_v5
- wrapcheck # Too verbose for internal packages
- varnamelen # Short names like db, id are idiomatic Go
# Deprecated: the warning is attached to the old name, so it is
# silenced by disabling that name, not by enabling the successor.
- wsl # Deprecated, replaced by wsl_v5
- gomodguard # Deprecated, replaced by gomodguard_v2
settings:
linters-settings:
lll:
line-length: 88
funlen:
@@ -34,65 +25,8 @@ linters:
max-complexity: 15
dupl:
threshold: 100
depguard:
# Test-support code must not be compiled into the shipped binary. A
# test-support package exists to hand a test privileges the program
# itself must never have, so a file that is not a test must not import
# one. Test files, and the files inside a package whose directory name
# ends in `test`, are where that code belongs, and are exempt.
#
# The deny list below is the one part of this file a repository is
# expected to extend, and the only part it may. depguard matches an
# import path against a list of prefixes, so it cannot be told "any path
# whose last segment ends in test"; a repository's own test-support
# packages have to be named here one at a time, by full import path,
# under a module path that differs from repository to repository. Add
# them; change nothing else.
rules:
test-support:
list-mode: lax
files:
- "$all"
- "!$test"
- "!**/*test/**"
deny:
- pkg: net/http/httptest
desc: >-
Test-support code belongs in test files and in packages whose
directory name ends in test, not in the shipped binary.
# Only decisions already recorded in the Go package defaults are
# listed here. Every entry matches the module path exactly.
gomodguard_v2:
blocked:
- module: github.com/rs/zerolog
recommendations:
- log/slog
reason: "Structured logging is stdlib log/slog."
# One entry per pre-fork module path, because the later releases
# are separate paths. A prefix match would be shorter but would
# also reach github.com/go-redis/redismock, the test double for
# the successor these entries recommend.
- module: github.com/go-redis/redis
recommendations:
- github.com/redis/go-redis/v9
reason: "Pre-fork module; use the maintained go-redis v9."
- module: github.com/go-redis/redis/v7
recommendations:
- github.com/redis/go-redis/v9
reason: "Pre-fork module; use the maintained go-redis v9."
- module: github.com/go-redis/redis/v8
recommendations:
- github.com/redis/go-redis/v9
reason: "Pre-fork module; use the maintained go-redis v9."
- module: github.com/sergi/go-diff
recommendations:
- github.com/aymanbagabas/go-udiff
reason: "No unified diff output; use go-udiff."
- module: github.com/hexops/gotextdiff
recommendations:
- github.com/aymanbagabas/go-udiff
reason: "Unmaintained fork; use go-udiff."
issues:
exclude-use-default: false
max-issues-per-linter: 0
max-same-issues: 0
+38 -15
View File
@@ -1,30 +1,53 @@
# Thin shims; the implementations live in backend/script/ (see the
# Entrypoints section of README.md). There is no check, hooks or docker
# target here: the root Makefile's check covers this directory, its
# hooks target installs the repo's only pre-commit hook, and its docker
# target builds the one image, which contains this backend.
UNAME_S := $(shell uname -s)
VERSION := $(shell git describe --always --dirty)
BUILDARCH := $(shell uname -m)
BINARY := netwatch-server
.PHONY: all build test lint fmt fmt-check run clean
GOLDFLAGS += -X main.Version=$(VERSION)
GOLDFLAGS += -X main.Buildarch=$(BUILDARCH)
ifeq ($(UNAME_S),Darwin)
GOFLAGS := -ldflags "$(GOLDFLAGS)"
else
GOFLAGS = -ldflags "-linkmode external -extldflags -static $(GOLDFLAGS)"
endif
.PHONY: all build test lint fmt fmt-check check docker hooks run clean
all: build
build:
@script/build
build: ./$(BINARY)
./$(BINARY): $(shell find . -name '*.go' -type f) go.mod go.sum
go build -o $@ $(GOFLAGS) ./cmd/netwatch-server/
test:
@script/test
timeout 30 go test ./...
lint:
@script/lint
golangci-lint run ./...
fmt:
@script/fmt
go fmt ./...
fmt-check:
@script/fmt-check
@test -z "$$(gofmt -l .)" || \
(echo "Files not formatted:"; gofmt -l .; exit 1)
run:
@script/run
check: test lint fmt-check
docker:
timeout 300 docker build -t netwatch-server -f ../Dockerfile.backend ..
hooks:
@printf '#!/bin/sh\ncd backend && make check\n' > \
$$(git rev-parse --show-toplevel)/.git/hooks/pre-commit
@chmod +x \
$$(git rev-parse --show-toplevel)/.git/hooks/pre-commit
@echo "Pre-commit hook installed"
run: build
./$(BINARY)
clean:
@script/clean
rm -f ./$(BINARY)
+12 -173
View File
@@ -4,53 +4,18 @@ SPA and persists them as zstd-compressed JSONL files on disk.
## Getting Started
From this directory:
```bash
# Build and run locally
make run
```
From the repo root, whose `Dockerfile` builds the one image that ships this
backend behind nginx (see [Container image](#container-image)):
```bash
# Run tests, lint, and format check over the frontend and this backend
# Run tests, lint, and format check
make check
# Build the image: nginx, the frontend and this backend
make docker
docker run -p 8080:8080 netwatch
# Docker
docker build -t netwatch-server .
docker run -p 8080:8080 netwatch-server
```
## Entrypoints
This directory follows the same
[Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all)
pattern as the repo root: the targets in `backend/Makefile` are thin shims over
`backend/script/`. The root `Dockerfile` runs them, and the root scripts call
`test`, `fmt` and `fmt-check`:
- `script/build` — compile the static `netwatch-server` binary with its version
stamped in. The version is `VERSION` from the environment; when that is unset
or empty, it falls back to `git describe` inside a git checkout, then to `dev`
- `script/test` — run the Go tests with the race detector and coverage. Go's
`-timeout 30s` bounds the tests, not their compile. If they fail, they run
again with `-v` for the details, and the script fails. The race detector needs
a C compiler
- `script/lint` — check `.golangci.yml` against its pinned sha256, then run
golangci-lint. It runs inside the golangci-lint image of the lint stage of the
root `Dockerfile`; from a checkout, run `make lint` at the repo root, which
builds that stage
- `script/fmt` — format the Go sources (writes)
- `script/fmt-check` — check Go formatting (read-only)
- `script/run` — build and run the server locally
- `script/clean` — remove build artifacts
There is no `check`, `hooks` or `docker` target here: the root `make check`
covers this directory, the root `make hooks` installs the repo's only pre-commit
hook, and the root `make docker` builds the image that contains this backend.
## Rationale
The NetWatch frontend collects latency measurements from the browser but has no
@@ -60,9 +25,8 @@ flushes them to compressed files on disk for later analysis.
## Design
The server is structured as an `fx`-wired Go application under
`cmd/netwatch-server/`. Internal packages in `internal/` follow standard Go
project layout:
The server is structured as an `fx`-wired Go application under `cmd/netwatch-server/`.
Internal packages in `internal/` follow standard Go project layout:
- **`config`**: Loads configuration from environment variables and config files
via Viper.
@@ -79,148 +43,23 @@ project layout:
### Configuration
| Variable | Default | Description |
| ---------------------- | -------------------- | -------------------------------------------------------------------------------------------------------- |
| `BIND_ADDRESS` | empty | IP address to listen on; empty listens on every interface |
| ---------- | ------------------ | --------------------------------- |
| `PORT` | `8080` | HTTP listen port |
| `DATA_DIR` | `./data/reports` | Directory for compressed reports |
| `DATA_DIR_MAX_BYTES` | `1073741824` (1 GiB) | Most bytes of report files kept in `DATA_DIR`, oldest deleted first; see [Report limits](#report-limits) |
| `DEBUG` | `false` | Enable debug logging |
| `TRUSTED_PROXIES` | loopback + RFC1918 | Comma-separated CIDRs whose `X-Forwarded-For` / `X-Real-IP` headers are trusted for client IP resolution |
| `REPORTS_PER_MINUTE` | `60` | Reports each client address may send a minute; see [Report limits](#report-limits) |
| `CORS_ALLOWED_ORIGINS` | empty | Comma-separated origins whose pages may call the API; see [CORS](#cors) |
| `METRICS_USERNAME` | empty | Basic auth user name for `/metrics`; see [Metrics](#metrics) |
| `METRICS_PASSWORD` | empty | Basic auth password for `/metrics`; see [Metrics](#metrics) |
| `SENTRY_DSN` | empty | DSN of the Sentry project to send errors to; see [Sentry](#sentry) |
`TRUSTED_PROXIES` defaults to
`127.0.0.1/32,::1/128,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16`. The loopback
entries cover a reverse proxy on the same host. A request whose direct peer is
outside this set has its forwarded headers ignored, and the direct peer is
logged and rate-limited instead. The container image does not use this default;
see [Container image](#container-image).
A variable set to a value the server cannot use, such as `PORT=abc`,
`DEBUG=maybe` or a `BIND_ADDRESS` that is not an IP address, stops it from
starting, with an error naming the variable. An empty variable counts as unset.
### Container image
The root `Dockerfile` builds one image in which nginx listens on the public port
8080, serves the frontend, and proxies `/api/`, `/.well-known/healthcheck` and
`/metrics` to this server. The image's entrypoint, `bin/entrypoint.sh`, starts
the server as user `netwatch` (uid 1000) with `BIND_ADDRESS=127.0.0.1` and
`PORT=8081`, so only nginx reaches it, and with `TRUSTED_PROXIES=127.0.0.1/32`,
so it takes the client address nginx passes on and no other. `DATA_DIR` is
`/data/reports`, on the `/data` volume; before starting the server, the
entrypoint creates it and gives it and `/data` to `netwatch` with
`netwatch-server prepare-data-dir`, which acts on nothing outside `/data`. nginx
replaces the security headers this server sets with those in the root
`security-headers.conf`, so those are what clients of the image see.
The container's own `TRUSTED_PROXIES` goes to nginx instead: IP addresses or
CIDRs, separated by commas, of the reverse proxies in front of the container.
nginx takes the client address from `X-Forwarded-For` only on a request from one
of them. Unset or empty, nginx trusts no proxy, and the client address is the
one each request comes from, so every client behind a proxy shares one rate
limit. An entry that is not an IP address or CIDR, such as a hostname or
`1.2.3`, stops the container at start with an error naming `TRUSTED_PROXIES`:
the entrypoint checks each entry with `netwatch-server check-cidr`, which parses
it as this server parses its own `TRUSTED_PROXIES`.
### Report storage
Reports are written as `reports-<timestamp>-<number>.jsonl.zst` files in
`DATA_DIR`. The timestamp is in UTC to the millisecond, so the names sort by
time. The number starts at 1 when the server starts and goes up by one for each
file the server starts to write, so two files written in the same millisecond
still get different names. A failed write uses up its number and leaves a gap in
the numbers: its file, if it was created, is removed. The file stays, counted
toward `DATA_DIR_MAX_BYTES` from the next start, only if removing it fails too.
Reports are written as `reports-<timestamp>.jsonl.zst` files in `DATA_DIR`.
Each file contains one JSON object per line, compressed with zstd. Files are
created with `O_EXCL` to prevent overwrites.
### Report limits
`POST /api/v1/reports` takes reports from anyone who can reach it, without
credentials, so it is bounded instead. Both refusals below answer with the same
`{"status":"error"}` body as any other error.
- **Rate limit.** Each client address, resolved through `TRUSTED_PROXIES`, may
send `REPORTS_PER_MINUTE` reports a minute; past that it gets 429 with
`Retry-After: 60`. The minute slides: reports from the minute before still
count, fading out over the current one, so an address is sure never to be
refused only while it sends at most half of `REPORTS_PER_MINUTE` in any 60
seconds. The page sends one report a minute from each open tab, so the default
of 60 refuses nothing from up to 30 tabs behind one address, such as a
household or an office sharing it, however their reports bunch up. Report
responses also carry `X-RateLimit-Limit`, `X-RateLimit-Remaining` and
`X-RateLimit-Reset` headers.
- **Size cap.** The report files in `DATA_DIR` may total at most
`DATA_DIR_MAX_BYTES`, counting the files already there at start. Reports
waiting in memory count at their uncompressed size until they are written;
those lost to a failed write stop counting, and the part of its file written
is removed. When a report would take the total past the cap, the oldest report
files are deleted to make room, and each deletion is logged with the file's
name and size; a file still being written is never deleted. A report is
refused with 507, and nothing of it is stored, only when the reports waiting
to be written fill the cap on their own, and then no file is deleted. At
start, report files past the cap, as after lowering it, are deleted the same
way. So the cap is how much of the newest reports is kept: the default of 1
GiB is small enough for any host; set it to the space you can give `DATA_DIR`.
### CORS
The page calls the API from the origin it is served from, so by default the
server sends no CORS headers, and browsers let no other origin's pages call it.
To serve the page from elsewhere, list that origin in `CORS_ALLOWED_ORIGINS`
(for example `https://netwatch.example.com`); pages from a listed origin may
`GET` and `POST` with a `Content-Type` header. Each entry must be a plain
origin, `scheme://host` with an optional `:port`, as browsers send it: no path,
not even a trailing `/`, and no `*`. Any other entry stops the server from
starting, with an error naming `CORS_ALLOWED_ORIGINS`.
### Metrics
With both `METRICS_USERNAME` and `METRICS_PASSWORD` set, the server serves
Prometheus metrics at `GET /metrics` to requests with those as their basic auth
credentials, and answers any other with 401. For each request that reaches the
health check or `POST /api/v1/reports`, those the rate limit refuses included,
the metrics record its duration and response size, labelled with its path,
method and status; they also count those requests in progress, and include Go's
runtime and process metrics. No other request is recorded: not those to
`/metrics` itself, and not those answered before they reach either route, such
as a CORS preflight, or a request refused with 404 for a path no route has, 405
for a method its route does not take, or 413 for declaring a body length over
the 1 MiB limit. A report whose body goes over the limit without declaring its
length reaches the route, is answered 413 there, and is recorded with that
status. Clients can make up any number of paths and methods, and each would add
labels to the metrics for as long as the server runs. With neither set, nothing
is recorded and `/metrics` answers 404. One without the other stops the server
from starting, with an error naming both; so does a `METRICS_USERNAME`
containing `:`, which basic auth cannot carry, with an error naming it.
`/metrics` is rate limited, so that its password cannot be guessed quickly: each
client address, resolved through `TRUSTED_PROXIES`, may make 60 requests to it a
minute, whatever their credentials. Past that it gets 429 with
`Retry-After: 60`, and its credentials are not checked. The minute slides as it
does for reports (see [Report limits](#report-limits)), so a scraper polling
every 2 seconds or less often is never refused. This allowance is apart from the
one for reports.
### Sentry
With `SENTRY_DSN` set, the server sends its errors to that Sentry project: each
panic in a handler is reported there, under the release `netwatch-server-`
followed by the server's version, and the request still gets 500 from the
server's panic recovery. On shutdown the server waits up to 2 seconds for Sentry
to finish sending. A DSN Sentry refuses stops the server from starting, with an
error naming `SENTRY_DSN`. With it empty, Sentry is not set up, and nothing is
sent to it.
## TODO
The to-do list, this backend's open work included, is [TODO.md](../TODO.md) at
the repo root.
- Add integration test that POSTs a report and verifies the compressed output
- Add report decompression/query endpoint
- Add metrics (Prometheus) for buffer size, flush count, report count
- Add retention policy to prune old report files
## License
+3 -50
View File
@@ -2,10 +2,6 @@
package main
import (
"fmt"
"os"
"os/user"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/handlers"
@@ -16,59 +12,21 @@ import (
"sneak.berlin/go/netwatch/internal/server"
"go.uber.org/fx"
"go.uber.org/fx/fxevent"
)
//nolint:gochecknoglobals // set via ldflags at build time
var (
Appname = "netwatch-server"
Version string
Buildarch string
)
func main() {
// "netwatch-server check-cidr CIDR" exits 1, with the error, if
// this server would refuse CIDR in its TRUSTED_PROXIES.
// bin/entrypoint.sh runs it on each entry it gives nginx.
if len(os.Args) == 3 && os.Args[1] == "check-cidr" {
_, err := middleware.ParseTrustedProxies(os.Args[2:])
if err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
return
}
// "netwatch-server prepare-data-dir DATA_DIR" gets DATA_DIR ready
// for the netwatch user, or exits 1 with the error; see
// reportbuf.PrepareDataDir. bin/entrypoint.sh runs it as root
// before it starts this server as that user.
if len(os.Args) == 3 && os.Args[1] == "prepare-data-dir" {
netwatch, err := user.Lookup("netwatch")
if err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
err = reportbuf.PrepareDataDir("/data", os.Args[2], netwatch)
if err != nil {
fmt.Fprintf(os.Stderr, "DATA_DIR '%s': %v\n", os.Args[2], err)
os.Exit(1)
}
return
}
globals.Appname = Appname
globals.Version = Version
globals.Buildarch = Buildarch
fx.New(
// fx logs each step of starting and stopping through the
// server's own logger, so off a terminal those lines are
// JSON like every other line.
fx.WithLogger(func(log *logger.Logger) fxevent.Logger {
return &fxevent.SlogLogger{Logger: log.Get()}
}),
fx.Provide(
config.New,
globals.New,
@@ -79,11 +37,6 @@ func main() {
reportbuf.New,
server.New,
),
fx.Invoke(
// First, so the name and version are logged even when
// a setting stops the start.
func(log *logger.Logger) { log.Identify() },
func(*server.Server) {},
),
fx.Invoke(func(*server.Server) {}),
).Run()
}
-249
View File
@@ -1,249 +0,0 @@
package main
import (
"bytes"
"context"
"encoding/json"
"net"
"net/http"
"os"
"os/exec"
"os/signal"
"path/filepath"
"strings"
"syscall"
"testing"
"time"
)
// The tests run main() in a child process, this test binary started
// again with runMainEnv set, because main() can exit its process and
// takes its settings from the environment.
const runMainEnv = "NETWATCH_SERVER_RUN_MAIN"
// childTimeout bounds each child's whole run; it is killed after it.
const childTimeout = 10 * time.Second
func TestMain(m *testing.M) {
if os.Getenv(runMainEnv) != "" {
// A SIGTERM that comes before fx catches it is dropped, not fatal.
signal.Notify(make(chan os.Signal, 1), syscall.SIGTERM)
main()
return
}
os.Exit(m.Run())
}
// TestOutputIsJSON: off a terminal, every line the server writes from
// start to stop is JSON, fx's own lines included.
func TestOutputIsJSON(t *testing.T) {
t.Parallel()
ctx, cancel := context.WithTimeout(t.Context(), childTimeout)
defer cancel()
port := freePort(ctx, t)
child, stdout, stderr := startServer(ctx, t, t.TempDir(), port)
waitForHealthcheck(ctx, t, port)
// The child drops a SIGTERM that comes before fx catches it (see
// TestMain), so send one every 100ms until the test ends. ctx
// bounds the wait: when it ends, the child is killed.
stop := make(chan struct{})
defer close(stop)
go func() {
for {
_ = child.Process.Signal(syscall.SIGTERM)
select {
case <-stop:
return
case <-time.After(100 * time.Millisecond):
}
}
}()
err := child.Wait()
if err != nil {
t.Fatalf("server exit = %v, want success", err)
}
requireJSONLines(t, stdout, stderr)
if !strings.Contains(stdout.String(), `"msg":"starting"`) {
t.Fatalf("no startup line in stdout:\n%s", stdout)
}
}
// TestMalformedConfigFileStopsTheStart: a config file the server finds
// but cannot read stops the start, and the error is logged as JSON.
func TestMalformedConfigFileStopsTheStart(t *testing.T) {
t.Parallel()
ctx, cancel := context.WithTimeout(t.Context(), childTimeout)
defer cancel()
home := t.TempDir()
dir := filepath.Join(home, ".config", "netwatch-server")
err := os.MkdirAll(dir, 0o750)
if err != nil {
t.Fatal(err)
}
err = os.WriteFile(filepath.Join(dir, "netwatch-server.yaml"),
[]byte("PORT: [8080\n"), 0o600)
if err != nil {
t.Fatal(err)
}
child, stdout, stderr := startServer(ctx, t, home, freePort(ctx, t))
err = child.Wait()
if child.ProcessState.ExitCode() != 1 {
t.Fatalf("server exit = %v, want exit status 1", err)
}
requireJSONLines(t, stdout, stderr)
if !strings.Contains(stdout.String(), "netwatch-server.yaml") {
t.Fatalf("no error naming the config file in stdout:\n%s", stdout)
}
}
// TestRefusedSentryDSNStopsTheStart: a SENTRY_DSN that Sentry refuses
// stops the start, and the error, naming SENTRY_DSN, is logged as JSON.
func TestRefusedSentryDSNStopsTheStart(t *testing.T) {
t.Setenv("SENTRY_DSN", "not-a-dsn")
ctx, cancel := context.WithTimeout(t.Context(), childTimeout)
defer cancel()
child, stdout, stderr := startServer(ctx, t, t.TempDir(), freePort(ctx, t))
err := child.Wait()
if child.ProcessState.ExitCode() != 1 {
t.Fatalf("server exit = %v, want exit status 1", err)
}
requireJSONLines(t, stdout, stderr)
if !strings.Contains(stdout.String(), "SENTRY_DSN") {
t.Fatalf("no error naming SENTRY_DSN in stdout:\n%s", stdout)
}
}
// startServer runs main() in a child process listening on
// 127.0.0.1:port, with home as its HOME and working directory and its
// data directory in home, so it touches nothing outside home. Its
// stdout and stderr go to the two buffers returned, which hold all of
// it once child.Wait returns. The child is killed when ctx ends, and
// killed and reaped when the test ends if nothing waited for it.
func startServer(
ctx context.Context,
t *testing.T,
home, port string,
) (*exec.Cmd, *bytes.Buffer, *bytes.Buffer) {
t.Helper()
self, err := os.Executable()
if err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
child := exec.CommandContext(ctx, self) //nolint:gosec // this test binary
child.Dir = home
child.Env = append(os.Environ(),
runMainEnv+"=1",
"HOME="+home,
"DATA_DIR="+filepath.Join(home, "data"),
"BIND_ADDRESS=127.0.0.1",
"PORT="+port,
)
child.Stdout = &stdout
child.Stderr = &stderr
err = child.Start()
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() {
if child.ProcessState == nil {
_ = child.Process.Kill()
_ = child.Wait()
}
})
return child, &stdout, &stderr
}
// freePort returns a TCP port on 127.0.0.1 that was free a moment ago.
func freePort(ctx context.Context, t *testing.T) string {
t.Helper()
var lc net.ListenConfig
l, err := lc.Listen(ctx, "tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
_ = l.Close()
_, port, err := net.SplitHostPort(l.Addr().String())
if err != nil {
t.Fatal(err)
}
return port
}
// waitForHealthcheck returns once the health check on port answers 200.
func waitForHealthcheck(ctx context.Context, t *testing.T, port string) {
t.Helper()
url := "http://127.0.0.1:" + port + "/.well-known/healthcheck"
for {
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
t.Fatal(err)
}
resp, err := http.DefaultClient.Do(req)
if err == nil {
_ = resp.Body.Close()
if resp.StatusCode == http.StatusOK {
return
}
}
select {
case <-ctx.Done():
t.Fatalf("health check never answered: %v", err)
case <-time.After(50 * time.Millisecond):
}
}
}
// requireJSONLines fails the test on each line of outs that is not
// JSON.
func requireJSONLines(t *testing.T, outs ...*bytes.Buffer) {
t.Helper()
for _, out := range outs {
for line := range strings.Lines(out.String()) {
if !json.Valid([]byte(line)) {
t.Errorf("line is not JSON: %s", line)
}
}
}
}
+3 -17
View File
@@ -3,42 +3,28 @@ module sneak.berlin/go/netwatch
go 1.25.5
require (
github.com/99designs/basicauth-go v0.0.0-20230316000542-bf6f9cbbf0f8
github.com/getsentry/sentry-go v0.49.0
github.com/go-chi/chi/v5 v5.2.5
github.com/go-chi/cors v1.2.2
github.com/go-chi/httprate v0.16.0
github.com/joho/godotenv v1.5.1
github.com/klauspost/compress v1.19.1
github.com/prometheus/client_golang v1.24.1
github.com/slok/go-http-metrics v0.13.0
github.com/klauspost/compress v1.18.4
github.com/spf13/viper v1.21.0
go.uber.org/fx v1.24.0
)
require (
github.com/beorn7/perks v1.0.1 // indirect
github.com/cespare/xxhash/v2 v2.3.0 // indirect
github.com/fsnotify/fsnotify v1.9.0 // indirect
github.com/go-viper/mapstructure/v2 v2.4.0 // indirect
github.com/klauspost/cpuid/v2 v2.2.10 // indirect
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
github.com/pelletier/go-toml/v2 v2.2.4 // indirect
github.com/prometheus/client_model v0.6.2 // indirect
github.com/prometheus/common v0.70.1 // indirect
github.com/prometheus/procfs v0.21.1 // indirect
github.com/sagikazarmark/locafero v0.11.0 // indirect
github.com/sourcegraph/conc v0.3.1-0.20240121214520-5f936abd7ae8 // indirect
github.com/spf13/afero v1.15.0 // indirect
github.com/spf13/cast v1.10.0 // indirect
github.com/spf13/pflag v1.0.10 // indirect
github.com/subosito/gotenv v1.6.0 // indirect
github.com/zeebo/xxh3 v1.0.2 // indirect
go.uber.org/dig v1.19.0 // indirect
go.uber.org/multierr v1.10.0 // indirect
go.uber.org/zap v1.26.0 // indirect
go.yaml.in/yaml/v3 v3.0.4 // indirect
golang.org/x/sys v0.47.0 // indirect
golang.org/x/text v0.40.0 // indirect
google.golang.org/protobuf v1.36.11 // indirect
golang.org/x/sys v0.29.0 // indirect
golang.org/x/text v0.28.0 // indirect
)
+18 -60
View File
@@ -1,65 +1,33 @@
github.com/99designs/basicauth-go v0.0.0-20230316000542-bf6f9cbbf0f8 h1:nMpu1t4amK3vJWBibQ5X/Nv0aXL+b69TQf2uK5PH7Go=
github.com/99designs/basicauth-go v0.0.0-20230316000542-bf6f9cbbf0f8/go.mod h1:3cARGAK9CfW3HoxCy1a0G4TKrdiKke8ftOMEOHyySYs=
github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM=
github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw=
github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs=
github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs=
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8=
github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0=
github.com/fsnotify/fsnotify v1.9.0 h1:2Ml+OJNzbYCTzsxtv8vKSFD9PbJjmhYF14k/jKC7S9k=
github.com/fsnotify/fsnotify v1.9.0/go.mod h1:8jBTzvmWwFyi3Pb8djgCCO5IBqzKJ/Jwo8TRcHyHii0=
github.com/getsentry/sentry-go v0.49.0 h1:Ehejknu1l023Ub7QoRBVLAI7g3Jnhqku4oWx4B4Sh5s=
github.com/getsentry/sentry-go v0.49.0/go.mod h1:nuMJAoCfe1u0Bts2ocyNI+TW8HT84vRMqwA5Qq/SKUI=
github.com/go-chi/chi/v5 v5.2.5 h1:Eg4myHZBjyvJmAFjFvWgrqDTXFyOzjj7YIm3L3mu6Ug=
github.com/go-chi/chi/v5 v5.2.5/go.mod h1:X7Gx4mteadT3eDOMTsXzmI4/rwUpOwBHLpAfupzFJP0=
github.com/go-chi/cors v1.2.2 h1:Jmey33TE+b+rB7fT8MUy1u0I4L+NARQlK6LhzKPSyQE=
github.com/go-chi/cors v1.2.2/go.mod h1:sSbTewc+6wYHBBCW7ytsFSn836hqM7JxpglAy2Vzc58=
github.com/go-chi/httprate v0.16.0 h1:8V5DH9j6pSK6UQoBsTpvMyFxycqaKEIToyPKzHJjUa8=
github.com/go-chi/httprate v0.16.0/go.mod h1:A8lo+qRhk+s9LiuP5saS7XCGDXRXMcrueq0NfIuCa/I=
github.com/go-errors/errors v1.4.2 h1:J6MZopCL4uSllY1OfXM374weqZFFItUbrImctkmUxIA=
github.com/go-errors/errors v1.4.2/go.mod h1:sIVyrIiJhuEF+Pj9Ebtd6P/rEYROXFi3BopGUQ5a5Og=
github.com/go-viper/mapstructure/v2 v2.4.0 h1:EBsztssimR/CONLSZZ04E8qAkxNYq4Qp9LvH92wZUgs=
github.com/go-viper/mapstructure/v2 v2.4.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM=
github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
github.com/google/go-cmp v0.6.0 h1:ofyhxvXcZhMsU5ulbFiLKl/XBFqE1GSq7atu8tAmTRI=
github.com/google/go-cmp v0.6.0/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY=
github.com/joho/godotenv v1.5.1 h1:7eLL/+HRGLY0ldzfGMeQkb7vMd0as4CfYvUVzLqw0N0=
github.com/joho/godotenv v1.5.1/go.mod h1:f4LDr5Voq0i2e/R5DDNOoa2zzDfwtkZa6DnEwAbqwq4=
github.com/klauspost/compress v1.19.1 h1:VsB4HPswih7mmZ8WleSFQ75c/Ui1M4trX5oAsJnhSlk=
github.com/klauspost/compress v1.19.1/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
github.com/klauspost/cpuid/v2 v2.2.10 h1:tBs3QSyvjDyFTq3uoc/9xFpCuOsJQFNPiAhYdw2skhE=
github.com/klauspost/cpuid/v2 v2.2.10/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0=
github.com/klauspost/compress v1.18.4 h1:RPhnKRAQ4Fh8zU2FY/6ZFDwTVTxgJ/EMydqSTzE9a2c=
github.com/klauspost/compress v1.18.4/go.mod h1:R0h/fSBs8DE4ENlcrlib3PsXS61voFxhIs2DeRhCvJ4=
github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE=
github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk=
github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc=
github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw=
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
github.com/pelletier/go-toml/v2 v2.2.4 h1:mye9XuhQ6gvn5h28+VilKrrPoQVanw5PMw/TB0t5Ec4=
github.com/pelletier/go-toml/v2 v2.2.4/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY=
github.com/pingcap/errors v0.11.4 h1:lFuQV/oaUMGcD2tqt+01ROSmJs75VG1ToEOkZIZ4nE4=
github.com/pingcap/errors v0.11.4/go.mod h1:Oi8TUi2kEtXXLMJk9l1cGmz20kV3TaQ0usTwv5KuLY8=
github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4=
github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0=
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 h1:Jamvg5psRIccs7FGNTlIRMkT8wgtp5eCXdBlqhYGL6U=
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk=
github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE=
github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY=
github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc=
github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI=
github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY=
github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ=
github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc=
github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/rogpeppe/go-internal v1.9.0 h1:73kH8U+JUqXU8lRuOHeVHaa/SZPifC7BkcraZVejAe8=
github.com/rogpeppe/go-internal v1.9.0/go.mod h1:WtVeX8xhTBvf0smdhujwtBcq4Qrzq/fJaraNFVN+nFs=
github.com/sagikazarmark/locafero v0.11.0 h1:1iurJgmM9G3PA/I+wWYIOw/5SyBtxapeHDcg+AAIFXc=
github.com/sagikazarmark/locafero v0.11.0/go.mod h1:nVIGvgyzw595SUSUE6tvCp3YYTeHs15MvlmU87WwIik=
github.com/slok/go-http-metrics v0.13.0 h1:lQDyJJx9wKhmbliyUsZ2l6peGnXRHjsjoqPt5VYzcP8=
github.com/slok/go-http-metrics v0.13.0/go.mod h1:HIr7t/HbN2sJaunvnt9wKP9xoBBVZFo1/KiHU3b0w+4=
github.com/sourcegraph/conc v0.3.1-0.20240121214520-5f936abd7ae8 h1:+jumHNA0Wrelhe64i8F6HNlS8pkoyMv5sreGx2Ry5Rw=
github.com/sourcegraph/conc v0.3.1-0.20240121214520-5f936abd7ae8/go.mod h1:3n1Cwaq1E1/1lhQhtRK2ts/ZwZEhjcQeJQ1RuC6Q/8U=
github.com/spf13/afero v1.15.0 h1:b/YBCLWAJdFWJTN9cLhiXXcD7mzKn9Dm86dNnfyQw1I=
@@ -70,38 +38,28 @@ github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk=
github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
github.com/spf13/viper v1.21.0 h1:x5S+0EU27Lbphp4UKm1C+1oQO+rKx36vfCoaVebLFSU=
github.com/spf13/viper v1.21.0/go.mod h1:P0lhsswPGWD/1lZJ9ny3fYnVqxiegrlNrEmgLjbTCAY=
github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY=
github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA=
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8=
github.com/subosito/gotenv v1.6.0/go.mod h1:Dk4QP5c2W3ibzajGcXpNraDfq2IrhjMIvMSWPKKo0FU=
github.com/zeebo/assert v1.3.0 h1:g7C04CbJuIDKNPFHmsk4hwZDO5O+kntRxzaUoNXj+IQ=
github.com/zeebo/assert v1.3.0/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0=
github.com/zeebo/xxh3 v1.0.2 h1:xZmwmqxHZA8AI603jOQ0tMqmBr9lPeFwGg6d+xy9DC0=
github.com/zeebo/xxh3 v1.0.2/go.mod h1:5NWz9Sef7zIDm2JHfFlcQvNekmcEl9ekUZQQKCYaDcA=
go.uber.org/dig v1.19.0 h1:BACLhebsYdpQ7IROQ1AGPjrXcP5dF80U3gKoFzbaq/4=
go.uber.org/dig v1.19.0/go.mod h1:Us0rSJiThwCv2GteUN0Q7OKvU7n5J4dxZ9JKUXozFdE=
go.uber.org/fx v1.24.0 h1:wE8mruvpg2kiiL1Vqd0CC+tr0/24XIB10Iwp2lLWzkg=
go.uber.org/fx v1.24.0/go.mod h1:AmDeGyS+ZARGKM4tlH4FY2Jr63VjbEDJHtqXTGP5hbo=
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE=
go.uber.org/goleak v1.2.0 h1:xqgm/S+aQvhWFTtR0XK3Jvg7z8kGV8P4X14IzwN3Eqk=
go.uber.org/goleak v1.2.0/go.mod h1:XJYK+MuIchqpmGmUSAzotztawfKvYLUIgg7guXrwVUo=
go.uber.org/multierr v1.10.0 h1:S0h4aNzvfcFsC3dRF1jLoaov7oRaKqRGC/pUEJ2yvPQ=
go.uber.org/multierr v1.10.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN80Y=
go.uber.org/zap v1.26.0 h1:sI7k6L95XOKS281NhVKOFCUNIvv9e0w4BF8N3u+tCRo=
go.uber.org/zap v1.26.0/go.mod h1:dtElttAiwGvoJ/vj4IwHBS/gXsEu/pZ50mUIRWuG0so=
go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc=
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE=
google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
golang.org/x/sys v0.29.0 h1:TPYlXGxvx1MGTn2GiZDhnjPA9wZzZeGKHHmKhHYvgaU=
golang.org/x/sys v0.29.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/text v0.28.0 h1:rhazDwis8INMIwQ4tpjLDzUhx6RlXqZNPEM0huQojng=
golang.org/x/text v0.28.0/go.mod h1:U8nCwOR8jO/marOQ0QbDiOngZVEBB7MAiitBuMjXiNU=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk=
gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q=
gopkg.in/check.v1 v1.0.0-20190902080502-41f04d3bba15 h1:YR8cESwS4TdDjEe65xsg0ogRM/Nc3DYOhEAlW+xobZo=
gopkg.in/check.v1 v1.0.0-20190902080502-41f04d3bba15/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
+5 -157
View File
@@ -4,13 +4,7 @@ package config
import (
"errors"
"fmt"
"log/slog"
"math"
"net/netip"
"net/url"
"strconv"
"strings"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/logger"
@@ -20,41 +14,6 @@ import (
"go.uber.org/fx"
)
// defaultTrustedProxies lists the networks whose forwarded
// headers are honoured by default: IPv4 and IPv6 loopback,
// for a reverse proxy on the same host, and the RFC1918
// ranges. The container image does not use it:
// bin/entrypoint.sh gives the server 127.0.0.1/32, since
// nginx is its only client there.
const defaultTrustedProxies = "127.0.0.1/32,::1/128," +
"10.0.0.0/8,172.16.0.0/12,192.168.0.0/16"
// Default limits on stored reports; backend/README.md gives the
// reasons for these values.
const (
defaultReportsPerMinute = 60
defaultDataDirMaxBytes = 1 << 30 // 1 GiB
)
var (
errNotPositive = errors.New("must be a positive whole number")
errNotOrigin = errors.New(
"must be an origin, scheme://host with an optional port",
)
errNotPort = errors.New("must be a port number, 1 to 65535")
errNotBool = errors.New("must be true or false")
errNotIP = errors.New("must be an IP address, or empty")
errMetricsCredentials = errors.New(
"METRICS_USERNAME and METRICS_PASSWORD must be set together, " +
"or neither",
)
errMetricsUsernameColon = errors.New(
"METRICS_USERNAME must not contain \":\", " +
"which basic auth cannot carry in a user name",
)
)
// Params defines the dependencies for Config.
type Params struct {
fx.In
@@ -65,25 +24,18 @@ type Params struct {
// Config holds the resolved application configuration.
type Config struct {
BindAddress string
CORSAllowedOrigins []string
DataDir string
DataDirMaxBytes int64
Debug bool
MetricsPassword string
MetricsUsername string
Port int
ReportsPerMinute int
SentryDSN string
TrustedProxies []string
log *slog.Logger
params *Params
}
// New loads configuration from env, .env files, and config
// files, returning a fully resolved Config. It fails, with an error
// naming the setting, on a value the server cannot use, and on a
// config file it finds but cannot read.
// files, returning a fully resolved Config.
func New(
_ fx.Lifecycle,
params Params,
@@ -98,64 +50,33 @@ func New(
viper.AutomaticEnv()
// An empty CORS_ALLOWED_ORIGINS allows no other origin.
viper.SetDefault("CORS_ALLOWED_ORIGINS", "")
viper.SetDefault("DATA_DIR", "./data/reports")
viper.SetDefault("DATA_DIR_MAX_BYTES", defaultDataDirMaxBytes)
viper.SetDefault("DEBUG", "false")
// An empty BIND_ADDRESS listens on every interface.
viper.SetDefault("BIND_ADDRESS", "")
viper.SetDefault("PORT", "8080")
viper.SetDefault("REPORTS_PER_MINUTE", defaultReportsPerMinute)
viper.SetDefault("SENTRY_DSN", "")
viper.SetDefault("METRICS_USERNAME", "")
viper.SetDefault("METRICS_PASSWORD", "")
viper.SetDefault("TRUSTED_PROXIES", defaultTrustedProxies)
err := viper.ReadInConfig()
if err != nil {
var notFound viper.ConfigFileNotFoundError
if !errors.As(err, &notFound) {
return nil, fmt.Errorf("config file %s: %w",
viper.ConfigFileUsed(), err)
log.Error("config file malformed", "error", err)
panic(err)
}
}
// Read with strconv: viper's GetInt and GetBool would read a value
// they cannot parse as 0 or false instead of failing.
port, err := strconv.Atoi(viper.GetString("PORT"))
if err != nil || port < 1 || port > math.MaxUint16 {
return nil, fmt.Errorf("PORT %q: %w",
viper.GetString("PORT"), errNotPort)
}
debug, err := strconv.ParseBool(viper.GetString("DEBUG"))
if err != nil {
return nil, fmt.Errorf("DEBUG %q: %w",
viper.GetString("DEBUG"), errNotBool)
}
s := &Config{
BindAddress: viper.GetString("BIND_ADDRESS"),
CORSAllowedOrigins: splitList(viper.GetString("CORS_ALLOWED_ORIGINS")),
DataDir: viper.GetString("DATA_DIR"),
DataDirMaxBytes: viper.GetInt64("DATA_DIR_MAX_BYTES"),
Debug: debug,
Debug: viper.GetBool("DEBUG"),
MetricsPassword: viper.GetString("METRICS_PASSWORD"),
MetricsUsername: viper.GetString("METRICS_USERNAME"),
Port: port,
ReportsPerMinute: viper.GetInt("REPORTS_PER_MINUTE"),
Port: viper.GetInt("PORT"),
SentryDSN: viper.GetString("SENTRY_DSN"),
TrustedProxies: splitList(viper.GetString("TRUSTED_PROXIES")),
log: log,
params: &params,
}
err = s.check()
if err != nil {
return nil, err
}
if s.Debug {
params.Logger.EnableDebugLogging()
s.log = params.Logger.Get()
@@ -163,76 +84,3 @@ func New(
return s, nil
}
// check fails with an error naming the first setting here whose value
// the server cannot use. New checks PORT and DEBUG as it reads them,
// and the middleware checks TRUSTED_PROXIES as it parses it.
func (s *Config) check() error {
// viper reads a value that is not a number as 0, so this also
// catches a mistyped setting.
if s.ReportsPerMinute <= 0 {
return fmt.Errorf("REPORTS_PER_MINUTE %q: %w",
viper.GetString("REPORTS_PER_MINUTE"), errNotPositive)
}
if s.DataDirMaxBytes <= 0 {
return fmt.Errorf("DATA_DIR_MAX_BYTES %q: %w",
viper.GetString("DATA_DIR_MAX_BYTES"), errNotPositive)
}
if s.BindAddress != "" {
_, err := netip.ParseAddr(s.BindAddress)
if err != nil {
return fmt.Errorf("BIND_ADDRESS %q: %w", s.BindAddress, errNotIP)
}
}
// The server records and serves metrics only with both set, so
// one alone is a mistake that would otherwise go unnoticed.
if (s.MetricsUsername == "") != (s.MetricsPassword == "") {
return errMetricsCredentials
}
// Basic auth splits the credentials at the first ":", so with one
// in the user name every request to /metrics would get 401.
if strings.Contains(s.MetricsUsername, ":") {
return errMetricsUsernameColon
}
return checkOrigins(s.CORSAllowedOrigins)
}
// checkOrigins fails on the first CORS_ALLOWED_ORIGINS entry that is
// not a plain origin, scheme://host with an optional port, as browsers
// send it; anything more, such as a trailing "/", would match no page.
// go-chi/cors reads a "*" anywhere in an entry as a wildcard, so no
// entry may contain one.
func checkOrigins(origins []string) error {
for _, origin := range origins {
u, err := url.Parse(origin)
if err != nil || u.Scheme == "" || u.Host == "" ||
strings.Contains(origin, "*") ||
origin != u.Scheme+"://"+u.Host {
return fmt.Errorf("CORS_ALLOWED_ORIGINS %q: %w",
origin, errNotOrigin)
}
}
return nil
}
// splitList turns a comma-separated setting into a trimmed
// slice, dropping empty entries.
func splitList(raw string) []string {
parts := strings.Split(raw, ",")
out := make([]string, 0, len(parts))
for _, p := range parts {
p = strings.TrimSpace(p)
if p != "" {
out = append(out, p)
}
}
return out
}
-147
View File
@@ -1,147 +0,0 @@
package config_test
import (
"strings"
"testing"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/logger"
"go.uber.org/fx"
)
// requireConfigError builds the config as main does and fails the
// test unless that fails with an error naming setting. It uses
// fx.New, because fxtest.New fails the test itself on an error.
func requireConfigError(t *testing.T, setting string) {
t.Helper()
app := fx.New(
fx.NopLogger,
fx.Provide(globals.New, logger.New, config.New),
fx.Invoke(func(*config.Config) {}),
)
err := app.Err()
if err == nil || !strings.Contains(err.Error(), setting) {
t.Fatalf("config error = %v, want one naming %s", err, setting)
}
}
// TestSettingsLoadAsGiven: valid values pass the checks and are used
// as given. bin/entrypoint.sh starts the server with these
// BIND_ADDRESS and PORT values.
func TestSettingsLoadAsGiven(t *testing.T) {
t.Setenv("BIND_ADDRESS", "127.0.0.1")
t.Setenv("PORT", "8081")
t.Setenv("DEBUG", "true")
var cfg *config.Config
app := fx.New(
fx.NopLogger,
fx.Provide(globals.New, logger.New, config.New),
fx.Populate(&cfg),
)
err := app.Err()
if err != nil {
t.Fatalf("config error = %v", err)
}
if cfg.BindAddress != "127.0.0.1" || cfg.Port != 8081 || !cfg.Debug {
t.Fatalf("BindAddress, Port, Debug = %q, %d, %t; "+
"want \"127.0.0.1\", 8081, true",
cfg.BindAddress, cfg.Port, cfg.Debug)
}
}
// TestPortMustBeAPortNumber: viper reads a value that is not a number
// as 0, on which the server would listen on a random port.
func TestPortMustBeAPortNumber(t *testing.T) {
for _, value := range []string{"abc", "0", "65536", "8080.5"} {
t.Run(value, func(t *testing.T) {
t.Setenv("PORT", value)
requireConfigError(t, "PORT")
})
}
}
// TestDebugMustBeTrueOrFalse: viper reads any other value, such as
// "yes", as false.
func TestDebugMustBeTrueOrFalse(t *testing.T) {
t.Setenv("DEBUG", "yes")
requireConfigError(t, "DEBUG")
}
// TestBindAddressMustBeAnIPAddress: a host name would be looked up
// only once the server starts listening, and a mistyped one would stop
// it then with an error that does not name the setting.
func TestBindAddressMustBeAnIPAddress(t *testing.T) {
t.Setenv("BIND_ADDRESS", "localhost")
requireConfigError(t, "BIND_ADDRESS")
}
// TestReportsPerMinuteMustBePositive: unchecked, zero would panic
// when the routes are built, and a negative rate would lift the
// limit.
func TestReportsPerMinuteMustBePositive(t *testing.T) {
t.Setenv("REPORTS_PER_MINUTE", "0")
requireConfigError(t, "REPORTS_PER_MINUTE")
}
// TestDataDirMaxBytesMustBeANumber: viper reads a value that is not
// a number, such as "1GB", as 0, which would refuse every report.
func TestDataDirMaxBytesMustBeANumber(t *testing.T) {
t.Setenv("DATA_DIR_MAX_BYTES", "1GB")
requireConfigError(t, "DATA_DIR_MAX_BYTES")
}
// TestMetricsCredentialsGoTogether: with only one of the two set, the
// server would quietly serve no metrics, so the start fails, naming
// both.
func TestMetricsCredentialsGoTogether(t *testing.T) {
for _, set := range []string{"METRICS_USERNAME", "METRICS_PASSWORD"} {
t.Run(set, func(t *testing.T) {
t.Setenv("METRICS_USERNAME", "")
t.Setenv("METRICS_PASSWORD", "")
t.Setenv(set, "prometheus")
requireConfigError(t, "METRICS_USERNAME")
requireConfigError(t, "METRICS_PASSWORD")
})
}
}
// TestMetricsUsernameMustNotContainColon: basic auth splits the
// credentials at the first ":", so such a user name would get 401 on
// every request to /metrics.
func TestMetricsUsernameMustNotContainColon(t *testing.T) {
t.Setenv("METRICS_USERNAME", "prom:etheus")
t.Setenv("METRICS_PASSWORD", "secret")
requireConfigError(t, "METRICS_USERNAME")
}
// TestCORSAllowedOriginsMustBeOrigins: "*" would let every origin in,
// and an entry that is not a plain origin would match no page.
func TestCORSAllowedOriginsMustBeOrigins(t *testing.T) {
for _, entry := range []string{
"*",
"https://*.netwatch.example",
"netwatch.example",
"https://netwatch.example/",
} {
t.Run(entry, func(t *testing.T) {
t.Setenv("CORS_ALLOWED_ORIGINS", entry)
requireConfigError(t, "CORS_ALLOWED_ORIGINS")
})
}
}
+4
View File
@@ -10,18 +10,22 @@ var (
Appname string
// Version is the git version tag.
Version string
// Buildarch is the build architecture.
Buildarch string
)
// Globals holds build-time metadata for the application.
type Globals struct {
Appname string
Version string
Buildarch string
}
// New creates a Globals instance from package-level variables.
func New(_ fx.Lifecycle) (*Globals, error) {
return &Globals{
Appname: Appname,
Buildarch: Buildarch,
Version: Version,
}, nil
}
-10
View File
@@ -1,10 +0,0 @@
package handlers
import "log/slog"
// NewForTest builds a Handlers around a report sink and logger,
// bypassing the fx graph so handler behaviour (including the
// storage failure path) is exercisable in unit tests.
func NewForTest(buf reportAppender, log *slog.Logger) *Handlers {
return &Handlers{buf: buf, log: log}
}
+1 -20
View File
@@ -18,13 +18,6 @@ import (
const jsonContentType = "application/json; charset=utf-8"
// reportAppender is the subset of the report buffer the handlers
// depend on. Defining it here keeps the storage failure path
// exercisable with a stub in tests.
type reportAppender interface {
Append(v any) error
}
// Params defines the dependencies for Handlers.
type Params struct {
fx.In
@@ -37,7 +30,7 @@ type Params struct {
// Handlers provides HTTP handler factories for all endpoints.
type Handlers struct {
buf reportAppender
buf *reportbuf.Buffer
hc *healthcheck.Healthcheck
log *slog.Logger
params *Params
@@ -79,15 +72,3 @@ func (s *Handlers) respondJSON(
}
}
}
// decodeJSON decodes the request body into v. The body is
// expected to already be bounded by the body-size middleware, so
// a caller can distinguish an over-limit body from malformed
// JSON by testing the returned error for *http.MaxBytesError.
func (s *Handlers) decodeJSON(
_ http.ResponseWriter,
r *http.Request,
v any,
) error {
return json.NewDecoder(r.Body).Decode(v)
}
@@ -0,0 +1,13 @@
package handlers_test
import (
"testing"
_ "sneak.berlin/go/netwatch/internal/handlers"
)
func TestImport(t *testing.T) {
t.Parallel()
// Compilation check — verifies the package parses
// and all imports resolve.
}
+1 -1
View File
@@ -6,6 +6,6 @@ import "net/http"
// endpoint.
func (s *Handlers) HandleHealthCheck() http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
s.respondJSON(w, r, s.hc.Healthcheck(), http.StatusOK)
s.respondJSON(w, r, s.hc.Check(), http.StatusOK)
}
}
@@ -1,120 +0,0 @@
package handlers_test
import (
"encoding/json"
"maps"
"net/http"
"net/http/httptest"
"slices"
"testing"
"time"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/handlers"
"sneak.berlin/go/netwatch/internal/healthcheck"
"sneak.berlin/go/netwatch/internal/logger"
"go.uber.org/fx/fxtest"
)
// newStartedHandlers builds Handlers with a real health check for the
// server named in g, and starts them, which records the time the
// uptime counts from.
func newStartedHandlers(t *testing.T, g *globals.Globals) *handlers.Handlers {
t.Helper()
lc := fxtest.NewLifecycle(t)
log, err := logger.New(lc, logger.Params{Globals: g})
if err != nil {
t.Fatalf("logger: %v", err)
}
hc, err := healthcheck.New(lc,
healthcheck.Params{Globals: g, Logger: log})
if err != nil {
t.Fatalf("health check: %v", err)
}
h, err := handlers.New(lc,
handlers.Params{Globals: g, Healthcheck: hc, Logger: log})
if err != nil {
t.Fatalf("handlers: %v", err)
}
lc.RequireStart()
t.Cleanup(lc.RequireStop)
return h
}
// TestHandleHealthCheck checks the health check's answer: 200, a JSON
// content type, and a JSON object with exactly the fields of
// healthcheck.HealthcheckResponse, carrying this server's name and
// version and an uptime counted from its start.
func TestHandleHealthCheck(t *testing.T) {
t.Parallel()
g := &globals.Globals{Appname: "netwatch-server", Version: "v1.2.3"}
h := newStartedHandlers(t, g)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/.well-known/healthcheck", http.NoBody)
h.HandleHealthCheck().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusOK)
}
contentType := rec.Header().Get("Content-Type")
if contentType != "application/json; charset=utf-8" {
t.Errorf("Content-Type = %q, want %q",
contentType, "application/json; charset=utf-8")
}
var body map[string]any
err := json.Unmarshal(rec.Body.Bytes(), &body)
if err != nil {
t.Fatalf("body not a JSON object: %v (%q)", err, rec.Body.String())
}
fields := []string{
"appname", "now", "status", "uptime_human", "uptime_seconds", "version",
}
if got := slices.Sorted(maps.Keys(body)); !slices.Equal(got, fields) {
t.Fatalf("fields = %v, want %v", got, fields)
}
for field, want := range map[string]string{
"appname": g.Appname, "status": "ok", "version": g.Version,
} {
if body[field] != want {
t.Errorf("%s = %v, want %q", field, body[field], want)
}
}
now, _ := body["now"].(string)
at, err := time.Parse(time.RFC3339Nano, now)
if err != nil || time.Since(at).Abs() > time.Minute {
t.Errorf("now = %q, want the current time in RFC 3339 (%v)", now, err)
}
// Started just now, so the uptime is well under a minute.
human, _ := body["uptime_human"].(string)
uptime, err := time.ParseDuration(human)
if err != nil || uptime > time.Minute {
t.Errorf("uptime_human = %q, want a duration under a minute (%v)",
human, err)
}
seconds, ok := body["uptime_seconds"].(float64)
if !ok || seconds < 0 || seconds > time.Minute.Seconds() {
t.Errorf("uptime_seconds = %v, want a number of seconds under a minute",
body["uptime_seconds"])
}
}
+29 -68
View File
@@ -2,13 +2,11 @@ package handlers
import (
"encoding/json"
"errors"
"net/http"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/reportbuf"
)
const maxReportBodyBytes = 1 << 20 // 1 MiB
type reportSample struct {
T int64 `json:"t"`
Latency *int `json:"latency"`
@@ -37,85 +35,48 @@ func (s *Handlers) HandleReport() http.HandlerFunc {
}
return func(w http.ResponseWriter, r *http.Request) {
r.Body = http.MaxBytesReader(
w, r.Body, maxReportBodyBytes,
)
var rpt report
err := s.decodeJSON(w, r, &rpt)
err := json.NewDecoder(r.Body).Decode(&rpt)
if err != nil {
s.respondJSON(w, r,
&response{Status: "error"},
s.decodeErrorStatus(err),
)
return
}
s.logReportReceived(rpt)
err = s.buf.Append(rpt)
if err != nil {
s.respondJSON(w, r,
&response{Status: "error"},
s.appendErrorStatus(err),
)
return
}
s.respondJSON(w, r, &response{Status: "ok"}, http.StatusOK)
}
}
// decodeErrorStatus logs a report decode failure and returns the
// status to send: 413 when the body exceeded the size limit,
// otherwise 400 for malformed JSON.
func (s *Handlers) decodeErrorStatus(err error) int {
var tooLarge *http.MaxBytesError
if errors.As(err, &tooLarge) {
s.log.Warn("report body too large", "limit_bytes", tooLarge.Limit)
return http.StatusRequestEntityTooLarge
}
// The decoder's error text can quote request bytes (a whole
// oversized number, for example), so it is bounded too.
s.log.Error("failed to decode report",
"error", logger.BoundedForLog(err.Error()),
"error", err,
)
s.respondJSON(w, r,
&response{Status: "error"},
http.StatusBadRequest,
)
return http.StatusBadRequest
return
}
// appendErrorStatus logs a failure to store a report and returns
// the status to send: 507 when the reports waiting to be written fill
// the size cap, otherwise 500.
func (s *Handlers) appendErrorStatus(err error) int {
if errors.Is(err, reportbuf.ErrFull) {
s.log.Warn("report refused: " +
"reports waiting to be written fill the size cap")
return http.StatusInsufficientStorage
}
s.log.Error("failed to buffer report", "error", err)
return http.StatusInternalServerError
}
// logReportReceived logs an accepted report. Untrusted fields are
// bounded (client_id, timestamp) or reduced to a length
// (geo_bytes) so the raw attacker-controlled body never reaches
// the log.
func (s *Handlers) logReportReceived(rpt report) {
totalSamples := 0
for _, h := range rpt.Hosts {
totalSamples += len(h.History)
}
s.log.Info("report received",
"client_id", logger.BoundedForLog(rpt.ClientID),
"timestamp", logger.BoundedForLog(rpt.Timestamp),
"client_id", rpt.ClientID,
"timestamp", rpt.Timestamp,
"host_count", len(rpt.Hosts),
"total_samples", totalSamples,
"geo_bytes", len(rpt.Geo),
"geo", string(rpt.Geo),
)
bufErr := s.buf.Append(rpt)
if bufErr != nil {
s.log.Error("failed to buffer report",
"error", bufErr,
)
}
s.respondJSON(w, r,
&response{Status: "ok"},
http.StatusOK,
)
}
}
-297
View File
@@ -1,297 +0,0 @@
package handlers_test
import (
"bytes"
"encoding/json"
"errors"
"io"
"log/slog"
"net/http"
"net/http/httptest"
"strings"
"testing"
"sneak.berlin/go/netwatch/internal/handlers"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/middleware"
"sneak.berlin/go/netwatch/internal/reportbuf"
)
var errStorageFailed = errors.New("storage failed")
// stubAppender drives the storage success/failure path without a
// real buffer or disk.
type stubAppender struct {
err error
}
func (s stubAppender) Append(any) error { return s.err }
func newTestHandlers(buf stubAppender, out io.Writer) *handlers.Handlers {
return handlers.NewForTest(buf, slog.New(slog.NewJSONHandler(out, nil)))
}
func decodeStatus(t *testing.T, body []byte) string {
t.Helper()
var resp struct {
Status string `json:"status"`
}
err := json.Unmarshal(body, &resp)
if err != nil {
t.Fatalf("response body not JSON: %v (%q)", err, body)
}
return resp.Status
}
// TestHandleReportAcceptsValidReports checks the answer to a valid
// report, one with no hosts and one shaped as the frontend sends them:
// 200 and {"status":"ok"} as JSON.
func TestHandleReportAcceptsValidReports(t *testing.T) {
t.Parallel()
tests := []struct {
name string
body string
}{
{
name: "no hosts",
body: `{"clientId":"c1","geo":null,"hosts":[],` +
`"timestamp":"2026-10-03T12:00:00.000Z"}`,
},
{
name: "a host with a latency and an error sample",
body: `{"clientId":"c1","geo":null,"hosts":[{` +
`"name":"Example","url":"https://example.com/",` +
`"status":"error","history":[` +
`{"t":1790000000000,"latency":42,"error":null},` +
`{"t":1790000003000,"latency":null,"error":"timeout"}]}],` +
`"timestamp":"2026-10-03T12:00:00.000Z"}`,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
h := newTestHandlers(stubAppender{}, io.Discard)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(tt.body),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusOK)
}
contentType := rec.Header().Get("Content-Type")
if contentType != "application/json; charset=utf-8" {
t.Errorf("Content-Type = %q, want %q",
contentType, "application/json; charset=utf-8")
}
if got := rec.Body.String(); got != "{\"status\":\"ok\"}\n" {
t.Errorf("body = %q, want %q", got, "{\"status\":\"ok\"}\n")
}
})
}
}
func TestHandleReportStorageFailureIsNon2xx(t *testing.T) {
t.Parallel()
h := newTestHandlers(stubAppender{err: errStorageFailed}, io.Discard)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{"clientId":"c1","hosts":[]}`),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code < 500 {
t.Fatalf("storage failure status = %d, want a 5xx", rec.Code)
}
if got := decodeStatus(t, rec.Body.Bytes()); got != "error" {
t.Fatalf("status field = %q, want %q", got, "error")
}
}
// TestHandleReportFullIs507 checks the answer when the reports waiting
// to be written fill the size cap: 507 and the usual error body, which
// tells the client nothing more.
func TestHandleReportFullIs507(t *testing.T) {
t.Parallel()
h := newTestHandlers(stubAppender{err: reportbuf.ErrFull}, io.Discard)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{"clientId":"c1","hosts":[]}`),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code != http.StatusInsufficientStorage {
t.Fatalf("status = %d, want %d",
rec.Code, http.StatusInsufficientStorage)
}
if got := rec.Body.String(); got != "{\"status\":\"error\"}\n" {
t.Errorf("body = %q, want %q", got, "{\"status\":\"error\"}\n")
}
}
func TestHandleReportMalformedJSONIs400(t *testing.T) {
t.Parallel()
h := newTestHandlers(stubAppender{}, io.Discard)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{not json`),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code != http.StatusBadRequest {
t.Fatalf("malformed status = %d, want %d",
rec.Code, http.StatusBadRequest)
}
}
func TestHandleReportOversizeIs413(t *testing.T) {
t.Parallel()
const limit = 32
h := newTestHandlers(stubAppender{}, io.Discard)
handler := (&middleware.Middleware{}).MaxBodyBytes(limit)(
h.HandleReport(),
)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{"clientId":"`+strings.Repeat("x", 200)+`"}`),
)
// No declared length, so only the middleware's read cap can
// stop this body.
req.ContentLength = -1
handler.ServeHTTP(rec, req)
if rec.Code != http.StatusRequestEntityTooLarge {
t.Fatalf("oversize status = %d, want %d",
rec.Code, http.StatusRequestEntityTooLarge)
}
}
func TestHandleReportDoesNotLogRawGeo(t *testing.T) {
t.Parallel()
const sentinel = "SENSITIVE-GEO-BLOB"
var logbuf bytes.Buffer
h := newTestHandlers(stubAppender{}, &logbuf)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(
`{"clientId":"c1","geo":{"raw":"`+sentinel+`"},"hosts":[]}`,
),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusOK)
}
if strings.Contains(logbuf.String(), sentinel) {
t.Fatal("raw geo bytes were written to the log")
}
if !strings.Contains(logbuf.String(), "geo_bytes") {
t.Fatal("expected a bounded geo_bytes field in the log")
}
}
func TestHandleReportLogsClientIDCutToBound(t *testing.T) {
t.Parallel()
long := strings.Repeat("c", 2*logger.MaxLoggedFieldBytes)
var logbuf bytes.Buffer
h := newTestHandlers(stubAppender{}, &logbuf)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(
`{"clientId":"`+long+`","timestamp":"`+long+`","hosts":[]}`,
),
)
h.HandleReport().ServeHTTP(rec, req)
var logged map[string]any
err := json.Unmarshal(logbuf.Bytes(), &logged)
if err != nil {
t.Fatalf("log line not JSON: %v (%q)", err, logbuf.String())
}
want := long[:logger.MaxLoggedFieldBytes]
if logged["client_id"] != want {
t.Fatalf("logged client_id not cut to %d bytes: %q",
logger.MaxLoggedFieldBytes, logged["client_id"])
}
if logged["timestamp"] != want {
t.Fatalf("logged timestamp not cut to %d bytes: %q",
logger.MaxLoggedFieldBytes, logged["timestamp"])
}
}
func TestHandleReportDecodeErrorLogIsBounded(t *testing.T) {
t.Parallel()
// A number too large for its int64 field makes the decoder's
// error text quote the whole number.
huge := strings.Repeat("9", 2*logger.MaxLoggedFieldBytes)
var logbuf bytes.Buffer
h := newTestHandlers(stubAppender{}, &logbuf)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{"hosts":[{"history":[{"t":`+huge+`}]}]}`),
)
h.HandleReport().ServeHTTP(rec, req)
if rec.Code != http.StatusBadRequest {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusBadRequest)
}
if strings.Contains(logbuf.String(), huge) {
t.Fatal("the whole oversized number was written to the log")
}
}
+8 -11
View File
@@ -30,17 +30,14 @@ type Healthcheck struct {
params *Params
}
// HealthcheckResponse is the JSON payload returned by the health
// check endpoint. Its name and its snake_case keys are the ones
// GO_HTTP_SERVER_CONVENTIONS.md gives.
//
//nolint:revive,tagliatelle // name and keys from the conventions
type HealthcheckResponse struct {
// Response is the JSON payload returned by the health check
// endpoint.
type Response struct {
Appname string `json:"appname"`
Now string `json:"now"`
Status string `json:"status"`
UptimeHuman string `json:"uptime_human"`
UptimeSeconds int64 `json:"uptime_seconds"`
UptimeHuman string `json:"uptimeHuman"`
UptimeSeconds int64 `json:"uptimeSeconds"`
Version string `json:"version"`
}
@@ -68,9 +65,9 @@ func New(
return s, nil
}
// Healthcheck returns the current health status of the application.
func (s *Healthcheck) Healthcheck() *HealthcheckResponse {
return &HealthcheckResponse{
// Check returns the current health status of the application.
func (s *Healthcheck) Check() *Response {
return &Response{
Appname: s.params.Globals.Appname,
Now: time.Now().UTC().Format(time.RFC3339Nano),
Status: "ok",
+1 -17
View File
@@ -5,28 +5,12 @@ package logger
import (
"log/slog"
"os"
"runtime"
"sneak.berlin/go/netwatch/internal/globals"
"go.uber.org/fx"
)
// MaxLoggedFieldBytes bounds untrusted text (request fields,
// header values, decode error text) before it is logged, so a
// caller cannot inflate log volume with an oversized value.
const MaxLoggedFieldBytes = 128
// BoundedForLog truncates an untrusted string to a fixed byte
// bound so an attacker-controlled field cannot dominate the log.
func BoundedForLog(s string) string {
if len(s) > MaxLoggedFieldBytes {
return s[:MaxLoggedFieldBytes]
}
return s
}
// Params defines the dependencies for Logger.
type Params struct {
fx.In
@@ -96,6 +80,6 @@ func (l *Logger) Identify() {
l.log.Info("starting",
"appname", l.params.Globals.Appname,
"version", l.params.Globals.Version,
"arch", runtime.GOARCH,
"buildarch", l.params.Globals.Buildarch,
)
}
@@ -1,31 +0,0 @@
package middleware
import (
"log/slog"
"net/http"
"net/netip"
)
// Test-only wrappers exposing unexported helpers to the
// external middleware_test package.
// NewWithLogger builds a Middleware around a logger for tests
// that exercise the logging paths without the fx graph.
func NewWithLogger(log *slog.Logger) *Middleware {
return &Middleware{log: log}
}
// NewWithTrustedProxies builds a Middleware that honours forwarded
// headers from the given networks, for tests of the client address
// paths without the fx graph.
func NewWithTrustedProxies(trusted []netip.Prefix) *Middleware {
return &Middleware{trustedProxies: trusted}
}
func ClientIP(
remoteAddr string,
header http.Header,
trusted []netip.Prefix,
) string {
return clientIP(remoteAddr, header, trusted)
}
+23 -287
View File
@@ -3,51 +3,22 @@
package middleware
import (
"errors"
"fmt"
"io"
"log/slog"
"net"
"net/http"
"net/netip"
"runtime/debug"
"strings"
"time"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/logger"
basicauth "github.com/99designs/basicauth-go"
"github.com/go-chi/chi/v5/middleware"
"github.com/go-chi/cors"
"github.com/go-chi/httprate"
"github.com/prometheus/client_golang/prometheus"
metrics "github.com/slok/go-http-metrics/metrics/prometheus"
ghmm "github.com/slok/go-http-metrics/middleware"
"github.com/slok/go-http-metrics/middleware/std"
"go.uber.org/fx"
)
const corsMaxAgeSec = 300
// jsonErrorBody is the body written for errors raised inside
// middleware, matching the {"status":"error"} shape the handlers
// return so clients see one error contract across the API.
const (
jsonContentType = "application/json; charset=utf-8"
jsonErrorBody = "{\"status\":\"error\"}\n"
)
// Security header values. The backend is a JSON API with no
// HTML surface, so the CSP forbids every resource type and
// framing outright.
const (
hstsValue = "max-age=31536000; includeSubDomains"
cspValue = "default-src 'none'; frame-ancestors 'none'"
permissionsPolicyValue = "camera=(), microphone=(), geolocation=()"
)
// Params defines the dependencies for Middleware.
type Params struct {
fx.In
@@ -61,7 +32,6 @@ type Params struct {
type Middleware struct {
log *slog.Logger
params *Params
trustedProxies []netip.Prefix
}
// New creates a Middleware instance.
@@ -69,40 +39,13 @@ func New(
_ fx.Lifecycle,
params Params,
) (*Middleware, error) {
trusted, err := ParseTrustedProxies(params.Config.TrustedProxies)
if err != nil {
return nil, err
}
s := new(Middleware)
s.params = &params
s.log = params.Logger.Get()
s.trustedProxies = trusted
return s, nil
}
// ParseTrustedProxies converts the TRUSTED_PROXIES entries into
// prefixes, failing fast on any malformed entry. Each entry must be
// a CIDR; a lone address is refused. "netwatch-server check-cidr"
// runs it too.
func ParseTrustedProxies(cidrs []string) ([]netip.Prefix, error) {
prefixes := make([]netip.Prefix, 0, len(cidrs))
for _, cidr := range cidrs {
prefix, err := netip.ParsePrefix(cidr)
if err != nil {
return nil, fmt.Errorf(
"TRUSTED_PROXIES %q: %w", cidr, err,
)
}
prefixes = append(prefixes, prefix.Masked())
}
return prefixes, nil
}
type loggingResponseWriter struct {
http.ResponseWriter
@@ -129,75 +72,8 @@ func ipFromHostPort(hostPort string) string {
return host
}
// clientIP resolves the caller's address. X-Forwarded-For and
// X-Real-IP are honoured only when the direct peer is a
// trusted proxy; otherwise the direct peer is returned so a
// spoofed header cannot forge the logged address.
func clientIP(
remoteAddr string,
header http.Header,
trusted []netip.Prefix,
) string {
peer := ipFromHostPort(remoteAddr)
if !addrInAny(peer, trusted) {
return peer
}
if xff := firstForwardedFor(header.Get("X-Forwarded-For")); xff != "" {
return xff
}
if xr := strings.TrimSpace(header.Get("X-Real-IP")); validIP(xr) {
return xr
}
return peer
}
// firstForwardedFor returns the left-most valid address in an
// X-Forwarded-For list (the original client), or "" if none.
func firstForwardedFor(value string) string {
for part := range strings.SplitSeq(value, ",") {
candidate := strings.TrimSpace(part)
if validIP(candidate) {
return candidate
}
}
return ""
}
func validIP(s string) bool {
_, err := netip.ParseAddr(s)
return err == nil
}
// addrInAny reports whether s parses as an address contained
// in any of the trusted prefixes.
func addrInAny(s string, trusted []netip.Prefix) bool {
addr, err := netip.ParseAddr(s)
if err != nil {
return false
}
addr = addr.Unmap()
for _, prefix := range trusted {
if prefix.Contains(addr) {
return true
}
}
return false
}
// Logging returns middleware that logs each request with
// timing, status code, and client information. Every string
// taken from the request is cut to logger.MaxLoggedFieldBytes,
// including the request ID, which chi takes from the client's
// X-Request-Id header when one is sent.
// timing, status code, and client information.
func (s *Middleware) Logging() func(http.Handler) http.Handler {
return func(next http.Handler) http.Handler {
return http.HandlerFunc(
@@ -210,19 +86,17 @@ func (s *Middleware) Logging() func(http.Handler) http.Handler {
latency := time.Since(start)
s.log.InfoContext(ctx, "request",
"request_start", start,
"method", logger.BoundedForLog(r.Method),
"url", logger.BoundedForLog(r.URL.String()),
"useragent", logger.BoundedForLog(r.UserAgent()),
"method", r.Method,
"url", r.URL.String(),
"useragent", r.UserAgent(),
"request_id",
logger.BoundedForLog(middleware.GetReqID(ctx)),
"referer", logger.BoundedForLog(r.Referer()),
"proto", logger.BoundedForLog(r.Proto),
ctx.Value(
middleware.RequestIDKey,
),
"referer", r.Referer(),
"proto", r.Proto,
"remote_ip",
logger.BoundedForLog(clientIP(
r.RemoteAddr,
r.Header,
s.trustedProxies,
)),
ipFromHostPort(r.RemoteAddr),
"status", lrw.statusCode,
"latency_ms",
latency.Milliseconds(),
@@ -235,159 +109,21 @@ func (s *Middleware) Logging() func(http.Handler) http.Handler {
}
}
// SecurityHeaders returns middleware that sets response
// security headers. It runs before CORS so the headers are
// present on preflight responses the CORS handler writes.
func (s *Middleware) SecurityHeaders() func(http.Handler) http.Handler {
return func(next http.Handler) http.Handler {
return http.HandlerFunc(
func(w http.ResponseWriter, r *http.Request) {
h := w.Header()
h.Set("Strict-Transport-Security", hstsValue)
h.Set("Content-Security-Policy", cspValue)
h.Set("X-Frame-Options", "DENY")
h.Set("X-Content-Type-Options", "nosniff")
h.Set("Referrer-Policy", "no-referrer")
h.Set("Permissions-Policy", permissionsPolicyValue)
next.ServeHTTP(w, r)
},
)
}
}
// writeJSONError writes the shared JSON error body with the
// given status. Used where middleware must reject a request
// before it reaches a handler.
func writeJSONError(w http.ResponseWriter, status int) {
w.Header().Set("Content-Type", jsonContentType)
w.WriteHeader(status)
_, _ = io.WriteString(w, jsonErrorBody)
}
// MaxBodyBytes returns middleware that caps the request body at
// limit bytes. A declared Content-Length over the limit is
// rejected immediately with 413. Bodies without a declared
// length (or that understate it) are capped as they are read, so
// a handler that reads the body sees a *http.MaxBytesError it can
// map to 413. Mounted again on a route group, it can only lower
// the limit: a cap applied earlier in the chain still holds.
func (s *Middleware) MaxBodyBytes(
limit int64,
) func(http.Handler) http.Handler {
return func(next http.Handler) http.Handler {
return http.HandlerFunc(
func(w http.ResponseWriter, r *http.Request) {
if r.ContentLength > limit {
writeJSONError(
w,
http.StatusRequestEntityTooLarge,
)
return
}
r.Body = http.MaxBytesReader(w, r.Body, limit)
next.ServeHTTP(w, r)
},
)
}
}
// Recoverer returns middleware that recovers from a panic in a
// downstream handler, logs the panic and stack trace through
// slog, and responds 500 with no body. http.ErrAbortHandler is
// re-panicked so the server can abort the response as intended.
func (s *Middleware) Recoverer() func(http.Handler) http.Handler {
return func(next http.Handler) http.Handler {
return http.HandlerFunc(
func(w http.ResponseWriter, r *http.Request) {
defer func() {
rec := recover()
if rec == nil {
return
}
err, ok := rec.(error)
if ok && errors.Is(err, http.ErrAbortHandler) {
panic(rec)
}
s.log.ErrorContext(r.Context(),
"panic recovered",
"panic", fmt.Sprintf("%v", rec),
"stack", string(debug.Stack()),
)
w.WriteHeader(http.StatusInternalServerError)
}()
next.ServeHTTP(w, r)
},
)
}
}
// CORS returns middleware that lets pages served from the given
// origins call the API. With no origins it adds no CORS headers at
// all, so only same-origin pages can use the API. That case must not
// reach cors.Handler, which treats an empty origin list as "allow
// every origin".
func (s *Middleware) CORS(
origins []string,
) func(http.Handler) http.Handler {
if len(origins) == 0 {
return func(next http.Handler) http.Handler { return next }
}
// CORS returns middleware that adds permissive CORS headers.
func (s *Middleware) CORS() func(http.Handler) http.Handler {
return cors.Handler(cors.Options{
AllowedOrigins: origins,
AllowedMethods: []string{http.MethodGet, http.MethodPost},
AllowedHeaders: []string{"Content-Type"},
AllowedOrigins: []string{"*"},
AllowedMethods: []string{
"GET", "POST", "PUT", "DELETE", "OPTIONS",
},
AllowedHeaders: []string{
"Accept",
"Authorization",
"Content-Type",
"X-CSRF-Token",
},
ExposedHeaders: []string{"Link"},
AllowCredentials: false,
MaxAge: corsMaxAgeSec,
})
}
// RateLimit returns middleware that allows each client address
// perMinute requests a minute and answers the rest with 429, the
// Retry-After header httprate sets, and the usual error body. The
// address is the one clientIP resolves, so clients behind the reverse
// proxy are limited one by one, not together as the proxy.
func (s *Middleware) RateLimit(
perMinute int,
) func(http.Handler) http.Handler {
return httprate.LimitBy(perMinute, time.Minute,
func(r *http.Request) (string, error) {
return clientIP(r.RemoteAddr, r.Header, s.trustedProxies), nil
},
httprate.WithLimitHandler(
func(w http.ResponseWriter, _ *http.Request) {
writeJSONError(w, http.StatusTooManyRequests)
},
),
)
}
// Metrics returns middleware that records each request's duration and
// response size, and the requests in progress, in registry. They are
// labelled by the request path.
func (s *Middleware) Metrics(
registry prometheus.Registerer,
) func(http.Handler) http.Handler {
mdlw := ghmm.New(ghmm.Config{
Recorder: metrics.NewRecorder(metrics.Config{Registry: registry}),
})
return std.HandlerProvider("", mdlw)
}
// MetricsAuth returns middleware that lets a request through only with
// METRICS_USERNAME and METRICS_PASSWORD as its basic auth credentials,
// and answers any other with 401.
func (s *Middleware) MetricsAuth() func(http.Handler) http.Handler {
return basicauth.New("metrics", map[string][]string{
s.params.Config.MetricsUsername: {s.params.Config.MetricsPassword},
})
}
@@ -1,571 +0,0 @@
package middleware_test
import (
"bytes"
"encoding/json"
"errors"
"log/slog"
"net/http"
"net/http/httptest"
"net/netip"
"strings"
"testing"
"testing/synctest"
"time"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/middleware"
chimiddleware "github.com/go-chi/chi/v5/middleware"
)
const (
// loopbackPeer is a remote address inside the trusted-proxy allowlist.
loopbackPeer = "127.0.0.1:5000"
// forwardedIP is the client address presented via X-Forwarded-For.
forwardedIP = "203.0.113.7"
// realIP is the client address presented via X-Real-IP.
realIP = "203.0.113.9"
)
func mustPrefixes(t *testing.T, cidrs ...string) []netip.Prefix {
t.Helper()
prefixes, err := middleware.ParseTrustedProxies(cidrs)
if err != nil {
t.Fatalf("ParseTrustedProxies(%v): %v", cidrs, err)
}
return prefixes
}
// TestParseTrustedProxiesRejectsMalformed includes entries nginx would
// read as another address or look up as a hostname, in the CIDR form
// bin/entrypoint.sh gives "netwatch-server check-cidr".
func TestParseTrustedProxiesRejectsMalformed(t *testing.T) {
t.Parallel()
for _, cidr := range []string{
"not-a-cidr", "10.0.0.1", "1.2.3/32", "172.30/32", "10/32",
"cafe/32", "999.1.1.1/32", "10.0.0.0/33", "::1/129",
"fe80::1%eth0/128",
} {
_, err := middleware.ParseTrustedProxies([]string{cidr})
if err == nil || !strings.Contains(err.Error(), "TRUSTED_PROXIES") {
t.Errorf("%q: error = %v, want one naming TRUSTED_PROXIES",
cidr, err)
}
}
}
func TestParseTrustedProxiesAcceptsCIDRs(t *testing.T) {
t.Parallel()
mustPrefixes(t, "172.17.0.1/32", "10.0.0.0/8", "2001:db8::1/128",
"2001:db8::/32", "::ffff:192.0.2.1/128")
}
type clientIPCase struct {
name string
remoteAddr string
xff string
xRealIP string
want string
}
func clientIPCases() []clientIPCase {
return []clientIPCase{
{
name: "trusted proxy uses forwarded-for",
remoteAddr: loopbackPeer,
xff: forwardedIP,
want: forwardedIP,
},
{
name: "trusted proxy uses left-most of chain",
remoteAddr: "10.1.2.3:5000",
xff: forwardedIP + ", 10.1.2.3",
want: forwardedIP,
},
{
name: "trusted proxy falls back to x-real-ip",
remoteAddr: loopbackPeer,
xRealIP: realIP,
want: realIP,
},
{
name: "untrusted peer ignores forwarded-for",
remoteAddr: "198.51.100.4:5000",
xff: forwardedIP,
want: "198.51.100.4",
},
{
name: "untrusted peer ignores x-real-ip",
remoteAddr: "198.51.100.4:5000",
xRealIP: realIP,
want: "198.51.100.4",
},
{
name: "trusted proxy with no headers uses peer",
remoteAddr: "10.1.2.3:5000",
want: "10.1.2.3",
},
{
name: "trusted proxy with garbage header uses peer",
remoteAddr: loopbackPeer,
xff: "not-an-ip",
want: "127.0.0.1",
},
}
}
func TestClientIP(t *testing.T) {
t.Parallel()
trusted := mustPrefixes(t, "127.0.0.1/32", "::1/128", "10.0.0.0/8")
for _, tc := range clientIPCases() {
t.Run(tc.name, func(t *testing.T) {
t.Parallel()
header := http.Header{}
if tc.xff != "" {
header.Set("X-Forwarded-For", tc.xff)
}
if tc.xRealIP != "" {
header.Set("X-Real-IP", tc.xRealIP)
}
got := middleware.ClientIP(tc.remoteAddr, header, trusted)
if got != tc.want {
t.Errorf("ClientIP() = %q, want %q", got, tc.want)
}
})
}
}
func TestSecurityHeaders(t *testing.T) {
t.Parallel()
handler := (&middleware.Middleware{}).SecurityHeaders()(
http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
w.WriteHeader(http.StatusOK)
}),
)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/", http.NoBody)
handler.ServeHTTP(rec, req)
want := map[string]string{
"Strict-Transport-Security": "max-age=31536000; includeSubDomains",
"Content-Security-Policy": "default-src 'none'; frame-ancestors 'none'",
"X-Frame-Options": "DENY",
"X-Content-Type-Options": "nosniff",
"Referrer-Policy": "no-referrer",
"Permissions-Policy": "camera=(), microphone=(), geolocation=()",
}
for name, value := range want {
if got := rec.Header().Get(name); got != value {
t.Errorf("header %s = %q, want %q", name, got, value)
}
}
}
// TestMaxBodyBytesRejectsOversizeOnNonReadingRoute confirms the
// limit is enforced even for a handler that never reads the body
// (for example the health check), via the Content-Length check.
func TestMaxBodyBytesRejectsOversizeOnNonReadingRoute(t *testing.T) {
t.Parallel()
const limit = 16
called := false
handler := (&middleware.Middleware{}).MaxBodyBytes(limit)(
http.HandlerFunc(func(_ http.ResponseWriter, _ *http.Request) {
called = true
}),
)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/.well-known/healthcheck",
strings.NewReader(strings.Repeat("x", limit+1)),
)
handler.ServeHTTP(rec, req)
if rec.Code != http.StatusRequestEntityTooLarge {
t.Fatalf("status = %d, want %d",
rec.Code, http.StatusRequestEntityTooLarge)
}
if called {
t.Fatal("handler ran despite oversize body")
}
if got := rec.Body.String(); got != "{\"status\":\"error\"}\n" {
t.Errorf("body = %q, want %q", got, "{\"status\":\"error\"}\n")
}
got := rec.Header().Get("Content-Type")
if got != "application/json; charset=utf-8" {
t.Errorf("Content-Type = %q, want a JSON content type", got)
}
}
func TestMaxBodyBytesAllowsWithinLimit(t *testing.T) {
t.Parallel()
const limit = 64
handler := (&middleware.Middleware{}).MaxBodyBytes(limit)(
http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
w.WriteHeader(http.StatusOK)
}),
)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{"clientId":"c1"}`),
)
handler.ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusOK)
}
}
func TestRecovererReturns500AndLogsThroughSlog(t *testing.T) {
t.Parallel()
var logbuf bytes.Buffer
mw := middleware.NewWithLogger(
slog.New(slog.NewJSONHandler(&logbuf, nil)),
)
handler := mw.Recoverer()(
http.HandlerFunc(func(_ http.ResponseWriter, _ *http.Request) {
panic("boom")
}),
)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/", http.NoBody)
handler.ServeHTTP(rec, req)
if rec.Code != http.StatusInternalServerError {
t.Fatalf("status = %d, want %d",
rec.Code, http.StatusInternalServerError)
}
var record map[string]any
err := json.Unmarshal(logbuf.Bytes(), &record)
if err != nil {
t.Fatalf("panic log is not one JSON record: %v (%q)", err, logbuf.String())
}
if record["msg"] != "panic recovered" || record["level"] != "ERROR" {
t.Errorf("log record = %v, want msg %q at level ERROR",
record, "panic recovered")
}
if record["panic"] != "boom" {
t.Errorf("panic field = %v, want %q", record["panic"], "boom")
}
stack, _ := record["stack"].(string)
if !strings.HasPrefix(stack, "goroutine ") {
t.Errorf("stack field = %q, want a stack trace", stack)
}
}
// TestRecovererRepanicsOnAbortHandler checks that a handler aborting
// with http.ErrAbortHandler is not treated as a crash: Recoverer
// panics again so the server aborts the response, and logs nothing.
func TestRecovererRepanicsOnAbortHandler(t *testing.T) {
t.Parallel()
var logbuf bytes.Buffer
mw := middleware.NewWithLogger(
slog.New(slog.NewJSONHandler(&logbuf, nil)),
)
handler := mw.Recoverer()(
http.HandlerFunc(func(_ http.ResponseWriter, _ *http.Request) {
panic(http.ErrAbortHandler)
}),
)
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/", http.NoBody)
var recovered any
func() {
defer func() { recovered = recover() }()
handler.ServeHTTP(rec, req)
}()
err, _ := recovered.(error)
if !errors.Is(err, http.ErrAbortHandler) {
t.Errorf("Recoverer panicked with %v, want http.ErrAbortHandler", recovered)
}
if logbuf.Len() != 0 {
t.Errorf("abort was logged: %q", logbuf.String())
}
}
// TestLoggingCutsRequestStringsToBound sends an over-long URL and
// over-long header values, and checks the request log writes each
// one cut to logger.MaxLoggedFieldBytes.
func TestLoggingCutsRequestStringsToBound(t *testing.T) {
t.Parallel()
long := strings.Repeat("a", 2*logger.MaxLoggedFieldBytes)
var logbuf bytes.Buffer
mw := middleware.NewWithLogger(
slog.New(slog.NewJSONHandler(&logbuf, nil)),
)
handler := chimiddleware.RequestID(mw.Logging()(okHandler()))
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/"+long, http.NoBody)
req.Header.Set("User-Agent", long)
req.Header.Set("Referer", long)
req.Header.Set("X-Request-Id", long)
handler.ServeHTTP(httptest.NewRecorder(), req)
var logged map[string]any
err := json.Unmarshal(logbuf.Bytes(), &logged)
if err != nil {
t.Fatalf("log line not JSON: %v (%q)", err, logbuf.String())
}
want := map[string]string{
"url": ("/" + long)[:logger.MaxLoggedFieldBytes],
"useragent": long[:logger.MaxLoggedFieldBytes],
"referer": long[:logger.MaxLoggedFieldBytes],
"request_id": long[:logger.MaxLoggedFieldBytes],
}
for field, value := range want {
if logged[field] != value {
t.Errorf("logged %s = %q, want it cut to %d bytes",
field, logged[field], logger.MaxLoggedFieldBytes)
}
}
}
// okHandler stands in for the route a middleware guards.
func okHandler() http.Handler {
return http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
w.WriteHeader(http.StatusOK)
})
}
// TestRateLimitRefusesPastAllowanceThenResets checks one client
// address: it may use its whole allowance at once, the next request
// is refused with 429, and later it may send again.
func TestRateLimitRefusesPastAllowanceThenResets(t *testing.T) {
t.Parallel()
// synctest runs this on a fake clock: time.Sleep returns at once,
// with the clock moved on.
synctest.Test(t, func(t *testing.T) {
const perMinute = 2
handler := (&middleware.Middleware{}).RateLimit(perMinute)(okHandler())
post := func() *httptest.ResponseRecorder {
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports", http.NoBody)
handler.ServeHTTP(rec, req)
return rec
}
for i := range perMinute {
if code := post().Code; code != http.StatusOK {
t.Fatalf("request %d: status = %d, want %d",
i+1, code, http.StatusOK)
}
}
rec := post()
if rec.Code != http.StatusTooManyRequests {
t.Fatalf("request past the allowance: status = %d, want %d",
rec.Code, http.StatusTooManyRequests)
}
if got := rec.Body.String(); got != "{\"status\":\"error\"}\n" {
t.Errorf("body = %q, want %q", got, "{\"status\":\"error\"}\n")
}
if got := rec.Header().Get("Retry-After"); got != "60" {
t.Fatalf("Retry-After = %q, want %q", got, "60")
}
// httprate also counts the previous minute's requests, fading
// them out over the current one, so two minutes on the whole
// allowance is back.
time.Sleep(2 * time.Minute)
for i := range perMinute {
if code := post().Code; code != http.StatusOK {
t.Fatalf("two minutes later, request %d: status = %d, want %d",
i+1, code, http.StatusOK)
}
}
})
}
// postForwarded sends handler a report from peer that names client in
// X-Forwarded-For, and returns the status.
func postForwarded(
t *testing.T,
handler http.Handler,
peer, client string,
) int {
t.Helper()
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports", http.NoBody)
req.RemoteAddr = peer
req.Header.Set("X-Forwarded-For", client)
handler.ServeHTTP(rec, req)
return rec.Code
}
// TestRateLimitIsPerForwardedClient checks that clients behind a
// trusted proxy each get their own allowance: the limit is keyed on
// the client address clientIP resolves, not on the proxy's.
func TestRateLimitIsPerForwardedClient(t *testing.T) {
t.Parallel()
const otherClient = "203.0.113.8"
mw := middleware.NewWithTrustedProxies(mustPrefixes(t, "127.0.0.1/32"))
handler := mw.RateLimit(1)(okHandler())
code := postForwarded(t, handler, loopbackPeer, forwardedIP)
if code != http.StatusOK {
t.Fatalf("first request: status = %d, want %d", code, http.StatusOK)
}
code = postForwarded(t, handler, loopbackPeer, forwardedIP)
if code != http.StatusTooManyRequests {
t.Fatalf("same client again: status = %d, want %d",
code, http.StatusTooManyRequests)
}
code = postForwarded(t, handler, loopbackPeer, otherClient)
if code != http.StatusOK {
t.Fatalf("other client behind the same proxy: status = %d, want %d",
code, http.StatusOK)
}
}
// TestRateLimitIgnoresForwardedForFromUntrustedPeer checks that a
// peer that is not a trusted proxy cannot get a fresh allowance by
// naming a different client in X-Forwarded-For on each request.
func TestRateLimitIgnoresForwardedForFromUntrustedPeer(t *testing.T) {
t.Parallel()
const untrustedPeer = "198.51.100.4:5000"
mw := middleware.NewWithTrustedProxies(mustPrefixes(t, "127.0.0.1/32"))
handler := mw.RateLimit(1)(okHandler())
code := postForwarded(t, handler, untrustedPeer, "203.0.113.8")
if code != http.StatusOK {
t.Fatalf("first request: status = %d, want %d", code, http.StatusOK)
}
code = postForwarded(t, handler, untrustedPeer, "203.0.113.9")
if code != http.StatusTooManyRequests {
t.Fatalf("same peer naming another client: status = %d, want %d",
code, http.StatusTooManyRequests)
}
}
// preflight sends cors the preflight request a browser makes before
// it POSTs JSON from origin.
func preflight(
t *testing.T,
cors func(http.Handler) http.Handler,
origin string,
) *httptest.ResponseRecorder {
t.Helper()
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodOptions, "/api/v1/reports", http.NoBody)
req.Header.Set("Origin", origin)
req.Header.Set("Access-Control-Request-Method", http.MethodPost)
req.Header.Set("Access-Control-Request-Headers", "content-type")
cors(okHandler()).ServeHTTP(rec, req)
return rec
}
// TestCORSWithoutOriginsAddsNoHeaders checks the default: with no
// origins configured, no origin is given any CORS header.
func TestCORSWithoutOriginsAddsNoHeaders(t *testing.T) {
t.Parallel()
rec := preflight(t,
(&middleware.Middleware{}).CORS(nil), "https://elsewhere.example")
for name := range rec.Header() {
if strings.HasPrefix(name, "Access-Control-") {
t.Errorf("CORS header %s set with no origins configured", name)
}
}
}
func TestCORSAllowsOnlyListedOrigins(t *testing.T) {
t.Parallel()
const listed = "https://netwatch.example"
cors := (&middleware.Middleware{}).CORS([]string{listed})
cases := []struct {
name string
origin string
want string
}{
{name: "listed origin allowed", origin: listed, want: listed},
{name: "other origin refused", origin: "https://elsewhere.example"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
t.Parallel()
rec := preflight(t, cors, tc.origin)
got := rec.Header().Get("Access-Control-Allow-Origin")
if got != tc.want {
t.Errorf("Access-Control-Allow-Origin = %q, want %q",
got, tc.want)
}
})
}
}
-150
View File
@@ -1,150 +0,0 @@
package reportbuf
import (
"errors"
"io/fs"
"os"
"os/user"
"path/filepath"
"strconv"
"syscall"
)
// ErrDataDirOutsideVolume is returned by PrepareDataDir for a DATA_DIR
// that is not the volume or a path below it, written in full.
var ErrDataDirOutsideVolume = errors.New(
"must be /data or a path below it, with no '.', '..' or extra '/'")
// PrepareDataDir gets dir, the server's DATA_DIR, ready for owner, the
// user the server runs as, so that a host directory mounted at volume,
// /data in the image, needs no preparing: it creates dir, gives volume
// and everything in it to owner, and gives volume and dir the mode the
// server gives a directory it creates. dir must be volume or a path
// below it, with no '.', '..', empty part or '/' at the end.
//
// bin/entrypoint.sh runs this as root, which would follow a symbolic
// link anywhere, so every step goes through an os.Root opened on
// volume: it follows a link only when it is written as a relative
// path that stays inside volume, and refuses any other. A process on
// the host can swap a link onto a path in volume at any moment while
// this runs. Even then, the os.Root checks each link as it reaches
// it. MkdirAll creates each directory inside a parent it already has
// open, never following a link at the name it creates, and follows a
// link on the path only as the os.Root allows, so a relative link
// inside volume can lead it to create directories elsewhere inside
// volume. Lchown never changes what a link points to, and the modes
// are set on directories already opened (see chmodDir), so the most
// that process can do is make a step fail or wait, or act on
// something else inside volume.
func PrepareDataDir(volume, dir string, owner *user.User) error {
// rel is dir as a path from volume; IsLocal is false for one that
// leads out of it.
rel, err := filepath.Rel(volume, dir)
if err != nil || dir != filepath.Clean(dir) || !filepath.IsLocal(rel) {
return ErrDataDirOutsideVolume
}
uid, err := strconv.Atoi(owner.Uid)
if err != nil {
return err
}
gid, err := strconv.Atoi(owner.Gid)
if err != nil {
return err
}
root, err := os.OpenRoot(volume)
if err != nil {
return err
}
defer func() { _ = root.Close() }()
err = root.MkdirAll(rel, dirPerms)
if err != nil {
return err
}
err = lchownAll(root, ".", uid, gid)
if err != nil {
return err
}
err = chmodDir(root, ".")
if err != nil {
return err
}
return chmodDir(root, rel)
}
// lchownAll gives name, a directory inside root, and everything in it
// to uid and gid. It reads each directory opened through root, not
// through root.FS(), which refuses a name that is not valid UTF-8, and
// calls Lchown on every entry, which gives a symbolic link itself to
// them, not what it points to. It goes into an entry only when the
// read found a directory there, so it follows no link it finds; one
// swapped in for that directory afterwards is followed only as the
// os.Root allows.
func lchownAll(root *os.Root, name string, uid, gid int) error {
err := root.Lchown(name, uid, gid)
if err != nil {
return err
}
dir, err := root.Open(name)
if err != nil {
return err
}
entries, err := dir.ReadDir(-1)
_ = dir.Close()
if err != nil {
return err
}
for _, entry := range entries {
entryName := filepath.Join(name, entry.Name())
if entry.IsDir() {
err = lchownAll(root, entryName, uid, gid)
} else {
err = root.Lchown(entryName, uid, gid)
}
if err != nil {
return err
}
}
return nil
}
// chmodDir gives name, a directory inside root, the mode the server
// gives a directory it creates. Root.Chmod would not hold: on Linux it
// checks that name is not a symbolic link, then sets the mode by name,
// following a link swapped in between. So chmodDir opens name through
// root and sets the mode on the open directory. It refuses anything
// but a directory: a directory has no second name (hard link), so the
// one opened is inside root, where any other file could be a hard link
// to one outside.
func chmodDir(root *os.Root, name string) error {
dir, err := root.Open(name)
if err != nil {
return err
}
defer func() { _ = dir.Close() }()
info, err := dir.Stat()
if err != nil {
return err
}
if !info.IsDir() {
return &fs.PathError{Op: "chmod", Path: name, Err: syscall.ENOTDIR}
}
return dir.Chmod(dirPerms)
}
-307
View File
@@ -1,307 +0,0 @@
package reportbuf_test
import (
"errors"
"io/fs"
"os"
"os/user"
"path/filepath"
"strconv"
"syscall"
"testing"
"sneak.berlin/go/netwatch/internal/reportbuf"
)
// reports is the last part of DATA_DIR in these tests, as in the
// image's /data/reports.
const reports = "reports"
// currentUser is the user the test runs as, the only owner a test not
// run as root can give files to.
func currentUser() *user.User {
return &user.User{
Uid: strconv.Itoa(os.Getuid()),
Gid: strconv.Itoa(os.Getgid()),
}
}
// tempDirMode700 is a new directory in a t.TempDir with mode 0700, so
// a test can tell that PrepareDataDir left its mode alone.
func tempDirMode700(t *testing.T) string {
t.Helper()
dir := filepath.Join(t.TempDir(), "d")
err := os.Mkdir(dir, 0o700)
if err != nil {
t.Fatal(err)
}
return dir
}
func requireMode(t *testing.T, path string, want fs.FileMode) {
t.Helper()
info, err := os.Stat(path)
if err != nil {
t.Fatal(err)
}
if info.Mode() != want {
t.Errorf("%s: mode %v, want %v", path, info.Mode(), want)
}
}
func requireMissing(t *testing.T, path string) {
t.Helper()
_, err := os.Lstat(path)
if !errors.Is(err, fs.ErrNotExist) {
t.Errorf("%s: created, or Lstat failed: %v", path, err)
}
}
func requireOwner(t *testing.T, path string, uid, gid uint32) {
t.Helper()
info, err := os.Lstat(path)
if err != nil {
t.Fatal(err)
}
stat, _ := info.Sys().(*syscall.Stat_t)
if stat.Uid != uid || stat.Gid != gid {
t.Errorf("%s: owner %d:%d, want %d:%d", path, stat.Uid, stat.Gid,
uid, gid)
}
}
func TestPrepareDataDirCreatesDataDir(t *testing.T) {
t.Parallel()
volume := t.TempDir()
dir := filepath.Join(volume, "a", reports)
err := reportbuf.PrepareDataDir(volume, dir, currentUser())
if err != nil {
t.Fatal(err)
}
requireMode(t, volume, fs.ModeDir|0o750)
requireMode(t, dir, fs.ModeDir|0o750)
}
// TestPrepareDataDirSetsModeOfExistingDataDir: a DATA_DIR already on
// the host with another mode gets the mode too, not only a new one.
func TestPrepareDataDirSetsModeOfExistingDataDir(t *testing.T) {
t.Parallel()
volume := t.TempDir()
dir := filepath.Join(volume, reports)
err := os.Mkdir(dir, 0o700)
if err != nil {
t.Fatal(err)
}
err = reportbuf.PrepareDataDir(volume, dir, currentUser())
if err != nil {
t.Fatal(err)
}
requireMode(t, dir, fs.ModeDir|0o750)
}
func TestPrepareDataDirTakesTheVolumeItself(t *testing.T) {
t.Parallel()
volume := t.TempDir()
err := reportbuf.PrepareDataDir(volume, volume, currentUser())
if err != nil {
t.Fatal(err)
}
requireMode(t, volume, fs.ModeDir|0o750)
}
// TestPrepareDataDirRefusesDataDirOutsideVolume covers a DATA_DIR that
// is relative, outside the volume, or not written in full.
func TestPrepareDataDirRefusesDataDirOutsideVolume(t *testing.T) {
t.Parallel()
volume := tempDirMode700(t)
for _, dir := range []string{
reports, volume + "/../new", volume + "/", volume + "//" + reports,
volume + "/./" + reports, volume + "/" + reports + "/..", volume + "x",
"/etc",
} {
err := reportbuf.PrepareDataDir(volume, dir, currentUser())
if !errors.Is(err, reportbuf.ErrDataDirOutsideVolume) {
t.Errorf("%q: error = %v, want ErrDataDirOutsideVolume", dir, err)
}
}
requireMissing(t, filepath.Join(filepath.Dir(volume), "new"))
requireMode(t, volume, fs.ModeDir|0o700)
}
// TestPrepareDataDirRefusesLinkOutOfVolume puts a symbolic link to a
// directory outside the volume on the path to DATA_DIR, written as a
// full path and as one that climbs out with '..', and as DATA_DIR
// itself, where the mode would be set through it.
func TestPrepareDataDirRefusesLinkOutOfVolume(t *testing.T) {
t.Parallel()
for _, tc := range []struct {
name string
climbsOut bool
link, dir string
}{
{"full path", false, "x", "x/reports"},
{"climbs out", true, "x", "x/reports"},
{"DATA_DIR itself", false, reports, reports},
} {
t.Run(tc.name, func(t *testing.T) {
t.Parallel()
outside := tempDirMode700(t)
volume := t.TempDir()
climbOut, err := filepath.Rel(volume, outside)
if err != nil {
t.Fatal(err)
}
target := outside
if tc.climbsOut {
target = climbOut
}
err = os.Symlink(target, filepath.Join(volume, tc.link))
if err != nil {
t.Fatal(err)
}
err = reportbuf.PrepareDataDir(volume,
filepath.Join(volume, tc.dir), currentUser())
if err == nil {
t.Error("no error")
}
requireMissing(t, filepath.Join(outside, reports))
requireMode(t, outside, fs.ModeDir|0o700)
})
}
}
// TestPrepareDataDirRefusesDanglingLink: DATA_DIR is a symbolic link
// to a name in the volume that does not exist, which is not created.
func TestPrepareDataDirRefusesDanglingLink(t *testing.T) {
t.Parallel()
volume := t.TempDir()
err := os.Symlink("missing", filepath.Join(volume, reports))
if err != nil {
t.Fatal(err)
}
err = reportbuf.PrepareDataDir(volume, filepath.Join(volume, reports),
currentUser())
if err == nil {
t.Error("no error")
}
requireMissing(t, filepath.Join(volume, "missing"))
}
// TestPrepareDataDirTakesDirectoryNamedInLatin1: a host directory can
// hold names that are not valid UTF-8, here "café" written in Latin-1.
// A test not run as root can only check that PrepareDataDir goes into
// such a directory and gives it, and what it holds, to the current
// user.
func TestPrepareDataDirTakesDirectoryNamedInLatin1(t *testing.T) {
t.Parallel()
volume := t.TempDir()
latin1 := filepath.Join(volume, "caf\xe9")
err := os.Mkdir(latin1, 0o700)
if err != nil {
t.Fatal(err)
}
err = os.WriteFile(filepath.Join(latin1, "f"), nil, 0o600)
if err != nil {
t.Fatal(err)
}
owner := currentUser()
err = reportbuf.PrepareDataDir(volume, filepath.Join(volume, reports),
owner)
if err != nil {
t.Fatal(err)
}
uid, _ := strconv.ParseUint(owner.Uid, 10, 32)
gid, _ := strconv.ParseUint(owner.Gid, 10, 32)
requireOwner(t, latin1, uint32(uid), uint32(gid))
requireOwner(t, filepath.Join(latin1, "f"), uint32(uid), uint32(gid))
}
// TestPrepareDataDirGivesVolumeToOwner gives everything in the volume
// to a uid and gid that own nothing, which only root can do. A
// symbolic link in the volume to a directory outside it is given to
// them itself; what it points to is left as it was.
func TestPrepareDataDirGivesVolumeToOwner(t *testing.T) {
t.Parallel()
if os.Geteuid() != 0 {
t.Skip("only root can give files to another uid")
}
outside := t.TempDir()
volume := t.TempDir()
old := filepath.Join(volume, "old")
err := os.WriteFile(filepath.Join(outside, "f"), nil, 0o600)
if err != nil {
t.Fatal(err)
}
err = os.Mkdir(old, 0o700)
if err != nil {
t.Fatal(err)
}
err = os.WriteFile(filepath.Join(old, "f"), nil, 0o600)
if err != nil {
t.Fatal(err)
}
err = os.Symlink(outside, filepath.Join(old, "link"))
if err != nil {
t.Fatal(err)
}
err = reportbuf.PrepareDataDir(volume, filepath.Join(volume, reports),
&user.User{Uid: "4242", Gid: "4343"})
if err != nil {
t.Fatal(err)
}
for _, path := range []string{
volume, filepath.Join(volume, reports), old,
filepath.Join(old, "f"), filepath.Join(old, "link"),
} {
requireOwner(t, path, 4242, 4343)
}
requireOwner(t, outside, 0, 0)
requireOwner(t, filepath.Join(outside, "f"), 0, 0)
}
-29
View File
@@ -1,29 +0,0 @@
package reportbuf
import (
"os"
"time"
)
// FlushSizeThreshold exposes the buffer size at which Append starts
// writing a report file to the external tests.
const FlushSizeThreshold = flushSizeThreshold
// Flush writes the buffered reports to a file now, as the periodic
// flush does, so tests need not wait a minute for it.
func (b *Buffer) Flush() error {
return b.flushLocked()
}
// StopClock makes every report file the buffer writes from now on
// carry the timestamp at, as if all were written in one millisecond.
func (b *Buffer) StopClock(at time.Time) {
b.now = func() time.Time { return at }
}
// OnFileCreated makes the buffer call fn with each report file it
// writes from now on, once the file is created and before anything is
// written to it.
func (b *Buffer) OnFileCreated(fn func(f *os.File)) {
b.fileCreated = fn
}
+32 -252
View File
@@ -1,22 +1,17 @@
// Package reportbuf accumulates telemetry reports in memory
// and periodically flushes them to zstd-compressed JSONL files,
// deleting the oldest files to keep them under a size cap.
// and periodically flushes them to zstd-compressed JSONL files.
package reportbuf
import (
"bytes"
"context"
"encoding/json"
"errors"
"fmt"
"io/fs"
"log/slog"
"os"
"path/filepath"
"slices"
"strings"
"sync"
"sync/atomic"
"time"
"sneak.berlin/go/netwatch/internal/config"
@@ -32,18 +27,8 @@ const (
defaultDataDir = "./data/reports"
dirPerms fs.FileMode = 0o750
filePerms fs.FileMode = 0o640
// Report files are named filePrefix + timestamp + "-" + number +
// fileSuffix; see writeFile.
filePrefix = "reports-"
fileSuffix = ".jsonl.zst"
)
// ErrFull is returned by Append when the reports waiting to be
// written leave no room for the report under the configured maximum
// size, however many report files are deleted.
var ErrFull = errors.New("reports waiting to be written fill the size cap")
// Params defines the dependencies for Buffer.
type Params struct {
fx.In
@@ -52,45 +37,14 @@ type Params struct {
Logger *logger.Logger
}
// reportFile is a report file that may be deleted to make room, with
// the size it counts for in usedBytes.
type reportFile struct {
name string
size int64
}
// Buffer accumulates JSON lines in memory and flushes them
// to zstd-compressed files on disk.
type Buffer struct {
buf bytes.Buffer
dataDir string
done chan struct{}
// fileCreated is called with each report file once it is
// created, before anything is written to it: it does nothing,
// except in tests that hold the write open or make it fail.
fileCreated func(f *os.File)
// files are the report files that may be deleted to make room,
// in name order, which is oldest first: those in dataDir at
// start, and each one this buffer writes, put in at its place by
// name once it is complete, even when an older file's write
// completes after a newer one's. A file still being written is
// not among them. filesBytes is their total size.
files []reportFile
filesBytes int64
log *slog.Logger
maxBytes int64
mu sync.Mutex
// now is the clock report files are named by: time.Now, except
// in tests that need two flushes to share a timestamp.
now func() time.Time
// seq numbers the report files, so that two named in the same
// millisecond still get different names.
seq atomic.Uint64
stopOnce sync.Once
// usedBytes is what Append checks against maxBytes: the size
// of the report files in dataDir, plus the reports waiting to
// be written to one at their uncompressed size.
usedBytes int64
}
// New creates a Buffer and registers lifecycle hooks to
@@ -107,10 +61,7 @@ func New(
b := &Buffer{
dataDir: dir,
done: make(chan struct{}),
fileCreated: func(*os.File) {},
log: params.Logger.Get(),
maxBytes: params.Config.DataDirMaxBytes,
now: time.Now,
}
lc.Append(fx.Hook{
@@ -120,46 +71,15 @@ func New(
return fmt.Errorf("create data dir: %w", err)
}
// Report files left by earlier runs count too, and are
// the first to be deleted to make room.
files, err := reportFiles(b.dataDir)
if err != nil {
return err
}
b.mu.Lock()
b.files = files
for _, f := range files {
b.filesBytes += f.size
}
b.usedBytes = b.filesBytes
// The files may be past the cap, if it was lowered since
// the last run.
b.deleteOldestFiles(0)
b.mu.Unlock()
go b.flushLoop()
return nil
},
OnStop: func(_ context.Context) error {
// stopOnce makes OnStop idempotent: a second
// invocation must not close an already-closed channel
// (which would panic) or flush again.
var err error
b.stopOnce.Do(func() {
close(b.done)
err = b.flushLocked()
})
b.flushLocked()
// A failed final flush fails the stop, so the process
// exits non-zero.
return err
return nil
},
})
@@ -167,30 +87,15 @@ func New(
}
// Append marshals v as a single JSON line and appends it to
// the buffer. If the line would take usedBytes past maxBytes, the
// oldest report files are deleted to make room; it stores nothing
// and returns ErrFull if that cannot make room. If the buffer
// reaches the size threshold, it is drained and written to disk
// asynchronously.
// the buffer. If the buffer reaches the size threshold, it is
// drained and written to disk asynchronously.
func (b *Buffer) Append(v any) error {
line, err := json.Marshal(v)
if err != nil {
return fmt.Errorf("marshal report: %w", err)
}
lineBytes := int64(len(line)) + 1 // with its newline
b.mu.Lock()
b.deleteOldestFiles(lineBytes)
if b.usedBytes+lineBytes > b.maxBytes {
b.mu.Unlock()
return ErrFull
}
b.usedBytes += lineBytes
b.buf.Write(line)
b.buf.WriteByte('\n')
@@ -198,12 +103,7 @@ func (b *Buffer) Append(v any) error {
data := b.drainBuf()
b.mu.Unlock()
go func() {
writeErr := b.writeFile(data)
if writeErr != nil {
b.log.Error("flush reports failed", "error", writeErr)
}
}()
go b.writeFile(data)
return nil
}
@@ -213,37 +113,6 @@ func (b *Buffer) Append(v any) error {
return nil
}
// deleteOldestFiles deletes report files, oldest first, until n more
// bytes fit under maxBytes. It deletes none when the reports waiting
// to be written leave no room for n even with every file gone, since
// that would lose the files for nothing. The caller must hold b.mu.
func (b *Buffer) deleteOldestFiles(n int64) {
for b.usedBytes+n > b.maxBytes && len(b.files) > 0 {
if b.usedBytes-b.filesBytes+n > b.maxBytes {
return
}
f := b.files[0]
b.files = b.files[1:]
b.filesBytes -= f.size
// A file already gone, deleted by hand, has freed its room too.
err := os.Remove(filepath.Join(b.dataDir, f.name))
if err != nil && !errors.Is(err, fs.ErrNotExist) {
// The file is still there, so it still counts. It is
// not tried again until the next start.
b.log.Error("delete report file failed",
"file", f.name, "error", err)
continue
}
b.usedBytes -= f.size
b.log.Info("deleted report file to make room",
"file", f.name, "bytes", f.size)
}
}
// flushLoop runs a ticker that periodically flushes buffered
// data to disk until the done channel is closed.
func (b *Buffer) flushLoop() {
@@ -253,10 +122,7 @@ func (b *Buffer) flushLoop() {
for {
select {
case <-ticker.C:
err := b.flushLocked()
if err != nil {
b.log.Error("flush reports failed", "error", err)
}
b.flushLocked()
case <-b.done:
return
}
@@ -265,19 +131,19 @@ func (b *Buffer) flushLoop() {
// flushLocked acquires the lock, drains the buffer, and
// writes the data to a compressed file.
func (b *Buffer) flushLocked() error {
func (b *Buffer) flushLocked() {
b.mu.Lock()
if b.buf.Len() == 0 {
b.mu.Unlock()
return nil
return
}
data := b.drainBuf()
b.mu.Unlock()
return b.writeFile(data)
b.writeFile(data)
}
// drainBuf copies the buffer contents and resets it.
@@ -290,130 +156,44 @@ func (b *Buffer) drainBuf() []byte {
return data
}
// writeFile writes data, reports drained from the buffer, to a new
// timestamped zstd-compressed JSONL file in the data directory.
func (b *Buffer) writeFile(data []byte) error {
// The timestamp comes first, so the names sort by time; the number
// after it tells apart files named in the same millisecond.
ts := b.now().UTC().Format("2006-01-02T15-04-05.000Z")
name := fmt.Sprintf("%s%s-%d%s", filePrefix, ts, b.seq.Add(1), fileSuffix)
// writeFile creates a timestamped zstd-compressed JSONL file
// in the data directory.
func (b *Buffer) writeFile(data []byte) {
ts := time.Now().UTC().Format("2006-01-02T15-04-05.000Z")
name := fmt.Sprintf("reports-%s.jsonl.zst", ts)
path := filepath.Join(b.dataDir, name)
size, err := b.createFile(filepath.Join(b.dataDir, name), data)
// The reports no longer wait to be written, so they stop counting
// at their uncompressed size. If the write failed they are lost;
// otherwise they count as the file, which from here on may be
// deleted to make room.
b.mu.Lock()
defer b.mu.Unlock()
b.usedBytes -= int64(len(data))
if err != nil {
return err
}
b.usedBytes += size
b.filesBytes += size
// At its place by name, not at the end: another write, of a newer
// file, may have completed while this one was being written.
i, _ := slices.BinarySearchFunc(b.files, name,
func(f reportFile, target string) int {
return strings.Compare(f.name, target)
})
b.files = slices.Insert(b.files, i, reportFile{name: name, size: size})
return nil
}
// createFile creates the file at path holding data compressed with
// zstd, and returns its size. If the write fails once the file is
// created, it removes the file, so that a failed write leaves nothing
// behind to take room.
func (b *Buffer) createFile(path string, data []byte) (int64, error) {
// path is built from the operator-supplied dataDir plus a
// generated timestamp and number, so it carries no external input.
f, err := os.OpenFile( //nolint:gosec // see comment above
f, err := os.OpenFile( //nolint:gosec // path built from controlled dataDir + timestamp
path,
os.O_WRONLY|os.O_CREATE|os.O_EXCL,
filePerms,
)
if err != nil {
return 0, fmt.Errorf("create report file: %w", err)
b.log.Error("create report file", "error", err)
return
}
b.fileCreated(f)
defer func() { _ = f.Close() }()
size, err := writeCompressed(f, data)
if err != nil {
_ = f.Close()
return 0, errors.Join(err, os.Remove(path))
}
err = f.Close()
if err != nil {
err = fmt.Errorf("close report file: %w", err)
return 0, errors.Join(err, os.Remove(path))
}
return size, nil
}
// writeCompressed writes data to f compressed with zstd, and returns
// the size of f.
func writeCompressed(f *os.File, data []byte) (int64, error) {
enc, err := zstd.NewWriter(f)
if err != nil {
return 0, fmt.Errorf("create zstd encoder: %w", err)
b.log.Error("create zstd encoder", "error", err)
return
}
_, err = enc.Write(data)
if err != nil {
_, writeErr := enc.Write(data)
if writeErr != nil {
b.log.Error("write compressed data", "error", writeErr)
_ = enc.Close()
return 0, fmt.Errorf("write compressed data: %w", err)
return
}
err = enc.Close()
if err != nil {
return 0, fmt.Errorf("close zstd encoder: %w", err)
closeErr := enc.Close()
if closeErr != nil {
b.log.Error("close zstd encoder", "error", closeErr)
}
info, err := f.Stat()
if err != nil {
return 0, fmt.Errorf("stat report file: %w", err)
}
return info.Size(), nil
}
// reportFiles returns the report files in dir, oldest first:
// os.ReadDir sorts them by name, and the names sort by time.
func reportFiles(dir string) ([]reportFile, error) {
entries, err := os.ReadDir(dir)
if err != nil {
return nil, fmt.Errorf("read data dir: %w", err)
}
files := make([]reportFile, 0, len(entries))
for _, entry := range entries {
name := entry.Name()
if !strings.HasPrefix(name, filePrefix) ||
!strings.HasSuffix(name, fileSuffix) {
continue
}
info, err := entry.Info()
if err != nil {
return nil, fmt.Errorf("stat report file: %w", err)
}
files = append(files, reportFile{name: name, size: info.Size()})
}
return files, nil
}
+5 -911
View File
@@ -1,919 +1,13 @@
package reportbuf_test
import (
"encoding/json"
"errors"
"fmt"
"io/fs"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
"sync"
"sync/atomic"
"testing"
"time"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/reportbuf"
"github.com/klauspost/compress/zstd"
"go.uber.org/fx"
"go.uber.org/fx/fxtest"
_ "sneak.berlin/go/netwatch/internal/reportbuf"
)
// TestFlushOnShutdown proves the flush-on-shutdown path: a
// report appended after start but before the periodic flush
// window must reach disk when the fx lifecycle stops. This is
// the exact case that silent data loss on restart used to
// destroy.
func TestFlushOnShutdown(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
var buf *reportbuf.Buffer
app := fxtest.New(t,
fx.Provide(
globals.New,
logger.New,
config.New,
reportbuf.New,
),
fx.Populate(&buf),
)
app.RequireStart()
err := buf.Append(map[string]string{"probe": "shutdown"})
if err != nil {
t.Fatalf("append report: %v", err)
}
// RequireStop runs the reportbuf OnStop hook, which is the
// only code path that flushes buffered reports on shutdown.
app.RequireStop()
if !hasReportFile(t, dir) {
t.Fatal("no report file on disk after shutdown; " +
"the buffered report was lost")
}
}
// TestFailedFinalFlushFailsStop proves a final flush that cannot
// write its file makes the stop fail, which makes the process
// exit non-zero instead of dropping the buffered reports silently.
func TestFailedFinalFlushFailsStop(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
var buf *reportbuf.Buffer
app := fxtest.New(t,
fx.Provide(
globals.New,
logger.New,
config.New,
reportbuf.New,
),
fx.Populate(&buf),
)
app.RequireStart()
err := buf.Append(map[string]string{"probe": "shutdown"})
if err != nil {
t.Fatalf("append report: %v", err)
}
// Removing the data directory leaves the final flush nowhere to
// write. A read-only directory would not do: tests run as root
// in the backend image, and root ignores the read-only bit.
err = os.RemoveAll(dir)
if err != nil {
t.Fatalf("remove data dir: %v", err)
}
err = app.Stop(t.Context())
if !errors.Is(err, fs.ErrNotExist) {
t.Fatalf("stop error = %v, want the final flush's error", err)
}
}
// startBuffer starts a Buffer through fx, as main does, with the
// DATA_DIR and DATA_DIR_MAX_BYTES the calling test has set.
func startBuffer(t *testing.T) *reportbuf.Buffer {
t.Helper()
var buf *reportbuf.Buffer
app := fxtest.New(t,
fx.Provide(
globals.New,
logger.New,
config.New,
reportbuf.New,
),
fx.Populate(&buf),
)
app.RequireStart()
t.Cleanup(app.RequireStop)
return buf
}
// lineBytes is what one report takes in the buffer: its JSON and a
// newline.
func lineBytes(t *testing.T, report any) int {
t.Helper()
line, err := json.Marshal(report)
if err != nil {
t.Fatalf("marshal report: %v", err)
}
return len(line) + 1
}
// TestAppendPastCapIsRefused fills the cap with a report not yet
// written. The next report is refused, and the report file already in
// DATA_DIR is kept: it is smaller than a report, so deleting it could
// not make room.
func TestAppendPastCapIsRefused(t *testing.T) {
report := map[string]string{"id": "cap"}
dir := t.TempDir()
earlier := reportFilePath(dir, 1)
writeBytes(t, earlier, 1)
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(1+lineBytes(t, report)))
buf := startBuffer(t)
err := buf.Append(report)
if err != nil {
t.Fatalf("report that fills the cap exactly: %v", err)
}
err = buf.Append(report)
if !errors.Is(err, reportbuf.ErrFull) {
t.Fatalf("report past the cap: error = %v, want ErrFull", err)
}
if !exists(t, earlier) {
t.Fatal("report file deleted, though that could not make room")
}
}
// TestOldestReportFileDeletedFirst starts on a data directory holding
// report files from an earlier run, and a file that is not a report,
// which neither counts nor is ever deleted. Nothing is deleted while
// there is room; then only the oldest report file is.
func TestOldestReportFileDeletedFirst(t *testing.T) {
const fileBytes = 100
report := map[string]string{"id": "oldest"}
dir := t.TempDir()
oldest := reportFilePath(dir, 1)
kept := []string{
reportFilePath(dir, 2),
reportFilePath(dir, 3),
filepath.Join(dir, "notes.txt"),
}
writeBytes(t, oldest, fileBytes)
for _, path := range kept {
writeBytes(t, path, fileBytes)
}
t.Setenv("DATA_DIR", dir)
// Room for the three report files and one report.
t.Setenv("DATA_DIR_MAX_BYTES",
strconv.Itoa(3*fileBytes+lineBytes(t, report)))
buf := startBuffer(t)
err := buf.Append(report)
if err != nil {
t.Fatalf("report that fills the cap exactly: %v", err)
}
if !exists(t, oldest) {
t.Fatal("oldest report file deleted while there was room")
}
err = buf.Append(report)
if err != nil {
t.Fatalf("report past the cap: %v", err)
}
if exists(t, oldest) {
t.Fatal("oldest report file kept when room was needed")
}
for _, path := range kept {
if !exists(t, path) {
t.Fatalf("%s deleted; only the oldest report file should be", path)
}
}
}
// TestStartDeletesFilesPastCap starts on report files past the cap,
// as after the cap is lowered: the oldest are deleted until the rest
// fit.
func TestStartDeletesFilesPastCap(t *testing.T) {
const fileBytes = 100
dir := t.TempDir()
oldest := reportFilePath(dir, 1)
kept := []string{reportFilePath(dir, 2), reportFilePath(dir, 3)}
writeBytes(t, oldest, fileBytes)
for _, path := range kept {
writeBytes(t, path, fileBytes)
}
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(len(kept)*fileBytes))
startBuffer(t)
if exists(t, oldest) {
t.Fatal("oldest report file kept, though the files were past the cap")
}
for _, path := range kept {
if !exists(t, path) {
t.Fatalf("%s deleted, though the rest fit without it", path)
}
}
}
// TestWrittenReportsCountAtFileSize checks that once reports are
// written, they count as their compressed file, not their
// uncompressed size, which frees room under the cap.
func TestWrittenReportsCountAtFileSize(t *testing.T) {
// Repetitive, so its file is far smaller than its JSON.
report := map[string]string{"id": strings.Repeat("a", 1000)}
size := lineBytes(t, report)
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
// Room for the report twice over only if the first one counts
// at its file's size by the time the second arrives.
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(2*size-1))
buf := startBuffer(t)
err := buf.Append(report)
if err != nil {
t.Fatalf("first report: %v", err)
}
err = buf.Flush()
if err != nil {
t.Fatalf("flush: %v", err)
}
err = buf.Append(report)
if err != nil {
t.Fatalf("second report, after the first was written: %v", err)
}
if !hasReportFile(t, dir) {
t.Fatal("first report's file deleted, though the second fit beside it")
}
}
// TestWrittenFilesDeletedToMakeRoom writes one report file after
// another under a small cap. Every report must be taken; files are
// deleted only when the report would not fit beside them, and the
// files kept leave room for it.
func TestWrittenFilesDeletedToMakeRoom(t *testing.T) {
const (
maxBytes = 200
// One file each, which take far more than maxBytes together.
reports = 50
)
report := map[string]string{"id": "written"}
size := int64(lineBytes(t, report))
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(maxBytes))
buf := startBuffer(t)
for range reports {
before := reportFilesBytes(t, dir)
err := buf.Append(report)
if err != nil {
t.Fatalf("with %d bytes of report files: %v", before, err)
}
after := reportFilesBytes(t, dir)
if before+size <= maxBytes && after != before {
t.Fatalf("files deleted, though the report fit beside "+
"their %d bytes", before)
}
if after+size > maxBytes {
t.Fatalf("%d bytes of report files kept, leaving no room "+
"for the report", after)
}
err = buf.Flush()
if err != nil {
t.Fatalf("flush: %v", err)
}
}
}
// TestFailedDeletionStillCounts makes deleting the oldest report file
// fail. It is still there, so it still takes room, and the next oldest
// is deleted in its place.
func TestFailedDeletionStillCounts(t *testing.T) {
const fileBytes = 100
report := map[string]string{"id": "stuck"}
dir := t.TempDir()
stuck := reportFilePath(dir, 1)
next := reportFilePath(dir, 2)
newest := reportFilePath(dir, 3)
for _, path := range []string{stuck, next, newest} {
writeBytes(t, path, fileBytes)
}
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(3*fileBytes))
buf := startBuffer(t)
// A directory that is not empty cannot be deleted, even by root,
// which the tests run as in the backend image.
err := os.Remove(stuck)
if err != nil {
t.Fatalf("remove %s: %v", stuck, err)
}
err = os.Mkdir(stuck, 0o750)
if err != nil {
t.Fatalf("make directory %s: %v", stuck, err)
}
writeBytes(t, filepath.Join(stuck, "file"), 1)
err = buf.Append(report)
if err != nil {
t.Fatalf("report past the cap: %v", err)
}
if exists(t, next) {
t.Fatal("next oldest report file kept: the failed deletion " +
"counted as making room")
}
if !exists(t, newest) {
t.Fatal("newest report file deleted, though deleting one made room")
}
}
// TestFileDeletedByHandFreesRoom deletes the oldest report file by
// hand after start. When room is needed, its room counts as freed, so
// no other file is deleted.
func TestFileDeletedByHandFreesRoom(t *testing.T) {
const fileBytes = 100
report := map[string]string{"id": "by-hand"}
dir := t.TempDir()
gone := reportFilePath(dir, 1)
kept := reportFilePath(dir, 2)
writeBytes(t, gone, fileBytes)
writeBytes(t, kept, fileBytes)
t.Setenv("DATA_DIR", dir)
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(2*fileBytes))
buf := startBuffer(t)
err := os.Remove(gone)
if err != nil {
t.Fatalf("remove %s: %v", gone, err)
}
err = buf.Append(report)
if err != nil {
t.Fatalf("report past the cap: %v", err)
}
if !exists(t, kept) {
t.Fatal("report file deleted, though the one deleted by hand " +
"had made room")
}
}
// TestFileBeingWrittenIsNeverDeleted holds the write of one report file
// open while a second write completes, then sends a report that needs
// room. Deleting either file would make it, and the one being written is
// the older, but only the complete one may be deleted. Once the first
// write is complete, its file is deleted when room is needed.
func TestFileBeingWrittenIsNeverDeleted(t *testing.T) {
report := map[string]string{"id": "writing"}
t.Setenv("DATA_DIR", t.TempDir())
// Room for two reports waiting to be written, but not for two
// beside a report file.
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(2*lineBytes(t, report)))
buf := startBuffer(t)
created := make(chan string)
release := make(chan struct{})
buf.OnFileCreated(func(f *os.File) {
created <- f.Name()
<-release
})
err := buf.Append(report)
if err != nil {
t.Fatalf("first report: %v", err)
}
flushed := make(chan error)
go func() { flushed <- buf.Flush() }()
writing := <-created
// Only the first write is held; the second goes through, and
// so do the writes after it, the final one at stop included.
var complete string
buf.OnFileCreated(func(f *os.File) { complete = f.Name() })
// Errorf, not Fatalf, until the first write is released, so that a
// failure here does not leave it held.
err = buf.Append(report)
if err != nil {
t.Errorf("second report: %v", err)
}
err = buf.Flush()
if err != nil {
t.Errorf("flush of the second report: %v", err)
}
err = buf.Append(report)
if err != nil {
t.Errorf("report that needs room: %v", err)
}
if !exists(t, writing) {
t.Error("report file deleted while it was being written")
}
if exists(t, complete) {
t.Error("complete report file kept, though room was needed")
}
close(release)
err = <-flushed
if err != nil {
t.Fatalf("flush of the first report: %v", err)
}
err = buf.Append(report)
if err != nil {
t.Fatalf("report after the first write was complete: %v", err)
}
if exists(t, writing) {
t.Fatal("complete report file kept when room was needed")
}
}
// TestFailedWriteStopsCounting makes a write fail once its file is
// created. Its reports are lost, so they stop counting, and the part of
// the file written is removed, so it takes no room.
func TestFailedWriteStopsCounting(t *testing.T) {
report := map[string]string{"id": "failed"}
t.Setenv("DATA_DIR", t.TempDir())
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(lineBytes(t, report)))
buf := startBuffer(t)
var failed string
// Closing the file under the write makes the write fail.
buf.OnFileCreated(func(f *os.File) {
failed = f.Name()
_ = f.Close()
})
err := buf.Append(report)
if err != nil {
t.Fatalf("report that fills the cap exactly: %v", err)
}
err = buf.Flush()
if err == nil {
t.Fatal("flush succeeded, though its file was closed under it")
}
if exists(t, failed) {
t.Fatal("file of the failed write kept")
}
// Writes from here on, the final one at stop included, succeed.
buf.OnFileCreated(func(*os.File) {})
err = buf.Append(report)
if err != nil {
t.Fatalf("report after the failed write: %v", err)
}
}
// TestConcurrentAppendsStopAtCap appends from many goroutines at once
// with room for exactly roomFor reports: exactly that many must be
// taken, which holds only if Append checks and counts each report
// under one lock.
func TestConcurrentAppendsStopAtCap(t *testing.T) {
const (
roomFor = 5
senders = 50
)
// Large, so each Append takes long enough for the senders to
// overlap while the cap is reached.
report := map[string]string{"id": strings.Repeat("a", 1_000_000)}
t.Setenv("DATA_DIR", t.TempDir())
t.Setenv("DATA_DIR_MAX_BYTES",
strconv.Itoa(roomFor*lineBytes(t, report)))
buf := startBuffer(t)
var (
taken atomic.Int64
wg sync.WaitGroup
)
start := make(chan struct{})
for range senders {
wg.Go(func() {
<-start
err := buf.Append(report)
if err == nil {
taken.Add(1)
} else if !errors.Is(err, reportbuf.ErrFull) {
t.Errorf("append: %v", err)
}
})
}
close(start)
wg.Wait()
if got := taken.Load(); got != roomFor {
t.Fatalf("%d reports taken, want %d", got, roomFor)
}
}
// TestTwoFlushesInOneMillisecond flushes twice within one millisecond,
// as a flush for size and the final flush at shutdown can: each flush
// must write a file of its own, and the files must hold every report.
func TestTwoFlushesInOneMillisecond(t *testing.T) {
const flushes = 2
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
buf := startBuffer(t)
buf.StopClock(time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC))
for id := 1; id <= flushes; id++ {
err := buf.Append(map[string]int{"id": id})
if err != nil {
t.Fatalf("append report %d: %v", id, err)
}
err = buf.Flush()
if err != nil {
t.Fatalf("flush %d: %v", id, err)
}
}
files, err := readReportFiles(dir)
if err != nil {
t.Fatalf("read report files: %v", err)
}
if len(files) != flushes {
t.Fatalf("%d report files after %d flushes", len(files), flushes)
}
for id := 1; id <= flushes; id++ {
want := fmt.Sprintf(`{"id":%d}`+"\n", id)
if !slices.Contains(files, want) {
t.Fatalf("no report file holds report %d alone", id)
}
}
}
// TestReportFileHoldsTheLinesAppended flushes three reports and reads
// their file back: it must decompress to exactly their JSON lines, in
// the order they were appended.
func TestReportFileHoldsTheLinesAppended(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
buf := startBuffer(t)
for _, id := range []int{1, 2, 3} {
err := buf.Append(map[string]int{"id": id})
if err != nil {
t.Fatalf("append report %d: %v", id, err)
}
}
err := buf.Flush()
if err != nil {
t.Fatalf("flush: %v", err)
}
files, err := readReportFiles(dir)
if err != nil {
t.Fatalf("read report files: %v", err)
}
want := `{"id":1}` + "\n" + `{"id":2}` + "\n" + `{"id":3}` + "\n"
if len(files) != 1 || files[0] != want {
t.Fatalf("report files = %q, want one holding %q", files, want)
}
}
// TestFlushAtSizeThreshold appends reports until the buffer holds
// FlushSizeThreshold bytes. The append that gets it there must write
// them all to one report file, with no call to Flush and the periodic
// flush a minute away, and no earlier append may write one.
func TestFlushAtSizeThreshold(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
buf := startBuffer(t)
// Large reports, so the threshold takes a few hundred appends.
pad := strings.Repeat("a", 64<<10)
var appended strings.Builder
for id := 0; appended.Len() < reportbuf.FlushSizeThreshold; id++ {
report := map[string]any{"id": id, "pad": pad}
err := buf.Append(report)
if err != nil {
t.Fatalf("append report %d: %v", id, err)
}
line, err := json.Marshal(report)
if err != nil {
t.Fatalf("marshal report %d: %v", id, err)
}
appended.Write(line)
appended.WriteByte('\n')
}
// Append writes the file in the background, so wait for it.
deadline := time.Now().Add(10 * time.Second)
for {
files, err := readReportFiles(dir)
if err == nil && len(files) == 1 && files[0] == appended.String() {
return
}
if time.Now().After(deadline) {
t.Fatalf("%d report files (error: %v), want one holding the "+
"%d bytes appended", len(files), err, appended.Len())
}
time.Sleep(10 * time.Millisecond)
}
}
// reportFilesBytes returns the total size of the report files in dir.
func reportFilesBytes(t *testing.T, dir string) int64 {
t.Helper()
paths, err := filepath.Glob(filepath.Join(dir, "reports-*.jsonl.zst"))
if err != nil {
t.Fatalf("list report files: %v", err)
}
var total int64
for _, path := range paths {
info, statErr := os.Stat(path)
if statErr != nil {
t.Fatalf("stat %s: %v", path, statErr)
}
total += info.Size()
}
return total
}
// readReportFiles returns the decompressed contents of each report
// file in dir. A file still being written does not decompress, so it
// gives an error.
func readReportFiles(dir string) ([]string, error) {
files := os.DirFS(dir)
names, err := fs.Glob(files, "reports-*.jsonl.zst")
if err != nil {
return nil, fmt.Errorf("list report files: %w", err)
}
dec, err := zstd.NewReader(nil)
if err != nil {
return nil, fmt.Errorf("create zstd decoder: %w", err)
}
defer dec.Close()
contents := make([]string, 0, len(names))
for _, name := range names {
compressed, readErr := fs.ReadFile(files, name)
if readErr != nil {
return nil, fmt.Errorf("read %s: %w", name, readErr)
}
data, decErr := dec.DecodeAll(compressed, nil)
if decErr != nil {
return nil, fmt.Errorf("decompress %s: %w", name, decErr)
}
contents = append(contents, string(data))
}
return contents, nil
}
// reportFilePath returns the path in dir of a report file named as
// written on the given day of January 2026, so that a lower day sorts
// as older.
func reportFilePath(dir string, day int) string {
return filepath.Join(dir,
fmt.Sprintf("reports-2026-01-%02dT00-00-00.000Z-1.jsonl.zst", day))
}
func writeBytes(t *testing.T, path string, n int) {
t.Helper()
err := os.WriteFile(path, make([]byte, n), 0o600)
if err != nil {
t.Fatalf("write %s: %v", path, err)
}
}
func exists(t *testing.T, path string) bool {
t.Helper()
_, err := os.Stat(path)
if errors.Is(err, fs.ErrNotExist) {
return false
}
if err != nil {
t.Fatalf("stat %s: %v", path, err)
}
return true
}
func hasReportFile(t *testing.T, dir string) bool {
t.Helper()
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatalf("read data dir: %v", err)
}
for _, e := range entries {
if strings.HasSuffix(e.Name(), ".jsonl.zst") {
info, statErr := e.Info()
if statErr != nil {
t.Fatalf("stat %s: %v", e.Name(), statErr)
}
if info.Size() > 0 {
return true
}
}
}
return false
}
// TestFilesDeletedOldestFirstWhenWritesOverlap holds the write of an
// older report file open until a newer one's write completes, then
// releases it. When room is needed, the older file is deleted first,
// though its write was the last to complete.
func TestFilesDeletedOldestFirstWhenWritesOverlap(t *testing.T) {
const maxBytes = 1000
report := map[string]string{"id": "overlap"}
t.Setenv("DATA_DIR", t.TempDir())
t.Setenv("DATA_DIR_MAX_BYTES", strconv.Itoa(maxBytes))
buf := startBuffer(t)
created := make(chan string)
release := make(chan struct{})
buf.OnFileCreated(func(f *os.File) {
created <- f.Name()
<-release
})
err := buf.Append(report)
if err != nil {
t.Fatalf("older report: %v", err)
}
flushed := make(chan error)
go func() { flushed <- buf.Flush() }()
older := <-created
// Only the older write is held; the newer one goes through, and
// so do the writes after it, the final one at stop included.
var newer string
buf.OnFileCreated(func(f *os.File) { newer = f.Name() })
// Errorf, not Fatalf, until the older write is released, so that a
// failure here does not leave it held.
err = buf.Append(report)
if err != nil {
t.Errorf("newer report: %v", err)
}
err = buf.Flush()
if err != nil {
t.Errorf("flush of the newer report: %v", err)
}
close(release)
err = <-flushed
if err != nil {
t.Fatalf("flush of the older report: %v", err)
}
info, err := os.Stat(newer)
if err != nil {
t.Fatalf("stat %s: %v", newer, err)
}
// A report that fits beside the newer file alone, so deleting the
// older one makes exactly the room it needs.
pad := maxBytes - int(info.Size()) - lineBytes(t, map[string]string{"id": ""})
err = buf.Append(map[string]string{"id": strings.Repeat("a", pad)})
if err != nil {
t.Fatalf("report that needs room: %v", err)
}
if exists(t, older) {
t.Fatal("older report file kept when room was needed")
}
if !exists(t, newer) {
t.Fatal("newer report file deleted before the older one")
}
func TestImport(t *testing.T) {
t.Parallel()
// Compilation check — verifies the package parses
// and all imports resolve.
}
-23
View File
@@ -1,23 +0,0 @@
package server
import "github.com/go-chi/chi/v5"
// Router exposes the router to the external tests, which add routes
// of their own to it after SetupRoutes.
func (s *Server) Router() *chi.Mux {
return s.router
}
// MaxRequestBodyBytes exposes the router-wide body limit to the
// external tests.
const MaxRequestBodyBytes = maxRequestBodyBytes
// MetricsRequestsPerMinute exposes the /metrics rate limit to the
// external tests.
const MetricsRequestsPerMinute = metricsRequestsPerMinute
// ListenAddr exposes the address the server listens on to the
// external tests.
func (s *Server) ListenAddr() string {
return s.newHTTPServer().Addr
}
+13 -41
View File
@@ -2,70 +2,42 @@ package server
import (
"errors"
"net"
"fmt"
"net/http"
"runtime"
"strconv"
"time"
"go.uber.org/fx"
)
const (
readTimeout = 10 * time.Second
readHeaderTimeout = 5 * time.Second
idleTimeout = 60 * time.Second
writeTimeout = 10 * time.Second
maxHeaderBytes = 1 << 20 // 1 MiB
// requestTimeout (routes.go) is the single per-request
// processing budget, enforced by chi's middleware.Timeout.
// writeTimeout must exceed that budget so a handler can write
// its 503 when the chi timeout fires; if it were shorter the
// server would abort the write first and the chi budget would
// be unreachable dead configuration.
writeTimeout = requestTimeout + 5*time.Second
)
// newHTTPServer constructs the http.Server. It performs no I/O
// and does not start listening.
func (s *Server) newHTTPServer() *http.Server {
listenAddr := net.JoinHostPort(
s.params.Config.BindAddress,
strconv.Itoa(s.params.Config.Port),
)
func (s *Server) serveUntilShutdown() {
listenAddr := fmt.Sprintf(":%d", s.params.Config.Port)
return &http.Server{
s.httpServer = &http.Server{
Addr: listenAddr,
Handler: s,
MaxHeaderBytes: maxHeaderBytes,
ReadTimeout: readTimeout,
ReadHeaderTimeout: readHeaderTimeout,
WriteTimeout: writeTimeout,
IdleTimeout: idleTimeout,
}
}
// listenAndServe runs the listener until the server is shut
// down. A genuine listen failure (not the expected
// ErrServerClosed from a clean shutdown) requests process
// shutdown through fx with a non-zero exit code, so the failure
// is visible to any supervisor.
func (s *Server) listenAndServe() {
s.SetupRoutes()
s.log.Info("http begin listen",
"listenaddr", s.httpServer.Addr,
"listenaddr", listenAddr,
"version", s.params.Globals.Version,
"arch", runtime.GOARCH,
"buildarch", s.params.Globals.Buildarch,
)
err := s.httpServer.ListenAndServe()
if err == nil || errors.Is(err, http.ErrServerClosed) {
return
}
if err != nil && !errors.Is(err, http.ErrServerClosed) {
s.log.Error("listen error", "error", err)
shutdownErr := s.shutdowner.Shutdown(fx.ExitCode(1))
if shutdownErr != nil {
s.log.Error("request shutdown failed", "error", shutdownErr)
if s.cancelFunc != nil {
s.cancelFunc()
}
}
}
-32
View File
@@ -1,32 +0,0 @@
package server_test
import "testing"
// TestListenAddress checks that the server listens on BIND_ADDRESS
// and PORT, and on port 8080 on every interface when neither is set.
// The container image sets both, to keep the backend on loopback
// behind nginx.
func TestListenAddress(t *testing.T) {
tests := []struct {
bindAddress string
port string
want string
}{
{bindAddress: "", port: "", want: ":8080"},
{bindAddress: "127.0.0.1", port: "8081", want: "127.0.0.1:8081"},
{bindAddress: "::1", port: "8081", want: "[::1]:8081"},
}
for _, tt := range tests {
t.Run(tt.want, func(t *testing.T) {
// t.Setenv rules out t.Parallel.
t.Setenv("BIND_ADDRESS", tt.bindAddress)
t.Setenv("PORT", tt.port)
got := newServer(t).ListenAddr()
if got != tt.want {
t.Errorf("listen address = %q, want %q", got, tt.want)
}
})
}
}
+8 -62
View File
@@ -3,83 +3,29 @@ package server
import (
"time"
sentryhttp "github.com/getsentry/sentry-go/http"
"github.com/go-chi/chi/v5"
"github.com/go-chi/chi/v5/middleware"
"github.com/prometheus/client_golang/prometheus"
"github.com/prometheus/client_golang/prometheus/collectors"
"github.com/prometheus/client_golang/prometheus/promhttp"
)
const (
requestTimeout = 60 * time.Second
// maxRequestBodyBytes caps every request body. A route group
// can mount s.mw.MaxBodyBytes with a smaller value to lower
// its bound, but cannot raise it: this cap runs first.
maxRequestBodyBytes int64 = 1 << 20 // 1 MiB
// metricsRequestsPerMinute is how many requests to /metrics each
// client address may make a minute, whatever their credentials. A
// scraper polling every 2 seconds sends half of it, which httprate
// never refuses.
metricsRequestsPerMinute = 60
)
const requestTimeout = 60 * time.Second
// SetupRoutes configures the chi router with middleware and
// all application routes.
func (s *Server) SetupRoutes() {
s.router = chi.NewRouter()
s.router.Use(s.mw.Recoverer())
s.router.Use(middleware.Recoverer)
s.router.Use(middleware.RequestID)
s.router.Use(s.mw.Logging())
s.router.Use(s.mw.SecurityHeaders())
s.router.Use(s.mw.CORS(s.params.Config.CORSAllowedOrigins))
s.router.Use(s.mw.MaxBodyBytes(maxRequestBodyBytes))
s.router.Use(s.mw.CORS())
s.router.Use(middleware.Timeout(requestTimeout))
// Sentry reports a panic, then panics again, so that s.mw.Recoverer
// still answers 500.
if s.params.Config.SentryDSN != "" {
s.router.Use(sentryhttp.New(sentryhttp.Options{Repanic: true}).Handle)
}
// The metrics go in a registry of this server's own, not in
// Prometheus' default one, which takes them only once per process.
registry := prometheus.NewRegistry()
registry.MustRegister(
collectors.NewGoCollector(),
collectors.NewProcessCollector(collectors.ProcessCollectorOpts{}),
s.router.Get(
"/.well-known/healthcheck",
s.h.HandleHealthCheck(),
)
// Requests are measured only once chi has matched them to one of
// these routes, by path and method. The metrics are labelled with
// both, which any client can make up, so measuring every request
// would let clients add labels without bound. A Route here would
// be matched by its path prefix alone, so each path is given in
// full.
s.router.Group(func(r chi.Router) {
// config.New refuses one of the two credentials without the
// other.
if s.params.Config.MetricsUsername != "" {
r.Use(s.mw.Metrics(registry))
}
r.Get("/.well-known/healthcheck", s.h.HandleHealthCheck())
r.With(s.mw.RateLimit(s.params.Config.ReportsPerMinute)).
Post("/api/v1/reports", s.h.HandleReport())
s.router.Route("/api/v1", func(r chi.Router) {
r.Post("/reports", s.h.HandleReport())
})
// The rate limit comes before the basic auth, so a client past it
// gets 429 and its password is not checked.
if s.params.Config.MetricsUsername != "" {
s.router.With(
s.mw.RateLimit(metricsRequestsPerMinute),
s.mw.MetricsAuth(),
).Get("/metrics", promhttp.HandlerFor(
registry, promhttp.HandlerOpts{},
).ServeHTTP)
}
}
-342
View File
@@ -1,342 +0,0 @@
package server_test
import (
"io"
"net/http"
"net/http/httptest"
"strings"
"testing"
"time"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/handlers"
"sneak.berlin/go/netwatch/internal/healthcheck"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/middleware"
"sneak.berlin/go/netwatch/internal/reportbuf"
"sneak.berlin/go/netwatch/internal/server"
"github.com/getsentry/sentry-go"
"go.uber.org/fx"
"go.uber.org/fx/fxtest"
)
// newServer builds a Server from the same constructors as main,
// configured from the environment. It is never started, so nothing
// listens.
func newServer(t *testing.T) *server.Server {
t.Helper()
var srv *server.Server
app := fxtest.New(t,
fx.Provide(
config.New,
globals.New,
handlers.New,
healthcheck.New,
logger.New,
middleware.New,
reportbuf.New,
server.New,
),
fx.Populate(&srv),
)
err := app.Err()
if err != nil {
t.Fatalf("build server: %v", err)
}
return srv
}
// TestReportsAreRateLimited checks that POST /api/v1/reports is
// behind the per-address rate limit, set here to two a minute.
func TestReportsAreRateLimited(t *testing.T) {
t.Setenv("REPORTS_PER_MINUTE", "2")
srv := newServer(t)
srv.SetupRoutes()
post := func() int {
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodPost, "/api/v1/reports",
strings.NewReader(`{"clientId":"c1","hosts":[]}`),
)
srv.ServeHTTP(rec, req)
return rec.Code
}
for i := range 2 {
if code := post(); code != http.StatusOK {
t.Fatalf("report %d: status = %d, want %d",
i+1, code, http.StatusOK)
}
}
if code := post(); code != http.StatusTooManyRequests {
t.Fatalf("third report in a minute: status = %d, want %d",
code, http.StatusTooManyRequests)
}
}
// TestCORSAllowedOriginsReachTheRouter checks that an origin listed in
// CORS_ALLOWED_ORIGINS is allowed by the router, not only when handed
// to the CORS middleware directly.
func TestCORSAllowedOriginsReachTheRouter(t *testing.T) {
const origin = "https://netwatch.example:8443"
t.Setenv("CORS_ALLOWED_ORIGINS", origin)
srv := newServer(t)
srv.SetupRoutes()
// The preflight a browser sends before it POSTs JSON from origin.
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodOptions, "/api/v1/reports", http.NoBody)
req.Header.Set("Origin", origin)
req.Header.Set("Access-Control-Request-Method", http.MethodPost)
req.Header.Set("Access-Control-Request-Headers", "content-type")
srv.ServeHTTP(rec, req)
got := rec.Header().Get("Access-Control-Allow-Origin")
if got != origin {
t.Fatalf("Access-Control-Allow-Origin = %q, want %q", got, origin)
}
}
// TestNoMetricsWithoutCredentials: with neither metrics setting set,
// there is no /metrics.
func TestNoMetricsWithoutCredentials(t *testing.T) {
t.Setenv("METRICS_USERNAME", "")
t.Setenv("METRICS_PASSWORD", "")
srv := newServer(t)
srv.SetupRoutes()
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/metrics", http.NoBody)
srv.ServeHTTP(rec, req)
if rec.Code != http.StatusNotFound {
t.Fatalf("status = %d, want %d", rec.Code, http.StatusNotFound)
}
}
// TestMetricsBehindBasicAuth: with both metrics settings set, /metrics
// answers only with them as basic auth credentials, and shows a
// request to a route but not one to a path no route has.
func TestMetricsBehindBasicAuth(t *testing.T) {
t.Setenv("METRICS_USERNAME", "prometheus")
t.Setenv("METRICS_PASSWORD", "right")
srv := newServer(t)
srv.SetupRoutes()
get := func(path, username, password string) *httptest.ResponseRecorder {
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, path, http.NoBody)
if username != "" {
req.SetBasicAuth(username, password)
}
srv.ServeHTTP(rec, req)
return rec
}
get("/.well-known/healthcheck", "", "")
get("/api/v1/no-such-route", "", "")
for _, creds := range [][2]string{
{"", ""},
{"prometheus", "wrong"},
{"someone", "right"},
} {
rec := get("/metrics", creds[0], creds[1])
if rec.Code != http.StatusUnauthorized {
t.Errorf("credentials %q: status = %d, want %d",
creds, rec.Code, http.StatusUnauthorized)
}
}
rec := get("/metrics", "prometheus", "right")
if rec.Code != http.StatusOK {
t.Fatalf("right credentials: status = %d, want %d",
rec.Code, http.StatusOK)
}
body := rec.Body.String()
if !strings.Contains(body, `handler="/.well-known/healthcheck"`) {
t.Errorf("metrics show no health check request:\n%s", body)
}
if strings.Contains(body, "no-such-route") {
t.Errorf("metrics show a request to a path no route has:\n%s", body)
}
if !strings.Contains(body, "go_goroutines") {
t.Errorf("metrics show no Go runtime metrics:\n%s", body)
}
}
// TestMetricsAreRateLimited: a client that has used up its /metrics
// allowance on wrong passwords gets 429 even with the right one, which
// is then not checked, while another client behind the same nginx
// still gets in.
func TestMetricsAreRateLimited(t *testing.T) {
t.Setenv("METRICS_USERNAME", "prometheus")
t.Setenv("METRICS_PASSWORD", "right")
// As in the container: nginx connects from loopback and names the
// client in X-Forwarded-For.
t.Setenv("TRUSTED_PROXIES", "127.0.0.1/32")
srv := newServer(t)
srv.SetupRoutes()
get := func(client, password string) int {
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/metrics", http.NoBody)
req.RemoteAddr = "127.0.0.1:40000"
req.Header.Set("X-Forwarded-For", client)
req.SetBasicAuth("prometheus", password)
srv.ServeHTTP(rec, req)
return rec.Code
}
for i := range server.MetricsRequestsPerMinute {
if code := get("203.0.113.7", "wrong"); code != http.StatusUnauthorized {
t.Fatalf("guess %d: status = %d, want %d",
i+1, code, http.StatusUnauthorized)
}
}
if code := get("203.0.113.7", "right"); code != http.StatusTooManyRequests {
t.Fatalf("right password past the limit: status = %d, want %d",
code, http.StatusTooManyRequests)
}
if code := get("203.0.113.8", "right"); code != http.StatusOK {
t.Fatalf("another client: status = %d, want %d",
code, http.StatusOK)
}
}
// TestMetricsInTwoServers: two servers in one process can both have
// metrics on.
func TestMetricsInTwoServers(t *testing.T) {
t.Setenv("METRICS_USERNAME", "prometheus")
t.Setenv("METRICS_PASSWORD", "right")
for range 2 {
newServer(t).SetupRoutes()
}
}
// TestSentry: with SENTRY_DSN empty there is no Sentry client. With it
// pointing at a local server standing in for Sentry, a panic in a
// handler reaches that server, and the request still gets the 500 from
// the panic recovery.
func TestSentry(t *testing.T) {
const panicMessage = "handler panic for TestSentry"
// sentry.Init sets the client for the whole process; take it away
// again so that no other test reports to Sentry.
t.Cleanup(func() { sentry.CurrentHub().BindClient(nil) })
t.Setenv("SENTRY_DSN", "")
newServer(t)
if sentry.CurrentHub().Client() != nil {
t.Fatal("a Sentry client exists with SENTRY_DSN empty")
}
// The body of the first request the stand-in for Sentry receives.
received := make(chan string, 1)
sentryServer := httptest.NewServer(http.HandlerFunc(
func(_ http.ResponseWriter, r *http.Request) {
body, _ := io.ReadAll(r.Body)
select {
case received <- string(body):
default:
}
},
))
defer sentryServer.Close()
t.Setenv("SENTRY_DSN",
"http://key@"+sentryServer.Listener.Addr().String()+"/1")
srv := newServer(t)
srv.SetupRoutes()
srv.Router().Get("/panic", func(http.ResponseWriter, *http.Request) {
panic(panicMessage)
})
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/panic", http.NoBody)
srv.ServeHTTP(rec, req)
if rec.Code != http.StatusInternalServerError {
t.Fatalf("status = %d, want %d",
rec.Code, http.StatusInternalServerError)
}
// Sentry sends from a goroutine of its own.
select {
case body := <-received:
if !strings.Contains(body, panicMessage) {
t.Fatalf("the Sentry server received no report of the panic:\n%s",
body)
}
case <-time.After(5 * time.Second):
t.Fatal("nothing reached the Sentry server")
}
}
// TestHealthCheckRejectsOversizeBody sends the health check, which
// never reads its body, a body one byte over the limit. Only the
// router-wide body limit can reject it.
func TestHealthCheckRejectsOversizeBody(t *testing.T) {
t.Parallel()
srv := newServer(t)
srv.SetupRoutes()
rec := httptest.NewRecorder()
req := httptest.NewRequestWithContext(t.Context(),
http.MethodGet, "/.well-known/healthcheck",
strings.NewReader(
strings.Repeat("x", int(server.MaxRequestBodyBytes)+1),
),
)
srv.ServeHTTP(rec, req)
if rec.Code != http.StatusRequestEntityTooLarge {
t.Fatalf("status = %d, want %d",
rec.Code, http.StatusRequestEntityTooLarge)
}
if got := rec.Body.String(); got != "{\"status\":\"error\"}\n" {
t.Errorf("body = %q, want %q", got, "{\"status\":\"error\"}\n")
}
got := rec.Header().Get("Content-Type")
if got != "application/json; charset=utf-8" {
t.Errorf("Content-Type = %q, want a JSON content type", got)
}
}
+68 -63
View File
@@ -1,15 +1,15 @@
// Package server provides the HTTP server lifecycle,
// including startup, routing, and graceful shutdown. The
// process lifetime is owned by fx: shutdown is requested
// through fx.Shutdowner so every component's OnStop hook runs
// in dependency order.
// including startup, routing, signal handling, and graceful
// shutdown.
package server
import (
"context"
"fmt"
"log/slog"
"net/http"
"os"
"os/signal"
"syscall"
"time"
"sneak.berlin/go/netwatch/internal/config"
@@ -18,15 +18,10 @@ import (
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/middleware"
"github.com/getsentry/sentry-go"
"github.com/go-chi/chi/v5"
"go.uber.org/fx"
)
// sentryFlushTimeout is how long shutdown waits for Sentry to send
// what it still holds.
const sentryFlushTimeout = 2 * time.Second
// Params defines the dependencies for Server.
type Params struct {
fx.In
@@ -36,18 +31,19 @@ type Params struct {
Handlers *handlers.Handlers
Logger *logger.Logger
Middleware *middleware.Middleware
Shutdowner fx.Shutdowner
}
// Server is the top-level HTTP server orchestrator.
type Server struct {
cancelFunc context.CancelFunc
exitCode int
h *handlers.Handlers
httpServer *http.Server
log *slog.Logger
mw *middleware.Middleware
params Params
router *chi.Mux
shutdowner fx.Shutdowner
startupTime time.Time
}
// New creates a Server and registers lifecycle hooks for
@@ -61,30 +57,23 @@ func New(
s.mw = params.Middleware
s.h = params.Handlers
s.log = params.Logger.Get()
s.shutdowner = params.Shutdowner
err := s.enableSentry()
if err != nil {
return nil, err
}
lc.Append(fx.Hook{
OnStart: func(_ context.Context) error {
// Build the router and http.Server synchronously
// here, before spawning the serving goroutine, so
// httpServer is fully constructed by the time OnStop
// (or an early signal) can read it. fx guarantees
// OnStart returns before OnStop runs, so no
// synchronization or nil check is needed at shutdown.
s.SetupRoutes()
s.httpServer = s.newHTTPServer()
s.startupTime = time.Now().UTC()
go s.listenAndServe()
go func() { //nolint:contextcheck // fx OnStart ctx is startup-only; run() creates its own
s.run()
}()
return nil
},
OnStop: func(ctx context.Context) error {
return s.shutdown(ctx)
OnStop: func(_ context.Context) error {
if s.cancelFunc != nil {
s.cancelFunc()
}
return nil
},
})
@@ -99,44 +88,60 @@ func (s *Server) ServeHTTP(
s.router.ServeHTTP(w, r)
}
// enableSentry sets Sentry up when SENTRY_DSN is set, so that
// SetupRoutes can report panics to it. With SENTRY_DSN empty it does
// nothing. A DSN Sentry refuses stops the start.
func (s *Server) enableSentry() error {
if s.params.Config.SentryDSN == "" {
return nil
func (s *Server) run() {
exitCode := s.serve()
os.Exit(exitCode)
}
err := sentry.Init(sentry.ClientOptions{
Dsn: s.params.Config.SentryDSN,
Release: s.params.Globals.Appname + "-" + s.params.Globals.Version,
})
func (s *Server) serve() int {
var ctx context.Context //nolint:wsl // ctx must be declared before multi-assign
ctx, s.cancelFunc = context.WithCancel(
context.Background(),
)
go func() {
c := make(chan os.Signal, 1)
signal.Ignore(syscall.SIGPIPE)
signal.Notify(c, os.Interrupt, syscall.SIGTERM)
sig := <-c
s.log.Info("signal received", "signal", sig)
if s.cancelFunc != nil {
s.cancelFunc()
}
}()
go func() {
s.serveUntilShutdown()
}()
<-ctx.Done()
s.cleanShutdown()
return s.exitCode
}
const shutdownTimeout = 5 * time.Second
func (s *Server) cleanShutdown() {
s.exitCode = 0
ctxShutdown, shutdownCancel := context.WithTimeout(
context.Background(),
shutdownTimeout,
)
defer shutdownCancel()
err := s.httpServer.Shutdown(ctxShutdown)
if err != nil {
return fmt.Errorf("SENTRY_DSN: %w", err)
}
s.log.Info("sentry error reporting activated")
return nil
}
// shutdown gracefully stops the HTTP server within the
// deadline of the context fx provides for OnStop, then gives
// Sentry, if set up, time to send what it still holds.
func (s *Server) shutdown(ctx context.Context) error {
err := s.httpServer.Shutdown(ctx)
if s.params.Config.SentryDSN != "" {
sentry.Flush(sentryFlushTimeout)
}
if err != nil {
s.log.Error("server clean shutdown failed", "error", err)
return err
s.log.Error(
"server clean shutdown failed",
"error", err,
)
}
s.log.Info("server stopped")
return nil
}
-21
View File
@@ -1,21 +0,0 @@
#!/bin/sh
# script/build: compile the static netwatch-server binary into the
# backend project root, with its version stamped in.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
# VERSION comes from the environment (the root Dockerfile passes its
# ARG VERSION in). Unset or empty, it is git describe, or "dev" where
# there is no git or no repository history.
version="${VERSION:-$(git describe --always --dirty 2>/dev/null || echo dev)}"
CGO_ENABLED=0 go build -trimpath \
-ldflags "-s -w -X main.Version=$version" \
-o netwatch-server ./cmd/netwatch-server/
}
main "$@"
-12
View File
@@ -1,12 +0,0 @@
#!/bin/sh
# script/clean: remove build artifacts.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
rm -f netwatch-server
}
main "$@"
-12
View File
@@ -1,12 +0,0 @@
#!/bin/sh
# script/fmt: format the Go sources (writes).
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
go fmt ./...
}
main "$@"
-18
View File
@@ -1,18 +0,0 @@
#!/bin/sh
# script/fmt-check: check Go formatting (read-only). Same scope as
# script/fmt, but fails instead of writing.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
unformatted="$(gofmt -l .)"
if [ -n "$unformatted" ]; then
echo "Files not formatted:" >&2
echo "$unformatted" >&2
exit 1
fi
}
main "$@"
-50
View File
@@ -1,50 +0,0 @@
#!/bin/sh
# script/lint: run golangci-lint over the backend. This runs inside the
# lint stage of the root Dockerfile, whose digest-pinned golangci-lint
# image provides the linter; nothing installs golangci-lint on the host.
# From a checkout, run `make lint` at the repo root, which builds that
# stage.
#
# .golangci.yml is standardized org-wide and must never be edited here
# (REPO_POLICIES.md). Its last silent drift replaced the v2 schema with
# v1 keys, which left every threshold in the file inert while the build
# stayed green. So the file is first checked against the canonical
# copy's sha256: a local comparison, no network, nothing unpinned.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
# The sha256 of the org standard .golangci.yml. When that file changes in
# sneak/prompts and is copied here again, this changes with it.
GOLANGCI_CONFIG_SHA256="a79b63a254602a5318db5d0e9a06bc71b84bf0c1d896305229d8bfed1d1b1776"
main() {
cd "$ROOT"
if [ ! -f .golangci.yml ]; then
echo "backend/.golangci.yml is missing. Copy the org standard verbatim" >&2
echo "from https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml" >&2
exit 1
fi
actual="$(sha256sum .golangci.yml | cut -d' ' -f1)"
if [ -z "$actual" ]; then
echo "sha256sum is missing or printed no hash, so" >&2
echo "backend/.golangci.yml could not be checked." >&2
exit 1
fi
if [ "$actual" != "$GOLANGCI_CONFIG_SHA256" ]; then
echo "backend/.golangci.yml does not match GOLANGCI_CONFIG_SHA256" >&2
echo "in backend/script/lint." >&2
echo " expected $GOLANGCI_CONFIG_SHA256" >&2
echo " actual $actual" >&2
echo "Compare it with the org standard," >&2
echo "https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml" >&2
echo "- If they differ, it was edited here: restore the org standard" >&2
echo " verbatim. Do not edit it." >&2
echo "- If they are the same, the org standard changed: set" >&2
echo " GOLANGCI_CONFIG_SHA256 in backend/script/lint to the actual hash." >&2
exit 1
fi
golangci-lint run ./...
}
main "$@"
-13
View File
@@ -1,13 +0,0 @@
#!/bin/sh
# script/run: build and run netwatch-server locally.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
"$ROOT/script/build"
exec ./netwatch-server "$@"
}
main "$@"
-20
View File
@@ -1,20 +0,0 @@
#!/bin/sh
# script/test: run the backend test suite with the race detector and
# coverage. Go's own -timeout bounds the tests and not their compile,
# so a cold build cache cannot fail it. The race detector needs cgo,
# and so a C compiler. If the tests fail, they run again with -v for
# the details, and the script fails even if that run passes.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
go test -timeout 30s -race -cover ./... || {
echo "--- Rerunning with -v for details ---"
go test -timeout 30s -race -v ./...
exit 1
}
}
main "$@"
-129
View File
@@ -1,129 +0,0 @@
#!/bin/sh
# The container's entrypoint: runs netwatch-server and nginx side by
# side. TERM or INT stops both, and the container exits 0 if both exit
# cleanly. If either exits on its own, the other is stopped too and the
# container exits non-zero, so the platform restarts it instead of
# leaving it half up.
#
# No set -e: kill and wait return non-zero here in normal operation.
set -u
# PORT is the public port nginx listens on, 8080 when unset or empty.
# nginx would take a value such as localhost or unix:/tmp/x.sock as an
# address and start anyway, and reports a bad port without naming
# PORT, so a value that is not a usable port stops the container here,
# before either process starts.
export PORT="${PORT:-8080}"
case "$PORT" in
*[!0-9]*)
echo "entrypoint: PORT must be a port number, not '$PORT'" >&2
exit 1
;;
esac
# The length is checked first because, for a number too big for it,
# the shell's test prints an error and is false, so the range checks
# alone would let it through.
if [ "${#PORT}" -gt 5 ] || [ "$PORT" -lt 1 ] || [ "$PORT" -gt 65535 ]; then
echo "entrypoint: PORT must be from 1 to 65535, not '$PORT'" >&2
exit 1
fi
if [ "$PORT" -eq 8081 ]; then
echo "entrypoint: PORT cannot be 8081, netwatch-server listens there" >&2
exit 1
fi
# TRUSTED_PROXIES names the reverse proxies in front of the container,
# as IP addresses or CIDRs separated by commas. nginx takes the client
# address from X-Forwarded-For only on a request from one of them, so
# unset or empty, it trusts no one. nginx.conf includes the file written
# here, one set_real_ip_from line per entry.
#
# nginx looks up an entry it cannot read as an address as a hostname,
# and trusts what it finds (1.2.3 is found as 1.2.0.3). So each entry
# is made a CIDR, a lone address getting /128 if it is IPv6 and /32 if
# not, and netwatch-server checks it with the parsing it gives its own
# TRUSTED_PROXIES. Its error, naming the CIDR, is dropped for the one
# below, naming the entry as written. set -f keeps a * in an entry from
# becoming a list of file names.
TRUSTED_PROXIES="${TRUSTED_PROXIES:-}"
set -f
for proxy in $(printf '%s' "$TRUSTED_PROXIES" | tr ',' ' '); do
case "$proxy" in
*/*) cidr="$proxy" ;;
*:*) cidr="$proxy/128" ;;
*) cidr="$proxy/32" ;;
esac
if ! netwatch-server check-cidr "$cidr" 2> /dev/null; then
echo "entrypoint: TRUSTED_PROXIES must be IP addresses or CIDRs" \
"separated by commas; '$proxy' is neither" >&2
exit 1
fi
echo "set_real_ip_from $cidr;"
done > /etc/nginx/trusted-proxies.conf
# netwatch-server keeps its report files in DATA_DIR, on the /data
# volume, which may be a host directory owned by root or by another
# uid. Here, as root, netwatch-server prepare-data-dir creates DATA_DIR
# and gives /data and everything in it to the netwatch user, so the
# host directory needs no preparing. It stops the start, naming
# DATA_DIR, unless DATA_DIR is /data or a path below it, and it acts on
# nothing outside /data, whatever symbolic links it meets there.
export DATA_DIR="${DATA_DIR:-/data/reports}"
netwatch-server prepare-data-dir "$DATA_DIR" || exit 1
# A stop signal is only noted here; the loop below acts on it.
stop_requested=""
trap 'stop_requested=yes' TERM INT
# netwatch-server runs as the netwatch user and listens on loopback
# only, on a port other than the public one; nginx.conf proxies to this
# address. Its only client is nginx, so it takes the client address
# nginx passes on from 127.0.0.1 alone, whatever TRUSTED_PROXIES the
# container has. The netwatch user has no login shell, hence -s
# /bin/sh. busybox su replaces itself with the command instead of
# staying on as its parent, so $! is the server's own PID.
BIND_ADDRESS=127.0.0.1 PORT=8081 TRUSTED_PROXIES=127.0.0.1/32 \
su -s /bin/sh netwatch -c 'exec netwatch-server' &
backend=$!
# nginx starts through the nginx image's own entrypoint, which applies
# the image's start-up configuration and then replaces itself with
# nginx. Part of that start-up configuration renders nginx.conf into
# conf.d with nginx listening on PORT. NGINX_ENVSUBST_FILTER limits
# that rendering to PORT: a variable nginx itself uses, such as $uri,
# would otherwise be replaced by an environment variable of the same
# name.
NGINX_ENVSUBST_FILTER='^PORT$' \
/docker-entrypoint.sh nginx -g 'daemon off;' &
nginx=$!
running() {
kill -0 "$1" 2>/dev/null
}
# POSIX sh cannot wait for whichever of two children exits first, so
# look once a second. The shell collects a child that has exited while
# it runs sleep, and running() is false for that child from then on.
while [ -z "$stop_requested" ] && running "$backend" && running "$nginx"; do
sleep 1
done
# Stop both, then wait until neither is left.
kill -TERM "$backend" "$nginx" 2>/dev/null
while running "$backend" || running "$nginx"; do
sleep 1
done
wait "$backend"
backend_status=$?
wait "$nginx"
nginx_status=$?
echo "entrypoint: netwatch-server exited $backend_status," \
"nginx exited $nginx_status"
# Success is a requested stop that both processes exited cleanly from.
if [ -n "$stop_requested" ] && [ "$backend_status" -eq 0 ] &&
[ "$nginx_status" -eq 0 ]; then
exit 0
fi
exit 1
-27
View File
@@ -1,27 +0,0 @@
// eslint's recommended rules over every JavaScript file in the repo.
// script/frontend-lint runs it, in the frontend-lint stage of Dockerfile.
import js from "@eslint/js";
import globals from "globals";
import { defineConfig } from "eslint/config";
export default defineConfig([
js.configs.recommended,
{
// The page, and facts.js, which the viewport test runs inside it.
files: ["src/**/*.js", "test/viewport/facts.js"],
languageOptions: {
globals: {
...globals.browser,
// vite.config.js defines these when it builds the page.
__COMMIT_HASH__: "readonly",
__COMMIT_FULL__: "readonly",
},
},
},
{
// Run by node: the tests, the viewport harness and these configs.
files: ["test/**/*.js", "*.js"],
ignores: ["test/viewport/facts.js"],
languageOptions: { globals: globals.node },
},
]);
-3
View File
@@ -9,9 +9,6 @@
type="image/svg+xml"
href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'><text y='.9em' font-size='90'>📡</text></svg>"
/>
<!-- Linked here, not imported by src/main.js, so the unit tests can
import that module in Node, which cannot import CSS. -->
<link rel="stylesheet" href="/src/styles.css" />
</head>
<body class="bg-gray-900 text-white min-h-screen">
<div id="app"></div>
+5 -54
View File
@@ -1,27 +1,14 @@
# A template: the nginx image renders it into conf.d at container start,
# filling in PORT and nothing else. bin/entrypoint.sh sets PORT and that
# limit.
server {
listen ${PORT};
listen 8080;
server_name _;
# Keep the nginx version out of the Server header and error pages.
server_tokens off;
# The security headers, on every response. An add_header in a
# location drops every add_header from here, so a location with one
# of its own includes this file again.
include /etc/nginx/security-headers.conf;
root /usr/share/nginx/html;
index index.html;
# The client address comes from X-Forwarded-For only on a request
# from the reverse proxies in TRUSTED_PROXIES: bin/entrypoint.sh
# writes one set_real_ip_from line for each into this file, and
# leaves it empty when TRUSTED_PROXIES is unset, so that by default
# the client address is the one each request comes from.
include /etc/nginx/trusted-proxies.conf;
# Trust RFC1918 reverse proxies for X-Forwarded-For
set_real_ip_from 10.0.0.0/8;
set_real_ip_from 172.16.0.0/12;
set_real_ip_from 192.168.0.0/16;
real_ip_header X-Forwarded-For;
real_ip_recursive on;
@@ -37,41 +24,5 @@ server {
location /assets/ {
expires 1y;
add_header Cache-Control "public, immutable";
include /etc/nginx/security-headers.conf;
}
# netwatch-server, the Go backend, runs in the same container and
# listens on loopback only: bin/entrypoint.sh starts it on
# 127.0.0.1:8081. These headers go with every request passed to it.
# X-Forwarded-For carries only the client address, as resolved by
# the real IP settings above, and not the chain the request came
# with: the backend takes the first entry, which a client can write.
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $remote_addr;
proxy_set_header X-Forwarded-Proto $scheme;
# netwatch-server sets the same security headers on its own
# responses. Its copies are dropped so that each header goes out
# once, as security-headers.conf sets it.
proxy_hide_header Strict-Transport-Security;
proxy_hide_header Content-Security-Policy;
proxy_hide_header X-Frame-Options;
proxy_hide_header X-Content-Type-Options;
proxy_hide_header Referrer-Policy;
proxy_hide_header Permissions-Policy;
location /api/ {
proxy_pass http://127.0.0.1:8081;
}
location = /.well-known/healthcheck {
proxy_pass http://127.0.0.1:8081;
}
# The backend's Prometheus metrics, behind its own basic auth. Unless
# METRICS_USERNAME and METRICS_PASSWORD are set, it answers 404.
location = /metrics {
proxy_pass http://127.0.0.1:8081;
}
}
+1 -6
View File
@@ -6,19 +6,14 @@
"scripts": {
"dev": "vite",
"build": "vite build",
"preview": "vite preview",
"test": "node --test test/unit/*.test.js"
"preview": "vite preview"
},
"license": "MIT",
"devDependencies": {
"@eslint/js": "^10.0.1",
"@tailwindcss/vite": "^4.1.18",
"autoprefixer": "^10.4.23",
"eslint": "^10.12.0",
"globals": "^17.13.0",
"postcss": "^8.5.6",
"prettier": "^3.8.1",
"puppeteer-core": "25.5.0",
"tailwindcss": "^4.1.18",
"vite": "^7.3.1"
}
-30
View File
@@ -1,30 +0,0 @@
#!/bin/sh
# script/add-dependency: add a frontend package, or move one to another
# version, with yarn add, which changes package.json and yarn.lock
# together; then install from yarn.lock with --frozen-lockfile, as
# script/bootstrap does, to show it installs as written. --dev because
# no frontend package is needed when the page runs: it ships as the
# built dist/.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
usage() {
echo "usage: make add-dependency PACKAGE=<name>@<version>" >&2
exit 2
}
main() {
# Exactly one package. A value beginning with - would reach yarn as
# an option; yarn would quietly drop all but the first of several
# packages given in one value.
[ "$#" -eq 1 ] || usage
case "$1" in
"" | -* | *[[:space:]]*) usage ;;
esac
cd "$ROOT"
yarn add --dev "$1"
yarn install --frozen-lockfile
}
main "$@"
-285
View File
@@ -1,285 +0,0 @@
#!/bin/sh
# script/bootstrap: install all dependencies needed to build and develop
# this repo. Idempotent: every install is guarded by a check so already
# installed tools are skipped. Base tooling comes from nix, apt, brew,
# or apk (detected in that order); assumes nothing is present. Node is
# used directly if it is at least NODE_MIN_VERSION; otherwise it is
# installed at a pinned version via nvm (installing nvm itself first,
# from a hash-verified release archive, never curl | sh). Go, with its
# gofmt, is used directly if it is at least the version backend/go.mod
# asks for; otherwise the pinned Go release is installed from its
# hash-verified archive.
#
# What this script installs outside the system package manager lives
# under $HOME and is linked into ~/.local/bin, where make and the git
# hook find it once that directory is on PATH. Nothing in ~/.local/bin
# that this script did not create is ever replaced.
#
# golangci-lint is not installed: make lint runs it in Docker, which
# this script does not install either.
#
# Unlike the org model: Go and gcc for backend/, a newer node for eslint.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
# Pinned versions, 2026-07-06
NODE_VERSION="22.17.0"
# The oldest node the frontend's dependencies accept: the "engines"
# field of eslint 10.12.0, the most demanding of them, asks for 22.13.0
# or newer, 2026-10-03. An older installed node is not used.
NODE_MIN_VERSION="22.13.0"
NVM_VERSION="0.40.3"
# sha256 of https://github.com/nvm-sh/nvm/archive/refs/tags/v0.40.3.tar.gz
NVM_SHA256="5f4d6aaa04a177dc93c985e31dbc411ab6b8c6e1e21d8015dbc1372625fcd1d0"
YARN_VERSION="1.22.22"
# The Go inside the golang:1.25-alpine image Dockerfile builds the
# backend with, 2026-08-09. The archive hashes are in ensure_go.
GO_VERSION="1.25.7"
BIN_DIR="$HOME/.local/bin"
TOOLCHAIN="$HOME/.local/share/$("$ROOT/script/projectname")/toolchain"
PKGMGR=""
SUDO=""
APT_UPDATED=""
detect_pkgmgr() {
[ -n "$PKGMGR" ] && return 0
if command -v nix-env >/dev/null 2>&1; then
PKGMGR="nix"
elif command -v apt-get >/dev/null 2>&1; then
PKGMGR="apt"
elif command -v brew >/dev/null 2>&1; then
PKGMGR="brew"
elif command -v apk >/dev/null 2>&1; then
PKGMGR="apk"
else
echo "bootstrap: no supported package manager (nix, apt, brew, apk)" >&2
exit 1
fi
if [ "$PKGMGR" = "apt" ]; then
export DEBIAN_FRONTEND=noninteractive
if [ "$(id -u)" != "0" ]; then
SUDO="sudo"
fi
fi
}
# pkg_install <nix-attr> <apt-pkgs> <brew-formula> <apk-pkgs>: the apt
# and apk arguments may each list several packages, separated by spaces.
pkg_install() {
detect_pkgmgr
case "$PKGMGR" in
nix) nix-env -iA "nixpkgs.$1" ;;
apt)
if [ -z "$APT_UPDATED" ]; then
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get update
APT_UPDATED=1
fi
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get install -y $2
;;
brew) brew install "$3" ;;
apk) apk add --no-cache $4 ;;
esac
}
missing() {
! command -v "$1" >/dev/null 2>&1
}
# verify_sha256 <file> <expected-hash>
verify_sha256() {
if command -v sha256sum >/dev/null 2>&1; then
actual="$(sha256sum "$1" | cut -d' ' -f1)"
else
actual="$(shasum -a 256 "$1" | cut -d' ' -f1)"
fi
if [ "$actual" != "$2" ]; then
echo "bootstrap: sha256 mismatch for $1" >&2
echo " expected: $2" >&2
echo " actual: $actual" >&2
exit 1
fi
}
# link_bin <target> <name>: make an installed tool reachable as
# $BIN_DIR/<name>. Only a symlink this script made, one pointing into
# $TOOLCHAIN or ~/.nvm, is ever replaced; if anything else is already
# there, bootstrap stops.
link_bin() {
link="$BIN_DIR/$2"
if [ -L "$link" ] || [ -e "$link" ]; then
case "$(readlink "$link" || true)" in
"$TOOLCHAIN"/* | "$HOME"/.nvm/*) ;;
*)
echo "bootstrap: $link was not created by this script;" >&2
echo " remove or rename it, then re-run bootstrap" >&2
exit 1
;;
esac
fi
mkdir -p "$BIN_DIR"
ln -sf "$1" "$link"
}
# nvm is a bash script; run a command in a bash with nvm loaded
nvm_sh() {
bash -c ". \"\$HOME/.nvm/nvm.sh\" && $*"
}
ensure_nvm() {
[ -s "$HOME/.nvm/nvm.sh" ] && return 0
# nvm prerequisites; nvm itself requires bash
if missing bash; then pkg_install bash bash bash bash; fi
if missing curl; then pkg_install curl curl curl curl; fi
if missing git; then pkg_install git git git git; fi
tmp="$(mktemp -d)"
curl -fsSL -o "$tmp/nvm.tar.gz" \
"https://github.com/nvm-sh/nvm/archive/refs/tags/v${NVM_VERSION}.tar.gz"
verify_sha256 "$tmp/nvm.tar.gz" "$NVM_SHA256"
mkdir -p "$HOME/.nvm"
tar -xzf "$tmp/nvm.tar.gz" -C "$HOME/.nvm" --strip-components=1
rm -rf "$tmp"
}
# node_ok: the node on PATH is at least NODE_MIN_VERSION. node itself
# compares the two: major, then minor, then patch.
node_ok() {
if missing node; then return 1; fi
node -e '
const have = process.versions.node.split(".").map(Number);
const want = process.argv[1].split(".").map(Number);
for (let i = 0; i < 3; i++) {
if (have[i] !== want[i]) process.exit(have[i] > want[i] ? 0 : 1);
}
' "$NODE_MIN_VERSION"
}
# ensure_node: unless node_ok, install NODE_VERSION and link its node.
ensure_node() {
if node_ok; then return 0; fi
ensure_nvm
nvm_sh "nvm install $NODE_VERSION"
link_bin "$HOME/.nvm/versions/node/v$NODE_VERSION/bin/node" node
}
# ensure_yarn: corepack writes its shims (pnpm and yarnpkg as well as
# yarn) into $TOOLCHAIN rather than next to itself, and the npm fallback
# installs there too; only yarn is linked.
ensure_yarn() {
if ! missing yarn; then return 0; fi
shims="$TOOLCHAIN/corepack-shims"
mkdir -p "$shims"
if ! missing corepack; then
corepack enable --install-directory "$shims"
corepack prepare "yarn@$YARN_VERSION" --activate
elif [ -s "$HOME/.nvm/nvm.sh" ]; then
nvm_sh "nvm use $NODE_VERSION >/dev/null && \
corepack enable --install-directory \"$shims\" && \
corepack prepare yarn@$YARN_VERSION --activate"
else
npm install -g --prefix "$TOOLCHAIN/npm-global" "yarn@$YARN_VERSION"
shims="$TOOLCHAIN/npm-global/bin"
fi
link_bin "$shims/yarn" yarn
}
# go_ok: the go on PATH has its gofmt beside it (a Go release ships the
# two together) and is at least the version backend/go.mod asks for.
# GOTOOLCHAIN=local makes an older go fail here instead of fetching a
# newer toolchain for itself.
go_ok() {
if missing go; then return 1; fi
[ -x "$(dirname "$(command -v go)")/gofmt" ] || return 1
(cd "$ROOT/backend" && GOTOOLCHAIN=local go list -m >/dev/null 2>&1)
}
# ensure_go: unless go_ok, install GO_VERSION and link its go and gofmt.
# They are linked on every run that needs them, so a deleted link is put
# back, and the archive is unpacked again if either binary is missing.
ensure_go() {
if go_ok; then return 0; fi
go_dir="$TOOLCHAIN/go-$GO_VERSION"
if [ ! -x "$go_dir/bin/go" ] || [ ! -x "$go_dir/bin/gofmt" ]; then
# sha256 of each archive, from https://go.dev/dl/?mode=json
case "$(uname -s)-$(uname -m)" in
Linux-x86_64)
plat="linux-amd64"
sha="12e6d6a191091ae27dc31f6efc630e3a3b8ba409baf3573d955b196fdf086005"
;;
Linux-aarch64)
plat="linux-arm64"
sha="ba611a53534135a81067240eff9508cd7e256c560edd5d8c2fef54f083c07129"
;;
Darwin-x86_64)
plat="darwin-amd64"
sha="bf5050a2152f4053837b886e8d9640c829dbacbc3370f913351eb0904cb706f5"
;;
Darwin-arm64)
plat="darwin-arm64"
sha="ff18369ffad05c57d5bed888b660b31385f3c913670a83ef557cdfd98ea9ae1b"
;;
*)
echo "bootstrap: no pinned Go release for this platform" >&2
exit 1
;;
esac
if missing curl; then pkg_install curl curl curl curl; fi
mkdir -p "$TOOLCHAIN"
curl -fsSL -o "$go_dir.tar.gz" \
"https://go.dev/dl/go$GO_VERSION.$plat.tar.gz"
verify_sha256 "$go_dir.tar.gz" "$sha"
# Unpacked beside its final place and then moved there, so an
# interrupted run never leaves a partial Go that looks complete.
rm -rf "$go_dir.partial"
mkdir "$go_dir.partial"
tar -xzf "$go_dir.tar.gz" -C "$go_dir.partial" --strip-components=1
rm -rf "$go_dir" "$go_dir.tar.gz"
mv "$go_dir.partial" "$go_dir"
fi
link_bin "$go_dir/bin/go" go
link_bin "$go_dir/bin/gofmt" gofmt
}
main() {
cd "$ROOT"
# Tools linked on an earlier run count as installed, and tools linked
# on this run are found by the steps after it.
path_hint=""
case ":$PATH:" in
*":$BIN_DIR:"*) ;;
*) path_hint=yes ;;
esac
PATH="$BIN_DIR:$PATH"
if missing make; then pkg_install gnumake make make make; fi
if missing git; then pkg_install git git git git; fi
# The race detector in make test needs cgo, which Go turns on only
# when it finds its C compiler, gcc on Linux. apt and apk ship the C
# library headers apart from gcc.
if missing gcc; then
pkg_install gcc "gcc libc6-dev" gcc "gcc musl-dev"
fi
ensure_node
ensure_yarn
yarn install --frozen-lockfile
ensure_go
(cd "$ROOT/backend" && go mod download)
if missing docker; then
echo "bootstrap: docker not found; make lint, and so make check" >&2
echo " and the pre-commit hook, need it to run the linters" >&2
fi
if [ -n "$path_hint" ] && [ -d "$BIN_DIR" ]; then
echo "bootstrap: add $BIN_DIR to the front of your PATH, e.g." >&2
echo " export PATH=\"\$HOME/.local/bin:\$PATH\"" >&2
fi
echo "bootstrap complete"
}
main "$@"
-13
View File
@@ -1,13 +0,0 @@
#!/bin/sh
# script/build: build the frontend for production into dist/. The Go
# backend is built by backend/script/build.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
yarn build
}
main "$@"
-14
View File
@@ -1,14 +0,0 @@
#!/bin/sh
# script/check: run all checks (test, lint, fmt-check). Our own
# extension to scripts-to-rule-them-all. Must not modify any files.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
main() {
"$SCRIPT_DIR/test"
"$SCRIPT_DIR/lint"
"$SCRIPT_DIR/fmt-check"
}
main "$@"
-28
View File
@@ -1,28 +0,0 @@
#!/bin/sh
# script/cibuild: run the CI build. It bootstraps first: a CI runner
# checks out and runs this and nothing else, and script/fmt-check runs
# the formatter on the host, which a pristine checkout cannot do.
# --no-cache for the same reason as script/docker: the gate phases the
# final stage depends on are RUN steps, and a cached one is a check that
# did not run.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() {
cd "$ROOT"
"$SCRIPT_DIR/bootstrap"
"$SCRIPT_DIR/check"
# Own line: a failing command substitution inside an argument does
# not trip `set -e`, so the inline form degrades silently to an
# empty constant. The VERSION build argument takes precedence over
# the version a build stage derives from the .git in the context.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
--build-arg VERSION="$version" \
-t "$("$SCRIPT_DIR/projectname")" .
}
main "$@"
-13
View File
@@ -1,13 +0,0 @@
#!/bin/sh
# script/dev: run the frontend's Vite dev server. It proxies /api to a
# netwatch-server running locally (see vite.config.js).
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
yarn dev
}
main "$@"
-24
View File
@@ -1,24 +0,0 @@
#!/bin/sh
# script/docker: build the Docker image tagged with the project name.
# Identical in all repos; the tag comes from script/projectname.
# --no-cache because the gate phases the final stage depends on are RUN
# steps, and a cached one is a check that did not run.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() {
cd "$ROOT"
# Own line: a failing command substitution inside an argument does
# not trip `set -e`, so the inline form degrades silently to an
# empty constant. The VERSION build argument takes precedence over
# the version a build stage derives from the .git in the context.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
--build-arg VERSION="$version" \
-t "$("$SCRIPT_DIR/projectname")" .
}
main "$@"
-15
View File
@@ -1,15 +0,0 @@
#!/bin/sh
# script/fmt: format the whole repo (writes): prettier over everything
# it understands, then gofmt over the Go backend.
# The org model formats only markdown; this repo also has JS and Go.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
"$ROOT/script/frontend-fmt"
"$ROOT/backend/script/fmt"
}
main "$@"
-15
View File
@@ -1,15 +0,0 @@
#!/bin/sh
# script/fmt-check: check formatting across the whole repo (read-only).
# Same scope as script/fmt, but fails instead of writing.
# The org model checks only markdown; this repo also has JS and Go.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
"$ROOT/script/frontend-fmt-check"
"$ROOT/backend/script/fmt-check"
}
main "$@"
-18
View File
@@ -1,18 +0,0 @@
#!/bin/sh
# script/frontend-check: run the frontend tests and format check only.
# This exists for the frontend stage of Dockerfile, a node image with
# neither Go nor Docker; the Dockerfile's frontend-lint stage runs the
# frontend linter, and its lint and builder stages gate the backend.
# Everywhere else, use script/check, which covers the whole repo. Must
# not modify any files.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
"$ROOT/script/frontend-test"
"$ROOT/script/frontend-fmt-check"
}
main "$@"
-14
View File
@@ -1,14 +0,0 @@
#!/bin/sh
# script/frontend-fmt: format the frontend and every other file prettier
# understands, repo-wide (writes), the markdown in backend/ included.
# Prettier does not read Go; backend/script/fmt formats the Go sources.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
yarn prettier --write .
}
main "$@"
-13
View File
@@ -1,13 +0,0 @@
#!/bin/sh
# script/frontend-fmt-check: check prettier formatting (read-only). Same
# scope as script/frontend-fmt, but fails instead of writing.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
yarn prettier --check .
}
main "$@"
-15
View File
@@ -1,15 +0,0 @@
#!/bin/sh
# script/frontend-lint: run eslint over the frontend. This runs inside
# the frontend-lint stage of Dockerfile, the digest-pinned node image
# with the packages from yarn.lock. From a checkout, run `make lint`,
# which builds that stage.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
yarn eslint .
}
main "$@"
-23
View File
@@ -1,23 +0,0 @@
#!/bin/sh
# script/frontend-test: run the frontend test suite: the unit tests in
# test/unit/, through the test script in package.json, then the
# production build, which fails on broken code. The tests print a dot
# each; if any fails, they run again with every test listed, and the
# script fails even if that run passes. NODE_OPTIONS chooses the
# reporter because yarn adds its arguments after the test files, where
# node would take a reporter option for one more file.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
NODE_OPTIONS=--test-reporter=dot timeout 30 yarn --silent run test || {
echo "--- Rerunning with every test listed for details ---"
NODE_OPTIONS=--test-reporter=spec timeout 30 yarn --silent run test
exit 1
}
timeout 30 yarn build
}
main "$@"
-110
View File
@@ -1,110 +0,0 @@
#!/bin/sh
# script/frontend-viewport-test: verify the responsive layout of the built
# frontend in a real browser engine.
#
# Builds dist/, serves it with the same nginx image and the same nginx.conf
# the shipping container uses, points a containerised headless Chrome at it
# over CDP, and asserts on computed layout at every viewport width derived
# from the app's own CSS. See test/viewport/README.md for what this covers
# and what it cannot.
#
# Deliberately not part of script/check: it needs Docker and takes far
# longer than the 20s budget make test has to stay inside.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
# chromedp/headless-shell 151.0.7922.109, 2026-08-09
BROWSER_IMAGE="chromedp/headless-shell@sha256:2d349b544a1ea6b5b5fd7c0fe99215ff662339c57407ee2e8c0a11af93516b04"
# nginx:stable-alpine, 2026-02-22 (the digest Dockerfile ships)
SERVER_IMAGE="nginx@sha256:15e96e59aa3b0aada3a121296e3bce117721f42d88f5f64217ef4b18f458c6ab"
# node:22-alpine, 2026-02-22 (the digest Dockerfile builds with)
NODE_IMAGE="node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34"
RUN_ID="$$-$(date +%s)"
NETWORK="netwatch-viewport-$RUN_ID"
SERVER="netwatch-viewport-server-$RUN_ID"
BROWSER="netwatch-viewport-browser-$RUN_ID"
HARNESS="netwatch-viewport-harness-$RUN_ID"
ARTIFACT_DIR="$ROOT/tmp/viewport"
# Every container is named and removed here, including the harness itself:
# `timeout` below kills the `docker run` client, not the container it
# started, and an unnamed survivor keeps the --internal network in use so
# `docker network rm` fails too. This host runs many sessions at once and
# neither may be left behind.
cleanup() {
docker rm -f "$HARNESS" > /dev/null 2>&1 || true
docker rm -f "$BROWSER" > /dev/null 2>&1 || true
docker rm -f "$SERVER" > /dev/null 2>&1 || true
docker network rm "$NETWORK" > /dev/null 2>&1 || true
}
trap cleanup EXIT INT TERM
main() {
cd "$ROOT"
# Test what ships: the production build, not a dev server.
"$ROOT/script/frontend-test"
if [ ! -f "$ROOT/dist/index.html" ]; then
echo "frontend-viewport-test: dist/index.html missing after build" >&2
exit 1
fi
mkdir -p "$ARTIFACT_DIR"
# An --internal network has no route off the host, so the browser
# cannot reach the real internet no matter what the page asks for.
# Latency probes are answered by the harness instead. This also means
# no port can be published from it, which is why the harness itself
# runs as a third container on the same network rather than on the
# host.
docker network create --internal "$NETWORK" > /dev/null
# nginx.conf is a template: the image renders it over its own
# default.conf, with the same port and limit bin/entrypoint.sh uses.
# The empty file it includes trusts no proxy, as bin/entrypoint.sh
# writes it when TRUSTED_PROXIES is unset. nginx.conf also includes
# the security headers, so the page runs under the shipped policy.
docker run -d --rm --name "$SERVER" \
--network "$NETWORK" --network-alias netwatch \
-e PORT=8080 -e NGINX_ENVSUBST_FILTER='^PORT$' \
-v "$ROOT/dist:/usr/share/nginx/html:ro" \
-v "$ROOT/nginx.conf:/etc/nginx/templates/default.conf.template:ro" \
-v /dev/null:/etc/nginx/trusted-proxies.conf:ro \
-v "$ROOT/security-headers.conf:/etc/nginx/security-headers.conf:ro" \
"$SERVER_IMAGE" > /dev/null
# The image's own entrypoint already exposes CDP on 9222 and passes
# --no-sandbox, so only extra flags belong here; re-specifying the
# debugging port collides with it and leaves the endpoint bound to
# loopback only. --hide-scrollbars keeps innerWidth equal to
# clientWidth, so the overflow assertion has no scrollbar-sized slack
# to hide behind, and matches the overlay scrollbars phones use.
docker run -d --rm --name "$BROWSER" --init --shm-size=1g \
--network "$NETWORK" \
"$BROWSER_IMAGE" \
--hide-scrollbars \
> /dev/null
# Chrome refuses DevTools requests whose Host header is neither
# localhost nor an IP address, so dial the container by address rather
# than by its network alias.
browser_ip="$(docker inspect \
-f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' \
"$BROWSER")"
timeout 900 docker run --rm --init --name "$HARNESS" \
--network "$NETWORK" \
--user "$(id -u):$(id -g)" \
-v "$ROOT:/app" \
-w /app \
-e NETWATCH_ROOT=/app \
-e NETWATCH_BASE_URL=http://netwatch:8080 \
-e "NETWATCH_CDP_URL=http://$browser_ip:9222" \
-e NETWATCH_ARTIFACT_DIR=/app/tmp/viewport \
"$NODE_IMAGE" \
node test/viewport/harness.js
}
main "$@"
-16
View File
@@ -1,16 +0,0 @@
#!/bin/sh
# script/install-precommit: install the git pre-commit hook that runs
# script/precommit. Our own extension to scripts-to-rule-them-all.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
hook=".git/hooks/pre-commit"
printf '#!/bin/sh\nset -e\nscript/precommit\n' > .git/hooks/pre-commit
chmod +x .git/hooks/pre-commit
echo "pre-commit hook installed: runs script/precommit"
}
main "$@"
-23
View File
@@ -1,23 +0,0 @@
#!/bin/sh
# script/lint: lint the whole repo: eslint over the frontend, then the Go
# linter over backend/.
#
# No linter runs on the host: this builds the frontend-lint and lint
# stages of Dockerfile, the digest-pinned node and golangci-lint images.
# The first runs eslint; the second runs the backend's fmt-check and
# lint targets. --no-cache makes each linter really run every time
# rather than reuse an earlier result, and each stage is built for its
# checks alone, so no image is kept.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
timeout 300 docker build --no-cache --target frontend-lint \
--output type=cacheonly .
timeout 300 docker build --no-cache --target lint \
--output type=cacheonly .
}
main "$@"
-12
View File
@@ -1,12 +0,0 @@
#!/bin/sh
# script/precommit: run by the git pre-commit hook; fails the commit if
# checks fail. Our own extension to scripts-to-rule-them-all.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
main() {
"$SCRIPT_DIR/check"
}
main "$@"
-12
View File
@@ -1,12 +0,0 @@
#!/bin/sh
# script/projectname: output the name of this project. Our own
# extension to scripts-to-rule-them-all. Other scripts that need the
# name (e.g. script/docker) call this, so they can stay identical
# across all repos.
set -eu
main() {
echo "netwatch"
}
main "$@"
-13
View File
@@ -1,13 +0,0 @@
#!/bin/sh
# script/setup: set up the repo for development after a fresh clone:
# installs dependencies and the git pre-commit hook.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
main() {
"$SCRIPT_DIR/bootstrap"
"$SCRIPT_DIR/install-precommit"
}
main "$@"
-17
View File
@@ -1,17 +0,0 @@
#!/bin/sh
# script/test: run the test suite for the whole repo: the frontend at
# the repo root, then the Go backend in backend/. Each half has its own
# 30-second limit, and there is none around both: from a cold Go build
# cache, compiling the backend's tests with the race detector can take
# 30 seconds on its own, and Go's -timeout leaves the compile out.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
script/frontend-test
backend/script/test
}
main "$@"
-13
View File
@@ -1,13 +0,0 @@
#!/bin/sh
# script/tidy: run go mod tidy in backend/, which adds the modules the
# Go sources import, drops those they no longer do, and updates go.sum.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT/backend"
go mod tidy
}
main "$@"
-24
View File
@@ -1,24 +0,0 @@
# The security headers REPO_POLICIES.md requires on every response.
# nginx.conf includes this file, which Dockerfile copies to
# /etc/nginx/security-headers.conf. always sends each header on error
# responses too.
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
# Scripts and styles load only from the page's own origin. Inline ones
# are blocked, style attributes in markup included, so style elements
# through classes or element.style. data: images are for the favicon
# in index.html. connect-src is * because the browser checks each probe in
# src/main.js against it, and also every redirect the probe follows,
# and several of those hosts redirect to others; a list of hosts here
# would block those probes. It also covers the reports the page sends
# to its own origin.
add_header Content-Security-Policy "default-src 'self'; connect-src *; img-src 'self' data:; object-src 'none'; base-uri 'none'; form-action 'none'; frame-ancestors 'none'" always;
add_header X-Frame-Options DENY always;
add_header X-Content-Type-Options nosniff always;
# The probed hosts are not told where the page is served from.
add_header Referrer-Policy no-referrer always;
add_header Permissions-Policy "accelerometer=(), camera=(), display-capture=(), geolocation=(), gyroscope=(), magnetometer=(), microphone=(), midi=(), payment=(), usb=()" always;
+92 -323
View File
@@ -1,24 +1,20 @@
import "./styles.css";
// --- Configuration -----------------------------------------------------------
// Timing, axis labels, and display constants. A target check times out
// after requestTimeout, 80% of updateInterval, so a round's checks have
// all finished before the next round is due; latency above maxLatency is
// recorded as a timeout. The sparkline Y-axis is capped at
// Timing, axis labels, and display constants. Latency above maxLatency is
// clamped to "unreachable". The sparkline Y-axis is capped at
// graphMaxLatency — values above it pin to the top of the chart but still
// display their real value in the latency figure. The history buffer holds
// maxHistoryPoints samples (historyDuration / updateInterval).
// reportInterval is how often collected samples are POSTed to the backend.
// The interval menu changes updateInterval while the page runs; the
// getters compute their values from it each time they are read.
export const CONFIG = {
const CONFIG = {
updateInterval: 3000,
maxHistoryPoints: 100,
reportInterval: 60000,
get historyDuration() {
return (this.maxHistoryPoints * this.updateInterval) / 1000;
},
get requestTimeout() {
return this.updateInterval * 0.8;
return Math.min(this.updateInterval - 100, 3000);
},
get maxLatency() {
return this.requestTimeout;
@@ -153,7 +149,7 @@ function formatUTCTimestamp(date) {
// --- Duration Formatting -----------------------------------------------------
export function humanDuration(seconds) {
function humanDuration(seconds) {
const h = Math.floor(seconds / 3600);
const m = Math.floor((seconds % 3600) / 60);
const s = seconds % 60;
@@ -199,14 +195,11 @@ async function detectGateway() {
// --- App State ---------------------------------------------------------------
export class HostState {
class HostState {
constructor(host, pinned = false) {
this.name = host.name;
this.url = host.url;
// Each entry is either a check's result, { timestamp, latency,
// error }, or a round skipped while paused, { timestamp,
// latency: null, paused: true }.
this.history = [];
this.history = []; // { timestamp, latency, paused }
this.lastLatency = null;
this.status = "pending"; // 'online' | 'offline' | 'error' | 'pending'
this.pinned = pinned;
@@ -268,7 +261,7 @@ export class HostState {
}
}
export class AppState {
class AppState {
constructor(localHosts) {
this.wan = WAN_HOSTS.map(
(h) => new HostState(h, h.name === "datavi.be"),
@@ -276,10 +269,6 @@ export class AppState {
this.local = localHosts.map((h) => new HostState(h));
this.paused = false;
this.tickCount = 0;
// The recovery probe's timer, null while it is not running, and the
// checks it started last.
this._recoveryProbeId = null;
this._recoveryProbeChecks = null;
}
get allHosts() {
@@ -357,171 +346,14 @@ export class AppState {
}
}
// --- Reporting ---------------------------------------------------------------
// A random UUIDv4. `crypto.randomUUID` exists only in secure contexts
// (HTTPS or localhost); over plain HTTP to any other host — the normal LAN
// deployment — it is undefined, so feature-detect it and otherwise build the
// id from `crypto.getRandomValues`, which is available in insecure contexts.
function randomId() {
if (typeof crypto !== "undefined" && crypto.randomUUID) {
return crypto.randomUUID();
}
const bytes = new Uint8Array(16);
crypto.getRandomValues(bytes);
bytes[6] = (bytes[6] & 0x0f) | 0x40; // version 4
bytes[8] = (bytes[8] & 0x3f) | 0x80; // variant 1
const hex = [...bytes].map((b) => b.toString(16).padStart(2, "0"));
return (
hex.slice(0, 4).join("") +
"-" +
hex.slice(4, 6).join("") +
"-" +
hex.slice(6, 8).join("") +
"-" +
hex.slice(8, 10).join("") +
"-" +
hex.slice(10, 16).join("")
);
}
// A random id identifying this browser across reports. Generated once and
// kept in localStorage; if storage is unavailable (e.g. private mode) a
// fresh id is used for this session only.
function getClientId() {
const key = "netwatch-client-id";
try {
let id = localStorage.getItem(key);
if (!id) {
id = randomId();
localStorage.setItem(key, id);
}
return id;
} catch {
return randomId();
}
}
// Build the delta report body the backend decodes, plus the new per-host
// high-water marks. Pure function of the passed state: `hosts` is an array
// of { name, url, status, history }, `since` maps a host url to the Unix-ms
// timestamp of the last sample already reported for it, and `now` is a Date.
// Only non-paused samples newer than the mark are included. Returns null
// when no host has an unreported sample.
export function buildReport(hosts, clientId, now, since) {
const reportHosts = [];
const marks = new Map();
for (const host of hosts) {
const mark = since.get(host.url) ?? 0;
const samples = [];
let high = mark;
for (const p of host.history) {
if (p.paused) continue;
if (p.timestamp <= mark) continue;
samples.push({
t: p.timestamp,
latency: p.latency,
error: p.error ?? null,
});
if (p.timestamp > high) high = p.timestamp;
}
if (samples.length === 0) continue;
reportHosts.push({
name: host.name,
url: host.url,
status: host.status,
history: samples,
});
marks.set(host.url, high);
}
if (reportHosts.length === 0) return null;
return {
body: {
clientId,
geo: null,
hosts: reportHosts,
timestamp: now.toISOString(),
},
marks,
};
}
// Periodically POSTs unreported samples to the same-origin backend. Holds
// the per-host high-water marks so each report is a delta; marks only
// advance on a delivered report, so a failed POST simply re-sends those
// samples next interval (bounded by the history window — whatever has since
// fallen out is dropped). Failure is quiet: one debug line per outage, one
// on recovery, never an alert, never a tight retry loop.
//
// Only one report is ever in flight, and it is abandoned after half the
// interval, so a slow POST can neither overlap the next report (which would
// carry the same samples) nor stall reporting for good.
class Reporter {
constructor(state, clientId, intervalMs) {
this.state = state;
this.clientId = clientId;
this.intervalMs = intervalMs;
this.marks = new Map();
this.failing = false;
this.sending = false;
this.timerId = null;
}
start() {
if (this.timerId) return;
this.timerId = setInterval(() => this.flush(), this.intervalMs);
}
async flush() {
if (this.sending || this.state.paused) return;
const report = buildReport(
this.state.allHosts,
this.clientId,
new Date(),
this.marks,
);
if (!report) return;
this.sending = true;
try {
const resp = await fetch("/api/v1/reports", {
method: "POST",
headers: { "Content-Type": "application/json" },
credentials: "omit",
body: JSON.stringify(report.body),
signal: AbortSignal.timeout(this.intervalMs / 2),
});
if (!resp.ok) throw new Error(`HTTP ${resp.status}`);
for (const [url, t] of report.marks) {
const current = this.marks.get(url) ?? 0;
this.marks.set(url, Math.max(current, t));
}
if (this.failing) {
log.debug("Report delivery recovered");
this.failing = false;
}
} catch (err) {
if (!this.failing) {
log.debug(`Report delivery failed: ${err.message}`);
this.failing = true;
}
} finally {
this.sending = false;
}
}
}
// --- Latency Measurement -----------------------------------------------------
// Checks one target. The check times out after CONFIG.requestTimeout; the
// caller can give it up sooner through the optional signal, which also ends
// it as a timeout.
export async function measureLatency(url, signal) {
async function measureLatency(url) {
const controller = new AbortController();
const timeoutId = setTimeout(
() => controller.abort(),
CONFIG.requestTimeout,
);
signal?.addEventListener("abort", () => controller.abort());
const targetUrl = new URL(url);
targetUrl.searchParams.set("_cb", Date.now().toString());
@@ -555,7 +387,7 @@ export async function measureLatency(url, signal) {
// --- Color Helpers -----------------------------------------------------------
export function latencyHex(latency) {
function latencyHex(latency) {
if (latency === null) return "#6b7280";
if (latency < 50) return "#22c55e";
if (latency < 100) return "#84cc16";
@@ -564,7 +396,7 @@ export function latencyHex(latency) {
return "#ef4444";
}
export function latencyClass(latency, status) {
function latencyClass(latency, status) {
if (status === "offline" || status === "error" || latency === null)
return "text-gray-500";
if (latency < 50) return "text-green-500";
@@ -589,8 +421,8 @@ class SparklineRenderer {
const ch = h - m.top - m.bottom;
ctx.clearRect(0, 0, w, h);
SparklineRenderer._drawYAxis(ctx, w, m, ch);
SparklineRenderer._drawXAxis(ctx, h, m, cw);
SparklineRenderer._drawYAxis(ctx, w, h, m, ch);
SparklineRenderer._drawXAxis(ctx, w, h, m, cw);
const len = history.length;
const pw = cw / (CONFIG.maxHistoryPoints - 1);
@@ -606,7 +438,7 @@ class SparklineRenderer {
SparklineRenderer._drawTip(ctx, history, getX, getY);
}
static _drawYAxis(ctx, w, m, ch) {
static _drawYAxis(ctx, w, h, m, ch) {
ctx.font = "300 12px monospace";
ctx.textAlign = "right";
ctx.textBaseline = "middle";
@@ -623,7 +455,7 @@ class SparklineRenderer {
}
}
static _drawXAxis(ctx, h, m, cw) {
static _drawXAxis(ctx, w, h, m, cw) {
ctx.textAlign = "center";
ctx.textBaseline = "top";
for (const tick of CONFIG.xAxisTicks) {
@@ -705,24 +537,7 @@ class SparklineRenderer {
// --- UI Renderer -------------------------------------------------------------
// The per-host status line must stay wrappable: its populated content is
// wider than the host column at a 320px viewport, and `whitespace-nowrap`
// here overflows the element and forces the whole document to scroll
// horizontally.
const STATUS_TEXT_CLASS = "status-text text-xs text-right col-span-2 mt-5";
// Escapes text for HTML, so it shows as written inside an element or a
// quoted attribute and is never read as markup.
function escapeHTML(text) {
return text
.replaceAll("&", "&amp;")
.replaceAll("<", "&lt;")
.replaceAll(">", "&gt;")
.replaceAll('"', "&quot;")
.replaceAll("'", "&#39;");
}
export function hostRowHTML(host, index, showPin = true) {
function hostRowHTML(host, index, showPin = true) {
const pinColor = host.pinned
? "text-blue-500"
: "text-gray-600 hover:text-gray-400";
@@ -740,14 +555,14 @@ export function hostRowHTML(host, index, showPin = true) {
${pinBtn}
<div class="w-[420px] flex-shrink-0 grid grid-cols-[minmax(0,1fr)_auto] items-center">
<div class="flex items-center gap-2 min-w-[200px]">
<div class="w-3 h-3 rounded-full flex-shrink-0 bg-[#6b7280]"></div>
<span class="font-medium text-white truncate">${escapeHTML(host.name)}</span>
<div class="w-3 h-3 rounded-full flex-shrink-0" style="background-color: ${latencyHex(null)}"></div>
<span class="font-medium text-white truncate">${host.name}</span>
</div>
<div class="latency-value text-4xl font-bold tabular-nums text-right mt-3" data-host="${index}">
<span class="text-gray-500">---</span>
</div>
<a href="${escapeHTML(host.url)}" target="_blank" rel="noopener" class="text-xs text-gray-500 truncate block col-span-2 -mt-2">${escapeHTML(host.url)}</a>
<div class="${STATUS_TEXT_CLASS} text-gray-500" data-host="${index}">waiting...</div>
<a href="${host.url}" target="_blank" rel="noopener" class="text-xs text-gray-500 truncate block col-span-2 -mt-2">${host.url}</a>
<div class="status-text text-xs text-gray-500 whitespace-nowrap text-right col-span-2 mt-5" data-host="${index}">waiting...</div>
</div>
<div class="flex-grow sparkline-container rounded overflow-hidden border border-gray-700/30">
<canvas class="sparkline-canvas w-full" data-host="${index}" height="${CONFIG.canvasHeight}"></canvas>
@@ -856,7 +671,7 @@ function buildUI(state) {
</p>
<p class="mt-2"><a href="https://git.eeqj.de/sneak/netwatch/commit/${__COMMIT_FULL__}" target="_blank" rel="noopener" class="text-gray-600 hover:text-gray-400">${__COMMIT_HASH__}</a></p>
<p class="mt-2">
<label class="debug-toggle-label cursor-pointer">
<label class="cursor-pointer">
<input type="checkbox" id="debug-toggle" class="mr-1">
<span>Debug log</span>
</label>
@@ -873,26 +688,6 @@ function buildUI(state) {
// --- UI Updaters -------------------------------------------------------------
// Renders `min 1ms / med 2ms / avg 3ms / max 4ms`. Each label, value and
// trailing separator is one unbreakable unit, so wrapping only ever happens
// between stats and a wrapped line never starts with a separator.
function statusStatsHTML(stats) {
return stats
.map(([label, value], i) => {
const sep =
i < stats.length - 1
? ` <span class="text-gray-500">/</span>`
: "";
return (
`<span class="whitespace-nowrap">` +
`<span class="text-gray-400">${label} </span>` +
`<span class="${latencyClass(value, "online")}">${value}ms</span>` +
`${sep}</span>`
);
})
.join(" ");
}
function updateHostRow(host, index) {
const latencyEl = document.querySelector(
`.latency-value[data-host="${index}"]`,
@@ -917,22 +712,28 @@ function updateHostRow(host, index) {
const min = host.minLatency();
const max = host.maxLatency();
if (host.status === "online" && avg !== null) {
statusEl.innerHTML = statusStatsHTML([
["min", min],
["med", med],
["avg", avg],
["max", max],
]);
statusEl.className = STATUS_TEXT_CLASS;
statusEl.innerHTML =
`<span class="text-gray-400">min </span><span class="${latencyClass(min, "online")}">${min}ms</span>` +
` <span class="text-gray-500">/</span> ` +
`<span class="text-gray-400">med </span><span class="${latencyClass(med, "online")}">${med}ms</span>` +
` <span class="text-gray-500">/</span> ` +
`<span class="text-gray-400">avg </span><span class="${latencyClass(avg, "online")}">${avg}ms</span>` +
` <span class="text-gray-500">/</span> ` +
`<span class="text-gray-400">max </span><span class="${latencyClass(max, "online")}">${max}ms</span>`;
statusEl.className =
"status-text text-xs whitespace-nowrap text-right col-span-2 mt-5";
} else if (host.status === "offline") {
statusEl.textContent = "unreachable";
statusEl.className = `${STATUS_TEXT_CLASS} text-red-400`;
statusEl.className =
"status-text text-xs text-red-400 whitespace-nowrap text-right col-span-2 mt-5";
} else if (host.status === "error") {
statusEl.textContent = "timeout";
statusEl.className = `${STATUS_TEXT_CLASS} text-orange-400`;
statusEl.className =
"status-text text-xs text-orange-400 whitespace-nowrap text-right col-span-2 mt-5";
} else {
statusEl.textContent = "connecting...";
statusEl.className = `${STATUS_TEXT_CLASS} text-gray-500`;
statusEl.className =
"status-text text-xs text-gray-500 whitespace-nowrap text-right col-span-2 mt-5";
}
SparklineRenderer.draw(canvas, host.history);
@@ -1062,16 +863,14 @@ function renderDebugLog() {
info: "text-gray-300",
debug: "text-gray-500",
};
el.replaceChildren(
...debugLog.map((entry) => {
el.innerHTML = debugLog
.map((entry) => {
const ts = formatUTCTimestamp(entry.timestamp);
const cls = levelColors[entry.level] || "text-gray-400";
const lvl = entry.level.toUpperCase().padEnd(7);
const line = document.createElement("div");
line.className = levelColors[entry.level] || "text-gray-400";
line.textContent = `${ts} ${lvl} ${entry.message}`;
return line;
}),
);
return `<div class="${cls}">${ts} ${lvl} ${entry.message}</div>`;
})
.join("");
el.scrollTop = el.scrollHeight;
}
@@ -1127,7 +926,7 @@ function sortAndRebuildWAN(state) {
// --- Main Loop ---------------------------------------------------------------
export async function tick(state, signal, onOffline) {
async function tick(state, onOffline) {
const ts = Date.now();
if (state.paused) {
@@ -1148,27 +947,12 @@ export async function tick(state, signal, onOffline) {
log.debug(`Tick #${state.tickCount + 1} started`);
// Each host's row shows its result as soon as its check ends. The
// result is discarded if by then the user has paused or the next round
// has given up this one's checks, and in the first tick (tickCount is
// still 0), which is discarded as a whole below. The row is looked up
// when the check ends, as a pin click may have re-sorted the rows since
// the round started.
await Promise.all(
state.allHosts.map(async (host) => {
const r = await measureLatency(host.url, signal);
if (state.paused || signal.aborted || state.tickCount === 0) {
return;
}
host.pushSample(ts, r);
updateHostRow(host, state.allHosts.indexOf(host));
log.debug(`${host.name}: ${r.error ? r.error : r.latency + "ms"}`);
}),
const results = await Promise.all(
state.allHosts.map((h) => measureLatency(h.url)),
);
// User may have paused, or the next round may have given up this
// one's checks, while awaiting results — skip the rest of the round
if (state.paused || signal.aborted) return;
// User may have paused while awaiting results — discard them
if (state.paused) return;
state.tickCount++;
@@ -1178,9 +962,12 @@ export async function tick(state, signal, onOffline) {
return;
}
// Redraw every row: if the user paused and resumed during this round,
// rows whose check ended before the resume still read "paused"
state.allHosts.forEach((host, i) => updateHostRow(host, i));
state.allHosts.forEach((host, i) => {
const r = results[i];
host.pushSample(ts, r);
updateHostRow(host, i);
log.debug(`${host.name}: ${r.error ? r.error : r.latency + "ms"}`);
});
// Sort after the first real check, then every 10 ticks thereafter
if (state.tickCount === 2 || state.tickCount % 10 === 1) {
@@ -1206,10 +993,9 @@ export async function tick(state, signal, onOffline) {
// --- Recovery Probe ----------------------------------------------------------
// When offline, check 4 random WAN hosts every 500ms, giving up the checks
// started 500ms before, so at most 4 are ever waiting. As soon as one
// answers, stop probing and start a new round at once.
function startRecoveryProbe(state, startRounds) {
// When offline, rapidly poll 4 random WAN hosts every 500ms. As soon as any
// responds, stop probing and fire a normal tick to refresh all hosts.
function startRecoveryProbe(state, triggerTick) {
if (state._recoveryProbeId) return; // already running
const candidates = [...state.wan];
for (let i = candidates.length - 1; i > 0; i--) {
@@ -1220,18 +1006,15 @@ function startRecoveryProbe(state, startRounds) {
log.notice(
`Recovery probe started (${canaries.map((h) => h.name).join(", ")})`,
);
state._recoveryProbeId = setInterval(() => {
state._recoveryProbeId = setInterval(async () => {
if (state.paused) return;
state._recoveryProbeChecks?.abort();
const checks = new AbortController();
state._recoveryProbeChecks = checks;
for (const host of canaries) {
measureLatency(host.url, checks.signal).then((r) => {
if (r.error !== null || checks.signal.aborted) return;
const results = await Promise.all(
canaries.map((h) => measureLatency(h.url)),
);
if (results.some((r) => r.error === null)) {
log.notice("Recovery probe: connectivity detected");
stopRecoveryProbe(state);
startRounds();
});
triggerTick();
}
}, 500);
}
@@ -1240,13 +1023,12 @@ function stopRecoveryProbe(state) {
if (state._recoveryProbeId) {
clearInterval(state._recoveryProbeId);
state._recoveryProbeId = null;
state._recoveryProbeChecks?.abort();
}
}
// --- Pause / Resume ----------------------------------------------------------
export function greyOutUI(state) {
function greyOutUI(state) {
// Grey out all host rows
state.allHosts.forEach((host, i) => {
const latencyEl = document.querySelector(
@@ -1262,7 +1044,8 @@ export function greyOutUI(state) {
}
if (statusEl) {
statusEl.textContent = "paused";
statusEl.className = `${STATUS_TEXT_CLASS} text-gray-500`;
statusEl.className =
"status-text text-xs text-gray-500 whitespace-nowrap text-right col-span-2 mt-5";
}
// Grey out the status dot
const row = document.querySelector(`.host-row[data-index="${i}"]`);
@@ -1348,6 +1131,25 @@ function handleResize(state) {
async function init() {
log.info("NetWatch starting");
// Mobile detection — show a friendly message and bail out early
if (window.innerWidth < 768) {
const app = document.getElementById("app");
app.innerHTML = `
<div class="mx-auto px-[5%] py-8">
<header class="mb-8">
<h1 class="text-3xl font-bold text-white"><a href="https://git.eeqj.de/sneak/netwatch" target="_blank" rel="noopener" class="underline decoration-dashed decoration-gray-500 underline-offset-4">NetWatch</a> by <a href="https://sneak.berlin" target="_blank" rel="noopener" class="text-blue-400 underline hover:text-blue-300">@sneak</a></h1>
<p class="text-gray-400 mt-2">Real-time network connectivity monitor</p>
</header>
<div style="display:flex;align-items:center;justify-content:center;min-height:50vh;">
<div style="background:#1f2937;border:1px solid #374151;border-radius:12px;padding:2rem 1.5rem;text-align:center;max-width:90vw;">
<p style="font-size:1.25rem;color:#e5e7eb;margin:0;">Not yet available on mobile.</p>
</div>
</div>
</div>`;
log.info("Mobile viewport detected — skipping monitor startup");
return;
}
// Probe common gateway IPs to find the local router
const gateway = await detectGateway();
const localHosts = [LOCAL_CPE];
@@ -1360,19 +1162,6 @@ async function init() {
buildUI(state);
log.info("UI built, starting tick loop");
// Reporting is best-effort: any failure setting it up (e.g. no usable
// crypto for the client id) must never stop the monitor from probing.
try {
const reporter = new Reporter(
state,
getClientId(),
CONFIG.reportInterval,
);
reporter.start();
} catch (err) {
log.error(`Reporting disabled: ${err.message}`);
}
document
.getElementById("pause-btn")
.addEventListener("click", () => togglePause(state));
@@ -1417,34 +1206,18 @@ async function init() {
updateClocks();
setInterval(updateClocks, 1000);
// Rounds never overlap: a round first gives up the last round's checks
// if they are still waiting, and the last round then records nothing
// more. At a steady interval they never are, as they time out at 80% of
// it; they can be when a round starts early, after an interval change
// or when the recovery probe finds a target answering.
let roundChecks = new AbortController();
function doTick() {
roundChecks.abort();
roundChecks = new AbortController();
tick(state, roundChecks.signal, () =>
startRecoveryProbe(state, startRounds),
);
tick(state, () => startRecoveryProbe(state, doTick));
}
// Starts a round now and then one every CONFIG.updateInterval.
let tickIntervalId;
function startRounds() {
clearInterval(tickIntervalId);
doTick();
tickIntervalId = setInterval(doTick, CONFIG.updateInterval);
}
startRounds();
let tickIntervalId = setInterval(doTick, CONFIG.updateInterval);
document
.getElementById("interval-select")
.addEventListener("change", (e) => {
const newInterval = parseInt(e.target.value, 10);
clearInterval(tickIntervalId);
CONFIG.updateInterval = newInterval;
log.notice(
`Interval changed to ${humanDuration(newInterval / 1000)}, history reset`,
@@ -1493,20 +1266,16 @@ async function init() {
// Start immediately with new interval
stopRecoveryProbe(state);
startRounds();
doTick();
tickIntervalId = setInterval(doTick, CONFIG.updateInterval);
});
window.addEventListener("resize", () => handleResize(state));
setTimeout(() => handleResize(state), 100);
}
// Bootstrap only when loaded as the page: a real DOM containing the #app
// mount point this module renders into. Importing the module in a unit test
// (which has no #app) runs nothing, so its exports can be tested in isolation.
if (typeof document !== "undefined" && document.getElementById("app")) {
if (document.readyState === "loading") {
document.addEventListener("DOMContentLoaded", init);
} else {
init();
}
}
-126
View File
@@ -14,38 +14,6 @@ body {
ui-monospace, SFMono-Regular, "SF Mono", Menlo, Consolas, monospace;
}
/* ---- Minimum tap targets ----------------------------------------------
Every interactive control is at least 44x44 CSS px (Apple HIG, WCAG 2.2
SC 2.5.5). Not scoped to a breakpoint or to `pointer: coarse`: a large
phone in landscape is above the 768px breakpoint and still a touch
device. */
/* The button grows to 44x44 while the negative margins keep its layout
footprint at the 16x16 of the icon inside it, so row height and the
icon's position are unchanged. */
.pin-btn {
display: flex;
align-items: center;
justify-content: center;
width: 2.75rem;
height: 2.75rem;
margin: -0.875rem;
}
/* A select paints its own background and border, so it has to actually be
44 tall rather than borrow the trick above. */
#interval-select {
min-height: 2.75rem;
}
/* The tappable target for #debug-toggle is the label wrapping it. */
.debug-toggle-label {
display: inline-flex;
align-items: center;
justify-content: center;
min-height: 2.75rem;
}
.sparkline-container {
background: linear-gradient(
to bottom,
@@ -53,97 +21,3 @@ body {
rgba(255, 255, 255, 0) 100%
);
}
/* ---- Mobile responsive layout (portrait / narrow viewports) ---- */
@media (max-width: 768px) {
/* Header: stack title and controls vertically */
header .flex.items-center.justify-between {
flex-direction: column;
align-items: flex-start !important;
gap: 1rem;
}
header .flex.flex-col.items-end {
align-items: flex-start !important;
flex-direction: row;
flex-wrap: wrap;
gap: 0.75rem;
}
/* Pause button: smaller on mobile, but not below the tap-target floor */
#pause-btn {
padding: 0.5rem 1rem;
min-height: 2.75rem;
}
#pause-btn svg {
width: 1.25rem;
height: 1.25rem;
}
#pause-text {
font-size: 0.875rem;
}
/* Summary box: wrap into a grid for readability */
#summary {
display: flex;
flex-wrap: wrap;
gap: 0.25rem 0.5rem;
justify-content: center;
line-height: 1.6;
}
/* Hide the pipe separators on mobile */
#summary .text-gray-600.mx-3 {
display: none;
}
/* Host row: stack vertically */
.host-row .flex.items-center.gap-4 {
flex-direction: column;
align-items: stretch !important;
gap: 0.5rem;
}
/* Info section: full width, remove fixed width */
.host-row .w-\[420px\] {
width: 100% !important;
display: grid;
grid-template-columns: 1fr auto;
align-items: center;
}
/* Host name row with dot */
.host-row .flex.items-center.gap-2.min-w-\[200px\] {
min-width: 0;
}
/* Latency value: slightly smaller on mobile */
.host-row .latency-value {
font-size: 1.875rem;
line-height: 2.25rem;
}
/* Sparkline: full width below the info */
.host-row .sparkline-container {
width: 100%;
flex-shrink: 0;
}
/* Pin button: inline with the host info */
.host-row .pin-btn {
position: absolute;
right: 0.5rem;
top: 0.5rem;
}
.host-row {
position: relative;
}
/* Footer legend: wrap nicely */
footer p {
line-height: 1.8;
}
}
-416
View File
@@ -1,416 +0,0 @@
// Unit tests for src/main.js, run with Node's built-in test runner by the
// test script in package.json. Importing the module does not start the
// page.
import { beforeEach, test } from "node:test";
import assert from "node:assert/strict";
import {
AppState,
CONFIG,
greyOutUI,
hostRowHTML,
HostState,
humanDuration,
latencyClass,
latencyHex,
measureLatency,
tick,
} from "../../src/main.js";
// There is no page here, so the tests stand in for it. The debug log looks
// for its panel by id and finds none. Each element of a host's row that
// tick or greyOutUI draws into is a plain object, made the first time a
// test looks it up and kept in elements under its selector until the next
// test starts. As on a page, writing its text replaces its markup; the
// status dot greyOutUI looks for in it is not there. Drawing a sparkline
// does nothing; it looks for the pixel ratio on window and finds none.
let elements;
beforeEach(() => {
elements = {};
});
const doNothing = () => {};
const canvasContext = {
clearRect: doNothing,
beginPath: doNothing,
moveTo: doNothing,
lineTo: doNothing,
stroke: doNothing,
fill: doNothing,
fillRect: doNothing,
fillText: doNothing,
arc: doNothing,
};
globalThis.window = {};
globalThis.document = {
getElementById: () => null,
querySelector: (selector) =>
(elements[selector] ??= {
getContext: () => canvasContext,
querySelector: () => null,
set textContent(text) {
this.innerHTML = text;
},
}),
};
// What tick last wrote into the latency figure in host's row, or undefined
// if it has written nothing there.
function latencyFigure(state, host) {
const index = state.allHosts.indexOf(host);
return elements[`.latency-value[data-host="${index}"]`]?.innerHTML;
}
// What was last written into the status text in host's row.
function statusText(state, host) {
const index = state.allHosts.indexOf(host);
return elements[`.status-text[data-host="${index}"]`]?.innerHTML;
}
// Mocks the clock for test t, so that a check lasting seconds takes no real
// time, and replaces fetch with targets that each answer after
// answerAfter(url) milliseconds of that clock, or never when that is
// Infinity. Both are restored when the test ends.
function mockTargets(t, answerAfter) {
t.mock.timers.enable({ apis: ["setTimeout", "Date"] });
t.mock.method(performance, "now", () => Date.now());
t.mock.method(
globalThis,
"fetch",
(url, { signal }) =>
new Promise((resolve, reject) => {
if (answerAfter(url) !== Infinity) {
setTimeout(resolve, answerAfter(url));
}
signal.addEventListener("abort", () => reject(signal.reason));
}),
);
}
// The result of check if it has ended, otherwise "still waiting".
function settled(check) {
return Promise.race([
check,
new Promise((resolve) => setImmediate(resolve, "still waiting")),
]);
}
for (const interval of [10000, 30000]) {
const timeout = interval * 0.8;
// Over 3 seconds, which the timeout used to be capped at.
const slowAnswer = timeout - 1000;
test(`at a ${interval}ms interval, an answer after ${slowAnswer}ms is recorded with its real time`, async (t) => {
CONFIG.updateInterval = interval;
mockTargets(t, () => slowAnswer);
const check = measureLatency("https://target.test");
t.mock.timers.tick(slowAnswer);
assert.deepEqual(await settled(check), {
latency: slowAnswer,
error: null,
});
});
test(`at a ${interval}ms interval, a target that never answers is recorded as a timeout after ${timeout}ms`, async (t) => {
CONFIG.updateInterval = interval;
mockTargets(t, () => Infinity);
const check = measureLatency("https://target.test");
t.mock.timers.tick(timeout - 1);
assert.equal(await settled(check), "still waiting");
t.mock.timers.tick(1);
assert.deepEqual(await settled(check), {
latency: null,
error: "timeout",
});
});
}
test("at a 30000ms interval, a target answering after 1000ms shows in its row while another target's check is still waiting", async (t) => {
CONFIG.updateInterval = 30000;
const state = new AppState([
{ name: "Answering", url: "https://answering.test" },
]);
const answering = state.local[0];
const waiting = state.wan[0];
// No target but the answering one ever answers.
mockTargets(t, (url) => (url.startsWith(answering.url) ? 1000 : Infinity));
// The third tick: the first is discarded as a whole, and the second ends
// by sorting the rows, which rebuilds a page that is not here.
state.tickCount = 2;
const round = tick(state, new AbortController().signal);
t.mock.timers.tick(1000);
assert.equal(await settled(round), "still waiting");
assert.match(latencyFigure(state, answering), />1000</);
assert.equal(latencyFigure(state, waiting), undefined);
assert.equal(state.tickCount, 2);
// The round ends, once, when the last check times out.
t.mock.timers.tick(CONFIG.requestTimeout - 1000);
assert.notEqual(await settled(round), "still waiting");
assert.equal(state.tickCount, 3);
});
// In the next three tests, the answering target's check is still waiting
// when something happens after which its result must not show.
test("at a 30000ms interval, a check still waiting when its round is given up does not show in its row", async (t) => {
CONFIG.updateInterval = 30000;
const state = new AppState([
{ name: "Answering", url: "https://answering.test" },
]);
const answering = state.local[0];
mockTargets(t, (url) => (url.startsWith(answering.url) ? 1000 : Infinity));
state.tickCount = 2;
const roundChecks = new AbortController();
const round = tick(state, roundChecks.signal);
t.mock.timers.tick(500);
// As a round started early does to the last round's checks.
roundChecks.abort();
assert.notEqual(await settled(round), "still waiting");
assert.equal(latencyFigure(state, answering), undefined);
});
test("at a 30000ms interval, a check still waiting when the user pauses does not show in its row", async (t) => {
CONFIG.updateInterval = 30000;
const state = new AppState([
{ name: "Answering", url: "https://answering.test" },
]);
const answering = state.local[0];
mockTargets(t, (url) => (url.startsWith(answering.url) ? 1000 : Infinity));
state.tickCount = 2;
const round = tick(state, new AbortController().signal);
t.mock.timers.tick(500);
state.paused = true;
t.mock.timers.tick(500);
assert.equal(await settled(round), "still waiting");
assert.equal(latencyFigure(state, answering), undefined);
});
test("at a 30000ms interval, a check in the first round does not show in its row", async (t) => {
CONFIG.updateInterval = 30000;
const state = new AppState([
{ name: "Answering", url: "https://answering.test" },
]);
const answering = state.local[0];
mockTargets(t, (url) => (url.startsWith(answering.url) ? 1000 : Infinity));
const round = tick(state, new AbortController().signal);
t.mock.timers.tick(1000);
assert.equal(await settled(round), "still waiting");
assert.equal(latencyFigure(state, answering), undefined);
});
test("at a 30000ms interval, after the user pauses and resumes during a round, no row reads paused once its last check ends", async (t) => {
CONFIG.updateInterval = 30000;
const state = new AppState([
{ name: "Answering", url: "https://answering.test" },
]);
const answering = state.local[0];
mockTargets(t, (url) => (url.startsWith(answering.url) ? 1000 : Infinity));
state.tickCount = 2;
const round = tick(state, new AbortController().signal);
t.mock.timers.tick(1000);
assert.equal(await settled(round), "still waiting");
// The user pauses, which greys out every row, and resumes, which leaves
// the rows as they are, as togglePause does. The answering target's
// check has already ended, so only the redraw of every row at the end
// of the round can take "paused" out of its row.
state.paused = true;
greyOutUI(state);
state.paused = false;
assert.equal(statusText(state, answering), "paused");
// The round ends when the last check times out.
t.mock.timers.tick(CONFIG.requestTimeout - 1000);
assert.notEqual(await settled(round), "still waiting");
for (const host of state.allHosts) {
assert.notEqual(statusText(state, host), "paused", host.name);
}
});
// The page shows &lt; &gt; &quot; &amp; and &#39; in a row's markup as
// < > " & and '.
test(`a target whose name and URL hold < > " & and ' shows those characters in its row`, () => {
const host = new HostState({
name: `<b>"x" & 'y'</b>`,
url: `https://x.test/<b>?a="x"&b='y'`,
});
const row = hostRowHTML(host, 0);
assert.doesNotMatch(row, /<b>/);
assert.ok(
row.includes(
">&lt;b&gt;&quot;x&quot; &amp; &#39;y&#39;&lt;/b&gt;</span>",
),
);
assert.ok(
row.includes(
'href="https://x.test/&lt;b&gt;?a=&quot;x&quot;&amp;b=&#39;y&#39;"',
),
);
assert.ok(
row.includes(
">https://x.test/&lt;b&gt;?a=&quot;x&quot;&amp;b=&#39;y&#39;</a>",
),
);
});
for (const [seconds, text] of [
[0, "0s"],
[1, "1s"],
[59, "59s"],
[60, "1m"],
[61, "1m1s"],
[3599, "59m59s"],
[3600, "1h"],
[3601, "1h1s"],
[3660, "1h1m"],
[3661, "1h1m1s"],
]) {
test(`humanDuration writes ${seconds} seconds as ${text}`, () => {
assert.equal(humanDuration(seconds), text);
});
}
// The colour of a target's latency figure, as a class, and of its sparkline
// for an answer after latency ms, as the colour coding in README.md gives
// them. The rows sit either side of each boundary.
for (const [latency, figure, sparkline] of [
[0, "text-green-500", "#22c55e"],
[49, "text-green-500", "#22c55e"],
[50, "text-lime-500", "#84cc16"],
[99, "text-lime-500", "#84cc16"],
[100, "text-yellow-500", "#eab308"],
[199, "text-yellow-500", "#eab308"],
[200, "text-orange-500", "#f97316"],
[499, "text-orange-500", "#f97316"],
[500, "text-red-500", "#ef4444"],
]) {
test(`an answer after ${latency}ms has a ${figure} figure and a ${sparkline} sparkline`, () => {
assert.equal(latencyClass(latency, "online"), figure);
assert.equal(latencyHex(latency), sparkline);
});
}
test("a check that timed out or found its target unreachable has a grey figure and sparkline", () => {
assert.equal(latencyClass(null, "error"), "text-gray-500");
assert.equal(latencyClass(null, "offline"), "text-gray-500");
assert.equal(latencyHex(null), "#6b7280");
});
// A target whose checks, in turn, answered after each of latencies ms, or,
// for null, found it unreachable.
function hostAfter(latencies) {
const host = new HostState({ name: "Target", url: "https://target.test" });
for (const latency of latencies) {
host.pushSample(
Date.now(),
latency === null
? { latency: null, error: "unreachable" }
: { latency, error: null },
);
}
return host;
}
for (const { history, latencies, statistics } of [
{
history: "no checks",
latencies: [],
statistics: { min: null, max: null, average: null, median: null },
},
{
history: "only unreachable checks",
latencies: [null, null, null],
statistics: { min: null, max: null, average: null, median: null },
},
{
history: "three answers and an unreachable check",
latencies: [30, null, 10, 20],
statistics: { min: 10, max: 30, average: 20, median: 20 },
},
{
// The median of an even number of answers is the mean of the
// middle two. It and the average, 23.75, are rounded.
history: "four answers",
latencies: [10, 40, 20, 25],
statistics: { min: 10, max: 40, average: 24, median: 23 },
},
{
// Sorted as text rather than as numbers, these answers would put
// 100 in the middle. The average, 39.67, is rounded.
history: "answers with different numbers of digits",
latencies: [100, 9, 10],
statistics: { min: 9, max: 100, average: 40, median: 10 },
},
]) {
test(`a target's min, max, average and median latency over ${history}`, () => {
const host = hostAfter(latencies);
assert.deepEqual(
{
min: host.minLatency(),
max: host.maxLatency(),
average: host.averageLatency(),
median: host.medianLatency(),
},
statistics,
);
});
}
// An app state in which, of the WAN targets, the first timedOut timed out,
// the next unreachable were found unreachable, the next answered answered
// after latency ms, and the rest have not been checked yet.
function stateAfter({ timedOut, unreachable, answered, latency }) {
const state = new AppState([]);
const results = [
...Array(timedOut).fill({ latency: null, error: "timeout" }),
...Array(unreachable).fill({ latency: null, error: "unreachable" }),
...Array(answered).fill({ latency, error: null }),
];
results.forEach((result, i) => state.wan[i].pushSample(Date.now(), result));
return state;
}
// For each number the health is decided by, the rows put it one under, at
// and one over its threshold.
for (const { timedOut, unreachable = 0, answered, latency, health } of [
// Offline: more than 10 timed out and at most 4 answered.
{ timedOut: 9, answered: 4, latency: 30, health: "degraded" },
{ timedOut: 10, answered: 4, latency: 30, health: "degraded" },
{ timedOut: 11, answered: 4, latency: 30, health: "offline" },
{ timedOut: 11, answered: 3, latency: 30, health: "offline" },
{ timedOut: 11, answered: 5, latency: 30, health: "degraded" },
// Otherwise degraded: more than 4 timed out.
{ timedOut: 3, answered: 10, latency: 30, health: "healthy" },
{ timedOut: 4, answered: 10, latency: 30, health: "healthy" },
{ timedOut: 5, answered: 10, latency: 30, health: "degraded" },
// Otherwise slow: more than 3 answered after more than 1000ms.
{ timedOut: 0, answered: 4, latency: 999, health: "healthy" },
{ timedOut: 0, answered: 4, latency: 1000, health: "healthy" },
{ timedOut: 0, answered: 4, latency: 1001, health: "slow" },
{ timedOut: 0, answered: 2, latency: 1001, health: "healthy" },
{ timedOut: 0, answered: 3, latency: 1001, health: "healthy" },
// A target found unreachable counts as one that timed out.
{
timedOut: 5,
unreachable: 6,
answered: 4,
latency: 30,
health: "offline",
},
{
timedOut: 0,
unreachable: 5,
answered: 10,
latency: 30,
health: "degraded",
},
]) {
test(`with ${timedOut} WAN targets timed out, ${unreachable} found unreachable and ${answered} answering after ${latency}ms, the health is ${health}`, () => {
const state = stateAfter({ timedOut, unreachable, answered, latency });
assert.equal(state.healthStatus(), health);
});
}
-114
View File
@@ -1,114 +0,0 @@
# Responsive-layout harness
Automated verification of the responsive layout that landed in #5. Run it with:
```bash
make frontend-viewport-test
```
It builds `dist/`, serves it from the same digest-pinned `nginx` image and the
same `nginx.conf` the shipping container uses, drives a digest-pinned headless
Chrome against it over CDP, and asserts on computed layout at every viewport
width derived from the app's own CSS. Screenshots land in `tmp/viewport/`
alongside a `results.json`; they are artifacts for a human to look at when
something fails, not the evidence. The assertions are the evidence.
The target is deliberately outside `make check`: it needs Docker and takes
minutes, and `make test` has to stay under 20 seconds.
## How the widths are chosen
Not from a list of phone models. `viewports.js` parses the `@media` conditions
out of `src/styles.css` and scans `src/main.js` and `index.html` for Tailwind
responsive prefixes, then tests every breakpoint it finds at one pixel below it,
exactly on it, and one pixel above it. A generic 375px "phone" test sails
straight past an off-by-one at a media query boundary; `max-width: 768px`
matches _at_ 768, and the sweep pins down which side of that line each layout is
on.
Nothing hardcodes 768. Add a second media block or start using `md:` classes and
the new breakpoint is covered without this directory being touched. The app is
desktop-first today (all narrow rules live in `max-width` blocks); a
`min-width`-only, mobile-first set is handled as its inverse, and a set that
mixes the two makes the run fail loudly rather than test the right widths with
the wrong expectation. Four further viewports are fixed anchors, each with a
stated reason: a 320px floor, a 1280px desktop baseline, and two phone-landscape
sizes straddling the breakpoint for the rotation case.
## What it asserts
- **app-rendered** — enough host rows exist and enough of them show a numeric
latency. This one exists so the rest cannot pass vacuously against a blank
page.
- **no-horizontal-overflow** — `documentElement.scrollWidth` fits the layout
viewport, with the widest offending element named.
- **nothing-past-viewport-edge** — no visible element's box extends past the
viewport edge.
- **no-clipped-text** — nothing hides text behind `overflow: hidden`. Deliberate
ellipsis truncation (Tailwind's `truncate`, used on host names and URLs) is
excluded: it is a design choice, not breakage.
- **tap-targets-44px** — every interactive control is at least 44x44 CSS px on
touch viewports, _and_ each selector in the control list matched at least the
number of visible elements it declares: one of each single control, and one
pin button per WAN host row. The second half is what stops the check passing
vacuously: with size alone, a renamed class would take its controls out of the
measured set and the check would report "all 0 controls are at least 44x44"
and pass. See below.
- **host-rows-stacked / host-rows-side-by-side** — the rows genuinely reflow.
Computed `flex-direction` _and_ the actual geometry are checked, and in the
narrow layout the info block and the sparkline must each occupy essentially
the full row width. A row that merely shrank its 420px column would fail.
- **probing-still-runs / gateway-detection-still-runs** — narrow viewports keep
probing and keep detecting the gateway. The mobile early-return path proposed
in #8 was rejected; this is what would catch it coming back.
### The tap-target threshold
44x44 CSS px. That is the figure in Apple's Human Interface Guidelines and in
WCAG 2.2 SC 2.5.5 "Target Size (Enhanced)". WCAG 2.2 SC 2.5.8 (level AA) sets a
lower 24x24 floor, but that floor comes with a spacing exception these controls
do not qualify for — the pin buttons sit directly against the host name they
belong to.
## Determinism
The browser container runs on an `--internal` docker network and has no route to
the internet, so the app's latency probes cannot reach anything real. The
harness answers them itself from a fixed delay table, with a deterministic
fraction failed outright, so the rows render a realistic spread of one-, two-
and three-digit latencies plus some unreachable rows. That spread is what the
layout has to survive; a `---` placeholder in every row would not exercise it.
## What this cannot verify
Real limits, so nobody re-parks this issue as needing hardware:
- **Non-Chromium engines.** This is Chrome. iOS Safari is WebKit and cannot be
emulated by it; Safari-specific bugs (viewport units under a collapsing URL
bar, `-webkit-fill-available`, form control metrics) will not show up here.
- **Real touch input.** `hasTouch` emulation changes what the page is told, not
how a finger behaves. Gesture handling, scroll momentum, double-tap zoom and
hover-state fallbacks on touch are out of scope.
- **Physical pixel density and rendering.** `deviceScaleFactor` is set, but
subpixel antialiasing, OLED colour rendering and actual legibility at a given
physical size are not measurable here.
- **Fonts.** The container has DejaVu, not the platform's own UI monospace. Text
metrics are therefore close to, but not identical to, a real device — a layout
that fits here by a few pixels might not there.
- **On-device performance.** Canvas sparkline redraw cost, battery, and
behaviour on a slow radio are not measured.
- **Browser chrome.** The address bar, safe-area insets and notch cutouts are
not simulated.
Everything else this issue was actually about — does the layout reflow, does
anything overflow, is content clipped, are the controls big enough — is a
function of viewport width and CSS, and is covered above.
## Relation to the unit tests
Complementary layers, not two stacks. The unit tests in `test/unit/`, which
`make test` runs with Node's built-in test runner, exercise the functions
`src/main.js` exports in-process with no browser. This harness exercises
rendered layout in a real engine and is the only thing here that can see a media
query. Neither replaces the other; assertions about computed styles and element
geometry belong here, assertions about functions belong in `test/unit/`.
-246
View File
@@ -1,246 +0,0 @@
// Pass/fail decisions for the responsive-layout harness.
//
// Kept in node rather than in the page so that a failure can be reported
// with the measurements that produced it. Every check runs at every
// viewport; none of them short-circuits, so one failure does not hide the
// rest.
// Minimum tap target, in CSS pixels. 44x44 is the figure in Apple's Human
// Interface Guidelines and in WCAG 2.2 SC 2.5.5 "Target Size (Enhanced)".
// WCAG 2.2 SC 2.5.8 (level AA) sets a lower 24x24 floor, but that floor
// comes with a spacing exception these controls do not qualify for: the
// pin buttons sit directly against the host name they belong to. Held at
// 44 deliberately.
export const MIN_TAP_TARGET_PX = 44;
// The controls named in the definition of done, plus the pause button.
// Each carries the smallest number of *visible* instances the page has to
// contain, worked out from the facts gathered from that page, for the
// tap-target oracle to be measuring every control it should.
//
// Without those floors the check is inert: `undersized` is empty both when
// every control is large enough and when the selectors have gone stale and
// matched nothing, and the pass condition cannot tell those apart. A single
// combined floor would not be enough either — 26 pin buttons would cover
// for all three singleton controls vanishing at once — so the floor is per
// selector, and one stale selector out of four fails the check.
export const INTERACTIVE_CONTROLS = [
{ selector: "#pause-btn", minCount: () => 1 },
{ selector: "#interval-select", minCount: () => 1 },
// One per WAN host row, so pin buttons missing from even one row fail
// the check rather than only a drop below some fixed number.
{ selector: ".pin-btn", minCount: (facts) => facts.wanRowCount },
{ selector: "#debug-toggle", minCount: () => 1 },
];
export const INTERACTIVE_SELECTORS = INTERACTIVE_CONTROLS.map(
(control) => control.selector,
);
// A host row is only "reflowed" if it stacked *and* went full width.
// A row that merely shrank its 420px info column would keep
// flex-direction: row, and a row that stacked but left the info column at
// its fixed width would fail the width test.
const FULL_WIDTH_FRACTION = 0.9;
function summarise(items, format, limit = 3) {
const shown = items.slice(0, limit).map(format).join("; ");
const rest = items.length > limit ? ` (+${items.length - limit} more)` : "";
return shown + rest;
}
// Collapse an overflow report to the elements actually responsible.
// Identical elements (24 host rows all doing the same thing) are counted
// rather than listed, and the deepest ones come first, since every
// ancestor of an overflowing element also reports as overflowing.
function deepestOffenders(entries) {
const byElement = new Map();
for (const entry of entries) {
const reach = entry.reach ?? entry.right;
const existing = byElement.get(entry.el);
if (existing) {
existing.count += 1;
existing.reach = Math.max(existing.reach, reach);
} else {
byElement.set(entry.el, { ...entry, reach, count: 1 });
}
}
return [...byElement.values()].sort(
(a, b) => b.depth - a.depth || b.reach - a.reach,
);
}
function checkRowLayout(row, expectStacked) {
if (expectStacked) {
if (row.flexDirection !== "column") {
return `row ${row.index}: flex-direction is ${row.flexDirection}, expected column`;
}
if (row.sparkline.top < row.info.bottom - 1) {
return `row ${row.index}: sparkline top ${row.sparkline.top} is above info bottom ${row.info.bottom} — still side by side`;
}
const minWidth = row.containerWidth * FULL_WIDTH_FRACTION;
if (row.info.width < minWidth) {
return `row ${row.index}: info block is ${row.info.width}px of ${row.containerWidth}px — shrunk, not reflowed`;
}
if (row.sparkline.width < minWidth) {
return `row ${row.index}: sparkline is ${row.sparkline.width}px of ${row.containerWidth}px — shrunk, not reflowed`;
}
return null;
}
if (row.flexDirection !== "row") {
return `row ${row.index}: flex-direction is ${row.flexDirection}, expected row`;
}
if (row.sparkline.left < row.info.right - 1) {
return `row ${row.index}: sparkline left ${row.sparkline.left} overlaps info right ${row.info.right} — not side by side`;
}
return null;
}
export function evaluateChecks(facts, viewport, probes) {
const checks = [];
const check = (name, ok, detail) => checks.push({ name, ok, detail });
// Guard against the whole harness passing vacuously because the page
// never rendered. Everything below is only meaningful if this holds.
check(
"app-rendered",
facts.wanRowCount >= 10 && facts.numericLatencies >= 5,
`${facts.wanRowCount} WAN host rows, ${facts.numericLatencies} showing a numeric latency`,
);
const viewportWidth = Math.min(facts.innerWidth, facts.documentClientWidth);
const culprits = deepestOffenders([
...facts.overflowing,
...facts.contentOverflowing,
]);
check(
"no-horizontal-overflow",
facts.documentScrollWidth <= viewportWidth,
`documentElement.scrollWidth ${facts.documentScrollWidth} vs viewport ${viewportWidth}` +
(culprits.length === 0
? ""
: "; widest content: " +
summarise(
culprits,
(c) =>
`${c.el} reaches ${Math.round(c.reach)}px${c.count > 1 ? ` (x${c.count})` : ""}`,
)),
);
check(
"nothing-past-viewport-edge",
facts.overflowing.length === 0,
facts.overflowing.length === 0
? "no element extends past the viewport"
: `${facts.overflowing.length} element(s) past the edge: ` +
summarise(
facts.overflowing,
(o) => `${o.el} spans ${o.left}..${o.right}`,
),
);
check(
"no-clipped-text",
facts.clipped.length === 0,
facts.clipped.length === 0
? "no element hides text behind overflow (deliberate ellipsis excluded)"
: `${facts.clipped.length} element(s) clipping text: ` +
summarise(
facts.clipped,
(c) =>
`${c.el} scrollWidth ${c.scrollWidth} > clientWidth ${c.clientWidth}`,
),
);
if (viewport.touch) {
// Presence first: a selector that matches nothing contributes no
// undersized targets, so without this the check would report
// "all 0 controls are at least 44x44" and pass.
const seen = new Map();
for (const target of facts.tapTargets) {
seen.set(target.selector, (seen.get(target.selector) ?? 0) + 1);
}
const missing = INTERACTIVE_CONTROLS.filter(
(control) =>
(seen.get(control.selector) ?? 0) < control.minCount(facts),
);
const undersized = facts.tapTargets.filter(
(t) => t.width < MIN_TAP_TARGET_PX || t.height < MIN_TAP_TARGET_PX,
);
const bySelector = new Map();
for (const target of undersized) {
const existing = bySelector.get(target.selector);
if (!existing || target.width * target.height < existing.area) {
bySelector.set(target.selector, {
...target,
area: target.width * target.height,
count: (existing?.count ?? 0) + 1,
});
} else {
existing.count += 1;
}
}
const detail = [];
if (missing.length > 0) {
detail.push(
"oracle is not measuring every control: " +
summarise(
missing,
(c) =>
`${c.selector} matched ${seen.get(c.selector) ?? 0} visible element(s), expected at least ${c.minCount(facts)}`,
4,
),
);
}
detail.push(
undersized.length === 0
? `${facts.tapTargets.length} controls measured, all at least ${MIN_TAP_TARGET_PX}x${MIN_TAP_TARGET_PX}`
: `${undersized.length} of ${facts.tapTargets.length} controls below ${MIN_TAP_TARGET_PX}x${MIN_TAP_TARGET_PX}: ` +
summarise(
[...bySelector.values()],
(t) =>
`${t.selector} ${t.width}x${t.height}${t.count > 1 ? ` (x${t.count})` : ""}`,
4,
),
);
check(
`tap-targets-${MIN_TAP_TARGET_PX}px`,
missing.length === 0 && undersized.length === 0,
detail.join("; "),
);
}
const badRows = facts.rows
.map((row) => checkRowLayout(row, viewport.expectStacked))
.filter(Boolean);
check(
viewport.expectStacked ? "host-rows-stacked" : "host-rows-side-by-side",
facts.rows.length > 0 && badRows.length === 0,
facts.rows.length === 0
? "no host rows were measured"
: badRows.length === 0
? `all ${facts.rows.length} rows laid out as expected`
: `${badRows.length} of ${facts.rows.length} rows wrong: ` +
summarise(badRows, (r) => r),
);
// The mobile early-return path proposed in #8 was rejected: narrow
// viewports must keep probing and keep detecting the gateway, not
// quietly skip work.
check(
"probing-still-runs",
probes.attempted > 0,
`${probes.attempted} outbound probe requests issued`,
);
check(
"gateway-detection-still-runs",
facts.gatewayDetected,
facts.gatewayDetected
? "Local Gateway row present"
: "no Local Gateway row — gateway detection did not run or did not complete",
);
return checks;
}
-186
View File
@@ -1,186 +0,0 @@
// Layout facts collected from inside the page.
//
// This function is serialised and evaluated in the browser, so it must be
// entirely self-contained: no imports, no closures over module scope. It
// only *measures*; every pass/fail decision is made back in node by
// checks.js, so failures can be reported with real numbers attached.
export function collectLayoutFacts(options) {
const describe = (el) => {
const id = el.id ? "#" + el.id : "";
const classes =
typeof el.className === "string" && el.className.trim()
? "." + el.className.trim().split(/\s+/).slice(0, 3).join(".")
: "";
return el.tagName.toLowerCase() + id + classes;
};
const round = (n) => Math.round(n * 10) / 10;
// Overflow propagates up every ancestor, so a single wide element
// reports as body, #app, the row, and so on. Depth lets the report
// name the deepest — that is, the actual — offender.
const depthOf = (el) => {
let depth = 0;
for (let node = el.parentElement; node; node = node.parentElement) {
depth++;
}
return depth;
};
const isVisible = (el) => {
const style = getComputedStyle(el);
if (style.display === "none") return false;
if (style.visibility === "hidden") return false;
const rect = el.getBoundingClientRect();
return rect.width > 0 && rect.height > 0;
};
const innerWidth = window.innerWidth;
const clientWidth = document.documentElement.clientWidth;
// Under mobile emulation Chrome lets window.innerWidth *grow* to the
// width of overflowing content, exactly as a phone zooms out to fit a
// too-wide page. Measuring against it would therefore hide the
// overflow it is supposed to expose: at a 320px device width a page
// that spills to 350 reports innerWidth 350 and looks clean. Every
// comparison below is against the layout viewport instead.
const viewportWidth = Math.min(innerWidth, clientWidth);
const elements = Array.from(document.querySelectorAll("body *"));
// Elements sticking out past the right (or left) edge of the viewport.
// The document-level scrollWidth check says *that* the page overflows;
// this says *what* is doing it.
const overflowing = [];
// Elements clipping their own text. Deliberate ellipsis truncation
// (Tailwind's `truncate`) is opt-in and excluded: it is a design
// choice, not breakage.
const clipped = [];
// Elements whose content spills out of their own box without being
// clipped, past the right edge of the viewport. A block element is
// only ever as wide as its container, so text overflowing it has no
// element rect of its own to catch — but it is exactly what drags
// documentElement.scrollWidth past the viewport width, so without
// this the page-level overflow failure has nothing to point at.
const contentOverflowing = [];
for (const el of elements) {
if (!isVisible(el)) continue;
const rect = el.getBoundingClientRect();
if (rect.right > viewportWidth + 1 || rect.left < -1) {
overflowing.push({
el: describe(el),
depth: depthOf(el),
left: round(rect.left),
right: round(rect.right),
});
}
const style = getComputedStyle(el);
const clips =
style.overflowX === "hidden" || style.overflowX === "clip";
const ellipsis = style.textOverflow === "ellipsis";
const hasText = el.textContent.trim().length > 0;
const spills =
el.clientWidth > 0 && el.scrollWidth > el.clientWidth + 1;
if (clips && !ellipsis && hasText && spills) {
clipped.push({
el: describe(el),
scrollWidth: el.scrollWidth,
clientWidth: el.clientWidth,
});
}
if (
!clips &&
spills &&
rect.left + el.scrollWidth > viewportWidth + 1
) {
contentOverflowing.push({
el: describe(el),
depth: depthOf(el),
scrollWidth: el.scrollWidth,
clientWidth: el.clientWidth,
reach: round(rect.left + el.scrollWidth),
});
}
}
// Interactive controls. The measured target is the nearest thing that
// is genuinely tappable — for a checkbox that is the <label> wrapping
// it, which is larger than the box itself and is what a finger hits.
const tapTargets = [];
for (const selector of options.interactiveSelectors) {
for (const el of document.querySelectorAll(selector)) {
if (!isVisible(el)) continue;
const target = el.closest("button, a, label, select") || el;
const rect = target.getBoundingClientRect();
tapTargets.push({
selector,
el: describe(target),
width: round(rect.width),
height: round(rect.height),
});
}
}
// Host rows. The question is not "did it get narrower" but "did it
// reflow": the info block and the sparkline must end up stacked
// vertically and full width in the narrow layout, and side by side in
// the wide one. Both the computed flex-direction and the actual
// geometry are recorded so a row that claims to be a column but is
// still laid out side by side cannot slip through.
const rows = [];
for (const row of document.querySelectorAll(".host-row")) {
const inner = row.firstElementChild;
if (!inner) continue;
const sparkline = inner.querySelector(".sparkline-container");
const info = sparkline ? sparkline.previousElementSibling : null;
if (!sparkline || !info) continue;
const innerStyle = getComputedStyle(inner);
const innerRect = inner.getBoundingClientRect();
const infoRect = info.getBoundingClientRect();
const sparkRect = sparkline.getBoundingClientRect();
rows.push({
index: row.dataset.index,
flexDirection: innerStyle.flexDirection,
containerWidth: round(innerRect.width),
info: {
left: round(infoRect.left),
right: round(infoRect.right),
bottom: round(infoRect.bottom),
width: round(infoRect.width),
},
sparkline: {
left: round(sparkRect.left),
top: round(sparkRect.top),
width: round(sparkRect.width),
},
});
}
const localRows = Array.from(
document.querySelectorAll("#local-hosts .host-row"),
);
return {
innerWidth,
// innerWidth includes any classic scrollbar, clientWidth does not.
// Reported separately so the overflow check can hold itself to the
// narrower of the two rather than to whichever one is more
// forgiving.
documentClientWidth: document.documentElement.clientWidth,
documentScrollWidth: document.documentElement.scrollWidth,
overflowing,
contentOverflowing,
clipped,
tapTargets,
rows,
// WAN host rows only: each has a pin button, and the tap-target
// check expects one per row. The local host rows have none.
wanRowCount: document.querySelectorAll("#wan-hosts .host-row").length,
numericLatencies: Array.from(
document.querySelectorAll(".latency-value"),
).filter((el) => /\d/.test(el.textContent)).length,
gatewayDetected: localRows.some((row) =>
row.textContent.includes("Local Gateway"),
),
};
}
-262
View File
@@ -1,262 +0,0 @@
// Responsive-layout harness.
//
// Drives the built frontend in a real, containerised, digest-pinned Chrome
// over CDP and asserts on computed layout at every viewport width derived
// from the app's own CSS. Screenshots are written alongside as artifacts;
// they are not the evidence, the assertions are.
//
// This is not meant to be run by hand. `make frontend-viewport-test` brings
// up the browser and the web server and then runs this; every input it
// needs arrives in the environment.
import { mkdirSync, writeFileSync } from "node:fs";
import { join } from "node:path";
import puppeteer from "puppeteer-core";
import { collectLayoutFacts } from "./facts.js";
import { evaluateChecks, INTERACTIVE_SELECTORS } from "./checks.js";
import { deriveViewports } from "./viewports.js";
// page.waitForFunction and page.evaluate run the functions given them in
// the page, where these are defined.
/* global document, requestAnimationFrame */
function required(name) {
const value = process.env[name];
if (!value) {
throw new Error(
`${name} is not set; run this via script/frontend-viewport-test`,
);
}
return value;
}
const ROOT = required("NETWATCH_ROOT");
const BASE_URL = required("NETWATCH_BASE_URL");
const CDP_URL = required("NETWATCH_CDP_URL");
const ARTIFACT_DIR = required("NETWATCH_ARTIFACT_DIR");
const BROWSER_TIMEOUT_MS = 60000;
const PAGE_TIMEOUT_MS = 30000;
// Canned responses for the app's outbound latency probes. The browser
// container sits on an --internal docker network and physically cannot
// reach the internet, so nothing here is about blocking traffic; it is
// about determinism. Real probes would render 24 rows of whatever the
// network happened to be doing. These delays make the rows show a
// realistic spread of value widths — one, two and three digit latencies,
// plus some unreachable rows — because that spread is what the layout has
// to survive.
const PROBE_DELAYS_MS = [2, 45, 123, 456, 780];
// One in every UNREACHABLE_MODULUS probes is failed outright so that the
// offline row rendering is exercised too.
const UNREACHABLE_MODULUS = 7;
// The gateway candidate that "answers", so gateway detection succeeds and
// the Local Gateway row renders. Matches GATEWAY_CANDIDATES in src/main.js.
const RESPONSIVE_GATEWAY = "http://192.168.1.1";
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
function stableHash(text) {
let hash = 0;
for (let i = 0; i < text.length; i++) {
hash = (hash * 31 + text.charCodeAt(i)) | 0;
}
return Math.abs(hash);
}
async function connectBrowser() {
const deadline = Date.now() + BROWSER_TIMEOUT_MS;
for (;;) {
try {
const response = await fetch(`${CDP_URL}/json/version`);
const info = await response.json();
// The endpoint advertises whatever Host it was reached on;
// pin it back to the address we actually dialled.
const endpoint = new URL(info.webSocketDebuggerUrl);
endpoint.host = new URL(CDP_URL).host;
const browser = await puppeteer.connect({
browserWSEndpoint: endpoint.toString(),
protocolTimeout: BROWSER_TIMEOUT_MS,
});
return { browser, version: info.Browser };
} catch (error) {
if (Date.now() > deadline) {
throw new Error(`browser never came up: ${error}`, {
cause: error,
});
}
await sleep(250);
}
}
}
function installProbeResponder(page, probes) {
const respond = (request, delayMs) =>
sleep(delayMs).then(() =>
request.respond({
status: 200,
contentType: "text/plain",
body: "",
}),
);
page.on("request", (request) => {
const url = request.url();
const settle = async () => {
if (url.startsWith(BASE_URL) || url.startsWith("data:")) {
return request.continue();
}
probes.attempted++;
if (url.startsWith(RESPONSIVE_GATEWAY)) {
probes.fulfilled++;
return respond(request, 5);
}
const hash = stableHash(url);
if (hash % UNREACHABLE_MODULUS === 0) {
probes.failed++;
return request.abort("connectionfailed");
}
probes.fulfilled++;
return respond(
request,
PROBE_DELAYS_MS[hash % PROBE_DELAYS_MS.length],
);
};
// The page may be torn down while a delayed response is pending;
// that is not a harness failure.
settle().catch(() => {});
});
}
async function runViewport(browser, viewport) {
const page = await browser.newPage();
const probes = { attempted: 0, fulfilled: 0, failed: 0 };
try {
page.setDefaultTimeout(PAGE_TIMEOUT_MS);
await page.setRequestInterception(true);
installProbeResponder(page, probes);
await page.setViewport({
width: viewport.width,
height: viewport.height,
deviceScaleFactor: viewport.deviceScaleFactor,
isMobile: viewport.touch,
hasTouch: viewport.touch,
isLandscape: viewport.width > viewport.height,
});
await page.goto(BASE_URL, { waitUntil: "load" });
await page.waitForSelector(".host-row");
// The app discards its first tick as a cold start, so rows only
// carry real values from the second one. Changing the interval
// restarts the loop at 1s, which reaches a populated UI without
// waiting out two default 3s intervals — and exercises the
// interval dropdown while we are at it.
await page.select("#interval-select", "1000");
await page.waitForFunction(
() =>
Array.from(document.querySelectorAll(".latency-value")).filter(
(el) => /\d/.test(el.textContent),
).length >= 5,
);
// Let the resize/redraw handlers settle before measuring.
await page.evaluate(
() =>
new Promise((resolve) =>
requestAnimationFrame(() => requestAnimationFrame(resolve)),
),
);
const facts = await page.evaluate(collectLayoutFacts, {
interactiveSelectors: INTERACTIVE_SELECTORS,
});
const screenshot = join(
ARTIFACT_DIR,
`${viewport.width}x${viewport.height}-${viewport.name}.png`,
);
await page.screenshot({ path: screenshot, fullPage: true });
return {
viewport,
probes,
facts,
screenshot,
checks: evaluateChecks(facts, viewport, probes),
};
} finally {
await page.close().catch(() => {});
}
}
function report(results, conditions, browserVersion) {
const label = (viewport) =>
`${viewport.width}x${viewport.height}`.padEnd(9) +
" " +
viewport.name.padEnd(24);
console.log(`browser: ${browserVersion}`);
console.log(`served from: ${BASE_URL} (built dist/)`);
console.log(
"breakpoints: " +
conditions
.map((c) => `${c.type}-width ${c.px}px (${c.source})`)
.join(", "),
);
console.log("");
let passed = 0;
let failed = 0;
for (const result of results) {
const bad = result.checks.filter((c) => !c.ok);
passed += result.checks.length - bad.length;
failed += bad.length;
// Passing viewports get one line. Detail is for failures.
console.log(
`${bad.length === 0 ? "PASS" : "FAIL"} ${label(result.viewport)} ` +
`${result.checks.length - bad.length}/${result.checks.length} checks` +
`${result.viewport.expectStacked ? " [narrow layout expected]" : ""}`,
);
for (const check of bad) {
console.log(` ${check.name}: ${check.detail}`);
}
if (bad.length > 0) {
console.log(` why this width: ${result.viewport.why}`);
console.log(` screenshot: ${result.screenshot}`);
}
}
console.log("");
console.log(
`${results.length} viewports, ${passed + failed} checks: ` +
`${passed} passed, ${failed} failed`,
);
console.log(`artifacts: ${ARTIFACT_DIR}`);
return failed;
}
async function main() {
const { conditions, viewports } = deriveViewports(ROOT);
mkdirSync(ARTIFACT_DIR, { recursive: true });
const { browser, version } = await connectBrowser();
const results = [];
try {
for (const viewport of viewports) {
results.push(await runViewport(browser, viewport));
}
} finally {
await browser.disconnect().catch(() => {});
}
writeFileSync(
join(ARTIFACT_DIR, "results.json"),
JSON.stringify({ browser: version, conditions, results }, null, 2) +
"\n",
);
const failed = report(results, conditions, version);
process.exitCode = failed === 0 ? 0 : 1;
}
await main();
-224
View File
@@ -1,224 +0,0 @@
// Viewport derivation for the responsive-layout harness.
//
// The widths tested are read out of the CSS the application actually
// ships, not taken from a list of popular phone models. A generic 375px
// "phone" test sails straight past an off-by-one error at a media query
// boundary, which is the classic way a responsive layout breaks, so
// every breakpoint found in the sources is probed three times: one pixel
// below it, exactly on it, and one pixel above it.
//
// Nothing here hardcodes 768. If someone adds a second media block or
// starts using Tailwind responsive prefixes, that breakpoint starts
// being covered without this file being edited.
import { readFileSync } from "node:fs";
import { join } from "node:path";
// Tailwind CSS v4 default breakpoints, in rem. The app currently uses
// none of these prefixes, so the whole table is inert until someone
// writes an `md:`-prefixed utility class.
const TAILWIND_BREAKPOINT_REM = {
sm: 40,
md: 48,
lg: 64,
xl: 80,
"2xl": 96,
};
// The app does not override the root font size, so rem and em in media
// queries resolve against the browser default.
const ROOT_FONT_SIZE_PX = 16;
// Extract every min-width / max-width condition from the @media blocks in
// a stylesheet. Returns e.g. [{ type: "max", px: 768, source: "..." }].
export function mediaConditionsFromCss(css, source) {
const conditions = [];
for (const block of css.matchAll(/@media([^{]+)\{/g)) {
const features = block[1].matchAll(
/\(\s*(min|max)-width\s*:\s*([\d.]+)(px|rem|em)\s*\)/g,
);
for (const feature of features) {
const scale = feature[3] === "px" ? 1 : ROOT_FONT_SIZE_PX;
conditions.push({
type: feature[1],
px: Math.round(Number(feature[2]) * scale),
source,
});
}
}
return conditions;
}
// Extract the breakpoints implied by Tailwind responsive prefixes used in
// markup. A prefix only counts when it opens a utility class, so `text-sm`
// does not masquerade as the `sm:` breakpoint.
export function mediaConditionsFromMarkup(sources) {
const conditions = [];
for (const { path, text } of sources) {
for (const [name, rem] of Object.entries(TAILWIND_BREAKPOINT_REM)) {
const used = new RegExp(
`(^|["'\\s])${name}:[a-z0-9[\\](),_./%-]+`,
"m",
).test(text);
if (used) {
conditions.push({
type: "min",
px: rem * ROOT_FONT_SIZE_PX,
source: path,
});
}
}
}
return conditions;
}
// Whether a given width should be rendering the app's narrow (stacked)
// layout.
//
// Two breakpoint styles can be answered from the condition list alone:
//
// - Desktop-first, which is what the app ships today: the wide layout is
// unconditional and every narrow rule lives in a `max-width` block, so
// a width is narrow exactly when one of those blocks matches. Note that
// `max-width: 768px` matches *at* 768 — getting this inclusive boundary
// wrong in either direction is what the three-widths-per-breakpoint
// sweep exists to catch.
// - Mobile-first, which is what Tailwind's `sm:`/`md:` prefixes are: the
// stacked layout is the unconditional base and a `min-width` block is
// what widens it, so a width is narrow exactly when it sits below every
// `min-width` breakpoint.
//
// A mix of the two cannot be resolved from the breakpoints alone — which
// block owns the host-row reflow is a property of the rules inside it, not
// of the condition — so this throws rather than guessing. Guessing is how
// the wrong expectation gets applied at the right widths and the whole
// sweep quietly verifies nothing.
export function expectsStackedLayout(width, conditions) {
const kinds = new Set(conditions.map((c) => c.type));
for (const kind of kinds) {
if (kind !== "max" && kind !== "min") {
throw new Error(
`unsupported media condition type "${kind}" in ` +
"expectsStackedLayout (test/viewport/viewports.js)",
);
}
}
if (kinds.has("max") && kinds.has("min")) {
throw new Error(
"the app now mixes max-width and min-width breakpoints (" +
conditions
.map((c) => `${c.type}-width ${c.px}px in ${c.source}`)
.join(", ") +
"), so which layout a width should be showing can no longer " +
"be inferred from the breakpoint list; teach " +
"expectsStackedLayout in test/viewport/viewports.js which " +
"block owns the host-row reflow",
);
}
if (kinds.has("min")) {
return !conditions.some((c) => width >= c.px);
}
return conditions.some((c) => width <= c.px);
}
// Viewports that are not derived from a breakpoint. Each one is here for
// a stated reason; none of them is a stand-in for "a phone".
const ANCHOR_VIEWPORTS = [
{
name: "floor-portrait",
width: 320,
height: 568,
deviceScaleFactor: 2,
touch: true,
why: "320px is the narrowest viewport still in mainstream use; nothing has to work below it",
},
{
name: "phone-landscape-narrow",
width: 667,
height: 375,
deviceScaleFactor: 2,
touch: true,
why: "phone rotated to landscape, still inside the narrow layout",
},
{
name: "phone-landscape-wide",
width: 844,
height: 390,
deviceScaleFactor: 3,
touch: true,
why: "large phone rotated to landscape: crosses into the wide layout while still being a touch device",
},
{
name: "desktop",
width: 1280,
height: 800,
deviceScaleFactor: 1,
touch: false,
why: "desktop baseline",
},
];
export function deriveViewports(root) {
const conditions = [
...mediaConditionsFromCss(
readFileSync(join(root, "src/styles.css"), "utf8"),
"src/styles.css",
),
...mediaConditionsFromMarkup([
{
path: "src/main.js",
text: readFileSync(join(root, "src/main.js"), "utf8"),
},
{
path: "index.html",
text: readFileSync(join(root, "index.html"), "utf8"),
},
]),
];
if (conditions.length === 0) {
throw new Error(
"no responsive breakpoints found in src/styles.css, src/main.js or " +
"index.html — either the responsive layout was deleted or this " +
"derivation has stopped matching the sources",
);
}
const viewports = new Map();
const add = (viewport) => {
const key = `${viewport.width}x${viewport.height}`;
if (!viewports.has(key)) viewports.set(key, viewport);
};
for (const condition of conditions) {
for (const [offset, label] of [
[-1, "below"],
[0, "at"],
[+1, "above"],
]) {
const width = condition.px + offset;
add({
name: `${condition.type}-width-${condition.px}-${label}`,
width,
// Tall enough that the whole app is laid out in one column
// without the viewport height influencing wrapping.
height: 1024,
deviceScaleFactor: 2,
touch: true,
why: `${offset === 0 ? "exactly on" : `1px ${label}`} the ${condition.type}-width: ${condition.px}px breakpoint declared in ${condition.source}`,
});
}
}
for (const anchor of ANCHOR_VIEWPORTS) add(anchor);
return {
conditions,
viewports: [...viewports.values()]
.map((viewport) => ({
...viewport,
expectStacked: expectsStackedLayout(viewport.width, conditions),
}))
.sort((a, b) => a.width - b.width || a.height - b.height),
};
}
-7
View File
@@ -7,13 +7,6 @@ const commitFull = execSync("git rev-parse HEAD").toString().trim();
export default defineConfig({
plugins: [tailwindcss()],
server: {
// Proxy /api to a locally running netwatch-server so `yarn dev`
// exercises the real report-posting path.
proxy: {
"/api": "http://127.0.0.1:8080",
},
},
define: {
__COMMIT_HASH__: JSON.stringify(commitHash),
__COMMIT_FULL__: JSON.stringify(commitFull),
+201 -884
View File
File diff suppressed because it is too large Load Diff