Compare commits
22
Commits
main
...
ec99b98d2b
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ec99b98d2b | ||
|
|
32704c601e | ||
|
|
7322f936e4 | ||
|
|
6422d9fa0c | ||
|
|
057e0bd9a9 | ||
|
|
87a3bc115e | ||
|
|
d28a59023a | ||
|
|
1ac24669d3 | ||
|
|
6187ac8503 | ||
|
|
f2a9e90625 | ||
|
|
df9e23d503 | ||
|
|
3898daad4e | ||
|
|
08f060045b | ||
|
|
658aadbb81 | ||
|
|
efaf79c4e3 | ||
|
|
63d62b7bc1 | ||
|
|
cb2374f033 | ||
|
|
6d8ae7f592 | ||
|
|
211fdad9c0 | ||
|
|
594e7a504b | ||
|
|
54014c88c8 | ||
|
|
b0cd884019 |
@@ -0,0 +1,32 @@
|
|||||||
|
# Docker does not read .gitignore, and a pattern here matches from the root of
|
||||||
|
# the build context only: a pattern meant for every directory needs `**/`.
|
||||||
|
|
||||||
|
# .git is sent without its config. Without a VERSION build argument the
|
||||||
|
# stage that compiles runs `git describe --tags --always` on .git, which
|
||||||
|
# does not need .git/config; that file can hold a credential, such as a
|
||||||
|
# password in a remote URL or the token the CI checkout step stores there.
|
||||||
|
.git/config
|
||||||
|
|
||||||
|
# Local build and debug output: `make build`, `make run`, `make asupdate`, test
|
||||||
|
# binaries, coverage profiles and source archives.
|
||||||
|
/bin
|
||||||
|
/log.txt
|
||||||
|
/out
|
||||||
|
/pkg/asinfo/asdata.json
|
||||||
|
**/*.tar.zst
|
||||||
|
**/*.test
|
||||||
|
**/*.out
|
||||||
|
**/*.tmp
|
||||||
|
|
||||||
|
# Local databases and secrets. The image carries a source archive of the whole
|
||||||
|
# build context, so these would otherwise ship inside it.
|
||||||
|
**/*.db
|
||||||
|
**/*.db-journal
|
||||||
|
**/*.db-wal
|
||||||
|
**/.env
|
||||||
|
|
||||||
|
# A local Go workspace points at directories outside the build context.
|
||||||
|
/go.work
|
||||||
|
/go.work.sum
|
||||||
|
|
||||||
|
**/.DS_Store
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
root = true
|
||||||
|
|
||||||
|
[*]
|
||||||
|
indent_style = space
|
||||||
|
indent_size = 4
|
||||||
|
end_of_line = lf
|
||||||
|
charset = utf-8
|
||||||
|
trim_trailing_whitespace = true
|
||||||
|
insert_final_newline = true
|
||||||
|
|
||||||
|
[Makefile]
|
||||||
|
indent_style = tab
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
name: check
|
||||||
|
on: [push]
|
||||||
|
jobs:
|
||||||
|
check:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
# actions/checkout v4.2.2, 2026-02-28
|
||||||
|
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||||
|
- run: script/cibuild
|
||||||
+59
-6
@@ -1,5 +1,22 @@
|
|||||||
|
# Lint stage — fast feedback on formatting and lint issues.
|
||||||
|
# The golangci-lint image bundles Go, gcc and make, so it can run go vet on
|
||||||
|
# the CGO sqlite package and golangci-lint without extra installs.
|
||||||
|
# golangci/golangci-lint:v2.7.2 (Go 1.25.5), 2026-09-21
|
||||||
|
FROM golangci/golangci-lint@sha256:5d6d5c70a61f1356adfd9dd6316ce286799fefc9d743421356ff1b00842368ba AS lint
|
||||||
|
|
||||||
|
WORKDIR /src
|
||||||
|
|
||||||
|
COPY go.mod go.sum ./
|
||||||
|
RUN go mod download
|
||||||
|
|
||||||
|
COPY . .
|
||||||
|
|
||||||
|
RUN make fmt-check
|
||||||
|
RUN make lint
|
||||||
|
|
||||||
# Build stage
|
# Build stage
|
||||||
FROM golang:1.24-bookworm AS builder
|
# golang:1.24-bookworm, 2026-09-21
|
||||||
|
FROM golang@sha256:1a6d4452c65dea36aac2e2d606b01b4a029ec90cc1ae53890540ce6173ea77ac AS builder
|
||||||
|
|
||||||
# Install build dependencies (zstd for archive, gcc for CGO/sqlite3)
|
# Install build dependencies (zstd for archive, gcc for CGO/sqlite3)
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
@@ -10,14 +27,36 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
|
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
|
|
||||||
|
# Force BuildKit to run the lint stage before compiling or testing.
|
||||||
|
COPY --from=lint /src/go.sum /dev/null
|
||||||
|
|
||||||
# Copy everything
|
# Copy everything
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# Vendor dependencies (must be after copying source)
|
# Vendor dependencies (must be after copying source)
|
||||||
RUN go mod download && go mod vendor
|
RUN go mod download && go mod vendor
|
||||||
|
|
||||||
# Build the binary with CGO enabled (required for sqlite3)
|
# Run the test suite in the build stage: -race needs cgo and the C compiler
|
||||||
RUN CGO_ENABLED=1 GOOS=linux go build -o /routewatch ./cmd/routewatch
|
# installed above. The suite is offline (the live-feed test is opt-in).
|
||||||
|
RUN make test
|
||||||
|
|
||||||
|
# Build the binary with CGO enabled (required for sqlite3). The version the
|
||||||
|
# page footer shows is the VERSION build argument when one is given, otherwise
|
||||||
|
# `git describe --tags --always` of the .git in the build context (git comes
|
||||||
|
# with this image): the tag on a tagged commit, tag-N-gHASH after one, the
|
||||||
|
# short commit when no tag is reachable. A context that carries .git and still
|
||||||
|
# yields no version fails the build. The footer links to the full commit.
|
||||||
|
ARG VERSION
|
||||||
|
RUN version="${VERSION:-$(git describe --tags --always || echo unknown)}"; \
|
||||||
|
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
||||||
|
[ "$version" = unknown ]; }; then \
|
||||||
|
echo "no version could be derived although the build context carries .git" >&2; \
|
||||||
|
exit 1; \
|
||||||
|
fi; \
|
||||||
|
CGO_ENABLED=1 GOOS=linux go build -o /routewatch -ldflags "\
|
||||||
|
-X git.eeqj.de/sneak/routewatch/internal/version.GitRevision=$(git rev-parse --verify HEAD || echo unknown) \
|
||||||
|
-X git.eeqj.de/sneak/routewatch/internal/version.GitRevisionShort=$version" \
|
||||||
|
./cmd/routewatch
|
||||||
|
|
||||||
# Create source archive with vendored dependencies
|
# Create source archive with vendored dependencies
|
||||||
RUN tar --zstd -cf /routewatch-source.tar.zst \
|
RUN tar --zstd -cf /routewatch-source.tar.zst \
|
||||||
@@ -26,7 +65,8 @@ RUN tar --zstd -cf /routewatch-source.tar.zst \
|
|||||||
.
|
.
|
||||||
|
|
||||||
# Runtime stage
|
# Runtime stage
|
||||||
FROM debian:bookworm-slim
|
# debian:bookworm-slim, 2026-09-21
|
||||||
|
FROM debian@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251
|
||||||
|
|
||||||
# Install runtime dependencies
|
# Install runtime dependencies
|
||||||
# - ca-certificates: for HTTPS connections
|
# - ca-certificates: for HTTPS connections
|
||||||
@@ -53,13 +93,26 @@ RUN chown -R routewatch:routewatch /app
|
|||||||
|
|
||||||
ENV XDG_DATA_HOME=/var/lib
|
ENV XDG_DATA_HOME=/var/lib
|
||||||
|
|
||||||
|
# Cap the Go heap at 1.5 GiB so the runtime collects harder before the
|
||||||
|
# container's memory limit is reached. setpriv in the entrypoint preserves this
|
||||||
|
# the way it does XDG_DATA_HOME above.
|
||||||
|
ENV GOMEMLIMIT=1536MiB
|
||||||
|
|
||||||
|
# Cap glibc's malloc arenas. The SQLite C library allocates and frees millions
|
||||||
|
# of small page-cache chunks from many threads; glibc otherwise creates up to
|
||||||
|
# eight arenas per core (hundreds on a large host) and keeps each arena's freed
|
||||||
|
# chunks resident, so process RSS climbs far above SQLite's live heap and never
|
||||||
|
# comes back down. Two arenas keep that retained memory bounded; database writes
|
||||||
|
# are already serialized, so the lost allocator concurrency costs nothing here.
|
||||||
|
ENV MALLOC_ARENA_MAX=2
|
||||||
|
|
||||||
# Expose HTTP port
|
# Expose HTTP port
|
||||||
EXPOSE 8080
|
EXPOSE 8080
|
||||||
|
|
||||||
COPY ./entrypoint.sh /entrypoint.sh
|
COPY ./entrypoint.sh /entrypoint.sh
|
||||||
|
|
||||||
# Health check using the health endpoint
|
# Health check using the health endpoint, on the port PORT names
|
||||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
||||||
CMD curl -sf http://localhost:8080/.well-known/healthcheck.json || exit 1
|
CMD curl -sf "http://localhost:${PORT:-8080}/.well-known/healthcheck.json" || exit 1
|
||||||
|
|
||||||
ENTRYPOINT ["/bin/bash", "/entrypoint.sh" ]
|
ENTRYPOINT ["/bin/bash", "/entrypoint.sh" ]
|
||||||
|
|||||||
@@ -0,0 +1,21 @@
|
|||||||
|
MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2026 Jeffrey Paul <sneak@sneak.berlin>
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
of this software and associated documentation files (the "Software"), to deal
|
||||||
|
in the Software without restriction, including without limitation the rights
|
||||||
|
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
copies of the Software, and to permit persons to whom the Software is
|
||||||
|
furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be included in all
|
||||||
|
copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
@@ -2,7 +2,7 @@ export DEBUG = routewatch
|
|||||||
|
|
||||||
# Git revision for version embedding
|
# Git revision for version embedding
|
||||||
GIT_REVISION := $(shell git rev-parse HEAD 2>/dev/null || echo "unknown")
|
GIT_REVISION := $(shell git rev-parse HEAD 2>/dev/null || echo "unknown")
|
||||||
GIT_REVISION_SHORT := $(shell git rev-parse --short HEAD 2>/dev/null || echo "unknown")
|
GIT_REVISION_SHORT := $(shell git describe --tags --always 2>/dev/null || echo "unknown")
|
||||||
VERSION_PKG := git.eeqj.de/sneak/routewatch/internal/version
|
VERSION_PKG := git.eeqj.de/sneak/routewatch/internal/version
|
||||||
LDFLAGS := -X $(VERSION_PKG).GitRevision=$(GIT_REVISION) -X $(VERSION_PKG).GitRevisionShort=$(GIT_REVISION_SHORT)
|
LDFLAGS := -X $(VERSION_PKG).GitRevision=$(GIT_REVISION) -X $(VERSION_PKG).GitRevisionShort=$(GIT_REVISION_SHORT)
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,9 @@
|
|||||||
# RouteWatch
|
# RouteWatch
|
||||||
|
|
||||||
RouteWatch is a real-time BGP routing table monitor that streams BGP UPDATE messages from the RIPE RIS Live service, maintains a live routing table in SQLite, and provides HTTP APIs for querying routing information.
|
RouteWatch is an MIT-licensed Go daemon by @sneak that monitors the BGP routing
|
||||||
|
table in real time: it streams BGP UPDATE messages from the RIPE RIS Live
|
||||||
|
service, maintains a live routing table in SQLite, and provides HTTP APIs for
|
||||||
|
querying routing information.
|
||||||
|
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
@@ -139,7 +142,10 @@ routewatch/
|
|||||||
- **Backpressure**: Probabilistic message dropping when queues exceed 50% capacity
|
- **Backpressure**: Probabilistic message dropping when queues exceed 50% capacity
|
||||||
- **Graceful Shutdown**: 60-second timeout, flushes all pending batches
|
- **Graceful Shutdown**: 60-second timeout, flushes all pending batches
|
||||||
- **Reconnection**: Exponential backoff (5s-320s) with reset after 30s of stable connection
|
- **Reconnection**: Exponential backoff (5s-320s) with reset after 30s of stable connection
|
||||||
- **IPv4 Optimization**: IP ranges stored as uint32 for O(1) lookups
|
- **IP Lookup**: the most specific live route is found by looking up the
|
||||||
|
address's prefix at each mask length, longest first, in the prefix index
|
||||||
|
(at most 33 lookups for IPv4, 129 for IPv6); prefixes are stored in the
|
||||||
|
text form Go's `net/netip` prints
|
||||||
|
|
||||||
### Database Schema
|
### Database Schema
|
||||||
|
|
||||||
@@ -151,7 +157,7 @@ prefixes_v6(id, prefix, mask_length, first_seen, last_seen)
|
|||||||
|
|
||||||
-- Live routing tables (one per IP version)
|
-- Live routing tables (one per IP version)
|
||||||
live_routes_v4(id, prefix, mask_length, origin_asn, peer_ip, as_path,
|
live_routes_v4(id, prefix, mask_length, origin_asn, peer_ip, as_path,
|
||||||
next_hop, last_updated, v4_ip_start, v4_ip_end)
|
next_hop, last_updated)
|
||||||
live_routes_v6(id, prefix, mask_length, origin_asn, peer_ip, as_path,
|
live_routes_v6(id, prefix, mask_length, origin_asn, peer_ip, as_path,
|
||||||
next_hop, last_updated)
|
next_hop, last_updated)
|
||||||
|
|
||||||
@@ -165,13 +171,83 @@ bgp_peers(id, peer_ip, peer_asn, last_message_type, last_seen)
|
|||||||
Configuration is handled via environment variables and OS-specific paths:
|
Configuration is handled via environment variables and OS-specific paths:
|
||||||
|
|
||||||
| Variable | Default | Description |
|
| Variable | Default | Description |
|
||||||
|----------|---------|-------------|
|
|----------|----------|-------------|
|
||||||
| `PORT` | `8080` | HTTP server port |
|
| `PORT` | `8080` | HTTP server port, a whole number from 1 to 65535 |
|
||||||
| `DEBUG` | (empty) | Set to `routewatch` for debug logging |
|
| `DEBUG` | (empty) | Set to `routewatch` for debug logging |
|
||||||
|
| `XDG_DATA_HOME` | `/var/lib` (in the Docker image) | Base of the state directory; must be an absolute path |
|
||||||
|
| `GOMEMLIMIT` | `1536MiB` (in the Docker image) | Go soft memory limit; see Memory |
|
||||||
|
| `MALLOC_ARENA_MAX` | `2` (in the Docker image) | glibc malloc arena cap, a positive whole number; see Memory |
|
||||||
|
|
||||||
|
A variable that is set to an invalid value stops the start with an error and a
|
||||||
|
non-zero exit. An empty variable counts as unset.
|
||||||
|
|
||||||
State directory (database location):
|
State directory (database location):
|
||||||
- macOS: `~/Library/Application Support/routewatch/`
|
- macOS: `~/Library/Application Support/routewatch/`
|
||||||
- Linux: `/var/lib/routewatch/` or `~/.local/share/routewatch/`
|
- Linux: `/var/lib/berlin.sneak.app.routewatch/` when running as root,
|
||||||
|
otherwise `$XDG_DATA_HOME/berlin.sneak.app.routewatch/` (with `XDG_DATA_HOME`
|
||||||
|
unset, `~/.local/share/berlin.sneak.app.routewatch/`). In the Docker image
|
||||||
|
this is `/var/lib/berlin.sneak.app.routewatch/`.
|
||||||
|
|
||||||
|
## Memory
|
||||||
|
|
||||||
|
The daemon holds a live routing table, so its memory grows with the size of the
|
||||||
|
data it tracks. The image sets ceilings that keep it inside a 5 GiB container.
|
||||||
|
|
||||||
|
Budget:
|
||||||
|
- Go heap: a 1.5 GiB soft limit (`GOMEMLIMIT=1536MiB`, set in the image).
|
||||||
|
- SQLite: at most 640 MiB of page cache across the connection pool (64 MiB per
|
||||||
|
connection, 10 connections) and a 1.5 GiB hard heap limit for the C library.
|
||||||
|
- glibc allocator: the SQLite C library runs on glibc `malloc`, which frees
|
||||||
|
page-cache chunks back to per-arena free lists rather than to the kernel, so
|
||||||
|
process RSS tracks the high-water mark of those arenas, not SQLite's live
|
||||||
|
heap. glibc creates up to eight arenas per core, so on a many-core host the
|
||||||
|
retained memory — and thus RSS — grows with the core count. The image sets
|
||||||
|
`MALLOC_ARENA_MAX=2` to bound it; the two-arena cap costs nothing here because
|
||||||
|
database writes are already serialized.
|
||||||
|
- About 0.2 GiB for everything else in the runtime.
|
||||||
|
|
||||||
|
Run the container with a memory limit of 5 GiB and swap disabled:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run --memory=5g --memory-swap=5g ...
|
||||||
|
```
|
||||||
|
|
||||||
|
or the equivalent in your deployment tool. This leaves headroom above the
|
||||||
|
ceilings for spikes and the kernel page cache.
|
||||||
|
|
||||||
|
Override the Go soft limit by setting `GOMEMLIMIT` in the environment (for
|
||||||
|
example `-e GOMEMLIMIT=1GiB`); this replaces the image default. `MALLOC_ARENA_MAX`
|
||||||
|
can be overridden the same way, but raising it lets RSS climb again on a
|
||||||
|
many-core host.
|
||||||
|
|
||||||
|
What happens at each limit:
|
||||||
|
- Go soft limit: as the heap approaches `GOMEMLIMIT`, the runtime runs garbage
|
||||||
|
collection more aggressively rather than growing further.
|
||||||
|
- SQLite: at a 1 GiB soft heap limit it recycles its page cache instead of
|
||||||
|
allocating more; at the 1.5 GiB hard heap limit a statement fails with an
|
||||||
|
out-of-memory error, and the handler logs it and drops that batch. The process
|
||||||
|
keeps running.
|
||||||
|
- Handler queues: each of the four handler queues holds at most 20,000 messages.
|
||||||
|
When a queue fills, the streamer drops messages instead of blocking.
|
||||||
|
|
||||||
|
With `DEBUG=routewatch` the daemon logs a `System stats` line every 60 seconds
|
||||||
|
with the goroutine count and Go memory figures.
|
||||||
|
|
||||||
|
## Running under upaas
|
||||||
|
|
||||||
|
What the [upaas](https://git.eeqj.de/sneak/upaas) app needs:
|
||||||
|
|
||||||
|
- Container port: `8080`.
|
||||||
|
- Volume: one, at container path `/var/lib/berlin.sneak.app.routewatch`.
|
||||||
|
- Environment: nothing is required. Leave `XDG_DATA_HOME`, `GOMEMLIMIT` and
|
||||||
|
`MALLOC_ARENA_MAX` at the image's values. `DEBUG=routewatch` is optional and
|
||||||
|
adds the `System stats` memory line to the log.
|
||||||
|
- Memory Limit: `5g`, the 5 GiB limit from Memory above. upaas sets no swap
|
||||||
|
limit, so on a host with swap Docker allows the same amount of swap again.
|
||||||
|
- Health check: the image's `HEALTHCHECK` requests
|
||||||
|
`/.well-known/healthcheck.json` on the container port. upaas reads the
|
||||||
|
container's health 60 seconds after a deploy and fails the deploy unless it
|
||||||
|
is `healthy`.
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
@@ -204,7 +280,9 @@ them. We provide:
|
|||||||
- `script/projectname` — print the project name (used for the Docker
|
- `script/projectname` — print the project name (used for the Docker
|
||||||
image tag)
|
image tag)
|
||||||
- `script/test` — run the test suite
|
- `script/test` — run the test suite
|
||||||
(`go test -timeout 30s -race -cover ./...`, verbose rerun on failure)
|
(`go test -short -timeout 30s -race -cover ./...`, verbose rerun on
|
||||||
|
failure). The `-short` flag skips the live-network integration test so the
|
||||||
|
default run is deterministic and offline.
|
||||||
- `script/lint` — run `go vet ./...` and `golangci-lint run`
|
- `script/lint` — run `go vet ./...` and `golangci-lint run`
|
||||||
- `script/fmt` — format all code (writes)
|
- `script/fmt` — format all code (writes)
|
||||||
- `script/fmt-check` — check formatting (read-only)
|
- `script/fmt-check` — check formatting (read-only)
|
||||||
@@ -218,6 +296,15 @@ them. We provide:
|
|||||||
- `script/install-precommit` — install the git pre-commit hook that
|
- `script/install-precommit` — install the git pre-commit hook that
|
||||||
runs `script/precommit`
|
runs `script/precommit`
|
||||||
|
|
||||||
|
The live-network integration test `TestRouteWatchLiveFeed` streams the RIPE
|
||||||
|
RIS feed for a few seconds and is skipped in short mode. To run it on demand,
|
||||||
|
invoke `go test` directly without `-short`:
|
||||||
|
`go test -run TestRouteWatchLiveFeed ./internal/routewatch/`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
See LICENSE file.
|
MIT. See [`LICENSE`](LICENSE).
|
||||||
|
|
||||||
|
## Author
|
||||||
|
|
||||||
|
[@sneak](https://sneak.berlin)
|
||||||
|
|||||||
@@ -10,19 +10,90 @@
|
|||||||
|
|
||||||
# Status
|
# Status
|
||||||
|
|
||||||
pre-1.0. No git tags. Runs in production-style Docker deployment, but
|
pre-1.0. No git tags. The Docker build runs the format check, the linter
|
||||||
the policy compliance branch (repo-policies-compliance, make check
|
and the tests, and the Gitea workflow runs that build on every push. The
|
||||||
passing, clean tree) is unmerged to main and the CI workflow is missing.
|
image sets memory ceilings for a 5 GiB container (README "Memory") and the
|
||||||
|
README says how to run it under upaas (README "Running under upaas"). A
|
||||||
|
35-hour run of `3898daa` on the live feed peaked at about 1 GiB, without a
|
||||||
|
container memory limit.
|
||||||
|
|
||||||
# Next Step
|
# Next Step
|
||||||
|
|
||||||
Merge repo-policies-compliance into main (3 commits: policy files and
|
`next` waits for sneak to merge it to `main` through
|
||||||
.gitignore, Makefile targets fmt-check/check/docker/hooks, gofmt pass),
|
https://git.eeqj.de/sneak/routewatch/pulls/6. After that, setting
|
||||||
then add .gitea/workflows/check.yml as a small follow-up commit so CI
|
routewatch up under upaas on fsn1app1 and deploying it are his
|
||||||
runs make check on main.
|
(https://git.eeqj.de/sneak/routewatch/issues/31), and so is the run under a
|
||||||
|
real 5 GiB limit (https://git.eeqj.de/sneak/routewatch/issues/3).
|
||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
|
- 2026-10-03: `/api/v1/stats` serves the prefix distribution from memory,
|
||||||
|
seeded at startup and adjusted on every live-route write, so a request no
|
||||||
|
longer reads every live route (closes #30)
|
||||||
|
- 2026-10-03: looking up an IP address no longer reads every IPv6 route:
|
||||||
|
both families find the most specific live route with at most 33 or 129
|
||||||
|
lookups on the prefix index, and the IPv4 range columns are gone. Prefixes
|
||||||
|
from the feed are stored in one text form, so an IPv6 withdrawal, which the
|
||||||
|
feed sends uncompressed, now removes its route (closes #48)
|
||||||
|
- 2026-10-02: a plain `docker build .` stamps the commit's tag or short
|
||||||
|
commit (`git describe --tags --always`) into the page footer instead of
|
||||||
|
`unknown`: `.dockerignore` sends `.git` without `.git/config`, a `VERSION`
|
||||||
|
build argument takes precedence, and `make build` stamps the same value
|
||||||
|
(closes #46)
|
||||||
|
- 2026-09-29: the entrypoint creates the data directory if it is missing and
|
||||||
|
stops the start if a step fails; README "Running under upaas" no longer
|
||||||
|
asks for the host directory to be created first (closes #42)
|
||||||
|
- 2026-09-29: `.dockerignore` keeps `.git`, local build output, local
|
||||||
|
databases and `.env` out of the Docker build context, and so out of the
|
||||||
|
source archive in the image (closes #39)
|
||||||
|
- 2026-09-29: README first line names the MIT license and the author;
|
||||||
|
the License section now says MIT and links `LICENSE`, and an Author
|
||||||
|
section was added; this file brought up to date (closes #38)
|
||||||
|
- 2026-09-29: MIT `LICENSE` (closes #1)
|
||||||
|
- 2026-09-28: stopping the daemon while the feed is flowing no longer
|
||||||
|
panics with "send on closed channel": the read loop checks for a stop
|
||||||
|
just before handing a message to the handler queues, and a second
|
||||||
|
`Stop` no longer closes the queues again (closes #34)
|
||||||
|
- 2026-09-28: `docker stop` no longer kills the daemon 2 seconds after the
|
||||||
|
stop signal: the entrypoint switches to the `routewatch` user with
|
||||||
|
`setpriv` instead of `runuser`, so the daemon receives the signal itself
|
||||||
|
and gets the whole wait `docker stop` allows, up to its own 60-second
|
||||||
|
limit (closes #33)
|
||||||
|
- 2026-09-28: ready to run under upaas: a set but invalid `PORT`,
|
||||||
|
`XDG_DATA_HOME` or `MALLOC_ARENA_MAX` stops the start, the health
|
||||||
|
check follows `PORT`, README "Running under upaas" section (closes
|
||||||
|
#31)
|
||||||
|
- 2026-09-22: realtime in-memory database statistics: counts seeded at
|
||||||
|
startup and adjusted on every write, oldest/newest route timestamps via
|
||||||
|
index-end lookups; `/api/v1/stats` no longer scans the tables (closes
|
||||||
|
#27)
|
||||||
|
- 2026-09-21: batch writes take the write lock when their transaction
|
||||||
|
begins (`_txlock=immediate`), so they wait out a WAL checkpoint instead
|
||||||
|
of failing with "database is locked" (closes #25)
|
||||||
|
- 2026-09-21: `MALLOC_ARENA_MAX=2` in the image caps glibc malloc arenas,
|
||||||
|
so memory outside the Go runtime no longer grows with the core count
|
||||||
|
(closes #23)
|
||||||
|
- 2026-09-21: `GOMEMLIMIT=1536MiB` in the image; README Memory section
|
||||||
|
with the memory budget and the 5 GiB container limit (closes #13)
|
||||||
|
- 2026-09-21: the four handler queues hold at most 20,000 messages each,
|
||||||
|
down from 100,000 (closes #11)
|
||||||
|
- 2026-09-21: two goroutine leaks fixed: the stats handlers after a
|
||||||
|
timeout and the streamer's tickers on every reconnect (closes #12)
|
||||||
|
- 2026-09-21: parsed RIS messages no longer keep the unused `Community`
|
||||||
|
and `Raw` fields (closes #9)
|
||||||
|
- 2026-09-21: `.editorconfig`, and a Gitea workflow that runs
|
||||||
|
`script/cibuild` on every push (closes #14)
|
||||||
|
- 2026-09-21: the peering handler's AS-path map holds at most 500,000
|
||||||
|
paths and is swapped for an empty one every 30 seconds instead of copied
|
||||||
|
(closes #10)
|
||||||
|
- 2026-09-21: SQLite memory bounded across the whole connection pool: a
|
||||||
|
64 MiB page cache on each connection, 1 GiB soft and 1.5 GiB hard heap
|
||||||
|
limits (closes #8)
|
||||||
|
- 2026-09-21: the Docker build runs the format check, the linter and the
|
||||||
|
tests, linting in a separate stage on a golangci-lint image pinned by
|
||||||
|
digest (closes #5)
|
||||||
|
- 2026-09-21: `make test` skips the live-network feed test (`-short`), so
|
||||||
|
`make check` no longer depends on the network (closes #2)
|
||||||
- 2026-07-07 Adopted scripts-to-rule-them-all: `script/` entrypoints,
|
- 2026-07-07 Adopted scripts-to-rule-them-all: `script/` entrypoints,
|
||||||
Makefile shims, README Entrypoints section
|
Makefile shims, README Entrypoints section
|
||||||
- 2026-02-22: repo policy compliance: required policy files, .gitignore
|
- 2026-02-22: repo policy compliance: required policy files, .gitignore
|
||||||
@@ -41,8 +112,6 @@ runs make check on main.
|
|||||||
|
|
||||||
# Future Steps
|
# Future Steps
|
||||||
|
|
||||||
- Verify main is green after the merge: make check locally and the new
|
- Production memory under 5 GiB: whether to test under a real 5 GiB
|
||||||
CI workflow passing
|
container limit on fsn1app1 is open for sneak
|
||||||
- Review stale remote branches fix-min-time-calculation and
|
(https://git.eeqj.de/sneak/routewatch/issues/3)
|
||||||
optimize-sqlite-settings: land or delete
|
|
||||||
- Clean up the tmp/ directory at the repo root: gitignore or remove
|
|
||||||
|
|||||||
+14
-1
@@ -1,7 +1,20 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# glibc silently ignores a malformed MALLOC_ARENA_MAX, so refuse it here.
|
||||||
|
if [[ -n "${MALLOC_ARENA_MAX:-}" && ! "$MALLOC_ARENA_MAX" =~ ^[1-9][0-9]*$ ]]; then
|
||||||
|
echo "MALLOC_ARENA_MAX must be a positive whole number, got '$MALLOC_ARENA_MAX'" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Give the data directory to the routewatch user before the daemon starts,
|
||||||
|
# whether it is missing, an empty root-owned mount, or holds another uid's files.
|
||||||
|
mkdir -p /var/lib/berlin.sneak.app.routewatch
|
||||||
cd /var/lib/berlin.sneak.app.routewatch
|
cd /var/lib/berlin.sneak.app.routewatch
|
||||||
chown -R routewatch:routewatch .
|
chown -R routewatch:routewatch .
|
||||||
chmod 700 .
|
chmod 700 .
|
||||||
|
|
||||||
exec runuser -u routewatch -- /app/routewatch
|
# setpriv replaces itself with the daemon, so the daemon receives the stop
|
||||||
|
# signal directly. runuser would stay in between and kill the daemon 2 seconds
|
||||||
|
# after passing the signal on.
|
||||||
|
exec setpriv --reuid=routewatch --regid=routewatch --init-groups -- /app/routewatch
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
|
"strconv"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -18,6 +19,12 @@ const (
|
|||||||
|
|
||||||
// defaultRouteExpirationMinutes is the default route expiration timeout in minutes
|
// defaultRouteExpirationMinutes is the default route expiration timeout in minutes
|
||||||
defaultRouteExpirationMinutes = 5
|
defaultRouteExpirationMinutes = 5
|
||||||
|
|
||||||
|
// defaultPort is the HTTP port used when PORT is not set
|
||||||
|
defaultPort = 8080
|
||||||
|
|
||||||
|
// maxPort is the highest TCP port number
|
||||||
|
maxPort = 65535
|
||||||
)
|
)
|
||||||
|
|
||||||
// Config holds configuration for the entire application
|
// Config holds configuration for the entire application
|
||||||
@@ -25,6 +32,9 @@ type Config struct {
|
|||||||
// StateDir is the directory for all application state (database, snapshots)
|
// StateDir is the directory for all application state (database, snapshots)
|
||||||
StateDir string
|
StateDir string
|
||||||
|
|
||||||
|
// Port is the TCP port the HTTP server listens on
|
||||||
|
Port int
|
||||||
|
|
||||||
// MaxRuntime is the maximum runtime (0 = run forever)
|
// MaxRuntime is the maximum runtime (0 = run forever)
|
||||||
MaxRuntime time.Duration
|
MaxRuntime time.Duration
|
||||||
|
|
||||||
@@ -43,8 +53,14 @@ func New() (*Config, error) {
|
|||||||
return nil, fmt.Errorf("failed to determine state directory: %w", err)
|
return nil, fmt.Errorf("failed to determine state directory: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
port, err := getPort()
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
return &Config{
|
return &Config{
|
||||||
StateDir: stateDir,
|
StateDir: stateDir,
|
||||||
|
Port: port,
|
||||||
MaxRuntime: 0, // Run forever by default
|
MaxRuntime: 0, // Run forever by default
|
||||||
EnableBatchedDatabaseWrites: true, // Enable batching by default
|
EnableBatchedDatabaseWrites: true, // Enable batching by default
|
||||||
RouteExpirationTimeout: defaultRouteExpirationMinutes * time.Minute, // For active route monitoring
|
RouteExpirationTimeout: defaultRouteExpirationMinutes * time.Minute, // For active route monitoring
|
||||||
@@ -69,13 +85,20 @@ func getStateDirectory() (string, error) {
|
|||||||
return filepath.Join(home, "Library", "Application Support", AppIdentifier), nil
|
return filepath.Join(home, "Library", "Application Support", AppIdentifier), nil
|
||||||
|
|
||||||
case "linux", "freebsd", "openbsd", "netbsd":
|
case "linux", "freebsd", "openbsd", "netbsd":
|
||||||
|
// The XDG spec requires an absolute path; a relative one would put
|
||||||
|
// the database somewhere unexpected.
|
||||||
|
xdgData := os.Getenv("XDG_DATA_HOME")
|
||||||
|
if xdgData != "" && !filepath.IsAbs(xdgData) {
|
||||||
|
return "", fmt.Errorf("XDG_DATA_HOME must be an absolute path, got %q", xdgData)
|
||||||
|
}
|
||||||
|
|
||||||
// Unix-like: /var/lib/berlin.sneak.app.routewatch if root, else XDG_DATA_HOME
|
// Unix-like: /var/lib/berlin.sneak.app.routewatch if root, else XDG_DATA_HOME
|
||||||
if os.Geteuid() == 0 {
|
if os.Geteuid() == 0 {
|
||||||
return filepath.Join("/var/lib", AppIdentifier), nil
|
return filepath.Join("/var/lib", AppIdentifier), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Check XDG_DATA_HOME first
|
// Check XDG_DATA_HOME first
|
||||||
if xdgData := os.Getenv("XDG_DATA_HOME"); xdgData != "" {
|
if xdgData != "" {
|
||||||
return filepath.Join(xdgData, AppIdentifier), nil
|
return filepath.Join(xdgData, AppIdentifier), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -92,6 +115,22 @@ func getStateDirectory() (string, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// getPort returns the HTTP port from PORT, or defaultPort when PORT is not set
|
||||||
|
func getPort() (int, error) {
|
||||||
|
value := os.Getenv("PORT")
|
||||||
|
if value == "" {
|
||||||
|
return defaultPort, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ParseUint, unlike Atoi, refuses a sign: the health check URL cannot use "+9090"
|
||||||
|
port, err := strconv.ParseUint(value, 10, 0)
|
||||||
|
if err != nil || port < 1 || port > maxPort {
|
||||||
|
return 0, fmt.Errorf("PORT must be a whole number from 1 to %d, got %q", maxPort, value)
|
||||||
|
}
|
||||||
|
|
||||||
|
return int(port), nil
|
||||||
|
}
|
||||||
|
|
||||||
// EnsureDirectories creates all necessary directories if they don't exist
|
// EnsureDirectories creates all necessary directories if they don't exist
|
||||||
func (c *Config) EnsureDirectories() error {
|
func (c *Config) EnsureDirectories() error {
|
||||||
// Ensure state directory exists
|
// Ensure state directory exists
|
||||||
|
|||||||
@@ -0,0 +1,62 @@
|
|||||||
|
package config
|
||||||
|
|
||||||
|
import (
|
||||||
|
"runtime"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestNewReadsPort(t *testing.T) {
|
||||||
|
tests := map[string]int{
|
||||||
|
"": defaultPort,
|
||||||
|
"1": 1,
|
||||||
|
"9090": 9090,
|
||||||
|
"65535": 65535,
|
||||||
|
}
|
||||||
|
|
||||||
|
for value, want := range tests {
|
||||||
|
t.Run(value, func(t *testing.T) {
|
||||||
|
t.Setenv("PORT", value)
|
||||||
|
t.Setenv("XDG_DATA_HOME", "")
|
||||||
|
|
||||||
|
cfg, err := New()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("New() with PORT=%q: %v", value, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if cfg.Port != want {
|
||||||
|
t.Errorf("New() with PORT=%q: Port = %d, want %d", value, cfg.Port, want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNewRefusesInvalidPort(t *testing.T) {
|
||||||
|
for _, value := range []string{"0", "65536", "-1", "+9090", "http", "80.5"} {
|
||||||
|
t.Run(value, func(t *testing.T) {
|
||||||
|
t.Setenv("PORT", value)
|
||||||
|
t.Setenv("XDG_DATA_HOME", "")
|
||||||
|
|
||||||
|
if _, err := New(); err == nil {
|
||||||
|
t.Errorf("New() with PORT=%q returned no error", value)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNewRefusesRelativeXDGDataHome(t *testing.T) {
|
||||||
|
if runtime.GOOS == "darwin" {
|
||||||
|
t.Skip("macOS does not read XDG_DATA_HOME")
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Setenv("PORT", "")
|
||||||
|
|
||||||
|
t.Setenv("XDG_DATA_HOME", "relative/path")
|
||||||
|
if _, err := New(); err == nil {
|
||||||
|
t.Error("New() with a relative XDG_DATA_HOME returned no error")
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Setenv("XDG_DATA_HOME", "/var/lib")
|
||||||
|
if _, err := New(); err != nil {
|
||||||
|
t.Errorf("New() with XDG_DATA_HOME=/var/lib: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,203 @@
|
|||||||
|
package database
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"sync"
|
||||||
|
)
|
||||||
|
|
||||||
|
// liveCounts holds the running row counts and the prefix distribution that the
|
||||||
|
// stats endpoints report. They are seeded once at startup from the tables and
|
||||||
|
// then adjusted on every write, so a stats read serves them from memory instead
|
||||||
|
// of running a query over the tables. The COUNT(*) scans (issue 27) and then
|
||||||
|
// the prefix distribution query (issue 30) each grew with the database until
|
||||||
|
// they took the whole request timeout and made /api/v1/stats return 500.
|
||||||
|
//
|
||||||
|
// A single mutex guards all fields so the stats reader takes a consistent
|
||||||
|
// snapshot at one instant and writers, which already run under the database
|
||||||
|
// write lock, adjust the counts after their transaction commits.
|
||||||
|
type liveCounts struct {
|
||||||
|
mu sync.RWMutex
|
||||||
|
asns int
|
||||||
|
prefixesV4 int
|
||||||
|
prefixesV6 int
|
||||||
|
peerings int
|
||||||
|
peers int
|
||||||
|
routesV4 int
|
||||||
|
routesV6 int
|
||||||
|
// The prefix distribution: for each mask length, the number of distinct
|
||||||
|
// prefixes that have at least one live route.
|
||||||
|
distributionV4 [ipv4Bits + 1]int
|
||||||
|
distributionV6 [ipv6Bits + 1]int
|
||||||
|
}
|
||||||
|
|
||||||
|
// seed sets every count to the value read from the tables at startup. It runs
|
||||||
|
// before any writer, so it needs no coordination with the adjust methods.
|
||||||
|
func (c *liveCounts) seed(asns, prefixesV4, prefixesV6, peerings, peers, routesV4, routesV6 int,
|
||||||
|
distributionV4, distributionV6 []PrefixDistribution) {
|
||||||
|
c.mu.Lock()
|
||||||
|
defer c.mu.Unlock()
|
||||||
|
|
||||||
|
c.asns = asns
|
||||||
|
c.prefixesV4 = prefixesV4
|
||||||
|
c.prefixesV6 = prefixesV6
|
||||||
|
c.peerings = peerings
|
||||||
|
c.peers = peers
|
||||||
|
c.routesV4 = routesV4
|
||||||
|
c.routesV6 = routesV6
|
||||||
|
for _, entry := range distributionV4 {
|
||||||
|
addAtMaskLength(c.distributionV4[:], entry.MaskLength, entry.Count)
|
||||||
|
}
|
||||||
|
for _, entry := range distributionV6 {
|
||||||
|
addAtMaskLength(c.distributionV6[:], entry.MaskLength, entry.Count)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// addASNs adds n to the ASN count.
|
||||||
|
func (c *liveCounts) addASNs(n int) {
|
||||||
|
c.mu.Lock()
|
||||||
|
c.asns += n
|
||||||
|
c.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// addPrefixes adds to the IPv4 and IPv6 prefix counts.
|
||||||
|
func (c *liveCounts) addPrefixes(v4, v6 int) {
|
||||||
|
c.mu.Lock()
|
||||||
|
c.prefixesV4 += v4
|
||||||
|
c.prefixesV6 += v6
|
||||||
|
c.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// addPeerings adds n to the peering count.
|
||||||
|
func (c *liveCounts) addPeerings(n int) {
|
||||||
|
c.mu.Lock()
|
||||||
|
c.peerings += n
|
||||||
|
c.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// addPeers adds n to the BGP peer count.
|
||||||
|
func (c *liveCounts) addPeers(n int) {
|
||||||
|
c.mu.Lock()
|
||||||
|
c.peers += n
|
||||||
|
c.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// addRoutes adds to the IPv4 and IPv6 live-route counts. Deletions pass
|
||||||
|
// negative values.
|
||||||
|
func (c *liveCounts) addRoutes(v4, v6 int) {
|
||||||
|
c.mu.Lock()
|
||||||
|
c.routesV4 += v4
|
||||||
|
c.routesV6 += v6
|
||||||
|
c.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// addToDistribution adds n to the IPv4 and IPv6 prefix distributions once for
|
||||||
|
// each listed mask length. A write lists the mask lengths of the prefixes it
|
||||||
|
// gave their first live route with n = 1, and of the prefixes it left with no
|
||||||
|
// live route with n = -1.
|
||||||
|
func (c *liveCounts) addToDistribution(maskLengthsV4, maskLengthsV6 []int, n int) {
|
||||||
|
c.mu.Lock()
|
||||||
|
defer c.mu.Unlock()
|
||||||
|
|
||||||
|
for _, maskLength := range maskLengthsV4 {
|
||||||
|
addAtMaskLength(c.distributionV4[:], maskLength, n)
|
||||||
|
}
|
||||||
|
for _, maskLength := range maskLengthsV6 {
|
||||||
|
addAtMaskLength(c.distributionV6[:], maskLength, n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// addAtMaskLength adds n to counts[maskLength]. A mask length the array has no
|
||||||
|
// entry for is ignored, so a malformed route cannot crash the daemon.
|
||||||
|
func addAtMaskLength(counts []int, maskLength, n int) {
|
||||||
|
if maskLength >= 0 && maskLength < len(counts) {
|
||||||
|
counts[maskLength] += n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// fill copies the counts into a Stats, including the derived totals, under a
|
||||||
|
// single read lock so the reader sees one consistent snapshot.
|
||||||
|
func (c *liveCounts) fill(s *Stats) {
|
||||||
|
c.mu.RLock()
|
||||||
|
defer c.mu.RUnlock()
|
||||||
|
|
||||||
|
s.ASNs = c.asns
|
||||||
|
s.IPv4Prefixes = c.prefixesV4
|
||||||
|
s.IPv6Prefixes = c.prefixesV6
|
||||||
|
s.Prefixes = c.prefixesV4 + c.prefixesV6
|
||||||
|
s.Peerings = c.peerings
|
||||||
|
s.Peers = c.peers
|
||||||
|
s.IPv4Routes = c.routesV4
|
||||||
|
s.IPv6Routes = c.routesV6
|
||||||
|
s.LiveRoutes = c.routesV4 + c.routesV6
|
||||||
|
s.IPv4PrefixDistribution = distributionList(c.distributionV4[:])
|
||||||
|
s.IPv6PrefixDistribution = distributionList(c.distributionV6[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
// distributionList lists the mask lengths that have at least one prefix, in
|
||||||
|
// ascending order, the way the distribution query returns them.
|
||||||
|
func distributionList(counts []int) []PrefixDistribution {
|
||||||
|
var list []PrefixDistribution
|
||||||
|
for maskLength, count := range counts {
|
||||||
|
if count > 0 {
|
||||||
|
list = append(list, PrefixDistribution{MaskLength: maskLength, Count: count})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return list
|
||||||
|
}
|
||||||
|
|
||||||
|
// countRows returns the number of rows in the named table. It is used only at
|
||||||
|
// startup to seed the in-memory counters, so a full COUNT(*) is acceptable.
|
||||||
|
func (d *Database) countRows(ctx context.Context, table string) (int, error) {
|
||||||
|
var n int
|
||||||
|
// table is one of a fixed set of literals below, never external input.
|
||||||
|
if err := d.db.QueryRowContext(ctx, "SELECT COUNT(*) FROM "+table).Scan(&n); err != nil {
|
||||||
|
return 0, fmt.Errorf("failed to count %s: %w", table, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return n, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// seedCounts reads the current row counts and prefix distribution from the
|
||||||
|
// tables into the in-memory counters. It runs once at startup, before the
|
||||||
|
// streamer begins writing.
|
||||||
|
func (d *Database) seedCounts(ctx context.Context) error {
|
||||||
|
asns, err := d.countRows(ctx, "asns")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
prefixesV4, err := d.countRows(ctx, "prefixes_v4")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
prefixesV6, err := d.countRows(ctx, "prefixes_v6")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
peerings, err := d.countRows(ctx, "peerings")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
peers, err := d.countRows(ctx, "bgp_peers")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
routesV4, err := d.countRows(ctx, "live_routes_v4")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
routesV6, err := d.countRows(ctx, "live_routes_v6")
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
distributionV4, distributionV6, err := d.GetPrefixDistributionContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
d.counts.seed(asns, prefixesV4, prefixesV6, peerings, peers, routesV4, routesV6,
|
||||||
|
distributionV4, distributionV6)
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,509 @@
|
|||||||
|
package database
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"slices"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/config"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
|
"github.com/google/uuid"
|
||||||
|
)
|
||||||
|
|
||||||
|
// mkV4Route builds an IPv4 live route with its mask length taken from the
|
||||||
|
// prefix.
|
||||||
|
func mkV4Route(t *testing.T, prefix string, asn int, ts time.Time) *LiveRoute {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
maskLength, err := prefixMaskLength(prefix)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("prefixMaskLength(%s): %v", prefix, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return &LiveRoute{
|
||||||
|
ID: uuid.New(),
|
||||||
|
Prefix: prefix,
|
||||||
|
MaskLength: maskLength,
|
||||||
|
IPVersion: ipVersionV4,
|
||||||
|
OriginASN: asn,
|
||||||
|
PeerIP: "192.0.2.1",
|
||||||
|
ASPath: []int{asn},
|
||||||
|
NextHop: "192.0.2.254",
|
||||||
|
LastUpdated: ts,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// mkV6Route builds an IPv6 live route.
|
||||||
|
func mkV6Route(prefix string, asn int, ts time.Time) *LiveRoute {
|
||||||
|
return &LiveRoute{
|
||||||
|
ID: uuid.New(),
|
||||||
|
Prefix: prefix,
|
||||||
|
MaskLength: 32,
|
||||||
|
IPVersion: ipVersionV6,
|
||||||
|
OriginASN: asn,
|
||||||
|
PeerIP: "2001:db8::1",
|
||||||
|
ASPath: []int{asn},
|
||||||
|
NextHop: "2001:db8::ffff",
|
||||||
|
LastUpdated: ts,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLiveCountsTrackWritesInRealtime checks that the stats counts start at
|
||||||
|
// zero, reflect each write the moment it commits (no recompute, no timer), do
|
||||||
|
// not move when a route is merely re-announced, and drop when a route is
|
||||||
|
// deleted. These counts are what /api/v1/stats reports; before this change the
|
||||||
|
// endpoint recomputed them with a COUNT(*) over each table on every request.
|
||||||
|
func TestLiveCountsTrackWritesInRealtime(t *testing.T) {
|
||||||
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
|
|
||||||
|
db, err := New(cfg, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
empty, err := db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext on empty database: %v", err)
|
||||||
|
}
|
||||||
|
if empty.ASNs != 0 || empty.Prefixes != 0 || empty.Peerings != 0 ||
|
||||||
|
empty.Peers != 0 || empty.LiveRoutes != 0 {
|
||||||
|
t.Fatalf("empty database counts nonzero: %+v", empty)
|
||||||
|
}
|
||||||
|
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
|
||||||
|
if err := db.GetOrCreateASNBatch(map[int]time.Time{64500: ts, 64501: ts}); err != nil {
|
||||||
|
t.Fatalf("GetOrCreateASNBatch: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.UpdatePrefixesBatch(map[string]time.Time{
|
||||||
|
"198.51.100.0/24": ts,
|
||||||
|
"2001:db8::/32": ts,
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("UpdatePrefixesBatch: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.UpdatePeerBatch(map[string]PeerUpdate{
|
||||||
|
"192.0.2.1": {PeerIP: "192.0.2.1", PeerASN: 64500, MessageType: "UPDATE", Timestamp: ts},
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("UpdatePeerBatch: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.RecordPeering(64500, 64501, ts); err != nil {
|
||||||
|
t.Fatalf("RecordPeering: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
routes := []*LiveRoute{
|
||||||
|
mkV4Route(t, "198.51.100.0/24", 64500, ts),
|
||||||
|
mkV4Route(t, "203.0.113.0/24", 64501, ts.Add(time.Minute)),
|
||||||
|
mkV6Route("2001:db8::/32", 64502, ts.Add(2*time.Minute)),
|
||||||
|
}
|
||||||
|
if err := db.UpsertLiveRouteBatch(routes); err != nil {
|
||||||
|
t.Fatalf("UpsertLiveRouteBatch: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
stats, err := db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext: %v", err)
|
||||||
|
}
|
||||||
|
assertCounts(t, "after inserts", stats, wantCounts{
|
||||||
|
asns: 2, prefixes: 2, peerings: 1, peers: 1,
|
||||||
|
ipv4Routes: 2, ipv6Routes: 1, liveRoutes: 3,
|
||||||
|
})
|
||||||
|
|
||||||
|
// Re-announcing the same routes is an update, not an insert: counts hold.
|
||||||
|
if err := db.UpsertLiveRouteBatch(routes); err != nil {
|
||||||
|
t.Fatalf("UpsertLiveRouteBatch (re-announce): %v", err)
|
||||||
|
}
|
||||||
|
stats, err = db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext: %v", err)
|
||||||
|
}
|
||||||
|
assertCounts(t, "after re-announce", stats, wantCounts{
|
||||||
|
asns: 2, prefixes: 2, peerings: 1, peers: 1,
|
||||||
|
ipv4Routes: 2, ipv6Routes: 1, liveRoutes: 3,
|
||||||
|
})
|
||||||
|
|
||||||
|
// A withdrawal removes one route.
|
||||||
|
if err := db.DeleteLiveRouteBatch([]LiveRouteDeletion{
|
||||||
|
{Prefix: "203.0.113.0/24", OriginASN: 64501, PeerIP: "192.0.2.1", IPVersion: ipVersionV4},
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("DeleteLiveRouteBatch: %v", err)
|
||||||
|
}
|
||||||
|
stats, err = db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext: %v", err)
|
||||||
|
}
|
||||||
|
assertCounts(t, "after delete", stats, wantCounts{
|
||||||
|
asns: 2, prefixes: 2, peerings: 1, peers: 1,
|
||||||
|
ipv4Routes: 1, ipv6Routes: 1, liveRoutes: 2,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLiveCountsSeededFromDatabaseAtStartup writes rows, reopens the same
|
||||||
|
// database file, and checks the counts and the prefix distribution come back
|
||||||
|
// from the seed scan rather than starting at zero.
|
||||||
|
func TestLiveCountsSeededFromDatabaseAtStartup(t *testing.T) {
|
||||||
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
|
|
||||||
|
db, err := New(cfg, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
if err := db.GetOrCreateASNBatch(map[int]time.Time{64500: ts, 64501: ts, 64502: ts}); err != nil {
|
||||||
|
t.Fatalf("GetOrCreateASNBatch: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.UpsertLiveRouteBatch([]*LiveRoute{
|
||||||
|
mkV4Route(t, "198.51.100.0/24", 64500, ts),
|
||||||
|
mkV6Route("2001:db8::/32", 64502, ts),
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("UpsertLiveRouteBatch: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.Close(); err != nil {
|
||||||
|
t.Fatalf("Close: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
reopened, err := New(cfg, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to reopen database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = reopened.Close() }()
|
||||||
|
|
||||||
|
stats, err := reopened.GetStatsContext(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext after reopen: %v", err)
|
||||||
|
}
|
||||||
|
if stats.ASNs != 3 {
|
||||||
|
t.Errorf("seeded ASNs = %d, want 3", stats.ASNs)
|
||||||
|
}
|
||||||
|
if stats.IPv4Routes != 1 || stats.IPv6Routes != 1 || stats.LiveRoutes != 2 {
|
||||||
|
t.Errorf("seeded routes = (v4 %d, v6 %d, total %d), want (1, 1, 2)",
|
||||||
|
stats.IPv4Routes, stats.IPv6Routes, stats.LiveRoutes)
|
||||||
|
}
|
||||||
|
assertDistribution(t, "seeded IPv4 distribution", stats.IPv4PrefixDistribution,
|
||||||
|
[]PrefixDistribution{{MaskLength: 24, Count: 1}})
|
||||||
|
assertDistribution(t, "seeded IPv6 distribution", stats.IPv6PrefixDistribution,
|
||||||
|
[]PrefixDistribution{{MaskLength: 32, Count: 1}})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPrefixDistributionTracksWrites checks that the prefix distribution the
|
||||||
|
// stats read reports stays exact across each kind of live-route write, and that
|
||||||
|
// after every step it equals what the distribution query reads from the route
|
||||||
|
// tables. The steps run once through the batch methods the prefix handler uses
|
||||||
|
// and once through the single-route methods.
|
||||||
|
func TestPrefixDistributionTracksWrites(t *testing.T) {
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
shared := mkV4Route(t, "198.51.100.0/24", 64500, ts)
|
||||||
|
sharedSecondPeer := mkV4Route(t, "198.51.100.0/24", 64500, ts)
|
||||||
|
sharedSecondPeer.PeerIP = "192.0.2.2"
|
||||||
|
other := mkV4Route(t, "203.0.113.0/24", 64501, ts)
|
||||||
|
wide := mkV4Route(t, "172.16.0.0/16", 64502, ts)
|
||||||
|
v6 := mkV6Route("2001:db8::/32", 64503, ts)
|
||||||
|
|
||||||
|
all := []PrefixDistribution{{MaskLength: 16, Count: 1}, {MaskLength: 24, Count: 2}}
|
||||||
|
v6Only := []PrefixDistribution{{MaskLength: 32, Count: 1}}
|
||||||
|
|
||||||
|
steps := []struct {
|
||||||
|
name string
|
||||||
|
announce []*LiveRoute
|
||||||
|
withdraw []*LiveRoute
|
||||||
|
wantV4 []PrefixDistribution
|
||||||
|
wantV6 []PrefixDistribution
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "new routes",
|
||||||
|
announce: []*LiveRoute{shared, other, wide, v6},
|
||||||
|
wantV4: all, wantV6: v6Only,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "re-announcement",
|
||||||
|
announce: []*LiveRoute{shared, other, wide, v6},
|
||||||
|
wantV4: all, wantV6: v6Only,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "second peer announces a prefix that has a live route",
|
||||||
|
announce: []*LiveRoute{sharedSecondPeer},
|
||||||
|
wantV4: all, wantV6: v6Only,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "withdrawal of a route that is not the last for its prefix",
|
||||||
|
withdraw: []*LiveRoute{shared},
|
||||||
|
wantV4: all, wantV6: v6Only,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "withdrawal of the last route for a prefix",
|
||||||
|
withdraw: []*LiveRoute{sharedSecondPeer},
|
||||||
|
wantV4: []PrefixDistribution{{MaskLength: 16, Count: 1}, {MaskLength: 24, Count: 1}},
|
||||||
|
wantV6: v6Only,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "withdrawal of every remaining route",
|
||||||
|
withdraw: []*LiveRoute{other, wide, v6},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "two peers announce a new prefix together",
|
||||||
|
announce: []*LiveRoute{shared, sharedSecondPeer},
|
||||||
|
wantV4: []PrefixDistribution{{MaskLength: 24, Count: 1}},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "both routes for a prefix withdrawn together",
|
||||||
|
withdraw: []*LiveRoute{shared, sharedSecondPeer},
|
||||||
|
},
|
||||||
|
// The feed often withdraws a route that is not live. That must not take
|
||||||
|
// the prefix out of the distribution. A count wrongly taken below zero is
|
||||||
|
// left out of the answer, so the next step shows it: its announcement
|
||||||
|
// at the same mask length would then not be counted.
|
||||||
|
{
|
||||||
|
name: "withdrawal of routes that are not live",
|
||||||
|
withdraw: []*LiveRoute{other, v6},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "announcement after a withdrawal of routes that are not live",
|
||||||
|
announce: []*LiveRoute{other, v6},
|
||||||
|
wantV4: []PrefixDistribution{{MaskLength: 24, Count: 1}},
|
||||||
|
wantV6: v6Only,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, batch := range []bool{true, false} {
|
||||||
|
name := "single-route writes"
|
||||||
|
if batch {
|
||||||
|
name = "batch writes"
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
db, err := New(&config.Config{StateDir: t.TempDir()}, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
|
ctx := context.Background()
|
||||||
|
for _, step := range steps {
|
||||||
|
if err := announceRoutes(db, batch, step.announce); err != nil {
|
||||||
|
t.Fatalf("%s: announce: %v", step.name, err)
|
||||||
|
}
|
||||||
|
if err := withdrawRoutes(db, batch, step.withdraw); err != nil {
|
||||||
|
t.Fatalf("%s: withdraw: %v", step.name, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
stats, err := db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: GetStatsContext: %v", step.name, err)
|
||||||
|
}
|
||||||
|
queryV4, queryV6, err := db.GetPrefixDistributionContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s: GetPrefixDistributionContext: %v", step.name, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
assertDistribution(t, step.name+": IPv4 distribution", stats.IPv4PrefixDistribution, step.wantV4)
|
||||||
|
assertDistribution(t, step.name+": IPv6 distribution", stats.IPv6PrefixDistribution, step.wantV6)
|
||||||
|
assertDistribution(t, step.name+": IPv4 distribution query", queryV4, step.wantV4)
|
||||||
|
assertDistribution(t, step.name+": IPv6 distribution query", queryV6, step.wantV6)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// announceRoutes writes routes in one UpsertLiveRouteBatch, or with one
|
||||||
|
// UpsertLiveRoute each.
|
||||||
|
func announceRoutes(db *Database, batch bool, routes []*LiveRoute) error {
|
||||||
|
if batch {
|
||||||
|
return db.UpsertLiveRouteBatch(routes)
|
||||||
|
}
|
||||||
|
for _, route := range routes {
|
||||||
|
if err := db.UpsertLiveRoute(route); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// withdrawRoutes removes routes in one DeleteLiveRouteBatch, or with one
|
||||||
|
// DeleteLiveRoute each. It names each route by prefix and peer only, with no
|
||||||
|
// origin ASN, as a withdrawal from the feed does when its message carries no AS
|
||||||
|
// path.
|
||||||
|
func withdrawRoutes(db *Database, batch bool, routes []*LiveRoute) error {
|
||||||
|
if !batch {
|
||||||
|
for _, route := range routes {
|
||||||
|
if err := db.DeleteLiveRoute(route.Prefix, 0, route.PeerIP); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
deletions := make([]LiveRouteDeletion, 0, len(routes))
|
||||||
|
for _, route := range routes {
|
||||||
|
deletions = append(deletions, LiveRouteDeletion{
|
||||||
|
Prefix: route.Prefix, PeerIP: route.PeerIP, IPVersion: route.IPVersion,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
return db.DeleteLiveRouteBatch(deletions)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStatsRouteTimestamps checks the oldest/newest route timestamps are read
|
||||||
|
// from the right rows across both tables and parse into time.Time. The old
|
||||||
|
// MIN/MAX union query read its result into *time.Time, which the driver could
|
||||||
|
// not parse, so it logged a warning every call and left both timestamps nil.
|
||||||
|
func TestStatsRouteTimestamps(t *testing.T) {
|
||||||
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
|
|
||||||
|
db, err := New(cfg, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
empty, err := db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext on empty database: %v", err)
|
||||||
|
}
|
||||||
|
if empty.OldestRoute != nil || empty.NewestRoute != nil {
|
||||||
|
t.Fatalf("empty database timestamps = (%v, %v), want (nil, nil)",
|
||||||
|
empty.OldestRoute, empty.NewestRoute)
|
||||||
|
}
|
||||||
|
|
||||||
|
base := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
oldest := base
|
||||||
|
newest := base.Add(2 * time.Minute)
|
||||||
|
if err := db.UpsertLiveRouteBatch([]*LiveRoute{
|
||||||
|
mkV4Route(t, "198.51.100.0/24", 64500, base.Add(time.Minute)),
|
||||||
|
mkV4Route(t, "203.0.113.0/24", 64501, oldest),
|
||||||
|
mkV6Route("2001:db8::/32", 64502, newest),
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("UpsertLiveRouteBatch: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
stats, err := db.GetStatsContext(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext: %v", err)
|
||||||
|
}
|
||||||
|
if stats.OldestRoute == nil || !stats.OldestRoute.Equal(oldest) {
|
||||||
|
t.Errorf("OldestRoute = %v, want %v", stats.OldestRoute, oldest)
|
||||||
|
}
|
||||||
|
if stats.NewestRoute == nil || !stats.NewestRoute.Equal(newest) {
|
||||||
|
t.Errorf("NewestRoute = %v, want %v", stats.NewestRoute, newest)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLiveCountsConcurrentReadWrite runs writers and stats readers at once so
|
||||||
|
// the race detector proves the counters are safe under concurrent use.
|
||||||
|
func TestLiveCountsConcurrentReadWrite(t *testing.T) {
|
||||||
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
|
|
||||||
|
db, err := New(cfg, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
|
||||||
|
const writers = 4
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
|
||||||
|
wg.Add(writers)
|
||||||
|
for w := range writers {
|
||||||
|
go func(base int) {
|
||||||
|
defer wg.Done()
|
||||||
|
for i := range 25 {
|
||||||
|
asn := 65000 + base*100 + i
|
||||||
|
route := mkV6Route("2001:db8::/32", asn, ts)
|
||||||
|
route.PeerIP = "2001:db8::" + uuid.NewString()
|
||||||
|
if err := db.UpsertLiveRoute(route); err != nil {
|
||||||
|
t.Errorf("UpsertLiveRoute: %v", err)
|
||||||
|
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}(w)
|
||||||
|
}
|
||||||
|
|
||||||
|
var readerWG sync.WaitGroup
|
||||||
|
readerWG.Add(1)
|
||||||
|
stop := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
defer readerWG.Done()
|
||||||
|
for {
|
||||||
|
select {
|
||||||
|
case <-stop:
|
||||||
|
return
|
||||||
|
default:
|
||||||
|
if _, err := db.GetStatsContext(context.Background()); err != nil {
|
||||||
|
t.Errorf("GetStatsContext: %v", err)
|
||||||
|
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
wg.Wait()
|
||||||
|
close(stop)
|
||||||
|
readerWG.Wait()
|
||||||
|
|
||||||
|
stats, err := db.GetStatsContext(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetStatsContext: %v", err)
|
||||||
|
}
|
||||||
|
if want := writers * 25; stats.IPv6Routes != want {
|
||||||
|
t.Errorf("IPv6Routes = %d, want %d", stats.IPv6Routes, want)
|
||||||
|
}
|
||||||
|
// Every writer announced the same prefix, so it counts once.
|
||||||
|
assertDistribution(t, "IPv6 distribution", stats.IPv6PrefixDistribution,
|
||||||
|
[]PrefixDistribution{{MaskLength: 32, Count: 1}})
|
||||||
|
}
|
||||||
|
|
||||||
|
type wantCounts struct {
|
||||||
|
asns int
|
||||||
|
prefixes int
|
||||||
|
peerings int
|
||||||
|
peers int
|
||||||
|
ipv4Routes int
|
||||||
|
ipv6Routes int
|
||||||
|
liveRoutes int
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertCounts(t *testing.T, when string, got Stats, want wantCounts) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
if got.ASNs != want.asns {
|
||||||
|
t.Errorf("%s: ASNs = %d, want %d", when, got.ASNs, want.asns)
|
||||||
|
}
|
||||||
|
if got.Prefixes != want.prefixes {
|
||||||
|
t.Errorf("%s: Prefixes = %d, want %d", when, got.Prefixes, want.prefixes)
|
||||||
|
}
|
||||||
|
if got.Peerings != want.peerings {
|
||||||
|
t.Errorf("%s: Peerings = %d, want %d", when, got.Peerings, want.peerings)
|
||||||
|
}
|
||||||
|
if got.Peers != want.peers {
|
||||||
|
t.Errorf("%s: Peers = %d, want %d", when, got.Peers, want.peers)
|
||||||
|
}
|
||||||
|
if got.IPv4Routes != want.ipv4Routes {
|
||||||
|
t.Errorf("%s: IPv4Routes = %d, want %d", when, got.IPv4Routes, want.ipv4Routes)
|
||||||
|
}
|
||||||
|
if got.IPv6Routes != want.ipv6Routes {
|
||||||
|
t.Errorf("%s: IPv6Routes = %d, want %d", when, got.IPv6Routes, want.ipv6Routes)
|
||||||
|
}
|
||||||
|
if got.LiveRoutes != want.liveRoutes {
|
||||||
|
t.Errorf("%s: LiveRoutes = %d, want %d", when, got.LiveRoutes, want.liveRoutes)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func assertDistribution(t *testing.T, what string, got, want []PrefixDistribution) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
if !slices.Equal(got, want) {
|
||||||
|
t.Errorf("%s = %v, want %v", what, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
+550
-540
File diff suppressed because it is too large
Load Diff
+220
-279
@@ -1,301 +1,242 @@
|
|||||||
package database
|
package database
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"net"
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"net/netip"
|
||||||
|
"sync"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/config"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
|
"github.com/google/uuid"
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestIPToUint32(t *testing.T) {
|
// tempStoreMemory is the PRAGMA temp_store value meaning "hold temp B-trees in
|
||||||
tests := []struct {
|
// memory"; the DSN change must leave temp_store below this so they spill to disk.
|
||||||
name string
|
const tempStoreMemory = 2
|
||||||
ip string
|
|
||||||
expected uint32
|
// heldConnections is how many pooled connections the pragma test holds open at
|
||||||
}{
|
// once so each is a distinct SQLite connection that parsed the DSN.
|
||||||
{
|
const heldConnections = 5
|
||||||
name: "Simple IP",
|
|
||||||
ip: "192.168.1.1",
|
// testPeerIP is the peer every route in the IP lookup test is learned from.
|
||||||
expected: 3232235777, // 192<<24 + 168<<16 + 1<<8 + 1
|
const testPeerIP = "192.0.2.254"
|
||||||
},
|
|
||||||
{
|
// Parameters for the checkpoint-contention regression test.
|
||||||
name: "Minimum IP",
|
const (
|
||||||
ip: "0.0.0.0",
|
// contentionIterations is how many batch writes race the checkpoint loop.
|
||||||
expected: 0,
|
contentionIterations = 400
|
||||||
},
|
// contendedASNCount is the small set of ASNs the batches reuse, so most
|
||||||
{
|
// batches update existing rows and exercise the read-before-write path.
|
||||||
name: "Maximum IP",
|
contendedASNCount = 16
|
||||||
ip: "255.255.255.255",
|
// asnSecondBand offsets a second ASN per batch so each batch writes more
|
||||||
expected: 4294967295,
|
// than one row.
|
||||||
},
|
asnSecondBand = 100
|
||||||
{
|
)
|
||||||
name: "10.0.0.0",
|
|
||||||
ip: "10.0.0.0",
|
// TestGetIPInfoFindsMostSpecificLiveRoute stores nested live prefixes for both
|
||||||
expected: 167772160,
|
// families and checks that a lookup returns the most specific one covering the
|
||||||
},
|
// address, ErrNoRoute when none covers it, and the next less specific prefix
|
||||||
{
|
// once the only route of the most specific one is withdrawn.
|
||||||
name: "172.16.0.0",
|
func TestGetIPInfoFindsMostSpecificLiveRoute(t *testing.T) {
|
||||||
ip: "172.16.0.0",
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
expected: 2886729728,
|
|
||||||
},
|
db, err := New(cfg, logger.New())
|
||||||
{
|
if err != nil {
|
||||||
name: "8.8.8.8",
|
t.Fatalf("failed to create database: %v", err)
|
||||||
ip: "8.8.8.8",
|
}
|
||||||
expected: 134744072,
|
defer func() { _ = db.Close() }()
|
||||||
},
|
|
||||||
{
|
// Nested live prefixes, each originated by its own AS.
|
||||||
name: "1.2.3.4",
|
origins := map[string]int{
|
||||||
ip: "1.2.3.4",
|
"10.0.0.0/8": 64500,
|
||||||
expected: 16909060,
|
"10.1.0.0/16": 64501,
|
||||||
},
|
"10.1.2.0/24": 64502,
|
||||||
|
"2001:db8::/32": 64500,
|
||||||
|
"2001:db8:1::/48": 64501,
|
||||||
|
"2001:db8:1:2::/64": 64502,
|
||||||
|
}
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
routes := make([]*LiveRoute, 0, len(origins))
|
||||||
|
for prefix, asn := range origins {
|
||||||
|
routes = append(routes, &LiveRoute{
|
||||||
|
ID: uuid.New(),
|
||||||
|
Prefix: prefix,
|
||||||
|
MaskLength: netip.MustParsePrefix(prefix).Bits(),
|
||||||
|
IPVersion: detectIPVersion(prefix),
|
||||||
|
OriginASN: asn,
|
||||||
|
PeerIP: testPeerIP,
|
||||||
|
ASPath: []int{asn},
|
||||||
|
NextHop: testPeerIP,
|
||||||
|
LastUpdated: ts,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
if err := db.UpsertLiveRouteBatch(routes); err != nil {
|
||||||
|
t.Fatalf("UpsertLiveRouteBatch: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
for _, tt := range tests {
|
// lookup checks that ip resolves to the live prefix want, or to ErrNoRoute
|
||||||
t.Run(tt.name, func(t *testing.T) {
|
// when want is empty.
|
||||||
ip := net.ParseIP(tt.ip)
|
lookup := func(ip, want string) {
|
||||||
if ip == nil {
|
t.Helper()
|
||||||
t.Fatalf("Failed to parse IP: %s", tt.ip)
|
|
||||||
|
info, err := db.GetIPInfo(ip)
|
||||||
|
if want == "" {
|
||||||
|
if !errors.Is(err, ErrNoRoute) {
|
||||||
|
t.Errorf("GetIPInfo(%s) = %+v, %v; want ErrNoRoute", ip, info, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
result := ipToUint32(ip)
|
return
|
||||||
if result != tt.expected {
|
}
|
||||||
t.Errorf("ipToUint32(%s) = %d, want %d", tt.ip, result, tt.expected)
|
if err != nil {
|
||||||
}
|
t.Errorf("GetIPInfo(%s): %v", ip, err)
|
||||||
|
|
||||||
// Test with IPv4-mapped IPv6 address
|
return
|
||||||
ip6 := net.ParseIP(tt.ip).To16()
|
}
|
||||||
if ip6 != nil {
|
if info.Netblock != want || info.MaskLength != netip.MustParsePrefix(want).Bits() ||
|
||||||
result6 := ipToUint32(ip6)
|
info.ASN != origins[want] {
|
||||||
if result6 != tt.expected {
|
t.Errorf("GetIPInfo(%s) = %s (mask %d) AS%d, want %s AS%d",
|
||||||
t.Errorf("ipToUint32(%s as IPv6) = %d, want %d", tt.ip, result6, tt.expected)
|
ip, info.Netblock, info.MaskLength, info.ASN, want, origins[want])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
})
|
|
||||||
|
lookup("10.1.2.3", "10.1.2.0/24")
|
||||||
|
lookup("::ffff:10.1.2.3", "10.1.2.0/24")
|
||||||
|
lookup("10.1.3.4", "10.1.0.0/16")
|
||||||
|
lookup("10.2.0.1", "10.0.0.0/8")
|
||||||
|
lookup("192.0.2.1", "")
|
||||||
|
lookup("2001:db8:1:2::3", "2001:db8:1:2::/64")
|
||||||
|
lookup("2001:db8:1:3::4", "2001:db8:1::/48")
|
||||||
|
lookup("2001:db8:2::1", "2001:db8::/32")
|
||||||
|
lookup("2001:db9::1", "")
|
||||||
|
|
||||||
|
err = db.DeleteLiveRouteBatch([]LiveRouteDeletion{
|
||||||
|
{Prefix: "10.1.2.0/24", OriginASN: 64502, PeerIP: testPeerIP, IPVersion: ipVersionV4},
|
||||||
|
{Prefix: "2001:db8:1:2::/64", OriginASN: 64502, PeerIP: testPeerIP, IPVersion: ipVersionV6},
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("DeleteLiveRouteBatch: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
lookup("10.1.2.3", "10.1.0.0/16")
|
||||||
|
lookup("2001:db8:1:2::3", "2001:db8:1::/48")
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestConnectionPoolPragmas holds several pooled connections open at once and
|
||||||
|
// checks each one carries the per-connection settings from the DSN, plus the
|
||||||
|
// process-wide hard heap limit.
|
||||||
|
func TestConnectionPoolPragmas(t *testing.T) {
|
||||||
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
|
|
||||||
|
db, err := New(cfg, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
// Hold distinct connections open simultaneously so the pool must open a new
|
||||||
|
// one (each parsing the DSN) rather than hand back the same connection.
|
||||||
|
conns := make([]*sql.Conn, 0, heldConnections)
|
||||||
|
defer func() {
|
||||||
|
for _, c := range conns {
|
||||||
|
_ = c.Close()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
for i := 0; i < heldConnections; i++ {
|
||||||
|
c, err := db.db.Conn(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to open connection %d: %v", i, err)
|
||||||
|
}
|
||||||
|
conns = append(conns, c)
|
||||||
|
}
|
||||||
|
|
||||||
|
for i, c := range conns {
|
||||||
|
var cacheSize int
|
||||||
|
if err := c.QueryRowContext(ctx, "PRAGMA cache_size").Scan(&cacheSize); err != nil {
|
||||||
|
t.Fatalf("conn %d: failed to read cache_size: %v", i, err)
|
||||||
|
}
|
||||||
|
if cacheSize != sqliteCacheSizeKiB {
|
||||||
|
t.Errorf("conn %d: cache_size = %d, want %d", i, cacheSize, sqliteCacheSizeKiB)
|
||||||
|
}
|
||||||
|
|
||||||
|
var busyTimeout int
|
||||||
|
if err := c.QueryRowContext(ctx, "PRAGMA busy_timeout").Scan(&busyTimeout); err != nil {
|
||||||
|
t.Fatalf("conn %d: failed to read busy_timeout: %v", i, err)
|
||||||
|
}
|
||||||
|
if busyTimeout != sqliteBusyTimeoutMs {
|
||||||
|
t.Errorf("conn %d: busy_timeout = %d, want %d", i, busyTimeout, sqliteBusyTimeoutMs)
|
||||||
|
}
|
||||||
|
|
||||||
|
var tempStore int
|
||||||
|
if err := c.QueryRowContext(ctx, "PRAGMA temp_store").Scan(&tempStore); err != nil {
|
||||||
|
t.Fatalf("conn %d: failed to read temp_store: %v", i, err)
|
||||||
|
}
|
||||||
|
if tempStore == tempStoreMemory {
|
||||||
|
t.Errorf("conn %d: temp_store = %d, want anything but %d (MEMORY)", i, tempStore, tempStoreMemory)
|
||||||
|
}
|
||||||
|
|
||||||
|
var hardHeapLimit int64
|
||||||
|
if err := c.QueryRowContext(ctx, "PRAGMA hard_heap_limit").Scan(&hardHeapLimit); err != nil {
|
||||||
|
t.Fatalf("conn %d: failed to read hard_heap_limit: %v", i, err)
|
||||||
|
}
|
||||||
|
if hardHeapLimit != sqliteHardHeapLimitBytes {
|
||||||
|
t.Errorf("conn %d: hard_heap_limit = %d, want %d", i, hardHeapLimit, sqliteHardHeapLimitBytes)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCalculateIPv4Range(t *testing.T) {
|
// TestBatchWriteDuringCheckpoint reproduces issue #25. A batch write reads
|
||||||
tests := []struct {
|
// (SELECT) before it writes (INSERT/UPDATE). Under the default deferred locking
|
||||||
name string
|
// the transaction begins as a reader and, when the maintainer's WAL checkpoint
|
||||||
cidr string
|
// holds the write lock, its upgrade to writer fails immediately with "database
|
||||||
wantStart uint32
|
// is locked" without honouring busy_timeout, dropping the batch. With
|
||||||
wantEnd uint32
|
// _txlock=immediate the transaction takes the write lock at BEGIN and waits, so
|
||||||
wantErr bool
|
// no batch is dropped. The checkpoint runs without the Database mutex, exactly
|
||||||
}{
|
// as the background maintainer does in production.
|
||||||
{
|
func TestBatchWriteDuringCheckpoint(t *testing.T) {
|
||||||
name: "Single IP /32",
|
cfg := &config.Config{StateDir: t.TempDir()}
|
||||||
cidr: "192.168.1.1/32",
|
|
||||||
wantStart: 3232235777,
|
db, err := New(cfg, logger.New())
|
||||||
wantEnd: 3232235777,
|
if err != nil {
|
||||||
},
|
t.Fatalf("failed to create database: %v", err)
|
||||||
{
|
|
||||||
name: "Class C /24",
|
|
||||||
cidr: "192.168.1.0/24",
|
|
||||||
wantStart: 3232235776, // 192.168.1.0
|
|
||||||
wantEnd: 3232236031, // 192.168.1.255
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Class B /16",
|
|
||||||
cidr: "192.168.0.0/16",
|
|
||||||
wantStart: 3232235520, // 192.168.0.0
|
|
||||||
wantEnd: 3232301055, // 192.168.255.255
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Class A /8",
|
|
||||||
cidr: "10.0.0.0/8",
|
|
||||||
wantStart: 167772160, // 10.0.0.0
|
|
||||||
wantEnd: 184549375, // 10.255.255.255
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Entire IPv4 space /0",
|
|
||||||
cidr: "0.0.0.0/0",
|
|
||||||
wantStart: 0,
|
|
||||||
wantEnd: 4294967295,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Small subnet /30",
|
|
||||||
cidr: "192.168.1.0/30",
|
|
||||||
wantStart: 3232235776, // 192.168.1.0
|
|
||||||
wantEnd: 3232235779, // 192.168.1.3
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Medium subnet /20",
|
|
||||||
cidr: "172.16.0.0/20",
|
|
||||||
wantStart: 2886729728, // 172.16.0.0
|
|
||||||
wantEnd: 2886733823, // 172.16.15.255
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Private range 172.16/12",
|
|
||||||
cidr: "172.16.0.0/12",
|
|
||||||
wantStart: 2886729728, // 172.16.0.0
|
|
||||||
wantEnd: 2887778303, // 172.31.255.255
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Google DNS /29",
|
|
||||||
cidr: "8.8.8.8/29",
|
|
||||||
wantStart: 134744072, // 8.8.8.8 (network is actually 8.8.8.8 with /29)
|
|
||||||
wantEnd: 134744079, // 8.8.8.15
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Non-zero host bits",
|
|
||||||
cidr: "192.168.1.5/24",
|
|
||||||
wantStart: 3232235776, // 192.168.1.0 (network address)
|
|
||||||
wantEnd: 3232236031, // 192.168.1.255
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Invalid CIDR",
|
|
||||||
cidr: "192.168.1.1/33",
|
|
||||||
wantErr: true,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Invalid IP",
|
|
||||||
cidr: "256.256.256.256/24",
|
|
||||||
wantErr: true,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "IPv6 CIDR",
|
|
||||||
cidr: "2001:db8::/32",
|
|
||||||
wantErr: true,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Empty CIDR",
|
|
||||||
cidr: "",
|
|
||||||
wantErr: true,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "Missing mask",
|
|
||||||
cidr: "192.168.1.1",
|
|
||||||
wantErr: true,
|
|
||||||
},
|
|
||||||
}
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
for _, tt := range tests {
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
t.Run(tt.name, func(t *testing.T) {
|
|
||||||
start, end, err := CalculateIPv4Range(tt.cidr)
|
|
||||||
|
|
||||||
if tt.wantErr {
|
var wg sync.WaitGroup
|
||||||
if err == nil {
|
wg.Add(1)
|
||||||
t.Errorf("CalculateIPv4Range(%s) expected error, got nil", tt.cidr)
|
go func() {
|
||||||
}
|
defer wg.Done()
|
||||||
|
for {
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
return
|
return
|
||||||
|
default:
|
||||||
|
_ = db.Checkpoint(ctx) // errors are the checkpoint's own to absorb
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
if err != nil {
|
ts := time.Now().UTC()
|
||||||
t.Errorf("CalculateIPv4Range(%s) unexpected error: %v", tt.cidr, err)
|
for i := 0; i < contentionIterations; i++ {
|
||||||
return
|
asns := map[int]time.Time{
|
||||||
}
|
i % contendedASNCount: ts,
|
||||||
|
(i % contendedASNCount) + asnSecondBand: ts,
|
||||||
if start != tt.wantStart {
|
}
|
||||||
t.Errorf("CalculateIPv4Range(%s) start = %d, want %d", tt.cidr, start, tt.wantStart)
|
if err := db.GetOrCreateASNBatch(asns); err != nil {
|
||||||
}
|
cancel()
|
||||||
|
wg.Wait()
|
||||||
if end != tt.wantEnd {
|
t.Fatalf("batch write failed under checkpoint contention: %v", err)
|
||||||
t.Errorf("CalculateIPv4Range(%s) end = %d, want %d", tt.cidr, end, tt.wantEnd)
|
}
|
||||||
}
|
|
||||||
|
|
||||||
// Verify that start <= end
|
|
||||||
if start > end {
|
|
||||||
t.Errorf("CalculateIPv4Range(%s) start (%d) > end (%d)", tt.cidr, start, end)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify the range size matches the CIDR mask
|
|
||||||
if !tt.wantErr && tt.cidr != "" {
|
|
||||||
_, ipNet, _ := net.ParseCIDR(tt.cidr)
|
|
||||||
if ipNet != nil {
|
|
||||||
ones, bits := ipNet.Mask.Size()
|
|
||||||
expectedSize := uint32(1) << uint(bits-ones)
|
|
||||||
actualSize := end - start + 1
|
|
||||||
if actualSize != expectedSize {
|
|
||||||
t.Errorf("CalculateIPv4Range(%s) range size = %d, want %d", tt.cidr, actualSize, expectedSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestIPv4RangeIntegration(t *testing.T) {
|
|
||||||
// Test that our functions work correctly together
|
|
||||||
tests := []struct {
|
|
||||||
name string
|
|
||||||
cidr string
|
|
||||||
testIPs []string
|
|
||||||
shouldContain []bool
|
|
||||||
}{
|
|
||||||
{
|
|
||||||
name: "192.168.1.0/24",
|
|
||||||
cidr: "192.168.1.0/24",
|
|
||||||
testIPs: []string{
|
|
||||||
"192.168.1.0",
|
|
||||||
"192.168.1.1",
|
|
||||||
"192.168.1.255",
|
|
||||||
"192.168.0.255",
|
|
||||||
"192.168.2.0",
|
|
||||||
},
|
|
||||||
shouldContain: []bool{true, true, true, false, false},
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "10.0.0.0/8",
|
|
||||||
cidr: "10.0.0.0/8",
|
|
||||||
testIPs: []string{
|
|
||||||
"10.0.0.0",
|
|
||||||
"10.255.255.255",
|
|
||||||
"10.1.2.3",
|
|
||||||
"9.255.255.255",
|
|
||||||
"11.0.0.0",
|
|
||||||
},
|
|
||||||
shouldContain: []bool{true, true, true, false, false},
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "172.16.0.0/12",
|
|
||||||
cidr: "172.16.0.0/12",
|
|
||||||
testIPs: []string{
|
|
||||||
"172.16.0.0",
|
|
||||||
"172.31.255.255",
|
|
||||||
"172.20.1.1",
|
|
||||||
"172.15.255.255",
|
|
||||||
"172.32.0.0",
|
|
||||||
},
|
|
||||||
shouldContain: []bool{true, true, true, false, false},
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tt := range tests {
|
|
||||||
t.Run(tt.name, func(t *testing.T) {
|
|
||||||
start, end, err := CalculateIPv4Range(tt.cidr)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("Failed to calculate range for %s: %v", tt.cidr, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
for i, testIP := range tt.testIPs {
|
|
||||||
ip := net.ParseIP(testIP)
|
|
||||||
if ip == nil {
|
|
||||||
t.Fatalf("Failed to parse test IP: %s", testIP)
|
|
||||||
}
|
|
||||||
|
|
||||||
ipUint := ipToUint32(ip)
|
|
||||||
contained := ipUint >= start && ipUint <= end
|
|
||||||
|
|
||||||
if contained != tt.shouldContain[i] {
|
|
||||||
t.Errorf("IP %s in range %s: got %v, want %v", testIP, tt.cidr, contained, tt.shouldContain[i])
|
|
||||||
}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func BenchmarkIPToUint32(b *testing.B) {
|
|
||||||
ip := net.ParseIP("192.168.1.1")
|
|
||||||
b.ResetTimer()
|
|
||||||
|
|
||||||
for i := 0; i < b.N; i++ {
|
|
||||||
_ = ipToUint32(ip)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func BenchmarkCalculateIPv4Range(b *testing.B) {
|
|
||||||
cidr := "192.168.0.0/16"
|
|
||||||
b.ResetTimer()
|
|
||||||
|
|
||||||
for i := 0; i < b.N; i++ {
|
|
||||||
_, _, _ = CalculateIPv4Range(cidr)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
cancel()
|
||||||
|
wg.Wait()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -18,6 +18,8 @@ type Stats struct {
|
|||||||
Peers int
|
Peers int
|
||||||
FileSizeBytes int64
|
FileSizeBytes int64
|
||||||
LiveRoutes int
|
LiveRoutes int
|
||||||
|
IPv4Routes int
|
||||||
|
IPv6Routes int
|
||||||
OldestRoute *time.Time
|
OldestRoute *time.Time
|
||||||
NewestRoute *time.Time
|
NewestRoute *time.Time
|
||||||
IPv4PrefixDistribution []PrefixDistribution
|
IPv4PrefixDistribution []PrefixDistribution
|
||||||
@@ -61,8 +63,6 @@ type Store interface {
|
|||||||
GetLiveRouteCountsContext(ctx context.Context) (ipv4Count, ipv6Count int, err error)
|
GetLiveRouteCountsContext(ctx context.Context) (ipv4Count, ipv6Count int, err error)
|
||||||
|
|
||||||
// IP lookup operations
|
// IP lookup operations
|
||||||
GetASInfoForIP(ip string) (*ASInfo, error)
|
|
||||||
GetASInfoForIPContext(ctx context.Context, ip string) (*ASInfo, error)
|
|
||||||
GetIPInfo(ip string) (*IPInfo, error)
|
GetIPInfo(ip string) (*IPInfo, error)
|
||||||
GetIPInfoContext(ctx context.Context, ip string) (*IPInfo, error)
|
GetIPInfoContext(ctx context.Context, ip string) (*IPInfo, error)
|
||||||
|
|
||||||
|
|||||||
@@ -77,9 +77,6 @@ type LiveRoute struct {
|
|||||||
ASPath []int `json:"as_path"`
|
ASPath []int `json:"as_path"`
|
||||||
NextHop string `json:"next_hop"`
|
NextHop string `json:"next_hop"`
|
||||||
LastUpdated time.Time `json:"last_updated"`
|
LastUpdated time.Time `json:"last_updated"`
|
||||||
// IPv4 range fields for fast lookups (nil for IPv6)
|
|
||||||
V4IPStart *uint32 `json:"v4_ip_start,omitempty"`
|
|
||||||
V4IPEnd *uint32 `json:"v4_ip_end,omitempty"`
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// PrefixDistribution represents the distribution of prefixes by mask length
|
// PrefixDistribution represents the distribution of prefixes by mask length
|
||||||
@@ -88,16 +85,6 @@ type PrefixDistribution struct {
|
|||||||
Count int `json:"count"`
|
Count int `json:"count"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// ASInfo represents AS information for an IP lookup (legacy format)
|
|
||||||
type ASInfo struct {
|
|
||||||
ASN int `json:"asn"`
|
|
||||||
Handle string `json:"handle"`
|
|
||||||
Description string `json:"description"`
|
|
||||||
Prefix string `json:"prefix"`
|
|
||||||
LastUpdated time.Time `json:"last_updated"`
|
|
||||||
Age string `json:"age"`
|
|
||||||
}
|
|
||||||
|
|
||||||
// IPInfo represents comprehensive IP information for the /ip endpoint
|
// IPInfo represents comprehensive IP information for the /ip endpoint
|
||||||
type IPInfo struct {
|
type IPInfo struct {
|
||||||
IP string `json:"ip"`
|
IP string `json:"ip"`
|
||||||
|
|||||||
@@ -107,9 +107,6 @@ CREATE TABLE IF NOT EXISTS live_routes_v4 (
|
|||||||
as_path TEXT NOT NULL, -- JSON array
|
as_path TEXT NOT NULL, -- JSON array
|
||||||
next_hop TEXT NOT NULL,
|
next_hop TEXT NOT NULL,
|
||||||
last_updated DATETIME NOT NULL,
|
last_updated DATETIME NOT NULL,
|
||||||
-- IPv4 range columns for fast lookups
|
|
||||||
ip_start INTEGER NOT NULL, -- Start of IPv4 range as 32-bit unsigned int
|
|
||||||
ip_end INTEGER NOT NULL, -- End of IPv4 range as 32-bit unsigned int
|
|
||||||
UNIQUE(prefix, origin_asn, peer_ip)
|
UNIQUE(prefix, origin_asn, peer_ip)
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -123,7 +120,6 @@ CREATE TABLE IF NOT EXISTS live_routes_v6 (
|
|||||||
as_path TEXT NOT NULL, -- JSON array
|
as_path TEXT NOT NULL, -- JSON array
|
||||||
next_hop TEXT NOT NULL,
|
next_hop TEXT NOT NULL,
|
||||||
last_updated DATETIME NOT NULL,
|
last_updated DATETIME NOT NULL,
|
||||||
-- Note: IPv6 doesn't use integer range columns
|
|
||||||
UNIQUE(prefix, origin_asn, peer_ip)
|
UNIQUE(prefix, origin_asn, peer_ip)
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -132,8 +128,6 @@ CREATE INDEX IF NOT EXISTS idx_live_routes_v4_prefix ON live_routes_v4(prefix);
|
|||||||
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_mask_length ON live_routes_v4(mask_length);
|
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_mask_length ON live_routes_v4(mask_length);
|
||||||
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_origin_asn ON live_routes_v4(origin_asn);
|
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_origin_asn ON live_routes_v4(origin_asn);
|
||||||
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_last_updated ON live_routes_v4(last_updated);
|
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_last_updated ON live_routes_v4(last_updated);
|
||||||
-- Indexes for IPv4 range queries
|
|
||||||
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_ip_range ON live_routes_v4(ip_start, ip_end);
|
|
||||||
-- Index to optimize prefix distribution queries
|
-- Index to optimize prefix distribution queries
|
||||||
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_mask_prefix ON live_routes_v4(mask_length, prefix);
|
CREATE INDEX IF NOT EXISTS idx_live_routes_v4_mask_prefix ON live_routes_v4(mask_length, prefix);
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
package database
|
package database
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"fmt"
|
||||||
|
"net"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"github.com/google/uuid"
|
"github.com/google/uuid"
|
||||||
@@ -18,3 +20,14 @@ func detectIPVersion(prefix string) int {
|
|||||||
|
|
||||||
return ipVersionV4
|
return ipVersionV4
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// prefixMaskLength returns the mask length of a prefix such as 192.0.2.0/24.
|
||||||
|
func prefixMaskLength(prefix string) (int, error) {
|
||||||
|
_, network, err := net.ParseCIDR(prefix)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("invalid prefix %s: %w", prefix, err)
|
||||||
|
}
|
||||||
|
maskLength, _ := network.Mask.Size()
|
||||||
|
|
||||||
|
return maskLength, nil
|
||||||
|
}
|
||||||
|
|||||||
+22
-18
@@ -63,24 +63,28 @@ type RISLiveMessage struct {
|
|||||||
// the actual BGP update data including AS path, communities, announcements,
|
// the actual BGP update data including AS path, communities, announcements,
|
||||||
// and withdrawals.
|
// and withdrawals.
|
||||||
type RISMessage struct {
|
type RISMessage struct {
|
||||||
Type string `json:"type"`
|
Type string `json:"type"`
|
||||||
Timestamp float64 `json:"timestamp"`
|
Timestamp float64 `json:"timestamp"`
|
||||||
ParsedTimestamp time.Time `json:"-"` // Parsed from Timestamp field
|
ParsedTimestamp time.Time `json:"-"` // Parsed from Timestamp field
|
||||||
Peer string `json:"peer"`
|
Peer string `json:"peer"`
|
||||||
PeerASN string `json:"peer_asn"`
|
PeerASN string `json:"peer_asn"`
|
||||||
ID string `json:"id"`
|
ID string `json:"id"`
|
||||||
Host string `json:"host"`
|
Host string `json:"host"`
|
||||||
RRC string `json:"rrc,omitempty"`
|
RRC string `json:"rrc,omitempty"`
|
||||||
MrtTime float64 `json:"mrt_time,omitempty"`
|
MrtTime float64 `json:"mrt_time,omitempty"`
|
||||||
SocketTime float64 `json:"socket_time,omitempty"`
|
SocketTime float64 `json:"socket_time,omitempty"`
|
||||||
Path ASPath `json:"path,omitempty"`
|
Path ASPath `json:"path,omitempty"`
|
||||||
Community [][]int `json:"community,omitempty"`
|
// Community and Raw are present in the feed but read by no handler.
|
||||||
Origin string `json:"origin,omitempty"`
|
// They are the largest fields on a message that lives in up to four
|
||||||
MED *int `json:"med,omitempty"`
|
// handler queues, so json:"-" keeps them out of the decoded message
|
||||||
LocalPref *int `json:"local_pref,omitempty"`
|
// to save queue memory. Do not decode them without a consumer.
|
||||||
Announcements []RISAnnouncement `json:"announcements,omitempty"`
|
Community [][]int `json:"-"`
|
||||||
Withdrawals []string `json:"withdrawals,omitempty"`
|
Origin string `json:"origin,omitempty"`
|
||||||
Raw string `json:"raw,omitempty"`
|
MED *int `json:"med,omitempty"`
|
||||||
|
LocalPref *int `json:"local_pref,omitempty"`
|
||||||
|
Announcements []RISAnnouncement `json:"announcements,omitempty"`
|
||||||
|
Withdrawals []string `json:"withdrawals,omitempty"`
|
||||||
|
Raw string `json:"-"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// RISAnnouncement represents a BGP route announcement within a RIS message.
|
// RISAnnouncement represents a BGP route announcement within a RIS message.
|
||||||
|
|||||||
@@ -0,0 +1,67 @@
|
|||||||
|
package ristypes
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bufio"
|
||||||
|
"encoding/json"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// messageExamplesPath is the captured RIS Live feed used as decode fixtures,
|
||||||
|
// one JSON message per line.
|
||||||
|
const messageExamplesPath = "../../docs/message-examples.json"
|
||||||
|
|
||||||
|
// TestDecodeDropsCommunityAndRaw decodes every captured message the way the
|
||||||
|
// streamer does and checks that the fields no handler reads (Community, Raw)
|
||||||
|
// stay empty while the fields handlers use (Path, Announcements) still decode.
|
||||||
|
func TestDecodeDropsCommunityAndRaw(t *testing.T) {
|
||||||
|
f, err := os.Open(messageExamplesPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("open fixtures: %v", err)
|
||||||
|
}
|
||||||
|
defer f.Close()
|
||||||
|
|
||||||
|
var messages, withPath, withAnnouncements int
|
||||||
|
|
||||||
|
scanner := bufio.NewScanner(f)
|
||||||
|
scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024)
|
||||||
|
for scanner.Scan() {
|
||||||
|
line := scanner.Bytes()
|
||||||
|
if len(line) == 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
var wrapper RISLiveMessage
|
||||||
|
if err := json.Unmarshal(line, &wrapper); err != nil {
|
||||||
|
t.Fatalf("unmarshal message %d: %v", messages+1, err)
|
||||||
|
}
|
||||||
|
messages++
|
||||||
|
|
||||||
|
msg := wrapper.Data
|
||||||
|
if msg.Community != nil {
|
||||||
|
t.Errorf("message %d: Community decoded, want empty: %v", messages, msg.Community)
|
||||||
|
}
|
||||||
|
if msg.Raw != "" {
|
||||||
|
t.Errorf("message %d: Raw decoded, want empty", messages)
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(msg.Path) > 0 {
|
||||||
|
withPath++
|
||||||
|
}
|
||||||
|
if len(msg.Announcements) > 0 {
|
||||||
|
withAnnouncements++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := scanner.Err(); err != nil {
|
||||||
|
t.Fatalf("scan fixtures: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if messages == 0 {
|
||||||
|
t.Fatal("no messages decoded from fixtures")
|
||||||
|
}
|
||||||
|
// The fixtures include announcement messages; a used field must still decode,
|
||||||
|
// otherwise an empty Community/Raw would prove nothing.
|
||||||
|
if withPath == 0 || withAnnouncements == 0 {
|
||||||
|
t.Fatalf("used fields did not decode: withPath=%d withAnnouncements=%d", withPath, withAnnouncements)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -232,25 +232,6 @@ func (m *mockStore) GetLiveRouteCountsContext(ctx context.Context) (ipv4Count, i
|
|||||||
return m.GetLiveRouteCounts()
|
return m.GetLiveRouteCounts()
|
||||||
}
|
}
|
||||||
|
|
||||||
// GetASInfoForIP mock implementation
|
|
||||||
func (m *mockStore) GetASInfoForIP(ip string) (*database.ASInfo, error) {
|
|
||||||
// Simple mock - return a test AS
|
|
||||||
now := time.Now()
|
|
||||||
return &database.ASInfo{
|
|
||||||
ASN: 15169,
|
|
||||||
Handle: "GOOGLE",
|
|
||||||
Description: "Google LLC",
|
|
||||||
Prefix: "8.8.8.0/24",
|
|
||||||
LastUpdated: now.Add(-5 * time.Minute),
|
|
||||||
Age: "5m0s",
|
|
||||||
}, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// GetASInfoForIPContext mock implementation with context support
|
|
||||||
func (m *mockStore) GetASInfoForIPContext(ctx context.Context, ip string) (*database.ASInfo, error) {
|
|
||||||
return m.GetASInfoForIP(ip)
|
|
||||||
}
|
|
||||||
|
|
||||||
// GetASDetails mock implementation
|
// GetASDetails mock implementation
|
||||||
func (m *mockStore) GetASDetails(asn int) (*database.ASN, []database.LiveRoute, error) {
|
func (m *mockStore) GetASDetails(asn int) (*database.ASN, []database.LiveRoute, error) {
|
||||||
m.mu.Lock()
|
m.mu.Lock()
|
||||||
@@ -426,6 +407,9 @@ func (m *mockStore) Ping(ctx context.Context) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestRouteWatchLiveFeed(t *testing.T) {
|
func TestRouteWatchLiveFeed(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("skipping live RIPE RIS network feed test in short mode; run without -short to include it")
|
||||||
|
}
|
||||||
|
|
||||||
// Create mock database
|
// Create mock database
|
||||||
mockDB := newMockStore()
|
mockDB := newMockStore()
|
||||||
@@ -447,7 +431,7 @@ func TestRouteWatchLiveFeed(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Create server
|
// Create server
|
||||||
srv := server.New(mockDB, s, logger)
|
srv := server.New(mockDB, s, logger, cfg)
|
||||||
|
|
||||||
// Create RouteWatch with 5 second limit
|
// Create RouteWatch with 5 second limit
|
||||||
deps := Dependencies{
|
deps := Dependencies{
|
||||||
|
|||||||
@@ -10,9 +10,11 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
// asHandlerQueueSize is the queue capacity for ASN operations
|
// asHandlerQueueSize is the queue capacity for ASN operations, about 4
|
||||||
// DO NOT set this higher than 100000 without explicit instructions
|
// seconds of feed at peak. The streamer drops rather than blocks when a
|
||||||
asHandlerQueueSize = 100000
|
// queue is full, so this bounds memory. Batches still flush on a timer
|
||||||
|
// (asnBatchTimeout), so a queue smaller than asnBatchSize is fine.
|
||||||
|
asHandlerQueueSize = 20000
|
||||||
|
|
||||||
// asnBatchSize is the number of ASN operations to batch together
|
// asnBatchSize is the number of ASN operations to batch together
|
||||||
asnBatchSize = 30000
|
asnBatchSize = 30000
|
||||||
|
|||||||
@@ -14,8 +14,11 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
// peerHandlerQueueSize is the queue capacity for peer tracking operations
|
// peerHandlerQueueSize is the queue capacity for peer tracking operations,
|
||||||
peerHandlerQueueSize = 100000
|
// about 4 seconds of feed at peak. The streamer drops rather than blocks
|
||||||
|
// when a queue is full, so this bounds memory. Batches still flush on a
|
||||||
|
// timer (peerBatchTimeout).
|
||||||
|
peerHandlerQueueSize = 20000
|
||||||
|
|
||||||
// peerBatchSize is the number of peer updates to batch together
|
// peerBatchSize is the number of peer updates to batch together
|
||||||
peerBatchSize = 10000
|
peerBatchSize = 10000
|
||||||
|
|||||||
@@ -11,29 +11,25 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
// peeringHandlerQueueSize defines the buffer capacity for the peering
|
// peeringHandlerQueueSize is the buffer capacity for the peering handler's
|
||||||
// handler's message queue. This should be large enough to handle bursts
|
// message queue, about 4 seconds of feed at peak. The streamer drops
|
||||||
// of BGP UPDATE messages without blocking.
|
// rather than blocks when a queue is full, so this bounds memory.
|
||||||
peeringHandlerQueueSize = 100000
|
peeringHandlerQueueSize = 20000
|
||||||
|
|
||||||
// minPathLengthForPeering specifies the minimum number of ASNs required
|
// minPathLengthForPeering specifies the minimum number of ASNs required
|
||||||
// in a BGP AS path to extract peering relationships. A path with fewer
|
// in a BGP AS path to extract peering relationships. A path with fewer
|
||||||
// than 2 ASNs cannot contain any peering information.
|
// than 2 ASNs cannot contain any peering information.
|
||||||
minPathLengthForPeering = 2
|
minPathLengthForPeering = 2
|
||||||
|
|
||||||
// pathExpirationTime determines how long AS paths are kept in memory
|
|
||||||
// before being eligible for pruning. Paths older than this are removed
|
|
||||||
// to prevent unbounded memory growth.
|
|
||||||
pathExpirationTime = 30 * time.Minute
|
|
||||||
|
|
||||||
// peeringProcessInterval controls how frequently the handler processes
|
// peeringProcessInterval controls how frequently the handler processes
|
||||||
// accumulated AS paths and extracts peering relationships to store
|
// accumulated AS paths and extracts peering relationships to store
|
||||||
// in the database.
|
// in the database.
|
||||||
peeringProcessInterval = 30 * time.Second
|
peeringProcessInterval = 30 * time.Second
|
||||||
|
|
||||||
// pathPruneInterval determines how often the handler checks for and
|
// maxTrackedPaths bounds how many distinct AS paths are held in memory
|
||||||
// removes expired AS paths from memory.
|
// between processing runs. Once the map is full, further new paths are
|
||||||
pathPruneInterval = 5 * time.Minute
|
// dropped and counted until the next run empties it.
|
||||||
|
maxTrackedPaths = 500000
|
||||||
)
|
)
|
||||||
|
|
||||||
// PeeringHandler processes BGP UPDATE messages to extract and track
|
// PeeringHandler processes BGP UPDATE messages to extract and track
|
||||||
@@ -46,17 +42,17 @@ type PeeringHandler struct {
|
|||||||
logger *logger.Logger
|
logger *logger.Logger
|
||||||
|
|
||||||
// In-memory AS path tracking
|
// In-memory AS path tracking
|
||||||
mu sync.RWMutex
|
mu sync.Mutex
|
||||||
asPaths map[string]time.Time // key is JSON-encoded AS path
|
asPaths map[string]time.Time // key is JSON-encoded AS path
|
||||||
|
droppedPaths int // paths dropped because the map was full
|
||||||
|
|
||||||
stopCh chan struct{}
|
stopCh chan struct{}
|
||||||
}
|
}
|
||||||
|
|
||||||
// NewPeeringHandler creates and initializes a new PeeringHandler with the
|
// NewPeeringHandler creates and initializes a new PeeringHandler with the
|
||||||
// provided database store and logger. It starts two background goroutines:
|
// provided database store and logger. It starts one background goroutine that
|
||||||
// one for periodic processing of accumulated AS paths into peering records,
|
// periodically processes accumulated AS paths into peering records. The
|
||||||
// and one for pruning expired paths from memory. The handler begins
|
// handler begins processing immediately upon creation.
|
||||||
// processing immediately upon creation.
|
|
||||||
func NewPeeringHandler(db database.Store, logger *logger.Logger) *PeeringHandler {
|
func NewPeeringHandler(db database.Store, logger *logger.Logger) *PeeringHandler {
|
||||||
h := &PeeringHandler{
|
h := &PeeringHandler{
|
||||||
db: db,
|
db: db,
|
||||||
@@ -65,9 +61,8 @@ func NewPeeringHandler(db database.Store, logger *logger.Logger) *PeeringHandler
|
|||||||
stopCh: make(chan struct{}),
|
stopCh: make(chan struct{}),
|
||||||
}
|
}
|
||||||
|
|
||||||
// Start the periodic processing goroutines
|
// Start the periodic processing goroutine
|
||||||
go h.processLoop()
|
go h.processLoop()
|
||||||
go h.pruneLoop()
|
|
||||||
|
|
||||||
return h
|
return h
|
||||||
}
|
}
|
||||||
@@ -106,9 +101,18 @@ func (h *PeeringHandler) HandleMessage(msg *ristypes.RISMessage) {
|
|||||||
|
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
key := string(pathJSON)
|
||||||
|
|
||||||
h.mu.Lock()
|
h.mu.Lock()
|
||||||
h.asPaths[string(pathJSON)] = timestamp
|
if _, exists := h.asPaths[key]; exists {
|
||||||
|
// Already tracked: refresh its timestamp.
|
||||||
|
h.asPaths[key] = timestamp
|
||||||
|
} else if len(h.asPaths) >= maxTrackedPaths {
|
||||||
|
// Map is full; drop this new path and count it.
|
||||||
|
h.droppedPaths++
|
||||||
|
} else {
|
||||||
|
h.asPaths[key] = timestamp
|
||||||
|
}
|
||||||
h.mu.Unlock()
|
h.mu.Unlock()
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -130,41 +134,6 @@ func (h *PeeringHandler) processLoop() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// pruneLoop runs periodically to remove old AS paths
|
|
||||||
func (h *PeeringHandler) pruneLoop() {
|
|
||||||
ticker := time.NewTicker(pathPruneInterval)
|
|
||||||
defer ticker.Stop()
|
|
||||||
|
|
||||||
for {
|
|
||||||
select {
|
|
||||||
case <-ticker.C:
|
|
||||||
h.prunePaths()
|
|
||||||
case <-h.stopCh:
|
|
||||||
return
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// prunePaths removes AS paths older than pathExpirationTime
|
|
||||||
func (h *PeeringHandler) prunePaths() {
|
|
||||||
cutoff := time.Now().Add(-pathExpirationTime)
|
|
||||||
var removed int
|
|
||||||
|
|
||||||
h.mu.Lock()
|
|
||||||
for pathKey, timestamp := range h.asPaths {
|
|
||||||
if timestamp.Before(cutoff) {
|
|
||||||
delete(h.asPaths, pathKey)
|
|
||||||
removed++
|
|
||||||
}
|
|
||||||
}
|
|
||||||
pathCount := len(h.asPaths)
|
|
||||||
h.mu.Unlock()
|
|
||||||
|
|
||||||
if removed > 0 {
|
|
||||||
h.logger.Debug("Pruned old AS paths", "removed", removed, "remaining", pathCount)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ProcessPeeringsNow triggers immediate processing of all accumulated AS
|
// ProcessPeeringsNow triggers immediate processing of all accumulated AS
|
||||||
// paths into peering records. This bypasses the normal periodic processing
|
// paths into peering records. This bypasses the normal periodic processing
|
||||||
// schedule and is primarily intended for testing purposes.
|
// schedule and is primarily intended for testing purposes.
|
||||||
@@ -174,15 +143,16 @@ func (h *PeeringHandler) ProcessPeeringsNow() {
|
|||||||
|
|
||||||
// processPeerings extracts peerings from AS paths and writes to database
|
// processPeerings extracts peerings from AS paths and writes to database
|
||||||
func (h *PeeringHandler) processPeerings() {
|
func (h *PeeringHandler) processPeerings() {
|
||||||
// Take a snapshot of current AS paths
|
// Take the accumulated paths and replace the map with a fresh empty one
|
||||||
h.mu.RLock()
|
// under the lock. Each path is processed exactly once and the memory is
|
||||||
pathsCopy := make(map[string]time.Time, len(h.asPaths))
|
// released, so the map never grows past a single interval's traffic.
|
||||||
for k, v := range h.asPaths {
|
h.mu.Lock()
|
||||||
pathsCopy[k] = v
|
paths := h.asPaths
|
||||||
}
|
h.asPaths = make(map[string]time.Time)
|
||||||
h.mu.RUnlock()
|
dropped := h.droppedPaths
|
||||||
|
h.mu.Unlock()
|
||||||
|
|
||||||
if len(pathsCopy) == 0 {
|
if len(paths) == 0 {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -192,7 +162,7 @@ func (h *PeeringHandler) processPeerings() {
|
|||||||
}
|
}
|
||||||
peerings := make(map[peeringKey]time.Time)
|
peerings := make(map[peeringKey]time.Time)
|
||||||
|
|
||||||
for pathJSON, timestamp := range pathsCopy {
|
for pathJSON, timestamp := range paths {
|
||||||
var path []int
|
var path []int
|
||||||
if err := json.Unmarshal([]byte(pathJSON), &path); err != nil {
|
if err := json.Unmarshal([]byte(pathJSON), &path); err != nil {
|
||||||
h.logger.Error("Failed to decode AS path", "error", err)
|
h.logger.Error("Failed to decode AS path", "error", err)
|
||||||
@@ -241,15 +211,16 @@ func (h *PeeringHandler) processPeerings() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
h.logger.Info("Processed AS peerings",
|
h.logger.Info("Processed AS peerings",
|
||||||
"paths", len(pathsCopy),
|
"paths", len(paths),
|
||||||
"unique_peerings", len(peerings),
|
"unique_peerings", len(peerings),
|
||||||
"success", successCount,
|
"success", successCount,
|
||||||
|
"dropped_paths", dropped,
|
||||||
"duration", time.Since(start),
|
"duration", time.Since(start),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Stop gracefully shuts down the handler by signaling the background
|
// Stop gracefully shuts down the handler by signaling the background
|
||||||
// goroutines to stop and performing a final synchronous processing of
|
// goroutine to stop and performing a final synchronous processing of
|
||||||
// any remaining AS paths. This ensures no peering data is lost during
|
// any remaining AS paths. This ensures no peering data is lost during
|
||||||
// shutdown.
|
// shutdown.
|
||||||
func (h *PeeringHandler) Stop() {
|
func (h *PeeringHandler) Stop() {
|
||||||
|
|||||||
@@ -0,0 +1,163 @@
|
|||||||
|
package routewatch
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"strconv"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/database"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/ristypes"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
testASNA = 64500
|
||||||
|
testASNB = 64501
|
||||||
|
testASNC = 64502
|
||||||
|
)
|
||||||
|
|
||||||
|
// recordingStore wraps mockStore to count every RecordPeering call, so a
|
||||||
|
// test can tell how many times a peering was written across separate runs.
|
||||||
|
type recordingStore struct {
|
||||||
|
*mockStore
|
||||||
|
|
||||||
|
mu sync.Mutex
|
||||||
|
calls int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingStore) RecordPeering(asA, asB int, ts time.Time) error {
|
||||||
|
r.mu.Lock()
|
||||||
|
r.calls++
|
||||||
|
r.mu.Unlock()
|
||||||
|
|
||||||
|
return r.mockStore.RecordPeering(asA, asB, ts)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingStore) callCount() int {
|
||||||
|
r.mu.Lock()
|
||||||
|
defer r.mu.Unlock()
|
||||||
|
|
||||||
|
return r.calls
|
||||||
|
}
|
||||||
|
|
||||||
|
// newTestHandler builds a PeeringHandler without starting the periodic
|
||||||
|
// processing goroutine, so tests drive processing explicitly.
|
||||||
|
func newTestHandler(db database.Store) *PeeringHandler {
|
||||||
|
return &PeeringHandler{
|
||||||
|
db: db,
|
||||||
|
logger: logger.New(),
|
||||||
|
asPaths: make(map[string]time.Time),
|
||||||
|
stopCh: make(chan struct{}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func snapshot(h *PeeringHandler) (tracked, dropped int) {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
|
||||||
|
return len(h.asPaths), h.droppedPaths
|
||||||
|
}
|
||||||
|
|
||||||
|
func pathKey(t *testing.T, asns ...int) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
b, err := json.Marshal(ristypes.ASPath(asns))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to marshal path: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return string(b)
|
||||||
|
}
|
||||||
|
|
||||||
|
func handle(h *PeeringHandler, ts time.Time, asns ...int) {
|
||||||
|
h.HandleMessage(&ristypes.RISMessage{
|
||||||
|
Path: ristypes.ASPath(asns),
|
||||||
|
ParsedTimestamp: ts,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPeeringHandlerProcessesEachRunAndEmpties verifies that a processing run
|
||||||
|
// empties the path map (the swap) and that a path seen again after a run is
|
||||||
|
// recorded in the next run too.
|
||||||
|
func TestPeeringHandlerProcessesEachRunAndEmpties(t *testing.T) {
|
||||||
|
store := &recordingStore{mockStore: newMockStore()}
|
||||||
|
h := newTestHandler(store)
|
||||||
|
|
||||||
|
now := time.Now().UTC()
|
||||||
|
|
||||||
|
// Run 1: one path, one peering recorded, map emptied afterwards.
|
||||||
|
handle(h, now, testASNA, testASNB)
|
||||||
|
h.ProcessPeeringsNow()
|
||||||
|
|
||||||
|
if tracked, _ := snapshot(h); tracked != 0 {
|
||||||
|
t.Fatalf("map not empty after first run: %d paths remain", tracked)
|
||||||
|
}
|
||||||
|
if got := store.callCount(); got != 1 {
|
||||||
|
t.Fatalf("want 1 RecordPeering call after first run, got %d", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Run 2: the same path again is recorded again (RecordPeering upserts).
|
||||||
|
handle(h, now.Add(time.Second), testASNA, testASNB)
|
||||||
|
h.ProcessPeeringsNow()
|
||||||
|
|
||||||
|
if tracked, _ := snapshot(h); tracked != 0 {
|
||||||
|
t.Fatalf("map not empty after second run: %d paths remain", tracked)
|
||||||
|
}
|
||||||
|
if got := store.callCount(); got != 2 {
|
||||||
|
t.Fatalf("want 2 RecordPeering calls after second run, got %d", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPeeringHandlerCapDropsAndCounts verifies that a full map drops new paths
|
||||||
|
// and counts them, while a path already tracked is refreshed rather than
|
||||||
|
// dropped.
|
||||||
|
func TestPeeringHandlerCapDropsAndCounts(t *testing.T) {
|
||||||
|
store := &recordingStore{mockStore: newMockStore()}
|
||||||
|
h := newTestHandler(store)
|
||||||
|
|
||||||
|
now := time.Now().UTC()
|
||||||
|
|
||||||
|
// Fill the map to exactly maxTrackedPaths, including one real path key so
|
||||||
|
// the "already tracked" branch can be exercised. The filler keys are never
|
||||||
|
// processed in this test, so their contents do not matter.
|
||||||
|
existing := pathKey(t, testASNA, testASNB)
|
||||||
|
|
||||||
|
h.mu.Lock()
|
||||||
|
h.asPaths[existing] = now
|
||||||
|
for i := 0; len(h.asPaths) < maxTrackedPaths; i++ {
|
||||||
|
h.asPaths[strconv.Itoa(i)] = now
|
||||||
|
}
|
||||||
|
h.mu.Unlock()
|
||||||
|
|
||||||
|
// A new path is dropped and counted because the map is full.
|
||||||
|
handle(h, now.Add(time.Second), testASNA, testASNC)
|
||||||
|
|
||||||
|
tracked, dropped := snapshot(h)
|
||||||
|
if tracked != maxTrackedPaths {
|
||||||
|
t.Fatalf("want map size %d after drop, got %d", maxTrackedPaths, tracked)
|
||||||
|
}
|
||||||
|
if dropped != 1 {
|
||||||
|
t.Fatalf("want dropped count 1, got %d", dropped)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A path already tracked is refreshed, not dropped.
|
||||||
|
refreshed := now.Add(2 * time.Second)
|
||||||
|
handle(h, refreshed, testASNA, testASNB)
|
||||||
|
|
||||||
|
tracked, dropped = snapshot(h)
|
||||||
|
if tracked != maxTrackedPaths {
|
||||||
|
t.Fatalf("want map size %d after refresh, got %d", maxTrackedPaths, tracked)
|
||||||
|
}
|
||||||
|
if dropped != 1 {
|
||||||
|
t.Fatalf("want dropped count still 1 after refresh, got %d", dropped)
|
||||||
|
}
|
||||||
|
|
||||||
|
h.mu.Lock()
|
||||||
|
gotTS := h.asPaths[existing]
|
||||||
|
h.mu.Unlock()
|
||||||
|
if !gotTS.Equal(refreshed) {
|
||||||
|
t.Fatalf("existing path timestamp not refreshed: want %v, got %v", refreshed, gotTS)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -2,6 +2,7 @@ package routewatch
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"net"
|
"net"
|
||||||
|
"net/netip"
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
"time"
|
"time"
|
||||||
@@ -14,9 +15,11 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
// prefixHandlerQueueSize is the queue capacity for prefix tracking operations
|
// prefixHandlerQueueSize is the queue capacity for prefix tracking
|
||||||
// DO NOT set this higher than 100000 without explicit instructions
|
// operations, about 4 seconds of feed at peak. The streamer drops rather
|
||||||
prefixHandlerQueueSize = 100000
|
// than blocks when a queue is full, so this bounds memory. Batches still
|
||||||
|
// flush on a timer (prefixBatchTimeout).
|
||||||
|
prefixHandlerQueueSize = 20000
|
||||||
|
|
||||||
// prefixBatchSize is the number of prefix updates to batch together
|
// prefixBatchSize is the number of prefix updates to batch together
|
||||||
prefixBatchSize = 25000
|
prefixBatchSize = 25000
|
||||||
@@ -106,7 +109,7 @@ func (h *PrefixHandler) HandleMessage(msg *ristypes.RISMessage) {
|
|||||||
for _, announcement := range msg.Announcements {
|
for _, announcement := range msg.Announcements {
|
||||||
for _, prefix := range announcement.Prefixes {
|
for _, prefix := range announcement.Prefixes {
|
||||||
h.batch = append(h.batch, prefixUpdate{
|
h.batch = append(h.batch, prefixUpdate{
|
||||||
prefix: prefix,
|
prefix: canonicalPrefix(prefix),
|
||||||
originASN: originASN,
|
originASN: originASN,
|
||||||
peer: msg.Peer,
|
peer: msg.Peer,
|
||||||
messageType: "announcement",
|
messageType: "announcement",
|
||||||
@@ -123,7 +126,7 @@ func (h *PrefixHandler) HandleMessage(msg *ristypes.RISMessage) {
|
|||||||
// Process withdrawals
|
// Process withdrawals
|
||||||
for _, prefix := range msg.Withdrawals {
|
for _, prefix := range msg.Withdrawals {
|
||||||
h.batch = append(h.batch, prefixUpdate{
|
h.batch = append(h.batch, prefixUpdate{
|
||||||
prefix: prefix,
|
prefix: canonicalPrefix(prefix),
|
||||||
originASN: originASN, // Use the originASN from path if available
|
originASN: originASN, // Use the originASN from path if available
|
||||||
peer: msg.Peer,
|
peer: msg.Peer,
|
||||||
messageType: "withdrawal",
|
messageType: "withdrawal",
|
||||||
@@ -262,6 +265,21 @@ func (h *PrefixHandler) flushBatchLocked() {
|
|||||||
h.lastFlush = time.Now()
|
h.lastFlush = time.Now()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// canonicalPrefix returns prefix in the text form net/netip prints, the form
|
||||||
|
// the IP lookup builds when it looks a prefix up. The feed sends IPv6
|
||||||
|
// announcements compressed ("2001:db8::/32") but IPv6 withdrawals uncompressed
|
||||||
|
// ("2001:db8:0:0:0:0:0:0/32"); stored as received, a withdrawal would not match
|
||||||
|
// the route its announcement stored. A prefix that does not parse is returned
|
||||||
|
// unchanged, and the batch flush reports it.
|
||||||
|
func canonicalPrefix(prefix string) string {
|
||||||
|
p, err := netip.ParsePrefix(prefix)
|
||||||
|
if err != nil {
|
||||||
|
return prefix
|
||||||
|
}
|
||||||
|
|
||||||
|
return p.Masked().String()
|
||||||
|
}
|
||||||
|
|
||||||
// parseCIDR extracts the mask length and IP version from a prefix string
|
// parseCIDR extracts the mask length and IP version from a prefix string
|
||||||
func parseCIDR(prefix string) (maskLength int, ipVersion int, err error) {
|
func parseCIDR(prefix string) (maskLength int, ipVersion int, err error) {
|
||||||
_, ipNet, err := net.ParseCIDR(prefix)
|
_, ipNet, err := net.ParseCIDR(prefix)
|
||||||
@@ -313,20 +331,6 @@ func (h *PrefixHandler) processAnnouncement(_ *database.Prefix, update prefixUpd
|
|||||||
LastUpdated: update.timestamp,
|
LastUpdated: update.timestamp,
|
||||||
}
|
}
|
||||||
|
|
||||||
// For IPv4, calculate the IP range
|
|
||||||
if ipVersion == ipv4Version {
|
|
||||||
start, end, err := database.CalculateIPv4Range(update.prefix)
|
|
||||||
if err == nil {
|
|
||||||
liveRoute.V4IPStart = &start
|
|
||||||
liveRoute.V4IPEnd = &end
|
|
||||||
} else {
|
|
||||||
h.logger.Error("Failed to calculate IPv4 range",
|
|
||||||
"prefix", update.prefix,
|
|
||||||
"error", err,
|
|
||||||
)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if err := h.db.UpsertLiveRoute(liveRoute); err != nil {
|
if err := h.db.UpsertLiveRoute(liveRoute); err != nil {
|
||||||
h.logger.Error("Failed to upsert live route",
|
h.logger.Error("Failed to upsert live route",
|
||||||
"prefix", update.prefix,
|
"prefix", update.prefix,
|
||||||
@@ -370,20 +374,6 @@ func (h *PrefixHandler) createLiveRoute(update prefixUpdate) *database.LiveRoute
|
|||||||
LastUpdated: update.timestamp,
|
LastUpdated: update.timestamp,
|
||||||
}
|
}
|
||||||
|
|
||||||
// For IPv4, calculate the IP range
|
|
||||||
if ipVersion == ipv4Version {
|
|
||||||
start, end, err := database.CalculateIPv4Range(update.prefix)
|
|
||||||
if err == nil {
|
|
||||||
liveRoute.V4IPStart = &start
|
|
||||||
liveRoute.V4IPEnd = &end
|
|
||||||
} else {
|
|
||||||
h.logger.Error("Failed to calculate IPv4 range",
|
|
||||||
"prefix", update.prefix,
|
|
||||||
"error", err,
|
|
||||||
)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return liveRoute
|
return liveRoute
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -423,20 +413,6 @@ func (h *PrefixHandler) processAnnouncementDirect(update prefixUpdate) {
|
|||||||
LastUpdated: update.timestamp,
|
LastUpdated: update.timestamp,
|
||||||
}
|
}
|
||||||
|
|
||||||
// For IPv4, calculate the IP range
|
|
||||||
if ipVersion == ipv4Version {
|
|
||||||
start, end, err := database.CalculateIPv4Range(update.prefix)
|
|
||||||
if err == nil {
|
|
||||||
liveRoute.V4IPStart = &start
|
|
||||||
liveRoute.V4IPEnd = &end
|
|
||||||
} else {
|
|
||||||
h.logger.Error("Failed to calculate IPv4 range",
|
|
||||||
"prefix", update.prefix,
|
|
||||||
"error", err,
|
|
||||||
)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if err := h.db.UpsertLiveRoute(liveRoute); err != nil {
|
if err := h.db.UpsertLiveRoute(liveRoute); err != nil {
|
||||||
h.logger.Error("Failed to upsert live route",
|
h.logger.Error("Failed to upsert live route",
|
||||||
"prefix", update.prefix,
|
"prefix", update.prefix,
|
||||||
|
|||||||
@@ -0,0 +1,74 @@
|
|||||||
|
package routewatch
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/config"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/database"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/ristypes"
|
||||||
|
)
|
||||||
|
|
||||||
|
const testPeerIP = "2001:db8:ffff::1"
|
||||||
|
|
||||||
|
// TestPrefixHandlerStoresPrefixesTheIPLookupFinds runs announcements and
|
||||||
|
// withdrawals through the prefix handler into a real database and looks the
|
||||||
|
// addresses up. The prefixes are written the way the feed sends them: IPv6
|
||||||
|
// announcements compressed, IPv6 withdrawals uncompressed. The withdrawal must
|
||||||
|
// remove the route the announcement stored, and the IP lookup must find the
|
||||||
|
// stored prefix.
|
||||||
|
func TestPrefixHandlerStoresPrefixesTheIPLookupFinds(t *testing.T) {
|
||||||
|
db, err := database.New(&config.Config{StateDir: t.TempDir()}, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("failed to create database: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
|
// Built without NewPrefixHandler's flush timer, so the test flushes itself.
|
||||||
|
h := &PrefixHandler{db: db, logger: logger.New()}
|
||||||
|
flush := func() {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
h.flushBatchLocked()
|
||||||
|
}
|
||||||
|
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
h.HandleMessage(&ristypes.RISMessage{
|
||||||
|
Peer: testPeerIP,
|
||||||
|
Path: ristypes.ASPath{testASNA, testASNB},
|
||||||
|
ParsedTimestamp: ts,
|
||||||
|
Announcements: []ristypes.RISAnnouncement{{
|
||||||
|
NextHop: testPeerIP,
|
||||||
|
Prefixes: []string{"2001:db8:1::/48", "192.0.2.0/24"},
|
||||||
|
}},
|
||||||
|
})
|
||||||
|
flush()
|
||||||
|
|
||||||
|
for ip, want := range map[string]string{
|
||||||
|
"2001:db8:1::1": "2001:db8:1::/48",
|
||||||
|
"192.0.2.1": "192.0.2.0/24",
|
||||||
|
} {
|
||||||
|
info, err := db.GetIPInfo(ip)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetIPInfo(%s) after announcement: %v", ip, err)
|
||||||
|
}
|
||||||
|
if info.Netblock != want || info.ASN != testASNB {
|
||||||
|
t.Errorf("GetIPInfo(%s) = %s AS%d, want %s AS%d", ip, info.Netblock, info.ASN, want, testASNB)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
h.HandleMessage(&ristypes.RISMessage{
|
||||||
|
Peer: testPeerIP,
|
||||||
|
ParsedTimestamp: ts.Add(time.Minute),
|
||||||
|
Withdrawals: []string{"2001:db8:1:0:0:0:0:0/48", "192.0.2.0/24"},
|
||||||
|
})
|
||||||
|
flush()
|
||||||
|
|
||||||
|
for _, ip := range []string{"2001:db8:1::1", "192.0.2.1"} {
|
||||||
|
if info, err := db.GetIPInfo(ip); !errors.Is(err, database.ErrNoRoute) {
|
||||||
|
t.Errorf("GetIPInfo(%s) after withdrawal = %+v, %v; want ErrNoRoute", ip, info, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+14
-24
@@ -179,9 +179,11 @@ func (s *Server) handleStatusJSON() http.HandlerFunc {
|
|||||||
|
|
||||||
metrics := s.streamer.GetMetrics()
|
metrics := s.streamer.GetMetrics()
|
||||||
|
|
||||||
// Get database stats with timeout
|
// Get database stats with timeout. The channels are buffered so the
|
||||||
statsChan := make(chan database.Stats)
|
// goroutine's send never blocks if the timeout wins and nothing here
|
||||||
errChan := make(chan error)
|
// receives; otherwise it would block forever and leak.
|
||||||
|
statsChan := make(chan database.Stats, 1)
|
||||||
|
errChan := make(chan error, 1)
|
||||||
|
|
||||||
go func() {
|
go func() {
|
||||||
dbStats, err := s.db.GetStatsContext(ctx)
|
dbStats, err := s.db.GetStatsContext(ctx)
|
||||||
@@ -217,13 +219,6 @@ func (s *Server) handleStatusJSON() http.HandlerFunc {
|
|||||||
|
|
||||||
const bitsPerMegabit = 1000000.0
|
const bitsPerMegabit = 1000000.0
|
||||||
|
|
||||||
// Get route counts from database
|
|
||||||
ipv4Routes, ipv6Routes, err := s.db.GetLiveRouteCountsContext(ctx)
|
|
||||||
if err != nil {
|
|
||||||
s.logger.Warn("Failed to get live route counts", "error", err)
|
|
||||||
// Continue with zero counts
|
|
||||||
}
|
|
||||||
|
|
||||||
// Get route update metrics
|
// Get route update metrics
|
||||||
routeMetrics := s.streamer.GetMetricsTracker().GetRouteMetrics()
|
routeMetrics := s.streamer.GetMetricsTracker().GetRouteMetrics()
|
||||||
|
|
||||||
@@ -257,8 +252,8 @@ func (s *Server) handleStatusJSON() http.HandlerFunc {
|
|||||||
Peers: dbStats.Peers,
|
Peers: dbStats.Peers,
|
||||||
DatabaseSizeBytes: dbStats.FileSizeBytes,
|
DatabaseSizeBytes: dbStats.FileSizeBytes,
|
||||||
LiveRoutes: dbStats.LiveRoutes,
|
LiveRoutes: dbStats.LiveRoutes,
|
||||||
IPv4Routes: ipv4Routes,
|
IPv4Routes: dbStats.IPv4Routes,
|
||||||
IPv6Routes: ipv6Routes,
|
IPv6Routes: dbStats.IPv6Routes,
|
||||||
OldestRoute: dbStats.OldestRoute,
|
OldestRoute: dbStats.OldestRoute,
|
||||||
NewestRoute: dbStats.NewestRoute,
|
NewestRoute: dbStats.NewestRoute,
|
||||||
IPv4UpdatesPerSec: routeMetrics.IPv4UpdatesPerSec,
|
IPv4UpdatesPerSec: routeMetrics.IPv4UpdatesPerSec,
|
||||||
@@ -398,9 +393,11 @@ func (s *Server) handleStats() http.HandlerFunc {
|
|||||||
|
|
||||||
metrics := s.streamer.GetMetrics()
|
metrics := s.streamer.GetMetrics()
|
||||||
|
|
||||||
// Get database stats with timeout
|
// Get database stats with timeout. The channels are buffered so the
|
||||||
statsChan := make(chan database.Stats)
|
// goroutine's send never blocks if the timeout wins and nothing here
|
||||||
errChan := make(chan error)
|
// receives; otherwise it would block forever and leak.
|
||||||
|
statsChan := make(chan database.Stats, 1)
|
||||||
|
errChan := make(chan error, 1)
|
||||||
|
|
||||||
go func() {
|
go func() {
|
||||||
dbStats, err := s.db.GetStatsContext(ctx)
|
dbStats, err := s.db.GetStatsContext(ctx)
|
||||||
@@ -435,13 +432,6 @@ func (s *Server) handleStats() http.HandlerFunc {
|
|||||||
|
|
||||||
const bitsPerMegabit = 1000000.0
|
const bitsPerMegabit = 1000000.0
|
||||||
|
|
||||||
// Get route counts from database
|
|
||||||
ipv4Routes, ipv6Routes, err := s.db.GetLiveRouteCountsContext(ctx)
|
|
||||||
if err != nil {
|
|
||||||
s.logger.Warn("Failed to get live route counts", "error", err)
|
|
||||||
// Continue with zero counts
|
|
||||||
}
|
|
||||||
|
|
||||||
// Get route update metrics
|
// Get route update metrics
|
||||||
routeMetrics := s.streamer.GetMetricsTracker().GetRouteMetrics()
|
routeMetrics := s.streamer.GetMetricsTracker().GetRouteMetrics()
|
||||||
|
|
||||||
@@ -533,8 +523,8 @@ func (s *Server) handleStats() http.HandlerFunc {
|
|||||||
Peers: dbStats.Peers,
|
Peers: dbStats.Peers,
|
||||||
DatabaseSizeBytes: dbStats.FileSizeBytes,
|
DatabaseSizeBytes: dbStats.FileSizeBytes,
|
||||||
LiveRoutes: dbStats.LiveRoutes,
|
LiveRoutes: dbStats.LiveRoutes,
|
||||||
IPv4Routes: ipv4Routes,
|
IPv4Routes: dbStats.IPv4Routes,
|
||||||
IPv6Routes: ipv6Routes,
|
IPv6Routes: dbStats.IPv6Routes,
|
||||||
OldestRoute: dbStats.OldestRoute,
|
OldestRoute: dbStats.OldestRoute,
|
||||||
NewestRoute: dbStats.NewestRoute,
|
NewestRoute: dbStats.NewestRoute,
|
||||||
IPv4UpdatesPerSec: routeMetrics.IPv4UpdatesPerSec,
|
IPv4UpdatesPerSec: routeMetrics.IPv4UpdatesPerSec,
|
||||||
|
|||||||
@@ -0,0 +1,177 @@
|
|||||||
|
package server
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"runtime"
|
||||||
|
"slices"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/config"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/database"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/metrics"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/streamer"
|
||||||
|
"github.com/google/uuid"
|
||||||
|
)
|
||||||
|
|
||||||
|
// blockingStatsDB embeds database.Store (left nil) and overrides only
|
||||||
|
// GetStatsContext, which blocks until release is closed. The stats handlers
|
||||||
|
// call it in a goroutine; every other Store method is unused on the timeout
|
||||||
|
// path and would panic if called.
|
||||||
|
type blockingStatsDB struct {
|
||||||
|
database.Store
|
||||||
|
release chan struct{}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (d blockingStatsDB) GetStatsContext(_ context.Context) (database.Stats, error) {
|
||||||
|
<-d.release
|
||||||
|
|
||||||
|
return database.Stats{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStatsHandlersDoNotLeakOnTimeout drives each stats handler repeatedly with
|
||||||
|
// a request whose context times out before the database responds, then releases
|
||||||
|
// the blocked queries and asserts the goroutine count returns to its starting
|
||||||
|
// value. Before the fix the per-request goroutine sent on an unbuffered channel
|
||||||
|
// that nothing received once the timeout won, so it blocked forever and every
|
||||||
|
// poll leaked one goroutine.
|
||||||
|
func TestStatsHandlersDoNotLeakOnTimeout(t *testing.T) {
|
||||||
|
release := make(chan struct{})
|
||||||
|
db := blockingStatsDB{release: release}
|
||||||
|
s := New(db, streamer.New(logger.New(), metrics.New()), logger.New(), &config.Config{})
|
||||||
|
|
||||||
|
handlers := map[string]http.HandlerFunc{
|
||||||
|
"status.json": s.handleStatusJSON(),
|
||||||
|
"stats": s.handleStats(),
|
||||||
|
}
|
||||||
|
|
||||||
|
baseline := settledGoroutineCount()
|
||||||
|
|
||||||
|
const (
|
||||||
|
iterations = 20
|
||||||
|
requestTimeout = 50 * time.Millisecond
|
||||||
|
)
|
||||||
|
for _, handler := range handlers {
|
||||||
|
for range iterations {
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), requestTimeout)
|
||||||
|
req := httptest.NewRequest(http.MethodGet, "/", nil).WithContext(ctx)
|
||||||
|
handler(httptest.NewRecorder(), req)
|
||||||
|
cancel()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Let the blocked queries finish; with buffered channels each goroutine's
|
||||||
|
// send now succeeds and the goroutine exits.
|
||||||
|
close(release)
|
||||||
|
|
||||||
|
if !waitForGoroutines(baseline) {
|
||||||
|
t.Fatalf("goroutines did not return to baseline %d, got %d",
|
||||||
|
baseline, runtime.NumGoroutine())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStatsHandlersAnswerFromMemory checks that both stats handlers answer 200
|
||||||
|
// with the live route counts and the prefix distribution while the database is
|
||||||
|
// closed, so that any query would fail: the request path reads them from
|
||||||
|
// memory. The prefix distribution query it used to run read every live route
|
||||||
|
// and, on a large database, took the whole 4-second deadline, so
|
||||||
|
// /api/v1/stats answered 500 (https://git.eeqj.de/sneak/routewatch/issues/30).
|
||||||
|
// The oldest and newest route times still come from one-row lookups at the ends
|
||||||
|
// of an index; with the database closed they are left out of the answer.
|
||||||
|
func TestStatsHandlersAnswerFromMemory(t *testing.T) {
|
||||||
|
db, err := database.New(&config.Config{StateDir: t.TempDir()}, logger.New())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("database.New: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||||
|
if err := db.UpsertLiveRouteBatch([]*database.LiveRoute{
|
||||||
|
{
|
||||||
|
ID: uuid.New(), Prefix: "198.51.100.0/24", MaskLength: 24, IPVersion: 4,
|
||||||
|
OriginASN: 64500, PeerIP: "192.0.2.1", ASPath: []int{64500}, NextHop: "192.0.2.1",
|
||||||
|
LastUpdated: ts,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
ID: uuid.New(), Prefix: "2001:db8::/32", MaskLength: 32, IPVersion: 6,
|
||||||
|
OriginASN: 64501, PeerIP: "2001:db8::1", ASPath: []int{64501}, NextHop: "2001:db8::1",
|
||||||
|
LastUpdated: ts,
|
||||||
|
},
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatalf("UpsertLiveRouteBatch: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.Close(); err != nil {
|
||||||
|
t.Fatalf("Close: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
s := New(db, streamer.New(logger.New(), metrics.New()), logger.New(), &config.Config{})
|
||||||
|
handlers := map[string]http.HandlerFunc{
|
||||||
|
"status.json": s.handleStatusJSON(),
|
||||||
|
"stats": s.handleStats(),
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, handler := range handlers {
|
||||||
|
rec := httptest.NewRecorder()
|
||||||
|
handler(rec, httptest.NewRequest(http.MethodGet, "/", nil))
|
||||||
|
if rec.Code != http.StatusOK {
|
||||||
|
t.Errorf("%s: status %d, want %d; body %s", name, rec.Code, http.StatusOK, rec.Body)
|
||||||
|
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
var body struct {
|
||||||
|
Data struct {
|
||||||
|
IPv4Routes int `json:"ipv4_routes"`
|
||||||
|
IPv6Routes int `json:"ipv6_routes"`
|
||||||
|
IPv4PrefixDistribution []database.PrefixDistribution `json:"ipv4_prefix_distribution"`
|
||||||
|
IPv6PrefixDistribution []database.PrefixDistribution `json:"ipv6_prefix_distribution"`
|
||||||
|
} `json:"data"`
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(rec.Body.Bytes(), &body); err != nil {
|
||||||
|
t.Fatalf("%s: decoding the answer: %v", name, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if body.Data.IPv4Routes != 1 || body.Data.IPv6Routes != 1 {
|
||||||
|
t.Errorf("%s: routes = (v4 %d, v6 %d), want (1, 1)", name, body.Data.IPv4Routes, body.Data.IPv6Routes)
|
||||||
|
}
|
||||||
|
wantV4 := []database.PrefixDistribution{{MaskLength: 24, Count: 1}}
|
||||||
|
if !slices.Equal(body.Data.IPv4PrefixDistribution, wantV4) {
|
||||||
|
t.Errorf("%s: IPv4 distribution = %v, want %v", name, body.Data.IPv4PrefixDistribution, wantV4)
|
||||||
|
}
|
||||||
|
wantV6 := []database.PrefixDistribution{{MaskLength: 32, Count: 1}}
|
||||||
|
if !slices.Equal(body.Data.IPv6PrefixDistribution, wantV6) {
|
||||||
|
t.Errorf("%s: IPv6 distribution = %v, want %v", name, body.Data.IPv6PrefixDistribution, wantV6)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// settledGoroutineCount lets transient goroutines finish, then reports the
|
||||||
|
// current count.
|
||||||
|
func settledGoroutineCount() int {
|
||||||
|
prev := runtime.NumGoroutine()
|
||||||
|
for range 20 {
|
||||||
|
time.Sleep(10 * time.Millisecond)
|
||||||
|
cur := runtime.NumGoroutine()
|
||||||
|
if cur == prev {
|
||||||
|
return cur
|
||||||
|
}
|
||||||
|
prev = cur
|
||||||
|
}
|
||||||
|
|
||||||
|
return prev
|
||||||
|
}
|
||||||
|
|
||||||
|
// waitForGoroutines waits until the goroutine count drops to target or below.
|
||||||
|
func waitForGoroutines(target int) bool {
|
||||||
|
for range 100 {
|
||||||
|
if runtime.NumGoroutine() <= target {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
time.Sleep(10 * time.Millisecond)
|
||||||
|
}
|
||||||
|
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -4,9 +4,10 @@ package server
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"net/http"
|
"net/http"
|
||||||
"os"
|
"strconv"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/config"
|
||||||
"git.eeqj.de/sneak/routewatch/internal/database"
|
"git.eeqj.de/sneak/routewatch/internal/database"
|
||||||
"git.eeqj.de/sneak/routewatch/internal/logger"
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
"git.eeqj.de/sneak/routewatch/internal/streamer"
|
"git.eeqj.de/sneak/routewatch/internal/streamer"
|
||||||
@@ -33,16 +34,18 @@ type Server struct {
|
|||||||
db database.Store
|
db database.Store
|
||||||
streamer *streamer.Streamer
|
streamer *streamer.Streamer
|
||||||
logger *logger.Logger
|
logger *logger.Logger
|
||||||
|
port int
|
||||||
srv *http.Server
|
srv *http.Server
|
||||||
asnFetcher ASNFetcher
|
asnFetcher ASNFetcher
|
||||||
}
|
}
|
||||||
|
|
||||||
// New creates a new HTTP server
|
// New creates a new HTTP server
|
||||||
func New(db database.Store, streamer *streamer.Streamer, logger *logger.Logger) *Server {
|
func New(db database.Store, streamer *streamer.Streamer, logger *logger.Logger, cfg *config.Config) *Server {
|
||||||
s := &Server{
|
s := &Server{
|
||||||
db: db,
|
db: db,
|
||||||
streamer: streamer,
|
streamer: streamer,
|
||||||
logger: logger,
|
logger: logger,
|
||||||
|
port: cfg.Port,
|
||||||
}
|
}
|
||||||
|
|
||||||
s.setupRoutes()
|
s.setupRoutes()
|
||||||
@@ -52,11 +55,6 @@ func New(db database.Store, streamer *streamer.Streamer, logger *logger.Logger)
|
|||||||
|
|
||||||
// Start starts the HTTP server
|
// Start starts the HTTP server
|
||||||
func (s *Server) Start() error {
|
func (s *Server) Start() error {
|
||||||
port := os.Getenv("PORT")
|
|
||||||
if port == "" {
|
|
||||||
port = "8080"
|
|
||||||
}
|
|
||||||
|
|
||||||
const (
|
const (
|
||||||
readHeaderTimeout = 40 * time.Second
|
readHeaderTimeout = 40 * time.Second
|
||||||
readTimeout = 60 * time.Second
|
readTimeout = 60 * time.Second
|
||||||
@@ -65,7 +63,7 @@ func (s *Server) Start() error {
|
|||||||
)
|
)
|
||||||
|
|
||||||
s.srv = &http.Server{
|
s.srv = &http.Server{
|
||||||
Addr: ":" + port,
|
Addr: ":" + strconv.Itoa(s.port),
|
||||||
Handler: s.router,
|
Handler: s.router,
|
||||||
ReadHeaderTimeout: readHeaderTimeout,
|
ReadHeaderTimeout: readHeaderTimeout,
|
||||||
ReadTimeout: readTimeout,
|
ReadTimeout: readTimeout,
|
||||||
@@ -73,7 +71,7 @@ func (s *Server) Start() error {
|
|||||||
IdleTimeout: idleTimeout,
|
IdleTimeout: idleTimeout,
|
||||||
}
|
}
|
||||||
|
|
||||||
s.logger.Info("Starting HTTP server", "port", port, "addr", s.srv.Addr)
|
s.logger.Info("Starting HTTP server", "port", s.port, "addr", s.srv.Addr)
|
||||||
|
|
||||||
// Start in goroutine but log when actually listening
|
// Start in goroutine but log when actually listening
|
||||||
go func() {
|
go func() {
|
||||||
|
|||||||
@@ -106,6 +106,7 @@ type handlerInfo struct {
|
|||||||
type Streamer struct {
|
type Streamer struct {
|
||||||
logger *logger.Logger
|
logger *logger.Logger
|
||||||
client *http.Client
|
client *http.Client
|
||||||
|
url string
|
||||||
handlers []*handlerInfo
|
handlers []*handlerInfo
|
||||||
rawHandler RawMessageHandler
|
rawHandler RawMessageHandler
|
||||||
mu sync.RWMutex
|
mu sync.RWMutex
|
||||||
@@ -124,6 +125,7 @@ type Streamer struct {
|
|||||||
func New(logger *logger.Logger, metrics *metrics.Tracker) *Streamer {
|
func New(logger *logger.Logger, metrics *metrics.Tracker) *Streamer {
|
||||||
return &Streamer{
|
return &Streamer{
|
||||||
logger: logger,
|
logger: logger,
|
||||||
|
url: risLiveURL,
|
||||||
client: &http.Client{
|
client: &http.Client{
|
||||||
Timeout: 0, // No timeout for streaming
|
Timeout: 0, // No timeout for streaming
|
||||||
Transport: &http.Transport{
|
Transport: &http.Transport{
|
||||||
@@ -208,9 +210,14 @@ func (s *Streamer) Start() error {
|
|||||||
// the connection status in metrics. This method is safe to call multiple times.
|
// the connection status in metrics. This method is safe to call multiple times.
|
||||||
func (s *Streamer) Stop() {
|
func (s *Streamer) Stop() {
|
||||||
s.mu.Lock()
|
s.mu.Lock()
|
||||||
if s.cancel != nil {
|
if s.cancel == nil {
|
||||||
s.cancel()
|
// Not started, or already stopped: closing the queues again would panic.
|
||||||
|
s.mu.Unlock()
|
||||||
|
|
||||||
|
return
|
||||||
}
|
}
|
||||||
|
s.cancel()
|
||||||
|
s.cancel = nil
|
||||||
// Close all handler queues to signal workers to stop
|
// Close all handler queues to signal workers to stop
|
||||||
for _, info := range s.handlers {
|
for _, info := range s.handlers {
|
||||||
close(info.queue)
|
close(info.queue)
|
||||||
@@ -463,7 +470,14 @@ func (s *Streamer) streamWithReconnect(ctx context.Context) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *Streamer) stream(ctx context.Context) error {
|
func (s *Streamer) stream(ctx context.Context) error {
|
||||||
req, err := http.NewRequestWithContext(ctx, "GET", risLiveURL, nil)
|
// connCtx is scoped to this single connection: cancelling it when stream
|
||||||
|
// returns stops the ticker goroutines below, so a reconnect does not leak
|
||||||
|
// them. Without this they would live until the streamer's lifetime context
|
||||||
|
// is cancelled, leaking two per reconnect.
|
||||||
|
connCtx, connCancel := context.WithCancel(ctx)
|
||||||
|
defer connCancel()
|
||||||
|
|
||||||
|
req, err := http.NewRequestWithContext(ctx, "GET", s.url, nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("failed to create request: %w", err)
|
return fmt.Errorf("failed to create request: %w", err)
|
||||||
}
|
}
|
||||||
@@ -516,7 +530,7 @@ func (s *Streamer) stream(ctx context.Context) error {
|
|||||||
select {
|
select {
|
||||||
case <-metricsTicker.C:
|
case <-metricsTicker.C:
|
||||||
s.logMetrics()
|
s.logMetrics()
|
||||||
case <-ctx.Done():
|
case <-connCtx.Done():
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -536,7 +550,7 @@ func (s *Streamer) stream(ctx context.Context) error {
|
|||||||
s.metrics.RecordWireBytes(delta)
|
s.metrics.RecordWireBytes(delta)
|
||||||
lastWireBytes = currentBytes
|
lastWireBytes = currentBytes
|
||||||
}
|
}
|
||||||
case <-ctx.Done():
|
case <-connCtx.Done():
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -651,8 +665,15 @@ func (s *Streamer) stream(ctx context.Context) error {
|
|||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
// Dispatch to interested handlers
|
// Dispatch to interested handlers. Stop cancels ctx and closes the
|
||||||
|
// queues under the write lock, so if ctx is cancelled here, under the
|
||||||
|
// read lock, the queues are closed and must not be sent to.
|
||||||
s.mu.RLock()
|
s.mu.RLock()
|
||||||
|
if ctx.Err() != nil {
|
||||||
|
s.mu.RUnlock()
|
||||||
|
|
||||||
|
return ctx.Err()
|
||||||
|
}
|
||||||
for _, info := range s.handlers {
|
for _, info := range s.handlers {
|
||||||
if !info.handler.WantsMessage(msg.Type) {
|
if !info.handler.WantsMessage(msg.Type) {
|
||||||
continue
|
continue
|
||||||
|
|||||||
@@ -1,10 +1,18 @@
|
|||||||
package streamer
|
package streamer
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"io"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"runtime"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
"git.eeqj.de/sneak/routewatch/internal/logger"
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
||||||
"git.eeqj.de/sneak/routewatch/internal/metrics"
|
"git.eeqj.de/sneak/routewatch/internal/metrics"
|
||||||
|
"git.eeqj.de/sneak/routewatch/internal/ristypes"
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestNewStreamer(t *testing.T) {
|
func TestNewStreamer(t *testing.T) {
|
||||||
@@ -32,3 +40,107 @@ func TestNewStreamer(t *testing.T) {
|
|||||||
t.Error("metrics tracker not set correctly")
|
t.Error("metrics tracker not set correctly")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestStreamDoesNotLeakTickersAcrossReconnects drives many short-lived
|
||||||
|
// connections (each stream call is one reconnect cycle) and asserts the
|
||||||
|
// goroutine count returns to its starting value. Each connection starts two
|
||||||
|
// ticker goroutines; before the fix they lived until the streamer's lifetime
|
||||||
|
// context was cancelled, so every reconnect leaked two.
|
||||||
|
func TestStreamDoesNotLeakTickersAcrossReconnects(t *testing.T) {
|
||||||
|
// The handler returns immediately, so the response body is empty and each
|
||||||
|
// stream call ends at once, standing in for a dropped connection.
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(_ http.ResponseWriter, _ *http.Request) {}))
|
||||||
|
defer srv.Close()
|
||||||
|
|
||||||
|
s := New(logger.New(), metrics.New())
|
||||||
|
s.url = srv.URL
|
||||||
|
|
||||||
|
// One warm-up connection so any persistent HTTP transport goroutine exists
|
||||||
|
// before we take the baseline.
|
||||||
|
if err := s.stream(context.Background()); err != nil {
|
||||||
|
t.Fatalf("warm-up stream returned error: %v", err)
|
||||||
|
}
|
||||||
|
s.client.CloseIdleConnections()
|
||||||
|
|
||||||
|
baseline := settledGoroutineCount()
|
||||||
|
|
||||||
|
const reconnects = 20
|
||||||
|
for range reconnects {
|
||||||
|
if err := s.stream(context.Background()); err != nil {
|
||||||
|
t.Fatalf("stream returned error: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
s.client.CloseIdleConnections()
|
||||||
|
|
||||||
|
if !waitForGoroutines(baseline) {
|
||||||
|
t.Fatalf("goroutines did not return to baseline %d after %d reconnects, got %d",
|
||||||
|
baseline, reconnects, runtime.NumGoroutine())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// updateHandler wants UPDATE messages and does nothing with them.
|
||||||
|
type updateHandler struct{}
|
||||||
|
|
||||||
|
func (updateHandler) WantsMessage(messageType string) bool { return messageType == "UPDATE" }
|
||||||
|
func (updateHandler) HandleMessage(*ristypes.RISMessage) {}
|
||||||
|
func (updateHandler) QueueCapacity() int { return 10 }
|
||||||
|
|
||||||
|
// TestStopBeforeMessageReachesQueues stops the streamer after the read loop
|
||||||
|
// has checked for cancellation but before it hands the message to the handler
|
||||||
|
// queues. That is the gap a stop from another goroutine can land in, and it
|
||||||
|
// used to end in "send on closed channel". The raw handler runs in that gap on
|
||||||
|
// the read loop itself, so calling Stop from it hits the gap every time.
|
||||||
|
func TestStopBeforeMessageReachesQueues(t *testing.T) {
|
||||||
|
const line = `{"type":"ris_message","data":{"type":"UPDATE","peer":"192.0.2.1",` +
|
||||||
|
`"peer_asn":"64496","timestamp":1700000000}}` + "\n"
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||||
|
_, _ = io.WriteString(w, line)
|
||||||
|
}))
|
||||||
|
defer srv.Close()
|
||||||
|
|
||||||
|
s := New(logger.New(), metrics.New())
|
||||||
|
s.url = srv.URL
|
||||||
|
s.RegisterHandler(updateHandler{})
|
||||||
|
s.RegisterRawHandler(func(string) { s.Stop() })
|
||||||
|
|
||||||
|
// Start would run the stream in the background, where the test cannot
|
||||||
|
// wait for it. Setting cancel as Start does lets Stop cancel the stream
|
||||||
|
// run here instead.
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
s.cancel = cancel
|
||||||
|
|
||||||
|
if err := s.stream(ctx); !errors.Is(err, context.Canceled) {
|
||||||
|
t.Fatalf("stream returned %v, want %v", err, context.Canceled)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A second Stop must not close the queues again.
|
||||||
|
s.Stop()
|
||||||
|
}
|
||||||
|
|
||||||
|
// settledGoroutineCount lets transient goroutines finish, then reports the
|
||||||
|
// current count.
|
||||||
|
func settledGoroutineCount() int {
|
||||||
|
prev := runtime.NumGoroutine()
|
||||||
|
for range 20 {
|
||||||
|
time.Sleep(10 * time.Millisecond)
|
||||||
|
cur := runtime.NumGoroutine()
|
||||||
|
if cur == prev {
|
||||||
|
return cur
|
||||||
|
}
|
||||||
|
prev = cur
|
||||||
|
}
|
||||||
|
|
||||||
|
return prev
|
||||||
|
}
|
||||||
|
|
||||||
|
// waitForGoroutines waits until the goroutine count drops to target or below.
|
||||||
|
func waitForGoroutines(target int) bool {
|
||||||
|
for range 100 {
|
||||||
|
if runtime.NumGoroutine() <= target {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
time.Sleep(10 * time.Millisecond)
|
||||||
|
}
|
||||||
|
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|||||||
@@ -7,7 +7,8 @@ package version
|
|||||||
var (
|
var (
|
||||||
// GitRevision is the git commit hash
|
// GitRevision is the git commit hash
|
||||||
GitRevision = "unknown"
|
GitRevision = "unknown"
|
||||||
// GitRevisionShort is the short git commit hash (7 chars)
|
// GitRevisionShort is the version the page footer shows: the tag or
|
||||||
|
// short commit hash from `git describe --tags --always`
|
||||||
GitRevisionShort = "unknown"
|
GitRevisionShort = "unknown"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
+11
-2
@@ -1,7 +1,8 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/docker: build the Docker image tagged with the project name.
|
# script/docker: build the Docker image tagged with the project name.
|
||||||
# Identical in all repos; the tag comes from script/projectname.
|
# Identical in all repos; the tag comes from script/projectname.
|
||||||
# Generic: needs no adaptation.
|
# --no-cache because the gate phases the final stage depends on are RUN
|
||||||
|
# steps, and a cached one is a check that did not run.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||||
@@ -9,7 +10,15 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
|||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
docker build -t "$("$SCRIPT_DIR/projectname")" .
|
# Own line: a failing command substitution inside an argument does
|
||||||
|
# not trip `set -e`, so the inline form degrades silently to an
|
||||||
|
# empty constant. The VERSION build argument takes precedence over
|
||||||
|
# the version a build stage derives from the .git in the context.
|
||||||
|
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
|
||||||
|
[ -n "$version" ] || version="unknown"
|
||||||
|
docker build --no-cache \
|
||||||
|
--build-arg VERSION="$version" \
|
||||||
|
-t "$("$SCRIPT_DIR/projectname")" .
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
+2
-2
@@ -7,9 +7,9 @@ ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
|||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
go test -timeout 30s -race -cover ./... || {
|
go test -short -timeout 30s -race -cover ./... || {
|
||||||
echo "--- Rerunning with -v for details ---"
|
echo "--- Rerunning with -v for details ---"
|
||||||
go test -timeout 30s -race -v ./...
|
go test -short -timeout 30s -race -v ./...
|
||||||
exit 1
|
exit 1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user