2 Commits
Author SHA1 Message Date
clawbot 27c4b84c03 watcher: notify NS query failure and recovery (closes #104)
check / check (push) Successful in 1m21s
LookupAllRecords now returns each nameserver's response, so the
watcher saves its status: ok when it answered, NXDOMAIN and no records
included, and error with the reason when it timed out, answered
SERVFAIL or REFUSED, or could not be reached. A nameserver that starts
failing sends NS Failure and one that answers again sends NS Recovery.
A failing nameserver is left out of the record change and
inconsistency comparisons. The resolver used to report REFUSED and
network errors as an answer with no records; they are now errors. A
lookup cut short by its context now returns an error instead of a
failure of the nameserver it was querying.

Model: opus-5-5
2026-10-01 18:47:17 +00:00
clawbot 5493e28480 notify: make the shutdown tests fail with the right message (closes #116)
check / check (push) Successful in 1m27s
drainSlack stood for three things: the deadline given to a drain that
should finish early, the watchdog on a drain that should time out, and
the wait for a delivery to reach the test server. It is now three
constants, each commented with what it bounds and why it is 2s; no
value changed. The idle-drain failure printed that deadline instead of
idleDrainBound, the bound it checks. The cancelled-context test now
also requires the drain to return within idleDrainBound and to log its
debug line, so a drain that logs nothing no longer passes;
newLoggingService records debug level for this.

Model: opus-5-5
2026-10-01 20:42:21 +02:00
3 changed files with 148 additions and 31 deletions
+2
View File
@@ -21,6 +21,8 @@ Rationale, Design, TODO, License, Author) if any are still missing.
- 2026-10-01: a nameserver that does not answer is saved as `error` with the
reason, and NS failure and NS recovery are notified (closes #104).
- 2026-10-01: notify shutdown tests use one timing constant per meaning, name
the bound they check, and require the drain's debug line (closes #116).
- 2026-09-29: the live-DNS test package is renamed `internal/livednstest` and
added to the `test-support` `deny` list in `.golangci.yml`, so `make lint`
fails when program code imports it (closes #164).
+83 -28
View File
@@ -33,10 +33,29 @@ const (
// out.
drainDeadline = 50 * time.Millisecond
// drainSlack is the upper bound on how long a bounded
// drain may take; generous enough for a loaded CI box,
// still far below the 20s test ceiling.
drainSlack = 2 * time.Second
// timeoutDrainBound is how long a drain given drainDeadline
// may take to return before the test gives up on it. At
// forty times drainDeadline it leaves ample room for
// scheduling delay on a loaded box under -race, yet it is far
// below the test binary's -timeout, so a drain that its
// deadline does not bound fails that one test instead of
// hanging the package.
timeoutDrainBound = 2 * time.Second
// longDrainDeadline is the deadline given to a drain that is
// expected to finish well before it: when the in-flight
// delivery completes after inFlightHold, or at once when
// nothing is in flight. It is far above inFlightHold, so
// those drains never reach it, and four times
// idleDrainBound, so an idle drain that waited for its
// deadline instead of returning fails that bound.
longDrainDeadline = 2 * time.Second
// reachEndpointTimeout is how long a submitted delivery may
// take to reach the test server. That normally takes a few
// milliseconds; the margin is for a loaded box under -race,
// and only a failing run ever waits this long.
reachEndpointTimeout = 2 * time.Second
// settleDelay is how long to wait before asserting that
// something did *not* happen.
@@ -46,10 +65,11 @@ const (
// nothing in flight. It is deliberately far above the cost
// of the goroutine hop through inFlight.Wait() — which
// reached 57ms on a loaded box under -race with the package's
// parallel tests — and far below drainSlack, the deadline
// such a drain is given. A drain that blocked until its
// deadline instead of returning on the WaitGroup therefore
// still fails this bound, but scheduling delay alone cannot.
// parallel tests — and far below longDrainDeadline, the
// deadline such a drain is given. A drain that blocked until
// its deadline instead of returning on the WaitGroup
// therefore still fails this bound, but scheduling delay
// alone cannot.
idleDrainBound = 500 * time.Millisecond
)
@@ -75,12 +95,14 @@ func (sb *syncBuffer) String() string {
}
// newLoggingService returns a Service writing JSON logs into
// the returned buffer.
// the returned buffer, debug level included.
func newLoggingService(
transport http.RoundTripper,
) (*notify.Service, *syncBuffer) {
logs := &syncBuffer{}
handler := slog.NewJSONHandler(logs, nil)
handler := slog.NewJSONHandler(
logs, &slog.HandlerOptions{Level: slog.LevelDebug},
)
return notify.NewTestServiceWithLogger(transport, handler),
logs
@@ -134,7 +156,7 @@ func TestDrainWaitsForInFlightDelivery(t *testing.T) {
// drain begins.
select {
case <-entered:
case <-time.After(drainSlack):
case <-time.After(reachEndpointTimeout):
t.Fatal("delivery never reached the endpoint")
}
@@ -151,7 +173,7 @@ func TestDrainWaitsForInFlightDelivery(t *testing.T) {
defer timer.Stop()
ctx, cancel := context.WithTimeout(
context.Background(), drainSlack,
context.Background(), longDrainDeadline,
)
defer cancel()
@@ -257,11 +279,11 @@ func TestDrainBoundedByContextDeadline(t *testing.T) {
select {
case <-returned:
case <-time.After(drainSlack):
case <-time.After(timeoutDrainBound):
t.Fatalf(
"drain did not return within %v; its %v deadline "+
"did not bound it",
drainSlack, drainDeadline,
timeoutDrainBound, drainDeadline,
)
}
@@ -333,7 +355,7 @@ func TestDrainRefusesNewDeliveries(t *testing.T) {
svc.SetMattermostWebhookURL(target)
ctx, cancel := context.WithTimeout(
context.Background(), drainSlack,
context.Background(), longDrainDeadline,
)
defer cancel()
@@ -446,7 +468,7 @@ func TestNewRegistersDrainingStopHook(t *testing.T) {
select {
case <-entered:
case <-time.After(drainSlack):
case <-time.After(reachEndpointTimeout):
t.Fatal("delivery never reached the endpoint")
}
@@ -456,7 +478,7 @@ func TestNewRegistersDrainingStopHook(t *testing.T) {
defer timer.Stop()
ctx, cancel := context.WithTimeout(
context.Background(), drainSlack,
context.Background(), longDrainDeadline,
)
defer cancel()
@@ -487,7 +509,7 @@ func TestDrainWithoutDeliveriesReturnsImmediately(t *testing.T) {
start := time.Now()
ctx, cancel := context.WithTimeout(
context.Background(), drainSlack,
context.Background(), longDrainDeadline,
)
defer cancel()
@@ -495,9 +517,9 @@ func TestDrainWithoutDeliveriesReturnsImmediately(t *testing.T) {
if elapsed := time.Since(start); elapsed > idleDrainBound {
t.Errorf(
"drain of an idle service took %v, want well "+
"under its %v deadline",
elapsed, drainSlack,
"drain of an idle service took %v, want at most "+
"%v; its deadline was %v",
elapsed, idleDrainBound, longDrainDeadline,
)
}
}
@@ -505,10 +527,11 @@ func TestDrainWithoutDeliveriesReturnsImmediately(t *testing.T) {
// TestDrainWithCancelledContextDoesNotWarn verifies that an
// OnStop context that is already dead on entry does not produce
// an "abandoning them" warning when there was nothing in flight
// to abandon. The expired context wins the select immediately,
// so only the outstanding count can tell the difference between
// a genuine timeout and a shutdown that had simply already run
// out of time with no work left.
// to abandon, and that the drain returns and says at debug level
// that nothing was in flight. The expired context wins the
// select immediately, so only the outstanding count can tell the
// difference between a genuine timeout and a shutdown that had
// simply already run out of time with no work left.
func TestDrainWithCancelledContextDoesNotWarn(t *testing.T) {
t.Parallel()
@@ -517,11 +540,43 @@ func TestDrainWithCancelledContextDoesNotWarn(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
svc.Drain(ctx)
// A watchdog, as in TestDrainBoundedByContextDeadline, so
// that a drain which never returns fails here instead of
// hanging the package.
returned := make(chan struct{})
if output := logs.String(); strings.Contains(
output, `"level":"WARN"`,
go func() {
defer close(returned)
svc.Drain(ctx)
}()
select {
case <-returned:
case <-time.After(idleDrainBound):
t.Fatalf(
"drain with nothing in flight and a cancelled "+
"context did not return within %v",
idleDrainBound,
)
}
output := logs.String()
// The absence of a warning alone would also pass if the drain
// logged nothing at all, so require the debug line it writes
// when it finds nothing outstanding.
if !strings.Contains(
output, "all in-flight notifications completed",
) {
t.Errorf(
"drain did not log that nothing was in flight; "+
"log output: %s",
output,
)
}
if strings.Contains(output, `"level":"WARN"`) {
t.Errorf(
"drain with nothing in flight warned about "+
"abandoned deliveries; log output: %s",
+63 -3
View File
@@ -2,11 +2,13 @@ package watcher_test
import (
"context"
"fmt"
"log/slog"
"strings"
"testing"
"time"
"sneak.berlin/go/dnswatcher/internal/livednstest"
"sneak.berlin/go/dnswatcher/internal/resolver"
"sneak.berlin/go/dnswatcher/internal/state"
"sneak.berlin/go/dnswatcher/internal/watcher"
@@ -152,7 +154,7 @@ func TestNSFailureAndRecoveryAlerts(t *testing.T) {
}
}
func TestNSFailureAlertNamesNameserverAndReason(t *testing.T) {
func TestNSFailureAlertNamesHostnameNameserverAndReason(t *testing.T) {
t.Parallel()
records := map[string][]string{"A": {ip1}}
@@ -172,9 +174,12 @@ func TestNSFailureAlertNamesNameserverAndReason(t *testing.T) {
}
msg := notifications[0].Message
if !strings.Contains(msg, nsA) ||
if !strings.Contains(msg, host) || !strings.Contains(msg, nsA) ||
!strings.Contains(msg, failed().Error) {
t.Errorf("message %q does not name %s and the reason", msg, nsA)
t.Errorf(
"message %q does not name %s, %s and the reason",
msg, host, nsA,
)
}
}
@@ -207,3 +212,58 @@ func TestNameserverThatNeverAnswers(t *testing.T) {
)
}
}
// TestNameserverThatAnswersNXDOMAIN asks a real nameserver about a name
// that does not exist and checks what the watcher saves for it: NXDOMAIN
// is an answer, so the nameserver is saved as ok with no error.
func TestNameserverThatAnswersNXDOMAIN(t *testing.T) {
t.Parallel()
res := resolver.NewFromLogger(slog.Default())
name := "this-surely-does-not-exist-xyz." + testDomain
var (
ns string
resp *resolver.NameserverResponse
)
livednstest.Retry(t, "QueryNameserver("+name+")", func(ctx context.Context) error {
nameservers, err := res.LookupNS(ctx, testDomain)
if err != nil {
return err
}
ns = nameservers[0]
resp, err = res.QueryNameserver(ctx, ns, name)
if err != nil {
return err
}
// A timeout or a failure is no answer to check.
if resp.Status == resolver.StatusTimeout ||
resp.Status == resolver.StatusError {
return fmt.Errorf(
"%w: %s: %s", livednstest.ErrNoAnswer, ns, resp.Error,
)
}
return nil
})
if resp.Status != resolver.StatusNXDomain {
t.Fatalf("%s answered %q for %s, want NXDOMAIN", ns, resp.Status, name)
}
hs := watcher.BuildHostnameState(
map[string]*resolver.NameserverResponse{ns: resp}, time.Now(),
)
got := hs.RecordsByNameserver[ns]
if got.Status != "ok" || got.Error != "" {
t.Errorf(
"saved status %q, error %q; want status ok with no error",
got.Status, got.Error,
)
}
}