fix(server): shut down through fx so buffered reports flush (closes #22)
check / check (push) Failing after 0s

The server ran os.Exit at the end of its own goroutine, which raced fx's
teardown and could kill the process before reportbuf's OnStop flushed the
buffer — losing up to a full flush window of telemetry on every restart,
silently and with exit 0. The server now requests shutdown through
fx.Shutdowner, so fx runs every OnStop in dependency order.

The http.Server is built synchronously in OnStart before the serving
goroutine starts, so shutdown can no longer race or nil-deref it; the
field is never written and read from two goroutines without a
happens-before edge. A listen failure now exits non-zero via
fx.ExitCode(1). reportbuf's OnStop is guarded by sync.Once. writeTimeout
now exceeds the chi per-request budget, with a comment, so that budget is
reachable. Dead startupTime, exitCode, and cancelFunc fields are gone.

A new test buffers a report and asserts it reaches disk after the fx
lifecycle stops.

Model: opus-4-8
This commit is contained in:
2026-09-21 12:49:59 +00:00
parent f7c7f92e27
commit dd762ce759
5 changed files with 154 additions and 100 deletions
+13 -7
View File
@@ -40,11 +40,12 @@ type Params struct {
// Buffer accumulates JSON lines in memory and flushes them
// to zstd-compressed files on disk.
type Buffer struct {
buf bytes.Buffer
dataDir string
done chan struct{}
log *slog.Logger
mu sync.Mutex
buf bytes.Buffer
dataDir string
done chan struct{}
log *slog.Logger
mu sync.Mutex
stopOnce sync.Once
}
// New creates a Buffer and registers lifecycle hooks to
@@ -76,8 +77,13 @@ func New(
return nil
},
OnStop: func(_ context.Context) error {
close(b.done)
b.flushLocked()
// stopOnce makes OnStop idempotent: a second
// invocation must not close an already-closed channel
// (which would panic) or flush again.
b.stopOnce.Do(func() {
close(b.done)
b.flushLocked()
})
return nil
},
+69 -5
View File
@@ -1,13 +1,77 @@
package reportbuf_test
import (
"os"
"strings"
"testing"
_ "sneak.berlin/go/netwatch/internal/reportbuf"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/reportbuf"
"go.uber.org/fx"
"go.uber.org/fx/fxtest"
)
func TestImport(t *testing.T) {
t.Parallel()
// Compilation check — verifies the package parses
// and all imports resolve.
// TestFlushOnShutdown proves the flush-on-shutdown path: a
// report appended after start but before the periodic flush
// window must reach disk when the fx lifecycle stops. This is
// the exact case that silent data loss on restart used to
// destroy.
func TestFlushOnShutdown(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
var buf *reportbuf.Buffer
app := fxtest.New(t,
fx.Provide(
globals.New,
logger.New,
config.New,
reportbuf.New,
),
fx.Populate(&buf),
)
app.RequireStart()
err := buf.Append(map[string]string{"probe": "shutdown"})
if err != nil {
t.Fatalf("append report: %v", err)
}
// RequireStop runs the reportbuf OnStop hook, which is the
// only code path that flushes buffered reports on shutdown.
app.RequireStop()
if !hasReportFile(t, dir) {
t.Fatal("no report file on disk after shutdown; " +
"the buffered report was lost")
}
}
func hasReportFile(t *testing.T, dir string) bool {
t.Helper()
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatalf("read data dir: %v", err)
}
for _, e := range entries {
if strings.HasSuffix(e.Name(), ".jsonl.zst") {
info, statErr := e.Info()
if statErr != nil {
t.Fatalf("stat %s: %v", e.Name(), statErr)
}
if info.Size() > 0 {
return true
}
}
}
return false
}