fix(server): shut down through fx so buffered reports flush (closes #22)
check / check (push) Successful in 1m11s

The server ran os.Exit at the end of its own goroutine, racing fx's
teardown and sometimes killing the process before reportbuf's OnStop
flushed — silently losing a full flush window of telemetry on every
restart, at exit 0. Shutdown now goes through fx.Shutdowner, so every
OnStop runs in order.

The http.Server is built synchronously in OnStart before the serving
goroutine, so shutdown can no longer race or nil-deref it. A listen
failure exits non-zero via fx.ExitCode(1). reportbuf's OnStop is guarded
by sync.Once. writeTimeout now exceeds the chi per-request budget so that
budget is reachable. Dead startupTime, exitCode, and cancelFunc fields
are gone. A new test asserts a buffered report reaches disk after the
lifecycle stops.

Model: opus-4-8
This commit was merged in pull request #57.
This commit is contained in:
2026-09-21 18:47:12 +02:00
parent 14eb376d79
commit d7cf010e00
5 changed files with 154 additions and 100 deletions
+13 -7
View File
@@ -40,11 +40,12 @@ type Params struct {
// Buffer accumulates JSON lines in memory and flushes them
// to zstd-compressed files on disk.
type Buffer struct {
buf bytes.Buffer
dataDir string
done chan struct{}
log *slog.Logger
mu sync.Mutex
buf bytes.Buffer
dataDir string
done chan struct{}
log *slog.Logger
mu sync.Mutex
stopOnce sync.Once
}
// New creates a Buffer and registers lifecycle hooks to
@@ -76,8 +77,13 @@ func New(
return nil
},
OnStop: func(_ context.Context) error {
close(b.done)
b.flushLocked()
// stopOnce makes OnStop idempotent: a second
// invocation must not close an already-closed channel
// (which would panic) or flush again.
b.stopOnce.Do(func() {
close(b.done)
b.flushLocked()
})
return nil
},
+69 -5
View File
@@ -1,13 +1,77 @@
package reportbuf_test
import (
"os"
"strings"
"testing"
_ "sneak.berlin/go/netwatch/internal/reportbuf"
"sneak.berlin/go/netwatch/internal/config"
"sneak.berlin/go/netwatch/internal/globals"
"sneak.berlin/go/netwatch/internal/logger"
"sneak.berlin/go/netwatch/internal/reportbuf"
"go.uber.org/fx"
"go.uber.org/fx/fxtest"
)
func TestImport(t *testing.T) {
t.Parallel()
// Compilation check — verifies the package parses
// and all imports resolve.
// TestFlushOnShutdown proves the flush-on-shutdown path: a
// report appended after start but before the periodic flush
// window must reach disk when the fx lifecycle stops. This is
// the exact case that silent data loss on restart used to
// destroy.
func TestFlushOnShutdown(t *testing.T) {
dir := t.TempDir()
t.Setenv("DATA_DIR", dir)
var buf *reportbuf.Buffer
app := fxtest.New(t,
fx.Provide(
globals.New,
logger.New,
config.New,
reportbuf.New,
),
fx.Populate(&buf),
)
app.RequireStart()
err := buf.Append(map[string]string{"probe": "shutdown"})
if err != nil {
t.Fatalf("append report: %v", err)
}
// RequireStop runs the reportbuf OnStop hook, which is the
// only code path that flushes buffered reports on shutdown.
app.RequireStop()
if !hasReportFile(t, dir) {
t.Fatal("no report file on disk after shutdown; " +
"the buffered report was lost")
}
}
func hasReportFile(t *testing.T, dir string) bool {
t.Helper()
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatalf("read data dir: %v", err)
}
for _, e := range entries {
if strings.HasSuffix(e.Name(), ".jsonl.zst") {
info, statErr := e.Info()
if statErr != nil {
t.Fatalf("stat %s: %v", e.Name(), statErr)
}
if info.Size() > 0 {
return true
}
}
}
return false
}