Clamp the HTTP drain by the tail-hook reserve (closes #170)
check / check (push) Successful in 3m21s

The server's stop hook bounded the drain by ShutdownTimeout alone, so
once the archive sweeper or retention reaper had spent part of the fx
stop budget, a request held open could use up the reserve and fx
skipped every hook after the server, the database close included.
The drain now gets the shorter of ShutdownTimeout and what is left
less TailHookReserve, the clamp the Sentry flush already has.

The headroom test also sweeps the time earlier hooks spent; a new
test holds a request open against a stop context with only the
reserve left. The README and the reserve's comment say how the
reserve is derived and what a slow sweeper now costs.

Model: opus-5-5
This commit is contained in:
2026-10-02 16:05:09 +00:00
parent bf3df0312b
commit 6aa9907c8e
6 changed files with 180 additions and 36 deletions
+10
View File
@@ -1,6 +1,8 @@
package server
import (
"context"
"log/slog"
"net/http"
"testing"
@@ -37,6 +39,14 @@ func SentryClientOptionsForTest(
return sentryClientOptions(dsn, release)
}
// CleanShutdownForTest runs the server's stop hook, cleanShutdown,
// against hs: a server the test started itself, so it can hold a
// request open across the drain. Sentry is off.
func CleanShutdownForTest(ctx context.Context, hs *http.Server) {
s := &Server{log: slog.New(slog.DiscardHandler), httpServer: hs}
s.cleanShutdown(ctx)
}
// newServerForTest builds a Server through New, as the application
// does, on a lifecycle that is never started: the hooks New adds to
// it never run, so nothing listens.
+26 -3
View File
@@ -39,6 +39,12 @@ const (
// refuses to spend, leaving it for the hooks that run after the
// server: the delivery engine, the healthcheck, the webhook DB
// manager and the database close.
//
// Its value is not tuned to those hooks, which take about a
// millisecond between them. It is what the 5s fx stop timeout in
// cmd/webhooker leaves after a full ShutdownTimeout drain, so a
// drain that starts on a full budget still gets all of
// ShutdownTimeout.
TailHookReserve = 2 * time.Second
// sentryFlushTimeout is the longest wait for Sentry to flush
@@ -59,6 +65,16 @@ const (
// key off it, and a zero exit would read as a deliberate stop.
const StartupFailureExitCode = 1
// DrainBudget reports how long the HTTP drain may wait for in-flight
// requests when remaining is the time left on the fx stop context as
// the server's stop hook starts. The hooks before the server can
// already have spent part of the budget, so the drain takes its time
// out of what they left, never out of TailHookReserve. Zero or less
// means no wait at all.
func DrainBudget(remaining time.Duration) time.Duration {
return min(ShutdownTimeout, remaining-TailHookReserve)
}
// SentryFlushBudget reports how long the Sentry flush may run when
// remaining is the time left on the fx stop context after the HTTP
// drain. sentry.Flush takes a bare duration and honours no context,
@@ -261,10 +277,17 @@ func (s *Server) cleanupForExit() {
s.log.Info("cleaning up")
}
// cleanShutdown drains the HTTP server and flushes Sentry inside what
// is left of the fx stop budget. A context carrying no deadline — a
// caller outside the fx lifecycle — gets the full ShutdownTimeout.
func (s *Server) cleanShutdown(ctx context.Context) {
ctxShutdown, shutdownCancel := context.WithTimeout(
ctx, ShutdownTimeout,
)
drain := ShutdownTimeout
if deadline, ok := ctx.Deadline(); ok {
drain = DrainBudget(time.Until(deadline))
}
ctxShutdown, shutdownCancel := context.WithTimeout(ctx, drain)
defer shutdownCancel()
err := s.httpServer.Shutdown(ctxShutdown)
+90
View File
@@ -1,6 +1,9 @@
package server_test
import (
"context"
"net/http"
"net/http/httptest"
"testing"
"time"
@@ -8,6 +11,93 @@ import (
"sneak.berlin/go/webhooker/internal/server"
)
// TestDrainBudget covers the clamp that keeps the HTTP drain from
// spending the tail hooks' share of the fx stop budget when the hooks
// before the server have already used part of it.
func TestDrainBudget(t *testing.T) {
t.Parallel()
tests := []struct {
name string
remaining time.Duration
want time.Duration
}{
{
name: "only the reserve is left",
remaining: server.TailHookReserve,
want: 0,
},
{
name: "earlier hooks spent part of the budget",
remaining: server.TailHookReserve + time.Second,
want: time.Second,
},
{
name: "capped at the nominal timeout",
remaining: time.Hour,
want: server.ShutdownTimeout,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
require.Equal(t, tt.want, server.DrainBudget(tt.remaining))
})
}
}
// TestCleanShutdown_LeavesTailHookReserve stops the server with a
// request still in flight, after the hooks before it have spent all
// of the stop budget but TailHookReserve. The drain must give up at
// once rather than wait for the request: what is left belongs to the
// hooks after the server, the database close among them. A drain
// bounded only by ShutdownTimeout waits until the stop context
// expires, and fx then skips those hooks.
func TestCleanShutdown_LeavesTailHookReserve(t *testing.T) {
t.Parallel()
entered := make(chan struct{})
release := make(chan struct{})
ts := httptest.NewServer(http.HandlerFunc(
func(http.ResponseWriter, *http.Request) {
close(entered)
<-release
},
))
// Cleanups run last first: the request is released before Close,
// which waits for it.
t.Cleanup(ts.Close)
t.Cleanup(func() { close(release) })
req, err := http.NewRequestWithContext(
t.Context(), http.MethodGet, ts.URL, nil,
)
require.NoError(t, err)
go func() {
resp, err := ts.Client().Do(req)
if err == nil {
_ = resp.Body.Close()
}
}()
<-entered
stopCtx, cancel := context.WithTimeout(
t.Context(), server.TailHookReserve,
)
defer cancel()
server.CleanShutdownForTest(stopCtx, ts.Config)
require.NoError(
t, stopCtx.Err(), "the drain spent the tail hooks' reserve",
)
}
// TestSentryFlushBudget covers the clamp that keeps the Sentry flush
// from spending the tail hooks' share of the fx stop budget.
// sentry.Flush ignores the stop context, so without the clamp a