Clamp the HTTP drain by the tail-hook reserve (closes #170)
check / check (push) Waiting to run

The HTTP drain at shutdown waited up to ShutdownTimeout regardless of how much of the stop budget earlier hooks had used, so a slow archive sweeper or retention reaper could eat the reserve the hooks after the server need, and the database close was skipped. The drain now waits at most the shorter of ShutdownTimeout and what is left of the budget less TailHookReserve, as the Sentry flush already does. The reserve is documented as derived from the two timeouts. Tests cover earlier hooks having spent part of the budget, on a clock that host speed cannot move, and pin that a drain on the full budget gets all of ShutdownTimeout.

Model: opus-5-5
This commit was merged in pull request #454.
This commit is contained in:
2026-10-02 20:11:43 +02:00
parent f82b730c31
commit 0945831442
6 changed files with 231 additions and 36 deletions
+10
View File
@@ -1,6 +1,8 @@
package server
import (
"context"
"log/slog"
"net/http"
"testing"
@@ -37,6 +39,14 @@ func SentryClientOptionsForTest(
return sentryClientOptions(dsn, release)
}
// CleanShutdownForTest runs the server's stop hook, cleanShutdown,
// against hs: a server the test started itself, so it can hold a
// request open across the drain. Sentry is off.
func CleanShutdownForTest(ctx context.Context, hs *http.Server) {
s := &Server{log: slog.New(slog.DiscardHandler), httpServer: hs}
s.cleanShutdown(ctx)
}
// newServerForTest builds a Server through New, as the application
// does, on a lifecycle that is never started: the hooks New adds to
// it never run, so nothing listens.
+26 -3
View File
@@ -39,6 +39,12 @@ const (
// refuses to spend, leaving it for the hooks that run after the
// server: the delivery engine, the healthcheck, the webhook DB
// manager and the database close.
//
// Its value is not tuned to those hooks, which take about a
// millisecond between them. It is what the 5s fx stop timeout in
// cmd/webhooker leaves after a full ShutdownTimeout drain, so a
// drain that starts on a full budget still gets all of
// ShutdownTimeout.
TailHookReserve = 2 * time.Second
// sentryFlushTimeout is the longest wait for Sentry to flush
@@ -59,6 +65,16 @@ const (
// key off it, and a zero exit would read as a deliberate stop.
const StartupFailureExitCode = 1
// DrainBudget reports how long the HTTP drain may wait for in-flight
// requests when remaining is the time left on the fx stop context as
// the server's stop hook starts. The hooks before the server can
// already have spent part of the budget, so the drain takes its time
// out of what they left, never out of TailHookReserve. Zero or less
// means no wait at all.
func DrainBudget(remaining time.Duration) time.Duration {
return min(ShutdownTimeout, remaining-TailHookReserve)
}
// SentryFlushBudget reports how long the Sentry flush may run when
// remaining is the time left on the fx stop context after the HTTP
// drain. sentry.Flush takes a bare duration and honours no context,
@@ -261,10 +277,17 @@ func (s *Server) cleanupForExit() {
s.log.Info("cleaning up")
}
// cleanShutdown drains the HTTP server and flushes Sentry inside what
// is left of the fx stop budget. A context carrying no deadline — a
// caller outside the fx lifecycle — gets the full ShutdownTimeout.
func (s *Server) cleanShutdown(ctx context.Context) {
ctxShutdown, shutdownCancel := context.WithTimeout(
ctx, ShutdownTimeout,
)
drain := ShutdownTimeout
if deadline, ok := ctx.Deadline(); ok {
drain = DrainBudget(time.Until(deadline))
}
ctxShutdown, shutdownCancel := context.WithTimeout(ctx, drain)
defer shutdownCancel()
err := s.httpServer.Shutdown(ctxShutdown)
+135
View File
@@ -1,13 +1,148 @@
package server_test
import (
"context"
"net"
"net/http"
"testing"
"testing/synctest"
"time"
"github.com/stretchr/testify/require"
"sneak.berlin/go/webhooker/internal/server"
)
// TestDrainBudget covers the clamp that keeps the HTTP drain from
// spending the tail hooks' share of the fx stop budget when the hooks
// before the server have already used part of it.
func TestDrainBudget(t *testing.T) {
t.Parallel()
tests := []struct {
name string
remaining time.Duration
want time.Duration
}{
{
name: "only the reserve is left",
remaining: server.TailHookReserve,
want: 0,
},
{
name: "earlier hooks spent part of the budget",
remaining: server.TailHookReserve + time.Second,
want: time.Second,
},
{
name: "capped at the nominal timeout",
remaining: time.Hour,
want: server.ShutdownTimeout,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
require.Equal(t, tt.want, server.DrainBudget(tt.remaining))
})
}
}
// TestCleanShutdown_LeavesTailHookReserve stops the server with a
// request still in flight, after the hooks before it have spent all
// of the stop budget but TailHookReserve. The drain must give up at
// once rather than wait for the request: what is left belongs to the
// hooks after the server, the database close among them. A drain
// bounded only by ShutdownTimeout waits until the stop context
// expires, and fx then skips those hooks.
//
// The test runs in a synctest bubble, whose clock moves only while
// every goroutine in it is blocked, so a drain that gives up at once
// leaves the stop context unexpired however slow the host is. The
// request travels over net.Pipe because a goroutine waiting on a
// real socket would stop that clock from moving at all.
func TestCleanShutdown_LeavesTailHookReserve(t *testing.T) {
t.Parallel()
synctest.Test(t, func(t *testing.T) {
entered := make(chan struct{})
release := make(chan struct{})
hs := &http.Server{
Handler: http.HandlerFunc(
func(http.ResponseWriter, *http.Request) {
close(entered)
<-release
},
),
ReadHeaderTimeout: time.Second,
}
srvConn, cliConn := net.Pipe()
listener := pipeListener{
conns: make(chan net.Conn, 1),
closed: make(chan struct{}),
}
listener.conns <- srvConn
go func() { _ = hs.Serve(listener) }()
// Cleanups run last first: the handler returns, then closing
// the client end ends the server's write of the response.
t.Cleanup(func() { _ = cliConn.Close() })
t.Cleanup(func() { close(release) })
_, err := cliConn.Write(
[]byte("GET / HTTP/1.1\r\nHost: webhooker.test\r\n\r\n"),
)
require.NoError(t, err)
<-entered
stopCtx, cancel := context.WithTimeout(
t.Context(), server.TailHookReserve,
)
defer cancel()
server.CleanShutdownForTest(stopCtx, hs)
require.NoError(
t, stopCtx.Err(), "the drain spent the tail hooks' reserve",
)
})
}
// pipeListener is the net.Listener http.Server.Serve needs to serve
// the server end of a net.Pipe: Accept returns that one connection,
// then waits until Close, as a real listener with no more clients
// does.
type pipeListener struct {
conns chan net.Conn
closed chan struct{}
}
func (l pipeListener) Accept() (net.Conn, error) {
select {
case conn := <-l.conns:
return conn, nil
case <-l.closed:
return nil, net.ErrClosed
}
}
func (l pipeListener) Close() error {
close(l.closed)
return nil
}
// Addr is never called by http.Server.Serve.
func (pipeListener) Addr() net.Addr {
return nil
}
// TestSentryFlushBudget covers the clamp that keeps the Sentry flush
// from spending the tail hooks' share of the fx stop budget.
// sentry.Flush ignores the stop context, so without the clamp a