Set fx.StopTimeout inside the container stop grace (closes #134)
All checks were successful
check / check (push) Successful in 3m3s

fx defaults to a 15s stop timeout and the Dockerfile sets no grace
override, so Docker SIGKILLed at 10s and the bounded shutdown #130 built
— including the log line that tells an operator a component is wedged —
was unreachable in the image this repo produces.

Sets fx.StopTimeout to 5s, and lowers the HTTP drain to 3s so a
full-length drain no longer exhausts the whole sequence budget and skip
every later hook, database close included. The Sentry flush, which runs
in the same hook and honours no context, is clamped to the remaining
stop budget less a 2s tail reserve, so a stalled flush drops Sentry
events rather than the database close.

Also fixes a latent coin flip in the shared stop-hook waiter, which
reported "shutdown timed out" about half the time for a component that
drained cleanly against an already-expired context.

Independently reviewed three times. The final reviewer derived a
stronger invariant than the implementation claims — the server hook's
absolute end is bounded at stopTimeout minus the reserve regardless of
drain length or of time consumed by preceding hooks — and confirmed the
guard's 10ms sweep cannot step over the maximum, since both breakpoints
land on its grid. Both Sentry probe arms, the docker stop demo and every
mutation were reproduced independently.

Known residual, filed separately: the HTTP drain itself is not clamped
by the reserve, so slow preceding hooks can still jointly exhaust the
budget. Demonstrated with a 2.2s sweeper delay.
This commit was merged in pull request #159.
This commit is contained in:
2026-08-18 00:12:51 +02:00
parent c3b6623be1
commit bef9986542
8 changed files with 401 additions and 10 deletions

View File

@@ -0,0 +1,21 @@
package lifecycle
import (
"context"
"log/slog"
)
// WaitDone exposes waitDone to the external test package. Only the
// unexported waiter can be handed a channel that is already closed
// before the call, which is the state the preamble exists for;
// through WaitForShutdown the waiter goroutine may or may not have
// closed the channel yet, so the case is not reachable
// deterministically from outside.
func WaitDone(
ctx context.Context,
log *slog.Logger,
component string,
done <-chan struct{},
) error {
return waitDone(ctx, log, component, done)
}

View File

@@ -38,6 +38,29 @@ func WaitForShutdown(
wg.Wait()
}()
return waitDone(ctx, log, component, done)
}
// waitDone waits for done to close, bounded by ctx.
//
// The non-blocking preamble is load-bearing. When the component has
// already drained and ctx has already expired, both cases of the
// bounded select are ready and Go picks between them uniformly at
// random, so a clean shutdown would be reported as a timeout about
// half the time. Draining wins: the goroutines are gone, and there
// is nothing left for the operator to act on.
func waitDone(
ctx context.Context,
log *slog.Logger,
component string,
done <-chan struct{},
) error {
select {
case <-done:
return nil
default:
}
select {
case <-done:
return nil

View File

@@ -37,6 +37,57 @@ func TestWaitForShutdown_DrainedGroup(t *testing.T) {
)
}
// racePasses is how many times the both-cases-ready race is run.
// Without the preamble each pass is an independent coin flip, so
// the probability of the whole loop passing by luck is 2^-N: at
// this N the test is deterministic in practice, and it involves no
// wall-clock waiting at all.
const racePasses = 1000
// TestWaitDone_DrainedBeforeExpiredContext covers the case where a
// component drained cleanly but the stop context had already
// expired. Both select cases are ready, and Go chooses among ready
// cases uniformly at random, so the drained case must be settled by
// the preamble before the bounded select ever runs.
func TestWaitDone_DrainedBeforeExpiredContext(t *testing.T) {
t.Parallel()
done := make(chan struct{})
close(done)
ctx, cancel := context.WithCancel(context.Background())
cancel()
for pass := range racePasses {
require.NoErrorf(
t,
lifecycle.WaitDone(
ctx, discardLogger(), "test component", done,
),
"pass %d reported a timeout for a drained component",
pass,
)
}
}
// TestWaitDone_ExpiredContext pins the other side of the preamble:
// an expired context with a component that has not drained is still
// a timeout.
func TestWaitDone_ExpiredContext(t *testing.T) {
t.Parallel()
ctx, cancel := context.WithCancel(context.Background())
cancel()
err := lifecycle.WaitDone(
ctx, discardLogger(), "test component",
make(chan struct{}),
)
require.ErrorIs(t, err, context.Canceled)
require.ErrorContains(t, err, "test component")
}
func TestWaitForShutdown_ContextExpires(t *testing.T) {
t.Parallel()

View File

@@ -24,15 +24,48 @@ import (
)
const (
// shutdownTimeout is the maximum time to wait for the HTTP
// ShutdownTimeout is the maximum time to wait for the HTTP
// server to finish in-flight requests during shutdown.
shutdownTimeout = 5 * time.Second
//
// It must stay strictly below the fx stop timeout in
// cmd/webhooker, which bounds the whole stop sequence: a drain
// that used the entire sequence budget would leave nothing for
// the hooks that run after the server, including the database
// close. It is exported so that relationship can be tested.
ShutdownTimeout = 3 * time.Second
// sentryFlushTimeout is the maximum time to wait for Sentry
// to flush pending events during shutdown.
// TailHookReserve is the share of the fx stop budget this hook
// refuses to spend, leaving it for the hooks that run after the
// server: the delivery engine, the healthcheck, the webhook DB
// manager and the database close.
TailHookReserve = 2 * time.Second
// sentryFlushTimeout is the longest wait for Sentry to flush
// pending events during shutdown, before the remaining stop
// budget is taken into account.
sentryFlushTimeout = 2 * time.Second
// minSentryFlush is the shortest flush worth attempting. Below
// it the remaining budget goes to the tail hooks instead.
minSentryFlush = 250 * time.Millisecond
)
// SentryFlushBudget reports how long the Sentry flush may run when
// remaining is the time left on the fx stop context after the HTTP
// drain. sentry.Flush takes a bare duration and honours no context,
// so this clamp is the only thing keeping a stalled flush from
// spending the tail hooks' share of the budget on top of a
// full-length drain. TailHookReserve is held back, and anything
// under minSentryFlush is skipped rather than attempted uselessly.
func SentryFlushBudget(remaining time.Duration) time.Duration {
budget := min(remaining-TailHookReserve, sentryFlushTimeout)
if budget < minSentryFlush {
return 0
}
return budget
}
//nolint:revive // ServerParams is a standard fx naming convention.
type ServerParams struct {
fx.In
@@ -164,7 +197,7 @@ func (s *Server) cleanShutdown(ctx context.Context) {
s.exitCode = 0
ctxShutdown, shutdownCancel := context.WithTimeout(
ctx, shutdownTimeout,
ctx, ShutdownTimeout,
)
defer shutdownCancel()
@@ -178,10 +211,31 @@ func (s *Server) cleanShutdown(ctx context.Context) {
s.cleanupForExit()
if s.sentryEnabled {
sentry.Flush(sentryFlushTimeout)
s.flushSentry(ctx)
}
}
// flushSentry drains Sentry's queue inside what is left of the fx
// stop budget. A context carrying no deadline — a caller outside the
// fx lifecycle — gets the full timeout.
func (s *Server) flushSentry(ctx context.Context) {
flush := sentryFlushTimeout
if deadline, ok := ctx.Deadline(); ok {
flush = SentryFlushBudget(time.Until(deadline))
}
if flush <= 0 {
s.log.Warn(
"skipping sentry flush, stop budget exhausted",
)
return
}
sentry.Flush(flush)
}
func (s *Server) configure() {
// identify ourselves in the logs
s.params.Logger.Identify()

View File

@@ -0,0 +1,59 @@
package server_test
import (
"testing"
"time"
"github.com/stretchr/testify/require"
"sneak.berlin/go/webhooker/internal/server"
)
// TestSentryFlushBudget covers the clamp that keeps the Sentry flush
// from spending the tail hooks' share of the fx stop budget.
// sentry.Flush ignores the stop context, so without the clamp a
// stalled flush adds its whole timeout on top of the HTTP drain.
func TestSentryFlushBudget(t *testing.T) {
t.Parallel()
tests := []struct {
name string
remaining time.Duration
want time.Duration
}{
{
name: "full drain leaves only the reserve",
remaining: server.TailHookReserve,
want: 0,
},
{
name: "expired budget",
remaining: -time.Second,
want: 0,
},
{
name: "sliver above the reserve is not worth it",
remaining: server.TailHookReserve + 10*time.Millisecond,
want: 0,
},
{
name: "partial flush when some room is left",
remaining: server.TailHookReserve + time.Second,
want: time.Second,
},
{
name: "capped at the nominal timeout",
remaining: time.Hour,
want: 2 * time.Second,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
require.Equal(
t, tt.want, server.SentryFlushBudget(tt.remaining),
)
})
}
}