Deploy: main into prod (#343)
check / check (push) Successful in 3m32s

Brings `prod`, which upaas deploys, up to `main` at `9cf9cdd`, the merge of #321. `prod` was cut from `main` at `251cb3d` (1.0.0b1).

What it deploys is everything listed in #321. For running it:

- With `WEBHOOKER_ENVIRONMENT` unset, the instance runs as `prod` and sends no `Access-Control-Allow-Origin: *`.
- Each event database gains its new indexes the first time it is opened after the upgrade.
- `webhooker_delivery_retries_total` no longer counts a circuit breaker holding back a delivery that is already `retrying`.

Not in this PR yet: #340, in which the container sets its own data directory owner and mode before start. It is in progress on `next`. Once it reaches `main`, this PR carries it, because the PR follows `main`.

Model: opus-5-5
Co-authored-by: Jeffrey Paul <1+sneak@noreply.example.org>
Reviewed-on: #343
This commit is contained in:
2026-09-29 02:48:03 -07:00
committed by sneak
co-authored by sneak
parent 251cb3d3d3
commit 5f84d891cf
36 changed files with 1574 additions and 221 deletions
+97 -25
View File
@@ -362,6 +362,15 @@ func (e *Engine) start() {
// stop cancels the worker pool's context and waits for the pool
// to drain, bounded by the stop hook's context: a wedged worker
// must not hang the process past fx's stop timeout.
//
// Once the pool has drained it closes the archive writers, so a
// clean stop leaves no archive -wal behind. Nothing else holds a
// writer for long by then: the archive sweeper stops before the
// engine, and deleting a webhook only closes one. If the pool did
// not drain in time, the writers are left open, as a kill would
// leave them. Closing them would wait for any write in progress,
// and a worker still running would then open new writers that
// nothing closes, so it gains nothing over a kill.
func (e *Engine) stop(ctx context.Context) error {
e.log.Info("delivery engine stopping")
@@ -376,6 +385,8 @@ func (e *Engine) stop(ctx context.Context) error {
return err
}
e.dbTarget.evictAll()
e.log.Info("delivery engine stopped")
return nil
@@ -438,6 +449,31 @@ func (e *Engine) processNewTask(
return
}
// Restart recovery can send and release this delivery before the
// receiver's Notify queues it. Ownership cannot refuse a delivery
// nobody holds, so the row decides whether it still needs sending.
row, err := e.loadDelivery(webhookDB, task.DeliveryID)
if err != nil {
e.log.Error(
"failed to load delivery",
"delivery_id", task.DeliveryID,
"error", err,
)
return
}
if row.Status != database.DeliveryStatusPending {
e.log.Info(
"delivery already handled, not sent again",
"delivery_id", task.DeliveryID,
"event_id", task.EventID,
"status", row.Status,
)
return
}
event := buildEventFromTask(task)
event, err = e.hydrateEvent(
@@ -482,7 +518,7 @@ func (e *Engine) processRetryTask(
return
}
d, err := e.loadRetryDelivery(
d, err := e.loadDelivery(
webhookDB, task.DeliveryID,
)
if err != nil {
@@ -703,9 +739,7 @@ func (e *Engine) recoverSingleRetry(
// webhook on one bad read would be a far larger fault than
// the strand it is meant to clear.
if errors.Is(err, gorm.ErrRecordNotFound) {
e.failMissingTargetRetry(
webhookDB, webhookID, d,
)
e.failMissingTarget(webhookDB, webhookID, d)
return
}
@@ -1108,9 +1142,7 @@ func (e *Engine) sweepSingleRetry(
// Deleted is terminal, unreadable is not; see
// recoverSingleRetry.
if errors.Is(err, gorm.ErrRecordNotFound) {
e.failMissingTargetRetry(
webhookDB, webhookID, d,
)
e.failMissingTarget(webhookDB, webhookID, d)
return
}
@@ -1224,19 +1256,19 @@ func (e *Engine) failUnretryableRetry(
e.failDelivery(webhookDB, d, target.Type, reason)
}
// failMissingTargetRetry terminally fails an orphaned retrying
// delivery whose target row is gone. Both restart recovery and the
// periodic sweep call it, so the transition exists once.
// failMissingTarget terminally fails a recovered delivery, pending or
// retrying, whose target row is gone. Restart recovery and the periodic
// sweep call it for both statuses, so the transition exists once.
//
// Until it existed both paths logged the failed lookup and returned,
// which left the delivery retrying for the life of the database and
// Until it existed those paths logged the failed lookup and moved on,
// which left the delivery where it was for the life of the database and
// the sweep repeating the same error every minute forever. Failing it
// with a recorded reason is the treatment the other orphaned-retry
// cases already get, so all of them read alike in the event log.
//
// Logged at warn rather than error: a deleted target is an operator
// action, not a system fault.
func (e *Engine) failMissingTargetRetry(
func (e *Engine) failMissingTarget(
webhookDB *gorm.DB,
webhookID string,
d *database.Delivery,
@@ -1249,13 +1281,37 @@ func (e *Engine) failMissingTargetRetry(
defer e.inflight.release(d.ID)
// The batch was read before ownership was taken, and a worker may
// have settled the delivery and let it go in between. Only a row
// still in the status the batch read is failed.
row, err := e.loadDelivery(webhookDB, d.ID)
if err != nil {
e.log.Error(
"failed to load delivery",
"delivery_id", d.ID,
"error", err,
)
return
}
if row.Status != d.Status {
e.log.Debug(
"delivery already handled, not failed",
"delivery_id", d.ID,
"status", row.Status,
)
return
}
targetType, reason := e.missingTargetReason(d.TargetID)
e.log.Warn(
"failing orphaned retrying delivery: "+
"its target no longer exists",
"failing recovered delivery: its target no longer exists",
"webhook_id", webhookID,
"delivery_id", d.ID,
"status", d.Status,
"target_id", d.TargetID,
"target_type", targetType,
)
@@ -1289,15 +1345,14 @@ func (e *Engine) missingTargetReason(
if err != nil {
return "", fmt.Sprintf(
"target %s no longer exists; the delivery "+
"cannot be retried and has been failed "+
"terminally",
"has been failed terminally",
targetID,
)
}
return target.Type, fmt.Sprintf(
"target %q (type %s) was deleted; the delivery "+
"cannot be retried and has been failed terminally",
"has been failed terminally",
target.Name, target.Type,
)
}
@@ -1643,7 +1698,7 @@ func (e *Engine) hydrateEvent(
return event, nil
}
func (e *Engine) loadRetryDelivery(
func (e *Engine) loadDelivery(
webhookDB *gorm.DB, deliveryID string,
) (*database.Delivery, error) {
var d database.Delivery
@@ -1996,13 +2051,30 @@ func (e *Engine) sendRecoveredDeliveries(
target, ok := targetMap[deliveries[i].TargetID]
if !ok {
e.log.Error(
"target not found for delivery",
"delivery_id", deliveries[i].ID,
"target_id", deliveries[i].TargetID,
)
// A missing entry does not mean the target is gone: the
// map is also empty when its query failed. Only a lookup
// that finds no row ends the delivery; any other error
// leaves it pending for the next sweep. See
// recoverSingleRetry.
var err error
continue
target, err = e.loadTarget(deliveries[i].TargetID)
if errors.Is(err, gorm.ErrRecordNotFound) {
e.failMissingTarget(webhookDB, webhookID, &deliveries[i])
continue
}
if err != nil {
e.log.Error(
"failed to load target for recovered delivery",
"delivery_id", deliveries[i].ID,
"target_id", deliveries[i].TargetID,
"error", err,
)
continue
}
}
if !e.takeForRedispatch(