Add a statistics pane to the webhook page (closes #368)
check / check (push) Waiting to run

Each webhook's event database keeps running totals: one row for its
events, with when the newest arrived, and one row per target for its
deliveries, delivered and failed, each with what retention removed.
Every write to them shares the transaction of the rows it counts, and
a delivery already delivered or failed is not settled again.
Deliveries get a finished_at column; it and target_id end the status
index, so each target's deliveries finished in a window come from one
index-range query grouped by target. Retention deletes 1000 expired
events per transaction, pausing 200 ms between them so other writers
get in, and stops between them on shutdown. The pane is its own
template, its figures in tables.

The schema changes in place with nothing back-filled, so an existing
database must be recreated.

Model: opus-5-5
This commit is contained in:
2026-10-02 01:26:42 +00:00
committed by sneak
parent bfdbc937c6
commit 926d34b4b8
23 changed files with 2033 additions and 118 deletions
+117 -52
View File
@@ -18,6 +18,19 @@ import (
// computation.
const hoursPerDay = 24
// reapBatchSize is how many expired events one retention transaction
// deletes. A transaction holds the event database's write lock, which
// the receiver and the delivery workers wait for, so a large prune is
// split into transactions each short enough to finish well inside the
// busy timeout.
const reapBatchSize = 1000
// reapBatchPause is how long retention waits after one batch before
// starting the next. A writer waiting for the write lock checks for it
// again after at most 100 ms, so a longer pause lets it in between two
// batches instead of only after the whole prune.
const reapBatchPause = 200 * time.Millisecond
// RetentionReaperParams holds the fx dependencies for the
// RetentionReaper.
type RetentionReaperParams struct {
@@ -187,13 +200,15 @@ func (r *RetentionReaper) sweep(ctx context.Context) {
continue
}
r.reapWebhook(wh.ID, wh.RetentionDays)
r.reapWebhook(ctx, wh.ID, wh.RetentionDays)
}
}
// reapWebhook removes every expired event (and its dependents) from a
// single webhook's database.
// single webhook's database, or as many as it reaches before ctx is
// cancelled.
func (r *RetentionReaper) reapWebhook(
ctx context.Context,
webhookID string,
retentionDays int,
) {
@@ -213,7 +228,7 @@ func (r *RetentionReaper) reapWebhook(
return
}
deleted, err := reapExpired(db, cutoff)
deleted, err := reapExpired(ctx, db, cutoff)
if err != nil {
r.log.Error(
"retention sweep: failed to reap expired events",
@@ -265,57 +280,107 @@ func retentionCutoff(
), true
}
// reapExpired hard-deletes, in foreign-key-safe order, the delivery
// results, deliveries, and events associated with events older than
// cutoff. Deletes are unscoped so rows are physically removed rather
// than soft-deleted, reclaiming disk. It returns the number of events
// deleted.
func reapExpired(db *gorm.DB, cutoff time.Time) (int64, error) {
// Fresh subqueries are built per statement to avoid reusing a
// mutated builder across executions.
expiredEventIDs := func() *gorm.DB {
return db.Model(&Event{}).
Select("id").
Where("created_at < ?", cutoff)
}
expiredDeliveryIDs := func() *gorm.DB {
return db.Model(&Delivery{}).
Select("id").
Where("event_id IN (?)", expiredEventIDs())
}
// reapExpired hard-deletes the events older than cutoff, with their
// deliveries and delivery results, reapBatchSize events per
// transaction with reapBatchPause between transactions, until none is
// left. Once ctx is cancelled it returns after the batch in hand,
// leaving the rest to the next sweep, so stopping the app does not
// wait for a long prune. It returns the number of events deleted.
func reapExpired(
ctx context.Context, db *gorm.DB, cutoff time.Time,
) (int64, error) {
var total int64
// 1. Delivery results whose delivery belongs to an expired event.
res := db.Unscoped().
Where("delivery_id IN (?)", expiredDeliveryIDs()).
Delete(&DeliveryResult{})
if res.Error != nil {
return 0, fmt.Errorf(
"deleting expired delivery results: %w",
res.Error,
)
}
for {
var eventIDs []string
// 2. Deliveries belonging to an expired event.
del := db.Unscoped().
Where("event_id IN (?)", expiredEventIDs()).
Delete(&Delivery{})
if del.Error != nil {
return 0, fmt.Errorf(
"deleting expired deliveries: %w",
del.Error,
)
}
err := db.Transaction(func(tx *gorm.DB) error {
err := tx.Unscoped().Model(&Event{}).
Where("created_at < ?", cutoff).
Limit(reapBatchSize).
Pluck("id", &eventIDs).Error
if err != nil {
return fmt.Errorf("selecting expired events: %w", err)
}
// 3. The expired events themselves.
ev := db.Unscoped().
Where("created_at < ?", cutoff).
Delete(&Event{})
if ev.Error != nil {
return 0, fmt.Errorf(
"deleting expired events: %w",
ev.Error,
)
}
if len(eventIDs) == 0 {
return nil
}
return ev.RowsAffected, nil
return deleteEvents(tx, eventIDs)
})
if err != nil {
return total, err
}
total += int64(len(eventIDs))
if len(eventIDs) < reapBatchSize {
return total, nil
}
select {
case <-ctx.Done():
return total, nil
case <-time.After(reapBatchPause):
}
}
}
// deleteEvents hard-deletes the given events and, in foreign-key-safe
// order before them, their delivery results and deliveries, then adds
// what it deleted to the running totals. It runs on reapExpired's
// transaction, so the totals change exactly when the rows do. Deletes
// are unscoped so rows are physically removed rather than
// soft-deleted, reclaiming disk.
func deleteEvents(tx *gorm.DB, eventIDs []string) error {
// 1. The delivery results of the events' deliveries.
err := tx.Unscoped().
Where("delivery_id IN (?)", tx.Unscoped().Model(&Delivery{}).
Select("id").
Where("event_id IN ?", eventIDs)).
Delete(&DeliveryResult{}).Error
if err != nil {
return fmt.Errorf("deleting expired delivery results: %w", err)
}
// 2. The events' deliveries, after counting them, and the failed
// ones among them, per target. The status is tested in the select
// list rather than the WHERE clause: there, SQLite would read every
// failed delivery the webhook has through the status index,
// instead of only these through the event_id index.
var removed []TargetTotals
err = tx.Unscoped().Model(&Delivery{}).
Select("target_id, count(*) AS deliveries_removed, "+
"count(CASE WHEN status = ? THEN 1 END) AS failed_removed",
DeliveryStatusFailed).
Where("event_id IN ?", eventIDs).
Group("target_id").
Find(&removed).Error
if err != nil {
return fmt.Errorf("counting expired deliveries: %w", err)
}
err = tx.Unscoped().
Where("event_id IN ?", eventIDs).
Delete(&Delivery{}).Error
if err != nil {
return fmt.Errorf("deleting expired deliveries: %w", err)
}
// 3. The events themselves.
ev := tx.Unscoped().Where("id IN ?", eventIDs).Delete(&Event{})
if ev.Error != nil {
return fmt.Errorf("deleting expired events: %w", ev.Error)
}
for i := range removed {
err = AddTargetTotals(tx, removed[i])
if err != nil {
return err
}
}
return AddEventTotals(tx, EventTotals{EventsRemoved: ev.RowsAffected})
}