All checks were successful
check / check (push) Successful in 3m3s
Capturing real webhook traffic and firing it repeatedly at a backend under development is a primary function of this service, and per-delivery replay cannot do it: it only ever resolves the delivery's own original target, so a target created for a dev backend has no prior delivery and nothing can be replayed to it. The event log now offers a per-event Resubmit action. It stores a NEW event copying the stored one's method, headers, body and content type verbatim, and fans it out to the webhook's currently ACTIVE targets, resolved fresh by the query the receiver uses -- so a target created long after the original event arrived receives it. The original event's deliveries have no bearing on where the copy goes, inactive targets are skipped as the receiver skips them, and the action is repeatable: replay's in-flight refusal is deliberately not ported, because firing one captured event over and over is the point. The receiver and the resubmit path share one construction and one fan-out site. An eventSource value carries where the fields came from, live request or stored event, and createAndFanOut writes the event and its pending deliveries in one transaction and hands the tasks to the same Notifier, so a resubmitted delivery is retried, SSRF-guarded and circuit-broken exactly as a first one is. buildDeliveryTasks returns an error instead of writing a response, which is what lets both callers share it. The stored event is read once, before the write transaction, with a cast to blob, so a body over delivery.MaxInlineBodySize is copied byte for byte and the engine loads it from the new event row. A nullable resubmitted_from_id records provenance -- empty for an event that arrived on the receiver -- and the event log reports the relationship in both directions, without which the log is unreadable after a few resubmits of one event. The route sits in the owned-source group, so auth, CSRF and the body cap apply, with its own rate limit bucket and an events_resubmitted_total counter. Inbound signature verification is not re-run: there is no inbound signature to check on a copy an authenticated, CSRF-protected operator action submits. Per-delivery replay is unchanged; it serves recovery, which resubmit does not replace. The README claimed in four places that replay was unimplemented, one of them telling the operator that a delivery stranded by a target type change was lost; all four are corrected and resubmit is documented beside replay.
400 lines
13 KiB
Go
400 lines
13 KiB
Go
// Package metrics defines the Prometheus collectors describing
|
|
// webhooker's delivery pipeline: how many events arrive, how many
|
|
// deliveries are attempted, how they end, how long they take, how
|
|
// deep the queues are, and how many circuit breakers are open.
|
|
//
|
|
// The inbound HTTP metrics come from the go-http-metrics recorder in
|
|
// internal/middleware and land on prometheus.DefaultRegisterer. These
|
|
// collectors register there too, so both surfaces are gathered by the
|
|
// one promhttp handler mounted on the authenticated /metrics route.
|
|
package metrics
|
|
|
|
import (
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/prometheus/client_golang/prometheus"
|
|
"github.com/prometheus/client_golang/prometheus/promauto"
|
|
"sneak.berlin/go/webhooker/internal/database"
|
|
)
|
|
|
|
// namespace prefixes every collector defined here.
|
|
const namespace = "webhooker"
|
|
|
|
// targetTypeLabel is the only label any delivery metric carries, and
|
|
// cardinality is the whole reason for that.
|
|
//
|
|
// A target type is one of four compile-time constants, so the label
|
|
// domain is bounded by construction. Target ids, event ids and
|
|
// entrypoint ids are not: they are UUIDs minted per operator action
|
|
// or per inbound request, a series is never reclaimed once it exists,
|
|
// and labelling by any of them makes /metrics a memory leak that
|
|
// grows with traffic. normalizeTargetType enforces the bound at every
|
|
// call site — a type the registry does not know collapses into
|
|
// unknownTargetType rather than minting a series of its own.
|
|
const targetTypeLabel = "target_type"
|
|
|
|
// unknownTargetType is the bucket for a target type outside the known
|
|
// set, so an unrecognised value cannot mint a new series.
|
|
const unknownTargetType = "unknown"
|
|
|
|
// Delivery duration buckets, exponential from 5ms so the last bucket
|
|
// (about 98s) sits above the 30s outbound HTTP client timeout.
|
|
const (
|
|
durationBucketStart = 0.005
|
|
durationBucketFactor = 3
|
|
durationBucketCount = 10
|
|
)
|
|
|
|
// knownTargetTypes is the fixed label domain: the target types the
|
|
// delivery engine implements.
|
|
//
|
|
//nolint:gochecknoglobals // the label domain, built once per process
|
|
var knownTargetTypes = []database.TargetType{
|
|
database.TargetTypeHTTP,
|
|
database.TargetTypeDatabase,
|
|
database.TargetTypeLog,
|
|
database.TargetTypeSlack,
|
|
}
|
|
|
|
// defaultSet is the process-wide metric set, registered on the same
|
|
// registry the HTTP middleware and the /metrics handler already use.
|
|
// It is built on first use rather than in an init so that a test
|
|
// binary that never touches metrics never registers them.
|
|
//
|
|
//nolint:gochecknoglobals // one process-wide registration, by design
|
|
var defaultSet = sync.OnceValue(func() *Set {
|
|
return New(prometheus.DefaultRegisterer)
|
|
})
|
|
|
|
// Default returns the process-wide metric set.
|
|
func Default() *Set {
|
|
return defaultSet()
|
|
}
|
|
|
|
// Set is one registered group of webhooker's delivery collectors.
|
|
// Production uses the single Default set; tests build their own
|
|
// against a private registry so assertions are not disturbed by
|
|
// deliveries other tests are making concurrently.
|
|
type Set struct {
|
|
eventsReceived prometheus.Counter
|
|
deliveryAttempts *prometheus.CounterVec
|
|
deliveriesSucceeded *prometheus.CounterVec
|
|
deliveriesFailed *prometheus.CounterVec
|
|
deliveryRetries *prometheus.CounterVec
|
|
deliveryReplays *prometheus.CounterVec
|
|
eventsResubmitted prometheus.Counter
|
|
deliveryDuration *prometheus.HistogramVec
|
|
deliveriesPending *prometheus.GaugeVec
|
|
deliveriesRetrying *prometheus.GaugeVec
|
|
circuitBreakersOpen *prometheus.GaugeVec
|
|
}
|
|
|
|
// New registers a full set of delivery collectors on reg and returns
|
|
// it. It panics if reg already holds them, which is the intended
|
|
// behaviour for a duplicate registration.
|
|
func New(reg prometheus.Registerer) *Set {
|
|
factory := promauto.With(reg)
|
|
|
|
s := &Set{
|
|
eventsReceived: factory.NewCounter(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "events_received_total",
|
|
Help: "Webhook events received and " +
|
|
"stored, so the receive and deliver " +
|
|
"sides can be compared.",
|
|
},
|
|
),
|
|
deliveryDuration: factory.NewHistogramVec(
|
|
prometheus.HistogramOpts{
|
|
Namespace: namespace,
|
|
Name: "delivery_duration_seconds",
|
|
Help: "Wall time of a single delivery " +
|
|
"attempt, by target type.",
|
|
Buckets: prometheus.ExponentialBuckets(
|
|
durationBucketStart,
|
|
durationBucketFactor,
|
|
durationBucketCount,
|
|
),
|
|
},
|
|
[]string{targetTypeLabel},
|
|
),
|
|
}
|
|
|
|
s.registerCounters(factory)
|
|
s.registerGauges(factory)
|
|
s.initSeries()
|
|
|
|
return s
|
|
}
|
|
|
|
// EventReceived counts one inbound webhook event stored.
|
|
func (s *Set) EventReceived() {
|
|
s.eventsReceived.Inc()
|
|
}
|
|
|
|
// DeliveryAttempted counts one delivery attempt dispatched to a
|
|
// target.
|
|
func (s *Set) DeliveryAttempted(t database.TargetType) {
|
|
s.deliveryAttempts.
|
|
WithLabelValues(normalizeTargetType(t)).
|
|
Inc()
|
|
}
|
|
|
|
// ObserveDeliveryDuration records how long one delivery attempt took.
|
|
func (s *Set) ObserveDeliveryDuration(
|
|
t database.TargetType, d time.Duration,
|
|
) {
|
|
s.deliveryDuration.
|
|
WithLabelValues(normalizeTargetType(t)).
|
|
Observe(d.Seconds())
|
|
}
|
|
|
|
// DeliveryReplayed counts one delivery an operator replayed from the
|
|
// event log.
|
|
//
|
|
// A replay runs the ordinary engine path, so it already moves the
|
|
// attempt, outcome and duration series exactly as a first delivery
|
|
// does — deliberately, since a replay is a real delivery and hiding it
|
|
// from those would misreport the pipeline. This counter is the one
|
|
// place the two are distinguishable, and it carries the existing
|
|
// target-type label rather than adding a replay dimension to every
|
|
// other series.
|
|
func (s *Set) DeliveryReplayed(t database.TargetType) {
|
|
s.deliveryReplays.
|
|
WithLabelValues(normalizeTargetType(t)).
|
|
Inc()
|
|
}
|
|
|
|
// EventResubmitted counts one stored event an operator re-injected
|
|
// from the event log.
|
|
//
|
|
// It counts the operator action once, not the deliveries it fans out
|
|
// to: those already move the attempt, outcome and duration series, and
|
|
// the new event moves events_received_total, since it is a stored
|
|
// event that the delivery side will be compared against. This counter
|
|
// is what separates a resubmitted event from a received one.
|
|
//
|
|
// It carries no labels. The only label available at the call site
|
|
// would be the route pattern, which has exactly one value and so would
|
|
// distinguish nothing; the target types the event fans out to belong
|
|
// to the delivery series, not to this one.
|
|
func (s *Set) EventResubmitted() {
|
|
s.eventsResubmitted.Inc()
|
|
}
|
|
|
|
// DeliveryStatusChanged counts a delivery's transition into a new
|
|
// status. The mapping from status to counter lives here, next to the
|
|
// collectors, so the engine has a single call for every transition it
|
|
// persists. A move back to pending is not an outcome and counts
|
|
// nothing.
|
|
func (s *Set) DeliveryStatusChanged(
|
|
t database.TargetType, status database.DeliveryStatus,
|
|
) {
|
|
label := normalizeTargetType(t)
|
|
|
|
switch status {
|
|
case database.DeliveryStatusDelivered:
|
|
s.deliveriesSucceeded.WithLabelValues(label).Inc()
|
|
case database.DeliveryStatusFailed:
|
|
s.deliveriesFailed.WithLabelValues(label).Inc()
|
|
case database.DeliveryStatusRetrying:
|
|
s.deliveryRetries.WithLabelValues(label).Inc()
|
|
case database.DeliveryStatusPending:
|
|
}
|
|
}
|
|
|
|
// SetQueueDepths publishes the pending and retrying queue depths from
|
|
// one sample. Every label in the queue domain is written on every
|
|
// call, so a type whose queue has drained reads zero instead of
|
|
// holding its last value forever.
|
|
func (s *Set) SetQueueDepths(
|
|
pending, retrying map[database.TargetType]int,
|
|
) {
|
|
pendingByLabel := foldToLabels(pending)
|
|
retryingByLabel := foldToLabels(retrying)
|
|
|
|
for _, label := range queueDepthLabels() {
|
|
s.deliveriesPending.WithLabelValues(label).
|
|
Set(float64(pendingByLabel[label]))
|
|
s.deliveriesRetrying.WithLabelValues(label).
|
|
Set(float64(retryingByLabel[label]))
|
|
}
|
|
}
|
|
|
|
// queueDepthLabels is the label domain of the two queue-depth gauges:
|
|
// the known target types plus unknown.
|
|
//
|
|
// Unknown is a real bucket here, not a safety net. A delivery queued
|
|
// against a target that has since been deleted carries a target id no
|
|
// longer in the targets table, so the sample resolves it to the empty
|
|
// type; folding it into unknown is what keeps that backlog visible.
|
|
// Dropping it would hide the one queue nobody is watching.
|
|
func queueDepthLabels() []string {
|
|
labels := make([]string, 0, len(knownTargetTypes)+1)
|
|
|
|
for _, t := range knownTargetTypes {
|
|
labels = append(labels, string(t))
|
|
}
|
|
|
|
return append(labels, unknownTargetType)
|
|
}
|
|
|
|
// foldToLabels collapses a per-target-type count onto the bounded
|
|
// label domain, summing everything outside the known set into
|
|
// unknown.
|
|
func foldToLabels(
|
|
counts map[database.TargetType]int,
|
|
) map[string]int {
|
|
byLabel := make(map[string]int, len(counts))
|
|
|
|
for t, n := range counts {
|
|
byLabel[normalizeTargetType(t)] += n
|
|
}
|
|
|
|
return byLabel
|
|
}
|
|
|
|
// SetCircuitBreakersOpen publishes how many of a target type's
|
|
// circuit breakers are currently open.
|
|
func (s *Set) SetCircuitBreakersOpen(
|
|
t database.TargetType, open int,
|
|
) {
|
|
s.circuitBreakersOpen.
|
|
WithLabelValues(normalizeTargetType(t)).
|
|
Set(float64(open))
|
|
}
|
|
|
|
func (s *Set) registerCounters(factory promauto.Factory) {
|
|
s.deliveryAttempts = factory.NewCounterVec(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "delivery_attempts_total",
|
|
Help: "Delivery attempts dispatched to a " +
|
|
"target, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.deliveriesSucceeded = factory.NewCounterVec(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "deliveries_succeeded_total",
|
|
Help: "Deliveries that reached the delivered " +
|
|
"state, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.deliveriesFailed = factory.NewCounterVec(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "deliveries_failed_total",
|
|
Help: "Deliveries that failed terminally and " +
|
|
"will not be retried, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.deliveryRetries = factory.NewCounterVec(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "delivery_retries_total",
|
|
Help: "Deliveries put back into the retrying " +
|
|
"state, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.deliveryReplays = factory.NewCounterVec(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "delivery_replays_total",
|
|
Help: "Deliveries an operator replayed from the " +
|
|
"event log, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.eventsResubmitted = factory.NewCounter(
|
|
prometheus.CounterOpts{
|
|
Namespace: namespace,
|
|
Name: "events_resubmitted_total",
|
|
Help: "Stored events an operator re-injected from " +
|
|
"the event log as new events.",
|
|
},
|
|
)
|
|
}
|
|
|
|
func (s *Set) registerGauges(factory promauto.Factory) {
|
|
s.deliveriesPending = factory.NewGaugeVec(
|
|
prometheus.GaugeOpts{
|
|
Namespace: namespace,
|
|
Name: "deliveries_pending",
|
|
Help: "Deliveries currently in the pending " +
|
|
"state, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.deliveriesRetrying = factory.NewGaugeVec(
|
|
prometheus.GaugeOpts{
|
|
Namespace: namespace,
|
|
Name: "deliveries_retrying",
|
|
Help: "Deliveries currently in the retrying " +
|
|
"state, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
|
|
s.circuitBreakersOpen = factory.NewGaugeVec(
|
|
prometheus.GaugeOpts{
|
|
Namespace: namespace,
|
|
Name: "circuit_breakers_open",
|
|
Help: "Delivery circuit breakers currently " +
|
|
"open, by target type.",
|
|
},
|
|
[]string{targetTypeLabel},
|
|
)
|
|
}
|
|
|
|
// initSeries materialises every known-target-type series at zero, so
|
|
// a dashboard and an alert rule see a target type that has not
|
|
// delivered yet rather than a missing series.
|
|
//
|
|
// The queue-depth gauges additionally get their unknown series, which
|
|
// holds deliveries queued against a deleted target. That backlog can
|
|
// predate the process — it is read out of the databases, not counted
|
|
// from transitions — so its series has to exist from the first scrape
|
|
// rather than appearing only once a backlog has already built up.
|
|
func (s *Set) initSeries() {
|
|
for _, t := range knownTargetTypes {
|
|
label := string(t)
|
|
|
|
s.deliveryAttempts.WithLabelValues(label)
|
|
s.deliveriesSucceeded.WithLabelValues(label)
|
|
s.deliveriesFailed.WithLabelValues(label)
|
|
s.deliveryRetries.WithLabelValues(label)
|
|
s.deliveryReplays.WithLabelValues(label)
|
|
s.deliveriesPending.WithLabelValues(label)
|
|
s.deliveriesRetrying.WithLabelValues(label)
|
|
s.circuitBreakersOpen.WithLabelValues(label)
|
|
}
|
|
|
|
s.deliveriesPending.WithLabelValues(unknownTargetType)
|
|
s.deliveriesRetrying.WithLabelValues(unknownTargetType)
|
|
}
|
|
|
|
// normalizeTargetType maps a target type onto the bounded label
|
|
// domain, collapsing anything outside it to unknownTargetType.
|
|
func normalizeTargetType(t database.TargetType) string {
|
|
for _, known := range knownTargetTypes {
|
|
if t == known {
|
|
return string(known)
|
|
}
|
|
}
|
|
|
|
return unknownTargetType
|
|
}
|