check / check (push) Successful in 2m52s
Each handler queue held 100,000 message pointers; a message is retained until the slowest handler drains it, so all four full was a derived worst case near 800 MiB. Twenty thousand is about four seconds of feed at peak and caps that at roughly 160 MiB. The streamer already drops rather than blocks on a full queue, so the smaller bound is safe. Batch sizes are unchanged. The largest, asnBatchSize, is 30,000 and now exceeds its queue, but each queued message contributes every ASN in its path, and every handler also flushes on its own timer regardless of fill, so batches still flush and no size change is warranted. Model: opus-4-8
506 lines
13 KiB
Go
506 lines
13 KiB
Go
package routewatch
|
|
|
|
import (
|
|
"net"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"git.eeqj.de/sneak/routewatch/internal/database"
|
|
"git.eeqj.de/sneak/routewatch/internal/logger"
|
|
"git.eeqj.de/sneak/routewatch/internal/metrics"
|
|
"git.eeqj.de/sneak/routewatch/internal/ristypes"
|
|
"github.com/google/uuid"
|
|
)
|
|
|
|
const (
|
|
// prefixHandlerQueueSize is the queue capacity for prefix tracking
|
|
// operations, about 4 seconds of feed at peak. The streamer drops rather
|
|
// than blocks when a queue is full, so this bounds memory. Batches still
|
|
// flush on a timer (prefixBatchTimeout).
|
|
prefixHandlerQueueSize = 20000
|
|
|
|
// prefixBatchSize is the number of prefix updates to batch together
|
|
prefixBatchSize = 25000
|
|
|
|
// prefixBatchTimeout is the maximum time to wait before flushing a batch
|
|
// DO NOT reduce this timeout - larger batches are more efficient
|
|
prefixBatchTimeout = 1 * time.Second
|
|
|
|
// IP version constants
|
|
ipv4Version = 4
|
|
ipv6Version = 6
|
|
)
|
|
|
|
// PrefixHandler tracks BGP prefixes and maintains a live routing table in the database.
|
|
// Routes are added on announcement and deleted on withdrawal.
|
|
type PrefixHandler struct {
|
|
db database.Store
|
|
logger *logger.Logger
|
|
metrics *metrics.Tracker
|
|
|
|
// Batching
|
|
mu sync.Mutex
|
|
batch []prefixUpdate
|
|
lastFlush time.Time
|
|
stopCh chan struct{}
|
|
wg sync.WaitGroup
|
|
}
|
|
|
|
type prefixUpdate struct {
|
|
prefix string
|
|
originASN int
|
|
peer string
|
|
messageType string
|
|
timestamp time.Time
|
|
path []int
|
|
}
|
|
|
|
// NewPrefixHandler creates a new batched prefix tracking handler
|
|
func NewPrefixHandler(db database.Store, logger *logger.Logger) *PrefixHandler {
|
|
h := &PrefixHandler{
|
|
db: db,
|
|
logger: logger,
|
|
batch: make([]prefixUpdate, 0, prefixBatchSize),
|
|
lastFlush: time.Now(),
|
|
stopCh: make(chan struct{}),
|
|
}
|
|
|
|
// Start the flush timer goroutine
|
|
h.wg.Add(1)
|
|
go h.flushLoop()
|
|
|
|
return h
|
|
}
|
|
|
|
// SetMetricsTracker sets the metrics tracker for recording route updates
|
|
func (h *PrefixHandler) SetMetricsTracker(metrics *metrics.Tracker) {
|
|
h.metrics = metrics
|
|
}
|
|
|
|
// WantsMessage returns true if this handler wants to process messages of the given type
|
|
func (h *PrefixHandler) WantsMessage(messageType string) bool {
|
|
// We only care about UPDATE messages for the routing table
|
|
return messageType == "UPDATE"
|
|
}
|
|
|
|
// QueueCapacity returns the desired queue capacity for this handler
|
|
func (h *PrefixHandler) QueueCapacity() int {
|
|
// Batching allows us to use a larger queue
|
|
return prefixHandlerQueueSize
|
|
}
|
|
|
|
// HandleMessage processes a message to track prefix information
|
|
func (h *PrefixHandler) HandleMessage(msg *ristypes.RISMessage) {
|
|
// Use the pre-parsed timestamp
|
|
timestamp := msg.ParsedTimestamp
|
|
|
|
// Get origin ASN from path (last element)
|
|
var originASN int
|
|
if len(msg.Path) > 0 {
|
|
originASN = msg.Path[len(msg.Path)-1]
|
|
}
|
|
|
|
h.mu.Lock()
|
|
defer h.mu.Unlock()
|
|
|
|
// Process announcements
|
|
for _, announcement := range msg.Announcements {
|
|
for _, prefix := range announcement.Prefixes {
|
|
h.batch = append(h.batch, prefixUpdate{
|
|
prefix: prefix,
|
|
originASN: originASN,
|
|
peer: msg.Peer,
|
|
messageType: "announcement",
|
|
timestamp: timestamp,
|
|
path: msg.Path,
|
|
})
|
|
// Record announcement in metrics
|
|
if h.metrics != nil {
|
|
h.metrics.RecordAnnouncement()
|
|
}
|
|
}
|
|
}
|
|
|
|
// Process withdrawals
|
|
for _, prefix := range msg.Withdrawals {
|
|
h.batch = append(h.batch, prefixUpdate{
|
|
prefix: prefix,
|
|
originASN: originASN, // Use the originASN from path if available
|
|
peer: msg.Peer,
|
|
messageType: "withdrawal",
|
|
timestamp: timestamp,
|
|
path: msg.Path,
|
|
})
|
|
// Record withdrawal in metrics
|
|
if h.metrics != nil {
|
|
h.metrics.RecordWithdrawal()
|
|
}
|
|
}
|
|
|
|
// Check if we need to flush
|
|
if len(h.batch) >= prefixBatchSize {
|
|
h.flushBatchLocked()
|
|
}
|
|
}
|
|
|
|
// flushLoop runs in a goroutine and periodically flushes batches
|
|
func (h *PrefixHandler) flushLoop() {
|
|
defer h.wg.Done()
|
|
ticker := time.NewTicker(prefixBatchTimeout)
|
|
defer ticker.Stop()
|
|
|
|
for {
|
|
select {
|
|
case <-ticker.C:
|
|
h.mu.Lock()
|
|
if time.Since(h.lastFlush) >= prefixBatchTimeout {
|
|
h.flushBatchLocked()
|
|
}
|
|
h.mu.Unlock()
|
|
case <-h.stopCh:
|
|
// Final flush
|
|
h.mu.Lock()
|
|
h.flushBatchLocked()
|
|
h.mu.Unlock()
|
|
|
|
return
|
|
}
|
|
}
|
|
}
|
|
|
|
// flushBatchLocked flushes the prefix batch to the database (must be called with mutex held)
|
|
func (h *PrefixHandler) flushBatchLocked() {
|
|
if len(h.batch) == 0 {
|
|
return
|
|
}
|
|
|
|
startTime := time.Now()
|
|
batchSize := len(h.batch)
|
|
|
|
// Group updates by prefix to deduplicate
|
|
// For each prefix, keep the latest update
|
|
prefixMap := make(map[string]prefixUpdate)
|
|
for _, update := range h.batch {
|
|
key := update.prefix
|
|
if existing, ok := prefixMap[key]; !ok || update.timestamp.After(existing.timestamp) {
|
|
prefixMap[key] = update
|
|
}
|
|
}
|
|
|
|
// Collect routes to upsert and delete
|
|
var routesToUpsert []*database.LiveRoute
|
|
var routesToDelete []database.LiveRouteDeletion
|
|
|
|
// Collect unique prefixes to update
|
|
prefixesToUpdate := make(map[string]time.Time)
|
|
|
|
for _, update := range prefixMap {
|
|
// Track prefix for both announcements and withdrawals
|
|
if _, exists := prefixesToUpdate[update.prefix]; !exists || update.timestamp.After(prefixesToUpdate[update.prefix]) {
|
|
prefixesToUpdate[update.prefix] = update.timestamp
|
|
}
|
|
|
|
if update.messageType == "announcement" && update.originASN > 0 {
|
|
// Create live route for batch upsert
|
|
route := h.createLiveRoute(update)
|
|
if route != nil {
|
|
routesToUpsert = append(routesToUpsert, route)
|
|
}
|
|
} else if update.messageType == "withdrawal" {
|
|
// Parse CIDR to get IP version
|
|
_, ipVersion, err := parseCIDR(update.prefix)
|
|
if err != nil {
|
|
h.logger.Error("Failed to parse CIDR for withdrawal", "prefix", update.prefix, "error", err)
|
|
|
|
continue
|
|
}
|
|
|
|
// Create deletion record for batch delete
|
|
routesToDelete = append(routesToDelete, database.LiveRouteDeletion{
|
|
Prefix: update.prefix,
|
|
OriginASN: update.originASN,
|
|
PeerIP: update.peer,
|
|
IPVersion: ipVersion,
|
|
})
|
|
}
|
|
}
|
|
|
|
// Process batch operations
|
|
successCount := 0
|
|
if len(routesToUpsert) > 0 {
|
|
if err := h.db.UpsertLiveRouteBatch(routesToUpsert); err != nil {
|
|
h.logger.Error("Failed to upsert route batch", "error", err, "count", len(routesToUpsert))
|
|
} else {
|
|
successCount += len(routesToUpsert)
|
|
}
|
|
}
|
|
|
|
if len(routesToDelete) > 0 {
|
|
if err := h.db.DeleteLiveRouteBatch(routesToDelete); err != nil {
|
|
h.logger.Error("Failed to delete route batch", "error", err, "count", len(routesToDelete))
|
|
} else {
|
|
successCount += len(routesToDelete)
|
|
}
|
|
}
|
|
|
|
// Update prefix tables
|
|
if len(prefixesToUpdate) > 0 {
|
|
if err := h.db.UpdatePrefixesBatch(prefixesToUpdate); err != nil {
|
|
h.logger.Error("Failed to update prefix batch", "error", err, "count", len(prefixesToUpdate))
|
|
}
|
|
}
|
|
|
|
elapsed := time.Since(startTime)
|
|
h.logger.Debug("Flushed prefix batch",
|
|
"batch_size", batchSize,
|
|
"unique_prefixes", len(prefixMap),
|
|
"success", successCount,
|
|
"duration_ms", elapsed.Milliseconds(),
|
|
)
|
|
|
|
// Clear batch
|
|
h.batch = h.batch[:0]
|
|
h.lastFlush = time.Now()
|
|
}
|
|
|
|
// parseCIDR extracts the mask length and IP version from a prefix string
|
|
func parseCIDR(prefix string) (maskLength int, ipVersion int, err error) {
|
|
_, ipNet, err := net.ParseCIDR(prefix)
|
|
if err != nil {
|
|
return 0, 0, err
|
|
}
|
|
|
|
ones, _ := ipNet.Mask.Size()
|
|
if strings.Contains(prefix, ":") {
|
|
return ones, ipv6Version, nil
|
|
}
|
|
|
|
return ones, ipv4Version, nil
|
|
}
|
|
|
|
// processAnnouncement handles storing an announcement in the database
|
|
// nolint:unused // kept for potential future use
|
|
func (h *PrefixHandler) processAnnouncement(_ *database.Prefix, update prefixUpdate) {
|
|
// Parse CIDR to get mask length
|
|
maskLength, ipVersion, err := parseCIDR(update.prefix)
|
|
if err != nil {
|
|
h.logger.Error("Failed to parse CIDR",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
|
|
return
|
|
}
|
|
|
|
// Track route update metrics
|
|
if h.metrics != nil {
|
|
if ipVersion == ipv4Version {
|
|
h.metrics.RecordIPv4Update()
|
|
} else {
|
|
h.metrics.RecordIPv6Update()
|
|
}
|
|
}
|
|
|
|
// Create live route record
|
|
liveRoute := &database.LiveRoute{
|
|
ID: uuid.New(),
|
|
Prefix: update.prefix,
|
|
MaskLength: maskLength,
|
|
IPVersion: ipVersion,
|
|
OriginASN: update.originASN,
|
|
PeerIP: update.peer,
|
|
ASPath: update.path,
|
|
NextHop: update.peer, // Using peer as next hop
|
|
LastUpdated: update.timestamp,
|
|
}
|
|
|
|
// For IPv4, calculate the IP range
|
|
if ipVersion == ipv4Version {
|
|
start, end, err := database.CalculateIPv4Range(update.prefix)
|
|
if err == nil {
|
|
liveRoute.V4IPStart = &start
|
|
liveRoute.V4IPEnd = &end
|
|
} else {
|
|
h.logger.Error("Failed to calculate IPv4 range",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
|
|
if err := h.db.UpsertLiveRoute(liveRoute); err != nil {
|
|
h.logger.Error("Failed to upsert live route",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
|
|
// createLiveRoute creates a LiveRoute from a prefix update
|
|
func (h *PrefixHandler) createLiveRoute(update prefixUpdate) *database.LiveRoute {
|
|
// Parse CIDR to get mask length
|
|
maskLength, ipVersion, err := parseCIDR(update.prefix)
|
|
if err != nil {
|
|
h.logger.Error("Failed to parse CIDR",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
|
|
return nil
|
|
}
|
|
|
|
// Track route update metrics
|
|
if h.metrics != nil {
|
|
if ipVersion == ipv4Version {
|
|
h.metrics.RecordIPv4Update()
|
|
} else {
|
|
h.metrics.RecordIPv6Update()
|
|
}
|
|
}
|
|
|
|
// Create live route record
|
|
liveRoute := &database.LiveRoute{
|
|
ID: uuid.New(),
|
|
Prefix: update.prefix,
|
|
MaskLength: maskLength,
|
|
IPVersion: ipVersion,
|
|
OriginASN: update.originASN,
|
|
PeerIP: update.peer,
|
|
ASPath: update.path,
|
|
NextHop: update.peer, // Using peer as next hop
|
|
LastUpdated: update.timestamp,
|
|
}
|
|
|
|
// For IPv4, calculate the IP range
|
|
if ipVersion == ipv4Version {
|
|
start, end, err := database.CalculateIPv4Range(update.prefix)
|
|
if err == nil {
|
|
liveRoute.V4IPStart = &start
|
|
liveRoute.V4IPEnd = &end
|
|
} else {
|
|
h.logger.Error("Failed to calculate IPv4 range",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
|
|
return liveRoute
|
|
}
|
|
|
|
// processAnnouncementDirect handles storing an announcement directly without prefix table
|
|
// nolint:unused // kept for potential future use
|
|
func (h *PrefixHandler) processAnnouncementDirect(update prefixUpdate) {
|
|
// Parse CIDR to get mask length
|
|
maskLength, ipVersion, err := parseCIDR(update.prefix)
|
|
if err != nil {
|
|
h.logger.Error("Failed to parse CIDR",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
|
|
return
|
|
}
|
|
|
|
// Track route update metrics
|
|
if h.metrics != nil {
|
|
if ipVersion == ipv4Version {
|
|
h.metrics.RecordIPv4Update()
|
|
} else {
|
|
h.metrics.RecordIPv6Update()
|
|
}
|
|
}
|
|
|
|
// Create live route record
|
|
liveRoute := &database.LiveRoute{
|
|
ID: uuid.New(),
|
|
Prefix: update.prefix,
|
|
MaskLength: maskLength,
|
|
IPVersion: ipVersion,
|
|
OriginASN: update.originASN,
|
|
PeerIP: update.peer,
|
|
ASPath: update.path,
|
|
NextHop: update.peer, // Using peer as next hop
|
|
LastUpdated: update.timestamp,
|
|
}
|
|
|
|
// For IPv4, calculate the IP range
|
|
if ipVersion == ipv4Version {
|
|
start, end, err := database.CalculateIPv4Range(update.prefix)
|
|
if err == nil {
|
|
liveRoute.V4IPStart = &start
|
|
liveRoute.V4IPEnd = &end
|
|
} else {
|
|
h.logger.Error("Failed to calculate IPv4 range",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
|
|
if err := h.db.UpsertLiveRoute(liveRoute); err != nil {
|
|
h.logger.Error("Failed to upsert live route",
|
|
"prefix", update.prefix,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
|
|
// processWithdrawalDirect handles removing a route directly without prefix table
|
|
// nolint:unused // kept for potential future use
|
|
func (h *PrefixHandler) processWithdrawalDirect(update prefixUpdate) {
|
|
// For withdrawals, we need to delete the route from live_routes
|
|
if update.originASN > 0 {
|
|
if err := h.db.DeleteLiveRoute(update.prefix, update.originASN, update.peer); err != nil {
|
|
h.logger.Error("Failed to delete live route",
|
|
"prefix", update.prefix,
|
|
"origin_asn", update.originASN,
|
|
"peer", update.peer,
|
|
"error", err,
|
|
)
|
|
}
|
|
} else {
|
|
// If no origin ASN, just delete all routes for this prefix from this peer
|
|
if err := h.db.DeleteLiveRoute(update.prefix, 0, update.peer); err != nil {
|
|
h.logger.Error("Failed to delete live route (no origin ASN)",
|
|
"prefix", update.prefix,
|
|
"peer", update.peer,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
}
|
|
|
|
// processWithdrawal handles removing a route from the live routing table
|
|
// nolint:unused // kept for potential future use
|
|
func (h *PrefixHandler) processWithdrawal(_ *database.Prefix, update prefixUpdate) {
|
|
// For withdrawals, we need to delete the route from live_routes
|
|
// Since we have the origin ASN from the update, we can delete the specific route
|
|
if update.originASN > 0 {
|
|
if err := h.db.DeleteLiveRoute(update.prefix, update.originASN, update.peer); err != nil {
|
|
h.logger.Error("Failed to delete live route",
|
|
"prefix", update.prefix,
|
|
"origin_asn", update.originASN,
|
|
"peer", update.peer,
|
|
"error", err,
|
|
)
|
|
}
|
|
} else {
|
|
// If no origin ASN, just delete all routes for this prefix from this peer
|
|
if err := h.db.DeleteLiveRoute(update.prefix, 0, update.peer); err != nil {
|
|
h.logger.Error("Failed to delete live route (no origin ASN)",
|
|
"prefix", update.prefix,
|
|
"peer", update.peer,
|
|
"error", err,
|
|
)
|
|
}
|
|
}
|
|
}
|
|
|
|
// Stop gracefully stops the handler and flushes remaining batches
|
|
func (h *PrefixHandler) Stop() {
|
|
close(h.stopCh)
|
|
h.wg.Wait()
|
|
}
|