Keep counts not written by the request's deadline in memory (closes #224)
check / check (push) Waiting to run

TestService_Get_ReturnsByItsDeadline failed on a busy host because its
request, past its deadline, still waited to write its miss count, a write
with no deadline; in pixad that write waits for the one database
connection every request shares. Each count write now keeps the request's
deadline but not its cancellation. A count not written by then is kept in
memory, where Stats includes it, and one goroutine of the cache writes
those counts one UPDATE at a time, and once more at shutdown before the
database closes. The test phase also runs go test with -parallel 4: on a
busy host, as many tests at once as there are CPUs wait so long to be
scheduled that a timed request can fail.

Model: opus-5-5
This commit is contained in:
2026-10-08 15:44:39 +00:00
parent bdde021b45
commit 410c167c70
8 changed files with 584 additions and 29 deletions
+71 -8
View File
@@ -11,6 +11,8 @@ import (
"io"
"log/slog"
"path/filepath"
"sync"
"sync/atomic"
"time"
lru "github.com/hashicorp/golang-lru/v2"
@@ -79,6 +81,26 @@ type Cache struct {
evictionDone chan struct{}
evictionCancel context.CancelFunc
// The pending counts: hits, misses, upstream fetches and transforms
// whose write to the database missed its request's deadline, kept
// here until they are written by the goroutine that
// StartPendingCountWrites starts. As for eviction, the channels are
// created in NewCache: pendingCountsAdded wakes that goroutine, and
// pendingCountsDone is closed when it returns. pendingCountsCancel,
// set by StartPendingCountWrites, cancels its context, which tells it
// to finish.
pendingHits atomic.Int64
pendingMisses atomic.Int64
pendingUpstreamFetches atomic.Int64
pendingUpstreamFetchBytes atomic.Int64
pendingTransforms atomic.Int64
pendingCountsAdded chan struct{}
pendingCountsDone chan struct{}
pendingCountsCancel context.CancelFunc
// Held through each write of the pending counts, and by Stats to read them
pendingCountsWriteMutex sync.Mutex
// metaCache holds the content types of the variants most recently
// stored or served, so a hit does not read the variant's .meta file.
// It never stands in for the variant file, which is always opened.
@@ -139,6 +161,9 @@ func newCache(
metaCache: metaCache,
contentLocks: newContentLock(),
pendingCountsAdded: make(chan struct{}, 1),
pendingCountsDone: make(chan struct{}),
reconciliationPageSize: defaultReconciliationPageSize,
}
@@ -474,12 +499,21 @@ func (c *Cache) CleanExpired(ctx context.Context) error {
func (c *Cache) Stats(ctx context.Context) (*CacheStats, error) {
var stats CacheStats
// So that no write of the pending counts falls between the two reads
c.pendingCountsWriteMutex.Lock()
// Fetch hit/miss counts from the stats table
err := c.db.QueryRowContext(ctx, `
SELECT hit_count, miss_count
FROM cache_stats WHERE id = 1
`).Scan(&stats.HitCount, &stats.MissCount)
// Hits and misses not yet written count too (see StartPendingCountWrites)
stats.HitCount += c.pendingHits.Load()
stats.MissCount += c.pendingMisses.Load()
c.pendingCountsWriteMutex.Unlock()
if err != nil && !errors.Is(err, sql.ErrNoRows) {
return nil, fmt.Errorf("failed to get cache stats: %w", err)
}
@@ -511,19 +545,30 @@ func (c *Cache) Stats(ctx context.Context) (*CacheStats, error) {
}
// IncrementStats counts a cache hit or miss, and an upstream fetch that read
// fetchBytes bytes, as IncrementUpstreamFetch does.
// fetchBytes bytes, as IncrementUpstreamFetch does. Like the other Increment
// methods, it writes to the database with ctx's deadline but not its
// cancellation, so an ended request is still counted but does not wait for
// the database past its deadline; a count not written by then becomes a
// pending count (see StartPendingCountWrites).
func (c *Cache) IncrementStats(ctx context.Context, hit bool, fetchBytes int64) {
countCtx, cancel := withoutCancelKeepingDeadline(ctx)
defer cancel()
var err error
pendingCount := &c.pendingMisses
if hit {
_, err = c.db.ExecContext(ctx, `
pendingCount = &c.pendingHits
_, err = c.db.ExecContext(countCtx, `
UPDATE cache_stats
SET hit_count = hit_count + 1,
last_updated_at = CURRENT_TIMESTAMP
WHERE id = 1
`)
} else {
_, err = c.db.ExecContext(ctx, `
_, err = c.db.ExecContext(countCtx, `
UPDATE cache_stats
SET miss_count = miss_count + 1,
last_updated_at = CURRENT_TIMESTAMP
@@ -531,7 +576,10 @@ func (c *Cache) IncrementStats(ctx context.Context, hit bool, fetchBytes int64)
`)
}
if err != nil {
switch {
case errors.Is(err, context.DeadlineExceeded):
c.addPendingCount(pendingCount, 1)
case err != nil:
c.log.Warn("failed to count cache hit or miss", "hit", hit, "error", err)
}
@@ -545,14 +593,22 @@ func (c *Cache) IncrementUpstreamFetch(ctx context.Context, fetchBytes int64) {
return
}
_, err := c.db.ExecContext(ctx, `
countCtx, cancel := withoutCancelKeepingDeadline(ctx)
defer cancel()
_, err := c.db.ExecContext(countCtx, `
UPDATE cache_stats
SET upstream_fetch_count = upstream_fetch_count + 1,
upstream_fetch_bytes = upstream_fetch_bytes + ?,
last_updated_at = CURRENT_TIMESTAMP
WHERE id = 1
`, fetchBytes)
if err != nil {
switch {
case errors.Is(err, context.DeadlineExceeded):
c.addPendingCount(&c.pendingUpstreamFetches, 1)
c.addPendingCount(&c.pendingUpstreamFetchBytes, fetchBytes)
case err != nil:
c.log.Warn("failed to count upstream fetch",
"fetch_bytes", fetchBytes, "error", err)
}
@@ -560,13 +616,20 @@ func (c *Cache) IncrementUpstreamFetch(ctx context.Context, fetchBytes int64) {
// IncrementTransformCount counts one image transcoded by the image processor.
func (c *Cache) IncrementTransformCount(ctx context.Context) {
_, err := c.db.ExecContext(ctx, `
countCtx, cancel := withoutCancelKeepingDeadline(ctx)
defer cancel()
_, err := c.db.ExecContext(countCtx, `
UPDATE cache_stats
SET transform_count = transform_count + 1,
last_updated_at = CURRENT_TIMESTAMP
WHERE id = 1
`)
if err != nil {
switch {
case errors.Is(err, context.DeadlineExceeded):
c.addPendingCount(&c.pendingTransforms, 1)
case err != nil:
c.log.Warn("failed to count transform", "error", err)
}
}