Delete the oldest report files to stay under the size cap (closes #54)
check / check (push) Successful in 1m55s
check / check (push) Successful in 1m55s
When a report would take the report files past DATA_DIR_MAX_BYTES, reportbuf now deletes the oldest report files until it fits, and does the same at start when files left by an earlier run are already past it. A file joins the files that may be deleted, at its place by name, only once it is completely written, so a file still being written is never deleted. A report is refused with 507 only when the reports waiting to be written fill the cap on their own, and then no file is deleted. The reports of a failed write stop counting, and the part of its file written is removed. A file whose deletion fails keeps counting; one already deleted by hand counts as freed. Model: opus-5-5
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
// Package reportbuf accumulates telemetry reports in memory
|
||||
// and periodically flushes them to zstd-compressed JSONL files.
|
||||
// and periodically flushes them to zstd-compressed JSONL files,
|
||||
// deleting the oldest files to keep them under a size cap.
|
||||
package reportbuf
|
||||
|
||||
import (
|
||||
@@ -12,6 +13,7 @@ import (
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
@@ -37,9 +39,10 @@ const (
|
||||
fileSuffix = ".jsonl.zst"
|
||||
)
|
||||
|
||||
// ErrFull is returned by Append when storing the report would
|
||||
// take the report files past the configured maximum size.
|
||||
var ErrFull = errors.New("report files at their size cap")
|
||||
// ErrFull is returned by Append when the reports waiting to be
|
||||
// written leave no room for the report under the configured maximum
|
||||
// size, however many report files are deleted.
|
||||
var ErrFull = errors.New("reports waiting to be written fill the size cap")
|
||||
|
||||
// Params defines the dependencies for Buffer.
|
||||
type Params struct {
|
||||
@@ -49,15 +52,34 @@ type Params struct {
|
||||
Logger *logger.Logger
|
||||
}
|
||||
|
||||
// reportFile is a report file that may be deleted to make room, with
|
||||
// the size it counts for in usedBytes.
|
||||
type reportFile struct {
|
||||
name string
|
||||
size int64
|
||||
}
|
||||
|
||||
// Buffer accumulates JSON lines in memory and flushes them
|
||||
// to zstd-compressed files on disk.
|
||||
type Buffer struct {
|
||||
buf bytes.Buffer
|
||||
dataDir string
|
||||
done chan struct{}
|
||||
log *slog.Logger
|
||||
maxBytes int64
|
||||
mu sync.Mutex
|
||||
buf bytes.Buffer
|
||||
dataDir string
|
||||
done chan struct{}
|
||||
// fileCreated is called with each report file once it is
|
||||
// created, before anything is written to it: it does nothing,
|
||||
// except in tests that hold the write open or make it fail.
|
||||
fileCreated func(f *os.File)
|
||||
// files are the report files that may be deleted to make room,
|
||||
// in name order, which is oldest first: those in dataDir at
|
||||
// start, and each one this buffer writes, put in at its place by
|
||||
// name once it is complete, even when an older file's write
|
||||
// completes after a newer one's. A file still being written is
|
||||
// not among them. filesBytes is their total size.
|
||||
files []reportFile
|
||||
filesBytes int64
|
||||
log *slog.Logger
|
||||
maxBytes int64
|
||||
mu sync.Mutex
|
||||
// now is the clock report files are named by: time.Now, except
|
||||
// in tests that need two flushes to share a timestamp.
|
||||
now func() time.Time
|
||||
@@ -66,8 +88,8 @@ type Buffer struct {
|
||||
seq atomic.Uint64
|
||||
stopOnce sync.Once
|
||||
// usedBytes is what Append checks against maxBytes: the size
|
||||
// of the report files in dataDir, plus the reports not yet
|
||||
// written to one at their uncompressed size.
|
||||
// of the report files in dataDir, plus the reports waiting to
|
||||
// be written to one at their uncompressed size.
|
||||
usedBytes int64
|
||||
}
|
||||
|
||||
@@ -83,11 +105,12 @@ func New(
|
||||
}
|
||||
|
||||
b := &Buffer{
|
||||
dataDir: dir,
|
||||
done: make(chan struct{}),
|
||||
log: params.Logger.Get(),
|
||||
maxBytes: params.Config.DataDirMaxBytes,
|
||||
now: time.Now,
|
||||
dataDir: dir,
|
||||
done: make(chan struct{}),
|
||||
fileCreated: func(*os.File) {},
|
||||
log: params.Logger.Get(),
|
||||
maxBytes: params.Config.DataDirMaxBytes,
|
||||
now: time.Now,
|
||||
}
|
||||
|
||||
lc.Append(fx.Hook{
|
||||
@@ -97,12 +120,28 @@ func New(
|
||||
return fmt.Errorf("create data dir: %w", err)
|
||||
}
|
||||
|
||||
// Report files left by earlier runs count too.
|
||||
b.usedBytes, err = reportFilesSize(b.dataDir)
|
||||
// Report files left by earlier runs count too, and are
|
||||
// the first to be deleted to make room.
|
||||
files, err := reportFiles(b.dataDir)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
b.mu.Lock()
|
||||
|
||||
b.files = files
|
||||
for _, f := range files {
|
||||
b.filesBytes += f.size
|
||||
}
|
||||
|
||||
b.usedBytes = b.filesBytes
|
||||
|
||||
// The files may be past the cap, if it was lowered since
|
||||
// the last run.
|
||||
b.deleteOldestFiles(0)
|
||||
|
||||
b.mu.Unlock()
|
||||
|
||||
go b.flushLoop()
|
||||
|
||||
return nil
|
||||
@@ -128,9 +167,10 @@ func New(
|
||||
}
|
||||
|
||||
// Append marshals v as a single JSON line and appends it to
|
||||
// the buffer. It stores nothing and returns ErrFull if the line
|
||||
// would take usedBytes past maxBytes. If the buffer reaches the
|
||||
// size threshold, it is drained and written to disk
|
||||
// the buffer. If the line would take usedBytes past maxBytes, the
|
||||
// oldest report files are deleted to make room; it stores nothing
|
||||
// and returns ErrFull if that cannot make room. If the buffer
|
||||
// reaches the size threshold, it is drained and written to disk
|
||||
// asynchronously.
|
||||
func (b *Buffer) Append(v any) error {
|
||||
line, err := json.Marshal(v)
|
||||
@@ -142,6 +182,8 @@ func (b *Buffer) Append(v any) error {
|
||||
|
||||
b.mu.Lock()
|
||||
|
||||
b.deleteOldestFiles(lineBytes)
|
||||
|
||||
if b.usedBytes+lineBytes > b.maxBytes {
|
||||
b.mu.Unlock()
|
||||
|
||||
@@ -171,6 +213,37 @@ func (b *Buffer) Append(v any) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// deleteOldestFiles deletes report files, oldest first, until n more
|
||||
// bytes fit under maxBytes. It deletes none when the reports waiting
|
||||
// to be written leave no room for n even with every file gone, since
|
||||
// that would lose the files for nothing. The caller must hold b.mu.
|
||||
func (b *Buffer) deleteOldestFiles(n int64) {
|
||||
for b.usedBytes+n > b.maxBytes && len(b.files) > 0 {
|
||||
if b.usedBytes-b.filesBytes+n > b.maxBytes {
|
||||
return
|
||||
}
|
||||
|
||||
f := b.files[0]
|
||||
b.files = b.files[1:]
|
||||
b.filesBytes -= f.size
|
||||
|
||||
// A file already gone, deleted by hand, has freed its room too.
|
||||
err := os.Remove(filepath.Join(b.dataDir, f.name))
|
||||
if err != nil && !errors.Is(err, fs.ErrNotExist) {
|
||||
// The file is still there, so it still counts. It is
|
||||
// not tried again until the next start.
|
||||
b.log.Error("delete report file failed",
|
||||
"file", f.name, "error", err)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
b.usedBytes -= f.size
|
||||
b.log.Info("deleted report file to make room",
|
||||
"file", f.name, "bytes", f.size)
|
||||
}
|
||||
}
|
||||
|
||||
// flushLoop runs a ticker that periodically flushes buffered
|
||||
// data to disk until the done channel is closed.
|
||||
func (b *Buffer) flushLoop() {
|
||||
@@ -217,15 +290,48 @@ func (b *Buffer) drainBuf() []byte {
|
||||
return data
|
||||
}
|
||||
|
||||
// writeFile creates a timestamped zstd-compressed JSONL file
|
||||
// in the data directory.
|
||||
// writeFile writes data, reports drained from the buffer, to a new
|
||||
// timestamped zstd-compressed JSONL file in the data directory.
|
||||
func (b *Buffer) writeFile(data []byte) error {
|
||||
// The timestamp comes first, so the names sort by time; the number
|
||||
// after it tells apart files named in the same millisecond.
|
||||
ts := b.now().UTC().Format("2006-01-02T15-04-05.000Z")
|
||||
name := fmt.Sprintf("%s%s-%d%s", filePrefix, ts, b.seq.Add(1), fileSuffix)
|
||||
path := filepath.Join(b.dataDir, name)
|
||||
|
||||
size, err := b.createFile(filepath.Join(b.dataDir, name), data)
|
||||
|
||||
// The reports no longer wait to be written, so they stop counting
|
||||
// at their uncompressed size. If the write failed they are lost;
|
||||
// otherwise they count as the file, which from here on may be
|
||||
// deleted to make room.
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
|
||||
b.usedBytes -= int64(len(data))
|
||||
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
b.usedBytes += size
|
||||
b.filesBytes += size
|
||||
|
||||
// At its place by name, not at the end: another write, of a newer
|
||||
// file, may have completed while this one was being written.
|
||||
i, _ := slices.BinarySearchFunc(b.files, name,
|
||||
func(f reportFile, target string) int {
|
||||
return strings.Compare(f.name, target)
|
||||
})
|
||||
b.files = slices.Insert(b.files, i, reportFile{name: name, size: size})
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// createFile creates the file at path holding data compressed with
|
||||
// zstd, and returns its size. If the write fails once the file is
|
||||
// created, it removes the file, so that a failed write leaves nothing
|
||||
// behind to take room.
|
||||
func (b *Buffer) createFile(path string, data []byte) (int64, error) {
|
||||
// path is built from the operator-supplied dataDir plus a
|
||||
// generated timestamp and number, so it carries no external input.
|
||||
f, err := os.OpenFile( //nolint:gosec // see comment above
|
||||
@@ -234,61 +340,65 @@ func (b *Buffer) writeFile(data []byte) error {
|
||||
filePerms,
|
||||
)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create report file: %w", err)
|
||||
return 0, fmt.Errorf("create report file: %w", err)
|
||||
}
|
||||
|
||||
// Closes the file on the early returns below. The success
|
||||
// path closes it explicitly to check the error; closing it
|
||||
// a second time here is harmless.
|
||||
defer func() { _ = f.Close() }()
|
||||
b.fileCreated(f)
|
||||
|
||||
size, err := writeCompressed(f, data)
|
||||
if err != nil {
|
||||
_ = f.Close()
|
||||
|
||||
return 0, errors.Join(err, os.Remove(path))
|
||||
}
|
||||
|
||||
err = f.Close()
|
||||
if err != nil {
|
||||
err = fmt.Errorf("close report file: %w", err)
|
||||
|
||||
return 0, errors.Join(err, os.Remove(path))
|
||||
}
|
||||
|
||||
return size, nil
|
||||
}
|
||||
|
||||
// writeCompressed writes data to f compressed with zstd, and returns
|
||||
// the size of f.
|
||||
func writeCompressed(f *os.File, data []byte) (int64, error) {
|
||||
enc, err := zstd.NewWriter(f)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create zstd encoder: %w", err)
|
||||
return 0, fmt.Errorf("create zstd encoder: %w", err)
|
||||
}
|
||||
|
||||
_, err = enc.Write(data)
|
||||
if err != nil {
|
||||
_ = enc.Close()
|
||||
|
||||
return fmt.Errorf("write compressed data: %w", err)
|
||||
return 0, fmt.Errorf("write compressed data: %w", err)
|
||||
}
|
||||
|
||||
err = enc.Close()
|
||||
if err != nil {
|
||||
return fmt.Errorf("close zstd encoder: %w", err)
|
||||
return 0, fmt.Errorf("close zstd encoder: %w", err)
|
||||
}
|
||||
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
return fmt.Errorf("stat report file: %w", err)
|
||||
return 0, fmt.Errorf("stat report file: %w", err)
|
||||
}
|
||||
|
||||
err = f.Close()
|
||||
if err != nil {
|
||||
return fmt.Errorf("close report file: %w", err)
|
||||
}
|
||||
|
||||
// The reports counted at their uncompressed size while they
|
||||
// waited; now they count as the file. After a failed write they
|
||||
// stay counted as they were, which errs toward refusing reports
|
||||
// early rather than letting the files pass the cap.
|
||||
b.mu.Lock()
|
||||
b.usedBytes += info.Size() - int64(len(data))
|
||||
b.mu.Unlock()
|
||||
|
||||
return nil
|
||||
return info.Size(), nil
|
||||
}
|
||||
|
||||
// reportFilesSize returns the total size of the report files in
|
||||
// dir.
|
||||
func reportFilesSize(dir string) (int64, error) {
|
||||
// reportFiles returns the report files in dir, oldest first:
|
||||
// os.ReadDir sorts them by name, and the names sort by time.
|
||||
func reportFiles(dir string) ([]reportFile, error) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read data dir: %w", err)
|
||||
return nil, fmt.Errorf("read data dir: %w", err)
|
||||
}
|
||||
|
||||
var total int64
|
||||
files := make([]reportFile, 0, len(entries))
|
||||
|
||||
for _, entry := range entries {
|
||||
name := entry.Name()
|
||||
@@ -299,11 +409,11 @@ func reportFilesSize(dir string) (int64, error) {
|
||||
|
||||
info, err := entry.Info()
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("stat report file: %w", err)
|
||||
return nil, fmt.Errorf("stat report file: %w", err)
|
||||
}
|
||||
|
||||
total += info.Size()
|
||||
files = append(files, reportFile{name: name, size: info.Size()})
|
||||
}
|
||||
|
||||
return total, nil
|
||||
return files, nil
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user