check / check (push) Waiting to run
The scanner added a changed file's bytes again for each new chunk and counted a file as unchanged for each chunk already stored. It now counts files and bytes once, in the scan phase, and counts its own uploads, so a --cron run, which has no progress reporter, records them. The blob count no longer adds earlier paths' blobs again. The snapshots row now stores the size of all files in total_size and the referenced blobs' sizes in blob_size, blob_uncompressed_size and compression_ratio, as docs/DATAMODEL.md says. Those sizes come from one query, and a failed query fails the snapshot. DATAMODEL.md now says chunk_count and blob_count count what the run added. Removed UpdateSnapshotStats and GetCountBySnapshot, which nothing calls any more. Model: opus-5-5
286 lines
8.8 KiB
Go
286 lines
8.8 KiB
Go
package vaultik_test
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/spf13/afero"
|
|
"github.com/stretchr/testify/assert"
|
|
"github.com/stretchr/testify/require"
|
|
"sneak.berlin/go/vaultik/internal/config"
|
|
"sneak.berlin/go/vaultik/internal/database"
|
|
"sneak.berlin/go/vaultik/internal/log"
|
|
"sneak.berlin/go/vaultik/internal/storage"
|
|
"sneak.berlin/go/vaultik/internal/storage/faultstore"
|
|
"sneak.berlin/go/vaultik/internal/ui"
|
|
"sneak.berlin/go/vaultik/internal/vaultik"
|
|
)
|
|
|
|
// These tests cover https://git.eeqj.de/sneak/vaultik/issues/225: the
|
|
// summary printed after a backup, and the statistics stored in the
|
|
// snapshots table, count each file, byte and upload once, and a --cron
|
|
// run records its uploads.
|
|
|
|
// summaryUploadDelay slows every blob upload, so a run's upload time is
|
|
// at least this long per blob even on a local store.
|
|
const summaryUploadDelay = 20 * time.Millisecond
|
|
|
|
// summaryEnv is a backup setup whose user-facing output is kept in out.
|
|
type summaryEnv struct {
|
|
v *vaultik.Vaultik
|
|
db *database.DB
|
|
repos *database.Repositories
|
|
out *bytes.Buffer
|
|
|
|
// aPath is a.bin, whose content copy.bin repeats; aSize is its size
|
|
// and totalSize the size of all three source files.
|
|
aPath string
|
|
aSize int64
|
|
totalSize int64
|
|
}
|
|
|
|
// newSummaryEnv writes src/one/a.bin, src/one/small.txt and
|
|
// src/two/copy.bin, a copy of a.bin. Every chunk of copy.bin is therefore
|
|
// already stored by the time the backup reaches it.
|
|
//
|
|
// The snapshot names "first" and "second" back up src; "split" backs up
|
|
// src/one and src/two as two paths.
|
|
func newSummaryEnv(t *testing.T) *summaryEnv {
|
|
t.Helper()
|
|
|
|
log.Initialize(log.Config{})
|
|
|
|
fs := afero.NewOsFs()
|
|
tempDir := t.TempDir()
|
|
srcDir := filepath.Join(tempDir, "src")
|
|
dirOne := filepath.Join(srcDir, "one")
|
|
dirTwo := filepath.Join(srcDir, "two")
|
|
dbPath := filepath.Join(tempDir, "index.sqlite")
|
|
ctx := context.Background()
|
|
|
|
aContent := bytesPattern("a-", int(3*faultChunkSize))
|
|
smallContent := []byte("hello vaultik")
|
|
files := map[string][]byte{
|
|
filepath.Join(dirOne, "a.bin"): aContent,
|
|
filepath.Join(dirOne, "small.txt"): smallContent,
|
|
filepath.Join(dirTwo, "copy.bin"): aContent,
|
|
}
|
|
|
|
for path, content := range files {
|
|
require.NoError(t, fs.MkdirAll(filepath.Dir(path), 0o755))
|
|
require.NoError(t, afero.WriteFile(fs, path, content, 0o644))
|
|
}
|
|
|
|
cfg := faultTestConfig()
|
|
cfg.IndexPath = dbPath
|
|
cfg.ChunkSize = config.Size(faultChunkSize)
|
|
cfg.Snapshots = map[string]config.SnapshotConfig{
|
|
"first": {Paths: []string{srcDir}},
|
|
"second": {Paths: []string{srcDir}},
|
|
"split": {Paths: []string{dirOne, dirTwo}},
|
|
}
|
|
|
|
inner, err := storage.NewFileStorer(filepath.Join(tempDir, "remote"))
|
|
require.NoError(t, err)
|
|
|
|
store := faultstore.New(inner)
|
|
store.OnPut = func(key string) faultstore.PutAction {
|
|
if strings.HasPrefix(key, "blobs/") {
|
|
time.Sleep(summaryUploadDelay)
|
|
}
|
|
|
|
return faultstore.PutNormal
|
|
}
|
|
|
|
db, err := database.New(ctx, dbPath)
|
|
require.NoError(t, err)
|
|
t.Cleanup(func() { _ = db.Close() })
|
|
|
|
repos := database.NewRepositories(db)
|
|
out := &bytes.Buffer{}
|
|
v := newBackupVaultik(ctx, cfg, store, repos, db, fs)
|
|
v.UI = ui.NewWithColor(out, false)
|
|
|
|
return &summaryEnv{
|
|
v: v,
|
|
db: db,
|
|
repos: repos,
|
|
out: out,
|
|
aPath: filepath.Join(dirOne, "a.bin"),
|
|
aSize: int64(len(aContent)),
|
|
totalSize: int64(2*len(aContent) + len(smallContent)),
|
|
}
|
|
}
|
|
|
|
// backUp runs a backup of the named snapshot and returns its output.
|
|
func (e *summaryEnv) backUp(t *testing.T, name string, cron bool) string {
|
|
t.Helper()
|
|
|
|
e.out.Reset()
|
|
require.NoError(t, e.v.CreateSnapshot(&vaultik.SnapshotCreateOptions{
|
|
Cron: cron,
|
|
Snapshots: []string{name},
|
|
}))
|
|
|
|
return e.out.String()
|
|
}
|
|
|
|
// snapshot returns the local snapshots row of the snapshot named name.
|
|
func (e *summaryEnv) snapshot(t *testing.T, name string) *database.Snapshot {
|
|
t.Helper()
|
|
|
|
ctx := context.Background()
|
|
snap, err := e.repos.Snapshots.GetByID(ctx,
|
|
localSnapshotID(ctx, t, e.repos, name))
|
|
require.NoError(t, err)
|
|
require.NotNil(t, snap)
|
|
|
|
return snap
|
|
}
|
|
|
|
// uploads returns how many blobs the snapshot uploaded and their
|
|
// total size, as recorded in the uploads table.
|
|
func (e *summaryEnv) uploads(t *testing.T, snapshotID string) (int64, int64) {
|
|
t.Helper()
|
|
|
|
var count, size int64
|
|
|
|
err := e.db.Conn().QueryRowContext(context.Background(), `
|
|
SELECT COUNT(*), COALESCE(SUM(size), 0)
|
|
FROM uploads WHERE snapshot_id = ?`, snapshotID).Scan(&count, &size)
|
|
require.NoError(t, err)
|
|
|
|
return count, size
|
|
}
|
|
|
|
// referencedBlobSizes returns the compressed and uncompressed sizes of
|
|
// all blobs the snapshot references.
|
|
func (e *summaryEnv) referencedBlobSizes(
|
|
t *testing.T, snapshotID string,
|
|
) (int64, int64) {
|
|
t.Helper()
|
|
|
|
var compressed, uncompressed int64
|
|
|
|
err := e.db.Conn().QueryRowContext(context.Background(), `
|
|
SELECT COALESCE(SUM(b.compressed_size), 0),
|
|
COALESCE(SUM(b.uncompressed_size), 0)
|
|
FROM snapshot_blobs sb JOIN blobs b ON b.blob_hash = sb.blob_hash
|
|
WHERE sb.snapshot_id = ?`, snapshotID).Scan(&compressed, &uncompressed)
|
|
require.NoError(t, err)
|
|
|
|
return compressed, uncompressed
|
|
}
|
|
|
|
// filesLine returns the summary's line of file counts.
|
|
func filesLine(examined, backedUp, unchanged int) string {
|
|
return fmt.Sprintf("Files: %d examined, %d backed up, %d unchanged.",
|
|
examined, backedUp, unchanged)
|
|
}
|
|
|
|
// dataLine returns the summary's line of byte counts.
|
|
func (e *summaryEnv) dataLine(total, backedUp int64) string {
|
|
return fmt.Sprintf("Data: %s total (%s backed up).",
|
|
e.v.UI.Size(total), e.v.UI.Size(backedUp))
|
|
}
|
|
|
|
// A first backup stores copy.bin's chunks while backing up a.bin, so
|
|
// copy.bin's chunks are deduplicated within the run. Each file and byte
|
|
// is still counted once.
|
|
//
|
|
//nolint:paralleltest // installs the global logger via log.Initialize
|
|
func TestSnapshotSummaryFirstRun(t *testing.T) {
|
|
env := newSummaryEnv(t)
|
|
|
|
summary := env.backUp(t, "first", false)
|
|
|
|
assert.Contains(t, summary, filesLine(3, 3, 0))
|
|
assert.Contains(t, summary, env.dataLine(env.totalSize, env.totalSize))
|
|
|
|
snap := env.snapshot(t, "first")
|
|
uploadCount, uploadBytes := env.uploads(t, snap.ID.String())
|
|
require.Positive(t, uploadCount)
|
|
|
|
assert.Contains(t, summary, fmt.Sprintf("Upload: %d blobs, %s in ",
|
|
uploadCount, env.v.UI.Size(uploadBytes)))
|
|
|
|
assert.Equal(t, int64(3), snap.FileCount)
|
|
assert.Equal(t, env.totalSize, snap.TotalSize)
|
|
assert.Equal(t, uploadCount, snap.BlobCount)
|
|
assert.Equal(t, uploadBytes, snap.UploadBytes)
|
|
}
|
|
|
|
// An incremental backup where a.bin's mtime changed but its content did
|
|
// not: a.bin is backed up again and every one of its chunks is already
|
|
// stored.
|
|
//
|
|
//nolint:paralleltest // installs the global logger via log.Initialize
|
|
func TestSnapshotSummaryIncrementalRunWithDeduplicatedChunks(t *testing.T) {
|
|
env := newSummaryEnv(t)
|
|
|
|
env.backUp(t, "first", false)
|
|
|
|
later := time.Now().Add(time.Hour)
|
|
require.NoError(t, os.Chtimes(env.aPath, later, later))
|
|
|
|
summary := env.backUp(t, "second", false)
|
|
|
|
assert.Contains(t, summary, filesLine(3, 1, 2))
|
|
assert.Contains(t, summary, env.dataLine(env.totalSize, env.aSize))
|
|
assert.NotContains(t, summary, "Upload:")
|
|
|
|
snap := env.snapshot(t, "second")
|
|
compressed, uncompressed := env.referencedBlobSizes(t, snap.ID.String())
|
|
require.Positive(t, compressed)
|
|
|
|
assert.Equal(t, env.totalSize, snap.TotalSize)
|
|
assert.Zero(t, snap.ChunkCount)
|
|
assert.Zero(t, snap.BlobCount)
|
|
assert.Zero(t, snap.UploadBytes)
|
|
assert.Equal(t, compressed, snap.BlobSize,
|
|
"blob_size must total the blobs the snapshot references")
|
|
assert.Equal(t, uncompressed, snap.BlobUncompressedSize)
|
|
}
|
|
|
|
// Under --cron the progress reporter is off; the upload figures must
|
|
// still reach the summary and the snapshots row. The snapshot has two
|
|
// paths, each backed up by its own scan.
|
|
//
|
|
//nolint:paralleltest // installs the global logger via log.Initialize
|
|
func TestSnapshotSummaryCronRunRecordsUploads(t *testing.T) {
|
|
env := newSummaryEnv(t)
|
|
|
|
summary := env.backUp(t, "split", true)
|
|
|
|
snap := env.snapshot(t, "split")
|
|
uploadCount, uploadBytes := env.uploads(t, snap.ID.String())
|
|
require.Positive(t, uploadCount)
|
|
|
|
assert.Contains(t, summary, filesLine(3, 3, 0))
|
|
assert.Contains(t, summary, env.dataLine(env.totalSize, env.totalSize))
|
|
assert.Contains(t, summary, fmt.Sprintf("Upload: %d blobs, %s in ",
|
|
uploadCount, env.v.UI.Size(uploadBytes)))
|
|
|
|
assert.Equal(t, env.totalSize, snap.TotalSize)
|
|
assert.Equal(t, uploadCount, snap.BlobCount,
|
|
"blob_count must count each blob once, however many paths the "+
|
|
"snapshot has")
|
|
assert.Equal(t, uploadBytes, snap.UploadBytes)
|
|
assert.GreaterOrEqual(t, snap.UploadDurationMs,
|
|
uploadCount*summaryUploadDelay.Milliseconds())
|
|
|
|
compressed, uncompressed := env.referencedBlobSizes(t, snap.ID.String())
|
|
require.Positive(t, uncompressed)
|
|
|
|
assert.Equal(t, compressed, snap.BlobSize)
|
|
assert.Equal(t, uncompressed, snap.BlobUncompressedSize)
|
|
assert.InDelta(t, float64(compressed)/float64(uncompressed),
|
|
snap.CompressionRatio, 1e-9)
|
|
}
|