2 Commits
Author SHA1 Message Date
sneak 5b3752db19 Count each file, byte and upload once in backup statistics (closes #225)
check / check (push) Successful in 17m14s
The scanner added a changed file's bytes again for each new chunk and
counted a file as unchanged for each chunk already stored. It now
counts files and bytes once, in the scan phase, and counts its own
uploads, so a --cron run, which has no progress reporter, records
them. The blob count no longer adds earlier paths' blobs again.

The snapshots row now stores the size of all files in total_size and
the referenced blobs' sizes in blob_size, blob_uncompressed_size and
compression_ratio, as docs/DATAMODEL.md says. Those sizes come from
one query, and a failed query fails the snapshot. DATAMODEL.md now
says chunk_count and blob_count count what the run added.

Removed the uncalled UpdateSnapshotStats.

Model: opus-5-5
2026-10-06 23:10:53 +00:00
clawbot b06f992152 Start the progress reporter once per snapshot, not once per path (closes #253)
check / check (push) Successful in 19m54s
A backup without --cron of a snapshot with two or more paths panicked
with "close of closed channel". Scan runs once per path, and it started
the progress reporter and deferred its Stop each time; Stop closes the
reporter's signal channel, so the second path's Stop panicked.
scanAllDirectories now starts the reporter before the first path and
stops it after the last, and Scan no longer starts or stops it. A new
test backs up a two-path snapshot with the reporter on and restores
both paths.

Model: opus-5-5
2026-10-07 00:12:10 +02:00
6 changed files with 191 additions and 37 deletions
+8
View File
@@ -35,6 +35,14 @@ the tag exists and is exercised; what is left is merging `next` to
the snapshot references, and `docs/DATAMODEL.md` now says
`chunk_count` and `blob_count` count what the run added.
- 2026-10-06: Made a backup without `--cron` of a snapshot with two or
more `paths` complete instead of panicking with `close of closed
channel` ([issue #253](https://git.eeqj.de/sneak/vaultik/issues/253)).
`Scan` runs once per path and started and stopped the progress
reporter each time, and a second stop panics. The reporter is now
started and stopped once per snapshot, around the scans of all its
paths.
- 2026-10-06: Made a restore path argument select only that path and
what is beneath it
([issue #223](https://git.eeqj.de/sneak/vaultik/issues/223)). The
+24
View File
@@ -544,6 +544,30 @@ func (r *SnapshotRepository) GetSnapshotTotalCompressedSize(
return totalSize, nil
}
// GetSnapshotBlobSizes returns the total compressed and uncompressed sizes
// of all blobs referenced by a snapshot.
func (r *SnapshotRepository) GetSnapshotBlobSizes(
ctx context.Context, snapshotID string,
) (int64, int64, error) {
query := `
SELECT COALESCE(SUM(b.compressed_size), 0),
COALESCE(SUM(b.uncompressed_size), 0)
FROM snapshot_blobs sb
JOIN blobs b ON sb.blob_hash = b.blob_hash
WHERE sb.snapshot_id = ?
`
var compressed, uncompressed int64
err := r.db.conn.QueryRowContext(ctx, query, snapshotID).Scan(
&compressed, &uncompressed)
if err != nil {
return 0, 0, fmt.Errorf("querying snapshot blob sizes: %w", err)
}
return compressed, uncompressed, nil
}
// GetSnapshotUncompressedChunkSize returns the sum of plaintext sizes of all unique
// chunks referenced by a snapshot (via snapshot_files → file_chunks → chunks).
func (r *SnapshotRepository) GetSnapshotUncompressedChunkSize(
+59
View File
@@ -145,6 +145,65 @@ func TestSnapshotRepositoryUpdateCounts(t *testing.T) {
}
}
// GetSnapshotBlobSizes totals the blobs the snapshot references, and only
// those.
func TestSnapshotRepositoryGetSnapshotBlobSizes(t *testing.T) {
t.Parallel()
db, cleanup := setupTestDB(t)
defer cleanup()
ctx := context.Background()
repos := database.NewRepositories(db)
snapshot := &database.Snapshot{
ID: "2024-01-03T12:00:00Z",
Hostname: testHostname,
VaultikVersion: testVersion,
StartedAt: time.Now().Truncate(time.Second),
}
err := repos.Snapshots.Create(ctx, nil, snapshot)
if err != nil {
t.Fatalf("failed to create snapshot: %v", err)
}
blobs := []*database.Blob{
{Hash: "referenced-1", CompressedSize: 10, UncompressedSize: 100},
{Hash: "referenced-2", CompressedSize: 20, UncompressedSize: 200},
{Hash: "unreferenced", CompressedSize: 40, UncompressedSize: 400},
}
for _, blob := range blobs {
blob.ID = types.NewBlobID()
blob.CreatedTS = time.Now().Truncate(time.Second)
err = repos.Blobs.Create(ctx, nil, blob)
if err != nil {
t.Fatalf("failed to create blob %s: %v", blob.Hash, err)
}
}
for _, blob := range blobs[:2] {
err = repos.Snapshots.AddBlob(ctx, nil, snapshot.ID.String(),
blob.ID, blob.Hash)
if err != nil {
t.Fatalf("failed to add blob %s to snapshot: %v", blob.Hash, err)
}
}
compressed, uncompressed, err := repos.Snapshots.GetSnapshotBlobSizes(
ctx, snapshot.ID.String())
if err != nil {
t.Fatalf("failed to get snapshot blob sizes: %v", err)
}
if compressed != 30 || uncompressed != 300 {
t.Errorf("blob sizes: got %d and %d, want 30 and 300",
compressed, uncompressed)
}
}
func TestSnapshotRepositoryListRecent(t *testing.T) {
t.Parallel()
+7 -6
View File
@@ -227,12 +227,6 @@ func (s *Scanner) Scan(
log.Debug("No storage configured, blobs will not be uploaded")
}
// Start progress reporting if enabled
if s.progress != nil {
s.progress.Start()
defer s.progress.Stop()
}
// Phase 0: Repair any state left by an interrupted previous run, then
// load known files and chunks from the database into memory for fast
// lookup.
@@ -302,6 +296,13 @@ func (s *Scanner) Scan(
return result, nil
}
// GetProgress returns the progress reporter for this scanner, or nil when
// progress is off. Scan neither starts nor stops it: the caller does,
// once for all the paths it scans, because a second Stop panics.
func (s *Scanner) GetProgress() *ProgressReporter {
return s.progress
}
// loadDatabaseState loads known files and chunks from the database into
// memory for fast lookup. This avoids per-file and per-chunk database
// queries during the scan and process phases.
+22 -31
View File
@@ -189,6 +189,11 @@ type snapshotStats struct {
totalBytesUploaded int64
totalBlobsUploaded int
uploadDuration time.Duration
// The sizes of all blobs the snapshot references, set by
// finalizeSnapshotMetadata once snapshot_blobs is populated.
blobSize int64
blobUncompressedSize int64
}
// createNamedSnapshot creates a single named snapshot
@@ -279,6 +284,11 @@ func (v *Vaultik) resolveSnapshotPaths(snapName string) ([]string, error) {
func (v *Vaultik) scanAllDirectories(
scanner *snapshot.Scanner, resolvedDirs []string, snapshotID string,
) (*snapshotStats, error) {
if progress := scanner.GetProgress(); progress != nil {
progress.Start()
defer progress.Stop()
}
stats := &snapshotStats{}
for i, dir := range resolvedDirs {
@@ -342,7 +352,11 @@ func (v *Vaultik) finalizeSnapshotMetadata(
return fmt.Errorf("populating snapshot blobs: %w", err)
}
blobSize, blobUncompressedSize := v.getSnapshotBlobSizes(snapshotID)
stats.blobSize, stats.blobUncompressedSize, err =
v.Repositories.Snapshots.GetSnapshotBlobSizes(v.ctx, snapshotID)
if err != nil {
return fmt.Errorf("getting snapshot blob sizes: %w", err)
}
extStats := snapshot.ExtendedBackupStats{
BackupStats: snapshot.BackupStats{
@@ -352,8 +366,8 @@ func (v *Vaultik) finalizeSnapshotMetadata(
BlobsCreated: stats.totalBlobs,
BytesUploaded: stats.totalBytesUploaded,
},
BlobSize: blobSize,
BlobUncompressedSize: blobUncompressedSize,
BlobSize: stats.blobSize,
BlobUncompressedSize: stats.blobUncompressedSize,
CompressionLevel: v.Config.CompressionLevel,
UploadDurationMs: stats.uploadDuration.Milliseconds(),
}
@@ -397,12 +411,10 @@ func (v *Vaultik) printSnapshotSummary(
totalFilesChanged := stats.totalFiles - stats.totalFilesSkipped
totalBytesAll := stats.totalBytes + stats.totalBytesSkipped
// Get total blob sizes from database
compressedSize, uncompressedSize := v.getSnapshotBlobSizes(snapshotID)
var compressionRatio float64
if uncompressedSize > 0 {
compressionRatio = float64(compressedSize) / float64(uncompressedSize)
if stats.blobUncompressedSize > 0 {
compressionRatio = float64(stats.blobSize) /
float64(stats.blobUncompressedSize)
} else {
compressionRatio = 1.0
}
@@ -430,8 +442,8 @@ func (v *Vaultik) printSnapshotSummary(
if stats.totalBlobsUploaded > 0 {
v.UI.Detailf("Storage: %s compressed from %s (%.2fx ratio).",
v.UI.Size(compressedSize),
v.UI.Size(uncompressedSize),
v.UI.Size(stats.blobSize),
v.UI.Size(stats.blobUncompressedSize),
compressionRatio)
v.UI.Detailf("Upload: %d blobs, %s in %s (%s).",
stats.totalBlobsUploaded,
@@ -443,27 +455,6 @@ func (v *Vaultik) printSnapshotSummary(
v.UI.Detailf("Snapshot create duration: %s.", v.UI.Duration(snapshotDuration))
}
// getSnapshotBlobSizes returns total compressed and uncompressed blob
// sizes for a snapshot.
func (v *Vaultik) getSnapshotBlobSizes(snapshotID string) (int64, int64) {
var compressed, uncompressed int64
blobHashes, err := v.Repositories.Snapshots.GetBlobHashes(v.ctx, snapshotID)
if err != nil {
return 0, 0
}
for _, hash := range blobHashes {
blob, err := v.Repositories.Blobs.GetByHash(v.ctx, hash)
if err == nil && blob != nil {
compressed += blob.CompressedSize
uncompressed += blob.UncompressedSize
}
}
return compressed, uncompressed
}
// SnapshotPurgeOptions contains options for the snapshot purge command.
type SnapshotPurgeOptions struct {
KeepLatest bool // Keep only the most recent snapshot per name
@@ -0,0 +1,71 @@
package vaultik_test
import (
"context"
"maps"
"path/filepath"
"testing"
"github.com/spf13/afero"
"github.com/stretchr/testify/require"
"sneak.berlin/go/vaultik/internal/config"
"sneak.berlin/go/vaultik/internal/database"
"sneak.berlin/go/vaultik/internal/log"
"sneak.berlin/go/vaultik/internal/storage"
"sneak.berlin/go/vaultik/internal/vaultik"
)
// A backup without --cron runs the progress reporter while one scanner
// scans each path of the snapshot in turn. See
// https://git.eeqj.de/sneak/vaultik/issues/253.
//
//nolint:paralleltest // installs the global logger via log.Initialize
func TestBackupWithoutCronOfTwoPathSnapshotRestoresBothPaths(t *testing.T) {
log.Initialize(log.Config{})
const snapshotName = "data"
fs := afero.NewOsFs()
tempDir := t.TempDir()
firstDir := filepath.Join(tempDir, "first")
secondDir := filepath.Join(tempDir, "second")
storeDir := filepath.Join(tempDir, "remote")
restoreDir := filepath.Join(tempDir, "restored")
dbPath := filepath.Join(tempDir, "index.sqlite")
ctx := context.Background()
files := writeFaultSourceTree(t, fs, firstDir)
maps.Copy(files, writeFaultSourceTree(t, fs, secondDir))
cfg := faultTestConfig()
cfg.IndexPath = dbPath
cfg.ChunkSize = config.Size(faultChunkSize)
cfg.Snapshots = map[string]config.SnapshotConfig{
snapshotName: {Paths: []string{firstDir, secondDir}},
}
store, err := storage.NewFileStorer(storeDir)
require.NoError(t, err)
db, err := database.New(ctx, dbPath)
require.NoError(t, err)
repos := database.NewRepositories(db)
v := newBackupVaultik(ctx, cfg, store, repos, db, fs)
require.NoError(t, v.CreateSnapshot(&vaultik.SnapshotCreateOptions{
Snapshots: []string{snapshotName},
}))
id := localSnapshotID(ctx, t, repos, snapshotName)
require.NoError(t, db.Close())
reader := newReaderVaultik(ctx, cfg, store, nil, fs)
require.NoError(t, reader.Restore(&vaultik.RestoreOptions{
SnapshotID: id,
TargetDir: restoreDir,
Verify: true,
}))
assertRestoredTree(t, fs, restoreDir, files)
}