Mark a snapshot complete only after its metadata export succeeds (closes #177)
finalizeSnapshotMetadata marked the snapshot complete and then exported its metadata. A crash after completion but before/during the export left the local index showing the snapshot complete while the destination had no manifest or database, and PruneDatabase (which drops only NULL completed_at rows) kept it: a silently unrestorable snapshot. Reorder so completion is recorded last. CompleteSnapshot is split into PopulateSnapshotBlobs (before the export) and MarkSnapshotComplete (after it). An interrupted export now leaves the snapshot incomplete, so the next run PruneDatabase drops it and re-backs-up the data; the reverse tiny window leaves a restorable snapshot the index reports honestly as remote-only. Update REPOSTRUCTURE.md guarantee 4 and the ARCHITECTURE.md flow. Add a fault-injection test driving the full create path. Model: opus-4-8
This commit was merged in pull request #200.
This commit is contained in:
@@ -14,6 +14,7 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
"sneak.berlin/go/vaultik/internal/config"
|
||||
"sneak.berlin/go/vaultik/internal/database"
|
||||
"sneak.berlin/go/vaultik/internal/globals"
|
||||
"sneak.berlin/go/vaultik/internal/log"
|
||||
"sneak.berlin/go/vaultik/internal/snapshot"
|
||||
"sneak.berlin/go/vaultik/internal/storage"
|
||||
@@ -417,9 +418,10 @@ func TestBackupRetryAfterInterruptedUploadIsRestorable(t *testing.T) {
|
||||
// database is uploaded but before the manifest. The destination is left
|
||||
// with blobs and a database but no manifest. verify and snapshot list
|
||||
// must report the damage honestly rather than crashing or passing.
|
||||
// Automatic detection and repair of this partial state on the next run
|
||||
// is tracked in https://git.eeqj.de/sneak/vaultik/issues/177 and is not
|
||||
// asserted here.
|
||||
// Automatic detection and repair of this partial state on the next run is
|
||||
// covered by TestBackupCompletesOnlyAfterMetadataExport
|
||||
// (https://git.eeqj.de/sneak/vaultik/issues/177); this test exercises the
|
||||
// lower-level export path in isolation.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestBackupSurvivesMetadataExportInterruption(t *testing.T) {
|
||||
@@ -496,6 +498,186 @@ func TestBackupSurvivesMetadataExportInterruption(t *testing.T) {
|
||||
"snapshot list must tolerate a partially-exported snapshot")
|
||||
}
|
||||
|
||||
// Scenario 2, repair: the process dies during the metadata export of a
|
||||
// full backup run. Because completion is recorded only after the export
|
||||
// succeeds (finalizeSnapshotMetadata), the interrupted snapshot is left
|
||||
// incomplete rather than silently marked complete without metadata at the
|
||||
// destination. Rerunning the backup must then prune the incomplete
|
||||
// snapshot, produce a snapshot whose destination metadata and local index
|
||||
// agree, and restore. See https://git.eeqj.de/sneak/vaultik/issues/177.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestBackupCompletesOnlyAfterMetadataExport(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
restoreDir := filepath.Join(tempDir, "restored")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
testFiles := writeFaultSourceTree(t, fs, dataDir)
|
||||
|
||||
// A full-backup config: the fault-test defaults plus the fields the
|
||||
// production create path reads (index location, chunk size, and the
|
||||
// named snapshot to back up).
|
||||
cfg := faultTestConfig()
|
||||
cfg.IndexPath = dbPath
|
||||
cfg.ChunkSize = config.Size(faultChunkSize)
|
||||
cfg.Snapshots = map[string]config.SnapshotConfig{
|
||||
"data": {Paths: []string{dataDir}},
|
||||
}
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
// failManifest is on for the first backup and off for the retry, so the
|
||||
// manifest upload fails once — interrupting the export mid-way — then
|
||||
// succeeds.
|
||||
failManifest := true
|
||||
store := faultstore.New(inner)
|
||||
store.OnPut = func(key string) faultstore.PutAction {
|
||||
if failManifest && strings.HasSuffix(key, "manifest.json.zst") {
|
||||
return faultstore.PutFail
|
||||
}
|
||||
|
||||
return faultstore.PutNormal
|
||||
}
|
||||
|
||||
v := newBackupVaultik(ctx, cfg, store, repos, db, fs)
|
||||
opts := &vaultik.SnapshotCreateOptions{Cron: true}
|
||||
|
||||
// First run: the export fails at the manifest upload, so the whole
|
||||
// create fails and the snapshot is left incomplete.
|
||||
require.Error(t, v.CreateSnapshot(opts),
|
||||
"backup must fail when the metadata export is interrupted")
|
||||
|
||||
incompletes, err := repos.Snapshots.GetIncompleteSnapshots(ctx)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, incompletes, 1,
|
||||
"an interrupted export must leave exactly one incomplete snapshot")
|
||||
|
||||
afterFirst, err := repos.Snapshots.ListRecent(ctx, listRecentTestLimit)
|
||||
require.NoError(t, err)
|
||||
|
||||
for _, s := range afterFirst {
|
||||
require.Nil(t, s.CompletedAt,
|
||||
"no snapshot may be marked complete before its metadata is exported")
|
||||
}
|
||||
|
||||
// Second run on the same index and destination: the retry succeeds.
|
||||
failManifest = false
|
||||
|
||||
require.NoError(t, v.CreateSnapshot(opts),
|
||||
"a retry after an interrupted export must succeed")
|
||||
|
||||
assertRetryConsistentAndRestorable(
|
||||
ctx, t, cfg, inner, repos, db, fs, restoreDir, testFiles)
|
||||
}
|
||||
|
||||
// assertRetryConsistentAndRestorable checks the end state after the retry
|
||||
// backup in TestBackupCompletesOnlyAfterMetadataExport: the interrupted
|
||||
// snapshot is pruned, exactly one completed snapshot remains, its metadata
|
||||
// is at the destination, and it restores to the original tree.
|
||||
func assertRetryConsistentAndRestorable(
|
||||
ctx context.Context, t *testing.T, cfg *config.Config,
|
||||
inner storage.Storer, repos *database.Repositories, db *database.DB,
|
||||
fs afero.Fs, restoreDir string, testFiles map[string][]byte,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
incompletes, err := repos.Snapshots.GetIncompleteSnapshots(ctx)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, incompletes,
|
||||
"the next run's prune must drop the interrupted snapshot")
|
||||
|
||||
local, err := repos.Snapshots.ListRecent(ctx, listRecentTestLimit)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, local, 1, "exactly one snapshot must remain after the retry")
|
||||
|
||||
final := local[0]
|
||||
require.NotNil(t, final.CompletedAt, "the retry's snapshot must be complete")
|
||||
|
||||
// The destination and the local index agree: the completed snapshot has
|
||||
// both its metadata objects at the destination.
|
||||
key := snapshot.RemoteSnapshotKey(final.ID.String())
|
||||
_, err = inner.Stat(ctx, "metadata/"+key+"/manifest.json.zst")
|
||||
require.NoError(t, err, "the completed snapshot's manifest must be at the destination")
|
||||
_, err = inner.Stat(ctx, "metadata/"+key+"/db.zst.age")
|
||||
require.NoError(t, err, "the completed snapshot's database must be at the destination")
|
||||
|
||||
require.NoError(t, db.Close())
|
||||
|
||||
// The snapshot restores from the destination alone.
|
||||
reader := newReaderVaultik(ctx, cfg, inner, nil, fs)
|
||||
require.NoError(t, reader.Restore(&vaultik.RestoreOptions{
|
||||
SnapshotID: final.ID.String(),
|
||||
TargetDir: restoreDir,
|
||||
Verify: true,
|
||||
}), "the retry's snapshot must be restorable")
|
||||
|
||||
assertRestoredTree(t, fs, restoreDir, testFiles)
|
||||
}
|
||||
|
||||
// listRecentTestLimit is a generous cap for the handful of snapshots these
|
||||
// tests create when reading the local index directly.
|
||||
const listRecentTestLimit = 100
|
||||
|
||||
// newBackupVaultik builds a Vaultik that runs the full create path
|
||||
// (CreateSnapshot) writing through storer, wiring the same scanner factory
|
||||
// and snapshot manager the production dependency graph provides.
|
||||
func newBackupVaultik(
|
||||
ctx context.Context, cfg *config.Config, storer storage.Storer,
|
||||
repos *database.Repositories, db *database.DB, fs afero.Fs,
|
||||
) *vaultik.Vaultik {
|
||||
v := &vaultik.Vaultik{
|
||||
Globals: &globals.Globals{Version: "v", Commit: "g"},
|
||||
Config: cfg,
|
||||
DB: db,
|
||||
Repositories: repos,
|
||||
Storage: storer,
|
||||
SnapshotManager: newFaultSnapshotManager(fs, storer, cfg, repos),
|
||||
ScannerFactory: faultScannerFactory(cfg, repos, storer),
|
||||
Fs: fs,
|
||||
Stdout: io.Discard,
|
||||
Stderr: io.Discard,
|
||||
UI: ui.NewWithColor(io.Discard, false),
|
||||
}
|
||||
v.SetContext(ctx)
|
||||
|
||||
return v
|
||||
}
|
||||
|
||||
// faultScannerFactory mirrors the production provideScannerFactory, binding
|
||||
// the scanner to the given store, repositories, and config so a full
|
||||
// create-path backup writes through the fault-injecting store.
|
||||
func faultScannerFactory(
|
||||
cfg *config.Config, repos *database.Repositories, storer storage.Storer,
|
||||
) snapshot.ScannerFactory {
|
||||
return func(params snapshot.ScannerParams) *snapshot.Scanner {
|
||||
return snapshot.NewScanner(snapshot.ScannerConfig{
|
||||
FS: params.Fs,
|
||||
Storage: storer,
|
||||
ChunkSize: faultChunkSize,
|
||||
MaxBlobSize: faultMaxBlobSize,
|
||||
CompressionLevel: cfg.CompressionLevel,
|
||||
AgeRecipients: cfg.AgeRecipients,
|
||||
Repositories: repos,
|
||||
EnableProgress: params.EnableProgress,
|
||||
UI: params.UI,
|
||||
Exclude: params.Exclude,
|
||||
SkipErrors: params.SkipErrors,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Scenario 5: the restore target runs out of space mid-file. Restore
|
||||
// must fail with an out-of-space error, and must not leave a truncated
|
||||
// file at the target path presenting as a complete restore. Restore
|
||||
|
||||
Reference in New Issue
Block a user