diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 6428d8f..309186f 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -353,7 +353,7 @@ CreateSnapshot(opts) ## Deduplication Strategy -1. **File-level**: Files unchanged since last backup are skipped (metadata comparison: size, mtime, mode, uid, gid) +1. **File-level**: Files unchanged since last backup are skipped (metadata comparison: size, mtime, mode, uid, gid), unless the file lists a chunk that no uploaded blob holds; such a file is re-chunked 2. **Chunk-level**: Chunks are content-addressed by SHA256 hash. If a chunk hash already exists in the database, the chunk data is not re-uploaded. diff --git a/README.md b/README.md index 0bab63c..ecd7622 100644 --- a/README.md +++ b/README.md @@ -40,7 +40,8 @@ Features: * modern encryption ([age](https://age-encryption.org/), X25519 + ChaCha20-Poly1305) * content-defined chunking with deduplication (FastCDC) -* incremental backups (only changed files are re-chunked) +* incremental backups (a file is re-chunked only when it changed or a + chunk it lists is held by no uploaded blob) * multithreaded zstd compression at configurable levels * content-addressed immutable storage * local state tracking in SQLite (enables write-only incremental backups) @@ -499,8 +500,9 @@ format does and does not protect. * Content-defined chunking using the FastCDC algorithm * Average chunk size: configurable (default 10MB) -* Deduplication at file level (unchanged files skipped) and chunk level - (identical chunks across files stored once) +* Deduplication at file level (unchanged files skipped, unless a chunk + the file lists is held by no uploaded blob) and chunk level (identical + chunks across files stored once) * Multiple chunks packed into blobs to reduce object count ### encryption diff --git a/TODO.md b/TODO.md index c490f04..fef4869 100644 --- a/TODO.md +++ b/TODO.md @@ -22,6 +22,15 @@ the tag exists and is exercised; what is left is merging `next` to # Completed Steps +- 2026-10-06: Made a backup re-chunk a known file that lists a chunk no + uploaded blob holds + ([issue #214](https://git.eeqj.de/sneak/vaultik/issues/214)). File + rows are shared by every snapshot and updated in place, so removing or + pruning the only snapshot that referenced a changed file's current + blob left an older snapshot keeping a row that matched the disk while + no blob held its chunks. The next backup skipped the file and + completed a snapshot that could not restore it. + - 2026-09-22: Routed the last direct-to-stdout command output through `internal/ui` ([issue #149](https://git.eeqj.de/sneak/vaultik/issues/149)). The diff --git a/docs/DATAMODEL.md b/docs/DATAMODEL.md index 7338c0d..37e128e 100644 --- a/docs/DATAMODEL.md +++ b/docs/DATAMODEL.md @@ -195,6 +195,7 @@ Tracks blob upload metrics. 1. **Change Detection** - `SELECT * FROM files WHERE path = ?` - Get previous file metadata - Compare mtime, size, mode to detect changes + - Re-chunk a file that lists a chunk no uploaded blob holds, even when its metadata is unchanged - Skip unchanged files but still add to `snapshot_files` 2. **Chunk Reuse** diff --git a/internal/database/files.go b/internal/database/files.go index 8fd69e7..f9d8022 100644 --- a/internal/database/files.go +++ b/internal/database/files.go @@ -266,6 +266,56 @@ func (r *FileRepository) ListByPrefix( return files, rows.Err() } +// ListIDsWithChunksNotInUploadedBlobs returns the IDs of the files whose +// path starts with prefix and that list at least one chunk held by no +// blob whose upload has completed (uploaded_ts set). A new snapshot +// cannot reference such a chunk, so a backup must not treat the file as +// unchanged even when its metadata matches the file on disk. +func (r *FileRepository) ListIDsWithChunksNotInUploadedBlobs( + ctx context.Context, prefix string, +) ([]types.FileID, error) { + query := ` + SELECT DISTINCT f.id + FROM files f + JOIN file_chunks fc ON fc.file_id = f.id + WHERE f.path LIKE ? || '%' + AND NOT EXISTS ( + SELECT 1 + FROM blob_chunks bc + JOIN blobs b ON bc.blob_id = b.id + WHERE bc.chunk_hash = fc.chunk_hash + AND b.uploaded_ts IS NOT NULL + ) + ` + + rows, err := r.db.conn.QueryContext(ctx, query, prefix) + if err != nil { + return nil, fmt.Errorf("querying files: %w", err) + } + + defer func() { + err := rows.Close() + if err != nil { + Fatalf("failed to close rows: %v", err) + } + }() + + var ids []types.FileID + + for rows.Next() { + var id types.FileID + + err := rows.Scan(&id) + if err != nil { + return nil, fmt.Errorf("scanning file ID: %w", err) + } + + ids = append(ids, id) + } + + return ids, rows.Err() +} + // ListAll returns all files in the database func (r *FileRepository) ListAll(ctx context.Context) ([]*File, error) { query := ` diff --git a/internal/snapshot/scanner.go b/internal/snapshot/scanner.go index a0f6a9b..71f8fd8 100644 --- a/internal/snapshot/scanner.go +++ b/internal/snapshot/scanner.go @@ -73,6 +73,11 @@ type Scanner struct { knownChunks map[string]struct{} knownChunksMu sync.RWMutex + // filesToRechunk holds the IDs of known files that list a chunk no + // uploaded blob holds; they are re-chunked even when their metadata + // is unchanged. + filesToRechunk map[types.FileID]struct{} + // Pending chunk hashes - chunks that have been added to packer but not // yet committed to DB. When a blob finalizes, the committed chunks are // removed from this set. @@ -325,9 +330,42 @@ func (s *Scanner) loadDatabaseState( s.ui.Completef("Loaded %s known chunks from local index database.", s.ui.Count(len(s.knownChunks))) + err = s.loadFilesToRechunk(ctx, path) + if err != nil { + return nil, fmt.Errorf("loading files to re-chunk: %w", err) + } + return knownFiles, nil } +// loadFilesToRechunk loads the IDs of known files under path that list a +// chunk no uploaded blob holds. A file row is shared by every snapshot +// that lists the file and is updated in place when the file changes, +// while a blob row is deleted once no snapshot references it. Removing +// the only snapshot that references a changed file's current blob +// therefore leaves an older snapshot keeping a file row whose metadata +// matches the disk while no blob the local index records as uploaded +// holds its chunks, so a new snapshot cannot reference them. The dropped +// blob can still be in remote storage until prune removes it. +func (s *Scanner) loadFilesToRechunk(ctx context.Context, path string) error { + ids, err := s.repos.Files.ListIDsWithChunksNotInUploadedBlobs(ctx, path) + if err != nil { + return fmt.Errorf("listing files: %w", err) + } + + s.filesToRechunk = make(map[types.FileID]struct{}, len(ids)) + for _, id := range ids { + s.filesToRechunk[id] = struct{}{} + } + + if len(ids) > 0 { + log.Info("Re-chunking known files whose chunks are not all "+ + "in uploaded blobs", "files", len(ids)) + } + + return nil +} + // repairInterruptedBlobs discards blob rows left by a previous run whose // upload never completed. Such a blob has its chunks, blob_chunks, and // blobs rows committed to the local index before the upload is attempted, @@ -1208,6 +1246,11 @@ func (s *Scanner) checkFileInMemory( return file, true } + // No uploaded blob holds one of its chunks (see loadFilesToRechunk) + if _, rechunk := s.filesToRechunk[fileID]; rechunk { + return file, true + } + // Check if file has changed if existingFile.Size != file.Size || existingFile.MTime.Unix() != file.MTime.Unix() || diff --git a/internal/vaultik/changed_file_restore_test.go b/internal/vaultik/changed_file_restore_test.go new file mode 100644 index 0000000..de61170 --- /dev/null +++ b/internal/vaultik/changed_file_restore_test.go @@ -0,0 +1,234 @@ +package vaultik_test + +import ( + "context" + "crypto/rand" + "path/filepath" + "slices" + "strings" + "testing" + + "github.com/spf13/afero" + "github.com/stretchr/testify/require" + "sneak.berlin/go/vaultik/internal/chunker" + "sneak.berlin/go/vaultik/internal/config" + "sneak.berlin/go/vaultik/internal/database" + "sneak.berlin/go/vaultik/internal/log" + "sneak.berlin/go/vaultik/internal/storage" + "sneak.berlin/go/vaultik/internal/storage/faultstore" + "sneak.berlin/go/vaultik/internal/vaultik" +) + +// These tests cover https://git.eeqj.de/sneak/vaultik/issues/214: once +// the only snapshot referencing a changed file's current blob is +// dropped, an older snapshot still keeps the file's row, and the next +// backup must re-chunk the file instead of treating it as unchanged. +// +// Each backup run uses its own snapshot name over the same source +// directory, so the snapshot IDs differ without waiting for their +// one-second timestamps to tick over. File rows are keyed by path, so +// the runs share them exactly as runs of one snapshot name would. + +// changedFileConfig returns a full-backup config with the snapshot names +// "first", "second" and "third", all backing up dataDir. +func changedFileConfig(dataDir, dbPath string) *config.Config { + source := config.SnapshotConfig{Paths: []string{dataDir}} + + cfg := faultTestConfig() + cfg.IndexPath = dbPath + cfg.ChunkSize = config.Size(faultChunkSize) + cfg.Snapshots = map[string]config.SnapshotConfig{ + "first": source, + "second": source, + "third": source, + } + + return cfg +} + +// backUp runs a full backup of the named snapshot. +func backUp(v *vaultik.Vaultik, name string) error { + return v.CreateSnapshot(&vaultik.SnapshotCreateOptions{ + Cron: true, + Snapshots: []string{name}, + }) +} + +// appendedFileSize is the size of the file before the tests append to it. +// It is twice the largest chunk the chunker cuts, so the file's first +// chunk ends before the appended bytes and stays in a blob the first +// snapshot still references, while its new chunks are only in the blob +// that gets dropped. +const appendedFileSize = 2 * chunker.ChunkSizeSpread * faultChunkSize + +// appendRandomBytes appends n random bytes to the file at path, creating +// the file if it does not exist, and records the new content in files. +// Random content gives the chunker real cut points. +func appendRandomBytes( + t *testing.T, fs afero.Fs, files map[string][]byte, path string, n int64, +) { + t.Helper() + + added := make([]byte, n) + _, err := rand.Read(added) + require.NoError(t, err) + + content := slices.Concat(files[path], added) + require.NoError(t, afero.WriteFile(fs, path, content, 0o644)) + + files[path] = content +} + +// localSnapshotID returns the ID of the one snapshot in the local index +// that was backed up under name. +func localSnapshotID( + ctx context.Context, t *testing.T, + repos *database.Repositories, name string, +) string { + t.Helper() + + snapshots, err := repos.Snapshots.ListRecent(ctx, listRecentTestLimit) + require.NoError(t, err) + + var ids []string + + for _, s := range snapshots { + if strings.Contains(s.ID.String(), "_"+name+"_") { + ids = append(ids, s.ID.String()) + } + } + + require.Lenf(t, ids, 1, "expected one local snapshot named %q", name) + + return ids[0] +} + +// assertThirdSnapshotRestores closes the local index, then restores the +// snapshot named "third" from store alone and byte-compares every file +// against files. +func assertThirdSnapshotRestores( + ctx context.Context, t *testing.T, cfg *config.Config, + store storage.Storer, repos *database.Repositories, db *database.DB, + fs afero.Fs, restoreDir string, files map[string][]byte, +) { + t.Helper() + + id := localSnapshotID(ctx, t, repos, "third") + require.NoError(t, db.Close()) + + reader := newReaderVaultik(ctx, cfg, store, nil, fs) + require.NoError(t, reader.Restore(&vaultik.RestoreOptions{ + SnapshotID: id, + TargetDir: restoreDir, + Verify: true, + }), "the backup after the drop must restore the changed file") + + assertRestoredTree(t, fs, restoreDir, files) +} + +// Trigger 1: bytes are appended to the file, a second snapshot backs it +// up, and that snapshot is removed. The first snapshot keeps the file row, +// which now lists the appended content's chunks, while removal drops the +// blob that held them. +// +//nolint:paralleltest // installs the global logger via log.Initialize +func TestBackupAfterRemovingNewestSnapshotRestoresChangedFile(t *testing.T) { + log.Initialize(log.Config{}) + + fs := afero.NewOsFs() + tempDir := t.TempDir() + dataDir := filepath.Join(tempDir, "src") + storeDir := filepath.Join(tempDir, "remote") + restoreDir := filepath.Join(tempDir, "restored") + dbPath := filepath.Join(tempDir, "index.sqlite") + changedPath := filepath.Join(dataDir, "appended.bin") + + ctx := context.Background() + files := writeFaultSourceTree(t, fs, dataDir) + cfg := changedFileConfig(dataDir, dbPath) + + appendRandomBytes(t, fs, files, changedPath, appendedFileSize) + + store, err := storage.NewFileStorer(storeDir) + require.NoError(t, err) + + db, err := database.New(ctx, dbPath) + require.NoError(t, err) + + repos := database.NewRepositories(db) + v := newBackupVaultik(ctx, cfg, store, repos, db, fs) + + require.NoError(t, backUp(v, "first")) + + appendRandomBytes(t, fs, files, changedPath, faultChunkSize) + require.NoError(t, backUp(v, "second")) + + _, err = v.RemoveSnapshot(localSnapshotID(ctx, t, repos, "second"), + &vaultik.RemoveOptions{Force: true}) + require.NoError(t, err) + + require.NoError(t, backUp(v, "third")) + + assertThirdSnapshotRestores( + ctx, t, cfg, store, repos, db, fs, restoreDir, files) +} + +// Trigger 2: bytes are appended to the file and the run that backs it up +// is interrupted at the manifest upload, after its blobs were uploaded. +// The next run's prune drops that incomplete snapshot and its blob, while +// the first snapshot keeps the file row, which now lists the appended +// content's chunks. +// +//nolint:paralleltest // installs the global logger via log.Initialize +func TestBackupAfterInterruptedRunRestoresChangedFile(t *testing.T) { + log.Initialize(log.Config{}) + + fs := afero.NewOsFs() + tempDir := t.TempDir() + dataDir := filepath.Join(tempDir, "src") + storeDir := filepath.Join(tempDir, "remote") + restoreDir := filepath.Join(tempDir, "restored") + dbPath := filepath.Join(tempDir, "index.sqlite") + changedPath := filepath.Join(dataDir, "appended.bin") + + ctx := context.Background() + files := writeFaultSourceTree(t, fs, dataDir) + cfg := changedFileConfig(dataDir, dbPath) + + appendRandomBytes(t, fs, files, changedPath, appendedFileSize) + + inner, err := storage.NewFileStorer(storeDir) + require.NoError(t, err) + + failManifest := false + store := faultstore.New(inner) + store.OnPut = func(key string) faultstore.PutAction { + if failManifest && strings.HasSuffix(key, "manifest.json.zst") { + return faultstore.PutFail + } + + return faultstore.PutNormal + } + + db, err := database.New(ctx, dbPath) + require.NoError(t, err) + + repos := database.NewRepositories(db) + v := newBackupVaultik(ctx, cfg, store, repos, db, fs) + + require.NoError(t, backUp(v, "first")) + + appendRandomBytes(t, fs, files, changedPath, faultChunkSize) + + failManifest = true + + require.Error(t, backUp(v, "second"), + "the run must fail when its manifest upload fails") + + failManifest = false + + require.NoError(t, backUp(v, "third")) + + assertThirdSnapshotRestores( + ctx, t, cfg, inner, repos, db, fs, restoreDir, files) +}