Compare commits
3
Commits
3441eec48c
...
5707a03b9a
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5707a03b9a | ||
|
|
38ebfd843a | ||
|
|
6b7517a4dc |
@@ -25,6 +25,32 @@ release" is exactly the contradiction
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- 2026-09-21: Stopped an interrupted blob upload from making a later
|
||||
backup deduplicate against data that was never stored
|
||||
([issue #148](https://git.eeqj.de/sneak/vaultik/issues/148)). The
|
||||
packer commits a blob's `chunks`, `blob_chunks`, and `blobs` rows
|
||||
before the upload is attempted, so a failed upload left chunk rows
|
||||
behind and the next run skipped re-uploading them, producing a
|
||||
snapshot that reported success but could not be restored. A run now
|
||||
deduplicates only against chunks held by a blob whose `uploaded_ts` is
|
||||
set, and at startup drops any un-uploaded blob rows (and the chunks
|
||||
they orphan) so the affected data is re-chunked and re-uploaded. Blobs
|
||||
recorded with no remote backend are marked uploaded so this invariant
|
||||
holds uniformly.
|
||||
|
||||
- 2026-09-22: Made restore refuse any snapshot path that would write
|
||||
outside the target directory
|
||||
([issue #154](https://git.eeqj.de/sneak/vaultik/issues/154)).
|
||||
`restoreFile` and `verifyRestoredFiles` joined the stored path onto the
|
||||
target with no containment check, so a `..` segment or an absolute path
|
||||
escaped the target and a restored symlink could redirect a later child
|
||||
write anywhere on disk. Every stored path is now rejected unless
|
||||
`filepath.IsLocal` accepts it with the leading separator removed, and
|
||||
each existing ancestor directory below the target is `Lstat`ed to refuse
|
||||
descending through a symlink; honest symlinks pointing outside the tree
|
||||
are still written verbatim. age decryption proves a snapshot is
|
||||
readable, not honest, and restore usually runs as root.
|
||||
|
||||
- 2026-09-21: Stopped `--json` from silencing stderr diagnostics
|
||||
([issue #112](https://git.eeqj.de/sneak/vaultik/issues/112)). `--json`
|
||||
used to be folded into `Quiet`, which pinned the log level to `WARN`,
|
||||
|
||||
@@ -208,6 +208,30 @@ func (r *BlobRepository) DeleteOrphaned(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// DeleteUnuploaded deletes blob rows whose upload never completed
|
||||
// (uploaded_ts IS NULL) and returns how many were removed. Their
|
||||
// blob_chunks rows are removed by the ON DELETE CASCADE foreign key.
|
||||
// A blob is only ever attached to a snapshot once its upload has been
|
||||
// recorded, so an un-uploaded blob is never referenced by a completed
|
||||
// snapshot: dropping it discards chunk rows that point at data which
|
||||
// was never stored remotely, so the affected content is re-chunked and
|
||||
// re-uploaded on the next run.
|
||||
func (r *BlobRepository) DeleteUnuploaded(ctx context.Context) (int64, error) {
|
||||
query := `DELETE FROM blobs WHERE uploaded_ts IS NULL`
|
||||
|
||||
result, err := r.db.ExecWithLog(ctx, query)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("deleting un-uploaded blobs: %w", err)
|
||||
}
|
||||
|
||||
rowsAffected, _ := result.RowsAffected()
|
||||
if rowsAffected > 0 {
|
||||
log.Debug("Deleted un-uploaded blobs", "count", rowsAffected)
|
||||
}
|
||||
|
||||
return rowsAffected, nil
|
||||
}
|
||||
|
||||
// getOne fetches a single blob row matched on the given column, or
|
||||
// (nil, nil) when no row matches.
|
||||
func (r *BlobRepository) getOne(
|
||||
|
||||
@@ -7,12 +7,32 @@ import (
|
||||
|
||||
// List returns every chunk in the index, ordered by chunk hash.
|
||||
func (r *ChunkRepository) List(ctx context.Context) ([]*Chunk, error) {
|
||||
query := `
|
||||
return r.list(ctx, `
|
||||
SELECT chunk_hash, size
|
||||
FROM chunks
|
||||
ORDER BY chunk_hash
|
||||
`
|
||||
`)
|
||||
}
|
||||
|
||||
// ListInUploadedBlobs returns the chunks that are stored in a blob whose
|
||||
// upload has completed (uploaded_ts set), ordered by chunk hash. These
|
||||
// are the only chunks a backup may safely deduplicate against: a chunk
|
||||
// recorded solely in a blob that was never uploaded refers to data that
|
||||
// is not in remote storage, so trusting it would silently drop that data
|
||||
// from later snapshots.
|
||||
func (r *ChunkRepository) ListInUploadedBlobs(ctx context.Context) ([]*Chunk, error) {
|
||||
return r.list(ctx, `
|
||||
SELECT DISTINCT c.chunk_hash, c.size
|
||||
FROM chunks c
|
||||
JOIN blob_chunks bc ON c.chunk_hash = bc.chunk_hash
|
||||
JOIN blobs b ON bc.blob_id = b.id
|
||||
WHERE b.uploaded_ts IS NOT NULL
|
||||
ORDER BY c.chunk_hash
|
||||
`)
|
||||
}
|
||||
|
||||
// list runs a chunk-selecting query and scans the (chunk_hash, size) rows.
|
||||
func (r *ChunkRepository) list(ctx context.Context, query string) ([]*Chunk, error) {
|
||||
rows, err := r.db.conn.QueryContext(ctx, query)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("querying chunks: %w", err)
|
||||
|
||||
@@ -220,7 +220,14 @@ func (s *Scanner) Scan(
|
||||
defer s.progress.Stop()
|
||||
}
|
||||
|
||||
// Phase 0: Load known files and chunks from database into memory for fast lookup
|
||||
// Phase 0: Repair any state left by an interrupted previous run, then
|
||||
// load known files and chunks from the database into memory for fast
|
||||
// lookup.
|
||||
err := s.repairInterruptedBlobs(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
knownFiles, err := s.loadDatabaseState(ctx, path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -317,6 +324,38 @@ func (s *Scanner) loadDatabaseState(
|
||||
return knownFiles, nil
|
||||
}
|
||||
|
||||
// repairInterruptedBlobs discards blob rows left by a previous run whose
|
||||
// upload never completed. Such a blob has its chunks, blob_chunks, and
|
||||
// blobs rows committed to the local index before the upload is attempted,
|
||||
// so a crash or dropped connection mid-upload leaves them behind while the
|
||||
// data never reaches remote storage. Deduplicating against those chunks on
|
||||
// a later run would produce a snapshot that reports success but cannot be
|
||||
// restored. Dropping the un-uploaded blobs (their blob_chunks cascade) and
|
||||
// then any chunks left unreferenced forces the affected data to be
|
||||
// re-chunked and re-uploaded this run. A blob is attached to a snapshot
|
||||
// only once its upload is recorded, so this never touches a completed
|
||||
// snapshot's data.
|
||||
func (s *Scanner) repairInterruptedBlobs(ctx context.Context) error {
|
||||
removed, err := s.repos.Blobs.DeleteUnuploaded(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("removing un-uploaded blob records: %w", err)
|
||||
}
|
||||
|
||||
if removed == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
log.Warn("Discarded blob records from an interrupted previous run; "+
|
||||
"their data will be re-uploaded", "blobs", removed)
|
||||
|
||||
err = s.repos.Chunks.DeleteOrphaned(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("removing orphaned chunks: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// summarizeScanPhase calculates total size to process, updates progress tracking,
|
||||
// and prints the scan phase summary with file counts and sizes
|
||||
func (s *Scanner) summarizeScanPhase(
|
||||
@@ -392,11 +431,14 @@ func (s *Scanner) loadKnownFiles(
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// loadKnownChunks loads all known chunk hashes from the database into a
|
||||
// map for fast lookup. This avoids per-chunk database queries during file
|
||||
// processing.
|
||||
// loadKnownChunks loads the chunk hashes safe to deduplicate against into
|
||||
// an in-memory map for fast lookup, avoiding per-chunk database queries
|
||||
// during file processing. Only chunks held by a blob whose upload
|
||||
// completed are loaded: a chunk left behind by an interrupted upload
|
||||
// refers to data that never reached remote storage, and deduplicating
|
||||
// against it would silently produce an unrestorable snapshot.
|
||||
func (s *Scanner) loadKnownChunks(ctx context.Context) error {
|
||||
chunks, err := s.repos.Chunks.List(ctx)
|
||||
chunks, err := s.repos.Chunks.ListInUploadedBlobs(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("listing chunks: %w", err)
|
||||
}
|
||||
@@ -1401,7 +1443,17 @@ func (s *Scanner) finalizeProcessPhase(ctx context.Context, result *ScanResult)
|
||||
return fmt.Errorf("parsing blob ID: %w", err)
|
||||
}
|
||||
|
||||
// With no remote backend the blob's lifecycle ends here, so
|
||||
// mark it uploaded in the same transaction that attaches it to
|
||||
// the snapshot. This keeps the invariant that any blob a
|
||||
// snapshot references has uploaded_ts set, so deduplication and
|
||||
// interrupted-run repair treat these blobs as trustworthy.
|
||||
err = s.repos.WithTx(ctx, func(ctx context.Context, tx *sql.Tx) error {
|
||||
err := s.repos.Blobs.UpdateUploaded(ctx, tx, b.ID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marking blob uploaded: %w", err)
|
||||
}
|
||||
|
||||
return s.repos.Snapshots.AddBlob(ctx, tx, s.snapshotID, blobID,
|
||||
types.BlobHash(b.Hash))
|
||||
})
|
||||
|
||||
@@ -0,0 +1,209 @@
|
||||
// Package faultstore provides a storage.Storer wrapper that injects
|
||||
// faults on demand, so tests can reproduce the failure modes a real
|
||||
// backend exhibits: an upload that fails partway, a backend that reports
|
||||
// success while storing nothing, and reads that return corrupt or
|
||||
// truncated bytes. It is the seam called for by the fault-injection
|
||||
// tests (sneak/vaultik issue 72) and is meant to be reused by future
|
||||
// tests rather than re-implemented per case.
|
||||
//
|
||||
// The wrapper delegates every method to the inner Storer. Two hooks
|
||||
// change that: OnPut decides the fate of each write, and OnGet decides
|
||||
// how each read's bytes are returned. Both are keyed by the object key,
|
||||
// so a test can fault only blobs, only metadata, or a single object.
|
||||
package faultstore
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
|
||||
"sneak.berlin/go/vaultik/internal/storage"
|
||||
)
|
||||
|
||||
// ErrInjectedUpload is returned by a Put the OnPut hook chose to fail.
|
||||
var ErrInjectedUpload = errors.New("faultstore: injected upload failure")
|
||||
|
||||
// PutAction is the disposition OnPut assigns to a write.
|
||||
type PutAction int
|
||||
|
||||
const (
|
||||
// PutNormal writes through to the inner Storer.
|
||||
PutNormal PutAction = iota
|
||||
// PutFail reads part of the stream, then fails without storing the
|
||||
// object — a network upload that dies partway through.
|
||||
PutFail
|
||||
// PutSwallow reports success but stores nothing — a backend that
|
||||
// lies about durability.
|
||||
PutSwallow
|
||||
)
|
||||
|
||||
// GetFault is how OnGet chooses to damage a read.
|
||||
type GetFault int
|
||||
|
||||
const (
|
||||
// GetNormal returns the stored bytes unchanged.
|
||||
GetNormal GetFault = iota
|
||||
// GetCorrupt flips a byte so the returned object no longer matches
|
||||
// what was stored.
|
||||
GetCorrupt
|
||||
// GetTruncate returns a short read: the object's bytes cut off
|
||||
// before the end.
|
||||
GetTruncate
|
||||
)
|
||||
|
||||
// Storer wraps an inner storage.Storer with fault-injection hooks. A
|
||||
// zero-valued hook means "no fault": construct with New and set only the
|
||||
// hook a test needs.
|
||||
type Storer struct {
|
||||
inner storage.Storer
|
||||
|
||||
// OnPut, when set, is consulted before every Put and
|
||||
// PutWithProgress with the object key.
|
||||
OnPut func(key string) PutAction
|
||||
|
||||
// OnGet, when set, is consulted for every Get with the object key
|
||||
// and damages the returned bytes accordingly.
|
||||
OnGet func(key string) GetFault
|
||||
}
|
||||
|
||||
// New wraps inner. inner must be non-nil.
|
||||
func New(inner storage.Storer) *Storer {
|
||||
return &Storer{inner: inner}
|
||||
}
|
||||
|
||||
// midStreamBytes is how far a PutFail reads before failing, enough to be
|
||||
// past the start of any real blob without depending on the blob's size.
|
||||
const midStreamBytes = 512
|
||||
|
||||
// Put stores data unless OnPut faults the write.
|
||||
func (f *Storer) Put(ctx context.Context, key string, data io.Reader) error {
|
||||
handled, err := f.injectPut(key, data)
|
||||
if handled {
|
||||
return err
|
||||
}
|
||||
|
||||
return f.inner.Put(ctx, key, data)
|
||||
}
|
||||
|
||||
// PutWithProgress stores data unless OnPut faults the write.
|
||||
func (f *Storer) PutWithProgress(
|
||||
ctx context.Context, key string, data io.Reader,
|
||||
size int64, progress storage.ProgressCallback,
|
||||
) error {
|
||||
handled, err := f.injectPut(key, data)
|
||||
if handled {
|
||||
return err
|
||||
}
|
||||
|
||||
return f.inner.PutWithProgress(ctx, key, data, size, progress)
|
||||
}
|
||||
|
||||
// Get retrieves data, damaging it if OnGet faults the read.
|
||||
func (f *Storer) Get(ctx context.Context, key string) (io.ReadCloser, error) {
|
||||
rc, err := f.inner.Get(ctx, key)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
fault := GetNormal
|
||||
if f.OnGet != nil {
|
||||
fault = f.OnGet(key)
|
||||
}
|
||||
|
||||
if fault == GetNormal {
|
||||
return rc, nil
|
||||
}
|
||||
|
||||
data, err := io.ReadAll(rc)
|
||||
_ = rc.Close()
|
||||
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return io.NopCloser(bytes.NewReader(damage(fault, data))), nil
|
||||
}
|
||||
|
||||
// damage returns a faulted copy of the stored bytes. GetCorrupt flips a
|
||||
// byte in the middle so decryption authentication fails; GetTruncate
|
||||
// drops the final byte so the read ends short. Both are no-ops on empty
|
||||
// input, which cannot be damaged into something distinguishable.
|
||||
func damage(fault GetFault, data []byte) []byte {
|
||||
out := make([]byte, len(data))
|
||||
copy(out, data)
|
||||
|
||||
if len(out) == 0 {
|
||||
return out
|
||||
}
|
||||
|
||||
switch fault {
|
||||
case GetCorrupt:
|
||||
out[len(out)/2] ^= 0xff
|
||||
case GetTruncate:
|
||||
out = out[:len(out)-1]
|
||||
case GetNormal:
|
||||
}
|
||||
|
||||
return out
|
||||
}
|
||||
|
||||
// Stat delegates unchanged.
|
||||
func (f *Storer) Stat(ctx context.Context, key string) (*storage.ObjectInfo, error) {
|
||||
return f.inner.Stat(ctx, key)
|
||||
}
|
||||
|
||||
// Delete delegates unchanged.
|
||||
func (f *Storer) Delete(ctx context.Context, key string) error {
|
||||
return f.inner.Delete(ctx, key)
|
||||
}
|
||||
|
||||
// List delegates unchanged.
|
||||
func (f *Storer) List(ctx context.Context, prefix string) ([]string, error) {
|
||||
return f.inner.List(ctx, prefix)
|
||||
}
|
||||
|
||||
// ListStream delegates unchanged.
|
||||
func (f *Storer) ListStream(
|
||||
ctx context.Context, prefix string,
|
||||
) <-chan storage.ObjectInfo {
|
||||
return f.inner.ListStream(ctx, prefix)
|
||||
}
|
||||
|
||||
// Info delegates unchanged.
|
||||
func (f *Storer) Info() storage.Info {
|
||||
return f.inner.Info()
|
||||
}
|
||||
|
||||
func (f *Storer) putAction(key string) PutAction {
|
||||
if f.OnPut == nil {
|
||||
return PutNormal
|
||||
}
|
||||
|
||||
return f.OnPut(key)
|
||||
}
|
||||
|
||||
// injectPut handles the non-normal write dispositions. It reports
|
||||
// whether it handled the write and, if so, with what error.
|
||||
func (f *Storer) injectPut(key string, data io.Reader) (bool, error) {
|
||||
switch f.putAction(key) {
|
||||
case PutFail:
|
||||
// Consume part of the stream so the failure lands mid-transfer,
|
||||
// the way a dropped connection would, then error without
|
||||
// storing anything.
|
||||
_, _ = io.CopyN(io.Discard, data, midStreamBytes)
|
||||
|
||||
return true, fmt.Errorf("%w for %q", ErrInjectedUpload, key)
|
||||
case PutSwallow:
|
||||
// A lying backend still drains the request body, then keeps
|
||||
// nothing.
|
||||
_, _ = io.Copy(io.Discard, data)
|
||||
|
||||
return true, nil
|
||||
case PutNormal:
|
||||
return false, nil
|
||||
default:
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,647 @@
|
||||
package vaultik_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/spf13/afero"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"sneak.berlin/go/vaultik/internal/config"
|
||||
"sneak.berlin/go/vaultik/internal/database"
|
||||
"sneak.berlin/go/vaultik/internal/log"
|
||||
"sneak.berlin/go/vaultik/internal/snapshot"
|
||||
"sneak.berlin/go/vaultik/internal/storage"
|
||||
"sneak.berlin/go/vaultik/internal/storage/faultstore"
|
||||
"sneak.berlin/go/vaultik/internal/ui"
|
||||
"sneak.berlin/go/vaultik/internal/vaultik"
|
||||
)
|
||||
|
||||
// These tests cover the failure modes a backup tool must survive:
|
||||
// interrupted uploads, an interrupted metadata export, corrupt and
|
||||
// truncated reads, a full restore disk, and a backend that reports
|
||||
// success while storing nothing. Faults are injected through the
|
||||
// storage.Storer seam (internal/storage/faultstore), never by patching
|
||||
// production code. Each test asserts on the observable end state — what
|
||||
// is in the index, what is at the destination, what the user is told —
|
||||
// not merely that an error was returned. See
|
||||
// https://git.eeqj.de/sneak/vaultik/issues/72.
|
||||
//
|
||||
// Object-level write atomicity (no partial blob object left behind) is
|
||||
// covered by the file:// backend's atomic-write work
|
||||
// (https://git.eeqj.de/sneak/vaultik/issues/130) and is not re-tested
|
||||
// here; these tests target the layers above the backend.
|
||||
//
|
||||
// The tests run serially, not with t.Parallel: each calls
|
||||
// log.Initialize, which replaces the package-global logger, and a
|
||||
// backup or restore running concurrently reads that same logger. Under
|
||||
// -race the two collide. Running one at a time is the same choice
|
||||
// prune_count_test.go already makes for the same reason.
|
||||
|
||||
const (
|
||||
faultChunkSize = int64(64 * 1024)
|
||||
faultMaxBlobSize = int64(256 * 1024)
|
||||
)
|
||||
|
||||
// faultTestConfig returns the config shared by the fault-injection
|
||||
// tests: a real recipient/secret keypair so blobs are genuinely
|
||||
// encrypted, and a blob size limit the restore sweeper can divide.
|
||||
func faultTestConfig() *config.Config {
|
||||
return &config.Config{
|
||||
AgeRecipients: []string{testAgePublicKey},
|
||||
AgeSecretKey: testAgeSecretKey,
|
||||
CompressionLevel: 3,
|
||||
Hostname: testHostname,
|
||||
BlobSizeLimit: config.Size(faultMaxBlobSize),
|
||||
}
|
||||
}
|
||||
|
||||
// writeFaultSourceTree writes a spread of file sizes that forces several
|
||||
// chunks across more than one blob, so a fault landing on a single blob
|
||||
// still leaves other data intact. Returns the expected content by path.
|
||||
func writeFaultSourceTree(
|
||||
t *testing.T, fs afero.Fs, dataDir string,
|
||||
) map[string][]byte {
|
||||
t.Helper()
|
||||
|
||||
files := map[string][]byte{
|
||||
filepath.Join(dataDir, "small.txt"): []byte("hello vaultik"),
|
||||
filepath.Join(dataDir, "a.bin"): bytesPattern("a-", int(faultChunkSize*3)),
|
||||
filepath.Join(dataDir, "sub", "b.bin"): bytesPattern("b-", int(faultChunkSize*3)),
|
||||
filepath.Join(dataDir, "sub", "c.bin"): bytesPattern("c-", int(faultChunkSize*2)),
|
||||
}
|
||||
|
||||
for path, content := range files {
|
||||
require.NoError(t, fs.MkdirAll(filepath.Dir(path), 0o755))
|
||||
require.NoError(t, afero.WriteFile(fs, path, content, 0o644))
|
||||
}
|
||||
|
||||
return files
|
||||
}
|
||||
|
||||
// newFaultScanner builds a scanner writing through the given storer.
|
||||
func newFaultScanner(
|
||||
fs afero.Fs, storer storage.Storer,
|
||||
cfg *config.Config, repos *database.Repositories,
|
||||
) *snapshot.Scanner {
|
||||
return snapshot.NewScanner(snapshot.ScannerConfig{
|
||||
FS: fs,
|
||||
Storage: storer,
|
||||
ChunkSize: faultChunkSize,
|
||||
MaxBlobSize: faultMaxBlobSize,
|
||||
CompressionLevel: cfg.CompressionLevel,
|
||||
AgeRecipients: cfg.AgeRecipients,
|
||||
Repositories: repos,
|
||||
})
|
||||
}
|
||||
|
||||
// newFaultSnapshotManager builds a snapshot manager writing through the
|
||||
// given storer.
|
||||
func newFaultSnapshotManager(
|
||||
fs afero.Fs, storer storage.Storer,
|
||||
cfg *config.Config, repos *database.Repositories,
|
||||
) *snapshot.SnapshotManager {
|
||||
sm := snapshot.NewSnapshotManager(snapshot.SnapshotManagerParams{
|
||||
Repos: repos,
|
||||
Storage: storer,
|
||||
Config: cfg,
|
||||
})
|
||||
sm.SetFilesystem(fs)
|
||||
|
||||
return sm
|
||||
}
|
||||
|
||||
// fullFaultBackup runs a complete backup (create, scan, complete,
|
||||
// export) through storer and returns the snapshot ID.
|
||||
func fullFaultBackup(
|
||||
ctx context.Context, t *testing.T, fs afero.Fs, storer storage.Storer,
|
||||
cfg *config.Config, repos *database.Repositories,
|
||||
dataDir, dbPath, name string,
|
||||
) string {
|
||||
t.Helper()
|
||||
|
||||
sm := newFaultSnapshotManager(fs, storer, cfg, repos)
|
||||
scanner := newFaultScanner(fs, storer, cfg, repos)
|
||||
|
||||
id, err := sm.CreateSnapshotWithName(ctx, cfg.Hostname, name, "v", "g")
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = scanner.Scan(ctx, dataDir, id)
|
||||
require.NoError(t, err)
|
||||
|
||||
require.NoError(t, sm.CompleteSnapshot(ctx, id))
|
||||
require.NoError(t, sm.ExportSnapshotMetadata(ctx, dbPath, id))
|
||||
|
||||
return id
|
||||
}
|
||||
|
||||
// newReaderVaultik builds a Vaultik that reads (restore/verify) through
|
||||
// storer, with the given repositories (nil is fine for restore/verify,
|
||||
// which read metadata from storage).
|
||||
func newReaderVaultik(
|
||||
ctx context.Context, cfg *config.Config, storer storage.Storer,
|
||||
repos *database.Repositories, fs afero.Fs,
|
||||
) *vaultik.Vaultik {
|
||||
v := &vaultik.Vaultik{
|
||||
Config: cfg,
|
||||
Storage: storer,
|
||||
Repositories: repos,
|
||||
Fs: fs,
|
||||
Stdout: io.Discard,
|
||||
Stderr: io.Discard,
|
||||
UI: ui.NewWithColor(io.Discard, false),
|
||||
}
|
||||
v.SetContext(ctx)
|
||||
|
||||
return v
|
||||
}
|
||||
|
||||
// Scenario 3: a stored blob's bytes are flipped before restore reads
|
||||
// them. Restore must fail loudly, and no file must be left on the
|
||||
// restore target holding corrupt content.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestRestoreRejectsCorruptBlob(t *testing.T) {
|
||||
assertRestoreRejectsDamagedBlob(t, faultstore.GetCorrupt, "corrupt")
|
||||
}
|
||||
|
||||
// Scenario 4: a stored blob is truncated before restore reads it. Same
|
||||
// contract as the corrupt case.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestRestoreRejectsTruncatedBlob(t *testing.T) {
|
||||
assertRestoreRejectsDamagedBlob(t, faultstore.GetTruncate, "truncated")
|
||||
}
|
||||
|
||||
// assertRestoreRejectsDamagedBlob backs up the source tree, then restores
|
||||
// through a store that damages every blob read with the given fault, and
|
||||
// asserts restore fails naming a blob and leaves no file on the target
|
||||
// holding wrong bytes. Metadata reads are returned intact so the failure
|
||||
// is isolated to the blob.
|
||||
func assertRestoreRejectsDamagedBlob(
|
||||
t *testing.T, fault faultstore.GetFault, name string,
|
||||
) {
|
||||
t.Helper()
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
restoreDir := filepath.Join(tempDir, "restored")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
cfg := faultTestConfig()
|
||||
testFiles := writeFaultSourceTree(t, fs, dataDir)
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
id := fullFaultBackup(ctx, t, fs, inner, cfg, repos, dataDir, dbPath, name)
|
||||
require.NoError(t, db.Close())
|
||||
|
||||
faultStore := faultstore.New(inner)
|
||||
faultStore.OnGet = func(key string) faultstore.GetFault {
|
||||
if strings.HasPrefix(key, "blobs/") {
|
||||
return fault
|
||||
}
|
||||
|
||||
return faultstore.GetNormal
|
||||
}
|
||||
|
||||
v := newReaderVaultik(ctx, cfg, faultStore, nil, fs)
|
||||
err = v.Restore(&vaultik.RestoreOptions{SnapshotID: id, TargetDir: restoreDir})
|
||||
|
||||
require.Error(t, err, "restore must fail on a damaged blob")
|
||||
assert.Contains(t, err.Error(), "blob",
|
||||
"error should name the blob that failed")
|
||||
assertNoCorruptFiles(t, fs, restoreDir, testFiles)
|
||||
}
|
||||
|
||||
// Scenario 6: the backend accepts blob uploads and reports success but
|
||||
// stores nothing. verify --deep must catch it.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestDeepVerifyCatchesLyingBackend(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
cfg := faultTestConfig()
|
||||
|
||||
writeFaultSourceTree(t, fs, dataDir)
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Blob uploads are swallowed; metadata uploads land, so verify can
|
||||
// download the manifest and database and then discover the blobs are
|
||||
// absent.
|
||||
lying := faultstore.New(inner)
|
||||
lying.OnPut = func(key string) faultstore.PutAction {
|
||||
if strings.HasPrefix(key, "blobs/") {
|
||||
return faultstore.PutSwallow
|
||||
}
|
||||
|
||||
return faultstore.PutNormal
|
||||
}
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
id := fullFaultBackup(ctx, t, fs, lying, cfg, repos, dataDir, dbPath, "lying")
|
||||
require.NoError(t, db.Close())
|
||||
|
||||
// No blob objects were actually written.
|
||||
blobKeys, err := inner.List(ctx, "blobs/")
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, blobKeys, "lying backend should have stored no blobs")
|
||||
|
||||
// Read back through the honest underlying store.
|
||||
v := newReaderVaultik(ctx, cfg, inner, nil, fs)
|
||||
err = v.VerifySnapshotWithOptions(id, &vaultik.VerifyOptions{Deep: true})
|
||||
require.Error(t, err, "deep verify must catch a backend that stored nothing")
|
||||
}
|
||||
|
||||
// Scenario 1a: a blob upload fails partway through. The interrupted run
|
||||
// must not record the blob as uploaded, must not reference it from the
|
||||
// snapshot, and must leave no blob object at the destination.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestInterruptedBlobUploadRecordsNoUploadedBlob(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
cfg := faultTestConfig()
|
||||
|
||||
writeFaultSourceTree(t, fs, dataDir)
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
// Every blob upload fails partway through. The scan must surface it.
|
||||
fault := faultstore.New(inner)
|
||||
fault.OnPut = func(key string) faultstore.PutAction {
|
||||
if strings.HasPrefix(key, "blobs/") {
|
||||
return faultstore.PutFail
|
||||
}
|
||||
|
||||
return faultstore.PutNormal
|
||||
}
|
||||
|
||||
sm := newFaultSnapshotManager(fs, fault, cfg, repos)
|
||||
scanner := newFaultScanner(fs, fault, cfg, repos)
|
||||
|
||||
id, err := sm.CreateSnapshotWithName(ctx, cfg.Hostname, "interrupted", "v", "g")
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = scanner.Scan(ctx, dataDir, id)
|
||||
require.Error(t, err, "scan must fail when a blob upload fails")
|
||||
|
||||
// No blob may claim to be uploaded.
|
||||
blobs, err := repos.Blobs.GetAll(ctx)
|
||||
require.NoError(t, err)
|
||||
|
||||
for _, b := range blobs {
|
||||
assert.Nilf(t, b.UploadedTS,
|
||||
"blob %s marked uploaded after a failed upload", b.Hash)
|
||||
}
|
||||
|
||||
// The snapshot may reference no blobs, and the destination holds none.
|
||||
hashes, err := repos.Snapshots.GetBlobHashes(ctx, id)
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, hashes, "interrupted snapshot must reference no blobs")
|
||||
|
||||
blobKeys, err := inner.List(ctx, "blobs/")
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, blobKeys, "no blob object may survive at the destination")
|
||||
}
|
||||
|
||||
// Scenario 1b: after an interrupted upload, a retry on the same local
|
||||
// index must produce a restorable snapshot. The interrupted run leaves
|
||||
// the blob's chunk rows in the index; the fix for
|
||||
// https://git.eeqj.de/sneak/vaultik/issues/148 discards those un-uploaded
|
||||
// blob rows at the start of the next scan and deduplicates only against
|
||||
// chunks in a blob that was actually uploaded, so the retry re-chunks and
|
||||
// re-uploads the affected data instead of silently referencing data that
|
||||
// never reached storage.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestBackupRetryAfterInterruptedUploadIsRestorable(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
restoreDir := filepath.Join(tempDir, "restored")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
cfg := faultTestConfig()
|
||||
testFiles := writeFaultSourceTree(t, fs, dataDir)
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
// Attempt 1: every blob upload fails.
|
||||
fault := faultstore.New(inner)
|
||||
fault.OnPut = func(key string) faultstore.PutAction {
|
||||
if strings.HasPrefix(key, "blobs/") {
|
||||
return faultstore.PutFail
|
||||
}
|
||||
|
||||
return faultstore.PutNormal
|
||||
}
|
||||
|
||||
sm := newFaultSnapshotManager(fs, fault, cfg, repos)
|
||||
scanner := newFaultScanner(fs, fault, cfg, repos)
|
||||
|
||||
id1, err := sm.CreateSnapshotWithName(ctx, cfg.Hostname, "interrupted", "v", "g")
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = scanner.Scan(ctx, dataDir, id1)
|
||||
require.Error(t, err)
|
||||
|
||||
// Retry on the same local index with a working backend.
|
||||
id2 := fullFaultBackup(ctx, t, fs, inner, cfg, repos, dataDir, dbPath, "retry")
|
||||
require.NoError(t, db.Close())
|
||||
|
||||
v := newReaderVaultik(ctx, cfg, inner, nil, fs)
|
||||
require.NoError(t, v.Restore(&vaultik.RestoreOptions{
|
||||
SnapshotID: id2,
|
||||
TargetDir: restoreDir,
|
||||
Verify: true,
|
||||
}), "retry after an interrupted upload must produce a restorable snapshot")
|
||||
|
||||
assertRestoredTree(t, fs, restoreDir, testFiles)
|
||||
}
|
||||
|
||||
// Scenario 2: the process dies during the metadata export, after the
|
||||
// database is uploaded but before the manifest. The destination is left
|
||||
// with blobs and a database but no manifest. verify and snapshot list
|
||||
// must report the damage honestly rather than crashing or passing.
|
||||
// Automatic detection and repair of this partial state on the next run
|
||||
// is tracked in https://git.eeqj.de/sneak/vaultik/issues/177 and is not
|
||||
// asserted here.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestBackupSurvivesMetadataExportInterruption(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
cfg := faultTestConfig()
|
||||
|
||||
writeFaultSourceTree(t, fs, dataDir)
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
// Back up and complete with a working backend.
|
||||
sm := newFaultSnapshotManager(fs, inner, cfg, repos)
|
||||
scanner := newFaultScanner(fs, inner, cfg, repos)
|
||||
|
||||
id, err := sm.CreateSnapshotWithName(ctx, cfg.Hostname, "export", "v", "g")
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = scanner.Scan(ctx, dataDir, id)
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, sm.CompleteSnapshot(ctx, id))
|
||||
|
||||
// Export through a backend that fails only the manifest upload. The
|
||||
// database uploads first and lands; the manifest does not.
|
||||
fault := faultstore.New(inner)
|
||||
fault.OnPut = func(key string) faultstore.PutAction {
|
||||
if strings.HasSuffix(key, "manifest.json.zst") {
|
||||
return faultstore.PutFail
|
||||
}
|
||||
|
||||
return faultstore.PutNormal
|
||||
}
|
||||
|
||||
smFault := newFaultSnapshotManager(fs, fault, cfg, repos)
|
||||
|
||||
err = smFault.ExportSnapshotMetadata(ctx, dbPath, id)
|
||||
require.Error(t, err, "export must fail when the manifest upload fails")
|
||||
|
||||
// The destination is in the partial state the scenario describes.
|
||||
key := snapshot.RemoteSnapshotKey(id)
|
||||
|
||||
_, err = inner.Stat(ctx, "metadata/"+key+"/db.zst.age")
|
||||
require.NoError(t, err, "database should have been uploaded before the manifest")
|
||||
|
||||
_, err = inner.Stat(ctx, "metadata/"+key+"/manifest.json.zst")
|
||||
require.ErrorIs(t, err, storage.ErrNotFound, "manifest upload should not have landed")
|
||||
|
||||
// verify must fail loudly for this snapshot, in both modes.
|
||||
reader := newReaderVaultik(ctx, cfg, inner, repos, fs)
|
||||
|
||||
deepOpts := &vaultik.VerifyOptions{Deep: true}
|
||||
require.Error(t, reader.VerifySnapshotWithOptions(id, deepOpts),
|
||||
"deep verify must report the missing manifest")
|
||||
|
||||
shallowOpts := &vaultik.VerifyOptions{Deep: false}
|
||||
require.Error(t, reader.VerifySnapshotWithOptions(id, shallowOpts),
|
||||
"shallow verify must report the missing manifest")
|
||||
|
||||
// snapshot list must not crash on the partial snapshot.
|
||||
require.NoError(t, reader.ListSnapshots(false),
|
||||
"snapshot list must tolerate a partially-exported snapshot")
|
||||
}
|
||||
|
||||
// Scenario 5: the restore target runs out of space mid-file. Restore
|
||||
// must fail with an out-of-space error, and must not leave a truncated
|
||||
// file at the target path presenting as a complete restore. Restore
|
||||
// today writes each file straight to its final path and does not remove
|
||||
// it when a write fails, so the truncated file survives; deleting it is
|
||||
// tracked by https://git.eeqj.de/sneak/vaultik/issues/163. Skipped until
|
||||
// that lands, so the destination assertion below is recorded rather than
|
||||
// dropped.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestRestoreReportsDiskFull(t *testing.T) {
|
||||
t.Skip("blocked on https://git.eeqj.de/sneak/vaultik/issues/163: " +
|
||||
"a disk-full write leaves a truncated file at the target path " +
|
||||
"instead of removing it")
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
osFS := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
dataDir := filepath.Join(tempDir, "src")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
restoreDir := filepath.Join(tempDir, "restored")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
ctx := context.Background()
|
||||
cfg := faultTestConfig()
|
||||
|
||||
testFiles := writeFaultSourceTree(t, osFS, dataDir)
|
||||
|
||||
inner, err := storage.NewFileStorer(storeDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
db, err := database.New(ctx, dbPath)
|
||||
require.NoError(t, err)
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
|
||||
id := fullFaultBackup(ctx, t, osFS, inner, cfg, repos, dataDir, dbPath, "diskfull")
|
||||
require.NoError(t, db.Close())
|
||||
|
||||
// Restore onto a filesystem that allows only a few bytes of file
|
||||
// content: enough to create files, far too little to hold them.
|
||||
budget := int64(8)
|
||||
quota := "aFS{Fs: osFS, remaining: &budget}
|
||||
|
||||
v := newReaderVaultik(ctx, cfg, inner, nil, quota)
|
||||
err = v.Restore(&vaultik.RestoreOptions{SnapshotID: id, TargetDir: restoreDir})
|
||||
|
||||
require.Error(t, err, "restore must fail when the target disk is full")
|
||||
assert.Contains(t, err.Error(), errNoSpace.Error(),
|
||||
"restore error should surface the out-of-space cause")
|
||||
|
||||
// The failure must not leave a truncated file behind presenting as a
|
||||
// complete restore: any file at the target must hold the original
|
||||
// bytes, or be absent.
|
||||
assertNoCorruptFiles(t, osFS, restoreDir, testFiles)
|
||||
}
|
||||
|
||||
// assertRestoredTree byte-compares every restored file against the
|
||||
// original.
|
||||
func assertRestoredTree(
|
||||
t *testing.T, fs afero.Fs, restoreDir string, testFiles map[string][]byte,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
for origPath, expected := range testFiles {
|
||||
restoredPath := filepath.Join(restoreDir, origPath)
|
||||
got, err := afero.ReadFile(fs, restoredPath)
|
||||
require.NoErrorf(t, err, "restored file missing: %s", origPath)
|
||||
require.Equalf(t, expected, got, "restored content mismatch for %s", origPath)
|
||||
}
|
||||
}
|
||||
|
||||
// errNoSpace is the out-of-space error quotaFS returns once its byte
|
||||
// budget is exhausted, mirroring a real ENOSPC.
|
||||
var errNoSpace = errors.New("no space left on device")
|
||||
|
||||
// quotaFS is an afero.Fs whose files may write only a fixed total number
|
||||
// of content bytes before failing, simulating a full restore target. It
|
||||
// wraps the interface so every method except Create delegates to the
|
||||
// real filesystem; only file writes are capped.
|
||||
type quotaFS struct {
|
||||
afero.Fs
|
||||
|
||||
remaining *int64
|
||||
}
|
||||
|
||||
//nolint:ireturn // afero.Fs.Create's signature requires returning afero.File.
|
||||
func (q *quotaFS) Create(name string) (afero.File, error) {
|
||||
f, err := q.Fs.Create(name)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return "aFile{File: f, remaining: q.remaining}, nil
|
||||
}
|
||||
|
||||
// quotaFile fails writes once the shared byte budget is exhausted.
|
||||
type quotaFile struct {
|
||||
afero.File
|
||||
|
||||
remaining *int64
|
||||
}
|
||||
|
||||
func (q *quotaFile) Write(p []byte) (int, error) {
|
||||
if *q.remaining <= 0 {
|
||||
return 0, errNoSpace
|
||||
}
|
||||
|
||||
allowed := min(int64(len(p)), *q.remaining)
|
||||
|
||||
n, err := q.File.Write(p[:allowed])
|
||||
*q.remaining -= int64(n)
|
||||
|
||||
if err != nil {
|
||||
return n, err
|
||||
}
|
||||
|
||||
if int64(n) < int64(len(p)) {
|
||||
return n, errNoSpace
|
||||
}
|
||||
|
||||
return n, nil
|
||||
}
|
||||
|
||||
// assertNoCorruptFiles fails if any file that made it to the restore
|
||||
// target holds content that differs from the original: a failed restore
|
||||
// may leave a file absent, but must never leave wrong bytes presenting
|
||||
// as the real file.
|
||||
func assertNoCorruptFiles(
|
||||
t *testing.T, fs afero.Fs, restoreDir string, testFiles map[string][]byte,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
for origPath, expected := range testFiles {
|
||||
restoredPath := filepath.Join(restoreDir, origPath)
|
||||
|
||||
got, err := afero.ReadFile(fs, restoredPath)
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
continue
|
||||
}
|
||||
|
||||
require.NoError(t, err)
|
||||
}
|
||||
|
||||
assert.Equalf(t, expected, got,
|
||||
"restored file %s holds corrupt content", origPath)
|
||||
}
|
||||
}
|
||||
+91
-11
@@ -11,6 +11,7 @@ import (
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"filippo.io/age"
|
||||
@@ -30,10 +31,12 @@ var (
|
||||
"Set the VAULTIK_AGE_SECRET_KEY environment variable to your " +
|
||||
"age private key:\n" +
|
||||
" export VAULTIK_AGE_SECRET_KEY='AGE-SECRET-KEY-...'")
|
||||
errBlobMissingFromIndex = errors.New("blob hash missing from blob index")
|
||||
errChunkNotInAnyBlob = errors.New("chunk not found in any blob")
|
||||
errBlobIDNotInHashIndex = errors.New("blob id missing from hash index")
|
||||
errShortChunkRead = errors.New("short read")
|
||||
errBlobMissingFromIndex = errors.New("blob hash missing from blob index")
|
||||
errChunkNotInAnyBlob = errors.New("chunk not found in any blob")
|
||||
errBlobIDNotInHashIndex = errors.New("blob id missing from hash index")
|
||||
errShortChunkRead = errors.New("short read")
|
||||
errRestorePathEscapesTarget = errors.New(
|
||||
"refusing to restore path outside the target directory")
|
||||
)
|
||||
|
||||
// restoreDirMode is the permission mode for directories created while
|
||||
@@ -760,13 +763,85 @@ type restoreSession struct {
|
||||
runningAsRoot bool
|
||||
}
|
||||
|
||||
// containedRestorePath resolves rel — a path read from the snapshot
|
||||
// database — to its location under targetDir and confirms the write will
|
||||
// stay inside the target.
|
||||
//
|
||||
// age decryption proves a snapshot is readable, not that it is honest, so
|
||||
// every stored path is treated as hostile. rel is rejected unless
|
||||
// filepath.IsLocal accepts it once the leading separator is stripped:
|
||||
// stored paths are absolute and the join to targetDir drops that
|
||||
// separator, so "/etc/passwd" is judged as the relative "etc/passwd" it
|
||||
// becomes on disk. This bars "..", absolute, and empty paths.
|
||||
//
|
||||
// A stored symlink whose target points outside the tree is still honest
|
||||
// (and restored verbatim), but a later entry must not be written through
|
||||
// it. Each existing ancestor directory below the target is therefore
|
||||
// Lstat'ed and a symlink among them is refused. The leaf itself is not
|
||||
// traversed: honest snapshots restore symlinks at leaf positions, and the
|
||||
// unique-path constraint keeps a leaf from being both a symlink and a
|
||||
// regular file. The target directory itself may be a symlink; only
|
||||
// components below it are checked.
|
||||
func containedRestorePath(fs afero.Fs, targetDir, rel string) (string, error) {
|
||||
local := strings.TrimPrefix(rel, string(filepath.Separator))
|
||||
if !filepath.IsLocal(local) {
|
||||
return "", fmt.Errorf("%w: %s", errRestorePathEscapesTarget, rel)
|
||||
}
|
||||
|
||||
local = filepath.Clean(local)
|
||||
targetPath := filepath.Join(targetDir, local)
|
||||
|
||||
relDir := filepath.Dir(local)
|
||||
if relDir == "." {
|
||||
return targetPath, nil
|
||||
}
|
||||
|
||||
current := targetDir
|
||||
for component := range strings.SplitSeq(relDir, string(filepath.Separator)) {
|
||||
current = filepath.Join(current, component)
|
||||
|
||||
info, err := lstatIfPossible(fs, current)
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
continue
|
||||
}
|
||||
|
||||
return "", fmt.Errorf("checking restore path %s: %w", current, err)
|
||||
}
|
||||
|
||||
if info.Mode()&os.ModeSymlink != 0 {
|
||||
return "", fmt.Errorf("%w: %s descends through symlink %s",
|
||||
errRestorePathEscapesTarget, rel, current)
|
||||
}
|
||||
}
|
||||
|
||||
return targetPath, nil
|
||||
}
|
||||
|
||||
// lstatIfPossible performs a symlink-aware stat when the filesystem
|
||||
// supports it. afero.OsFs does; MemMapFs, which has no symlinks, reports
|
||||
// that Lstat was not used and its result never carries ModeSymlink.
|
||||
func lstatIfPossible(fs afero.Fs, name string) (os.FileInfo, error) {
|
||||
if lstater, ok := fs.(afero.Lstater); ok {
|
||||
info, _, err := lstater.LstatIfPossible(name)
|
||||
|
||||
return info, err
|
||||
}
|
||||
|
||||
return fs.Stat(name)
|
||||
}
|
||||
|
||||
// restoreFile dispatches to the right per-kind restorer.
|
||||
func (s *restoreSession) restoreFile(file *database.File) error {
|
||||
targetPath := filepath.Join(s.opts.TargetDir, file.Path.String())
|
||||
targetPath, err := containedRestorePath(
|
||||
s.v.Fs, s.opts.TargetDir, file.Path.String())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
parentDir := filepath.Dir(targetPath)
|
||||
|
||||
err := s.v.Fs.MkdirAll(parentDir, restoreDirMode)
|
||||
err = s.v.Fs.MkdirAll(parentDir, restoreDirMode)
|
||||
if err != nil {
|
||||
return fmt.Errorf("creating parent directory: %w", err)
|
||||
}
|
||||
@@ -1062,17 +1137,22 @@ func (v *Vaultik) verifyRestoredFiles(
|
||||
return ctx.Err()
|
||||
}
|
||||
|
||||
targetPath := filepath.Join(targetDir, file.Path.String())
|
||||
targetPath, err := containedRestorePath(v.Fs, targetDir, file.Path.String())
|
||||
if err == nil {
|
||||
var bytesVerified int64
|
||||
|
||||
bytesVerified, err = v.verifyFile(ctx, repos, file, targetPath)
|
||||
if err == nil {
|
||||
result.FilesVerified++
|
||||
result.BytesVerified += bytesVerified
|
||||
}
|
||||
}
|
||||
|
||||
bytesVerified, err := v.verifyFile(ctx, repos, file, targetPath)
|
||||
if err != nil {
|
||||
log.Error("File verification failed", "path", file.Path, "error", err)
|
||||
|
||||
result.FilesFailed++
|
||||
result.FailedFiles = append(result.FailedFiles, file.Path.String())
|
||||
} else {
|
||||
result.FilesVerified++
|
||||
result.BytesVerified += bytesVerified
|
||||
}
|
||||
|
||||
bytesProcessed += file.Size
|
||||
|
||||
@@ -0,0 +1,175 @@
|
||||
package vaultik //nolint:testpackage // drives unexported restore internals
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/afero"
|
||||
"github.com/stretchr/testify/require"
|
||||
"sneak.berlin/go/vaultik/internal/config"
|
||||
"sneak.berlin/go/vaultik/internal/database"
|
||||
"sneak.berlin/go/vaultik/internal/log"
|
||||
"sneak.berlin/go/vaultik/internal/types"
|
||||
"sneak.berlin/go/vaultik/internal/ui"
|
||||
)
|
||||
|
||||
// These tests exercise the path-containment guard that keeps restore from
|
||||
// writing outside its target directory. age decryption proves only that a
|
||||
// snapshot is readable, not that its recorded paths are honest, so restore
|
||||
// treats every stored path as hostile: a compromised backed-up host could
|
||||
// forge a snapshot that decrypts cleanly, and restore usually runs as root.
|
||||
//
|
||||
// They drive restoreAllFiles directly (rather than the full Restore, which
|
||||
// downloads and decrypts the metadata database from storage) so a snapshot
|
||||
// database with adversarial rows can be handed to the restore loop without
|
||||
// the surrounding blob/storage machinery. Directory and symlink entries
|
||||
// carry no chunks, so no blobs are needed.
|
||||
|
||||
// containmentDirMode marks a File row as a directory for the restore loop.
|
||||
const containmentDirMode = uint32(os.ModeDir | 0o755)
|
||||
|
||||
// newContainmentVaultik builds the minimal Vaultik needed to run
|
||||
// restoreAllFiles against fs.
|
||||
func newContainmentVaultik(ctx context.Context, fs afero.Fs) *Vaultik {
|
||||
v := &Vaultik{
|
||||
Config: &config.Config{
|
||||
BlobSizeLimit: config.Size(10 * 1024 * 1024),
|
||||
},
|
||||
Fs: fs,
|
||||
Stdout: io.Discard,
|
||||
Stderr: io.Discard,
|
||||
UI: ui.NewWithColor(io.Discard, false),
|
||||
}
|
||||
v.SetContext(ctx)
|
||||
|
||||
return v
|
||||
}
|
||||
|
||||
// makeFiles inserts the given rows into a fresh in-memory snapshot database
|
||||
// and returns them (with IDs assigned) plus the repositories.
|
||||
func makeFiles(
|
||||
ctx context.Context, t *testing.T, rows []*database.File,
|
||||
) ([]*database.File, *database.Repositories) {
|
||||
t.Helper()
|
||||
|
||||
db, err := database.New(ctx, filepath.Join(t.TempDir(), "index.sqlite"))
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
repos := database.NewRepositories(db)
|
||||
for _, f := range rows {
|
||||
require.NoError(t, repos.Files.Create(ctx, nil, f))
|
||||
}
|
||||
|
||||
return rows, repos
|
||||
}
|
||||
|
||||
func TestRestoreRejectsPathTraversal(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
// rows are inserted in order; the escape entry is restored after
|
||||
// any entry it depends on (the symlink case needs its link first).
|
||||
rows func(outsideDir string) []*database.File
|
||||
// escaped is the path, outside the target, that must not appear.
|
||||
escaped func(tempDir, outsideDir string) string
|
||||
}{
|
||||
{
|
||||
name: "relative dotdot",
|
||||
rows: func(_ string) []*database.File {
|
||||
return []*database.File{{
|
||||
Path: "../escaped-relative",
|
||||
Mode: containmentDirMode,
|
||||
}}
|
||||
},
|
||||
escaped: func(tempDir, _ string) string {
|
||||
return filepath.Join(tempDir, "escaped-relative")
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "absolute with dotdot",
|
||||
rows: func(_ string) []*database.File {
|
||||
return []*database.File{{
|
||||
Path: "/a/../../escaped-absolute",
|
||||
Mode: containmentDirMode,
|
||||
}}
|
||||
},
|
||||
escaped: func(tempDir, _ string) string {
|
||||
return filepath.Join(tempDir, "escaped-absolute")
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "child through symlink",
|
||||
rows: func(outsideDir string) []*database.File {
|
||||
return []*database.File{
|
||||
// Restored first: an in-target symlink pointing out.
|
||||
{Path: "linkdir", LinkTarget: types.FilePath(outsideDir)},
|
||||
// Restored second: a child written through that link.
|
||||
{Path: "linkdir/child", Mode: containmentDirMode},
|
||||
}
|
||||
},
|
||||
escaped: func(_, outsideDir string) string {
|
||||
return filepath.Join(outsideDir, "child")
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
ctx := context.Background()
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
targetDir := filepath.Join(tempDir, "target")
|
||||
outsideDir := filepath.Join(tempDir, "outside")
|
||||
require.NoError(t, fs.MkdirAll(outsideDir, 0o755))
|
||||
|
||||
rows, repos := makeFiles(ctx, t, tc.rows(outsideDir))
|
||||
v := newContainmentVaultik(ctx, fs)
|
||||
|
||||
_, err := v.restoreAllFiles(rows, repos,
|
||||
&RestoreOptions{TargetDir: targetDir}, nil, nil)
|
||||
|
||||
require.ErrorIs(t, err, errRestorePathEscapesTarget)
|
||||
|
||||
escaped := tc.escaped(tempDir, outsideDir)
|
||||
_, statErr := os.Lstat(escaped)
|
||||
require.Truef(t, os.IsNotExist(statErr),
|
||||
"restore wrote outside the target at %s", escaped)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestoreAllowsSymlinkPointingOutsideTree confirms the guard does not
|
||||
// over-block: an honest snapshot may contain a symlink whose target lies
|
||||
// outside the restored tree, and it must still be restored verbatim.
|
||||
func TestRestoreAllowsSymlinkPointingOutsideTree(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
t.Parallel()
|
||||
|
||||
ctx := context.Background()
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
targetDir := filepath.Join(tempDir, "target")
|
||||
linkTarget := filepath.Join(tempDir, "outside", "data")
|
||||
|
||||
rows, repos := makeFiles(ctx, t, []*database.File{
|
||||
{Path: "goodlink", LinkTarget: types.FilePath(linkTarget), MTime: time.Unix(0, 0)},
|
||||
})
|
||||
v := newContainmentVaultik(ctx, fs)
|
||||
|
||||
_, err := v.restoreAllFiles(rows, repos,
|
||||
&RestoreOptions{TargetDir: targetDir}, nil, nil)
|
||||
require.NoError(t, err)
|
||||
|
||||
got, err := os.Readlink(filepath.Join(targetDir, "goodlink"))
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, linkTarget, got)
|
||||
}
|
||||
Reference in New Issue
Block a user