Compare commits

...
2 Commits
Author SHA1 Message Date
sneak eca82738af List the files under a path by path, not by string prefix (closes #223)
check / check (push) Successful in 12m35s
FileRepository.ListByPrefix matched with SQL LIKE: a plain string
prefix that ignores ASCII case and treats _ and % as wildcards.
Restoring /home/u/doc also restored doc2, DOC and doc.txt.bak, and a
backup counted the files of a longer sibling path as deleted. It is
now ListUnderPath, which returns the file at the path and every file
whose path starts with the path plus a slash, compared exactly. A
trailing slash is ignored, so "/" still lists every file.

Three tests relied on string-prefix matching and now name a full path
or call ListAll.

ListIDsWithChunksNotInUploadedBlobs keeps its LIKE: it only adds file
IDs the scan never looks up.

Model: opus-5-5
2026-10-06 19:17:19 +00:00
clawbot f59086c0e5 Return an error, not a panic, on a malformed snapshot database (closes #231)
check / check (push) Successful in 13m39s
Restore cut chunk hashes from the snapshot database to 16 characters
for its error messages, so a shorter hash panicked. Those messages now
use shortHash. Under --verify, a file_chunks row with no chunks row was
dereferenced, and the chunk size from the database was allocated in
one piece, so a negative or huge size panicked. A missing row is now an
error, a negative size is rejected, and each chunk is hashed by
streaming it from the restored file.

A restored file shorter than its chunks now fails verify as a short
read instead of an unexpected EOF.

Model: opus-5-5
2026-10-06 21:16:28 +02:00
13 changed files with 376 additions and 39 deletions
+20
View File
@@ -22,6 +22,26 @@ the tag exists and is exercised; what is left is merging `next` to
# Completed Steps
- 2026-10-06: Made a restore path argument select only that path and
what is beneath it
([issue #223](https://git.eeqj.de/sneak/vaultik/issues/223)). The
lookup matched with SQL `LIKE`, so `/home/u/doc` also restored
`doc2`, `DOC` and `doc.txt.bak`, and a `_` or `%` in the path acted
as a wildcard. A backup used the same lookup to load the known files
of each configured path, so files of a longer sibling path were
counted as deleted. `FileRepository.ListUnderPath`, which replaces
`ListByPrefix`, returns the file at the path and every file whose
path starts with the path plus `/`, compared case-sensitively.
- 2026-10-06: Made restore return an error instead of panicking on a
malformed snapshot database
([issue #231](https://git.eeqj.de/sneak/vaultik/issues/231)). A chunk
hash shorter than 16 characters crashed the error message naming it,
and `--verify` dereferenced a missing `chunks` row and allocated
whatever chunk size the database gave. Those messages now go through
`shortHash`, a missing row is an error, and `--verify` rejects a
negative size and hashes each chunk as a stream.
- 2026-10-06: Made `s3://bucket/prefix` and `s3://bucket/prefix/` the same
destination ([issue #222](https://git.eeqj.de/sneak/vaultik/issues/222)).
The S3 client put the prefix directly in front of each key, so a prefix
+11 -6
View File
@@ -228,19 +228,24 @@ func (r *FileRepository) DeleteByID(
return nil
}
// ListByPrefix returns all files whose path starts with prefix, ordered by
// path.
func (r *FileRepository) ListByPrefix(
ctx context.Context, prefix string,
// ListUnderPath returns the file at path and every file beneath it,
// ordered by path. Paths are compared case-sensitively, and a trailing
// slash on path is ignored, so "/" lists every file.
func (r *FileRepository) ListUnderPath(
ctx context.Context, path string,
) ([]*File, error) {
path = strings.TrimRight(path, "/")
dirPrefix := path + "/"
// LIKE would ignore ASCII case and treat _ and % in path as wildcards.
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
FROM files
WHERE path LIKE ? || '%'
WHERE path = ? OR substr(path, 1, length(?)) = ?
ORDER BY path
`
rows, err := r.db.conn.QueryContext(ctx, query, prefix)
rows, err := r.db.conn.QueryContext(ctx, query, path, dirPrefix, dirPrefix)
if err != nil {
return nil, fmt.Errorf("querying files: %w", err)
}
+78
View File
@@ -5,10 +5,12 @@ import (
"database/sql"
"errors"
"os"
"slices"
"testing"
"time"
"sneak.berlin/go/vaultik/internal/database"
"sneak.berlin/go/vaultik/internal/types"
)
// errTestRollback is the sentinel returned from transaction bodies to
@@ -134,6 +136,82 @@ func TestFileRepositoryListDelete(t *testing.T) {
}
}
func TestFileRepositoryListUnderPath(t *testing.T) {
t.Parallel()
db, cleanup := setupTestDB(t)
defer cleanup()
ctx := context.Background()
repo := database.NewFileRepository(db)
const (
docDir = "/home/u/doc"
docFile = "/home/u/doc/a.txt"
)
// In path order, so the root case can expect all of them as listed.
paths := []string{
"/home/u/50%/x.txt",
"/home/u/50percent/y.txt",
"/home/u/DOC/c.txt",
"/home/u/a_b/x.txt",
"/home/u/axb/y.txt",
docDir,
"/home/u/doc.txt.bak",
docFile,
"/home/u/doc/sub/b.txt",
"/home/u/doc2/b.txt",
}
for _, path := range paths {
err := repo.Create(ctx, nil, &database.File{
Path: types.FilePath(path),
MTime: time.Now().Truncate(time.Second),
Mode: 0644,
})
if err != nil {
t.Fatalf("failed to create %s: %v", path, err)
}
}
docTree := []string{docDir, docFile, "/home/u/doc/sub/b.txt"}
tests := []struct {
name string
path string
want []string
}{
{"directory", docDir, docTree},
{"directory with trailing slash", docDir + "/", docTree},
{"directory differing only in case", "/home/u/DOC",
[]string{"/home/u/DOC/c.txt"}},
{"file", docFile, []string{docFile}},
{"underscore is literal", "/home/u/a_b",
[]string{"/home/u/a_b/x.txt"}},
{"percent is literal", "/home/u/50%",
[]string{"/home/u/50%/x.txt"}},
{"root", "/", paths},
}
for _, tt := range tests {
files, err := repo.ListUnderPath(ctx, tt.path)
if err != nil {
t.Fatalf("%s: failed to list files: %v", tt.name, err)
}
got := make([]string, 0, len(files))
for _, f := range files {
got = append(got, f.Path.String())
}
if !slices.Equal(got, tt.want) {
t.Errorf("%s: listing %q got %q, want %q",
tt.name, tt.path, got, tt.want)
}
}
}
func TestFileRepositorySymlink(t *testing.T) {
t.Parallel()
@@ -824,7 +824,7 @@ func TestTransactionIsolation(t *testing.T) {
}
// Verify the file was not created (transaction rolled back)
files, err := repos.Files.ListByPrefix(ctx, "/tx-test")
files, err := repos.Files.ListUnderPath(ctx, "/tx-test.txt")
if err != nil {
t.Fatal(err)
}
@@ -916,7 +916,7 @@ func TestConcurrentOrphanedCleanup(t *testing.T) {
}
// Verify correct files were deleted
files, err := repos.Files.ListByPrefix(ctx, "/concurrent-")
files, err := repos.Files.ListAll(ctx)
if err != nil {
t.Fatal(err)
}
+1 -1
View File
@@ -147,7 +147,7 @@ func TestOrphanedFileCleanupDebug(t *testing.T) {
t.Logf("Files count after cleanup: %d", count)
// List remaining files
files, err := repos.Files.ListByPrefix(ctx, "/")
files, err := repos.Files.ListUnderPath(ctx, "/")
if err != nil {
t.Fatal(err)
}
@@ -442,12 +442,12 @@ func TestLargeDatasets(t *testing.T) {
createLargeDatasetFiles(t, repos, snapshot.ID.String(), fileCount)
})
// Test ListByPrefix performance
// Test ListUnderPath performance
//nolint:paralleltest // phases share one database and are order-dependent
t.Run("list by prefix performance", func(t *testing.T) {
t.Run("list under path performance", func(t *testing.T) {
start := time.Now()
files, err := repos.Files.ListByPrefix(ctx, "/large/")
files, err := repos.Files.ListUnderPath(ctx, "/large/")
if err != nil {
t.Fatal(err)
}
@@ -472,7 +472,7 @@ func TestLargeDatasets(t *testing.T) {
t.Logf("Cleaned up orphaned files in %v", time.Since(start))
// Verify correct number remain
files, err := repos.Files.ListByPrefix(ctx, "/large/")
files, err := repos.Files.ListUnderPath(ctx, "/large/")
if err != nil {
t.Fatal(err)
}
+1 -1
View File
@@ -71,7 +71,7 @@ func verifyBackupFiles(
) {
t.Helper()
files, err := repos.Files.ListByPrefix(ctx, "")
files, err := repos.Files.ListAll(ctx)
if err != nil {
t.Fatalf("Failed to list files: %v", err)
}
+6 -4
View File
@@ -456,14 +456,16 @@ func (s *Scanner) finalizeScanResult(ctx context.Context, result *ScanResult) {
result.EndTime = time.Now().UTC()
}
// loadKnownFiles loads all known files from the database into a map for fast lookup
// This avoids per-file database queries during the scan phase
// loadKnownFiles loads the known files at and beneath path from the
// database into a map for fast lookup. Every loaded file the scan does
// not find is counted as deleted. This avoids per-file database queries
// during the scan phase.
func (s *Scanner) loadKnownFiles(
ctx context.Context, path string,
) (map[string]*database.File, error) {
files, err := s.repos.Files.ListByPrefix(ctx, path)
files, err := s.repos.Files.ListUnderPath(ctx, path)
if err != nil {
return nil, fmt.Errorf("listing files by prefix: %w", err)
return nil, fmt.Errorf("listing files under %s: %w", path, err)
}
result := make(map[string]*database.File, len(files))
+1 -1
View File
@@ -71,7 +71,7 @@ func verifySimpleScanDatabase(
t.Helper()
// Verify files in database - includes regular files and directories
files, err := repos.Files.ListByPrefix(ctx, "/source")
files, err := repos.Files.ListUnderPath(ctx, "/source")
if err != nil {
t.Fatalf("failed to list files: %v", err)
}
+1 -1
View File
@@ -249,7 +249,7 @@ func verifyEndToEndBackupState(
assert.Positive(t, blobUploads, "Should upload at least one blob")
// Verify files in database
files, err := repos.Files.ListByPrefix(ctx, "/home/user")
files, err := repos.Files.ListUnderPath(ctx, "/home/user")
require.NoError(t, err)
// Count only regular files (not directories)
regularFiles := 0
+29 -18
View File
@@ -42,6 +42,7 @@ var (
errChunkNotInAnyBlob = errors.New("chunk not found in any blob")
errBlobIDNotInHashIndex = errors.New("blob id missing from hash index")
errShortChunkRead = errors.New("short read")
errChunkRowMissing = errors.New("chunk has no row in the chunks table")
errRestorePathEscapesTarget = errors.New(
"refusing to restore path outside the target directory")
errTrailingRestoreData = errors.New(
@@ -829,10 +830,9 @@ func (v *Vaultik) getFilesToRestore(
// Normalize the filter path
filter = filepath.Clean(filter)
// Get files with this prefix
files, err := repos.Files.ListByPrefix(ctx, filter)
files, err := repos.Files.ListUnderPath(ctx, filter)
if err != nil {
return nil, fmt.Errorf("listing files with prefix %s: %w", filter, err)
return nil, fmt.Errorf("listing files under %s: %w", filter, err)
}
for _, file := range files {
@@ -1267,7 +1267,7 @@ func (s *restoreSession) writeFileChunks(
blobChunk, ok := s.chunkToBlobMap[chunkHashStr]
if !ok {
return bytesWritten, timings, fmt.Errorf(
"%w: %s", errChunkNotInAnyBlob, chunkHashStr[:16])
"%w: %s", errChunkNotInAnyBlob, shortHash(chunkHashStr))
}
blobHash, ok := s.blobIDToHash[blobChunk.BlobID.String()]
@@ -1284,7 +1284,7 @@ func (s *restoreSession) writeFileChunks(
if err != nil {
return bytesWritten, timings, fmt.Errorf(
"reading chunk %s from cached blob %s: %w",
fc.ChunkHash[:16], blobHash[:16], err)
shortHash(chunkHashStr), shortHash(blobHash), err)
}
t0 = time.Now()
@@ -1482,33 +1482,44 @@ func (v *Vaultik) verifyFile(
chunk, err := repos.Chunks.GetByHash(ctx, fc.ChunkHash.String())
if err != nil {
return bytesVerified, fmt.Errorf("getting chunk %s: %w",
fc.ChunkHash.String()[:16], err)
shortHash(fc.ChunkHash.String()), err)
}
// Read chunk data from file
chunkData := make([]byte, chunk.Size)
n, err := io.ReadFull(f, chunkData)
if err != nil {
return bytesVerified, fmt.Errorf("reading chunk data: %w", err)
if chunk == nil {
return bytesVerified, fmt.Errorf("%w: %s",
errChunkRowMissing, shortHash(fc.ChunkHash.String()))
}
if int64(n) != chunk.Size {
// chunk.Size comes from the snapshot database, which is not
// trusted: reject a negative size, and hash the chunk by
// streaming it rather than allocating that many bytes.
if chunk.Size < 0 {
return bytesVerified, fmt.Errorf("%w: chunk %d size %d",
errNegativeChunkLength, fc.Idx, chunk.Size)
}
hasher := sha256.New()
n, err := io.CopyN(hasher, f, chunk.Size)
if errors.Is(err, io.EOF) {
return bytesVerified, fmt.Errorf("%w: expected %d bytes, got %d",
errShortChunkRead, chunk.Size, n)
}
// Calculate hash and compare
hash := sha256.Sum256(chunkData)
actualHash := hex.EncodeToString(hash[:])
if err != nil {
return bytesVerified, fmt.Errorf("reading chunk data: %w", err)
}
actualHash := hex.EncodeToString(hasher.Sum(nil))
expectedHash := fc.ChunkHash.String()
if actualHash != expectedHash {
return bytesVerified, fmt.Errorf("%w: chunk %d: expected %s, got %s",
errChunkHashMismatch, fc.Idx, expectedHash[:16], actualHash[:16])
errChunkHashMismatch, fc.Idx,
shortHash(expectedHash), shortHash(actualHash))
}
bytesVerified += int64(n)
bytesVerified += n
}
// The stored chunks account for the whole file, so the reader must
@@ -0,0 +1,221 @@
package vaultik //nolint:testpackage // drives unexported restore and verify steps
import (
"context"
"math"
"path/filepath"
"strings"
"testing"
"time"
"github.com/spf13/afero"
"github.com/stretchr/testify/require"
"sneak.berlin/go/vaultik/internal/database"
"sneak.berlin/go/vaultik/internal/types"
)
// These tests feed restore and --verify a snapshot database written by
// hand, as a damaged or hostile store could serve one. Each malformed row
// must end in an error, not a panic.
// shortChunkHash is shorter than the hash prefix that error messages print.
const shortChunkHash = "abc"
// restoredFileContent is the content of the restored file under verify.
const restoredFileContent = "xyz"
// craftedSnapshotDB opens an empty snapshot database in a temp directory.
func craftedSnapshotDB(t *testing.T) (*database.DB, *database.Repositories) {
t.Helper()
db, err := database.New(context.Background(),
filepath.Join(t.TempDir(), "snapshot.db"))
require.NoError(t, err)
t.Cleanup(func() { _ = db.Close() })
return db, database.NewRepositories(db)
}
// craftedFile adds a regular file whose only chunk has the given hash.
// Adding the chunks row, if any, is left to the caller.
func craftedFile(
t *testing.T, repos *database.Repositories, chunkHash string,
) *database.File {
t.Helper()
ctx := context.Background()
file := &database.File{
Path: "/src/f",
MTime: time.Now().UTC(),
Size: int64(len(restoredFileContent)),
Mode: 0o644,
}
require.NoError(t, repos.Files.Create(ctx, nil, file))
require.NoError(t, repos.FileChunks.Create(ctx, nil, &database.FileChunk{
FileID: file.ID,
ChunkHash: types.ChunkHash(chunkHash),
}))
return file
}
// TestRestoreShortChunkHashInNoBlob proves a file whose short chunk hash
// has no blob_chunks row fails restore planning and the chunk write with
// an error.
func TestRestoreShortChunkHashInNoBlob(t *testing.T) {
t.Parallel()
ctx := context.Background()
_, repos := craftedSnapshotDB(t)
require.NoError(t, repos.Chunks.Create(ctx, nil,
&database.Chunk{ChunkHash: shortChunkHash, Size: 3}))
file := craftedFile(t, repos, shortChunkHash)
v := NewForTesting(nil)
chunkToBlobMap, err := v.buildChunkToBlobMap(ctx, repos)
require.NoError(t, err)
_, err = newRestorePlan(ctx, repos, []*database.File{file},
chunkToBlobMap, map[string]string{})
require.ErrorIs(t, err, errPlanChunkMissing)
fileChunks, err := repos.FileChunks.GetByFileID(ctx, file.ID)
require.NoError(t, err)
out, err := afero.NewMemMapFs().Create("out")
require.NoError(t, err)
session := &restoreSession{
v: v.Vaultik, ctx: ctx, chunkToBlobMap: chunkToBlobMap,
}
_, _, err = session.writeFileChunks(out, fileChunks)
require.ErrorIs(t, err, errChunkNotInAnyBlob)
}
// TestRestoreShortChunkHashReadPastBlobEnd proves a short chunk hash
// whose blob_chunks row reads past the end of its blob fails the chunk
// write with an error.
func TestRestoreShortChunkHashReadPastBlobEnd(t *testing.T) {
t.Parallel()
ctx := context.Background()
_, repos := craftedSnapshotDB(t)
blobHash := strings.Repeat("b", blobHashHexLen)
blob := &database.Blob{
ID: types.NewBlobID(),
Hash: types.BlobHash(blobHash),
CreatedTS: time.Now().UTC(),
}
require.NoError(t, repos.Blobs.Create(ctx, nil, blob))
require.NoError(t, repos.Chunks.Create(ctx, nil,
&database.Chunk{ChunkHash: shortChunkHash, Size: 3}))
require.NoError(t, repos.BlobChunks.Create(ctx, nil, &database.BlobChunk{
BlobID: blob.ID,
ChunkHash: shortChunkHash,
Length: 100,
}))
file := craftedFile(t, repos, shortChunkHash)
cache, err := newBlobDiskCache(1 << 20)
require.NoError(t, err)
t.Cleanup(func() { _ = cache.Close() })
require.NoError(t, cache.Put(blobHash, []byte("abc")))
v := NewForTesting(nil)
chunkToBlobMap, err := v.buildChunkToBlobMap(ctx, repos)
require.NoError(t, err)
_, blobIDToHash, err := v.buildBlobIndexes(repos)
require.NoError(t, err)
fileChunks, err := repos.FileChunks.GetByFileID(ctx, file.ID)
require.NoError(t, err)
out, err := afero.NewMemMapFs().Create("out")
require.NoError(t, err)
session := &restoreSession{
v: v.Vaultik,
ctx: ctx,
chunkToBlobMap: chunkToBlobMap,
blobIDToHash: blobIDToHash,
blobCache: cache,
}
_, _, err = session.writeFileChunks(out, fileChunks)
require.ErrorIs(t, err, errCacheReadBeyondBlob)
}
// TestVerifyFileMalformedChunkRow proves --verify returns an error for a
// chunk with no chunks row, a short chunk hash, and a chunk size from the
// database that is negative or larger than the restored file.
func TestVerifyFileMalformedChunkRow(t *testing.T) {
t.Parallel()
fullHash := types.ChunkHash(strings.Repeat("c", blobHashHexLen))
tests := []struct {
name string
hash types.ChunkHash
chunk *database.Chunk // nil adds no chunks row
want error
}{
{
name: "missing chunk row",
hash: fullHash,
want: errChunkRowMissing,
},
{
name: "short hash",
hash: shortChunkHash,
chunk: &database.Chunk{ChunkHash: shortChunkHash, Size: 3},
want: errChunkHashMismatch,
},
{
name: "size larger than the file",
hash: fullHash,
chunk: &database.Chunk{ChunkHash: fullHash, Size: math.MaxInt64},
want: errShortChunkRead,
},
{
name: "negative size",
hash: fullHash,
chunk: &database.Chunk{ChunkHash: fullHash, Size: -1},
want: errNegativeChunkLength,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
t.Parallel()
ctx := context.Background()
db, repos := craftedSnapshotDB(t)
if tt.chunk == nil {
// A crafted database need not satisfy its foreign keys.
_, err := db.Conn().ExecContext(ctx, "PRAGMA foreign_keys = OFF")
require.NoError(t, err)
} else {
require.NoError(t, repos.Chunks.Create(ctx, nil, tt.chunk))
}
file := craftedFile(t, repos, tt.hash.String())
v := NewForTesting(nil)
v.Fs = afero.NewMemMapFs()
require.NoError(t, afero.WriteFile(v.Fs, "/restore/f",
[]byte(restoredFileContent), 0o600))
_, err := v.verifyFile(ctx, repos, file, "/restore/f")
require.ErrorIs(t, err, tt.want)
})
}
}
+1 -1
View File
@@ -75,7 +75,7 @@ func newRestorePlan(
bc, ok := chunkToBlobMap[fc.ChunkHash.String()]
if !ok {
return nil, fmt.Errorf("planning %s: %w: %s",
f.Path, errPlanChunkMissing, fc.ChunkHash.String()[:16])
f.Path, errPlanChunkMissing, shortHash(fc.ChunkHash.String()))
}
hash, ok := blobIDToHash[bc.BlobID.String()]