1 Commits
Author SHA1 Message Date
sneak 2de79a0897 Report a prune count that could not be read as unknown, not 0 (closes #96)
check / check (pull_request) Failing after 0s
PruneDatabase read seven table counts with the error discarded, so a
query that could not run silently became 0 and the before/after delta
computed from it looked like real work.

Each read now goes through a helper that logs at warn on failure and
returns nil; nil renders as "unknown", never "0", so an empty table is
distinguishable from one that could not be queried. Deltas built from an
unknown count are themselves unknown. The failure is surfaced, not
propagated, so the command's failure conditions are unchanged. These
counts have no --json output — under --json the summary is suppressed
entirely — so nothing there can show a false 0.

Model: opus-4-8
2026-09-21 18:05:05 +00:00
5 changed files with 181 additions and 162 deletions
+9 -9
View File
@@ -25,15 +25,15 @@ release" is exactly the contradiction
# Completed Steps
- 2026-09-21: Fixed `verify --deep` reporting healthy snapshots as
corrupt. Its final blob-integrity check hashed the encrypted
downloaded bytes with a single SHA256 and compared that to the blob
ID, which is the double SHA256 of the plaintext, so the two could
never match. It now hashes the decompressed plaintext and compares the
double SHA256. Added a test that backs up a real snapshot, deep-verifies
it, then flips a byte in one stored blob and confirms deep verification
then fails
([issue #131](https://git.eeqj.de/sneak/vaultik/issues/131)).
- 2026-09-21: Stopped `prune` from reporting a failed row count as 0
([issue #96](https://git.eeqj.de/sneak/vaultik/issues/96)). The seven
`getTableCount` reads in `PruneDatabase` discarded their error, so a
query that could not run became a plausible `0` and the before/after
delta computed from it looked like real work. Each read now logs at
warn on failure and renders as `unknown`, never `0`, so an empty table
is distinguishable from one that could not be queried. The counts have
no `--json` representation — under `--json` the summary is suppressed
entirely — so nothing there can show a false `0`.
- 2026-09-21: Made `snapshot create` VACUUM the per-snapshot metadata
database through the `modernc.org/sqlite` driver instead of shelling
-108
View File
@@ -1,108 +0,0 @@
package vaultik_test
import (
"context"
"io"
"os"
"path/filepath"
"testing"
"github.com/spf13/afero"
"github.com/stretchr/testify/require"
"sneak.berlin/go/vaultik/internal/log"
"sneak.berlin/go/vaultik/internal/ui"
"sneak.berlin/go/vaultik/internal/vaultik"
)
// TestDeepVerifyAcceptsHealthyAndRejectsCorruptBlob backs up a real
// snapshot with the on-disk storage backend, runs deep verification on
// it, then flips a byte inside one stored blob and runs deep
// verification again. A healthy snapshot must pass; a corrupted blob
// must fail. The healthy case is the regression guard: deep
// verification used to hash the encrypted blob bytes and compare them
// to the blob's ID (the double SHA256 of the plaintext), so it reported
// every healthy blob as corrupt.
func TestDeepVerifyAcceptsHealthyAndRejectsCorruptBlob(t *testing.T) {
log.Initialize(log.Config{})
t.Parallel()
fs := afero.NewOsFs()
tempDir := t.TempDir()
dataDir := filepath.Join(tempDir, "source")
storeDir := filepath.Join(tempDir, "remote")
dbPath := filepath.Join(tempDir, "index.sqlite")
chunkSize := int64(64 * 1024)
maxBlobSize := int64(512 * 1024)
// One file large enough to span several chunks within a single blob.
require.NoError(t, fs.MkdirAll(dataDir, 0o755))
require.NoError(t, afero.WriteFile(fs,
filepath.Join(dataDir, "data.bin"),
bytesPattern("deep-", int(chunkSize*3)), 0o644))
ctx := context.Background()
// runFileStorageBackup writes a real snapshot to storeDir and closes
// the source index, so verification runs from remote bytes only.
cfg, storer, snapshotID := runFileStorageBackup(
ctx, t, fs, dataDir, storeDir, dbPath, chunkSize, maxBlobSize)
newVerifier := func() *vaultik.Vaultik {
v := &vaultik.Vaultik{
Config: cfg,
Storage: storer,
Fs: fs,
Stdout: io.Discard,
Stderr: io.Discard,
UI: ui.NewWithColor(io.Discard, false),
}
v.SetContext(ctx)
return v
}
require.NoError(t,
newVerifier().RunDeepVerify(snapshotID, &vaultik.VerifyOptions{Deep: true}),
"deep verify should pass on a healthy snapshot")
// Flip a byte inside one blob without changing its length, so the
// blob-existence and size checks still pass and verification reaches
// the blob-content stage.
corruptOneBlob(t, fs, filepath.Join(storeDir, "blobs"))
require.Error(t,
newVerifier().RunDeepVerify(snapshotID, &vaultik.VerifyOptions{Deep: true}),
"deep verify should fail on a corrupted blob")
}
// corruptOneBlob flips a middle byte of the first blob file found under
// blobsDir, leaving the file length unchanged.
func corruptOneBlob(t *testing.T, fs afero.Fs, blobsDir string) {
t.Helper()
var blobPath string
err := afero.Walk(fs, blobsDir,
func(path string, info os.FileInfo, err error) error {
if err != nil {
return err
}
if blobPath == "" && !info.IsDir() {
blobPath = path
}
return nil
})
require.NoError(t, err)
require.NotEmpty(t, blobPath, "expected at least one blob on disk")
data, err := afero.ReadFile(fs, blobPath)
require.NoError(t, err)
require.NotEmpty(t, data)
data[len(data)/2] ^= 0xff
require.NoError(t, afero.WriteFile(fs, blobPath, data, 0o644))
}
+79
View File
@@ -0,0 +1,79 @@
package vaultik //nolint:testpackage // exercises unexported count helpers
import (
"context"
"testing"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"sneak.berlin/go/vaultik/internal/database"
"sneak.berlin/go/vaultik/internal/log"
)
// TestTableCountForReportSurfacesReadFailure is the regression guard for
// the discarded-error bug: getTableCount for a table its query cannot
// resolve must not silently become 0. A count that could not be read is
// reported as unknown, which a reader can tell apart from an empty table.
//
//nolint:paralleltest // installs the global logger via log.Initialize
func TestTableCountForReportSurfacesReadFailure(t *testing.T) {
log.Initialize(log.Config{})
ctx := context.Background()
db, err := database.New(ctx, ":memory:")
require.NoError(t, err)
t.Cleanup(func() { _ = db.Close() })
v := &Vaultik{DB: db}
v.SetContext(ctx)
// A table present in the schema reads as a real count.
blobs := v.tableCountForReport("blobs")
require.NotNil(t, blobs, "an existing table must read as a real count")
assert.Equal(t, int64(0), *blobs)
// A syntactically valid name the sanitizer accepts but whose table
// the query cannot resolve is the exact shape #96 describes: a
// would-be loud failure that used to be discarded into a 0.
_, err = v.getTableCount("snapshots_missing")
require.Error(t, err, "a query against a nonexistent table must fail")
missing := v.tableCountForReport("snapshots_missing")
assert.Nil(t, missing, "a failed read is unknown, not a count")
// The rendered count for a failed read must say unknown, never 0.
assert.Equal(t, countUnknown, countText(missing))
assert.NotEqual(t, "0", countText(missing))
}
// TestCountTextDistinguishesEmptyFromUnknown pins the distinction the
// output has to preserve: 0 means the table was empty, "unknown" means
// the count could not be read.
func TestCountTextDistinguishesEmptyFromUnknown(t *testing.T) {
t.Parallel()
zero := int64(0)
seven := int64(7)
assert.Equal(t, "0", countText(&zero))
assert.Equal(t, "7", countText(&seven))
assert.Equal(t, countUnknown, countText(nil))
}
// TestCountDiffUnknownWhenEitherSideUnknown checks that a delta computed
// from an unreadable count is itself unknown rather than a plausible
// number.
func TestCountDiffUnknownWhenEitherSideUnknown(t *testing.T) {
t.Parallel()
before := int64(10)
after := int64(3)
require.NotNil(t, countDiff(&before, &after))
assert.Equal(t, int64(7), *countDiff(&before, &after))
assert.Nil(t, countDiff(nil, &after), "unknown before yields unknown delta")
assert.Nil(t, countDiff(&before, nil), "unknown after yields unknown delta")
assert.Nil(t, countDiff(nil, nil))
}
+79 -26
View File
@@ -8,6 +8,7 @@ import (
"path/filepath"
"regexp"
"sort"
"strconv"
"strings"
"time"
@@ -1540,12 +1541,17 @@ func (v *Vaultik) outputRemoveJSON(result *RemoveResult) error {
return encoder.Encode(result)
}
// PruneResult contains statistics about the prune operation
// PruneResult contains statistics about the prune operation.
// SnapshotsDeleted counts snapshots actually deleted. FilesDeleted,
// ChunksDeleted, and BlobsDeleted are derived from before/after row
// counts of the local index; each is nil when a count could not be read,
// so an unreadable count is reported as unknown rather than silently
// as 0.
type PruneResult struct {
SnapshotsDeleted int64
FilesDeleted int64
ChunksDeleted int64
BlobsDeleted int64
FilesDeleted *int64
ChunksDeleted *int64
BlobsDeleted *int64
}
// PruneDatabase removes incomplete snapshots and orphaned files, chunks,
@@ -1560,7 +1566,7 @@ func (v *Vaultik) PruneDatabase() (*PruneResult, error) {
result := &PruneResult{}
// Snapshot counts before deletion of incompletes.
snapshotCountBefore, _ := v.getTableCount("snapshots")
snapshotCountBefore := v.tableCountForReport("snapshots")
// First, delete any incomplete snapshots
incompleteSnapshots, err := v.Repositories.Snapshots.GetIncompleteSnapshots(v.ctx)
@@ -1575,9 +1581,9 @@ func (v *Vaultik) PruneDatabase() (*PruneResult, error) {
}
// Get counts before cleanup for reporting
fileCountBefore, _ := v.getTableCount("files")
chunkCountBefore, _ := v.getTableCount("chunks")
blobCountBefore, _ := v.getTableCount("blobs")
fileCountBefore := v.tableCountForReport("files")
chunkCountBefore := v.tableCountForReport("chunks")
blobCountBefore := v.tableCountForReport("blobs")
// Run the cleanup
err = v.SnapshotManager.CleanupOrphanedData(v.ctx)
@@ -1586,36 +1592,83 @@ func (v *Vaultik) PruneDatabase() (*PruneResult, error) {
}
// Get counts after cleanup
fileCountAfter, _ := v.getTableCount("files")
chunkCountAfter, _ := v.getTableCount("chunks")
blobCountAfter, _ := v.getTableCount("blobs")
fileCountAfter := v.tableCountForReport("files")
chunkCountAfter := v.tableCountForReport("chunks")
blobCountAfter := v.tableCountForReport("blobs")
result.FilesDeleted = fileCountBefore - fileCountAfter
result.ChunksDeleted = chunkCountBefore - chunkCountAfter
result.BlobsDeleted = blobCountBefore - blobCountAfter
result.FilesDeleted = countDiff(fileCountBefore, fileCountAfter)
result.ChunksDeleted = countDiff(chunkCountBefore, chunkCountAfter)
result.BlobsDeleted = countDiff(blobCountBefore, blobCountAfter)
log.Info("Local database prune complete",
"incomplete_snapshots", result.SnapshotsDeleted,
"orphaned_files", result.FilesDeleted,
"orphaned_chunks", result.ChunksDeleted,
"orphaned_blobs", result.BlobsDeleted,
"orphaned_files", countText(result.FilesDeleted),
"orphaned_chunks", countText(result.ChunksDeleted),
"orphaned_blobs", countText(result.BlobsDeleted),
)
snapshotCountAfter := snapshotCountBefore - result.SnapshotsDeleted
// Snapshots remaining after removing the incomplete ones; unknown if
// the pre-prune snapshot count could not be read.
snapshotsRemain := countDiff(snapshotCountBefore, &result.SnapshotsDeleted)
v.UI.Completef("Pruned local index database.")
v.UI.Detailf("Incomplete snapshots: %d removed (%d remain).",
result.SnapshotsDeleted, snapshotCountAfter)
v.UI.Detailf("Orphaned files: %d removed (%d remain).",
result.FilesDeleted, fileCountAfter)
v.UI.Detailf("Orphaned chunks: %d removed (%d remain).",
result.ChunksDeleted, chunkCountAfter)
v.UI.Detailf("Orphaned blobs: %d removed (%d remain).",
result.BlobsDeleted, blobCountAfter)
v.UI.Detailf("Incomplete snapshots: %s removed (%s remain).",
countText(&result.SnapshotsDeleted), countText(snapshotsRemain))
v.UI.Detailf("Orphaned files: %s removed (%s remain).",
countText(result.FilesDeleted), countText(fileCountAfter))
v.UI.Detailf("Orphaned chunks: %s removed (%s remain).",
countText(result.ChunksDeleted), countText(chunkCountAfter))
v.UI.Detailf("Orphaned blobs: %s removed (%s remain).",
countText(result.BlobsDeleted), countText(blobCountAfter))
return result, nil
}
// countUnknown is what a count reads as when its query could not be run,
// distinct from "0", which means the table really was empty.
const countUnknown = "unknown"
// tableCountForReport returns the row count of a table for the prune
// summary, or nil if the count could not be read. A read failure is
// logged at warn — visible even under --json, which routes warnings to
// stderr — and then rendered as unknown rather than silently becoming 0,
// so a broken query is a visible failure instead of a plausible wrong
// number.
func (v *Vaultik) tableCountForReport(tableName string) *int64 {
count, err := v.getTableCount(tableName)
if err != nil {
log.Warn("could not read table row count for prune summary",
"table", tableName, "error", err)
return nil
}
return &count
}
// countDiff returns before-after, or nil if either count is unknown so
// that an unreadable count does not collapse into a plausible delta.
func countDiff(before, after *int64) *int64 {
if before == nil || after == nil {
return nil
}
diff := *before - *after
return &diff
}
// countText renders a count that may be unknown: nil (the read failed)
// becomes "unknown", never "0", so a reader can tell an empty table from
// one that could not be queried.
func countText(count *int64) string {
if count == nil {
return countUnknown
}
return strconv.FormatInt(*count, 10)
}
// validTableNameRe matches table names containing only lowercase
// alphanumeric characters and underscores.
var validTableNameRe = regexp.MustCompile(`^[a-z0-9_]+$`)
+14 -19
View File
@@ -344,8 +344,12 @@ func (v *Vaultik) verifyBlob(blobInfo snapshot.BlobInfo, db *sql.DB) error {
return fmt.Errorf("failed to get decryptor: %w", err)
}
// Decrypt blob
decryptedReader, err := decryptor.DecryptStream(reader)
// Hash the encrypted blob data as it streams through to decryption
blobHasher := sha256.New()
teeReader := io.TeeReader(reader, blobHasher)
// Decrypt blob (reading through teeReader to hash encrypted data)
decryptedReader, err := decryptor.DecryptStream(teeReader)
if err != nil {
return fmt.Errorf("failed to decrypt: %w", err)
}
@@ -357,19 +361,12 @@ func (v *Vaultik) verifyBlob(blobInfo snapshot.BlobInfo, db *sql.DB) error {
}
defer decompressor.Close()
// A blob's hash — its remote name — is the double SHA256 of its
// decompressed plaintext (see blobgen.Writer.Sum256), not of the
// encrypted bytes. Hash the plaintext as chunk verification streams
// it, then compare on completion.
plaintextHasher := sha256.New()
hashedStream := io.TeeReader(decompressor, plaintextHasher)
chunkCount, err := v.verifyBlobChunks(db, blobInfo.Hash, hashedStream)
chunkCount, err := v.verifyBlobChunks(db, blobInfo.Hash, decompressor)
if err != nil {
return err
}
err = v.verifyBlobFinalIntegrity(hashedStream, plaintextHasher, blobInfo.Hash)
err = v.verifyBlobFinalIntegrity(decompressor, blobHasher, blobInfo.Hash)
if err != nil {
return err
}
@@ -473,13 +470,14 @@ func (v *Vaultik) verifyBlobChunks(
}
// verifyBlobFinalIntegrity checks that no trailing data exists in the
// decompressed stream and that the blob hash matches the expected value.
// decompressed stream and that the encrypted blob hash matches the
// expected value.
func (v *Vaultik) verifyBlobFinalIntegrity(
plaintext io.Reader, plaintextHasher hash.Hash, expectedHash string,
decompressor io.Reader, blobHasher hash.Hash, expectedHash string,
) error {
// Verify no remaining data in blob - if the chunk list is accurate,
// the blob should be fully consumed.
remaining, err := io.Copy(io.Discard, plaintext)
remaining, err := io.Copy(io.Discard, decompressor)
if err != nil {
return fmt.Errorf("failed to check for remaining blob data: %w", err)
}
@@ -488,11 +486,8 @@ func (v *Vaultik) verifyBlobFinalIntegrity(
return fmt.Errorf("%w: %d bytes", errTrailingBlobData, remaining)
}
// The blob hash is the double SHA256 of its plaintext content.
firstHash := plaintextHasher.Sum(nil)
secondHash := sha256.Sum256(firstHash)
calculatedBlobHash := hex.EncodeToString(secondHash[:])
// Verify blob hash matches the encrypted data we downloaded
calculatedBlobHash := hex.EncodeToString(blobHasher.Sum(nil))
if calculatedBlobHash != expectedHash {
return fmt.Errorf("%w: calculated %s, expected %s",
errBlobHashMismatch, calculatedBlobHash, expectedHash)