Compare commits
3
Commits
2de79a0897
...
0e57ea874a
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0e57ea874a | ||
|
|
07ef3a1c78 | ||
|
|
c423d13191 |
+5
-5
@@ -72,11 +72,11 @@ RUN [ -n "$CHECK_EPOCH" ] || exit 1
|
||||
# running, and exits 0 reporting `0 issues.` on a tree the real config
|
||||
# fails. Demonstrated on this repo at this pin, recorded on
|
||||
# https://git.eeqj.de/sneak/vaultik/pulls/114: with a planted
|
||||
# over-length line, `script/lint` exits 1 naming the `lll` finding with
|
||||
# `linters:` and exits 0 with `linterz:`. A set-but-ineffective config
|
||||
# quietly falling back to defaults is precisely the false-green class
|
||||
# this gate exists to eliminate, so it must not sit in the gate's own
|
||||
# configuration.
|
||||
# over-length line, `script/lint` exits 1 naming the `revive` finding
|
||||
# with `linters:` and exits 0 with `linterz:`. A set-but-ineffective
|
||||
# config quietly falling back to defaults is precisely the false-green
|
||||
# class this gate exists to eliminate, so it must not sit in the gate's
|
||||
# own configuration.
|
||||
#
|
||||
# `config verify` catches it, and it does so OFFLINE at this pinned
|
||||
# version -- verified, not assumed. Under `docker run --network none`
|
||||
|
||||
@@ -25,6 +25,25 @@ release" is exactly the contradiction
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- 2026-09-21: Stopped `prune` from reporting a failed row count as 0
|
||||
([issue #96](https://git.eeqj.de/sneak/vaultik/issues/96)). The seven
|
||||
`getTableCount` reads in `PruneDatabase` discarded their error, so a
|
||||
query that could not run became a plausible `0` and the before/after
|
||||
delta computed from it looked like real work. Each read now logs at
|
||||
warn on failure and renders as `unknown`, never `0`, so an empty table
|
||||
is distinguishable from one that could not be queried. The counts have
|
||||
no `--json` representation — under `--json` the summary is suppressed
|
||||
entirely — so nothing there can show a false `0`.
|
||||
- 2026-09-21: Fixed `verify --deep` reporting healthy snapshots as
|
||||
corrupt. Its final blob-integrity check hashed the encrypted
|
||||
downloaded bytes with a single SHA256 and compared that to the blob
|
||||
ID, which is the double SHA256 of the plaintext, so the two could
|
||||
never match. It now hashes the decompressed plaintext and compares the
|
||||
double SHA256. Added a test that backs up a real snapshot, deep-verifies
|
||||
it, then flips a byte in one stored blob and confirms deep verification
|
||||
then fails
|
||||
([issue #131](https://git.eeqj.de/sneak/vaultik/issues/131)).
|
||||
|
||||
- 2026-09-21: Made `snapshot create` VACUUM the per-snapshot metadata
|
||||
database through the `modernc.org/sqlite` driver instead of shelling
|
||||
out to the external `sqlite` command-line binary (issue #120). A
|
||||
@@ -78,10 +97,11 @@ release" is exactly the contradiction
|
||||
into each check command, and a fresh `$(date +%s%N)$$` per invocation
|
||||
computed as a bare assignment. `cmd/vaultik/lintdocker_test.go`
|
||||
parses both Dockerfiles and both scripts and fails if any part of
|
||||
that is dropped, because every way of losing it is silent. Its
|
||||
host-lint assertion is structural — no script runs `golangci-lint`
|
||||
except through `docker` — rather than a search for the one retired
|
||||
variable name, which nothing could ever reintroduce.
|
||||
that is dropped, because every way of losing it is silent. No test
|
||||
asserts that no script runs the host linter: `script/lint` is the one
|
||||
lint entry point and runs `golangci-lint` only inside the container,
|
||||
and keeping it that way is a review matter, not something a test
|
||||
proves.
|
||||
|
||||
The product `Dockerfile` lost its lint stage rather than gaining a
|
||||
second linter pin: `make lint` is now `docker build`, so the stage
|
||||
|
||||
@@ -28,6 +28,11 @@ import (
|
||||
// -- that a real finding actually fails the build -- is verified by
|
||||
// hand against a deliberately broken tree, recorded on the pull
|
||||
// request.
|
||||
//
|
||||
// One property is deliberately NOT tested here: that no script runs the
|
||||
// linter on the host. script/lint is the only lint entry point, and it
|
||||
// runs golangci-lint only inside the container; keeping it that way is a
|
||||
// review matter, not something a test in this file establishes.
|
||||
|
||||
// The files under guard, relative to the repository root.
|
||||
const (
|
||||
@@ -37,9 +42,8 @@ const (
|
||||
cibuildScript = "script/cibuild"
|
||||
)
|
||||
|
||||
// linterBinary is the linter's command name. Every occurrence of it in
|
||||
// executable shell in this repo must be inside a docker invocation; see
|
||||
// TestNoHostLintPathRemains.
|
||||
// linterBinary is the linter's command name, used to locate the
|
||||
// config-verify and lint steps in Dockerfile.lint.
|
||||
const linterBinary = "golangci-lint"
|
||||
|
||||
// checkEpochARG is the declaration, with no default value. A default
|
||||
@@ -219,90 +223,6 @@ func TestCibuildBuildsBothDockerfilesWithFreshEpochs(t *testing.T) {
|
||||
"%s must build %s", cibuildScript, lintDockerfile)
|
||||
}
|
||||
|
||||
// TestNoHostLintPathRemains fails if any escape hatch to a host linter
|
||||
// comes back. The owner's ruling is that every lint run happens inside
|
||||
// a container; a PATH binary that happens to match the pinned version
|
||||
// is a different build reached by a different code path, and admitting
|
||||
// it is what lets a local pass disagree with CI.
|
||||
//
|
||||
// This asserts the PROPERTY -- no script invokes the linter except
|
||||
// through docker -- rather than the absence of any particular variable
|
||||
// name. An earlier version of this test looked only for the literal
|
||||
// VAULTIK_LINT_IN_CONTAINER, the name of the hatch that was removed
|
||||
// alongside it, so nothing could ever trip it again: a hatch under any
|
||||
// other name left it passing. A structural test that passes on a broken
|
||||
// tree is worse than no test, because it is what a later reader trusts
|
||||
// instead of re-deriving the invariant.
|
||||
//
|
||||
// script/lint-fix is not exempted. It is the one script that runs the
|
||||
// linter as a container rather than as a build step, but it still runs
|
||||
// it in one, so the same property holds of it.
|
||||
func TestNoHostLintPathRemains(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
root := repoRoot(t)
|
||||
|
||||
entries, err := os.ReadDir(filepath.Join(root, "script"))
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, entries, "no scripts found to scan")
|
||||
|
||||
for _, entry := range entries {
|
||||
if entry.IsDir() {
|
||||
continue
|
||||
}
|
||||
|
||||
name := filepath.Join("script", entry.Name())
|
||||
for _, line := range shellCode(readRepoFile(t, name)) {
|
||||
assertLinterIsContainerised(t, name, line)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// assertLinterIsContainerised fails if the line runs the linter without
|
||||
// handing it to docker first. Position matters: docker has to come
|
||||
// before the binary, or the line is running the host linter and merely
|
||||
// mentioning docker afterwards.
|
||||
func assertLinterIsContainerised(t *testing.T, name, line string) {
|
||||
t.Helper()
|
||||
|
||||
at := strings.Index(line, linterBinary)
|
||||
if at < 0 {
|
||||
return
|
||||
}
|
||||
|
||||
docker := strings.Index(line, "docker")
|
||||
|
||||
assert.True(t, docker >= 0 && docker < at,
|
||||
"%s runs %s on the host; every lint run happens in a container"+
|
||||
" (line: %s)", name, linterBinary, line)
|
||||
}
|
||||
|
||||
// TestShellCodeSeesCodeAndNotProse keeps the scanner above honest. It
|
||||
// has to ignore comments and here-document bodies, because script/lint
|
||||
// and script/bootstrap both NAME golangci-lint in prose -- in comments,
|
||||
// and in the error text they print -- precisely to say that the host
|
||||
// binary is never used. A scanner that went blind, by over-eager
|
||||
// stripping or by failing to join continuation lines, would make
|
||||
// TestNoHostLintPathRemains pass on everything.
|
||||
func TestShellCodeSeesCodeAndNotProse(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
script := strings.Join([]string{
|
||||
"#!/bin/sh",
|
||||
"# a comment naming golangci-lint",
|
||||
"cat >&2 <<EOF",
|
||||
"prose naming golangci-lint, printed not executed",
|
||||
"EOF",
|
||||
"docker run --rm \\",
|
||||
" \"$image\" \\",
|
||||
" golangci-lint run ./...",
|
||||
}, "\n")
|
||||
|
||||
assert.Equal(t,
|
||||
[]string{"cat >&2 <<EOF", `docker run --rm "$image" golangci-lint run ./...`},
|
||||
shellCode(script))
|
||||
}
|
||||
|
||||
// assertEpochExpandedInto fails unless some instruction runs the named
|
||||
// command with the epoch expanded into it. Expansion, not mere
|
||||
// declaration: an ARG that no instruction references is not guaranteed
|
||||
@@ -407,70 +327,6 @@ func indexContaining(found []string, want string) int {
|
||||
return -1
|
||||
}
|
||||
|
||||
// shellCode returns a POSIX shell script's executable lines: comments
|
||||
// dropped, here-document bodies dropped, and backslash continuations
|
||||
// joined so a multi-line command is a single string. Whitespace is
|
||||
// collapsed, as it is for Dockerfile instructions.
|
||||
//
|
||||
// Both exclusions are load-bearing rather than tidiness. The scripts
|
||||
// name golangci-lint in prose to state that the host binary is never
|
||||
// used, and joining continuations is what lets the one legitimate
|
||||
// container invocation -- script/lint-fix's `docker run`, whose linter
|
||||
// command sits several lines below the word `docker` -- be recognised
|
||||
// as containerised.
|
||||
func shellCode(contents string) []string {
|
||||
var (
|
||||
out []string
|
||||
joined string
|
||||
terminate string
|
||||
)
|
||||
|
||||
for line := range strings.SplitSeq(contents, "\n") {
|
||||
trimmed := strings.TrimSpace(line)
|
||||
|
||||
if terminate != "" {
|
||||
if trimmed == terminate {
|
||||
terminate = ""
|
||||
}
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
if joined == "" && (trimmed == "" || strings.HasPrefix(trimmed, "#")) {
|
||||
continue
|
||||
}
|
||||
|
||||
joined += strings.TrimSuffix(trimmed, `\`) + " "
|
||||
if strings.HasSuffix(trimmed, `\`) {
|
||||
continue
|
||||
}
|
||||
|
||||
joined = strings.Join(strings.Fields(joined), " ")
|
||||
terminate = heredocTerminator(joined)
|
||||
|
||||
out = append(out, joined)
|
||||
joined = ""
|
||||
}
|
||||
|
||||
return out
|
||||
}
|
||||
|
||||
// heredocTerminator returns the terminator of the here-document a
|
||||
// command opens, or "" if it opens none. Only the first on a line is
|
||||
// recognised; nothing in script/ opens two.
|
||||
func heredocTerminator(line string) string {
|
||||
_, after, opens := strings.Cut(line, "<<")
|
||||
if !opens {
|
||||
return ""
|
||||
}
|
||||
|
||||
// `<<-` strips leading tabs from the body; the terminator word is
|
||||
// the same either way, and callers compare against trimmed lines.
|
||||
word, _, _ := strings.Cut(strings.TrimPrefix(after, "-"), " ")
|
||||
|
||||
return strings.Trim(word, `'"`)
|
||||
}
|
||||
|
||||
// readRepoFile reads a file by its path relative to the repository
|
||||
// root.
|
||||
func readRepoFile(t *testing.T, name string) string {
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
package vaultik_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/spf13/afero"
|
||||
"github.com/stretchr/testify/require"
|
||||
"sneak.berlin/go/vaultik/internal/log"
|
||||
"sneak.berlin/go/vaultik/internal/ui"
|
||||
"sneak.berlin/go/vaultik/internal/vaultik"
|
||||
)
|
||||
|
||||
// TestDeepVerifyAcceptsHealthyAndRejectsCorruptBlob backs up a real
|
||||
// snapshot with the on-disk storage backend, runs deep verification on
|
||||
// it, then flips a byte inside one stored blob and runs deep
|
||||
// verification again. A healthy snapshot must pass; a corrupted blob
|
||||
// must fail. The healthy case is the regression guard: deep
|
||||
// verification used to hash the encrypted blob bytes and compare them
|
||||
// to the blob's ID (the double SHA256 of the plaintext), so it reported
|
||||
// every healthy blob as corrupt.
|
||||
func TestDeepVerifyAcceptsHealthyAndRejectsCorruptBlob(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
t.Parallel()
|
||||
|
||||
fs := afero.NewOsFs()
|
||||
tempDir := t.TempDir()
|
||||
|
||||
dataDir := filepath.Join(tempDir, "source")
|
||||
storeDir := filepath.Join(tempDir, "remote")
|
||||
dbPath := filepath.Join(tempDir, "index.sqlite")
|
||||
|
||||
chunkSize := int64(64 * 1024)
|
||||
maxBlobSize := int64(512 * 1024)
|
||||
|
||||
// One file large enough to span several chunks within a single blob.
|
||||
require.NoError(t, fs.MkdirAll(dataDir, 0o755))
|
||||
require.NoError(t, afero.WriteFile(fs,
|
||||
filepath.Join(dataDir, "data.bin"),
|
||||
bytesPattern("deep-", int(chunkSize*3)), 0o644))
|
||||
|
||||
ctx := context.Background()
|
||||
|
||||
// runFileStorageBackup writes a real snapshot to storeDir and closes
|
||||
// the source index, so verification runs from remote bytes only.
|
||||
cfg, storer, snapshotID := runFileStorageBackup(
|
||||
ctx, t, fs, dataDir, storeDir, dbPath, chunkSize, maxBlobSize)
|
||||
|
||||
newVerifier := func() *vaultik.Vaultik {
|
||||
v := &vaultik.Vaultik{
|
||||
Config: cfg,
|
||||
Storage: storer,
|
||||
Fs: fs,
|
||||
Stdout: io.Discard,
|
||||
Stderr: io.Discard,
|
||||
UI: ui.NewWithColor(io.Discard, false),
|
||||
}
|
||||
v.SetContext(ctx)
|
||||
|
||||
return v
|
||||
}
|
||||
|
||||
require.NoError(t,
|
||||
newVerifier().RunDeepVerify(snapshotID, &vaultik.VerifyOptions{Deep: true}),
|
||||
"deep verify should pass on a healthy snapshot")
|
||||
|
||||
// Flip a byte inside one blob without changing its length, so the
|
||||
// blob-existence and size checks still pass and verification reaches
|
||||
// the blob-content stage.
|
||||
corruptOneBlob(t, fs, filepath.Join(storeDir, "blobs"))
|
||||
|
||||
require.Error(t,
|
||||
newVerifier().RunDeepVerify(snapshotID, &vaultik.VerifyOptions{Deep: true}),
|
||||
"deep verify should fail on a corrupted blob")
|
||||
}
|
||||
|
||||
// corruptOneBlob flips a middle byte of the first blob file found under
|
||||
// blobsDir, leaving the file length unchanged.
|
||||
func corruptOneBlob(t *testing.T, fs afero.Fs, blobsDir string) {
|
||||
t.Helper()
|
||||
|
||||
var blobPath string
|
||||
|
||||
err := afero.Walk(fs, blobsDir,
|
||||
func(path string, info os.FileInfo, err error) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if blobPath == "" && !info.IsDir() {
|
||||
blobPath = path
|
||||
}
|
||||
|
||||
return nil
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, blobPath, "expected at least one blob on disk")
|
||||
|
||||
data, err := afero.ReadFile(fs, blobPath)
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, data)
|
||||
|
||||
data[len(data)/2] ^= 0xff
|
||||
require.NoError(t, afero.WriteFile(fs, blobPath, data, 0o644))
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package vaultik //nolint:testpackage // exercises unexported count helpers
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"sneak.berlin/go/vaultik/internal/database"
|
||||
"sneak.berlin/go/vaultik/internal/log"
|
||||
)
|
||||
|
||||
// TestTableCountForReportSurfacesReadFailure is the regression guard for
|
||||
// the discarded-error bug: getTableCount for a table its query cannot
|
||||
// resolve must not silently become 0. A count that could not be read is
|
||||
// reported as unknown, which a reader can tell apart from an empty table.
|
||||
//
|
||||
//nolint:paralleltest // installs the global logger via log.Initialize
|
||||
func TestTableCountForReportSurfacesReadFailure(t *testing.T) {
|
||||
log.Initialize(log.Config{})
|
||||
|
||||
ctx := context.Background()
|
||||
|
||||
db, err := database.New(ctx, ":memory:")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
v := &Vaultik{DB: db}
|
||||
v.SetContext(ctx)
|
||||
|
||||
// A table present in the schema reads as a real count.
|
||||
blobs := v.tableCountForReport("blobs")
|
||||
require.NotNil(t, blobs, "an existing table must read as a real count")
|
||||
assert.Equal(t, int64(0), *blobs)
|
||||
|
||||
// A syntactically valid name the sanitizer accepts but whose table
|
||||
// the query cannot resolve is the exact shape #96 describes: a
|
||||
// would-be loud failure that used to be discarded into a 0.
|
||||
_, err = v.getTableCount("snapshots_missing")
|
||||
require.Error(t, err, "a query against a nonexistent table must fail")
|
||||
|
||||
missing := v.tableCountForReport("snapshots_missing")
|
||||
assert.Nil(t, missing, "a failed read is unknown, not a count")
|
||||
|
||||
// The rendered count for a failed read must say unknown, never 0.
|
||||
assert.Equal(t, countUnknown, countText(missing))
|
||||
assert.NotEqual(t, "0", countText(missing))
|
||||
}
|
||||
|
||||
// TestCountTextDistinguishesEmptyFromUnknown pins the distinction the
|
||||
// output has to preserve: 0 means the table was empty, "unknown" means
|
||||
// the count could not be read.
|
||||
func TestCountTextDistinguishesEmptyFromUnknown(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
zero := int64(0)
|
||||
seven := int64(7)
|
||||
|
||||
assert.Equal(t, "0", countText(&zero))
|
||||
assert.Equal(t, "7", countText(&seven))
|
||||
assert.Equal(t, countUnknown, countText(nil))
|
||||
}
|
||||
|
||||
// TestCountDiffUnknownWhenEitherSideUnknown checks that a delta computed
|
||||
// from an unreadable count is itself unknown rather than a plausible
|
||||
// number.
|
||||
func TestCountDiffUnknownWhenEitherSideUnknown(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
before := int64(10)
|
||||
after := int64(3)
|
||||
|
||||
require.NotNil(t, countDiff(&before, &after))
|
||||
assert.Equal(t, int64(7), *countDiff(&before, &after))
|
||||
|
||||
assert.Nil(t, countDiff(nil, &after), "unknown before yields unknown delta")
|
||||
assert.Nil(t, countDiff(&before, nil), "unknown after yields unknown delta")
|
||||
assert.Nil(t, countDiff(nil, nil))
|
||||
}
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -1540,12 +1541,17 @@ func (v *Vaultik) outputRemoveJSON(result *RemoveResult) error {
|
||||
return encoder.Encode(result)
|
||||
}
|
||||
|
||||
// PruneResult contains statistics about the prune operation
|
||||
// PruneResult contains statistics about the prune operation.
|
||||
// SnapshotsDeleted counts snapshots actually deleted. FilesDeleted,
|
||||
// ChunksDeleted, and BlobsDeleted are derived from before/after row
|
||||
// counts of the local index; each is nil when a count could not be read,
|
||||
// so an unreadable count is reported as unknown rather than silently
|
||||
// as 0.
|
||||
type PruneResult struct {
|
||||
SnapshotsDeleted int64
|
||||
FilesDeleted int64
|
||||
ChunksDeleted int64
|
||||
BlobsDeleted int64
|
||||
FilesDeleted *int64
|
||||
ChunksDeleted *int64
|
||||
BlobsDeleted *int64
|
||||
}
|
||||
|
||||
// PruneDatabase removes incomplete snapshots and orphaned files, chunks,
|
||||
@@ -1560,7 +1566,7 @@ func (v *Vaultik) PruneDatabase() (*PruneResult, error) {
|
||||
result := &PruneResult{}
|
||||
|
||||
// Snapshot counts before deletion of incompletes.
|
||||
snapshotCountBefore, _ := v.getTableCount("snapshots")
|
||||
snapshotCountBefore := v.tableCountForReport("snapshots")
|
||||
|
||||
// First, delete any incomplete snapshots
|
||||
incompleteSnapshots, err := v.Repositories.Snapshots.GetIncompleteSnapshots(v.ctx)
|
||||
@@ -1575,9 +1581,9 @@ func (v *Vaultik) PruneDatabase() (*PruneResult, error) {
|
||||
}
|
||||
|
||||
// Get counts before cleanup for reporting
|
||||
fileCountBefore, _ := v.getTableCount("files")
|
||||
chunkCountBefore, _ := v.getTableCount("chunks")
|
||||
blobCountBefore, _ := v.getTableCount("blobs")
|
||||
fileCountBefore := v.tableCountForReport("files")
|
||||
chunkCountBefore := v.tableCountForReport("chunks")
|
||||
blobCountBefore := v.tableCountForReport("blobs")
|
||||
|
||||
// Run the cleanup
|
||||
err = v.SnapshotManager.CleanupOrphanedData(v.ctx)
|
||||
@@ -1586,36 +1592,83 @@ func (v *Vaultik) PruneDatabase() (*PruneResult, error) {
|
||||
}
|
||||
|
||||
// Get counts after cleanup
|
||||
fileCountAfter, _ := v.getTableCount("files")
|
||||
chunkCountAfter, _ := v.getTableCount("chunks")
|
||||
blobCountAfter, _ := v.getTableCount("blobs")
|
||||
fileCountAfter := v.tableCountForReport("files")
|
||||
chunkCountAfter := v.tableCountForReport("chunks")
|
||||
blobCountAfter := v.tableCountForReport("blobs")
|
||||
|
||||
result.FilesDeleted = fileCountBefore - fileCountAfter
|
||||
result.ChunksDeleted = chunkCountBefore - chunkCountAfter
|
||||
result.BlobsDeleted = blobCountBefore - blobCountAfter
|
||||
result.FilesDeleted = countDiff(fileCountBefore, fileCountAfter)
|
||||
result.ChunksDeleted = countDiff(chunkCountBefore, chunkCountAfter)
|
||||
result.BlobsDeleted = countDiff(blobCountBefore, blobCountAfter)
|
||||
|
||||
log.Info("Local database prune complete",
|
||||
"incomplete_snapshots", result.SnapshotsDeleted,
|
||||
"orphaned_files", result.FilesDeleted,
|
||||
"orphaned_chunks", result.ChunksDeleted,
|
||||
"orphaned_blobs", result.BlobsDeleted,
|
||||
"orphaned_files", countText(result.FilesDeleted),
|
||||
"orphaned_chunks", countText(result.ChunksDeleted),
|
||||
"orphaned_blobs", countText(result.BlobsDeleted),
|
||||
)
|
||||
|
||||
snapshotCountAfter := snapshotCountBefore - result.SnapshotsDeleted
|
||||
// Snapshots remaining after removing the incomplete ones; unknown if
|
||||
// the pre-prune snapshot count could not be read.
|
||||
snapshotsRemain := countDiff(snapshotCountBefore, &result.SnapshotsDeleted)
|
||||
|
||||
v.UI.Completef("Pruned local index database.")
|
||||
v.UI.Detailf("Incomplete snapshots: %d removed (%d remain).",
|
||||
result.SnapshotsDeleted, snapshotCountAfter)
|
||||
v.UI.Detailf("Orphaned files: %d removed (%d remain).",
|
||||
result.FilesDeleted, fileCountAfter)
|
||||
v.UI.Detailf("Orphaned chunks: %d removed (%d remain).",
|
||||
result.ChunksDeleted, chunkCountAfter)
|
||||
v.UI.Detailf("Orphaned blobs: %d removed (%d remain).",
|
||||
result.BlobsDeleted, blobCountAfter)
|
||||
v.UI.Detailf("Incomplete snapshots: %s removed (%s remain).",
|
||||
countText(&result.SnapshotsDeleted), countText(snapshotsRemain))
|
||||
v.UI.Detailf("Orphaned files: %s removed (%s remain).",
|
||||
countText(result.FilesDeleted), countText(fileCountAfter))
|
||||
v.UI.Detailf("Orphaned chunks: %s removed (%s remain).",
|
||||
countText(result.ChunksDeleted), countText(chunkCountAfter))
|
||||
v.UI.Detailf("Orphaned blobs: %s removed (%s remain).",
|
||||
countText(result.BlobsDeleted), countText(blobCountAfter))
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// countUnknown is what a count reads as when its query could not be run,
|
||||
// distinct from "0", which means the table really was empty.
|
||||
const countUnknown = "unknown"
|
||||
|
||||
// tableCountForReport returns the row count of a table for the prune
|
||||
// summary, or nil if the count could not be read. A read failure is
|
||||
// logged at warn — visible even under --json, which routes warnings to
|
||||
// stderr — and then rendered as unknown rather than silently becoming 0,
|
||||
// so a broken query is a visible failure instead of a plausible wrong
|
||||
// number.
|
||||
func (v *Vaultik) tableCountForReport(tableName string) *int64 {
|
||||
count, err := v.getTableCount(tableName)
|
||||
if err != nil {
|
||||
log.Warn("could not read table row count for prune summary",
|
||||
"table", tableName, "error", err)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
return &count
|
||||
}
|
||||
|
||||
// countDiff returns before-after, or nil if either count is unknown so
|
||||
// that an unreadable count does not collapse into a plausible delta.
|
||||
func countDiff(before, after *int64) *int64 {
|
||||
if before == nil || after == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
diff := *before - *after
|
||||
|
||||
return &diff
|
||||
}
|
||||
|
||||
// countText renders a count that may be unknown: nil (the read failed)
|
||||
// becomes "unknown", never "0", so a reader can tell an empty table from
|
||||
// one that could not be queried.
|
||||
func countText(count *int64) string {
|
||||
if count == nil {
|
||||
return countUnknown
|
||||
}
|
||||
|
||||
return strconv.FormatInt(*count, 10)
|
||||
}
|
||||
|
||||
// validTableNameRe matches table names containing only lowercase
|
||||
// alphanumeric characters and underscores.
|
||||
var validTableNameRe = regexp.MustCompile(`^[a-z0-9_]+$`)
|
||||
|
||||
+19
-14
@@ -344,12 +344,8 @@ func (v *Vaultik) verifyBlob(blobInfo snapshot.BlobInfo, db *sql.DB) error {
|
||||
return fmt.Errorf("failed to get decryptor: %w", err)
|
||||
}
|
||||
|
||||
// Hash the encrypted blob data as it streams through to decryption
|
||||
blobHasher := sha256.New()
|
||||
teeReader := io.TeeReader(reader, blobHasher)
|
||||
|
||||
// Decrypt blob (reading through teeReader to hash encrypted data)
|
||||
decryptedReader, err := decryptor.DecryptStream(teeReader)
|
||||
// Decrypt blob
|
||||
decryptedReader, err := decryptor.DecryptStream(reader)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to decrypt: %w", err)
|
||||
}
|
||||
@@ -361,12 +357,19 @@ func (v *Vaultik) verifyBlob(blobInfo snapshot.BlobInfo, db *sql.DB) error {
|
||||
}
|
||||
defer decompressor.Close()
|
||||
|
||||
chunkCount, err := v.verifyBlobChunks(db, blobInfo.Hash, decompressor)
|
||||
// A blob's hash — its remote name — is the double SHA256 of its
|
||||
// decompressed plaintext (see blobgen.Writer.Sum256), not of the
|
||||
// encrypted bytes. Hash the plaintext as chunk verification streams
|
||||
// it, then compare on completion.
|
||||
plaintextHasher := sha256.New()
|
||||
hashedStream := io.TeeReader(decompressor, plaintextHasher)
|
||||
|
||||
chunkCount, err := v.verifyBlobChunks(db, blobInfo.Hash, hashedStream)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = v.verifyBlobFinalIntegrity(decompressor, blobHasher, blobInfo.Hash)
|
||||
err = v.verifyBlobFinalIntegrity(hashedStream, plaintextHasher, blobInfo.Hash)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -470,14 +473,13 @@ func (v *Vaultik) verifyBlobChunks(
|
||||
}
|
||||
|
||||
// verifyBlobFinalIntegrity checks that no trailing data exists in the
|
||||
// decompressed stream and that the encrypted blob hash matches the
|
||||
// expected value.
|
||||
// decompressed stream and that the blob hash matches the expected value.
|
||||
func (v *Vaultik) verifyBlobFinalIntegrity(
|
||||
decompressor io.Reader, blobHasher hash.Hash, expectedHash string,
|
||||
plaintext io.Reader, plaintextHasher hash.Hash, expectedHash string,
|
||||
) error {
|
||||
// Verify no remaining data in blob - if the chunk list is accurate,
|
||||
// the blob should be fully consumed.
|
||||
remaining, err := io.Copy(io.Discard, decompressor)
|
||||
remaining, err := io.Copy(io.Discard, plaintext)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to check for remaining blob data: %w", err)
|
||||
}
|
||||
@@ -486,8 +488,11 @@ func (v *Vaultik) verifyBlobFinalIntegrity(
|
||||
return fmt.Errorf("%w: %d bytes", errTrailingBlobData, remaining)
|
||||
}
|
||||
|
||||
// Verify blob hash matches the encrypted data we downloaded
|
||||
calculatedBlobHash := hex.EncodeToString(blobHasher.Sum(nil))
|
||||
// The blob hash is the double SHA256 of its plaintext content.
|
||||
firstHash := plaintextHasher.Sum(nil)
|
||||
secondHash := sha256.Sum256(firstHash)
|
||||
calculatedBlobHash := hex.EncodeToString(secondHash[:])
|
||||
|
||||
if calculatedBlobHash != expectedHash {
|
||||
return fmt.Errorf("%w: calculated %s, expected %s",
|
||||
errBlobHashMismatch, calculatedBlobHash, expectedHash)
|
||||
|
||||
Reference in New Issue
Block a user