Files
sfdupes/scan_test.go
sneak a668e8aab8
check / check (push) Failing after 1s
Test the -x filesystem-boundary rejection branch (closes #17)
The -x descent rules in subdirJob had no test executing the actual
rejection: the only case proved -x skips nothing when everything is on
one filesystem. Add tests that call subdirJob directly with a fabricated
operand device, so no second real filesystem is needed:

- reject descent when the subdirectory is on a different device;
- bypass the check when the operand's own device is unknown;
- warn and refuse descent when the entry's Info() fails;
- cross the boundary by default (no -x).

Tests only; scan.go is unchanged.

Model: opus-4-8
2026-09-21 07:28:19 +00:00

1210 lines
29 KiB
Go

package main
import (
"bytes"
"context"
"crypto/sha256"
"database/sql"
"encoding/hex"
"errors"
"fmt"
"io/fs"
"os"
"path/filepath"
"runtime"
"slices"
"strconv"
"strings"
"testing"
"time"
)
// writeFile creates a file with the given content and returns its path.
func writeFile(t *testing.T, dir, name string, data []byte) string {
t.Helper()
p := filepath.Join(dir, name)
err := os.MkdirAll(filepath.Dir(p), 0o750)
if err != nil {
t.Fatal(err)
}
err = os.WriteFile(p, data, 0o600)
if err != nil {
t.Fatal(err)
}
return p
}
// hexSum returns the lowercase-hex SHA-256 of data.
func hexSum(data []byte) string {
s := sha256.Sum256(data)
return hex.EncodeToString(s[:])
}
// pattern returns n bytes of deterministic content seeded by tag.
func pattern(tag byte, n int) []byte {
data := make([]byte, n)
for i := range data {
data[i] = tag ^ byte(i)
}
return data
}
func TestHashHeadTail(t *testing.T) {
t.Parallel()
dir := t.TempDir()
cases := []struct {
name string
data []byte
}{
{"empty", nil},
{"one-byte", []byte("x")},
{"under-one-chunk", pattern(1, chunk-1)},
{"exactly-one-chunk", pattern(2, chunk)},
{"overlapping-reads", pattern(3, chunk+chunk/2)},
{"exactly-two-chunks", pattern(4, 2*chunk)},
{"beyond-two-chunks", pattern(5, 3*chunk)},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
t.Parallel()
p := writeFile(t, dir, c.name, c.data)
head, tail, err := hashHeadTail(p, int64(len(c.data)))
if err != nil {
t.Fatalf("hashHeadTail: %v", err)
}
n := min(chunk, len(c.data))
if want := hexSum(c.data[:n]); head != want {
t.Errorf("head = %s, want %s", head, want)
}
if want := hexSum(c.data[len(c.data)-n:]); tail != want {
t.Errorf("tail = %s, want %s", tail, want)
}
})
}
}
func TestHashHeadTailErrors(t *testing.T) {
t.Parallel()
dir := t.TempDir()
_, _, err := hashHeadTail(filepath.Join(dir, "missing"), 1)
if err == nil {
t.Error("no error for a missing file")
}
// A zero-length file has constant hashes and is never opened: even
// a missing path succeeds.
head, tail, err := hashHeadTail(filepath.Join(dir, "missing"), 0)
if err != nil || head != emptyHash || tail != emptyHash {
t.Errorf("empty: head=%q tail=%q err=%v, want constant hashes",
head, tail, err)
}
// A file that shrank between the stat and hash passes: reading at
// the stat-reported size must fail rather than emit wrong hashes.
p := writeFile(t, dir, "shrunk", []byte("tiny"))
_, _, err = hashHeadTail(p, int64(2*chunk))
if err == nil {
t.Error("no error when the stat size exceeds the file size")
}
}
// collectWalk runs a walk over roots and returns the emitted records
// and the number of warning events.
func collectWalk(t *testing.T, roots []string, oneFS bool,
workers int,
) ([]fileRec, int) {
t.Helper()
var (
recs []fileRec
errs int
)
for ev := range startWalk(t.Context(), roots, oneFS, workers) {
if ev.fail {
errs++
continue
}
recs = append(recs, ev.rec)
}
return recs, errs
}
// walkedPaths returns the sorted paths of the walked records.
func walkedPaths(recs []fileRec) []string {
paths := make([]string, 0, len(recs))
for _, r := range recs {
paths = append(paths, r.path)
}
slices.Sort(paths)
return paths
}
func TestWalk(t *testing.T) {
t.Parallel()
dir := t.TempDir()
want := []string{
writeFile(t, dir, "a.txt", []byte("a")),
writeFile(t, dir, "sub/b.txt", []byte("bb")),
writeFile(t, dir, "sub/deeper/c.txt", []byte("ccc")),
}
slices.Sort(want)
// Files under a .zfs directory must never be walked.
writeFile(t, dir, ".zfs/snapshot/hourly/a.txt", []byte("a"))
// Symlinks are skipped, not followed.
err := os.Symlink(want[0], filepath.Join(dir, "link"))
if err != nil {
t.Fatal(err)
}
recs, errs := collectWalk(t, []string{dir}, false, 4)
if errs != 0 {
t.Fatalf("errs = %d, want 0", errs)
}
if got := walkedPaths(recs); !slices.Equal(got, want) {
t.Fatalf("paths = %q, want %q", got, want)
}
// The walk stats each file as it is discovered: every record must
// carry the real size and a plausible mtime.
for _, r := range recs {
if r.size < 1 || r.size > 3 {
t.Errorf("%s: size = %d, want 1..3", r.path, r.size)
}
if r.mtime <= 0 {
t.Errorf("%s: mtime = %d, want positive", r.path, r.mtime)
}
}
}
func TestWalkDeepAndWide(t *testing.T) {
t.Parallel()
// Exercise the dispatcher with more directories than workers and
// with nesting deeper than the worker count.
dir := t.TempDir()
deep := "deep" + strings.Repeat("/d", 30)
want := make([]string, 0, 41)
want = append(want, writeFile(t, dir, deep+"/f", []byte("x")))
for i := range 40 {
want = append(want, writeFile(t, dir,
fmt.Sprintf("wide/%02d/f", i), []byte("y")))
}
slices.Sort(want)
recs, errs := collectWalk(t, []string{dir}, false, 8)
if errs != 0 {
t.Fatalf("errs = %d, want 0", errs)
}
if got := walkedPaths(recs); !slices.Equal(got, want) {
t.Fatalf("walked %d paths, want %d", len(got), len(want))
}
}
func TestWalkMultipleRoots(t *testing.T) {
t.Parallel()
rootA := t.TempDir()
rootB := t.TempDir()
want := []string{
writeFile(t, rootA, "a1", []byte("1")),
writeFile(t, rootA, "sub/a2", []byte("2")),
writeFile(t, rootB, "b1", []byte("3")),
}
slices.Sort(want)
// Operands are enumerated concurrently by the shared pool; order
// is unspecified.
recs, errs := collectWalk(t, []string{rootA, rootB}, false, 4)
if errs != 0 {
t.Fatalf("errs = %d, want 0", errs)
}
if got := walkedPaths(recs); !slices.Equal(got, want) {
t.Fatalf("paths = %q, want %q", got, want)
}
}
func TestWalkFileAndSymlinkOperands(t *testing.T) {
t.Parallel()
dir := t.TempDir()
f := writeFile(t, dir, "plain", []byte("data"))
link := filepath.Join(dir, "link")
err := os.Symlink(f, link)
if err != nil {
t.Fatal(err)
}
// A regular-file operand is emitted as itself, statted.
recs, errs := collectWalk(t, []string{f}, false, 2)
if errs != 0 || len(recs) != 1 || recs[0].path != f || recs[0].size != 4 {
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
}
// A symlink operand is not followed and yields nothing.
recs, errs = collectWalk(t, []string{link}, false, 2)
if errs != 0 || len(recs) != 0 {
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
}
}
func TestWalkOneFilesystemSameFS(t *testing.T) {
t.Parallel()
// Everything in one filesystem: -x must not skip anything.
dir := t.TempDir()
want := []string{
writeFile(t, dir, "a", []byte("a")),
writeFile(t, dir, "sub/deep/b", []byte("b")),
}
recs, errs := collectWalk(t, []string{dir}, true, 4)
if errs != 0 {
t.Fatalf("errs = %d, want 0", errs)
}
if got := walkedPaths(recs); !slices.Equal(got, want) {
t.Fatalf("paths = %q, want %q", got, want)
}
}
func TestDeviceOfInfo(t *testing.T) {
t.Parallel()
dir := t.TempDir()
fi1, err := os.Lstat(dir)
if err != nil {
t.Fatal(err)
}
fi2, err := os.Lstat(dir)
if err != nil {
t.Fatal(err)
}
dev1, ok1 := deviceOfInfo(fi1)
dev2, ok2 := deviceOfInfo(fi2)
if !ok1 || !ok2 || dev1 != dev2 {
t.Fatalf("deviceOfInfo unstable: %d/%v vs %d/%v",
dev1, ok1, dev2, ok2)
}
}
// dirEntryFor returns the fs.DirEntry for name within dir, obtained via
// the same os.ReadDir the walk uses, so it carries a real Info().
func dirEntryFor(t *testing.T, dir, name string) fs.DirEntry {
t.Helper()
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatal(err)
}
for _, e := range entries {
if e.Name() == name {
return e
}
}
t.Fatalf("entry %q not found in %q", name, dir)
return nil
}
// callSubdirJob runs subdirJob against e under parent, collecting any
// warning events it emits (subdirJob emits at most one).
func callSubdirJob(t *testing.T, p string, e fs.DirEntry,
parent dirJob, oneFS bool,
) (dirJob, bool, []walkEvent) {
t.Helper()
events := make(chan walkEvent, 1)
job, ok := subdirJob(t.Context(), p, e, parent, oneFS, events)
close(events)
var evs []walkEvent
for ev := range events {
evs = append(evs, ev)
}
return job, ok, evs
}
// TestSubdirJobOneFilesystem exercises the -x boundary check in
// subdirJob directly, so no second real filesystem is needed. The
// subdirectory's real device is compared against a fabricated operand
// device.
func TestSubdirJobOneFilesystem(t *testing.T) {
t.Parallel()
dir := t.TempDir()
sub := filepath.Join(dir, "sub")
err := os.Mkdir(sub, 0o750)
if err != nil {
t.Fatal(err)
}
info, err := os.Lstat(sub)
if err != nil {
t.Fatal(err)
}
dev, ok := deviceOfInfo(info)
if !ok {
t.Skip("platform exposes no device id")
}
// A device the subdirectory is not on, standing in for an operand
// rooted on a different filesystem.
otherDev := dev + 1
e := dirEntryFor(t, dir, "sub")
cases := []struct {
name string
oneFS bool
parent dirJob
wantOK bool
}{
// -x on, subdirectory on a different device than its operand:
// descent is refused.
{"reject across boundary", true,
dirJob{rootDev: otherDev, rootDevOK: true}, false},
// -x on but the operand's own device is unknown: the boundary
// check is bypassed and descent proceeds.
{"bypass when root device unknown", true,
dirJob{rootDev: otherDev, rootDevOK: false}, true},
// Default (no -x): boundaries are crossed even onto a different
// device.
{"cross by default", false,
dirJob{rootDev: otherDev, rootDevOK: true}, true},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
t.Parallel()
job, ok, evs := callSubdirJob(t, sub, e, tc.parent, tc.oneFS)
if ok != tc.wantOK {
t.Fatalf("accepted = %v, want %v", ok, tc.wantOK)
}
if len(evs) != 0 {
t.Fatalf("unexpected events: %+v", evs)
}
if ok && job.path != sub {
t.Fatalf("job.path = %q, want %q", job.path, sub)
}
})
}
}
// errInfoUnavailable is returned by errDirEntry.Info().
var errInfoUnavailable = errors.New("info unavailable")
// errDirEntry is a directory entry whose Info() always fails, driving
// subdirJob's stat-error branch deterministically.
type errDirEntry struct{ name string }
func (e errDirEntry) Name() string { return e.name }
func (errDirEntry) IsDir() bool { return true }
func (errDirEntry) Type() fs.FileMode {
return fs.ModeDir
}
func (errDirEntry) Info() (fs.FileInfo, error) {
return nil, errInfoUnavailable
}
// TestSubdirJobStatError asserts that when a subdirectory's Info()
// fails under -x, subdirJob warns and refuses descent.
func TestSubdirJobStatError(t *testing.T) {
t.Parallel()
p := "/does/not/matter/sub"
_, ok, evs := callSubdirJob(t, p, errDirEntry{name: "sub"},
dirJob{rootDev: 1, rootDevOK: true}, true)
if ok {
t.Fatal("descent accepted after stat error, want refused")
}
if len(evs) != 1 || !evs[0].fail || !strings.Contains(evs[0].warn, p) {
t.Fatalf("want one warning naming the path, got %+v", evs)
}
}
// buildSmokeTree recreates the README smoke-test filesystem layout
// with deterministic content and returns the tree root.
func buildSmokeTree(t *testing.T) string {
t.Helper()
dir := t.TempDir()
one := pattern(10, 2000)
f1 := pattern(30, 3000)
f2 := pattern(40, 100)
writeFile(t, dir, "a/one.bin", one)
writeFile(t, dir, "b/copy.bin", one)
writeFile(t, dir, "b/copy2.bin", one)
// Same size as one.bin, different content.
writeFile(t, dir, "a/unique.bin", pattern(20, 2000))
writeFile(t, dir, "tiny1", []byte("x"))
writeFile(t, dir, "tiny2", []byte("x"))
writeFile(t, dir, "tiny3", []byte("y"))
writeFile(t, dir, "empty1", nil)
writeFile(t, dir, "empty2", nil)
writeFile(t, dir, "t1/f1", f1)
writeFile(t, dir, "t1/sub/f2", f2)
writeFile(t, dir, "t2/f1", f1)
writeFile(t, dir, "t2/sub/f2", f2)
writeFile(t, dir, "t3/f1", f1)
writeFile(t, dir, "t3/sub/f2renamed", f2)
return dir
}
// smokeTreeFiles is the number of regular files buildSmokeTree creates.
const smokeTreeFiles = 15
// syncTree synchronizes the database with the given roots and returns
// the scan stats.
func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
t.Helper()
st, err := syncScan(t.Context(), db, roots, 4, false)
if err != nil {
t.Fatalf("syncScan: %v", err)
}
return st
}
// dbRecords returns every record currently in the database.
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
t.Helper()
recs, err := loadFileRows(t.Context(), db)
if err != nil {
t.Fatal(err)
}
return recs
}
// recordByPath finds the record with the given path.
func recordByPath(t *testing.T, recs []scanRec, path string) scanRec {
t.Helper()
for _, r := range recs {
if r.path == path {
return r
}
}
t.Fatalf("no record for %q", path)
return scanRec{}
}
// recordPaths returns the sorted paths of recs.
func recordPaths(recs []scanRec) []string {
paths := make([]string, 0, len(recs))
for _, r := range recs {
paths = append(paths, r.path)
}
slices.Sort(paths)
return paths
}
// assertSmokeDupeGroups checks the file-level duplicate groups for the
// smoke tree rooted at dir.
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
t.Helper()
groups := collectDupeGroups(parsed)
if len(groups) != 5 {
t.Fatalf("len(groups) = %d, want 5", len(groups))
}
wantSizes := []int64{3000, 2000, 100, 1, 0}
for i, g := range groups {
if g.size != wantSizes[i] {
t.Errorf("groups[%d].size = %d, want %d",
i, g.size, wantSizes[i])
}
}
wantF1 := []string{
filepath.Join(dir, "t1/f1"),
filepath.Join(dir, "t2/f1"),
filepath.Join(dir, "t3/f1"),
}
if !slices.Equal(groups[0].paths, wantF1) {
t.Errorf("groups[0].paths = %q, want %q", groups[0].paths, wantF1)
}
}
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
// tree rooted at dir.
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
t.Helper()
super, dirs := buildHierarchy(parsed)
super.compute()
tg := collectTreeGroups(dirs, super)
if len(tg) != 1 {
t.Fatalf("len(tree groups) = %d, want 1", len(tg))
}
wantTrees := []string{filepath.Join(dir, "t1"), filepath.Join(dir, "t2")}
if got := groupPaths(tg)[0]; !slices.Equal(got, wantTrees) {
t.Fatalf("tree group = %q, want %q", got, wantTrees)
}
if tg[0][0].fileCount != 2 || tg[0][0].totalSize != 3100 {
t.Fatalf("tree totals: %d files %d bytes, want 2 3100",
tg[0][0].fileCount, tg[0][0].totalSize)
}
}
func TestScanPipeline(t *testing.T) {
t.Parallel()
dir := buildSmokeTree(t)
db := openTestDB(t)
st := syncTree(t, db, dir)
if st != (scanStats{added: smokeTreeFiles}) {
t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles)
}
parsed := dbRecords(t, db)
if len(parsed) != smokeTreeFiles {
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
}
assertSmokeDupeGroups(t, dir, parsed)
assertSmokeTreeGroups(t, dir, parsed)
}
func TestSyncScanUnchangedReuse(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
a := writeFile(t, dir, "a.bin", pattern(1, 500))
writeFile(t, dir, "b.bin", pattern(2, 600))
st := syncTree(t, db, dir)
if st != (scanStats{added: 2}) {
t.Fatalf("first scan stats = %+v, want 2 added", st)
}
// An immediate rescan reuses every record without reading file
// contents. Prove the files are not re-read by corrupting a stored
// hash and observing that it survives the rescan.
_, err := db.ExecContext(context.Background(),
"UPDATE files SET head = 'sentinel' WHERE path = ?", []byte(a))
if err != nil {
t.Fatal(err)
}
st = syncTree(t, db, dir)
if st != (scanStats{unchanged: 2}) {
t.Fatalf("rescan stats = %+v, want 2 unchanged", st)
}
if r := recordByPath(t, dbRecords(t, db), a); r.head != "sentinel" {
t.Fatalf("head = %q, want sentinel (file must not be re-read)",
r.head)
}
}
func TestSyncScanMtimeBump(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
a := writeFile(t, dir, "a.bin", pattern(1, 500))
syncTree(t, db, dir)
// Bump the mtime forward: the file must be re-hashed even though
// its size is unchanged.
future := time.Now().Add(time.Hour)
err := os.Chtimes(a, future, future)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, dir)
if st != (scanStats{updated: 1}) {
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
}
if r := recordByPath(t, dbRecords(t, db), a); r.mtime != future.Unix() {
t.Fatalf("mtime = %d, want %d", r.mtime, future.Unix())
}
}
func TestSyncScanAddRemove(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
a := writeFile(t, dir, "a.bin", pattern(1, 500))
writeFile(t, dir, "b.bin", pattern(2, 600))
syncTree(t, db, dir)
// Add one file, remove another.
c := writeFile(t, dir, "c.bin", pattern(3, 700))
err := os.Remove(a)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, dir)
if st != (scanStats{added: 1, removed: 1, unchanged: 1}) {
t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged",
st)
}
want := []string{filepath.Join(dir, "b.bin"), c}
if got := recordPaths(dbRecords(t, db)); !slices.Equal(got, want) {
t.Fatalf("paths = %q, want %q", got, want)
}
}
func TestSyncScanSizeChange(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
p := writeFile(t, dir, "f", pattern(1, 100))
syncTree(t, db, dir)
// Rewrite with a different size but force the mtime back to the
// recorded value: the size mismatch alone must trigger a re-hash.
old := recordByPath(t, dbRecords(t, db), p)
writeFile(t, dir, "f", pattern(1, 200))
mt := time.Unix(old.mtime, 0)
err := os.Chtimes(p, mt, mt)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, dir)
if st.updated != 1 {
t.Fatalf("stats = %+v, want 1 updated", st)
}
if got := recordByPath(t, dbRecords(t, db), p); got.size != 200 {
t.Fatalf("size = %d, want 200", got.size)
}
}
func TestSyncScanScope(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
writeFile(t, dir, "a/keep", pattern(1, 10))
gone := writeFile(t, dir, "b/gone", pattern(2, 10))
syncTree(t, db, dir)
// Deleting a file outside the rescanned root must not remove its
// record: records outside the scanned operands are untouched.
err := os.Remove(gone)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, filepath.Join(dir, "a"))
if st.removed != 0 || st.unchanged != 1 {
t.Fatalf("subtree stats = %+v, want 0 removed 1 unchanged", st)
}
if got := recordPaths(dbRecords(t, db)); len(got) != 2 {
t.Fatalf("records = %q, want both retained", got)
}
// Rescanning the parent now removes the vanished file's record.
st = syncTree(t, db, dir)
if st.removed != 1 {
t.Fatalf("parent stats = %+v, want 1 removed", st)
}
}
func TestSyncScanRemovesNonRegular(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
p := writeFile(t, dir, "f", pattern(1, 10))
keep := writeFile(t, dir, "g", pattern(2, 10))
syncTree(t, db, dir)
// Replace the file with a symlink: it is no longer walked, so its
// record must be deleted.
err := os.Remove(p)
if err != nil {
t.Fatal(err)
}
err = os.Symlink(keep, p)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, dir)
if st.removed != 1 || st.unchanged != 1 {
t.Fatalf("stats = %+v, want 1 removed 1 unchanged", st)
}
}
func TestSyncScanOverlappingRoots(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
writeFile(t, dir, "sub/f", pattern(1, 10))
// A file reachable via two overlapping operands is deduplicated
// by path in the shared walk and processed once.
st := syncTree(t, db, dir, filepath.Join(dir, "sub"))
if st != (scanStats{added: 1}) {
t.Fatalf("stats = %+v, want 1 added", st)
}
if got := recordPaths(dbRecords(t, db)); len(got) != 1 {
t.Fatalf("records = %q, want exactly one", got)
}
}
func TestScanSkipsUniqueSizes(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
a := writeFile(t, dir, "a.bin", pattern(1, 500))
writeFile(t, dir, "b.bin", pattern(2, 600))
// Neither size is shared, so neither file is read: both records
// are written without hashes and no duplicates are reported.
st := syncTree(t, db, dir)
if st != (scanStats{added: 2}) {
t.Fatalf("stats = %+v, want 2 added", st)
}
recs := dbRecords(t, db)
for _, r := range recs {
if r.head != "" || r.tail != "" {
t.Errorf("%s: head = %q tail = %q, want unhashed",
r.path, r.head, r.tail)
}
}
if groups := collectDupeGroups(recs); len(groups) != 0 {
t.Fatalf("groups = %+v, want none from unhashed records", groups)
}
// A new same-size file makes 500 a shared size: the next scan
// hashes both the new file and the previously unhashed unchanged
// one, and they group as duplicates.
c := writeFile(t, dir, "c.bin", pattern(1, 500))
st = syncTree(t, db, dir)
if st != (scanStats{added: 1, updated: 1, unchanged: 1}) {
t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged",
st)
}
groups := collectDupeGroups(dbRecords(t, db))
if len(groups) != 1 {
t.Fatalf("groups = %+v, want the a/c pair", groups)
}
if want := []string{a, c}; !slices.Equal(groups[0].paths, want) {
t.Fatalf("group paths = %q, want %q", groups[0].paths, want)
}
}
func TestTreesUnhashedNeverEqual(t *testing.T) {
t.Parallel()
// Two trees identical except for unhashed same-name, same-size
// files (possible when the trees were scanned separately) must not
// compare equal: unhashed content is unknown.
shared := pattern(1, 100)
recs := []scanRec{
{path: "/x/t1/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
{path: "/x/t2/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
{path: "/x/t1/u", size: 50},
{path: "/x/t2/u", size: 50},
}
super, dirs := buildHierarchy(recs)
super.compute()
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
t.Fatalf("tree groups = %d, want 0 (unhashed files differ)",
len(tg))
}
}
func TestScanHardlinksReadOnce(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
a := writeFile(t, dir, "a.bin", pattern(1, 300))
b := filepath.Join(dir, "b.bin")
err := os.Link(a, b)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, dir)
if st != (scanStats{added: 2}) {
t.Fatalf("stats = %+v, want 2 added", st)
}
// Both paths share the single read's hashes and group together.
recs := dbRecords(t, db)
ra := recordByPath(t, recs, a)
rb := recordByPath(t, recs, b)
if ra.head == "" || ra.head != rb.head || ra.tail != rb.tail {
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
}
if groups := collectDupeGroups(recs); len(groups) != 1 {
t.Fatalf("groups = %+v, want the hardlink pair", groups)
}
}
func TestScanHardlinkRunFailsTogether(t *testing.T) {
t.Parallel()
dir := t.TempDir()
db := openTestDB(t)
a := writeFile(t, dir, "a.bin", pattern(1, 300))
err := os.Link(a, filepath.Join(dir, "b.bin"))
if err != nil {
t.Fatal(err)
}
// Unreadable inode: the run's single read fails, so both paths are
// skipped — proof that hard links are read once, not per path.
err = os.Chmod(a, 0)
if err != nil {
t.Fatal(err)
}
st := syncTree(t, db, dir)
if st.skipped != 2 || st.added != 0 {
t.Fatalf("stats = %+v, want both hardlink paths skipped", st)
}
}
// injectedWriteFailure is the message the injected database trigger
// aborts with, so the test can recognize its own failure in the error
// the scan reports.
const injectedWriteFailure = "injected write failure"
// hashLeakFiles is the size of the fixture for the hash-phase failure
// test. The batch commit inside the hash phase is what fails, so the
// tree must hold more than updateBatchSize files for the failure to
// happen at all, and the surplus over that is what is still queued
// when it does. That surplus is 2*workQueueDepth, which jobs, results
// and the workers in flight between them absorb exactly, so the feeder
// itself drains and exits; what an abandoned pool leaves parked is
// every worker, each holding a result nobody will ever receive, plus
// the goroutine waiting on them. That is what this test detects, and
// its margin over detecting nothing at all is the worker count —
// worth knowing before changing hashLeakWorkers or workQueueDepth.
const hashLeakFiles = updateBatchSize + 2*workQueueDepth
// hashLeakWorkers is the worker count for that scan: a fixed, modest
// number keeps the leak deterministic on any machine.
const hashLeakWorkers = 4
// goroutineSettle bounds how long a goroutine count is given to come
// back down to its target. Only a failing run ever waits this long.
const goroutineSettle = 5 * time.Second
// goroutinePoll is the interval between goroutine-count samples.
const goroutinePoll = 10 * time.Millisecond
// writeEmptyFiles creates n empty files directly in dir. Zero-length
// files are never opened by the hasher — their hashes are constant —
// so a fixture this size costs directory entries and no read I/O,
// while still queueing n runs through the hash pool.
func writeEmptyFiles(t *testing.T, dir string, n int) {
t.Helper()
for i := range n {
err := os.WriteFile(
filepath.Join(dir, strconv.Itoa(i)), nil, 0o600)
if err != nil {
t.Fatal(err)
}
}
}
// injectWriteFailure creates a scan database at path carrying the real
// schema plus a trigger that aborts every insert. Reads are untouched,
// so a scan loads its index and walks normally and then fails on the
// first record it tries to commit — a genuine database write failure
// partway through the hash phase.
func injectWriteFailure(t *testing.T, path string) {
t.Helper()
db, err := openScanDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
_, err = db.ExecContext(t.Context(),
"CREATE TRIGGER refuse_insert BEFORE INSERT ON files "+
"BEGIN SELECT RAISE(ABORT, '"+injectedWriteFailure+"'); END")
if err != nil {
t.Fatal(err)
}
err = db.Close()
if err != nil {
t.Fatal(err)
}
}
// baselineGoroutines waits for the goroutine count to stop moving and
// returns it. Handles closed by earlier tests take a moment to reap
// their driver goroutines, so a single sample would make the baseline
// itself flaky.
func baselineGoroutines(t *testing.T) int {
t.Helper()
deadline := time.Now().Add(goroutineSettle)
last := runtime.NumGoroutine()
for time.Now().Before(deadline) {
time.Sleep(goroutinePoll)
n := runtime.NumGoroutine()
if n == last {
return n
}
last = n
}
return last
}
// settledGoroutines polls runtime.NumGoroutine until it is back at or
// below want and returns the last count seen. Polling, rather than one
// sample after a fixed sleep, is what keeps this from being a race
// between the assertion and goroutines that are already exiting.
func settledGoroutines(t *testing.T, want int) int {
t.Helper()
deadline := time.Now().Add(goroutineSettle)
for {
n := runtime.NumGoroutine()
if n <= want || time.Now().After(deadline) {
return n
}
time.Sleep(goroutinePoll)
}
}
// TestScanHashWriteFailureUnwindsPool drives the scan entry point
// against a database that refuses every write. The hash phase gives up
// partway through with thousands of runs still queued, which used to
// leave the feeder parked on a full job channel and every worker parked
// on a full result channel for the life of the process.
func TestScanHashWriteFailureUnwindsPool(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
dir := t.TempDir()
writeEmptyFiles(t, dir, hashLeakFiles)
injectWriteFailure(t, path)
base := baselineGoroutines(t)
var stderr bytes.Buffer
code := run([]string{
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
}, &stderr)
if code != exitFatal {
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
code, exitFatal, stderr.String())
}
if !strings.Contains(stderr.String(), injectedWriteFailure) {
t.Errorf("stderr = %q, want the injected write failure",
stderr.String())
}
if got := settledGoroutines(t, base); got > base {
t.Errorf("goroutines = %d after the failed scan, want %d back",
got, base)
}
}
func TestHashRuns(t *testing.T) {
t.Parallel()
rec := func(path string, dev, ino uint64) fileRec {
return fileRec{path: path, dev: dev, ino: ino}
}
runs := hashRuns([]fileRec{
rec("/c", 1, 7),
rec("/a", 1, 7),
rec("/b", 1, 9),
// No inode identity: never merged, even with matching zeros.
rec("/z1", 0, 0),
rec("/z2", 0, 0),
})
got := make([][]string, 0, len(runs))
for _, run := range runs {
paths := make([]string, 0, len(run))
for _, r := range run {
paths = append(paths, r.path)
}
got = append(got, paths)
}
want := [][]string{{"/z1"}, {"/z2"}, {"/a", "/c"}, {"/b"}}
if !slices.EqualFunc(got, want, slices.Equal) {
t.Fatalf("runs = %v, want %v", got, want)
}
}
func TestPruneRoots(t *testing.T) {
t.Parallel()
// Duplicates and operands under other operands are dropped; /cc is
// not under /c (sibling with a shared prefix).
got := pruneRoots([]string{"/a/b", "/a", "/c", "/a", "/a/b/c", "/cc"})
want := []string{"/a", "/c", "/cc"}
if !slices.Equal(got, want) {
t.Fatalf("pruneRoots = %q, want %q", got, want)
}
}
func TestReportsNeverTouchFilesystem(t *testing.T) {
t.Parallel()
dir := buildSmokeTree(t)
db := openTestDB(t)
syncTree(t, db, dir)
// Remove the scanned tree entirely; the analysis must be
// unaffected because it reads the database alone.
err := os.RemoveAll(dir)
if err != nil {
t.Fatal(err)
}
recs := dbRecords(t, db)
assertSmokeDupeGroups(t, dir, recs)
assertSmokeTreeGroups(t, dir, recs)
}
func TestUnderRoot(t *testing.T) {
t.Parallel()
const abRoot = "/a/b"
cases := []struct {
path string
root string
want bool
}{
{"/a/b/c", abRoot, true},
{abRoot, abRoot, true},
{"/a/bc", abRoot, false},
{"/a", abRoot, false},
{"/x/y", "/", true},
{"/", "/", true},
}
for _, c := range cases {
if got := underRoot(c.path, c.root); got != c.want {
t.Errorf("underRoot(%q, %q) = %v, want %v",
c.path, c.root, got, c.want)
}
}
}