Files
sfdupes/trees_test.go
T
sneak bd8d41b174
check / check (push) Successful in 1m49s
Stream report and trees instead of loading every record (closes #14)
report now has SQLite group the records and put the rows in report
order, helped by a new files_signature index on (size, head, tail,
content), and writes each row as it reads it. trees reads the records
in path order, where all the paths under a directory come together, so
it computes each directory's digest as soon as the stream leaves it and
keeps only its path, parent, digest and totals. Output is unchanged.

The tests that called the removed in-memory grouping functions now group
records stored in a database. A new test checks that both commands give
the same output whatever order the records were inserted in.

Model: opus-5-5
2026-10-04 02:04:37 +00:00

336 lines
8.4 KiB
Go

package main
import (
"bytes"
"database/sql"
"slices"
"testing"
)
// Signature hashes shared by the smoke-test records.
const (
f1Head = "f1h"
f1Tail = "f1t"
f1Content = "f1c"
f2Head = "f2h"
f2Tail = "f2t"
f2Content = "f2c"
)
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
// /d/t2 are identical, /d/t3 differs from them only by one filename.
func smokeTreeRecs() []scanRec {
return []scanRec{
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
path: "/d/t3/sub/f2renamed"},
}
}
// dbTree builds the directory hierarchy from the records in db the way
// trees does, and returns the super-root and every directory.
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
t.Helper()
tree := newTreeBuilder()
err := loadFileRows(t.Context(), db, tree.add)
if err != nil {
t.Fatal(err)
}
return tree.finish()
}
// treeOf writes recs into a fresh database and builds the directory
// hierarchy from it the way trees does.
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dbTree(t, db)
}
// nodeByPath finds the directory node with the given path.
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
t.Helper()
for _, d := range dirs {
if d.path == path {
return d
}
}
t.Fatalf("no directory node with path %q", path)
return nil
}
// groupPaths flattens tree groups into their member path lists.
func groupPaths(groups [][]*treeNode) [][]string {
out := make([][]string, 0, len(groups))
for _, g := range groups {
paths := make([]string, 0, len(g))
for _, n := range g {
paths = append(paths, n.path)
}
out = append(out, paths)
}
return out
}
func TestTreeCounts(t *testing.T) {
t.Parallel()
_, dirs := treeOf(t, smokeTreeRecs())
d := nodeByPath(t, dirs, "/d")
if d.fileCount != 6 || d.totalSize != 9300 {
t.Errorf("/d: fileCount %d size %d, want 6 9300",
d.fileCount, d.totalSize)
}
t1 := nodeByPath(t, dirs, "/d/t1")
if t1.fileCount != 2 || t1.totalSize != 3100 {
t.Errorf("/d/t1: fileCount %d size %d, want 2 3100",
t1.fileCount, t1.totalSize)
}
sub := nodeByPath(t, dirs, "/d/t1/sub")
if sub.fileCount != 1 || sub.totalSize != 100 {
t.Errorf("/d/t1/sub: fileCount %d size %d, want 1 100",
sub.fileCount, sub.totalSize)
}
}
func TestTreeRootPath(t *testing.T) {
t.Parallel()
// The root directory's path is "/", never empty, and its
// children's paths start with a single slash.
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/srv"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
}
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
t.Parallel()
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
// directory must still be built once, whole, so /a matches /c.
recs := make([]scanRec, 0, 8)
for _, top := range []string{"/a", "/c"} {
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
content := "c"
if p == "/b-x/f" {
content = "other"
}
recs = append(recs, scanRec{
size: 1, head: "h", tail: "t", content: content, path: top + p,
})
}
}
super, dirs := treeOf(t, recs)
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
groups := collectTreeGroups(dirs, super)
gotGroups := groupPaths(groups)
wantGroups := [][]string{{"/a", "/c"}}
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
}
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
t.Errorf("group totals: %d files %d bytes, want 4 4",
groups[0][0].fileCount, groups[0][0].totalSize)
}
}
func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stdout, stderr bytes.Buffer
code := run([]string{cmdTrees}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
want := "first\tdupe\tfiles\tsize\n" +
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
if got := stdout.String(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}
func TestTreeDigests(t *testing.T) {
t.Parallel()
_, dirs := treeOf(t, smokeTreeRecs())
t1 := nodeByPath(t, dirs, "/d/t1")
t2 := nodeByPath(t, dirs, "/d/t2")
t3 := nodeByPath(t, dirs, "/d/t3")
if t1.digest != t2.digest {
t.Error("identical trees /d/t1 and /d/t2 have different digests")
}
// t3 differs only in a filename; names are part of the digest.
if t1.digest == t3.digest {
t.Error("/d/t3 digest equals /d/t1 despite a renamed file")
}
sub1 := nodeByPath(t, dirs, "/d/t1/sub")
sub3 := nodeByPath(t, dirs, "/d/t3/sub")
if sub1.digest == sub3.digest {
t.Error("subdirs with differently-named files share a digest")
}
}
func TestTreeDigestContentSensitivity(t *testing.T) {
t.Parallel()
const sharedTail = "same"
recs := []scanRec{
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
}
_, dirs := treeOf(t, recs)
a := nodeByPath(t, dirs, "/r/a")
b := nodeByPath(t, dirs, "/r/b")
if a.digest == b.digest {
t.Error("trees with different file content share a digest")
}
}
func TestCollectTreeGroupsMaximal(t *testing.T) {
t.Parallel()
super, dirs := treeOf(t, smokeTreeRecs())
groups := collectTreeGroups(dirs, super)
want := [][]string{{"/d/t1", "/d/t2"}}
if got := groupPaths(groups); !slices.EqualFunc(got, want,
slices.Equal) {
t.Fatalf("groups = %v, want %v", got, want)
}
// The /d/t1/sub vs /d/t2/sub group must be suppressed as implied
// by its parents' group; the winning group reports one copy's
// recursive totals.
if groups[0][0].fileCount != 2 || groups[0][0].totalSize != 3100 {
t.Errorf("group totals: %d files %d bytes, want 2 3100",
groups[0][0].fileCount, groups[0][0].totalSize)
}
}
func TestCollectTreeGroupsDeterministic(t *testing.T) {
t.Parallel()
recs := smokeTreeRecs()
super, dirs := treeOf(t, recs)
forward := groupPaths(collectTreeGroups(dirs, super))
reversed := slices.Clone(recs)
slices.Reverse(reversed)
superR, dirsR := treeOf(t, reversed)
backward := groupPaths(collectTreeGroups(dirsR, superR))
if !slices.EqualFunc(forward, backward, slices.Equal) {
t.Fatalf("output depends on record order: %v vs %v",
forward, backward)
}
}
func TestCollectTreeGroupsSiblings(t *testing.T) {
t.Parallel()
// Identical sibling dirs share a parent, so their group cannot be
// implied by a parent group and must be reported.
recs := []scanRec{
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
}
super, dirs := treeOf(t, recs)
got := groupPaths(collectTreeGroups(dirs, super))
want := [][]string{{"/p/x1", "/p/x2"}}
if !slices.EqualFunc(got, want, slices.Equal) {
t.Fatalf("groups = %v, want %v", got, want)
}
}
func TestCollectTreeGroupsDifferingParents(t *testing.T) {
t.Parallel()
// /p/a and /q/b contain identical x subtrees, but /p/a has an
// extra file, so the parents' digests differ and the x group must
// be reported.
recs := []scanRec{
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
}
super, dirs := treeOf(t, recs)
got := groupPaths(collectTreeGroups(dirs, super))
want := [][]string{{"/p/a/x", "/q/b/x"}}
if !slices.EqualFunc(got, want, slices.Equal) {
t.Fatalf("groups = %v, want %v", got, want)
}
}