check / check (push) Successful in 1m49s
report now has SQLite group the records and put the rows in report order, helped by a new files_signature index on (size, head, tail, content), and writes each row as it reads it. trees reads the records in path order, where all the paths under a directory come together, so it computes each directory's digest as soon as the stream leaves it and keeps only its path, parent, digest and totals. Output is unchanged. The tests that called the removed in-memory grouping functions now group records stored in a database. A new test checks that both commands give the same output whatever order the records were inserted in. Model: opus-5-5
371 lines
9.3 KiB
Go
371 lines
9.3 KiB
Go
package main
|
|
|
|
import (
|
|
"bytes"
|
|
"database/sql"
|
|
"io"
|
|
"os"
|
|
"path/filepath"
|
|
"slices"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// awkwardDir is a directory name holding every byte the reports escape.
|
|
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
|
|
|
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
|
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
|
// sorts after it, so awkwardDir coming first shows that sorting uses
|
|
// the raw path.
|
|
func awkwardPairRecs() []scanRec {
|
|
return []scanRec{
|
|
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
|
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
|
}
|
|
}
|
|
|
|
// seedDatabase writes recs into a fresh database and returns its path.
|
|
func seedDatabase(t *testing.T, recs []scanRec) string {
|
|
t.Helper()
|
|
|
|
path := testDBPath(t)
|
|
|
|
db, err := openScanDatabase(t.Context(), path)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
err = applyChanges(t.Context(), db, recs, nil, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
err = db.Close()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
return path
|
|
}
|
|
|
|
// dupeGroup is one duplicate group as report reads it: the size, and
|
|
// the paths in report order, first path first.
|
|
type dupeGroup struct {
|
|
size int64
|
|
paths []string
|
|
}
|
|
|
|
// dupeGroups returns the duplicate groups report reads from db, in
|
|
// report order.
|
|
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
|
t.Helper()
|
|
|
|
var groups []dupeGroup
|
|
|
|
_, err := loadDupeRows(t.Context(), db,
|
|
func(first, path string, size int64) error {
|
|
if path == first {
|
|
groups = append(groups, dupeGroup{size: size})
|
|
}
|
|
|
|
g := &groups[len(groups)-1]
|
|
g.paths = append(g.paths, path)
|
|
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
return groups
|
|
}
|
|
|
|
// dupeGroupsOf writes recs into a fresh database and returns the
|
|
// duplicate groups report reads from it.
|
|
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
|
t.Helper()
|
|
|
|
db := openTestDB(t)
|
|
|
|
err := applyChanges(t.Context(), db, recs, nil, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
return dupeGroups(t, db)
|
|
}
|
|
|
|
func TestRunReportEscapesPaths(t *testing.T) {
|
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
|
|
var stdout, stderr bytes.Buffer
|
|
|
|
code := run([]string{cmdReport}, &stdout, &stderr)
|
|
if code != exitOK {
|
|
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
|
code, exitOK, stderr.String())
|
|
}
|
|
|
|
want := "first\tdupe\tsize\n" +
|
|
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
|
if got := stdout.String(); got != want {
|
|
t.Errorf("stdout = %q, want %q", got, want)
|
|
}
|
|
}
|
|
|
|
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
|
// README §Constraints: identical database contents give identical
|
|
// output, whatever order the records were inserted in.
|
|
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
|
recs = append(recs,
|
|
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
|
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
|
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
|
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
|
scanRec{size: 50, path: "/x/unhashed"},
|
|
)
|
|
|
|
reversed := slices.Clone(recs)
|
|
slices.Reverse(reversed)
|
|
|
|
for _, name := range []string{cmdReport, cmdTrees} {
|
|
t.Run(name, func(t *testing.T) {
|
|
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
|
|
|
forward := runStdout(t, name)
|
|
|
|
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
|
|
|
backward := runStdout(t, name)
|
|
|
|
if strings.Count(forward, "\n") < 3 {
|
|
t.Errorf("stdout = %q, want at least two rows", forward)
|
|
}
|
|
|
|
if forward != backward {
|
|
t.Errorf("stdout depends on insertion order: %q vs %q",
|
|
forward, backward)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// runStdout runs the subcommand name and returns its stdout, failing
|
|
// the test unless it succeeds.
|
|
func runStdout(t *testing.T, name string) string {
|
|
t.Helper()
|
|
|
|
var stdout, stderr bytes.Buffer
|
|
|
|
code := run([]string{name}, &stdout, &stderr)
|
|
if code != exitOK {
|
|
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
|
name, code, exitOK, stderr.String())
|
|
}
|
|
|
|
return stdout.String()
|
|
}
|
|
|
|
func TestEscapePath(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
cases := map[string]string{
|
|
"/srv/plain": "/srv/plain",
|
|
"/a\tb": `/a\tb`,
|
|
"/a\nb": `/a\nb`,
|
|
"/a\rb": `/a\rb`,
|
|
`/a\b`: `/a\\b`,
|
|
`/a\tb`: `/a\\tb`,
|
|
"/not-utf8\xff": "/not-utf8\xff",
|
|
}
|
|
for in, want := range cases {
|
|
if got := escapePath(in); got != want {
|
|
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestWarnfEscapes checks that a warning naming a path that holds a
|
|
// newline is still one line.
|
|
//
|
|
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
func TestWarnfEscapes(t *testing.T) {
|
|
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
saved := os.Stderr
|
|
os.Stderr = f
|
|
|
|
t.Cleanup(func() {
|
|
os.Stderr = saved
|
|
|
|
_ = f.Close()
|
|
})
|
|
|
|
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
|
|
|
_, err = f.Seek(0, io.SeekStart)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
got, err := io.ReadAll(f)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
want := `stat /d/a\nb: gone` + "\n"
|
|
if string(got) != want {
|
|
t.Errorf("warning = %q, want %q", got, want)
|
|
}
|
|
}
|
|
|
|
func TestDupeGroups(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
recs := []scanRec{
|
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
|
// Same size as the /z group but a different head hash.
|
|
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
|
// A singleton signature must not form a group.
|
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
|
}
|
|
|
|
groups := dupeGroupsOf(t, recs)
|
|
if len(groups) != 2 {
|
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
|
}
|
|
|
|
if groups[0].size != 4000 ||
|
|
!slices.Equal(groups[0].paths, []string{"/big/1", "/big/2"}) {
|
|
t.Errorf("groups[0] = %+v, want size 4000, paths /big/1 /big/2",
|
|
groups[0])
|
|
}
|
|
|
|
if groups[1].size != 100 ||
|
|
!slices.Equal(groups[1].paths, []string{"/z/a", "/z/b", "/z/c"}) {
|
|
t.Errorf("groups[1] = %+v, want size 100, paths /z/a /z/b /z/c",
|
|
groups[1])
|
|
}
|
|
}
|
|
|
|
func TestDupeGroupsContentSeparates(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
// Same size, head, and tail, but different content hashes: the final
|
|
// rung keeps them apart, so no group forms. Matching content groups.
|
|
// Records without a content hash never group, not even with each
|
|
// other.
|
|
recs := []scanRec{
|
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
|
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
|
{size: 100, head: "h", tail: "t", path: "/d"},
|
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
|
}
|
|
|
|
groups := dupeGroupsOf(t, recs)
|
|
if len(groups) != 1 {
|
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
|
len(groups))
|
|
}
|
|
|
|
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
|
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
|
}
|
|
}
|
|
|
|
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
// mtime is informational only; records differing only in mtime
|
|
// still group together.
|
|
recs := []scanRec{
|
|
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
|
}
|
|
|
|
groups := dupeGroupsOf(t, recs)
|
|
if len(groups) != 1 {
|
|
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
|
}
|
|
}
|
|
|
|
func TestDupeGroupsTieBreak(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
recs := []scanRec{
|
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
|
}
|
|
|
|
groups := dupeGroupsOf(t, recs)
|
|
if len(groups) != 2 {
|
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
|
}
|
|
|
|
// Equal sizes: ordered by first path ascending.
|
|
if groups[0].paths[0] != "/alpha/1" || groups[1].paths[0] != "/beta/1" {
|
|
t.Fatalf("tie-break order wrong: %q then %q",
|
|
groups[0].paths[0], groups[1].paths[0])
|
|
}
|
|
}
|
|
|
|
func TestDupeGroupsDeterministic(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
recs := []scanRec{
|
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
|
}
|
|
|
|
forward := dupeGroupsOf(t, recs)
|
|
|
|
reversed := slices.Clone(recs)
|
|
slices.Reverse(reversed)
|
|
|
|
backward := dupeGroupsOf(t, reversed)
|
|
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
|
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
|
}) {
|
|
t.Fatalf("output depends on record order: %+v vs %+v",
|
|
forward, backward)
|
|
}
|
|
}
|
|
|
|
func TestHumanBytes(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
cases := []struct {
|
|
n int64
|
|
want string
|
|
}{
|
|
{0, "0 B"},
|
|
{1, "1 B"},
|
|
{1023, "1023 B"},
|
|
{1024, "1.0 KiB"},
|
|
{1536, "1.5 KiB"},
|
|
{1 << 20, "1.0 MiB"},
|
|
{5 << 30, "5.0 GiB"},
|
|
{1 << 40, "1.0 TiB"},
|
|
{1 << 50, "1.0 PiB"},
|
|
{1 << 60, "1.0 EiB"},
|
|
}
|
|
for _, c := range cases {
|
|
if got := humanBytes(c.n); got != c.want {
|
|
t.Errorf("humanBytes(%d) = %q, want %q", c.n, got, c.want)
|
|
}
|
|
}
|
|
}
|