package main import ( "bytes" "io" "os" "path/filepath" "slices" "testing" ) // awkwardDir is a directory name holding every byte the reports escape. const awkwardDir = "/d/\tone\ntwo\rthree\\four" // awkwardPairRecs is a duplicate pair in sibling directories /d/A and // awkwardDir. A raw tab sorts before "A" but its escaped form `\t` // sorts after it, so awkwardDir coming first shows that sorting uses // the raw path. func awkwardPairRecs() []scanRec { return []scanRec{ {size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"}, {size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"}, } } // seedDatabase writes recs into a fresh database and returns its path. func seedDatabase(t *testing.T, recs []scanRec) string { t.Helper() path := testDBPath(t) db, err := openScanDatabase(t.Context(), path) if err != nil { t.Fatal(err) } err = applyChanges(t.Context(), db, recs, nil, nil) if err != nil { t.Fatal(err) } err = db.Close() if err != nil { t.Fatal(err) } return path } func TestRunReportEscapesPaths(t *testing.T) { t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) var stderr bytes.Buffer stdout := captureStdout(t) code := run([]string{cmdReport}, &stderr) if code != exitOK { t.Fatalf("run(report) = %d, want %d; stderr: %s", code, exitOK, stderr.String()) } want := "first\tdupe\tsize\n" + `/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n" if got := stdout(); got != want { t.Errorf("stdout = %q, want %q", got, want) } } func TestEscapePath(t *testing.T) { t.Parallel() cases := map[string]string{ "/srv/plain": "/srv/plain", "/a\tb": `/a\tb`, "/a\nb": `/a\nb`, "/a\rb": `/a\rb`, `/a\b`: `/a\\b`, `/a\tb`: `/a\\tb`, "/not-utf8\xff": "/not-utf8\xff", } for in, want := range cases { if got := escapePath(in); got != want { t.Errorf("escapePath(%q) = %q, want %q", in, got, want) } } } // TestWarnfEscapes checks that a warning naming a path that holds a // newline is still one line. // //nolint:paralleltest // replaces the process-wide os.Stderr func TestWarnfEscapes(t *testing.T) { f, err := os.Create(filepath.Join(t.TempDir(), "stderr")) if err != nil { t.Fatal(err) } saved := os.Stderr os.Stderr = f t.Cleanup(func() { os.Stderr = saved _ = f.Close() }) (&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone") _, err = f.Seek(0, io.SeekStart) if err != nil { t.Fatal(err) } got, err := io.ReadAll(f) if err != nil { t.Fatal(err) } want := `stat /d/a\nb: gone` + "\n" if string(got) != want { t.Errorf("warning = %q, want %q", got, want) } } func TestCollectDupeGroups(t *testing.T) { t.Parallel() recs := []scanRec{ {size: 100, head: "h", tail: "t", content: "c", path: "/z/b"}, {size: 100, head: "h", tail: "t", content: "c", path: "/z/a"}, {size: 100, head: "h", tail: "t", content: "c", path: "/z/c"}, {size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"}, {size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"}, // Same size as the /z group but a different head hash. {size: 100, head: "other", tail: "t", content: "c", path: "/z/d"}, // A singleton signature must not form a group. {size: 7, head: "u", tail: "u", content: "u", path: "/lonely"}, } groups := collectDupeGroups(recs) if len(groups) != 2 { t.Fatalf("len(groups) = %d, want 2", len(groups)) } if groups[0].size != 4000 || !slices.Equal(groups[0].paths, []string{"/big/1", "/big/2"}) { t.Errorf("groups[0] = %+v, want size 4000, paths /big/1 /big/2", groups[0]) } if groups[1].size != 100 || !slices.Equal(groups[1].paths, []string{"/z/a", "/z/b", "/z/c"}) { t.Errorf("groups[1] = %+v, want size 100, paths /z/a /z/b /z/c", groups[1]) } } func TestCollectDupeGroupsContentSeparates(t *testing.T) { t.Parallel() // Same size, head, and tail, but different content hashes: the final // rung keeps them apart, so no group forms. Matching content groups. // Records without a content hash never group, not even with each // other. recs := []scanRec{ {size: 100, head: "h", tail: "t", content: "c1", path: "/a"}, {size: 100, head: "h", tail: "t", content: "c2", path: "/b"}, {size: 100, head: "h", tail: "t", content: "c1", path: "/c"}, {size: 100, head: "h", tail: "t", path: "/d"}, {size: 100, head: "h", tail: "t", path: "/e"}, } groups := collectDupeGroups(recs) if len(groups) != 1 { t.Fatalf("len(groups) = %d, want 1 (only the matching content)", len(groups)) } if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) { t.Errorf("group paths = %q, want /a /c", groups[0].paths) } } func TestCollectDupeGroupsMtimeExcluded(t *testing.T) { t.Parallel() // mtime is informational only; records differing only in mtime // still group together. recs := []scanRec{ {size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"}, {size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"}, } groups := collectDupeGroups(recs) if len(groups) != 1 { t.Fatalf("len(groups) = %d, want 1", len(groups)) } } func TestCollectDupeGroupsTieBreak(t *testing.T) { t.Parallel() recs := []scanRec{ {size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"}, {size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"}, {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"}, {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"}, } groups := collectDupeGroups(recs) if len(groups) != 2 { t.Fatalf("len(groups) = %d, want 2", len(groups)) } // Equal sizes: ordered by first path ascending. if groups[0].paths[0] != "/alpha/1" || groups[1].paths[0] != "/beta/1" { t.Fatalf("tie-break order wrong: %q then %q", groups[0].paths[0], groups[1].paths[0]) } } func TestCollectDupeGroupsDeterministic(t *testing.T) { t.Parallel() recs := []scanRec{ {size: 1, head: "a", tail: "a", content: "a", path: "/p/1"}, {size: 1, head: "a", tail: "a", content: "a", path: "/p/2"}, {size: 2, head: "b", tail: "b", content: "b", path: "/q/1"}, {size: 2, head: "b", tail: "b", content: "b", path: "/q/2"}, } forward := collectDupeGroups(recs) reversed := slices.Clone(recs) slices.Reverse(reversed) backward := collectDupeGroups(reversed) if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool { return a.size == b.size && slices.Equal(a.paths, b.paths) }) { t.Fatalf("output depends on record order: %+v vs %+v", forward, backward) } } func TestHumanBytes(t *testing.T) { t.Parallel() cases := []struct { n int64 want string }{ {0, "0 B"}, {1, "1 B"}, {1023, "1023 B"}, {1024, "1.0 KiB"}, {1536, "1.5 KiB"}, {1 << 20, "1.0 MiB"}, {5 << 30, "5.0 GiB"}, {1 << 40, "1.0 TiB"}, {1 << 50, "1.0 PiB"}, {1 << 60, "1.0 EiB"}, } for _, c := range cases { if got := humanBytes(c.n); got != c.want { t.Errorf("humanBytes(%d) = %q, want %q", c.n, got, c.want) } } }