check / check (push) Successful in 1m29s
A path holding a tab or newline split a row of the report or trees output. The path columns of both now write a backslash, tab, newline and carriage return as \\, \t, \n and \r; every other byte is written unchanged. Grouping and sorting still use the stored path. Warnings on stderr are escaped the same way in warnf, so each stays one line. In trees, the root directory's node now has the path "/" instead of an empty string, and its children's paths start with a single slash. README states the rule under "Report output format". Model: opus-5-5
271 lines
6.8 KiB
Go
271 lines
6.8 KiB
Go
package main
|
|
|
|
import (
|
|
"bytes"
|
|
"io"
|
|
"os"
|
|
"path/filepath"
|
|
"slices"
|
|
"testing"
|
|
)
|
|
|
|
// awkwardDir is a directory name holding every byte the reports escape.
|
|
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
|
|
|
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
|
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
|
// sorts after it, so awkwardDir coming first shows that sorting uses
|
|
// the raw path.
|
|
func awkwardPairRecs() []scanRec {
|
|
return []scanRec{
|
|
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
|
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
|
}
|
|
}
|
|
|
|
// seedDatabase writes recs into a fresh database and returns its path.
|
|
func seedDatabase(t *testing.T, recs []scanRec) string {
|
|
t.Helper()
|
|
|
|
path := testDBPath(t)
|
|
|
|
db, err := openScanDatabase(t.Context(), path)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
err = applyChanges(t.Context(), db, recs, nil, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
err = db.Close()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
return path
|
|
}
|
|
|
|
func TestRunReportEscapesPaths(t *testing.T) {
|
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
|
|
var stderr bytes.Buffer
|
|
|
|
stdout := captureStdout(t)
|
|
|
|
code := run([]string{cmdReport}, &stderr)
|
|
if code != exitOK {
|
|
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
|
code, exitOK, stderr.String())
|
|
}
|
|
|
|
want := "first\tdupe\tsize\n" +
|
|
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
|
if got := stdout(); got != want {
|
|
t.Errorf("stdout = %q, want %q", got, want)
|
|
}
|
|
}
|
|
|
|
func TestEscapePath(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
cases := map[string]string{
|
|
"/srv/plain": "/srv/plain",
|
|
"/a\tb": `/a\tb`,
|
|
"/a\nb": `/a\nb`,
|
|
"/a\rb": `/a\rb`,
|
|
`/a\b`: `/a\\b`,
|
|
`/a\tb`: `/a\\tb`,
|
|
"/not-utf8\xff": "/not-utf8\xff",
|
|
}
|
|
for in, want := range cases {
|
|
if got := escapePath(in); got != want {
|
|
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestWarnfEscapes checks that a warning naming a path that holds a
|
|
// newline is still one line.
|
|
//
|
|
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
func TestWarnfEscapes(t *testing.T) {
|
|
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
saved := os.Stderr
|
|
os.Stderr = f
|
|
|
|
t.Cleanup(func() {
|
|
os.Stderr = saved
|
|
|
|
_ = f.Close()
|
|
})
|
|
|
|
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
|
|
|
_, err = f.Seek(0, io.SeekStart)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
got, err := io.ReadAll(f)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
want := `stat /d/a\nb: gone` + "\n"
|
|
if string(got) != want {
|
|
t.Errorf("warning = %q, want %q", got, want)
|
|
}
|
|
}
|
|
|
|
func TestCollectDupeGroups(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
recs := []scanRec{
|
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
|
// Same size as the /z group but a different head hash.
|
|
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
|
// A singleton signature must not form a group.
|
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
|
}
|
|
|
|
groups := collectDupeGroups(recs)
|
|
if len(groups) != 2 {
|
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
|
}
|
|
|
|
if groups[0].size != 4000 ||
|
|
!slices.Equal(groups[0].paths, []string{"/big/1", "/big/2"}) {
|
|
t.Errorf("groups[0] = %+v, want size 4000, paths /big/1 /big/2",
|
|
groups[0])
|
|
}
|
|
|
|
if groups[1].size != 100 ||
|
|
!slices.Equal(groups[1].paths, []string{"/z/a", "/z/b", "/z/c"}) {
|
|
t.Errorf("groups[1] = %+v, want size 100, paths /z/a /z/b /z/c",
|
|
groups[1])
|
|
}
|
|
}
|
|
|
|
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
// Same size, head, and tail, but different content hashes: the final
|
|
// rung keeps them apart, so no group forms. Matching content groups.
|
|
// Records without a content hash never group, not even with each
|
|
// other.
|
|
recs := []scanRec{
|
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
|
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
|
{size: 100, head: "h", tail: "t", path: "/d"},
|
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
|
}
|
|
|
|
groups := collectDupeGroups(recs)
|
|
if len(groups) != 1 {
|
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
|
len(groups))
|
|
}
|
|
|
|
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
|
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
|
}
|
|
}
|
|
|
|
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
// mtime is informational only; records differing only in mtime
|
|
// still group together.
|
|
recs := []scanRec{
|
|
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
|
}
|
|
|
|
groups := collectDupeGroups(recs)
|
|
if len(groups) != 1 {
|
|
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
|
}
|
|
}
|
|
|
|
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
recs := []scanRec{
|
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
|
}
|
|
|
|
groups := collectDupeGroups(recs)
|
|
if len(groups) != 2 {
|
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
|
}
|
|
|
|
// Equal sizes: ordered by first path ascending.
|
|
if groups[0].paths[0] != "/alpha/1" || groups[1].paths[0] != "/beta/1" {
|
|
t.Fatalf("tie-break order wrong: %q then %q",
|
|
groups[0].paths[0], groups[1].paths[0])
|
|
}
|
|
}
|
|
|
|
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
recs := []scanRec{
|
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
|
}
|
|
|
|
forward := collectDupeGroups(recs)
|
|
|
|
reversed := slices.Clone(recs)
|
|
slices.Reverse(reversed)
|
|
|
|
backward := collectDupeGroups(reversed)
|
|
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
|
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
|
}) {
|
|
t.Fatalf("output depends on record order: %+v vs %+v",
|
|
forward, backward)
|
|
}
|
|
}
|
|
|
|
func TestHumanBytes(t *testing.T) {
|
|
t.Parallel()
|
|
|
|
cases := []struct {
|
|
n int64
|
|
want string
|
|
}{
|
|
{0, "0 B"},
|
|
{1, "1 B"},
|
|
{1023, "1023 B"},
|
|
{1024, "1.0 KiB"},
|
|
{1536, "1.5 KiB"},
|
|
{1 << 20, "1.0 MiB"},
|
|
{5 << 30, "5.0 GiB"},
|
|
{1 << 40, "1.0 TiB"},
|
|
{1 << 50, "1.0 PiB"},
|
|
{1 << 60, "1.0 EiB"},
|
|
}
|
|
for _, c := range cases {
|
|
if got := humanBytes(c.n); got != c.want {
|
|
t.Errorf("humanBytes(%d) = %q, want %q", c.n, got, c.want)
|
|
}
|
|
}
|
|
}
|