Files
sfdupes/report_test.go
T
clawbot c887f80f57
check / check (push) Successful in 1m29s
Escape tab, newline, CR and backslash in report paths (closes #7)
A path holding a tab or newline split a row of the report or trees
output. The path columns of both now write a backslash, tab, newline
and carriage return as \\, \t, \n and \r; every other byte is written
unchanged. Grouping and sorting still use the stored path. Warnings
on stderr are escaped the same way in warnf, so each stays one line.

In trees, the root directory's node now has the path "/" instead of
an empty string, and its children's paths start with a single slash.

README states the rule under "Report output format".

Model: opus-5-5
2026-10-03 15:30:37 +02:00

271 lines
6.8 KiB
Go

package main
import (
"bytes"
"io"
"os"
"path/filepath"
"slices"
"testing"
)
// awkwardDir is a directory name holding every byte the reports escape.
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
// sorts after it, so awkwardDir coming first shows that sorting uses
// the raw path.
func awkwardPairRecs() []scanRec {
return []scanRec{
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
}
}
// seedDatabase writes recs into a fresh database and returns its path.
func seedDatabase(t *testing.T, recs []scanRec) string {
t.Helper()
path := testDBPath(t)
db, err := openScanDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
err = applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
err = db.Close()
if err != nil {
t.Fatal(err)
}
return path
}
func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stderr bytes.Buffer
stdout := captureStdout(t)
code := run([]string{cmdReport}, &stderr)
if code != exitOK {
t.Fatalf("run(report) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
want := "first\tdupe\tsize\n" +
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}
func TestEscapePath(t *testing.T) {
t.Parallel()
cases := map[string]string{
"/srv/plain": "/srv/plain",
"/a\tb": `/a\tb`,
"/a\nb": `/a\nb`,
"/a\rb": `/a\rb`,
`/a\b`: `/a\\b`,
`/a\tb`: `/a\\tb`,
"/not-utf8\xff": "/not-utf8\xff",
}
for in, want := range cases {
if got := escapePath(in); got != want {
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
}
}
}
// TestWarnfEscapes checks that a warning naming a path that holds a
// newline is still one line.
//
//nolint:paralleltest // replaces the process-wide os.Stderr
func TestWarnfEscapes(t *testing.T) {
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
if err != nil {
t.Fatal(err)
}
saved := os.Stderr
os.Stderr = f
t.Cleanup(func() {
os.Stderr = saved
_ = f.Close()
})
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
_, err = f.Seek(0, io.SeekStart)
if err != nil {
t.Fatal(err)
}
got, err := io.ReadAll(f)
if err != nil {
t.Fatal(err)
}
want := `stat /d/a\nb: gone` + "\n"
if string(got) != want {
t.Errorf("warning = %q, want %q", got, want)
}
}
func TestCollectDupeGroups(t *testing.T) {
t.Parallel()
recs := []scanRec{
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
// Same size as the /z group but a different head hash.
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
// A singleton signature must not form a group.
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
}
groups := collectDupeGroups(recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
if groups[0].size != 4000 ||
!slices.Equal(groups[0].paths, []string{"/big/1", "/big/2"}) {
t.Errorf("groups[0] = %+v, want size 4000, paths /big/1 /big/2",
groups[0])
}
if groups[1].size != 100 ||
!slices.Equal(groups[1].paths, []string{"/z/a", "/z/b", "/z/c"}) {
t.Errorf("groups[1] = %+v, want size 100, paths /z/a /z/b /z/c",
groups[1])
}
}
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
t.Parallel()
// Same size, head, and tail, but different content hashes: the final
// rung keeps them apart, so no group forms. Matching content groups.
// Records without a content hash never group, not even with each
// other.
recs := []scanRec{
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
{size: 100, head: "h", tail: "t", path: "/d"},
{size: 100, head: "h", tail: "t", path: "/e"},
}
groups := collectDupeGroups(recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
len(groups))
}
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
}
}
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
t.Parallel()
// mtime is informational only; records differing only in mtime
// still group together.
recs := []scanRec{
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
}
groups := collectDupeGroups(recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1", len(groups))
}
}
func TestCollectDupeGroupsTieBreak(t *testing.T) {
t.Parallel()
recs := []scanRec{
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
}
groups := collectDupeGroups(recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
// Equal sizes: ordered by first path ascending.
if groups[0].paths[0] != "/alpha/1" || groups[1].paths[0] != "/beta/1" {
t.Fatalf("tie-break order wrong: %q then %q",
groups[0].paths[0], groups[1].paths[0])
}
}
func TestCollectDupeGroupsDeterministic(t *testing.T) {
t.Parallel()
recs := []scanRec{
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
}
forward := collectDupeGroups(recs)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
backward := collectDupeGroups(reversed)
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
return a.size == b.size && slices.Equal(a.paths, b.paths)
}) {
t.Fatalf("output depends on record order: %+v vs %+v",
forward, backward)
}
}
func TestHumanBytes(t *testing.T) {
t.Parallel()
cases := []struct {
n int64
want string
}{
{0, "0 B"},
{1, "1 B"},
{1023, "1023 B"},
{1024, "1.0 KiB"},
{1536, "1.5 KiB"},
{1 << 20, "1.0 MiB"},
{5 << 30, "5.0 GiB"},
{1 << 40, "1.0 TiB"},
{1 << 50, "1.0 PiB"},
{1 << 60, "1.0 EiB"},
}
for _, c := range cases {
if got := humanBytes(c.n); got != c.want {
t.Errorf("humanBytes(%d) = %q, want %q", c.n, got, c.want)
}
}
}