Files
sfdupes/report_test.go
T
clawbot 696565cb12
check / check (push) Waiting to run
Store mtime to the nanosecond so a same-second rewrite is re-hashed (closes #12)
scan recorded mtime in whole seconds, so a file rewritten in place at
the same size within the same second as its recorded mtime was classed
unchanged and kept its old hashes. The files table keeps mtime as whole
Unix seconds and gains mtime_nsec, the nanoseconds within that second.
scan holds the mtime as a time.Time and decides "newer" with After, so
any time a filesystem can record, one after 2262 included, compares in
the right order. The walk, a file given as an operand, and the content
phase's recheck all move over. PRAGMA user_version stays 1, per the
owner's ruling. README states what both columns hold.

Model: opus-5-5
2026-10-07 22:02:22 +00:00

405 lines
10 KiB
Go

package main
import (
"bytes"
"database/sql"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"time"
)
// awkwardDir is a directory name holding every byte the reports escape.
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
// sorts after it, so awkwardDir coming first shows that sorting uses
// the raw path.
func awkwardPairRecs() []scanRec {
return []scanRec{
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
}
}
// seedDatabase writes recs into a fresh database and returns its path.
func seedDatabase(t *testing.T, recs []scanRec) string {
t.Helper()
path := testDBPath(t)
db, err := openScanDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
err = applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
err = db.Close()
if err != nil {
t.Fatal(err)
}
return path
}
// dupeGroup is one duplicate group as report reads it: the size, and
// the paths in report order, first path first.
type dupeGroup struct {
size int64
paths []string
}
// dupeGroups returns the duplicate groups report reads from db, in
// report order.
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
t.Helper()
var groups []dupeGroup
_, err := loadDupeRows(t.Context(), db,
func(first, path string, size int64) error {
if path == first {
groups = append(groups, dupeGroup{size: size})
}
g := &groups[len(groups)-1]
g.paths = append(g.paths, path)
return nil
})
if err != nil {
t.Fatal(err)
}
return groups
}
// dupeGroupsOf writes recs into a fresh database and returns the
// duplicate groups report reads from it.
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dupeGroups(t, db)
}
func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stdout, stderr bytes.Buffer
code := run([]string{cmdReport}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(report) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
want := "first\tdupe\tsize\n" +
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
if got := stdout.String(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}
func TestReportStdoutFailsWhileReading(t *testing.T) {
// Each row holds two paths longer than dir, so the report is more
// than twice the stdout buffer and stdout fails while rows are
// still being read, not at the final flush.
dir := "/" + strings.Repeat("d", 4096)
recs := make([]scanRec, ioBufSize/len(dir))
for i := range recs {
recs[i] = scanRec{
size: 1, head: "h", tail: "t", content: "c",
path: fmt.Sprintf("%s/%d", dir, i),
}
}
t.Setenv(databaseEnv, seedDatabase(t, recs))
err := runReport(t.Context(), failingWriter{})
if !errors.Is(err, errWriteFailed) ||
!strings.HasPrefix(err.Error(), "write stdout: ") {
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
}
}
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
// README §Constraints: identical database contents give identical
// output, whatever order the records were inserted in.
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
recs = append(recs,
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
scanRec{size: 50, path: "/x/unhashed"},
)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, recs))
forward := runStdout(t, name)
t.Setenv(databaseEnv, seedDatabase(t, reversed))
backward := runStdout(t, name)
if strings.Count(forward, "\n") < 3 {
t.Errorf("stdout = %q, want at least two rows", forward)
}
if forward != backward {
t.Errorf("stdout depends on insertion order: %q vs %q",
forward, backward)
}
})
}
}
// runStdout runs the subcommand name and returns its stdout, failing
// the test unless it succeeds.
func runStdout(t *testing.T, name string) string {
t.Helper()
var stdout, stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
}
return stdout.String()
}
func TestEscapePath(t *testing.T) {
t.Parallel()
cases := map[string]string{
"/srv/plain": "/srv/plain",
"/a\tb": `/a\tb`,
"/a\nb": `/a\nb`,
"/a\rb": `/a\rb`,
`/a\b`: `/a\\b`,
`/a\tb`: `/a\\tb`,
"/not-utf8\xff": "/not-utf8\xff",
}
for in, want := range cases {
if got := escapePath(in); got != want {
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
}
}
}
// TestWarnfEscapes checks that a warning naming a path that holds a
// newline is still one line.
//
//nolint:paralleltest // replaces the process-wide os.Stderr
func TestWarnfEscapes(t *testing.T) {
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
if err != nil {
t.Fatal(err)
}
saved := os.Stderr
os.Stderr = f
t.Cleanup(func() {
os.Stderr = saved
_ = f.Close()
})
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
_, err = f.Seek(0, io.SeekStart)
if err != nil {
t.Fatal(err)
}
got, err := io.ReadAll(f)
if err != nil {
t.Fatal(err)
}
want := `stat /d/a\nb: gone` + "\n"
if string(got) != want {
t.Errorf("warning = %q, want %q", got, want)
}
}
func TestDupeGroups(t *testing.T) {
t.Parallel()
recs := []scanRec{
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
// Same size as the /z group but a different head hash.
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
// A singleton signature must not form a group.
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
}
groups := dupeGroupsOf(t, recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
if groups[0].size != 4000 ||
!slices.Equal(groups[0].paths, []string{"/big/1", "/big/2"}) {
t.Errorf("groups[0] = %+v, want size 4000, paths /big/1 /big/2",
groups[0])
}
if groups[1].size != 100 ||
!slices.Equal(groups[1].paths, []string{"/z/a", "/z/b", "/z/c"}) {
t.Errorf("groups[1] = %+v, want size 100, paths /z/a /z/b /z/c",
groups[1])
}
}
func TestDupeGroupsContentSeparates(t *testing.T) {
t.Parallel()
// Same size, head, and tail, but different content hashes: the final
// rung keeps them apart, so no group forms. Matching content groups.
// Records without a content hash never group, not even with each
// other.
recs := []scanRec{
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
{size: 100, head: "h", tail: "t", path: "/d"},
{size: 100, head: "h", tail: "t", path: "/e"},
}
groups := dupeGroupsOf(t, recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
len(groups))
}
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
}
}
func TestDupeGroupsMtimeExcluded(t *testing.T) {
t.Parallel()
// mtime is informational only; records differing only in mtime
// still group together.
recs := []scanRec{
{
size: 9, mtime: time.Unix(100, 0), head: "h", tail: "t",
content: "c", path: "/m/1",
},
{
size: 9, mtime: time.Unix(200, 0), head: "h", tail: "t",
content: "c", path: "/m/2",
},
}
groups := dupeGroupsOf(t, recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1", len(groups))
}
}
func TestDupeGroupsTieBreak(t *testing.T) {
t.Parallel()
// The hashes sort opposite to the first paths, so ordering the
// groups by hash instead of by first path fails this test.
recs := []scanRec{
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
}
groups := dupeGroupsOf(t, recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
// Equal sizes: ordered by first path ascending.
if groups[0].paths[0] != "/alpha/1" || groups[1].paths[0] != "/beta/1" {
t.Fatalf("tie-break order wrong: %q then %q",
groups[0].paths[0], groups[1].paths[0])
}
}
func TestDupeGroupsDeterministic(t *testing.T) {
t.Parallel()
recs := []scanRec{
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
}
forward := dupeGroupsOf(t, recs)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
backward := dupeGroupsOf(t, reversed)
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
return a.size == b.size && slices.Equal(a.paths, b.paths)
}) {
t.Fatalf("output depends on record order: %+v vs %+v",
forward, backward)
}
}
func TestHumanBytes(t *testing.T) {
t.Parallel()
cases := []struct {
n int64
want string
}{
{0, "0 B"},
{1, "1 B"},
{1023, "1023 B"},
{1024, "1.0 KiB"},
{1536, "1.5 KiB"},
{1 << 20, "1.0 MiB"},
{5 << 30, "5.0 GiB"},
{1 << 40, "1.0 TiB"},
{1 << 50, "1.0 PiB"},
{1 << 60, "1.0 EiB"},
}
for _, c := range cases {
if got := humanBytes(c.n); got != c.want {
t.Errorf("humanBytes(%d) = %q, want %q", c.n, got, c.want)
}
}
}