Compare commits

1 Commits
Author SHA1 Message Date
sneak d8ca0c1f6c Warn about and skip non-regular and .zfs operands, keeping their records (closes #9)
check / check (push) Successful in 1m8s
A symlink, socket, FIFO or device-node operand, or a directory operand
named .zfs, was silently ignored yet stayed in the scanned operands, so
the update phase deleted every record stored beneath it. Such an
operand now gets a one-line warning, counts as skipped, and is dropped
before overlapping operands are pruned and the database index is
loaded: another operand beneath it is still scanned, and the records
beneath it count as outside the scanned operands and are not deleted,
unless it lies under another operand. The exit status stays 0. An
operand that turns into one of these after that check is warned about
and skipped by the walk instead. README "scan mode" and "Rules for the
walk" say so.

Model: opus-5-5
2026-10-03 15:49:35 +00:00
9 changed files with 81 additions and 142 deletions
-12
View File
@@ -590,18 +590,6 @@ Additional requirements:
- `2`: usage error (including `scan` with no `PATH` operand and
`report`/`trees` with any positional argument).
A stdout write failure, such as a full disk, is reported in one line on
stderr and exits 1. Two cases never reach sfdupes as a failed write:
- When the reader of a stdout pipe exits early, as in
`sfdupes report | head`, the next write ends sfdupes with `SIGPIPE`,
quietly and without a summary, the way `cat` or `sort` end. The
shell reports the signal (status 141 in most shells), not exit 1.
- When stdout is closed outright (`sfdupes report >&-`), the Go
runtime opens `/dev/null` in its place before sfdupes starts, so
the output is discarded and the run succeeds, as with
`> /dev/null`.
## Entrypoints
This repository adheres to the
-4
View File
@@ -33,10 +33,6 @@
operands, keeping the records beneath them (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/9)
- test stdout write failures in `report` and `trees`; README states that
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
- `report` and `trees` open the database read-only, and `scan` leaves it
out of WAL mode, so reading needs only read access (2026-10-03, closes
https://git.eeqj.de/sneak/sfdupes/issues/8)
+7 -12
View File
@@ -57,27 +57,22 @@ var errNoSubcommand = errors.New("no subcommand")
var Version = "dev"
func main() {
// Once the reader of a stdout pipe has gone, as in "sfdupes report |
// head", the Go runtime ends the process with SIGPIPE on the next
// write instead of returning an error (README "Error handling").
// Registering for SIGPIPE with os/signal would change that.
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
os.Exit(run(os.Args[1:], os.Stderr))
}
// run executes args against the command tree and returns the process
// exit code. It is the program's single exit point: the subcommands
// return their errors instead of exiting, so every deferred cleanup —
// above all closing the database, which checkpoints the SQLite WAL —
// runs before the process ends. The report and trees subcommands write
// their data to stdout.
func run(args []string, stdout, stderr io.Writer) int {
// runs before the process ends.
func run(args []string, stderr io.Writer) int {
// A nil slice makes cobra fall back to os.Args, which would let a
// test binary's own flags reach the command tree.
if args == nil {
args = []string{}
}
root := newRootCommand(stdout, stderr)
root := newRootCommand(stderr)
root.SetArgs(args)
err := root.Execute()
@@ -103,7 +98,7 @@ func run(args []string, stdout, stderr io.Writer) int {
// newRootCommand builds the command tree. Everything on stdout is
// machine-readable data; all human-facing output (help, usage, errors)
// goes to stderr.
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
func newRootCommand(stderr io.Writer) *cobra.Command {
root := &cobra.Command{
Use: "sfdupes",
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
@@ -145,7 +140,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
Short: "Read the scan database and print the file-level duplicates report",
Args: cobra.NoArgs,
RunE: runE(func(ctx context.Context, _ []string) error {
return runReport(ctx, stdout)
return runReport(ctx)
}),
}
@@ -154,7 +149,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
Short: "Read the scan database and print the duplicate-tree report",
Args: cobra.NoArgs,
RunE: runE(func(ctx context.Context, _ []string) error {
return runTrees(ctx, stdout)
return runTrees(ctx)
}),
}
+59 -100
View File
@@ -82,23 +82,32 @@ func makeReadOnly(t *testing.T, path string) {
})
}
// captureStderr redirects os.Stderr to a file for the rest of the test
// and returns a function reading back everything written to it. scan
// writes its warnings and summary straight to os.Stderr, not to the
// stderr writer run is given.
func captureStderr(t *testing.T) func() string {
// captureStdout redirects os.Stdout to a file for the rest of the test
// and returns a function reading back everything written to it. Only
// machine-readable data belongs on stdout (README design goal 4), so
// the tests assert on it directly.
func captureStdout(t *testing.T) func() string {
t.Helper()
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
return captureStream(t, &os.Stdout)
}
// captureStream redirects *stream (os.Stdout or os.Stderr) to a file
// for the rest of the test and returns a function reading back
// everything written to it.
func captureStream(t *testing.T, stream **os.File) func() string {
t.Helper()
f, err := os.Create(filepath.Join(t.TempDir(), "capture"))
if err != nil {
t.Fatal(err)
}
saved := os.Stderr
os.Stderr = f
saved := *stream
*stream = f
t.Cleanup(func() {
os.Stderr = saved
*stream = saved
_ = f.Close()
})
@@ -198,15 +207,17 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
args = append(args, t.TempDir())
}
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run(args, &stdout, &stderr)
stdout := captureStdout(t)
code := run(args, &stderr)
if code != exitFatal {
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
}
assertNoSidecars(t, path)
assertFatalOutput(t, stderr.String(), stdout.String())
assertFatalOutput(t, stderr.String(), stdout())
// Proof that the failure happened after the open: only a
// query against the opened database can report this.
@@ -224,16 +235,18 @@ func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
// must not dump the usage text.
t.Setenv(databaseEnv, testDBPath(t))
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
stdout := captureStdout(t)
missing := filepath.Join(t.TempDir(), "nope")
code := run([]string{cmdScan, missing}, &stdout, &stderr)
code := run([]string{cmdScan, missing}, &stderr)
if code != exitFatal {
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
}
assertFatalOutput(t, stderr.String(), stdout.String())
assertFatalOutput(t, stderr.String(), stdout())
}
// assertFatalOutput checks that a fatal error was reported the way
@@ -277,9 +290,11 @@ func TestRunUsageErrors(t *testing.T) {
// path that does not exist.
t.Setenv(databaseEnv, testDBPath(t))
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run(tc.args, &stdout, &stderr)
stdout := captureStdout(t)
code := run(tc.args, &stderr)
if code != exitUsage {
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
}
@@ -288,7 +303,7 @@ func TestRunUsageErrors(t *testing.T) {
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
}
if got := stdout.String(); got != "" {
if got := stdout(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
}
})
@@ -297,9 +312,9 @@ func TestRunUsageErrors(t *testing.T) {
// TestRunHelpAndVersionSucceed checks that the two informational flags
// exit 0 and keep their human-facing output on stderr.
//
//nolint:paralleltest // captureStdout replaces the process-wide os.Stdout
func TestRunHelpAndVersionSucceed(t *testing.T) {
t.Parallel()
assertHumanOutput(t, "--help")
assertHumanOutput(t, "--version")
}
@@ -310,9 +325,11 @@ func TestRunHelpAndVersionSucceed(t *testing.T) {
func assertHumanOutput(t *testing.T, arg string) {
t.Helper()
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run([]string{arg}, &stdout, &stderr)
stdout := captureStdout(t)
code := run([]string{arg}, &stderr)
if code != exitOK {
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
}
@@ -321,7 +338,7 @@ func assertHumanOutput(t *testing.T, arg string) {
t.Errorf("run(%s) wrote nothing to stderr", arg)
}
if got := stdout.String(); got != "" {
if got := stdout(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
}
}
@@ -360,17 +377,16 @@ func scanFixture(t *testing.T) []string {
func scanOK(t *testing.T, operands ...string) string {
t.Helper()
var stdout bytes.Buffer
stdout := captureStdout(t)
stderr := captureStream(t, &os.Stderr)
stderr := captureStderr(t)
code := run(append([]string{cmdScan}, operands...), &stdout, os.Stderr)
code := run(append([]string{cmdScan}, operands...), os.Stderr)
if code != exitOK {
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
operands, code, exitOK, stderr())
}
if got := stdout.String(); got != "" {
if got := stdout(); got != "" {
t.Errorf("scan stdout = %q, want nothing (data only)", got)
}
@@ -488,16 +504,18 @@ func TestRunReportSucceeds(t *testing.T) {
dupes := scanFixture(t)
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run([]string{cmdReport}, &stdout, &stderr)
stdout := captureStdout(t)
code := run([]string{cmdReport}, &stderr)
if code != exitOK {
t.Fatalf("run(report) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
if got := stdout.String(); got != want {
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
@@ -510,9 +528,11 @@ func TestRunTreesSucceeds(t *testing.T) {
dupes := scanFixture(t)
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run([]string{cmdTrees}, &stdout, &stderr)
stdout := captureStdout(t)
code := run([]string{cmdTrees}, &stderr)
if code != exitOK {
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
@@ -522,7 +542,7 @@ func TestRunTreesSucceeds(t *testing.T) {
// trees of each other.
want := "first\tdupe\tfiles\tsize\n" +
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
if got := stdout.String(); got != want {
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
@@ -549,9 +569,11 @@ func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
}
for name, want := range cases {
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
stdout := captureStdout(t)
code := run([]string{name}, &stderr)
if code != exitOK {
t.Errorf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
@@ -559,71 +581,8 @@ func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
continue
}
if got := stdout.String(); got != want {
if got := stdout(); got != want {
t.Errorf("%s stdout = %q, want %q", name, got, want)
}
}
}
func TestRunStdoutClosedIsFatal(t *testing.T) {
// README §Error handling: a stdout write failure exits 1, reported
// in one line on stderr.
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
t.Setenv(databaseEnv, testDBPath(t))
scanFixture(t)
stdout, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
if err != nil {
t.Fatal(err)
}
err = stdout.Close()
if err != nil {
t.Fatal(err)
}
var stderr bytes.Buffer
code := run([]string{name}, stdout, &stderr)
if code != exitFatal {
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
}
got := stderr.String()
if !strings.HasPrefix(got, "sfdupes: write stdout: ") ||
!strings.Contains(got, os.ErrClosed.Error()) ||
strings.Count(got, "\n") != 1 {
t.Errorf("stderr = %q, want one line reporting the "+
"failed stdout write", got)
}
})
}
}
// errWriteFailed is the error failingWriter returns.
var errWriteFailed = errors.New("write failed")
// failingWriter is a stdout that fails every write.
type failingWriter struct{}
func (failingWriter) Write([]byte) (int, error) { return 0, errWriteFailed }
func TestStdoutWriteErrorPropagates(t *testing.T) {
t.Setenv(databaseEnv, testDBPath(t))
scanFixture(t)
cases := map[string]func(context.Context, io.Writer) error{
cmdReport: runReport,
cmdTrees: runTrees,
}
for name, fn := range cases {
err := fn(t.Context(), failingWriter{})
if !errors.Is(err, errWriteFailed) {
t.Errorf("%s: error = %v, want %v", name, err, errWriteFailed)
}
}
}
+2 -3
View File
@@ -4,7 +4,6 @@ import (
"bufio"
"context"
"fmt"
"io"
"os"
"slices"
"strings"
@@ -66,7 +65,7 @@ type dupeGroup struct {
// from the database and prints the file-level duplicates report as TSV
// on stdout. It never touches the scanned filesystem; its only I/O is
// the database, stdout, and stderr.
func runReport(ctx context.Context, stdout io.Writer) error {
func runReport(ctx context.Context) error {
recs, err := loadRecords(ctx)
if err != nil {
return err
@@ -74,7 +73,7 @@ func runReport(ctx context.Context, stdout io.Writer) error {
dupes := collectDupeGroups(recs)
out := bufio.NewWriterSize(stdout, ioBufSize)
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
if err != nil {
+5 -3
View File
@@ -50,9 +50,11 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run([]string{cmdReport}, &stdout, &stderr)
stdout := captureStdout(t)
code := run([]string{cmdReport}, &stderr)
if code != exitOK {
t.Fatalf("run(report) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
@@ -60,7 +62,7 @@ func TestRunReportEscapesPaths(t *testing.T) {
want := "first\tdupe\tsize\n" +
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
if got := stdout.String(); got != want {
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}
+1 -2
View File
@@ -7,7 +7,6 @@ import (
"database/sql"
"encoding/hex"
"fmt"
"io"
"os"
"path/filepath"
"runtime"
@@ -1551,7 +1550,7 @@ func TestScanHashWriteFailureUnwindsPool(t *testing.T) {
code := run([]string{
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
}, io.Discard, &stderr)
}, &stderr)
if code != exitFatal {
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
code, exitFatal, stderr.String())
+2 -3
View File
@@ -5,7 +5,6 @@ import (
"context"
"crypto/sha256"
"fmt"
"io"
"os"
"slices"
"strconv"
@@ -37,7 +36,7 @@ type treeNode struct {
// maximal duplicate-tree groups as TSV on stdout. It never touches the
// scanned filesystem; its only I/O is the database, stdout, and
// stderr.
func runTrees(ctx context.Context, stdout io.Writer) error {
func runTrees(ctx context.Context) error {
recs, err := loadRecords(ctx)
if err != nil {
return err
@@ -48,7 +47,7 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
dupes := collectTreeGroups(allDirs, super)
out := bufio.NewWriterSize(stdout, ioBufSize)
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
if err != nil {
+5 -3
View File
@@ -108,9 +108,11 @@ func TestBuildHierarchyRootPath(t *testing.T) {
func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
code := run([]string{cmdTrees}, &stdout, &stderr)
stdout := captureStdout(t)
code := run([]string{cmdTrees}, &stderr)
if code != exitOK {
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
@@ -118,7 +120,7 @@ func TestRunTreesEscapesPaths(t *testing.T) {
want := "first\tdupe\tfiles\tsize\n" +
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
if got := stdout.String(); got != want {
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}