Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d8ca0c1f6c |
@@ -263,6 +263,18 @@ duplicates another or lies under another is dropped before walking,
|
||||
so every file is reached exactly once and produces one database
|
||||
record.
|
||||
|
||||
An operand that is a symlink (never followed, not even as an operand),
|
||||
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
||||
scanned. `scan` prints a one-line warning naming the path and what it
|
||||
is, counts it as skipped, and drops it from the scanned operands before
|
||||
reading the database. Another operand beneath it is still scanned. The
|
||||
records stored beneath it are not deleted: they are treated like any
|
||||
other record outside the scanned operands, including the content-phase
|
||||
exception below. If it lies under another operand, they are under that
|
||||
operand instead, and are deleted like any other record there that this
|
||||
scan did not verify. This is not an error: a scan whose every operand
|
||||
is dropped walks nothing and exits 0.
|
||||
|
||||
`scan` synchronizes the database with the filesystem state under the
|
||||
scanned operands:
|
||||
|
||||
@@ -368,9 +380,12 @@ during the hash phase:
|
||||
Rules for the walk:
|
||||
|
||||
- Only regular files. Skip directories, symlinks (do not follow,
|
||||
including symlink operands), sockets, FIFOs, and device nodes.
|
||||
including symlink operands), sockets, FIFOs, and device nodes. An
|
||||
operand that is a symlink, socket, FIFO, or device node is dropped
|
||||
as described in "`scan` mode" above.
|
||||
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
|
||||
walking them would list every file once per snapshot).
|
||||
walking them would list every file once per snapshot), not even
|
||||
when it is an operand; such an operand is dropped the same way.
|
||||
- Filesystem boundaries are crossed by default. With `-x`
|
||||
(long form `--one-file-system`, following the GNU `du`/`rsync`
|
||||
convention), never descend into a directory on a different
|
||||
@@ -381,9 +396,11 @@ Rules for the walk:
|
||||
path, and continue. Per-file errors never abort the run; the final
|
||||
summary reports how many were skipped. As specified above, a
|
||||
skipped path that has a database record from an earlier scan loses
|
||||
that record, unless it failed only in the content phase; an
|
||||
unreadable directory subtree likewise loses its records (accepted:
|
||||
the database mirrors what the latest scan could actually verify).
|
||||
that record, unless it failed only in the content phase, or is an
|
||||
operand dropped before the database was read that lies under no
|
||||
other operand; an unreadable directory subtree likewise loses its
|
||||
records (accepted: the database mirrors what the latest scan could
|
||||
actually verify).
|
||||
|
||||
Concurrency: the walk phase (which also stats files), the hash phase,
|
||||
and the content phase each use a worker pool of `--workers` workers
|
||||
@@ -573,18 +590,6 @@ Additional requirements:
|
||||
- `2`: usage error (including `scan` with no `PATH` operand and
|
||||
`report`/`trees` with any positional argument).
|
||||
|
||||
A stdout write failure, such as a full disk, is reported in one line on
|
||||
stderr and exits 1. Two cases never reach sfdupes as a failed write:
|
||||
|
||||
- When the reader of a stdout pipe exits early, as in
|
||||
`sfdupes report | head`, the next write ends sfdupes with `SIGPIPE`,
|
||||
quietly and without a summary, the way `cat` or `sort` end. The
|
||||
shell reports the signal (status 141 in most shells), not exit 1.
|
||||
- When stdout is closed outright (`sfdupes report >&-`), the Go
|
||||
runtime opens `/dev/null` in its place before sfdupes starts, so
|
||||
the output is discarded and the run succeeds, as with
|
||||
`> /dev/null`.
|
||||
|
||||
## Entrypoints
|
||||
|
||||
This repository adheres to the
|
||||
|
||||
@@ -29,9 +29,9 @@
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- test stdout write failures in `report` and `trees`; README states that
|
||||
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
|
||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
|
||||
- warn about and skip symlink, socket, FIFO, device and `.zfs`
|
||||
operands, keeping the records beneath them (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||
|
||||
- `report` and `trees` open the database read-only, and `scan` leaves it
|
||||
out of WAL mode, so reading needs only read access (2026-10-03, closes
|
||||
|
||||
@@ -57,27 +57,22 @@ var errNoSubcommand = errors.New("no subcommand")
|
||||
var Version = "dev"
|
||||
|
||||
func main() {
|
||||
// Once the reader of a stdout pipe has gone, as in "sfdupes report |
|
||||
// head", the Go runtime ends the process with SIGPIPE on the next
|
||||
// write instead of returning an error (README "Error handling").
|
||||
// Registering for SIGPIPE with os/signal would change that.
|
||||
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
|
||||
os.Exit(run(os.Args[1:], os.Stderr))
|
||||
}
|
||||
|
||||
// run executes args against the command tree and returns the process
|
||||
// exit code. It is the program's single exit point: the subcommands
|
||||
// return their errors instead of exiting, so every deferred cleanup —
|
||||
// above all closing the database, which checkpoints the SQLite WAL —
|
||||
// runs before the process ends. The report and trees subcommands write
|
||||
// their data to stdout.
|
||||
func run(args []string, stdout, stderr io.Writer) int {
|
||||
// runs before the process ends.
|
||||
func run(args []string, stderr io.Writer) int {
|
||||
// A nil slice makes cobra fall back to os.Args, which would let a
|
||||
// test binary's own flags reach the command tree.
|
||||
if args == nil {
|
||||
args = []string{}
|
||||
}
|
||||
|
||||
root := newRootCommand(stdout, stderr)
|
||||
root := newRootCommand(stderr)
|
||||
root.SetArgs(args)
|
||||
|
||||
err := root.Execute()
|
||||
@@ -103,7 +98,7 @@ func run(args []string, stdout, stderr io.Writer) int {
|
||||
// newRootCommand builds the command tree. Everything on stdout is
|
||||
// machine-readable data; all human-facing output (help, usage, errors)
|
||||
// goes to stderr.
|
||||
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
||||
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
root := &cobra.Command{
|
||||
Use: "sfdupes",
|
||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||
@@ -145,7 +140,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
||||
Short: "Read the scan database and print the file-level duplicates report",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||
return runReport(ctx, stdout)
|
||||
return runReport(ctx)
|
||||
}),
|
||||
}
|
||||
|
||||
@@ -154,7 +149,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
||||
Short: "Read the scan database and print the duplicate-tree report",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||
return runTrees(ctx, stdout)
|
||||
return runTrees(ctx)
|
||||
}),
|
||||
}
|
||||
|
||||
|
||||
+204
-92
@@ -82,6 +82,59 @@ func makeReadOnly(t *testing.T, path string) {
|
||||
})
|
||||
}
|
||||
|
||||
// captureStdout redirects os.Stdout to a file for the rest of the test
|
||||
// and returns a function reading back everything written to it. Only
|
||||
// machine-readable data belongs on stdout (README design goal 4), so
|
||||
// the tests assert on it directly.
|
||||
func captureStdout(t *testing.T) func() string {
|
||||
t.Helper()
|
||||
|
||||
return captureStream(t, &os.Stdout)
|
||||
}
|
||||
|
||||
// captureStream redirects *stream (os.Stdout or os.Stderr) to a file
|
||||
// for the rest of the test and returns a function reading back
|
||||
// everything written to it.
|
||||
func captureStream(t *testing.T, stream **os.File) func() string {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "capture"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := *stream
|
||||
*stream = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
*stream = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
|
||||
return func() string {
|
||||
// Read what has been written without disturbing the write
|
||||
// offset, so the capture can be inspected more than once.
|
||||
size, err := f.Seek(0, io.SeekCurrent)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if size == 0 {
|
||||
return ""
|
||||
}
|
||||
|
||||
b := make([]byte, size)
|
||||
|
||||
_, err = f.ReadAt(b, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return string(b)
|
||||
}
|
||||
}
|
||||
|
||||
// brokenDatabase writes a database that opens cleanly and passes the
|
||||
// schema-version check but has no files table, so the first query
|
||||
// fails with the database already open: a fatal error on a path that
|
||||
@@ -154,15 +207,17 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||
args = append(args, t.TempDir())
|
||||
}
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run(args, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run(args, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
||||
assertFatalOutput(t, stderr.String(), stdout())
|
||||
|
||||
// Proof that the failure happened after the open: only a
|
||||
// query against the opened database can report this.
|
||||
@@ -180,16 +235,18 @@ func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
|
||||
// must not dump the usage text.
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
missing := filepath.Join(t.TempDir(), "nope")
|
||||
|
||||
code := run([]string{cmdScan, missing}, &stdout, &stderr)
|
||||
code := run([]string{cmdScan, missing}, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
||||
}
|
||||
|
||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
||||
assertFatalOutput(t, stderr.String(), stdout())
|
||||
}
|
||||
|
||||
// assertFatalOutput checks that a fatal error was reported the way
|
||||
@@ -233,9 +290,11 @@ func TestRunUsageErrors(t *testing.T) {
|
||||
// path that does not exist.
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run(tc.args, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run(tc.args, &stderr)
|
||||
if code != exitUsage {
|
||||
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
||||
}
|
||||
@@ -244,7 +303,7 @@ func TestRunUsageErrors(t *testing.T) {
|
||||
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
if got := stdout(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
})
|
||||
@@ -253,9 +312,9 @@ func TestRunUsageErrors(t *testing.T) {
|
||||
|
||||
// TestRunHelpAndVersionSucceed checks that the two informational flags
|
||||
// exit 0 and keep their human-facing output on stderr.
|
||||
//
|
||||
//nolint:paralleltest // captureStdout replaces the process-wide os.Stdout
|
||||
func TestRunHelpAndVersionSucceed(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
assertHumanOutput(t, "--help")
|
||||
assertHumanOutput(t, "--version")
|
||||
}
|
||||
@@ -266,9 +325,11 @@ func TestRunHelpAndVersionSucceed(t *testing.T) {
|
||||
func assertHumanOutput(t *testing.T, arg string) {
|
||||
t.Helper()
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{arg}, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{arg}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
||||
}
|
||||
@@ -277,7 +338,7 @@ func assertHumanOutput(t *testing.T, arg string) {
|
||||
t.Errorf("run(%s) wrote nothing to stderr", arg)
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
if got := stdout(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
}
|
||||
@@ -306,19 +367,30 @@ func scanFixture(t *testing.T) []string {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
scanOK(t, dir)
|
||||
|
||||
code := run([]string{cmdScan, dir}, &stdout, &stderr)
|
||||
return dupes
|
||||
}
|
||||
|
||||
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
||||
// nothing on stdout, and returns everything it printed to stderr.
|
||||
func scanOK(t *testing.T, operands ...string) string {
|
||||
t.Helper()
|
||||
|
||||
stdout := captureStdout(t)
|
||||
stderr := captureStream(t, &os.Stderr)
|
||||
|
||||
code := run(append([]string{cmdScan}, operands...), os.Stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
||||
operands, code, exitOK, stderr())
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
if got := stdout(); got != "" {
|
||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
return dupes
|
||||
return stderr()
|
||||
}
|
||||
|
||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
@@ -329,22 +401,121 @@ func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Scanning a directory through the symlink stores a record beneath
|
||||
// the symlink's own path for a file beneath its target.
|
||||
scanOK(t, filepath.Join(link, "sub"))
|
||||
|
||||
assertOperandSkipped(t, path, link, "symlink",
|
||||
filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// link is dropped as a symlink, but link/sub must still be scanned,
|
||||
// not dropped as lying under link.
|
||||
scanOK(t, link, filepath.Join(link, "sub"))
|
||||
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
zfs := filepath.Join(t.TempDir(), ".zfs")
|
||||
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
||||
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
||||
|
||||
// An operand beneath a .zfs directory is walked, because it is not
|
||||
// itself named .zfs.
|
||||
scanOK(t, snapshot)
|
||||
|
||||
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
||||
}
|
||||
|
||||
// assertOperandSkipped scans operand alone and checks that it is skipped
|
||||
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
||||
// record for kept, which an earlier scan stored beneath operand, still
|
||||
// in the database at dbPath.
|
||||
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
||||
kept string,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
stderr := scanOK(t, operand)
|
||||
|
||||
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
||||
if !strings.Contains(stderr, warning) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
||||
}
|
||||
|
||||
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
||||
"0 unchanged), 1 skipped\n"
|
||||
if !strings.Contains(stderr, summary) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
||||
}
|
||||
|
||||
db, err := openDB(dbPath, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), kept)
|
||||
}
|
||||
|
||||
func TestRunReportSucceeds(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdReport}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
||||
if got := stdout.String(); got != want {
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
@@ -357,9 +528,11 @@ func TestRunTreesSucceeds(t *testing.T) {
|
||||
|
||||
dupes := scanFixture(t)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdTrees}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
@@ -369,7 +542,7 @@ func TestRunTreesSucceeds(t *testing.T) {
|
||||
// trees of each other.
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
||||
if got := stdout.String(); got != want {
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
@@ -396,9 +569,11 @@ func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
||||
}
|
||||
|
||||
for name, want := range cases {
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{name}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
@@ -406,71 +581,8 @@ func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
||||
continue
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != want {
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("%s stdout = %q, want %q", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunStdoutClosedIsFatal(t *testing.T) {
|
||||
// README §Error handling: a stdout write failure exits 1, reported
|
||||
// in one line on stderr.
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
scanFixture(t)
|
||||
|
||||
stdout, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = stdout.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
|
||||
}
|
||||
|
||||
got := stderr.String()
|
||||
if !strings.HasPrefix(got, "sfdupes: write stdout: ") ||
|
||||
!strings.Contains(got, os.ErrClosed.Error()) ||
|
||||
strings.Count(got, "\n") != 1 {
|
||||
t.Errorf("stderr = %q, want one line reporting the "+
|
||||
"failed stdout write", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// errWriteFailed is the error failingWriter returns.
|
||||
var errWriteFailed = errors.New("write failed")
|
||||
|
||||
// failingWriter is a stdout that fails every write.
|
||||
type failingWriter struct{}
|
||||
|
||||
func (failingWriter) Write([]byte) (int, error) { return 0, errWriteFailed }
|
||||
|
||||
func TestStdoutWriteErrorPropagates(t *testing.T) {
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
scanFixture(t)
|
||||
|
||||
cases := map[string]func(context.Context, io.Writer) error{
|
||||
cmdReport: runReport,
|
||||
cmdTrees: runTrees,
|
||||
}
|
||||
|
||||
for name, fn := range cases {
|
||||
err := fn(t.Context(), failingWriter{})
|
||||
if !errors.Is(err, errWriteFailed) {
|
||||
t.Errorf("%s: error = %v, want %v", name, err, errWriteFailed)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,7 +4,6 @@ import (
|
||||
"bufio"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"slices"
|
||||
"strings"
|
||||
@@ -66,7 +65,7 @@ type dupeGroup struct {
|
||||
// from the database and prints the file-level duplicates report as TSV
|
||||
// on stdout. It never touches the scanned filesystem; its only I/O is
|
||||
// the database, stdout, and stderr.
|
||||
func runReport(ctx context.Context, stdout io.Writer) error {
|
||||
func runReport(ctx context.Context) error {
|
||||
recs, err := loadRecords(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -74,7 +73,7 @@ func runReport(ctx context.Context, stdout io.Writer) error {
|
||||
|
||||
dupes := collectDupeGroups(recs)
|
||||
|
||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||
|
||||
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
||||
if err != nil {
|
||||
|
||||
+5
-3
@@ -50,9 +50,11 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||
func TestRunReportEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdReport}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
@@ -60,7 +62,7 @@ func TestRunReportEscapesPaths(t *testing.T) {
|
||||
|
||||
want := "first\tdupe\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
||||
if got := stdout.String(); got != want {
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -211,14 +211,18 @@ type scanState struct {
|
||||
// in the content hash of every record of headTailMin or more whose
|
||||
// size, head, and tail match another record's). Records outside the
|
||||
// roots are never touched, except that the content phase fills in
|
||||
// their content hash.
|
||||
// their content hash. Operands the walk cannot start from are dropped
|
||||
// first, so the records beneath them count as outside the roots unless
|
||||
// they lie under another root.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
roots = pruneRoots(roots)
|
||||
|
||||
s := &scanState{db: db}
|
||||
|
||||
// Types are checked before pruning so that an operand under a
|
||||
// dropped one is still scanned, not dropped as lying under it.
|
||||
roots = pruneRoots(s.walkableRoots(roots))
|
||||
|
||||
err := s.loadIndex(ctx, roots)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
@@ -254,6 +258,35 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
return s.st, s.contentPhase(ctx, workers)
|
||||
}
|
||||
|
||||
// walkableRoots returns the operands the walk can start from: regular
|
||||
// files, and directories not named .zfs. Every other operand is warned
|
||||
// about, counted as skipped, and dropped. A dropped operand is no
|
||||
// longer a root, so the records stored beneath it count as outside the
|
||||
// roots and are not deleted as unverified, unless it lies under another
|
||||
// root. An operand that fails lstat here is kept, and the walk warns
|
||||
// about it.
|
||||
func (s *scanState) walkableRoots(roots []string) []string {
|
||||
kept := make([]string, 0, len(roots))
|
||||
|
||||
for _, root := range roots {
|
||||
fi, err := os.Lstat(root)
|
||||
if err == nil {
|
||||
warn := operandWarning(root, fi)
|
||||
if warn != "" {
|
||||
s.st.skipped++
|
||||
|
||||
fmt.Fprintln(os.Stderr, escapePath(warn))
|
||||
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
kept = append(kept, root)
|
||||
}
|
||||
|
||||
return kept
|
||||
}
|
||||
|
||||
// loadIndex indexes the database records under the scan roots for
|
||||
// change detection and collects the sizes of every record outside
|
||||
// them: out-of-scope records join the size census so a scanned file
|
||||
@@ -790,11 +823,43 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
|
||||
}
|
||||
}
|
||||
|
||||
// operandWarning returns the one-line warning for an operand the walk
|
||||
// does not start from, naming the path and what it is, or "" for one it
|
||||
// does: a regular file, or a directory not named .zfs. Symlinks are
|
||||
// never followed, including as operands.
|
||||
func operandWarning(root string, fi fs.FileInfo) string {
|
||||
var kind string
|
||||
|
||||
switch mode := fi.Mode(); {
|
||||
case mode.IsRegular():
|
||||
return ""
|
||||
case mode.IsDir():
|
||||
if filepath.Base(root) != ".zfs" {
|
||||
return ""
|
||||
}
|
||||
|
||||
kind = ".zfs directory"
|
||||
case mode&fs.ModeSymlink != 0:
|
||||
kind = "symlink"
|
||||
case mode&fs.ModeSocket != 0:
|
||||
kind = "socket"
|
||||
case mode&fs.ModeNamedPipe != 0:
|
||||
kind = "FIFO"
|
||||
case mode&fs.ModeDevice != 0:
|
||||
kind = "device node"
|
||||
default:
|
||||
kind = "non-regular file"
|
||||
}
|
||||
|
||||
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
|
||||
}
|
||||
|
||||
// seedRoot turns one PATH operand into the walk's starting state: a
|
||||
// regular-file operand is statted and emitted directly, a directory
|
||||
// operand becomes an initial job, and a symlink or other non-regular
|
||||
// operand yields nothing (symlinks are never followed, including as
|
||||
// operands).
|
||||
// regular-file operand is statted and emitted directly, and a directory
|
||||
// operand becomes an initial job. walkableRoots has already dropped
|
||||
// every other operand. One that has changed into something else since
|
||||
// is warned about and skipped here; it is still a root, so the records
|
||||
// stored beneath it are deleted as unverified.
|
||||
func seedRoot(ctx context.Context, root string,
|
||||
events chan<- walkEvent,
|
||||
) []dirJob {
|
||||
@@ -808,30 +873,30 @@ func seedRoot(ctx context.Context, root string,
|
||||
return nil
|
||||
}
|
||||
|
||||
switch {
|
||||
case fi.IsDir():
|
||||
if filepath.Base(root) == ".zfs" {
|
||||
return nil
|
||||
}
|
||||
warn := operandWarning(root, fi)
|
||||
if warn != "" {
|
||||
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
if fi.IsDir() {
|
||||
dev, ok := deviceOfInfo(fi)
|
||||
|
||||
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
||||
case fi.Mode().IsRegular():
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
path: root,
|
||||
size: fi.Size(),
|
||||
mtime: fi.ModTime().Unix(),
|
||||
dev: dev,
|
||||
ino: ino,
|
||||
}})
|
||||
|
||||
return nil
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
path: root,
|
||||
size: fi.Size(),
|
||||
mtime: fi.ModTime().Unix(),
|
||||
dev: dev,
|
||||
ino: ino,
|
||||
}})
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// startWalkWorkers starts the walk worker pool. Each worker processes
|
||||
|
||||
+5
-4
@@ -7,7 +7,6 @@ import (
|
||||
"database/sql"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
@@ -857,9 +856,11 @@ func TestWalkFileAndSymlinkOperands(t *testing.T) {
|
||||
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
||||
}
|
||||
|
||||
// A symlink operand is not followed and yields nothing.
|
||||
// A symlink operand that reaches the walk (it became one after
|
||||
// walkableRoots checked it) is not followed: it yields a warning and
|
||||
// no records.
|
||||
recs, errs = collectWalk(t, []string{link}, false, 2)
|
||||
if errs != 0 || len(recs) != 0 {
|
||||
if errs != 1 || len(recs) != 0 {
|
||||
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
||||
}
|
||||
}
|
||||
@@ -1549,7 +1550,7 @@ func TestScanHashWriteFailureUnwindsPool(t *testing.T) {
|
||||
|
||||
code := run([]string{
|
||||
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
|
||||
}, io.Discard, &stderr)
|
||||
}, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||
code, exitFatal, stderr.String())
|
||||
|
||||
@@ -5,7 +5,6 @@ import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"slices"
|
||||
"strconv"
|
||||
@@ -37,7 +36,7 @@ type treeNode struct {
|
||||
// maximal duplicate-tree groups as TSV on stdout. It never touches the
|
||||
// scanned filesystem; its only I/O is the database, stdout, and
|
||||
// stderr.
|
||||
func runTrees(ctx context.Context, stdout io.Writer) error {
|
||||
func runTrees(ctx context.Context) error {
|
||||
recs, err := loadRecords(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -48,7 +47,7 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
|
||||
|
||||
dupes := collectTreeGroups(allDirs, super)
|
||||
|
||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||
|
||||
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
||||
if err != nil {
|
||||
|
||||
+5
-3
@@ -108,9 +108,11 @@ func TestBuildHierarchyRootPath(t *testing.T) {
|
||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdTrees}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
@@ -118,7 +120,7 @@ func TestRunTreesEscapesPaths(t *testing.T) {
|
||||
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
||||
if got := stdout.String(); got != want {
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user