Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3fcf4993e2 |
@@ -255,11 +255,9 @@ An operand that is a symlink (never followed, not even as an operand),
|
|||||||
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
||||||
scanned. `scan` prints a one-line warning naming the path and what it
|
scanned. `scan` prints a one-line warning naming the path and what it
|
||||||
is, counts it as skipped, and drops it from the scanned operands before
|
is, counts it as skipped, and drops it from the scanned operands before
|
||||||
reading the database. The records stored beneath it are left as they
|
reading the database, so the records stored beneath it are left as
|
||||||
are, unless it lies under another operand: then they are deleted like
|
they are. This is not an error: a scan whose every operand is dropped
|
||||||
any other record under that operand that this scan did not verify.
|
walks nothing and exits 0.
|
||||||
This is not an error: a scan whose every operand is dropped walks
|
|
||||||
nothing and exits 0.
|
|
||||||
|
|
||||||
`scan` synchronizes the database with the filesystem state under the
|
`scan` synchronizes the database with the filesystem state under the
|
||||||
scanned operands:
|
scanned operands:
|
||||||
@@ -382,9 +380,8 @@ Rules for the walk:
|
|||||||
path, and continue. Per-file errors never abort the run; the final
|
path, and continue. Per-file errors never abort the run; the final
|
||||||
summary reports how many were skipped. As specified above, a
|
summary reports how many were skipped. As specified above, a
|
||||||
skipped path that has a database record from an earlier scan loses
|
skipped path that has a database record from an earlier scan loses
|
||||||
that record, unless it failed only in the content phase, or is an
|
that record, unless it failed only in the content phase or is an
|
||||||
operand dropped before the database was read that lies under no
|
operand dropped before the database was read; an
|
||||||
other operand; an
|
|
||||||
unreadable directory subtree likewise loses its records (accepted:
|
unreadable directory subtree likewise loses its records (accepted:
|
||||||
the database mirrors what the latest scan could actually verify).
|
the database mirrors what the latest scan could actually verify).
|
||||||
|
|
||||||
@@ -443,15 +440,6 @@ first dupe size
|
|||||||
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
||||||
```
|
```
|
||||||
|
|
||||||
Paths are raw bytes and may hold any byte except NUL, so the path
|
|
||||||
columns (`first` and `dupe`) are escaped to keep every row one line of
|
|
||||||
tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a
|
|
||||||
newline as `\n`, and a carriage return as `\r`. Every other byte is
|
|
||||||
written unchanged, including bytes that are not valid UTF-8. Undoing
|
|
||||||
those four escapes gives back the stored path. Grouping and ordering
|
|
||||||
use the stored path, not the escaped one. The warnings `scan` prints on
|
|
||||||
stderr are escaped the same way, so each warning is one line.
|
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate groups, number of
|
Summary to stderr: records read, number of duplicate groups, number of
|
||||||
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
||||||
rows) in human units.
|
rows) in human units.
|
||||||
@@ -523,9 +511,6 @@ first dupe files size
|
|||||||
/srv/a/project /srv/backup/project 3417 104857600
|
/srv/a/project /srv/backup/project 3417 104857600
|
||||||
```
|
```
|
||||||
|
|
||||||
The `first` and `dupe` paths are escaped as described under "Report
|
|
||||||
output format". The root directory's path is `/`.
|
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate-tree groups,
|
Summary to stderr: records read, number of duplicate-tree groups,
|
||||||
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
||||||
all dupe rows) in human units.
|
all dupe rows) in human units.
|
||||||
|
|||||||
@@ -33,10 +33,6 @@
|
|||||||
operands, keeping the records beneath them (2026-10-03,
|
operands, keeping the records beneath them (2026-10-03,
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||||
|
|
||||||
- escape tabs, newlines, carriage returns and backslashes in report,
|
|
||||||
trees and warning paths; the root directory's path is `/`
|
|
||||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
|
|
||||||
|
|
||||||
- stamp the git tag or short commit in a plain `docker build .`
|
- stamp the git tag or short commit in a plain `docker build .`
|
||||||
instead of `dev` (2026-10-02, branch `next`, closes
|
instead of `dev` (2026-10-02, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
||||||
|
|||||||
+1
-3
@@ -106,8 +106,6 @@ func (p *progress) increment() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||||
// The whole message is escaped like a report's path columns, so a path
|
|
||||||
// holding a newline cannot split the warning.
|
|
||||||
func (p *progress) warnf(format string, args ...any) {
|
func (p *progress) warnf(format string, args ...any) {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
@@ -117,7 +115,7 @@ func (p *progress) warnf(format string, args ...any) {
|
|||||||
_ = p.bar.Clear()
|
_ = p.bar.Clear()
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, escapePath(fmt.Sprintf(format, args...)))
|
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display.
|
// finish terminates the pass's display.
|
||||||
|
|||||||
@@ -87,7 +87,7 @@ func runReport(ctx context.Context) error {
|
|||||||
for _, g := range dupes {
|
for _, g := range dupes {
|
||||||
for _, p := range g.paths[1:] {
|
for _, p := range g.paths[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||||
escapePath(g.paths[0]), escapePath(p), g.size)
|
g.paths[0], p, g.size)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -156,21 +156,6 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
|||||||
return dupes
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
// escapePath returns a path as it is written in a report column (README
|
|
||||||
// "Report output format"): a backslash, tab, newline or carriage return
|
|
||||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
|
||||||
// Grouping and sorting use the raw path, never this form.
|
|
||||||
func escapePath(p string) string {
|
|
||||||
// Most paths need no escaping; skip building a replacer for them.
|
|
||||||
if !strings.ContainsAny(p, "\\\t\n\r") {
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
return strings.NewReplacer(
|
|
||||||
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
|
||||||
).Replace(p)
|
|
||||||
}
|
|
||||||
|
|
||||||
// humanBytes formats a byte count in human units (binary prefixes).
|
// humanBytes formats a byte count in human units (binary prefixes).
|
||||||
func humanBytes(n int64) string {
|
func humanBytes(n int64) string {
|
||||||
const unit = 1024
|
const unit = 1024
|
||||||
|
|||||||
-118
@@ -1,128 +1,10 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"io"
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
// awkwardDir is a directory name holding every byte the reports escape.
|
|
||||||
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
|
||||||
|
|
||||||
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
|
||||||
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
|
||||||
// sorts after it, so awkwardDir coming first shows that sorting uses
|
|
||||||
// the raw path.
|
|
||||||
func awkwardPairRecs() []scanRec {
|
|
||||||
return []scanRec{
|
|
||||||
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
|
||||||
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// seedDatabase writes recs into a fresh database and returns its path.
|
|
||||||
func seedDatabase(t *testing.T, recs []scanRec) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = db.Close()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return path
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportEscapesPaths(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
|
|
||||||
code := run([]string{cmdReport}, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
|
||||||
code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" +
|
|
||||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
|
||||||
if got := stdout(); got != want {
|
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestEscapePath(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
cases := map[string]string{
|
|
||||||
"/srv/plain": "/srv/plain",
|
|
||||||
"/a\tb": `/a\tb`,
|
|
||||||
"/a\nb": `/a\nb`,
|
|
||||||
"/a\rb": `/a\rb`,
|
|
||||||
`/a\b`: `/a\\b`,
|
|
||||||
`/a\tb`: `/a\\tb`,
|
|
||||||
"/not-utf8\xff": "/not-utf8\xff",
|
|
||||||
}
|
|
||||||
for in, want := range cases {
|
|
||||||
if got := escapePath(in); got != want {
|
|
||||||
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestWarnfEscapes checks that a warning naming a path that holds a
|
|
||||||
// newline is still one line.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
||||||
func TestWarnfEscapes(t *testing.T) {
|
|
||||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
saved := os.Stderr
|
|
||||||
os.Stderr = f
|
|
||||||
|
|
||||||
t.Cleanup(func() {
|
|
||||||
os.Stderr = saved
|
|
||||||
|
|
||||||
_ = f.Close()
|
|
||||||
})
|
|
||||||
|
|
||||||
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
|
||||||
|
|
||||||
_, err = f.Seek(0, io.SeekStart)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got, err := io.ReadAll(f)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := `stat /d/a\nb: gone` + "\n"
|
|
||||||
if string(got) != want {
|
|
||||||
t.Errorf("warning = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestCollectDupeGroups(t *testing.T) {
|
func TestCollectDupeGroups(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
@@ -211,8 +211,7 @@ type scanState struct {
|
|||||||
// size, head, and tail match another record's). Records outside the
|
// size, head, and tail match another record's). Records outside the
|
||||||
// roots are never touched, except that the content phase fills in
|
// roots are never touched, except that the content phase fills in
|
||||||
// their content hash. Operands the walk cannot start from are dropped
|
// their content hash. Operands the walk cannot start from are dropped
|
||||||
// first, so the records beneath them count as outside the roots unless
|
// first, so the records beneath them count as outside the roots.
|
||||||
// they lie under another root.
|
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) (scanStats, error) {
|
) (scanStats, error) {
|
||||||
@@ -259,9 +258,8 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
// files, and directories not named .zfs. Every other operand is warned
|
// files, and directories not named .zfs. Every other operand is warned
|
||||||
// about, counted as skipped, and dropped. A dropped operand is no
|
// about, counted as skipped, and dropped. A dropped operand is no
|
||||||
// longer a root, so the records stored beneath it are left as they are
|
// longer a root, so the records stored beneath it are left as they are
|
||||||
// instead of being deleted as unverified, unless it lies under another
|
// instead of being deleted as unverified. An operand that fails lstat
|
||||||
// root. An operand that fails lstat here is kept, and the walk warns
|
// here is kept, and the walk warns about it.
|
||||||
// about it.
|
|
||||||
func (s *scanState) walkableRoots(roots []string) []string {
|
func (s *scanState) walkableRoots(roots []string) []string {
|
||||||
kept := make([]string, 0, len(roots))
|
kept := make([]string, 0, len(roots))
|
||||||
|
|
||||||
@@ -272,7 +270,7 @@ func (s *scanState) walkableRoots(roots []string) []string {
|
|||||||
if warn != "" {
|
if warn != "" {
|
||||||
s.st.skipped++
|
s.st.skipped++
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, escapePath(warn))
|
fmt.Fprintln(os.Stderr, warn)
|
||||||
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -62,8 +62,7 @@ func runTrees(ctx context.Context) error {
|
|||||||
first := g[0]
|
first := g[0]
|
||||||
for _, n := range g[1:] {
|
for _, n := range g[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||||
escapePath(first.path), escapePath(n.path),
|
first.path, n.path, first.fileCount, first.totalSize)
|
||||||
first.fileCount, first.totalSize)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -88,8 +87,8 @@ func runTrees(ctx context.Context) error {
|
|||||||
|
|
||||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||||
// absolute paths the first component is empty, which becomes the
|
// absolute paths the first component is empty, which simply becomes a
|
||||||
// top-level node with path "/". It returns the super-root and every
|
// top-level node representing "/". It returns the super-root and every
|
||||||
// directory node created.
|
// directory node created.
|
||||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||||
super := &treeNode{}
|
super := &treeNode{}
|
||||||
@@ -103,17 +102,9 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
|||||||
for _, c := range comps[:len(comps)-1] {
|
for _, c := range comps[:len(comps)-1] {
|
||||||
child := node.dirs[c]
|
child := node.dirs[c]
|
||||||
if child == nil {
|
if child == nil {
|
||||||
childPath := node.path + "/" + c
|
childPath := c
|
||||||
|
if node != super {
|
||||||
// The root directory's path is "/", not empty, and its
|
childPath = node.path + "/" + c
|
||||||
// children's paths start with one slash, not two.
|
|
||||||
switch {
|
|
||||||
case node == super && c == "":
|
|
||||||
childPath = "/"
|
|
||||||
case node == super:
|
|
||||||
childPath = c
|
|
||||||
case node.path == "/":
|
|
||||||
childPath = "/" + c
|
|
||||||
}
|
}
|
||||||
|
|
||||||
child = &treeNode{path: childPath, parent: node}
|
child = &treeNode{path: childPath, parent: node}
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
@@ -85,46 +84,6 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestBuildHierarchyRootPath(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// The root directory's path is "/", never empty, and its
|
|
||||||
// children's paths start with a single slash.
|
|
||||||
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
|
|
||||||
|
|
||||||
got := make([]string, 0, len(dirs))
|
|
||||||
for _, d := range dirs {
|
|
||||||
got = append(got, d.path)
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(got)
|
|
||||||
|
|
||||||
want := []string{"/", "/srv"}
|
|
||||||
if !slices.Equal(got, want) {
|
|
||||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
|
|
||||||
code := run([]string{cmdTrees}, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
|
||||||
code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "first\tdupe\tfiles\tsize\n" +
|
|
||||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
|
||||||
if got := stdout(); got != want {
|
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestTreeDigests(t *testing.T) {
|
func TestTreeDigests(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user