Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f829542b3b | ||
|
|
c887f80f57 |
@@ -255,9 +255,11 @@ An operand that is a symlink (never followed, not even as an operand),
|
|||||||
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
||||||
scanned. `scan` prints a one-line warning naming the path and what it
|
scanned. `scan` prints a one-line warning naming the path and what it
|
||||||
is, counts it as skipped, and drops it from the scanned operands before
|
is, counts it as skipped, and drops it from the scanned operands before
|
||||||
reading the database, so the records stored beneath it are left as
|
reading the database. The records stored beneath it are left as they
|
||||||
they are. This is not an error: a scan whose every operand is dropped
|
are, unless it lies under another operand: then they are deleted like
|
||||||
walks nothing and exits 0.
|
any other record under that operand that this scan did not verify.
|
||||||
|
This is not an error: a scan whose every operand is dropped walks
|
||||||
|
nothing and exits 0.
|
||||||
|
|
||||||
`scan` synchronizes the database with the filesystem state under the
|
`scan` synchronizes the database with the filesystem state under the
|
||||||
scanned operands:
|
scanned operands:
|
||||||
@@ -380,8 +382,9 @@ Rules for the walk:
|
|||||||
path, and continue. Per-file errors never abort the run; the final
|
path, and continue. Per-file errors never abort the run; the final
|
||||||
summary reports how many were skipped. As specified above, a
|
summary reports how many were skipped. As specified above, a
|
||||||
skipped path that has a database record from an earlier scan loses
|
skipped path that has a database record from an earlier scan loses
|
||||||
that record, unless it failed only in the content phase or is an
|
that record, unless it failed only in the content phase, or is an
|
||||||
operand dropped before the database was read; an
|
operand dropped before the database was read that lies under no
|
||||||
|
other operand; an
|
||||||
unreadable directory subtree likewise loses its records (accepted:
|
unreadable directory subtree likewise loses its records (accepted:
|
||||||
the database mirrors what the latest scan could actually verify).
|
the database mirrors what the latest scan could actually verify).
|
||||||
|
|
||||||
@@ -440,6 +443,15 @@ first dupe size
|
|||||||
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Paths are raw bytes and may hold any byte except NUL, so the path
|
||||||
|
columns (`first` and `dupe`) are escaped to keep every row one line of
|
||||||
|
tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a
|
||||||
|
newline as `\n`, and a carriage return as `\r`. Every other byte is
|
||||||
|
written unchanged, including bytes that are not valid UTF-8. Undoing
|
||||||
|
those four escapes gives back the stored path. Grouping and ordering
|
||||||
|
use the stored path, not the escaped one. The warnings `scan` prints on
|
||||||
|
stderr are escaped the same way, so each warning is one line.
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate groups, number of
|
Summary to stderr: records read, number of duplicate groups, number of
|
||||||
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
||||||
rows) in human units.
|
rows) in human units.
|
||||||
@@ -511,6 +523,9 @@ first dupe files size
|
|||||||
/srv/a/project /srv/backup/project 3417 104857600
|
/srv/a/project /srv/backup/project 3417 104857600
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The `first` and `dupe` paths are escaped as described under "Report
|
||||||
|
output format". The root directory's path is `/`.
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate-tree groups,
|
Summary to stderr: records read, number of duplicate-tree groups,
|
||||||
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
||||||
all dupe rows) in human units.
|
all dupe rows) in human units.
|
||||||
|
|||||||
@@ -33,6 +33,10 @@
|
|||||||
operands, keeping the records beneath them (2026-10-03,
|
operands, keeping the records beneath them (2026-10-03,
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||||
|
|
||||||
|
- escape tabs, newlines, carriage returns and backslashes in report,
|
||||||
|
trees and warning paths; the root directory's path is `/`
|
||||||
|
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
|
||||||
|
|
||||||
- stamp the git tag or short commit in a plain `docker build .`
|
- stamp the git tag or short commit in a plain `docker build .`
|
||||||
instead of `dev` (2026-10-02, branch `next`, closes
|
instead of `dev` (2026-10-02, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
||||||
|
|||||||
+3
-1
@@ -106,6 +106,8 @@ func (p *progress) increment() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||||
|
// The whole message is escaped like a report's path columns, so a path
|
||||||
|
// holding a newline cannot split the warning.
|
||||||
func (p *progress) warnf(format string, args ...any) {
|
func (p *progress) warnf(format string, args ...any) {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
@@ -115,7 +117,7 @@ func (p *progress) warnf(format string, args ...any) {
|
|||||||
_ = p.bar.Clear()
|
_ = p.bar.Clear()
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
fmt.Fprintln(os.Stderr, escapePath(fmt.Sprintf(format, args...)))
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display.
|
// finish terminates the pass's display.
|
||||||
|
|||||||
@@ -87,7 +87,7 @@ func runReport(ctx context.Context) error {
|
|||||||
for _, g := range dupes {
|
for _, g := range dupes {
|
||||||
for _, p := range g.paths[1:] {
|
for _, p := range g.paths[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||||
g.paths[0], p, g.size)
|
escapePath(g.paths[0]), escapePath(p), g.size)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -156,6 +156,21 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
|||||||
return dupes
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// escapePath returns a path as it is written in a report column (README
|
||||||
|
// "Report output format"): a backslash, tab, newline or carriage return
|
||||||
|
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
||||||
|
// Grouping and sorting use the raw path, never this form.
|
||||||
|
func escapePath(p string) string {
|
||||||
|
// Most paths need no escaping; skip building a replacer for them.
|
||||||
|
if !strings.ContainsAny(p, "\\\t\n\r") {
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
return strings.NewReplacer(
|
||||||
|
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
||||||
|
).Replace(p)
|
||||||
|
}
|
||||||
|
|
||||||
// humanBytes formats a byte count in human units (binary prefixes).
|
// humanBytes formats a byte count in human units (binary prefixes).
|
||||||
func humanBytes(n int64) string {
|
func humanBytes(n int64) string {
|
||||||
const unit = 1024
|
const unit = 1024
|
||||||
|
|||||||
+118
@@ -1,10 +1,128 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// awkwardDir is a directory name holding every byte the reports escape.
|
||||||
|
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
||||||
|
|
||||||
|
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
||||||
|
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
||||||
|
// sorts after it, so awkwardDir coming first shows that sorting uses
|
||||||
|
// the raw path.
|
||||||
|
func awkwardPairRecs() []scanRec {
|
||||||
|
return []scanRec{
|
||||||
|
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
||||||
|
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// seedDatabase writes recs into a fresh database and returns its path.
|
||||||
|
func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
path := testDBPath(t)
|
||||||
|
|
||||||
|
db, err := openScanDatabase(t.Context(), path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = applyChanges(t.Context(), db, recs, nil, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = db.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return path
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunReportEscapesPaths(t *testing.T) {
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||||
|
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{cmdReport}, &stderr)
|
||||||
|
if code != exitOK {
|
||||||
|
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||||
|
code, exitOK, stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
want := "first\tdupe\tsize\n" +
|
||||||
|
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
||||||
|
if got := stdout(); got != want {
|
||||||
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEscapePath(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
cases := map[string]string{
|
||||||
|
"/srv/plain": "/srv/plain",
|
||||||
|
"/a\tb": `/a\tb`,
|
||||||
|
"/a\nb": `/a\nb`,
|
||||||
|
"/a\rb": `/a\rb`,
|
||||||
|
`/a\b`: `/a\\b`,
|
||||||
|
`/a\tb`: `/a\\tb`,
|
||||||
|
"/not-utf8\xff": "/not-utf8\xff",
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
if got := escapePath(in); got != want {
|
||||||
|
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestWarnfEscapes checks that a warning naming a path that holds a
|
||||||
|
// newline is still one line.
|
||||||
|
//
|
||||||
|
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||||
|
func TestWarnfEscapes(t *testing.T) {
|
||||||
|
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
saved := os.Stderr
|
||||||
|
os.Stderr = f
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
os.Stderr = saved
|
||||||
|
|
||||||
|
_ = f.Close()
|
||||||
|
})
|
||||||
|
|
||||||
|
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
||||||
|
|
||||||
|
_, err = f.Seek(0, io.SeekStart)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
got, err := io.ReadAll(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := `stat /d/a\nb: gone` + "\n"
|
||||||
|
if string(got) != want {
|
||||||
|
t.Errorf("warning = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroups(t *testing.T) {
|
func TestCollectDupeGroups(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
@@ -211,7 +211,8 @@ type scanState struct {
|
|||||||
// size, head, and tail match another record's). Records outside the
|
// size, head, and tail match another record's). Records outside the
|
||||||
// roots are never touched, except that the content phase fills in
|
// roots are never touched, except that the content phase fills in
|
||||||
// their content hash. Operands the walk cannot start from are dropped
|
// their content hash. Operands the walk cannot start from are dropped
|
||||||
// first, so the records beneath them count as outside the roots.
|
// first, so the records beneath them count as outside the roots unless
|
||||||
|
// they lie under another root.
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) (scanStats, error) {
|
) (scanStats, error) {
|
||||||
@@ -258,8 +259,9 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
// files, and directories not named .zfs. Every other operand is warned
|
// files, and directories not named .zfs. Every other operand is warned
|
||||||
// about, counted as skipped, and dropped. A dropped operand is no
|
// about, counted as skipped, and dropped. A dropped operand is no
|
||||||
// longer a root, so the records stored beneath it are left as they are
|
// longer a root, so the records stored beneath it are left as they are
|
||||||
// instead of being deleted as unverified. An operand that fails lstat
|
// instead of being deleted as unverified, unless it lies under another
|
||||||
// here is kept, and the walk warns about it.
|
// root. An operand that fails lstat here is kept, and the walk warns
|
||||||
|
// about it.
|
||||||
func (s *scanState) walkableRoots(roots []string) []string {
|
func (s *scanState) walkableRoots(roots []string) []string {
|
||||||
kept := make([]string, 0, len(roots))
|
kept := make([]string, 0, len(roots))
|
||||||
|
|
||||||
@@ -270,7 +272,7 @@ func (s *scanState) walkableRoots(roots []string) []string {
|
|||||||
if warn != "" {
|
if warn != "" {
|
||||||
s.st.skipped++
|
s.st.skipped++
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, warn)
|
fmt.Fprintln(os.Stderr, escapePath(warn))
|
||||||
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -62,7 +62,8 @@ func runTrees(ctx context.Context) error {
|
|||||||
first := g[0]
|
first := g[0]
|
||||||
for _, n := range g[1:] {
|
for _, n := range g[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||||
first.path, n.path, first.fileCount, first.totalSize)
|
escapePath(first.path), escapePath(n.path),
|
||||||
|
first.fileCount, first.totalSize)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -87,8 +88,8 @@ func runTrees(ctx context.Context) error {
|
|||||||
|
|
||||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||||
// absolute paths the first component is empty, which simply becomes a
|
// absolute paths the first component is empty, which becomes the
|
||||||
// top-level node representing "/". It returns the super-root and every
|
// top-level node with path "/". It returns the super-root and every
|
||||||
// directory node created.
|
// directory node created.
|
||||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||||
super := &treeNode{}
|
super := &treeNode{}
|
||||||
@@ -102,9 +103,17 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
|||||||
for _, c := range comps[:len(comps)-1] {
|
for _, c := range comps[:len(comps)-1] {
|
||||||
child := node.dirs[c]
|
child := node.dirs[c]
|
||||||
if child == nil {
|
if child == nil {
|
||||||
childPath := c
|
childPath := node.path + "/" + c
|
||||||
if node != super {
|
|
||||||
childPath = node.path + "/" + c
|
// The root directory's path is "/", not empty, and its
|
||||||
|
// children's paths start with one slash, not two.
|
||||||
|
switch {
|
||||||
|
case node == super && c == "":
|
||||||
|
childPath = "/"
|
||||||
|
case node == super:
|
||||||
|
childPath = c
|
||||||
|
case node.path == "/":
|
||||||
|
childPath = "/" + c
|
||||||
}
|
}
|
||||||
|
|
||||||
child = &treeNode{path: childPath, parent: node}
|
child = &treeNode{path: childPath, parent: node}
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
@@ -84,6 +85,46 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestBuildHierarchyRootPath(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
// The root directory's path is "/", never empty, and its
|
||||||
|
// children's paths start with a single slash.
|
||||||
|
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
|
||||||
|
|
||||||
|
got := make([]string, 0, len(dirs))
|
||||||
|
for _, d := range dirs {
|
||||||
|
got = append(got, d.path)
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(got)
|
||||||
|
|
||||||
|
want := []string{"/", "/srv"}
|
||||||
|
if !slices.Equal(got, want) {
|
||||||
|
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||||
|
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{cmdTrees}, &stderr)
|
||||||
|
if code != exitOK {
|
||||||
|
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||||
|
code, exitOK, stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
want := "first\tdupe\tfiles\tsize\n" +
|
||||||
|
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
||||||
|
if got := stdout(); got != want {
|
||||||
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestTreeDigests(t *testing.T) {
|
func TestTreeDigests(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user