Compare commits
3
Commits
main
...
f829542b3b
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f829542b3b | ||
|
|
c887f80f57 | ||
|
|
c9bf22d483 |
+6
-1
@@ -1,4 +1,9 @@
|
|||||||
.git
|
# .git is sent without its config. Without a VERSION build argument the
|
||||||
|
# stage that compiles runs `git describe --tags --always` on .git, which
|
||||||
|
# does not need .git/config; that file can hold a credential, such as a
|
||||||
|
# password in a remote URL or the token the CI checkout step stores there.
|
||||||
|
.git/config
|
||||||
|
|
||||||
.claude
|
.claude
|
||||||
.DS_Store
|
.DS_Store
|
||||||
sfdupes
|
sfdupes
|
||||||
|
|||||||
@@ -6,4 +6,6 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
# actions/checkout v4.2.2, 2026-02-22
|
# actions/checkout v4.2.2, 2026-02-22
|
||||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
- run: script/cibuild
|
- run: script/cibuild
|
||||||
|
|||||||
+13
-1
@@ -108,7 +108,19 @@ ARG CHECK_EPOCH
|
|||||||
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||||
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||||
|
|
||||||
RUN make build
|
# The version stamped into the binary: the VERSION build argument when
|
||||||
|
# one is given, otherwise `git describe --tags --always` of the .git in
|
||||||
|
# the build context (git is installed by script/bootstrap above). A
|
||||||
|
# context that carries .git and still yields no version fails the build;
|
||||||
|
# with neither, as from a source tarball, it is "dev".
|
||||||
|
ARG VERSION
|
||||||
|
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
||||||
|
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
||||||
|
[ "$version" = unknown ]; }; then \
|
||||||
|
echo "no version could be derived although the build context carries .git" >&2; \
|
||||||
|
exit 1; \
|
||||||
|
fi; \
|
||||||
|
make build VERSION="$version"
|
||||||
|
|
||||||
# Runtime stage
|
# Runtime stage
|
||||||
# alpine:3.22, 2026-07-23
|
# alpine:3.22, 2026-07-23
|
||||||
|
|||||||
@@ -251,6 +251,16 @@ duplicates another or lies under another is dropped before walking,
|
|||||||
so every file is reached exactly once and produces one database
|
so every file is reached exactly once and produces one database
|
||||||
record.
|
record.
|
||||||
|
|
||||||
|
An operand that is a symlink (never followed, not even as an operand),
|
||||||
|
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
||||||
|
scanned. `scan` prints a one-line warning naming the path and what it
|
||||||
|
is, counts it as skipped, and drops it from the scanned operands before
|
||||||
|
reading the database. The records stored beneath it are left as they
|
||||||
|
are, unless it lies under another operand: then they are deleted like
|
||||||
|
any other record under that operand that this scan did not verify.
|
||||||
|
This is not an error: a scan whose every operand is dropped walks
|
||||||
|
nothing and exits 0.
|
||||||
|
|
||||||
`scan` synchronizes the database with the filesystem state under the
|
`scan` synchronizes the database with the filesystem state under the
|
||||||
scanned operands:
|
scanned operands:
|
||||||
|
|
||||||
@@ -356,9 +366,12 @@ during the hash phase:
|
|||||||
Rules for the walk:
|
Rules for the walk:
|
||||||
|
|
||||||
- Only regular files. Skip directories, symlinks (do not follow,
|
- Only regular files. Skip directories, symlinks (do not follow,
|
||||||
including symlink operands), sockets, FIFOs, and device nodes.
|
including symlink operands), sockets, FIFOs, and device nodes. An
|
||||||
|
operand that is a symlink, socket, FIFO, or device node is dropped
|
||||||
|
as described in "`scan` mode" above.
|
||||||
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
|
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
|
||||||
walking them would list every file once per snapshot).
|
walking them would list every file once per snapshot), not even
|
||||||
|
when it is an operand; such an operand is dropped the same way.
|
||||||
- Filesystem boundaries are crossed by default. With `-x`
|
- Filesystem boundaries are crossed by default. With `-x`
|
||||||
(long form `--one-file-system`, following the GNU `du`/`rsync`
|
(long form `--one-file-system`, following the GNU `du`/`rsync`
|
||||||
convention), never descend into a directory on a different
|
convention), never descend into a directory on a different
|
||||||
@@ -369,7 +382,9 @@ Rules for the walk:
|
|||||||
path, and continue. Per-file errors never abort the run; the final
|
path, and continue. Per-file errors never abort the run; the final
|
||||||
summary reports how many were skipped. As specified above, a
|
summary reports how many were skipped. As specified above, a
|
||||||
skipped path that has a database record from an earlier scan loses
|
skipped path that has a database record from an earlier scan loses
|
||||||
that record, unless it failed only in the content phase; an
|
that record, unless it failed only in the content phase, or is an
|
||||||
|
operand dropped before the database was read that lies under no
|
||||||
|
other operand; an
|
||||||
unreadable directory subtree likewise loses its records (accepted:
|
unreadable directory subtree likewise loses its records (accepted:
|
||||||
the database mirrors what the latest scan could actually verify).
|
the database mirrors what the latest scan could actually verify).
|
||||||
|
|
||||||
@@ -428,6 +443,15 @@ first dupe size
|
|||||||
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Paths are raw bytes and may hold any byte except NUL, so the path
|
||||||
|
columns (`first` and `dupe`) are escaped to keep every row one line of
|
||||||
|
tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a
|
||||||
|
newline as `\n`, and a carriage return as `\r`. Every other byte is
|
||||||
|
written unchanged, including bytes that are not valid UTF-8. Undoing
|
||||||
|
those four escapes gives back the stored path. Grouping and ordering
|
||||||
|
use the stored path, not the escaped one. The warnings `scan` prints on
|
||||||
|
stderr are escaped the same way, so each warning is one line.
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate groups, number of
|
Summary to stderr: records read, number of duplicate groups, number of
|
||||||
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
||||||
rows) in human units.
|
rows) in human units.
|
||||||
@@ -499,6 +523,9 @@ first dupe files size
|
|||||||
/srv/a/project /srv/backup/project 3417 104857600
|
/srv/a/project /srv/backup/project 3417 104857600
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The `first` and `dupe` paths are escaped as described under "Report
|
||||||
|
output format". The root directory's path is `/`.
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate-tree groups,
|
Summary to stderr: records read, number of duplicate-tree groups,
|
||||||
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
||||||
all dupe rows) in human units.
|
all dupe rows) in human units.
|
||||||
|
|||||||
@@ -29,6 +29,25 @@
|
|||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
|
- warn about and skip symlink, socket, FIFO, device and `.zfs`
|
||||||
|
operands, keeping the records beneath them (2026-10-03,
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||||
|
|
||||||
|
- escape tabs, newlines, carriage returns and backslashes in report,
|
||||||
|
trees and warning paths; the root directory's path is `/`
|
||||||
|
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
|
||||||
|
|
||||||
|
- stamp the git tag or short commit in a plain `docker build .`
|
||||||
|
instead of `dev` (2026-10-02, branch `next`, closes
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
||||||
|
sends `.git`, without `.git/config`, and the `Dockerfile` build
|
||||||
|
stage takes the `VERSION` build argument when one is given,
|
||||||
|
otherwise `git describe --tags --always` of that `.git`. The build
|
||||||
|
fails if the context carries `.git` and the version still comes out
|
||||||
|
empty, `dev` or `unknown`. The CI checkout step fetches the full
|
||||||
|
history (`fetch-depth: 0`) so CI sees the tag and stamps the same
|
||||||
|
value as `make build`.
|
||||||
|
|
||||||
- replace the 1 KiB end-window sampling with the head/tail plus
|
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||||
content-hash ladder (2026-09-22, branch `next`, closes
|
content-hash ladder (2026-09-22, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||||
|
|||||||
+96
-9
@@ -51,16 +51,25 @@ func assertNoSidecars(t *testing.T, path string) {
|
|||||||
func captureStdout(t *testing.T) func() string {
|
func captureStdout(t *testing.T) func() string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
f, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
return captureStream(t, &os.Stdout)
|
||||||
|
}
|
||||||
|
|
||||||
|
// captureStream redirects *stream (os.Stdout or os.Stderr) to a file
|
||||||
|
// for the rest of the test and returns a function reading back
|
||||||
|
// everything written to it.
|
||||||
|
func captureStream(t *testing.T, stream **os.File) func() string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
f, err := os.Create(filepath.Join(t.TempDir(), "capture"))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
saved := os.Stdout
|
saved := *stream
|
||||||
os.Stdout = f
|
*stream = f
|
||||||
|
|
||||||
t.Cleanup(func() {
|
t.Cleanup(func() {
|
||||||
os.Stdout = saved
|
*stream = saved
|
||||||
|
|
||||||
_ = f.Close()
|
_ = f.Close()
|
||||||
})
|
})
|
||||||
@@ -320,21 +329,30 @@ func scanFixture(t *testing.T) []string {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
scanOK(t, dir)
|
||||||
|
|
||||||
|
return dupes
|
||||||
|
}
|
||||||
|
|
||||||
|
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
||||||
|
// nothing on stdout, and returns everything it printed to stderr.
|
||||||
|
func scanOK(t *testing.T, operands ...string) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
stdout := captureStdout(t)
|
||||||
|
stderr := captureStream(t, &os.Stderr)
|
||||||
|
|
||||||
code := run([]string{cmdScan, dir}, &stderr)
|
code := run(append([]string{cmdScan}, operands...), os.Stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
||||||
code, exitOK, stderr.String())
|
operands, code, exitOK, stderr())
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stdout(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
return dupes
|
return stderr()
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||||
@@ -345,6 +363,75 @@ func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
|||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
||||||
|
path := testDBPath(t)
|
||||||
|
t.Setenv(databaseEnv, path)
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||||
|
|
||||||
|
link := filepath.Join(dir, "link")
|
||||||
|
|
||||||
|
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Scanning a directory through the symlink stores a record beneath
|
||||||
|
// the symlink's own path for a file beneath its target.
|
||||||
|
scanOK(t, filepath.Join(link, "sub"))
|
||||||
|
|
||||||
|
assertOperandSkipped(t, path, link, "symlink",
|
||||||
|
filepath.Join(link, "sub", "f"))
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
||||||
|
path := testDBPath(t)
|
||||||
|
t.Setenv(databaseEnv, path)
|
||||||
|
|
||||||
|
zfs := filepath.Join(t.TempDir(), ".zfs")
|
||||||
|
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
||||||
|
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
||||||
|
|
||||||
|
// An operand beneath a .zfs directory is walked, because it is not
|
||||||
|
// itself named .zfs.
|
||||||
|
scanOK(t, snapshot)
|
||||||
|
|
||||||
|
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// assertOperandSkipped scans operand alone and checks that it is skipped
|
||||||
|
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
||||||
|
// record for kept, which an earlier scan stored beneath operand, still
|
||||||
|
// in the database at dbPath.
|
||||||
|
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
||||||
|
kept string,
|
||||||
|
) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
stderr := scanOK(t, operand)
|
||||||
|
|
||||||
|
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
||||||
|
if !strings.Contains(stderr, warning) {
|
||||||
|
t.Errorf("stderr = %q, want %q", stderr, warning)
|
||||||
|
}
|
||||||
|
|
||||||
|
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
||||||
|
"0 unchanged), 1 skipped\n"
|
||||||
|
if !strings.Contains(stderr, summary) {
|
||||||
|
t.Errorf("stderr = %q, want %q", stderr, summary)
|
||||||
|
}
|
||||||
|
|
||||||
|
db, err := openDB(dbPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Cleanup(func() { _ = db.Close() })
|
||||||
|
|
||||||
|
recordByPath(t, dbRecords(t, db), kept)
|
||||||
|
}
|
||||||
|
|
||||||
func TestRunReportSucceeds(t *testing.T) {
|
func TestRunReportSucceeds(t *testing.T) {
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
t.Setenv(databaseEnv, path)
|
t.Setenv(databaseEnv, path)
|
||||||
|
|||||||
+3
-1
@@ -106,6 +106,8 @@ func (p *progress) increment() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||||
|
// The whole message is escaped like a report's path columns, so a path
|
||||||
|
// holding a newline cannot split the warning.
|
||||||
func (p *progress) warnf(format string, args ...any) {
|
func (p *progress) warnf(format string, args ...any) {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
@@ -115,7 +117,7 @@ func (p *progress) warnf(format string, args ...any) {
|
|||||||
_ = p.bar.Clear()
|
_ = p.bar.Clear()
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
fmt.Fprintln(os.Stderr, escapePath(fmt.Sprintf(format, args...)))
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display.
|
// finish terminates the pass's display.
|
||||||
|
|||||||
@@ -87,7 +87,7 @@ func runReport(ctx context.Context) error {
|
|||||||
for _, g := range dupes {
|
for _, g := range dupes {
|
||||||
for _, p := range g.paths[1:] {
|
for _, p := range g.paths[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||||
g.paths[0], p, g.size)
|
escapePath(g.paths[0]), escapePath(p), g.size)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -156,6 +156,21 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
|||||||
return dupes
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// escapePath returns a path as it is written in a report column (README
|
||||||
|
// "Report output format"): a backslash, tab, newline or carriage return
|
||||||
|
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
||||||
|
// Grouping and sorting use the raw path, never this form.
|
||||||
|
func escapePath(p string) string {
|
||||||
|
// Most paths need no escaping; skip building a replacer for them.
|
||||||
|
if !strings.ContainsAny(p, "\\\t\n\r") {
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
return strings.NewReplacer(
|
||||||
|
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
||||||
|
).Replace(p)
|
||||||
|
}
|
||||||
|
|
||||||
// humanBytes formats a byte count in human units (binary prefixes).
|
// humanBytes formats a byte count in human units (binary prefixes).
|
||||||
func humanBytes(n int64) string {
|
func humanBytes(n int64) string {
|
||||||
const unit = 1024
|
const unit = 1024
|
||||||
|
|||||||
+118
@@ -1,10 +1,128 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// awkwardDir is a directory name holding every byte the reports escape.
|
||||||
|
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
||||||
|
|
||||||
|
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
||||||
|
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
||||||
|
// sorts after it, so awkwardDir coming first shows that sorting uses
|
||||||
|
// the raw path.
|
||||||
|
func awkwardPairRecs() []scanRec {
|
||||||
|
return []scanRec{
|
||||||
|
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
||||||
|
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// seedDatabase writes recs into a fresh database and returns its path.
|
||||||
|
func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
path := testDBPath(t)
|
||||||
|
|
||||||
|
db, err := openScanDatabase(t.Context(), path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = applyChanges(t.Context(), db, recs, nil, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = db.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return path
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunReportEscapesPaths(t *testing.T) {
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||||
|
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{cmdReport}, &stderr)
|
||||||
|
if code != exitOK {
|
||||||
|
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||||
|
code, exitOK, stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
want := "first\tdupe\tsize\n" +
|
||||||
|
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
||||||
|
if got := stdout(); got != want {
|
||||||
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEscapePath(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
cases := map[string]string{
|
||||||
|
"/srv/plain": "/srv/plain",
|
||||||
|
"/a\tb": `/a\tb`,
|
||||||
|
"/a\nb": `/a\nb`,
|
||||||
|
"/a\rb": `/a\rb`,
|
||||||
|
`/a\b`: `/a\\b`,
|
||||||
|
`/a\tb`: `/a\\tb`,
|
||||||
|
"/not-utf8\xff": "/not-utf8\xff",
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
if got := escapePath(in); got != want {
|
||||||
|
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestWarnfEscapes checks that a warning naming a path that holds a
|
||||||
|
// newline is still one line.
|
||||||
|
//
|
||||||
|
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||||
|
func TestWarnfEscapes(t *testing.T) {
|
||||||
|
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
saved := os.Stderr
|
||||||
|
os.Stderr = f
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
os.Stderr = saved
|
||||||
|
|
||||||
|
_ = f.Close()
|
||||||
|
})
|
||||||
|
|
||||||
|
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
||||||
|
|
||||||
|
_, err = f.Seek(0, io.SeekStart)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
got, err := io.ReadAll(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := `stat /d/a\nb: gone` + "\n"
|
||||||
|
if string(got) != want {
|
||||||
|
t.Errorf("warning = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroups(t *testing.T) {
|
func TestCollectDupeGroups(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
@@ -210,14 +210,16 @@ type scanState struct {
|
|||||||
// in the content hash of every record of headTailMin or more whose
|
// in the content hash of every record of headTailMin or more whose
|
||||||
// size, head, and tail match another record's). Records outside the
|
// size, head, and tail match another record's). Records outside the
|
||||||
// roots are never touched, except that the content phase fills in
|
// roots are never touched, except that the content phase fills in
|
||||||
// their content hash.
|
// their content hash. Operands the walk cannot start from are dropped
|
||||||
|
// first, so the records beneath them count as outside the roots unless
|
||||||
|
// they lie under another root.
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) (scanStats, error) {
|
) (scanStats, error) {
|
||||||
roots = pruneRoots(roots)
|
|
||||||
|
|
||||||
s := &scanState{db: db}
|
s := &scanState{db: db}
|
||||||
|
|
||||||
|
roots = pruneRoots(s.walkableRoots(roots))
|
||||||
|
|
||||||
err := s.loadIndex(ctx, roots)
|
err := s.loadIndex(ctx, roots)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return s.st, err
|
return s.st, err
|
||||||
@@ -253,6 +255,35 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
return s.st, s.contentPhase(ctx, workers)
|
return s.st, s.contentPhase(ctx, workers)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// walkableRoots returns the operands the walk can start from: regular
|
||||||
|
// files, and directories not named .zfs. Every other operand is warned
|
||||||
|
// about, counted as skipped, and dropped. A dropped operand is no
|
||||||
|
// longer a root, so the records stored beneath it are left as they are
|
||||||
|
// instead of being deleted as unverified, unless it lies under another
|
||||||
|
// root. An operand that fails lstat here is kept, and the walk warns
|
||||||
|
// about it.
|
||||||
|
func (s *scanState) walkableRoots(roots []string) []string {
|
||||||
|
kept := make([]string, 0, len(roots))
|
||||||
|
|
||||||
|
for _, root := range roots {
|
||||||
|
fi, err := os.Lstat(root)
|
||||||
|
if err == nil {
|
||||||
|
warn := operandWarning(root, fi)
|
||||||
|
if warn != "" {
|
||||||
|
s.st.skipped++
|
||||||
|
|
||||||
|
fmt.Fprintln(os.Stderr, escapePath(warn))
|
||||||
|
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
kept = append(kept, root)
|
||||||
|
}
|
||||||
|
|
||||||
|
return kept
|
||||||
|
}
|
||||||
|
|
||||||
// loadIndex indexes the database records under the scan roots for
|
// loadIndex indexes the database records under the scan roots for
|
||||||
// change detection and collects the sizes of every record outside
|
// change detection and collects the sizes of every record outside
|
||||||
// them: out-of-scope records join the size census so a scanned file
|
// them: out-of-scope records join the size census so a scanned file
|
||||||
@@ -789,11 +820,43 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// operandWarning returns the one-line warning for an operand the walk
|
||||||
|
// does not start from, naming the path and what it is, or "" for one it
|
||||||
|
// does: a regular file, or a directory not named .zfs. Symlinks are
|
||||||
|
// never followed, including as operands.
|
||||||
|
func operandWarning(root string, fi fs.FileInfo) string {
|
||||||
|
var kind string
|
||||||
|
|
||||||
|
switch mode := fi.Mode(); {
|
||||||
|
case mode.IsRegular():
|
||||||
|
return ""
|
||||||
|
case mode.IsDir():
|
||||||
|
if filepath.Base(root) != ".zfs" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
kind = ".zfs directory"
|
||||||
|
case mode&fs.ModeSymlink != 0:
|
||||||
|
kind = "symlink"
|
||||||
|
case mode&fs.ModeSocket != 0:
|
||||||
|
kind = "socket"
|
||||||
|
case mode&fs.ModeNamedPipe != 0:
|
||||||
|
kind = "FIFO"
|
||||||
|
case mode&fs.ModeDevice != 0:
|
||||||
|
kind = "device node"
|
||||||
|
default:
|
||||||
|
kind = "non-regular file"
|
||||||
|
}
|
||||||
|
|
||||||
|
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
|
||||||
|
}
|
||||||
|
|
||||||
// seedRoot turns one PATH operand into the walk's starting state: a
|
// seedRoot turns one PATH operand into the walk's starting state: a
|
||||||
// regular-file operand is statted and emitted directly, a directory
|
// regular-file operand is statted and emitted directly, and a directory
|
||||||
// operand becomes an initial job, and a symlink or other non-regular
|
// operand becomes an initial job. walkableRoots has already dropped
|
||||||
// operand yields nothing (symlinks are never followed, including as
|
// every other operand. One that has changed into something else since
|
||||||
// operands).
|
// is warned about and skipped here; it is still a root, so the records
|
||||||
|
// stored beneath it are deleted as unverified.
|
||||||
func seedRoot(ctx context.Context, root string,
|
func seedRoot(ctx context.Context, root string,
|
||||||
events chan<- walkEvent,
|
events chan<- walkEvent,
|
||||||
) []dirJob {
|
) []dirJob {
|
||||||
@@ -807,30 +870,30 @@ func seedRoot(ctx context.Context, root string,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
switch {
|
warn := operandWarning(root, fi)
|
||||||
case fi.IsDir():
|
if warn != "" {
|
||||||
if filepath.Base(root) == ".zfs" {
|
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if fi.IsDir() {
|
||||||
dev, ok := deviceOfInfo(fi)
|
dev, ok := deviceOfInfo(fi)
|
||||||
|
|
||||||
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
||||||
case fi.Mode().IsRegular():
|
|
||||||
dev, ino := inodeOfInfo(fi)
|
|
||||||
|
|
||||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
|
||||||
path: root,
|
|
||||||
size: fi.Size(),
|
|
||||||
mtime: fi.ModTime().Unix(),
|
|
||||||
dev: dev,
|
|
||||||
ino: ino,
|
|
||||||
}})
|
|
||||||
|
|
||||||
return nil
|
|
||||||
default:
|
|
||||||
return nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
dev, ino := inodeOfInfo(fi)
|
||||||
|
|
||||||
|
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||||
|
path: root,
|
||||||
|
size: fi.Size(),
|
||||||
|
mtime: fi.ModTime().Unix(),
|
||||||
|
dev: dev,
|
||||||
|
ino: ino,
|
||||||
|
}})
|
||||||
|
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// startWalkWorkers starts the walk worker pool. Each worker processes
|
// startWalkWorkers starts the walk worker pool. Each worker processes
|
||||||
|
|||||||
+4
-2
@@ -856,9 +856,11 @@ func TestWalkFileAndSymlinkOperands(t *testing.T) {
|
|||||||
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
||||||
}
|
}
|
||||||
|
|
||||||
// A symlink operand is not followed and yields nothing.
|
// A symlink operand that reaches the walk (it became one after
|
||||||
|
// walkableRoots checked it) is not followed: it yields a warning and
|
||||||
|
// no records.
|
||||||
recs, errs = collectWalk(t, []string{link}, false, 2)
|
recs, errs = collectWalk(t, []string{link}, false, 2)
|
||||||
if errs != 0 || len(recs) != 0 {
|
if errs != 1 || len(recs) != 0 {
|
||||||
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+4
-4
@@ -16,10 +16,10 @@
|
|||||||
# implies the repo is green.
|
# implies the repo is green.
|
||||||
#
|
#
|
||||||
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||||
# invalidated by changed content, and a merge commit's tree is
|
# invalidated only by changed content, and a rebuild of an unchanged
|
||||||
# byte-identical to the branch head it merges, so without a fresh value
|
# checkout sends the same content, so without a fresh value here Docker
|
||||||
# here Docker serves the gate layers from cache and the build reports a
|
# serves the gate layers from cache and the build reports a green it
|
||||||
# green it never earned. Passing the current epoch invalidates the gate
|
# never earned. Passing the current epoch invalidates the gate
|
||||||
# layers on every run while leaving the pinned base images and
|
# layers on every run while leaving the pinned base images and
|
||||||
# go mod download cached; see the Dockerfile for the placement.
|
# go mod download cached; see the Dockerfile for the placement.
|
||||||
set -eu
|
set -eu
|
||||||
|
|||||||
@@ -62,7 +62,8 @@ func runTrees(ctx context.Context) error {
|
|||||||
first := g[0]
|
first := g[0]
|
||||||
for _, n := range g[1:] {
|
for _, n := range g[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||||
first.path, n.path, first.fileCount, first.totalSize)
|
escapePath(first.path), escapePath(n.path),
|
||||||
|
first.fileCount, first.totalSize)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -87,8 +88,8 @@ func runTrees(ctx context.Context) error {
|
|||||||
|
|
||||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||||
// absolute paths the first component is empty, which simply becomes a
|
// absolute paths the first component is empty, which becomes the
|
||||||
// top-level node representing "/". It returns the super-root and every
|
// top-level node with path "/". It returns the super-root and every
|
||||||
// directory node created.
|
// directory node created.
|
||||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||||
super := &treeNode{}
|
super := &treeNode{}
|
||||||
@@ -102,9 +103,17 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
|||||||
for _, c := range comps[:len(comps)-1] {
|
for _, c := range comps[:len(comps)-1] {
|
||||||
child := node.dirs[c]
|
child := node.dirs[c]
|
||||||
if child == nil {
|
if child == nil {
|
||||||
childPath := c
|
childPath := node.path + "/" + c
|
||||||
if node != super {
|
|
||||||
childPath = node.path + "/" + c
|
// The root directory's path is "/", not empty, and its
|
||||||
|
// children's paths start with one slash, not two.
|
||||||
|
switch {
|
||||||
|
case node == super && c == "":
|
||||||
|
childPath = "/"
|
||||||
|
case node == super:
|
||||||
|
childPath = c
|
||||||
|
case node.path == "/":
|
||||||
|
childPath = "/" + c
|
||||||
}
|
}
|
||||||
|
|
||||||
child = &treeNode{path: childPath, parent: node}
|
child = &treeNode{path: childPath, parent: node}
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
@@ -84,6 +85,46 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestBuildHierarchyRootPath(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
// The root directory's path is "/", never empty, and its
|
||||||
|
// children's paths start with a single slash.
|
||||||
|
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
|
||||||
|
|
||||||
|
got := make([]string, 0, len(dirs))
|
||||||
|
for _, d := range dirs {
|
||||||
|
got = append(got, d.path)
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(got)
|
||||||
|
|
||||||
|
want := []string{"/", "/srv"}
|
||||||
|
if !slices.Equal(got, want) {
|
||||||
|
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||||
|
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{cmdTrees}, &stderr)
|
||||||
|
if code != exitOK {
|
||||||
|
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||||
|
code, exitOK, stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
want := "first\tdupe\tfiles\tsize\n" +
|
||||||
|
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
||||||
|
if got := stdout(); got != want {
|
||||||
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestTreeDigests(t *testing.T) {
|
func TestTreeDigests(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user