Author SHA1 Message Date
sneak c2804d088b Merge pull request 'next' (#50) from next into main
check / check (push) Successful in 1m1s
Reviewed-on: #50
2026-09-26 15:27:29 +02:00
16 changed files with 90 additions and 687 deletions
+1 -6
View File
@@ -1,9 +1,4 @@
# .git is sent without its config. Without a VERSION build argument the
# stage that compiles runs `git describe --tags --always` on .git, which
# does not need .git/config; that file can hold a credential, such as a
# password in a remote URL or the token the CI checkout step stores there.
.git/config
.git
.claude
.DS_Store
sfdupes
-2
View File
@@ -6,6 +6,4 @@ jobs:
steps:
# actions/checkout v4.2.2, 2026-02-22
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
with:
fetch-depth: 0
- run: script/cibuild
+1 -13
View File
@@ -108,19 +108,7 @@ ARG CHECK_EPOCH
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
# The version stamped into the binary: the VERSION build argument when
# one is given, otherwise `git describe --tags --always` of the .git in
# the build context (git is installed by script/bootstrap above). A
# context that carries .git and still yields no version fails the build;
# with neither, as from a source tarball, it is "dev".
ARG VERSION
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
[ "$version" = unknown ]; }; then \
echo "no version could be derived although the build context carries .git" >&2; \
exit 1; \
fi; \
make build VERSION="$version"
RUN make build
# Runtime stage
# alpine:3.22, 2026-07-23
+15 -56
View File
@@ -152,28 +152,16 @@ All three subcommands operate on a single SQLite database file:
use. `report` and `trees` require an existing database; a missing
database file is a fatal error (exit 1) telling the user to run
`scan` first.
- While `scan` runs, the database is in WAL journal mode with a busy
timeout, so running a report while a cron `scan` is in progress is
safe. The filesystem is authoritative; the database is an
eventually-consistent reflection of it. Hashed records are
committed in batched transactions while the scan is still running
(keeping the WAL small and letting concurrent reports observe
progress), so a report may see a scan's changes partially applied,
and a scan that dies partway leaves a valid database holding
everything hashed so far; the next scan skips those records and
converges toward the filesystem.
- `scan` switches the database back to rollback-journal mode when it
closes it, so between scans the database file alone holds the whole
database. Each switch needs the database to itself: a `scan` that
starts while a report is still reading waits for it up to the
10-second busy timeout, then fails; a `scan` that ends while a
report has the database open warns and leaves the database in WAL
mode until the next scan.
- `report` and `trees` open the database read-only and need only read
access to the database file, and no write access to its directory.
While the database is in WAL mode they also read the `-wal` and
`-shm` files beside it, which SQLite creates with the database
file's permissions.
- The database uses WAL journal mode and a busy timeout, so running a
report while a cron `scan` is in progress is safe. The filesystem
is authoritative; the database is an eventually-consistent
reflection of it. Hashed records are committed in batched
transactions while the scan is still running (keeping the WAL
small and letting concurrent reports observe progress), so a
report may see a scan's changes partially applied, and a scan
that dies partway leaves a valid database holding everything
hashed so far; the next scan skips those records and converges
toward the filesystem.
- Schema (`PRAGMA user_version` is the schema version, currently 1; a
database with any other version is a fatal error):
@@ -263,18 +251,6 @@ duplicates another or lies under another is dropped before walking,
so every file is reached exactly once and produces one database
record.
An operand that is a symlink (never followed, not even as an operand),
socket, FIFO, or device node, or a directory named `.zfs`, is not
scanned. `scan` prints a one-line warning naming the path and what it
is, counts it as skipped, and drops it from the scanned operands before
reading the database. Another operand beneath it is still scanned. The
records stored beneath it are not deleted: they are treated like any
other record outside the scanned operands, including the content-phase
exception below. If it lies under another operand, they are under that
operand instead, and are deleted like any other record there that this
scan did not verify. This is not an error: a scan whose every operand
is dropped walks nothing and exits 0.
`scan` synchronizes the database with the filesystem state under the
scanned operands:
@@ -380,12 +356,9 @@ during the hash phase:
Rules for the walk:
- Only regular files. Skip directories, symlinks (do not follow,
including symlink operands), sockets, FIFOs, and device nodes. An
operand that is a symlink, socket, FIFO, or device node is dropped
as described in "`scan` mode" above.
including symlink operands), sockets, FIFOs, and device nodes.
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
walking them would list every file once per snapshot), not even
when it is an operand; such an operand is dropped the same way.
walking them would list every file once per snapshot).
- Filesystem boundaries are crossed by default. With `-x`
(long form `--one-file-system`, following the GNU `du`/`rsync`
convention), never descend into a directory on a different
@@ -396,11 +369,9 @@ Rules for the walk:
path, and continue. Per-file errors never abort the run; the final
summary reports how many were skipped. As specified above, a
skipped path that has a database record from an earlier scan loses
that record, unless it failed only in the content phase, or is an
operand dropped before the database was read that lies under no
other operand; an unreadable directory subtree likewise loses its
records (accepted: the database mirrors what the latest scan could
actually verify).
that record, unless it failed only in the content phase; an
unreadable directory subtree likewise loses its records (accepted:
the database mirrors what the latest scan could actually verify).
Concurrency: the walk phase (which also stats files), the hash phase,
and the content phase each use a worker pool of `--workers` workers
@@ -457,15 +428,6 @@ first dupe size
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
```
Paths are raw bytes and may hold any byte except NUL, so the path
columns (`first` and `dupe`) are escaped to keep every row one line of
tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a
newline as `\n`, and a carriage return as `\r`. Every other byte is
written unchanged, including bytes that are not valid UTF-8. Undoing
those four escapes gives back the stored path. Grouping and ordering
use the stored path, not the escaped one. The warnings `scan` prints on
stderr are escaped the same way, so each warning is one line.
Summary to stderr: records read, number of duplicate groups, number of
dupe files, and total reclaimable bytes (sum of `size` over all dupe
rows) in human units.
@@ -537,9 +499,6 @@ first dupe files size
/srv/a/project /srv/backup/project 3417 104857600
```
The `first` and `dupe` paths are escaped as described under "Report
output format". The root directory's path is `/`.
Summary to stderr: records read, number of duplicate-tree groups,
number of dupe trees, and total reclaimable bytes (sum of `size` over
all dupe rows) in human units.
-23
View File
@@ -29,29 +29,6 @@
# Completed Steps
- warn about and skip symlink, socket, FIFO, device and `.zfs`
operands, keeping the records beneath them (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/9)
- `report` and `trees` open the database read-only, and `scan` leaves it
out of WAL mode, so reading needs only read access (2026-10-03, closes
https://git.eeqj.de/sneak/sfdupes/issues/8)
- escape tabs, newlines, carriage returns and backslashes in report,
trees and warning paths; the root directory's path is `/`
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
- stamp the git tag or short commit in a plain `docker build .`
instead of `dev` (2026-10-02, branch `next`, closes
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
sends `.git`, without `.git/config`, and the `Dockerfile` build
stage takes the `VERSION` build argument when one is given,
otherwise `git describe --tags --always` of that `.git`. The build
fails if the context carries `.git` and the version still comes out
empty, `dev` or `unknown`. The CI checkout step fetches the full
history (`fetch-depth: 0`) so CI sees the tag and stamps the same
value as `make build`.
- replace the 1 KiB end-window sampling with the head/tail plus
content-hash ladder (2026-09-22, branch `next`, closes
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
+11 -37
View File
@@ -74,24 +74,16 @@ func databasePath() string {
return defaultDatabasePath
}
// scanParams are the connection parameters for scan: read-write, with
// WAL journaling and a busy timeout, so a report can run while a cron
// scan is in progress. closeScanDatabase leaves WAL mode again.
const scanParams = "_pragma=busy_timeout(10000)" +
"&_pragma=journal_mode(WAL)" +
"&_pragma=synchronous(NORMAL)"
// openDB opens the SQLite database at path with WAL journaling and a
// busy timeout, so a report can run while a cron scan is in progress.
// It does not create or verify the schema.
func openDB(path string) (*sql.DB, error) {
dsn := "file:" + path +
"?_pragma=busy_timeout(10000)" +
"&_pragma=journal_mode(WAL)" +
"&_pragma=synchronous(NORMAL)"
// reportParams are the connection parameters for report and trees:
// read-only, with the same busy timeout. They set no journal mode,
// because setting one is a write.
const reportParams = "mode=ro" +
"&_pragma=busy_timeout(10000)" +
"&_pragma=query_only(1)"
// openDB opens the SQLite database at path with the connection
// parameters params. It does not create or verify the schema.
func openDB(path, params string) (*sql.DB, error) {
db, err := sql.Open("sqlite", "file:"+path+"?"+params)
db, err := sql.Open("sqlite", dsn)
if err != nil {
return nil, fmt.Errorf("open database %s: %w", path, err)
}
@@ -112,7 +104,7 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
return nil, fmt.Errorf("create database directory: %w", err)
}
db, err := openDB(path, scanParams)
db, err := openDB(path)
if err != nil {
return nil, err
}
@@ -127,24 +119,6 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
return db, nil
}
// closeScanDatabase switches the database at path from WAL back to
// rollback-journal mode and closes it. Out of WAL mode the database
// file alone holds the whole database, so a reader needs no -wal or
// -shm file beside it, nor write access to create them. The switch
// fails while a report has the database open; the database then stays
// in WAL mode, still readable, until a later scan closes it.
func closeScanDatabase(ctx context.Context, db *sql.DB, path string) {
// Runs on the way out of a cancelled scan too.
_, err := db.ExecContext(context.WithoutCancel(ctx),
"PRAGMA journal_mode = DELETE")
if err != nil {
fmt.Fprintf(os.Stderr, "scan: database %s left in WAL mode: %v\n",
path, err)
}
_ = db.Close()
}
// openReportDatabase opens an existing database for the report and
// trees subcommands. A missing database file is an error directing the
// user to run scan first; the schema version must match exactly.
@@ -160,7 +134,7 @@ func openReportDatabase(ctx context.Context,
return nil, fmt.Errorf("database: %w", err)
}
db, err := openDB(path, reportParams)
db, err := openDB(path)
if err != nil {
return nil, err
}
-44
View File
@@ -5,7 +5,6 @@ import (
"database/sql"
"errors"
"fmt"
"os"
"path/filepath"
"slices"
"strings"
@@ -131,49 +130,6 @@ func TestOpenReportDatabaseOK(t *testing.T) {
_ = db.Close()
}
func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
t.Parallel()
// A report holding the database open stops scan from taking it out
// of WAL mode. The -wal and -shm files must then stay beside it, so
// that a later report still needs only read access.
path := testDBPath(t)
scanDB, err := openScanDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
reportDB, err := openReportDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
closeScanDatabase(t.Context(), scanDB, path)
_ = reportDB.Close()
_, err = os.Stat(path + "-wal")
if err != nil {
t.Fatalf("no -wal left: the switch out of WAL mode was not "+
"stopped: %v", err)
}
makeReadOnly(t, path)
reportDB, err = openReportDatabase(t.Context(), path)
if err != nil {
t.Fatalf("openReportDatabase: %v", err)
}
defer func() { _ = reportDB.Close() }()
_, err = loadFileRows(t.Context(), reportDB)
if err != nil {
t.Fatalf("loadFileRows: %v", err)
}
}
func TestApplyChangesRoundTrip(t *testing.T) {
t.Parallel()
+16 -207
View File
@@ -44,44 +44,6 @@ func assertNoSidecars(t *testing.T, path string) {
}
}
// makeReadOnly takes write permission away from the database at path,
// from any WAL sidecar beside it, and from their directory, as for a
// user reading a database that a root cron scan keeps. Root ignores
// file permissions, so it skips the test when run as root.
func makeReadOnly(t *testing.T, path string) {
t.Helper()
if os.Geteuid() == 0 {
t.Skip("root ignores file permissions")
}
err := os.Chmod(path, 0o400)
if err != nil {
t.Fatal(err)
}
for _, suffix := range walSuffixes {
err = os.Chmod(path+suffix, 0o400)
if err != nil && !errors.Is(err, fs.ErrNotExist) {
t.Fatal(err)
}
}
dir := filepath.Dir(path)
//nolint:gosec // reaching the database needs the search bit
err = os.Chmod(dir, 0o500)
if err != nil {
t.Fatal(err)
}
// Runs before t.TempDir's own cleanup, which must delete the files.
t.Cleanup(func() {
//nolint:gosec // removing the directory needs its search bit back
_ = os.Chmod(dir, 0o700)
})
}
// captureStdout redirects os.Stdout to a file for the rest of the test
// and returns a function reading back everything written to it. Only
// machine-readable data belongs on stdout (README design goal 4), so
@@ -89,25 +51,16 @@ func makeReadOnly(t *testing.T, path string) {
func captureStdout(t *testing.T) func() string {
t.Helper()
return captureStream(t, &os.Stdout)
}
// captureStream redirects *stream (os.Stdout or os.Stderr) to a file
// for the rest of the test and returns a function reading back
// everything written to it.
func captureStream(t *testing.T, stream **os.File) func() string {
t.Helper()
f, err := os.Create(filepath.Join(t.TempDir(), "capture"))
f, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
if err != nil {
t.Fatal(err)
}
saved := *stream
*stream = f
saved := os.Stdout
os.Stdout = f
t.Cleanup(func() {
*stream = saved
os.Stdout = saved
_ = f.Close()
})
@@ -138,13 +91,13 @@ func captureStream(t *testing.T, stream **os.File) func() string {
// brokenDatabase writes a database that opens cleanly and passes the
// schema-version check but has no files table, so the first query
// fails with the database already open: a fatal error on a path that
// owns an open database. It closes the database the way scan does.
// owns an open database.
func brokenDatabase(t *testing.T) string {
t.Helper()
path := testDBPath(t)
db, err := openDB(path, scanParams)
db, err := openDB(path)
if err != nil {
t.Fatal(err)
}
@@ -155,7 +108,10 @@ func brokenDatabase(t *testing.T) string {
t.Fatal(err)
}
closeScanDatabase(t.Context(), db, path)
err = db.Close()
if err != nil {
t.Fatal(err)
}
return path
}
@@ -188,10 +144,7 @@ func TestOpenDatabaseKeepsWALWhileOpen(t *testing.T) {
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
// Every subcommand that owns an open database must close it when
// it fails: no os.Exit between the open and the return. The
// sidecar check is evidence of the close only for scan: report and
// trees only read a database that is out of WAL mode, which leaves
// nothing on disk whether they close it or not.
// it fails: no os.Exit between the open and the return.
cases := map[string][]string{
cmdScan: {cmdScan},
cmdReport: {cmdReport},
@@ -367,30 +320,21 @@ func scanFixture(t *testing.T) []string {
t.Fatal(err)
}
scanOK(t, dir)
return dupes
}
// scanOK runs scan over operands, fails the test unless it exits 0 with
// nothing on stdout, and returns everything it printed to stderr.
func scanOK(t *testing.T, operands ...string) string {
t.Helper()
var stderr bytes.Buffer
stdout := captureStdout(t)
stderr := captureStream(t, &os.Stderr)
code := run(append([]string{cmdScan}, operands...), os.Stderr)
code := run([]string{cmdScan, dir}, &stderr)
if code != exitOK {
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
operands, code, exitOK, stderr())
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
if got := stdout(); got != "" {
t.Errorf("scan stdout = %q, want nothing (data only)", got)
}
return stderr()
return dupes
}
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
@@ -401,103 +345,6 @@ func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
assertNoSidecars(t, path)
}
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
dir := t.TempDir()
writeFile(t, dir, "target/sub/f", pattern(1, 10))
link := filepath.Join(dir, "link")
err := os.Symlink(filepath.Join(dir, "target"), link)
if err != nil {
t.Fatal(err)
}
// Scanning a directory through the symlink stores a record beneath
// the symlink's own path for a file beneath its target.
scanOK(t, filepath.Join(link, "sub"))
assertOperandSkipped(t, path, link, "symlink",
filepath.Join(link, "sub", "f"))
}
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
dir := t.TempDir()
writeFile(t, dir, "target/sub/f", pattern(1, 10))
link := filepath.Join(dir, "link")
err := os.Symlink(filepath.Join(dir, "target"), link)
if err != nil {
t.Fatal(err)
}
// link is dropped as a symlink, but link/sub must still be scanned,
// not dropped as lying under link.
scanOK(t, link, filepath.Join(link, "sub"))
db, err := openDB(path, reportParams)
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = db.Close() })
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
}
func TestRunScanSkipsZFSOperand(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
zfs := filepath.Join(t.TempDir(), ".zfs")
snapshot := filepath.Join(zfs, "snapshot", "hourly")
f := writeFile(t, snapshot, "f", pattern(1, 10))
// An operand beneath a .zfs directory is walked, because it is not
// itself named .zfs.
scanOK(t, snapshot)
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
}
// assertOperandSkipped scans operand alone and checks that it is skipped
// as kind: a warning naming it, one skip in the summary, exit 0, and the
// record for kept, which an earlier scan stored beneath operand, still
// in the database at dbPath.
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
kept string,
) {
t.Helper()
stderr := scanOK(t, operand)
warning := "walk " + operand + ": skipping " + kind + " operand\n"
if !strings.Contains(stderr, warning) {
t.Errorf("stderr = %q, want %q", stderr, warning)
}
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
"0 unchanged), 1 skipped\n"
if !strings.Contains(stderr, summary) {
t.Errorf("stderr = %q, want %q", stderr, summary)
}
db, err := openDB(dbPath, reportParams)
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = db.Close() })
recordByPath(t, dbRecords(t, db), kept)
}
func TestRunReportSucceeds(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
@@ -548,41 +395,3 @@ func TestRunTreesSucceeds(t *testing.T) {
assertNoSidecars(t, path)
}
func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
// README §Database: report and trees need only read access to the
// database file. With its directory read-only as well, SQLite
// cannot create any file beside it.
path := testDBPath(t)
t.Setenv(databaseEnv, path)
dupes := scanFixture(t)
assertNoSidecars(t, path)
makeReadOnly(t, path)
cases := map[string]string{
cmdReport: "first\tdupe\tsize\n" +
dupes[0] + "\t" + dupes[1] + "\t300\n",
cmdTrees: "first\tdupe\tfiles\tsize\n" +
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) +
"\t1\t300\n",
}
for name, want := range cases {
var stderr bytes.Buffer
stdout := captureStdout(t)
code := run([]string{name}, &stderr)
if code != exitOK {
t.Errorf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
continue
}
if got := stdout(); got != want {
t.Errorf("%s stdout = %q, want %q", name, got, want)
}
}
}
+1 -3
View File
@@ -106,8 +106,6 @@ func (p *progress) increment() {
}
// warnf prints a one-line warning to stderr without corrupting the bar.
// The whole message is escaped like a report's path columns, so a path
// holding a newline cannot split the warning.
func (p *progress) warnf(format string, args ...any) {
if p == nil {
return
@@ -117,7 +115,7 @@ func (p *progress) warnf(format string, args ...any) {
_ = p.bar.Clear()
}
fmt.Fprintln(os.Stderr, escapePath(fmt.Sprintf(format, args...)))
fmt.Fprintf(os.Stderr, format+"\n", args...)
}
// finish terminates the pass's display.
+4 -19
View File
@@ -31,9 +31,9 @@ type scanRec struct {
// loadRecords opens the database and reads every file record for the
// report and trees subcommands. Any database problem — including a
// missing database — is fatal. The error is returned rather than
// exiting, so that the deferred close always runs; the database is
// closed before the caller formats its output, so it stays closed even
// if that output fails.
// exiting, so that the deferred close — which checkpoints the SQLite
// WAL — always runs; the database is closed before the caller formats
// its output, so it stays closed even if that output fails.
func loadRecords(ctx context.Context) ([]scanRec, error) {
dbPath := databasePath()
@@ -87,7 +87,7 @@ func runReport(ctx context.Context) error {
for _, g := range dupes {
for _, p := range g.paths[1:] {
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(g.paths[0]), escapePath(p), g.size)
g.paths[0], p, g.size)
if err != nil {
return fmt.Errorf("write stdout: %w", err)
}
@@ -156,21 +156,6 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
return dupes
}
// escapePath returns a path as it is written in a report column (README
// "Report output format"): a backslash, tab, newline or carriage return
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
// Grouping and sorting use the raw path, never this form.
func escapePath(p string) string {
// Most paths need no escaping; skip building a replacer for them.
if !strings.ContainsAny(p, "\\\t\n\r") {
return p
}
return strings.NewReplacer(
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
).Replace(p)
}
// humanBytes formats a byte count in human units (binary prefixes).
func humanBytes(n int64) string {
const unit = 1024
-118
View File
@@ -1,128 +1,10 @@
package main
import (
"bytes"
"io"
"os"
"path/filepath"
"slices"
"testing"
)
// awkwardDir is a directory name holding every byte the reports escape.
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
// sorts after it, so awkwardDir coming first shows that sorting uses
// the raw path.
func awkwardPairRecs() []scanRec {
return []scanRec{
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
}
}
// seedDatabase writes recs into a fresh database and returns its path.
func seedDatabase(t *testing.T, recs []scanRec) string {
t.Helper()
path := testDBPath(t)
db, err := openScanDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
err = applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
err = db.Close()
if err != nil {
t.Fatal(err)
}
return path
}
func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stderr bytes.Buffer
stdout := captureStdout(t)
code := run([]string{cmdReport}, &stderr)
if code != exitOK {
t.Fatalf("run(report) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
want := "first\tdupe\tsize\n" +
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}
func TestEscapePath(t *testing.T) {
t.Parallel()
cases := map[string]string{
"/srv/plain": "/srv/plain",
"/a\tb": `/a\tb`,
"/a\nb": `/a\nb`,
"/a\rb": `/a\rb`,
`/a\b`: `/a\\b`,
`/a\tb`: `/a\\tb`,
"/not-utf8\xff": "/not-utf8\xff",
}
for in, want := range cases {
if got := escapePath(in); got != want {
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
}
}
}
// TestWarnfEscapes checks that a warning naming a path that holds a
// newline is still one line.
//
//nolint:paralleltest // replaces the process-wide os.Stderr
func TestWarnfEscapes(t *testing.T) {
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
if err != nil {
t.Fatal(err)
}
saved := os.Stderr
os.Stderr = f
t.Cleanup(func() {
os.Stderr = saved
_ = f.Close()
})
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
_, err = f.Seek(0, io.SeekStart)
if err != nil {
t.Fatal(err)
}
got, err := io.ReadAll(f)
if err != nil {
t.Fatal(err)
}
want := `stat /d/a\nb: gone` + "\n"
if string(got) != want {
t.Errorf("warning = %q, want %q", got, want)
}
}
func TestCollectDupeGroups(t *testing.T) {
t.Parallel()
+29 -95
View File
@@ -87,9 +87,8 @@ type fileMeta struct {
// hash only when its size, head, and tail match another file's. Flag
// parsing and the at-least-one-operand check are done by cobra. Errors
// are returned rather than exiting, so that the deferred close — which
// takes the database out of WAL mode — always runs. Cancelling ctx
// unwinds the worker pools and aborts the scan with the context's
// error.
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
// worker pools and aborts the scan with the context's error.
func runScan(ctx context.Context, roots []string, workers int,
oneFS bool,
) error {
@@ -109,7 +108,7 @@ func runScan(ctx context.Context, roots []string, workers int,
return err
}
defer closeScanDatabase(ctx, db, dbPath)
defer func() { _ = db.Close() }()
st, err := syncScan(ctx, db, roots, workers, oneFS)
if err != nil {
@@ -211,17 +210,13 @@ type scanState struct {
// in the content hash of every record of headTailMin or more whose
// size, head, and tail match another record's). Records outside the
// roots are never touched, except that the content phase fills in
// their content hash. Operands the walk cannot start from are dropped
// first, so the records beneath them count as outside the roots unless
// they lie under another root.
// their content hash.
func syncScan(ctx context.Context, db *sql.DB, roots []string,
workers int, oneFS bool,
) (scanStats, error) {
s := &scanState{db: db}
roots = pruneRoots(roots)
// Types are checked before pruning so that an operand under a
// dropped one is still scanned, not dropped as lying under it.
roots = pruneRoots(s.walkableRoots(roots))
s := &scanState{db: db}
err := s.loadIndex(ctx, roots)
if err != nil {
@@ -258,35 +253,6 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
return s.st, s.contentPhase(ctx, workers)
}
// walkableRoots returns the operands the walk can start from: regular
// files, and directories not named .zfs. Every other operand is warned
// about, counted as skipped, and dropped. A dropped operand is no
// longer a root, so the records stored beneath it count as outside the
// roots and are not deleted as unverified, unless it lies under another
// root. An operand that fails lstat here is kept, and the walk warns
// about it.
func (s *scanState) walkableRoots(roots []string) []string {
kept := make([]string, 0, len(roots))
for _, root := range roots {
fi, err := os.Lstat(root)
if err == nil {
warn := operandWarning(root, fi)
if warn != "" {
s.st.skipped++
fmt.Fprintln(os.Stderr, escapePath(warn))
continue
}
}
kept = append(kept, root)
}
return kept
}
// loadIndex indexes the database records under the scan roots for
// change detection and collects the sizes of every record outside
// them: out-of-scope records join the size census so a scanned file
@@ -823,43 +789,11 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
}
}
// operandWarning returns the one-line warning for an operand the walk
// does not start from, naming the path and what it is, or "" for one it
// does: a regular file, or a directory not named .zfs. Symlinks are
// never followed, including as operands.
func operandWarning(root string, fi fs.FileInfo) string {
var kind string
switch mode := fi.Mode(); {
case mode.IsRegular():
return ""
case mode.IsDir():
if filepath.Base(root) != ".zfs" {
return ""
}
kind = ".zfs directory"
case mode&fs.ModeSymlink != 0:
kind = "symlink"
case mode&fs.ModeSocket != 0:
kind = "socket"
case mode&fs.ModeNamedPipe != 0:
kind = "FIFO"
case mode&fs.ModeDevice != 0:
kind = "device node"
default:
kind = "non-regular file"
}
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
}
// seedRoot turns one PATH operand into the walk's starting state: a
// regular-file operand is statted and emitted directly, and a directory
// operand becomes an initial job. walkableRoots has already dropped
// every other operand. One that has changed into something else since
// is warned about and skipped here; it is still a root, so the records
// stored beneath it are deleted as unverified.
// regular-file operand is statted and emitted directly, a directory
// operand becomes an initial job, and a symlink or other non-regular
// operand yields nothing (symlinks are never followed, including as
// operands).
func seedRoot(ctx context.Context, root string,
events chan<- walkEvent,
) []dirJob {
@@ -873,30 +807,30 @@ func seedRoot(ctx context.Context, root string,
return nil
}
warn := operandWarning(root, fi)
if warn != "" {
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
switch {
case fi.IsDir():
if filepath.Base(root) == ".zfs" {
return nil
}
return nil
}
if fi.IsDir() {
dev, ok := deviceOfInfo(fi)
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
case fi.Mode().IsRegular():
dev, ino := inodeOfInfo(fi)
sendEvent(ctx, events, walkEvent{rec: fileRec{
path: root,
size: fi.Size(),
mtime: fi.ModTime().Unix(),
dev: dev,
ino: ino,
}})
return nil
default:
return nil
}
dev, ino := inodeOfInfo(fi)
sendEvent(ctx, events, walkEvent{rec: fileRec{
path: root,
size: fi.Size(),
mtime: fi.ModTime().Unix(),
dev: dev,
ino: ino,
}})
return nil
}
// startWalkWorkers starts the walk worker pool. Each worker processes
+2 -4
View File
@@ -856,11 +856,9 @@ func TestWalkFileAndSymlinkOperands(t *testing.T) {
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
}
// A symlink operand that reaches the walk (it became one after
// walkableRoots checked it) is not followed: it yields a warning and
// no records.
// A symlink operand is not followed and yields nothing.
recs, errs = collectWalk(t, []string{link}, false, 2)
if errs != 1 || len(recs) != 0 {
if errs != 0 || len(recs) != 0 {
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
}
}
+4 -4
View File
@@ -16,10 +16,10 @@
# implies the repo is green.
#
# That implication holds only because of CHECK_EPOCH. A COPY layer is
# invalidated only by changed content, and a rebuild of an unchanged
# checkout sends the same content, so without a fresh value here Docker
# serves the gate layers from cache and the build reports a green it
# never earned. Passing the current epoch invalidates the gate
# invalidated by changed content, and a merge commit's tree is
# byte-identical to the branch head it merges, so without a fresh value
# here Docker serves the gate layers from cache and the build reports a
# green it never earned. Passing the current epoch invalidates the gate
# layers on every run while leaving the pinned base images and
# go mod download cached; see the Dockerfile for the placement.
set -eu
+6 -15
View File
@@ -62,8 +62,7 @@ func runTrees(ctx context.Context) error {
first := g[0]
for _, n := range g[1:] {
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
escapePath(first.path), escapePath(n.path),
first.fileCount, first.totalSize)
first.path, n.path, first.fileCount, first.totalSize)
if err != nil {
return fmt.Errorf("write stdout: %w", err)
}
@@ -88,8 +87,8 @@ func runTrees(ctx context.Context) error {
// buildHierarchy reconstructs the directory hierarchy from the record
// paths under a synthetic super-root. Paths are split on "/"; for
// absolute paths the first component is empty, which becomes the
// top-level node with path "/". It returns the super-root and every
// absolute paths the first component is empty, which simply becomes a
// top-level node representing "/". It returns the super-root and every
// directory node created.
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
super := &treeNode{}
@@ -103,17 +102,9 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
for _, c := range comps[:len(comps)-1] {
child := node.dirs[c]
if child == nil {
childPath := node.path + "/" + c
// The root directory's path is "/", not empty, and its
// children's paths start with one slash, not two.
switch {
case node == super && c == "":
childPath = "/"
case node == super:
childPath = c
case node.path == "/":
childPath = "/" + c
childPath := c
if node != super {
childPath = node.path + "/" + c
}
child = &treeNode{path: childPath, parent: node}
-41
View File
@@ -1,7 +1,6 @@
package main
import (
"bytes"
"slices"
"testing"
)
@@ -85,46 +84,6 @@ func TestBuildHierarchyCounts(t *testing.T) {
}
}
func TestBuildHierarchyRootPath(t *testing.T) {
t.Parallel()
// The root directory's path is "/", never empty, and its
// children's paths start with a single slash.
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/srv"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
}
func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
var stderr bytes.Buffer
stdout := captureStdout(t)
code := run([]string{cmdTrees}, &stderr)
if code != exitOK {
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
code, exitOK, stderr.String())
}
want := "first\tdupe\tfiles\tsize\n" +
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
if got := stdout(); got != want {
t.Errorf("stdout = %q, want %q", got, want)
}
}
func TestTreeDigests(t *testing.T) {
t.Parallel()