Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d8ca0c1f6c | ||
|
|
d63d3cc7fc | ||
|
|
c887f80f57 | ||
|
|
c9bf22d483 |
@@ -152,16 +152,28 @@ All three subcommands operate on a single SQLite database file:
|
||||
use. `report` and `trees` require an existing database; a missing
|
||||
database file is a fatal error (exit 1) telling the user to run
|
||||
`scan` first.
|
||||
- The database uses WAL journal mode and a busy timeout, so running a
|
||||
report while a cron `scan` is in progress is safe. The filesystem
|
||||
is authoritative; the database is an eventually-consistent
|
||||
reflection of it. Hashed records are committed in batched
|
||||
transactions while the scan is still running (keeping the WAL
|
||||
small and letting concurrent reports observe progress), so a
|
||||
report may see a scan's changes partially applied, and a scan
|
||||
that dies partway leaves a valid database holding everything
|
||||
hashed so far; the next scan skips those records and converges
|
||||
toward the filesystem.
|
||||
- While `scan` runs, the database is in WAL journal mode with a busy
|
||||
timeout, so running a report while a cron `scan` is in progress is
|
||||
safe. The filesystem is authoritative; the database is an
|
||||
eventually-consistent reflection of it. Hashed records are
|
||||
committed in batched transactions while the scan is still running
|
||||
(keeping the WAL small and letting concurrent reports observe
|
||||
progress), so a report may see a scan's changes partially applied,
|
||||
and a scan that dies partway leaves a valid database holding
|
||||
everything hashed so far; the next scan skips those records and
|
||||
converges toward the filesystem.
|
||||
- `scan` switches the database back to rollback-journal mode when it
|
||||
closes it, so between scans the database file alone holds the whole
|
||||
database. Each switch needs the database to itself: a `scan` that
|
||||
starts while a report is still reading waits for it up to the
|
||||
10-second busy timeout, then fails; a `scan` that ends while a
|
||||
report has the database open warns and leaves the database in WAL
|
||||
mode until the next scan.
|
||||
- `report` and `trees` open the database read-only and need only read
|
||||
access to the database file, and no write access to its directory.
|
||||
While the database is in WAL mode they also read the `-wal` and
|
||||
`-shm` files beside it, which SQLite creates with the database
|
||||
file's permissions.
|
||||
- Schema (`PRAGMA user_version` is the schema version, currently 1; a
|
||||
database with any other version is a fatal error):
|
||||
|
||||
@@ -251,6 +263,18 @@ duplicates another or lies under another is dropped before walking,
|
||||
so every file is reached exactly once and produces one database
|
||||
record.
|
||||
|
||||
An operand that is a symlink (never followed, not even as an operand),
|
||||
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
||||
scanned. `scan` prints a one-line warning naming the path and what it
|
||||
is, counts it as skipped, and drops it from the scanned operands before
|
||||
reading the database. Another operand beneath it is still scanned. The
|
||||
records stored beneath it are not deleted: they are treated like any
|
||||
other record outside the scanned operands, including the content-phase
|
||||
exception below. If it lies under another operand, they are under that
|
||||
operand instead, and are deleted like any other record there that this
|
||||
scan did not verify. This is not an error: a scan whose every operand
|
||||
is dropped walks nothing and exits 0.
|
||||
|
||||
`scan` synchronizes the database with the filesystem state under the
|
||||
scanned operands:
|
||||
|
||||
@@ -356,9 +380,12 @@ during the hash phase:
|
||||
Rules for the walk:
|
||||
|
||||
- Only regular files. Skip directories, symlinks (do not follow,
|
||||
including symlink operands), sockets, FIFOs, and device nodes.
|
||||
including symlink operands), sockets, FIFOs, and device nodes. An
|
||||
operand that is a symlink, socket, FIFO, or device node is dropped
|
||||
as described in "`scan` mode" above.
|
||||
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
|
||||
walking them would list every file once per snapshot).
|
||||
walking them would list every file once per snapshot), not even
|
||||
when it is an operand; such an operand is dropped the same way.
|
||||
- Filesystem boundaries are crossed by default. With `-x`
|
||||
(long form `--one-file-system`, following the GNU `du`/`rsync`
|
||||
convention), never descend into a directory on a different
|
||||
@@ -369,9 +396,11 @@ Rules for the walk:
|
||||
path, and continue. Per-file errors never abort the run; the final
|
||||
summary reports how many were skipped. As specified above, a
|
||||
skipped path that has a database record from an earlier scan loses
|
||||
that record, unless it failed only in the content phase; an
|
||||
unreadable directory subtree likewise loses its records (accepted:
|
||||
the database mirrors what the latest scan could actually verify).
|
||||
that record, unless it failed only in the content phase, or is an
|
||||
operand dropped before the database was read that lies under no
|
||||
other operand; an unreadable directory subtree likewise loses its
|
||||
records (accepted: the database mirrors what the latest scan could
|
||||
actually verify).
|
||||
|
||||
Concurrency: the walk phase (which also stats files), the hash phase,
|
||||
and the content phase each use a worker pool of `--workers` workers
|
||||
@@ -428,6 +457,15 @@ first dupe size
|
||||
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
||||
```
|
||||
|
||||
Paths are raw bytes and may hold any byte except NUL, so the path
|
||||
columns (`first` and `dupe`) are escaped to keep every row one line of
|
||||
tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a
|
||||
newline as `\n`, and a carriage return as `\r`. Every other byte is
|
||||
written unchanged, including bytes that are not valid UTF-8. Undoing
|
||||
those four escapes gives back the stored path. Grouping and ordering
|
||||
use the stored path, not the escaped one. The warnings `scan` prints on
|
||||
stderr are escaped the same way, so each warning is one line.
|
||||
|
||||
Summary to stderr: records read, number of duplicate groups, number of
|
||||
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
||||
rows) in human units.
|
||||
@@ -499,6 +537,9 @@ first dupe files size
|
||||
/srv/a/project /srv/backup/project 3417 104857600
|
||||
```
|
||||
|
||||
The `first` and `dupe` paths are escaped as described under "Report
|
||||
output format". The root directory's path is `/`.
|
||||
|
||||
Summary to stderr: records read, number of duplicate-tree groups,
|
||||
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
||||
all dupe rows) in human units.
|
||||
|
||||
@@ -29,6 +29,18 @@
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- warn about and skip symlink, socket, FIFO, device and `.zfs`
|
||||
operands, keeping the records beneath them (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||
|
||||
- `report` and `trees` open the database read-only, and `scan` leaves it
|
||||
out of WAL mode, so reading needs only read access (2026-10-03, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/8)
|
||||
|
||||
- escape tabs, newlines, carriage returns and backslashes in report,
|
||||
trees and warning paths; the root directory's path is `/`
|
||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
|
||||
|
||||
- stamp the git tag or short commit in a plain `docker build .`
|
||||
instead of `dev` (2026-10-02, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
||||
|
||||
@@ -74,16 +74,24 @@ func databasePath() string {
|
||||
return defaultDatabasePath
|
||||
}
|
||||
|
||||
// openDB opens the SQLite database at path with WAL journaling and a
|
||||
// busy timeout, so a report can run while a cron scan is in progress.
|
||||
// It does not create or verify the schema.
|
||||
func openDB(path string) (*sql.DB, error) {
|
||||
dsn := "file:" + path +
|
||||
"?_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=journal_mode(WAL)" +
|
||||
"&_pragma=synchronous(NORMAL)"
|
||||
// scanParams are the connection parameters for scan: read-write, with
|
||||
// WAL journaling and a busy timeout, so a report can run while a cron
|
||||
// scan is in progress. closeScanDatabase leaves WAL mode again.
|
||||
const scanParams = "_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=journal_mode(WAL)" +
|
||||
"&_pragma=synchronous(NORMAL)"
|
||||
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
// reportParams are the connection parameters for report and trees:
|
||||
// read-only, with the same busy timeout. They set no journal mode,
|
||||
// because setting one is a write.
|
||||
const reportParams = "mode=ro" +
|
||||
"&_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=query_only(1)"
|
||||
|
||||
// openDB opens the SQLite database at path with the connection
|
||||
// parameters params. It does not create or verify the schema.
|
||||
func openDB(path, params string) (*sql.DB, error) {
|
||||
db, err := sql.Open("sqlite", "file:"+path+"?"+params)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open database %s: %w", path, err)
|
||||
}
|
||||
@@ -104,7 +112,7 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||
return nil, fmt.Errorf("create database directory: %w", err)
|
||||
}
|
||||
|
||||
db, err := openDB(path)
|
||||
db, err := openDB(path, scanParams)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -119,6 +127,24 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// closeScanDatabase switches the database at path from WAL back to
|
||||
// rollback-journal mode and closes it. Out of WAL mode the database
|
||||
// file alone holds the whole database, so a reader needs no -wal or
|
||||
// -shm file beside it, nor write access to create them. The switch
|
||||
// fails while a report has the database open; the database then stays
|
||||
// in WAL mode, still readable, until a later scan closes it.
|
||||
func closeScanDatabase(ctx context.Context, db *sql.DB, path string) {
|
||||
// Runs on the way out of a cancelled scan too.
|
||||
_, err := db.ExecContext(context.WithoutCancel(ctx),
|
||||
"PRAGMA journal_mode = DELETE")
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "scan: database %s left in WAL mode: %v\n",
|
||||
path, err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
}
|
||||
|
||||
// openReportDatabase opens an existing database for the report and
|
||||
// trees subcommands. A missing database file is an error directing the
|
||||
// user to run scan first; the schema version must match exactly.
|
||||
@@ -134,7 +160,7 @@ func openReportDatabase(ctx context.Context,
|
||||
return nil, fmt.Errorf("database: %w", err)
|
||||
}
|
||||
|
||||
db, err := openDB(path)
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
+44
@@ -5,6 +5,7 @@ import (
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
@@ -130,6 +131,49 @@ func TestOpenReportDatabaseOK(t *testing.T) {
|
||||
_ = db.Close()
|
||||
}
|
||||
|
||||
func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A report holding the database open stops scan from taking it out
|
||||
// of WAL mode. The -wal and -shm files must then stay beside it, so
|
||||
// that a later report still needs only read access.
|
||||
path := testDBPath(t)
|
||||
|
||||
scanDB, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
reportDB, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
closeScanDatabase(t.Context(), scanDB, path)
|
||||
|
||||
_ = reportDB.Close()
|
||||
|
||||
_, err = os.Stat(path + "-wal")
|
||||
if err != nil {
|
||||
t.Fatalf("no -wal left: the switch out of WAL mode was not "+
|
||||
"stopped: %v", err)
|
||||
}
|
||||
|
||||
makeReadOnly(t, path)
|
||||
|
||||
reportDB, err = openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("openReportDatabase: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = reportDB.Close() }()
|
||||
|
||||
_, err = loadFileRows(t.Context(), reportDB)
|
||||
if err != nil {
|
||||
t.Fatalf("loadFileRows: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
|
||||
+207
-16
@@ -44,6 +44,44 @@ func assertNoSidecars(t *testing.T, path string) {
|
||||
}
|
||||
}
|
||||
|
||||
// makeReadOnly takes write permission away from the database at path,
|
||||
// from any WAL sidecar beside it, and from their directory, as for a
|
||||
// user reading a database that a root cron scan keeps. Root ignores
|
||||
// file permissions, so it skips the test when run as root.
|
||||
func makeReadOnly(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
|
||||
if os.Geteuid() == 0 {
|
||||
t.Skip("root ignores file permissions")
|
||||
}
|
||||
|
||||
err := os.Chmod(path, 0o400)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for _, suffix := range walSuffixes {
|
||||
err = os.Chmod(path+suffix, 0o400)
|
||||
if err != nil && !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
dir := filepath.Dir(path)
|
||||
|
||||
//nolint:gosec // reaching the database needs the search bit
|
||||
err = os.Chmod(dir, 0o500)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Runs before t.TempDir's own cleanup, which must delete the files.
|
||||
t.Cleanup(func() {
|
||||
//nolint:gosec // removing the directory needs its search bit back
|
||||
_ = os.Chmod(dir, 0o700)
|
||||
})
|
||||
}
|
||||
|
||||
// captureStdout redirects os.Stdout to a file for the rest of the test
|
||||
// and returns a function reading back everything written to it. Only
|
||||
// machine-readable data belongs on stdout (README design goal 4), so
|
||||
@@ -51,16 +89,25 @@ func assertNoSidecars(t *testing.T, path string) {
|
||||
func captureStdout(t *testing.T) func() string {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||
return captureStream(t, &os.Stdout)
|
||||
}
|
||||
|
||||
// captureStream redirects *stream (os.Stdout or os.Stderr) to a file
|
||||
// for the rest of the test and returns a function reading back
|
||||
// everything written to it.
|
||||
func captureStream(t *testing.T, stream **os.File) func() string {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "capture"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stdout
|
||||
os.Stdout = f
|
||||
saved := *stream
|
||||
*stream = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stdout = saved
|
||||
*stream = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
@@ -91,13 +138,13 @@ func captureStdout(t *testing.T) func() string {
|
||||
// brokenDatabase writes a database that opens cleanly and passes the
|
||||
// schema-version check but has no files table, so the first query
|
||||
// fails with the database already open: a fatal error on a path that
|
||||
// owns an open database.
|
||||
// owns an open database. It closes the database the way scan does.
|
||||
func brokenDatabase(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openDB(path)
|
||||
db, err := openDB(path, scanParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -108,10 +155,7 @@ func brokenDatabase(t *testing.T) string {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = db.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
closeScanDatabase(t.Context(), db, path)
|
||||
|
||||
return path
|
||||
}
|
||||
@@ -144,7 +188,10 @@ func TestOpenDatabaseKeepsWALWhileOpen(t *testing.T) {
|
||||
|
||||
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||
// Every subcommand that owns an open database must close it when
|
||||
// it fails: no os.Exit between the open and the return.
|
||||
// it fails: no os.Exit between the open and the return. The
|
||||
// sidecar check is evidence of the close only for scan: report and
|
||||
// trees only read a database that is out of WAL mode, which leaves
|
||||
// nothing on disk whether they close it or not.
|
||||
cases := map[string][]string{
|
||||
cmdScan: {cmdScan},
|
||||
cmdReport: {cmdReport},
|
||||
@@ -320,21 +367,30 @@ func scanFixture(t *testing.T) []string {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
scanOK(t, dir)
|
||||
|
||||
return dupes
|
||||
}
|
||||
|
||||
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
||||
// nothing on stdout, and returns everything it printed to stderr.
|
||||
func scanOK(t *testing.T, operands ...string) string {
|
||||
t.Helper()
|
||||
|
||||
stdout := captureStdout(t)
|
||||
stderr := captureStream(t, &os.Stderr)
|
||||
|
||||
code := run([]string{cmdScan, dir}, &stderr)
|
||||
code := run(append([]string{cmdScan}, operands...), os.Stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
||||
operands, code, exitOK, stderr())
|
||||
}
|
||||
|
||||
if got := stdout(); got != "" {
|
||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
return dupes
|
||||
return stderr()
|
||||
}
|
||||
|
||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
@@ -345,6 +401,103 @@ func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Scanning a directory through the symlink stores a record beneath
|
||||
// the symlink's own path for a file beneath its target.
|
||||
scanOK(t, filepath.Join(link, "sub"))
|
||||
|
||||
assertOperandSkipped(t, path, link, "symlink",
|
||||
filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// link is dropped as a symlink, but link/sub must still be scanned,
|
||||
// not dropped as lying under link.
|
||||
scanOK(t, link, filepath.Join(link, "sub"))
|
||||
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
zfs := filepath.Join(t.TempDir(), ".zfs")
|
||||
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
||||
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
||||
|
||||
// An operand beneath a .zfs directory is walked, because it is not
|
||||
// itself named .zfs.
|
||||
scanOK(t, snapshot)
|
||||
|
||||
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
||||
}
|
||||
|
||||
// assertOperandSkipped scans operand alone and checks that it is skipped
|
||||
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
||||
// record for kept, which an earlier scan stored beneath operand, still
|
||||
// in the database at dbPath.
|
||||
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
||||
kept string,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
stderr := scanOK(t, operand)
|
||||
|
||||
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
||||
if !strings.Contains(stderr, warning) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
||||
}
|
||||
|
||||
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
||||
"0 unchanged), 1 skipped\n"
|
||||
if !strings.Contains(stderr, summary) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
||||
}
|
||||
|
||||
db, err := openDB(dbPath, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), kept)
|
||||
}
|
||||
|
||||
func TestRunReportSucceeds(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
@@ -395,3 +548,41 @@ func TestRunTreesSucceeds(t *testing.T) {
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
||||
// README §Database: report and trees need only read access to the
|
||||
// database file. With its directory read-only as well, SQLite
|
||||
// cannot create any file beside it.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
assertNoSidecars(t, path)
|
||||
makeReadOnly(t, path)
|
||||
|
||||
cases := map[string]string{
|
||||
cmdReport: "first\tdupe\tsize\n" +
|
||||
dupes[0] + "\t" + dupes[1] + "\t300\n",
|
||||
cmdTrees: "first\tdupe\tfiles\tsize\n" +
|
||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) +
|
||||
"\t1\t300\n",
|
||||
}
|
||||
|
||||
for name, want := range cases {
|
||||
var stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{name}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("%s stdout = %q, want %q", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+3
-1
@@ -106,6 +106,8 @@ func (p *progress) increment() {
|
||||
}
|
||||
|
||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||
// The whole message is escaped like a report's path columns, so a path
|
||||
// holding a newline cannot split the warning.
|
||||
func (p *progress) warnf(format string, args ...any) {
|
||||
if p == nil {
|
||||
return
|
||||
@@ -115,7 +117,7 @@ func (p *progress) warnf(format string, args ...any) {
|
||||
_ = p.bar.Clear()
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
||||
fmt.Fprintln(os.Stderr, escapePath(fmt.Sprintf(format, args...)))
|
||||
}
|
||||
|
||||
// finish terminates the pass's display.
|
||||
|
||||
@@ -31,9 +31,9 @@ type scanRec struct {
|
||||
// loadRecords opens the database and reads every file record for the
|
||||
// report and trees subcommands. Any database problem — including a
|
||||
// missing database — is fatal. The error is returned rather than
|
||||
// exiting, so that the deferred close — which checkpoints the SQLite
|
||||
// WAL — always runs; the database is closed before the caller formats
|
||||
// its output, so it stays closed even if that output fails.
|
||||
// exiting, so that the deferred close always runs; the database is
|
||||
// closed before the caller formats its output, so it stays closed even
|
||||
// if that output fails.
|
||||
func loadRecords(ctx context.Context) ([]scanRec, error) {
|
||||
dbPath := databasePath()
|
||||
|
||||
@@ -87,7 +87,7 @@ func runReport(ctx context.Context) error {
|
||||
for _, g := range dupes {
|
||||
for _, p := range g.paths[1:] {
|
||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||
g.paths[0], p, g.size)
|
||||
escapePath(g.paths[0]), escapePath(p), g.size)
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
@@ -156,6 +156,21 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||
return dupes
|
||||
}
|
||||
|
||||
// escapePath returns a path as it is written in a report column (README
|
||||
// "Report output format"): a backslash, tab, newline or carriage return
|
||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
||||
// Grouping and sorting use the raw path, never this form.
|
||||
func escapePath(p string) string {
|
||||
// Most paths need no escaping; skip building a replacer for them.
|
||||
if !strings.ContainsAny(p, "\\\t\n\r") {
|
||||
return p
|
||||
}
|
||||
|
||||
return strings.NewReplacer(
|
||||
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
||||
).Replace(p)
|
||||
}
|
||||
|
||||
// humanBytes formats a byte count in human units (binary prefixes).
|
||||
func humanBytes(n int64) string {
|
||||
const unit = 1024
|
||||
|
||||
+118
@@ -1,10 +1,128 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// awkwardDir is a directory name holding every byte the reports escape.
|
||||
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
||||
|
||||
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
||||
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
||||
// sorts after it, so awkwardDir coming first shows that sorting uses
|
||||
// the raw path.
|
||||
func awkwardPairRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
||||
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
||||
}
|
||||
}
|
||||
|
||||
// seedDatabase writes recs into a fresh database and returns its path.
|
||||
func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||
t.Helper()
|
||||
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = db.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return path
|
||||
}
|
||||
|
||||
func TestRunReportEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdReport}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEscapePath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cases := map[string]string{
|
||||
"/srv/plain": "/srv/plain",
|
||||
"/a\tb": `/a\tb`,
|
||||
"/a\nb": `/a\nb`,
|
||||
"/a\rb": `/a\rb`,
|
||||
`/a\b`: `/a\\b`,
|
||||
`/a\tb`: `/a\\tb`,
|
||||
"/not-utf8\xff": "/not-utf8\xff",
|
||||
}
|
||||
for in, want := range cases {
|
||||
if got := escapePath(in); got != want {
|
||||
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWarnfEscapes checks that a warning naming a path that holds a
|
||||
// newline is still one line.
|
||||
//
|
||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||
func TestWarnfEscapes(t *testing.T) {
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stderr
|
||||
os.Stderr = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stderr = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
|
||||
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
||||
|
||||
_, err = f.Seek(0, io.SeekStart)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got, err := io.ReadAll(f)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
want := `stat /d/a\nb: gone` + "\n"
|
||||
if string(got) != want {
|
||||
t.Errorf("warning = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroups(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
|
||||
@@ -87,8 +87,9 @@ type fileMeta struct {
|
||||
// hash only when its size, head, and tail match another file's. Flag
|
||||
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||
// are returned rather than exiting, so that the deferred close — which
|
||||
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||
// worker pools and aborts the scan with the context's error.
|
||||
// takes the database out of WAL mode — always runs. Cancelling ctx
|
||||
// unwinds the worker pools and aborts the scan with the context's
|
||||
// error.
|
||||
func runScan(ctx context.Context, roots []string, workers int,
|
||||
oneFS bool,
|
||||
) error {
|
||||
@@ -108,7 +109,7 @@ func runScan(ctx context.Context, roots []string, workers int,
|
||||
return err
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
defer closeScanDatabase(ctx, db, dbPath)
|
||||
|
||||
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
||||
if err != nil {
|
||||
@@ -210,14 +211,18 @@ type scanState struct {
|
||||
// in the content hash of every record of headTailMin or more whose
|
||||
// size, head, and tail match another record's). Records outside the
|
||||
// roots are never touched, except that the content phase fills in
|
||||
// their content hash.
|
||||
// their content hash. Operands the walk cannot start from are dropped
|
||||
// first, so the records beneath them count as outside the roots unless
|
||||
// they lie under another root.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
roots = pruneRoots(roots)
|
||||
|
||||
s := &scanState{db: db}
|
||||
|
||||
// Types are checked before pruning so that an operand under a
|
||||
// dropped one is still scanned, not dropped as lying under it.
|
||||
roots = pruneRoots(s.walkableRoots(roots))
|
||||
|
||||
err := s.loadIndex(ctx, roots)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
@@ -253,6 +258,35 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
return s.st, s.contentPhase(ctx, workers)
|
||||
}
|
||||
|
||||
// walkableRoots returns the operands the walk can start from: regular
|
||||
// files, and directories not named .zfs. Every other operand is warned
|
||||
// about, counted as skipped, and dropped. A dropped operand is no
|
||||
// longer a root, so the records stored beneath it count as outside the
|
||||
// roots and are not deleted as unverified, unless it lies under another
|
||||
// root. An operand that fails lstat here is kept, and the walk warns
|
||||
// about it.
|
||||
func (s *scanState) walkableRoots(roots []string) []string {
|
||||
kept := make([]string, 0, len(roots))
|
||||
|
||||
for _, root := range roots {
|
||||
fi, err := os.Lstat(root)
|
||||
if err == nil {
|
||||
warn := operandWarning(root, fi)
|
||||
if warn != "" {
|
||||
s.st.skipped++
|
||||
|
||||
fmt.Fprintln(os.Stderr, escapePath(warn))
|
||||
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
kept = append(kept, root)
|
||||
}
|
||||
|
||||
return kept
|
||||
}
|
||||
|
||||
// loadIndex indexes the database records under the scan roots for
|
||||
// change detection and collects the sizes of every record outside
|
||||
// them: out-of-scope records join the size census so a scanned file
|
||||
@@ -789,11 +823,43 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
|
||||
}
|
||||
}
|
||||
|
||||
// operandWarning returns the one-line warning for an operand the walk
|
||||
// does not start from, naming the path and what it is, or "" for one it
|
||||
// does: a regular file, or a directory not named .zfs. Symlinks are
|
||||
// never followed, including as operands.
|
||||
func operandWarning(root string, fi fs.FileInfo) string {
|
||||
var kind string
|
||||
|
||||
switch mode := fi.Mode(); {
|
||||
case mode.IsRegular():
|
||||
return ""
|
||||
case mode.IsDir():
|
||||
if filepath.Base(root) != ".zfs" {
|
||||
return ""
|
||||
}
|
||||
|
||||
kind = ".zfs directory"
|
||||
case mode&fs.ModeSymlink != 0:
|
||||
kind = "symlink"
|
||||
case mode&fs.ModeSocket != 0:
|
||||
kind = "socket"
|
||||
case mode&fs.ModeNamedPipe != 0:
|
||||
kind = "FIFO"
|
||||
case mode&fs.ModeDevice != 0:
|
||||
kind = "device node"
|
||||
default:
|
||||
kind = "non-regular file"
|
||||
}
|
||||
|
||||
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
|
||||
}
|
||||
|
||||
// seedRoot turns one PATH operand into the walk's starting state: a
|
||||
// regular-file operand is statted and emitted directly, a directory
|
||||
// operand becomes an initial job, and a symlink or other non-regular
|
||||
// operand yields nothing (symlinks are never followed, including as
|
||||
// operands).
|
||||
// regular-file operand is statted and emitted directly, and a directory
|
||||
// operand becomes an initial job. walkableRoots has already dropped
|
||||
// every other operand. One that has changed into something else since
|
||||
// is warned about and skipped here; it is still a root, so the records
|
||||
// stored beneath it are deleted as unverified.
|
||||
func seedRoot(ctx context.Context, root string,
|
||||
events chan<- walkEvent,
|
||||
) []dirJob {
|
||||
@@ -807,30 +873,30 @@ func seedRoot(ctx context.Context, root string,
|
||||
return nil
|
||||
}
|
||||
|
||||
switch {
|
||||
case fi.IsDir():
|
||||
if filepath.Base(root) == ".zfs" {
|
||||
return nil
|
||||
}
|
||||
warn := operandWarning(root, fi)
|
||||
if warn != "" {
|
||||
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
if fi.IsDir() {
|
||||
dev, ok := deviceOfInfo(fi)
|
||||
|
||||
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
||||
case fi.Mode().IsRegular():
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
path: root,
|
||||
size: fi.Size(),
|
||||
mtime: fi.ModTime().Unix(),
|
||||
dev: dev,
|
||||
ino: ino,
|
||||
}})
|
||||
|
||||
return nil
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
path: root,
|
||||
size: fi.Size(),
|
||||
mtime: fi.ModTime().Unix(),
|
||||
dev: dev,
|
||||
ino: ino,
|
||||
}})
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// startWalkWorkers starts the walk worker pool. Each worker processes
|
||||
|
||||
+4
-2
@@ -856,9 +856,11 @@ func TestWalkFileAndSymlinkOperands(t *testing.T) {
|
||||
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
||||
}
|
||||
|
||||
// A symlink operand is not followed and yields nothing.
|
||||
// A symlink operand that reaches the walk (it became one after
|
||||
// walkableRoots checked it) is not followed: it yields a warning and
|
||||
// no records.
|
||||
recs, errs = collectWalk(t, []string{link}, false, 2)
|
||||
if errs != 0 || len(recs) != 0 {
|
||||
if errs != 1 || len(recs) != 0 {
|
||||
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -62,7 +62,8 @@ func runTrees(ctx context.Context) error {
|
||||
first := g[0]
|
||||
for _, n := range g[1:] {
|
||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||
first.path, n.path, first.fileCount, first.totalSize)
|
||||
escapePath(first.path), escapePath(n.path),
|
||||
first.fileCount, first.totalSize)
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
@@ -87,8 +88,8 @@ func runTrees(ctx context.Context) error {
|
||||
|
||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||
// absolute paths the first component is empty, which simply becomes a
|
||||
// top-level node representing "/". It returns the super-root and every
|
||||
// absolute paths the first component is empty, which becomes the
|
||||
// top-level node with path "/". It returns the super-root and every
|
||||
// directory node created.
|
||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||
super := &treeNode{}
|
||||
@@ -102,9 +103,17 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||
for _, c := range comps[:len(comps)-1] {
|
||||
child := node.dirs[c]
|
||||
if child == nil {
|
||||
childPath := c
|
||||
if node != super {
|
||||
childPath = node.path + "/" + c
|
||||
childPath := node.path + "/" + c
|
||||
|
||||
// The root directory's path is "/", not empty, and its
|
||||
// children's paths start with one slash, not two.
|
||||
switch {
|
||||
case node == super && c == "":
|
||||
childPath = "/"
|
||||
case node == super:
|
||||
childPath = c
|
||||
case node.path == "/":
|
||||
childPath = "/" + c
|
||||
}
|
||||
|
||||
child = &treeNode{path: childPath, parent: node}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"slices"
|
||||
"testing"
|
||||
)
|
||||
@@ -84,6 +85,46 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildHierarchyRootPath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The root directory's path is "/", never empty, and its
|
||||
// children's paths start with a single slash.
|
||||
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
|
||||
|
||||
got := make([]string, 0, len(dirs))
|
||||
for _, d := range dirs {
|
||||
got = append(got, d.path)
|
||||
}
|
||||
|
||||
slices.Sort(got)
|
||||
|
||||
want := []string{"/", "/srv"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdTrees}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
||||
if got := stdout(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeDigests(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user