Compare commits
2
Commits
bd8d41b174
...
7cbb2d7f20
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7cbb2d7f20 | ||
|
|
9abf81535a |
@@ -94,7 +94,15 @@ Goals, in order:
|
|||||||
millions of files, ~150 TB filesystem, possibly slow or busy disks
|
millions of files, ~150 TB filesystem, possibly slow or busy disks
|
||||||
(ZFS pool under resilver). Holding one small record (path, size,
|
(ZFS pool under resilver). Holding one small record (path, size,
|
||||||
mtime) per file in memory during a scan is acceptable; holding
|
mtime) per file in memory during a scan is acceptable; holding
|
||||||
every file's hashes is not (they stay in the database).
|
every file's hashes is not (they stay in the database). The
|
||||||
|
reporting commands do not hold every file's hashes either: `report`
|
||||||
|
lets SQLite group and order the records and writes each row as it
|
||||||
|
reads it, so its memory does not grow with the database, and
|
||||||
|
`trees` reads the records in path order and keeps each directory's
|
||||||
|
path, digest and totals, plus the hashes of only the files in the
|
||||||
|
directories holding the record being read, so its memory grows with
|
||||||
|
the number of directories and with the size of the largest
|
||||||
|
directory.
|
||||||
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
||||||
scan maintains a persistent database; an unchanged file is never
|
scan maintains a persistent database; an unchanged file is never
|
||||||
read again on a rescan, except to compute its content hash once a
|
read again on a rescan, except to compute its content hash once a
|
||||||
@@ -113,7 +121,8 @@ Goals, in order:
|
|||||||
`sfdupes`.
|
`sfdupes`.
|
||||||
- Dependencies: standard library, `github.com/spf13/cobra` for the
|
- Dependencies: standard library, `github.com/spf13/cobra` for the
|
||||||
CLI, **one progress-bar library**
|
CLI, **one progress-bar library**
|
||||||
(`github.com/schollz/progressbar/v3`), **one SQLite driver**
|
(`github.com/schollz/progressbar/v3`), `golang.org/x/term` to tell
|
||||||
|
whether stderr is a terminal, **one SQLite driver**
|
||||||
(`modernc.org/sqlite`, pure Go, so builds keep cgo disabled), and
|
(`modernc.org/sqlite`, pure Go, so builds keep cgo disabled), and
|
||||||
`golang.org/x/sys` for `flock(2)` (the scan lock, see "Database").
|
`golang.org/x/sys` for `flock(2)` (the scan lock, see "Database").
|
||||||
`github.com/spf13/viper` is permitted if configuration-file support
|
`github.com/spf13/viper` is permitted if configuration-file support
|
||||||
@@ -183,7 +192,10 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
starts while a report is still reading waits for it up to the
|
starts while a report is still reading waits for it up to the
|
||||||
10-second busy timeout, then fails; a `scan` that ends while a
|
10-second busy timeout, then fails; a `scan` that ends while a
|
||||||
report has the database open warns and leaves the database in WAL
|
report has the database open warns and leaves the database in WAL
|
||||||
mode until the next scan.
|
mode until the next scan. `report` writes each row as it reads it,
|
||||||
|
so it is still reading while its output is paused (a pager, a
|
||||||
|
stalled pipe), and a `scan` started then fails after the busy
|
||||||
|
timeout.
|
||||||
- `report` and `trees` open the database read-only and need only read
|
- `report` and `trees` open the database read-only and need only read
|
||||||
access to the database file, and no write access to its directory.
|
access to the database file, and no write access to its directory.
|
||||||
While the database is in WAL mode they also read the `-wal` and
|
While the database is in WAL mode they also read the `-wal` and
|
||||||
@@ -201,6 +213,7 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||||
) WITHOUT ROWID;
|
) WITHOUT ROWID;
|
||||||
|
CREATE INDEX files_signature ON files (size, head, tail, content);
|
||||||
```
|
```
|
||||||
|
|
||||||
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
||||||
@@ -215,7 +228,8 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
until the content phase of a scan (see "`scan` mode" below) has
|
until the content phase of a scan (see "`scan` mode" below) has
|
||||||
read the file. A record with an empty `content` is never part of a
|
read the file. A record with an empty `content` is never part of a
|
||||||
duplicate group, though it still defines the file for tree
|
duplicate group, though it still defines the file for tree
|
||||||
reconstruction.
|
reconstruction. The `files_signature` index lets SQLite group the
|
||||||
|
records by signature for `report` without sorting the whole table.
|
||||||
|
|
||||||
### Duplicate detection
|
### Duplicate detection
|
||||||
|
|
||||||
@@ -443,9 +457,12 @@ arguments.
|
|||||||
|
|
||||||
**`report` must never touch the filesystem being analyzed.** It does not
|
**`report` must never touch the filesystem being analyzed.** It does not
|
||||||
stat, open, or otherwise access any path that appears in the records; its
|
stat, open, or otherwise access any path that appears in the records; its
|
||||||
only I/O is reading the database and writing stdout/stderr. It must
|
only I/O is reading the database, writing stdout/stderr, and the
|
||||||
produce identical output whether or not the scanned filesystem is still
|
temporary file SQLite sorts in when the duplicate rows do not fit in
|
||||||
mounted.
|
memory. SQLite puts that file in `$SQLITE_TMPDIR` or `$TMPDIR` when set,
|
||||||
|
otherwise in `/var/tmp` (or `/tmp`), and deletes it as soon as it has
|
||||||
|
opened it. `report` must produce identical output whether or not the
|
||||||
|
scanned filesystem is still mounted.
|
||||||
|
|
||||||
Processing:
|
Processing:
|
||||||
|
|
||||||
@@ -589,10 +606,16 @@ hash: [12345/98765] 12% |████ | 92 files/s elapsed 2:32 eta 17:54
|
|||||||
|
|
||||||
Additional requirements:
|
Additional requirements:
|
||||||
|
|
||||||
- When stderr is not a TTY, do not emit ANSI redraws: print a plain
|
- When stderr is not a terminal (a pipe, a file, `/dev/null`), do not
|
||||||
one-line progress update no more often than every 5 seconds instead.
|
emit ANSI redraws: print a plain one-line progress update the moment
|
||||||
|
each phase starts, then no more often than every 5 seconds.
|
||||||
- Progress updates are driven from the main goroutine and must be
|
- Progress updates are driven from the main goroutine and must be
|
||||||
non-blocking with respect to the worker pool.
|
non-blocking with respect to the worker pool. On a terminal the
|
||||||
|
spinner-style displays also redraw on their own several times a
|
||||||
|
second, so their count and elapsed time stay current while a phase
|
||||||
|
waits for its next item.
|
||||||
|
- A warning printed during a phase always lands on a line of its own,
|
||||||
|
never inside the progress display.
|
||||||
- `report` and `trees` modes need no progress display, only their
|
- `report` and `trees` modes need no progress display, only their
|
||||||
stderr summaries.
|
stderr summaries.
|
||||||
|
|
||||||
|
|||||||
@@ -29,6 +29,14 @@
|
|||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
|
- `report` and `trees` stream the records instead of holding them all in
|
||||||
|
memory; the schema gains the `files_signature` index (2026-10-04,
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
||||||
|
|
||||||
|
- progress prints at once on a non-terminal, uses a real terminal test,
|
||||||
|
and prints warnings through a spinner instead of racing its redraw
|
||||||
|
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13)
|
||||||
|
|
||||||
- warn about and skip symlink, socket, FIFO, device and `.zfs`
|
- warn about and skip symlink, socket, FIFO, device and `.zfs`
|
||||||
operands, keeping the records beneath them (2026-10-03,
|
operands, keeping the records beneath them (2026-10-03,
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||||
|
|||||||
@@ -51,6 +51,12 @@ CREATE TABLE files (
|
|||||||
) WITHOUT ROWID
|
) WITHOUT ROWID
|
||||||
`
|
`
|
||||||
|
|
||||||
|
// createIndexSQL indexes the records by signature, so report can have
|
||||||
|
// SQLite group them without sorting the whole table.
|
||||||
|
const createIndexSQL = `
|
||||||
|
CREATE INDEX files_signature ON files (size, head, tail, content)
|
||||||
|
`
|
||||||
|
|
||||||
// upsertSQL inserts one file record, replacing any existing record for
|
// upsertSQL inserts one file record, replacing any existing record for
|
||||||
// the same path.
|
// the same path.
|
||||||
const upsertSQL = `
|
const upsertSQL = `
|
||||||
@@ -256,6 +262,11 @@ func createSchema(ctx context.Context, db *sql.DB) error {
|
|||||||
return fmt.Errorf("create schema: %w", err)
|
return fmt.Errorf("create schema: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
_, err = db.ExecContext(ctx, createIndexSQL)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("create schema: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
_, err = db.ExecContext(ctx,
|
_, err = db.ExecContext(ctx,
|
||||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -277,18 +288,18 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
|||||||
return v, nil
|
return v, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadFileRows reads every record from the files table.
|
// loadFileRows streams every record to fn in path order: byte order,
|
||||||
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
// which is the order of the primary key, so SQLite does not sort.
|
||||||
|
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, head, tail, content FROM files")
|
"SELECT path, size, mtime, head, tail, content FROM files "+
|
||||||
|
"ORDER BY path")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read records: %w", err)
|
return fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = rows.Close() }()
|
defer func() { _ = rows.Close() }()
|
||||||
|
|
||||||
var recs []scanRec
|
|
||||||
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
@@ -298,19 +309,92 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
|||||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||||
&r.content)
|
&r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read record: %w", err)
|
return fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
r.path = string(path)
|
r.path = string(path)
|
||||||
recs = append(recs, r)
|
fn(r)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = rows.Err()
|
err = rows.Err()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read records: %w", err)
|
return fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
return recs, nil
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// dupeRowsSQL selects every record in a duplicate group, with the
|
||||||
|
// group's first path. A group is the records with a content hash that
|
||||||
|
// share a size, head, tail, and content, when there are two or more of
|
||||||
|
// them. The rows come in report order: groups by size descending, then
|
||||||
|
// by first path, and each group's paths ascending.
|
||||||
|
const dupeRowsSQL = `
|
||||||
|
SELECT g.first, f.path, f.size
|
||||||
|
FROM files AS f
|
||||||
|
JOIN (
|
||||||
|
SELECT size, head, tail, content, MIN(path) AS first
|
||||||
|
FROM files
|
||||||
|
WHERE content <> ''
|
||||||
|
GROUP BY size, head, tail, content
|
||||||
|
HAVING COUNT(*) > 1
|
||||||
|
) AS g USING (size, head, tail, content)
|
||||||
|
ORDER BY f.size DESC, g.first, f.path
|
||||||
|
`
|
||||||
|
|
||||||
|
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
|
||||||
|
// number of records in the database. The count and the rows are read
|
||||||
|
// in one transaction, so they agree while a scan is committing. An
|
||||||
|
// error from fn stops the reading and is returned as it is.
|
||||||
|
func loadDupeRows(ctx context.Context, db *sql.DB,
|
||||||
|
fn func(first, path string, size int64) error,
|
||||||
|
) (int, error) {
|
||||||
|
// Everything goes through tx: the report connection is the only
|
||||||
|
// one, so a query on db would wait for tx forever.
|
||||||
|
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
defer func() { _ = tx.Rollback() }()
|
||||||
|
|
||||||
|
var records int
|
||||||
|
|
||||||
|
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
defer func() { _ = rows.Close() }()
|
||||||
|
|
||||||
|
for rows.Next() {
|
||||||
|
var (
|
||||||
|
first, path []byte
|
||||||
|
size int64
|
||||||
|
)
|
||||||
|
|
||||||
|
err = rows.Scan(&first, &path, &size)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("read record: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = fn(string(first), string(path), size)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
err = rows.Err()
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return records, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadFileMeta streams every record's path, size, mtime, and whether
|
// loadFileMeta streams every record's path, size, mtime, and whether
|
||||||
|
|||||||
+10
-25
@@ -8,7 +8,6 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -73,9 +72,8 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
|
|||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
recs, err := loadFileRows(t.Context(), db)
|
if recs := dbRecords(t, db); len(recs) != 0 {
|
||||||
if err != nil || len(recs) != 0 {
|
t.Fatalf("records = %v, want none", recs)
|
||||||
t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -168,7 +166,7 @@ func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
|
|||||||
|
|
||||||
defer func() { _ = reportDB.Close() }()
|
defer func() { _ = reportDB.Close() }()
|
||||||
|
|
||||||
_, err = loadFileRows(t.Context(), reportDB)
|
err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("loadFileRows: %v", err)
|
t.Fatalf("loadFileRows: %v", err)
|
||||||
}
|
}
|
||||||
@@ -196,15 +194,8 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
got, err := loadFileRows(t.Context(), db)
|
// The records come back in path order, which is the order of recs.
|
||||||
if err != nil {
|
got := dbRecords(t, db)
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.SortFunc(got, func(a, b scanRec) int {
|
|
||||||
return strings.Compare(a.path, b.path)
|
|
||||||
})
|
|
||||||
|
|
||||||
if !slices.Equal(got, recs) {
|
if !slices.Equal(got, recs) {
|
||||||
t.Fatalf("rows = %+v, want %+v", got, recs)
|
t.Fatalf("rows = %+v, want %+v", got, recs)
|
||||||
}
|
}
|
||||||
@@ -221,11 +212,7 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
got, err = loadFileRows(t.Context(), db)
|
got = dbRecords(t, db)
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(got) != 1 || got[0] != upd {
|
if len(got) != 1 || got[0] != upd {
|
||||||
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
||||||
}
|
}
|
||||||
@@ -254,9 +241,8 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
got, err := loadFileRows(t.Context(), db)
|
if got := dbRecords(t, db); len(got) != n {
|
||||||
if err != nil || len(got) != n {
|
t.Fatalf("records = %d, want %d", len(got), n)
|
||||||
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
deletes := make([]string, 0, n)
|
deletes := make([]string, 0, n)
|
||||||
@@ -270,8 +256,7 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges deletes: %v", err)
|
t.Fatalf("applyChanges deletes: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
got, err = loadFileRows(t.Context(), db)
|
if got := dbRecords(t, db); len(got) != 0 {
|
||||||
if err != nil || len(got) != 0 {
|
t.Fatalf("records = %d, want 0", len(got))
|
||||||
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ require (
|
|||||||
github.com/schollz/progressbar/v3 v3.19.1
|
github.com/schollz/progressbar/v3 v3.19.1
|
||||||
github.com/spf13/cobra v1.10.2
|
github.com/spf13/cobra v1.10.2
|
||||||
golang.org/x/sys v0.46.0
|
golang.org/x/sys v0.46.0
|
||||||
|
golang.org/x/term v0.44.0
|
||||||
modernc.org/sqlite v1.54.0
|
modernc.org/sqlite v1.54.0
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -19,7 +20,6 @@ require (
|
|||||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||||
github.com/rivo/uniseg v0.4.7 // indirect
|
github.com/rivo/uniseg v0.4.7 // indirect
|
||||||
github.com/spf13/pflag v1.0.9 // indirect
|
github.com/spf13/pflag v1.0.9 // indirect
|
||||||
golang.org/x/term v0.44.0 // indirect
|
|
||||||
modernc.org/libc v1.74.1 // indirect
|
modernc.org/libc v1.74.1 // indirect
|
||||||
modernc.org/mathutil v1.7.1 // indirect
|
modernc.org/mathutil v1.7.1 // indirect
|
||||||
modernc.org/memory v1.11.0 // indirect
|
modernc.org/memory v1.11.0 // indirect
|
||||||
|
|||||||
+37
-16
@@ -6,6 +6,7 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/schollz/progressbar/v3"
|
"github.com/schollz/progressbar/v3"
|
||||||
|
"golang.org/x/term"
|
||||||
)
|
)
|
||||||
|
|
||||||
// plainInterval is the minimum time between progress lines when stderr
|
// plainInterval is the minimum time between progress lines when stderr
|
||||||
@@ -24,24 +25,23 @@ const percentScale = 100
|
|||||||
|
|
||||||
// stderrIsTTY reports whether stderr is attached to a terminal.
|
// stderrIsTTY reports whether stderr is attached to a terminal.
|
||||||
func stderrIsTTY() bool {
|
func stderrIsTTY() bool {
|
||||||
fi, err := os.Stderr.Stat()
|
return term.IsTerminal(int(os.Stderr.Fd()))
|
||||||
if err != nil {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
|
|
||||||
return fi.Mode()&os.ModeCharDevice != 0
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// progress renders one scan pass's progress on stderr. On a TTY it
|
// progress renders one scan pass's progress on stderr. On a TTY it
|
||||||
// delegates to the progressbar library (spinner style when the total is
|
// delegates to the progressbar library (spinner style when the total is
|
||||||
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
||||||
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
||||||
// one-line update no more often than every plainInterval.
|
// one-line update as the pass starts, then no more often than every
|
||||||
|
// plainInterval.
|
||||||
//
|
//
|
||||||
// All methods must be called from the main goroutine only. A nil
|
// All methods must be called from the main goroutine only. On a TTY
|
||||||
// *progress is a valid no-display receiver: every method is a no-op,
|
// the library also redraws a spinner from its own goroutine, several
|
||||||
// so batched database flushes during the streaming pass can reuse the
|
// times a second, so its count and elapsed time stay current while a
|
||||||
// update-pass helpers without rendering anything.
|
// pass waits for its next item. A nil *progress is a valid
|
||||||
|
// no-display receiver: every method is a no-op, so batched database
|
||||||
|
// flushes during the streaming pass can reuse the update-pass helpers
|
||||||
|
// without rendering anything.
|
||||||
type progress struct {
|
type progress struct {
|
||||||
label string
|
label string
|
||||||
total int64 // -1 when unknown (walk pass)
|
total int64 // -1 when unknown (walk pass)
|
||||||
@@ -53,10 +53,22 @@ type progress struct {
|
|||||||
|
|
||||||
func newProgress(label string, total int64) *progress {
|
func newProgress(label string, total int64) *progress {
|
||||||
p := &progress{label: label, total: total, start: time.Now()}
|
p := &progress{label: label, total: total, start: time.Now()}
|
||||||
if !stderrIsTTY() {
|
if stderrIsTTY() {
|
||||||
|
p.bar = newBar(label, total)
|
||||||
|
|
||||||
return p
|
return p
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Print the zero state at once: the first item may take minutes,
|
||||||
|
// and a pass must never look hung.
|
||||||
|
p.last = p.start
|
||||||
|
fmt.Fprintln(os.Stderr, p.plainLine())
|
||||||
|
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
// newBar builds the TTY display for newProgress.
|
||||||
|
func newBar(label string, total int64) *progressbar.ProgressBar {
|
||||||
opts := []progressbar.Option{
|
opts := []progressbar.Option{
|
||||||
progressbar.OptionSetWriter(os.Stderr),
|
progressbar.OptionSetWriter(os.Stderr),
|
||||||
progressbar.OptionSetDescription(label),
|
progressbar.OptionSetDescription(label),
|
||||||
@@ -81,9 +93,7 @@ func newProgress(label string, total int64) *progress {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
p.bar = progressbar.NewOptions64(total, opts...)
|
return progressbar.NewOptions64(total, opts...)
|
||||||
|
|
||||||
return p
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// increment records one completed item and refreshes the display.
|
// increment records one completed item and refreshes the display.
|
||||||
@@ -113,11 +123,22 @@ func (p *progress) warnf(format string, args ...any) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
msg := escapePath(fmt.Sprintf(format, args...))
|
||||||
|
|
||||||
|
if p.bar != nil && p.total < 0 {
|
||||||
|
// The library also redraws a spinner from its own goroutine, so
|
||||||
|
// a direct write could land inside a redraw. The bar prints the
|
||||||
|
// warning itself, just before its next redraw.
|
||||||
|
_, _ = progressbar.Bprintln(p.bar, msg)
|
||||||
|
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
if p.bar != nil {
|
if p.bar != nil {
|
||||||
_ = p.bar.Clear()
|
_ = p.bar.Clear()
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, escapePath(fmt.Sprintf(format, args...)))
|
fmt.Fprintln(os.Stderr, msg)
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display.
|
// finish terminates the pass's display.
|
||||||
|
|||||||
@@ -0,0 +1,161 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// spinnerIdle comfortably outlasts the 100ms interval at which the
|
||||||
|
// progressbar library redraws a spinner from its own goroutine.
|
||||||
|
const spinnerIdle = 500 * time.Millisecond
|
||||||
|
|
||||||
|
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||||
|
func TestStderrIsTTYFalseForNonTerminals(t *testing.T) {
|
||||||
|
r, pipe, err := os.Pipe()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
regular, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
devNull, err := os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
saved := os.Stderr
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
os.Stderr = saved
|
||||||
|
|
||||||
|
for _, f := range []*os.File{r, pipe, regular, devNull} {
|
||||||
|
_ = f.Close()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
cases := map[string]*os.File{
|
||||||
|
"a pipe": pipe,
|
||||||
|
"a regular file": regular,
|
||||||
|
os.DevNull: devNull,
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, f := range cases {
|
||||||
|
os.Stderr = f
|
||||||
|
|
||||||
|
if stderrIsTTY() {
|
||||||
|
t.Errorf("stderrIsTTY() = true with stderr on %s", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestNewProgressPrintsBeforeFirstItem checks that each pass shows its
|
||||||
|
// zero state the moment it starts when stderr is not a terminal, and
|
||||||
|
// that the next line still waits for plainInterval.
|
||||||
|
//
|
||||||
|
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||||
|
func TestNewProgressPrintsBeforeFirstItem(t *testing.T) {
|
||||||
|
stderr := captureStderr(t)
|
||||||
|
|
||||||
|
newProgress("walk", -1).increment()
|
||||||
|
newProgress("hash", 10).increment()
|
||||||
|
|
||||||
|
want := "walk: 0 files, elapsed 0s\n" +
|
||||||
|
"hash: [0/10] 0% 0 files/s elapsed 0s eta ?\n"
|
||||||
|
if got := stderr(); got != want {
|
||||||
|
t.Errorf("stderr = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// newWalkSpinner returns the walk pass's terminal display, writing to
|
||||||
|
// os.Stderr whether or not it is a terminal, and stops the library's
|
||||||
|
// redraws when the test ends.
|
||||||
|
func newWalkSpinner(t *testing.T) *progress {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
p := &progress{
|
||||||
|
label: "walk", total: -1, start: time.Now(),
|
||||||
|
bar: newBar("walk", -1),
|
||||||
|
}
|
||||||
|
t.Cleanup(p.finish)
|
||||||
|
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestProgressWarningsOnOwnLines drives the terminal display of the walk
|
||||||
|
// pass through a run of warnings with no items between them, as when the
|
||||||
|
// walk meets many unreadable paths, for several of the spinner's
|
||||||
|
// redraws: every warning must land on a line of its own, never inside a
|
||||||
|
// redraw.
|
||||||
|
//
|
||||||
|
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||||
|
func TestProgressWarningsOnOwnLines(t *testing.T) {
|
||||||
|
stderr := captureStderr(t)
|
||||||
|
p := newWalkSpinner(t)
|
||||||
|
|
||||||
|
// No pause between warnings: one written straight to stderr is
|
||||||
|
// garbled only if a redraw lands while it is being written.
|
||||||
|
issued := 0
|
||||||
|
for start := time.Now(); time.Since(start) < spinnerIdle; issued++ {
|
||||||
|
p.warnf("warning")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The spinner prints the warnings at its next redraw.
|
||||||
|
time.Sleep(spinnerIdle)
|
||||||
|
|
||||||
|
// A terminal shows each line as the text after its last carriage
|
||||||
|
// return.
|
||||||
|
shown := 0
|
||||||
|
|
||||||
|
for line := range strings.SplitSeq(stderr(), "\n") {
|
||||||
|
if !strings.Contains(line, "warning") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
shown++
|
||||||
|
|
||||||
|
if text := line[strings.LastIndex(line, "\r")+1:]; text != "warning" {
|
||||||
|
t.Errorf("terminal shows %q, want %q", text, "warning")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if shown != issued {
|
||||||
|
t.Errorf("%d warning lines, want %d", shown, issued)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSpinnerShowsCountAfterBurst checks that once a burst of items
|
||||||
|
// faster than the redraw limit is over, the walk display shows every
|
||||||
|
// item completed while it waits for the next one.
|
||||||
|
//
|
||||||
|
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||||
|
func TestSpinnerShowsCountAfterBurst(t *testing.T) {
|
||||||
|
stderr := captureStderr(t)
|
||||||
|
p := newWalkSpinner(t)
|
||||||
|
|
||||||
|
for range 50 {
|
||||||
|
p.increment()
|
||||||
|
}
|
||||||
|
|
||||||
|
time.Sleep(spinnerIdle)
|
||||||
|
|
||||||
|
// A terminal shows the last frame drawn. The library starts each
|
||||||
|
// frame with a carriage return and erases the previous one with
|
||||||
|
// spaces first.
|
||||||
|
var shown string
|
||||||
|
|
||||||
|
for frame := range strings.SplitSeq(stderr(), "\r") {
|
||||||
|
if strings.TrimSpace(frame) != "" {
|
||||||
|
shown = frame
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if !strings.Contains(shown, "(50/-,") {
|
||||||
|
t.Errorf("terminal shows %q, want a count of 50", shown)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -6,7 +6,6 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
"os"
|
"os"
|
||||||
"slices"
|
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -29,50 +28,21 @@ type scanRec struct {
|
|||||||
path string
|
path string
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadRecords opens the database and reads every file record for the
|
// runReport implements the report subcommand: it prints the file-level
|
||||||
// report and trees subcommands. Any database problem — including a
|
// duplicates report as TSV on stdout. SQLite groups and orders the
|
||||||
// missing database — is fatal. The error is returned rather than
|
// records, and each row is written as it is read, so no group is held
|
||||||
// exiting, so that the deferred close always runs; the database is
|
// in memory. It never touches the scanned filesystem; its only I/O is
|
||||||
// closed before the caller formats its output, so it stays closed even
|
// the database (with SQLite's temporary sort file), stdout, and stderr.
|
||||||
// if that output fails.
|
// Any database problem, including a missing database, is fatal.
|
||||||
func loadRecords(ctx context.Context) ([]scanRec, error) {
|
func runReport(ctx context.Context, stdout io.Writer) error {
|
||||||
dbPath := databasePath()
|
dbPath := databasePath()
|
||||||
|
|
||||||
db, err := openReportDatabase(ctx, dbPath)
|
db, err := openReportDatabase(ctx, dbPath)
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
recs, err := loadFileRows(ctx, db)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("database %s: %w", dbPath, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return recs, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
|
||||||
// head hash, tail hash, and content hash. paths is sorted
|
|
||||||
// lexicographically; the first entry is the group's "first", the rest
|
|
||||||
// are dupes.
|
|
||||||
type dupeGroup struct {
|
|
||||||
size int64
|
|
||||||
paths []string
|
|
||||||
}
|
|
||||||
|
|
||||||
// runReport implements the report subcommand: it reads every record
|
|
||||||
// from the database and prints the file-level duplicates report as TSV
|
|
||||||
// on stdout. It never touches the scanned filesystem; its only I/O is
|
|
||||||
// the database, stdout, and stderr.
|
|
||||||
func runReport(ctx context.Context, stdout io.Writer) error {
|
|
||||||
recs, err := loadRecords(ctx)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
dupes := collectDupeGroups(recs)
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||||
|
|
||||||
@@ -81,21 +51,36 @@ func runReport(ctx context.Context, stdout io.Writer) error {
|
|||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
dupeFiles := 0
|
var (
|
||||||
|
groups, dupeFiles int
|
||||||
|
reclaimable int64
|
||||||
|
writeErr error
|
||||||
|
)
|
||||||
|
|
||||||
var reclaimable int64
|
records, err := loadDupeRows(ctx, db,
|
||||||
|
func(first, path string, size int64) error {
|
||||||
|
// A group's first path is its first row; every other
|
||||||
|
// path is a dupe.
|
||||||
|
if path == first {
|
||||||
|
groups++
|
||||||
|
|
||||||
for _, g := range dupes {
|
return nil
|
||||||
for _, p := range g.paths[1:] {
|
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
|
||||||
escapePath(g.paths[0]), escapePath(p), g.size)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||||
|
escapePath(first), escapePath(path), size)
|
||||||
dupeFiles++
|
dupeFiles++
|
||||||
reclaimable += g.size
|
reclaimable += size
|
||||||
}
|
|
||||||
|
return writeErr
|
||||||
|
})
|
||||||
|
|
||||||
|
if writeErr != nil {
|
||||||
|
return fmt.Errorf("write stdout: %w", writeErr)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("database %s: %w", dbPath, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = out.Flush()
|
err = out.Flush()
|
||||||
@@ -106,57 +91,11 @@ func runReport(ctx context.Context, stdout io.Writer) error {
|
|||||||
fmt.Fprintf(os.Stderr,
|
fmt.Fprintf(os.Stderr,
|
||||||
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
||||||
"%s reclaimable\n",
|
"%s reclaimable\n",
|
||||||
len(recs), len(dupes), dupeFiles, humanBytes(reclaimable))
|
records, groups, dupeFiles, humanBytes(reclaimable))
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectDupeGroups groups records by signature and returns every group
|
|
||||||
// with two or more paths, each group's paths sorted lexicographically,
|
|
||||||
// groups ordered by size descending then by first path ascending.
|
|
||||||
func collectDupeGroups(recs []scanRec) []dupeGroup {
|
|
||||||
groups := make(map[fileSig][]string)
|
|
||||||
|
|
||||||
for _, r := range recs {
|
|
||||||
// A record without a content hash has unknown content and is
|
|
||||||
// never reported as a duplicate (README "Database").
|
|
||||||
if r.content == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
k := fileSig{
|
|
||||||
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
|
||||||
}
|
|
||||||
groups[k] = append(groups[k], r.path)
|
|
||||||
}
|
|
||||||
|
|
||||||
var dupes []dupeGroup
|
|
||||||
|
|
||||||
for k, paths := range groups {
|
|
||||||
if len(paths) < minGroupSize {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(paths)
|
|
||||||
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
|
|
||||||
}
|
|
||||||
|
|
||||||
// Biggest reclaimable space first; ties broken by first path.
|
|
||||||
slices.SortFunc(dupes, func(a, b dupeGroup) int {
|
|
||||||
if a.size != b.size {
|
|
||||||
if a.size > b.size {
|
|
||||||
return -1
|
|
||||||
}
|
|
||||||
|
|
||||||
return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
return strings.Compare(a.paths[0], b.paths[0])
|
|
||||||
})
|
|
||||||
|
|
||||||
return dupes
|
|
||||||
}
|
|
||||||
|
|
||||||
// escapePath returns a path as it is written in a report column (README
|
// escapePath returns a path as it is written in a report column (README
|
||||||
// "Report output format"): a backslash, tab, newline or carriage return
|
// "Report output format"): a backslash, tab, newline or carriage return
|
||||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
||||||
|
|||||||
+144
-15
@@ -2,10 +2,14 @@ package main
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -47,6 +51,53 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
|
|||||||
return path
|
return path
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dupeGroup is one duplicate group as report reads it: the size, and
|
||||||
|
// the paths in report order, first path first.
|
||||||
|
type dupeGroup struct {
|
||||||
|
size int64
|
||||||
|
paths []string
|
||||||
|
}
|
||||||
|
|
||||||
|
// dupeGroups returns the duplicate groups report reads from db, in
|
||||||
|
// report order.
|
||||||
|
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
var groups []dupeGroup
|
||||||
|
|
||||||
|
_, err := loadDupeRows(t.Context(), db,
|
||||||
|
func(first, path string, size int64) error {
|
||||||
|
if path == first {
|
||||||
|
groups = append(groups, dupeGroup{size: size})
|
||||||
|
}
|
||||||
|
|
||||||
|
g := &groups[len(groups)-1]
|
||||||
|
g.paths = append(g.paths, path)
|
||||||
|
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return groups
|
||||||
|
}
|
||||||
|
|
||||||
|
// dupeGroupsOf writes recs into a fresh database and returns the
|
||||||
|
// duplicate groups report reads from it.
|
||||||
|
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
|
||||||
|
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return dupeGroups(t, db)
|
||||||
|
}
|
||||||
|
|
||||||
func TestRunReportEscapesPaths(t *testing.T) {
|
func TestRunReportEscapesPaths(t *testing.T) {
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||||
|
|
||||||
@@ -65,6 +116,82 @@ func TestRunReportEscapesPaths(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestReportStdoutFailsWhileReading(t *testing.T) {
|
||||||
|
// Each row holds two paths longer than dir, so the report is more
|
||||||
|
// than twice the stdout buffer and stdout fails while rows are
|
||||||
|
// still being read, not at the final flush.
|
||||||
|
dir := "/" + strings.Repeat("d", 4096)
|
||||||
|
|
||||||
|
var recs []scanRec
|
||||||
|
for i := range ioBufSize / len(dir) {
|
||||||
|
recs = append(recs, scanRec{
|
||||||
|
size: 1, head: "h", tail: "t", content: "c",
|
||||||
|
path: fmt.Sprintf("%s/%d", dir, i),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||||
|
|
||||||
|
err := runReport(t.Context(), failingWriter{})
|
||||||
|
if !errors.Is(err, errWriteFailed) ||
|
||||||
|
!strings.HasPrefix(err.Error(), "write stdout: ") {
|
||||||
|
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
||||||
|
// README §Constraints: identical database contents give identical
|
||||||
|
// output, whatever order the records were inserted in.
|
||||||
|
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
||||||
|
recs = append(recs,
|
||||||
|
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
||||||
|
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
||||||
|
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
||||||
|
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
||||||
|
scanRec{size: 50, path: "/x/unhashed"},
|
||||||
|
)
|
||||||
|
|
||||||
|
reversed := slices.Clone(recs)
|
||||||
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
|
for _, name := range []string{cmdReport, cmdTrees} {
|
||||||
|
t.Run(name, func(t *testing.T) {
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||||
|
|
||||||
|
forward := runStdout(t, name)
|
||||||
|
|
||||||
|
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
||||||
|
|
||||||
|
backward := runStdout(t, name)
|
||||||
|
|
||||||
|
if strings.Count(forward, "\n") < 3 {
|
||||||
|
t.Errorf("stdout = %q, want at least two rows", forward)
|
||||||
|
}
|
||||||
|
|
||||||
|
if forward != backward {
|
||||||
|
t.Errorf("stdout depends on insertion order: %q vs %q",
|
||||||
|
forward, backward)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// runStdout runs the subcommand name and returns its stdout, failing
|
||||||
|
// the test unless it succeeds.
|
||||||
|
func runStdout(t *testing.T, name string) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
|
||||||
|
code := run([]string{name}, &stdout, &stderr)
|
||||||
|
if code != exitOK {
|
||||||
|
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
||||||
|
name, code, exitOK, stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
return stdout.String()
|
||||||
|
}
|
||||||
|
|
||||||
func TestEscapePath(t *testing.T) {
|
func TestEscapePath(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -121,7 +248,7 @@ func TestWarnfEscapes(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroups(t *testing.T) {
|
func TestDupeGroups(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
@@ -136,7 +263,7 @@ func TestCollectDupeGroups(t *testing.T) {
|
|||||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroupsOf(t, recs)
|
||||||
if len(groups) != 2 {
|
if len(groups) != 2 {
|
||||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||||
}
|
}
|
||||||
@@ -154,7 +281,7 @@ func TestCollectDupeGroups(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
func TestDupeGroupsContentSeparates(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// Same size, head, and tail, but different content hashes: the final
|
// Same size, head, and tail, but different content hashes: the final
|
||||||
@@ -169,7 +296,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
|||||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroupsOf(t, recs)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||||
len(groups))
|
len(groups))
|
||||||
@@ -180,7 +307,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// mtime is informational only; records differing only in mtime
|
// mtime is informational only; records differing only in mtime
|
||||||
@@ -190,23 +317,25 @@ func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
|||||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroupsOf(t, recs)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
func TestDupeGroupsTieBreak(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
// The hashes sort opposite to the first paths, so ordering the
|
||||||
|
// groups by hash instead of by first path fails this test.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
|
||||||
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
|
||||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
|
||||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroupsOf(t, recs)
|
||||||
if len(groups) != 2 {
|
if len(groups) != 2 {
|
||||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||||
}
|
}
|
||||||
@@ -218,7 +347,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
func TestDupeGroupsDeterministic(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
@@ -228,12 +357,12 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
|||||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
forward := collectDupeGroups(recs)
|
forward := dupeGroupsOf(t, recs)
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
reversed := slices.Clone(recs)
|
||||||
slices.Reverse(reversed)
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
backward := collectDupeGroups(reversed)
|
backward := dupeGroupsOf(t, reversed)
|
||||||
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
||||||
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
||||||
}) {
|
}) {
|
||||||
|
|||||||
+25
-24
@@ -400,7 +400,7 @@ func TestScanContentGate(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||||
t.Fatalf("groups = %+v, want only the identical pair %q",
|
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||||
groups, same)
|
groups, same)
|
||||||
@@ -435,7 +435,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
|
|||||||
want := []string{a, b}
|
want := []string{a, b}
|
||||||
slices.Sort(want)
|
slices.Sort(want)
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
}
|
}
|
||||||
@@ -462,7 +462,7 @@ func TestScanContentWithinOperand(t *testing.T) {
|
|||||||
|
|
||||||
want := []string{stored, added}
|
want := []string{stored, added}
|
||||||
|
|
||||||
groups := collectDupeGroups(dbRecords(t, db))
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
}
|
}
|
||||||
@@ -520,7 +520,7 @@ func TestScanContentStalePartners(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
if groups := dupeGroups(t, db); len(groups) != 0 {
|
||||||
t.Errorf("groups = %+v, want none", groups)
|
t.Errorf("groups = %+v, want none", groups)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -571,7 +571,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
|||||||
|
|
||||||
// The stored records lie outside the operand and are left as they
|
// The stored records lie outside the operand and are left as they
|
||||||
// are, so they still group with each other, but not with the copy.
|
// are, so they still group with each other, but not with the copy.
|
||||||
groups := collectDupeGroups(recs)
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||||
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||||
}
|
}
|
||||||
@@ -620,7 +620,7 @@ func TestScanContentReadFailure(t *testing.T) {
|
|||||||
want := []string{a, b}
|
want := []string{a, b}
|
||||||
slices.Sort(want)
|
slices.Sort(want)
|
||||||
|
|
||||||
groups := collectDupeGroups(dbRecords(t, db))
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||||
groups, want)
|
groups, want)
|
||||||
@@ -956,11 +956,16 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
|
|||||||
return st
|
return st
|
||||||
}
|
}
|
||||||
|
|
||||||
// dbRecords returns every record currently in the database.
|
// dbRecords returns every record currently in the database, in path
|
||||||
|
// order.
|
||||||
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
|
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
recs, err := loadFileRows(t.Context(), db)
|
var recs []scanRec
|
||||||
|
|
||||||
|
err := loadFileRows(t.Context(), db, func(r scanRec) {
|
||||||
|
recs = append(recs, r)
|
||||||
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
@@ -997,10 +1002,10 @@ func recordPaths(recs []scanRec) []string {
|
|||||||
|
|
||||||
// assertSmokeDupeGroups checks the file-level duplicate groups for the
|
// assertSmokeDupeGroups checks the file-level duplicate groups for the
|
||||||
// smoke tree rooted at dir.
|
// smoke tree rooted at dir.
|
||||||
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
|
func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
groups := collectDupeGroups(parsed)
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 5 {
|
if len(groups) != 5 {
|
||||||
t.Fatalf("len(groups) = %d, want 5", len(groups))
|
t.Fatalf("len(groups) = %d, want 5", len(groups))
|
||||||
}
|
}
|
||||||
@@ -1025,11 +1030,10 @@ func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
|
|||||||
|
|
||||||
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
|
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
|
||||||
// tree rooted at dir.
|
// tree rooted at dir.
|
||||||
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
|
func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
super, dirs := buildHierarchy(parsed)
|
super, dirs := dbTree(t, db)
|
||||||
super.compute()
|
|
||||||
|
|
||||||
tg := collectTreeGroups(dirs, super)
|
tg := collectTreeGroups(dirs, super)
|
||||||
if len(tg) != 1 {
|
if len(tg) != 1 {
|
||||||
@@ -1063,8 +1067,8 @@ func TestScanPipeline(t *testing.T) {
|
|||||||
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
|
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertSmokeDupeGroups(t, dir, parsed)
|
assertSmokeDupeGroups(t, dir, db)
|
||||||
assertSmokeTreeGroups(t, dir, parsed)
|
assertSmokeTreeGroups(t, dir, db)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSyncScanUnchangedReuse(t *testing.T) {
|
func TestSyncScanUnchangedReuse(t *testing.T) {
|
||||||
@@ -1298,7 +1302,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
if groups := dupeGroups(t, db); len(groups) != 0 {
|
||||||
t.Fatalf("groups = %+v, want none from unhashed records", groups)
|
t.Fatalf("groups = %+v, want none from unhashed records", groups)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1313,7 +1317,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
st)
|
st)
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(dbRecords(t, db))
|
groups := dupeGroups(t, db)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("groups = %+v, want the a/c pair", groups)
|
t.Fatalf("groups = %+v, want the a/c pair", groups)
|
||||||
}
|
}
|
||||||
@@ -1349,8 +1353,7 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
for name, unknown := range cases {
|
for name, unknown := range cases {
|
||||||
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
super, dirs := treeOf(t, append(slices.Clone(shared), unknown...))
|
||||||
super.compute()
|
|
||||||
|
|
||||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||||
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||||
@@ -1387,7 +1390,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
|
|||||||
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
|
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := collectDupeGroups(recs); len(groups) != 1 {
|
if groups := dupeGroups(t, db); len(groups) != 1 {
|
||||||
t.Fatalf("groups = %+v, want the hardlink pair", groups)
|
t.Fatalf("groups = %+v, want the hardlink pair", groups)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1628,10 +1631,8 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
recs := dbRecords(t, db)
|
assertSmokeDupeGroups(t, dir, db)
|
||||||
|
assertSmokeTreeGroups(t, dir, db)
|
||||||
assertSmokeDupeGroups(t, dir, recs)
|
|
||||||
assertSmokeTreeGroups(t, dir, recs)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestUnderRoot(t *testing.T) {
|
func TestUnderRoot(t *testing.T) {
|
||||||
|
|||||||
@@ -12,39 +12,48 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
// fileSig is a file's duplicate signature; mtime is excluded.
|
// treeNode is one directory reconstructed from the record paths.
|
||||||
type fileSig struct {
|
|
||||||
size int64
|
|
||||||
head string
|
|
||||||
tail string
|
|
||||||
content string
|
|
||||||
}
|
|
||||||
|
|
||||||
// treeNode is one directory reconstructed from the scan stream.
|
|
||||||
type treeNode struct {
|
type treeNode struct {
|
||||||
path string
|
path string
|
||||||
parent *treeNode
|
parent *treeNode
|
||||||
dirs map[string]*treeNode
|
// entries holds the serialized child entries until the digest is
|
||||||
files map[string]fileSig
|
// computed from them, and is then dropped.
|
||||||
|
entries []string
|
||||||
digest [sha256.Size]byte
|
digest [sha256.Size]byte
|
||||||
fileCount int64
|
fileCount int64
|
||||||
totalSize int64
|
totalSize int64
|
||||||
}
|
}
|
||||||
|
|
||||||
// runTrees implements the trees subcommand: it reads every record from
|
// runTrees implements the trees subcommand: it reads every record from
|
||||||
// the database, reconstructs the directory hierarchy from the record
|
// the database in path order, reconstructs the directory hierarchy from
|
||||||
// paths, computes a Merkle-style digest per directory, and prints
|
// the record paths, computes a Merkle-style digest per directory, and
|
||||||
// maximal duplicate-tree groups as TSV on stdout. It never touches the
|
// prints maximal duplicate-tree groups as TSV on stdout. It never
|
||||||
// scanned filesystem; its only I/O is the database, stdout, and
|
// touches the scanned filesystem; its only I/O is the database, stdout,
|
||||||
// stderr.
|
// and stderr. Any database problem, including a missing database, is
|
||||||
|
// fatal.
|
||||||
func runTrees(ctx context.Context, stdout io.Writer) error {
|
func runTrees(ctx context.Context, stdout io.Writer) error {
|
||||||
recs, err := loadRecords(ctx)
|
dbPath := databasePath()
|
||||||
|
|
||||||
|
db, err := openReportDatabase(ctx, dbPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
super, allDirs := buildHierarchy(recs)
|
defer func() { _ = db.Close() }()
|
||||||
super.compute()
|
|
||||||
|
records := 0
|
||||||
|
tree := newTreeBuilder()
|
||||||
|
|
||||||
|
err = loadFileRows(ctx, db, func(r scanRec) {
|
||||||
|
records++
|
||||||
|
|
||||||
|
tree.add(r)
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("database %s: %w", dbPath, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
super, allDirs := tree.finish()
|
||||||
|
|
||||||
dupes := collectTreeGroups(allDirs, super)
|
dupes := collectTreeGroups(allDirs, super)
|
||||||
|
|
||||||
@@ -82,73 +91,128 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
|
|||||||
fmt.Fprintf(os.Stderr,
|
fmt.Fprintf(os.Stderr,
|
||||||
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
||||||
"%s reclaimable\n",
|
"%s reclaimable\n",
|
||||||
len(recs), len(dupes), dupeTrees, humanBytes(reclaimable))
|
records, len(dupes), dupeTrees, humanBytes(reclaimable))
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
// treeBuilder reconstructs the directory hierarchy from records added
|
||||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
// in path order, under a synthetic super-root. Paths are split on "/";
|
||||||
// absolute paths the first component is empty, which becomes the
|
// for absolute paths the first component is empty, which becomes the
|
||||||
// top-level node with path "/". It returns the super-root and every
|
// top-level directory with path "/". In path order all the paths under
|
||||||
// directory node created.
|
// one directory come together, so a directory is complete once a path
|
||||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
// outside it is added: its digest is computed then and its entries are
|
||||||
|
// dropped. Only the directories holding the latest path keep entries.
|
||||||
|
type treeBuilder struct {
|
||||||
|
super *treeNode
|
||||||
|
// open lists the directories holding the latest path, outermost
|
||||||
|
// first, starting with the super-root; names[i] is open[i]'s name.
|
||||||
|
open []*treeNode
|
||||||
|
names []string
|
||||||
|
// dirs lists every completed directory.
|
||||||
|
dirs []*treeNode
|
||||||
|
}
|
||||||
|
|
||||||
|
func newTreeBuilder() *treeBuilder {
|
||||||
super := &treeNode{}
|
super := &treeNode{}
|
||||||
|
|
||||||
var allDirs []*treeNode
|
return &treeBuilder{
|
||||||
|
super: super,
|
||||||
|
open: []*treeNode{super},
|
||||||
|
names: []string{""},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
for _, r := range recs {
|
// add adds one record. Each record must come after the previous one in
|
||||||
comps := strings.Split(r.path, "/")
|
// path order (byte order); otherwise a completed directory would be
|
||||||
|
// started again as a second directory with the same path.
|
||||||
|
func (b *treeBuilder) add(r scanRec) {
|
||||||
|
comps := strings.Split(r.path, "/")
|
||||||
|
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
|
||||||
|
|
||||||
node := super
|
// Keep the open directories that hold this path; complete the rest.
|
||||||
for _, c := range comps[:len(comps)-1] {
|
depth := 1
|
||||||
child := node.dirs[c]
|
for depth < len(b.open) && depth <= len(dirNames) &&
|
||||||
if child == nil {
|
b.names[depth] == dirNames[depth-1] {
|
||||||
childPath := node.path + "/" + c
|
depth++
|
||||||
|
|
||||||
// The root directory's path is "/", not empty, and its
|
|
||||||
// children's paths start with one slash, not two.
|
|
||||||
switch {
|
|
||||||
case node == super && c == "":
|
|
||||||
childPath = "/"
|
|
||||||
case node == super:
|
|
||||||
childPath = c
|
|
||||||
case node.path == "/":
|
|
||||||
childPath = "/" + c
|
|
||||||
}
|
|
||||||
|
|
||||||
child = &treeNode{path: childPath, parent: node}
|
|
||||||
if node.dirs == nil {
|
|
||||||
node.dirs = make(map[string]*treeNode)
|
|
||||||
}
|
|
||||||
|
|
||||||
node.dirs[c] = child
|
|
||||||
allDirs = append(allDirs, child)
|
|
||||||
}
|
|
||||||
|
|
||||||
node = child
|
|
||||||
}
|
|
||||||
|
|
||||||
if node.files == nil {
|
|
||||||
node.files = make(map[string]fileSig)
|
|
||||||
}
|
|
||||||
|
|
||||||
sig := fileSig{
|
|
||||||
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
|
||||||
}
|
|
||||||
|
|
||||||
// A record without a content hash has unknown content (README
|
|
||||||
// "Database"): give it a signature no other file can share, so
|
|
||||||
// trees containing it never compare equal. Real hashes are
|
|
||||||
// hex, so the NUL-prefixed form cannot collide.
|
|
||||||
if sig.content == "" {
|
|
||||||
sig.content = "unhashed\x00" + r.path
|
|
||||||
}
|
|
||||||
|
|
||||||
node.files[comps[len(comps)-1]] = sig
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return super, allDirs
|
b.closeTo(depth)
|
||||||
|
|
||||||
|
for _, c := range dirNames[depth-1:] {
|
||||||
|
b.openDir(c)
|
||||||
|
}
|
||||||
|
|
||||||
|
dir := b.open[len(b.open)-1]
|
||||||
|
dir.entries = append(dir.entries, fileEntry(name, r))
|
||||||
|
dir.fileCount++
|
||||||
|
dir.totalSize += r.size
|
||||||
|
}
|
||||||
|
|
||||||
|
// openDir starts the directory called name inside the innermost open
|
||||||
|
// one.
|
||||||
|
func (b *treeBuilder) openDir(name string) {
|
||||||
|
parent := b.open[len(b.open)-1]
|
||||||
|
path := parent.path + "/" + name
|
||||||
|
|
||||||
|
// The root directory's path is "/", not empty, and its children's
|
||||||
|
// paths start with one slash, not two.
|
||||||
|
switch {
|
||||||
|
case parent == b.super && name == "":
|
||||||
|
path = "/"
|
||||||
|
case parent == b.super:
|
||||||
|
path = name
|
||||||
|
case parent.path == "/":
|
||||||
|
path = "/" + name
|
||||||
|
}
|
||||||
|
|
||||||
|
b.open = append(b.open, &treeNode{path: path, parent: parent})
|
||||||
|
b.names = append(b.names, name)
|
||||||
|
}
|
||||||
|
|
||||||
|
// closeTo completes the open directories after the first n, innermost
|
||||||
|
// first: each one's digest is computed and entered in its parent along
|
||||||
|
// with its totals.
|
||||||
|
func (b *treeBuilder) closeTo(n int) {
|
||||||
|
for len(b.open) > n {
|
||||||
|
last := len(b.open) - 1
|
||||||
|
dir, name := b.open[last], b.names[last]
|
||||||
|
b.open, b.names = b.open[:last], b.names[:last]
|
||||||
|
|
||||||
|
dir.computeDigest()
|
||||||
|
|
||||||
|
dir.parent.entries = append(dir.parent.entries,
|
||||||
|
"d\x00"+name+"\x00"+string(dir.digest[:]))
|
||||||
|
dir.parent.fileCount += dir.fileCount
|
||||||
|
dir.parent.totalSize += dir.totalSize
|
||||||
|
b.dirs = append(b.dirs, dir)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// finish completes every open directory and returns the super-root and
|
||||||
|
// every directory.
|
||||||
|
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
|
||||||
|
b.closeTo(1)
|
||||||
|
|
||||||
|
return b.super, b.dirs
|
||||||
|
}
|
||||||
|
|
||||||
|
// fileEntry serializes a file child for its directory's digest: its
|
||||||
|
// name and its signature (size, head, tail, content); mtime is
|
||||||
|
// excluded.
|
||||||
|
func fileEntry(name string, r scanRec) string {
|
||||||
|
content := r.content
|
||||||
|
|
||||||
|
// A record without a content hash has unknown content (README
|
||||||
|
// "Database"): give it a signature no other file can share, so
|
||||||
|
// trees containing it never compare equal. Real hashes are hex, so
|
||||||
|
// the NUL-prefixed form cannot collide.
|
||||||
|
if content == "" {
|
||||||
|
content = "unhashed\x00" + r.path
|
||||||
|
}
|
||||||
|
|
||||||
|
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
|
||||||
|
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectTreeGroups groups directories by digest and returns every
|
// collectTreeGroups groups directories by digest and returns every
|
||||||
@@ -190,38 +254,22 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
|
|||||||
return dupes
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
// compute fills in digest, fileCount, and totalSize for n and all of
|
// computeDigest sets n's digest and drops its entries. A directory's
|
||||||
// its descendants. A directory's digest is the SHA-256 of its child
|
// digest is the SHA-256 of its child entries — files serialized with
|
||||||
// entries — files serialized with name and signature, subdirectories
|
// name and signature, subdirectories with name and recursive digest —
|
||||||
// with name and recursive digest — sorted byte-lexicographically.
|
// sorted byte-lexicographically. Filenames cannot contain NUL or "/",
|
||||||
// Filenames cannot contain NUL or "/", so NUL delimiters are
|
// so NUL delimiters are unambiguous.
|
||||||
// unambiguous.
|
func (n *treeNode) computeDigest() {
|
||||||
func (n *treeNode) compute() {
|
slices.Sort(n.entries)
|
||||||
entries := make([]string, 0, len(n.dirs)+len(n.files))
|
|
||||||
for name, sig := range n.files {
|
|
||||||
entries = append(entries,
|
|
||||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
|
||||||
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
|
||||||
n.fileCount++
|
|
||||||
n.totalSize += sig.size
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, child := range n.dirs {
|
|
||||||
child.compute()
|
|
||||||
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
|
|
||||||
n.fileCount += child.fileCount
|
|
||||||
n.totalSize += child.totalSize
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(entries)
|
|
||||||
|
|
||||||
h := sha256.New()
|
h := sha256.New()
|
||||||
for _, e := range entries {
|
for _, e := range n.entries {
|
||||||
h.Write([]byte(e))
|
h.Write([]byte(e))
|
||||||
h.Write([]byte{0})
|
h.Write([]byte{0})
|
||||||
}
|
}
|
||||||
|
|
||||||
copy(n.digest[:], h.Sum(nil))
|
copy(n.digest[:], h.Sum(nil))
|
||||||
|
n.entries = nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
||||||
|
|||||||
+92
-19
@@ -2,6 +2,7 @@ package main
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
|
"database/sql"
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
@@ -30,6 +31,36 @@ func smokeTreeRecs() []scanRec {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dbTree builds the directory hierarchy from the records in db the way
|
||||||
|
// trees does, and returns the super-root and every directory.
|
||||||
|
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
tree := newTreeBuilder()
|
||||||
|
|
||||||
|
err := loadFileRows(t.Context(), db, tree.add)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return tree.finish()
|
||||||
|
}
|
||||||
|
|
||||||
|
// treeOf writes recs into a fresh database and builds the directory
|
||||||
|
// hierarchy from it the way trees does.
|
||||||
|
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
|
||||||
|
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return dbTree(t, db)
|
||||||
|
}
|
||||||
|
|
||||||
// nodeByPath finds the directory node with the given path.
|
// nodeByPath finds the directory node with the given path.
|
||||||
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
@@ -60,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string {
|
|||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestBuildHierarchyCounts(t *testing.T) {
|
func TestTreeCounts(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
_, dirs := treeOf(t, smokeTreeRecs())
|
||||||
super.compute()
|
|
||||||
|
|
||||||
d := nodeByPath(t, dirs, "/d")
|
d := nodeByPath(t, dirs, "/d")
|
||||||
if d.fileCount != 6 || d.totalSize != 9300 {
|
if d.fileCount != 6 || d.totalSize != 9300 {
|
||||||
@@ -85,12 +115,12 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestBuildHierarchyRootPath(t *testing.T) {
|
func TestTreeRootPath(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// The root directory's path is "/", never empty, and its
|
// The root directory's path is "/", never empty, and its
|
||||||
// children's paths start with a single slash.
|
// children's paths start with a single slash.
|
||||||
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
|
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
|
||||||
|
|
||||||
got := make([]string, 0, len(dirs))
|
got := make([]string, 0, len(dirs))
|
||||||
for _, d := range dirs {
|
for _, d := range dirs {
|
||||||
@@ -105,6 +135,56 @@ func TestBuildHierarchyRootPath(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
|
||||||
|
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
|
||||||
|
// directory must still be built once, whole, so /a matches /c.
|
||||||
|
recs := make([]scanRec, 0, 8)
|
||||||
|
|
||||||
|
for _, top := range []string{"/a", "/c"} {
|
||||||
|
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
|
||||||
|
content := "c"
|
||||||
|
if p == "/b-x/f" {
|
||||||
|
content = "other"
|
||||||
|
}
|
||||||
|
|
||||||
|
recs = append(recs, scanRec{
|
||||||
|
size: 1, head: "h", tail: "t", content: content, path: top + p,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
super, dirs := treeOf(t, recs)
|
||||||
|
|
||||||
|
got := make([]string, 0, len(dirs))
|
||||||
|
for _, d := range dirs {
|
||||||
|
got = append(got, d.path)
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(got)
|
||||||
|
|
||||||
|
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
|
||||||
|
if !slices.Equal(got, want) {
|
||||||
|
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
|
||||||
|
groups := collectTreeGroups(dirs, super)
|
||||||
|
|
||||||
|
gotGroups := groupPaths(groups)
|
||||||
|
wantGroups := [][]string{{"/a", "/c"}}
|
||||||
|
|
||||||
|
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
|
||||||
|
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
|
||||||
|
}
|
||||||
|
|
||||||
|
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
|
||||||
|
t.Errorf("group totals: %d files %d bytes, want 4 4",
|
||||||
|
groups[0][0].fileCount, groups[0][0].totalSize)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||||
|
|
||||||
@@ -126,8 +206,7 @@ func TestRunTreesEscapesPaths(t *testing.T) {
|
|||||||
func TestTreeDigests(t *testing.T) {
|
func TestTreeDigests(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
_, dirs := treeOf(t, smokeTreeRecs())
|
||||||
super.compute()
|
|
||||||
|
|
||||||
t1 := nodeByPath(t, dirs, "/d/t1")
|
t1 := nodeByPath(t, dirs, "/d/t1")
|
||||||
t2 := nodeByPath(t, dirs, "/d/t2")
|
t2 := nodeByPath(t, dirs, "/d/t2")
|
||||||
@@ -160,8 +239,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
_, dirs := treeOf(t, recs)
|
||||||
super.compute()
|
|
||||||
|
|
||||||
a := nodeByPath(t, dirs, "/r/a")
|
a := nodeByPath(t, dirs, "/r/a")
|
||||||
b := nodeByPath(t, dirs, "/r/b")
|
b := nodeByPath(t, dirs, "/r/b")
|
||||||
@@ -174,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
super, dirs := treeOf(t, smokeTreeRecs())
|
||||||
super.compute()
|
|
||||||
|
|
||||||
groups := collectTreeGroups(dirs, super)
|
groups := collectTreeGroups(dirs, super)
|
||||||
|
|
||||||
@@ -199,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
|
|||||||
|
|
||||||
recs := smokeTreeRecs()
|
recs := smokeTreeRecs()
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := treeOf(t, recs)
|
||||||
super.compute()
|
|
||||||
|
|
||||||
forward := groupPaths(collectTreeGroups(dirs, super))
|
forward := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
reversed := slices.Clone(recs)
|
||||||
slices.Reverse(reversed)
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
superR, dirsR := buildHierarchy(reversed)
|
superR, dirsR := treeOf(t, reversed)
|
||||||
superR.compute()
|
|
||||||
|
|
||||||
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
||||||
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
||||||
@@ -227,8 +302,7 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
|||||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := treeOf(t, recs)
|
||||||
super.compute()
|
|
||||||
|
|
||||||
got := groupPaths(collectTreeGroups(dirs, super))
|
got := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
@@ -250,8 +324,7 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
|||||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := treeOf(t, recs)
|
||||||
super.compute()
|
|
||||||
|
|
||||||
got := groupPaths(collectTreeGroups(dirs, super))
|
got := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user