Stream report and trees instead of loading every record (closes #14) #76

Merged
clawbot merged 1 commits from issue-14-stream-reports into next 2026-10-04 06:47:25 +02:00
9 changed files with 576 additions and 297 deletions
Showing only changes of commit dbc9afb4ef - Show all commits
+22 -6
View File
@@ -94,7 +94,15 @@ Goals, in order:
millions of files, ~150 TB filesystem, possibly slow or busy disks millions of files, ~150 TB filesystem, possibly slow or busy disks
(ZFS pool under resilver). Holding one small record (path, size, (ZFS pool under resilver). Holding one small record (path, size,
mtime) per file in memory during a scan is acceptable; holding mtime) per file in memory during a scan is acceptable; holding
every file's hashes is not (they stay in the database). every file's hashes is not (they stay in the database). The
reporting commands do not hold every file's hashes either: `report`
lets SQLite group and order the records and writes each row as it
reads it, so its memory does not grow with the database, and
`trees` reads the records in path order and keeps each directory's
path, digest and totals, plus the hashes of only the files in the
directories holding the record being read, so its memory grows with
the number of directories and with the size of the largest
directory.
3. **Scan incrementally, analyze offline.** The expensive filesystem 3. **Scan incrementally, analyze offline.** The expensive filesystem
scan maintains a persistent database; an unchanged file is never scan maintains a persistent database; an unchanged file is never
read again on a rescan, except to compute its content hash once a read again on a rescan, except to compute its content hash once a
@@ -184,7 +192,10 @@ All three subcommands operate on a single SQLite database file:
starts while a report is still reading waits for it up to the starts while a report is still reading waits for it up to the
10-second busy timeout, then fails; a `scan` that ends while a 10-second busy timeout, then fails; a `scan` that ends while a
report has the database open warns and leaves the database in WAL report has the database open warns and leaves the database in WAL
mode until the next scan. mode until the next scan. `report` writes each row as it reads it,
so it is still reading while its output is paused (a pager, a
stalled pipe), and a `scan` started then fails after the busy
timeout.
- `report` and `trees` open the database read-only and need only read - `report` and `trees` open the database read-only and need only read
access to the database file, and no write access to its directory. access to the database file, and no write access to its directory.
While the database is in WAL mode they also read the `-wal` and While the database is in WAL mode they also read the `-wal` and
@@ -202,6 +213,7 @@ All three subcommands operate on a single SQLite database file:
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
) WITHOUT ROWID; ) WITHOUT ROWID;
CREATE INDEX files_signature ON files (size, head, tail, content);
``` ```
Paths are stored as BLOBs because Unix paths are raw bytes, not Paths are stored as BLOBs because Unix paths are raw bytes, not
@@ -216,7 +228,8 @@ All three subcommands operate on a single SQLite database file:
until the content phase of a scan (see "`scan` mode" below) has until the content phase of a scan (see "`scan` mode" below) has
read the file. A record with an empty `content` is never part of a read the file. A record with an empty `content` is never part of a
duplicate group, though it still defines the file for tree duplicate group, though it still defines the file for tree
reconstruction. reconstruction. The `files_signature` index lets SQLite group the
records by signature for `report` without sorting the whole table.
### Duplicate detection ### Duplicate detection
@@ -444,9 +457,12 @@ arguments.
**`report` must never touch the filesystem being analyzed.** It does not **`report` must never touch the filesystem being analyzed.** It does not
stat, open, or otherwise access any path that appears in the records; its stat, open, or otherwise access any path that appears in the records; its
only I/O is reading the database and writing stdout/stderr. It must only I/O is reading the database, writing stdout/stderr, and the
produce identical output whether or not the scanned filesystem is still temporary file SQLite sorts in when the duplicate rows do not fit in
mounted. memory. SQLite puts that file in `$SQLITE_TMPDIR` or `$TMPDIR` when set,
otherwise in `/var/tmp` (or `/tmp`), and deletes it as soon as it has
opened it. `report` must produce identical output whether or not the
scanned filesystem is still mounted.
Processing: Processing:
+4
View File
@@ -29,6 +29,10 @@
# Completed Steps # Completed Steps
- `report` and `trees` stream the records instead of holding them all in
memory; the schema gains the `files_signature` index (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/14)
- progress prints at once on a non-terminal, uses a real terminal test, - progress prints at once on a non-terminal, uses a real terminal test,
and prints warnings through a spinner instead of racing its redraw and prints warnings through a spinner instead of racing its redraw
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13) (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13)
+94 -10
View File
@@ -51,6 +51,12 @@ CREATE TABLE files (
) WITHOUT ROWID ) WITHOUT ROWID
` `
// createIndexSQL indexes the records by signature, so report can have
// SQLite group them without sorting the whole table.
const createIndexSQL = `
CREATE INDEX files_signature ON files (size, head, tail, content)
`
// upsertSQL inserts one file record, replacing any existing record for // upsertSQL inserts one file record, replacing any existing record for
// the same path. // the same path.
const upsertSQL = ` const upsertSQL = `
@@ -256,6 +262,11 @@ func createSchema(ctx context.Context, db *sql.DB) error {
return fmt.Errorf("create schema: %w", err) return fmt.Errorf("create schema: %w", err)
} }
_, err = db.ExecContext(ctx, createIndexSQL)
if err != nil {
return fmt.Errorf("create schema: %w", err)
}
_, err = db.ExecContext(ctx, _, err = db.ExecContext(ctx,
"PRAGMA user_version = "+strconv.Itoa(schemaVersion)) "PRAGMA user_version = "+strconv.Itoa(schemaVersion))
if err != nil { if err != nil {
@@ -277,18 +288,18 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
return v, nil return v, nil
} }
// loadFileRows reads every record from the files table. // loadFileRows streams every record to fn in path order: byte order,
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) { // which is the order of the primary key, so SQLite does not sort.
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
rows, err := db.QueryContext(ctx, rows, err := db.QueryContext(ctx,
"SELECT path, size, mtime, head, tail, content FROM files") "SELECT path, size, mtime, head, tail, content FROM files "+
"ORDER BY path")
if err != nil { if err != nil {
return nil, fmt.Errorf("read records: %w", err) return fmt.Errorf("read records: %w", err)
} }
defer func() { _ = rows.Close() }() defer func() { _ = rows.Close() }()
var recs []scanRec
for rows.Next() { for rows.Next() {
var ( var (
path []byte path []byte
@@ -298,19 +309,92 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
&r.content) &r.content)
if err != nil { if err != nil {
return nil, fmt.Errorf("read record: %w", err) return fmt.Errorf("read record: %w", err)
} }
r.path = string(path) r.path = string(path)
recs = append(recs, r) fn(r)
} }
err = rows.Err() err = rows.Err()
if err != nil { if err != nil {
return nil, fmt.Errorf("read records: %w", err) return fmt.Errorf("read records: %w", err)
} }
return recs, nil return nil
}
// dupeRowsSQL selects every record in a duplicate group, with the
// group's first path. A group is the records with a content hash that
// share a size, head, tail, and content, when there are two or more of
// them. The rows come in report order: groups by size descending, then
// by first path, and each group's paths ascending.
const dupeRowsSQL = `
SELECT g.first, f.path, f.size
FROM files AS f
JOIN (
SELECT size, head, tail, content, MIN(path) AS first
FROM files
WHERE content <> ''
GROUP BY size, head, tail, content
HAVING COUNT(*) > 1
) AS g USING (size, head, tail, content)
ORDER BY f.size DESC, g.first, f.path
`
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
// number of records in the database. The count and the rows are read
// in one transaction, so they agree while a scan is committing. An
// error from fn stops the reading and is returned as it is.
func loadDupeRows(ctx context.Context, db *sql.DB,
fn func(first, path string, size int64) error,
) (int, error) {
// Everything goes through tx: the report connection is the only
// one, so a query on db would wait for tx forever.
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
defer func() { _ = tx.Rollback() }()
var records int
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
defer func() { _ = rows.Close() }()
for rows.Next() {
var (
first, path []byte
size int64
)
err = rows.Scan(&first, &path, &size)
if err != nil {
return 0, fmt.Errorf("read record: %w", err)
}
err = fn(string(first), string(path), size)
if err != nil {
return 0, err
}
}
err = rows.Err()
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
return records, nil
} }
// loadFileMeta streams every record's path, size, mtime, and whether // loadFileMeta streams every record's path, size, mtime, and whether
+10 -25
View File
@@ -8,7 +8,6 @@ import (
"os" "os"
"path/filepath" "path/filepath"
"slices" "slices"
"strings"
"testing" "testing"
) )
@@ -73,9 +72,8 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
defer func() { _ = db.Close() }() defer func() { _ = db.Close() }()
recs, err := loadFileRows(t.Context(), db) if recs := dbRecords(t, db); len(recs) != 0 {
if err != nil || len(recs) != 0 { t.Fatalf("records = %v, want none", recs)
t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
} }
} }
@@ -168,7 +166,7 @@ func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
defer func() { _ = reportDB.Close() }() defer func() { _ = reportDB.Close() }()
_, err = loadFileRows(t.Context(), reportDB) err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
if err != nil { if err != nil {
t.Fatalf("loadFileRows: %v", err) t.Fatalf("loadFileRows: %v", err)
} }
@@ -196,15 +194,8 @@ func TestApplyChangesRoundTrip(t *testing.T) {
t.Fatalf("applyChanges: %v", err) t.Fatalf("applyChanges: %v", err)
} }
got, err := loadFileRows(t.Context(), db) // The records come back in path order, which is the order of recs.
if err != nil { got := dbRecords(t, db)
t.Fatal(err)
}
slices.SortFunc(got, func(a, b scanRec) int {
return strings.Compare(a.path, b.path)
})
if !slices.Equal(got, recs) { if !slices.Equal(got, recs) {
t.Fatalf("rows = %+v, want %+v", got, recs) t.Fatalf("rows = %+v, want %+v", got, recs)
} }
@@ -221,11 +212,7 @@ func TestApplyChangesRoundTrip(t *testing.T) {
t.Fatalf("applyChanges: %v", err) t.Fatalf("applyChanges: %v", err)
} }
got, err = loadFileRows(t.Context(), db) got = dbRecords(t, db)
if err != nil {
t.Fatal(err)
}
if len(got) != 1 || got[0] != upd { if len(got) != 1 || got[0] != upd {
t.Fatalf("rows = %+v, want just %+v", got, upd) t.Fatalf("rows = %+v, want just %+v", got, upd)
} }
@@ -254,9 +241,8 @@ func TestApplyChangesBatching(t *testing.T) {
t.Fatalf("applyChanges: %v", err) t.Fatalf("applyChanges: %v", err)
} }
got, err := loadFileRows(t.Context(), db) if got := dbRecords(t, db); len(got) != n {
if err != nil || len(got) != n { t.Fatalf("records = %d, want %d", len(got), n)
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
} }
deletes := make([]string, 0, n) deletes := make([]string, 0, n)
@@ -270,8 +256,7 @@ func TestApplyChangesBatching(t *testing.T) {
t.Fatalf("applyChanges deletes: %v", err) t.Fatalf("applyChanges deletes: %v", err)
} }
got, err = loadFileRows(t.Context(), db) if got := dbRecords(t, db); len(got) != 0 {
if err != nil || len(got) != 0 { t.Fatalf("records = %d, want 0", len(got))
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
} }
} }
+34 -95
View File
@@ -6,7 +6,6 @@ import (
"fmt" "fmt"
"io" "io"
"os" "os"
"slices"
"strings" "strings"
) )
@@ -29,50 +28,21 @@ type scanRec struct {
path string path string
} }
// loadRecords opens the database and reads every file record for the // runReport implements the report subcommand: it prints the file-level
// report and trees subcommands. Any database problem — including a // duplicates report as TSV on stdout. SQLite groups and orders the
// missing database — is fatal. The error is returned rather than // records, and each row is written as it is read, so no group is held
// exiting, so that the deferred close always runs; the database is // in memory. It never touches the scanned filesystem; its only I/O is
// closed before the caller formats its output, so it stays closed even // the database (with SQLite's temporary sort file), stdout, and stderr.
// if that output fails. // Any database problem, including a missing database, is fatal.
func loadRecords(ctx context.Context) ([]scanRec, error) { func runReport(ctx context.Context, stdout io.Writer) error {
dbPath := databasePath() dbPath := databasePath()
db, err := openReportDatabase(ctx, dbPath) db, err := openReportDatabase(ctx, dbPath)
if err != nil {
return nil, err
}
defer func() { _ = db.Close() }()
recs, err := loadFileRows(ctx, db)
if err != nil {
return nil, fmt.Errorf("database %s: %w", dbPath, err)
}
return recs, nil
}
// dupeGroup is one set of candidate-duplicate files: identical size,
// head hash, tail hash, and content hash. paths is sorted
// lexicographically; the first entry is the group's "first", the rest
// are dupes.
type dupeGroup struct {
size int64
paths []string
}
// runReport implements the report subcommand: it reads every record
// from the database and prints the file-level duplicates report as TSV
// on stdout. It never touches the scanned filesystem; its only I/O is
// the database, stdout, and stderr.
func runReport(ctx context.Context, stdout io.Writer) error {
recs, err := loadRecords(ctx)
if err != nil { if err != nil {
return err return err
} }
dupes := collectDupeGroups(recs) defer func() { _ = db.Close() }()
out := bufio.NewWriterSize(stdout, ioBufSize) out := bufio.NewWriterSize(stdout, ioBufSize)
@@ -81,21 +51,36 @@ func runReport(ctx context.Context, stdout io.Writer) error {
return fmt.Errorf("write stdout: %w", err) return fmt.Errorf("write stdout: %w", err)
} }
dupeFiles := 0 var (
groups, dupeFiles int
reclaimable int64
writeErr error
)
var reclaimable int64 records, err := loadDupeRows(ctx, db,
func(first, path string, size int64) error {
// A group's first path is its first row; every other
// path is a dupe.
if path == first {
groups++
for _, g := range dupes { return nil
for _, p := range g.paths[1:] {
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(g.paths[0]), escapePath(p), g.size)
if err != nil {
return fmt.Errorf("write stdout: %w", err)
} }
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(first), escapePath(path), size)
dupeFiles++ dupeFiles++
reclaimable += g.size reclaimable += size
}
return writeErr
})
if writeErr != nil {
return fmt.Errorf("write stdout: %w", writeErr)
}
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
} }
err = out.Flush() err = out.Flush()
@@ -106,57 +91,11 @@ func runReport(ctx context.Context, stdout io.Writer) error {
fmt.Fprintf(os.Stderr, fmt.Fprintf(os.Stderr,
"report: %d records read, %d duplicate groups, %d dupe files, "+ "report: %d records read, %d duplicate groups, %d dupe files, "+
"%s reclaimable\n", "%s reclaimable\n",
len(recs), len(dupes), dupeFiles, humanBytes(reclaimable)) records, groups, dupeFiles, humanBytes(reclaimable))
return nil return nil
} }
// collectDupeGroups groups records by signature and returns every group
// with two or more paths, each group's paths sorted lexicographically,
// groups ordered by size descending then by first path ascending.
func collectDupeGroups(recs []scanRec) []dupeGroup {
groups := make(map[fileSig][]string)
for _, r := range recs {
// A record without a content hash has unknown content and is
// never reported as a duplicate (README "Database").
if r.content == "" {
continue
}
k := fileSig{
size: r.size, head: r.head, tail: r.tail, content: r.content,
}
groups[k] = append(groups[k], r.path)
}
var dupes []dupeGroup
for k, paths := range groups {
if len(paths) < minGroupSize {
continue
}
slices.Sort(paths)
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
}
// Biggest reclaimable space first; ties broken by first path.
slices.SortFunc(dupes, func(a, b dupeGroup) int {
if a.size != b.size {
if a.size > b.size {
return -1
}
return 1
}
return strings.Compare(a.paths[0], b.paths[0])
})
return dupes
}
// escapePath returns a path as it is written in a report column (README // escapePath returns a path as it is written in a report column (README
// "Report output format"): a backslash, tab, newline or carriage return // "Report output format"): a backslash, tab, newline or carriage return
// becomes \\, \t, \n or \r, and every other byte is kept as it is. // becomes \\, \t, \n or \r, and every other byte is kept as it is.
+144 -15
View File
@@ -2,10 +2,14 @@ package main
import ( import (
"bytes" "bytes"
"database/sql"
"errors"
"fmt"
"io" "io"
"os" "os"
"path/filepath" "path/filepath"
"slices" "slices"
"strings"
"testing" "testing"
) )
@@ -47,6 +51,53 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
return path return path
} }
// dupeGroup is one duplicate group as report reads it: the size, and
// the paths in report order, first path first.
type dupeGroup struct {
size int64
paths []string
}
// dupeGroups returns the duplicate groups report reads from db, in
// report order.
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
t.Helper()
var groups []dupeGroup
_, err := loadDupeRows(t.Context(), db,
func(first, path string, size int64) error {
if path == first {
groups = append(groups, dupeGroup{size: size})
}
g := &groups[len(groups)-1]
g.paths = append(g.paths, path)
return nil
})
if err != nil {
t.Fatal(err)
}
return groups
}
// dupeGroupsOf writes recs into a fresh database and returns the
// duplicate groups report reads from it.
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dupeGroups(t, db)
}
func TestRunReportEscapesPaths(t *testing.T) { func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -65,6 +116,82 @@ func TestRunReportEscapesPaths(t *testing.T) {
} }
} }
func TestReportStdoutFailsWhileReading(t *testing.T) {
// Each row holds two paths longer than dir, so the report is more
// than twice the stdout buffer and stdout fails while rows are
// still being read, not at the final flush.
dir := "/" + strings.Repeat("d", 4096)
recs := make([]scanRec, ioBufSize/len(dir))
for i := range recs {
recs[i] = scanRec{
size: 1, head: "h", tail: "t", content: "c",
path: fmt.Sprintf("%s/%d", dir, i),
}
}
t.Setenv(databaseEnv, seedDatabase(t, recs))
err := runReport(t.Context(), failingWriter{})
if !errors.Is(err, errWriteFailed) ||
!strings.HasPrefix(err.Error(), "write stdout: ") {
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
}
}
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
// README §Constraints: identical database contents give identical
// output, whatever order the records were inserted in.
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
recs = append(recs,
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
scanRec{size: 50, path: "/x/unhashed"},
)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, recs))
forward := runStdout(t, name)
t.Setenv(databaseEnv, seedDatabase(t, reversed))
backward := runStdout(t, name)
if strings.Count(forward, "\n") < 3 {
t.Errorf("stdout = %q, want at least two rows", forward)
}
if forward != backward {
t.Errorf("stdout depends on insertion order: %q vs %q",
forward, backward)
}
})
}
}
// runStdout runs the subcommand name and returns its stdout, failing
// the test unless it succeeds.
func runStdout(t *testing.T, name string) string {
t.Helper()
var stdout, stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
}
return stdout.String()
}
func TestEscapePath(t *testing.T) { func TestEscapePath(t *testing.T) {
t.Parallel() t.Parallel()
@@ -121,7 +248,7 @@ func TestWarnfEscapes(t *testing.T) {
} }
} }
func TestCollectDupeGroups(t *testing.T) { func TestDupeGroups(t *testing.T) {
t.Parallel() t.Parallel()
recs := []scanRec{ recs := []scanRec{
@@ -136,7 +263,7 @@ func TestCollectDupeGroups(t *testing.T) {
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"}, {size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
} }
groups := collectDupeGroups(recs) groups := dupeGroupsOf(t, recs)
if len(groups) != 2 { if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups)) t.Fatalf("len(groups) = %d, want 2", len(groups))
} }
@@ -154,7 +281,7 @@ func TestCollectDupeGroups(t *testing.T) {
} }
} }
func TestCollectDupeGroupsContentSeparates(t *testing.T) { func TestDupeGroupsContentSeparates(t *testing.T) {
t.Parallel() t.Parallel()
// Same size, head, and tail, but different content hashes: the final // Same size, head, and tail, but different content hashes: the final
@@ -169,7 +296,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
{size: 100, head: "h", tail: "t", path: "/e"}, {size: 100, head: "h", tail: "t", path: "/e"},
} }
groups := collectDupeGroups(recs) groups := dupeGroupsOf(t, recs)
if len(groups) != 1 { if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1 (only the matching content)", t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
len(groups)) len(groups))
@@ -180,7 +307,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
} }
} }
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) { func TestDupeGroupsMtimeExcluded(t *testing.T) {
t.Parallel() t.Parallel()
// mtime is informational only; records differing only in mtime // mtime is informational only; records differing only in mtime
@@ -190,23 +317,25 @@ func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"}, {size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
} }
groups := collectDupeGroups(recs) groups := dupeGroupsOf(t, recs)
if len(groups) != 1 { if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1", len(groups)) t.Fatalf("len(groups) = %d, want 1", len(groups))
} }
} }
func TestCollectDupeGroupsTieBreak(t *testing.T) { func TestDupeGroupsTieBreak(t *testing.T) {
t.Parallel() t.Parallel()
// The hashes sort opposite to the first paths, so ordering the
// groups by hash instead of by first path fails this test.
recs := []scanRec{ recs := []scanRec{
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"}, {size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"}, {size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"}, {size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"}, {size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
} }
groups := collectDupeGroups(recs) groups := dupeGroupsOf(t, recs)
if len(groups) != 2 { if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups)) t.Fatalf("len(groups) = %d, want 2", len(groups))
} }
@@ -218,7 +347,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
} }
} }
func TestCollectDupeGroupsDeterministic(t *testing.T) { func TestDupeGroupsDeterministic(t *testing.T) {
t.Parallel() t.Parallel()
recs := []scanRec{ recs := []scanRec{
@@ -228,12 +357,12 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"}, {size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
} }
forward := collectDupeGroups(recs) forward := dupeGroupsOf(t, recs)
reversed := slices.Clone(recs) reversed := slices.Clone(recs)
slices.Reverse(reversed) slices.Reverse(reversed)
backward := collectDupeGroups(reversed) backward := dupeGroupsOf(t, reversed)
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool { if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
return a.size == b.size && slices.Equal(a.paths, b.paths) return a.size == b.size && slices.Equal(a.paths, b.paths)
}) { }) {
+25 -24
View File
@@ -400,7 +400,7 @@ func TestScanContentGate(t *testing.T) {
} }
} }
groups := collectDupeGroups(recs) groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) { if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
t.Fatalf("groups = %+v, want only the identical pair %q", t.Fatalf("groups = %+v, want only the identical pair %q",
groups, same) groups, same)
@@ -435,7 +435,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
want := []string{a, b} want := []string{a, b}
slices.Sort(want) slices.Sort(want)
groups := collectDupeGroups(recs) groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want) t.Fatalf("groups = %+v, want the pair %q", groups, want)
} }
@@ -462,7 +462,7 @@ func TestScanContentWithinOperand(t *testing.T) {
want := []string{stored, added} want := []string{stored, added}
groups := collectDupeGroups(dbRecords(t, db)) groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want) t.Fatalf("groups = %+v, want the pair %q", groups, want)
} }
@@ -520,7 +520,7 @@ func TestScanContentStalePartners(t *testing.T) {
} }
} }
if groups := collectDupeGroups(recs); len(groups) != 0 { if groups := dupeGroups(t, db); len(groups) != 0 {
t.Errorf("groups = %+v, want none", groups) t.Errorf("groups = %+v, want none", groups)
} }
} }
@@ -571,7 +571,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
// The stored records lie outside the operand and are left as they // The stored records lie outside the operand and are left as they
// are, so they still group with each other, but not with the copy. // are, so they still group with each other, but not with the copy.
groups := collectDupeGroups(recs) groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) { if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored) t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
} }
@@ -620,7 +620,7 @@ func TestScanContentReadFailure(t *testing.T) {
want := []string{a, b} want := []string{a, b}
slices.Sort(want) slices.Sort(want)
groups := collectDupeGroups(dbRecords(t, db)) groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q after the retry", t.Fatalf("groups = %+v, want the pair %q after the retry",
groups, want) groups, want)
@@ -956,11 +956,16 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
return st return st
} }
// dbRecords returns every record currently in the database. // dbRecords returns every record currently in the database, in path
// order.
func dbRecords(t *testing.T, db *sql.DB) []scanRec { func dbRecords(t *testing.T, db *sql.DB) []scanRec {
t.Helper() t.Helper()
recs, err := loadFileRows(t.Context(), db) var recs []scanRec
err := loadFileRows(t.Context(), db, func(r scanRec) {
recs = append(recs, r)
})
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
@@ -997,10 +1002,10 @@ func recordPaths(recs []scanRec) []string {
// assertSmokeDupeGroups checks the file-level duplicate groups for the // assertSmokeDupeGroups checks the file-level duplicate groups for the
// smoke tree rooted at dir. // smoke tree rooted at dir.
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) { func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
t.Helper() t.Helper()
groups := collectDupeGroups(parsed) groups := dupeGroups(t, db)
if len(groups) != 5 { if len(groups) != 5 {
t.Fatalf("len(groups) = %d, want 5", len(groups)) t.Fatalf("len(groups) = %d, want 5", len(groups))
} }
@@ -1025,11 +1030,10 @@ func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke // assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
// tree rooted at dir. // tree rooted at dir.
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) { func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) {
t.Helper() t.Helper()
super, dirs := buildHierarchy(parsed) super, dirs := dbTree(t, db)
super.compute()
tg := collectTreeGroups(dirs, super) tg := collectTreeGroups(dirs, super)
if len(tg) != 1 { if len(tg) != 1 {
@@ -1063,8 +1067,8 @@ func TestScanPipeline(t *testing.T) {
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles) t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
} }
assertSmokeDupeGroups(t, dir, parsed) assertSmokeDupeGroups(t, dir, db)
assertSmokeTreeGroups(t, dir, parsed) assertSmokeTreeGroups(t, dir, db)
} }
func TestSyncScanUnchangedReuse(t *testing.T) { func TestSyncScanUnchangedReuse(t *testing.T) {
@@ -1298,7 +1302,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
} }
} }
if groups := collectDupeGroups(recs); len(groups) != 0 { if groups := dupeGroups(t, db); len(groups) != 0 {
t.Fatalf("groups = %+v, want none from unhashed records", groups) t.Fatalf("groups = %+v, want none from unhashed records", groups)
} }
@@ -1313,7 +1317,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
st) st)
} }
groups := collectDupeGroups(dbRecords(t, db)) groups := dupeGroups(t, db)
if len(groups) != 1 { if len(groups) != 1 {
t.Fatalf("groups = %+v, want the a/c pair", groups) t.Fatalf("groups = %+v, want the a/c pair", groups)
} }
@@ -1349,8 +1353,7 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
} }
for name, unknown := range cases { for name, unknown := range cases {
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...)) super, dirs := treeOf(t, append(slices.Clone(shared), unknown...))
super.compute()
if tg := collectTreeGroups(dirs, super); len(tg) != 0 { if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)", t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
@@ -1387,7 +1390,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb) t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
} }
if groups := collectDupeGroups(recs); len(groups) != 1 { if groups := dupeGroups(t, db); len(groups) != 1 {
t.Fatalf("groups = %+v, want the hardlink pair", groups) t.Fatalf("groups = %+v, want the hardlink pair", groups)
} }
} }
@@ -1628,10 +1631,8 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
t.Fatal(err) t.Fatal(err)
} }
recs := dbRecords(t, db) assertSmokeDupeGroups(t, dir, db)
assertSmokeTreeGroups(t, dir, db)
assertSmokeDupeGroups(t, dir, recs)
assertSmokeTreeGroups(t, dir, recs)
} }
func TestUnderRoot(t *testing.T) { func TestUnderRoot(t *testing.T) {
+151 -103
View File
@@ -12,39 +12,48 @@ import (
"strings" "strings"
) )
// fileSig is a file's duplicate signature; mtime is excluded. // treeNode is one directory reconstructed from the record paths.
type fileSig struct {
size int64
head string
tail string
content string
}
// treeNode is one directory reconstructed from the scan stream.
type treeNode struct { type treeNode struct {
path string path string
parent *treeNode parent *treeNode
dirs map[string]*treeNode // entries holds the serialized child entries until the digest is
files map[string]fileSig // computed from them, and is then dropped.
entries []string
digest [sha256.Size]byte digest [sha256.Size]byte
fileCount int64 fileCount int64
totalSize int64 totalSize int64
} }
// runTrees implements the trees subcommand: it reads every record from // runTrees implements the trees subcommand: it reads every record from
// the database, reconstructs the directory hierarchy from the record // the database in path order, reconstructs the directory hierarchy from
// paths, computes a Merkle-style digest per directory, and prints // the record paths, computes a Merkle-style digest per directory, and
// maximal duplicate-tree groups as TSV on stdout. It never touches the // prints maximal duplicate-tree groups as TSV on stdout. It never
// scanned filesystem; its only I/O is the database, stdout, and // touches the scanned filesystem; its only I/O is the database, stdout,
// stderr. // and stderr. Any database problem, including a missing database, is
// fatal.
func runTrees(ctx context.Context, stdout io.Writer) error { func runTrees(ctx context.Context, stdout io.Writer) error {
recs, err := loadRecords(ctx) dbPath := databasePath()
db, err := openReportDatabase(ctx, dbPath)
if err != nil { if err != nil {
return err return err
} }
super, allDirs := buildHierarchy(recs) defer func() { _ = db.Close() }()
super.compute()
records := 0
tree := newTreeBuilder()
err = loadFileRows(ctx, db, func(r scanRec) {
records++
tree.add(r)
})
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
}
super, allDirs := tree.finish()
dupes := collectTreeGroups(allDirs, super) dupes := collectTreeGroups(allDirs, super)
@@ -82,73 +91,128 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
fmt.Fprintf(os.Stderr, fmt.Fprintf(os.Stderr,
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+ "trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
"%s reclaimable\n", "%s reclaimable\n",
len(recs), len(dupes), dupeTrees, humanBytes(reclaimable)) records, len(dupes), dupeTrees, humanBytes(reclaimable))
return nil return nil
} }
// buildHierarchy reconstructs the directory hierarchy from the record // treeBuilder reconstructs the directory hierarchy from records added
// paths under a synthetic super-root. Paths are split on "/"; for // in path order, under a synthetic super-root. Paths are split on "/";
// absolute paths the first component is empty, which becomes the // for absolute paths the first component is empty, which becomes the
// top-level node with path "/". It returns the super-root and every // top-level directory with path "/". In path order all the paths under
// directory node created. // one directory come together, so a directory is complete once a path
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) { // outside it is added: its digest is computed then and its entries are
// dropped. Only the directories holding the latest path keep entries.
type treeBuilder struct {
super *treeNode
// open lists the directories holding the latest path, outermost
// first, starting with the super-root; names[i] is open[i]'s name.
open []*treeNode
names []string
// dirs lists every completed directory.
dirs []*treeNode
}
func newTreeBuilder() *treeBuilder {
super := &treeNode{} super := &treeNode{}
var allDirs []*treeNode return &treeBuilder{
super: super,
open: []*treeNode{super},
names: []string{""},
}
}
for _, r := range recs { // add adds one record. Each record must come after the previous one in
comps := strings.Split(r.path, "/") // path order (byte order); otherwise a completed directory would be
// started again as a second directory with the same path.
func (b *treeBuilder) add(r scanRec) {
comps := strings.Split(r.path, "/")
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
node := super // Keep the open directories that hold this path; complete the rest.
for _, c := range comps[:len(comps)-1] { depth := 1
child := node.dirs[c] for depth < len(b.open) && depth <= len(dirNames) &&
if child == nil { b.names[depth] == dirNames[depth-1] {
childPath := node.path + "/" + c depth++
// The root directory's path is "/", not empty, and its
// children's paths start with one slash, not two.
switch {
case node == super && c == "":
childPath = "/"
case node == super:
childPath = c
case node.path == "/":
childPath = "/" + c
}
child = &treeNode{path: childPath, parent: node}
if node.dirs == nil {
node.dirs = make(map[string]*treeNode)
}
node.dirs[c] = child
allDirs = append(allDirs, child)
}
node = child
}
if node.files == nil {
node.files = make(map[string]fileSig)
}
sig := fileSig{
size: r.size, head: r.head, tail: r.tail, content: r.content,
}
// A record without a content hash has unknown content (README
// "Database"): give it a signature no other file can share, so
// trees containing it never compare equal. Real hashes are
// hex, so the NUL-prefixed form cannot collide.
if sig.content == "" {
sig.content = "unhashed\x00" + r.path
}
node.files[comps[len(comps)-1]] = sig
} }
return super, allDirs b.closeTo(depth)
for _, c := range dirNames[depth-1:] {
b.openDir(c)
}
dir := b.open[len(b.open)-1]
dir.entries = append(dir.entries, fileEntry(name, r))
dir.fileCount++
dir.totalSize += r.size
}
// openDir starts the directory called name inside the innermost open
// one.
func (b *treeBuilder) openDir(name string) {
parent := b.open[len(b.open)-1]
path := parent.path + "/" + name
// The root directory's path is "/", not empty, and its children's
// paths start with one slash, not two.
switch {
case parent == b.super && name == "":
path = "/"
case parent == b.super:
path = name
case parent.path == "/":
path = "/" + name
}
b.open = append(b.open, &treeNode{path: path, parent: parent})
b.names = append(b.names, name)
}
// closeTo completes the open directories after the first n, innermost
// first: each one's digest is computed and entered in its parent along
// with its totals.
func (b *treeBuilder) closeTo(n int) {
for len(b.open) > n {
last := len(b.open) - 1
dir, name := b.open[last], b.names[last]
b.open, b.names = b.open[:last], b.names[:last]
dir.computeDigest()
dir.parent.entries = append(dir.parent.entries,
"d\x00"+name+"\x00"+string(dir.digest[:]))
dir.parent.fileCount += dir.fileCount
dir.parent.totalSize += dir.totalSize
b.dirs = append(b.dirs, dir)
}
}
// finish completes every open directory and returns the super-root and
// every directory.
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
b.closeTo(1)
return b.super, b.dirs
}
// fileEntry serializes a file child for its directory's digest: its
// name and its signature (size, head, tail, content); mtime is
// excluded.
func fileEntry(name string, r scanRec) string {
content := r.content
// A record without a content hash has unknown content (README
// "Database"): give it a signature no other file can share, so
// trees containing it never compare equal. Real hashes are hex, so
// the NUL-prefixed form cannot collide.
if content == "" {
content = "unhashed\x00" + r.path
}
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
} }
// collectTreeGroups groups directories by digest and returns every // collectTreeGroups groups directories by digest and returns every
@@ -190,38 +254,22 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
return dupes return dupes
} }
// compute fills in digest, fileCount, and totalSize for n and all of // computeDigest sets n's digest and drops its entries. A directory's
// its descendants. A directory's digest is the SHA-256 of its child // digest is the SHA-256 of its child entries — files serialized with
// entries — files serialized with name and signature, subdirectories // name and signature, subdirectories with name and recursive digest —
// with name and recursive digest — sorted byte-lexicographically. // sorted byte-lexicographically. Filenames cannot contain NUL or "/",
// Filenames cannot contain NUL or "/", so NUL delimiters are // so NUL delimiters are unambiguous.
// unambiguous. func (n *treeNode) computeDigest() {
func (n *treeNode) compute() { slices.Sort(n.entries)
entries := make([]string, 0, len(n.dirs)+len(n.files))
for name, sig := range n.files {
entries = append(entries,
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
n.fileCount++
n.totalSize += sig.size
}
for name, child := range n.dirs {
child.compute()
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
n.fileCount += child.fileCount
n.totalSize += child.totalSize
}
slices.Sort(entries)
h := sha256.New() h := sha256.New()
for _, e := range entries { for _, e := range n.entries {
h.Write([]byte(e)) h.Write([]byte(e))
h.Write([]byte{0}) h.Write([]byte{0})
} }
copy(n.digest[:], h.Sum(nil)) copy(n.digest[:], h.Sum(nil))
n.entries = nil
} }
// suppressed reports whether a duplicate-tree group is non-maximal: its // suppressed reports whether a duplicate-tree group is non-maximal: its
+92 -19
View File
@@ -2,6 +2,7 @@ package main
import ( import (
"bytes" "bytes"
"database/sql"
"slices" "slices"
"testing" "testing"
) )
@@ -30,6 +31,36 @@ func smokeTreeRecs() []scanRec {
} }
} }
// dbTree builds the directory hierarchy from the records in db the way
// trees does, and returns the super-root and every directory.
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
t.Helper()
tree := newTreeBuilder()
err := loadFileRows(t.Context(), db, tree.add)
if err != nil {
t.Fatal(err)
}
return tree.finish()
}
// treeOf writes recs into a fresh database and builds the directory
// hierarchy from it the way trees does.
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dbTree(t, db)
}
// nodeByPath finds the directory node with the given path. // nodeByPath finds the directory node with the given path.
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode { func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
t.Helper() t.Helper()
@@ -60,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string {
return out return out
} }
func TestBuildHierarchyCounts(t *testing.T) { func TestTreeCounts(t *testing.T) {
t.Parallel() t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs()) _, dirs := treeOf(t, smokeTreeRecs())
super.compute()
d := nodeByPath(t, dirs, "/d") d := nodeByPath(t, dirs, "/d")
if d.fileCount != 6 || d.totalSize != 9300 { if d.fileCount != 6 || d.totalSize != 9300 {
@@ -85,12 +115,12 @@ func TestBuildHierarchyCounts(t *testing.T) {
} }
} }
func TestBuildHierarchyRootPath(t *testing.T) { func TestTreeRootPath(t *testing.T) {
t.Parallel() t.Parallel()
// The root directory's path is "/", never empty, and its // The root directory's path is "/", never empty, and its
// children's paths start with a single slash. // children's paths start with a single slash.
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}}) _, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
got := make([]string, 0, len(dirs)) got := make([]string, 0, len(dirs))
for _, d := range dirs { for _, d := range dirs {
@@ -105,6 +135,56 @@ func TestBuildHierarchyRootPath(t *testing.T) {
} }
} }
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
t.Parallel()
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
// directory must still be built once, whole, so /a matches /c.
recs := make([]scanRec, 0, 8)
for _, top := range []string{"/a", "/c"} {
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
content := "c"
if p == "/b-x/f" {
content = "other"
}
recs = append(recs, scanRec{
size: 1, head: "h", tail: "t", content: content, path: top + p,
})
}
}
super, dirs := treeOf(t, recs)
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
groups := collectTreeGroups(dirs, super)
gotGroups := groupPaths(groups)
wantGroups := [][]string{{"/a", "/c"}}
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
}
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
t.Errorf("group totals: %d files %d bytes, want 4 4",
groups[0][0].fileCount, groups[0][0].totalSize)
}
}
func TestRunTreesEscapesPaths(t *testing.T) { func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -126,8 +206,7 @@ func TestRunTreesEscapesPaths(t *testing.T) {
func TestTreeDigests(t *testing.T) { func TestTreeDigests(t *testing.T) {
t.Parallel() t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs()) _, dirs := treeOf(t, smokeTreeRecs())
super.compute()
t1 := nodeByPath(t, dirs, "/d/t1") t1 := nodeByPath(t, dirs, "/d/t1")
t2 := nodeByPath(t, dirs, "/d/t2") t2 := nodeByPath(t, dirs, "/d/t2")
@@ -160,8 +239,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"}, {size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
} }
super, dirs := buildHierarchy(recs) _, dirs := treeOf(t, recs)
super.compute()
a := nodeByPath(t, dirs, "/r/a") a := nodeByPath(t, dirs, "/r/a")
b := nodeByPath(t, dirs, "/r/b") b := nodeByPath(t, dirs, "/r/b")
@@ -174,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
func TestCollectTreeGroupsMaximal(t *testing.T) { func TestCollectTreeGroupsMaximal(t *testing.T) {
t.Parallel() t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs()) super, dirs := treeOf(t, smokeTreeRecs())
super.compute()
groups := collectTreeGroups(dirs, super) groups := collectTreeGroups(dirs, super)
@@ -199,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
recs := smokeTreeRecs() recs := smokeTreeRecs()
super, dirs := buildHierarchy(recs) super, dirs := treeOf(t, recs)
super.compute()
forward := groupPaths(collectTreeGroups(dirs, super)) forward := groupPaths(collectTreeGroups(dirs, super))
reversed := slices.Clone(recs) reversed := slices.Clone(recs)
slices.Reverse(reversed) slices.Reverse(reversed)
superR, dirsR := buildHierarchy(reversed) superR, dirsR := treeOf(t, reversed)
superR.compute()
backward := groupPaths(collectTreeGroups(dirsR, superR)) backward := groupPaths(collectTreeGroups(dirsR, superR))
if !slices.EqualFunc(forward, backward, slices.Equal) { if !slices.EqualFunc(forward, backward, slices.Equal) {
@@ -227,8 +302,7 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"}, {size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
} }
super, dirs := buildHierarchy(recs) super, dirs := treeOf(t, recs)
super.compute()
got := groupPaths(collectTreeGroups(dirs, super)) got := groupPaths(collectTreeGroups(dirs, super))
@@ -250,8 +324,7 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"}, {size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
} }
super, dirs := buildHierarchy(recs) super, dirs := treeOf(t, recs)
super.compute()
got := groupPaths(collectTreeGroups(dirs, super)) got := groupPaths(collectTreeGroups(dirs, super))