diff --git a/README.md b/README.md index ae992d3..f9905fa 100644 --- a/README.md +++ b/README.md @@ -94,7 +94,15 @@ Goals, in order: millions of files, ~150 TB filesystem, possibly slow or busy disks (ZFS pool under resilver). Holding one small record (path, size, mtime) per file in memory during a scan is acceptable; holding - every file's hashes is not (they stay in the database). + every file's hashes is not (they stay in the database). The + reporting commands do not hold every file's hashes either: `report` + lets SQLite group and order the records and writes each row as it + reads it, so its memory does not grow with the database, and + `trees` reads the records in path order and keeps each directory's + path, digest and totals, plus the hashes of only the files in the + directories holding the record being read, so its memory grows with + the number of directories and with the size of the largest + directory. 3. **Scan incrementally, analyze offline.** The expensive filesystem scan maintains a persistent database; an unchanged file is never read again on a rescan, except to compute its content hash once a @@ -184,7 +192,10 @@ All three subcommands operate on a single SQLite database file: starts while a report is still reading waits for it up to the 10-second busy timeout, then fails; a `scan` that ends while a report has the database open warns and leaves the database in WAL - mode until the next scan. + mode until the next scan. `report` writes each row as it reads it, + so it is still reading while its output is paused (a pager, a + stalled pipe), and a `scan` started then fails after the busy + timeout. - `report` and `trees` open the database read-only and need only read access to the database file, and no write access to its directory. While the database is in WAL mode they also read the `-wal` and @@ -202,6 +213,7 @@ All three subcommands operate on a single SQLite database file: tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples ) WITHOUT ROWID; + CREATE INDEX files_signature ON files (size, head, tail, content); ``` Paths are stored as BLOBs because Unix paths are raw bytes, not @@ -216,7 +228,8 @@ All three subcommands operate on a single SQLite database file: until the content phase of a scan (see "`scan` mode" below) has read the file. A record with an empty `content` is never part of a duplicate group, though it still defines the file for tree - reconstruction. + reconstruction. The `files_signature` index lets SQLite group the + records by signature for `report` without sorting the whole table. ### Duplicate detection @@ -444,9 +457,12 @@ arguments. **`report` must never touch the filesystem being analyzed.** It does not stat, open, or otherwise access any path that appears in the records; its -only I/O is reading the database and writing stdout/stderr. It must -produce identical output whether or not the scanned filesystem is still -mounted. +only I/O is reading the database, writing stdout/stderr, and the +temporary file SQLite sorts in when the duplicate rows do not fit in +memory. SQLite puts that file in `$SQLITE_TMPDIR` or `$TMPDIR` when set, +otherwise in `/var/tmp` (or `/tmp`), and deletes it as soon as it has +opened it. `report` must produce identical output whether or not the +scanned filesystem is still mounted. Processing: diff --git a/TODO.md b/TODO.md index 95fefb7..18e9bc3 100644 --- a/TODO.md +++ b/TODO.md @@ -29,6 +29,10 @@ # Completed Steps +- `report` and `trees` stream the records instead of holding them all in + memory; the schema gains the `files_signature` index (2026-10-04, + https://git.eeqj.de/sneak/sfdupes/issues/14) + - progress prints at once on a non-terminal, uses a real terminal test, and prints warnings through a spinner instead of racing its redraw (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13) diff --git a/db.go b/db.go index 1c1ffac..7633ac0 100644 --- a/db.go +++ b/db.go @@ -51,6 +51,12 @@ CREATE TABLE files ( ) WITHOUT ROWID ` +// createIndexSQL indexes the records by signature, so report can have +// SQLite group them without sorting the whole table. +const createIndexSQL = ` +CREATE INDEX files_signature ON files (size, head, tail, content) +` + // upsertSQL inserts one file record, replacing any existing record for // the same path. const upsertSQL = ` @@ -256,6 +262,11 @@ func createSchema(ctx context.Context, db *sql.DB) error { return fmt.Errorf("create schema: %w", err) } + _, err = db.ExecContext(ctx, createIndexSQL) + if err != nil { + return fmt.Errorf("create schema: %w", err) + } + _, err = db.ExecContext(ctx, "PRAGMA user_version = "+strconv.Itoa(schemaVersion)) if err != nil { @@ -277,18 +288,18 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) { return v, nil } -// loadFileRows reads every record from the files table. -func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) { +// loadFileRows streams every record to fn in path order: byte order, +// which is the order of the primary key, so SQLite does not sort. +func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error { rows, err := db.QueryContext(ctx, - "SELECT path, size, mtime, head, tail, content FROM files") + "SELECT path, size, mtime, head, tail, content FROM files "+ + "ORDER BY path") if err != nil { - return nil, fmt.Errorf("read records: %w", err) + return fmt.Errorf("read records: %w", err) } defer func() { _ = rows.Close() }() - var recs []scanRec - for rows.Next() { var ( path []byte @@ -298,19 +309,92 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) { err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &r.content) if err != nil { - return nil, fmt.Errorf("read record: %w", err) + return fmt.Errorf("read record: %w", err) } r.path = string(path) - recs = append(recs, r) + fn(r) } err = rows.Err() if err != nil { - return nil, fmt.Errorf("read records: %w", err) + return fmt.Errorf("read records: %w", err) } - return recs, nil + return nil +} + +// dupeRowsSQL selects every record in a duplicate group, with the +// group's first path. A group is the records with a content hash that +// share a size, head, tail, and content, when there are two or more of +// them. The rows come in report order: groups by size descending, then +// by first path, and each group's paths ascending. +const dupeRowsSQL = ` +SELECT g.first, f.path, f.size +FROM files AS f +JOIN ( + SELECT size, head, tail, content, MIN(path) AS first + FROM files + WHERE content <> '' + GROUP BY size, head, tail, content + HAVING COUNT(*) > 1 +) AS g USING (size, head, tail, content) +ORDER BY f.size DESC, g.first, f.path +` + +// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the +// number of records in the database. The count and the rows are read +// in one transaction, so they agree while a scan is committing. An +// error from fn stops the reading and is returned as it is. +func loadDupeRows(ctx context.Context, db *sql.DB, + fn func(first, path string, size int64) error, +) (int, error) { + // Everything goes through tx: the report connection is the only + // one, so a query on db would wait for tx forever. + tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true}) + if err != nil { + return 0, fmt.Errorf("read records: %w", err) + } + + defer func() { _ = tx.Rollback() }() + + var records int + + err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records) + if err != nil { + return 0, fmt.Errorf("read records: %w", err) + } + + rows, err := tx.QueryContext(ctx, dupeRowsSQL) + if err != nil { + return 0, fmt.Errorf("read records: %w", err) + } + + defer func() { _ = rows.Close() }() + + for rows.Next() { + var ( + first, path []byte + size int64 + ) + + err = rows.Scan(&first, &path, &size) + if err != nil { + return 0, fmt.Errorf("read record: %w", err) + } + + err = fn(string(first), string(path), size) + if err != nil { + return 0, err + } + } + + err = rows.Err() + if err != nil { + return 0, fmt.Errorf("read records: %w", err) + } + + return records, nil } // loadFileMeta streams every record's path, size, mtime, and whether diff --git a/db_test.go b/db_test.go index bc66c9c..7bbb2b8 100644 --- a/db_test.go +++ b/db_test.go @@ -8,7 +8,6 @@ import ( "os" "path/filepath" "slices" - "strings" "testing" ) @@ -73,9 +72,8 @@ func TestOpenScanDatabaseCreates(t *testing.T) { defer func() { _ = db.Close() }() - recs, err := loadFileRows(t.Context(), db) - if err != nil || len(recs) != 0 { - t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err) + if recs := dbRecords(t, db); len(recs) != 0 { + t.Fatalf("records = %v, want none", recs) } } @@ -168,7 +166,7 @@ func TestCloseScanDatabaseWhileReportOpen(t *testing.T) { defer func() { _ = reportDB.Close() }() - _, err = loadFileRows(t.Context(), reportDB) + err = loadFileRows(t.Context(), reportDB, func(scanRec) {}) if err != nil { t.Fatalf("loadFileRows: %v", err) } @@ -196,15 +194,8 @@ func TestApplyChangesRoundTrip(t *testing.T) { t.Fatalf("applyChanges: %v", err) } - got, err := loadFileRows(t.Context(), db) - if err != nil { - t.Fatal(err) - } - - slices.SortFunc(got, func(a, b scanRec) int { - return strings.Compare(a.path, b.path) - }) - + // The records come back in path order, which is the order of recs. + got := dbRecords(t, db) if !slices.Equal(got, recs) { t.Fatalf("rows = %+v, want %+v", got, recs) } @@ -221,11 +212,7 @@ func TestApplyChangesRoundTrip(t *testing.T) { t.Fatalf("applyChanges: %v", err) } - got, err = loadFileRows(t.Context(), db) - if err != nil { - t.Fatal(err) - } - + got = dbRecords(t, db) if len(got) != 1 || got[0] != upd { t.Fatalf("rows = %+v, want just %+v", got, upd) } @@ -254,9 +241,8 @@ func TestApplyChangesBatching(t *testing.T) { t.Fatalf("applyChanges: %v", err) } - got, err := loadFileRows(t.Context(), db) - if err != nil || len(got) != n { - t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n) + if got := dbRecords(t, db); len(got) != n { + t.Fatalf("records = %d, want %d", len(got), n) } deletes := make([]string, 0, n) @@ -270,8 +256,7 @@ func TestApplyChangesBatching(t *testing.T) { t.Fatalf("applyChanges deletes: %v", err) } - got, err = loadFileRows(t.Context(), db) - if err != nil || len(got) != 0 { - t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err) + if got := dbRecords(t, db); len(got) != 0 { + t.Fatalf("records = %d, want 0", len(got)) } } diff --git a/report.go b/report.go index 68b4ecb..d2a0c7b 100644 --- a/report.go +++ b/report.go @@ -6,7 +6,6 @@ import ( "fmt" "io" "os" - "slices" "strings" ) @@ -29,50 +28,21 @@ type scanRec struct { path string } -// loadRecords opens the database and reads every file record for the -// report and trees subcommands. Any database problem — including a -// missing database — is fatal. The error is returned rather than -// exiting, so that the deferred close always runs; the database is -// closed before the caller formats its output, so it stays closed even -// if that output fails. -func loadRecords(ctx context.Context) ([]scanRec, error) { +// runReport implements the report subcommand: it prints the file-level +// duplicates report as TSV on stdout. SQLite groups and orders the +// records, and each row is written as it is read, so no group is held +// in memory. It never touches the scanned filesystem; its only I/O is +// the database (with SQLite's temporary sort file), stdout, and stderr. +// Any database problem, including a missing database, is fatal. +func runReport(ctx context.Context, stdout io.Writer) error { dbPath := databasePath() db, err := openReportDatabase(ctx, dbPath) - if err != nil { - return nil, err - } - - defer func() { _ = db.Close() }() - - recs, err := loadFileRows(ctx, db) - if err != nil { - return nil, fmt.Errorf("database %s: %w", dbPath, err) - } - - return recs, nil -} - -// dupeGroup is one set of candidate-duplicate files: identical size, -// head hash, tail hash, and content hash. paths is sorted -// lexicographically; the first entry is the group's "first", the rest -// are dupes. -type dupeGroup struct { - size int64 - paths []string -} - -// runReport implements the report subcommand: it reads every record -// from the database and prints the file-level duplicates report as TSV -// on stdout. It never touches the scanned filesystem; its only I/O is -// the database, stdout, and stderr. -func runReport(ctx context.Context, stdout io.Writer) error { - recs, err := loadRecords(ctx) if err != nil { return err } - dupes := collectDupeGroups(recs) + defer func() { _ = db.Close() }() out := bufio.NewWriterSize(stdout, ioBufSize) @@ -81,21 +51,36 @@ func runReport(ctx context.Context, stdout io.Writer) error { return fmt.Errorf("write stdout: %w", err) } - dupeFiles := 0 + var ( + groups, dupeFiles int + reclaimable int64 + writeErr error + ) - var reclaimable int64 + records, err := loadDupeRows(ctx, db, + func(first, path string, size int64) error { + // A group's first path is its first row; every other + // path is a dupe. + if path == first { + groups++ - for _, g := range dupes { - for _, p := range g.paths[1:] { - _, err = fmt.Fprintf(out, "%s\t%s\t%d\n", - escapePath(g.paths[0]), escapePath(p), g.size) - if err != nil { - return fmt.Errorf("write stdout: %w", err) + return nil } + _, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n", + escapePath(first), escapePath(path), size) dupeFiles++ - reclaimable += g.size - } + reclaimable += size + + return writeErr + }) + + if writeErr != nil { + return fmt.Errorf("write stdout: %w", writeErr) + } + + if err != nil { + return fmt.Errorf("database %s: %w", dbPath, err) } err = out.Flush() @@ -106,57 +91,11 @@ func runReport(ctx context.Context, stdout io.Writer) error { fmt.Fprintf(os.Stderr, "report: %d records read, %d duplicate groups, %d dupe files, "+ "%s reclaimable\n", - len(recs), len(dupes), dupeFiles, humanBytes(reclaimable)) + records, groups, dupeFiles, humanBytes(reclaimable)) return nil } -// collectDupeGroups groups records by signature and returns every group -// with two or more paths, each group's paths sorted lexicographically, -// groups ordered by size descending then by first path ascending. -func collectDupeGroups(recs []scanRec) []dupeGroup { - groups := make(map[fileSig][]string) - - for _, r := range recs { - // A record without a content hash has unknown content and is - // never reported as a duplicate (README "Database"). - if r.content == "" { - continue - } - - k := fileSig{ - size: r.size, head: r.head, tail: r.tail, content: r.content, - } - groups[k] = append(groups[k], r.path) - } - - var dupes []dupeGroup - - for k, paths := range groups { - if len(paths) < minGroupSize { - continue - } - - slices.Sort(paths) - dupes = append(dupes, dupeGroup{size: k.size, paths: paths}) - } - - // Biggest reclaimable space first; ties broken by first path. - slices.SortFunc(dupes, func(a, b dupeGroup) int { - if a.size != b.size { - if a.size > b.size { - return -1 - } - - return 1 - } - - return strings.Compare(a.paths[0], b.paths[0]) - }) - - return dupes -} - // escapePath returns a path as it is written in a report column (README // "Report output format"): a backslash, tab, newline or carriage return // becomes \\, \t, \n or \r, and every other byte is kept as it is. diff --git a/report_test.go b/report_test.go index 9aa092d..796dbbf 100644 --- a/report_test.go +++ b/report_test.go @@ -2,10 +2,14 @@ package main import ( "bytes" + "database/sql" + "errors" + "fmt" "io" "os" "path/filepath" "slices" + "strings" "testing" ) @@ -47,6 +51,53 @@ func seedDatabase(t *testing.T, recs []scanRec) string { return path } +// dupeGroup is one duplicate group as report reads it: the size, and +// the paths in report order, first path first. +type dupeGroup struct { + size int64 + paths []string +} + +// dupeGroups returns the duplicate groups report reads from db, in +// report order. +func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup { + t.Helper() + + var groups []dupeGroup + + _, err := loadDupeRows(t.Context(), db, + func(first, path string, size int64) error { + if path == first { + groups = append(groups, dupeGroup{size: size}) + } + + g := &groups[len(groups)-1] + g.paths = append(g.paths, path) + + return nil + }) + if err != nil { + t.Fatal(err) + } + + return groups +} + +// dupeGroupsOf writes recs into a fresh database and returns the +// duplicate groups report reads from it. +func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup { + t.Helper() + + db := openTestDB(t) + + err := applyChanges(t.Context(), db, recs, nil, nil) + if err != nil { + t.Fatal(err) + } + + return dupeGroups(t, db) +} + func TestRunReportEscapesPaths(t *testing.T) { t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) @@ -65,6 +116,82 @@ func TestRunReportEscapesPaths(t *testing.T) { } } +func TestReportStdoutFailsWhileReading(t *testing.T) { + // Each row holds two paths longer than dir, so the report is more + // than twice the stdout buffer and stdout fails while rows are + // still being read, not at the final flush. + dir := "/" + strings.Repeat("d", 4096) + + recs := make([]scanRec, ioBufSize/len(dir)) + for i := range recs { + recs[i] = scanRec{ + size: 1, head: "h", tail: "t", content: "c", + path: fmt.Sprintf("%s/%d", dir, i), + } + } + + t.Setenv(databaseEnv, seedDatabase(t, recs)) + + err := runReport(t.Context(), failingWriter{}) + if !errors.Is(err, errWriteFailed) || + !strings.HasPrefix(err.Error(), "write stdout: ") { + t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed) + } +} + +func TestRunReportsIgnoreInsertionOrder(t *testing.T) { + // README §Constraints: identical database contents give identical + // output, whatever order the records were inserted in. + recs := append(smokeTreeRecs(), awkwardPairRecs()...) + recs = append(recs, + scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"}, + scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"}, + scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"}, + scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"}, + scanRec{size: 50, path: "/x/unhashed"}, + ) + + reversed := slices.Clone(recs) + slices.Reverse(reversed) + + for _, name := range []string{cmdReport, cmdTrees} { + t.Run(name, func(t *testing.T) { + t.Setenv(databaseEnv, seedDatabase(t, recs)) + + forward := runStdout(t, name) + + t.Setenv(databaseEnv, seedDatabase(t, reversed)) + + backward := runStdout(t, name) + + if strings.Count(forward, "\n") < 3 { + t.Errorf("stdout = %q, want at least two rows", forward) + } + + if forward != backward { + t.Errorf("stdout depends on insertion order: %q vs %q", + forward, backward) + } + }) + } +} + +// runStdout runs the subcommand name and returns its stdout, failing +// the test unless it succeeds. +func runStdout(t *testing.T, name string) string { + t.Helper() + + var stdout, stderr bytes.Buffer + + code := run([]string{name}, &stdout, &stderr) + if code != exitOK { + t.Fatalf("run(%s) = %d, want %d; stderr: %s", + name, code, exitOK, stderr.String()) + } + + return stdout.String() +} + func TestEscapePath(t *testing.T) { t.Parallel() @@ -121,7 +248,7 @@ func TestWarnfEscapes(t *testing.T) { } } -func TestCollectDupeGroups(t *testing.T) { +func TestDupeGroups(t *testing.T) { t.Parallel() recs := []scanRec{ @@ -136,7 +263,7 @@ func TestCollectDupeGroups(t *testing.T) { {size: 7, head: "u", tail: "u", content: "u", path: "/lonely"}, } - groups := collectDupeGroups(recs) + groups := dupeGroupsOf(t, recs) if len(groups) != 2 { t.Fatalf("len(groups) = %d, want 2", len(groups)) } @@ -154,7 +281,7 @@ func TestCollectDupeGroups(t *testing.T) { } } -func TestCollectDupeGroupsContentSeparates(t *testing.T) { +func TestDupeGroupsContentSeparates(t *testing.T) { t.Parallel() // Same size, head, and tail, but different content hashes: the final @@ -169,7 +296,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) { {size: 100, head: "h", tail: "t", path: "/e"}, } - groups := collectDupeGroups(recs) + groups := dupeGroupsOf(t, recs) if len(groups) != 1 { t.Fatalf("len(groups) = %d, want 1 (only the matching content)", len(groups)) @@ -180,7 +307,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) { } } -func TestCollectDupeGroupsMtimeExcluded(t *testing.T) { +func TestDupeGroupsMtimeExcluded(t *testing.T) { t.Parallel() // mtime is informational only; records differing only in mtime @@ -190,23 +317,25 @@ func TestCollectDupeGroupsMtimeExcluded(t *testing.T) { {size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"}, } - groups := collectDupeGroups(recs) + groups := dupeGroupsOf(t, recs) if len(groups) != 1 { t.Fatalf("len(groups) = %d, want 1", len(groups)) } } -func TestCollectDupeGroupsTieBreak(t *testing.T) { +func TestDupeGroupsTieBreak(t *testing.T) { t.Parallel() + // The hashes sort opposite to the first paths, so ordering the + // groups by hash instead of by first path fails this test. recs := []scanRec{ - {size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"}, - {size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"}, - {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"}, - {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"}, + {size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"}, + {size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"}, + {size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"}, + {size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"}, } - groups := collectDupeGroups(recs) + groups := dupeGroupsOf(t, recs) if len(groups) != 2 { t.Fatalf("len(groups) = %d, want 2", len(groups)) } @@ -218,7 +347,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) { } } -func TestCollectDupeGroupsDeterministic(t *testing.T) { +func TestDupeGroupsDeterministic(t *testing.T) { t.Parallel() recs := []scanRec{ @@ -228,12 +357,12 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) { {size: 2, head: "b", tail: "b", content: "b", path: "/q/2"}, } - forward := collectDupeGroups(recs) + forward := dupeGroupsOf(t, recs) reversed := slices.Clone(recs) slices.Reverse(reversed) - backward := collectDupeGroups(reversed) + backward := dupeGroupsOf(t, reversed) if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool { return a.size == b.size && slices.Equal(a.paths, b.paths) }) { diff --git a/scan_test.go b/scan_test.go index 249989a..3b4e73b 100644 --- a/scan_test.go +++ b/scan_test.go @@ -400,7 +400,7 @@ func TestScanContentGate(t *testing.T) { } } - groups := collectDupeGroups(recs) + groups := dupeGroups(t, db) if len(groups) != 1 || !slices.Equal(groups[0].paths, same) { t.Fatalf("groups = %+v, want only the identical pair %q", groups, same) @@ -435,7 +435,7 @@ func TestScanContentAcrossOperands(t *testing.T) { want := []string{a, b} slices.Sort(want) - groups := collectDupeGroups(recs) + groups := dupeGroups(t, db) if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { t.Fatalf("groups = %+v, want the pair %q", groups, want) } @@ -462,7 +462,7 @@ func TestScanContentWithinOperand(t *testing.T) { want := []string{stored, added} - groups := collectDupeGroups(dbRecords(t, db)) + groups := dupeGroups(t, db) if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { t.Fatalf("groups = %+v, want the pair %q", groups, want) } @@ -520,7 +520,7 @@ func TestScanContentStalePartners(t *testing.T) { } } - if groups := collectDupeGroups(recs); len(groups) != 0 { + if groups := dupeGroups(t, db); len(groups) != 0 { t.Errorf("groups = %+v, want none", groups) } } @@ -571,7 +571,7 @@ func TestScanContentHashedStalePartners(t *testing.T) { // The stored records lie outside the operand and are left as they // are, so they still group with each other, but not with the copy. - groups := collectDupeGroups(recs) + groups := dupeGroups(t, db) if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) { t.Errorf("groups = %+v, want only the stored pair %q", groups, stored) } @@ -620,7 +620,7 @@ func TestScanContentReadFailure(t *testing.T) { want := []string{a, b} slices.Sort(want) - groups := collectDupeGroups(dbRecords(t, db)) + groups := dupeGroups(t, db) if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { t.Fatalf("groups = %+v, want the pair %q after the retry", groups, want) @@ -956,11 +956,16 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats { return st } -// dbRecords returns every record currently in the database. +// dbRecords returns every record currently in the database, in path +// order. func dbRecords(t *testing.T, db *sql.DB) []scanRec { t.Helper() - recs, err := loadFileRows(t.Context(), db) + var recs []scanRec + + err := loadFileRows(t.Context(), db, func(r scanRec) { + recs = append(recs, r) + }) if err != nil { t.Fatal(err) } @@ -997,10 +1002,10 @@ func recordPaths(recs []scanRec) []string { // assertSmokeDupeGroups checks the file-level duplicate groups for the // smoke tree rooted at dir. -func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) { +func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) { t.Helper() - groups := collectDupeGroups(parsed) + groups := dupeGroups(t, db) if len(groups) != 5 { t.Fatalf("len(groups) = %d, want 5", len(groups)) } @@ -1025,11 +1030,10 @@ func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) { // assertSmokeTreeGroups checks the duplicate-tree groups for the smoke // tree rooted at dir. -func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) { +func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) { t.Helper() - super, dirs := buildHierarchy(parsed) - super.compute() + super, dirs := dbTree(t, db) tg := collectTreeGroups(dirs, super) if len(tg) != 1 { @@ -1063,8 +1067,8 @@ func TestScanPipeline(t *testing.T) { t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles) } - assertSmokeDupeGroups(t, dir, parsed) - assertSmokeTreeGroups(t, dir, parsed) + assertSmokeDupeGroups(t, dir, db) + assertSmokeTreeGroups(t, dir, db) } func TestSyncScanUnchangedReuse(t *testing.T) { @@ -1298,7 +1302,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) { } } - if groups := collectDupeGroups(recs); len(groups) != 0 { + if groups := dupeGroups(t, db); len(groups) != 0 { t.Fatalf("groups = %+v, want none from unhashed records", groups) } @@ -1313,7 +1317,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) { st) } - groups := collectDupeGroups(dbRecords(t, db)) + groups := dupeGroups(t, db) if len(groups) != 1 { t.Fatalf("groups = %+v, want the a/c pair", groups) } @@ -1349,8 +1353,7 @@ func TestTreesUnhashedNeverEqual(t *testing.T) { } for name, unknown := range cases { - super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...)) - super.compute() + super, dirs := treeOf(t, append(slices.Clone(shared), unknown...)) if tg := collectTreeGroups(dirs, super); len(tg) != 0 { t.Errorf("%s: tree groups = %d, want 0 (the files may differ)", @@ -1387,7 +1390,7 @@ func TestScanHardlinksReadOnce(t *testing.T) { t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb) } - if groups := collectDupeGroups(recs); len(groups) != 1 { + if groups := dupeGroups(t, db); len(groups) != 1 { t.Fatalf("groups = %+v, want the hardlink pair", groups) } } @@ -1628,10 +1631,8 @@ func TestReportsNeverTouchFilesystem(t *testing.T) { t.Fatal(err) } - recs := dbRecords(t, db) - - assertSmokeDupeGroups(t, dir, recs) - assertSmokeTreeGroups(t, dir, recs) + assertSmokeDupeGroups(t, dir, db) + assertSmokeTreeGroups(t, dir, db) } func TestUnderRoot(t *testing.T) { diff --git a/trees.go b/trees.go index 735c5b5..5c1bbf7 100644 --- a/trees.go +++ b/trees.go @@ -12,39 +12,48 @@ import ( "strings" ) -// fileSig is a file's duplicate signature; mtime is excluded. -type fileSig struct { - size int64 - head string - tail string - content string -} - -// treeNode is one directory reconstructed from the scan stream. +// treeNode is one directory reconstructed from the record paths. type treeNode struct { - path string - parent *treeNode - dirs map[string]*treeNode - files map[string]fileSig + path string + parent *treeNode + // entries holds the serialized child entries until the digest is + // computed from them, and is then dropped. + entries []string digest [sha256.Size]byte fileCount int64 totalSize int64 } // runTrees implements the trees subcommand: it reads every record from -// the database, reconstructs the directory hierarchy from the record -// paths, computes a Merkle-style digest per directory, and prints -// maximal duplicate-tree groups as TSV on stdout. It never touches the -// scanned filesystem; its only I/O is the database, stdout, and -// stderr. +// the database in path order, reconstructs the directory hierarchy from +// the record paths, computes a Merkle-style digest per directory, and +// prints maximal duplicate-tree groups as TSV on stdout. It never +// touches the scanned filesystem; its only I/O is the database, stdout, +// and stderr. Any database problem, including a missing database, is +// fatal. func runTrees(ctx context.Context, stdout io.Writer) error { - recs, err := loadRecords(ctx) + dbPath := databasePath() + + db, err := openReportDatabase(ctx, dbPath) if err != nil { return err } - super, allDirs := buildHierarchy(recs) - super.compute() + defer func() { _ = db.Close() }() + + records := 0 + tree := newTreeBuilder() + + err = loadFileRows(ctx, db, func(r scanRec) { + records++ + + tree.add(r) + }) + if err != nil { + return fmt.Errorf("database %s: %w", dbPath, err) + } + + super, allDirs := tree.finish() dupes := collectTreeGroups(allDirs, super) @@ -82,73 +91,128 @@ func runTrees(ctx context.Context, stdout io.Writer) error { fmt.Fprintf(os.Stderr, "trees: %d records read, %d duplicate tree groups, %d dupe trees, "+ "%s reclaimable\n", - len(recs), len(dupes), dupeTrees, humanBytes(reclaimable)) + records, len(dupes), dupeTrees, humanBytes(reclaimable)) return nil } -// buildHierarchy reconstructs the directory hierarchy from the record -// paths under a synthetic super-root. Paths are split on "/"; for -// absolute paths the first component is empty, which becomes the -// top-level node with path "/". It returns the super-root and every -// directory node created. -func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) { +// treeBuilder reconstructs the directory hierarchy from records added +// in path order, under a synthetic super-root. Paths are split on "/"; +// for absolute paths the first component is empty, which becomes the +// top-level directory with path "/". In path order all the paths under +// one directory come together, so a directory is complete once a path +// outside it is added: its digest is computed then and its entries are +// dropped. Only the directories holding the latest path keep entries. +type treeBuilder struct { + super *treeNode + // open lists the directories holding the latest path, outermost + // first, starting with the super-root; names[i] is open[i]'s name. + open []*treeNode + names []string + // dirs lists every completed directory. + dirs []*treeNode +} + +func newTreeBuilder() *treeBuilder { super := &treeNode{} - var allDirs []*treeNode + return &treeBuilder{ + super: super, + open: []*treeNode{super}, + names: []string{""}, + } +} - for _, r := range recs { - comps := strings.Split(r.path, "/") +// add adds one record. Each record must come after the previous one in +// path order (byte order); otherwise a completed directory would be +// started again as a second directory with the same path. +func (b *treeBuilder) add(r scanRec) { + comps := strings.Split(r.path, "/") + dirNames, name := comps[:len(comps)-1], comps[len(comps)-1] - node := super - for _, c := range comps[:len(comps)-1] { - child := node.dirs[c] - if child == nil { - childPath := node.path + "/" + c - - // The root directory's path is "/", not empty, and its - // children's paths start with one slash, not two. - switch { - case node == super && c == "": - childPath = "/" - case node == super: - childPath = c - case node.path == "/": - childPath = "/" + c - } - - child = &treeNode{path: childPath, parent: node} - if node.dirs == nil { - node.dirs = make(map[string]*treeNode) - } - - node.dirs[c] = child - allDirs = append(allDirs, child) - } - - node = child - } - - if node.files == nil { - node.files = make(map[string]fileSig) - } - - sig := fileSig{ - size: r.size, head: r.head, tail: r.tail, content: r.content, - } - - // A record without a content hash has unknown content (README - // "Database"): give it a signature no other file can share, so - // trees containing it never compare equal. Real hashes are - // hex, so the NUL-prefixed form cannot collide. - if sig.content == "" { - sig.content = "unhashed\x00" + r.path - } - - node.files[comps[len(comps)-1]] = sig + // Keep the open directories that hold this path; complete the rest. + depth := 1 + for depth < len(b.open) && depth <= len(dirNames) && + b.names[depth] == dirNames[depth-1] { + depth++ } - return super, allDirs + b.closeTo(depth) + + for _, c := range dirNames[depth-1:] { + b.openDir(c) + } + + dir := b.open[len(b.open)-1] + dir.entries = append(dir.entries, fileEntry(name, r)) + dir.fileCount++ + dir.totalSize += r.size +} + +// openDir starts the directory called name inside the innermost open +// one. +func (b *treeBuilder) openDir(name string) { + parent := b.open[len(b.open)-1] + path := parent.path + "/" + name + + // The root directory's path is "/", not empty, and its children's + // paths start with one slash, not two. + switch { + case parent == b.super && name == "": + path = "/" + case parent == b.super: + path = name + case parent.path == "/": + path = "/" + name + } + + b.open = append(b.open, &treeNode{path: path, parent: parent}) + b.names = append(b.names, name) +} + +// closeTo completes the open directories after the first n, innermost +// first: each one's digest is computed and entered in its parent along +// with its totals. +func (b *treeBuilder) closeTo(n int) { + for len(b.open) > n { + last := len(b.open) - 1 + dir, name := b.open[last], b.names[last] + b.open, b.names = b.open[:last], b.names[:last] + + dir.computeDigest() + + dir.parent.entries = append(dir.parent.entries, + "d\x00"+name+"\x00"+string(dir.digest[:])) + dir.parent.fileCount += dir.fileCount + dir.parent.totalSize += dir.totalSize + b.dirs = append(b.dirs, dir) + } +} + +// finish completes every open directory and returns the super-root and +// every directory. +func (b *treeBuilder) finish() (*treeNode, []*treeNode) { + b.closeTo(1) + + return b.super, b.dirs +} + +// fileEntry serializes a file child for its directory's digest: its +// name and its signature (size, head, tail, content); mtime is +// excluded. +func fileEntry(name string, r scanRec) string { + content := r.content + + // A record without a content hash has unknown content (README + // "Database"): give it a signature no other file can share, so + // trees containing it never compare equal. Real hashes are hex, so + // the NUL-prefixed form cannot collide. + if content == "" { + content = "unhashed\x00" + r.path + } + + return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) + + "\x00" + r.head + "\x00" + r.tail + "\x00" + content } // collectTreeGroups groups directories by digest and returns every @@ -190,38 +254,22 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode { return dupes } -// compute fills in digest, fileCount, and totalSize for n and all of -// its descendants. A directory's digest is the SHA-256 of its child -// entries — files serialized with name and signature, subdirectories -// with name and recursive digest — sorted byte-lexicographically. -// Filenames cannot contain NUL or "/", so NUL delimiters are -// unambiguous. -func (n *treeNode) compute() { - entries := make([]string, 0, len(n.dirs)+len(n.files)) - for name, sig := range n.files { - entries = append(entries, - "f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+ - "\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content) - n.fileCount++ - n.totalSize += sig.size - } - - for name, child := range n.dirs { - child.compute() - entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:])) - n.fileCount += child.fileCount - n.totalSize += child.totalSize - } - - slices.Sort(entries) +// computeDigest sets n's digest and drops its entries. A directory's +// digest is the SHA-256 of its child entries — files serialized with +// name and signature, subdirectories with name and recursive digest — +// sorted byte-lexicographically. Filenames cannot contain NUL or "/", +// so NUL delimiters are unambiguous. +func (n *treeNode) computeDigest() { + slices.Sort(n.entries) h := sha256.New() - for _, e := range entries { + for _, e := range n.entries { h.Write([]byte(e)) h.Write([]byte{0}) } copy(n.digest[:], h.Sum(nil)) + n.entries = nil } // suppressed reports whether a duplicate-tree group is non-maximal: its diff --git a/trees_test.go b/trees_test.go index 5cc8817..969c438 100644 --- a/trees_test.go +++ b/trees_test.go @@ -2,6 +2,7 @@ package main import ( "bytes" + "database/sql" "slices" "testing" ) @@ -30,6 +31,36 @@ func smokeTreeRecs() []scanRec { } } +// dbTree builds the directory hierarchy from the records in db the way +// trees does, and returns the super-root and every directory. +func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) { + t.Helper() + + tree := newTreeBuilder() + + err := loadFileRows(t.Context(), db, tree.add) + if err != nil { + t.Fatal(err) + } + + return tree.finish() +} + +// treeOf writes recs into a fresh database and builds the directory +// hierarchy from it the way trees does. +func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) { + t.Helper() + + db := openTestDB(t) + + err := applyChanges(t.Context(), db, recs, nil, nil) + if err != nil { + t.Fatal(err) + } + + return dbTree(t, db) +} + // nodeByPath finds the directory node with the given path. func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode { t.Helper() @@ -60,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string { return out } -func TestBuildHierarchyCounts(t *testing.T) { +func TestTreeCounts(t *testing.T) { t.Parallel() - super, dirs := buildHierarchy(smokeTreeRecs()) - super.compute() + _, dirs := treeOf(t, smokeTreeRecs()) d := nodeByPath(t, dirs, "/d") if d.fileCount != 6 || d.totalSize != 9300 { @@ -85,12 +115,12 @@ func TestBuildHierarchyCounts(t *testing.T) { } } -func TestBuildHierarchyRootPath(t *testing.T) { +func TestTreeRootPath(t *testing.T) { t.Parallel() // The root directory's path is "/", never empty, and its // children's paths start with a single slash. - _, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}}) + _, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}}) got := make([]string, 0, len(dirs)) for _, d := range dirs { @@ -105,6 +135,56 @@ func TestBuildHierarchyRootPath(t *testing.T) { } } +func TestTreeNamesSortingBeforeSlash(t *testing.T) { + t.Parallel() + + // In path order "/a/b-x/f" and "/a/b.txt" come between the file + // "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each + // directory must still be built once, whole, so /a matches /c. + recs := make([]scanRec, 0, 8) + + for _, top := range []string{"/a", "/c"} { + for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} { + content := "c" + if p == "/b-x/f" { + content = "other" + } + + recs = append(recs, scanRec{ + size: 1, head: "h", tail: "t", content: content, path: top + p, + }) + } + } + + super, dirs := treeOf(t, recs) + + got := make([]string, 0, len(dirs)) + for _, d := range dirs { + got = append(got, d.path) + } + + slices.Sort(got) + + want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"} + if !slices.Equal(got, want) { + t.Fatalf("directory paths = %q, want %q", got, want) + } + + groups := collectTreeGroups(dirs, super) + + gotGroups := groupPaths(groups) + wantGroups := [][]string{{"/a", "/c"}} + + if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) { + t.Fatalf("groups = %v, want %v", gotGroups, wantGroups) + } + + if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 { + t.Errorf("group totals: %d files %d bytes, want 4 4", + groups[0][0].fileCount, groups[0][0].totalSize) + } +} + func TestRunTreesEscapesPaths(t *testing.T) { t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) @@ -126,8 +206,7 @@ func TestRunTreesEscapesPaths(t *testing.T) { func TestTreeDigests(t *testing.T) { t.Parallel() - super, dirs := buildHierarchy(smokeTreeRecs()) - super.compute() + _, dirs := treeOf(t, smokeTreeRecs()) t1 := nodeByPath(t, dirs, "/d/t1") t2 := nodeByPath(t, dirs, "/d/t2") @@ -160,8 +239,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) { {size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"}, } - super, dirs := buildHierarchy(recs) - super.compute() + _, dirs := treeOf(t, recs) a := nodeByPath(t, dirs, "/r/a") b := nodeByPath(t, dirs, "/r/b") @@ -174,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) { func TestCollectTreeGroupsMaximal(t *testing.T) { t.Parallel() - super, dirs := buildHierarchy(smokeTreeRecs()) - super.compute() + super, dirs := treeOf(t, smokeTreeRecs()) groups := collectTreeGroups(dirs, super) @@ -199,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) { recs := smokeTreeRecs() - super, dirs := buildHierarchy(recs) - super.compute() + super, dirs := treeOf(t, recs) forward := groupPaths(collectTreeGroups(dirs, super)) reversed := slices.Clone(recs) slices.Reverse(reversed) - superR, dirsR := buildHierarchy(reversed) - superR.compute() + superR, dirsR := treeOf(t, reversed) backward := groupPaths(collectTreeGroups(dirsR, superR)) if !slices.EqualFunc(forward, backward, slices.Equal) { @@ -227,8 +302,7 @@ func TestCollectTreeGroupsSiblings(t *testing.T) { {size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"}, } - super, dirs := buildHierarchy(recs) - super.compute() + super, dirs := treeOf(t, recs) got := groupPaths(collectTreeGroups(dirs, super)) @@ -250,8 +324,7 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) { {size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"}, } - super, dirs := buildHierarchy(recs) - super.compute() + super, dirs := treeOf(t, recs) got := groupPaths(collectTreeGroups(dirs, super))