Compare commits

1 Commits
Author SHA1 Message Date
sneak bd8d41b174 Stream report and trees instead of loading every record (closes #14)
check / check (push) Successful in 1m49s
report now has SQLite group the records and put the rows in report
order, helped by a new files_signature index on (size, head, tail,
content), and writes each row as it reads it. trees reads the records
in path order, where all the paths under a directory come together, so
it computes each directory's digest as soon as the stream leaves it and
keeps only its path, parent, digest and totals. Output is unchanged.

The tests that called the removed in-memory grouping functions now group
records stored in a database. A new test checks that both commands give
the same output whatever order the records were inserted in.

Model: opus-5-5
2026-10-04 02:04:37 +00:00
9 changed files with 540 additions and 292 deletions
+17 -5
View File
@@ -94,7 +94,14 @@ Goals, in order:
millions of files, ~150 TB filesystem, possibly slow or busy disks
(ZFS pool under resilver). Holding one small record (path, size,
mtime) per file in memory during a scan is acceptable; holding
every file's hashes is not (they stay in the database).
every file's hashes is not (they stay in the database). The
reporting commands hold no file's hashes either: `report` lets
SQLite group and order the records and writes each row as it reads
it, so its memory does not grow with the database, and `trees`
reads the records in path order and keeps each directory's path,
digest and totals, plus the files of the directories holding the
record being read, so its memory grows with the number of
directories, not files.
3. **Scan incrementally, analyze offline.** The expensive filesystem
scan maintains a persistent database; an unchanged file is never
read again on a rescan, except to compute its content hash once a
@@ -201,6 +208,7 @@ All three subcommands operate on a single SQLite database file:
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
) WITHOUT ROWID;
CREATE INDEX files_signature ON files (size, head, tail, content);
```
Paths are stored as BLOBs because Unix paths are raw bytes, not
@@ -215,7 +223,8 @@ All three subcommands operate on a single SQLite database file:
until the content phase of a scan (see "`scan` mode" below) has
read the file. A record with an empty `content` is never part of a
duplicate group, though it still defines the file for tree
reconstruction.
reconstruction. The `files_signature` index lets SQLite group the
records by signature for `report` without sorting the whole table.
### Duplicate detection
@@ -443,9 +452,12 @@ arguments.
**`report` must never touch the filesystem being analyzed.** It does not
stat, open, or otherwise access any path that appears in the records; its
only I/O is reading the database and writing stdout/stderr. It must
produce identical output whether or not the scanned filesystem is still
mounted.
only I/O is reading the database, writing stdout/stderr, and the
temporary file SQLite sorts in when the duplicate rows do not fit in
memory. SQLite puts that file in `$SQLITE_TMPDIR` or `$TMPDIR` when set,
otherwise in `/var/tmp` (or `/tmp`), and deletes it as soon as it has
opened it. `report` must produce identical output whether or not the
scanned filesystem is still mounted.
Processing:
+4
View File
@@ -29,6 +29,10 @@
# Completed Steps
- `report` and `trees` stream the records instead of holding them all in
memory; the schema gains the `files_signature` index (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/14)
- warn about and skip symlink, socket, FIFO, device and `.zfs`
operands, keeping the records beneath them (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/9)
+94 -10
View File
@@ -51,6 +51,12 @@ CREATE TABLE files (
) WITHOUT ROWID
`
// createIndexSQL indexes the records by signature, so report can have
// SQLite group them without sorting the whole table.
const createIndexSQL = `
CREATE INDEX files_signature ON files (size, head, tail, content)
`
// upsertSQL inserts one file record, replacing any existing record for
// the same path.
const upsertSQL = `
@@ -256,6 +262,11 @@ func createSchema(ctx context.Context, db *sql.DB) error {
return fmt.Errorf("create schema: %w", err)
}
_, err = db.ExecContext(ctx, createIndexSQL)
if err != nil {
return fmt.Errorf("create schema: %w", err)
}
_, err = db.ExecContext(ctx,
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
if err != nil {
@@ -277,18 +288,18 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
return v, nil
}
// loadFileRows reads every record from the files table.
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
// loadFileRows streams every record to fn in path order: byte order,
// which is the order of the primary key, so SQLite does not sort.
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
rows, err := db.QueryContext(ctx,
"SELECT path, size, mtime, head, tail, content FROM files")
"SELECT path, size, mtime, head, tail, content FROM files "+
"ORDER BY path")
if err != nil {
return nil, fmt.Errorf("read records: %w", err)
return fmt.Errorf("read records: %w", err)
}
defer func() { _ = rows.Close() }()
var recs []scanRec
for rows.Next() {
var (
path []byte
@@ -298,19 +309,92 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
&r.content)
if err != nil {
return nil, fmt.Errorf("read record: %w", err)
return fmt.Errorf("read record: %w", err)
}
r.path = string(path)
recs = append(recs, r)
fn(r)
}
err = rows.Err()
if err != nil {
return nil, fmt.Errorf("read records: %w", err)
return fmt.Errorf("read records: %w", err)
}
return recs, nil
return nil
}
// dupeRowsSQL selects every record in a duplicate group, with the
// group's first path. A group is the records with a content hash that
// share a size, head, tail, and content, when there are two or more of
// them. The rows come in report order: groups by size descending, then
// by first path, and each group's paths ascending.
const dupeRowsSQL = `
SELECT g.first, f.path, f.size
FROM files AS f
JOIN (
SELECT size, head, tail, content, MIN(path) AS first
FROM files
WHERE content <> ''
GROUP BY size, head, tail, content
HAVING COUNT(*) > 1
) AS g USING (size, head, tail, content)
ORDER BY f.size DESC, g.first, f.path
`
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
// number of records in the database. The count and the rows are read
// in one transaction, so they agree while a scan is committing. An
// error from fn stops the reading and is returned as it is.
func loadDupeRows(ctx context.Context, db *sql.DB,
fn func(first, path string, size int64) error,
) (int, error) {
// Everything goes through tx: the report connection is the only
// one, so a query on db would wait for tx forever.
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
defer func() { _ = tx.Rollback() }()
var records int
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
defer func() { _ = rows.Close() }()
for rows.Next() {
var (
first, path []byte
size int64
)
err = rows.Scan(&first, &path, &size)
if err != nil {
return 0, fmt.Errorf("read record: %w", err)
}
err = fn(string(first), string(path), size)
if err != nil {
return 0, err
}
}
err = rows.Err()
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
return records, nil
}
// loadFileMeta streams every record's path, size, mtime, and whether
+10 -25
View File
@@ -8,7 +8,6 @@ import (
"os"
"path/filepath"
"slices"
"strings"
"testing"
)
@@ -73,9 +72,8 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
defer func() { _ = db.Close() }()
recs, err := loadFileRows(t.Context(), db)
if err != nil || len(recs) != 0 {
t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
if recs := dbRecords(t, db); len(recs) != 0 {
t.Fatalf("records = %v, want none", recs)
}
}
@@ -168,7 +166,7 @@ func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
defer func() { _ = reportDB.Close() }()
_, err = loadFileRows(t.Context(), reportDB)
err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
if err != nil {
t.Fatalf("loadFileRows: %v", err)
}
@@ -196,15 +194,8 @@ func TestApplyChangesRoundTrip(t *testing.T) {
t.Fatalf("applyChanges: %v", err)
}
got, err := loadFileRows(t.Context(), db)
if err != nil {
t.Fatal(err)
}
slices.SortFunc(got, func(a, b scanRec) int {
return strings.Compare(a.path, b.path)
})
// The records come back in path order, which is the order of recs.
got := dbRecords(t, db)
if !slices.Equal(got, recs) {
t.Fatalf("rows = %+v, want %+v", got, recs)
}
@@ -221,11 +212,7 @@ func TestApplyChangesRoundTrip(t *testing.T) {
t.Fatalf("applyChanges: %v", err)
}
got, err = loadFileRows(t.Context(), db)
if err != nil {
t.Fatal(err)
}
got = dbRecords(t, db)
if len(got) != 1 || got[0] != upd {
t.Fatalf("rows = %+v, want just %+v", got, upd)
}
@@ -254,9 +241,8 @@ func TestApplyChangesBatching(t *testing.T) {
t.Fatalf("applyChanges: %v", err)
}
got, err := loadFileRows(t.Context(), db)
if err != nil || len(got) != n {
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
if got := dbRecords(t, db); len(got) != n {
t.Fatalf("records = %d, want %d", len(got), n)
}
deletes := make([]string, 0, n)
@@ -270,8 +256,7 @@ func TestApplyChangesBatching(t *testing.T) {
t.Fatalf("applyChanges deletes: %v", err)
}
got, err = loadFileRows(t.Context(), db)
if err != nil || len(got) != 0 {
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
if got := dbRecords(t, db); len(got) != 0 {
t.Fatalf("records = %d, want 0", len(got))
}
}
+33 -94
View File
@@ -6,7 +6,6 @@ import (
"fmt"
"io"
"os"
"slices"
"strings"
)
@@ -29,50 +28,21 @@ type scanRec struct {
path string
}
// loadRecords opens the database and reads every file record for the
// report and trees subcommands. Any database problem — including a
// missing database — is fatal. The error is returned rather than
// exiting, so that the deferred close always runs; the database is
// closed before the caller formats its output, so it stays closed even
// if that output fails.
func loadRecords(ctx context.Context) ([]scanRec, error) {
// runReport implements the report subcommand: it prints the file-level
// duplicates report as TSV on stdout. SQLite groups and orders the
// records, and each row is written as it is read, so no group is held
// in memory. It never touches the scanned filesystem; its only I/O is
// the database (with SQLite's temporary sort file), stdout, and stderr.
// Any database problem, including a missing database, is fatal.
func runReport(ctx context.Context, stdout io.Writer) error {
dbPath := databasePath()
db, err := openReportDatabase(ctx, dbPath)
if err != nil {
return nil, err
}
defer func() { _ = db.Close() }()
recs, err := loadFileRows(ctx, db)
if err != nil {
return nil, fmt.Errorf("database %s: %w", dbPath, err)
}
return recs, nil
}
// dupeGroup is one set of candidate-duplicate files: identical size,
// head hash, tail hash, and content hash. paths is sorted
// lexicographically; the first entry is the group's "first", the rest
// are dupes.
type dupeGroup struct {
size int64
paths []string
}
// runReport implements the report subcommand: it reads every record
// from the database and prints the file-level duplicates report as TSV
// on stdout. It never touches the scanned filesystem; its only I/O is
// the database, stdout, and stderr.
func runReport(ctx context.Context, stdout io.Writer) error {
recs, err := loadRecords(ctx)
if err != nil {
return err
}
dupes := collectDupeGroups(recs)
defer func() { _ = db.Close() }()
out := bufio.NewWriterSize(stdout, ioBufSize)
@@ -81,21 +51,36 @@ func runReport(ctx context.Context, stdout io.Writer) error {
return fmt.Errorf("write stdout: %w", err)
}
dupeFiles := 0
var (
groups, dupeFiles int
reclaimable int64
writeErr error
)
var reclaimable int64
records, err := loadDupeRows(ctx, db,
func(first, path string, size int64) error {
// A group's first path is its first row; every other
// path is a dupe.
if path == first {
groups++
for _, g := range dupes {
for _, p := range g.paths[1:] {
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(g.paths[0]), escapePath(p), g.size)
if err != nil {
return fmt.Errorf("write stdout: %w", err)
return nil
}
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(first), escapePath(path), size)
dupeFiles++
reclaimable += g.size
reclaimable += size
return writeErr
})
if writeErr != nil {
return fmt.Errorf("write stdout: %w", writeErr)
}
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
}
err = out.Flush()
@@ -106,57 +91,11 @@ func runReport(ctx context.Context, stdout io.Writer) error {
fmt.Fprintf(os.Stderr,
"report: %d records read, %d duplicate groups, %d dupe files, "+
"%s reclaimable\n",
len(recs), len(dupes), dupeFiles, humanBytes(reclaimable))
records, groups, dupeFiles, humanBytes(reclaimable))
return nil
}
// collectDupeGroups groups records by signature and returns every group
// with two or more paths, each group's paths sorted lexicographically,
// groups ordered by size descending then by first path ascending.
func collectDupeGroups(recs []scanRec) []dupeGroup {
groups := make(map[fileSig][]string)
for _, r := range recs {
// A record without a content hash has unknown content and is
// never reported as a duplicate (README "Database").
if r.content == "" {
continue
}
k := fileSig{
size: r.size, head: r.head, tail: r.tail, content: r.content,
}
groups[k] = append(groups[k], r.path)
}
var dupes []dupeGroup
for k, paths := range groups {
if len(paths) < minGroupSize {
continue
}
slices.Sort(paths)
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
}
// Biggest reclaimable space first; ties broken by first path.
slices.SortFunc(dupes, func(a, b dupeGroup) int {
if a.size != b.size {
if a.size > b.size {
return -1
}
return 1
}
return strings.Compare(a.paths[0], b.paths[0])
})
return dupes
}
// escapePath returns a path as it is written in a report column (README
// "Report output format"): a backslash, tab, newline or carriage return
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
+113 -11
View File
@@ -2,10 +2,12 @@ package main
import (
"bytes"
"database/sql"
"io"
"os"
"path/filepath"
"slices"
"strings"
"testing"
)
@@ -47,6 +49,53 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
return path
}
// dupeGroup is one duplicate group as report reads it: the size, and
// the paths in report order, first path first.
type dupeGroup struct {
size int64
paths []string
}
// dupeGroups returns the duplicate groups report reads from db, in
// report order.
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
t.Helper()
var groups []dupeGroup
_, err := loadDupeRows(t.Context(), db,
func(first, path string, size int64) error {
if path == first {
groups = append(groups, dupeGroup{size: size})
}
g := &groups[len(groups)-1]
g.paths = append(g.paths, path)
return nil
})
if err != nil {
t.Fatal(err)
}
return groups
}
// dupeGroupsOf writes recs into a fresh database and returns the
// duplicate groups report reads from it.
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dupeGroups(t, db)
}
func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -65,6 +114,59 @@ func TestRunReportEscapesPaths(t *testing.T) {
}
}
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
// README §Constraints: identical database contents give identical
// output, whatever order the records were inserted in.
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
recs = append(recs,
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
scanRec{size: 50, path: "/x/unhashed"},
)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, recs))
forward := runStdout(t, name)
t.Setenv(databaseEnv, seedDatabase(t, reversed))
backward := runStdout(t, name)
if strings.Count(forward, "\n") < 3 {
t.Errorf("stdout = %q, want at least two rows", forward)
}
if forward != backward {
t.Errorf("stdout depends on insertion order: %q vs %q",
forward, backward)
}
})
}
}
// runStdout runs the subcommand name and returns its stdout, failing
// the test unless it succeeds.
func runStdout(t *testing.T, name string) string {
t.Helper()
var stdout, stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
}
return stdout.String()
}
func TestEscapePath(t *testing.T) {
t.Parallel()
@@ -121,7 +223,7 @@ func TestWarnfEscapes(t *testing.T) {
}
}
func TestCollectDupeGroups(t *testing.T) {
func TestDupeGroups(t *testing.T) {
t.Parallel()
recs := []scanRec{
@@ -136,7 +238,7 @@ func TestCollectDupeGroups(t *testing.T) {
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
@@ -154,7 +256,7 @@ func TestCollectDupeGroups(t *testing.T) {
}
}
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
func TestDupeGroupsContentSeparates(t *testing.T) {
t.Parallel()
// Same size, head, and tail, but different content hashes: the final
@@ -169,7 +271,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
{size: 100, head: "h", tail: "t", path: "/e"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
len(groups))
@@ -180,7 +282,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
}
}
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
func TestDupeGroupsMtimeExcluded(t *testing.T) {
t.Parallel()
// mtime is informational only; records differing only in mtime
@@ -190,13 +292,13 @@ func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1", len(groups))
}
}
func TestCollectDupeGroupsTieBreak(t *testing.T) {
func TestDupeGroupsTieBreak(t *testing.T) {
t.Parallel()
recs := []scanRec{
@@ -206,7 +308,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
@@ -218,7 +320,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
}
}
func TestCollectDupeGroupsDeterministic(t *testing.T) {
func TestDupeGroupsDeterministic(t *testing.T) {
t.Parallel()
recs := []scanRec{
@@ -228,12 +330,12 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
}
forward := collectDupeGroups(recs)
forward := dupeGroupsOf(t, recs)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
backward := collectDupeGroups(reversed)
backward := dupeGroupsOf(t, reversed)
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
return a.size == b.size && slices.Equal(a.paths, b.paths)
}) {
+25 -24
View File
@@ -400,7 +400,7 @@ func TestScanContentGate(t *testing.T) {
}
}
groups := collectDupeGroups(recs)
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
t.Fatalf("groups = %+v, want only the identical pair %q",
groups, same)
@@ -435,7 +435,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
want := []string{a, b}
slices.Sort(want)
groups := collectDupeGroups(recs)
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want)
}
@@ -462,7 +462,7 @@ func TestScanContentWithinOperand(t *testing.T) {
want := []string{stored, added}
groups := collectDupeGroups(dbRecords(t, db))
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want)
}
@@ -520,7 +520,7 @@ func TestScanContentStalePartners(t *testing.T) {
}
}
if groups := collectDupeGroups(recs); len(groups) != 0 {
if groups := dupeGroups(t, db); len(groups) != 0 {
t.Errorf("groups = %+v, want none", groups)
}
}
@@ -571,7 +571,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
// The stored records lie outside the operand and are left as they
// are, so they still group with each other, but not with the copy.
groups := collectDupeGroups(recs)
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
}
@@ -620,7 +620,7 @@ func TestScanContentReadFailure(t *testing.T) {
want := []string{a, b}
slices.Sort(want)
groups := collectDupeGroups(dbRecords(t, db))
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q after the retry",
groups, want)
@@ -956,11 +956,16 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
return st
}
// dbRecords returns every record currently in the database.
// dbRecords returns every record currently in the database, in path
// order.
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
t.Helper()
recs, err := loadFileRows(t.Context(), db)
var recs []scanRec
err := loadFileRows(t.Context(), db, func(r scanRec) {
recs = append(recs, r)
})
if err != nil {
t.Fatal(err)
}
@@ -997,10 +1002,10 @@ func recordPaths(recs []scanRec) []string {
// assertSmokeDupeGroups checks the file-level duplicate groups for the
// smoke tree rooted at dir.
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
t.Helper()
groups := collectDupeGroups(parsed)
groups := dupeGroups(t, db)
if len(groups) != 5 {
t.Fatalf("len(groups) = %d, want 5", len(groups))
}
@@ -1025,11 +1030,10 @@ func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
// tree rooted at dir.
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) {
t.Helper()
super, dirs := buildHierarchy(parsed)
super.compute()
super, dirs := dbTree(t, db)
tg := collectTreeGroups(dirs, super)
if len(tg) != 1 {
@@ -1063,8 +1067,8 @@ func TestScanPipeline(t *testing.T) {
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
}
assertSmokeDupeGroups(t, dir, parsed)
assertSmokeTreeGroups(t, dir, parsed)
assertSmokeDupeGroups(t, dir, db)
assertSmokeTreeGroups(t, dir, db)
}
func TestSyncScanUnchangedReuse(t *testing.T) {
@@ -1298,7 +1302,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
}
}
if groups := collectDupeGroups(recs); len(groups) != 0 {
if groups := dupeGroups(t, db); len(groups) != 0 {
t.Fatalf("groups = %+v, want none from unhashed records", groups)
}
@@ -1313,7 +1317,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
st)
}
groups := collectDupeGroups(dbRecords(t, db))
groups := dupeGroups(t, db)
if len(groups) != 1 {
t.Fatalf("groups = %+v, want the a/c pair", groups)
}
@@ -1349,8 +1353,7 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
}
for name, unknown := range cases {
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
super.compute()
super, dirs := treeOf(t, append(slices.Clone(shared), unknown...))
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
@@ -1387,7 +1390,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
}
if groups := collectDupeGroups(recs); len(groups) != 1 {
if groups := dupeGroups(t, db); len(groups) != 1 {
t.Fatalf("groups = %+v, want the hardlink pair", groups)
}
}
@@ -1628,10 +1631,8 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
t.Fatal(err)
}
recs := dbRecords(t, db)
assertSmokeDupeGroups(t, dir, recs)
assertSmokeTreeGroups(t, dir, recs)
assertSmokeDupeGroups(t, dir, db)
assertSmokeTreeGroups(t, dir, db)
}
func TestUnderRoot(t *testing.T) {
+135 -87
View File
@@ -12,39 +12,48 @@ import (
"strings"
)
// fileSig is a file's duplicate signature; mtime is excluded.
type fileSig struct {
size int64
head string
tail string
content string
}
// treeNode is one directory reconstructed from the scan stream.
// treeNode is one directory reconstructed from the record paths.
type treeNode struct {
path string
parent *treeNode
dirs map[string]*treeNode
files map[string]fileSig
// entries holds the serialized child entries until the digest is
// computed from them, and is then dropped.
entries []string
digest [sha256.Size]byte
fileCount int64
totalSize int64
}
// runTrees implements the trees subcommand: it reads every record from
// the database, reconstructs the directory hierarchy from the record
// paths, computes a Merkle-style digest per directory, and prints
// maximal duplicate-tree groups as TSV on stdout. It never touches the
// scanned filesystem; its only I/O is the database, stdout, and
// stderr.
// the database in path order, reconstructs the directory hierarchy from
// the record paths, computes a Merkle-style digest per directory, and
// prints maximal duplicate-tree groups as TSV on stdout. It never
// touches the scanned filesystem; its only I/O is the database, stdout,
// and stderr. Any database problem, including a missing database, is
// fatal.
func runTrees(ctx context.Context, stdout io.Writer) error {
recs, err := loadRecords(ctx)
dbPath := databasePath()
db, err := openReportDatabase(ctx, dbPath)
if err != nil {
return err
}
super, allDirs := buildHierarchy(recs)
super.compute()
defer func() { _ = db.Close() }()
records := 0
tree := newTreeBuilder()
err = loadFileRows(ctx, db, func(r scanRec) {
records++
tree.add(r)
})
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
}
super, allDirs := tree.finish()
dupes := collectTreeGroups(allDirs, super)
@@ -82,73 +91,128 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
fmt.Fprintf(os.Stderr,
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
"%s reclaimable\n",
len(recs), len(dupes), dupeTrees, humanBytes(reclaimable))
records, len(dupes), dupeTrees, humanBytes(reclaimable))
return nil
}
// buildHierarchy reconstructs the directory hierarchy from the record
// paths under a synthetic super-root. Paths are split on "/"; for
// absolute paths the first component is empty, which becomes the
// top-level node with path "/". It returns the super-root and every
// directory node created.
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
// treeBuilder reconstructs the directory hierarchy from records added
// in path order, under a synthetic super-root. Paths are split on "/";
// for absolute paths the first component is empty, which becomes the
// top-level directory with path "/". In path order all the paths under
// one directory come together, so a directory is complete once a path
// outside it is added: its digest is computed then and its entries are
// dropped. Only the directories holding the latest path keep entries.
type treeBuilder struct {
super *treeNode
// open lists the directories holding the latest path, outermost
// first, starting with the super-root; names[i] is open[i]'s name.
open []*treeNode
names []string
// dirs lists every completed directory.
dirs []*treeNode
}
func newTreeBuilder() *treeBuilder {
super := &treeNode{}
var allDirs []*treeNode
return &treeBuilder{
super: super,
open: []*treeNode{super},
names: []string{""},
}
}
for _, r := range recs {
// add adds one record. Each record must come after the previous one in
// path order (byte order); otherwise a completed directory would be
// started again as a second directory with the same path.
func (b *treeBuilder) add(r scanRec) {
comps := strings.Split(r.path, "/")
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
node := super
for _, c := range comps[:len(comps)-1] {
child := node.dirs[c]
if child == nil {
childPath := node.path + "/" + c
// Keep the open directories that hold this path; complete the rest.
depth := 1
for depth < len(b.open) && depth <= len(dirNames) &&
b.names[depth] == dirNames[depth-1] {
depth++
}
// The root directory's path is "/", not empty, and its
// children's paths start with one slash, not two.
b.closeTo(depth)
for _, c := range dirNames[depth-1:] {
b.openDir(c)
}
dir := b.open[len(b.open)-1]
dir.entries = append(dir.entries, fileEntry(name, r))
dir.fileCount++
dir.totalSize += r.size
}
// openDir starts the directory called name inside the innermost open
// one.
func (b *treeBuilder) openDir(name string) {
parent := b.open[len(b.open)-1]
path := parent.path + "/" + name
// The root directory's path is "/", not empty, and its children's
// paths start with one slash, not two.
switch {
case node == super && c == "":
childPath = "/"
case node == super:
childPath = c
case node.path == "/":
childPath = "/" + c
case parent == b.super && name == "":
path = "/"
case parent == b.super:
path = name
case parent.path == "/":
path = "/" + name
}
child = &treeNode{path: childPath, parent: node}
if node.dirs == nil {
node.dirs = make(map[string]*treeNode)
b.open = append(b.open, &treeNode{path: path, parent: parent})
b.names = append(b.names, name)
}
node.dirs[c] = child
allDirs = append(allDirs, child)
// closeTo completes the open directories after the first n, innermost
// first: each one's digest is computed and entered in its parent along
// with its totals.
func (b *treeBuilder) closeTo(n int) {
for len(b.open) > n {
last := len(b.open) - 1
dir, name := b.open[last], b.names[last]
b.open, b.names = b.open[:last], b.names[:last]
dir.computeDigest()
dir.parent.entries = append(dir.parent.entries,
"d\x00"+name+"\x00"+string(dir.digest[:]))
dir.parent.fileCount += dir.fileCount
dir.parent.totalSize += dir.totalSize
b.dirs = append(b.dirs, dir)
}
}
node = child
// finish completes every open directory and returns the super-root and
// every directory.
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
b.closeTo(1)
return b.super, b.dirs
}
if node.files == nil {
node.files = make(map[string]fileSig)
}
sig := fileSig{
size: r.size, head: r.head, tail: r.tail, content: r.content,
}
// fileEntry serializes a file child for its directory's digest: its
// name and its signature (size, head, tail, content); mtime is
// excluded.
func fileEntry(name string, r scanRec) string {
content := r.content
// A record without a content hash has unknown content (README
// "Database"): give it a signature no other file can share, so
// trees containing it never compare equal. Real hashes are
// hex, so the NUL-prefixed form cannot collide.
if sig.content == "" {
sig.content = "unhashed\x00" + r.path
// trees containing it never compare equal. Real hashes are hex, so
// the NUL-prefixed form cannot collide.
if content == "" {
content = "unhashed\x00" + r.path
}
node.files[comps[len(comps)-1]] = sig
}
return super, allDirs
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
}
// collectTreeGroups groups directories by digest and returns every
@@ -190,38 +254,22 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
return dupes
}
// compute fills in digest, fileCount, and totalSize for n and all of
// its descendants. A directory's digest is the SHA-256 of its child
// entries — files serialized with name and signature, subdirectories
// with name and recursive digest — sorted byte-lexicographically.
// Filenames cannot contain NUL or "/", so NUL delimiters are
// unambiguous.
func (n *treeNode) compute() {
entries := make([]string, 0, len(n.dirs)+len(n.files))
for name, sig := range n.files {
entries = append(entries,
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
n.fileCount++
n.totalSize += sig.size
}
for name, child := range n.dirs {
child.compute()
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
n.fileCount += child.fileCount
n.totalSize += child.totalSize
}
slices.Sort(entries)
// computeDigest sets n's digest and drops its entries. A directory's
// digest is the SHA-256 of its child entries — files serialized with
// name and signature, subdirectories with name and recursive digest —
// sorted byte-lexicographically. Filenames cannot contain NUL or "/",
// so NUL delimiters are unambiguous.
func (n *treeNode) computeDigest() {
slices.Sort(n.entries)
h := sha256.New()
for _, e := range entries {
for _, e := range n.entries {
h.Write([]byte(e))
h.Write([]byte{0})
}
copy(n.digest[:], h.Sum(nil))
n.entries = nil
}
// suppressed reports whether a duplicate-tree group is non-maximal: its
+92 -19
View File
@@ -2,6 +2,7 @@ package main
import (
"bytes"
"database/sql"
"slices"
"testing"
)
@@ -30,6 +31,36 @@ func smokeTreeRecs() []scanRec {
}
}
// dbTree builds the directory hierarchy from the records in db the way
// trees does, and returns the super-root and every directory.
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
t.Helper()
tree := newTreeBuilder()
err := loadFileRows(t.Context(), db, tree.add)
if err != nil {
t.Fatal(err)
}
return tree.finish()
}
// treeOf writes recs into a fresh database and builds the directory
// hierarchy from it the way trees does.
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dbTree(t, db)
}
// nodeByPath finds the directory node with the given path.
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
t.Helper()
@@ -60,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string {
return out
}
func TestBuildHierarchyCounts(t *testing.T) {
func TestTreeCounts(t *testing.T) {
t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
_, dirs := treeOf(t, smokeTreeRecs())
d := nodeByPath(t, dirs, "/d")
if d.fileCount != 6 || d.totalSize != 9300 {
@@ -85,12 +115,12 @@ func TestBuildHierarchyCounts(t *testing.T) {
}
}
func TestBuildHierarchyRootPath(t *testing.T) {
func TestTreeRootPath(t *testing.T) {
t.Parallel()
// The root directory's path is "/", never empty, and its
// children's paths start with a single slash.
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
got := make([]string, 0, len(dirs))
for _, d := range dirs {
@@ -105,6 +135,56 @@ func TestBuildHierarchyRootPath(t *testing.T) {
}
}
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
t.Parallel()
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
// directory must still be built once, whole, so /a matches /c.
recs := make([]scanRec, 0, 8)
for _, top := range []string{"/a", "/c"} {
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
content := "c"
if p == "/b-x/f" {
content = "other"
}
recs = append(recs, scanRec{
size: 1, head: "h", tail: "t", content: content, path: top + p,
})
}
}
super, dirs := treeOf(t, recs)
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
groups := collectTreeGroups(dirs, super)
gotGroups := groupPaths(groups)
wantGroups := [][]string{{"/a", "/c"}}
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
}
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
t.Errorf("group totals: %d files %d bytes, want 4 4",
groups[0][0].fileCount, groups[0][0].totalSize)
}
}
func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -126,8 +206,7 @@ func TestRunTreesEscapesPaths(t *testing.T) {
func TestTreeDigests(t *testing.T) {
t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
_, dirs := treeOf(t, smokeTreeRecs())
t1 := nodeByPath(t, dirs, "/d/t1")
t2 := nodeByPath(t, dirs, "/d/t2")
@@ -160,8 +239,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
}
super, dirs := buildHierarchy(recs)
super.compute()
_, dirs := treeOf(t, recs)
a := nodeByPath(t, dirs, "/r/a")
b := nodeByPath(t, dirs, "/r/b")
@@ -174,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
func TestCollectTreeGroupsMaximal(t *testing.T) {
t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
super, dirs := treeOf(t, smokeTreeRecs())
groups := collectTreeGroups(dirs, super)
@@ -199,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
recs := smokeTreeRecs()
super, dirs := buildHierarchy(recs)
super.compute()
super, dirs := treeOf(t, recs)
forward := groupPaths(collectTreeGroups(dirs, super))
reversed := slices.Clone(recs)
slices.Reverse(reversed)
superR, dirsR := buildHierarchy(reversed)
superR.compute()
superR, dirsR := treeOf(t, reversed)
backward := groupPaths(collectTreeGroups(dirsR, superR))
if !slices.EqualFunc(forward, backward, slices.Equal) {
@@ -227,8 +302,7 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
}
super, dirs := buildHierarchy(recs)
super.compute()
super, dirs := treeOf(t, recs)
got := groupPaths(collectTreeGroups(dirs, super))
@@ -250,8 +324,7 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
}
super, dirs := buildHierarchy(recs)
super.compute()
super, dirs := treeOf(t, recs)
got := groupPaths(collectTreeGroups(dirs, super))