Stream report and trees instead of loading every record (closes #14)
check / check (push) Successful in 1m49s
check / check (push) Successful in 1m49s
report now has SQLite group the records and put the rows in report order, helped by a new files_signature index on (size, head, tail, content), and writes each row as it reads it. trees reads the records in path order, where all the paths under a directory come together, so it computes each directory's digest as soon as the stream leaves it and keeps only its path, parent, digest and totals. Output is unchanged. The tests that called the removed in-memory grouping functions now group records stored in a database. A new test checks that both commands give the same output whatever order the records were inserted in. Model: opus-5-5
This commit is contained in:
+113
-11
@@ -2,10 +2,12 @@ package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"database/sql"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -47,6 +49,53 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||
return path
|
||||
}
|
||||
|
||||
// dupeGroup is one duplicate group as report reads it: the size, and
|
||||
// the paths in report order, first path first.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
}
|
||||
|
||||
// dupeGroups returns the duplicate groups report reads from db, in
|
||||
// report order.
|
||||
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
||||
t.Helper()
|
||||
|
||||
var groups []dupeGroup
|
||||
|
||||
_, err := loadDupeRows(t.Context(), db,
|
||||
func(first, path string, size int64) error {
|
||||
if path == first {
|
||||
groups = append(groups, dupeGroup{size: size})
|
||||
}
|
||||
|
||||
g := &groups[len(groups)-1]
|
||||
g.paths = append(g.paths, path)
|
||||
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return groups
|
||||
}
|
||||
|
||||
// dupeGroupsOf writes recs into a fresh database and returns the
|
||||
// duplicate groups report reads from it.
|
||||
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return dupeGroups(t, db)
|
||||
}
|
||||
|
||||
func TestRunReportEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
@@ -65,6 +114,59 @@ func TestRunReportEscapesPaths(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
||||
// README §Constraints: identical database contents give identical
|
||||
// output, whatever order the records were inserted in.
|
||||
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
||||
recs = append(recs,
|
||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
||||
scanRec{size: 50, path: "/x/unhashed"},
|
||||
)
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||
|
||||
forward := runStdout(t, name)
|
||||
|
||||
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
||||
|
||||
backward := runStdout(t, name)
|
||||
|
||||
if strings.Count(forward, "\n") < 3 {
|
||||
t.Errorf("stdout = %q, want at least two rows", forward)
|
||||
}
|
||||
|
||||
if forward != backward {
|
||||
t.Errorf("stdout depends on insertion order: %q vs %q",
|
||||
forward, backward)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// runStdout runs the subcommand name and returns its stdout, failing
|
||||
// the test unless it succeeds.
|
||||
func runStdout(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
return stdout.String()
|
||||
}
|
||||
|
||||
func TestEscapePath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -121,7 +223,7 @@ func TestWarnfEscapes(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroups(t *testing.T) {
|
||||
func TestDupeGroups(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
@@ -136,7 +238,7 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 2 {
|
||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||
}
|
||||
@@ -154,7 +256,7 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||
func TestDupeGroupsContentSeparates(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Same size, head, and tail, but different content hashes: the final
|
||||
@@ -169,7 +271,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||
len(groups))
|
||||
@@ -180,7 +282,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// mtime is informational only; records differing only in mtime
|
||||
@@ -190,13 +292,13 @@ func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
func TestDupeGroupsTieBreak(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
@@ -206,7 +308,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 2 {
|
||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||
}
|
||||
@@ -218,7 +320,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||
func TestDupeGroupsDeterministic(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
@@ -228,12 +330,12 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||
}
|
||||
|
||||
forward := collectDupeGroups(recs)
|
||||
forward := dupeGroupsOf(t, recs)
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
backward := collectDupeGroups(reversed)
|
||||
backward := dupeGroupsOf(t, reversed)
|
||||
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
||||
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
||||
}) {
|
||||
|
||||
Reference in New Issue
Block a user