Stream report and trees instead of loading every record (closes #14)
check / check (push) Failing after 46s

report now has SQLite group the records and put the rows in report
order, helped by a new files_signature index on (size, head, tail,
content), and writes each row as it reads it. trees reads the records
in path order, where all the paths under a directory come together, so
it computes each directory's digest as soon as the stream leaves it and
keeps only its path, parent, digest and totals. Output is unchanged.

The tests that called the removed in-memory grouping functions now group
records stored in a database. New tests check that both commands give
the same output whatever order the records were inserted in, and that a
stdout failure partway through a long report is reported as one.

Model: opus-5-5
This commit is contained in:
2026-10-04 03:16:46 +00:00
parent 9abf81535a
commit 7cbb2d7f20
9 changed files with 576 additions and 297 deletions
+144 -15
View File
@@ -2,10 +2,14 @@ package main
import (
"bytes"
"database/sql"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"slices"
"strings"
"testing"
)
@@ -47,6 +51,53 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
return path
}
// dupeGroup is one duplicate group as report reads it: the size, and
// the paths in report order, first path first.
type dupeGroup struct {
size int64
paths []string
}
// dupeGroups returns the duplicate groups report reads from db, in
// report order.
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
t.Helper()
var groups []dupeGroup
_, err := loadDupeRows(t.Context(), db,
func(first, path string, size int64) error {
if path == first {
groups = append(groups, dupeGroup{size: size})
}
g := &groups[len(groups)-1]
g.paths = append(g.paths, path)
return nil
})
if err != nil {
t.Fatal(err)
}
return groups
}
// dupeGroupsOf writes recs into a fresh database and returns the
// duplicate groups report reads from it.
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dupeGroups(t, db)
}
func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -65,6 +116,82 @@ func TestRunReportEscapesPaths(t *testing.T) {
}
}
func TestReportStdoutFailsWhileReading(t *testing.T) {
// Each row holds two paths longer than dir, so the report is more
// than twice the stdout buffer and stdout fails while rows are
// still being read, not at the final flush.
dir := "/" + strings.Repeat("d", 4096)
var recs []scanRec
for i := range ioBufSize / len(dir) {
recs = append(recs, scanRec{
size: 1, head: "h", tail: "t", content: "c",
path: fmt.Sprintf("%s/%d", dir, i),
})
}
t.Setenv(databaseEnv, seedDatabase(t, recs))
err := runReport(t.Context(), failingWriter{})
if !errors.Is(err, errWriteFailed) ||
!strings.HasPrefix(err.Error(), "write stdout: ") {
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
}
}
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
// README §Constraints: identical database contents give identical
// output, whatever order the records were inserted in.
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
recs = append(recs,
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
scanRec{size: 50, path: "/x/unhashed"},
)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, recs))
forward := runStdout(t, name)
t.Setenv(databaseEnv, seedDatabase(t, reversed))
backward := runStdout(t, name)
if strings.Count(forward, "\n") < 3 {
t.Errorf("stdout = %q, want at least two rows", forward)
}
if forward != backward {
t.Errorf("stdout depends on insertion order: %q vs %q",
forward, backward)
}
})
}
}
// runStdout runs the subcommand name and returns its stdout, failing
// the test unless it succeeds.
func runStdout(t *testing.T, name string) string {
t.Helper()
var stdout, stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
}
return stdout.String()
}
func TestEscapePath(t *testing.T) {
t.Parallel()
@@ -121,7 +248,7 @@ func TestWarnfEscapes(t *testing.T) {
}
}
func TestCollectDupeGroups(t *testing.T) {
func TestDupeGroups(t *testing.T) {
t.Parallel()
recs := []scanRec{
@@ -136,7 +263,7 @@ func TestCollectDupeGroups(t *testing.T) {
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
@@ -154,7 +281,7 @@ func TestCollectDupeGroups(t *testing.T) {
}
}
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
func TestDupeGroupsContentSeparates(t *testing.T) {
t.Parallel()
// Same size, head, and tail, but different content hashes: the final
@@ -169,7 +296,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
{size: 100, head: "h", tail: "t", path: "/e"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
len(groups))
@@ -180,7 +307,7 @@ func TestCollectDupeGroupsContentSeparates(t *testing.T) {
}
}
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
func TestDupeGroupsMtimeExcluded(t *testing.T) {
t.Parallel()
// mtime is informational only; records differing only in mtime
@@ -190,23 +317,25 @@ func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1", len(groups))
}
}
func TestCollectDupeGroupsTieBreak(t *testing.T) {
func TestDupeGroupsTieBreak(t *testing.T) {
t.Parallel()
// The hashes sort opposite to the first paths, so ordering the
// groups by hash instead of by first path fails this test.
recs := []scanRec{
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
}
groups := collectDupeGroups(recs)
groups := dupeGroupsOf(t, recs)
if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups))
}
@@ -218,7 +347,7 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
}
}
func TestCollectDupeGroupsDeterministic(t *testing.T) {
func TestDupeGroupsDeterministic(t *testing.T) {
t.Parallel()
recs := []scanRec{
@@ -228,12 +357,12 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
}
forward := collectDupeGroups(recs)
forward := dupeGroupsOf(t, recs)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
backward := collectDupeGroups(reversed)
backward := dupeGroupsOf(t, reversed)
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
return a.size == b.size && slices.Equal(a.paths, b.paths)
}) {