Stream report and trees instead of loading every record (closes #14)
check / check (push) Successful in 1m49s

report now has SQLite group the records and put the rows in report
order, helped by a new files_signature index on (size, head, tail,
content), and writes each row as it reads it. trees reads the records
in path order, where all the paths under a directory come together, so
it computes each directory's digest as soon as the stream leaves it and
keeps only its path, parent, digest and totals. Output is unchanged.

The tests that called the removed in-memory grouping functions now group
records stored in a database. A new test checks that both commands give
the same output whatever order the records were inserted in.

Model: opus-5-5
This commit is contained in:
2026-10-04 02:04:37 +00:00
parent 2dd1194f33
commit bd8d41b174
9 changed files with 540 additions and 292 deletions
+25 -24
View File
@@ -400,7 +400,7 @@ func TestScanContentGate(t *testing.T) {
}
}
groups := collectDupeGroups(recs)
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
t.Fatalf("groups = %+v, want only the identical pair %q",
groups, same)
@@ -435,7 +435,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
want := []string{a, b}
slices.Sort(want)
groups := collectDupeGroups(recs)
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want)
}
@@ -462,7 +462,7 @@ func TestScanContentWithinOperand(t *testing.T) {
want := []string{stored, added}
groups := collectDupeGroups(dbRecords(t, db))
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want)
}
@@ -520,7 +520,7 @@ func TestScanContentStalePartners(t *testing.T) {
}
}
if groups := collectDupeGroups(recs); len(groups) != 0 {
if groups := dupeGroups(t, db); len(groups) != 0 {
t.Errorf("groups = %+v, want none", groups)
}
}
@@ -571,7 +571,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
// The stored records lie outside the operand and are left as they
// are, so they still group with each other, but not with the copy.
groups := collectDupeGroups(recs)
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
}
@@ -620,7 +620,7 @@ func TestScanContentReadFailure(t *testing.T) {
want := []string{a, b}
slices.Sort(want)
groups := collectDupeGroups(dbRecords(t, db))
groups := dupeGroups(t, db)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q after the retry",
groups, want)
@@ -956,11 +956,16 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
return st
}
// dbRecords returns every record currently in the database.
// dbRecords returns every record currently in the database, in path
// order.
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
t.Helper()
recs, err := loadFileRows(t.Context(), db)
var recs []scanRec
err := loadFileRows(t.Context(), db, func(r scanRec) {
recs = append(recs, r)
})
if err != nil {
t.Fatal(err)
}
@@ -997,10 +1002,10 @@ func recordPaths(recs []scanRec) []string {
// assertSmokeDupeGroups checks the file-level duplicate groups for the
// smoke tree rooted at dir.
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
t.Helper()
groups := collectDupeGroups(parsed)
groups := dupeGroups(t, db)
if len(groups) != 5 {
t.Fatalf("len(groups) = %d, want 5", len(groups))
}
@@ -1025,11 +1030,10 @@ func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
// tree rooted at dir.
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) {
t.Helper()
super, dirs := buildHierarchy(parsed)
super.compute()
super, dirs := dbTree(t, db)
tg := collectTreeGroups(dirs, super)
if len(tg) != 1 {
@@ -1063,8 +1067,8 @@ func TestScanPipeline(t *testing.T) {
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
}
assertSmokeDupeGroups(t, dir, parsed)
assertSmokeTreeGroups(t, dir, parsed)
assertSmokeDupeGroups(t, dir, db)
assertSmokeTreeGroups(t, dir, db)
}
func TestSyncScanUnchangedReuse(t *testing.T) {
@@ -1298,7 +1302,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
}
}
if groups := collectDupeGroups(recs); len(groups) != 0 {
if groups := dupeGroups(t, db); len(groups) != 0 {
t.Fatalf("groups = %+v, want none from unhashed records", groups)
}
@@ -1313,7 +1317,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
st)
}
groups := collectDupeGroups(dbRecords(t, db))
groups := dupeGroups(t, db)
if len(groups) != 1 {
t.Fatalf("groups = %+v, want the a/c pair", groups)
}
@@ -1349,8 +1353,7 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
}
for name, unknown := range cases {
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
super.compute()
super, dirs := treeOf(t, append(slices.Clone(shared), unknown...))
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
@@ -1387,7 +1390,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
}
if groups := collectDupeGroups(recs); len(groups) != 1 {
if groups := dupeGroups(t, db); len(groups) != 1 {
t.Fatalf("groups = %+v, want the hardlink pair", groups)
}
}
@@ -1628,10 +1631,8 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
t.Fatal(err)
}
recs := dbRecords(t, db)
assertSmokeDupeGroups(t, dir, recs)
assertSmokeTreeGroups(t, dir, recs)
assertSmokeDupeGroups(t, dir, db)
assertSmokeTreeGroups(t, dir, db)
}
func TestUnderRoot(t *testing.T) {