Stream report and trees instead of loading every record (closes #14)
check / check (push) Successful in 1m49s

report now has SQLite group the records and put the rows in report
order, helped by a new files_signature index on (size, head, tail,
content), and writes each row as it reads it. trees reads the records
in path order, where all the paths under a directory come together, so
it computes each directory's digest as soon as the stream leaves it and
keeps only its path, parent, digest and totals. Output is unchanged.

The tests that called the removed in-memory grouping functions now group
records stored in a database. A new test checks that both commands give
the same output whatever order the records were inserted in.

Model: opus-5-5
This commit is contained in:
2026-10-04 02:04:37 +00:00
parent 2dd1194f33
commit bd8d41b174
9 changed files with 540 additions and 292 deletions
+92 -19
View File
@@ -2,6 +2,7 @@ package main
import (
"bytes"
"database/sql"
"slices"
"testing"
)
@@ -30,6 +31,36 @@ func smokeTreeRecs() []scanRec {
}
}
// dbTree builds the directory hierarchy from the records in db the way
// trees does, and returns the super-root and every directory.
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
t.Helper()
tree := newTreeBuilder()
err := loadFileRows(t.Context(), db, tree.add)
if err != nil {
t.Fatal(err)
}
return tree.finish()
}
// treeOf writes recs into a fresh database and builds the directory
// hierarchy from it the way trees does.
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dbTree(t, db)
}
// nodeByPath finds the directory node with the given path.
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
t.Helper()
@@ -60,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string {
return out
}
func TestBuildHierarchyCounts(t *testing.T) {
func TestTreeCounts(t *testing.T) {
t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
_, dirs := treeOf(t, smokeTreeRecs())
d := nodeByPath(t, dirs, "/d")
if d.fileCount != 6 || d.totalSize != 9300 {
@@ -85,12 +115,12 @@ func TestBuildHierarchyCounts(t *testing.T) {
}
}
func TestBuildHierarchyRootPath(t *testing.T) {
func TestTreeRootPath(t *testing.T) {
t.Parallel()
// The root directory's path is "/", never empty, and its
// children's paths start with a single slash.
_, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
got := make([]string, 0, len(dirs))
for _, d := range dirs {
@@ -105,6 +135,56 @@ func TestBuildHierarchyRootPath(t *testing.T) {
}
}
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
t.Parallel()
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
// directory must still be built once, whole, so /a matches /c.
recs := make([]scanRec, 0, 8)
for _, top := range []string{"/a", "/c"} {
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
content := "c"
if p == "/b-x/f" {
content = "other"
}
recs = append(recs, scanRec{
size: 1, head: "h", tail: "t", content: content, path: top + p,
})
}
}
super, dirs := treeOf(t, recs)
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
groups := collectTreeGroups(dirs, super)
gotGroups := groupPaths(groups)
wantGroups := [][]string{{"/a", "/c"}}
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
}
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
t.Errorf("group totals: %d files %d bytes, want 4 4",
groups[0][0].fileCount, groups[0][0].totalSize)
}
}
func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -126,8 +206,7 @@ func TestRunTreesEscapesPaths(t *testing.T) {
func TestTreeDigests(t *testing.T) {
t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
_, dirs := treeOf(t, smokeTreeRecs())
t1 := nodeByPath(t, dirs, "/d/t1")
t2 := nodeByPath(t, dirs, "/d/t2")
@@ -160,8 +239,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
}
super, dirs := buildHierarchy(recs)
super.compute()
_, dirs := treeOf(t, recs)
a := nodeByPath(t, dirs, "/r/a")
b := nodeByPath(t, dirs, "/r/b")
@@ -174,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
func TestCollectTreeGroupsMaximal(t *testing.T) {
t.Parallel()
super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
super, dirs := treeOf(t, smokeTreeRecs())
groups := collectTreeGroups(dirs, super)
@@ -199,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
recs := smokeTreeRecs()
super, dirs := buildHierarchy(recs)
super.compute()
super, dirs := treeOf(t, recs)
forward := groupPaths(collectTreeGroups(dirs, super))
reversed := slices.Clone(recs)
slices.Reverse(reversed)
superR, dirsR := buildHierarchy(reversed)
superR.compute()
superR, dirsR := treeOf(t, reversed)
backward := groupPaths(collectTreeGroups(dirsR, superR))
if !slices.EqualFunc(forward, backward, slices.Equal) {
@@ -227,8 +302,7 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
}
super, dirs := buildHierarchy(recs)
super.compute()
super, dirs := treeOf(t, recs)
got := groupPaths(collectTreeGroups(dirs, super))
@@ -250,8 +324,7 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
}
super, dirs := buildHierarchy(recs)
super.compute()
super, dirs := treeOf(t, recs)
got := groupPaths(collectTreeGroups(dirs, super))