Files
sfdupes/report.go
T
sneak bd8d41b174
check / check (push) Waiting to run
Stream report and trees instead of loading every record (closes #14)
report now has SQLite group the records and put the rows in report
order, helped by a new files_signature index on (size, head, tail,
content), and writes each row as it reads it. trees reads the records
in path order, where all the paths under a directory come together, so
it computes each directory's digest as soon as the stream leaves it and
keeps only its path, parent, digest and totals. Output is unchanged.

The tests that called the removed in-memory grouping functions now group
records stored in a database. A new test checks that both commands give
the same output whatever order the records were inserted in.

Model: opus-5-5
2026-10-04 02:04:37 +00:00

129 lines
3.1 KiB
Go

package main
import (
"bufio"
"context"
"fmt"
"io"
"os"
"strings"
)
// ioBufSize is the buffer size for the buffered stdout writers.
const ioBufSize = 1 << 20
// minGroupSize is the smallest number of members that makes a
// duplicate group.
const minGroupSize = 2
// scanRec is one file record from the database. The signature (size,
// head, tail, content) is the duplicate key; mtime is informational
// only and used by scan for change detection.
type scanRec struct {
size int64
mtime int64
head string
tail string
content string
path string
}
// runReport implements the report subcommand: it prints the file-level
// duplicates report as TSV on stdout. SQLite groups and orders the
// records, and each row is written as it is read, so no group is held
// in memory. It never touches the scanned filesystem; its only I/O is
// the database (with SQLite's temporary sort file), stdout, and stderr.
// Any database problem, including a missing database, is fatal.
func runReport(ctx context.Context, stdout io.Writer) error {
dbPath := databasePath()
db, err := openReportDatabase(ctx, dbPath)
if err != nil {
return err
}
defer func() { _ = db.Close() }()
out := bufio.NewWriterSize(stdout, ioBufSize)
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
if err != nil {
return fmt.Errorf("write stdout: %w", err)
}
var (
groups, dupeFiles int
reclaimable int64
writeErr error
)
records, err := loadDupeRows(ctx, db,
func(first, path string, size int64) error {
// A group's first path is its first row; every other
// path is a dupe.
if path == first {
groups++
return nil
}
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(first), escapePath(path), size)
dupeFiles++
reclaimable += size
return writeErr
})
if writeErr != nil {
return fmt.Errorf("write stdout: %w", writeErr)
}
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
}
err = out.Flush()
if err != nil {
return fmt.Errorf("write stdout: %w", err)
}
fmt.Fprintf(os.Stderr,
"report: %d records read, %d duplicate groups, %d dupe files, "+
"%s reclaimable\n",
records, groups, dupeFiles, humanBytes(reclaimable))
return nil
}
// escapePath returns a path as it is written in a report column (README
// "Report output format"): a backslash, tab, newline or carriage return
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
// Grouping and sorting use the raw path, never this form.
func escapePath(p string) string {
// Most paths need no escaping; skip building a replacer for them.
if !strings.ContainsAny(p, "\\\t\n\r") {
return p
}
return strings.NewReplacer(
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
).Replace(p)
}
// humanBytes formats a byte count in human units (binary prefixes).
func humanBytes(n int64) string {
const unit = 1024
if n < unit {
return fmt.Sprintf("%d B", n)
}
div, exp := int64(unit), 0
for m := n / unit; m >= unit; m /= unit {
div *= unit
exp++
}
return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp])
}