check / check (push) Waiting to run
report now has SQLite group the records and put the rows in report order, helped by a new files_signature index on (size, head, tail, content), and writes each row as it reads it. trees reads the records in path order, where all the paths under a directory come together, so it computes each directory's digest as soon as the stream leaves it and keeps only its path, parent, digest and totals. Output is unchanged. The tests that called the removed in-memory grouping functions now group records stored in a database. A new test checks that both commands give the same output whatever order the records were inserted in. Model: opus-5-5
129 lines
3.1 KiB
Go
129 lines
3.1 KiB
Go
package main
|
|
|
|
import (
|
|
"bufio"
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"os"
|
|
"strings"
|
|
)
|
|
|
|
// ioBufSize is the buffer size for the buffered stdout writers.
|
|
const ioBufSize = 1 << 20
|
|
|
|
// minGroupSize is the smallest number of members that makes a
|
|
// duplicate group.
|
|
const minGroupSize = 2
|
|
|
|
// scanRec is one file record from the database. The signature (size,
|
|
// head, tail, content) is the duplicate key; mtime is informational
|
|
// only and used by scan for change detection.
|
|
type scanRec struct {
|
|
size int64
|
|
mtime int64
|
|
head string
|
|
tail string
|
|
content string
|
|
path string
|
|
}
|
|
|
|
// runReport implements the report subcommand: it prints the file-level
|
|
// duplicates report as TSV on stdout. SQLite groups and orders the
|
|
// records, and each row is written as it is read, so no group is held
|
|
// in memory. It never touches the scanned filesystem; its only I/O is
|
|
// the database (with SQLite's temporary sort file), stdout, and stderr.
|
|
// Any database problem, including a missing database, is fatal.
|
|
func runReport(ctx context.Context, stdout io.Writer) error {
|
|
dbPath := databasePath()
|
|
|
|
db, err := openReportDatabase(ctx, dbPath)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
out := bufio.NewWriterSize(stdout, ioBufSize)
|
|
|
|
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
|
if err != nil {
|
|
return fmt.Errorf("write stdout: %w", err)
|
|
}
|
|
|
|
var (
|
|
groups, dupeFiles int
|
|
reclaimable int64
|
|
writeErr error
|
|
)
|
|
|
|
records, err := loadDupeRows(ctx, db,
|
|
func(first, path string, size int64) error {
|
|
// A group's first path is its first row; every other
|
|
// path is a dupe.
|
|
if path == first {
|
|
groups++
|
|
|
|
return nil
|
|
}
|
|
|
|
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
|
escapePath(first), escapePath(path), size)
|
|
dupeFiles++
|
|
reclaimable += size
|
|
|
|
return writeErr
|
|
})
|
|
|
|
if writeErr != nil {
|
|
return fmt.Errorf("write stdout: %w", writeErr)
|
|
}
|
|
|
|
if err != nil {
|
|
return fmt.Errorf("database %s: %w", dbPath, err)
|
|
}
|
|
|
|
err = out.Flush()
|
|
if err != nil {
|
|
return fmt.Errorf("write stdout: %w", err)
|
|
}
|
|
|
|
fmt.Fprintf(os.Stderr,
|
|
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
|
"%s reclaimable\n",
|
|
records, groups, dupeFiles, humanBytes(reclaimable))
|
|
|
|
return nil
|
|
}
|
|
|
|
// escapePath returns a path as it is written in a report column (README
|
|
// "Report output format"): a backslash, tab, newline or carriage return
|
|
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
|
// Grouping and sorting use the raw path, never this form.
|
|
func escapePath(p string) string {
|
|
// Most paths need no escaping; skip building a replacer for them.
|
|
if !strings.ContainsAny(p, "\\\t\n\r") {
|
|
return p
|
|
}
|
|
|
|
return strings.NewReplacer(
|
|
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
|
).Replace(p)
|
|
}
|
|
|
|
// humanBytes formats a byte count in human units (binary prefixes).
|
|
func humanBytes(n int64) string {
|
|
const unit = 1024
|
|
if n < unit {
|
|
return fmt.Sprintf("%d B", n)
|
|
}
|
|
|
|
div, exp := int64(unit), 0
|
|
for m := n / unit; m >= unit; m /= unit {
|
|
div *= unit
|
|
exp++
|
|
}
|
|
|
|
return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp])
|
|
}
|