package main import ( "bufio" "context" "fmt" "io" "os" "strings" ) // ioBufSize is the buffer size for the buffered stdout writers. const ioBufSize = 1 << 20 // minGroupSize is the smallest number of members that makes a // duplicate group. const minGroupSize = 2 // scanRec is one file record from the database. The signature (size, // head, tail, content) is the duplicate key; mtime is informational // only and used by scan for change detection. type scanRec struct { size int64 mtime int64 head string tail string content string path string } // runReport implements the report subcommand: it prints the file-level // duplicates report as TSV on stdout. SQLite groups and orders the // records, and each row is written as it is read, so no group is held // in memory. It never touches the scanned filesystem; its only I/O is // the database (with SQLite's temporary sort file), stdout, and stderr. // Any database problem, including a missing database, is fatal. func runReport(ctx context.Context, stdout io.Writer) error { dbPath := databasePath() db, err := openReportDatabase(ctx, dbPath) if err != nil { return err } defer func() { _ = db.Close() }() out := bufio.NewWriterSize(stdout, ioBufSize) _, err = fmt.Fprintln(out, "first\tdupe\tsize") if err != nil { return fmt.Errorf("write stdout: %w", err) } var ( groups, dupeFiles int reclaimable int64 writeErr error ) records, err := loadDupeRows(ctx, db, func(first, path string, size int64) error { // A group's first path is its first row; every other // path is a dupe. if path == first { groups++ return nil } _, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n", escapePath(first), escapePath(path), size) dupeFiles++ reclaimable += size return writeErr }) if writeErr != nil { return fmt.Errorf("write stdout: %w", writeErr) } if err != nil { return fmt.Errorf("database %s: %w", dbPath, err) } err = out.Flush() if err != nil { return fmt.Errorf("write stdout: %w", err) } fmt.Fprintf(os.Stderr, "report: %d records read, %d duplicate groups, %d dupe files, "+ "%s reclaimable\n", records, groups, dupeFiles, humanBytes(reclaimable)) return nil } // escapePath returns a path as it is written in a report column (README // "Report output format"): a backslash, tab, newline or carriage return // becomes \\, \t, \n or \r, and every other byte is kept as it is. // Grouping and sorting use the raw path, never this form. func escapePath(p string) string { // Most paths need no escaping; skip building a replacer for them. if !strings.ContainsAny(p, "\\\t\n\r") { return p } return strings.NewReplacer( `\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`, ).Replace(p) } // humanBytes formats a byte count in human units (binary prefixes). func humanBytes(n int64) string { const unit = 1024 if n < unit { return fmt.Sprintf("%d B", n) } div, exp := int64(unit), 0 for m := n / unit; m >= unit; m /= unit { div *= unit exp++ } return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp]) }