diff --git a/.dockerignore b/.dockerignore index 3522d5b..fd9fbce 100644 --- a/.dockerignore +++ b/.dockerignore @@ -2,7 +2,6 @@ .claude .DS_Store sfdupes -files.dat *.log *.out *.test diff --git a/.gitignore b/.gitignore index fe37e88..56ba247 100644 --- a/.gitignore +++ b/.gitignore @@ -27,7 +27,6 @@ node_modules/ *.log # Local scan data -files.dat *.sqlite *.sqlite-shm *.sqlite-wal diff --git a/Dockerfile b/Dockerfile index 7e1f610..a6009f2 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,6 +1,6 @@ # Lint stage — fast feedback on formatting and lint issues -# golangci/golangci-lint:v2.12.2 (Debian-based), 2026-08-07 -FROM golangci/golangci-lint:v2.12.2@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint +# golangci/golangci-lint:v2.12.2, 2026-08-07 +FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint WORKDIR /src COPY go.mod go.sum ./ RUN go mod download diff --git a/Dockerfile.lint b/Dockerfile.lint index 836c725..aac1cb9 100644 --- a/Dockerfile.lint +++ b/Dockerfile.lint @@ -9,8 +9,8 @@ # stage of the main Dockerfile because script/lint must not depend on # the rest of that build; the two FROM lines are kept identical by # script/verify-lint-image-pin, run as a gate below. -# golangci/golangci-lint:v2.12.2 (Debian-based), 2026-08-07 -FROM golangci/golangci-lint:v2.12.2@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 +# golangci/golangci-lint:v2.12.2, 2026-08-07 +FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 WORKDIR /src diff --git a/Makefile b/Makefile index 57bd44e..67c7260 100644 --- a/Makefile +++ b/Makefile @@ -46,4 +46,4 @@ hooks: @script/install-precommit clean: - rm -f $(BINARY) files.dat + rm -f $(BINARY) diff --git a/README.md b/README.md index 182a50d..6508cbe 100644 --- a/README.md +++ b/README.md @@ -5,14 +5,20 @@ `sfdupes` is an MIT-licensed Go CLI tool by [@sneak](https://sneak.berlin) that quickly identifies *candidate* duplicate files — and, ultimately, entire duplicate directory trees — -across very large filesystems without reading full file contents. Files -are considered duplicates when they have identical size, identical -SHA-256 of their first 1024 bytes, and identical SHA-256 of their last -1024 bytes. This is a strong candidate signal, not proof of identical -content (the middle of the file is never read); the intended use is +across very large filesystems without reading every byte of every file. +Files are considered duplicates when their sizes are equal and they +agree on a short ladder of hashes. A file under 10 MiB is hashed in full +and compared directly. A larger file is gated first on the SHA-256 of +its first 64 KiB and of its last 64 KiB, and only when its size and +both of those match another file's is it read for a content hash to +compare — the SHA-256 of the whole file when it is under 50 MiB, or of +gigabyte-spaced 1 MiB samples when it is 50 MiB or larger. Below 50 MiB +the content hash is proof of identical content; at or above 50 MiB it is +a strong candidate signal rather than proof, because the gaps between +samples are never read. The intended use is finding duplicate downloads and duplicated directory trees on -multi-terabyte ZFS servers where reading every byte is prohibitively -expensive. `scan` maintains a persistent SQLite database of file +multi-terabyte ZFS servers where reading every byte of every file is +prohibitively expensive. `scan` maintains a persistent SQLite database of file signatures that survives between runs, so it can be run from cron and the reports can be generated at any time from the most recent scan. @@ -29,9 +35,11 @@ export SFDUPES_DATABASE="$HOME/.local/share/sfdupes/db.sqlite" ``` `scan` walks one or more filesystem trees and maintains one database -record per regular file (path, size, mtime, head hash, tail hash). The +record per regular file (path, size, mtime, head hash, tail hash, +content hash). The database persists between runs; a rescan only hashes files that are new -or changed, and removes records for files that no longer exist. +or changed, or that may have gained a duplicate since the last scan, +and removes records for files that no longer exist. `report` reads the database and prints the file-level duplicates report. `trees` reads the same database and prints the duplicate-tree report. A missing/invalid subcommand — or a `scan` invocation with no @@ -47,13 +55,19 @@ completed scan. Duplicate finders that hash entire files do not scale to the target environment: ~10 million files and ~150 TB on possibly slow or busy -disks (a ZFS pool under resilver). Reading at most 2 KiB per file — and -only from files whose size at least one other file shares, since a -size-unique file cannot be a duplicate — makes a full-filesystem sweep -tractable, and the signatures are kept in a persistent database, so -the expensive filesystem pass is incremental: a rescan re-hashes only -files whose recorded mtime or size changed, and all analysis happens -offline from the database alone. The end goal is +disks (a ZFS pool under resilver). sfdupes spends disk I/O only on files +whose size at least one other file shares, since a size-unique file +cannot be a duplicate. Of those, a file under 10 MiB is read in full; a +larger one has its cheap end windows read first, and is read for a +content hash only when its size and both end windows match another +file's — the whole file below 50 MiB, but only gigabyte-spaced samples +at or above 50 MiB, so the largest files are never read in full. This +keeps a full-filesystem sweep tractable, and the signatures are kept in +a persistent database, so the expensive filesystem pass is incremental: +a rescan re-hashes only files whose recorded mtime or size changed, plus +— for its content hash — a file of 10 MiB or more whose size and end +windows have come to match another file's. All analysis happens offline +from the database alone. The end goal is not individual files but whole duplicated trees — duplicate extractions, duplicate downloads, copied project trees — which an operator can consider removing as a unit. @@ -68,17 +82,24 @@ Goals, in order: downloads, copied project trees), so the operator can consider removing an entire subtree at once. File-level duplicate detection is the foundation; tree-level detection is built on top of it. -2. **Never read full file contents.** At most 2 KiB is read per file - (first and last 1024 bytes), and only files whose size at least - one other file shares are read at all — a size-unique file cannot - be a duplicate. Scale target: tens of millions of files, ~150 TB - filesystem, possibly slow or busy disks (ZFS pool under resilver). - Holding one small record (path, size, mtime) per file in memory - during a scan is acceptable; holding every file's hashes is not - (they stay in the database). +2. **Spend I/O in proportion to duplicate likelihood.** Only files + whose size at least one other file shares are read at all — a + size-unique file cannot be a duplicate. Those are compared by the + ladder in "Duplicate detection" below: a file under 10 MiB is hashed + in full, while a larger file is gated on cheap 64 KiB end windows + first, and gets a content hash only when its size and both end + windows match another file's. That hash reads the whole file below + 50 MiB but only gigabyte-spaced 1 MiB samples at or above it, so the + very largest files are still never read in full. Scale target: tens of + millions of files, ~150 TB filesystem, possibly slow or busy disks + (ZFS pool under resilver). Holding one small record (path, size, + mtime) per file in memory during a scan is acceptable; holding + every file's hashes is not (they stay in the database). 3. **Scan incrementally, analyze offline.** The expensive filesystem scan maintains a persistent database; an unchanged file is never - read again on a rescan. All analysis (`report`, `trees`) works from + read again on a rescan, except to compute its content hash once a + file of 10 MiB or more comes to match another on size and both end + windows. All analysis (`report`, `trees`) works from the database alone and must never touch the scanned filesystem again. `scan` is designed to be cronned; the reports run at any time against the last completed scan. @@ -146,21 +167,73 @@ All three subcommands operate on a single SQLite database file: ```sql CREATE TABLE files ( - path BLOB PRIMARY KEY, -- absolute path, raw bytes - size INTEGER NOT NULL, -- bytes, from lstat - mtime INTEGER NOT NULL, -- Unix seconds, from lstat - head TEXT NOT NULL, -- lowercase-hex SHA-256, first 1 KiB - tail TEXT NOT NULL -- lowercase-hex SHA-256, last 1 KiB + path BLOB PRIMARY KEY, -- absolute path, raw bytes + size INTEGER NOT NULL, -- bytes, from lstat + mtime INTEGER NOT NULL, -- Unix seconds, from lstat + head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB + tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB + content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples ) WITHOUT ROWID; ``` Paths are stored as BLOBs because Unix paths are raw bytes, not guaranteed UTF-8. `mtime` is used only for change detection; it is - not part of the duplicate key. `head` and `tail` are empty strings - when the file has never been hashed because its size was unique as - of the last scan that covered it; such records still define the - file for tree reconstruction but never participate in duplicate - groups. + not part of the duplicate key. For a file under 10 MiB `head`, `tail`, + and `content` all hold the whole-file hash (that range is hashed in + full, with no end windows); for a larger file `head` and `tail` hold + the first- and last-64 KiB hashes and `content` the whole-file or + sampled hash. All three are empty strings when the file has never + been hashed because its size was unique as of the last scan that + covered it. For a file of 10 MiB or more, `content` stays empty + until the content phase of a scan (see "`scan` mode" below) has + read the file. A record with an empty `content` is never part of a + duplicate group, though it still defines the file for tree + reconstruction. + +### Duplicate detection + +Two files are duplicates only when they agree on every rung of this +ladder; a mismatch at any rung means they are not duplicates. `scan` +stores each file's hashes, and `report` and `trees` group files by the +whole signature — size, `head`, `tail`, and `content` — so the grouping +is exactly this ladder applied across everything scanned into the +database, even across separate scans. + +1. **Size.** Files of different sizes are never compared. Only files + whose size at least one other file shares are hashed at all. +2. **Under 10 MiB: whole file.** A file smaller than 10 MiB is hashed + in full and compared directly, with no separate end-window step — + small files are cheap to read to the last byte, and doing so makes + the comparison exact. `head`, `tail`, and `content` all hold this + whole-file SHA-256, so such a file's signature is decided entirely + by its size and its content. +3. **10 MiB and above: head and tail.** For a larger file, the SHA-256 + of the first 64 KiB (`head`) and of the last 64 KiB (`tail`) are a + cheap gate that eliminates most same-size pairs before any bulk + reading: the content hash of the next two rungs is computed only for + a file whose size, `head`, and `tail` match another file's, whether + that file is scanned in the same run or stored by an earlier scan. + A stored file that first gains such a match in a later scan gets its + content hash then; until it has one, its `content` is empty and it + is not a duplicate. At 10 MiB and above the two windows never + overlap. +4. **10 MiB and above, content below 50 MiB.** The SHA-256 of the + entire file. Agreement here is proof of identical content (barring a + SHA-256 collision). +5. **10 MiB and above, content 50 MiB and above.** A sampled SHA-256: + the 1 MiB window at each gigabyte-aligned offset (0, 1 GiB, 2 GiB, … + while inside the file, the final window truncated at end of file) is + fed, in order, into one hash. This is **deliberately probabilistic** + — the gaps between samples are never read, so two large files that + agree on every sample are reported as duplicates without being read + in full. It is the price of never reading a 150 GB file end to end. + Because size is already part of the signature, only equal-size files + reach this rung, so their sample boundaries always align. + +`head`, `tail`, and `content` are one column each. A file below 10 MiB +and one at or above it never share a size, and neither do a file below +50 MiB and one at or above it, so a stored value is never ambiguous +between the whole-file, end-window, and sampled forms. ### `scan` mode @@ -183,10 +256,10 @@ scanned operands: - Only a file whose size at least one other file shares is ever read: a size-unique file cannot be a duplicate, so it is recorded - without hashes (`head` and `tail` empty). The size census covers - every file walked this scan plus every database record outside - the scanned operands, so a possible duplicate of a separately - scanned tree is still recognized. + without hashes (`head`, `tail`, and `content` empty). The size + census covers every file walked this scan plus every database + record outside the scanned operands, so a possible duplicate of a + separately scanned tree is still recognized. - A file not yet in the database is inserted: hashed when its size is shared, without hashes otherwise. - A file already in the database is **skipped without reading its @@ -195,7 +268,10 @@ scanned operands: makes a daily rescan cheap. Exception: an unchanged file whose record lacks hashes is hashed — and its record updated — once its size becomes shared, so hashing deferred by size-uniqueness - happens as soon as it could matter. + happens as soon as it could matter. Likewise, an unchanged file of + 10 MiB or more whose record has no `content` hash is read for one + by the content phase below once its size, `head`, and `tail` match + another record's. - A file whose mtime is newer than recorded, or whose size differs, is processed as if new: re-hashed, or recorded without hashes, per the shared-size rule. @@ -204,12 +280,18 @@ scanned operands: removes records for deleted files. It also removes records for paths that failed to stat or hash this run: the database only ever contains signatures verified by the most recent scan that covered - them (a subsequent successful scan re-adds such files). + them (a subsequent successful scan re-adds such files). A failure + in the content phase below removes nothing: the record is left as + it is. - Database records outside the scanned operands are untouched, so disjoint trees can be scanned on different schedules into the same - database. + database. The one exception is the content phase below: a stored + file of 10 MiB or more without a `content` hash is read for one, + wherever it lies, once its size, `head`, and `tail` match another + record's. If that file is gone or has changed since its record was + written, the record is left as it is. -`scan` runs **three sequential phases over the whole scan**. +`scan` runs **four sequential phases over the whole scan**. Parallelism lives inside each phase; batched database writes begin during the hash phase: @@ -229,12 +311,13 @@ during the hash phase: decides its fate. Size-unique files are never read: new or changed ones are recorded without hashes in the update phase, unchanged unhashed ones simply keep their records. Every file - with a shared size is hashed by the worker pool: read the first - `min(1024, size)` bytes and the last `min(1024, size)` bytes - (one read when `size <= 1024`, since the two windows coincide) - and compute the SHA-256 of each. Zero-length files have constant - hashes and are never opened. Files are hashed in **inode order** - (minimizing seeks on spinning disks), and paths that are hard + with a shared size is hashed by the worker pool as described in + "Duplicate detection" above: a file under 10 MiB in full, which + gives its `head`, `tail`, and `content` alike, and a larger file + only in its end windows, which give its `head` and `tail`; its + content hash is left to the content phase. Zero-length files have + constant hashes and are never opened. Files are hashed in **inode + order** (minimizing seeks on spinning disks), and paths that are hard links to the same inode are **read once**, all sharing the one result — a hard-link backup farm costs one read per inode, not per path. The phase total counts actual reads, so progress and @@ -246,6 +329,29 @@ during the hash phase: records for size-unique new and changed files, and the deletions for records the scan did not verify (vanished files, plus paths that failed to stat or hash). +4. **content** — find every record of 10 MiB or more without a + `content` hash whose size, `head`, and `tail` equal another + record's, anywhere in the database: records from this scan and + records stored by earlier scans, inside or outside the scanned + operands. SQLite finds them, so only the records to be read are + kept in memory, never every file's hashes. Every record sharing + their size, `head`, and `tail`, including one that already has a + `content` hash, has its file checked with `lstat` first. A file + that is gone, is no longer a regular file, or has changed (a + different size, or an mtime newer than recorded) keeps its record + as it is and does not count as a match for the others. Any other + `lstat` error is warned about and counted as skipped, with the same + result. If such a record has no `content` hash, it stays out of + duplicate groups; if it has one, it is still reported until a scan + covering its own tree updates or removes it. The files that pass + and have no `content` hash are read only if at least two of those + records pass, so a file whose only matches are stale costs no read; + a file that already has a `content` hash is never read again. + They are read by a worker pool as in the hash phase, in inode order + and once per inode, and their content hashes are committed in + batches. A failed read is warned about and counted as skipped; its + record keeps an empty `content`, so it is not a duplicate, and a + later scan tries again. Rules for the walk: @@ -263,18 +369,18 @@ Rules for the walk: path, and continue. Per-file errors never abort the run; the final summary reports how many were skipped. As specified above, a skipped path that has a database record from an earlier scan loses - that record; an unreadable directory subtree likewise loses its - records (accepted: the database mirrors what the latest scan could - actually verify). + that record, unless it failed only in the content phase; an + unreadable directory subtree likewise loses its records (accepted: + the database mirrors what the latest scan could actually verify). -Concurrency: the walk phase (which also stats files) and the hash -phase each use a worker pool of `--workers` workers (default -`runtime.NumCPU()`); the walk parallelizes across directories, -hashing across files. Both phases are seek-bound on spinning disks, -so raising `--workers` well past the core count can help on pools -with many spindles. The main goroutine owns partitioning, database -writes, and progress rendering; progress display must never block -the workers. +Concurrency: the walk phase (which also stats files), the hash phase, +and the content phase each use a worker pool of `--workers` workers +(default `runtime.NumCPU()`); the walk parallelizes across +directories, hashing across files. All three phases are seek-bound on +spinning disks, so raising `--workers` well past the core count can +help on pools with many spindles. The main goroutine owns +partitioning, database writes, and progress rendering; progress +display must never block the workers. `scan` writes nothing to stdout. The summary line on stderr reports the files seen this run broken down by disposition, plus skips: @@ -299,11 +405,11 @@ mounted. Processing: -- Records without hashes (size-unique when last scanned) are +- Records without a `content` hash (see "Database" above) are excluded: their content is unknown, so they are never reported as duplicates. - Group the remaining records by the key - `(size, head_hash, tail_hash)`. + `(size, head, tail, content)`. - Every group with two or more paths is a duplicate group. - Within each group, sort paths lexicographically (byte order). The first path is the group's `first`; every other path is a `dupe`. @@ -339,11 +445,11 @@ the paths in the records, split on `/`. Definitions: -- A file's **signature** is `(size, head_hash, tail_hash)` — mtime is - informational and excluded. An unhashed record (empty hashes) has +- A file's **signature** is `(size, head, tail, content)` — mtime is + informational and excluded. A record without a `content` hash has unknown content: its signature is treated as unique to that file, - so a tree containing an unhashed file never compares equal to any - other tree. + so a tree containing such a file never compares equal to any other + tree. - A directory's **digest** is a SHA-256 Merkle digest computed bottom-up: serialize the directory's child entries — for a file child, its name and signature; for a subdirectory child, its name @@ -406,10 +512,13 @@ Each phase gets its own display, rendered the moment the phase starts — a scan must never look hung. Loading the existing-record index (`load`) and the walk have no known totals while running: show a live count, rate, and elapsed time (spinner-style, no percentage or -ETA). The hash and update phases -have exact totals — only files that actually need hashing appear in -the hash total, so its ETA is meaningful. Required elements for the -bars with known totals: +ETA). The content phase's display (`content`) starts the same way, +counting the records checked while SQLite finds the files to read and +`lstat` checks them, then shows a bar once reading starts. The hash +and update phases, and the content phase's reads, have exact totals — +only files that actually need hashing appear in the hash and content +totals, so their ETAs are meaningful. Required elements for the bars +with known totals: - elapsed time - estimated time remaining @@ -636,9 +745,10 @@ Tracked in [TODO.md](TODO.md). ## Non-goals -- No full-content verification, no byte-for-byte compare, no deletion - or linking of duplicates. The reports are advisory; acting on them is - the user's job. +- No byte-for-byte compare, and no deletion or linking of + duplicates. Files that match are compared by a SHA-256 of the whole + file below 50 MiB, and only by samples at 50 MiB and over. The + reports are advisory; acting on them is the user's job. - No persistence beyond the SQLite database described above; no export/import formats. - No daemon or filesystem watcher; scheduling rescans is cron's job. diff --git a/TODO.md b/TODO.md index 235101c..7fd3083 100644 --- a/TODO.md +++ b/TODO.md @@ -29,6 +29,40 @@ # Completed Steps +- replace the 1 KiB end-window sampling with the head/tail plus + content-hash ladder (2026-09-22, branch `next`, closes + https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is + hashed in full and compared directly, with no end-window step — its + `head`, `tail`, and `content` all hold the whole-file hash. A file at + 10 MiB or above gets only the 64 KiB `head` and `tail` in the hash + phase; a new content phase, after the update phase, reads it for its + `content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB + samples at or above — only when its size, `head`, and `tail` match + another record's, from the same scan or stored by an earlier one, so + a stored file gains its content hash when it gains a match. A file + that is gone or has changed since its record was written is not + read. The `content` column is part of the version 1 schema. `report` + and `trees` group by the extended signature and leave out any record + without a `content` hash, so the ladder is applied across the whole + database. README "Duplicate detection" documents every rung including + the probabilistic large-file path. + +- remove the dead `files.dat` references from `Makefile`, `.gitignore` + and `.dockerignore` (2026-09-21, branch `next`, closes + https://git.eeqj.de/sneak/sfdupes/issues/22) + +- fix the lint-image pin comments and `FROM` form in `Dockerfile` and + `Dockerfile.lint` (2026-08-10, branch `next`, closes + https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false + `(Debian-based)` parenthetical (v2.12.1 was Debian too) and the + redundant tag, so both pins are the policy `# image:vX.Y.Z, + YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest + unchanged. `script/verify-lint-image-pin` parses those `FROM` lines + and still matches the tagless form; its advice line lost the now + meaningless "tag and digest". With no tag in either reference, a + tag-only disagreement no longer exists — a one-sided tag is caught as + a plain mismatch. + - run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10, branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the diff --git a/cancel_test.go b/cancel_test.go index b894228..f8c18d6 100644 --- a/cancel_test.go +++ b/cancel_test.go @@ -456,7 +456,7 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) { go func() { defer close(done) - hashWorker(cancelledContext(t), jobs, results) + hashWorker(cancelledContext(t), jobs, results, hashSignature) }() awaitReturn(t, done, "hashWorker") diff --git a/db.go b/db.go index 573d944..f732608 100644 --- a/db.go +++ b/db.go @@ -36,22 +36,24 @@ const dbDirPerm = 0o755 // BLOBs because Unix paths are raw bytes, not guaranteed UTF-8. const createTableSQL = ` CREATE TABLE files ( - path BLOB PRIMARY KEY, - size INTEGER NOT NULL, - mtime INTEGER NOT NULL, - head TEXT NOT NULL, - tail TEXT NOT NULL + path BLOB PRIMARY KEY, + size INTEGER NOT NULL, + mtime INTEGER NOT NULL, + head TEXT NOT NULL, + tail TEXT NOT NULL, + content TEXT NOT NULL ) WITHOUT ROWID ` // upsertSQL inserts one file record, replacing any existing record for // the same path. const upsertSQL = ` -INSERT INTO files (path, size, mtime, head, tail) -VALUES (?, ?, ?, ?, ?) +INSERT INTO files (path, size, mtime, head, tail, content) +VALUES (?, ?, ?, ?, ?, ?) ON CONFLICT (path) DO UPDATE SET size = excluded.size, mtime = excluded.mtime, - head = excluded.head, tail = excluded.tail + head = excluded.head, tail = excluded.tail, + content = excluded.content ` // errNoDatabase reports a missing database file for report/trees. @@ -205,7 +207,7 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) { // loadFileRows reads every record from the files table. func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) { rows, err := db.QueryContext(ctx, - "SELECT path, size, mtime, head, tail FROM files") + "SELECT path, size, mtime, head, tail, content FROM files") if err != nil { return nil, fmt.Errorf("read records: %w", err) } @@ -220,7 +222,8 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) { r scanRec ) - err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail) + err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, + &r.content) if err != nil { return nil, fmt.Errorf("read record: %w", err) } @@ -275,6 +278,62 @@ func loadFileMeta(ctx context.Context, db *sql.DB, return nil } +// contentCandidatesSQL selects every record of at least headTailMin +// bytes whose size, head, and tail equal another record's, in each +// group (the records sharing a size, head, and tail) where at least one +// record has no content hash, with whether each record has one. SQLite +// does the grouping, so no other record's hashes are loaded into +// memory; the rows come ordered by size, head, and tail, so each +// group's rows arrive together. +const contentCandidatesSQL = ` +SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> '' +FROM files AS f +JOIN ( + SELECT size, head, tail + FROM files + WHERE size >= ? AND head <> '' + GROUP BY size, head, tail + HAVING COUNT(*) > 1 AND SUM(content = '') > 0 +) AS g USING (size, head, tail) +ORDER BY size, head, tail +` + +// loadContentCandidates streams the rows of contentCandidatesSQL to fn: +// each record, without its content hash, and whether it has one. +func loadContentCandidates(ctx context.Context, db *sql.DB, + fn func(r scanRec, hashed bool), +) error { + rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin) + if err != nil { + return fmt.Errorf("read records: %w", err) + } + + defer func() { _ = rows.Close() }() + + for rows.Next() { + var ( + path []byte + r scanRec + hashed int64 + ) + + err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed) + if err != nil { + return fmt.Errorf("read record: %w", err) + } + + r.path = string(path) + fn(r, hashed != 0) + } + + err = rows.Err() + if err != nil { + return fmt.Errorf("read records: %w", err) + } + + return nil +} + // updateBatchSize is the number of record changes committed per // transaction during the update pass. The filesystem is authoritative // and the database an eventually-consistent reflection of it, so @@ -348,7 +407,7 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec, for _, r := range upserts { _, err = st.ExecContext(ctx, - []byte(r.path), r.size, r.mtime, r.head, r.tail) + []byte(r.path), r.size, r.mtime, r.head, r.tail, r.content) if err != nil { return fmt.Errorf("upsert %s: %w", r.path, err) } diff --git a/db_test.go b/db_test.go index 41ab6f2..b68bddd 100644 --- a/db_test.go +++ b/db_test.go @@ -136,10 +136,14 @@ func TestApplyChangesRoundTrip(t *testing.T) { db := openTestDB(t) // Paths may contain tabs and newlines; the database must store - // them byte-exactly. + // them byte-exactly. Every hash, content included, comes back as + // written. recs := []scanRec{ - {size: 2, mtime: 20, head: "h2", tail: "t2", path: "/a/tab\tnew\nline"}, - {size: 1, mtime: 10, head: "h1", tail: "t1", path: "/a/x"}, + { + size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2", + path: "/a/tab\tnew\nline", + }, + {size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"}, } err := applyChanges(t.Context(), db, recs, nil, @@ -163,7 +167,9 @@ func TestApplyChangesRoundTrip(t *testing.T) { // An upsert for an existing path updates in place; a delete // removes exactly its path. - upd := scanRec{size: 3, mtime: 30, head: "h3", tail: "t3", path: "/a/x"} + upd := scanRec{ + size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x", + } err = applyChanges(t.Context(), db, []scanRec{upd}, []string{"/a/tab\tnew\nline"}, newProgress("update", 2)) diff --git a/main.go b/main.go index e9012db..d2160c1 100644 --- a/main.go +++ b/main.go @@ -1,10 +1,14 @@ // Command sfdupes quickly identifies candidate duplicate files across -// very large filesystems without reading full file contents. Files are -// considered duplicates when they have identical size, identical SHA-256 -// of their first 1024 bytes, and identical SHA-256 of their last 1024 -// bytes. scan maintains a persistent SQLite database of file signatures -// (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) that the -// reporting subcommands read. +// very large filesystems without reading every byte of every file. +// Files are considered duplicates when their sizes are equal and they +// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is +// hashed in full. A larger file is compared on the hashes of its first +// and last 64 KiB, and only when those match another file's is its +// content hash computed and compared: of the whole file when it is +// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB +// or larger. scan maintains a persistent SQLite database of file +// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) +// that the reporting subcommands read. // // Usage: // @@ -97,7 +101,7 @@ func run(args []string, stderr io.Writer) int { func newRootCommand(stderr io.Writer) *cobra.Command { root := &cobra.Command{ Use: "sfdupes", - Short: "Find candidate duplicate files by size and head/tail SHA-256", + Short: "Find candidate duplicate files by size and head/tail/content SHA-256", Version: Version, Args: cobra.NoArgs, RunE: func(cmd *cobra.Command, _ []string) error { @@ -127,7 +131,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command { }), } scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(), - "concurrent workers for the walk and hash phases") + "concurrent workers for the walk, hash, and content phases") scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false, "do not cross filesystem boundaries") diff --git a/report.go b/report.go index 7f7d0f9..c5761b5 100644 --- a/report.go +++ b/report.go @@ -17,14 +17,15 @@ const ioBufSize = 1 << 20 const minGroupSize = 2 // scanRec is one file record from the database. The signature (size, -// head, tail) is the duplicate key; mtime is informational only and -// used by scan for change detection. +// head, tail, content) is the duplicate key; mtime is informational +// only and used by scan for change detection. type scanRec struct { - size int64 - mtime int64 - head string - tail string - path string + size int64 + mtime int64 + head string + tail string + content string + path string } // loadRecords opens the database and reads every file record for the @@ -52,8 +53,9 @@ func loadRecords(ctx context.Context) ([]scanRec, error) { } // dupeGroup is one set of candidate-duplicate files: identical size, -// head hash, and tail hash. paths is sorted lexicographically; the -// first entry is the group's "first", the rest are dupes. +// head hash, tail hash, and content hash. paths is sorted +// lexicographically; the first entry is the group's "first", the rest +// are dupes. type dupeGroup struct { size int64 paths []string @@ -115,14 +117,15 @@ func collectDupeGroups(recs []scanRec) []dupeGroup { groups := make(map[fileSig][]string) for _, r := range recs { - // A record without hashes (its size was unique when last - // scanned) has unknown content and is never reported as a - // duplicate. - if r.head == "" { + // A record without a content hash has unknown content and is + // never reported as a duplicate (README "Database"). + if r.content == "" { continue } - k := fileSig{size: r.size, head: r.head, tail: r.tail} + k := fileSig{ + size: r.size, head: r.head, tail: r.tail, content: r.content, + } groups[k] = append(groups[k], r.path) } diff --git a/report_test.go b/report_test.go index e540985..1aa7cc5 100644 --- a/report_test.go +++ b/report_test.go @@ -9,15 +9,15 @@ func TestCollectDupeGroups(t *testing.T) { t.Parallel() recs := []scanRec{ - {size: 100, head: "h", tail: "t", path: "/z/b"}, - {size: 100, head: "h", tail: "t", path: "/z/a"}, - {size: 100, head: "h", tail: "t", path: "/z/c"}, - {size: 4000, head: "H", tail: "T", path: "/big/2"}, - {size: 4000, head: "H", tail: "T", path: "/big/1"}, + {size: 100, head: "h", tail: "t", content: "c", path: "/z/b"}, + {size: 100, head: "h", tail: "t", content: "c", path: "/z/a"}, + {size: 100, head: "h", tail: "t", content: "c", path: "/z/c"}, + {size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"}, + {size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"}, // Same size as the /z group but a different head hash. - {size: 100, head: "other", tail: "t", path: "/z/d"}, + {size: 100, head: "other", tail: "t", content: "c", path: "/z/d"}, // A singleton signature must not form a group. - {size: 7, head: "u", tail: "u", path: "/lonely"}, + {size: 7, head: "u", tail: "u", content: "u", path: "/lonely"}, } groups := collectDupeGroups(recs) @@ -38,14 +38,40 @@ func TestCollectDupeGroups(t *testing.T) { } } +func TestCollectDupeGroupsContentSeparates(t *testing.T) { + t.Parallel() + + // Same size, head, and tail, but different content hashes: the final + // rung keeps them apart, so no group forms. Matching content groups. + // Records without a content hash never group, not even with each + // other. + recs := []scanRec{ + {size: 100, head: "h", tail: "t", content: "c1", path: "/a"}, + {size: 100, head: "h", tail: "t", content: "c2", path: "/b"}, + {size: 100, head: "h", tail: "t", content: "c1", path: "/c"}, + {size: 100, head: "h", tail: "t", path: "/d"}, + {size: 100, head: "h", tail: "t", path: "/e"}, + } + + groups := collectDupeGroups(recs) + if len(groups) != 1 { + t.Fatalf("len(groups) = %d, want 1 (only the matching content)", + len(groups)) + } + + if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) { + t.Errorf("group paths = %q, want /a /c", groups[0].paths) + } +} + func TestCollectDupeGroupsMtimeExcluded(t *testing.T) { t.Parallel() // mtime is informational only; records differing only in mtime // still group together. recs := []scanRec{ - {size: 9, mtime: 100, head: "h", tail: "t", path: "/m/1"}, - {size: 9, mtime: 200, head: "h", tail: "t", path: "/m/2"}, + {size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"}, + {size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"}, } groups := collectDupeGroups(recs) @@ -58,10 +84,10 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) { t.Parallel() recs := []scanRec{ - {size: 50, head: "b", tail: "b", path: "/beta/2"}, - {size: 50, head: "b", tail: "b", path: "/beta/1"}, - {size: 50, head: "a", tail: "a", path: "/alpha/2"}, - {size: 50, head: "a", tail: "a", path: "/alpha/1"}, + {size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"}, + {size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"}, + {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"}, + {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"}, } groups := collectDupeGroups(recs) @@ -80,10 +106,10 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) { t.Parallel() recs := []scanRec{ - {size: 1, head: "a", tail: "a", path: "/p/1"}, - {size: 1, head: "a", tail: "a", path: "/p/2"}, - {size: 2, head: "b", tail: "b", path: "/q/1"}, - {size: 2, head: "b", tail: "b", path: "/q/2"}, + {size: 1, head: "a", tail: "a", content: "a", path: "/p/1"}, + {size: 1, head: "a", tail: "a", content: "a", path: "/p/2"}, + {size: 2, head: "b", tail: "b", content: "b", path: "/q/1"}, + {size: 2, head: "b", tail: "b", content: "b", path: "/q/2"}, } forward := collectDupeGroups(recs) diff --git a/scan.go b/scan.go index a6d825e..6d7aaf4 100644 --- a/scan.go +++ b/scan.go @@ -6,7 +6,9 @@ import ( "crypto/sha256" "database/sql" "encoding/hex" + "errors" "fmt" + "io" "io/fs" "os" "path/filepath" @@ -16,8 +18,40 @@ import ( "syscall" ) -// chunk is the number of bytes hashed from each end of a file. -const chunk = 1024 +// The duplicate ladder (see hashSignature and README "Duplicate +// detection"). A same-size candidate below headTailMin is hashed in +// full and compared directly; a larger one is separated first by the +// hashes of its end windows, then by a content hash that is exact below +// wholeFileMax and deliberately sampled at or above it. The hash phase +// reads only the end windows of a larger file; the content phase reads +// it for its content hash only once its size, head, and tail match +// another file's. + +// headTailMin is the size threshold for the end-window gate. A file +// smaller than this is hashed in full directly, with no separate head +// and tail step: its head, tail, and content all carry the whole-file +// hash. A file this size or larger is separated first by its end +// windows. +const headTailMin = 10 * 1024 * 1024 + +// headTailWindow is the number of bytes hashed from each end of a file +// at or above headTailMin (the head and tail rungs). Because +// headTailMin is far larger than two windows, the head and tail windows +// never overlap. +const headTailWindow = 64 * 1024 + +// wholeFileMax is the size boundary between the two content rungs: a +// file strictly smaller than this is content-hashed in full; a file +// this size or larger is content-hashed by sampling. +const wholeFileMax = 50 * 1024 * 1024 + +// sampleStride is the spacing between content samples for large files: +// one window is read at each gigabyte-aligned offset (0, 1 GiB, ...). +const sampleStride = 1024 * 1024 * 1024 + +// sampleWindow is the number of bytes read at each large-file sample +// offset, truncated at end of file. +const sampleWindow = 1024 * 1024 // workQueueDepth bounds the job and result channels feeding the walk // and hash worker pools. @@ -44,16 +78,17 @@ type fileMeta struct { hashed bool } -// runScan implements the scan subcommand: three sequential phases — -// walk (which stats each file as it is discovered), hash, update — -// that synchronize the persistent database with the filesystem state -// under the PATH operands. Only files whose size at least one other -// file shares are ever hashed: a size-unique file cannot be a -// duplicate. Flag parsing and the at-least-one-operand check are done -// by cobra. Errors are returned rather than exiting, so that the -// deferred close — which checkpoints the SQLite WAL — always runs. -// Cancelling ctx unwinds the worker pools and aborts the scan with the -// context's error. +// runScan implements the scan subcommand: four sequential phases — +// walk (which stats each file as it is discovered), hash, update, +// content — that synchronize the persistent database with the +// filesystem state under the PATH operands. Only files whose size at +// least one other file shares are ever hashed: a size-unique file +// cannot be a duplicate. A file of headTailMin or more gets its content +// hash only when its size, head, and tail match another file's. Flag +// parsing and the at-least-one-operand check are done by cobra. Errors +// are returned rather than exiting, so that the deferred close — which +// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the +// worker pools and aborts the scan with the context's error. func runScan(ctx context.Context, roots []string, workers int, oneFS bool, ) error { @@ -166,13 +201,16 @@ type scanState struct { } // syncScan synchronizes the database with the filesystem under roots -// in three sequential phases: walk (enumerate and stat every file, +// in four sequential phases: walk (enumerate and stat every file, // building a complete size census), hash (read only the new or // changed — or previously unhashed — files whose size at least one // other file shares, committing results in batches as they arrive), -// and update (record the size-unique files without reading them, and -// delete the records the scan no longer verifies). Records outside -// the roots are never touched. +// update (record the size-unique files without reading them, and +// delete the records the scan no longer verifies), and content (fill +// in the content hash of every record of headTailMin or more whose +// size, head, and tail match another record's). Records outside the +// roots are never touched, except that the content phase fills in +// their content hash. func syncScan(ctx context.Context, db *sql.DB, roots []string, workers int, oneFS bool, ) (scanStats, error) { @@ -207,7 +245,12 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string, return s.st, err } - return s.st, s.updatePhase(ctx) + err = s.updatePhase(ctx) + if err != nil { + return s.st, err + } + + return s.st, s.contentPhase(ctx, workers) } // loadIndex indexes the database records under the scan roots for @@ -241,7 +284,8 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error { // walkPhase drains the walk, appending every walked file's size to // the census and resolving what it can immediately: an unchanged file -// whose record already has hashes needs nothing further. It returns +// whose record already has hashes needs nothing from the hash phase +// (the content phase may still fill in its content hash). It returns // the new-or-changed files and the unchanged files whose records lack // hashes; both remain candidates until the census decides whether // their sizes are shared. @@ -380,27 +424,40 @@ func sameInode(a, b fileRec) bool { return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino } -// hashPhase hashes every queued file with the worker pool — one read -// per inode run, in inode order — committing completed records to the -// database in batches as results arrive, so a long scan persists its -// progress as it goes (an interrupted scan resumes cheaply: the next -// run skips everything already recorded). The total counts actual -// reads, so the bar shows a real ETA. A run that fails to hash is -// warned about and skipped; stale records for its paths, if any, are -// deleted by the update phase. +// hashPhase hashes every queued file with hashSignature — the head and +// tail of a file of headTailMin or more, the whole file below that — +// committing completed records to the database in batches as results +// arrive, so a long scan persists its progress as it goes (an +// interrupted scan resumes cheaply: the next run skips everything +// already recorded). A run that fails to hash is warned about and +// skipped; stale records for its paths, if any, are deleted by the +// update phase. +func (s *scanState) hashPhase(ctx context.Context, workers int) error { + runs := hashRuns(s.toHash) + s.toHash = nil + + return s.readRuns(ctx, workers, "hash", runs, hashSignature, s.recordRun) +} + +// readRuns reads runs with the worker pool, one read per inode run, in +// the order given, under a progress display named label. The workers +// compute each run's hashes with hash, and each result goes to record; +// a run that fails to read is warned about and counted as skipped +// instead. The total counts actual reads, so the bar shows a real ETA. // // Returning early — a failed database write, or a cancelled scan — must // not strand the pool: the feeder would park forever on a full jobs // channel and every worker on a full results channel. The deferred stop // is what prevents that. -func (s *scanState) hashPhase(ctx context.Context, workers int) error { - runs := hashRuns(s.toHash) - s.toHash = nil - - pool := startHashPool(ctx, runs, workers) +func (s *scanState) readRuns(ctx context.Context, workers int, + label string, runs [][]fileRec, + hash func(path string, size int64) (string, string, string, error), + record func(ctx context.Context, r hashResult) error, +) error { + pool := startHashPool(ctx, runs, workers, hash) defer pool.stop() - prog := newProgress("hash", int64(len(runs))) + prog := newProgress(label, int64(len(runs))) defer prog.finish() for range runs { @@ -417,12 +474,12 @@ func (s *scanState) hashPhase(ctx context.Context, workers int) error { if r.err != nil { s.st.skipped += len(r.run) - prog.warnf("hash %s: %v", r.run[0].path, r.err) + prog.warnf("%s %s: %v", label, r.run[0].path, r.err) continue } - err := s.recordRun(ctx, r) + err := record(ctx, r) if err != nil { return err } @@ -439,14 +496,21 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error { s.resolve(rec.path) s.batch = append(s.batch, scanRec{ - size: rec.size, - mtime: rec.mtime, - head: r.head, - tail: r.tail, - path: rec.path, + size: rec.size, + mtime: rec.mtime, + head: r.head, + tail: r.tail, + content: r.content, + path: rec.path, }) } + return s.commitFullBatch(ctx) +} + +// commitFullBatch commits the running batch once it holds +// updateBatchSize records. +func (s *scanState) commitFullBatch(ctx context.Context) error { if len(s.batch) < updateBatchSize { return nil } @@ -504,6 +568,147 @@ func (s *scanState) updatePhase(ctx context.Context) error { return applyChanges(ctx, s.db, nil, deletes, prog) } +// contentPhase fills in the content hash of every record of headTailMin +// or more that lacks one and whose size, head, and tail equal another +// record's, anywhere in the database: records from this scan and +// records stored by earlier scans, inside or outside the roots. Only +// such a file can still be a duplicate, so no other file of headTailMin +// or more is read beyond its end windows. The files are read with the +// hash phase's worker pool and their records written back in batches. A +// failed read is warned about and counted as skipped; the record keeps +// its empty content, so it is never grouped, and a later scan tries +// again. +func (s *scanState) contentPhase(ctx context.Context, workers int) error { + toRead, recs, err := s.contentCandidates(ctx) + if err != nil { + return err + } + + err = s.readRuns(ctx, workers, "content", hashRuns(toRead), + hashContentOnly, func(ctx context.Context, r hashResult) error { + // Every path in the run keeps its record's head and tail + // and gains the one content hash read for the run. + for _, f := range r.run { + rec := recs[f.path] + rec.content = r.content + s.batch = append(s.batch, rec) + } + + return s.commitFullBatch(ctx) + }) + if err != nil { + return err + } + + return applyChanges(ctx, s.db, s.batch, nil, nil) +} + +// contentCandidates returns the files the content phase reads, and +// their records by path. Every record contentCandidatesSQL returns has +// its file checked with lstat, whether or not it already has a content +// hash: a file that is gone, is no longer a regular file, or has +// changed by the walk's rule keeps its record as it is and does not +// count as a match for the others, and any other lstat error is warned +// about and counted as skipped, with the same result. If such a record +// has no content hash, it stays out of duplicate groups; if it has one, +// it is still reported until a scan covering its own tree updates or +// removes it. The files of a group that pass and have no content hash +// are read only if at least minGroupSize of the group's files pass, so +// a group whose other members are all stale costs no reads. Only the +// records to be read are kept. +func (s *scanState) contentCandidates( + ctx context.Context, +) ([]fileRec, map[string]scanRec, error) { + // The query and the checks take real time on a large database; + // without a display the scan looks hung before the reads begin. + prog := newProgress("content", -1) + defer prog.finish() + + var ( + toRead []fileRec + first scanRec // the current group's first record + passed int // the current group's files that passed the check + unread []fileRec // those of them without a content hash + ) + + recs := make(map[string]scanRec) + + // endGroup queues the current group's files to read if at least + // minGroupSize of its files passed, and drops their records if not. + endGroup := func() { + if passed >= minGroupSize { + toRead = append(toRead, unread...) + } else { + for _, f := range unread { + delete(recs, f.path) + } + } + + passed, unread = 0, nil + } + + err := loadContentCandidates(ctx, s.db, func(r scanRec, hashed bool) { + prog.increment() + + if r.size != first.size || r.head != first.head || r.tail != first.tail { + endGroup() + + first = r + } + + f, ok, err := unchangedFile(r) + if err != nil { + s.st.skipped++ + + prog.warnf("content %s: %v", r.path, err) + } + + if !ok { + return + } + + passed++ + + if !hashed { + unread = append(unread, f) + recs[r.path] = r + } + }) + if err != nil { + return nil, nil, err + } + + endGroup() + + return toRead, recs, nil +} + +// unchangedFile lstats the file r names and returns it for reading if +// it is still the regular file r records: the same size, and an mtime +// no newer than recorded (the walk's change rule). A file that is gone +// or has changed reports false; any other lstat error is returned. +func unchangedFile(r scanRec) (fileRec, bool, error) { + fi, err := os.Lstat(r.path) + if errors.Is(err, fs.ErrNotExist) { + return fileRec{}, false, nil + } + + if err != nil { + return fileRec{}, false, err + } + + if !fi.Mode().IsRegular() || fi.Size() != r.size || + fi.ModTime().Unix() > r.mtime { + return fileRec{}, false, nil + } + + dev, ino := inodeOfInfo(fi) + + return fileRec{ + path: r.path, size: r.size, mtime: r.mtime, dev: dev, ino: ino, + }, true, nil +} + // underAnyRoot reports whether path is any of the roots or lies under // one of them. func underAnyRoot(path string, roots []string) bool { @@ -834,13 +1039,16 @@ func inodeOfInfo(fi fs.FileInfo) (uint64, uint64) { return statDev(st), st.Ino } -// hashResult carries one inode run's head/tail hashes (or the error -// that prevented hashing it) from the hash workers to the hash phase. +// hashResult carries the hashes computed for one inode run (or the +// error that prevented computing them) from the pool's workers to the +// phase that started the pool: head, tail, and content from +// hashSignature, content alone from hashContentOnly. type hashResult struct { - run []fileRec - head string - tail string - err error + run []fileRec + head string + tail string + content string + err error } // hashPool owns every goroutine of the hash worker pool: the feeder @@ -856,10 +1064,10 @@ type hashPool struct { } // startHashPool starts the feeder and the workers over runs. Workers -// hash each run's first path (all paths in a run are hard links to the -// same inode) and write one result per run. -func startHashPool(ctx context.Context, runs [][]fileRec, - workers int, +// hash each run's first path with hash (all paths in a run are hard +// links to the same inode) and write one result per run. +func startHashPool(ctx context.Context, runs [][]fileRec, workers int, + hash func(path string, size int64) (string, string, string, error), ) *hashPool { ctx, cancel := context.WithCancel(ctx) @@ -871,7 +1079,7 @@ func startHashPool(ctx context.Context, runs [][]fileRec, wg.Go(func() { feedHashJobs(ctx, runs, jobs) }) for range workers { - wg.Go(func() { hashWorker(ctx, jobs, results) }) + wg.Go(func() { hashWorker(ctx, jobs, results, hash) }) } done := make(chan struct{}) @@ -917,24 +1125,25 @@ func feedHashJobs(ctx context.Context, runs [][]fileRec, } } -// hashWorker hashes one inode run at a time until jobs is closed or the -// scan is cancelled. A cancelled worker drops the runs still queued -// instead of stopping its reads of jobs: the range must run out for the -// pool to tear down, and reading a file nobody wants the hash of only -// delays that. +// hashWorker hashes one inode run at a time with hash until jobs is +// closed or the scan is cancelled. A cancelled worker drops the runs +// still queued instead of stopping its reads of jobs: the range must +// run out for the pool to tear down, and reading a file nobody wants +// the hash of only delays that. func hashWorker(ctx context.Context, jobs <-chan []fileRec, results chan<- hashResult, + hash func(path string, size int64) (string, string, string, error), ) { for run := range jobs { if ctx.Err() != nil { continue } - head, tail, err := hashHeadTail(run[0].path, run[0].size) + head, tail, content, err := hash(run[0].path, run[0].size) select { case results <- hashResult{ - run: run, head: head, tail: tail, err: err, + run: run, head: head, tail: tail, content: content, err: err, }: case <-ctx.Done(): return @@ -942,55 +1151,151 @@ func hashWorker(ctx context.Context, jobs <-chan []fileRec, } } -// emptyHash is the lowercase-hex SHA-256 of the empty input: the head -// and tail hash of every zero-length file. +// emptyHash is the lowercase-hex SHA-256 of the empty input: the head, +// tail, and content hash of every zero-length file. const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" + "27ae41e4649b934ca495991b7852b855" -// hashHeadTail returns the lowercase-hex SHA-256 of the first -// min(chunk, size) bytes and of the last min(chunk, size) bytes of the -// file at path. The two reads overlap when size < 2*chunk. size is the -// value recorded when the file was statted; a zero-length file's -// hashes are constant, so it is never even opened. -func hashHeadTail(path string, size int64) (string, string, error) { +// hashSignature computes the hashes the hash phase records for a file +// whose size is shared; with the file size they form its duplicate +// signature. A file below headTailMin is hashed in full and its +// whole-file SHA-256 is returned as head, tail, and content alike — +// that range takes no separate end-window step. For a file at or above +// headTailMin only the head and tail are computed, the SHA-256 of its +// first and last headTailWindow bytes, and content is returned empty: +// the content phase computes it with hashContentOnly once the file's +// size, head, and tail match another file's. Two files are duplicates +// only when all four agree; any mismatch means not a duplicate. size +// is the value recorded when the file was statted; a zero-length file +// has constant hashes and is never opened. +func hashSignature(path string, size int64) (string, string, string, error) { if size == 0 { - return emptyHash, emptyHash, nil + return emptyHash, emptyHash, emptyHash, nil } //nolint:gosec // hashing operator-supplied paths is the tool's purpose f, err := os.Open(path) if err != nil { - return "", "", err + return "", "", "", err } defer func() { _ = f.Close() }() - n := min(int64(chunk), size) + // Below the threshold the whole file is hashed directly, with no + // end-window step: head and tail both carry the whole-file hash. + if size < int64(headTailMin) { + content, err := hashWhole(f, size) + if err != nil { + return "", "", "", err + } - buf := make([]byte, n) + return content, content, content, nil + } - _, err = f.ReadAt(buf, 0) + head, tail, err := hashEnds(f, size) + if err != nil { + return "", "", "", err + } + + return head, tail, "", nil +} + +// hashContentOnly returns the content hash of the file at path, which +// is at least headTailMin bytes: the content phase's read. head and +// tail are returned empty, because the content phase keeps the ones its +// records already hold. +func hashContentOnly(path string, size int64) (string, string, string, error) { + //nolint:gosec // hashing operator-supplied paths is the tool's purpose + f, err := os.Open(path) + if err != nil { + return "", "", "", err + } + + defer func() { _ = f.Close() }() + + content, err := hashContent(f, size) + + return "", "", content, err +} + +// hashEnds returns the SHA-256 of the first and last headTailWindow +// bytes of f. It is called only for files at least headTailMin, which +// is far larger than two windows, so the windows never overlap and both +// reads are always full. +func hashEnds(f *os.File, size int64) (string, string, error) { + buf := make([]byte, headTailWindow) + + _, err := f.ReadAt(buf, 0) if err != nil { return "", "", err } h := sha256.Sum256(buf) + head := hex.EncodeToString(h[:]) - // When the whole file fits in one chunk the tail window is exactly - // the bytes just read: reuse the head hash instead of issuing a - // second read for every small file. - if size <= int64(chunk) { - hh := hex.EncodeToString(h[:]) - - return hh, hh, nil - } - - _, err = f.ReadAt(buf, size-n) + _, err = f.ReadAt(buf, size-int64(headTailWindow)) if err != nil { return "", "", err } t := sha256.Sum256(buf) - return hex.EncodeToString(h[:]), hex.EncodeToString(t[:]), nil + return head, hex.EncodeToString(t[:]), nil +} + +// hashContent returns the content-rung hash of f: the SHA-256 of the +// whole file when it is smaller than wholeFileMax, or of sampled +// windows when it is that size or larger. +func hashContent(f *os.File, size int64) (string, error) { + if size >= int64(wholeFileMax) { + return hashSamples(f, size) + } + + return hashWhole(f, size) +} + +// hashWhole returns the SHA-256 of the entire file. A SectionReader is +// used so the read is independent of the offset left by any end-window +// reads. Reading fewer than size bytes means the file shrank between +// the stat and the hash; that is an error rather than a hash of content +// that no longer matches the recorded size. +func hashWhole(f *os.File, size int64) (string, error) { + h := sha256.New() + + n, err := io.Copy(h, io.NewSectionReader(f, 0, size)) + if err != nil { + return "", err + } + + if n != size { + return "", fmt.Errorf("read %d of %d bytes: %w", n, size, + io.ErrUnexpectedEOF) + } + + return hex.EncodeToString(h.Sum(nil)), nil +} + +// hashSamples feeds sampleWindow bytes at each gigabyte-aligned offset +// (0, sampleStride, 2*sampleStride, ... while inside the file), in +// order, into one hash, each window truncated at end of file. This is +// the probabilistic large-file rung: two files of equal size agreeing +// on every sample are reported as duplicates without every byte being +// read. Because size is part of the signature, files of different sizes +// never reach this comparison, so the sample boundaries always align. +func hashSamples(f *os.File, size int64) (string, error) { + h := sha256.New() + buf := make([]byte, sampleWindow) + + for off := int64(0); off < size; off += int64(sampleStride) { + n := min(int64(sampleWindow), size-off) + + _, err := f.ReadAt(buf[:n], off) + if err != nil { + return "", err + } + + h.Write(buf[:n]) + } + + return hex.EncodeToString(h.Sum(nil)), nil } diff --git a/scan_test.go b/scan_test.go index cc1fbf1..f47005d 100644 --- a/scan_test.go +++ b/scan_test.go @@ -53,7 +53,23 @@ func pattern(tag byte, n int) []byte { return data } -func TestHashHeadTail(t *testing.T) { +// sig returns a file's full signature (head, tail, content), failing the +// test on any error. +func sig(t *testing.T, path string, size int64) (string, string, string) { + t.Helper() + + head, tail, content, err := hashSignature(path, size) + if err != nil { + t.Fatalf("hashSignature %s: %v", path, err) + } + + return head, tail, content +} + +// TestHashSignatureBelowThreshold verifies that a file below headTailMin +// is hashed in full and compared directly: head, tail, and content all +// carry the whole-file SHA-256, with no separate end-window step. +func TestHashSignatureBelowThreshold(t *testing.T) { t.Parallel() dir := t.TempDir() @@ -62,13 +78,10 @@ func TestHashHeadTail(t *testing.T) { name string data []byte }{ - {"empty", nil}, {"one-byte", []byte("x")}, - {"under-one-chunk", pattern(1, chunk-1)}, - {"exactly-one-chunk", pattern(2, chunk)}, - {"overlapping-reads", pattern(3, chunk+chunk/2)}, - {"exactly-two-chunks", pattern(4, 2*chunk)}, - {"beyond-two-chunks", pattern(5, 3*chunk)}, + {"one-window", pattern(1, headTailWindow)}, + {"several-windows", pattern(2, 3*headTailWindow)}, + {"near-threshold", pattern(3, headTailMin-1)}, } for _, c := range cases { t.Run(c.name, func(t *testing.T) { @@ -76,49 +89,619 @@ func TestHashHeadTail(t *testing.T) { p := writeFile(t, dir, c.name, c.data) - head, tail, err := hashHeadTail(p, int64(len(c.data))) - if err != nil { - t.Fatalf("hashHeadTail: %v", err) - } + head, tail, content := sig(t, p, int64(len(c.data))) - n := min(chunk, len(c.data)) - if want := hexSum(c.data[:n]); head != want { - t.Errorf("head = %s, want %s", head, want) - } - - if want := hexSum(c.data[len(c.data)-n:]); tail != want { - t.Errorf("tail = %s, want %s", tail, want) + whole := hexSum(c.data) + if head != whole || tail != whole || content != whole { + t.Errorf("head=%s tail=%s content=%s, want all whole-file %s", + head, tail, content, whole) } }) } } -func TestHashHeadTailErrors(t *testing.T) { +// TestHashSignatureEnds exercises the head and tail rungs, which apply +// only to files at least headTailMin. Sparse files keep the fixtures +// cheap: a difference in the first window changes only head, a +// difference in the last window changes only tail, and a difference +// between the windows changes neither end hash but does change the +// whole-file content rung (the file is below wholeFileMax). +// hashSignature leaves the content hash of a file this size to the +// content phase, so that rung is checked through a scan. +func TestHashSignatureEnds(t *testing.T) { t.Parallel() dir := t.TempDir() - _, _, err := hashHeadTail(filepath.Join(dir, "missing"), 1) + // Between headTailMin and wholeFileMax: the end-window gate is active + // and the content rung is a whole-file hash. + const size = int64(headTailMin + 2*1024*1024) + + base := sparseFile(t, dir, "ends-base", size) + headDiff := sparseFile(t, dir, "ends-head", size) + tailDiff := sparseFile(t, dir, "ends-tail", size) + midDiff := sparseFile(t, dir, "ends-mid", size) + + pokeAt(t, headDiff, 0, []byte{1}) + pokeAt(t, tailDiff, size-1, []byte{1}) + pokeAt(t, midDiff, size/2, []byte{1}) + + bHead, bTail, bContent := sig(t, base, size) + if bContent != "" { + t.Errorf("content = %q, want none from the hash phase", bContent) + } + + h, tl, _ := sig(t, headDiff, size) + if h == bHead { + t.Error("a byte in the first window did not change head") + } + + if tl != bTail { + t.Error("a byte in the first window changed tail") + } + + h, tl, _ = sig(t, tailDiff, size) + if tl == bTail { + t.Error("a byte in the last window did not change tail") + } + + if h != bHead { + t.Error("a byte in the last window changed head") + } + + h, tl, _ = sig(t, midDiff, size) + if h != bHead || tl != bTail { + t.Error("a byte between the windows changed an end hash") + } + + // base and midDiff match on size, head, and tail, so the scan reads + // both for their content hashes. + c := scanContents(t, dir, base, midDiff) + if c[midDiff] == c[base] { + t.Error("whole-file content rung ignored a byte between the windows") + } +} + +func TestHashSignatureErrors(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + + // A missing file: an error, and every hash left empty. + head, tail, content, err := hashSignature(filepath.Join(dir, "missing"), 1) if err == nil { t.Error("no error for a missing file") } + if head != "" || tail != "" || content != "" { + t.Errorf("missing file returned hashes: %q %q %q", head, tail, content) + } + // A zero-length file has constant hashes and is never opened: even // a missing path succeeds. - head, tail, err := hashHeadTail(filepath.Join(dir, "missing"), 0) - if err != nil || head != emptyHash || tail != emptyHash { - t.Errorf("empty: head=%q tail=%q err=%v, want constant hashes", - head, tail, err) + head, tail, content, err = hashSignature(filepath.Join(dir, "missing"), 0) + if err != nil || + head != emptyHash || tail != emptyHash || content != emptyHash { + t.Errorf("empty: head=%q tail=%q content=%q err=%v, "+ + "want constant hashes", head, tail, content, err) } // A file that shrank between the stat and hash passes: reading at // the stat-reported size must fail rather than emit wrong hashes. p := writeFile(t, dir, "shrunk", []byte("tiny")) - _, _, err = hashHeadTail(p, int64(2*chunk)) + head, tail, content, err = hashSignature(p, int64(2*headTailWindow)) if err == nil { t.Error("no error when the stat size exceeds the file size") } + + if head != "" || tail != "" || content != "" { + t.Errorf("shrunk file returned hashes: %q %q %q", head, tail, content) + } +} + +// sparseFile creates a file that is logically size bytes long without +// allocating blocks for the hole, so multi-gigabyte cases stay cheap. +func sparseFile(t *testing.T, dir, name string, size int64) string { + t.Helper() + + p := filepath.Join(dir, name) + + f, err := os.Create(p) //nolint:gosec // test-controlled path + if err != nil { + t.Fatal(err) + } + + err = f.Truncate(size) + if err != nil { + t.Fatal(err) + } + + err = f.Close() + if err != nil { + t.Fatal(err) + } + + return p +} + +// pokeAt writes data into an existing file at off, leaving the rest of +// the file (a sparse hole) untouched. +func pokeAt(t *testing.T, path string, off int64, data []byte) { + t.Helper() + + f, err := os.OpenFile(path, os.O_WRONLY, 0o600) //nolint:gosec // test path + if err != nil { + t.Fatal(err) + } + + _, err = f.WriteAt(data, off) + if err != nil { + t.Fatal(err) + } + + err = f.Close() + if err != nil { + t.Fatal(err) + } +} + +// scanContents scans dir into a fresh database and returns the content +// hash recorded for each file, by path, failing the test if one of want +// has none. A file of headTailMin or more gets a content hash only when +// it is scanned with a file of the same size, head, and tail. +func scanContents(t *testing.T, dir string, + want ...string, +) map[string]string { + t.Helper() + + db := openTestDB(t) + syncTree(t, db, dir) + + contents := make(map[string]string) + for _, r := range dbRecords(t, db) { + contents[r.path] = r.content + } + + for _, p := range want { + if contents[p] == "" { + t.Fatalf("%s: no content hash", p) + } + } + + return contents +} + +// TestContentRungBoundary checks the 50 MiB boundary between the two +// content rungs: just below it the whole file is hashed and any byte +// difference shows; at the boundary only the gigabyte-spaced samples are +// hashed, so a difference outside a sample window is invisible. The +// files of each pair match on size, head, and tail, so the scan reads +// both for their content hashes. +func TestContentRungBoundary(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + + // A byte that lands outside the single [0, sampleWindow) sample a + // sub-gigabyte file has, but well inside the file. + const off = 10 * 1024 * 1024 + + // Just under the boundary: the whole-file rung sees the poked byte. + under := int64(wholeFileMax - 1) + underBase := sparseFile(t, dir, "under-base", under) + underPoked := sparseFile(t, dir, "under-poked", under) + + pokeAt(t, underPoked, off, []byte{1}) + + // At the boundary: only [0, sampleWindow) is sampled, so the poked + // byte at off is invisible and the two content hashes match. + at := int64(wholeFileMax) + atBase := sparseFile(t, dir, "at-base", at) + atPoked := sparseFile(t, dir, "at-poked", at) + + pokeAt(t, atPoked, off, []byte{1}) + + c := scanContents(t, dir, underBase, underPoked, atBase, atPoked) + + if c[underBase] == c[underPoked] { + t.Error("whole-file rung ignored a byte difference below wholeFileMax") + } + + if c[atBase] != c[atPoked] { + t.Error("sampled rung saw a byte outside every sample window") + } +} + +// TestContentRungMultiGigabyte exercises the sampled rung across several +// gigabytes using sparse files: a difference inside the third sample +// window (at offset 2*sampleStride) changes the hash, while a difference +// in the gap after it does not. The three files match on size, head, +// and tail, so the scan reads each for its content hash. +func TestContentRungMultiGigabyte(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + + // Three sample windows (offsets 0, 1 GiB, 2 GiB) plus a trailing gap + // that no sample covers. + size := int64(2*sampleStride + 2*sampleWindow) + thirdSample := int64(2 * sampleStride) + gap := thirdSample + int64(sampleWindow) + + base := sparseFile(t, dir, "g-base", size) + inSample := sparseFile(t, dir, "g-insample", size) + inGap := sparseFile(t, dir, "g-ingap", size) + + pokeAt(t, inSample, thirdSample, []byte{1}) + pokeAt(t, inGap, gap, []byte{1}) + + c := scanContents(t, dir, base, inSample, inGap) + + if c[inSample] == c[base] { + t.Error("sample at 2 GiB was not read: difference there was invisible") + } + + if c[inGap] != c[base] { + t.Error("a byte in an unsampled gap changed the content hash") + } +} + +// sparseFileWithoutMatch writes name in dir as a sparse file of size +// bytes, next to another file of that size whose first byte differs. A +// scan then reads the file's head and tail, since its size is shared, +// but finds no file matching them, so it gets no content hash. +func sparseFileWithoutMatch(t *testing.T, dir, name string, + size int64, +) string { + t.Helper() + + p := sparseFile(t, dir, name, size) + other := sparseFile(t, dir, name+"-other-head", size) + + pokeAt(t, other, 0, []byte{1}) + + return p +} + +// TestScanContentGate checks that a file of headTailMin or more is read +// for its content hash only when its size, head, and tail match another +// file's: a same-size pair whose heads differ and one whose tails differ +// get no content hash and are not reported, while an identical pair is +// read and reported. +func TestScanContentGate(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + db := openTestDB(t) + + // Three sizes, so that no pair meets another. + headA := sparseFile(t, dir, "head-a", headTailMin) + headB := sparseFile(t, dir, "head-b", headTailMin) + tailA := sparseFile(t, dir, "tail-a", headTailMin+1) + tailB := sparseFile(t, dir, "tail-b", headTailMin+1) + same := []string{ + sparseFile(t, dir, "same-a", headTailMin+2), + sparseFile(t, dir, "same-b", headTailMin+2), + } + + pokeAt(t, headB, 0, []byte{1}) + pokeAt(t, tailB, headTailMin, []byte{1}) // its last byte + + syncTree(t, db, dir) + + recs := dbRecords(t, db) + for _, p := range []string{headA, headB, tailA, tailB} { + r := recordByPath(t, recs, p) + if r.head == "" || r.tail == "" || r.content != "" { + t.Errorf("%s: head = %q tail = %q content = %q, "+ + "want head and tail only", p, r.head, r.tail, r.content) + } + } + + groups := collectDupeGroups(recs) + if len(groups) != 1 || !slices.Equal(groups[0].paths, same) { + t.Fatalf("groups = %+v, want only the identical pair %q", + groups, same) + } +} + +// TestScanContentAcrossOperands checks that a stored file gets its +// content hash when a later scan of a separate operand brings its +// match: tree A's file has a head and tail but no content hash until +// tree B, holding an identical file, is scanned. +func TestScanContentAcrossOperands(t *testing.T) { + t.Parallel() + + db := openTestDB(t) + a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin) + + syncTree(t, db, filepath.Dir(a)) + + if r := recordByPath(t, dbRecords(t, db), a); r.head == "" || r.content != "" { + t.Fatalf("after scanning A: %+v, want head and tail only", r) + } + + b := sparseFile(t, t.TempDir(), "b", headTailMin) + + syncTree(t, db, filepath.Dir(b)) + + recs := dbRecords(t, db) + if r := recordByPath(t, recs, a); r.content == "" { + t.Fatalf("after scanning B: %+v, want A's file content-hashed", r) + } + + want := []string{a, b} + slices.Sort(want) + + groups := collectDupeGroups(recs) + if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { + t.Fatalf("groups = %+v, want the pair %q", groups, want) + } +} + +// TestScanContentWithinOperand checks that a rescan adding a match next +// to an unchanged stored file gives the stored file its content hash, +// though the hash phase leaves it alone as unchanged. +func TestScanContentWithinOperand(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + db := openTestDB(t) + stored := sparseFileWithoutMatch(t, dir, "d1", headTailMin) + + syncTree(t, db, dir) + + added := sparseFile(t, dir, "d2", headTailMin) + + st := syncTree(t, db, dir) + if st != (scanStats{added: 1, unchanged: 2}) { + t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st) + } + + want := []string{stored, added} + + groups := collectDupeGroups(dbRecords(t, db)) + if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { + t.Fatalf("groups = %+v, want the pair %q", groups, want) + } +} + +// TestScanContentStalePartners checks that a stored file outside the +// operand that has vanished, or changed, since it was recorded is not +// read, and that its match inside the operand is not read either: the +// match has no other partner left, so neither gets a content hash and +// no duplicate is reported. +func TestScanContentStalePartners(t *testing.T) { + t.Parallel() + + db := openTestDB(t) + dirA := t.TempDir() + gone := sparseFileWithoutMatch(t, dirA, "gone", headTailMin) + changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin+1) + + syncTree(t, db, dirA) + + before := dbRecords(t, db) + + err := os.Remove(gone) + if err != nil { + t.Fatal(err) + } + + future := time.Now().Add(time.Hour) + + err = os.Chtimes(changed, future, future) + if err != nil { + t.Fatal(err) + } + + dirB := t.TempDir() + sparseFile(t, dirB, "gone-copy", headTailMin) + sparseFile(t, dirB, "changed-copy", headTailMin+1) + + st := syncTree(t, db, dirB) + if st != (scanStats{added: 2}) { + t.Errorf("stats = %+v, want 2 added and nothing skipped", st) + } + + recs := dbRecords(t, db) + for _, r := range recs { + if r.content != "" { + t.Errorf("%s: content = %q, want none: its only match is stale", + r.path, r.content) + } + } + + for _, old := range before { + if r := recordByPath(t, recs, old.path); r != old { + t.Errorf("record = %+v, want it left as %+v", r, old) + } + } + + if groups := collectDupeGroups(recs); len(groups) != 0 { + t.Errorf("groups = %+v, want none", groups) + } +} + +// TestScanContentHashedStalePartners checks that stored matches outside +// the operand that already have a content hash are checked like any +// other: once one has vanished and the other has changed, a copy of +// them scanned in another tree has no match left, so it is not read and +// is not reported as their duplicate. +func TestScanContentHashedStalePartners(t *testing.T) { + t.Parallel() + + db := openTestDB(t) + dirA := t.TempDir() + stored := []string{ + sparseFile(t, dirA, "changed", headTailMin), + sparseFile(t, dirA, "gone", headTailMin), + } + + // The two stored files match, so this scan gives both a content + // hash. + syncTree(t, db, dirA) + + err := os.Remove(stored[1]) + if err != nil { + t.Fatal(err) + } + + future := time.Now().Add(time.Hour) + + err = os.Chtimes(stored[0], future, future) + if err != nil { + t.Fatal(err) + } + + b := sparseFile(t, t.TempDir(), "copy", headTailMin) + + st := syncTree(t, db, filepath.Dir(b)) + if st != (scanStats{added: 1}) { + t.Errorf("stats = %+v, want 1 added and nothing skipped", st) + } + + recs := dbRecords(t, db) + if r := recordByPath(t, recs, b); r.content != "" { + t.Errorf("copy: content = %q, want none: its only matches are stale", + r.content) + } + + // The stored records lie outside the operand and are left as they + // are, so they still group with each other, but not with the copy. + groups := collectDupeGroups(recs) + if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) { + t.Errorf("groups = %+v, want only the stored pair %q", groups, stored) + } +} + +// TestScanContentReadFailure checks that a failed content read is +// counted as skipped and leaves the record without a content hash, and +// that a later scan tries the read again. +func TestScanContentReadFailure(t *testing.T) { + t.Parallel() + + db := openTestDB(t) + a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin) + + syncTree(t, db, filepath.Dir(a)) + + // lstat still works on the unreadable file, so it passes the check + // and fails only when it is read. + err := os.Chmod(a, 0) + if err != nil { + t.Fatal(err) + } + + dirB := t.TempDir() + b := sparseFile(t, dirB, "b", headTailMin) + + st := syncTree(t, db, dirB) + if st != (scanStats{added: 1, skipped: 1}) { + t.Fatalf("stats = %+v, want 1 added 1 skipped", st) + } + + if r := recordByPath(t, dbRecords(t, db), a); r.content != "" { + t.Fatalf("unreadable file: %+v, want no content hash", r) + } + + err = os.Chmod(a, 0o600) + if err != nil { + t.Fatal(err) + } + + st = syncTree(t, db, dirB) + if st != (scanStats{unchanged: 1}) { + t.Fatalf("rescan stats = %+v, want 1 unchanged", st) + } + + want := []string{a, b} + slices.Sort(want) + + groups := collectDupeGroups(dbRecords(t, db)) + if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { + t.Fatalf("groups = %+v, want the pair %q after the retry", + groups, want) + } +} + +// TestScanContentCheckError checks that a stored file the content phase +// cannot lstat, for a reason other than its being gone, is counted as +// skipped and does not count as a match. +func TestScanContentCheckError(t *testing.T) { + t.Parallel() + + db := openTestDB(t) + sub := filepath.Join(t.TempDir(), "sub") + + err := os.Mkdir(sub, 0o700) + if err != nil { + t.Fatal(err) + } + + sparseFileWithoutMatch(t, sub, "a", headTailMin) + syncTree(t, db, sub) + + // Without search permission on its directory, the stored file's + // lstat fails with permission denied. + err = os.Chmod(sub, 0) + if err != nil { + t.Fatal(err) + } + + t.Cleanup(func() { + //nolint:gosec // removing the directory needs its search bit back + _ = os.Chmod(sub, 0o700) + }) + + b := sparseFile(t, t.TempDir(), "b", headTailMin) + + st := syncTree(t, db, filepath.Dir(b)) + if st != (scanStats{added: 1, skipped: 1}) { + t.Fatalf("stats = %+v, want 1 added 1 skipped", st) + } + + if r := recordByPath(t, dbRecords(t, db), b); r.content != "" { + t.Errorf("b: content = %q, want none: its only match could not be "+ + "checked", r.content) + } +} + +// TestScanContentHardlinks checks that the content phase stores the +// content hash of a hard-linked file on every one of its links. +func TestScanContentHardlinks(t *testing.T) { + t.Parallel() + + dir := t.TempDir() + db := openTestDB(t) + a := sparseFile(t, dir, "a", headTailMin) + b := filepath.Join(dir, "b") + + err := os.Link(a, b) + if err != nil { + t.Fatal(err) + } + + c := sparseFile(t, dir, "copy", headTailMin) + + st := syncTree(t, db, dir) + if st != (scanStats{added: 3}) { + t.Fatalf("stats = %+v, want 3 added", st) + } + + recs := dbRecords(t, db) + + want := recordByPath(t, recs, c).content + if want == "" { + t.Fatal("the copy has no content hash") + } + + for _, p := range []string{a, b} { + if got := recordByPath(t, recs, p).content; got != want { + t.Errorf("%s: content = %q, want %q", p, got, want) + } + } } // collectWalk runs a walk over roots and returns the emitted records @@ -706,9 +1289,9 @@ func TestScanSkipsUniqueSizes(t *testing.T) { recs := dbRecords(t, db) for _, r := range recs { - if r.head != "" || r.tail != "" { - t.Errorf("%s: head = %q tail = %q, want unhashed", - r.path, r.head, r.tail) + if r.head != "" || r.tail != "" || r.content != "" { + t.Errorf("%s: head = %q tail = %q content = %q, want unhashed", + r.path, r.head, r.tail, r.content) } } @@ -740,23 +1323,36 @@ func TestScanSkipsUniqueSizes(t *testing.T) { func TestTreesUnhashedNeverEqual(t *testing.T) { t.Parallel() - // Two trees identical except for unhashed same-name, same-size - // files (possible when the trees were scanned separately) must not - // compare equal: unhashed content is unknown. - shared := pattern(1, 100) - recs := []scanRec{ - {path: "/x/t1/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)}, - {path: "/x/t2/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)}, - {path: "/x/t1/u", size: 50}, - {path: "/x/t2/u", size: 50}, + // Two trees identical except for same-name, same-size files without + // a content hash must not compare equal: their content is unknown. + // That holds for unhashed files (possible when the trees were + // scanned separately) and for files of headTailMin or more that + // have only a head and tail. + sum := hexSum(pattern(1, 100)) + shared := []scanRec{ + {path: "/x/t1/f1", size: 100, head: sum, tail: sum, content: sum}, + {path: "/x/t2/f1", size: 100, head: sum, tail: sum, content: sum}, } - super, dirs := buildHierarchy(recs) - super.compute() + cases := map[string][]scanRec{ + "unhashed": { + {path: "/x/t1/u", size: 50}, + {path: "/x/t2/u", size: 50}, + }, + "head and tail only": { + {path: "/x/t1/u", size: headTailMin, head: "h", tail: "t"}, + {path: "/x/t2/u", size: headTailMin, head: "h", tail: "t"}, + }, + } - if tg := collectTreeGroups(dirs, super); len(tg) != 0 { - t.Fatalf("tree groups = %d, want 0 (unhashed files differ)", - len(tg)) + for name, unknown := range cases { + super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...)) + super.compute() + + if tg := collectTreeGroups(dirs, super); len(tg) != 0 { + t.Errorf("%s: tree groups = %d, want 0 (the files may differ)", + name, len(tg)) + } } } diff --git a/script/verify-lint-image-pin b/script/verify-lint-image-pin index ca6e488..fa30722 100755 --- a/script/verify-lint-image-pin +++ b/script/verify-lint-image-pin @@ -71,9 +71,9 @@ main() { "the two pins disagree:" >&2 echo "verify-lint-image-pin: $LINT_DOCKERFILE: $lint_ref" >&2 echo "verify-lint-image-pin: $MAIN_DOCKERFILE: $main_ref" >&2 - echo "verify-lint-image-pin: bump both FROM lines together, tag and" \ - "digest, so script/lint and the Dockerfile lint stage keep" \ - "running the same linter" >&2 + echo "verify-lint-image-pin: bump both FROM lines together so" \ + "script/lint and the Dockerfile lint stage keep running the" \ + "same linter" >&2 exit 1 fi diff --git a/trees.go b/trees.go index fd05c0a..f5337fc 100644 --- a/trees.go +++ b/trees.go @@ -13,9 +13,10 @@ import ( // fileSig is a file's duplicate signature; mtime is excluded. type fileSig struct { - size int64 - head string - tail string + size int64 + head string + tail string + content string } // treeNode is one directory reconstructed from the scan stream. @@ -122,14 +123,16 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) { node.files = make(map[string]fileSig) } - sig := fileSig{size: r.size, head: r.head, tail: r.tail} + sig := fileSig{ + size: r.size, head: r.head, tail: r.tail, content: r.content, + } - // An unhashed record (its size was unique when last scanned) - // has unknown content: give it a signature no other file can - // share, so trees containing it never compare equal. Real - // heads are hex, so the NUL-prefixed form cannot collide. - if sig.head == "" { - sig.head = "unhashed\x00" + r.path + // A record without a content hash has unknown content (README + // "Database"): give it a signature no other file can share, so + // trees containing it never compare equal. Real hashes are + // hex, so the NUL-prefixed form cannot collide. + if sig.content == "" { + sig.content = "unhashed\x00" + r.path } node.files[comps[len(comps)-1]] = sig @@ -188,7 +191,7 @@ func (n *treeNode) compute() { for name, sig := range n.files { entries = append(entries, "f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+ - "\x00"+sig.head+"\x00"+sig.tail) + "\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content) n.fileCount++ n.totalSize += sig.size } diff --git a/trees_test.go b/trees_test.go index ab444c3..d0995ae 100644 --- a/trees_test.go +++ b/trees_test.go @@ -7,22 +7,25 @@ import ( // Signature hashes shared by the smoke-test records. const ( - f1Head = "f1h" - f1Tail = "f1t" - f2Head = "f2h" - f2Tail = "f2t" + f1Head = "f1h" + f1Tail = "f1t" + f1Content = "f1c" + f2Head = "f2h" + f2Tail = "f2t" + f2Content = "f2c" ) // smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and // /d/t2 are identical, /d/t3 differs from them only by one filename. func smokeTreeRecs() []scanRec { return []scanRec{ - {size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"}, - {size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"}, - {size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"}, - {size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"}, - {size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"}, - {size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"}, + {size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"}, + {size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"}, + {size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"}, + {size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"}, + {size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"}, + {size: 100, head: f2Head, tail: f2Tail, content: f2Content, + path: "/d/t3/sub/f2renamed"}, } } @@ -114,8 +117,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) { const sharedTail = "same" recs := []scanRec{ - {size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"}, - {size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"}, + {size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"}, + {size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"}, } super, dirs := buildHierarchy(recs) @@ -181,8 +184,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) { // Identical sibling dirs share a parent, so their group cannot be // implied by a parent group and must be reported. recs := []scanRec{ - {size: 10, head: "h", tail: "t", path: "/p/x1/f"}, - {size: 10, head: "h", tail: "t", path: "/p/x2/f"}, + {size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"}, + {size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"}, } super, dirs := buildHierarchy(recs) @@ -203,9 +206,9 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) { // extra file, so the parents' digests differ and the x group must // be reported. recs := []scanRec{ - {size: 10, head: "h", tail: "t", path: "/p/a/x/f"}, - {size: 99, head: "e", tail: "e", path: "/p/a/extra"}, - {size: 10, head: "h", tail: "t", path: "/q/b/x/f"}, + {size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"}, + {size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"}, + {size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"}, } super, dirs := buildHierarchy(recs)