@@ -2,7 +2,6 @@
|
||||
.claude
|
||||
.DS_Store
|
||||
sfdupes
|
||||
files.dat
|
||||
*.log
|
||||
*.out
|
||||
*.test
|
||||
|
||||
@@ -27,7 +27,6 @@ node_modules/
|
||||
*.log
|
||||
|
||||
# Local scan data
|
||||
files.dat
|
||||
*.sqlite
|
||||
*.sqlite-shm
|
||||
*.sqlite-wal
|
||||
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
# Lint stage — fast feedback on formatting and lint issues
|
||||
# golangci/golangci-lint:v2.12.2 (Debian-based), 2026-08-07
|
||||
FROM golangci/golangci-lint:v2.12.2@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
|
||||
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
|
||||
WORKDIR /src
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
|
||||
+2
-2
@@ -9,8 +9,8 @@
|
||||
# stage of the main Dockerfile because script/lint must not depend on
|
||||
# the rest of that build; the two FROM lines are kept identical by
|
||||
# script/verify-lint-image-pin, run as a gate below.
|
||||
# golangci/golangci-lint:v2.12.2 (Debian-based), 2026-08-07
|
||||
FROM golangci/golangci-lint:v2.12.2@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
|
||||
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
|
||||
|
||||
WORKDIR /src
|
||||
|
||||
|
||||
@@ -46,4 +46,4 @@ hooks:
|
||||
@script/install-precommit
|
||||
|
||||
clean:
|
||||
rm -f $(BINARY) files.dat
|
||||
rm -f $(BINARY)
|
||||
|
||||
@@ -5,14 +5,20 @@
|
||||
`sfdupes` is an MIT-licensed Go CLI tool by
|
||||
[@sneak](https://sneak.berlin) that quickly identifies *candidate*
|
||||
duplicate files — and, ultimately, entire duplicate directory trees —
|
||||
across very large filesystems without reading full file contents. Files
|
||||
are considered duplicates when they have identical size, identical
|
||||
SHA-256 of their first 1024 bytes, and identical SHA-256 of their last
|
||||
1024 bytes. This is a strong candidate signal, not proof of identical
|
||||
content (the middle of the file is never read); the intended use is
|
||||
across very large filesystems without reading every byte of every file.
|
||||
Files are considered duplicates when their sizes are equal and they
|
||||
agree on a short ladder of hashes. A file under 10 MiB is hashed in full
|
||||
and compared directly. A larger file is gated first on the SHA-256 of
|
||||
its first 64 KiB and of its last 64 KiB, and only when its size and
|
||||
both of those match another file's is it read for a content hash to
|
||||
compare — the SHA-256 of the whole file when it is under 50 MiB, or of
|
||||
gigabyte-spaced 1 MiB samples when it is 50 MiB or larger. Below 50 MiB
|
||||
the content hash is proof of identical content; at or above 50 MiB it is
|
||||
a strong candidate signal rather than proof, because the gaps between
|
||||
samples are never read. The intended use is
|
||||
finding duplicate downloads and duplicated directory trees on
|
||||
multi-terabyte ZFS servers where reading every byte is prohibitively
|
||||
expensive. `scan` maintains a persistent SQLite database of file
|
||||
multi-terabyte ZFS servers where reading every byte of every file is
|
||||
prohibitively expensive. `scan` maintains a persistent SQLite database of file
|
||||
signatures that survives between runs, so it can be run from cron and
|
||||
the reports can be generated at any time from the most recent scan.
|
||||
|
||||
@@ -29,9 +35,11 @@ export SFDUPES_DATABASE="$HOME/.local/share/sfdupes/db.sqlite"
|
||||
```
|
||||
|
||||
`scan` walks one or more filesystem trees and maintains one database
|
||||
record per regular file (path, size, mtime, head hash, tail hash). The
|
||||
record per regular file (path, size, mtime, head hash, tail hash,
|
||||
content hash). The
|
||||
database persists between runs; a rescan only hashes files that are new
|
||||
or changed, and removes records for files that no longer exist.
|
||||
or changed, or that may have gained a duplicate since the last scan,
|
||||
and removes records for files that no longer exist.
|
||||
`report` reads the database and prints the file-level duplicates
|
||||
report. `trees` reads the same database and prints the duplicate-tree
|
||||
report. A missing/invalid subcommand — or a `scan` invocation with no
|
||||
@@ -47,13 +55,19 @@ completed scan.
|
||||
|
||||
Duplicate finders that hash entire files do not scale to the target
|
||||
environment: ~10 million files and ~150 TB on possibly slow or busy
|
||||
disks (a ZFS pool under resilver). Reading at most 2 KiB per file — and
|
||||
only from files whose size at least one other file shares, since a
|
||||
size-unique file cannot be a duplicate — makes a full-filesystem sweep
|
||||
tractable, and the signatures are kept in a persistent database, so
|
||||
the expensive filesystem pass is incremental: a rescan re-hashes only
|
||||
files whose recorded mtime or size changed, and all analysis happens
|
||||
offline from the database alone. The end goal is
|
||||
disks (a ZFS pool under resilver). sfdupes spends disk I/O only on files
|
||||
whose size at least one other file shares, since a size-unique file
|
||||
cannot be a duplicate. Of those, a file under 10 MiB is read in full; a
|
||||
larger one has its cheap end windows read first, and is read for a
|
||||
content hash only when its size and both end windows match another
|
||||
file's — the whole file below 50 MiB, but only gigabyte-spaced samples
|
||||
at or above 50 MiB, so the largest files are never read in full. This
|
||||
keeps a full-filesystem sweep tractable, and the signatures are kept in
|
||||
a persistent database, so the expensive filesystem pass is incremental:
|
||||
a rescan re-hashes only files whose recorded mtime or size changed, plus
|
||||
— for its content hash — a file of 10 MiB or more whose size and end
|
||||
windows have come to match another file's. All analysis happens offline
|
||||
from the database alone. The end goal is
|
||||
not individual files but whole duplicated trees — duplicate
|
||||
extractions, duplicate downloads, copied project trees — which an
|
||||
operator can consider removing as a unit.
|
||||
@@ -68,17 +82,24 @@ Goals, in order:
|
||||
downloads, copied project trees), so the operator can consider
|
||||
removing an entire subtree at once. File-level duplicate detection is
|
||||
the foundation; tree-level detection is built on top of it.
|
||||
2. **Never read full file contents.** At most 2 KiB is read per file
|
||||
(first and last 1024 bytes), and only files whose size at least
|
||||
one other file shares are read at all — a size-unique file cannot
|
||||
be a duplicate. Scale target: tens of millions of files, ~150 TB
|
||||
filesystem, possibly slow or busy disks (ZFS pool under resilver).
|
||||
Holding one small record (path, size, mtime) per file in memory
|
||||
during a scan is acceptable; holding every file's hashes is not
|
||||
(they stay in the database).
|
||||
2. **Spend I/O in proportion to duplicate likelihood.** Only files
|
||||
whose size at least one other file shares are read at all — a
|
||||
size-unique file cannot be a duplicate. Those are compared by the
|
||||
ladder in "Duplicate detection" below: a file under 10 MiB is hashed
|
||||
in full, while a larger file is gated on cheap 64 KiB end windows
|
||||
first, and gets a content hash only when its size and both end
|
||||
windows match another file's. That hash reads the whole file below
|
||||
50 MiB but only gigabyte-spaced 1 MiB samples at or above it, so the
|
||||
very largest files are still never read in full. Scale target: tens of
|
||||
millions of files, ~150 TB filesystem, possibly slow or busy disks
|
||||
(ZFS pool under resilver). Holding one small record (path, size,
|
||||
mtime) per file in memory during a scan is acceptable; holding
|
||||
every file's hashes is not (they stay in the database).
|
||||
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
||||
scan maintains a persistent database; an unchanged file is never
|
||||
read again on a rescan. All analysis (`report`, `trees`) works from
|
||||
read again on a rescan, except to compute its content hash once a
|
||||
file of 10 MiB or more comes to match another on size and both end
|
||||
windows. All analysis (`report`, `trees`) works from
|
||||
the database alone and must never touch the scanned filesystem
|
||||
again. `scan` is designed to be cronned; the reports run at any
|
||||
time against the last completed scan.
|
||||
@@ -146,21 +167,73 @@ All three subcommands operate on a single SQLite database file:
|
||||
|
||||
```sql
|
||||
CREATE TABLE files (
|
||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||
size INTEGER NOT NULL, -- bytes, from lstat
|
||||
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
||||
head TEXT NOT NULL, -- lowercase-hex SHA-256, first 1 KiB
|
||||
tail TEXT NOT NULL -- lowercase-hex SHA-256, last 1 KiB
|
||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||
size INTEGER NOT NULL, -- bytes, from lstat
|
||||
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
||||
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||
) WITHOUT ROWID;
|
||||
```
|
||||
|
||||
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
||||
guaranteed UTF-8. `mtime` is used only for change detection; it is
|
||||
not part of the duplicate key. `head` and `tail` are empty strings
|
||||
when the file has never been hashed because its size was unique as
|
||||
of the last scan that covered it; such records still define the
|
||||
file for tree reconstruction but never participate in duplicate
|
||||
groups.
|
||||
not part of the duplicate key. For a file under 10 MiB `head`, `tail`,
|
||||
and `content` all hold the whole-file hash (that range is hashed in
|
||||
full, with no end windows); for a larger file `head` and `tail` hold
|
||||
the first- and last-64 KiB hashes and `content` the whole-file or
|
||||
sampled hash. All three are empty strings when the file has never
|
||||
been hashed because its size was unique as of the last scan that
|
||||
covered it. For a file of 10 MiB or more, `content` stays empty
|
||||
until the content phase of a scan (see "`scan` mode" below) has
|
||||
read the file. A record with an empty `content` is never part of a
|
||||
duplicate group, though it still defines the file for tree
|
||||
reconstruction.
|
||||
|
||||
### Duplicate detection
|
||||
|
||||
Two files are duplicates only when they agree on every rung of this
|
||||
ladder; a mismatch at any rung means they are not duplicates. `scan`
|
||||
stores each file's hashes, and `report` and `trees` group files by the
|
||||
whole signature — size, `head`, `tail`, and `content` — so the grouping
|
||||
is exactly this ladder applied across everything scanned into the
|
||||
database, even across separate scans.
|
||||
|
||||
1. **Size.** Files of different sizes are never compared. Only files
|
||||
whose size at least one other file shares are hashed at all.
|
||||
2. **Under 10 MiB: whole file.** A file smaller than 10 MiB is hashed
|
||||
in full and compared directly, with no separate end-window step —
|
||||
small files are cheap to read to the last byte, and doing so makes
|
||||
the comparison exact. `head`, `tail`, and `content` all hold this
|
||||
whole-file SHA-256, so such a file's signature is decided entirely
|
||||
by its size and its content.
|
||||
3. **10 MiB and above: head and tail.** For a larger file, the SHA-256
|
||||
of the first 64 KiB (`head`) and of the last 64 KiB (`tail`) are a
|
||||
cheap gate that eliminates most same-size pairs before any bulk
|
||||
reading: the content hash of the next two rungs is computed only for
|
||||
a file whose size, `head`, and `tail` match another file's, whether
|
||||
that file is scanned in the same run or stored by an earlier scan.
|
||||
A stored file that first gains such a match in a later scan gets its
|
||||
content hash then; until it has one, its `content` is empty and it
|
||||
is not a duplicate. At 10 MiB and above the two windows never
|
||||
overlap.
|
||||
4. **10 MiB and above, content below 50 MiB.** The SHA-256 of the
|
||||
entire file. Agreement here is proof of identical content (barring a
|
||||
SHA-256 collision).
|
||||
5. **10 MiB and above, content 50 MiB and above.** A sampled SHA-256:
|
||||
the 1 MiB window at each gigabyte-aligned offset (0, 1 GiB, 2 GiB, …
|
||||
while inside the file, the final window truncated at end of file) is
|
||||
fed, in order, into one hash. This is **deliberately probabilistic**
|
||||
— the gaps between samples are never read, so two large files that
|
||||
agree on every sample are reported as duplicates without being read
|
||||
in full. It is the price of never reading a 150 GB file end to end.
|
||||
Because size is already part of the signature, only equal-size files
|
||||
reach this rung, so their sample boundaries always align.
|
||||
|
||||
`head`, `tail`, and `content` are one column each. A file below 10 MiB
|
||||
and one at or above it never share a size, and neither do a file below
|
||||
50 MiB and one at or above it, so a stored value is never ambiguous
|
||||
between the whole-file, end-window, and sampled forms.
|
||||
|
||||
### `scan` mode
|
||||
|
||||
@@ -183,10 +256,10 @@ scanned operands:
|
||||
|
||||
- Only a file whose size at least one other file shares is ever
|
||||
read: a size-unique file cannot be a duplicate, so it is recorded
|
||||
without hashes (`head` and `tail` empty). The size census covers
|
||||
every file walked this scan plus every database record outside
|
||||
the scanned operands, so a possible duplicate of a separately
|
||||
scanned tree is still recognized.
|
||||
without hashes (`head`, `tail`, and `content` empty). The size
|
||||
census covers every file walked this scan plus every database
|
||||
record outside the scanned operands, so a possible duplicate of a
|
||||
separately scanned tree is still recognized.
|
||||
- A file not yet in the database is inserted: hashed when its size
|
||||
is shared, without hashes otherwise.
|
||||
- A file already in the database is **skipped without reading its
|
||||
@@ -195,7 +268,10 @@ scanned operands:
|
||||
makes a daily rescan cheap. Exception: an unchanged file whose
|
||||
record lacks hashes is hashed — and its record updated — once its
|
||||
size becomes shared, so hashing deferred by size-uniqueness
|
||||
happens as soon as it could matter.
|
||||
happens as soon as it could matter. Likewise, an unchanged file of
|
||||
10 MiB or more whose record has no `content` hash is read for one
|
||||
by the content phase below once its size, `head`, and `tail` match
|
||||
another record's.
|
||||
- A file whose mtime is newer than recorded, or whose size differs,
|
||||
is processed as if new: re-hashed, or recorded without hashes,
|
||||
per the shared-size rule.
|
||||
@@ -204,12 +280,18 @@ scanned operands:
|
||||
removes records for deleted files. It also removes records for
|
||||
paths that failed to stat or hash this run: the database only ever
|
||||
contains signatures verified by the most recent scan that covered
|
||||
them (a subsequent successful scan re-adds such files).
|
||||
them (a subsequent successful scan re-adds such files). A failure
|
||||
in the content phase below removes nothing: the record is left as
|
||||
it is.
|
||||
- Database records outside the scanned operands are untouched, so
|
||||
disjoint trees can be scanned on different schedules into the same
|
||||
database.
|
||||
database. The one exception is the content phase below: a stored
|
||||
file of 10 MiB or more without a `content` hash is read for one,
|
||||
wherever it lies, once its size, `head`, and `tail` match another
|
||||
record's. If that file is gone or has changed since its record was
|
||||
written, the record is left as it is.
|
||||
|
||||
`scan` runs **three sequential phases over the whole scan**.
|
||||
`scan` runs **four sequential phases over the whole scan**.
|
||||
Parallelism lives inside each phase; batched database writes begin
|
||||
during the hash phase:
|
||||
|
||||
@@ -229,12 +311,13 @@ during the hash phase:
|
||||
decides its fate. Size-unique files are never read: new or
|
||||
changed ones are recorded without hashes in the update phase,
|
||||
unchanged unhashed ones simply keep their records. Every file
|
||||
with a shared size is hashed by the worker pool: read the first
|
||||
`min(1024, size)` bytes and the last `min(1024, size)` bytes
|
||||
(one read when `size <= 1024`, since the two windows coincide)
|
||||
and compute the SHA-256 of each. Zero-length files have constant
|
||||
hashes and are never opened. Files are hashed in **inode order**
|
||||
(minimizing seeks on spinning disks), and paths that are hard
|
||||
with a shared size is hashed by the worker pool as described in
|
||||
"Duplicate detection" above: a file under 10 MiB in full, which
|
||||
gives its `head`, `tail`, and `content` alike, and a larger file
|
||||
only in its end windows, which give its `head` and `tail`; its
|
||||
content hash is left to the content phase. Zero-length files have
|
||||
constant hashes and are never opened. Files are hashed in **inode
|
||||
order** (minimizing seeks on spinning disks), and paths that are hard
|
||||
links to the same inode are **read once**, all sharing the one
|
||||
result — a hard-link backup farm costs one read per inode, not
|
||||
per path. The phase total counts actual reads, so progress and
|
||||
@@ -246,6 +329,29 @@ during the hash phase:
|
||||
records for size-unique new and changed files, and the deletions
|
||||
for records the scan did not verify (vanished files, plus paths
|
||||
that failed to stat or hash).
|
||||
4. **content** — find every record of 10 MiB or more without a
|
||||
`content` hash whose size, `head`, and `tail` equal another
|
||||
record's, anywhere in the database: records from this scan and
|
||||
records stored by earlier scans, inside or outside the scanned
|
||||
operands. SQLite finds them, so only the records to be read are
|
||||
kept in memory, never every file's hashes. Every record sharing
|
||||
their size, `head`, and `tail`, including one that already has a
|
||||
`content` hash, has its file checked with `lstat` first. A file
|
||||
that is gone, is no longer a regular file, or has changed (a
|
||||
different size, or an mtime newer than recorded) keeps its record
|
||||
as it is and does not count as a match for the others. Any other
|
||||
`lstat` error is warned about and counted as skipped, with the same
|
||||
result. If such a record has no `content` hash, it stays out of
|
||||
duplicate groups; if it has one, it is still reported until a scan
|
||||
covering its own tree updates or removes it. The files that pass
|
||||
and have no `content` hash are read only if at least two of those
|
||||
records pass, so a file whose only matches are stale costs no read;
|
||||
a file that already has a `content` hash is never read again.
|
||||
They are read by a worker pool as in the hash phase, in inode order
|
||||
and once per inode, and their content hashes are committed in
|
||||
batches. A failed read is warned about and counted as skipped; its
|
||||
record keeps an empty `content`, so it is not a duplicate, and a
|
||||
later scan tries again.
|
||||
|
||||
Rules for the walk:
|
||||
|
||||
@@ -263,18 +369,18 @@ Rules for the walk:
|
||||
path, and continue. Per-file errors never abort the run; the final
|
||||
summary reports how many were skipped. As specified above, a
|
||||
skipped path that has a database record from an earlier scan loses
|
||||
that record; an unreadable directory subtree likewise loses its
|
||||
records (accepted: the database mirrors what the latest scan could
|
||||
actually verify).
|
||||
that record, unless it failed only in the content phase; an
|
||||
unreadable directory subtree likewise loses its records (accepted:
|
||||
the database mirrors what the latest scan could actually verify).
|
||||
|
||||
Concurrency: the walk phase (which also stats files) and the hash
|
||||
phase each use a worker pool of `--workers` workers (default
|
||||
`runtime.NumCPU()`); the walk parallelizes across directories,
|
||||
hashing across files. Both phases are seek-bound on spinning disks,
|
||||
so raising `--workers` well past the core count can help on pools
|
||||
with many spindles. The main goroutine owns partitioning, database
|
||||
writes, and progress rendering; progress display must never block
|
||||
the workers.
|
||||
Concurrency: the walk phase (which also stats files), the hash phase,
|
||||
and the content phase each use a worker pool of `--workers` workers
|
||||
(default `runtime.NumCPU()`); the walk parallelizes across
|
||||
directories, hashing across files. All three phases are seek-bound on
|
||||
spinning disks, so raising `--workers` well past the core count can
|
||||
help on pools with many spindles. The main goroutine owns
|
||||
partitioning, database writes, and progress rendering; progress
|
||||
display must never block the workers.
|
||||
|
||||
`scan` writes nothing to stdout. The summary line on stderr reports the
|
||||
files seen this run broken down by disposition, plus skips:
|
||||
@@ -299,11 +405,11 @@ mounted.
|
||||
|
||||
Processing:
|
||||
|
||||
- Records without hashes (size-unique when last scanned) are
|
||||
- Records without a `content` hash (see "Database" above) are
|
||||
excluded: their content is unknown, so they are never reported as
|
||||
duplicates.
|
||||
- Group the remaining records by the key
|
||||
`(size, head_hash, tail_hash)`.
|
||||
`(size, head, tail, content)`.
|
||||
- Every group with two or more paths is a duplicate group.
|
||||
- Within each group, sort paths lexicographically (byte order). The
|
||||
first path is the group's `first`; every other path is a `dupe`.
|
||||
@@ -339,11 +445,11 @@ the paths in the records, split on `/`.
|
||||
|
||||
Definitions:
|
||||
|
||||
- A file's **signature** is `(size, head_hash, tail_hash)` — mtime is
|
||||
informational and excluded. An unhashed record (empty hashes) has
|
||||
- A file's **signature** is `(size, head, tail, content)` — mtime is
|
||||
informational and excluded. A record without a `content` hash has
|
||||
unknown content: its signature is treated as unique to that file,
|
||||
so a tree containing an unhashed file never compares equal to any
|
||||
other tree.
|
||||
so a tree containing such a file never compares equal to any other
|
||||
tree.
|
||||
- A directory's **digest** is a SHA-256 Merkle digest computed
|
||||
bottom-up: serialize the directory's child entries — for a file
|
||||
child, its name and signature; for a subdirectory child, its name
|
||||
@@ -406,10 +512,13 @@ Each phase gets its own display, rendered the moment the phase
|
||||
starts — a scan must never look hung. Loading the existing-record
|
||||
index (`load`) and the walk have no known totals while running: show
|
||||
a live count, rate, and elapsed time (spinner-style, no percentage or
|
||||
ETA). The hash and update phases
|
||||
have exact totals — only files that actually need hashing appear in
|
||||
the hash total, so its ETA is meaningful. Required elements for the
|
||||
bars with known totals:
|
||||
ETA). The content phase's display (`content`) starts the same way,
|
||||
counting the records checked while SQLite finds the files to read and
|
||||
`lstat` checks them, then shows a bar once reading starts. The hash
|
||||
and update phases, and the content phase's reads, have exact totals —
|
||||
only files that actually need hashing appear in the hash and content
|
||||
totals, so their ETAs are meaningful. Required elements for the bars
|
||||
with known totals:
|
||||
|
||||
- elapsed time
|
||||
- estimated time remaining
|
||||
@@ -636,9 +745,10 @@ Tracked in [TODO.md](TODO.md).
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No full-content verification, no byte-for-byte compare, no deletion
|
||||
or linking of duplicates. The reports are advisory; acting on them is
|
||||
the user's job.
|
||||
- No byte-for-byte compare, and no deletion or linking of
|
||||
duplicates. Files that match are compared by a SHA-256 of the whole
|
||||
file below 50 MiB, and only by samples at 50 MiB and over. The
|
||||
reports are advisory; acting on them is the user's job.
|
||||
- No persistence beyond the SQLite database described above; no
|
||||
export/import formats.
|
||||
- No daemon or filesystem watcher; scheduling rescans is cron's job.
|
||||
|
||||
@@ -29,6 +29,40 @@
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||
content-hash ladder (2026-09-22, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||
hashed in full and compared directly, with no end-window step — its
|
||||
`head`, `tail`, and `content` all hold the whole-file hash. A file at
|
||||
10 MiB or above gets only the 64 KiB `head` and `tail` in the hash
|
||||
phase; a new content phase, after the update phase, reads it for its
|
||||
`content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB
|
||||
samples at or above — only when its size, `head`, and `tail` match
|
||||
another record's, from the same scan or stored by an earlier one, so
|
||||
a stored file gains its content hash when it gains a match. A file
|
||||
that is gone or has changed since its record was written is not
|
||||
read. The `content` column is part of the version 1 schema. `report`
|
||||
and `trees` group by the extended signature and leave out any record
|
||||
without a `content` hash, so the ladder is applied across the whole
|
||||
database. README "Duplicate detection" documents every rung including
|
||||
the probabilistic large-file path.
|
||||
|
||||
- remove the dead `files.dat` references from `Makefile`, `.gitignore`
|
||||
and `.dockerignore` (2026-09-21, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||
|
||||
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the
|
||||
redundant tag, so both pins are the policy `# image:vX.Y.Z,
|
||||
YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest
|
||||
unchanged. `script/verify-lint-image-pin` parses those `FROM` lines
|
||||
and still matches the tagless form; its advice line lost the now
|
||||
meaningless "tag and digest". With no tag in either reference, a
|
||||
tag-only disagreement no longer exists — a one-sided tag is caught as
|
||||
a plain mismatch.
|
||||
|
||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint`
|
||||
(2026-08-10, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
|
||||
|
||||
+1
-1
@@ -456,7 +456,7 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
hashWorker(cancelledContext(t), jobs, results)
|
||||
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "hashWorker")
|
||||
|
||||
@@ -36,22 +36,24 @@ const dbDirPerm = 0o755
|
||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||
const createTableSQL = `
|
||||
CREATE TABLE files (
|
||||
path BLOB PRIMARY KEY,
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL
|
||||
path BLOB PRIMARY KEY,
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL,
|
||||
content TEXT NOT NULL
|
||||
) WITHOUT ROWID
|
||||
`
|
||||
|
||||
// upsertSQL inserts one file record, replacing any existing record for
|
||||
// the same path.
|
||||
const upsertSQL = `
|
||||
INSERT INTO files (path, size, mtime, head, tail)
|
||||
VALUES (?, ?, ?, ?, ?)
|
||||
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT (path) DO UPDATE SET
|
||||
size = excluded.size, mtime = excluded.mtime,
|
||||
head = excluded.head, tail = excluded.tail
|
||||
head = excluded.head, tail = excluded.tail,
|
||||
content = excluded.content
|
||||
`
|
||||
|
||||
// errNoDatabase reports a missing database file for report/trees.
|
||||
@@ -205,7 +207,7 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
||||
// loadFileRows reads every record from the files table.
|
||||
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head, tail FROM files")
|
||||
"SELECT path, size, mtime, head, tail, content FROM files")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
@@ -220,7 +222,8 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||
r scanRec
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail)
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||
&r.content)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
@@ -275,6 +278,62 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
return nil
|
||||
}
|
||||
|
||||
// contentCandidatesSQL selects every record of at least headTailMin
|
||||
// bytes whose size, head, and tail equal another record's, in each
|
||||
// group (the records sharing a size, head, and tail) where at least one
|
||||
// record has no content hash, with whether each record has one. SQLite
|
||||
// does the grouping, so no other record's hashes are loaded into
|
||||
// memory; the rows come ordered by size, head, and tail, so each
|
||||
// group's rows arrive together.
|
||||
const contentCandidatesSQL = `
|
||||
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail
|
||||
FROM files
|
||||
WHERE size >= ? AND head <> ''
|
||||
GROUP BY size, head, tail
|
||||
HAVING COUNT(*) > 1 AND SUM(content = '') > 0
|
||||
) AS g USING (size, head, tail)
|
||||
ORDER BY size, head, tail
|
||||
`
|
||||
|
||||
// loadContentCandidates streams the rows of contentCandidatesSQL to fn:
|
||||
// each record, without its content hash, and whether it has one.
|
||||
func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||
fn func(r scanRec, hashed bool),
|
||||
) error {
|
||||
rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
fn(r, hashed != 0)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// updateBatchSize is the number of record changes committed per
|
||||
// transaction during the update pass. The filesystem is authoritative
|
||||
// and the database an eventually-consistent reflection of it, so
|
||||
@@ -348,7 +407,7 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
||||
|
||||
for _, r := range upserts {
|
||||
_, err = st.ExecContext(ctx,
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail)
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||
}
|
||||
|
||||
+10
-4
@@ -136,10 +136,14 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
db := openTestDB(t)
|
||||
|
||||
// Paths may contain tabs and newlines; the database must store
|
||||
// them byte-exactly.
|
||||
// them byte-exactly. Every hash, content included, comes back as
|
||||
// written.
|
||||
recs := []scanRec{
|
||||
{size: 2, mtime: 20, head: "h2", tail: "t2", path: "/a/tab\tnew\nline"},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", path: "/a/x"},
|
||||
{
|
||||
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||
path: "/a/tab\tnew\nline",
|
||||
},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||
}
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil,
|
||||
@@ -163,7 +167,9 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
|
||||
// An upsert for an existing path updates in place; a delete
|
||||
// removes exactly its path.
|
||||
upd := scanRec{size: 3, mtime: 30, head: "h3", tail: "t3", path: "/a/x"}
|
||||
upd := scanRec{
|
||||
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
||||
|
||||
@@ -1,10 +1,14 @@
|
||||
// Command sfdupes quickly identifies candidate duplicate files across
|
||||
// very large filesystems without reading full file contents. Files are
|
||||
// considered duplicates when they have identical size, identical SHA-256
|
||||
// of their first 1024 bytes, and identical SHA-256 of their last 1024
|
||||
// bytes. scan maintains a persistent SQLite database of file signatures
|
||||
// (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) that the
|
||||
// reporting subcommands read.
|
||||
// very large filesystems without reading every byte of every file.
|
||||
// Files are considered duplicates when their sizes are equal and they
|
||||
// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is
|
||||
// hashed in full. A larger file is compared on the hashes of its first
|
||||
// and last 64 KiB, and only when those match another file's is its
|
||||
// content hash computed and compared: of the whole file when it is
|
||||
// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB
|
||||
// or larger. scan maintains a persistent SQLite database of file
|
||||
// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite)
|
||||
// that the reporting subcommands read.
|
||||
//
|
||||
// Usage:
|
||||
//
|
||||
@@ -97,7 +101,7 @@ func run(args []string, stderr io.Writer) int {
|
||||
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
root := &cobra.Command{
|
||||
Use: "sfdupes",
|
||||
Short: "Find candidate duplicate files by size and head/tail SHA-256",
|
||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||
Version: Version,
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
@@ -127,7 +131,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
}),
|
||||
}
|
||||
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
||||
"concurrent workers for the walk and hash phases")
|
||||
"concurrent workers for the walk, hash, and content phases")
|
||||
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
||||
"do not cross filesystem boundaries")
|
||||
|
||||
|
||||
@@ -17,14 +17,15 @@ const ioBufSize = 1 << 20
|
||||
const minGroupSize = 2
|
||||
|
||||
// scanRec is one file record from the database. The signature (size,
|
||||
// head, tail) is the duplicate key; mtime is informational only and
|
||||
// used by scan for change detection.
|
||||
// head, tail, content) is the duplicate key; mtime is informational
|
||||
// only and used by scan for change detection.
|
||||
type scanRec struct {
|
||||
size int64
|
||||
mtime int64
|
||||
head string
|
||||
tail string
|
||||
path string
|
||||
size int64
|
||||
mtime int64
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
path string
|
||||
}
|
||||
|
||||
// loadRecords opens the database and reads every file record for the
|
||||
@@ -52,8 +53,9 @@ func loadRecords(ctx context.Context) ([]scanRec, error) {
|
||||
}
|
||||
|
||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||
// head hash, and tail hash. paths is sorted lexicographically; the
|
||||
// first entry is the group's "first", the rest are dupes.
|
||||
// head hash, tail hash, and content hash. paths is sorted
|
||||
// lexicographically; the first entry is the group's "first", the rest
|
||||
// are dupes.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
@@ -115,14 +117,15 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||
groups := make(map[fileSig][]string)
|
||||
|
||||
for _, r := range recs {
|
||||
// A record without hashes (its size was unique when last
|
||||
// scanned) has unknown content and is never reported as a
|
||||
// duplicate.
|
||||
if r.head == "" {
|
||||
// A record without a content hash has unknown content and is
|
||||
// never reported as a duplicate (README "Database").
|
||||
if r.content == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
k := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
k := fileSig{
|
||||
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||
}
|
||||
groups[k] = append(groups[k], r.path)
|
||||
}
|
||||
|
||||
|
||||
+43
-17
@@ -9,15 +9,15 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/1"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
||||
// Same size as the /z group but a different head hash.
|
||||
{size: 100, head: "other", tail: "t", path: "/z/d"},
|
||||
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
||||
// A singleton signature must not form a group.
|
||||
{size: 7, head: "u", tail: "u", path: "/lonely"},
|
||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
@@ -38,14 +38,40 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Same size, head, and tail, but different content hashes: the final
|
||||
// rung keeps them apart, so no group forms. Matching content groups.
|
||||
// Records without a content hash never group, not even with each
|
||||
// other.
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
||||
{size: 100, head: "h", tail: "t", path: "/d"},
|
||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||
len(groups))
|
||||
}
|
||||
|
||||
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
||||
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// mtime is informational only; records differing only in mtime
|
||||
// still group together.
|
||||
recs := []scanRec{
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", path: "/m/2"},
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
@@ -58,10 +84,10 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/2"},
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/1"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/2"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/1"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
@@ -80,10 +106,10 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 1, head: "a", tail: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/2"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||
}
|
||||
|
||||
forward := collectDupeGroups(recs)
|
||||
|
||||
@@ -6,7 +6,9 @@ import (
|
||||
"crypto/sha256"
|
||||
"database/sql"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -16,8 +18,40 @@ import (
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// chunk is the number of bytes hashed from each end of a file.
|
||||
const chunk = 1024
|
||||
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||
// detection"). A same-size candidate below headTailMin is hashed in
|
||||
// full and compared directly; a larger one is separated first by the
|
||||
// hashes of its end windows, then by a content hash that is exact below
|
||||
// wholeFileMax and deliberately sampled at or above it. The hash phase
|
||||
// reads only the end windows of a larger file; the content phase reads
|
||||
// it for its content hash only once its size, head, and tail match
|
||||
// another file's.
|
||||
|
||||
// headTailMin is the size threshold for the end-window gate. A file
|
||||
// smaller than this is hashed in full directly, with no separate head
|
||||
// and tail step: its head, tail, and content all carry the whole-file
|
||||
// hash. A file this size or larger is separated first by its end
|
||||
// windows.
|
||||
const headTailMin = 10 * 1024 * 1024
|
||||
|
||||
// headTailWindow is the number of bytes hashed from each end of a file
|
||||
// at or above headTailMin (the head and tail rungs). Because
|
||||
// headTailMin is far larger than two windows, the head and tail windows
|
||||
// never overlap.
|
||||
const headTailWindow = 64 * 1024
|
||||
|
||||
// wholeFileMax is the size boundary between the two content rungs: a
|
||||
// file strictly smaller than this is content-hashed in full; a file
|
||||
// this size or larger is content-hashed by sampling.
|
||||
const wholeFileMax = 50 * 1024 * 1024
|
||||
|
||||
// sampleStride is the spacing between content samples for large files:
|
||||
// one window is read at each gigabyte-aligned offset (0, 1 GiB, ...).
|
||||
const sampleStride = 1024 * 1024 * 1024
|
||||
|
||||
// sampleWindow is the number of bytes read at each large-file sample
|
||||
// offset, truncated at end of file.
|
||||
const sampleWindow = 1024 * 1024
|
||||
|
||||
// workQueueDepth bounds the job and result channels feeding the walk
|
||||
// and hash worker pools.
|
||||
@@ -44,16 +78,17 @@ type fileMeta struct {
|
||||
hashed bool
|
||||
}
|
||||
|
||||
// runScan implements the scan subcommand: three sequential phases —
|
||||
// walk (which stats each file as it is discovered), hash, update —
|
||||
// that synchronize the persistent database with the filesystem state
|
||||
// under the PATH operands. Only files whose size at least one other
|
||||
// file shares are ever hashed: a size-unique file cannot be a
|
||||
// duplicate. Flag parsing and the at-least-one-operand check are done
|
||||
// by cobra. Errors are returned rather than exiting, so that the
|
||||
// deferred close — which checkpoints the SQLite WAL — always runs.
|
||||
// Cancelling ctx unwinds the worker pools and aborts the scan with the
|
||||
// context's error.
|
||||
// runScan implements the scan subcommand: four sequential phases —
|
||||
// walk (which stats each file as it is discovered), hash, update,
|
||||
// content — that synchronize the persistent database with the
|
||||
// filesystem state under the PATH operands. Only files whose size at
|
||||
// least one other file shares are ever hashed: a size-unique file
|
||||
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||
// hash only when its size, head, and tail match another file's. Flag
|
||||
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||
// are returned rather than exiting, so that the deferred close — which
|
||||
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||
// worker pools and aborts the scan with the context's error.
|
||||
func runScan(ctx context.Context, roots []string, workers int,
|
||||
oneFS bool,
|
||||
) error {
|
||||
@@ -166,13 +201,16 @@ type scanState struct {
|
||||
}
|
||||
|
||||
// syncScan synchronizes the database with the filesystem under roots
|
||||
// in three sequential phases: walk (enumerate and stat every file,
|
||||
// in four sequential phases: walk (enumerate and stat every file,
|
||||
// building a complete size census), hash (read only the new or
|
||||
// changed — or previously unhashed — files whose size at least one
|
||||
// other file shares, committing results in batches as they arrive),
|
||||
// and update (record the size-unique files without reading them, and
|
||||
// delete the records the scan no longer verifies). Records outside
|
||||
// the roots are never touched.
|
||||
// update (record the size-unique files without reading them, and
|
||||
// delete the records the scan no longer verifies), and content (fill
|
||||
// in the content hash of every record of headTailMin or more whose
|
||||
// size, head, and tail match another record's). Records outside the
|
||||
// roots are never touched, except that the content phase fills in
|
||||
// their content hash.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
@@ -207,7 +245,12 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
return s.st, s.updatePhase(ctx)
|
||||
err = s.updatePhase(ctx)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
return s.st, s.contentPhase(ctx, workers)
|
||||
}
|
||||
|
||||
// loadIndex indexes the database records under the scan roots for
|
||||
@@ -241,7 +284,8 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
||||
|
||||
// walkPhase drains the walk, appending every walked file's size to
|
||||
// the census and resolving what it can immediately: an unchanged file
|
||||
// whose record already has hashes needs nothing further. It returns
|
||||
// whose record already has hashes needs nothing from the hash phase
|
||||
// (the content phase may still fill in its content hash). It returns
|
||||
// the new-or-changed files and the unchanged files whose records lack
|
||||
// hashes; both remain candidates until the census decides whether
|
||||
// their sizes are shared.
|
||||
@@ -380,27 +424,40 @@ func sameInode(a, b fileRec) bool {
|
||||
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
||||
}
|
||||
|
||||
// hashPhase hashes every queued file with the worker pool — one read
|
||||
// per inode run, in inode order — committing completed records to the
|
||||
// database in batches as results arrive, so a long scan persists its
|
||||
// progress as it goes (an interrupted scan resumes cheaply: the next
|
||||
// run skips everything already recorded). The total counts actual
|
||||
// reads, so the bar shows a real ETA. A run that fails to hash is
|
||||
// warned about and skipped; stale records for its paths, if any, are
|
||||
// deleted by the update phase.
|
||||
// hashPhase hashes every queued file with hashSignature — the head and
|
||||
// tail of a file of headTailMin or more, the whole file below that —
|
||||
// committing completed records to the database in batches as results
|
||||
// arrive, so a long scan persists its progress as it goes (an
|
||||
// interrupted scan resumes cheaply: the next run skips everything
|
||||
// already recorded). A run that fails to hash is warned about and
|
||||
// skipped; stale records for its paths, if any, are deleted by the
|
||||
// update phase.
|
||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
runs := hashRuns(s.toHash)
|
||||
s.toHash = nil
|
||||
|
||||
return s.readRuns(ctx, workers, "hash", runs, hashSignature, s.recordRun)
|
||||
}
|
||||
|
||||
// readRuns reads runs with the worker pool, one read per inode run, in
|
||||
// the order given, under a progress display named label. The workers
|
||||
// compute each run's hashes with hash, and each result goes to record;
|
||||
// a run that fails to read is warned about and counted as skipped
|
||||
// instead. The total counts actual reads, so the bar shows a real ETA.
|
||||
//
|
||||
// Returning early — a failed database write, or a cancelled scan — must
|
||||
// not strand the pool: the feeder would park forever on a full jobs
|
||||
// channel and every worker on a full results channel. The deferred stop
|
||||
// is what prevents that.
|
||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
runs := hashRuns(s.toHash)
|
||||
s.toHash = nil
|
||||
|
||||
pool := startHashPool(ctx, runs, workers)
|
||||
func (s *scanState) readRuns(ctx context.Context, workers int,
|
||||
label string, runs [][]fileRec,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
record func(ctx context.Context, r hashResult) error,
|
||||
) error {
|
||||
pool := startHashPool(ctx, runs, workers, hash)
|
||||
defer pool.stop()
|
||||
|
||||
prog := newProgress("hash", int64(len(runs)))
|
||||
prog := newProgress(label, int64(len(runs)))
|
||||
defer prog.finish()
|
||||
|
||||
for range runs {
|
||||
@@ -417,12 +474,12 @@ func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
if r.err != nil {
|
||||
s.st.skipped += len(r.run)
|
||||
|
||||
prog.warnf("hash %s: %v", r.run[0].path, r.err)
|
||||
prog.warnf("%s %s: %v", label, r.run[0].path, r.err)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
err := s.recordRun(ctx, r)
|
||||
err := record(ctx, r)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -439,14 +496,21 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
||||
s.resolve(rec.path)
|
||||
|
||||
s.batch = append(s.batch, scanRec{
|
||||
size: rec.size,
|
||||
mtime: rec.mtime,
|
||||
head: r.head,
|
||||
tail: r.tail,
|
||||
path: rec.path,
|
||||
size: rec.size,
|
||||
mtime: rec.mtime,
|
||||
head: r.head,
|
||||
tail: r.tail,
|
||||
content: r.content,
|
||||
path: rec.path,
|
||||
})
|
||||
}
|
||||
|
||||
return s.commitFullBatch(ctx)
|
||||
}
|
||||
|
||||
// commitFullBatch commits the running batch once it holds
|
||||
// updateBatchSize records.
|
||||
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||
if len(s.batch) < updateBatchSize {
|
||||
return nil
|
||||
}
|
||||
@@ -504,6 +568,147 @@ func (s *scanState) updatePhase(ctx context.Context) error {
|
||||
return applyChanges(ctx, s.db, nil, deletes, prog)
|
||||
}
|
||||
|
||||
// contentPhase fills in the content hash of every record of headTailMin
|
||||
// or more that lacks one and whose size, head, and tail equal another
|
||||
// record's, anywhere in the database: records from this scan and
|
||||
// records stored by earlier scans, inside or outside the roots. Only
|
||||
// such a file can still be a duplicate, so no other file of headTailMin
|
||||
// or more is read beyond its end windows. The files are read with the
|
||||
// hash phase's worker pool and their records written back in batches. A
|
||||
// failed read is warned about and counted as skipped; the record keeps
|
||||
// its empty content, so it is never grouped, and a later scan tries
|
||||
// again.
|
||||
func (s *scanState) contentPhase(ctx context.Context, workers int) error {
|
||||
toRead, recs, err := s.contentCandidates(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = s.readRuns(ctx, workers, "content", hashRuns(toRead),
|
||||
hashContentOnly, func(ctx context.Context, r hashResult) error {
|
||||
// Every path in the run keeps its record's head and tail
|
||||
// and gains the one content hash read for the run.
|
||||
for _, f := range r.run {
|
||||
rec := recs[f.path]
|
||||
rec.content = r.content
|
||||
s.batch = append(s.batch, rec)
|
||||
}
|
||||
|
||||
return s.commitFullBatch(ctx)
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return applyChanges(ctx, s.db, s.batch, nil, nil)
|
||||
}
|
||||
|
||||
// contentCandidates returns the files the content phase reads, and
|
||||
// their records by path. Every record contentCandidatesSQL returns has
|
||||
// its file checked with lstat, whether or not it already has a content
|
||||
// hash: a file that is gone, is no longer a regular file, or has
|
||||
// changed by the walk's rule keeps its record as it is and does not
|
||||
// count as a match for the others, and any other lstat error is warned
|
||||
// about and counted as skipped, with the same result. If such a record
|
||||
// has no content hash, it stays out of duplicate groups; if it has one,
|
||||
// it is still reported until a scan covering its own tree updates or
|
||||
// removes it. The files of a group that pass and have no content hash
|
||||
// are read only if at least minGroupSize of the group's files pass, so
|
||||
// a group whose other members are all stale costs no reads. Only the
|
||||
// records to be read are kept.
|
||||
func (s *scanState) contentCandidates(
|
||||
ctx context.Context,
|
||||
) ([]fileRec, map[string]scanRec, error) {
|
||||
// The query and the checks take real time on a large database;
|
||||
// without a display the scan looks hung before the reads begin.
|
||||
prog := newProgress("content", -1)
|
||||
defer prog.finish()
|
||||
|
||||
var (
|
||||
toRead []fileRec
|
||||
first scanRec // the current group's first record
|
||||
passed int // the current group's files that passed the check
|
||||
unread []fileRec // those of them without a content hash
|
||||
)
|
||||
|
||||
recs := make(map[string]scanRec)
|
||||
|
||||
// endGroup queues the current group's files to read if at least
|
||||
// minGroupSize of its files passed, and drops their records if not.
|
||||
endGroup := func() {
|
||||
if passed >= minGroupSize {
|
||||
toRead = append(toRead, unread...)
|
||||
} else {
|
||||
for _, f := range unread {
|
||||
delete(recs, f.path)
|
||||
}
|
||||
}
|
||||
|
||||
passed, unread = 0, nil
|
||||
}
|
||||
|
||||
err := loadContentCandidates(ctx, s.db, func(r scanRec, hashed bool) {
|
||||
prog.increment()
|
||||
|
||||
if r.size != first.size || r.head != first.head || r.tail != first.tail {
|
||||
endGroup()
|
||||
|
||||
first = r
|
||||
}
|
||||
|
||||
f, ok, err := unchangedFile(r)
|
||||
if err != nil {
|
||||
s.st.skipped++
|
||||
|
||||
prog.warnf("content %s: %v", r.path, err)
|
||||
}
|
||||
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
|
||||
passed++
|
||||
|
||||
if !hashed {
|
||||
unread = append(unread, f)
|
||||
recs[r.path] = r
|
||||
}
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
|
||||
endGroup()
|
||||
|
||||
return toRead, recs, nil
|
||||
}
|
||||
|
||||
// unchangedFile lstats the file r names and returns it for reading if
|
||||
// it is still the regular file r records: the same size, and an mtime
|
||||
// no newer than recorded (the walk's change rule). A file that is gone
|
||||
// or has changed reports false; any other lstat error is returned.
|
||||
func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||
fi, err := os.Lstat(r.path)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fileRec{}, false, err
|
||||
}
|
||||
|
||||
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||
fi.ModTime().Unix() > r.mtime {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
return fileRec{
|
||||
path: r.path, size: r.size, mtime: r.mtime, dev: dev, ino: ino,
|
||||
}, true, nil
|
||||
}
|
||||
|
||||
// underAnyRoot reports whether path is any of the roots or lies under
|
||||
// one of them.
|
||||
func underAnyRoot(path string, roots []string) bool {
|
||||
@@ -834,13 +1039,16 @@ func inodeOfInfo(fi fs.FileInfo) (uint64, uint64) {
|
||||
return statDev(st), st.Ino
|
||||
}
|
||||
|
||||
// hashResult carries one inode run's head/tail hashes (or the error
|
||||
// that prevented hashing it) from the hash workers to the hash phase.
|
||||
// hashResult carries the hashes computed for one inode run (or the
|
||||
// error that prevented computing them) from the pool's workers to the
|
||||
// phase that started the pool: head, tail, and content from
|
||||
// hashSignature, content alone from hashContentOnly.
|
||||
type hashResult struct {
|
||||
run []fileRec
|
||||
head string
|
||||
tail string
|
||||
err error
|
||||
run []fileRec
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
err error
|
||||
}
|
||||
|
||||
// hashPool owns every goroutine of the hash worker pool: the feeder
|
||||
@@ -856,10 +1064,10 @@ type hashPool struct {
|
||||
}
|
||||
|
||||
// startHashPool starts the feeder and the workers over runs. Workers
|
||||
// hash each run's first path (all paths in a run are hard links to the
|
||||
// same inode) and write one result per run.
|
||||
func startHashPool(ctx context.Context, runs [][]fileRec,
|
||||
workers int,
|
||||
// hash each run's first path with hash (all paths in a run are hard
|
||||
// links to the same inode) and write one result per run.
|
||||
func startHashPool(ctx context.Context, runs [][]fileRec, workers int,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
) *hashPool {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
|
||||
@@ -871,7 +1079,7 @@ func startHashPool(ctx context.Context, runs [][]fileRec,
|
||||
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
||||
|
||||
for range workers {
|
||||
wg.Go(func() { hashWorker(ctx, jobs, results) })
|
||||
wg.Go(func() { hashWorker(ctx, jobs, results, hash) })
|
||||
}
|
||||
|
||||
done := make(chan struct{})
|
||||
@@ -917,24 +1125,25 @@ func feedHashJobs(ctx context.Context, runs [][]fileRec,
|
||||
}
|
||||
}
|
||||
|
||||
// hashWorker hashes one inode run at a time until jobs is closed or the
|
||||
// scan is cancelled. A cancelled worker drops the runs still queued
|
||||
// instead of stopping its reads of jobs: the range must run out for the
|
||||
// pool to tear down, and reading a file nobody wants the hash of only
|
||||
// delays that.
|
||||
// hashWorker hashes one inode run at a time with hash until jobs is
|
||||
// closed or the scan is cancelled. A cancelled worker drops the runs
|
||||
// still queued instead of stopping its reads of jobs: the range must
|
||||
// run out for the pool to tear down, and reading a file nobody wants
|
||||
// the hash of only delays that.
|
||||
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||
results chan<- hashResult,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
) {
|
||||
for run := range jobs {
|
||||
if ctx.Err() != nil {
|
||||
continue
|
||||
}
|
||||
|
||||
head, tail, err := hashHeadTail(run[0].path, run[0].size)
|
||||
head, tail, content, err := hash(run[0].path, run[0].size)
|
||||
|
||||
select {
|
||||
case results <- hashResult{
|
||||
run: run, head: head, tail: tail, err: err,
|
||||
run: run, head: head, tail: tail, content: content, err: err,
|
||||
}:
|
||||
case <-ctx.Done():
|
||||
return
|
||||
@@ -942,55 +1151,151 @@ func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||
}
|
||||
}
|
||||
|
||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head
|
||||
// and tail hash of every zero-length file.
|
||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head,
|
||||
// tail, and content hash of every zero-length file.
|
||||
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
||||
"27ae41e4649b934ca495991b7852b855"
|
||||
|
||||
// hashHeadTail returns the lowercase-hex SHA-256 of the first
|
||||
// min(chunk, size) bytes and of the last min(chunk, size) bytes of the
|
||||
// file at path. The two reads overlap when size < 2*chunk. size is the
|
||||
// value recorded when the file was statted; a zero-length file's
|
||||
// hashes are constant, so it is never even opened.
|
||||
func hashHeadTail(path string, size int64) (string, string, error) {
|
||||
// hashSignature computes the hashes the hash phase records for a file
|
||||
// whose size is shared; with the file size they form its duplicate
|
||||
// signature. A file below headTailMin is hashed in full and its
|
||||
// whole-file SHA-256 is returned as head, tail, and content alike —
|
||||
// that range takes no separate end-window step. For a file at or above
|
||||
// headTailMin only the head and tail are computed, the SHA-256 of its
|
||||
// first and last headTailWindow bytes, and content is returned empty:
|
||||
// the content phase computes it with hashContentOnly once the file's
|
||||
// size, head, and tail match another file's. Two files are duplicates
|
||||
// only when all four agree; any mismatch means not a duplicate. size
|
||||
// is the value recorded when the file was statted; a zero-length file
|
||||
// has constant hashes and is never opened.
|
||||
func hashSignature(path string, size int64) (string, string, string, error) {
|
||||
if size == 0 {
|
||||
return emptyHash, emptyHash, nil
|
||||
return emptyHash, emptyHash, emptyHash, nil
|
||||
}
|
||||
|
||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
defer func() { _ = f.Close() }()
|
||||
|
||||
n := min(int64(chunk), size)
|
||||
// Below the threshold the whole file is hashed directly, with no
|
||||
// end-window step: head and tail both carry the whole-file hash.
|
||||
if size < int64(headTailMin) {
|
||||
content, err := hashWhole(f, size)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
buf := make([]byte, n)
|
||||
return content, content, content, nil
|
||||
}
|
||||
|
||||
_, err = f.ReadAt(buf, 0)
|
||||
head, tail, err := hashEnds(f, size)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
return head, tail, "", nil
|
||||
}
|
||||
|
||||
// hashContentOnly returns the content hash of the file at path, which
|
||||
// is at least headTailMin bytes: the content phase's read. head and
|
||||
// tail are returned empty, because the content phase keeps the ones its
|
||||
// records already hold.
|
||||
func hashContentOnly(path string, size int64) (string, string, string, error) {
|
||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
defer func() { _ = f.Close() }()
|
||||
|
||||
content, err := hashContent(f, size)
|
||||
|
||||
return "", "", content, err
|
||||
}
|
||||
|
||||
// hashEnds returns the SHA-256 of the first and last headTailWindow
|
||||
// bytes of f. It is called only for files at least headTailMin, which
|
||||
// is far larger than two windows, so the windows never overlap and both
|
||||
// reads are always full.
|
||||
func hashEnds(f *os.File, size int64) (string, string, error) {
|
||||
buf := make([]byte, headTailWindow)
|
||||
|
||||
_, err := f.ReadAt(buf, 0)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
h := sha256.Sum256(buf)
|
||||
head := hex.EncodeToString(h[:])
|
||||
|
||||
// When the whole file fits in one chunk the tail window is exactly
|
||||
// the bytes just read: reuse the head hash instead of issuing a
|
||||
// second read for every small file.
|
||||
if size <= int64(chunk) {
|
||||
hh := hex.EncodeToString(h[:])
|
||||
|
||||
return hh, hh, nil
|
||||
}
|
||||
|
||||
_, err = f.ReadAt(buf, size-n)
|
||||
_, err = f.ReadAt(buf, size-int64(headTailWindow))
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
t := sha256.Sum256(buf)
|
||||
|
||||
return hex.EncodeToString(h[:]), hex.EncodeToString(t[:]), nil
|
||||
return head, hex.EncodeToString(t[:]), nil
|
||||
}
|
||||
|
||||
// hashContent returns the content-rung hash of f: the SHA-256 of the
|
||||
// whole file when it is smaller than wholeFileMax, or of sampled
|
||||
// windows when it is that size or larger.
|
||||
func hashContent(f *os.File, size int64) (string, error) {
|
||||
if size >= int64(wholeFileMax) {
|
||||
return hashSamples(f, size)
|
||||
}
|
||||
|
||||
return hashWhole(f, size)
|
||||
}
|
||||
|
||||
// hashWhole returns the SHA-256 of the entire file. A SectionReader is
|
||||
// used so the read is independent of the offset left by any end-window
|
||||
// reads. Reading fewer than size bytes means the file shrank between
|
||||
// the stat and the hash; that is an error rather than a hash of content
|
||||
// that no longer matches the recorded size.
|
||||
func hashWhole(f *os.File, size int64) (string, error) {
|
||||
h := sha256.New()
|
||||
|
||||
n, err := io.Copy(h, io.NewSectionReader(f, 0, size))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
if n != size {
|
||||
return "", fmt.Errorf("read %d of %d bytes: %w", n, size,
|
||||
io.ErrUnexpectedEOF)
|
||||
}
|
||||
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
// hashSamples feeds sampleWindow bytes at each gigabyte-aligned offset
|
||||
// (0, sampleStride, 2*sampleStride, ... while inside the file), in
|
||||
// order, into one hash, each window truncated at end of file. This is
|
||||
// the probabilistic large-file rung: two files of equal size agreeing
|
||||
// on every sample are reported as duplicates without every byte being
|
||||
// read. Because size is part of the signature, files of different sizes
|
||||
// never reach this comparison, so the sample boundaries always align.
|
||||
func hashSamples(f *os.File, size int64) (string, error) {
|
||||
h := sha256.New()
|
||||
buf := make([]byte, sampleWindow)
|
||||
|
||||
for off := int64(0); off < size; off += int64(sampleStride) {
|
||||
n := min(int64(sampleWindow), size-off)
|
||||
|
||||
_, err := f.ReadAt(buf[:n], off)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
h.Write(buf[:n])
|
||||
}
|
||||
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
+638
-42
@@ -53,7 +53,23 @@ func pattern(tag byte, n int) []byte {
|
||||
return data
|
||||
}
|
||||
|
||||
func TestHashHeadTail(t *testing.T) {
|
||||
// sig returns a file's full signature (head, tail, content), failing the
|
||||
// test on any error.
|
||||
func sig(t *testing.T, path string, size int64) (string, string, string) {
|
||||
t.Helper()
|
||||
|
||||
head, tail, content, err := hashSignature(path, size)
|
||||
if err != nil {
|
||||
t.Fatalf("hashSignature %s: %v", path, err)
|
||||
}
|
||||
|
||||
return head, tail, content
|
||||
}
|
||||
|
||||
// TestHashSignatureBelowThreshold verifies that a file below headTailMin
|
||||
// is hashed in full and compared directly: head, tail, and content all
|
||||
// carry the whole-file SHA-256, with no separate end-window step.
|
||||
func TestHashSignatureBelowThreshold(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
@@ -62,13 +78,10 @@ func TestHashHeadTail(t *testing.T) {
|
||||
name string
|
||||
data []byte
|
||||
}{
|
||||
{"empty", nil},
|
||||
{"one-byte", []byte("x")},
|
||||
{"under-one-chunk", pattern(1, chunk-1)},
|
||||
{"exactly-one-chunk", pattern(2, chunk)},
|
||||
{"overlapping-reads", pattern(3, chunk+chunk/2)},
|
||||
{"exactly-two-chunks", pattern(4, 2*chunk)},
|
||||
{"beyond-two-chunks", pattern(5, 3*chunk)},
|
||||
{"one-window", pattern(1, headTailWindow)},
|
||||
{"several-windows", pattern(2, 3*headTailWindow)},
|
||||
{"near-threshold", pattern(3, headTailMin-1)},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
@@ -76,49 +89,619 @@ func TestHashHeadTail(t *testing.T) {
|
||||
|
||||
p := writeFile(t, dir, c.name, c.data)
|
||||
|
||||
head, tail, err := hashHeadTail(p, int64(len(c.data)))
|
||||
if err != nil {
|
||||
t.Fatalf("hashHeadTail: %v", err)
|
||||
}
|
||||
head, tail, content := sig(t, p, int64(len(c.data)))
|
||||
|
||||
n := min(chunk, len(c.data))
|
||||
if want := hexSum(c.data[:n]); head != want {
|
||||
t.Errorf("head = %s, want %s", head, want)
|
||||
}
|
||||
|
||||
if want := hexSum(c.data[len(c.data)-n:]); tail != want {
|
||||
t.Errorf("tail = %s, want %s", tail, want)
|
||||
whole := hexSum(c.data)
|
||||
if head != whole || tail != whole || content != whole {
|
||||
t.Errorf("head=%s tail=%s content=%s, want all whole-file %s",
|
||||
head, tail, content, whole)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestHashHeadTailErrors(t *testing.T) {
|
||||
// TestHashSignatureEnds exercises the head and tail rungs, which apply
|
||||
// only to files at least headTailMin. Sparse files keep the fixtures
|
||||
// cheap: a difference in the first window changes only head, a
|
||||
// difference in the last window changes only tail, and a difference
|
||||
// between the windows changes neither end hash but does change the
|
||||
// whole-file content rung (the file is below wholeFileMax).
|
||||
// hashSignature leaves the content hash of a file this size to the
|
||||
// content phase, so that rung is checked through a scan.
|
||||
func TestHashSignatureEnds(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
_, _, err := hashHeadTail(filepath.Join(dir, "missing"), 1)
|
||||
// Between headTailMin and wholeFileMax: the end-window gate is active
|
||||
// and the content rung is a whole-file hash.
|
||||
const size = int64(headTailMin + 2*1024*1024)
|
||||
|
||||
base := sparseFile(t, dir, "ends-base", size)
|
||||
headDiff := sparseFile(t, dir, "ends-head", size)
|
||||
tailDiff := sparseFile(t, dir, "ends-tail", size)
|
||||
midDiff := sparseFile(t, dir, "ends-mid", size)
|
||||
|
||||
pokeAt(t, headDiff, 0, []byte{1})
|
||||
pokeAt(t, tailDiff, size-1, []byte{1})
|
||||
pokeAt(t, midDiff, size/2, []byte{1})
|
||||
|
||||
bHead, bTail, bContent := sig(t, base, size)
|
||||
if bContent != "" {
|
||||
t.Errorf("content = %q, want none from the hash phase", bContent)
|
||||
}
|
||||
|
||||
h, tl, _ := sig(t, headDiff, size)
|
||||
if h == bHead {
|
||||
t.Error("a byte in the first window did not change head")
|
||||
}
|
||||
|
||||
if tl != bTail {
|
||||
t.Error("a byte in the first window changed tail")
|
||||
}
|
||||
|
||||
h, tl, _ = sig(t, tailDiff, size)
|
||||
if tl == bTail {
|
||||
t.Error("a byte in the last window did not change tail")
|
||||
}
|
||||
|
||||
if h != bHead {
|
||||
t.Error("a byte in the last window changed head")
|
||||
}
|
||||
|
||||
h, tl, _ = sig(t, midDiff, size)
|
||||
if h != bHead || tl != bTail {
|
||||
t.Error("a byte between the windows changed an end hash")
|
||||
}
|
||||
|
||||
// base and midDiff match on size, head, and tail, so the scan reads
|
||||
// both for their content hashes.
|
||||
c := scanContents(t, dir, base, midDiff)
|
||||
if c[midDiff] == c[base] {
|
||||
t.Error("whole-file content rung ignored a byte between the windows")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHashSignatureErrors(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
// A missing file: an error, and every hash left empty.
|
||||
head, tail, content, err := hashSignature(filepath.Join(dir, "missing"), 1)
|
||||
if err == nil {
|
||||
t.Error("no error for a missing file")
|
||||
}
|
||||
|
||||
if head != "" || tail != "" || content != "" {
|
||||
t.Errorf("missing file returned hashes: %q %q %q", head, tail, content)
|
||||
}
|
||||
|
||||
// A zero-length file has constant hashes and is never opened: even
|
||||
// a missing path succeeds.
|
||||
head, tail, err := hashHeadTail(filepath.Join(dir, "missing"), 0)
|
||||
if err != nil || head != emptyHash || tail != emptyHash {
|
||||
t.Errorf("empty: head=%q tail=%q err=%v, want constant hashes",
|
||||
head, tail, err)
|
||||
head, tail, content, err = hashSignature(filepath.Join(dir, "missing"), 0)
|
||||
if err != nil ||
|
||||
head != emptyHash || tail != emptyHash || content != emptyHash {
|
||||
t.Errorf("empty: head=%q tail=%q content=%q err=%v, "+
|
||||
"want constant hashes", head, tail, content, err)
|
||||
}
|
||||
|
||||
// A file that shrank between the stat and hash passes: reading at
|
||||
// the stat-reported size must fail rather than emit wrong hashes.
|
||||
p := writeFile(t, dir, "shrunk", []byte("tiny"))
|
||||
|
||||
_, _, err = hashHeadTail(p, int64(2*chunk))
|
||||
head, tail, content, err = hashSignature(p, int64(2*headTailWindow))
|
||||
if err == nil {
|
||||
t.Error("no error when the stat size exceeds the file size")
|
||||
}
|
||||
|
||||
if head != "" || tail != "" || content != "" {
|
||||
t.Errorf("shrunk file returned hashes: %q %q %q", head, tail, content)
|
||||
}
|
||||
}
|
||||
|
||||
// sparseFile creates a file that is logically size bytes long without
|
||||
// allocating blocks for the hole, so multi-gigabyte cases stay cheap.
|
||||
func sparseFile(t *testing.T, dir, name string, size int64) string {
|
||||
t.Helper()
|
||||
|
||||
p := filepath.Join(dir, name)
|
||||
|
||||
f, err := os.Create(p) //nolint:gosec // test-controlled path
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = f.Truncate(size)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = f.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// pokeAt writes data into an existing file at off, leaving the rest of
|
||||
// the file (a sparse hole) untouched.
|
||||
func pokeAt(t *testing.T, path string, off int64, data []byte) {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.OpenFile(path, os.O_WRONLY, 0o600) //nolint:gosec // test path
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = f.WriteAt(data, off)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = f.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// scanContents scans dir into a fresh database and returns the content
|
||||
// hash recorded for each file, by path, failing the test if one of want
|
||||
// has none. A file of headTailMin or more gets a content hash only when
|
||||
// it is scanned with a file of the same size, head, and tail.
|
||||
func scanContents(t *testing.T, dir string,
|
||||
want ...string,
|
||||
) map[string]string {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
syncTree(t, db, dir)
|
||||
|
||||
contents := make(map[string]string)
|
||||
for _, r := range dbRecords(t, db) {
|
||||
contents[r.path] = r.content
|
||||
}
|
||||
|
||||
for _, p := range want {
|
||||
if contents[p] == "" {
|
||||
t.Fatalf("%s: no content hash", p)
|
||||
}
|
||||
}
|
||||
|
||||
return contents
|
||||
}
|
||||
|
||||
// TestContentRungBoundary checks the 50 MiB boundary between the two
|
||||
// content rungs: just below it the whole file is hashed and any byte
|
||||
// difference shows; at the boundary only the gigabyte-spaced samples are
|
||||
// hashed, so a difference outside a sample window is invisible. The
|
||||
// files of each pair match on size, head, and tail, so the scan reads
|
||||
// both for their content hashes.
|
||||
func TestContentRungBoundary(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
// A byte that lands outside the single [0, sampleWindow) sample a
|
||||
// sub-gigabyte file has, but well inside the file.
|
||||
const off = 10 * 1024 * 1024
|
||||
|
||||
// Just under the boundary: the whole-file rung sees the poked byte.
|
||||
under := int64(wholeFileMax - 1)
|
||||
underBase := sparseFile(t, dir, "under-base", under)
|
||||
underPoked := sparseFile(t, dir, "under-poked", under)
|
||||
|
||||
pokeAt(t, underPoked, off, []byte{1})
|
||||
|
||||
// At the boundary: only [0, sampleWindow) is sampled, so the poked
|
||||
// byte at off is invisible and the two content hashes match.
|
||||
at := int64(wholeFileMax)
|
||||
atBase := sparseFile(t, dir, "at-base", at)
|
||||
atPoked := sparseFile(t, dir, "at-poked", at)
|
||||
|
||||
pokeAt(t, atPoked, off, []byte{1})
|
||||
|
||||
c := scanContents(t, dir, underBase, underPoked, atBase, atPoked)
|
||||
|
||||
if c[underBase] == c[underPoked] {
|
||||
t.Error("whole-file rung ignored a byte difference below wholeFileMax")
|
||||
}
|
||||
|
||||
if c[atBase] != c[atPoked] {
|
||||
t.Error("sampled rung saw a byte outside every sample window")
|
||||
}
|
||||
}
|
||||
|
||||
// TestContentRungMultiGigabyte exercises the sampled rung across several
|
||||
// gigabytes using sparse files: a difference inside the third sample
|
||||
// window (at offset 2*sampleStride) changes the hash, while a difference
|
||||
// in the gap after it does not. The three files match on size, head,
|
||||
// and tail, so the scan reads each for its content hash.
|
||||
func TestContentRungMultiGigabyte(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
// Three sample windows (offsets 0, 1 GiB, 2 GiB) plus a trailing gap
|
||||
// that no sample covers.
|
||||
size := int64(2*sampleStride + 2*sampleWindow)
|
||||
thirdSample := int64(2 * sampleStride)
|
||||
gap := thirdSample + int64(sampleWindow)
|
||||
|
||||
base := sparseFile(t, dir, "g-base", size)
|
||||
inSample := sparseFile(t, dir, "g-insample", size)
|
||||
inGap := sparseFile(t, dir, "g-ingap", size)
|
||||
|
||||
pokeAt(t, inSample, thirdSample, []byte{1})
|
||||
pokeAt(t, inGap, gap, []byte{1})
|
||||
|
||||
c := scanContents(t, dir, base, inSample, inGap)
|
||||
|
||||
if c[inSample] == c[base] {
|
||||
t.Error("sample at 2 GiB was not read: difference there was invisible")
|
||||
}
|
||||
|
||||
if c[inGap] != c[base] {
|
||||
t.Error("a byte in an unsampled gap changed the content hash")
|
||||
}
|
||||
}
|
||||
|
||||
// sparseFileWithoutMatch writes name in dir as a sparse file of size
|
||||
// bytes, next to another file of that size whose first byte differs. A
|
||||
// scan then reads the file's head and tail, since its size is shared,
|
||||
// but finds no file matching them, so it gets no content hash.
|
||||
func sparseFileWithoutMatch(t *testing.T, dir, name string,
|
||||
size int64,
|
||||
) string {
|
||||
t.Helper()
|
||||
|
||||
p := sparseFile(t, dir, name, size)
|
||||
other := sparseFile(t, dir, name+"-other-head", size)
|
||||
|
||||
pokeAt(t, other, 0, []byte{1})
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// TestScanContentGate checks that a file of headTailMin or more is read
|
||||
// for its content hash only when its size, head, and tail match another
|
||||
// file's: a same-size pair whose heads differ and one whose tails differ
|
||||
// get no content hash and are not reported, while an identical pair is
|
||||
// read and reported.
|
||||
func TestScanContentGate(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
|
||||
// Three sizes, so that no pair meets another.
|
||||
headA := sparseFile(t, dir, "head-a", headTailMin)
|
||||
headB := sparseFile(t, dir, "head-b", headTailMin)
|
||||
tailA := sparseFile(t, dir, "tail-a", headTailMin+1)
|
||||
tailB := sparseFile(t, dir, "tail-b", headTailMin+1)
|
||||
same := []string{
|
||||
sparseFile(t, dir, "same-a", headTailMin+2),
|
||||
sparseFile(t, dir, "same-b", headTailMin+2),
|
||||
}
|
||||
|
||||
pokeAt(t, headB, 0, []byte{1})
|
||||
pokeAt(t, tailB, headTailMin, []byte{1}) // its last byte
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, p := range []string{headA, headB, tailA, tailB} {
|
||||
r := recordByPath(t, recs, p)
|
||||
if r.head == "" || r.tail == "" || r.content != "" {
|
||||
t.Errorf("%s: head = %q tail = %q content = %q, "+
|
||||
"want head and tail only", p, r.head, r.tail, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||
groups, same)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentAcrossOperands checks that a stored file gets its
|
||||
// content hash when a later scan of a separate operand brings its
|
||||
// match: tree A's file has a head and tail but no content hash until
|
||||
// tree B, holding an identical file, is scanned.
|
||||
func TestScanContentAcrossOperands(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||
|
||||
syncTree(t, db, filepath.Dir(a))
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), a); r.head == "" || r.content != "" {
|
||||
t.Fatalf("after scanning A: %+v, want head and tail only", r)
|
||||
}
|
||||
|
||||
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||
|
||||
syncTree(t, db, filepath.Dir(b))
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
if r := recordByPath(t, recs, a); r.content == "" {
|
||||
t.Fatalf("after scanning B: %+v, want A's file content-hashed", r)
|
||||
}
|
||||
|
||||
want := []string{a, b}
|
||||
slices.Sort(want)
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentWithinOperand checks that a rescan adding a match next
|
||||
// to an unchanged stored file gives the stored file its content hash,
|
||||
// though the hash phase leaves it alone as unchanged.
|
||||
func TestScanContentWithinOperand(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
stored := sparseFileWithoutMatch(t, dir, "d1", headTailMin)
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
added := sparseFile(t, dir, "d2", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st != (scanStats{added: 1, unchanged: 2}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
||||
}
|
||||
|
||||
want := []string{stored, added}
|
||||
|
||||
groups := collectDupeGroups(dbRecords(t, db))
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentStalePartners checks that a stored file outside the
|
||||
// operand that has vanished, or changed, since it was recorded is not
|
||||
// read, and that its match inside the operand is not read either: the
|
||||
// match has no other partner left, so neither gets a content hash and
|
||||
// no duplicate is reported.
|
||||
func TestScanContentStalePartners(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
dirA := t.TempDir()
|
||||
gone := sparseFileWithoutMatch(t, dirA, "gone", headTailMin)
|
||||
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin+1)
|
||||
|
||||
syncTree(t, db, dirA)
|
||||
|
||||
before := dbRecords(t, db)
|
||||
|
||||
err := os.Remove(gone)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
future := time.Now().Add(time.Hour)
|
||||
|
||||
err = os.Chtimes(changed, future, future)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
dirB := t.TempDir()
|
||||
sparseFile(t, dirB, "gone-copy", headTailMin)
|
||||
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
||||
|
||||
st := syncTree(t, db, dirB)
|
||||
if st != (scanStats{added: 2}) {
|
||||
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, r := range recs {
|
||||
if r.content != "" {
|
||||
t.Errorf("%s: content = %q, want none: its only match is stale",
|
||||
r.path, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
for _, old := range before {
|
||||
if r := recordByPath(t, recs, old.path); r != old {
|
||||
t.Errorf("record = %+v, want it left as %+v", r, old)
|
||||
}
|
||||
}
|
||||
|
||||
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||
t.Errorf("groups = %+v, want none", groups)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentHashedStalePartners checks that stored matches outside
|
||||
// the operand that already have a content hash are checked like any
|
||||
// other: once one has vanished and the other has changed, a copy of
|
||||
// them scanned in another tree has no match left, so it is not read and
|
||||
// is not reported as their duplicate.
|
||||
func TestScanContentHashedStalePartners(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
dirA := t.TempDir()
|
||||
stored := []string{
|
||||
sparseFile(t, dirA, "changed", headTailMin),
|
||||
sparseFile(t, dirA, "gone", headTailMin),
|
||||
}
|
||||
|
||||
// The two stored files match, so this scan gives both a content
|
||||
// hash.
|
||||
syncTree(t, db, dirA)
|
||||
|
||||
err := os.Remove(stored[1])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
future := time.Now().Add(time.Hour)
|
||||
|
||||
err = os.Chtimes(stored[0], future, future)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
||||
|
||||
st := syncTree(t, db, filepath.Dir(b))
|
||||
if st != (scanStats{added: 1}) {
|
||||
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
if r := recordByPath(t, recs, b); r.content != "" {
|
||||
t.Errorf("copy: content = %q, want none: its only matches are stale",
|
||||
r.content)
|
||||
}
|
||||
|
||||
// The stored records lie outside the operand and are left as they
|
||||
// are, so they still group with each other, but not with the copy.
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentReadFailure checks that a failed content read is
|
||||
// counted as skipped and leaves the record without a content hash, and
|
||||
// that a later scan tries the read again.
|
||||
func TestScanContentReadFailure(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||
|
||||
syncTree(t, db, filepath.Dir(a))
|
||||
|
||||
// lstat still works on the unreadable file, so it passes the check
|
||||
// and fails only when it is read.
|
||||
err := os.Chmod(a, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
dirB := t.TempDir()
|
||||
b := sparseFile(t, dirB, "b", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dirB)
|
||||
if st != (scanStats{added: 1, skipped: 1}) {
|
||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||
}
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), a); r.content != "" {
|
||||
t.Fatalf("unreadable file: %+v, want no content hash", r)
|
||||
}
|
||||
|
||||
err = os.Chmod(a, 0o600)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st = syncTree(t, db, dirB)
|
||||
if st != (scanStats{unchanged: 1}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
||||
}
|
||||
|
||||
want := []string{a, b}
|
||||
slices.Sort(want)
|
||||
|
||||
groups := collectDupeGroups(dbRecords(t, db))
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||
groups, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentCheckError checks that a stored file the content phase
|
||||
// cannot lstat, for a reason other than its being gone, is counted as
|
||||
// skipped and does not count as a match.
|
||||
func TestScanContentCheckError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
sub := filepath.Join(t.TempDir(), "sub")
|
||||
|
||||
err := os.Mkdir(sub, 0o700)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
sparseFileWithoutMatch(t, sub, "a", headTailMin)
|
||||
syncTree(t, db, sub)
|
||||
|
||||
// Without search permission on its directory, the stored file's
|
||||
// lstat fails with permission denied.
|
||||
err = os.Chmod(sub, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() {
|
||||
//nolint:gosec // removing the directory needs its search bit back
|
||||
_ = os.Chmod(sub, 0o700)
|
||||
})
|
||||
|
||||
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||
|
||||
st := syncTree(t, db, filepath.Dir(b))
|
||||
if st != (scanStats{added: 1, skipped: 1}) {
|
||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||
}
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), b); r.content != "" {
|
||||
t.Errorf("b: content = %q, want none: its only match could not be "+
|
||||
"checked", r.content)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentHardlinks checks that the content phase stores the
|
||||
// content hash of a hard-linked file on every one of its links.
|
||||
func TestScanContentHardlinks(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
a := sparseFile(t, dir, "a", headTailMin)
|
||||
b := filepath.Join(dir, "b")
|
||||
|
||||
err := os.Link(a, b)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
c := sparseFile(t, dir, "copy", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st != (scanStats{added: 3}) {
|
||||
t.Fatalf("stats = %+v, want 3 added", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
|
||||
want := recordByPath(t, recs, c).content
|
||||
if want == "" {
|
||||
t.Fatal("the copy has no content hash")
|
||||
}
|
||||
|
||||
for _, p := range []string{a, b} {
|
||||
if got := recordByPath(t, recs, p).content; got != want {
|
||||
t.Errorf("%s: content = %q, want %q", p, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// collectWalk runs a walk over roots and returns the emitted records
|
||||
@@ -706,9 +1289,9 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, r := range recs {
|
||||
if r.head != "" || r.tail != "" {
|
||||
t.Errorf("%s: head = %q tail = %q, want unhashed",
|
||||
r.path, r.head, r.tail)
|
||||
if r.head != "" || r.tail != "" || r.content != "" {
|
||||
t.Errorf("%s: head = %q tail = %q content = %q, want unhashed",
|
||||
r.path, r.head, r.tail, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -740,23 +1323,36 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
||||
func TestTreesUnhashedNeverEqual(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Two trees identical except for unhashed same-name, same-size
|
||||
// files (possible when the trees were scanned separately) must not
|
||||
// compare equal: unhashed content is unknown.
|
||||
shared := pattern(1, 100)
|
||||
recs := []scanRec{
|
||||
{path: "/x/t1/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
||||
{path: "/x/t2/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
||||
{path: "/x/t1/u", size: 50},
|
||||
{path: "/x/t2/u", size: 50},
|
||||
// Two trees identical except for same-name, same-size files without
|
||||
// a content hash must not compare equal: their content is unknown.
|
||||
// That holds for unhashed files (possible when the trees were
|
||||
// scanned separately) and for files of headTailMin or more that
|
||||
// have only a head and tail.
|
||||
sum := hexSum(pattern(1, 100))
|
||||
shared := []scanRec{
|
||||
{path: "/x/t1/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||
{path: "/x/t2/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
cases := map[string][]scanRec{
|
||||
"unhashed": {
|
||||
{path: "/x/t1/u", size: 50},
|
||||
{path: "/x/t2/u", size: 50},
|
||||
},
|
||||
"head and tail only": {
|
||||
{path: "/x/t1/u", size: headTailMin, head: "h", tail: "t"},
|
||||
{path: "/x/t2/u", size: headTailMin, head: "h", tail: "t"},
|
||||
},
|
||||
}
|
||||
|
||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||
t.Fatalf("tree groups = %d, want 0 (unhashed files differ)",
|
||||
len(tg))
|
||||
for name, unknown := range cases {
|
||||
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
||||
super.compute()
|
||||
|
||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||
name, len(tg))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -71,9 +71,9 @@ main() {
|
||||
"the two pins disagree:" >&2
|
||||
echo "verify-lint-image-pin: $LINT_DOCKERFILE: $lint_ref" >&2
|
||||
echo "verify-lint-image-pin: $MAIN_DOCKERFILE: $main_ref" >&2
|
||||
echo "verify-lint-image-pin: bump both FROM lines together, tag and" \
|
||||
"digest, so script/lint and the Dockerfile lint stage keep" \
|
||||
"running the same linter" >&2
|
||||
echo "verify-lint-image-pin: bump both FROM lines together so" \
|
||||
"script/lint and the Dockerfile lint stage keep running the" \
|
||||
"same linter" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
|
||||
@@ -13,9 +13,10 @@ import (
|
||||
|
||||
// fileSig is a file's duplicate signature; mtime is excluded.
|
||||
type fileSig struct {
|
||||
size int64
|
||||
head string
|
||||
tail string
|
||||
size int64
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
}
|
||||
|
||||
// treeNode is one directory reconstructed from the scan stream.
|
||||
@@ -122,14 +123,16 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||
node.files = make(map[string]fileSig)
|
||||
}
|
||||
|
||||
sig := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
sig := fileSig{
|
||||
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||
}
|
||||
|
||||
// An unhashed record (its size was unique when last scanned)
|
||||
// has unknown content: give it a signature no other file can
|
||||
// share, so trees containing it never compare equal. Real
|
||||
// heads are hex, so the NUL-prefixed form cannot collide.
|
||||
if sig.head == "" {
|
||||
sig.head = "unhashed\x00" + r.path
|
||||
// A record without a content hash has unknown content (README
|
||||
// "Database"): give it a signature no other file can share, so
|
||||
// trees containing it never compare equal. Real hashes are
|
||||
// hex, so the NUL-prefixed form cannot collide.
|
||||
if sig.content == "" {
|
||||
sig.content = "unhashed\x00" + r.path
|
||||
}
|
||||
|
||||
node.files[comps[len(comps)-1]] = sig
|
||||
@@ -188,7 +191,7 @@ func (n *treeNode) compute() {
|
||||
for name, sig := range n.files {
|
||||
entries = append(entries,
|
||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||
"\x00"+sig.head+"\x00"+sig.tail)
|
||||
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
||||
n.fileCount++
|
||||
n.totalSize += sig.size
|
||||
}
|
||||
|
||||
+20
-17
@@ -7,22 +7,25 @@ import (
|
||||
|
||||
// Signature hashes shared by the smoke-test records.
|
||||
const (
|
||||
f1Head = "f1h"
|
||||
f1Tail = "f1t"
|
||||
f2Head = "f2h"
|
||||
f2Tail = "f2t"
|
||||
f1Head = "f1h"
|
||||
f1Tail = "f1t"
|
||||
f1Content = "f1c"
|
||||
f2Head = "f2h"
|
||||
f2Tail = "f2t"
|
||||
f2Content = "f2c"
|
||||
)
|
||||
|
||||
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
||||
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
||||
func smokeTreeRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
|
||||
path: "/d/t3/sub/f2renamed"},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -114,8 +117,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
||||
const sharedTail = "same"
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"},
|
||||
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
@@ -181,8 +184,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
||||
// Identical sibling dirs share a parent, so their group cannot be
|
||||
// implied by a parent group and must be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x2/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
@@ -203,9 +206,9 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
||||
// extra file, so the parents' digests differ and the x group must
|
||||
// be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", path: "/q/b/x/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
|
||||
Reference in New Issue
Block a user