@@ -2,7 +2,6 @@
|
|||||||
.claude
|
.claude
|
||||||
.DS_Store
|
.DS_Store
|
||||||
sfdupes
|
sfdupes
|
||||||
files.dat
|
|
||||||
*.log
|
*.log
|
||||||
*.out
|
*.out
|
||||||
*.test
|
*.test
|
||||||
|
|||||||
@@ -27,7 +27,6 @@ node_modules/
|
|||||||
*.log
|
*.log
|
||||||
|
|
||||||
# Local scan data
|
# Local scan data
|
||||||
files.dat
|
|
||||||
*.sqlite
|
*.sqlite
|
||||||
*.sqlite-shm
|
*.sqlite-shm
|
||||||
*.sqlite-wal
|
*.sqlite-wal
|
||||||
|
|||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
# Lint stage — fast feedback on formatting and lint issues
|
# Lint stage — fast feedback on formatting and lint issues
|
||||||
# golangci/golangci-lint:v2.12.2 (Debian-based), 2026-08-07
|
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||||
FROM golangci/golangci-lint:v2.12.2@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
|
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
COPY go.mod go.sum ./
|
COPY go.mod go.sum ./
|
||||||
RUN go mod download
|
RUN go mod download
|
||||||
|
|||||||
+2
-2
@@ -9,8 +9,8 @@
|
|||||||
# stage of the main Dockerfile because script/lint must not depend on
|
# stage of the main Dockerfile because script/lint must not depend on
|
||||||
# the rest of that build; the two FROM lines are kept identical by
|
# the rest of that build; the two FROM lines are kept identical by
|
||||||
# script/verify-lint-image-pin, run as a gate below.
|
# script/verify-lint-image-pin, run as a gate below.
|
||||||
# golangci/golangci-lint:v2.12.2 (Debian-based), 2026-08-07
|
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||||
FROM golangci/golangci-lint:v2.12.2@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
|
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
|
||||||
|
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
|
|
||||||
|
|||||||
@@ -46,4 +46,4 @@ hooks:
|
|||||||
@script/install-precommit
|
@script/install-precommit
|
||||||
|
|
||||||
clean:
|
clean:
|
||||||
rm -f $(BINARY) files.dat
|
rm -f $(BINARY)
|
||||||
|
|||||||
@@ -5,14 +5,20 @@
|
|||||||
`sfdupes` is an MIT-licensed Go CLI tool by
|
`sfdupes` is an MIT-licensed Go CLI tool by
|
||||||
[@sneak](https://sneak.berlin) that quickly identifies *candidate*
|
[@sneak](https://sneak.berlin) that quickly identifies *candidate*
|
||||||
duplicate files — and, ultimately, entire duplicate directory trees —
|
duplicate files — and, ultimately, entire duplicate directory trees —
|
||||||
across very large filesystems without reading full file contents. Files
|
across very large filesystems without reading every byte of every file.
|
||||||
are considered duplicates when they have identical size, identical
|
Files are considered duplicates when their sizes are equal and they
|
||||||
SHA-256 of their first 1024 bytes, and identical SHA-256 of their last
|
agree on a short ladder of hashes. A file under 10 MiB is hashed in full
|
||||||
1024 bytes. This is a strong candidate signal, not proof of identical
|
and compared directly. A larger file is gated first on the SHA-256 of
|
||||||
content (the middle of the file is never read); the intended use is
|
its first 64 KiB and of its last 64 KiB, and only when its size and
|
||||||
|
both of those match another file's is it read for a content hash to
|
||||||
|
compare — the SHA-256 of the whole file when it is under 50 MiB, or of
|
||||||
|
gigabyte-spaced 1 MiB samples when it is 50 MiB or larger. Below 50 MiB
|
||||||
|
the content hash is proof of identical content; at or above 50 MiB it is
|
||||||
|
a strong candidate signal rather than proof, because the gaps between
|
||||||
|
samples are never read. The intended use is
|
||||||
finding duplicate downloads and duplicated directory trees on
|
finding duplicate downloads and duplicated directory trees on
|
||||||
multi-terabyte ZFS servers where reading every byte is prohibitively
|
multi-terabyte ZFS servers where reading every byte of every file is
|
||||||
expensive. `scan` maintains a persistent SQLite database of file
|
prohibitively expensive. `scan` maintains a persistent SQLite database of file
|
||||||
signatures that survives between runs, so it can be run from cron and
|
signatures that survives between runs, so it can be run from cron and
|
||||||
the reports can be generated at any time from the most recent scan.
|
the reports can be generated at any time from the most recent scan.
|
||||||
|
|
||||||
@@ -29,9 +35,11 @@ export SFDUPES_DATABASE="$HOME/.local/share/sfdupes/db.sqlite"
|
|||||||
```
|
```
|
||||||
|
|
||||||
`scan` walks one or more filesystem trees and maintains one database
|
`scan` walks one or more filesystem trees and maintains one database
|
||||||
record per regular file (path, size, mtime, head hash, tail hash). The
|
record per regular file (path, size, mtime, head hash, tail hash,
|
||||||
|
content hash). The
|
||||||
database persists between runs; a rescan only hashes files that are new
|
database persists between runs; a rescan only hashes files that are new
|
||||||
or changed, and removes records for files that no longer exist.
|
or changed, or that may have gained a duplicate since the last scan,
|
||||||
|
and removes records for files that no longer exist.
|
||||||
`report` reads the database and prints the file-level duplicates
|
`report` reads the database and prints the file-level duplicates
|
||||||
report. `trees` reads the same database and prints the duplicate-tree
|
report. `trees` reads the same database and prints the duplicate-tree
|
||||||
report. A missing/invalid subcommand — or a `scan` invocation with no
|
report. A missing/invalid subcommand — or a `scan` invocation with no
|
||||||
@@ -47,13 +55,19 @@ completed scan.
|
|||||||
|
|
||||||
Duplicate finders that hash entire files do not scale to the target
|
Duplicate finders that hash entire files do not scale to the target
|
||||||
environment: ~10 million files and ~150 TB on possibly slow or busy
|
environment: ~10 million files and ~150 TB on possibly slow or busy
|
||||||
disks (a ZFS pool under resilver). Reading at most 2 KiB per file — and
|
disks (a ZFS pool under resilver). sfdupes spends disk I/O only on files
|
||||||
only from files whose size at least one other file shares, since a
|
whose size at least one other file shares, since a size-unique file
|
||||||
size-unique file cannot be a duplicate — makes a full-filesystem sweep
|
cannot be a duplicate. Of those, a file under 10 MiB is read in full; a
|
||||||
tractable, and the signatures are kept in a persistent database, so
|
larger one has its cheap end windows read first, and is read for a
|
||||||
the expensive filesystem pass is incremental: a rescan re-hashes only
|
content hash only when its size and both end windows match another
|
||||||
files whose recorded mtime or size changed, and all analysis happens
|
file's — the whole file below 50 MiB, but only gigabyte-spaced samples
|
||||||
offline from the database alone. The end goal is
|
at or above 50 MiB, so the largest files are never read in full. This
|
||||||
|
keeps a full-filesystem sweep tractable, and the signatures are kept in
|
||||||
|
a persistent database, so the expensive filesystem pass is incremental:
|
||||||
|
a rescan re-hashes only files whose recorded mtime or size changed, plus
|
||||||
|
— for its content hash — a file of 10 MiB or more whose size and end
|
||||||
|
windows have come to match another file's. All analysis happens offline
|
||||||
|
from the database alone. The end goal is
|
||||||
not individual files but whole duplicated trees — duplicate
|
not individual files but whole duplicated trees — duplicate
|
||||||
extractions, duplicate downloads, copied project trees — which an
|
extractions, duplicate downloads, copied project trees — which an
|
||||||
operator can consider removing as a unit.
|
operator can consider removing as a unit.
|
||||||
@@ -68,17 +82,24 @@ Goals, in order:
|
|||||||
downloads, copied project trees), so the operator can consider
|
downloads, copied project trees), so the operator can consider
|
||||||
removing an entire subtree at once. File-level duplicate detection is
|
removing an entire subtree at once. File-level duplicate detection is
|
||||||
the foundation; tree-level detection is built on top of it.
|
the foundation; tree-level detection is built on top of it.
|
||||||
2. **Never read full file contents.** At most 2 KiB is read per file
|
2. **Spend I/O in proportion to duplicate likelihood.** Only files
|
||||||
(first and last 1024 bytes), and only files whose size at least
|
whose size at least one other file shares are read at all — a
|
||||||
one other file shares are read at all — a size-unique file cannot
|
size-unique file cannot be a duplicate. Those are compared by the
|
||||||
be a duplicate. Scale target: tens of millions of files, ~150 TB
|
ladder in "Duplicate detection" below: a file under 10 MiB is hashed
|
||||||
filesystem, possibly slow or busy disks (ZFS pool under resilver).
|
in full, while a larger file is gated on cheap 64 KiB end windows
|
||||||
Holding one small record (path, size, mtime) per file in memory
|
first, and gets a content hash only when its size and both end
|
||||||
during a scan is acceptable; holding every file's hashes is not
|
windows match another file's. That hash reads the whole file below
|
||||||
(they stay in the database).
|
50 MiB but only gigabyte-spaced 1 MiB samples at or above it, so the
|
||||||
|
very largest files are still never read in full. Scale target: tens of
|
||||||
|
millions of files, ~150 TB filesystem, possibly slow or busy disks
|
||||||
|
(ZFS pool under resilver). Holding one small record (path, size,
|
||||||
|
mtime) per file in memory during a scan is acceptable; holding
|
||||||
|
every file's hashes is not (they stay in the database).
|
||||||
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
||||||
scan maintains a persistent database; an unchanged file is never
|
scan maintains a persistent database; an unchanged file is never
|
||||||
read again on a rescan. All analysis (`report`, `trees`) works from
|
read again on a rescan, except to compute its content hash once a
|
||||||
|
file of 10 MiB or more comes to match another on size and both end
|
||||||
|
windows. All analysis (`report`, `trees`) works from
|
||||||
the database alone and must never touch the scanned filesystem
|
the database alone and must never touch the scanned filesystem
|
||||||
again. `scan` is designed to be cronned; the reports run at any
|
again. `scan` is designed to be cronned; the reports run at any
|
||||||
time against the last completed scan.
|
time against the last completed scan.
|
||||||
@@ -146,21 +167,73 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE TABLE files (
|
CREATE TABLE files (
|
||||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||||
size INTEGER NOT NULL, -- bytes, from lstat
|
size INTEGER NOT NULL, -- bytes, from lstat
|
||||||
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
||||||
head TEXT NOT NULL, -- lowercase-hex SHA-256, first 1 KiB
|
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
||||||
tail TEXT NOT NULL -- lowercase-hex SHA-256, last 1 KiB
|
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||||
|
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||||
) WITHOUT ROWID;
|
) WITHOUT ROWID;
|
||||||
```
|
```
|
||||||
|
|
||||||
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
||||||
guaranteed UTF-8. `mtime` is used only for change detection; it is
|
guaranteed UTF-8. `mtime` is used only for change detection; it is
|
||||||
not part of the duplicate key. `head` and `tail` are empty strings
|
not part of the duplicate key. For a file under 10 MiB `head`, `tail`,
|
||||||
when the file has never been hashed because its size was unique as
|
and `content` all hold the whole-file hash (that range is hashed in
|
||||||
of the last scan that covered it; such records still define the
|
full, with no end windows); for a larger file `head` and `tail` hold
|
||||||
file for tree reconstruction but never participate in duplicate
|
the first- and last-64 KiB hashes and `content` the whole-file or
|
||||||
groups.
|
sampled hash. All three are empty strings when the file has never
|
||||||
|
been hashed because its size was unique as of the last scan that
|
||||||
|
covered it. For a file of 10 MiB or more, `content` stays empty
|
||||||
|
until the content phase of a scan (see "`scan` mode" below) has
|
||||||
|
read the file. A record with an empty `content` is never part of a
|
||||||
|
duplicate group, though it still defines the file for tree
|
||||||
|
reconstruction.
|
||||||
|
|
||||||
|
### Duplicate detection
|
||||||
|
|
||||||
|
Two files are duplicates only when they agree on every rung of this
|
||||||
|
ladder; a mismatch at any rung means they are not duplicates. `scan`
|
||||||
|
stores each file's hashes, and `report` and `trees` group files by the
|
||||||
|
whole signature — size, `head`, `tail`, and `content` — so the grouping
|
||||||
|
is exactly this ladder applied across everything scanned into the
|
||||||
|
database, even across separate scans.
|
||||||
|
|
||||||
|
1. **Size.** Files of different sizes are never compared. Only files
|
||||||
|
whose size at least one other file shares are hashed at all.
|
||||||
|
2. **Under 10 MiB: whole file.** A file smaller than 10 MiB is hashed
|
||||||
|
in full and compared directly, with no separate end-window step —
|
||||||
|
small files are cheap to read to the last byte, and doing so makes
|
||||||
|
the comparison exact. `head`, `tail`, and `content` all hold this
|
||||||
|
whole-file SHA-256, so such a file's signature is decided entirely
|
||||||
|
by its size and its content.
|
||||||
|
3. **10 MiB and above: head and tail.** For a larger file, the SHA-256
|
||||||
|
of the first 64 KiB (`head`) and of the last 64 KiB (`tail`) are a
|
||||||
|
cheap gate that eliminates most same-size pairs before any bulk
|
||||||
|
reading: the content hash of the next two rungs is computed only for
|
||||||
|
a file whose size, `head`, and `tail` match another file's, whether
|
||||||
|
that file is scanned in the same run or stored by an earlier scan.
|
||||||
|
A stored file that first gains such a match in a later scan gets its
|
||||||
|
content hash then; until it has one, its `content` is empty and it
|
||||||
|
is not a duplicate. At 10 MiB and above the two windows never
|
||||||
|
overlap.
|
||||||
|
4. **10 MiB and above, content below 50 MiB.** The SHA-256 of the
|
||||||
|
entire file. Agreement here is proof of identical content (barring a
|
||||||
|
SHA-256 collision).
|
||||||
|
5. **10 MiB and above, content 50 MiB and above.** A sampled SHA-256:
|
||||||
|
the 1 MiB window at each gigabyte-aligned offset (0, 1 GiB, 2 GiB, …
|
||||||
|
while inside the file, the final window truncated at end of file) is
|
||||||
|
fed, in order, into one hash. This is **deliberately probabilistic**
|
||||||
|
— the gaps between samples are never read, so two large files that
|
||||||
|
agree on every sample are reported as duplicates without being read
|
||||||
|
in full. It is the price of never reading a 150 GB file end to end.
|
||||||
|
Because size is already part of the signature, only equal-size files
|
||||||
|
reach this rung, so their sample boundaries always align.
|
||||||
|
|
||||||
|
`head`, `tail`, and `content` are one column each. A file below 10 MiB
|
||||||
|
and one at or above it never share a size, and neither do a file below
|
||||||
|
50 MiB and one at or above it, so a stored value is never ambiguous
|
||||||
|
between the whole-file, end-window, and sampled forms.
|
||||||
|
|
||||||
### `scan` mode
|
### `scan` mode
|
||||||
|
|
||||||
@@ -183,10 +256,10 @@ scanned operands:
|
|||||||
|
|
||||||
- Only a file whose size at least one other file shares is ever
|
- Only a file whose size at least one other file shares is ever
|
||||||
read: a size-unique file cannot be a duplicate, so it is recorded
|
read: a size-unique file cannot be a duplicate, so it is recorded
|
||||||
without hashes (`head` and `tail` empty). The size census covers
|
without hashes (`head`, `tail`, and `content` empty). The size
|
||||||
every file walked this scan plus every database record outside
|
census covers every file walked this scan plus every database
|
||||||
the scanned operands, so a possible duplicate of a separately
|
record outside the scanned operands, so a possible duplicate of a
|
||||||
scanned tree is still recognized.
|
separately scanned tree is still recognized.
|
||||||
- A file not yet in the database is inserted: hashed when its size
|
- A file not yet in the database is inserted: hashed when its size
|
||||||
is shared, without hashes otherwise.
|
is shared, without hashes otherwise.
|
||||||
- A file already in the database is **skipped without reading its
|
- A file already in the database is **skipped without reading its
|
||||||
@@ -195,7 +268,10 @@ scanned operands:
|
|||||||
makes a daily rescan cheap. Exception: an unchanged file whose
|
makes a daily rescan cheap. Exception: an unchanged file whose
|
||||||
record lacks hashes is hashed — and its record updated — once its
|
record lacks hashes is hashed — and its record updated — once its
|
||||||
size becomes shared, so hashing deferred by size-uniqueness
|
size becomes shared, so hashing deferred by size-uniqueness
|
||||||
happens as soon as it could matter.
|
happens as soon as it could matter. Likewise, an unchanged file of
|
||||||
|
10 MiB or more whose record has no `content` hash is read for one
|
||||||
|
by the content phase below once its size, `head`, and `tail` match
|
||||||
|
another record's.
|
||||||
- A file whose mtime is newer than recorded, or whose size differs,
|
- A file whose mtime is newer than recorded, or whose size differs,
|
||||||
is processed as if new: re-hashed, or recorded without hashes,
|
is processed as if new: re-hashed, or recorded without hashes,
|
||||||
per the shared-size rule.
|
per the shared-size rule.
|
||||||
@@ -204,12 +280,18 @@ scanned operands:
|
|||||||
removes records for deleted files. It also removes records for
|
removes records for deleted files. It also removes records for
|
||||||
paths that failed to stat or hash this run: the database only ever
|
paths that failed to stat or hash this run: the database only ever
|
||||||
contains signatures verified by the most recent scan that covered
|
contains signatures verified by the most recent scan that covered
|
||||||
them (a subsequent successful scan re-adds such files).
|
them (a subsequent successful scan re-adds such files). A failure
|
||||||
|
in the content phase below removes nothing: the record is left as
|
||||||
|
it is.
|
||||||
- Database records outside the scanned operands are untouched, so
|
- Database records outside the scanned operands are untouched, so
|
||||||
disjoint trees can be scanned on different schedules into the same
|
disjoint trees can be scanned on different schedules into the same
|
||||||
database.
|
database. The one exception is the content phase below: a stored
|
||||||
|
file of 10 MiB or more without a `content` hash is read for one,
|
||||||
|
wherever it lies, once its size, `head`, and `tail` match another
|
||||||
|
record's. If that file is gone or has changed since its record was
|
||||||
|
written, the record is left as it is.
|
||||||
|
|
||||||
`scan` runs **three sequential phases over the whole scan**.
|
`scan` runs **four sequential phases over the whole scan**.
|
||||||
Parallelism lives inside each phase; batched database writes begin
|
Parallelism lives inside each phase; batched database writes begin
|
||||||
during the hash phase:
|
during the hash phase:
|
||||||
|
|
||||||
@@ -229,12 +311,13 @@ during the hash phase:
|
|||||||
decides its fate. Size-unique files are never read: new or
|
decides its fate. Size-unique files are never read: new or
|
||||||
changed ones are recorded without hashes in the update phase,
|
changed ones are recorded without hashes in the update phase,
|
||||||
unchanged unhashed ones simply keep their records. Every file
|
unchanged unhashed ones simply keep their records. Every file
|
||||||
with a shared size is hashed by the worker pool: read the first
|
with a shared size is hashed by the worker pool as described in
|
||||||
`min(1024, size)` bytes and the last `min(1024, size)` bytes
|
"Duplicate detection" above: a file under 10 MiB in full, which
|
||||||
(one read when `size <= 1024`, since the two windows coincide)
|
gives its `head`, `tail`, and `content` alike, and a larger file
|
||||||
and compute the SHA-256 of each. Zero-length files have constant
|
only in its end windows, which give its `head` and `tail`; its
|
||||||
hashes and are never opened. Files are hashed in **inode order**
|
content hash is left to the content phase. Zero-length files have
|
||||||
(minimizing seeks on spinning disks), and paths that are hard
|
constant hashes and are never opened. Files are hashed in **inode
|
||||||
|
order** (minimizing seeks on spinning disks), and paths that are hard
|
||||||
links to the same inode are **read once**, all sharing the one
|
links to the same inode are **read once**, all sharing the one
|
||||||
result — a hard-link backup farm costs one read per inode, not
|
result — a hard-link backup farm costs one read per inode, not
|
||||||
per path. The phase total counts actual reads, so progress and
|
per path. The phase total counts actual reads, so progress and
|
||||||
@@ -246,6 +329,29 @@ during the hash phase:
|
|||||||
records for size-unique new and changed files, and the deletions
|
records for size-unique new and changed files, and the deletions
|
||||||
for records the scan did not verify (vanished files, plus paths
|
for records the scan did not verify (vanished files, plus paths
|
||||||
that failed to stat or hash).
|
that failed to stat or hash).
|
||||||
|
4. **content** — find every record of 10 MiB or more without a
|
||||||
|
`content` hash whose size, `head`, and `tail` equal another
|
||||||
|
record's, anywhere in the database: records from this scan and
|
||||||
|
records stored by earlier scans, inside or outside the scanned
|
||||||
|
operands. SQLite finds them, so only the records to be read are
|
||||||
|
kept in memory, never every file's hashes. Every record sharing
|
||||||
|
their size, `head`, and `tail`, including one that already has a
|
||||||
|
`content` hash, has its file checked with `lstat` first. A file
|
||||||
|
that is gone, is no longer a regular file, or has changed (a
|
||||||
|
different size, or an mtime newer than recorded) keeps its record
|
||||||
|
as it is and does not count as a match for the others. Any other
|
||||||
|
`lstat` error is warned about and counted as skipped, with the same
|
||||||
|
result. If such a record has no `content` hash, it stays out of
|
||||||
|
duplicate groups; if it has one, it is still reported until a scan
|
||||||
|
covering its own tree updates or removes it. The files that pass
|
||||||
|
and have no `content` hash are read only if at least two of those
|
||||||
|
records pass, so a file whose only matches are stale costs no read;
|
||||||
|
a file that already has a `content` hash is never read again.
|
||||||
|
They are read by a worker pool as in the hash phase, in inode order
|
||||||
|
and once per inode, and their content hashes are committed in
|
||||||
|
batches. A failed read is warned about and counted as skipped; its
|
||||||
|
record keeps an empty `content`, so it is not a duplicate, and a
|
||||||
|
later scan tries again.
|
||||||
|
|
||||||
Rules for the walk:
|
Rules for the walk:
|
||||||
|
|
||||||
@@ -263,18 +369,18 @@ Rules for the walk:
|
|||||||
path, and continue. Per-file errors never abort the run; the final
|
path, and continue. Per-file errors never abort the run; the final
|
||||||
summary reports how many were skipped. As specified above, a
|
summary reports how many were skipped. As specified above, a
|
||||||
skipped path that has a database record from an earlier scan loses
|
skipped path that has a database record from an earlier scan loses
|
||||||
that record; an unreadable directory subtree likewise loses its
|
that record, unless it failed only in the content phase; an
|
||||||
records (accepted: the database mirrors what the latest scan could
|
unreadable directory subtree likewise loses its records (accepted:
|
||||||
actually verify).
|
the database mirrors what the latest scan could actually verify).
|
||||||
|
|
||||||
Concurrency: the walk phase (which also stats files) and the hash
|
Concurrency: the walk phase (which also stats files), the hash phase,
|
||||||
phase each use a worker pool of `--workers` workers (default
|
and the content phase each use a worker pool of `--workers` workers
|
||||||
`runtime.NumCPU()`); the walk parallelizes across directories,
|
(default `runtime.NumCPU()`); the walk parallelizes across
|
||||||
hashing across files. Both phases are seek-bound on spinning disks,
|
directories, hashing across files. All three phases are seek-bound on
|
||||||
so raising `--workers` well past the core count can help on pools
|
spinning disks, so raising `--workers` well past the core count can
|
||||||
with many spindles. The main goroutine owns partitioning, database
|
help on pools with many spindles. The main goroutine owns
|
||||||
writes, and progress rendering; progress display must never block
|
partitioning, database writes, and progress rendering; progress
|
||||||
the workers.
|
display must never block the workers.
|
||||||
|
|
||||||
`scan` writes nothing to stdout. The summary line on stderr reports the
|
`scan` writes nothing to stdout. The summary line on stderr reports the
|
||||||
files seen this run broken down by disposition, plus skips:
|
files seen this run broken down by disposition, plus skips:
|
||||||
@@ -299,11 +405,11 @@ mounted.
|
|||||||
|
|
||||||
Processing:
|
Processing:
|
||||||
|
|
||||||
- Records without hashes (size-unique when last scanned) are
|
- Records without a `content` hash (see "Database" above) are
|
||||||
excluded: their content is unknown, so they are never reported as
|
excluded: their content is unknown, so they are never reported as
|
||||||
duplicates.
|
duplicates.
|
||||||
- Group the remaining records by the key
|
- Group the remaining records by the key
|
||||||
`(size, head_hash, tail_hash)`.
|
`(size, head, tail, content)`.
|
||||||
- Every group with two or more paths is a duplicate group.
|
- Every group with two or more paths is a duplicate group.
|
||||||
- Within each group, sort paths lexicographically (byte order). The
|
- Within each group, sort paths lexicographically (byte order). The
|
||||||
first path is the group's `first`; every other path is a `dupe`.
|
first path is the group's `first`; every other path is a `dupe`.
|
||||||
@@ -339,11 +445,11 @@ the paths in the records, split on `/`.
|
|||||||
|
|
||||||
Definitions:
|
Definitions:
|
||||||
|
|
||||||
- A file's **signature** is `(size, head_hash, tail_hash)` — mtime is
|
- A file's **signature** is `(size, head, tail, content)` — mtime is
|
||||||
informational and excluded. An unhashed record (empty hashes) has
|
informational and excluded. A record without a `content` hash has
|
||||||
unknown content: its signature is treated as unique to that file,
|
unknown content: its signature is treated as unique to that file,
|
||||||
so a tree containing an unhashed file never compares equal to any
|
so a tree containing such a file never compares equal to any other
|
||||||
other tree.
|
tree.
|
||||||
- A directory's **digest** is a SHA-256 Merkle digest computed
|
- A directory's **digest** is a SHA-256 Merkle digest computed
|
||||||
bottom-up: serialize the directory's child entries — for a file
|
bottom-up: serialize the directory's child entries — for a file
|
||||||
child, its name and signature; for a subdirectory child, its name
|
child, its name and signature; for a subdirectory child, its name
|
||||||
@@ -406,10 +512,13 @@ Each phase gets its own display, rendered the moment the phase
|
|||||||
starts — a scan must never look hung. Loading the existing-record
|
starts — a scan must never look hung. Loading the existing-record
|
||||||
index (`load`) and the walk have no known totals while running: show
|
index (`load`) and the walk have no known totals while running: show
|
||||||
a live count, rate, and elapsed time (spinner-style, no percentage or
|
a live count, rate, and elapsed time (spinner-style, no percentage or
|
||||||
ETA). The hash and update phases
|
ETA). The content phase's display (`content`) starts the same way,
|
||||||
have exact totals — only files that actually need hashing appear in
|
counting the records checked while SQLite finds the files to read and
|
||||||
the hash total, so its ETA is meaningful. Required elements for the
|
`lstat` checks them, then shows a bar once reading starts. The hash
|
||||||
bars with known totals:
|
and update phases, and the content phase's reads, have exact totals —
|
||||||
|
only files that actually need hashing appear in the hash and content
|
||||||
|
totals, so their ETAs are meaningful. Required elements for the bars
|
||||||
|
with known totals:
|
||||||
|
|
||||||
- elapsed time
|
- elapsed time
|
||||||
- estimated time remaining
|
- estimated time remaining
|
||||||
@@ -636,9 +745,10 @@ Tracked in [TODO.md](TODO.md).
|
|||||||
|
|
||||||
## Non-goals
|
## Non-goals
|
||||||
|
|
||||||
- No full-content verification, no byte-for-byte compare, no deletion
|
- No byte-for-byte compare, and no deletion or linking of
|
||||||
or linking of duplicates. The reports are advisory; acting on them is
|
duplicates. Files that match are compared by a SHA-256 of the whole
|
||||||
the user's job.
|
file below 50 MiB, and only by samples at 50 MiB and over. The
|
||||||
|
reports are advisory; acting on them is the user's job.
|
||||||
- No persistence beyond the SQLite database described above; no
|
- No persistence beyond the SQLite database described above; no
|
||||||
export/import formats.
|
export/import formats.
|
||||||
- No daemon or filesystem watcher; scheduling rescans is cron's job.
|
- No daemon or filesystem watcher; scheduling rescans is cron's job.
|
||||||
|
|||||||
@@ -29,6 +29,40 @@
|
|||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
|
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||||
|
content-hash ladder (2026-09-22, branch `next`, closes
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||||
|
hashed in full and compared directly, with no end-window step — its
|
||||||
|
`head`, `tail`, and `content` all hold the whole-file hash. A file at
|
||||||
|
10 MiB or above gets only the 64 KiB `head` and `tail` in the hash
|
||||||
|
phase; a new content phase, after the update phase, reads it for its
|
||||||
|
`content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB
|
||||||
|
samples at or above — only when its size, `head`, and `tail` match
|
||||||
|
another record's, from the same scan or stored by an earlier one, so
|
||||||
|
a stored file gains its content hash when it gains a match. A file
|
||||||
|
that is gone or has changed since its record was written is not
|
||||||
|
read. The `content` column is part of the version 1 schema. `report`
|
||||||
|
and `trees` group by the extended signature and leave out any record
|
||||||
|
without a `content` hash, so the ladder is applied across the whole
|
||||||
|
database. README "Duplicate detection" documents every rung including
|
||||||
|
the probabilistic large-file path.
|
||||||
|
|
||||||
|
- remove the dead `files.dat` references from `Makefile`, `.gitignore`
|
||||||
|
and `.dockerignore` (2026-09-21, branch `next`, closes
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||||
|
|
||||||
|
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||||
|
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||||
|
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the
|
||||||
|
redundant tag, so both pins are the policy `# image:vX.Y.Z,
|
||||||
|
YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest
|
||||||
|
unchanged. `script/verify-lint-image-pin` parses those `FROM` lines
|
||||||
|
and still matches the tagless form; its advice line lost the now
|
||||||
|
meaningless "tag and digest". With no tag in either reference, a
|
||||||
|
tag-only disagreement no longer exists — a one-sided tag is caught as
|
||||||
|
a plain mismatch.
|
||||||
|
|
||||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint`
|
- run all linting in Docker via `Dockerfile.lint` and `script/lint`
|
||||||
(2026-08-10, branch `next`, closes
|
(2026-08-10, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
|
https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
|
||||||
|
|||||||
+1
-1
@@ -456,7 +456,7 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
|||||||
go func() {
|
go func() {
|
||||||
defer close(done)
|
defer close(done)
|
||||||
|
|
||||||
hashWorker(cancelledContext(t), jobs, results)
|
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||||
}()
|
}()
|
||||||
|
|
||||||
awaitReturn(t, done, "hashWorker")
|
awaitReturn(t, done, "hashWorker")
|
||||||
|
|||||||
@@ -36,22 +36,24 @@ const dbDirPerm = 0o755
|
|||||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||||
const createTableSQL = `
|
const createTableSQL = `
|
||||||
CREATE TABLE files (
|
CREATE TABLE files (
|
||||||
path BLOB PRIMARY KEY,
|
path BLOB PRIMARY KEY,
|
||||||
size INTEGER NOT NULL,
|
size INTEGER NOT NULL,
|
||||||
mtime INTEGER NOT NULL,
|
mtime INTEGER NOT NULL,
|
||||||
head TEXT NOT NULL,
|
head TEXT NOT NULL,
|
||||||
tail TEXT NOT NULL
|
tail TEXT NOT NULL,
|
||||||
|
content TEXT NOT NULL
|
||||||
) WITHOUT ROWID
|
) WITHOUT ROWID
|
||||||
`
|
`
|
||||||
|
|
||||||
// upsertSQL inserts one file record, replacing any existing record for
|
// upsertSQL inserts one file record, replacing any existing record for
|
||||||
// the same path.
|
// the same path.
|
||||||
const upsertSQL = `
|
const upsertSQL = `
|
||||||
INSERT INTO files (path, size, mtime, head, tail)
|
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||||
VALUES (?, ?, ?, ?, ?)
|
VALUES (?, ?, ?, ?, ?, ?)
|
||||||
ON CONFLICT (path) DO UPDATE SET
|
ON CONFLICT (path) DO UPDATE SET
|
||||||
size = excluded.size, mtime = excluded.mtime,
|
size = excluded.size, mtime = excluded.mtime,
|
||||||
head = excluded.head, tail = excluded.tail
|
head = excluded.head, tail = excluded.tail,
|
||||||
|
content = excluded.content
|
||||||
`
|
`
|
||||||
|
|
||||||
// errNoDatabase reports a missing database file for report/trees.
|
// errNoDatabase reports a missing database file for report/trees.
|
||||||
@@ -205,7 +207,7 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
|||||||
// loadFileRows reads every record from the files table.
|
// loadFileRows reads every record from the files table.
|
||||||
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, head, tail FROM files")
|
"SELECT path, size, mtime, head, tail, content FROM files")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read records: %w", err)
|
return nil, fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
@@ -220,7 +222,8 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
|||||||
r scanRec
|
r scanRec
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail)
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||||
|
&r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read record: %w", err)
|
return nil, fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
@@ -275,6 +278,62 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// contentCandidatesSQL selects every record of at least headTailMin
|
||||||
|
// bytes whose size, head, and tail equal another record's, in each
|
||||||
|
// group (the records sharing a size, head, and tail) where at least one
|
||||||
|
// record has no content hash, with whether each record has one. SQLite
|
||||||
|
// does the grouping, so no other record's hashes are loaded into
|
||||||
|
// memory; the rows come ordered by size, head, and tail, so each
|
||||||
|
// group's rows arrive together.
|
||||||
|
const contentCandidatesSQL = `
|
||||||
|
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||||
|
FROM files AS f
|
||||||
|
JOIN (
|
||||||
|
SELECT size, head, tail
|
||||||
|
FROM files
|
||||||
|
WHERE size >= ? AND head <> ''
|
||||||
|
GROUP BY size, head, tail
|
||||||
|
HAVING COUNT(*) > 1 AND SUM(content = '') > 0
|
||||||
|
) AS g USING (size, head, tail)
|
||||||
|
ORDER BY size, head, tail
|
||||||
|
`
|
||||||
|
|
||||||
|
// loadContentCandidates streams the rows of contentCandidatesSQL to fn:
|
||||||
|
// each record, without its content hash, and whether it has one.
|
||||||
|
func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||||
|
fn func(r scanRec, hashed bool),
|
||||||
|
) error {
|
||||||
|
rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
defer func() { _ = rows.Close() }()
|
||||||
|
|
||||||
|
for rows.Next() {
|
||||||
|
var (
|
||||||
|
path []byte
|
||||||
|
r scanRec
|
||||||
|
hashed int64
|
||||||
|
)
|
||||||
|
|
||||||
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read record: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
r.path = string(path)
|
||||||
|
fn(r, hashed != 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = rows.Err()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// updateBatchSize is the number of record changes committed per
|
// updateBatchSize is the number of record changes committed per
|
||||||
// transaction during the update pass. The filesystem is authoritative
|
// transaction during the update pass. The filesystem is authoritative
|
||||||
// and the database an eventually-consistent reflection of it, so
|
// and the database an eventually-consistent reflection of it, so
|
||||||
@@ -348,7 +407,7 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
|||||||
|
|
||||||
for _, r := range upserts {
|
for _, r := range upserts {
|
||||||
_, err = st.ExecContext(ctx,
|
_, err = st.ExecContext(ctx,
|
||||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail)
|
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||||
}
|
}
|
||||||
|
|||||||
+10
-4
@@ -136,10 +136,14 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
db := openTestDB(t)
|
db := openTestDB(t)
|
||||||
|
|
||||||
// Paths may contain tabs and newlines; the database must store
|
// Paths may contain tabs and newlines; the database must store
|
||||||
// them byte-exactly.
|
// them byte-exactly. Every hash, content included, comes back as
|
||||||
|
// written.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 2, mtime: 20, head: "h2", tail: "t2", path: "/a/tab\tnew\nline"},
|
{
|
||||||
{size: 1, mtime: 10, head: "h1", tail: "t1", path: "/a/x"},
|
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||||
|
path: "/a/tab\tnew\nline",
|
||||||
|
},
|
||||||
|
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||||
}
|
}
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil,
|
err := applyChanges(t.Context(), db, recs, nil,
|
||||||
@@ -163,7 +167,9 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
|
|
||||||
// An upsert for an existing path updates in place; a delete
|
// An upsert for an existing path updates in place; a delete
|
||||||
// removes exactly its path.
|
// removes exactly its path.
|
||||||
upd := scanRec{size: 3, mtime: 30, head: "h3", tail: "t3", path: "/a/x"}
|
upd := scanRec{
|
||||||
|
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||||
|
}
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||||
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
||||||
|
|||||||
@@ -1,10 +1,14 @@
|
|||||||
// Command sfdupes quickly identifies candidate duplicate files across
|
// Command sfdupes quickly identifies candidate duplicate files across
|
||||||
// very large filesystems without reading full file contents. Files are
|
// very large filesystems without reading every byte of every file.
|
||||||
// considered duplicates when they have identical size, identical SHA-256
|
// Files are considered duplicates when their sizes are equal and they
|
||||||
// of their first 1024 bytes, and identical SHA-256 of their last 1024
|
// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is
|
||||||
// bytes. scan maintains a persistent SQLite database of file signatures
|
// hashed in full. A larger file is compared on the hashes of its first
|
||||||
// (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) that the
|
// and last 64 KiB, and only when those match another file's is its
|
||||||
// reporting subcommands read.
|
// content hash computed and compared: of the whole file when it is
|
||||||
|
// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB
|
||||||
|
// or larger. scan maintains a persistent SQLite database of file
|
||||||
|
// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite)
|
||||||
|
// that the reporting subcommands read.
|
||||||
//
|
//
|
||||||
// Usage:
|
// Usage:
|
||||||
//
|
//
|
||||||
@@ -97,7 +101,7 @@ func run(args []string, stderr io.Writer) int {
|
|||||||
func newRootCommand(stderr io.Writer) *cobra.Command {
|
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||||
root := &cobra.Command{
|
root := &cobra.Command{
|
||||||
Use: "sfdupes",
|
Use: "sfdupes",
|
||||||
Short: "Find candidate duplicate files by size and head/tail SHA-256",
|
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||||
Version: Version,
|
Version: Version,
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||||
@@ -127,7 +131,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
|||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
||||||
"concurrent workers for the walk and hash phases")
|
"concurrent workers for the walk, hash, and content phases")
|
||||||
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
||||||
"do not cross filesystem boundaries")
|
"do not cross filesystem boundaries")
|
||||||
|
|
||||||
|
|||||||
@@ -17,14 +17,15 @@ const ioBufSize = 1 << 20
|
|||||||
const minGroupSize = 2
|
const minGroupSize = 2
|
||||||
|
|
||||||
// scanRec is one file record from the database. The signature (size,
|
// scanRec is one file record from the database. The signature (size,
|
||||||
// head, tail) is the duplicate key; mtime is informational only and
|
// head, tail, content) is the duplicate key; mtime is informational
|
||||||
// used by scan for change detection.
|
// only and used by scan for change detection.
|
||||||
type scanRec struct {
|
type scanRec struct {
|
||||||
size int64
|
size int64
|
||||||
mtime int64
|
mtime int64
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
path string
|
content string
|
||||||
|
path string
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadRecords opens the database and reads every file record for the
|
// loadRecords opens the database and reads every file record for the
|
||||||
@@ -52,8 +53,9 @@ func loadRecords(ctx context.Context) ([]scanRec, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||||
// head hash, and tail hash. paths is sorted lexicographically; the
|
// head hash, tail hash, and content hash. paths is sorted
|
||||||
// first entry is the group's "first", the rest are dupes.
|
// lexicographically; the first entry is the group's "first", the rest
|
||||||
|
// are dupes.
|
||||||
type dupeGroup struct {
|
type dupeGroup struct {
|
||||||
size int64
|
size int64
|
||||||
paths []string
|
paths []string
|
||||||
@@ -115,14 +117,15 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
|||||||
groups := make(map[fileSig][]string)
|
groups := make(map[fileSig][]string)
|
||||||
|
|
||||||
for _, r := range recs {
|
for _, r := range recs {
|
||||||
// A record without hashes (its size was unique when last
|
// A record without a content hash has unknown content and is
|
||||||
// scanned) has unknown content and is never reported as a
|
// never reported as a duplicate (README "Database").
|
||||||
// duplicate.
|
if r.content == "" {
|
||||||
if r.head == "" {
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
k := fileSig{size: r.size, head: r.head, tail: r.tail}
|
k := fileSig{
|
||||||
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
groups[k] = append(groups[k], r.path)
|
groups[k] = append(groups[k], r.path)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+43
-17
@@ -9,15 +9,15 @@ func TestCollectDupeGroups(t *testing.T) {
|
|||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 100, head: "h", tail: "t", path: "/z/b"},
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
||||||
{size: 100, head: "h", tail: "t", path: "/z/a"},
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
||||||
{size: 100, head: "h", tail: "t", path: "/z/c"},
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
||||||
{size: 4000, head: "H", tail: "T", path: "/big/2"},
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
||||||
{size: 4000, head: "H", tail: "T", path: "/big/1"},
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
||||||
// Same size as the /z group but a different head hash.
|
// Same size as the /z group but a different head hash.
|
||||||
{size: 100, head: "other", tail: "t", path: "/z/d"},
|
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
||||||
// A singleton signature must not form a group.
|
// A singleton signature must not form a group.
|
||||||
{size: 7, head: "u", tail: "u", path: "/lonely"},
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := collectDupeGroups(recs)
|
||||||
@@ -38,14 +38,40 @@ func TestCollectDupeGroups(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
// Same size, head, and tail, but different content hashes: the final
|
||||||
|
// rung keeps them apart, so no group forms. Matching content groups.
|
||||||
|
// Records without a content hash never group, not even with each
|
||||||
|
// other.
|
||||||
|
recs := []scanRec{
|
||||||
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
||||||
|
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
||||||
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
||||||
|
{size: 100, head: "h", tail: "t", path: "/d"},
|
||||||
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||||
|
}
|
||||||
|
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 {
|
||||||
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||||
|
len(groups))
|
||||||
|
}
|
||||||
|
|
||||||
|
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
||||||
|
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// mtime is informational only; records differing only in mtime
|
// mtime is informational only; records differing only in mtime
|
||||||
// still group together.
|
// still group together.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 9, mtime: 100, head: "h", tail: "t", path: "/m/1"},
|
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||||
{size: 9, mtime: 200, head: "h", tail: "t", path: "/m/2"},
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := collectDupeGroups(recs)
|
||||||
@@ -58,10 +84,10 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
|||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 50, head: "b", tail: "b", path: "/beta/2"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
||||||
{size: 50, head: "b", tail: "b", path: "/beta/1"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
||||||
{size: 50, head: "a", tail: "a", path: "/alpha/2"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
||||||
{size: 50, head: "a", tail: "a", path: "/alpha/1"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := collectDupeGroups(recs)
|
||||||
@@ -80,10 +106,10 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
|||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 1, head: "a", tail: "a", path: "/p/1"},
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
||||||
{size: 1, head: "a", tail: "a", path: "/p/2"},
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
||||||
{size: 2, head: "b", tail: "b", path: "/q/1"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
||||||
{size: 2, head: "b", tail: "b", path: "/q/2"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
forward := collectDupeGroups(recs)
|
forward := collectDupeGroups(recs)
|
||||||
|
|||||||
@@ -6,7 +6,9 @@ import (
|
|||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
@@ -16,8 +18,40 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
)
|
)
|
||||||
|
|
||||||
// chunk is the number of bytes hashed from each end of a file.
|
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||||
const chunk = 1024
|
// detection"). A same-size candidate below headTailMin is hashed in
|
||||||
|
// full and compared directly; a larger one is separated first by the
|
||||||
|
// hashes of its end windows, then by a content hash that is exact below
|
||||||
|
// wholeFileMax and deliberately sampled at or above it. The hash phase
|
||||||
|
// reads only the end windows of a larger file; the content phase reads
|
||||||
|
// it for its content hash only once its size, head, and tail match
|
||||||
|
// another file's.
|
||||||
|
|
||||||
|
// headTailMin is the size threshold for the end-window gate. A file
|
||||||
|
// smaller than this is hashed in full directly, with no separate head
|
||||||
|
// and tail step: its head, tail, and content all carry the whole-file
|
||||||
|
// hash. A file this size or larger is separated first by its end
|
||||||
|
// windows.
|
||||||
|
const headTailMin = 10 * 1024 * 1024
|
||||||
|
|
||||||
|
// headTailWindow is the number of bytes hashed from each end of a file
|
||||||
|
// at or above headTailMin (the head and tail rungs). Because
|
||||||
|
// headTailMin is far larger than two windows, the head and tail windows
|
||||||
|
// never overlap.
|
||||||
|
const headTailWindow = 64 * 1024
|
||||||
|
|
||||||
|
// wholeFileMax is the size boundary between the two content rungs: a
|
||||||
|
// file strictly smaller than this is content-hashed in full; a file
|
||||||
|
// this size or larger is content-hashed by sampling.
|
||||||
|
const wholeFileMax = 50 * 1024 * 1024
|
||||||
|
|
||||||
|
// sampleStride is the spacing between content samples for large files:
|
||||||
|
// one window is read at each gigabyte-aligned offset (0, 1 GiB, ...).
|
||||||
|
const sampleStride = 1024 * 1024 * 1024
|
||||||
|
|
||||||
|
// sampleWindow is the number of bytes read at each large-file sample
|
||||||
|
// offset, truncated at end of file.
|
||||||
|
const sampleWindow = 1024 * 1024
|
||||||
|
|
||||||
// workQueueDepth bounds the job and result channels feeding the walk
|
// workQueueDepth bounds the job and result channels feeding the walk
|
||||||
// and hash worker pools.
|
// and hash worker pools.
|
||||||
@@ -44,16 +78,17 @@ type fileMeta struct {
|
|||||||
hashed bool
|
hashed bool
|
||||||
}
|
}
|
||||||
|
|
||||||
// runScan implements the scan subcommand: three sequential phases —
|
// runScan implements the scan subcommand: four sequential phases —
|
||||||
// walk (which stats each file as it is discovered), hash, update —
|
// walk (which stats each file as it is discovered), hash, update,
|
||||||
// that synchronize the persistent database with the filesystem state
|
// content — that synchronize the persistent database with the
|
||||||
// under the PATH operands. Only files whose size at least one other
|
// filesystem state under the PATH operands. Only files whose size at
|
||||||
// file shares are ever hashed: a size-unique file cannot be a
|
// least one other file shares are ever hashed: a size-unique file
|
||||||
// duplicate. Flag parsing and the at-least-one-operand check are done
|
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||||
// by cobra. Errors are returned rather than exiting, so that the
|
// hash only when its size, head, and tail match another file's. Flag
|
||||||
// deferred close — which checkpoints the SQLite WAL — always runs.
|
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||||
// Cancelling ctx unwinds the worker pools and aborts the scan with the
|
// are returned rather than exiting, so that the deferred close — which
|
||||||
// context's error.
|
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||||
|
// worker pools and aborts the scan with the context's error.
|
||||||
func runScan(ctx context.Context, roots []string, workers int,
|
func runScan(ctx context.Context, roots []string, workers int,
|
||||||
oneFS bool,
|
oneFS bool,
|
||||||
) error {
|
) error {
|
||||||
@@ -166,13 +201,16 @@ type scanState struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// syncScan synchronizes the database with the filesystem under roots
|
// syncScan synchronizes the database with the filesystem under roots
|
||||||
// in three sequential phases: walk (enumerate and stat every file,
|
// in four sequential phases: walk (enumerate and stat every file,
|
||||||
// building a complete size census), hash (read only the new or
|
// building a complete size census), hash (read only the new or
|
||||||
// changed — or previously unhashed — files whose size at least one
|
// changed — or previously unhashed — files whose size at least one
|
||||||
// other file shares, committing results in batches as they arrive),
|
// other file shares, committing results in batches as they arrive),
|
||||||
// and update (record the size-unique files without reading them, and
|
// update (record the size-unique files without reading them, and
|
||||||
// delete the records the scan no longer verifies). Records outside
|
// delete the records the scan no longer verifies), and content (fill
|
||||||
// the roots are never touched.
|
// in the content hash of every record of headTailMin or more whose
|
||||||
|
// size, head, and tail match another record's). Records outside the
|
||||||
|
// roots are never touched, except that the content phase fills in
|
||||||
|
// their content hash.
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) (scanStats, error) {
|
) (scanStats, error) {
|
||||||
@@ -207,7 +245,12 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
return s.st, err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
return s.st, s.updatePhase(ctx)
|
err = s.updatePhase(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return s.st, err
|
||||||
|
}
|
||||||
|
|
||||||
|
return s.st, s.contentPhase(ctx, workers)
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadIndex indexes the database records under the scan roots for
|
// loadIndex indexes the database records under the scan roots for
|
||||||
@@ -241,7 +284,8 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
|||||||
|
|
||||||
// walkPhase drains the walk, appending every walked file's size to
|
// walkPhase drains the walk, appending every walked file's size to
|
||||||
// the census and resolving what it can immediately: an unchanged file
|
// the census and resolving what it can immediately: an unchanged file
|
||||||
// whose record already has hashes needs nothing further. It returns
|
// whose record already has hashes needs nothing from the hash phase
|
||||||
|
// (the content phase may still fill in its content hash). It returns
|
||||||
// the new-or-changed files and the unchanged files whose records lack
|
// the new-or-changed files and the unchanged files whose records lack
|
||||||
// hashes; both remain candidates until the census decides whether
|
// hashes; both remain candidates until the census decides whether
|
||||||
// their sizes are shared.
|
// their sizes are shared.
|
||||||
@@ -380,27 +424,40 @@ func sameInode(a, b fileRec) bool {
|
|||||||
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashPhase hashes every queued file with the worker pool — one read
|
// hashPhase hashes every queued file with hashSignature — the head and
|
||||||
// per inode run, in inode order — committing completed records to the
|
// tail of a file of headTailMin or more, the whole file below that —
|
||||||
// database in batches as results arrive, so a long scan persists its
|
// committing completed records to the database in batches as results
|
||||||
// progress as it goes (an interrupted scan resumes cheaply: the next
|
// arrive, so a long scan persists its progress as it goes (an
|
||||||
// run skips everything already recorded). The total counts actual
|
// interrupted scan resumes cheaply: the next run skips everything
|
||||||
// reads, so the bar shows a real ETA. A run that fails to hash is
|
// already recorded). A run that fails to hash is warned about and
|
||||||
// warned about and skipped; stale records for its paths, if any, are
|
// skipped; stale records for its paths, if any, are deleted by the
|
||||||
// deleted by the update phase.
|
// update phase.
|
||||||
|
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||||
|
runs := hashRuns(s.toHash)
|
||||||
|
s.toHash = nil
|
||||||
|
|
||||||
|
return s.readRuns(ctx, workers, "hash", runs, hashSignature, s.recordRun)
|
||||||
|
}
|
||||||
|
|
||||||
|
// readRuns reads runs with the worker pool, one read per inode run, in
|
||||||
|
// the order given, under a progress display named label. The workers
|
||||||
|
// compute each run's hashes with hash, and each result goes to record;
|
||||||
|
// a run that fails to read is warned about and counted as skipped
|
||||||
|
// instead. The total counts actual reads, so the bar shows a real ETA.
|
||||||
//
|
//
|
||||||
// Returning early — a failed database write, or a cancelled scan — must
|
// Returning early — a failed database write, or a cancelled scan — must
|
||||||
// not strand the pool: the feeder would park forever on a full jobs
|
// not strand the pool: the feeder would park forever on a full jobs
|
||||||
// channel and every worker on a full results channel. The deferred stop
|
// channel and every worker on a full results channel. The deferred stop
|
||||||
// is what prevents that.
|
// is what prevents that.
|
||||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
func (s *scanState) readRuns(ctx context.Context, workers int,
|
||||||
runs := hashRuns(s.toHash)
|
label string, runs [][]fileRec,
|
||||||
s.toHash = nil
|
hash func(path string, size int64) (string, string, string, error),
|
||||||
|
record func(ctx context.Context, r hashResult) error,
|
||||||
pool := startHashPool(ctx, runs, workers)
|
) error {
|
||||||
|
pool := startHashPool(ctx, runs, workers, hash)
|
||||||
defer pool.stop()
|
defer pool.stop()
|
||||||
|
|
||||||
prog := newProgress("hash", int64(len(runs)))
|
prog := newProgress(label, int64(len(runs)))
|
||||||
defer prog.finish()
|
defer prog.finish()
|
||||||
|
|
||||||
for range runs {
|
for range runs {
|
||||||
@@ -417,12 +474,12 @@ func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
|||||||
if r.err != nil {
|
if r.err != nil {
|
||||||
s.st.skipped += len(r.run)
|
s.st.skipped += len(r.run)
|
||||||
|
|
||||||
prog.warnf("hash %s: %v", r.run[0].path, r.err)
|
prog.warnf("%s %s: %v", label, r.run[0].path, r.err)
|
||||||
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
err := s.recordRun(ctx, r)
|
err := record(ctx, r)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -439,14 +496,21 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
|||||||
s.resolve(rec.path)
|
s.resolve(rec.path)
|
||||||
|
|
||||||
s.batch = append(s.batch, scanRec{
|
s.batch = append(s.batch, scanRec{
|
||||||
size: rec.size,
|
size: rec.size,
|
||||||
mtime: rec.mtime,
|
mtime: rec.mtime,
|
||||||
head: r.head,
|
head: r.head,
|
||||||
tail: r.tail,
|
tail: r.tail,
|
||||||
path: rec.path,
|
content: r.content,
|
||||||
|
path: rec.path,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return s.commitFullBatch(ctx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// commitFullBatch commits the running batch once it holds
|
||||||
|
// updateBatchSize records.
|
||||||
|
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||||
if len(s.batch) < updateBatchSize {
|
if len(s.batch) < updateBatchSize {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -504,6 +568,147 @@ func (s *scanState) updatePhase(ctx context.Context) error {
|
|||||||
return applyChanges(ctx, s.db, nil, deletes, prog)
|
return applyChanges(ctx, s.db, nil, deletes, prog)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// contentPhase fills in the content hash of every record of headTailMin
|
||||||
|
// or more that lacks one and whose size, head, and tail equal another
|
||||||
|
// record's, anywhere in the database: records from this scan and
|
||||||
|
// records stored by earlier scans, inside or outside the roots. Only
|
||||||
|
// such a file can still be a duplicate, so no other file of headTailMin
|
||||||
|
// or more is read beyond its end windows. The files are read with the
|
||||||
|
// hash phase's worker pool and their records written back in batches. A
|
||||||
|
// failed read is warned about and counted as skipped; the record keeps
|
||||||
|
// its empty content, so it is never grouped, and a later scan tries
|
||||||
|
// again.
|
||||||
|
func (s *scanState) contentPhase(ctx context.Context, workers int) error {
|
||||||
|
toRead, recs, err := s.contentCandidates(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
err = s.readRuns(ctx, workers, "content", hashRuns(toRead),
|
||||||
|
hashContentOnly, func(ctx context.Context, r hashResult) error {
|
||||||
|
// Every path in the run keeps its record's head and tail
|
||||||
|
// and gains the one content hash read for the run.
|
||||||
|
for _, f := range r.run {
|
||||||
|
rec := recs[f.path]
|
||||||
|
rec.content = r.content
|
||||||
|
s.batch = append(s.batch, rec)
|
||||||
|
}
|
||||||
|
|
||||||
|
return s.commitFullBatch(ctx)
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
return applyChanges(ctx, s.db, s.batch, nil, nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
// contentCandidates returns the files the content phase reads, and
|
||||||
|
// their records by path. Every record contentCandidatesSQL returns has
|
||||||
|
// its file checked with lstat, whether or not it already has a content
|
||||||
|
// hash: a file that is gone, is no longer a regular file, or has
|
||||||
|
// changed by the walk's rule keeps its record as it is and does not
|
||||||
|
// count as a match for the others, and any other lstat error is warned
|
||||||
|
// about and counted as skipped, with the same result. If such a record
|
||||||
|
// has no content hash, it stays out of duplicate groups; if it has one,
|
||||||
|
// it is still reported until a scan covering its own tree updates or
|
||||||
|
// removes it. The files of a group that pass and have no content hash
|
||||||
|
// are read only if at least minGroupSize of the group's files pass, so
|
||||||
|
// a group whose other members are all stale costs no reads. Only the
|
||||||
|
// records to be read are kept.
|
||||||
|
func (s *scanState) contentCandidates(
|
||||||
|
ctx context.Context,
|
||||||
|
) ([]fileRec, map[string]scanRec, error) {
|
||||||
|
// The query and the checks take real time on a large database;
|
||||||
|
// without a display the scan looks hung before the reads begin.
|
||||||
|
prog := newProgress("content", -1)
|
||||||
|
defer prog.finish()
|
||||||
|
|
||||||
|
var (
|
||||||
|
toRead []fileRec
|
||||||
|
first scanRec // the current group's first record
|
||||||
|
passed int // the current group's files that passed the check
|
||||||
|
unread []fileRec // those of them without a content hash
|
||||||
|
)
|
||||||
|
|
||||||
|
recs := make(map[string]scanRec)
|
||||||
|
|
||||||
|
// endGroup queues the current group's files to read if at least
|
||||||
|
// minGroupSize of its files passed, and drops their records if not.
|
||||||
|
endGroup := func() {
|
||||||
|
if passed >= minGroupSize {
|
||||||
|
toRead = append(toRead, unread...)
|
||||||
|
} else {
|
||||||
|
for _, f := range unread {
|
||||||
|
delete(recs, f.path)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
passed, unread = 0, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
err := loadContentCandidates(ctx, s.db, func(r scanRec, hashed bool) {
|
||||||
|
prog.increment()
|
||||||
|
|
||||||
|
if r.size != first.size || r.head != first.head || r.tail != first.tail {
|
||||||
|
endGroup()
|
||||||
|
|
||||||
|
first = r
|
||||||
|
}
|
||||||
|
|
||||||
|
f, ok, err := unchangedFile(r)
|
||||||
|
if err != nil {
|
||||||
|
s.st.skipped++
|
||||||
|
|
||||||
|
prog.warnf("content %s: %v", r.path, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
passed++
|
||||||
|
|
||||||
|
if !hashed {
|
||||||
|
unread = append(unread, f)
|
||||||
|
recs[r.path] = r
|
||||||
|
}
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
endGroup()
|
||||||
|
|
||||||
|
return toRead, recs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// unchangedFile lstats the file r names and returns it for reading if
|
||||||
|
// it is still the regular file r records: the same size, and an mtime
|
||||||
|
// no newer than recorded (the walk's change rule). A file that is gone
|
||||||
|
// or has changed reports false; any other lstat error is returned.
|
||||||
|
func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||||
|
fi, err := os.Lstat(r.path)
|
||||||
|
if errors.Is(err, fs.ErrNotExist) {
|
||||||
|
return fileRec{}, false, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
return fileRec{}, false, err
|
||||||
|
}
|
||||||
|
|
||||||
|
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||||
|
fi.ModTime().Unix() > r.mtime {
|
||||||
|
return fileRec{}, false, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
dev, ino := inodeOfInfo(fi)
|
||||||
|
|
||||||
|
return fileRec{
|
||||||
|
path: r.path, size: r.size, mtime: r.mtime, dev: dev, ino: ino,
|
||||||
|
}, true, nil
|
||||||
|
}
|
||||||
|
|
||||||
// underAnyRoot reports whether path is any of the roots or lies under
|
// underAnyRoot reports whether path is any of the roots or lies under
|
||||||
// one of them.
|
// one of them.
|
||||||
func underAnyRoot(path string, roots []string) bool {
|
func underAnyRoot(path string, roots []string) bool {
|
||||||
@@ -834,13 +1039,16 @@ func inodeOfInfo(fi fs.FileInfo) (uint64, uint64) {
|
|||||||
return statDev(st), st.Ino
|
return statDev(st), st.Ino
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashResult carries one inode run's head/tail hashes (or the error
|
// hashResult carries the hashes computed for one inode run (or the
|
||||||
// that prevented hashing it) from the hash workers to the hash phase.
|
// error that prevented computing them) from the pool's workers to the
|
||||||
|
// phase that started the pool: head, tail, and content from
|
||||||
|
// hashSignature, content alone from hashContentOnly.
|
||||||
type hashResult struct {
|
type hashResult struct {
|
||||||
run []fileRec
|
run []fileRec
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
err error
|
content string
|
||||||
|
err error
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashPool owns every goroutine of the hash worker pool: the feeder
|
// hashPool owns every goroutine of the hash worker pool: the feeder
|
||||||
@@ -856,10 +1064,10 @@ type hashPool struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// startHashPool starts the feeder and the workers over runs. Workers
|
// startHashPool starts the feeder and the workers over runs. Workers
|
||||||
// hash each run's first path (all paths in a run are hard links to the
|
// hash each run's first path with hash (all paths in a run are hard
|
||||||
// same inode) and write one result per run.
|
// links to the same inode) and write one result per run.
|
||||||
func startHashPool(ctx context.Context, runs [][]fileRec,
|
func startHashPool(ctx context.Context, runs [][]fileRec, workers int,
|
||||||
workers int,
|
hash func(path string, size int64) (string, string, string, error),
|
||||||
) *hashPool {
|
) *hashPool {
|
||||||
ctx, cancel := context.WithCancel(ctx)
|
ctx, cancel := context.WithCancel(ctx)
|
||||||
|
|
||||||
@@ -871,7 +1079,7 @@ func startHashPool(ctx context.Context, runs [][]fileRec,
|
|||||||
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
||||||
|
|
||||||
for range workers {
|
for range workers {
|
||||||
wg.Go(func() { hashWorker(ctx, jobs, results) })
|
wg.Go(func() { hashWorker(ctx, jobs, results, hash) })
|
||||||
}
|
}
|
||||||
|
|
||||||
done := make(chan struct{})
|
done := make(chan struct{})
|
||||||
@@ -917,24 +1125,25 @@ func feedHashJobs(ctx context.Context, runs [][]fileRec,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashWorker hashes one inode run at a time until jobs is closed or the
|
// hashWorker hashes one inode run at a time with hash until jobs is
|
||||||
// scan is cancelled. A cancelled worker drops the runs still queued
|
// closed or the scan is cancelled. A cancelled worker drops the runs
|
||||||
// instead of stopping its reads of jobs: the range must run out for the
|
// still queued instead of stopping its reads of jobs: the range must
|
||||||
// pool to tear down, and reading a file nobody wants the hash of only
|
// run out for the pool to tear down, and reading a file nobody wants
|
||||||
// delays that.
|
// the hash of only delays that.
|
||||||
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||||
results chan<- hashResult,
|
results chan<- hashResult,
|
||||||
|
hash func(path string, size int64) (string, string, string, error),
|
||||||
) {
|
) {
|
||||||
for run := range jobs {
|
for run := range jobs {
|
||||||
if ctx.Err() != nil {
|
if ctx.Err() != nil {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
head, tail, err := hashHeadTail(run[0].path, run[0].size)
|
head, tail, content, err := hash(run[0].path, run[0].size)
|
||||||
|
|
||||||
select {
|
select {
|
||||||
case results <- hashResult{
|
case results <- hashResult{
|
||||||
run: run, head: head, tail: tail, err: err,
|
run: run, head: head, tail: tail, content: content, err: err,
|
||||||
}:
|
}:
|
||||||
case <-ctx.Done():
|
case <-ctx.Done():
|
||||||
return
|
return
|
||||||
@@ -942,55 +1151,151 @@ func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head
|
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head,
|
||||||
// and tail hash of every zero-length file.
|
// tail, and content hash of every zero-length file.
|
||||||
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
||||||
"27ae41e4649b934ca495991b7852b855"
|
"27ae41e4649b934ca495991b7852b855"
|
||||||
|
|
||||||
// hashHeadTail returns the lowercase-hex SHA-256 of the first
|
// hashSignature computes the hashes the hash phase records for a file
|
||||||
// min(chunk, size) bytes and of the last min(chunk, size) bytes of the
|
// whose size is shared; with the file size they form its duplicate
|
||||||
// file at path. The two reads overlap when size < 2*chunk. size is the
|
// signature. A file below headTailMin is hashed in full and its
|
||||||
// value recorded when the file was statted; a zero-length file's
|
// whole-file SHA-256 is returned as head, tail, and content alike —
|
||||||
// hashes are constant, so it is never even opened.
|
// that range takes no separate end-window step. For a file at or above
|
||||||
func hashHeadTail(path string, size int64) (string, string, error) {
|
// headTailMin only the head and tail are computed, the SHA-256 of its
|
||||||
|
// first and last headTailWindow bytes, and content is returned empty:
|
||||||
|
// the content phase computes it with hashContentOnly once the file's
|
||||||
|
// size, head, and tail match another file's. Two files are duplicates
|
||||||
|
// only when all four agree; any mismatch means not a duplicate. size
|
||||||
|
// is the value recorded when the file was statted; a zero-length file
|
||||||
|
// has constant hashes and is never opened.
|
||||||
|
func hashSignature(path string, size int64) (string, string, string, error) {
|
||||||
if size == 0 {
|
if size == 0 {
|
||||||
return emptyHash, emptyHash, nil
|
return emptyHash, emptyHash, emptyHash, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||||
f, err := os.Open(path)
|
f, err := os.Open(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", err
|
return "", "", "", err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = f.Close() }()
|
defer func() { _ = f.Close() }()
|
||||||
|
|
||||||
n := min(int64(chunk), size)
|
// Below the threshold the whole file is hashed directly, with no
|
||||||
|
// end-window step: head and tail both carry the whole-file hash.
|
||||||
|
if size < int64(headTailMin) {
|
||||||
|
content, err := hashWhole(f, size)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
buf := make([]byte, n)
|
return content, content, content, nil
|
||||||
|
}
|
||||||
|
|
||||||
_, err = f.ReadAt(buf, 0)
|
head, tail, err := hashEnds(f, size)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
return head, tail, "", nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashContentOnly returns the content hash of the file at path, which
|
||||||
|
// is at least headTailMin bytes: the content phase's read. head and
|
||||||
|
// tail are returned empty, because the content phase keeps the ones its
|
||||||
|
// records already hold.
|
||||||
|
func hashContentOnly(path string, size int64) (string, string, string, error) {
|
||||||
|
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||||
|
f, err := os.Open(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
defer func() { _ = f.Close() }()
|
||||||
|
|
||||||
|
content, err := hashContent(f, size)
|
||||||
|
|
||||||
|
return "", "", content, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashEnds returns the SHA-256 of the first and last headTailWindow
|
||||||
|
// bytes of f. It is called only for files at least headTailMin, which
|
||||||
|
// is far larger than two windows, so the windows never overlap and both
|
||||||
|
// reads are always full.
|
||||||
|
func hashEnds(f *os.File, size int64) (string, string, error) {
|
||||||
|
buf := make([]byte, headTailWindow)
|
||||||
|
|
||||||
|
_, err := f.ReadAt(buf, 0)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", err
|
return "", "", err
|
||||||
}
|
}
|
||||||
|
|
||||||
h := sha256.Sum256(buf)
|
h := sha256.Sum256(buf)
|
||||||
|
head := hex.EncodeToString(h[:])
|
||||||
|
|
||||||
// When the whole file fits in one chunk the tail window is exactly
|
_, err = f.ReadAt(buf, size-int64(headTailWindow))
|
||||||
// the bytes just read: reuse the head hash instead of issuing a
|
|
||||||
// second read for every small file.
|
|
||||||
if size <= int64(chunk) {
|
|
||||||
hh := hex.EncodeToString(h[:])
|
|
||||||
|
|
||||||
return hh, hh, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = f.ReadAt(buf, size-n)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", err
|
return "", "", err
|
||||||
}
|
}
|
||||||
|
|
||||||
t := sha256.Sum256(buf)
|
t := sha256.Sum256(buf)
|
||||||
|
|
||||||
return hex.EncodeToString(h[:]), hex.EncodeToString(t[:]), nil
|
return head, hex.EncodeToString(t[:]), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashContent returns the content-rung hash of f: the SHA-256 of the
|
||||||
|
// whole file when it is smaller than wholeFileMax, or of sampled
|
||||||
|
// windows when it is that size or larger.
|
||||||
|
func hashContent(f *os.File, size int64) (string, error) {
|
||||||
|
if size >= int64(wholeFileMax) {
|
||||||
|
return hashSamples(f, size)
|
||||||
|
}
|
||||||
|
|
||||||
|
return hashWhole(f, size)
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashWhole returns the SHA-256 of the entire file. A SectionReader is
|
||||||
|
// used so the read is independent of the offset left by any end-window
|
||||||
|
// reads. Reading fewer than size bytes means the file shrank between
|
||||||
|
// the stat and the hash; that is an error rather than a hash of content
|
||||||
|
// that no longer matches the recorded size.
|
||||||
|
func hashWhole(f *os.File, size int64) (string, error) {
|
||||||
|
h := sha256.New()
|
||||||
|
|
||||||
|
n, err := io.Copy(h, io.NewSectionReader(f, 0, size))
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
if n != size {
|
||||||
|
return "", fmt.Errorf("read %d of %d bytes: %w", n, size,
|
||||||
|
io.ErrUnexpectedEOF)
|
||||||
|
}
|
||||||
|
|
||||||
|
return hex.EncodeToString(h.Sum(nil)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashSamples feeds sampleWindow bytes at each gigabyte-aligned offset
|
||||||
|
// (0, sampleStride, 2*sampleStride, ... while inside the file), in
|
||||||
|
// order, into one hash, each window truncated at end of file. This is
|
||||||
|
// the probabilistic large-file rung: two files of equal size agreeing
|
||||||
|
// on every sample are reported as duplicates without every byte being
|
||||||
|
// read. Because size is part of the signature, files of different sizes
|
||||||
|
// never reach this comparison, so the sample boundaries always align.
|
||||||
|
func hashSamples(f *os.File, size int64) (string, error) {
|
||||||
|
h := sha256.New()
|
||||||
|
buf := make([]byte, sampleWindow)
|
||||||
|
|
||||||
|
for off := int64(0); off < size; off += int64(sampleStride) {
|
||||||
|
n := min(int64(sampleWindow), size-off)
|
||||||
|
|
||||||
|
_, err := f.ReadAt(buf[:n], off)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
h.Write(buf[:n])
|
||||||
|
}
|
||||||
|
|
||||||
|
return hex.EncodeToString(h.Sum(nil)), nil
|
||||||
}
|
}
|
||||||
|
|||||||
+638
-42
@@ -53,7 +53,23 @@ func pattern(tag byte, n int) []byte {
|
|||||||
return data
|
return data
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestHashHeadTail(t *testing.T) {
|
// sig returns a file's full signature (head, tail, content), failing the
|
||||||
|
// test on any error.
|
||||||
|
func sig(t *testing.T, path string, size int64) (string, string, string) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
head, tail, content, err := hashSignature(path, size)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("hashSignature %s: %v", path, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return head, tail, content
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestHashSignatureBelowThreshold verifies that a file below headTailMin
|
||||||
|
// is hashed in full and compared directly: head, tail, and content all
|
||||||
|
// carry the whole-file SHA-256, with no separate end-window step.
|
||||||
|
func TestHashSignatureBelowThreshold(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
@@ -62,13 +78,10 @@ func TestHashHeadTail(t *testing.T) {
|
|||||||
name string
|
name string
|
||||||
data []byte
|
data []byte
|
||||||
}{
|
}{
|
||||||
{"empty", nil},
|
|
||||||
{"one-byte", []byte("x")},
|
{"one-byte", []byte("x")},
|
||||||
{"under-one-chunk", pattern(1, chunk-1)},
|
{"one-window", pattern(1, headTailWindow)},
|
||||||
{"exactly-one-chunk", pattern(2, chunk)},
|
{"several-windows", pattern(2, 3*headTailWindow)},
|
||||||
{"overlapping-reads", pattern(3, chunk+chunk/2)},
|
{"near-threshold", pattern(3, headTailMin-1)},
|
||||||
{"exactly-two-chunks", pattern(4, 2*chunk)},
|
|
||||||
{"beyond-two-chunks", pattern(5, 3*chunk)},
|
|
||||||
}
|
}
|
||||||
for _, c := range cases {
|
for _, c := range cases {
|
||||||
t.Run(c.name, func(t *testing.T) {
|
t.Run(c.name, func(t *testing.T) {
|
||||||
@@ -76,49 +89,619 @@ func TestHashHeadTail(t *testing.T) {
|
|||||||
|
|
||||||
p := writeFile(t, dir, c.name, c.data)
|
p := writeFile(t, dir, c.name, c.data)
|
||||||
|
|
||||||
head, tail, err := hashHeadTail(p, int64(len(c.data)))
|
head, tail, content := sig(t, p, int64(len(c.data)))
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("hashHeadTail: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
n := min(chunk, len(c.data))
|
whole := hexSum(c.data)
|
||||||
if want := hexSum(c.data[:n]); head != want {
|
if head != whole || tail != whole || content != whole {
|
||||||
t.Errorf("head = %s, want %s", head, want)
|
t.Errorf("head=%s tail=%s content=%s, want all whole-file %s",
|
||||||
}
|
head, tail, content, whole)
|
||||||
|
|
||||||
if want := hexSum(c.data[len(c.data)-n:]); tail != want {
|
|
||||||
t.Errorf("tail = %s, want %s", tail, want)
|
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestHashHeadTailErrors(t *testing.T) {
|
// TestHashSignatureEnds exercises the head and tail rungs, which apply
|
||||||
|
// only to files at least headTailMin. Sparse files keep the fixtures
|
||||||
|
// cheap: a difference in the first window changes only head, a
|
||||||
|
// difference in the last window changes only tail, and a difference
|
||||||
|
// between the windows changes neither end hash but does change the
|
||||||
|
// whole-file content rung (the file is below wholeFileMax).
|
||||||
|
// hashSignature leaves the content hash of a file this size to the
|
||||||
|
// content phase, so that rung is checked through a scan.
|
||||||
|
func TestHashSignatureEnds(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
|
|
||||||
_, _, err := hashHeadTail(filepath.Join(dir, "missing"), 1)
|
// Between headTailMin and wholeFileMax: the end-window gate is active
|
||||||
|
// and the content rung is a whole-file hash.
|
||||||
|
const size = int64(headTailMin + 2*1024*1024)
|
||||||
|
|
||||||
|
base := sparseFile(t, dir, "ends-base", size)
|
||||||
|
headDiff := sparseFile(t, dir, "ends-head", size)
|
||||||
|
tailDiff := sparseFile(t, dir, "ends-tail", size)
|
||||||
|
midDiff := sparseFile(t, dir, "ends-mid", size)
|
||||||
|
|
||||||
|
pokeAt(t, headDiff, 0, []byte{1})
|
||||||
|
pokeAt(t, tailDiff, size-1, []byte{1})
|
||||||
|
pokeAt(t, midDiff, size/2, []byte{1})
|
||||||
|
|
||||||
|
bHead, bTail, bContent := sig(t, base, size)
|
||||||
|
if bContent != "" {
|
||||||
|
t.Errorf("content = %q, want none from the hash phase", bContent)
|
||||||
|
}
|
||||||
|
|
||||||
|
h, tl, _ := sig(t, headDiff, size)
|
||||||
|
if h == bHead {
|
||||||
|
t.Error("a byte in the first window did not change head")
|
||||||
|
}
|
||||||
|
|
||||||
|
if tl != bTail {
|
||||||
|
t.Error("a byte in the first window changed tail")
|
||||||
|
}
|
||||||
|
|
||||||
|
h, tl, _ = sig(t, tailDiff, size)
|
||||||
|
if tl == bTail {
|
||||||
|
t.Error("a byte in the last window did not change tail")
|
||||||
|
}
|
||||||
|
|
||||||
|
if h != bHead {
|
||||||
|
t.Error("a byte in the last window changed head")
|
||||||
|
}
|
||||||
|
|
||||||
|
h, tl, _ = sig(t, midDiff, size)
|
||||||
|
if h != bHead || tl != bTail {
|
||||||
|
t.Error("a byte between the windows changed an end hash")
|
||||||
|
}
|
||||||
|
|
||||||
|
// base and midDiff match on size, head, and tail, so the scan reads
|
||||||
|
// both for their content hashes.
|
||||||
|
c := scanContents(t, dir, base, midDiff)
|
||||||
|
if c[midDiff] == c[base] {
|
||||||
|
t.Error("whole-file content rung ignored a byte between the windows")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHashSignatureErrors(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
// A missing file: an error, and every hash left empty.
|
||||||
|
head, tail, content, err := hashSignature(filepath.Join(dir, "missing"), 1)
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Error("no error for a missing file")
|
t.Error("no error for a missing file")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if head != "" || tail != "" || content != "" {
|
||||||
|
t.Errorf("missing file returned hashes: %q %q %q", head, tail, content)
|
||||||
|
}
|
||||||
|
|
||||||
// A zero-length file has constant hashes and is never opened: even
|
// A zero-length file has constant hashes and is never opened: even
|
||||||
// a missing path succeeds.
|
// a missing path succeeds.
|
||||||
head, tail, err := hashHeadTail(filepath.Join(dir, "missing"), 0)
|
head, tail, content, err = hashSignature(filepath.Join(dir, "missing"), 0)
|
||||||
if err != nil || head != emptyHash || tail != emptyHash {
|
if err != nil ||
|
||||||
t.Errorf("empty: head=%q tail=%q err=%v, want constant hashes",
|
head != emptyHash || tail != emptyHash || content != emptyHash {
|
||||||
head, tail, err)
|
t.Errorf("empty: head=%q tail=%q content=%q err=%v, "+
|
||||||
|
"want constant hashes", head, tail, content, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
// A file that shrank between the stat and hash passes: reading at
|
// A file that shrank between the stat and hash passes: reading at
|
||||||
// the stat-reported size must fail rather than emit wrong hashes.
|
// the stat-reported size must fail rather than emit wrong hashes.
|
||||||
p := writeFile(t, dir, "shrunk", []byte("tiny"))
|
p := writeFile(t, dir, "shrunk", []byte("tiny"))
|
||||||
|
|
||||||
_, _, err = hashHeadTail(p, int64(2*chunk))
|
head, tail, content, err = hashSignature(p, int64(2*headTailWindow))
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Error("no error when the stat size exceeds the file size")
|
t.Error("no error when the stat size exceeds the file size")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if head != "" || tail != "" || content != "" {
|
||||||
|
t.Errorf("shrunk file returned hashes: %q %q %q", head, tail, content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sparseFile creates a file that is logically size bytes long without
|
||||||
|
// allocating blocks for the hole, so multi-gigabyte cases stay cheap.
|
||||||
|
func sparseFile(t *testing.T, dir, name string, size int64) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
p := filepath.Join(dir, name)
|
||||||
|
|
||||||
|
f, err := os.Create(p) //nolint:gosec // test-controlled path
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = f.Truncate(size)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = f.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
// pokeAt writes data into an existing file at off, leaving the rest of
|
||||||
|
// the file (a sparse hole) untouched.
|
||||||
|
func pokeAt(t *testing.T, path string, off int64, data []byte) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
f, err := os.OpenFile(path, os.O_WRONLY, 0o600) //nolint:gosec // test path
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err = f.WriteAt(data, off)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = f.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// scanContents scans dir into a fresh database and returns the content
|
||||||
|
// hash recorded for each file, by path, failing the test if one of want
|
||||||
|
// has none. A file of headTailMin or more gets a content hash only when
|
||||||
|
// it is scanned with a file of the same size, head, and tail.
|
||||||
|
func scanContents(t *testing.T, dir string,
|
||||||
|
want ...string,
|
||||||
|
) map[string]string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
contents := make(map[string]string)
|
||||||
|
for _, r := range dbRecords(t, db) {
|
||||||
|
contents[r.path] = r.content
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, p := range want {
|
||||||
|
if contents[p] == "" {
|
||||||
|
t.Fatalf("%s: no content hash", p)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return contents
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestContentRungBoundary checks the 50 MiB boundary between the two
|
||||||
|
// content rungs: just below it the whole file is hashed and any byte
|
||||||
|
// difference shows; at the boundary only the gigabyte-spaced samples are
|
||||||
|
// hashed, so a difference outside a sample window is invisible. The
|
||||||
|
// files of each pair match on size, head, and tail, so the scan reads
|
||||||
|
// both for their content hashes.
|
||||||
|
func TestContentRungBoundary(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
// A byte that lands outside the single [0, sampleWindow) sample a
|
||||||
|
// sub-gigabyte file has, but well inside the file.
|
||||||
|
const off = 10 * 1024 * 1024
|
||||||
|
|
||||||
|
// Just under the boundary: the whole-file rung sees the poked byte.
|
||||||
|
under := int64(wholeFileMax - 1)
|
||||||
|
underBase := sparseFile(t, dir, "under-base", under)
|
||||||
|
underPoked := sparseFile(t, dir, "under-poked", under)
|
||||||
|
|
||||||
|
pokeAt(t, underPoked, off, []byte{1})
|
||||||
|
|
||||||
|
// At the boundary: only [0, sampleWindow) is sampled, so the poked
|
||||||
|
// byte at off is invisible and the two content hashes match.
|
||||||
|
at := int64(wholeFileMax)
|
||||||
|
atBase := sparseFile(t, dir, "at-base", at)
|
||||||
|
atPoked := sparseFile(t, dir, "at-poked", at)
|
||||||
|
|
||||||
|
pokeAt(t, atPoked, off, []byte{1})
|
||||||
|
|
||||||
|
c := scanContents(t, dir, underBase, underPoked, atBase, atPoked)
|
||||||
|
|
||||||
|
if c[underBase] == c[underPoked] {
|
||||||
|
t.Error("whole-file rung ignored a byte difference below wholeFileMax")
|
||||||
|
}
|
||||||
|
|
||||||
|
if c[atBase] != c[atPoked] {
|
||||||
|
t.Error("sampled rung saw a byte outside every sample window")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestContentRungMultiGigabyte exercises the sampled rung across several
|
||||||
|
// gigabytes using sparse files: a difference inside the third sample
|
||||||
|
// window (at offset 2*sampleStride) changes the hash, while a difference
|
||||||
|
// in the gap after it does not. The three files match on size, head,
|
||||||
|
// and tail, so the scan reads each for its content hash.
|
||||||
|
func TestContentRungMultiGigabyte(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
// Three sample windows (offsets 0, 1 GiB, 2 GiB) plus a trailing gap
|
||||||
|
// that no sample covers.
|
||||||
|
size := int64(2*sampleStride + 2*sampleWindow)
|
||||||
|
thirdSample := int64(2 * sampleStride)
|
||||||
|
gap := thirdSample + int64(sampleWindow)
|
||||||
|
|
||||||
|
base := sparseFile(t, dir, "g-base", size)
|
||||||
|
inSample := sparseFile(t, dir, "g-insample", size)
|
||||||
|
inGap := sparseFile(t, dir, "g-ingap", size)
|
||||||
|
|
||||||
|
pokeAt(t, inSample, thirdSample, []byte{1})
|
||||||
|
pokeAt(t, inGap, gap, []byte{1})
|
||||||
|
|
||||||
|
c := scanContents(t, dir, base, inSample, inGap)
|
||||||
|
|
||||||
|
if c[inSample] == c[base] {
|
||||||
|
t.Error("sample at 2 GiB was not read: difference there was invisible")
|
||||||
|
}
|
||||||
|
|
||||||
|
if c[inGap] != c[base] {
|
||||||
|
t.Error("a byte in an unsampled gap changed the content hash")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sparseFileWithoutMatch writes name in dir as a sparse file of size
|
||||||
|
// bytes, next to another file of that size whose first byte differs. A
|
||||||
|
// scan then reads the file's head and tail, since its size is shared,
|
||||||
|
// but finds no file matching them, so it gets no content hash.
|
||||||
|
func sparseFileWithoutMatch(t *testing.T, dir, name string,
|
||||||
|
size int64,
|
||||||
|
) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
p := sparseFile(t, dir, name, size)
|
||||||
|
other := sparseFile(t, dir, name+"-other-head", size)
|
||||||
|
|
||||||
|
pokeAt(t, other, 0, []byte{1})
|
||||||
|
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentGate checks that a file of headTailMin or more is read
|
||||||
|
// for its content hash only when its size, head, and tail match another
|
||||||
|
// file's: a same-size pair whose heads differ and one whose tails differ
|
||||||
|
// get no content hash and are not reported, while an identical pair is
|
||||||
|
// read and reported.
|
||||||
|
func TestScanContentGate(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
|
||||||
|
// Three sizes, so that no pair meets another.
|
||||||
|
headA := sparseFile(t, dir, "head-a", headTailMin)
|
||||||
|
headB := sparseFile(t, dir, "head-b", headTailMin)
|
||||||
|
tailA := sparseFile(t, dir, "tail-a", headTailMin+1)
|
||||||
|
tailB := sparseFile(t, dir, "tail-b", headTailMin+1)
|
||||||
|
same := []string{
|
||||||
|
sparseFile(t, dir, "same-a", headTailMin+2),
|
||||||
|
sparseFile(t, dir, "same-b", headTailMin+2),
|
||||||
|
}
|
||||||
|
|
||||||
|
pokeAt(t, headB, 0, []byte{1})
|
||||||
|
pokeAt(t, tailB, headTailMin, []byte{1}) // its last byte
|
||||||
|
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
for _, p := range []string{headA, headB, tailA, tailB} {
|
||||||
|
r := recordByPath(t, recs, p)
|
||||||
|
if r.head == "" || r.tail == "" || r.content != "" {
|
||||||
|
t.Errorf("%s: head = %q tail = %q content = %q, "+
|
||||||
|
"want head and tail only", p, r.head, r.tail, r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||||
|
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||||
|
groups, same)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentAcrossOperands checks that a stored file gets its
|
||||||
|
// content hash when a later scan of a separate operand brings its
|
||||||
|
// match: tree A's file has a head and tail but no content hash until
|
||||||
|
// tree B, holding an identical file, is scanned.
|
||||||
|
func TestScanContentAcrossOperands(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, filepath.Dir(a))
|
||||||
|
|
||||||
|
if r := recordByPath(t, dbRecords(t, db), a); r.head == "" || r.content != "" {
|
||||||
|
t.Fatalf("after scanning A: %+v, want head and tail only", r)
|
||||||
|
}
|
||||||
|
|
||||||
|
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, filepath.Dir(b))
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
if r := recordByPath(t, recs, a); r.content == "" {
|
||||||
|
t.Fatalf("after scanning B: %+v, want A's file content-hashed", r)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := []string{a, b}
|
||||||
|
slices.Sort(want)
|
||||||
|
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentWithinOperand checks that a rescan adding a match next
|
||||||
|
// to an unchanged stored file gives the stored file its content hash,
|
||||||
|
// though the hash phase leaves it alone as unchanged.
|
||||||
|
func TestScanContentWithinOperand(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
stored := sparseFileWithoutMatch(t, dir, "d1", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
added := sparseFile(t, dir, "d2", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dir)
|
||||||
|
if st != (scanStats{added: 1, unchanged: 2}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := []string{stored, added}
|
||||||
|
|
||||||
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentStalePartners checks that a stored file outside the
|
||||||
|
// operand that has vanished, or changed, since it was recorded is not
|
||||||
|
// read, and that its match inside the operand is not read either: the
|
||||||
|
// match has no other partner left, so neither gets a content hash and
|
||||||
|
// no duplicate is reported.
|
||||||
|
func TestScanContentStalePartners(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
dirA := t.TempDir()
|
||||||
|
gone := sparseFileWithoutMatch(t, dirA, "gone", headTailMin)
|
||||||
|
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin+1)
|
||||||
|
|
||||||
|
syncTree(t, db, dirA)
|
||||||
|
|
||||||
|
before := dbRecords(t, db)
|
||||||
|
|
||||||
|
err := os.Remove(gone)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
future := time.Now().Add(time.Hour)
|
||||||
|
|
||||||
|
err = os.Chtimes(changed, future, future)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
dirB := t.TempDir()
|
||||||
|
sparseFile(t, dirB, "gone-copy", headTailMin)
|
||||||
|
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{added: 2}) {
|
||||||
|
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
for _, r := range recs {
|
||||||
|
if r.content != "" {
|
||||||
|
t.Errorf("%s: content = %q, want none: its only match is stale",
|
||||||
|
r.path, r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, old := range before {
|
||||||
|
if r := recordByPath(t, recs, old.path); r != old {
|
||||||
|
t.Errorf("record = %+v, want it left as %+v", r, old)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||||
|
t.Errorf("groups = %+v, want none", groups)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentHashedStalePartners checks that stored matches outside
|
||||||
|
// the operand that already have a content hash are checked like any
|
||||||
|
// other: once one has vanished and the other has changed, a copy of
|
||||||
|
// them scanned in another tree has no match left, so it is not read and
|
||||||
|
// is not reported as their duplicate.
|
||||||
|
func TestScanContentHashedStalePartners(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
dirA := t.TempDir()
|
||||||
|
stored := []string{
|
||||||
|
sparseFile(t, dirA, "changed", headTailMin),
|
||||||
|
sparseFile(t, dirA, "gone", headTailMin),
|
||||||
|
}
|
||||||
|
|
||||||
|
// The two stored files match, so this scan gives both a content
|
||||||
|
// hash.
|
||||||
|
syncTree(t, db, dirA)
|
||||||
|
|
||||||
|
err := os.Remove(stored[1])
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
future := time.Now().Add(time.Hour)
|
||||||
|
|
||||||
|
err = os.Chtimes(stored[0], future, future)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
|
if st != (scanStats{added: 1}) {
|
||||||
|
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
if r := recordByPath(t, recs, b); r.content != "" {
|
||||||
|
t.Errorf("copy: content = %q, want none: its only matches are stale",
|
||||||
|
r.content)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The stored records lie outside the operand and are left as they
|
||||||
|
// are, so they still group with each other, but not with the copy.
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||||
|
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentReadFailure checks that a failed content read is
|
||||||
|
// counted as skipped and leaves the record without a content hash, and
|
||||||
|
// that a later scan tries the read again.
|
||||||
|
func TestScanContentReadFailure(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, filepath.Dir(a))
|
||||||
|
|
||||||
|
// lstat still works on the unreadable file, so it passes the check
|
||||||
|
// and fails only when it is read.
|
||||||
|
err := os.Chmod(a, 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
dirB := t.TempDir()
|
||||||
|
b := sparseFile(t, dirB, "b", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
if r := recordByPath(t, dbRecords(t, db), a); r.content != "" {
|
||||||
|
t.Fatalf("unreadable file: %+v, want no content hash", r)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = os.Chmod(a, 0o600)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
st = syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{unchanged: 1}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := []string{a, b}
|
||||||
|
slices.Sort(want)
|
||||||
|
|
||||||
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
|
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||||
|
groups, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentCheckError checks that a stored file the content phase
|
||||||
|
// cannot lstat, for a reason other than its being gone, is counted as
|
||||||
|
// skipped and does not count as a match.
|
||||||
|
func TestScanContentCheckError(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
sub := filepath.Join(t.TempDir(), "sub")
|
||||||
|
|
||||||
|
err := os.Mkdir(sub, 0o700)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sparseFileWithoutMatch(t, sub, "a", headTailMin)
|
||||||
|
syncTree(t, db, sub)
|
||||||
|
|
||||||
|
// Without search permission on its directory, the stored file's
|
||||||
|
// lstat fails with permission denied.
|
||||||
|
err = os.Chmod(sub, 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
//nolint:gosec // removing the directory needs its search bit back
|
||||||
|
_ = os.Chmod(sub, 0o700)
|
||||||
|
})
|
||||||
|
|
||||||
|
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
if r := recordByPath(t, dbRecords(t, db), b); r.content != "" {
|
||||||
|
t.Errorf("b: content = %q, want none: its only match could not be "+
|
||||||
|
"checked", r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentHardlinks checks that the content phase stores the
|
||||||
|
// content hash of a hard-linked file on every one of its links.
|
||||||
|
func TestScanContentHardlinks(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := sparseFile(t, dir, "a", headTailMin)
|
||||||
|
b := filepath.Join(dir, "b")
|
||||||
|
|
||||||
|
err := os.Link(a, b)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
c := sparseFile(t, dir, "copy", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dir)
|
||||||
|
if st != (scanStats{added: 3}) {
|
||||||
|
t.Fatalf("stats = %+v, want 3 added", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
|
||||||
|
want := recordByPath(t, recs, c).content
|
||||||
|
if want == "" {
|
||||||
|
t.Fatal("the copy has no content hash")
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, p := range []string{a, b} {
|
||||||
|
if got := recordByPath(t, recs, p).content; got != want {
|
||||||
|
t.Errorf("%s: content = %q, want %q", p, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectWalk runs a walk over roots and returns the emitted records
|
// collectWalk runs a walk over roots and returns the emitted records
|
||||||
@@ -706,9 +1289,9 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
|
|
||||||
recs := dbRecords(t, db)
|
recs := dbRecords(t, db)
|
||||||
for _, r := range recs {
|
for _, r := range recs {
|
||||||
if r.head != "" || r.tail != "" {
|
if r.head != "" || r.tail != "" || r.content != "" {
|
||||||
t.Errorf("%s: head = %q tail = %q, want unhashed",
|
t.Errorf("%s: head = %q tail = %q content = %q, want unhashed",
|
||||||
r.path, r.head, r.tail)
|
r.path, r.head, r.tail, r.content)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -740,23 +1323,36 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
func TestTreesUnhashedNeverEqual(t *testing.T) {
|
func TestTreesUnhashedNeverEqual(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// Two trees identical except for unhashed same-name, same-size
|
// Two trees identical except for same-name, same-size files without
|
||||||
// files (possible when the trees were scanned separately) must not
|
// a content hash must not compare equal: their content is unknown.
|
||||||
// compare equal: unhashed content is unknown.
|
// That holds for unhashed files (possible when the trees were
|
||||||
shared := pattern(1, 100)
|
// scanned separately) and for files of headTailMin or more that
|
||||||
recs := []scanRec{
|
// have only a head and tail.
|
||||||
{path: "/x/t1/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
sum := hexSum(pattern(1, 100))
|
||||||
{path: "/x/t2/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
shared := []scanRec{
|
||||||
{path: "/x/t1/u", size: 50},
|
{path: "/x/t1/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||||
{path: "/x/t2/u", size: 50},
|
{path: "/x/t2/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
cases := map[string][]scanRec{
|
||||||
super.compute()
|
"unhashed": {
|
||||||
|
{path: "/x/t1/u", size: 50},
|
||||||
|
{path: "/x/t2/u", size: 50},
|
||||||
|
},
|
||||||
|
"head and tail only": {
|
||||||
|
{path: "/x/t1/u", size: headTailMin, head: "h", tail: "t"},
|
||||||
|
{path: "/x/t2/u", size: headTailMin, head: "h", tail: "t"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
for name, unknown := range cases {
|
||||||
t.Fatalf("tree groups = %d, want 0 (unhashed files differ)",
|
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
||||||
len(tg))
|
super.compute()
|
||||||
|
|
||||||
|
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||||
|
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||||
|
name, len(tg))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -71,9 +71,9 @@ main() {
|
|||||||
"the two pins disagree:" >&2
|
"the two pins disagree:" >&2
|
||||||
echo "verify-lint-image-pin: $LINT_DOCKERFILE: $lint_ref" >&2
|
echo "verify-lint-image-pin: $LINT_DOCKERFILE: $lint_ref" >&2
|
||||||
echo "verify-lint-image-pin: $MAIN_DOCKERFILE: $main_ref" >&2
|
echo "verify-lint-image-pin: $MAIN_DOCKERFILE: $main_ref" >&2
|
||||||
echo "verify-lint-image-pin: bump both FROM lines together, tag and" \
|
echo "verify-lint-image-pin: bump both FROM lines together so" \
|
||||||
"digest, so script/lint and the Dockerfile lint stage keep" \
|
"script/lint and the Dockerfile lint stage keep running the" \
|
||||||
"running the same linter" >&2
|
"same linter" >&2
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|||||||
@@ -13,9 +13,10 @@ import (
|
|||||||
|
|
||||||
// fileSig is a file's duplicate signature; mtime is excluded.
|
// fileSig is a file's duplicate signature; mtime is excluded.
|
||||||
type fileSig struct {
|
type fileSig struct {
|
||||||
size int64
|
size int64
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
|
content string
|
||||||
}
|
}
|
||||||
|
|
||||||
// treeNode is one directory reconstructed from the scan stream.
|
// treeNode is one directory reconstructed from the scan stream.
|
||||||
@@ -122,14 +123,16 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
|||||||
node.files = make(map[string]fileSig)
|
node.files = make(map[string]fileSig)
|
||||||
}
|
}
|
||||||
|
|
||||||
sig := fileSig{size: r.size, head: r.head, tail: r.tail}
|
sig := fileSig{
|
||||||
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
|
|
||||||
// An unhashed record (its size was unique when last scanned)
|
// A record without a content hash has unknown content (README
|
||||||
// has unknown content: give it a signature no other file can
|
// "Database"): give it a signature no other file can share, so
|
||||||
// share, so trees containing it never compare equal. Real
|
// trees containing it never compare equal. Real hashes are
|
||||||
// heads are hex, so the NUL-prefixed form cannot collide.
|
// hex, so the NUL-prefixed form cannot collide.
|
||||||
if sig.head == "" {
|
if sig.content == "" {
|
||||||
sig.head = "unhashed\x00" + r.path
|
sig.content = "unhashed\x00" + r.path
|
||||||
}
|
}
|
||||||
|
|
||||||
node.files[comps[len(comps)-1]] = sig
|
node.files[comps[len(comps)-1]] = sig
|
||||||
@@ -188,7 +191,7 @@ func (n *treeNode) compute() {
|
|||||||
for name, sig := range n.files {
|
for name, sig := range n.files {
|
||||||
entries = append(entries,
|
entries = append(entries,
|
||||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||||
"\x00"+sig.head+"\x00"+sig.tail)
|
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
||||||
n.fileCount++
|
n.fileCount++
|
||||||
n.totalSize += sig.size
|
n.totalSize += sig.size
|
||||||
}
|
}
|
||||||
|
|||||||
+20
-17
@@ -7,22 +7,25 @@ import (
|
|||||||
|
|
||||||
// Signature hashes shared by the smoke-test records.
|
// Signature hashes shared by the smoke-test records.
|
||||||
const (
|
const (
|
||||||
f1Head = "f1h"
|
f1Head = "f1h"
|
||||||
f1Tail = "f1t"
|
f1Tail = "f1t"
|
||||||
f2Head = "f2h"
|
f1Content = "f1c"
|
||||||
f2Tail = "f2t"
|
f2Head = "f2h"
|
||||||
|
f2Tail = "f2t"
|
||||||
|
f2Content = "f2c"
|
||||||
)
|
)
|
||||||
|
|
||||||
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
||||||
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
||||||
func smokeTreeRecs() []scanRec {
|
func smokeTreeRecs() []scanRec {
|
||||||
return []scanRec{
|
return []scanRec{
|
||||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"},
|
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
|
||||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"},
|
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
|
||||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"},
|
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
|
||||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"},
|
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
|
||||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"},
|
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
|
||||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"},
|
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
|
||||||
|
path: "/d/t3/sub/f2renamed"},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -114,8 +117,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
const sharedTail = "same"
|
const sharedTail = "same"
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"},
|
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
|
||||||
{size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"},
|
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
@@ -181,8 +184,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
|||||||
// Identical sibling dirs share a parent, so their group cannot be
|
// Identical sibling dirs share a parent, so their group cannot be
|
||||||
// implied by a parent group and must be reported.
|
// implied by a parent group and must be reported.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 10, head: "h", tail: "t", path: "/p/x1/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
|
||||||
{size: 10, head: "h", tail: "t", path: "/p/x2/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
@@ -203,9 +206,9 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
|||||||
// extra file, so the parents' digests differ and the x group must
|
// extra file, so the parents' digests differ and the x group must
|
||||||
// be reported.
|
// be reported.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 10, head: "h", tail: "t", path: "/p/a/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
|
||||||
{size: 99, head: "e", tail: "e", path: "/p/a/extra"},
|
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
|
||||||
{size: 10, head: "h", tail: "t", path: "/q/b/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
|||||||
Reference in New Issue
Block a user